Files
2026-09-30 20:30:56 +03:00

17046 lines
756 KiB
Python

import os
import sys
import socket
import sqlite3
import requests
import json
import re
import subprocess
import concurrent.futures
import threading
import time
import multiprocessing
import tempfile
import shutil
import tarfile
import zipfile
import gzip
import bz2
import codecs
import lzma
import base64
import binascii
import copy
import hashlib
import io
import logging
import math
import stat
import uuid
import ctypes
import ipaddress
from email.utils import parsedate_to_datetime
from html.parser import HTMLParser
from collections import Counter
from contextlib import contextmanager
from contextvars import ContextVar
from dataclasses import dataclass, replace
from urllib.parse import parse_qsl, quote, urlencode, urljoin, urlsplit, urlunsplit
from datetime import datetime, timedelta, timezone
import urllib3.util.connection as urllib3_connection
import urllib3.exceptions as urllib3_exceptions
from paths import default_project_paths
from scanner_db import (
DOCKER_ADAPTIVE_PAYLOAD_CLASSES,
canonical_docker_layer_plan_bytes,
canonical_git_scan_plan_bytes,
normalize_target,
sanitize_endpoint,
sanitize_endpoint_host,
sanitize_postman_context,
target_status,
validate_docker_layer_plan,
validate_git_resolution,
)
from owned_process import OwnedProcess, run_owned
from process_identity import (
current_process_identity,
exact_process_identity_state,
open_process,
serialize_process_identity,
)
from janitor import JanitorBudget, bounded_remove_tree
from runtime_security import (
atomic_write_private_json,
canonical_path,
durable_replace,
durable_unlink,
ensure_private_directory,
harden_private_directory,
harden_private_file,
harden_private_tree,
is_reparse_point,
PrivateFileLock,
private_directory_ready,
private_file_ready,
read_private_json,
reject_reparse_components,
require_private_directory,
require_private_file,
sha256_file,
)
from target_identity import (
normalize_docker_digest,
normalize_huggingface_space_id,
parse_dockerhub_digest_target,
parse_docker_target,
postman_target_identity as semantic_postman_target_identity,
validate_docker_image_reference,
)
from docker_depth_experiment import (
DOCKER_DEPTH_SELECTOR_VERSION,
canonical_docker_depth_selection_evidence_hash,
canonical_selector_hash,
select_docker_layer_graphs,
validate_docker_images_per_repository,
)
from lifecycle_authority import (
CHILD_KIND_ENV,
REMOTE_WORKER_CODE_AUTHORITY_FILES,
verify_code_manifest,
require_active_supervisor_child,
resolve_manifest_executable,
strip_supervisor_credentials,
)
from keycheck_candidates import extract_candidates, extract_structured_candidates
from result_bundle import BundleReservation, ResultBundleWriter
from worker_contracts import (
AssignmentOutcome,
DiagnosticCategory,
DiagnosticExceptionContext,
DiagnosticHTTPContext,
DiagnosticKind,
DiagnosticProcessContext,
MAX_DIAGNOSTIC_BODY_BYTES,
MAX_DIAGNOSTIC_LOG_BYTES,
ScanOutcome,
WorkerPhase,
build_diagnostic_envelope,
diagnostic_material_bytes,
make_body_material,
make_log_material,
)
def configure_requests_networking():
if os.getenv('SCANNER_FORCE_IPV4', '1').strip().lower() in ('0', 'false', 'no', 'off'):
return
urllib3_connection.allowed_gai_family = lambda: socket.AF_INET
logger = logging.getLogger(__name__)
_runtime_initialized = False
_cleanup_registered = False
_client_scan_manifest = ContextVar('client_scan_manifest', default=None)
_client_scan_policy = ContextVar('client_scan_policy', default=None)
_client_remote_execution_kind = ContextVar(
'client_remote_execution_kind', default=None,
)
_client_scan_phase_callback = ContextVar(
'client_scan_phase_callback', default=None,
)
@contextmanager
def client_scan_launch_authority(manifest, expected_sha256=None):
verified = verify_code_manifest(
manifest,
expected_sha256=expected_sha256,
required_names=REMOTE_WORKER_CODE_AUTHORITY_FILES,
external_names=(),
)
token = _client_scan_manifest.set(verified)
try:
yield verified
finally:
_client_scan_manifest.reset(token)
@contextmanager
def client_scan_execution_policy(policy):
if not isinstance(policy, dict):
raise RuntimeError('remote scan execution policy is invalid')
token = _client_scan_policy.set(dict(policy))
try:
yield
finally:
_client_scan_policy.reset(token)
@contextmanager
def client_remote_execution_binding(planning_kind):
if _client_scan_manifest.get() is None:
raise RuntimeError('remote direct execution requires client launch authority')
kind = str(planning_kind or '')
if kind not in {'docker_direct_v1', 'huggingface_space_v1'}:
raise RuntimeError('remote direct execution kind is invalid')
token = _client_remote_execution_kind.set(kind)
try:
yield
finally:
_client_remote_execution_kind.reset(token)
@contextmanager
def client_scan_phase_events(callback):
if callback is not None and not callable(callback):
raise TypeError('scan phase callback must be callable')
token = _client_scan_phase_callback.set(callback)
try:
yield
finally:
_client_scan_phase_callback.reset(token)
def emit_client_scan_phase(phase, progress=None):
callback = _client_scan_phase_callback.get()
if callback is not None:
callback(phase, dict(progress or {}))
def _scan_policy_value(name, default):
policy = _client_scan_policy.get()
if policy is not None:
if name not in policy:
raise RuntimeError('remote scan execution policy is incomplete')
return policy[name]
return getattr(scan_config, name, default)
def initialize_scanner_runtime(*, preflight_complete=False, register_cleanup=True):
"""Apply process-global scanner setup only after lifecycle preflight."""
global _runtime_initialized, _cleanup_registered
if not preflight_complete:
raise RuntimeError('scanner runtime initialization requires completed lifecycle preflight')
if not _runtime_initialized:
configure_requests_networking()
logging.basicConfig(
level=logging.INFO,
format='%(asctime)s - %(levelname)s - %(message)s',
)
_runtime_initialized = True
# Stale tree ownership belongs to the isolated janitor, never an atexit hook.
_cleanup_registered = False
def require_scanner_runtime_initialized():
if not _runtime_initialized:
raise RuntimeError('scanner runtime is not initialized after lifecycle preflight')
def bool_setting(value, default=False):
if value is None:
return default
if isinstance(value, bool):
return value
return str(value).strip().lower() in ('1', 'true', 'yes', 'on')
def int_setting(value, default):
try:
return int(value)
except (TypeError, ValueError):
return default
def float_setting(value, default):
try:
return float(value)
except (TypeError, ValueError):
return default
def csv_items(value):
if not value:
return []
if isinstance(value, str):
return [item.strip() for item in value.split(',') if item.strip()]
return [str(item).strip() for item in value if str(item).strip()]
DEFAULT_DROP_DETECTORS = (
'Privacy',
'URI',
'JDBC',
'Postgres',
'MongoDB',
'SQLServer',
'Box',
'ZohoCRM',
'Accuweather',
'Roaring',
'Flatio',
'LinkPreview',
'RailwayApp',
)
class RateLimitError(Exception):
def __init__(
self, source, message, reset_at=None, category='rate_limit',
retryable=True, auth_related=True, diagnostic_http=None,
):
super().__init__(message)
self.source = source
self.reset_at = reset_at
self.category = category
self.retryable = retryable
self.auth_related = auth_related
self.diagnostic_http = (
dict(diagnostic_http) if isinstance(diagnostic_http, dict) else None
)
def response_message(response):
if response is None:
return ''
payload_bytes = bytearray()
try:
digest = hashlib.sha256()
original_size = 0
for chunk in response.iter_content(chunk_size=4096):
if not chunk:
continue
digest.update(chunk)
original_size += len(chunk)
remaining = MAX_DIAGNOSTIC_BODY_BYTES - len(payload_bytes)
if remaining > 0:
payload_bytes.extend(chunk[:remaining])
captured = bytes(payload_bytes)
material = make_body_material(captured)
if original_size > len(captured):
material = replace(
material,
original_size=original_size,
sha256=digest.hexdigest(),
truncated=True,
)
response._truf_diagnostic_body_material = material
response._truf_diagnostic_body = captured
response._truf_diagnostic_body_truncated = material.truncated
payload = json.loads(captured.decode('utf-8', errors='strict'))
if isinstance(payload, dict):
message = str(payload.get('message') or payload.get('error') or payload)
else:
message = str(payload)
return message.replace('\x00', '\\u0000')
except Exception:
captured = bytes(payload_bytes[:MAX_DIAGNOSTIC_BODY_BYTES])
response._truf_diagnostic_body = captured
return captured[:500].decode(
'utf-8', errors='replace'
).replace('\x00', '\\u0000')
def retry_after_reset(response):
if response is None:
return None
retry_after = response.headers.get('Retry-After')
if retry_after:
try:
return (datetime.now(timezone.utc) + timedelta(seconds=int(retry_after))).isoformat(timespec='seconds')
except (TypeError, ValueError):
return None
return None
def build_api_error(source, category, message, response=None, reset_at=None, retryable=True, auth_related=False):
if reset_at is None:
reset_at = retry_after_reset(response)
diagnostic_http = None
if response is not None and type(getattr(response, 'status_code', None)) is int:
headers = getattr(response, 'headers', {}) or {}
body_material = getattr(response, '_truf_diagnostic_body_material', None)
if body_material is None and (
not getattr(response, 'raw', None)
or getattr(response, '_content_consumed', False)
):
captured = getattr(response, 'content', b'')
body = captured if isinstance(captured, bytes) else str(captured).encode('utf-8')
body_material = make_body_material(body)
diagnostic_http = {
'operation': f'{source}-api',
'status_code': int(response.status_code),
'content_type': str(headers.get('Content-Type') or '') or None,
'request_id': str(
headers.get('X-Request-Id') or headers.get('X-GitHub-Request-Id') or ''
) or None,
'headers_b64': base64.b64encode(json.dumps(
{str(key): str(value) for key, value in headers.items()},
ensure_ascii=True, sort_keys=True, separators=(',', ':'),
).encode('ascii')).decode('ascii'),
}
if body_material is not None:
body = diagnostic_material_bytes(body_material)
diagnostic_http.update({
'body_b64': base64.b64encode(body).decode('ascii'),
'body_original_size': body_material.original_size,
'body_stored_size': body_material.stored_size,
'body_sha256': body_material.sha256,
'body_capture_truncated': body_material.truncated,
})
else:
diagnostic_http['body_capture_truncated'] = False
return RateLimitError(
source,
message,
reset_at=reset_at,
category=category,
retryable=retryable,
auth_related=auth_related,
diagnostic_http=diagnostic_http,
)
def github_api_error(response):
status = response.status_code if response is not None else None
message = response_message(response)
lower_message = message.lower()
reset_at = github_rate_limit_reset(response) or retry_after_reset(response)
if status == 401:
return build_api_error('github', 'auth_invalid', f'GitHub API authentication failed (HTTP 401): {message}', response, retryable=False, auth_related=True)
if status == 403:
remaining = response.headers.get('X-RateLimit-Remaining') if response is not None else None
if remaining == '0':
return build_api_error('github', 'rate_limit', f'GitHub API rate limit hit: {message}', response, reset_at, auth_related=True)
if 'secondary rate limit' in lower_message or 'abuse' in lower_message:
return build_api_error('github', 'secondary_rate_limit', f'GitHub secondary rate limit hit: {message}', response, reset_at, auth_related=True)
return build_api_error('github', 'auth_forbidden', f'GitHub API access forbidden (HTTP 403): {message}', response, retryable=False, auth_related=True)
if status == 429:
return build_api_error('github', 'rate_limit', f'GitHub API returned HTTP 429: {message}', response, reset_at, auth_related=True)
if status == 422:
return build_api_error('github', 'query_invalid', f'GitHub search query invalid (HTTP 422): {message}', response, retryable=False, auth_related=False)
if status == 404:
return build_api_error('github', 'not_found', f'GitHub API resource not found (HTTP 404): {message}', response, retryable=False, auth_related=False)
if status and status >= 500:
return build_api_error('github', 'server_error', f'GitHub API server error (HTTP {status}): {message}', response, auth_related=False)
return build_api_error('github', 'api', f'GitHub API error (HTTP {status}): {message}', response, auth_related=False)
def gitlab_api_error(response):
status = response.status_code if response is not None else None
message = response_message(response)
reset_at = gitlab_rate_limit_reset(response) or retry_after_reset(response)
if status == 401:
return build_api_error('gitlab', 'auth_invalid', f'GitLab API authentication failed (HTTP 401): {message}', response, retryable=False, auth_related=True)
if status == 403:
return build_api_error('gitlab', 'auth_forbidden', f'GitLab API access forbidden (HTTP 403): {message}', response, retryable=False, auth_related=True)
if status == 429:
return build_api_error('gitlab', 'rate_limit', f'GitLab API returned HTTP 429: {message}', response, reset_at, auth_related=True)
if status == 404:
return build_api_error('gitlab', 'not_found', f'GitLab API resource not found (HTTP 404): {message}', response, retryable=False, auth_related=False)
if status and status >= 500:
return build_api_error('gitlab', 'server_error', f'GitLab API server error (HTTP {status}): {message}', response, auth_related=False)
return build_api_error('gitlab', 'api', f'GitLab API error (HTTP {status}): {message}', response, auth_related=False)
# =====================
# GLOBAL CONFIGURATION
# =====================
class ScanConfig:
def __init__(self):
defaults = default_project_paths()
self.git_timeout = 900
self.docker_timeout = 1800
self.detectors = ""
self.exclude_detectors = os.getenv("TRUFFLEHOG_EXCLUDE_DETECTORS", "github.v1,gitlab.v1,GitHubOauth2")
self.no_verification = os.getenv("TRUFFLEHOG_NO_VERIFICATION", "0").strip().lower() in ("1", "true", "yes", "on")
self.strict_git_provider_token_filter = os.getenv(
"STRICT_GIT_PROVIDER_TOKEN_FILTER",
"1",
).strip().lower() not in ("0", "false", "no", "off")
self.drop_detectors = csv_items(os.getenv("SCANNER_DROP_DETECTORS"))
self.webhook_url = None
self.max_concurrent = min(10, max(1, multiprocessing.cpu_count() * 2))
self.trufflehog_path = os.getenv(
"TRUFFLEHOG_PATH",
defaults['trufflehog_path']
)
self.trufflehog_config = os.getenv("TRUFFLEHOG_CONFIG", "")
self.trufflehog_job_memory_limit_bytes = int_setting(
os.getenv("TRUFFLEHOG_JOB_MEMORY_LIMIT_BYTES"),
4096 * 1024 * 1024,
)
self.trufflehog_windows_job_cpu_weight = int_setting(
os.getenv("TRUFFLEHOG_WINDOWS_JOB_CPU_WEIGHT"), 0,
)
self.trufflehog_windows_memory_priority = int_setting(
os.getenv("TRUFFLEHOG_WINDOWS_MEMORY_PRIORITY"), 0,
)
self.trufflehog_stdout_max_mb = int_setting(os.getenv('TRUFFLEHOG_STDOUT_MAX_MB'), 32)
self.trufflehog_stderr_max_mb = int_setting(os.getenv('TRUFFLEHOG_STDERR_MAX_MB'), 8)
self.trufflehog_max_findings_per_target = int_setting(
os.getenv('TRUFFLEHOG_MAX_FINDINGS_PER_TARGET'), 20000,
)
self.work_dir = os.getenv("TRUFFLEHOG_WORK_DIR", defaults['work_dir'])
self.runtime_dir = defaults['runtime_dir']
self.results_dir = os.getenv("SCAN_RESULTS_DIR", defaults['results_dir'])
self.result_spool_dir = os.getenv("SCANNER_RESULT_SPOOL_DIR", defaults['result_spool_dir'])
self.result_spool_max_event_bytes = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MAX_EVENT_BYTES"), 192 * 1024 * 1024)
self.result_spool_max_events = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MAX_EVENTS"), 10000)
self.result_spool_max_total_bytes = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MAX_TOTAL_BYTES"), 2 * 1024 * 1024 * 1024)
self.result_spool_min_free_bytes = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MIN_FREE_BYTES"), 1024 * 1024 * 1024)
self.result_bundle_dir = os.getenv('SCANNER_RESULT_BUNDLE_DIR', defaults['result_bundle_dir'])
self.result_bundle_max_event_bytes = int_setting(
os.getenv('SCANNER_RESULT_BUNDLE_MAX_EVENT_BYTES'), 64 * 1024 * 1024,
)
self.scan_outbox_max_pending_items = int_setting(os.getenv('SCAN_OUTBOX_MAX_PENDING_ITEMS'), 10000)
self.scan_outbox_max_pending_bytes = int_setting(os.getenv('SCAN_OUTBOX_MAX_PENDING_BYTES'), 1024 * 1024 * 1024)
self.scan_outbox_max_pending_age_sec = int_setting(os.getenv('SCAN_OUTBOX_MAX_PENDING_AGE_SEC'), 24 * 60 * 60)
self.queue_dir = defaults['queue_dir']
self.keycheck_dir = defaults['keycheck_dir']
self.postman_cache_dir = defaults['postman_cache_dir']
self.postman_cache_max_items = int_setting(os.getenv('POSTMAN_CACHE_MAX_ITEMS'), 100000)
self.postman_cache_max_bytes = int_setting(os.getenv('POSTMAN_CACHE_MAX_BYTES'), 20 * 1024 * 1024 * 1024)
self.postman_cache_min_free_bytes = int_setting(os.getenv('POSTMAN_CACHE_MIN_FREE_BYTES'), 20 * 1024 * 1024 * 1024)
self.postman_cache_lock_timeout_sec = int_setting(os.getenv('POSTMAN_CACHE_LOCK_TIMEOUT_SEC'), 30)
self.postman_discovery_max_artifacts_per_cycle = int_setting(os.getenv('POSTMAN_DISCOVERY_MAX_ARTIFACTS_PER_CYCLE'), 1000)
self.postman_discovery_max_artifacts_per_page = int_setting(os.getenv('POSTMAN_DISCOVERY_MAX_ARTIFACTS_PER_PAGE'), 100)
self.postman_discovery_max_bytes_per_cycle = int_setting(os.getenv('POSTMAN_DISCOVERY_MAX_BYTES_PER_CYCLE'), 1024 * 1024 * 1024)
self.postman_discovery_max_elapsed_sec = float_setting(os.getenv('POSTMAN_DISCOVERY_MAX_ELAPSED_SEC'), 300.0)
self.postman_package_harvest_max_artifacts = int_setting(os.getenv('POSTMAN_PACKAGE_HARVEST_MAX_ARTIFACTS'), 100)
self.postman_package_harvest_max_bytes = int_setting(os.getenv('POSTMAN_PACKAGE_HARVEST_MAX_BYTES'), 128 * 1024 * 1024)
self.postman_package_harvest_max_elapsed_sec = float_setting(os.getenv('POSTMAN_PACKAGE_HARVEST_MAX_ELAPSED_SEC'), 30.0)
self.postman_context_max_input_bytes = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_INPUT_BYTES'), 16 * 1024 * 1024)
self.postman_context_max_nodes = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_NODES'), 100000)
self.postman_context_max_depth = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_DEPTH'), 64)
self.postman_context_max_scalar_bytes = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_SCALAR_BYTES'), 16 * 1024 * 1024)
self.postman_context_max_items = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_ITEMS'), 50000)
self.context_enrichment_max_source_bytes = int_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_SOURCE_BYTES'), 16 * 1024 * 1024)
self.context_enrichment_max_findings = int_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_FINDINGS'), 2000)
self.context_enrichment_max_postman_comparisons = int_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_POSTMAN_COMPARISONS'), 200000)
self.context_enrichment_max_elapsed_sec = float_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_ELAPSED_SEC'), 5.0)
self.trufflehog_diagnostic_max_lines = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_LINES'), 2000)
self.trufflehog_diagnostic_max_line_chars = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_LINE_CHARS'), 8192)
self.trufflehog_diagnostic_max_line_bytes = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_LINE_BYTES'), 8192)
self.trufflehog_diagnostic_max_errors = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_ERRORS'), 200)
self.trufflehog_diagnostic_max_warnings = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_WARNINGS'), 200)
self.trufflehog_diagnostic_max_unclassified = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_UNCLASSIFIED'), 20)
self.keycheck_input_max_line_bytes = int_setting(os.getenv('KEYCHECK_INPUT_MAX_LINE_BYTES'), 16 * 1024 * 1024)
self.keycheck_candidate_artifact_max_items = int_setting(os.getenv('KEYCHECK_CANDIDATE_ARTIFACT_MAX_ITEMS'), 2000)
self.keycheck_candidate_artifact_max_bytes = int_setting(os.getenv('KEYCHECK_CANDIDATE_ARTIFACT_MAX_BYTES'), 2 * 1024 * 1024)
self.keycheck_candidate_file_max_items = int_setting(os.getenv('KEYCHECK_CANDIDATE_FILE_MAX_ITEMS'), 100000)
self.keycheck_candidate_file_max_bytes = int_setting(os.getenv('KEYCHECK_CANDIDATE_FILE_MAX_BYTES'), 32 * 1024 * 1024)
self.keycheck_candidate_line_max_bytes = int_setting(os.getenv('KEYCHECK_CANDIDATE_LINE_MAX_BYTES'), 8192)
self.gharchive_cache_dir = defaults['gharchive_cache_dir']
self.gharchive_cache_max_items = int_setting(os.getenv('GHARCHIVE_CACHE_MAX_ITEMS'), 48)
self.gharchive_cache_max_bytes = int_setting(os.getenv('GHARCHIVE_CACHE_MAX_BYTES'), 8 * 1024 * 1024 * 1024)
self.gharchive_cache_min_free_bytes = int_setting(os.getenv('GHARCHIVE_CACHE_MIN_FREE_BYTES'), 5 * 1024 * 1024 * 1024)
self.gharchive_download_max_bytes = int_setting(os.getenv('GHARCHIVE_DOWNLOAD_MAX_BYTES'), 512 * 1024 * 1024)
self.gharchive_decompressed_max_bytes = int_setting(os.getenv('GHARCHIVE_DECOMPRESSED_MAX_BYTES'), 8 * 1024 * 1024 * 1024)
self.gharchive_max_events = int_setting(os.getenv('GHARCHIVE_MAX_EVENTS'), 5000000)
self.gharchive_max_line_bytes = int_setting(os.getenv('GHARCHIVE_MAX_LINE_BYTES'), 8 * 1024 * 1024)
self.gharchive_cache_lock_timeout_sec = int_setting(os.getenv('GHARCHIVE_CACHE_LOCK_TIMEOUT_SEC'), 600)
self.proxy_file = defaults['proxy_file']
self.api_proxy_enabled = bool_setting(os.getenv("SCANNER_API_PROXY_ENABLED"), False)
self.api_proxy_file = os.getenv("SCANNER_API_PROXY_FILE", self.proxy_file)
self.api_proxy_timeout = int_setting(os.getenv("SCANNER_API_PROXY_TIMEOUT"), 5)
self.api_proxy_max_retries = int_setting(os.getenv("SCANNER_API_PROXY_MAX_RETRIES"), 100)
self.api_proxy_retry_delay = int_setting(os.getenv("SCANNER_API_PROXY_RETRY_DELAY"), 5)
self.download_proxy_enabled = bool_setting(os.getenv("SCANNER_DOWNLOAD_PROXY_ENABLED"), False)
self.download_proxy_file = os.getenv("SCANNER_DOWNLOAD_PROXY_FILE", "")
self.max_active_scans = int_setting(os.getenv("SCANNER_MAX_ACTIVE_SCANS"), 0)
self.opportunistic_scan_slots = int_setting(
os.getenv("SCANNER_OPPORTUNISTIC_SCAN_SLOTS"), 0,
)
self.opportunistic_scan_sources = csv_items(
os.getenv("SCANNER_OPPORTUNISTIC_SCAN_SOURCES")
)
self.opportunistic_scan_reserve_overhead_bytes = int_setting(
os.getenv("SCANNER_OPPORTUNISTIC_SCAN_RESERVE_OVERHEAD_BYTES"),
1024 * 1024 * 1024,
)
self.opportunistic_scan_min_available_after_reserve_bytes = int_setting(
os.getenv("SCANNER_OPPORTUNISTIC_SCAN_MIN_AVAILABLE_AFTER_RESERVE_BYTES"),
4 * 1024 * 1024 * 1024,
)
self.opportunistic_scan_min_commit_after_reserve_bytes = int_setting(
os.getenv("SCANNER_OPPORTUNISTIC_SCAN_MIN_COMMIT_AFTER_RESERVE_BYTES"),
6 * 1024 * 1024 * 1024,
)
self.scan_limiter_db = os.getenv("SCANNER_SCAN_LIMITER_DB", os.path.join(defaults['state_dir'], 'scan_limiter.db'))
self.dockerhub_tag_cache_path = os.getenv("DOCKERHUB_TAG_CACHE_PATH", os.path.join(defaults['state_dir'], 'dockerhub_tag_cache.sqlite'))
self.dockerhub_tag_cache_ttl_sec = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_TTL_SEC"), 21600)
self.dockerhub_tag_negative_cache_ttl_sec = int_setting(os.getenv("DOCKERHUB_TAG_NEGATIVE_CACHE_TTL_SEC"), 3600)
self.dockerhub_tag_rate_limit_cache_ttl_sec = int_setting(os.getenv("DOCKERHUB_TAG_RATE_LIMIT_CACHE_TTL_SEC"), 1800)
self.dockerhub_tag_cache_max_rows = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MAX_ROWS"), 50000)
self.dockerhub_tag_cache_max_age_sec = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MAX_AGE_SEC"), 7 * 86400)
self.dockerhub_tag_cache_max_bytes = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MAX_BYTES"), 256 * 1024 * 1024)
self.dockerhub_tag_cache_min_free_bytes = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MIN_FREE_BYTES"), 512 * 1024 * 1024)
self.scan_slot_wait_sec = float_setting(os.getenv("SCANNER_SCAN_SLOT_WAIT_SEC"), 0.5)
self.scan_slot_wait_log_sec = int_setting(os.getenv("SCANNER_SCAN_SLOT_WAIT_LOG_SEC"), 30)
self.scan_slot_stale_sec = int_setting(os.getenv("SCANNER_SCAN_SLOT_STALE_SEC"), 7200)
self.low_space_cleanup_max_items = int_setting(os.getenv("SCANNER_LOW_SPACE_CLEANUP_MAX_ITEMS"), 50)
self.min_free_gb = float(os.getenv("TRUFFLEHOG_MIN_FREE_GB", "5"))
self.jsonl_rotation_enabled = bool_setting(os.getenv("SCANNER_JSONL_ROTATION_ENABLED"), False)
self.found_secrets_max_mb = int_setting(os.getenv("SCANNER_FOUND_SECRETS_MAX_MB"), 512)
self.scan_results_max_mb = int_setting(os.getenv("SCANNER_SCAN_RESULTS_MAX_MB"), 1024)
self.scan_errors_max_mb = int_setting(os.getenv("SCANNER_SCAN_ERRORS_MAX_MB"), 64)
self.scan_errors_keep = int_setting(os.getenv("SCANNER_SCAN_ERRORS_KEEP"), 5)
self.jsonl_lock_stale_sec = int_setting(os.getenv("SCANNER_JSONL_LOCK_STALE_SEC"), 300)
self.jsonl_max_segments = int_setting(os.getenv("SCANNER_JSONL_MAX_SEGMENTS"), 16)
self.jsonl_ledger_max_rows = int_setting(os.getenv("SCANNER_JSONL_LEDGER_MAX_ROWS"), 1000000)
self.jsonl_ledger_max_bytes = int_setting(os.getenv("SCANNER_JSONL_LEDGER_MAX_BYTES"), 512 * 1024 * 1024)
self.jsonl_legacy_index_max_bytes = int_setting(os.getenv("SCANNER_JSONL_LEGACY_INDEX_MAX_BYTES"), 16 * 1024 * 1024)
self.jsonl_tail_scan_max_bytes = int_setting(os.getenv("SCANNER_JSONL_TAIL_SCAN_MAX_BYTES"), 8 * 1024 * 1024)
self.jsonl_torn_quarantine_max_bytes = int_setting(os.getenv("SCANNER_JSONL_TORN_QUARANTINE_MAX_BYTES"), 64 * 1024)
scan_config = ScanConfig()
pending_temp_dirs = set()
pending_temp_lock = threading.Lock()
TEMP_OWNER_FILE = '.scanner-owner.json'
TEMP_OWNER_SCHEMA = 2
PENDING_TEMP_SCHEMA = 1
APPROVED_TEMP_PREFIXES = ('trufflehog-', 'trufflehog-run-', 'trufflehog-probe-', 'hg-run-', 'docker-config-', 'tmp-', 'worker-assignment-')
class ApiRequestError(Exception):
def __init__(
self, message, *, response=None, operation='provider-api',
capture_body=True,
):
super().__init__(message)
self.operation = str(operation)
self.status_code = None
self.content_type = None
self.request_id = None
self.body = None
self.body_original_size = None
self.body_stored_size = None
self.body_sha256 = None
self.body_capture_truncated = False
self.headers = None
if response is not None and type(getattr(response, 'status_code', None)) is int:
self.status_code = int(response.status_code)
headers = getattr(response, 'headers', {}) or {}
self.headers = {
str(key): str(value) for key, value in headers.items()
}
self.content_type = str(headers.get('Content-Type') or '') or None
self.request_id = str(
headers.get('X-Request-Id') or headers.get('X-GitHub-Request-Id') or ''
) or None
body_material = getattr(response, '_truf_diagnostic_body_material', None)
if body_material is not None:
self.body = diagnostic_material_bytes(body_material)
elif capture_body and (
not getattr(response, 'raw', None)
or getattr(response, '_content_consumed', False)
):
body = getattr(response, 'content', b'')
body = body if isinstance(body, bytes) else str(body).encode('utf-8')
body_material = make_body_material(body)
self.body = diagnostic_material_bytes(body_material)
if body_material is not None:
self.body_original_size = body_material.original_size
self.body_stored_size = body_material.stored_size
self.body_sha256 = body_material.sha256
self.body_capture_truncated = body_material.truncated
class GitLabDiscoveryTransportError(ApiRequestError):
pass
class DockerHubDiscoveryTransportError(ApiRequestError):
def __init__(
self, message, *, category='page_unavailable', retry_at=None,
remote_attempted=True, retryable=True,
):
super().__init__(message)
self.category = str(category or 'page_unavailable')
self.retry_at = retry_at
self.remote_attempted = bool(remote_attempted)
self.retryable = bool(retryable)
API_RETRY_STATUSES = {408, 500, 502, 503, 504}
API_RETRY_EXCEPTIONS = (
requests.exceptions.ProxyError,
requests.exceptions.ConnectionError,
requests.exceptions.ConnectTimeout,
requests.exceptions.ReadTimeout,
requests.exceptions.Timeout,
requests.exceptions.SSLError,
requests.exceptions.ChunkedEncodingError,
)
_api_proxy_lock = threading.Lock()
_api_proxy_cache_path = None
_api_proxy_cache_mtime = None
_api_proxy_cache_entries = []
_api_proxy_cache_index = 0
def _redacted_proxy_url(proxy_url):
try:
parsed = urlsplit(proxy_url)
if '@' not in parsed.netloc:
return proxy_url
host = parsed.hostname or ''
port = f':{parsed.port}' if parsed.port else ''
return urlunsplit((parsed.scheme, f'***:***@{host}{port}', parsed.path, parsed.query, parsed.fragment))
except Exception:
return '<proxy>'
def parse_proxy_line(line):
line = str(line or '').strip()
if not line or line.startswith('#'):
return None
if '://' in line:
proxy_url = line
else:
parts = line.split(':', 3)
if len(parts) == 2:
host, port = parts
proxy_url = f'http://{host}:{port}'
elif len(parts) == 4:
host, port, username, password = parts
credentials = f'{quote(username, safe="")}:{quote(password, safe="")}'
proxy_url = f'http://{credentials}@{host}:{port}'
else:
raise ValueError('expected host:port or host:port:username:password')
return {'http': proxy_url, 'https': proxy_url}
def load_proxy_entries(proxy_file):
entries = []
if not proxy_file or not os.path.exists(proxy_file):
return entries
with open(proxy_file, 'r', encoding='utf-8') as f:
for line_number, line in enumerate(f, 1):
try:
proxy = parse_proxy_line(line)
except ValueError as e:
logger.warning(f'Ignoring bad proxy line {proxy_file}:{line_number}: {e}')
continue
if proxy:
entries.append(proxy)
return entries
def next_api_proxy():
global _api_proxy_cache_path, _api_proxy_cache_mtime, _api_proxy_cache_entries, _api_proxy_cache_index
if not scan_config.api_proxy_enabled:
return None
proxy_file = scan_config.api_proxy_file or scan_config.proxy_file
try:
mtime = os.path.getmtime(proxy_file) if proxy_file else None
except OSError:
mtime = None
with _api_proxy_lock:
if proxy_file != _api_proxy_cache_path or mtime != _api_proxy_cache_mtime:
_api_proxy_cache_path = proxy_file
_api_proxy_cache_mtime = mtime
_api_proxy_cache_entries = load_proxy_entries(proxy_file)
_api_proxy_cache_index = 0
if _api_proxy_cache_entries:
logger.info(f'Loaded {len(_api_proxy_cache_entries)} API proxy entry(ies) from {proxy_file}')
if not _api_proxy_cache_entries:
raise ApiRequestError(f'API proxy is enabled but no valid proxies are loaded from {proxy_file}')
proxy = _api_proxy_cache_entries[_api_proxy_cache_index % len(_api_proxy_cache_entries)]
_api_proxy_cache_index += 1
return proxy
def _short_url(url):
try:
parsed = urlsplit(url)
return urlunsplit((parsed.scheme, parsed.netloc, parsed.path, '', ''))
except Exception:
return str(url)
def _log_api_retry(method, url, attempt, attempts, error, proxy):
if attempt != 1 and attempt % 10 != 0 and attempt != attempts:
return
proxy_url = None
if proxy:
proxy_url = proxy.get('https') or proxy.get('http')
proxy_part = f' via {_redacted_proxy_url(proxy_url)}' if proxy_url else ''
logger.warning(f'API {method} {_short_url(url)} failed ({attempt}/{attempts}){proxy_part}: {str(error)[:300]}')
def _direct_request(method, url, **kwargs):
# None removes merged proxy routes; no_proxy also blocks environment rebuilds
# on redirects. Keep Requests' certificate and streamed-response behavior.
kwargs['proxies'] = {'http': None, 'https': None, 'all': None, 'no_proxy': '*'}
return requests.request(method, url, **kwargs)
def api_request(
method, url, *, timeout=None, max_retries=None, retry_delay=None,
retry_statuses=None, deadline=None, use_proxy=None, **kwargs,
):
# False opts out; other values retain the configured discovery policy.
use_proxy = use_proxy is not False and scan_config.api_proxy_enabled
attempts = max(1, int(max_retries if max_retries is not None else (scan_config.api_proxy_max_retries if use_proxy else 1)))
delay = max(0, int(retry_delay if retry_delay is not None else scan_config.api_proxy_retry_delay))
retry_statuses = set(API_RETRY_STATUSES if retry_statuses is None else retry_statuses)
request_timeout = timeout
if use_proxy and scan_config.api_proxy_timeout is not None:
if isinstance(timeout, (tuple, list)):
connect_timeout, read_timeout = timeout
else:
connect_timeout = read_timeout = timeout
proxy_timeout = float(scan_config.api_proxy_timeout)
# Proxy connection limits must not replace the caller's read budget.
request_timeout = (
proxy_timeout if connect_timeout is None else min(proxy_timeout, float(connect_timeout)),
proxy_timeout if timeout is None else read_timeout,
)
last_error = None
for attempt in range(1, attempts + 1):
_raise_if_scan_slot_fatal()
remaining = None if deadline is None else float(deadline) - time.monotonic()
if remaining is not None and remaining <= 0:
raise ApiRequestError(f'API request deadline expired before attempt {attempt}: {method} {_short_url(url)}')
proxy = next_api_proxy() if use_proxy else None
request_kwargs = dict(kwargs)
if proxy:
request_kwargs['proxies'] = proxy
effective_timeout = request_timeout
if remaining is not None and effective_timeout is None:
effective_timeout = max(0.001, remaining)
elif remaining is not None and isinstance(effective_timeout, (int, float)):
effective_timeout = max(0.001, min(float(effective_timeout), remaining))
elif remaining is not None and isinstance(effective_timeout, (tuple, list)):
effective_timeout = tuple(
max(0.001, remaining if value is None else min(float(value), remaining))
for value in effective_timeout
)
try:
request = requests.request if use_proxy else _direct_request
response = request(method, url, timeout=effective_timeout, **request_kwargs)
if _scan_slot_fatal_event.is_set():
response.close()
_raise_if_scan_slot_fatal()
if deadline is not None and time.monotonic() >= float(deadline):
response.close()
raise ApiRequestError(f'API request deadline expired after response: {method} {_short_url(url)}')
if response.status_code in retry_statuses:
detail = ''
if not request_kwargs.get('stream'):
detail = response.text[:300] if response.text else ''
last_error = f'HTTP {response.status_code}' + (f': {detail}' if detail else '')
if attempt >= attempts:
failure = ApiRequestError(
f'API request failed after {attempts} attempt(s): '
f'{method} {_short_url(url)}: {last_error}',
response=response,
operation='provider-api-request',
capture_body=not bool(request_kwargs.get('stream')),
)
response.close()
raise failure
_log_api_retry(method, url, attempt, attempts, last_error, proxy)
response.close()
if deadline is not None and time.monotonic() + delay >= float(deadline):
raise ApiRequestError(f'API request deadline expired during retry: {method} {_short_url(url)}')
_wait_or_raise_scan_slot_fatal(delay)
continue
return response
except API_RETRY_EXCEPTIONS as e:
try:
url_has_query = bool(urlsplit(str(url)).query)
except ValueError:
url_has_query = True
safe_error = (
type(e).__name__
if request_kwargs.get('stream') or url_has_query
else str(e)[:300]
)
last_error = safe_error
_log_api_retry(method, url, attempt, attempts, safe_error, proxy)
if attempt >= attempts:
raise ApiRequestError(
f'API request failed after {attempts} attempt(s): '
f'{method} {_short_url(url)}: {safe_error}'
) from e
if deadline is not None and time.monotonic() + delay >= float(deadline):
raise ApiRequestError(f'API request deadline expired during retry: {method} {_short_url(url)}') from e
_wait_or_raise_scan_slot_fatal(delay)
raise ApiRequestError(f'API request failed after {attempts} attempt(s): {method} {_short_url(url)}: {last_error}')
SCAN_SLOT_SCHEMA = """
CREATE TABLE IF NOT EXISTS scan_slots (
slot_id TEXT PRIMARY KEY,
owner_pid INTEGER NOT NULL,
owner_thread INTEGER NOT NULL,
owner_source TEXT,
owner_creation_time TEXT,
owner_executable TEXT,
child_pid INTEGER,
child_creation_time TEXT,
child_executable TEXT,
slot_kind TEXT NOT NULL DEFAULT 'base' CHECK(slot_kind IN ('base', 'bonus')),
command TEXT,
acquired_at REAL NOT NULL,
updated_at REAL NOT NULL
);
CREATE TABLE IF NOT EXISTS scan_waiters (
waiter_id TEXT PRIMARY KEY,
owner_pid INTEGER NOT NULL,
owner_thread INTEGER NOT NULL,
owner_source TEXT NOT NULL,
owner_creation_time TEXT NOT NULL,
owner_executable TEXT NOT NULL,
enqueued_at REAL NOT NULL
);
CREATE INDEX IF NOT EXISTS idx_scan_waiters_fair
ON scan_waiters(owner_source, enqueued_at, waiter_id);
CREATE TABLE IF NOT EXISTS scan_source_fairness (
owner_source TEXT PRIMARY KEY,
last_granted_at REAL NOT NULL
);
"""
_scan_limiter_init_lock = threading.Lock()
_scan_limiter_initialized_paths = set()
_scan_slot_scope_local = threading.local()
_scan_slot_fatal_event = threading.Event()
_scan_slot_fatal_lock = threading.Lock()
_scan_slot_fatal_detail = None
class _ScanSlotScope:
def __init__(self, lease):
self.lease = lease
class ScanSlotFatalError(RuntimeError):
pass
def _set_scan_slot_fatal(detail):
global _scan_slot_fatal_detail
bounded = str(detail or 'scan-slot durability failure')[:1000]
with _scan_slot_fatal_lock:
if not _scan_slot_fatal_event.is_set():
_scan_slot_fatal_detail = bounded
_scan_slot_fatal_event.set()
def _raise_if_scan_slot_fatal():
if not _scan_slot_fatal_event.is_set():
return
with _scan_slot_fatal_lock:
detail = _scan_slot_fatal_detail
raise ScanSlotFatalError(detail or 'FATAL: scan-slot limiter is in an indeterminate state')
def _wait_or_raise_scan_slot_fatal(delay):
remaining = max(0.0, float(delay or 0))
while remaining > 0:
_raise_if_scan_slot_fatal()
interval = min(0.2, remaining)
time.sleep(interval)
remaining -= interval
_raise_if_scan_slot_fatal()
class ScanSlotLease:
DB_RETRY_ATTEMPTS = 5
RELEASE_PENDING_DB_ATTEMPTS = 2
DB_RETRY_DELAY_SEC = 0.1
MIN_HEARTBEAT_INTERVAL_SEC = 5.0
RELEASE_PENDING_INTERVAL_SEC = 1.0
HEARTBEAT_JOIN_TIMEOUT_SEC = 2.0
def __init__(self, slot_id, db_path, owner_pid=None, owner_thread=None):
self.slot_id = slot_id
self.db_path = db_path
self.owner_pid = int(os.getpid() if owner_pid is None else owner_pid)
self.owner_thread = int(threading.get_ident() if owner_thread is None else owner_thread)
self.child_pid = None
self.heartbeat_stop = threading.Event()
self.heartbeat_thread = None
self._heartbeat_wake = threading.Event()
self._state_lock = threading.Lock()
self._db_lock = threading.Lock()
self._release_call_lock = threading.Lock()
self._released = False
self._releasable = True
self._release_requested = False
self._release_pending = False
self._heartbeat_interval = 30.0
self._release_pending_interval = self.RELEASE_PENDING_INTERVAL_SEC
self._last_release_error_log_at = 0.0
@property
def released(self):
with self._state_lock:
return self._released
@property
def releasable(self):
with self._state_lock:
return self._releasable
@property
def release_pending(self):
with self._state_lock:
return self._release_pending
def _connect(self):
return sqlite3.connect(self.db_path, timeout=30)
def _close_connection(self, conn):
try:
conn.close()
except Exception as exc:
logger.warning('Unable to close scan-slot DB connection for %s: %s', self.slot_id, exc)
def _execute_update_once(self, sql, params):
conn = None
try:
conn = self._connect()
conn.execute('PRAGMA busy_timeout=30000')
cursor = conn.execute(sql, params)
conn.commit()
return cursor.rowcount != 0
finally:
if conn is not None:
self._close_connection(conn)
def _delete_slot_once(self):
conn = None
try:
conn = self._connect()
conn.execute('PRAGMA busy_timeout=30000')
cursor = conn.execute(
'''DELETE FROM scan_slots
WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''',
(self.slot_id, self.owner_pid, self.owner_thread),
)
if cursor.rowcount != 1:
existing = conn.execute(
'SELECT owner_pid, owner_thread FROM scan_slots WHERE slot_id = ?',
(self.slot_id,),
).fetchone()
if existing is not None:
raise RuntimeError(
f'scan slot identity changed from pid/thread '
f'{self.owner_pid}/{self.owner_thread} to {existing[0]}/{existing[1]}'
)
conn.commit()
return True
finally:
if conn is not None:
self._close_connection(conn)
def _retry_db_operation(self, operation, attempts):
attempts = max(1, int(attempts))
last_error = None
for attempt in range(1, attempts + 1):
try:
with self._db_lock:
return bool(operation()), None
except (sqlite3.Error, OSError) as exc:
last_error = exc
except Exception as exc:
last_error = exc
break
if attempt < attempts:
time.sleep(max(0.0, float(self.DB_RETRY_DELAY_SEC)) * attempt)
return False, last_error
def update(self, sql, params, attempts=None, log_failure=True):
success, last_error = self._retry_db_operation(
lambda: self._execute_update_once(sql, params),
self.DB_RETRY_ATTEMPTS if attempts is None else attempts,
)
if last_error is not None and log_failure:
logger.warning('Unable to update scan slot %s: %s', self.slot_id, last_error)
return success
def set_child_pid(self, child_pid):
if not child_pid:
return False
child_pid = int(child_pid)
with self._release_call_lock:
with self._state_lock:
if (
self._released or not self._releasable or self._release_requested
or not self.slot_id or not self.db_path
):
return False
identity = capture_process_identity(child_pid)
updated = self.update(
'''UPDATE scan_slots SET child_pid = ?, child_creation_time = ?, child_executable = ?, updated_at = ?
WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''',
(
child_pid,
identity.get('creation_time') if identity else None,
identity.get('executable') if identity else None,
time.time(),
self.slot_id,
self.owner_pid,
self.owner_thread,
),
)
if updated:
with self._state_lock:
self.child_pid = child_pid
return updated
def mark_non_releasable(self):
with self._release_call_lock:
with self._state_lock:
if self._released:
return
self._releasable = False
self._release_pending = False
self._heartbeat_wake.set()
def _heartbeat_update(self, attempts=None, log_failure=True):
return self.update(
'''UPDATE scan_slots SET updated_at = ?
WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''',
(time.time(), self.slot_id, self.owner_pid, self.owner_thread),
attempts=attempts,
log_failure=log_failure,
)
def start_heartbeat(self):
interval = max(
float(self.MIN_HEARTBEAT_INTERVAL_SEC),
float_setting(getattr(scan_config, 'scan_slot_heartbeat_sec', 30), 30),
)
release_interval = max(0.01, min(interval, float(self.RELEASE_PENDING_INTERVAL_SEC)))
with self._state_lock:
if self._released:
return False
if self.heartbeat_thread is not None and self.heartbeat_thread.is_alive():
return True
self._heartbeat_interval = interval
self._release_pending_interval = release_interval
thread = threading.Thread(
target=self._heartbeat_loop,
name=f'scan-slot-heartbeat-{self.slot_id[:8]}',
daemon=True,
)
self.heartbeat_thread = thread
try:
thread.start()
except Exception:
self.heartbeat_thread = None
raise
return True
def transfer_to_current_thread(self):
new_thread = int(threading.get_ident())
with self._release_call_lock:
with self._state_lock:
if self._released or self._release_requested or not self._releasable:
raise RuntimeError(f'scan slot {self.slot_id} cannot be transferred after release started')
if self.heartbeat_thread is not None:
raise RuntimeError(f'scan slot {self.slot_id} cannot be transferred after heartbeat start')
old_thread = self.owner_thread
if old_thread == new_thread:
return True
updated = self.update(
'''UPDATE scan_slots SET owner_thread = ?, updated_at = ?
WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''',
(new_thread, time.time(), self.slot_id, self.owner_pid, old_thread),
)
if not updated:
raise RuntimeError(f'scan slot {self.slot_id} ownership transfer was not confirmed')
with self._state_lock:
self.owner_thread = new_thread
return True
def _heartbeat_loop(self):
while True:
with self._state_lock:
if self._released:
return
pending = self._release_pending and self._releasable
interval = self._release_pending_interval if pending else self._heartbeat_interval
self._heartbeat_wake.wait(interval)
self._heartbeat_wake.clear()
with self._state_lock:
if self._released or self.heartbeat_stop.is_set():
return
pending = self._release_pending and self._releasable
if not pending:
self._heartbeat_update()
continue
with self._release_call_lock:
with self._state_lock:
pending = self._release_pending and self._releasable and not self._released
if not pending:
continue
success, last_error = self._retry_db_operation(
self._delete_slot_once,
self.RELEASE_PENDING_DB_ATTEMPTS,
)
if success:
completed = self._complete_release()
if completed:
logger.info('Released scan slot %s after background retry', self.slot_id)
return
self._log_pending_release_failure(last_error)
# Keep the exact owner row fresh when DELETE itself is temporarily unavailable.
self._heartbeat_update(attempts=1, log_failure=False)
def _log_pending_release_failure(self, error):
now = time.monotonic()
with self._state_lock:
if now - self._last_release_error_log_at < 30:
return
self._last_release_error_log_at = now
logger.error('Scan slot %s release remains pending and fail-closed: %s', self.slot_id, error)
def _complete_release(self):
with self._state_lock:
if self._released:
return False
self._released = True
self._release_pending = False
self.heartbeat_stop.set()
self._heartbeat_wake.set()
return True
def _join_heartbeat(self):
with self._state_lock:
thread = self.heartbeat_thread
if thread is None or thread is threading.current_thread() or not thread.is_alive():
return
thread.join(timeout=max(0.0, float(self.HEARTBEAT_JOIN_TIMEOUT_SEC)))
if thread.is_alive():
logger.warning('Scan-slot heartbeat did not stop promptly for %s', self.slot_id)
def release(self):
should_join = False
success = False
with self._release_call_lock:
with self._state_lock:
if self._released:
success = True
should_join = True
elif (
not self._releasable or self._release_requested
or not self.slot_id or not self.db_path
):
return
else:
self._release_requested = True
if not success:
success, last_error = self._retry_db_operation(
self._delete_slot_once,
self.DB_RETRY_ATTEMPTS,
)
if success:
self._complete_release()
should_join = True
else:
with self._state_lock:
pending = not self._released and self._releasable
if pending:
self._release_pending = True
self._last_release_error_log_at = time.monotonic()
if pending:
try:
self.start_heartbeat()
except Exception as exc:
logger.error('Unable to start pending-release heartbeat for scan slot %s: %s', self.slot_id, exc)
self._heartbeat_wake.set()
logger.error(
'Unable to release scan slot %s after %d attempts; '
'lease remains live and background retries will continue: %s',
self.slot_id, self.DB_RETRY_ATTEMPTS, last_error,
)
if should_join:
self._join_heartbeat()
def process_exists(pid):
try:
pid = int(pid)
except (TypeError, ValueError):
return False
if pid <= 0:
return False
if pid == os.getpid():
return True
if os.name == 'nt':
try:
import ctypes
process_query_limited_information = 0x1000
handle = ctypes.windll.kernel32.OpenProcess(process_query_limited_information, False, pid)
if handle:
ctypes.windll.kernel32.CloseHandle(handle)
return True
return False
except Exception:
return True
try:
os.kill(pid, 0)
return True
except ProcessLookupError:
return False
except PermissionError:
return True
except Exception:
return True
def capture_process_identity(pid):
try:
process = open_process(int(pid))
except Exception:
return None
try:
return {
'creation_time': str(process.identity.creation_time),
'executable': canonical_path(process.identity.executable),
}
finally:
process.close()
def exact_process_identity_live(pid, creation_time, executable):
if not pid:
return False
if not creation_time or not executable:
return None if process_exists(pid) else False
try:
process = open_process(int(pid))
except Exception:
return None if process_exists(pid) else False
try:
return bool(
process.is_running()
and str(process.identity.creation_time) == str(creation_time)
and canonical_path(process.identity.executable) == canonical_path(executable)
)
finally:
process.close()
def scan_limiter_enabled():
return int_setting(getattr(scan_config, 'max_active_scans', 0), 0) > 0
def scan_limiter_db_path():
path = getattr(scan_config, 'scan_limiter_db', '') or ''
if not path:
defaults = default_project_paths()
path = os.path.join(defaults['state_dir'], 'scan_limiter.db')
return path
def ensure_scan_limiter_db(path):
parent = os.path.dirname(path)
if parent:
os.makedirs(parent, exist_ok=True)
with _scan_limiter_init_lock:
if path in _scan_limiter_initialized_paths:
return
conn = sqlite3.connect(path, timeout=30)
try:
conn.execute('PRAGMA busy_timeout=30000')
conn.execute('PRAGMA journal_mode=WAL')
conn.executescript(SCAN_SLOT_SCHEMA)
conn.execute('BEGIN IMMEDIATE')
existing = {row[1] for row in conn.execute('PRAGMA table_info(scan_slots)').fetchall()}
if 'child_pid' not in existing:
conn.execute('ALTER TABLE scan_slots ADD COLUMN child_pid INTEGER')
for name, declaration in (
('owner_creation_time', 'TEXT'),
('owner_executable', 'TEXT'),
('child_creation_time', 'TEXT'),
('child_executable', 'TEXT'),
):
if name not in existing:
conn.execute(f'ALTER TABLE scan_slots ADD COLUMN {name} {declaration}')
if 'slot_kind' not in existing:
conn.execute(
"ALTER TABLE scan_slots ADD COLUMN slot_kind TEXT NOT NULL DEFAULT 'base'"
)
conn.execute(
"CREATE UNIQUE INDEX IF NOT EXISTS idx_scan_slots_single_bonus "
"ON scan_slots(slot_kind) WHERE slot_kind = 'bonus'"
)
conn.commit()
finally:
conn.close()
_scan_limiter_initialized_paths.add(path)
def connect_scan_limiter_db(path):
ensure_scan_limiter_db(path)
conn = sqlite3.connect(path, timeout=30)
conn.execute('PRAGMA busy_timeout=30000')
return conn
def cleanup_stale_scan_slots(conn, now, stale_sec):
columns = {row[1] for row in conn.execute('PRAGMA table_info(scan_slots)').fetchall()}
child_expr = 'child_pid' if 'child_pid' in columns else 'NULL AS child_pid'
owner_creation_expr = 'owner_creation_time' if 'owner_creation_time' in columns else 'NULL AS owner_creation_time'
owner_executable_expr = 'owner_executable' if 'owner_executable' in columns else 'NULL AS owner_executable'
child_creation_expr = 'child_creation_time' if 'child_creation_time' in columns else 'NULL AS child_creation_time'
child_executable_expr = 'child_executable' if 'child_executable' in columns else 'NULL AS child_executable'
rows = conn.execute(
f'''SELECT slot_id, owner_pid, {owner_creation_expr}, {owner_executable_expr},
{child_expr}, {child_creation_expr}, {child_executable_expr}, acquired_at, updated_at
FROM scan_slots'''
).fetchall()
for slot_id, owner_pid, owner_creation, owner_executable, child_pid, child_creation, child_executable, acquired_at, updated_at in rows:
heartbeat_age = now - float(updated_at or acquired_at or 0)
owner_live = exact_process_identity_live(owner_pid, owner_creation, owner_executable)
child_live = exact_process_identity_live(child_pid, child_creation, child_executable)
if owner_live is False and child_live is False:
conn.execute('DELETE FROM scan_slots WHERE slot_id = ?', (slot_id,))
elif heartbeat_age > stale_sec:
logger.warning(
'Stale scan-slot heartbeat remains capacity-blocking: slot=%s owner_live=%s child_live=%s age=%.0fs',
slot_id, owner_live, child_live, heartbeat_age,
)
if 'scan_waiters' in {
row[0] for row in conn.execute("SELECT name FROM sqlite_master WHERE type = 'table'").fetchall()
}:
waiters = conn.execute(
'''SELECT waiter_id, owner_pid, owner_creation_time, owner_executable
FROM scan_waiters'''
).fetchall()
for waiter_id, owner_pid, owner_creation, owner_executable in waiters:
if exact_process_identity_live(owner_pid, owner_creation, owner_executable) is False:
conn.execute('DELETE FROM scan_waiters WHERE waiter_id = ?', (waiter_id,))
def redact_scan_command_text(cmd):
text = ' '.join(str(part) for part in cmd)
text = re.sub(r'(https?://)[^\s/@:]+:[^\s/@]+@', r'\1***:***@', text)
text = re.sub(r'github_pat_[A-Za-z0-9_]+', 'github_pat_***', text)
text = re.sub(r'gh[pousr]_[A-Za-z0-9_]+', 'ghp_***', text)
text = re.sub(r'hf_[A-Za-z0-9]+', 'hf_***', text)
return text[:1000]
def windows_scan_capacity_snapshot():
if os.name != 'nt':
raise OSError('opportunistic scan capacity is supported only on Windows')
from ctypes import wintypes
class PerformanceInformation(ctypes.Structure):
_fields_ = [
('cb', wintypes.DWORD),
('CommitTotal', ctypes.c_size_t),
('CommitLimit', ctypes.c_size_t),
('CommitPeak', ctypes.c_size_t),
('PhysicalTotal', ctypes.c_size_t),
('PhysicalAvailable', ctypes.c_size_t),
('SystemCache', ctypes.c_size_t),
('KernelTotal', ctypes.c_size_t),
('KernelPaged', ctypes.c_size_t),
('KernelNonpaged', ctypes.c_size_t),
('PageSize', ctypes.c_size_t),
('HandleCount', wintypes.DWORD),
('ProcessCount', wintypes.DWORD),
('ThreadCount', wintypes.DWORD),
]
get_performance_info = ctypes.WinDLL('psapi', use_last_error=True).GetPerformanceInfo
get_performance_info.argtypes = [ctypes.POINTER(PerformanceInformation), wintypes.DWORD]
get_performance_info.restype = wintypes.BOOL
info = PerformanceInformation()
info.cb = ctypes.sizeof(info)
if not get_performance_info(ctypes.byref(info), info.cb):
raise ctypes.WinError(ctypes.get_last_error())
page_size = int(info.PageSize)
if page_size <= 0 or int(info.CommitLimit) < int(info.CommitTotal):
raise OSError('Windows returned invalid scan-capacity counters')
return {
'available_physical_bytes': int(info.PhysicalAvailable) * page_size,
'commit_headroom_bytes': (int(info.CommitLimit) - int(info.CommitTotal)) * page_size,
}
def opportunistic_scan_slot_allowed(source):
if max(0, min(1, int_setting(getattr(scan_config, 'opportunistic_scan_slots', 0), 0))) <= 0:
return False
eligible = {
item.lower() for item in csv_items(
getattr(scan_config, 'opportunistic_scan_sources', [])
)
}
if str(source or '').lower() not in eligible:
return False
try:
job_limit = int(getattr(scan_config, 'trufflehog_job_memory_limit_bytes', 0))
overhead = max(0, int(getattr(
scan_config, 'opportunistic_scan_reserve_overhead_bytes', 0,
)))
reserve = job_limit + overhead
if job_limit <= 0 or reserve <= 0:
return False
capacity = windows_scan_capacity_snapshot()
available_after = int(capacity['available_physical_bytes']) - reserve
commit_after = int(capacity['commit_headroom_bytes']) - reserve
minimum_available = max(0, int(getattr(
scan_config, 'opportunistic_scan_min_available_after_reserve_bytes', 0,
)))
minimum_commit = max(0, int(getattr(
scan_config, 'opportunistic_scan_min_commit_after_reserve_bytes', 0,
)))
return available_after >= minimum_available and commit_after >= minimum_commit
except (OSError, TypeError, ValueError, OverflowError, KeyError):
return False
def acquire_scan_slot(cmd, timeout_sec=None, wait=True, start_heartbeat=True):
_raise_if_scan_slot_fatal()
max_active = int_setting(getattr(scan_config, 'max_active_scans', 0), 0)
if max_active <= 0:
return None
db_path = scan_limiter_db_path()
wait_sec = max(0.1, float_setting(getattr(scan_config, 'scan_slot_wait_sec', 0.5), 0.5))
wait_log_sec = max(1, int_setting(getattr(scan_config, 'scan_slot_wait_log_sec', 30), 30))
stale_sec = max(
int_setting(getattr(scan_config, 'scan_slot_stale_sec', 7200), 7200),
int(timeout_sec or 0) + 300,
)
source = os.getenv('SCANNER_SOURCE') or (cmd[1] if len(cmd) > 1 else 'unknown')
command_text = redact_scan_command_text(cmd)
owner_pid = os.getpid()
owner_thread = threading.get_ident()
slot_id = f'{owner_pid}-{owner_thread}-{uuid.uuid4().hex}'
waiter_id = f'wait-{owner_pid}-{owner_thread}-{uuid.uuid4().hex}'
owner_identity = current_process_identity()
started_waiting = time.monotonic()
enqueued_at = time.time()
last_log_at = 0.0
waiter_registered = False
while True:
_raise_if_scan_slot_fatal()
now = time.time()
conn = None
slot_committed = False
try:
conn = connect_scan_limiter_db(db_path)
conn.execute('BEGIN IMMEDIATE')
_raise_if_scan_slot_fatal()
cleanup_stale_scan_slots(conn, now, stale_sec)
conn.execute(
'''INSERT OR IGNORE INTO scan_waiters(
waiter_id, owner_pid, owner_thread, owner_source,
owner_creation_time, owner_executable, enqueued_at
) VALUES (?, ?, ?, ?, ?, ?, ?)''',
(
waiter_id, owner_pid, owner_thread, str(source),
owner_identity.creation_time, canonical_path(owner_identity.executable),
enqueued_at,
),
)
waiter_registered = True
base_active, bonus_active = conn.execute(
"SELECT "
"SUM(CASE WHEN slot_kind = 'base' THEN 1 ELSE 0 END), "
"SUM(CASE WHEN slot_kind = 'bonus' THEN 1 ELSE 0 END) "
"FROM scan_slots"
).fetchone()
base_active = int(base_active or 0)
bonus_active = int(bonus_active or 0)
active = base_active + bonus_active
next_waiter = conn.execute(
'''SELECT w.waiter_id
FROM scan_waiters w
LEFT JOIN scan_source_fairness f ON f.owner_source = w.owner_source
ORDER BY COALESCE(f.last_granted_at, 0), w.enqueued_at, w.waiter_id
LIMIT 1'''
).fetchone()
slot_kind = None
if next_waiter and next_waiter[0] == waiter_id:
if base_active < max_active:
slot_kind = 'base'
elif bonus_active < max(0, min(1, int_setting(
getattr(scan_config, 'opportunistic_scan_slots', 0), 0,
))) and opportunistic_scan_slot_allowed(source):
slot_kind = 'bonus'
if slot_kind is not None:
conn.execute(
'''INSERT INTO scan_slots(
slot_id, owner_pid, owner_thread, owner_source, owner_creation_time,
owner_executable, slot_kind, command, acquired_at, updated_at
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''',
(
slot_id, owner_pid, owner_thread, str(source),
owner_identity.creation_time, canonical_path(owner_identity.executable),
slot_kind, command_text, now, now,
),
)
conn.execute('DELETE FROM scan_waiters WHERE waiter_id = ?', (waiter_id,))
conn.execute(
'''INSERT INTO scan_source_fairness(owner_source, last_granted_at)
VALUES (?, ?)
ON CONFLICT(owner_source) DO UPDATE SET last_granted_at = excluded.last_granted_at''',
(str(source), now),
)
conn.commit()
slot_committed = True
waiter_registered = False
waited = time.monotonic() - started_waiting
if waited >= wait_log_sec:
hard_limit = max_active + max(0, min(1, int_setting(
getattr(scan_config, 'opportunistic_scan_slots', 0), 0,
)))
logger.info(
f'Acquired {slot_kind} scan slot after waiting {waited:.0f}s '
f'({active + 1}/{hard_limit})'
)
lease = ScanSlotLease(slot_id, db_path, owner_pid=owner_pid, owner_thread=owner_thread)
if start_heartbeat:
try:
if not lease.start_heartbeat():
raise RuntimeError('scan-slot heartbeat did not start')
except BaseException as start_error:
logger.error(
'Scan-slot heartbeat failed to start for %s; synchronously removing the exact owner row: %s',
slot_id, start_error,
)
lease.release()
if not lease.released:
fatal_detail = (
'FATAL: scan-slot heartbeat startup failed and exact-owner rollback '
f'could not be confirmed for slot {slot_id}'
)
_set_scan_slot_fatal(fatal_detail)
logger.critical(
'FATAL scan-slot acquisition rollback is unconfirmed for %s; capacity remains fail-closed',
slot_id,
)
raise ScanSlotFatalError(fatal_detail) from start_error
raise
return lease
if not wait:
conn.execute('DELETE FROM scan_waiters WHERE waiter_id = ?', (waiter_id,))
conn.commit()
waiter_registered = False
return None
conn.commit()
if time.monotonic() - last_log_at >= wait_log_sec:
logger.info(f'Waiting for scan slot ({active}/{max_active} active)')
last_log_at = time.monotonic()
except sqlite3.OperationalError as e:
if slot_committed:
raise
if not wait:
return None
if time.monotonic() - last_log_at >= wait_log_sec:
logger.warning(f'Waiting for scan limiter DB lock: {str(e)}')
last_log_at = time.monotonic()
finally:
if conn is not None:
conn.close()
if _scan_slot_fatal_event.wait(wait_sec):
_raise_if_scan_slot_fatal()
def acquire_scan_slot_leases(cmd, count, timeout_sec=None):
count = max(0, int(count or 0))
if count <= 0 or not scan_limiter_enabled():
return []
leases = []
try:
first = acquire_scan_slot(cmd, timeout_sec, wait=True, start_heartbeat=False)
if first is not None:
leases.append(first)
while len(leases) < count:
lease = acquire_scan_slot(cmd, timeout_sec, wait=False, start_heartbeat=False)
if lease is None:
break
leases.append(lease)
return leases
except BaseException:
for lease in leases:
lease.release()
raise
@contextmanager
def scan_slot_scope(cmd, timeout_sec=None, lease=None):
"""Own one physical lease through the target's durable bundle handoff."""
if getattr(_scan_slot_scope_local, 'scope', None) is not None:
raise RuntimeError('scan slot scopes cannot be nested on one worker thread')
if lease is None:
lease = acquire_scan_slot(cmd, timeout_sec)
else:
try:
lease.transfer_to_current_thread()
if not lease.start_heartbeat():
raise RuntimeError('transferred scan-slot heartbeat did not start')
except BaseException:
lease.release()
raise
scope = _ScanSlotScope(lease)
_scan_slot_scope_local.scope = scope
try:
yield lease
finally:
try:
if lease and lease.releasable:
lease.release()
finally:
if getattr(_scan_slot_scope_local, 'scope', None) is scope:
del _scan_slot_scope_local.scope
def scoped_scan_slot_lease():
scope = getattr(_scan_slot_scope_local, 'scope', None)
return (scope is not None, scope.lease if scope is not None else None)
def get_pending_temp_file():
work_dir = get_work_dir()
if not work_dir:
return None
return os.path.join(work_dir, 'pending_cleanup.json')
def get_pending_temp_lock_file():
path = get_pending_temp_file()
return path + '.lock' if path else None
def _load_persisted_pending_temp_dirs_unlocked(path):
if not path or not os.path.exists(path):
return set()
try:
value = read_private_json(path)
except OSError:
return set()
if value.get('schema') != PENDING_TEMP_SCHEMA or not isinstance(value.get('paths'), list):
return set()
return {str(item) for item in value['paths'] if isinstance(item, str) and item.strip()}
def _persist_pending_temp_dirs_unlocked(path, paths):
normalized = sorted({str(item) for item in paths if str(item).strip()})
if normalized:
atomic_write_private_json(path, {'schema': PENDING_TEMP_SCHEMA, 'paths': normalized})
elif os.path.exists(path):
if not private_file_ready(path):
raise OSError(f'refusing to remove non-private pending cleanup list: {path}')
durable_unlink(path)
def load_persisted_pending_temp_dirs():
path = get_pending_temp_file()
lock_path = get_pending_temp_lock_file()
if not path or not lock_path:
return set()
with PrivateFileLock(lock_path):
return _load_persisted_pending_temp_dirs_unlocked(path)
def persist_pending_temp_dirs(paths):
path = get_pending_temp_file()
lock_path = get_pending_temp_lock_file()
if not path or not lock_path:
return
try:
with PrivateFileLock(lock_path):
_persist_pending_temp_dirs_unlocked(path, paths)
except OSError:
pass
def redact_secrets(text, secrets):
if not text:
return text
redacted = text
for secret in secrets:
if secret:
redacted = redacted.replace(secret, '***REDACTED***')
redacted = redacted.replace(quote(secret, safe=''), '***REDACTED***')
return redacted
def build_authenticated_git_url(repo_url, provider=None, token=None):
if not token:
return repo_url, []
parsed = urlsplit(repo_url)
if parsed.scheme != 'https' or not parsed.netloc:
return repo_url, []
hostname = (parsed.hostname or '').lower()
detected_provider = 'gitlab' if hostname == 'gitlab.com' else 'github' if hostname == 'github.com' else None
if provider and provider != detected_provider:
return repo_url, []
provider = detected_provider
if provider not in ('github', 'gitlab'):
return repo_url, []
return repo_url, [token]
def get_git_provider_and_path(repo_url, provider=None):
parsed = urlsplit(repo_url)
hostname = (parsed.hostname or '').lower()
path = parsed.path.strip('/')
if path.endswith('.git'):
path = path[:-4]
provider = provider or ('gitlab' if 'gitlab.' in hostname or hostname == 'gitlab.com' else 'github' if 'github.' in hostname or hostname == 'github.com' else None)
return provider, path
def recent_commit_boundary(repo_url, provider=None, token=None, max_age_days=None, lookup_pages=3):
if not max_age_days or max_age_days <= 0:
return {'since_commit': None, 'skip': False, 'reason': ''}
provider, repo_path = get_git_provider_and_path(repo_url, provider)
if provider not in ('github', 'gitlab') or not repo_path:
return {'since_commit': None, 'skip': True, 'reason': 'unsupported provider for commit age lookup'}
cutoff = datetime.now(timezone.utc) - timedelta(days=max_age_days)
since = cutoff.isoformat().replace('+00:00', 'Z')
headers = {'User-Agent': 'GitSecretsScanner/2.0'}
if token:
headers['Authorization'] = f'Bearer {token}'
commits = []
lookup_cap_reached = False
for page in range(1, max(1, lookup_pages) + 1):
try:
if provider == 'github':
url = f'https://api.github.com/repos/{repo_path}/commits'
params = {'since': since, 'per_page': 100, 'page': page}
else:
url = f'https://gitlab.com/api/v4/projects/{quote(repo_path, safe="")}/repository/commits'
params = {'since': since, 'per_page': 100, 'page': page}
response = api_request('GET', url, headers=headers, params=params, timeout=30)
if token and response.status_code in (401, 403):
anonymous = api_request(
'GET', url, headers={'User-Agent': 'GitSecretsScanner/2.0'},
params=params, timeout=30,
)
if anonymous.status_code < 400:
response = anonymous
response.raise_for_status()
page_commits = response.json()
if not page_commits:
break
commits.extend(page_commits)
if len(page_commits) < 100:
break
if page == max(1, lookup_pages):
lookup_cap_reached = True
except requests.exceptions.HTTPError as e:
api_error = github_api_error(e.response) if provider == 'github' else gitlab_api_error(e.response)
if api_error.category == 'not_found':
return {
'since_commit': None,
'skip': True,
'permanent': True,
'reason': f'commit age lookup found no repository: {str(api_error)[:300]}',
'error_category': api_error.category,
'auth_related': False,
}
return {
'since_commit': None,
'skip': False,
'error': True,
'reason': f'commit age lookup failed ({api_error.category}): {str(api_error)[:300]}',
'error_category': api_error.category,
'auth_related': bool(getattr(api_error, 'auth_related', True)),
}
except Exception as e:
if isinstance(e, ApiRequestError):
return {
'since_commit': None, 'skip': False, 'error': True,
'reason': f'commit age lookup transport failed: {str(e)[:300]}',
'error_category': 'network', 'auth_related': False,
}
return {
'since_commit': None,
'skip': False,
'error': True,
'reason': f'commit age lookup failed: {str(e)[:300]}',
'error_category': 'unknown',
'auth_related': False,
}
if not commits:
return {
'since_commit': None,
'skip': True,
'reason': f'no commits newer than {max_age_days} days'
}
if lookup_cap_reached:
return {
'since_commit': None,
'skip': False,
'reason': 'commit lookup cap reached; scanning without since-commit boundary',
'recent_commit_count': len(commits),
'cutoff': since,
}
oldest = commits[-1]
if provider == 'github':
parents = oldest.get('parents') or []
parent_sha = parents[0].get('sha') if parents else None
oldest_sha = oldest.get('sha')
else:
parents = oldest.get('parent_ids') or []
parent_sha = parents[0] if parents else None
oldest_sha = oldest.get('id')
return {
'since_commit': parent_sha,
'skip': False,
'reason': '',
'recent_commit_count': len(commits),
'cutoff': since,
'boundary_commit': oldest_sha,
}
def get_trufflehog_cmd():
"""Return configured TruffleHog executable path."""
return scan_config.trufflehog_path or "trufflehog"
def require_trufflehog_launch_authority(command=None):
child_kind = str(os.getenv(CHILD_KIND_ENV) or 'scanner').strip().lower()
if child_kind not in {'scanner', 'docker-shadow'}:
raise RuntimeError('TruffleHog launch requires scanner or Docker shadow authority')
manifest = _client_scan_manifest.get()
metadata = {'code_manifest': manifest, 'authority': 'remote-worker'} if manifest else None
if metadata is None:
metadata = require_active_supervisor_child(child_kind=child_kind, require_dsn=True)
manifest = metadata.get('code_manifest') or {}
expected = (manifest.get('executables') or {}).get('trufflehog') or {}
candidate = resolve_manifest_executable((command or [get_trufflehog_cmd()])[0])
if canonical_path(candidate) != canonical_path(expected.get('path') or ''):
raise RuntimeError('TruffleHog command does not match immutable supervisor authority')
values = list(command or [])
if '--config' in values:
try:
policy_path = canonical_path(values[values.index('--config') + 1])
except (IndexError, TypeError, ValueError) as exc:
raise RuntimeError('TruffleHog policy argument is incomplete') from exc
assets = manifest.get('assets') or {}
if policy_path not in {canonical_path(item.get('path') or '') for item in assets.values() if isinstance(item, dict)}:
raise RuntimeError('TruffleHog policy does not match immutable supervisor authority')
return metadata
def get_git_cmd():
"""Return manifested Git in runtime; uninitialized tests may resolve PATH."""
manifest = _client_scan_manifest.get()
if manifest:
return str(((manifest.get('executables') or {}).get('git') or {}).get('path') or '')
if not _runtime_initialized:
return shutil.which('git') or 'git'
if str(os.getenv(CHILD_KIND_ENV) or 'scanner').strip().lower() != 'scanner':
raise RuntimeError('Git clone launch requires scanner authority')
manifest = _client_scan_manifest.get()
if manifest:
metadata = {'code_manifest': manifest, 'authority': 'remote-worker'}
else:
metadata = require_active_supervisor_child(child_kind='scanner', require_dsn=True)
manifest = metadata.get('code_manifest') or {}
expected = (manifest.get('executables') or {}).get('git') or {}
path = expected.get('path')
if not isinstance(path, str) or not os.path.isabs(path):
raise RuntimeError('Git executable is absent from immutable supervisor authority')
return path
def prepend_client_git_environment(env):
manifest = _client_scan_manifest.get()
if manifest is None:
return env
path = str(((manifest.get('executables') or {}).get('git') or {}).get('path') or '')
if not os.path.isabs(path):
raise RuntimeError('remote worker Git executable is absent from immutable authority')
directory = os.path.dirname(path)
env['PATH'] = os.pathsep.join((directory, env.get('PATH', '')))
return env
def require_git_clone_launch_authority(cmd):
"""Authorize only checkout-free HTTPS clones, never general Git commands."""
if (
not isinstance(cmd, (list, tuple)) or len(cmd) != 7
or any(not isinstance(value, str) or not value or any(ord(ch) < 32 or ord(ch) == 127 for ch in value) for value in cmd)
or list(cmd[1:5]) != ['clone', '--no-checkout', '--no-recurse-submodules', '--']
):
raise RuntimeError('Git clone command does not match the allowed argv contract')
source, destination = cmd[5:]
try:
parsed = urlsplit(source)
valid_source = (
source.startswith('https://') and bool(parsed.hostname)
and parsed.username is None and parsed.password is None
and bool(parsed.path) and parsed.path.startswith('/')
and not parsed.query and not parsed.fragment
and not any(ch.isspace() for ch in source) and '\\' not in source
and '%' not in parsed.netloc and parsed.port != 0
)
except ValueError:
valid_source = False
if not valid_source:
raise RuntimeError('Git clone source must be credential-free absolute HTTPS without query or fragment')
if (
not os.path.isabs(destination) or destination.startswith('-')
or (os.name == 'nt' and not os.path.splitdrive(destination)[0])
):
raise RuntimeError('Git clone destination must be an absolute path')
if not os.path.isabs(cmd[0]):
raise RuntimeError('Git command does not match immutable supervisor authority')
if str(os.getenv(CHILD_KIND_ENV) or 'scanner').strip().lower() != 'scanner':
raise RuntimeError('Git clone launch requires scanner authority')
manifest = _client_scan_manifest.get()
if manifest:
metadata = {'code_manifest': manifest, 'authority': 'remote-worker'}
else:
metadata = require_active_supervisor_child(child_kind='scanner', require_dsn=True)
manifest = metadata.get('code_manifest') or {}
expected = (manifest.get('executables') or {}).get('git') or {}
# Authentication rehashes manifest contents, not just path/mtime identity.
if cmd[0] != expected.get('path'):
raise RuntimeError('Git command does not match immutable supervisor authority')
return metadata
def get_trufflehog_config(config_path=None):
"""Return configured TruffleHog custom detector config path, if any."""
return str(config_path if config_path is not None else getattr(scan_config, 'trufflehog_config', '') or '').strip()
def append_trufflehog_scan_args(cmd, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None):
"""Append common TruffleHog scan flags in one place."""
config_path = get_trufflehog_config(trufflehog_config)
if config_path:
cmd.extend(['--config', config_path])
if detectors:
cmd.extend(['--include-detectors', detectors])
if exclude_detectors:
cmd.extend(['--exclude-detectors', exclude_detectors])
if no_verification:
cmd.append('--no-verification')
return cmd
def get_work_dir():
"""Prepare and return the directory used for TruffleHog temporary data."""
require_scanner_runtime_initialized()
if not scan_config.work_dir:
raise RuntimeError('TruffleHog work_dir is required')
try:
return require_private_directory(scan_config.work_dir, create=False)
except OSError as exc:
raise RuntimeError(f'Unable to use private TruffleHog work_dir {scan_config.work_dir}: {exc}') from exc
def create_command_work_dir():
"""Create an isolated temporary directory for one TruffleHog subprocess."""
work_dir = get_work_dir()
ensure_work_dir_space()
path = tempfile.mkdtemp(prefix='trufflehog-run-', dir=work_dir)
try:
harden_private_directory(path)
if not write_temp_owner(path, ['scanner-workdir'], os.getpid(), required=True):
raise RuntimeError(f'Unable to write required temp owner marker for {path}')
return path
except Exception:
try_remove_tree(path, attempts=2, delay=0.2)
raise
def write_temp_owner(path, cmd=None, owner_pid=None, required=False, owner_identity=None):
require_scanner_runtime_initialized()
if not path:
return
try:
owner_pid = int((owner_identity or {}).get('pid') if isinstance(owner_identity, dict) else owner_pid or os.getpid())
parent_identity = current_process_identity()
if isinstance(owner_identity, dict):
owner = dict(owner_identity)
elif owner_pid == parent_identity.pid:
owner = serialize_process_identity(parent_identity)
else:
with open_process(owner_pid) as retained:
owner = serialize_process_identity(retained.identity)
work_root = canonical_path(get_work_dir())
candidate = canonical_path(path)
relative = os.path.relpath(candidate, work_root)
if relative == '.' or relative.startswith('..' + os.sep) or os.path.isabs(relative):
raise RuntimeError('temp owner marker path escapes configured work_dir')
parent = serialize_process_identity(parent_identity)
payload = {
'schema': TEMP_OWNER_SCHEMA,
'owner_pid': owner['pid'],
'owner_creation_time': owner['creation_time'],
'owner_executable': owner['executable'],
'parent_pid': parent['pid'],
'parent_creation_time': parent['creation_time'],
'parent_executable': parent['executable'],
'created_at': datetime.now(timezone.utc).isoformat(timespec='seconds'),
'root_kind': 'work',
'relative_path': relative.replace(os.sep, '/'),
'command': redact_command_args(cmd or [])[:64],
}
atomic_write_private_json(os.path.join(path, TEMP_OWNER_FILE), payload)
return True
except (OSError, ValueError) as e:
if required:
raise RuntimeError(f'Unable to write required temp owner marker for {path}: {e}') from e
logger.warning(f'Unable to write temp owner marker for {path}: {str(e)[:200]}')
return False
def redact_command_args(args):
sensitive_flags = {'--token', '--docker-token', '--password', '--api-key', '--secret'}
redacted = []
hide_next = False
for value in args or []:
text = str(value)
if hide_next:
redacted.append('***REDACTED***')
hide_next = False
continue
flag = text.split('=', 1)[0].lower()
if flag in sensitive_flags:
if '=' in text:
redacted.append(text.split('=', 1)[0] + '=***REDACTED***')
else:
redacted.append(text)
hide_next = True
continue
try:
parsed = urlsplit(text)
if parsed.scheme in ('http', 'https') and parsed.hostname and ('@' in parsed.netloc or parsed.password):
netloc = parsed.hostname
if parsed.port:
netloc += f':{parsed.port}'
else:
netloc = parsed.netloc
if parsed.scheme in ('http', 'https') and parsed.hostname:
query = []
for key, query_value in parse_qsl(parsed.query, keep_blank_values=True):
sensitive = any(part in key.lower() for part in ('password', 'passwd', 'pwd', 'token', 'secret', 'credential'))
query.append((key, '***REDACTED***' if sensitive else query_value))
text = urlunsplit((parsed.scheme, netloc, parsed.path, urlencode(query), parsed.fragment))
except ValueError:
pass
redacted.append(text)
return redacted
def read_temp_owner(path):
marker = os.path.join(path, TEMP_OWNER_FILE)
try:
if not private_file_ready(marker):
return {}
data = read_private_json(marker)
return data if isinstance(data, dict) else {}
except (OSError, ValueError):
return {}
def temp_dir_active(path):
owner = read_temp_owner(path)
if owner.get('schema') != TEMP_OWNER_SCHEMA:
return True
states = [
exact_process_identity_state(
owner.get(f'{prefix}_pid'),
owner.get(f'{prefix}_creation_time'),
owner.get(f'{prefix}_executable'),
)
for prefix in ('owner', 'parent')
]
return any(state in ('alive', 'unknown') for state in states)
def create_docker_config_dir():
work_dir = get_work_dir()
docker_config_root = os.path.join(work_dir, 'docker-config')
require_private_directory(docker_config_root, create=True)
path = tempfile.mkdtemp(prefix='docker-config-', dir=docker_config_root)
harden_private_directory(path)
write_temp_owner(path, ['docker-auth-config'], os.getpid())
return path
def force_remove_readonly(function, path, exc_info):
try:
reject_reparse_components(path)
os.chmod(path, stat.S_IWRITE)
function(path)
except Exception:
pass
def try_remove_tree(path, attempts=1, delay=0.0):
if not path:
return True
for attempt in range(attempts):
try:
budget = JanitorBudget(
max_candidates=1,
max_entries=10000,
max_bytes=1024 * 1024 * 1024,
max_seconds=5.0,
max_depth=64,
)
return bounded_remove_tree(path, budget)
except FileNotFoundError:
return True
except KeyboardInterrupt:
raise
except Exception:
if delay and attempt + 1 < attempts:
time.sleep(delay)
return False
def _shared_staging_owners(roots):
"""Resolve only private trees already owned by this scanner; never adopt input data."""
work_root = canonical_path(get_work_dir())
current = serialize_process_identity(current_process_identity())
owners = {}
for value in roots:
root = canonical_path(require_private_directory(value, create=False))
if root == work_root or os.path.commonpath((root, work_root)) != work_root:
raise RuntimeError('shared staging root escapes configured work_dir')
while root != work_root and not os.path.lexists(os.path.join(root, TEMP_OWNER_FILE)):
root = os.path.dirname(root)
if root == work_root:
raise RuntimeError('shared staging root has no authenticated owner')
if root in owners:
continue
require_private_directory(root, create=False)
marker = read_private_json(require_private_file(os.path.join(root, TEMP_OWNER_FILE)), max_bytes=65536)
relative = os.path.relpath(root, work_root).replace(os.sep, '/')
if marker.get('schema') != TEMP_OWNER_SCHEMA or marker.get('root_kind') != 'work' or marker.get('relative_path') != relative:
raise RuntimeError('shared staging owner marker does not match its private root')
if any(marker.get(f'{prefix}_{field}') != current[field]
for prefix in ('owner', 'parent') for field in ('pid', 'creation_time', 'executable')):
raise RuntimeError('shared staging root belongs to another owner')
if any(f'child_{field}' in marker for field in ('pid', 'creation_time', 'executable')):
if any(not marker.get(f'child_{field}') for field in ('pid', 'creation_time', 'executable')) or (
exact_process_identity_state(marker.get('child_pid'), marker.get('child_creation_time'),
marker.get('child_executable')) != 'dead'
):
raise RuntimeError('shared staging root has an unconfirmed child')
for field in ('pid', 'creation_time', 'executable'):
marker.pop(f'child_{field}', None)
owners[root] = marker
return list(owners.items())
def cleanup_command_work_dir(path):
if not path:
return
marker_path = os.path.join(path, TEMP_OWNER_FILE)
if os.path.lexists(marker_path):
try:
marker = read_private_json(require_private_file(marker_path), max_bytes=65536)
if any(f'child_{field}' in marker for field in ('pid', 'creation_time', 'executable')):
if any(not marker.get(f'child_{field}') for field in ('pid', 'creation_time', 'executable')) or (
exact_process_identity_state(marker.get('child_pid'), marker.get('child_creation_time'),
marker.get('child_executable')) != 'dead'
):
logger.warning('Retaining private command tree with an unconfirmed child')
return
except (OSError, TypeError, ValueError):
logger.warning('Retaining private command tree with unreadable ownership evidence')
return
if try_remove_tree(path, attempts=2, delay=0.2):
return
logger.info('Bounded immediate cleanup deferred to authenticated janitor: %s', path)
def approved_pending_temp_path(path, work_dir=None):
"""Validate scanner-owned placement without following an external path."""
if not path or not os.path.isabs(path):
return False
try:
work_root = work_dir or get_work_dir()
reject_reparse_components(work_root)
reject_reparse_components(path)
if is_reparse_point(path) or not os.path.isdir(path) or not private_directory_ready(path):
return False
work_root = canonical_path(work_root)
candidate = canonical_path(path)
if os.path.commonpath((work_root, candidate)) != work_root or candidate == work_root:
return False
relative = os.path.relpath(candidate, work_root)
except (OSError, ValueError):
return False
parts = relative.split(os.sep)
if len(parts) == 1:
approved_name = parts[0].startswith(APPROVED_TEMP_PREFIXES)
elif len(parts) == 2 and parts[0] == 'docker-config':
approved_name = parts[1].startswith('docker-config-')
elif len(parts) == 2 and parts[0] == 'hg':
approved_name = parts[1].startswith('hg-run-')
elif len(parts) == 2 and parts[0] == 'tmp':
approved_name = parts[1].startswith(APPROVED_TEMP_PREFIXES)
elif len(parts) == 3 and parts[:2] == ['tmp', 'docker-config']:
approved_name = parts[2].startswith('docker-config-')
else:
approved_name = False
if not approved_name:
return False
owner = read_temp_owner(candidate)
owner_pid = owner.get('owner_pid') or owner.get('parent_pid')
return bool(owner_pid) and not temp_dir_active(candidate)
def cleanup_pending_command_work_dirs(max_items=None, attempts=1, delay=0.0, log_failures=False):
if log_failures:
logger.info('Source-side pending temp cleanup is retired; the authenticated janitor owns recovery')
return 0
def cleanup_assignment_work_dir(max_items=256):
"""Bound one cooperative cleanup pass to this runner's private work root."""
work_root = get_work_dir()
report = {'enumerated': 0, 'removed': 0, 'retained': 0}
with os.scandir(work_root) as entries:
for entry in entries:
report['enumerated'] += 1
if report['enumerated'] > max(1, int(max_items)):
report['retained'] += 1
break
if (
entry.is_symlink()
or not entry.is_dir(follow_symlinks=False)
or not entry.name.startswith(APPROVED_TEMP_PREFIXES)
):
continue
if cleanup_command_work_dir(entry.path) is None and not os.path.exists(entry.path):
report['removed'] += 1
else:
report['retained'] += 1
return report
def ensure_work_dir_space():
work_dir = get_work_dir()
if not work_dir or scan_config.min_free_gb <= 0:
return
min_free_bytes = scan_config.min_free_gb * 1024 * 1024 * 1024
free_bytes = shutil.disk_usage(work_dir).free
if free_bytes >= min_free_bytes:
return
raise RuntimeError(
f"Not enough free space on {work_dir}: {free_bytes / (1024 ** 3):.2f} GB free, "
f"minimum is {scan_config.min_free_gb:.2f} GB; admission is closed without cleanup"
)
def cleanup_stale_temp_dirs(age_minutes=120, log=True, max_items=None):
"""Compatibility no-op; stale recovery is isolated in janitor.py."""
if log:
logger.info('Source-side stale temp cleanup is retired; the authenticated janitor owns recovery')
return 0
def get_results_dir():
"""Prepare and return the directory used for persisted scan output."""
require_scanner_runtime_initialized()
if not scan_config.results_dir:
raise RuntimeError('scan results directory is required')
try:
return require_private_directory(scan_config.results_dir, create=False)
except OSError as exc:
raise RuntimeError(f'Unable to use private scan results directory {scan_config.results_dir}: {exc}') from exc
def append_jsonl(path, payload):
lock = None
lock_path = f'{path}.lock'
try:
require_private_directory(os.path.dirname(os.path.abspath(path)), create=True)
if os.path.lexists(path):
reject_reparse_components(path)
lock = acquire_file_lock(lock_path, timeout_sec=30)
repair_jsonl_tail(path)
serialized = (json.dumps(payload, ensure_ascii=False, default=str) + '\n').encode('utf-8')
with open(path, 'ab') as f:
f.write(serialized)
f.flush()
os.fsync(f.fileno())
harden_private_file(path)
return True
except OSError as e:
if getattr(e, 'errno', None) == 28:
logger.error(f"No space left while writing {path}. Result was not persisted.")
else:
logger.error(f"Unable to write {path}: {str(e)}")
return False
finally:
if lock is not None:
release_file_lock(lock, lock_path)
def jsonl_manifest_path(path):
base, ext = os.path.splitext(path)
return f'{base}.manifest.json'
def load_jsonl_manifest(path):
manifest_path = jsonl_manifest_path(path)
try:
if os.path.getsize(manifest_path) > 1024 * 1024:
raise ValueError(f'JSONL manifest exceeds its bounded size: {manifest_path}')
with open(manifest_path, 'r', encoding='utf-8') as f:
data = json.load(f)
return data if isinstance(data, dict) else {}
except FileNotFoundError:
return {}
def write_jsonl_manifest(path, manifest):
manifest_path = jsonl_manifest_path(path)
tmp_path = f'{manifest_path}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.tmp'
with open(tmp_path, 'w', encoding='utf-8') as f:
json.dump(manifest, f, ensure_ascii=False, indent=2, sort_keys=True)
f.flush()
os.fsync(f.fileno())
harden_private_file(tmp_path)
os.replace(tmp_path, manifest_path)
harden_private_file(manifest_path)
def next_jsonl_segment_path(path, manifest):
base, ext = os.path.splitext(path)
seq = int(manifest.get('next_sequence') or 1)
while True:
segment = f'{base}.{seq:06d}{ext or ".jsonl"}'
if not os.path.exists(segment):
return segment, seq
seq += 1
def acquire_file_lock(lock_path, stale_sec=300, timeout_sec=30):
require_private_directory(os.path.dirname(os.path.abspath(lock_path)), create=True)
reject_reparse_components(os.path.dirname(os.path.abspath(lock_path)))
deadline = time.monotonic() + max(0.01, float(timeout_sec))
while True:
lock = PrivateFileLock(lock_path)
try:
return lock.acquire()
except BlockingIOError:
if time.monotonic() >= deadline:
raise TimeoutError(f'timed out acquiring lock {lock_path}')
time.sleep(min(0.05, max(0.0, deadline - time.monotonic())))
def release_file_lock(lock, lock_path):
try:
lock.release()
except (AttributeError, OSError):
return
class JsonlProjectionReconciliationRequired(RuntimeError):
pass
def jsonl_ledger_path(path):
base, _ = os.path.splitext(path)
return f'{base}.publication-ledger.sqlite3'
def _projection_segment_sequence(path):
parent = os.path.dirname(os.path.abspath(path))
base, extension = os.path.splitext(os.path.basename(path))
pattern = re.compile(rf'^{re.escape(base)}\.(\d{{6}}){re.escape(extension)}$')
output = []
inspect_limit = max(2, int(getattr(scan_config, 'jsonl_max_segments', 16)) + 1)
try:
with os.scandir(parent) as entries:
for entry in entries:
match = pattern.fullmatch(entry.name)
if not match:
continue
if entry.is_symlink() or is_reparse_point(entry.path) or not entry.is_file(follow_symlinks=False):
raise JsonlProjectionReconciliationRequired(f'unsafe JSONL segment entry: {entry.path}')
output.append((int(match.group(1)), entry.path))
if len(output) > inspect_limit:
raise JsonlProjectionReconciliationRequired(
f'JSONL physical segment count exceeds its bound for {path}; use offline reconciliation'
)
except FileNotFoundError:
return []
return sorted(output)
def _write_torn_tail_quarantine(path, payload):
limit = max(1, int(getattr(scan_config, 'jsonl_torn_quarantine_max_bytes', 64 * 1024)))
sample = bytes(payload[:limit])
quarantine = f'{path}.torn-tail.bin'
temporary = f'{quarantine}.{os.getpid()}.{threading.get_ident()}.tmp'
with open(temporary, 'wb') as handle:
handle.write(sample)
handle.flush()
os.fsync(handle.fileno())
harden_private_file(temporary)
durable_replace(temporary, quarantine)
harden_private_file(quarantine)
def repair_jsonl_tail(path):
"""Quarantine and remove one bounded unterminated tail before appending."""
if not os.path.exists(path):
return 0
reject_reparse_components(path)
size = os.path.getsize(path)
if size <= 0:
return 0
scan_limit = max(1, int(getattr(scan_config, 'jsonl_tail_scan_max_bytes', 8 * 1024 * 1024)))
with open(path, 'r+b') as handle:
handle.seek(-1, os.SEEK_END)
if handle.read(1) == b'\n':
return 0
start = max(0, size - scan_limit)
handle.seek(start)
tail = handle.read(size - start)
newline = tail.rfind(b'\n')
if newline < 0 and start:
raise JsonlProjectionReconciliationRequired(
f'JSONL tail exceeds the bounded repair window for {path}; use offline reconciliation'
)
truncate_at = start + newline + 1 if newline >= 0 else 0
torn = tail[newline + 1:] if newline >= 0 else tail
_write_torn_tail_quarantine(path, torn)
handle.truncate(truncate_at)
handle.flush()
os.fsync(handle.fileno())
logger.error('Quarantined and truncated %s torn byte(s) from %s', size - truncate_at, path)
return size - truncate_at
def _open_projection_ledger(path):
ledger_path = jsonl_ledger_path(path)
reject_reparse_components(os.path.dirname(os.path.abspath(ledger_path)))
if os.path.lexists(ledger_path):
reject_reparse_components(ledger_path)
if not private_file_ready(ledger_path):
raise JsonlProjectionReconciliationRequired(f'JSONL publication ledger is not private: {ledger_path}')
connection = sqlite3.connect(ledger_path, timeout=30)
try:
connection.execute('PRAGMA busy_timeout=30000')
connection.execute('PRAGMA journal_mode=DELETE')
connection.execute('PRAGMA synchronous=FULL')
connection.executescript('''
CREATE TABLE IF NOT EXISTS publication_identity (
identity_key TEXT NOT NULL,
identity_value TEXT NOT NULL,
payload_sha256 TEXT NOT NULL,
state TEXT NOT NULL,
file_name TEXT NOT NULL,
byte_offset INTEGER NOT NULL,
byte_length INTEGER NOT NULL,
created_at REAL NOT NULL,
updated_at REAL NOT NULL,
PRIMARY KEY(identity_key, identity_value)
);
CREATE TABLE IF NOT EXISTS publication_meta (
key TEXT PRIMARY KEY,
value TEXT NOT NULL
);
CREATE TABLE IF NOT EXISTS publication_identity_variant (
identity_key TEXT NOT NULL,
identity_value TEXT NOT NULL,
payload_sha256 TEXT NOT NULL,
file_name TEXT NOT NULL,
byte_offset INTEGER NOT NULL,
byte_length INTEGER NOT NULL,
created_at REAL NOT NULL,
PRIMARY KEY(identity_key, identity_value, payload_sha256)
);
CREATE TABLE IF NOT EXISTS reconciliation_issue (
id INTEGER PRIMARY KEY AUTOINCREMENT,
identity_key TEXT NOT NULL,
file_name TEXT NOT NULL,
file_device TEXT NOT NULL,
file_inode TEXT NOT NULL,
file_size INTEGER NOT NULL,
file_mtime_ns TEXT NOT NULL,
byte_offset INTEGER NOT NULL,
byte_length INTEGER NOT NULL,
record_sha256 TEXT NOT NULL,
classification TEXT NOT NULL,
status TEXT NOT NULL,
created_at REAL NOT NULL,
resolved_at REAL,
UNIQUE(identity_key, file_name, byte_offset, record_sha256)
);
CREATE TABLE IF NOT EXISTS reconciliation_variant_issue (
id INTEGER PRIMARY KEY AUTOINCREMENT,
identity_key TEXT NOT NULL,
identity_sha256 TEXT NOT NULL,
file_name TEXT NOT NULL,
file_device TEXT NOT NULL,
file_inode TEXT NOT NULL,
file_size INTEGER NOT NULL,
file_mtime_ns TEXT NOT NULL,
byte_offset INTEGER NOT NULL,
byte_length INTEGER NOT NULL,
payload_sha256 TEXT NOT NULL,
field_name_set_sha256 TEXT NOT NULL,
status TEXT NOT NULL,
created_at REAL NOT NULL,
resolved_at REAL,
UNIQUE(identity_key, identity_sha256, file_name, byte_offset, payload_sha256)
);
CREATE INDEX IF NOT EXISTS idx_publication_identity_state_created
ON publication_identity(state, created_at);
CREATE INDEX IF NOT EXISTS idx_publication_identity_file_state
ON publication_identity(file_name, state);
CREATE INDEX IF NOT EXISTS idx_publication_identity_variant_identity
ON publication_identity_variant(identity_key, identity_value);
CREATE INDEX IF NOT EXISTS idx_reconciliation_issue_status
ON reconciliation_issue(status, id);
CREATE INDEX IF NOT EXISTS idx_reconciliation_variant_issue_status
ON reconciliation_variant_issue(status, id);
''')
connection.execute(
'''INSERT OR IGNORE INTO publication_identity_variant (
identity_key, identity_value, payload_sha256, file_name,
byte_offset, byte_length, created_at
)
SELECT identity_key, identity_value, payload_sha256, file_name,
byte_offset, byte_length, created_at
FROM publication_identity WHERE state = 'appended' '''
)
row = connection.execute("SELECT value FROM publication_meta WHERE key = 'row_count'").fetchone()
if row is None:
count = int(connection.execute('SELECT COUNT(*) FROM publication_identity').fetchone()[0])
connection.execute(
"INSERT INTO publication_meta(key, value) VALUES ('row_count', ?)",
(str(count),),
)
connection.commit()
harden_private_file(ledger_path)
return connection
except BaseException:
connection.close()
raise
def _ledger_row_count(connection):
row = connection.execute("SELECT value FROM publication_meta WHERE key = 'row_count'").fetchone()
return max(0, int(row[0] if row else 0))
def _set_ledger_row_count(connection, count):
connection.execute(
"INSERT OR REPLACE INTO publication_meta(key, value) VALUES ('row_count', ?)",
(str(max(0, int(count))),),
)
def _bounded_projection_bytes(path, offset, length):
max_record = max(
1024 * 1024,
int(getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)) + 1,
)
if offset < 0 or length <= 0 or length > max_record:
return b''
try:
with open(path, 'rb') as handle:
handle.seek(offset)
return handle.read(length)
except OSError:
return b''
def _recover_prepared_publications(connection, path):
rows = connection.execute(
'''SELECT identity_key, identity_value, payload_sha256, file_name, byte_offset, byte_length
FROM publication_identity WHERE state = 'prepared' ORDER BY created_at LIMIT 2'''
).fetchall()
if len(rows) > 1:
raise JsonlProjectionReconciliationRequired('publication ledger contains multiple unresolved append states')
for identity_key, identity_value, digest, file_name, offset, length in rows:
candidate = os.path.join(os.path.dirname(os.path.abspath(path)), os.path.basename(file_name))
payload = _bounded_projection_bytes(candidate, int(offset), int(length))
if payload.endswith(b'\n') and hashlib.sha256(payload).hexdigest() == digest:
connection.execute(
'''UPDATE publication_identity SET state = 'appended', updated_at = ?
WHERE identity_key = ? AND identity_value = ? AND state = 'prepared' ''',
(time.time(), identity_key, identity_value),
)
connection.execute(
'''INSERT OR IGNORE INTO publication_identity_variant (
identity_key, identity_value, payload_sha256, file_name,
byte_offset, byte_length, created_at
) VALUES (?, ?, ?, ?, ?, ?, ?)''',
(identity_key, identity_value, digest, file_name, offset, length, time.time()),
)
else:
connection.execute(
'DELETE FROM publication_identity WHERE identity_key = ? AND identity_value = ? AND state = ?',
(identity_key, identity_value, 'prepared'),
)
_set_ledger_row_count(connection, _ledger_row_count(connection) - 1)
connection.commit()
def _ensure_projection_ledger_bootstrapped(connection, path, identity_key):
marker = f'bootstrapped:{identity_key}'
if connection.execute('SELECT 1 FROM publication_meta WHERE key = ?', (marker,)).fetchone():
return
candidates = projection_segment_paths(path)
total_bytes = sum(os.path.getsize(candidate) for candidate in candidates)
max_bytes = max(0, int(getattr(scan_config, 'jsonl_legacy_index_max_bytes', 16 * 1024 * 1024)))
if total_bytes > max_bytes:
raise JsonlProjectionReconciliationRequired(
f'existing JSONL history for {path} is unindexed ({total_bytes} bytes); '
'run the offline JSONL reconciliation procedure before publication'
)
row_count = _ledger_row_count(connection)
row_limit = max(1, int(getattr(scan_config, 'jsonl_ledger_max_rows', 1000000)))
for candidate in candidates:
offset = 0
with open(candidate, 'rb') as handle:
for raw_line in handle:
if not raw_line.endswith(b'\n'):
raise JsonlProjectionReconciliationRequired(f'unterminated closed JSONL record in {candidate}')
identity = _projection_identity_from_line(raw_line, identity_key, candidate)
if identity:
digest = hashlib.sha256(raw_line).hexdigest()
existing = connection.execute(
'''SELECT payload_sha256 FROM publication_identity
WHERE identity_key = ? AND identity_value = ?''',
(identity_key, identity),
).fetchone()
if existing and existing[0] != digest:
raise JsonlProjectionReconciliationRequired(
'conflicting identity in existing JSONL history: '
f'identity_key={identity_key} '
f'identity_sha256={hashlib.sha256(identity.encode("utf-8")).hexdigest()}'
)
if not existing:
if row_count >= row_limit:
raise JsonlProjectionReconciliationRequired('existing JSONL identities exceed the ledger row bound')
now = time.time()
connection.execute(
'''INSERT INTO publication_identity (
identity_key, identity_value, payload_sha256, state, file_name,
byte_offset, byte_length, created_at, updated_at
) VALUES (?, ?, ?, 'appended', ?, ?, ?, ?, ?)''',
(
identity_key, identity, digest, os.path.basename(candidate),
offset, len(raw_line), now, now,
),
)
connection.execute(
'''INSERT INTO publication_identity_variant (
identity_key, identity_value, payload_sha256, file_name,
byte_offset, byte_length, created_at
) VALUES (?, ?, ?, ?, ?, ?, ?)''',
(
identity_key, identity, digest, os.path.basename(candidate),
offset, len(raw_line), now,
),
)
row_count += 1
offset += len(raw_line)
_set_ledger_row_count(connection, row_count)
connection.execute('INSERT INTO publication_meta(key, value) VALUES (?, ?)', (marker, '1'))
connection.commit()
def _projection_identity_from_line(raw_line, identity_key, candidate):
if not raw_line.endswith(b'\n'):
raise JsonlProjectionReconciliationRequired(f'unterminated projection record in {candidate}')
if identity_key == 'error_row_id':
try:
identity, separator, _ = raw_line.partition(b'\t')
if not separator or not identity:
raise ValueError('missing error projection identity separator')
return identity.decode('utf-8')
except UnicodeDecodeError as exc:
raise JsonlProjectionReconciliationRequired(
f'invalid existing projection record in {candidate}'
) from exc
except ValueError as exc:
raise JsonlProjectionReconciliationRequired(
f'invalid existing projection record in {candidate}'
) from exc
try:
payload = json.loads(raw_line.decode('utf-8'))
except (UnicodeDecodeError, ValueError) as exc:
raise JsonlProjectionReconciliationRequired(
f'invalid existing JSONL record in {candidate}'
) from exc
return str(payload.get(identity_key) or '') if isinstance(payload, dict) else ''
def _bounded_projection_json_values(raw_line):
body = raw_line[:-1] if raw_line.endswith(b'\n') else raw_line
if not raw_line.endswith(b'\n'):
return None, 'unterminated_record'
try:
text = body.decode('utf-8')
except UnicodeDecodeError:
return None, 'invalid_utf8'
if text.startswith('\ufeff'):
return None, 'utf8_bom_prefix'
try:
value = json.loads(text)
return [{
'value': value,
'relative_offset': 0,
'byte_length': len(raw_line),
'payload_sha256': hashlib.sha256(raw_line).hexdigest(),
}], 'single_json'
except ValueError:
pass
decoder = json.JSONDecoder()
position = 0
parsed = []
while True:
while position < len(text) and text[position].isspace():
position += 1
if position >= len(text):
break
start = position
try:
value, position = decoder.raw_decode(text, position)
except json.JSONDecodeError:
parsed = []
break
parsed.append((start, position, value))
if len(parsed) > 1 and position >= len(text) and all(isinstance(item[2], dict) for item in parsed):
values = []
for start, end, value in parsed:
prefix_bytes = len(text[:start].encode('utf-8'))
serialized = text[start:end].encode('utf-8') + b'\n'
values.append({
'value': value,
'relative_offset': prefix_bytes,
'byte_length': len(serialized),
'payload_sha256': hashlib.sha256(serialized).hexdigest(),
})
return values, 'concatenated_json_objects'
if re.match(r'^[0-9]+,\s*', text):
return None, 'legacy_numeric_prefix_corrupt_json'
return None, 'invalid_json'
def _projection_field_name_set_sha256(value):
paths = []
def visit(item, prefix=''):
if isinstance(item, dict):
for key in sorted(map(str, item.keys())):
path = prefix + key
paths.append(path)
visit(item.get(key), path + '.')
elif isinstance(item, list):
paths.append(prefix + '[]')
for child in item[:32]:
visit(child, prefix + '[].')
visit(value)
return hashlib.sha256('\x00'.join(sorted(set(paths))).encode('utf-8')).hexdigest()
def _error_projection_field_name_set_sha256(raw_line):
try:
text = raw_line[:-1].decode('utf-8') if raw_line.endswith(b'\n') else raw_line.decode('utf-8')
tail = text.rsplit('\t', 1)[-1]
value = json.loads(tail)
except (UnicodeDecodeError, ValueError):
value = {}
return _projection_field_name_set_sha256(value)
def _save_projection_reconciliation_progress(ledger, prefix, file_index, offset, indexed_rows):
_set_ledger_row_count(ledger, indexed_rows)
for key, value in (
(prefix + 'file_index', str(file_index)),
(prefix + 'offset', str(offset)),
):
ledger.execute('INSERT OR REPLACE INTO publication_meta(key, value) VALUES (?, ?)', (key, value))
def _record_projection_reconciliation_issue(
ledger,
identity_key,
plan_entry,
byte_offset,
raw_line,
classification,
resolved=False,
byte_length=None,
record_sha256=None,
):
if raw_line is not None:
byte_length = len(raw_line)
record_sha256 = hashlib.sha256(raw_line).hexdigest()
byte_length = int(byte_length or 0)
digest = str(record_sha256 or '').strip().lower()
if byte_length <= 0 or not re.fullmatch(r'[a-f0-9]{64}', digest):
raise ValueError('projection reconciliation issue metadata is invalid')
now = time.time()
ledger.execute(
'''INSERT OR IGNORE INTO reconciliation_issue (
identity_key, file_name, file_device, file_inode, file_size, file_mtime_ns,
byte_offset, byte_length, record_sha256, classification, status, created_at, resolved_at
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''',
(
identity_key,
os.path.basename(plan_entry['path']),
str(plan_entry['device']),
str(plan_entry['inode']),
int(plan_entry['size']),
str(plan_entry['mtime_ns']),
int(byte_offset),
byte_length,
digest,
classification,
'resolved' if resolved else 'pending',
now,
now if resolved else None,
),
)
if resolved:
ledger.execute(
'''UPDATE reconciliation_issue SET status = 'resolved', resolved_at = COALESCE(resolved_at, ?)
WHERE identity_key = ? AND file_name = ? AND byte_offset = ? AND record_sha256 = ?''',
(now, identity_key, os.path.basename(plan_entry['path']), int(byte_offset), digest),
)
return ledger.execute(
'''SELECT id, status FROM reconciliation_issue
WHERE identity_key = ? AND file_name = ? AND byte_offset = ? AND record_sha256 = ?''',
(identity_key, os.path.basename(plan_entry['path']), int(byte_offset), digest),
).fetchone()
def _record_projection_variant_issue(
ledger,
identity_key,
identity,
plan_entry,
byte_offset,
byte_length,
payload_sha256,
field_name_set_sha256,
resolved=False,
):
identity_sha256 = hashlib.sha256(identity.encode('utf-8')).hexdigest()
now = time.time()
ledger.execute(
'''INSERT OR IGNORE INTO reconciliation_variant_issue (
identity_key, identity_sha256, file_name, file_device, file_inode,
file_size, file_mtime_ns, byte_offset, byte_length, payload_sha256,
field_name_set_sha256, status, created_at, resolved_at
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''',
(
identity_key,
identity_sha256,
os.path.basename(plan_entry['path']),
str(plan_entry['device']),
str(plan_entry['inode']),
int(plan_entry['size']),
str(plan_entry['mtime_ns']),
int(byte_offset),
int(byte_length),
payload_sha256,
field_name_set_sha256,
'resolved' if resolved else 'pending',
now,
now if resolved else None,
),
)
if resolved:
ledger.execute(
'''UPDATE reconciliation_variant_issue
SET status = 'resolved', resolved_at = COALESCE(resolved_at, ?)
WHERE identity_key = ? AND identity_sha256 = ? AND file_name = ?
AND byte_offset = ? AND payload_sha256 = ?''',
(
now, identity_key, identity_sha256, os.path.basename(plan_entry['path']),
int(byte_offset), payload_sha256,
),
)
return ledger.execute(
'''SELECT id, status FROM reconciliation_variant_issue
WHERE identity_key = ? AND identity_sha256 = ? AND file_name = ?
AND byte_offset = ? AND payload_sha256 = ?''',
(
identity_key, identity_sha256, os.path.basename(plan_entry['path']),
int(byte_offset), payload_sha256,
),
).fetchone()
def _stream_projection_record(handle, first_chunk, chunk_bytes=1024 * 1024):
digest = hashlib.sha256()
digest.update(first_chunk)
total = len(first_chunk)
newline_terminated = first_chunk.endswith(b'\n')
while not newline_terminated:
chunk = handle.readline(max(1, int(chunk_bytes)))
if not chunk:
break
digest.update(chunk)
total += len(chunk)
newline_terminated = chunk.endswith(b'\n')
return total, digest.hexdigest(), newline_terminated
def _projection_reconciliation_plan(path):
candidates = projection_segment_paths(path)
if not candidates:
_publish_empty_jsonl_generation(path)
candidates = [os.path.abspath(path)]
plan = []
for candidate in candidates:
require_private_file(candidate)
details = os.stat(candidate, follow_symlinks=False)
plan.append({
'path': os.path.abspath(candidate),
'device': int(getattr(details, 'st_dev', 0) or 0),
'inode': int(getattr(details, 'st_ino', 0) or 0),
'size': int(details.st_size),
'mtime_ns': int(getattr(details, 'st_mtime_ns', int(details.st_mtime * 1_000_000_000))),
})
return plan
def reconcile_projection_ledger_batch(
path,
identity_key,
max_rows=10000,
max_bytes=64 * 1024 * 1024,
max_seconds=30.0,
row_limit=None,
ledger_byte_limit=None,
max_record_bytes=None,
):
"""Build one bounded, resumable ledger batch without modifying JSONL history."""
if identity_key not in ('scan_event_id', 'finding_uid', 'error_row_id'):
raise ValueError('unsupported projection reconciliation identity')
path = os.path.abspath(path)
require_private_directory(os.path.dirname(path), create=False)
max_rows = max(1, int(max_rows))
max_bytes = max(1, int(max_bytes))
max_seconds = max(0.01, float(max_seconds))
row_limit = max(1, int(row_limit or getattr(scan_config, 'jsonl_ledger_max_rows', 1000000)))
ledger_byte_limit = max(
1024 * 1024,
int(ledger_byte_limit or getattr(scan_config, 'jsonl_ledger_max_bytes', 512 * 1024 * 1024)),
)
max_record_bytes = max(
1024 * 1024,
int(max_record_bytes or getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)),
)
lock_path = f'{path}.lock'
lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec)
ledger = None
try:
prefix = f'offline-reconcile:{identity_key}:'
marker = f'bootstrapped:{identity_key}'
ledger = _open_projection_ledger(path)
if ledger.execute('SELECT 1 FROM publication_meta WHERE key = ?', (marker,)).fetchone():
return {
'path': path, 'identity_key': identity_key, 'complete': True,
'batch_rows': 0, 'batch_bytes': 0, 'indexed_rows': _ledger_row_count(ledger),
}
plan = _projection_reconciliation_plan(path)
plan_json = json.dumps(plan, ensure_ascii=True, sort_keys=True, separators=(',', ':'))
existing_plan = ledger.execute(
'SELECT value FROM publication_meta WHERE key = ?', (prefix + 'plan',),
).fetchone()
if existing_plan and existing_plan[0] != plan_json:
raise JsonlProjectionReconciliationRequired(
f'projection history changed during offline reconciliation: {path}'
)
if not existing_plan:
ledger.execute(
'INSERT INTO publication_meta(key, value) VALUES (?, ?)',
(prefix + 'plan', plan_json),
)
file_index_row = ledger.execute(
'SELECT value FROM publication_meta WHERE key = ?', (prefix + 'file_index',),
).fetchone()
offset_row = ledger.execute(
'SELECT value FROM publication_meta WHERE key = ?', (prefix + 'offset',),
).fetchone()
file_index = max(0, int(file_index_row[0] if file_index_row else 0))
offset = max(0, int(offset_row[0] if offset_row else 0))
indexed_rows = _ledger_row_count(ledger)
batch_rows = 0
batch_bytes = 0
started = time.monotonic()
while file_index < len(plan):
candidate = plan[file_index]['path']
with open(candidate, 'rb') as handle:
handle.seek(offset)
while True:
raw_line = handle.readline(max_record_bytes + 1)
if not raw_line:
file_index += 1
offset = 0
break
if len(raw_line) > max_record_bytes:
byte_length, record_sha256, newline_terminated = _stream_projection_record(
handle, raw_line,
)
classification = 'oversized_record' if newline_terminated else 'unterminated_record'
issue = _record_projection_reconciliation_issue(
ledger,
identity_key,
plan[file_index],
offset,
None,
classification,
byte_length=byte_length,
record_sha256=record_sha256,
)
_save_projection_reconciliation_progress(
ledger, prefix, file_index, offset, indexed_rows,
)
ledger.commit()
if issue[1] != 'resolved':
raise JsonlProjectionReconciliationRequired(
'projection reconciliation issue requires explicit review: '
f'id={issue[0]} file={os.path.basename(candidate)} offset={offset} '
f'length={byte_length} sha256={record_sha256} '
f'classification={classification}'
)
offset += byte_length
batch_rows += 1
batch_bytes += byte_length
if (
batch_rows >= max_rows
or batch_bytes >= max_bytes
or time.monotonic() - started >= max_seconds
):
break
continue
if identity_key == 'error_row_id':
try:
identity = _projection_identity_from_line(raw_line, identity_key, candidate)
if len(identity) > 256 or any(
character in identity for character in ('\x00', '\r', '\n')
):
values = None
classification = 'invalid_error_projection'
else:
values = [{
'identity': identity,
'relative_offset': 0,
'byte_length': len(raw_line),
'payload_sha256': hashlib.sha256(raw_line).hexdigest(),
'field_name_set_sha256': _error_projection_field_name_set_sha256(raw_line),
}]
classification = 'error_projection'
except JsonlProjectionReconciliationRequired:
values = None
classification = (
'unterminated_record' if not raw_line.endswith(b'\n')
else 'invalid_error_projection'
)
else:
parsed_values, classification = _bounded_projection_json_values(raw_line)
values = None if parsed_values is None else [{
'identity': (
str(item['value'].get(identity_key) or '')
if isinstance(item['value'], dict) else ''
),
'relative_offset': item['relative_offset'],
'byte_length': item['byte_length'],
'payload_sha256': item['payload_sha256'],
'field_name_set_sha256': _projection_field_name_set_sha256(item['value']),
} for item in parsed_values]
if values is None:
issue = _record_projection_reconciliation_issue(
ledger,
identity_key,
plan[file_index],
offset,
raw_line,
classification,
)
_save_projection_reconciliation_progress(
ledger, prefix, file_index, offset, indexed_rows,
)
ledger.commit()
if issue[1] != 'resolved':
raise JsonlProjectionReconciliationRequired(
'projection reconciliation issue requires explicit review: '
f'id={issue[0]} file={os.path.basename(candidate)} offset={offset} '
f'length={len(raw_line)} sha256={hashlib.sha256(raw_line).hexdigest()} '
f'classification={classification}'
)
offset += len(raw_line)
batch_rows += 1
batch_bytes += len(raw_line)
if (
batch_rows >= max_rows
or batch_bytes >= max_bytes
or time.monotonic() - started >= max_seconds
):
break
continue
for value in values:
identity = value['identity']
if not identity:
continue
if len(identity) > 256 or any(character in identity for character in ('\x00', '\r', '\n')):
raise JsonlProjectionReconciliationRequired(
f'invalid {identity_key} in existing projection history: {candidate}:{offset}'
)
digest = value['payload_sha256']
existing = ledger.execute(
'''SELECT payload_sha256, state FROM publication_identity
WHERE identity_key = ? AND identity_value = ?''',
(identity_key, identity),
).fetchone()
variant = ledger.execute(
'''SELECT 1 FROM publication_identity_variant
WHERE identity_key = ? AND identity_value = ? AND payload_sha256 = ?''',
(identity_key, identity, digest),
).fetchone()
if variant:
continue
if existing:
if existing[1] != 'appended':
raise JsonlProjectionReconciliationRequired(
'publication ledger retained an unresolved historical identity state'
)
variant_offset = offset + int(value['relative_offset'])
issue = _record_projection_variant_issue(
ledger,
identity_key,
identity,
plan[file_index],
variant_offset,
int(value['byte_length']),
digest,
value['field_name_set_sha256'],
)
_save_projection_reconciliation_progress(
ledger, prefix, file_index, offset, indexed_rows,
)
ledger.commit()
raise JsonlProjectionReconciliationRequired(
'historical projection payload variant requires explicit review: '
f'id={issue[0]} file={os.path.basename(candidate)} '
f'offset={variant_offset} length={int(value["byte_length"])} '
f'identity_sha256={hashlib.sha256(identity.encode("utf-8")).hexdigest()} '
f'payload_sha256={digest}'
)
if not existing:
if indexed_rows >= row_limit:
raise JsonlProjectionReconciliationRequired(
f'existing JSONL identities exceed the ledger row bound: {row_limit}'
)
now = time.time()
ledger.execute(
'''INSERT INTO publication_identity (
identity_key, identity_value, payload_sha256, state, file_name,
byte_offset, byte_length, created_at, updated_at
) VALUES (?, ?, ?, 'appended', ?, ?, ?, ?, ?)''',
(
identity_key, identity, digest, os.path.basename(candidate),
offset + int(value['relative_offset']), int(value['byte_length']), now, now,
),
)
ledger.execute(
'''INSERT INTO publication_identity_variant (
identity_key, identity_value, payload_sha256, file_name,
byte_offset, byte_length, created_at
) VALUES (?, ?, ?, ?, ?, ?, ?)''',
(
identity_key, identity, digest, os.path.basename(candidate),
offset + int(value['relative_offset']), int(value['byte_length']), now,
),
)
indexed_rows += 1
offset += len(raw_line)
batch_rows += 1
batch_bytes += len(raw_line)
if (
batch_rows >= max_rows
or batch_bytes >= max_bytes
or time.monotonic() - started >= max_seconds
):
break
if batch_rows and (
batch_rows >= max_rows
or batch_bytes >= max_bytes
or time.monotonic() - started >= max_seconds
):
break
_save_projection_reconciliation_progress(ledger, prefix, file_index, offset, indexed_rows)
complete = file_index >= len(plan)
if complete:
ledger.execute('INSERT OR REPLACE INTO publication_meta(key, value) VALUES (?, ?)', (marker, '1'))
if os.path.getsize(jsonl_ledger_path(path)) > ledger_byte_limit:
raise JsonlProjectionReconciliationRequired(
f'JSONL identity ledger exceeds its {ledger_byte_limit} byte bound'
)
ledger.commit()
return {
'path': path,
'identity_key': identity_key,
'complete': complete,
'batch_rows': batch_rows,
'batch_bytes': batch_bytes,
'indexed_rows': indexed_rows,
'file_index': file_index,
'file_count': len(plan),
'byte_offset': offset,
}
except BaseException:
if ledger is not None:
ledger.rollback()
raise
finally:
if ledger is not None:
ledger.close()
if os.path.exists(jsonl_ledger_path(path)):
harden_private_file(jsonl_ledger_path(path))
release_file_lock(lock, lock_path)
def _review_projection_issue_record(
plan_entry,
identity_key,
byte_offset,
expected_sha256,
max_record_bytes,
expected_length=None,
expected_classification=None,
):
if byte_offset >= int(plan_entry['size']):
raise JsonlProjectionReconciliationRequired('projection issue offset is outside the immutable file')
with open(plan_entry['path'], 'rb') as handle:
if byte_offset:
handle.seek(byte_offset - 1)
if handle.read(1) != b'\n':
raise JsonlProjectionReconciliationRequired('projection issue offset is not a record boundary')
handle.seek(byte_offset)
raw_line = handle.readline(max_record_bytes + 1)
if not raw_line:
raise JsonlProjectionReconciliationRequired('projection issue record is absent')
if len(raw_line) > max_record_bytes:
byte_length, actual_sha256, newline_terminated = _stream_projection_record(handle, raw_line)
classification = 'oversized_record' if newline_terminated else 'unterminated_record'
raw_line = None
else:
byte_length = len(raw_line)
actual_sha256 = hashlib.sha256(raw_line).hexdigest()
if identity_key == 'error_row_id':
try:
identity = _projection_identity_from_line(raw_line, identity_key, plan_entry['path'])
if len(identity) > 256 or any(character in identity for character in ('\x00', '\r', '\n')):
raise ValueError('error projection identity exceeds its bounded format')
except (JsonlProjectionReconciliationRequired, ValueError):
classification = (
'unterminated_record' if not raw_line.endswith(b'\n')
else 'invalid_error_projection'
)
else:
raise JsonlProjectionReconciliationRequired(
'reviewed projection issue is a parseable error record and must be indexed'
)
else:
parsed_values, classification = _bounded_projection_json_values(raw_line)
if parsed_values is not None:
raise JsonlProjectionReconciliationRequired(
'reviewed projection issue contains bounded parseable JSON and must be indexed'
)
if actual_sha256 != expected_sha256:
raise JsonlProjectionReconciliationRequired('projection issue record SHA-256 does not match review')
if expected_length is not None and byte_length != int(expected_length):
raise JsonlProjectionReconciliationRequired('projection issue record length does not match review')
if expected_classification is not None and classification != str(expected_classification):
raise JsonlProjectionReconciliationRequired('projection issue classification does not match review')
return raw_line, byte_length, actual_sha256, classification
def approve_projection_reconciliation_issues(
path, identity_key, reviewed_issues, max_record_bytes=None, return_details=False,
):
"""Approve exact reviewed corrupt records without storing or changing their payloads."""
if identity_key not in ('scan_event_id', 'finding_uid', 'error_row_id'):
raise ValueError('unsupported projection reconciliation identity')
reviewed_issues = list(reviewed_issues or [])
if not reviewed_issues or len(reviewed_issues) > 100000:
raise ValueError('projection issue review count is outside its bound')
path = os.path.abspath(path)
max_record_bytes = max(
1024 * 1024,
int(max_record_bytes or getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)),
)
lock_path = f'{path}.lock'
lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec)
ledger = None
try:
plan = _projection_reconciliation_plan(path)
plan_by_name = {os.path.basename(item['path']): (index, item) for index, item in enumerate(plan)}
if len(plan_by_name) != len(plan):
raise JsonlProjectionReconciliationRequired('projection history contains ambiguous basenames')
normalized = []
seen = set()
for value in reviewed_issues:
requested_file = str(value.get('file') or value.get('basename') or '')
physical_file = os.path.basename(requested_file)
byte_offset = int(value.get('offset', value.get('byte_offset', -1)))
expected_sha256 = str(value.get('sha256') or '').strip().lower()
if (
not physical_file or physical_file != requested_file
or physical_file not in plan_by_name
or byte_offset < 0
or not re.fullmatch(r'[a-f0-9]{64}', expected_sha256)
):
raise ValueError('projection issue review metadata is invalid')
identity = (physical_file, byte_offset, expected_sha256)
if identity in seen:
raise ValueError('projection issue review contains duplicate metadata')
seen.add(identity)
normalized.append((
plan_by_name[physical_file][0],
physical_file,
byte_offset,
expected_sha256,
value.get('length', value.get('byte_length')),
value.get('classification'),
))
ledger = _open_projection_ledger(path)
details = []
classifications = Counter()
for sequence, (_, physical_file, byte_offset, expected_sha256, expected_length, expected_classification) in enumerate(sorted(normalized), 1):
plan_entry = plan_by_name[physical_file][1]
raw_line, byte_length, actual_sha256, classification = _review_projection_issue_record(
plan_entry,
identity_key,
byte_offset,
expected_sha256,
max_record_bytes,
expected_length=expected_length,
expected_classification=expected_classification,
)
issue = _record_projection_reconciliation_issue(
ledger,
identity_key,
plan_entry,
byte_offset,
raw_line,
classification,
resolved=True,
byte_length=byte_length,
record_sha256=actual_sha256,
)
classifications[classification] += 1
if return_details:
details.append({
'id': int(issue[0]), 'status': issue[1], 'file': physical_file,
'offset': byte_offset, 'length': byte_length,
'sha256': actual_sha256, 'classification': classification,
})
if sequence % 100 == 0:
ledger.commit()
ledger.commit()
report = {
'resolved_count': len(normalized),
'classifications': dict(sorted(classifications.items())),
}
if return_details:
report['details'] = details
return report
finally:
if ledger is not None:
ledger.close()
harden_private_file(jsonl_ledger_path(path))
release_file_lock(lock, lock_path)
def approve_projection_reconciliation_issue(
path, identity_key, physical_file, byte_offset, expected_sha256, max_record_bytes=None,
):
report = approve_projection_reconciliation_issues(
path,
identity_key,
[{
'file': physical_file,
'offset': byte_offset,
'sha256': expected_sha256,
}],
max_record_bytes=max_record_bytes,
return_details=True,
)
return report['details'][0]
def _apply_reviewed_projection_conflict_variant(
ledger,
handle,
plan_entry,
identity_key,
reviewed,
max_record_bytes,
):
file_name, offset, length, identity_sha256, payload_sha256, field_set_sha256 = reviewed
if offset >= int(plan_entry['size']):
raise JsonlProjectionReconciliationRequired('projection conflict offset is outside the immutable file')
if offset:
handle.seek(offset - 1)
if handle.read(1) != b'\n':
raise JsonlProjectionReconciliationRequired(
'projection conflict offset is not a record boundary'
)
handle.seek(offset)
raw_line = handle.readline(max_record_bytes + 1)
if not raw_line or len(raw_line) > max_record_bytes or len(raw_line) != length:
raise JsonlProjectionReconciliationRequired(
'projection conflict record is absent or does not match its reviewed length'
)
actual_payload_sha256 = hashlib.sha256(raw_line).hexdigest()
if identity_key == 'error_row_id':
identity = _projection_identity_from_line(raw_line, identity_key, plan_entry['path'])
actual_field_set_sha256 = _error_projection_field_name_set_sha256(raw_line)
else:
parsed_values, _ = _bounded_projection_json_values(raw_line)
if not parsed_values or len(parsed_values) != 1 or not isinstance(parsed_values[0]['value'], dict):
raise JsonlProjectionReconciliationRequired(
'projection conflict record is not one bounded JSON identity record'
)
identity = str(parsed_values[0]['value'].get(identity_key) or '')
actual_field_set_sha256 = _projection_field_name_set_sha256(parsed_values[0]['value'])
if (
not identity
or hashlib.sha256(identity.encode('utf-8')).hexdigest() != identity_sha256
or actual_payload_sha256 != payload_sha256
or actual_field_set_sha256 != field_set_sha256
):
raise JsonlProjectionReconciliationRequired(
'projection conflict record does not match reviewed identity/payload/schema hashes'
)
primary = ledger.execute(
'''SELECT state FROM publication_identity
WHERE identity_key = ? AND identity_value = ?''',
(identity_key, identity),
).fetchone()
if primary and primary[0] != 'appended':
raise JsonlProjectionReconciliationRequired(
'projection conflict review primary identity is unresolved'
)
if not primary:
row_count = _ledger_row_count(ledger)
row_limit = max(1, int(getattr(scan_config, 'jsonl_ledger_max_rows', 1000000)))
if row_count >= row_limit:
raise JsonlProjectionReconciliationRequired(
f'JSONL identity ledger reached its {row_limit} row bound'
)
now = time.time()
ledger.execute(
'''INSERT INTO publication_identity (
identity_key, identity_value, payload_sha256, state, file_name,
byte_offset, byte_length, created_at, updated_at
) VALUES (?, ?, ?, 'appended', ?, ?, ?, ?, ?)''',
(
identity_key, identity, payload_sha256, file_name,
offset, length, now, now,
),
)
_set_ledger_row_count(ledger, row_count + 1)
ledger.execute(
'''INSERT OR IGNORE INTO publication_identity_variant (
identity_key, identity_value, payload_sha256, file_name,
byte_offset, byte_length, created_at
) VALUES (?, ?, ?, ?, ?, ?, ?)''',
(identity_key, identity, payload_sha256, file_name, offset, length, time.time()),
)
_record_projection_variant_issue(
ledger,
identity_key,
identity,
plan_entry,
offset,
length,
payload_sha256,
field_set_sha256,
resolved=True,
)
def approve_projection_conflict_variants(
path,
identity_key,
reviewed_variants,
max_record_bytes=None,
commit_batch_size=250,
progress_callback=None,
):
if identity_key not in ('scan_event_id', 'finding_uid', 'error_row_id'):
raise ValueError('unsupported projection conflict identity')
reviewed_variants = list(reviewed_variants or [])
if not reviewed_variants or len(reviewed_variants) > 100000:
raise ValueError('projection conflict review count is outside its bound')
path = os.path.abspath(path)
max_record_bytes = max(
1024 * 1024,
int(max_record_bytes or getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)),
)
commit_batch_size = min(1000, max(1, int(commit_batch_size or 250)))
lock_path = f'{path}.lock'
lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec)
ledger = None
try:
plan = _projection_reconciliation_plan(path)
plan_by_name = {os.path.basename(item['path']): (index, item) for index, item in enumerate(plan)}
if len(plan_by_name) != len(plan):
raise JsonlProjectionReconciliationRequired('projection history contains ambiguous basenames')
normalized = []
seen = set()
for value in reviewed_variants:
file_name = str(value.get('file') or '')
offset = int(value.get('offset', -1))
length = int(value.get('length', 0))
identity_sha256 = str(value.get('identity_sha256') or '').lower()
payload_sha256 = str(value.get('payload_sha256') or '').lower()
field_set_sha256 = str(value.get('field_name_set_sha256') or '').lower()
if (
not file_name or os.path.basename(file_name) != file_name
or file_name not in plan_by_name or offset < 0 or length <= 0
or not re.fullmatch(r'[a-f0-9]{64}', identity_sha256)
or not re.fullmatch(r'[a-f0-9]{64}', payload_sha256)
or not re.fullmatch(r'[a-f0-9]{64}', field_set_sha256)
or value.get('classification') != 'historical_payload_variant'
):
raise ValueError('projection conflict review metadata is invalid')
key = (file_name, offset, identity_sha256, payload_sha256)
if key in seen:
raise ValueError('projection conflict review contains duplicate metadata')
seen.add(key)
normalized.append((
plan_by_name[file_name][0], file_name,
(file_name, offset, length, identity_sha256, payload_sha256, field_set_sha256),
))
ledger = _open_projection_ledger(path)
resolved_rows = ledger.execute(
'''SELECT identity_sha256, file_name, byte_offset, byte_length,
payload_sha256, field_name_set_sha256
FROM reconciliation_variant_issue
WHERE identity_key = ? AND status = 'resolved' ''',
(identity_key,),
).fetchall()
resolved = {
(row[1], int(row[2]), row[0], row[4]): (int(row[3]), row[5])
for row in resolved_rows
}
pending = []
already_resolved = 0
for plan_index, file_name, reviewed in sorted(normalized):
key = (file_name, reviewed[1], reviewed[3], reviewed[4])
expected = resolved.get(key)
if expected is not None:
if expected != (reviewed[2], reviewed[5]):
raise JsonlProjectionReconciliationRequired(
'resolved projection conflict metadata does not match this review manifest'
)
already_resolved += 1
continue
pending.append((plan_index, file_name, reviewed))
if progress_callback is not None:
progress_callback({
'reviewed': len(normalized),
'already_resolved': already_resolved,
'newly_resolved': 0,
})
newly_resolved = 0
grouped = {}
for plan_index, file_name, reviewed in pending:
grouped.setdefault((plan_index, file_name), []).append(reviewed)
for plan_index, file_name in sorted(grouped):
plan_entry = plan_by_name[file_name][1]
with open(plan_entry['path'], 'rb') as handle:
for reviewed in sorted(grouped[(plan_index, file_name)], key=lambda item: item[1]):
_apply_reviewed_projection_conflict_variant(
ledger, handle, plan_entry, identity_key, reviewed, max_record_bytes,
)
newly_resolved += 1
if newly_resolved % commit_batch_size == 0:
ledger.commit()
if progress_callback is not None:
progress_callback({
'reviewed': len(normalized),
'already_resolved': already_resolved,
'newly_resolved': newly_resolved,
})
ledger.commit()
if progress_callback is not None and newly_resolved % commit_batch_size:
progress_callback({
'reviewed': len(normalized),
'already_resolved': already_resolved,
'newly_resolved': newly_resolved,
})
return {
'resolved_variant_count': len(normalized),
'already_resolved': already_resolved,
'newly_resolved': newly_resolved,
}
finally:
if ledger is not None:
ledger.close()
harden_private_file(jsonl_ledger_path(path))
release_file_lock(lock, lock_path)
def _files_share_prefix(segment_path, current_path, size):
if size <= 0 or not os.path.isfile(current_path) or os.path.getsize(current_path) < size:
return False
remaining = size
with open(segment_path, 'rb') as segment, open(current_path, 'rb') as current:
while remaining:
amount = min(1024 * 1024, remaining)
left = segment.read(amount)
right = current.read(amount)
if left != right or not left:
return False
remaining -= len(left)
return remaining == 0
def _publish_empty_jsonl_generation(path):
temporary = (
f'{path}.{os.getpid()}.{threading.get_ident()}.'
f'{uuid.uuid4().hex}.empty.tmp'
)
try:
with open(temporary, 'xb') as handle:
handle.flush()
os.fsync(handle.fileno())
harden_private_file(temporary)
durable_replace(temporary, path)
harden_private_file(path)
finally:
if os.path.exists(temporary):
os.remove(temporary)
def _remove_file_prefix(path, size):
current_size = os.path.getsize(path)
if size <= 0 or current_size < size:
return
if current_size == size:
_publish_empty_jsonl_generation(path)
return
temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.prefix.tmp'
with open(path, 'rb') as source, open(temporary, 'wb') as destination:
source.seek(size)
shutil.copyfileobj(source, destination, 1024 * 1024)
destination.flush()
os.fsync(destination.fileno())
harden_private_file(temporary)
durable_replace(temporary, path)
harden_private_file(path)
def _segment_manifest_entry(path, sequence):
return {
'name': os.path.basename(path),
'path': os.path.abspath(path),
'bytes': int(os.path.getsize(path)),
'closed_at': datetime.now(timezone.utc).isoformat(timespec='seconds'),
'sequence': int(sequence),
}
def _jsonl_file_generation(path):
details = os.stat(path, follow_symlinks=False)
return {
'device': int(getattr(details, 'st_dev', 0) or 0),
'size': int(details.st_size),
'mtime_ns': int(getattr(details, 'st_mtime_ns', int(details.st_mtime * 1_000_000_000))),
'inode': int(getattr(details, 'st_ino', 0) or 0),
}
def _manifest_skip_matches_generation(path, manifest, skip_bytes):
signature = manifest.get('current_skip_signature') if isinstance(manifest, dict) else None
if not isinstance(signature, dict) or not os.path.isfile(path):
return False
try:
current = _jsonl_file_generation(path)
return (
int(skip_bytes) <= current['size']
and all(int(current[key]) == int(signature.get(key, -1)) for key in ('device', 'size', 'mtime_ns', 'inode'))
)
except (OSError, TypeError, ValueError):
return False
def reconcile_jsonl_segments(path):
"""Publish physical orphan segments before any active-file mutation."""
manifest = load_jsonl_manifest(path)
physical = _projection_segment_sequence(path)
existing = {
str(item.get('name') or os.path.basename(str(item.get('path') or ''))): item
for item in (manifest.get('segments') or []) if isinstance(item, dict)
}
normalized = []
missing = []
for sequence, segment_path in physical:
item = existing.get(os.path.basename(segment_path))
if item is None:
item = _segment_manifest_entry(segment_path, sequence)
missing.append((sequence, segment_path))
else:
item = dict(item, path=os.path.abspath(segment_path), name=os.path.basename(segment_path), sequence=sequence)
normalized.append(item)
skip_bytes = max(0, int(manifest.get('current_skip_bytes') or 0))
duplicate_path = None
duplicate_size = 0
candidates = list(reversed(missing))
if skip_bytes and physical:
candidates.insert(0, physical[-1])
for _, segment_path in candidates:
size = os.path.getsize(segment_path)
if _files_share_prefix(segment_path, path, size):
duplicate_path = segment_path
duplicate_size = size
break
changed = bool(missing) or normalized != (manifest.get('segments') or [])
effective_skip = duplicate_size if duplicate_path else 0
if changed or skip_bytes or manifest.get('current_skip_signature'):
next_sequence = max([sequence for sequence, _ in physical] or [0]) + 1
manifest.update({
'current': os.path.basename(path),
'current_path': os.path.abspath(path),
'next_sequence': next_sequence,
'segments': normalized,
'current_skip_bytes': effective_skip,
'current_skip_signature': _jsonl_file_generation(path) if effective_skip and os.path.isfile(path) else None,
'updated_at': datetime.now(timezone.utc).isoformat(timespec='seconds'),
})
write_jsonl_manifest(path, manifest)
if duplicate_path:
_remove_file_prefix(path, duplicate_size)
manifest['current_skip_bytes'] = 0
manifest['current_skip_signature'] = None
manifest['updated_at'] = datetime.now(timezone.utc).isoformat(timespec='seconds')
write_jsonl_manifest(path, manifest)
return manifest
def _segment_consumed_by_registered_keychecks(segment_path):
if os.path.basename(segment_path).lower().startswith('scan_results.'):
return True
keycheck_root = getattr(scan_config, 'keycheck_dir', '') or ''
if not os.path.isdir(keycheck_root):
return False
states = []
consumer_directories = 0
with os.scandir(keycheck_root) as entries:
for entry in entries:
if not entry.is_dir(follow_symlinks=False):
continue
consumer_directories += 1
if consumer_directories > 128:
return False
state_path = os.path.join(entry.path, 'input_state.json')
if not os.path.isfile(state_path):
return False
try:
if os.path.getsize(state_path) > 1024 * 1024:
return False
with open(state_path, 'r', encoding='utf-8') as handle:
states.append(json.load(handle))
except (OSError, ValueError):
return False
if not states or len(states) != consumer_directories:
return False
absolute = os.path.abspath(segment_path)
size = os.path.getsize(segment_path)
for state in states:
files = state.get('files') if isinstance(state, dict) and isinstance(state.get('files'), dict) else {}
record = files.get(absolute) or files.get(segment_path)
if not isinstance(record, dict) or int(record.get('offset', 0) or 0) < size:
return False
return True
def _prune_consumed_jsonl_segments(path, manifest, keep_below):
physical = _projection_segment_sequence(path)
removed = set()
while len(physical) >= keep_below and physical:
_, candidate = physical[0]
if not _segment_consumed_by_registered_keychecks(candidate):
break
reject_reparse_components(candidate)
os.remove(candidate)
removed.add(os.path.basename(candidate))
physical.pop(0)
if removed:
manifest['segments'] = [
item for item in (manifest.get('segments') or [])
if str(item.get('name') or os.path.basename(str(item.get('path') or ''))) not in removed
]
manifest['updated_at'] = datetime.now(timezone.utc).isoformat(timespec='seconds')
write_jsonl_manifest(path, manifest)
return physical
def rotate_jsonl_if_needed(path, max_bytes, ledger=None):
if not max_bytes or max_bytes <= 0:
return
if not os.path.lexists(path):
return
reject_reparse_components(path)
details = os.stat(path, follow_symlinks=False)
if not stat.S_ISREG(details.st_mode) or details.st_size < max_bytes:
return
repair_jsonl_tail(path)
manifest = reconcile_jsonl_segments(path)
max_segments = max(1, int(getattr(scan_config, 'jsonl_max_segments', 16)))
physical = _prune_consumed_jsonl_segments(path, manifest, max_segments)
if len(physical) >= max_segments:
raise JsonlProjectionReconciliationRequired(
f'JSONL segment retention bound reached for {path}; registered consumers must catch up '
'or offline reconciliation must retire acknowledged segments'
)
segment_path, seq = next_jsonl_segment_path(path, manifest)
size = os.path.getsize(path)
temporary = f'{segment_path}.{os.getpid()}.{threading.get_ident()}.tmp'
try:
with open(path, 'rb') as source, open(temporary, 'xb') as destination:
shutil.copyfileobj(source, destination, 1024 * 1024)
destination.flush()
os.fsync(destination.fileno())
harden_private_file(temporary)
os.replace(temporary, segment_path)
finally:
if os.path.exists(temporary):
os.remove(temporary)
harden_private_file(segment_path)
segments = manifest.get('segments') if isinstance(manifest.get('segments'), list) else []
segments.append(_segment_manifest_entry(segment_path, seq))
manifest.update({
'current': os.path.basename(path),
'current_path': path,
'next_sequence': seq + 1,
'max_bytes': int(max_bytes),
'segments': segments,
'current_skip_bytes': int(size),
'current_skip_signature': _jsonl_file_generation(path),
'updated_at': datetime.now(timezone.utc).isoformat(timespec='seconds'),
})
write_jsonl_manifest(path, manifest)
if ledger is not None:
ledger.execute(
"UPDATE publication_identity SET file_name = ? WHERE file_name = ? AND state = 'appended'",
(os.path.basename(segment_path), os.path.basename(path)),
)
ledger.execute(
"UPDATE publication_identity_variant SET file_name = ? WHERE file_name = ?",
(os.path.basename(segment_path), os.path.basename(path)),
)
ledger.commit()
_publish_empty_jsonl_generation(path)
manifest['current_skip_bytes'] = 0
manifest['current_skip_signature'] = None
manifest['updated_at'] = datetime.now(timezone.utc).isoformat(timespec='seconds')
write_jsonl_manifest(path, manifest)
logger.info(f'Rotated JSONL {path} -> {segment_path} ({size} bytes)')
def append_rotating_jsonl(path, payload, max_mb=None):
if not scan_config.jsonl_rotation_enabled:
return append_jsonl(path, payload)
max_bytes = int(max_mb or 0) * 1024 * 1024
serialized = json.dumps(payload, ensure_ascii=False, default=str) + '\n'
lock_path = f'{path}.lock'
fd = None
try:
parent = os.path.dirname(path)
if parent:
require_private_directory(parent, create=True)
fd = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec)
repair_jsonl_tail(path)
rotate_jsonl_if_needed(path, max_bytes)
with open(path, 'ab') as f:
f.write(serialized.encode('utf-8'))
f.flush()
os.fsync(f.fileno())
harden_private_file(path)
return True
except Exception as e:
logger.error(f'Unable to rotating-write {path}: {str(e)}')
return False
finally:
if fd is not None:
release_file_lock(fd, lock_path)
def projection_segment_paths(path):
paths = [segment_path for _, segment_path in _projection_segment_sequence(path)]
if os.path.isfile(path):
paths.append(path)
return paths
def jsonl_projection_contains(path, identity_key, identity_value):
expected = str(identity_value or '')
if not expected:
return False
ledger_path = jsonl_ledger_path(path)
if not os.path.isfile(ledger_path) or not private_file_ready(ledger_path):
return False
connection = sqlite3.connect(f'file:{ledger_path}?mode=ro', uri=True, timeout=5)
try:
row = connection.execute(
'''SELECT state FROM publication_identity
WHERE identity_key = ? AND identity_value = ?''',
(identity_key, expected),
).fetchone()
return bool(row and row[0] == 'appended')
finally:
connection.close()
def _append_projection_once_locked(
path, serialized, identity_key, identity_value, max_bytes, rotation_enabled=None,
):
repair_jsonl_tail(path)
reconcile_jsonl_segments(path)
if len(identity_key) > 64 or len(identity_value) > 256 or any(
character in identity_value for character in ('\x00', '\r', '\n')
):
raise JsonlProjectionReconciliationRequired('projection identity exceeds its bounded format')
max_record_bytes = max(1, int(getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)))
if len(serialized) > max_record_bytes:
raise JsonlProjectionReconciliationRequired('projection record exceeds its per-record byte bound')
ledger = _open_projection_ledger(path)
try:
_ensure_projection_ledger_bootstrapped(ledger, path, identity_key)
_recover_prepared_publications(ledger, path)
digest = hashlib.sha256(serialized).hexdigest()
row = ledger.execute(
'''SELECT payload_sha256, state FROM publication_identity
WHERE identity_key = ? AND identity_value = ?''',
(identity_key, identity_value),
).fetchone()
if row:
if row[1] != 'appended':
raise JsonlProjectionReconciliationRequired('publication ledger retained an unresolved append state')
variant = ledger.execute(
'''SELECT 1 FROM publication_identity_variant
WHERE identity_key = ? AND identity_value = ? AND payload_sha256 = ?''',
(identity_key, identity_value, digest),
).fetchone()
if variant:
return True
raise JsonlProjectionReconciliationRequired(
f'{identity_key} {identity_value} has a conflicting projection payload'
)
row_limit = max(1, int(getattr(scan_config, 'jsonl_ledger_max_rows', 1000000)))
ledger_byte_limit = max(1024 * 1024, int(getattr(scan_config, 'jsonl_ledger_max_bytes', 512 * 1024 * 1024)))
row_count = _ledger_row_count(ledger)
if row_count >= row_limit:
raise JsonlProjectionReconciliationRequired(
f'JSONL identity ledger reached its {row_limit} row bound; run offline reconciliation'
)
ledger_file = jsonl_ledger_path(path)
if os.path.getsize(ledger_file) + 8192 > ledger_byte_limit:
raise JsonlProjectionReconciliationRequired(
f'JSONL identity ledger reached its {ledger_byte_limit} byte bound; run offline reconciliation'
)
should_rotate = scan_config.jsonl_rotation_enabled if rotation_enabled is None else bool(rotation_enabled)
if should_rotate:
rotate_jsonl_if_needed(path, max_bytes, ledger=ledger)
if not os.path.exists(path):
with open(path, 'ab'):
pass
harden_private_file(path)
offset = os.path.getsize(path)
now = time.time()
ledger.execute(
'''INSERT INTO publication_identity (
identity_key, identity_value, payload_sha256, state, file_name,
byte_offset, byte_length, created_at, updated_at
) VALUES (?, ?, ?, 'prepared', ?, ?, ?, ?, ?)''',
(
identity_key, identity_value, digest, os.path.basename(path),
offset, len(serialized), now, now,
),
)
_set_ledger_row_count(ledger, row_count + 1)
ledger.commit()
with open(path, 'ab') as handle:
handle.write(serialized)
handle.flush()
os.fsync(handle.fileno())
harden_private_file(path)
ledger.execute(
'''UPDATE publication_identity SET state = 'appended', updated_at = ?
WHERE identity_key = ? AND identity_value = ? AND state = 'prepared' ''',
(time.time(), identity_key, identity_value),
)
ledger.execute(
'''INSERT INTO publication_identity_variant (
identity_key, identity_value, payload_sha256, file_name,
byte_offset, byte_length, created_at
) VALUES (?, ?, ?, ?, ?, ?, ?)''',
(
identity_key, identity_value, digest, os.path.basename(path),
offset, len(serialized), time.time(),
),
)
ledger.commit()
return True
finally:
ledger.close()
if os.path.exists(jsonl_ledger_path(path)):
harden_private_file(jsonl_ledger_path(path))
def append_rotating_jsonl_once(path, payload, identity_key, identity_value, max_mb=None):
if not identity_value:
logger.error(f'Unable to publish {path}: missing {identity_key}')
return False
max_bytes = int(max_mb or 0) * 1024 * 1024
lock_path = f'{path}.lock'
lock = None
try:
require_private_directory(os.path.dirname(os.path.abspath(path)), create=True)
lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec)
serialized = (json.dumps(payload, ensure_ascii=False, default=str) + '\n').encode('utf-8')
return _append_projection_once_locked(
path, serialized, identity_key, str(identity_value), max_bytes,
)
except Exception as exc:
logger.error(f'Unable to idempotently publish {path}: {exc}')
return False
finally:
if lock is not None:
release_file_lock(lock, lock_path)
def _rotate_bounded_text_log(path, keep, max_bytes=None):
if not os.path.exists(path) or os.path.getsize(path) <= 0:
return
if max_bytes and os.path.getsize(path) > int(max_bytes):
with open(path, 'r+b') as handle:
handle.truncate(int(max_bytes))
handle.flush()
os.fsync(handle.fileno())
segment, _ = next_jsonl_segment_path(path, {})
os.replace(path, segment)
harden_private_file(segment)
with open(path, 'ab'):
pass
harden_private_file(path)
segments = [candidate for candidate in projection_segment_paths(path) if candidate != path]
segments.sort(key=lambda candidate: os.path.getmtime(candidate), reverse=True)
for candidate in segments[max(0, int(keep or 0)):]:
reject_reparse_components(candidate)
os.remove(candidate)
def append_scan_errors_once(path, result, max_mb=None, keep=5):
event_id = str(result.get('scan_event_id') or '')
if not event_id:
logger.error(f'Unable to publish {path}: missing scan_event_id')
return False
errors = list(result.get('errors') or [])
if not errors:
return True
max_bytes = max(1, int(max_mb or 0) * 1024 * 1024)
timestamp = result.get('timestamp') or result.get('scan_started_at') or event_id
scan_type = result.get('scan_type', '')
target = result.get('target', '')
lock_path = f'{path}.lock'
lock = None
try:
require_private_directory(os.path.dirname(os.path.abspath(path)), create=True)
lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec)
repair_jsonl_tail(path)
for index, error in enumerate(errors, 1):
row_id = f'{event_id}:{index}'
line = f'{row_id}\t{timestamp}\t{scan_type}\t{target}\t{error}\n'.encode('utf-8', errors='replace')
if len(line) > max_bytes:
suffix = b'...[truncated]\n'
line = line[:max(0, max_bytes - len(suffix))] + suffix
if os.path.exists(path) and os.path.getsize(path) + len(line) > max_bytes:
_rotate_bounded_text_log(path, keep, max_bytes)
_append_projection_once_locked(
path, line, 'error_row_id', row_id, 0, rotation_enabled=False,
)
return True
except Exception as exc:
logger.error(f'Unable to idempotently publish {path}: {exc}')
return False
finally:
if lock is not None:
release_file_lock(lock, lock_path)
def parse_finding_datetime(value):
if not value:
return None
value = str(value).strip()
for date_format in ("%Y-%m-%d %H:%M:%S %z", "%Y-%m-%dT%H:%M:%SZ", "%Y-%m-%dT%H:%M:%S%z"):
try:
return datetime.strptime(value, date_format)
except ValueError:
continue
return None
def get_finding_timestamp(finding):
data = finding.get('SourceMetadata', {}).get('Data', {})
if not isinstance(data, dict):
return None
for source in data.values():
if isinstance(source, dict) and source.get('timestamp'):
return parse_finding_datetime(source.get('timestamp'))
return None
def filter_findings_by_age(findings, max_age_days=None):
if not max_age_days or max_age_days <= 0:
return findings, 0, 0
cutoff = datetime.now().astimezone() - timedelta(days=max_age_days)
kept = []
skipped_old = 0
skipped_missing = 0
for finding in findings:
timestamp = get_finding_timestamp(finding)
if timestamp is None:
skipped_missing += 1
continue
if timestamp >= cutoff:
kept.append(finding)
else:
skipped_old += 1
return kept, skipped_old, skipped_missing
def filter_dropped_detectors(findings):
drop = {
item.strip().lower()
for item in csv_items(_scan_policy_value('drop_detectors', []))
if item.strip()
}
if not drop:
return findings, 0, Counter()
kept = []
counts = Counter()
skipped = 0
for finding in findings or []:
detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '')
if detector.lower() in drop:
skipped += 1
counts[detector or '(unknown)'] += 1
continue
kept.append(finding)
return kept, skipped, counts
GITHUB_TOKEN_PREFIXES = ('ghp_', 'gho_', 'ghu_', 'ghs_', 'ghr_', 'github_pat_')
GITLAB_TOKEN_PREFIXES = ('glpat-', 'gloas-', 'glcbt-', 'glimt-', 'glrt-', 'glft-', 'glsoat-')
def finding_raw_values(finding):
values = []
for key in ('Raw', 'RawV2'):
value = finding.get(key)
if value:
values.append(str(value))
return values
def is_known_provider_token_shape(detector_name, finding):
values = finding_raw_values(finding)
detector = str(detector_name or '').lower()
if detector in ('github', 'githuboauth2'):
return any(value.startswith(GITHUB_TOKEN_PREFIXES) for value in values)
if detector == 'gitlab':
return any(value.startswith(GITLAB_TOKEN_PREFIXES) for value in values)
return True
def filter_noisy_findings(findings):
if not _scan_policy_value('strict_git_provider_token_filter', True):
return findings, 0
kept = []
skipped = 0
for finding in findings:
detector = finding.get('DetectorName') or ''
detector_key = str(detector).lower()
if detector_key in ('github', 'githuboauth2', 'gitlab') and not finding.get('Verified', False):
if not is_known_provider_token_shape(detector, finding):
skipped += 1
continue
kept.append(finding)
return kept, skipped
CUSTOM_DETECTOR_NAME_ALIASES = {
'xaicontextafter': 'Xai',
'zaiglmcontextafter': 'ZaiGLM',
}
def normalize_custom_detector_names(findings):
for finding in findings or []:
detector = str(finding.get('DetectorName') or '')
extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {}
custom_name = str(extra.get('name') or '').strip()
if detector.lower() == 'customregex' and custom_name:
finding.setdefault('OriginalDetectorName', detector)
finding['DetectorName'] = CUSTOM_DETECTOR_NAME_ALIASES.get(
custom_name.lower(), custom_name,
)
return findings
def apply_finding_filters(results, target_label='target', *, log_target=True):
emit_client_scan_phase('filtering')
findings = results.get('findings') or []
normalize_custom_detector_names(findings)
findings, dropped, dropped_counts = filter_dropped_detectors(findings)
if dropped:
results['findings'] = findings
results['dropped_detectors_count'] = int(results.get('dropped_detectors_count', 0) or 0) + dropped
results['dropped_detectors'] = dict(dropped_counts)
target_context = f' for {target_label}' if log_target else ''
logger.info(
f"Dropped {dropped} configured noise detector finding(s)"
f"{target_context}: {dict(dropped_counts)}"
)
filtered, skipped = filter_noisy_findings(findings)
if skipped:
results['findings'] = filtered
results['filtered_findings_count'] = int(results.get('filtered_findings_count', 0) or 0) + skipped
target_context = f' for {target_label}' if log_target else ''
logger.info(
f"Filtered {skipped} noisy unverified Git provider finding(s)"
f"{target_context}"
)
return results
def finding_source_location(finding):
metadata = finding.get('SourceMetadata') or {}
data = metadata.get('Data') if isinstance(metadata, dict) else {}
if not isinstance(data, dict):
return None, None
for source in data.values():
if not isinstance(source, dict):
continue
file_path = source.get('file') or source.get('path') or source.get('File')
line = source.get('line') or source.get('Line')
if file_path:
try:
line = int(line) if line else None
except (TypeError, ValueError):
line = None
return file_path, line
return None, None
def context_enrichment_budget():
return {
'started_at': time.monotonic(),
'max_source_bytes': max(0, int(getattr(scan_config, 'context_enrichment_max_source_bytes', 16 * 1024 * 1024))),
'max_findings': max(0, int(getattr(scan_config, 'context_enrichment_max_findings', 2000))),
'max_postman_comparisons': max(0, int(getattr(scan_config, 'context_enrichment_max_postman_comparisons', 200000))),
'max_elapsed_sec': max(0.0, float(getattr(scan_config, 'context_enrichment_max_elapsed_sec', 5.0))),
'source_bytes': 0,
'postman_comparisons': 0,
'finding_ids': set(),
'files': {},
}
def _context_budget_expired(budget):
return time.monotonic() - budget['started_at'] >= budget['max_elapsed_sec']
def _context_budget_claim_finding(budget, finding):
identity = id(finding)
if identity in budget['finding_ids']:
return True
if len(budget['finding_ids']) >= budget['max_findings']:
return False
budget['finding_ids'].add(identity)
return True
def _add_context_warning(results, warning_class, detail):
warning_class = f'context_enrichment_{warning_class}'
classes = list(results.get('warning_classes') or [])
if warning_class not in classes:
warnings = list(results.get('warnings') or [])
warnings.append(f'Optional context enrichment degraded: {detail}'[:400])
results['warnings'] = warnings
classes.append(warning_class)
results['warning_classes'] = sorted(set(classes))
results['context_enrichment_degraded'] = True
results['degraded'] = True
def _read_context_source(file_path, budget):
key = os.path.normcase(os.path.abspath(os.fspath(file_path)))
if key in budget['files']:
return budget['files'][key]
if _context_budget_expired(budget):
return None, False, 'elapsed'
try:
size = max(0, int(os.path.getsize(file_path)))
remaining = max(0, budget['max_source_bytes'] - budget['source_bytes'])
if remaining <= 0 and size:
return None, False, 'source_bytes'
amount = min(size, remaining)
with open(file_path, 'rb') as handle:
payload = handle.read(amount)
budget['source_bytes'] += len(payload)
complete = len(payload) == size
reason = None if complete else 'source_bytes' if size > remaining else 'read'
budget['files'][key] = (payload, complete, reason)
return payload, complete, reason
except (OSError, TypeError, ValueError):
return None, False, 'read'
def _nearby_context_from_lines(lines, file_path, line_number=None, radius=20, max_chars=12000):
if not lines:
return None
if line_number and line_number > 0:
start = max(0, line_number - radius - 1)
requested_end = line_number + radius
else:
start = 0
requested_end = radius * 2 + 1
if start >= len(lines):
return None
end = min(len(lines), requested_end)
return {
'file': file_path,
'line': line_number,
'start_line': start + 1,
'end_line': end,
'nearby': ''.join(lines[start:end])[:max_chars],
}
def read_nearby_context(file_path, line_number=None, radius=20, max_chars=12000):
if not file_path or not os.path.exists(file_path):
return None
if line_number and line_number > 0:
start = max(0, line_number - radius - 1)
requested_end = line_number + radius
else:
start = 0
requested_end = radius * 2 + 1
selected = []
actual_end = 0
try:
with open(file_path, 'r', encoding='utf-8', errors='replace') as f:
for index in range(requested_end):
line = f.readline(max_chars + 1)
if not line:
break
if not line.endswith('\n') and len(line) > max_chars:
while True:
remainder = f.readline(64 * 1024)
if not remainder or remainder.endswith('\n'):
break
actual_end = index + 1
if index >= start and sum(len(value) for value in selected) < max_chars:
selected.append(line[:max_chars])
except OSError:
return None
if actual_end == 0:
return None
snippet = ''.join(selected)[:max_chars]
return {
'file': file_path,
'line': line_number,
'start_line': start + 1,
'end_line': actual_end,
'nearby': snippet,
}
def attach_nearby_context(results, budget=None):
budget = budget or context_enrichment_budget()
grouped = {}
finding_budget_exhausted = False
try:
for finding in results.get('findings') or []:
if _context_budget_expired(budget):
_add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained')
return results
if not isinstance(finding, dict):
continue
file_path, line_number = finding_source_location(finding)
if not file_path:
continue
if not _context_budget_claim_finding(budget, finding):
finding_budget_exhausted = True
break
key = os.path.normcase(os.path.abspath(os.fspath(file_path)))
grouped.setdefault(key, {'path': file_path, 'findings': []})['findings'].append((finding, line_number))
for group in grouped.values():
if _context_budget_expired(budget):
_add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained')
return results
payload, complete, reason = _read_context_source(group['path'], budget)
if payload is None:
if reason in ('elapsed', 'source_bytes'):
dimension = 'elapsed-time' if reason == 'elapsed' else 'source-byte'
_add_context_warning(results, 'budget', f'{dimension} budget was exhausted; remaining findings were retained')
return results
_add_context_warning(results, 'failure', 'a nearby source file could not be read; findings were retained')
continue
lines = payload.decode('utf-8', errors='replace').splitlines(keepends=True)
for finding, line_number in group['findings']:
if _context_budget_expired(budget):
_add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained')
return results
context = _nearby_context_from_lines(lines, group['path'], line_number)
if context:
finding['ScannerContext'] = context
if not complete:
if reason == 'source_bytes':
_add_context_warning(results, 'budget', 'source-byte budget was exhausted; remaining findings were retained')
return results
_add_context_warning(results, 'failure', 'a nearby source file changed during its bounded read; findings were retained')
if finding_budget_exhausted:
_add_context_warning(results, 'budget', 'finding-count budget was exhausted; remaining findings were retained')
except Exception:
logger.warning('Optional nearby context enrichment failed; parsed findings were retained')
_add_context_warning(results, 'failure', 'nearby context parsing failed; findings were retained')
return results
QWEN_ROUTING_CONTEXT_RE = re.compile(
r'(?:DASHSCOPE_API_KEY|QWEN_API_KEY|dashscope|qwen|model[_-]?studio|bailian)',
re.IGNORECASE,
)
DEEPSEEK_ROUTING_CONTEXT_RE = re.compile(
r'(?:DEEPSEEK_API_KEY|deepseek|api\.deepseek\.com)',
re.IGNORECASE,
)
KIMI_ROUTING_CONTEXT_RE = re.compile(
r'(?:MOONSHOT_API_KEY|KIMI_API_KEY|api\.moonshot\.(?:ai|cn)|platform\.kimi\.(?:ai|com))',
re.IGNORECASE,
)
ZAI_ROUTING_CONTEXT_RE = re.compile(
r'(?:ZAI_API_KEY|GLM_API_KEY|ZHIPUAI_API_KEY|BIGMODEL_API_KEY|api\.z\.ai|'
r'open\.bigmodel\.cn|zhipuai|chatglm)',
re.IGNORECASE,
)
QWEN_ROUTING_DETECTORS = {'qwendashscope', 'qwen_dashscope', 'dashscope', 'qwen'}
QWEN_EXPLICIT_ROUTING_DETECTORS = {'qwendashscope', 'qwen_dashscope'}
DEEPSEEK_ROUTING_DETECTORS = {'deepseek', 'deepseekapikey', 'deepseek_api_key'}
DEEPSEEK_EXPLICIT_ROUTING_DETECTORS = {'deepseekapikey', 'deepseek_api_key'}
KIMI_ROUTING_DETECTORS = {'kimimoonshot', 'moonshotai', 'moonshot', 'kimi'}
KIMI_EXPLICIT_ROUTING_DETECTORS = {'kimimoonshot', 'moonshotai'}
ZAI_ROUTING_DETECTORS = {'zaiglm'}
ZAI_EXPLICIT_ROUTING_DETECTORS = {'zaiglm'}
AMBIGUOUS_QWEN_DEEPSEEK_HINT = 'ambiguous_qwen_deepseek'
AMBIGUOUS_GENERIC_SK_HINT = 'ambiguous_generic_sk'
GENERIC_SK_PROVIDERS = {'qwen', 'deepseek', 'kimi', 'zai'}
GENERIC_SK_PROVIDER_ORDER = ('deepseek', 'zai', 'qwen', 'kimi')
GENERIC_SK_PROVIDER_HINTS = {
*GENERIC_SK_PROVIDERS, AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT,
}
EXPLICIT_ASSIGNMENT_PROVIDER_HINT_SOURCE = 'explicit_assignment'
GENERIC_SK_ROUTING_RE = re.compile(r'^sk-[A-Za-z0-9][A-Za-z0-9_-]{20,505}$')
QWEN_SPECIFIC_ROUTING_RE = re.compile(r'^sk-sp-[A-Za-z0-9][A-Za-z0-9_-]{20,505}$')
ZAI_SPECIFIC_ROUTING_RE = re.compile(
r'^(?:zai-[A-Za-z0-9][A-Za-z0-9_-]{20,505}|'
r'[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{20,505})$'
)
def provider_routing_context_text(finding, max_context_chars=65536):
if not isinstance(finding, dict):
return ''
context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {}
parts = []
remaining = max(0, int(max_context_chars))
def add(value):
nonlocal remaining
if not value or remaining <= 0:
return
text = str(value)[:remaining]
parts.append(text)
remaining -= len(text)
for key in ('nearby', 'file'):
add(context.get(key))
metadata = finding.get('SourceMetadata') if isinstance(finding.get('SourceMetadata'), dict) else {}
data = metadata.get('Data') if isinstance(metadata.get('Data'), dict) else {}
for details in data.values():
if not isinstance(details, dict):
continue
for key in ('file', 'repository', 'repo', 'link', 'image'):
add(details.get(key))
return '\n'.join(parts)
def provider_routing_detector_names(finding):
if not isinstance(finding, dict):
return set()
detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '').strip().lower()
extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {}
custom_name = str(extra.get('name') or '').strip().lower()
return {name for name in (detector, custom_name) if name}
def is_qwen_routing_detector(finding):
return bool(provider_routing_detector_names(finding) & QWEN_ROUTING_DETECTORS)
def is_explicit_qwen_routing_detector(finding):
return bool(provider_routing_detector_names(finding) & QWEN_EXPLICIT_ROUTING_DETECTORS)
def is_deepseek_routing_detector(finding):
return bool(provider_routing_detector_names(finding) & DEEPSEEK_ROUTING_DETECTORS)
def is_explicit_deepseek_routing_detector(finding):
return bool(provider_routing_detector_names(finding) & DEEPSEEK_EXPLICIT_ROUTING_DETECTORS)
def is_kimi_routing_detector(finding):
return bool(provider_routing_detector_names(finding) & KIMI_ROUTING_DETECTORS)
def is_explicit_kimi_routing_detector(finding):
return bool(provider_routing_detector_names(finding) & KIMI_EXPLICIT_ROUTING_DETECTORS)
def is_zai_routing_detector(finding):
return bool(provider_routing_detector_names(finding) & ZAI_ROUTING_DETECTORS)
def is_explicit_zai_routing_detector(finding):
return bool(provider_routing_detector_names(finding) & ZAI_EXPLICIT_ROUTING_DETECTORS)
def is_generic_sk_routing_detector(finding):
return bool(
is_qwen_routing_detector(finding)
or is_deepseek_routing_detector(finding)
or is_kimi_routing_detector(finding)
or is_zai_routing_detector(finding)
)
def provider_routing_hint_evidence(hint):
if hint == AMBIGUOUS_QWEN_DEEPSEEK_HINT:
return {'qwen', 'deepseek'}
if hint == AMBIGUOUS_GENERIC_SK_HINT:
return set(GENERIC_SK_PROVIDERS)
return {hint} if hint in GENERIC_SK_PROVIDERS else set()
def ambiguous_provider_routing_hint(evidence):
evidence = set(evidence)
if evidence == {'qwen', 'deepseek'}:
return AMBIGUOUS_QWEN_DEEPSEEK_HINT
return AMBIGUOUS_GENERIC_SK_HINT
def explicit_provider_routing_evidence(finding):
evidence = set()
if is_explicit_qwen_routing_detector(finding):
evidence.add('qwen')
if is_explicit_deepseek_routing_detector(finding):
evidence.add('deepseek')
if is_explicit_kimi_routing_detector(finding):
evidence.add('kimi')
if is_explicit_zai_routing_detector(finding):
evidence.add('zai')
context = finding.get('ScannerContext') if isinstance(finding, dict) else None
if isinstance(context, dict) and context.get('provider_hint_source') == EXPLICIT_ASSIGNMENT_PROVIDER_HINT_SOURCE:
evidence.update(provider_routing_hint_evidence(context.get('provider_hint')))
return evidence
def provider_routing_evidence(finding, max_context_chars=65536):
if not isinstance(finding, dict):
return set()
context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {}
text = provider_routing_context_text(finding, max_context_chars)
evidence = explicit_provider_routing_evidence(finding)
if QWEN_ROUTING_CONTEXT_RE.search(text):
evidence.add('qwen')
if DEEPSEEK_ROUTING_CONTEXT_RE.search(text):
evidence.add('deepseek')
if KIMI_ROUTING_CONTEXT_RE.search(text):
evidence.add('kimi')
if ZAI_ROUTING_CONTEXT_RE.search(text):
evidence.add('zai')
evidence.update(provider_routing_hint_evidence(context.get('provider_hint')))
raw_values = finding_raw_values(finding)
if any(QWEN_SPECIFIC_ROUTING_RE.fullmatch(value) for value in raw_values):
evidence.add('qwen')
if any(ZAI_SPECIFIC_ROUTING_RE.fullmatch(value) for value in raw_values):
evidence.add('zai')
if (
not evidence
and is_generic_sk_routing_detector(finding)
and any(GENERIC_SK_ROUTING_RE.fullmatch(value) for value in raw_values)
):
evidence.update(GENERIC_SK_PROVIDERS)
return evidence
def derive_provider_routing_hint(
finding, max_context_chars=65536, evidence=None, explicit_evidence=None,
):
if not isinstance(finding, dict):
return ''
context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {}
evidence = set(evidence) if evidence is not None else provider_routing_evidence(finding, max_context_chars)
explicit_evidence = (
set(explicit_evidence)
if explicit_evidence is not None
else explicit_provider_routing_evidence(finding)
)
if len(explicit_evidence) > 1:
provider_hint = ambiguous_provider_routing_hint(explicit_evidence)
elif explicit_evidence:
provider_hint = next(iter(explicit_evidence))
elif len(evidence) > 1:
provider_hint = ambiguous_provider_routing_hint(evidence)
elif evidence:
provider_hint = next(iter(evidence))
else:
provider_hint = ''
persisted_context = dict(context)
if provider_hint:
persisted_context['provider_hint'] = provider_hint
else:
persisted_context.pop('provider_hint', None)
if explicit_evidence:
persisted_context['provider_hint_source'] = EXPLICIT_ASSIGNMENT_PROVIDER_HINT_SOURCE
else:
persisted_context.pop('provider_hint_source', None)
if provider_hint in (AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT):
persisted_context['provider_candidates'] = [
provider for provider in GENERIC_SK_PROVIDER_ORDER if provider in evidence
]
else:
persisted_context.pop('provider_candidates', None)
if persisted_context or 'ScannerContext' in finding:
finding['ScannerContext'] = persisted_context
return provider_hint
def strip_nearby_context_for_persistence(result):
findings = result.get('findings') or []
evidence_by_value = {}
explicit_evidence_by_value = {}
for finding in findings:
if not (
is_qwen_routing_detector(finding)
or is_deepseek_routing_detector(finding)
or is_kimi_routing_detector(finding)
or is_zai_routing_detector(finding)
):
continue
evidence = provider_routing_evidence(finding)
explicit_evidence = explicit_provider_routing_evidence(finding)
for value in finding_raw_values(finding):
evidence_by_value.setdefault(value, set()).update(evidence)
explicit_evidence_by_value.setdefault(value, set()).update(explicit_evidence)
for finding in findings:
is_routed_detector = (
is_qwen_routing_detector(finding)
or is_deepseek_routing_detector(finding)
or is_kimi_routing_detector(finding)
or is_zai_routing_detector(finding)
)
evidence = provider_routing_evidence(finding)
explicit_evidence = explicit_provider_routing_evidence(finding)
if is_routed_detector:
for value in finding_raw_values(finding):
evidence.update(evidence_by_value.get(value, ()))
explicit_evidence.update(explicit_evidence_by_value.get(value, ()))
derive_provider_routing_hint(
finding, evidence=evidence, explicit_evidence=explicit_evidence,
)
for finding in findings:
postman_context = finding.get('PostmanContext') if isinstance(finding, dict) else None
if isinstance(postman_context, dict):
finding['PostmanContext'] = sanitize_postman_context(postman_context)
context = finding.get('ScannerContext') if isinstance(finding, dict) else None
if isinstance(context, dict) and 'nearby' in context:
finding['ScannerContext'] = {key: value for key, value in context.items() if key != 'nearby'}
if is_foundry_detector(finding):
text = foundry_finding_text(finding)
endpoints = [normalize_foundry_endpoint(match) for match in POSTMAN_FOUNDRY_ENDPOINT_RE.findall(text)]
keys = foundry_candidate_keys(finding, text)
endpoint = next((item for item in endpoints if item), '')
key = keys[0] if keys else ''
if endpoint and key:
finding['Raw'] = key
finding['RawV2'] = f'{endpoint}:{key}'
finding['Redacted'] = f'{endpoint}:***REDACTED***'
return result
def finding_raw_secret_for_uid(finding):
for key in ('RawV2', 'Raw'):
value = finding.get(key)
if value:
return str(value)
structured = finding.get('StructuredData')
if isinstance(structured, dict):
for value in structured.values():
if isinstance(value, str) and value:
return value
return ''
def finding_location_for_uid(finding):
metadata = finding.get('SourceMetadata') if isinstance(finding, dict) else {}
data = metadata.get('Data') if isinstance(metadata, dict) else {}
if not isinstance(data, dict):
return '', '', ''
for details in data.values():
if not isinstance(details, dict):
continue
file_path = details.get('file') or details.get('path') or details.get('File') or ''
line_number = details.get('line') or details.get('Line') or ''
commit_hash = details.get('commit') or details.get('commitHash') or details.get('commit_hash') or ''
return str(file_path or ''), str(line_number or ''), str(commit_hash or '')
return '', '', ''
def sha256_json(value):
return hashlib.sha256(json.dumps(value, ensure_ascii=False, default=str, sort_keys=True).encode('utf-8', errors='replace')).hexdigest()
def _bounded_utf8(value, max_bytes):
text = ''.join(' ' if ord(character) < 32 else character for character in str(value or ''))
return text.encode('utf-8', errors='replace')[:max_bytes].decode('utf-8', errors='ignore')
def keycheck_input_line_limit():
return max(1024, int(getattr(scan_config, 'keycheck_input_max_line_bytes', 16 * 1024 * 1024)))
def finding_projection_payload(finding):
payload_bytes = json.dumps(finding, ensure_ascii=False, default=str).encode('utf-8')
line_limit = keycheck_input_line_limit()
if len(payload_bytes) + 1 <= line_limit:
return finding, False
raw_secret = finding_raw_secret_for_uid(finding)
metadata = finding.get('SourceMetadata') if isinstance(finding.get('SourceMetadata'), dict) else {}
data = metadata.get('Data') if isinstance(metadata.get('Data'), dict) else {}
source_type = next(iter(data), '')
file_path, line_number, commit_hash = finding_location_for_uid(finding)
marker = {
'finding_uid': _bounded_utf8(finding.get('finding_uid'), 256),
'DetectorName': _bounded_utf8(finding.get('DetectorName') or finding.get('DetectorType'), 64),
'finding_omitted': True,
'keycheck_uncheckable': True,
'omission_reason': 'oversized_finding',
'secret_sha256': hashlib.sha256(raw_secret.encode('utf-8', errors='replace')).hexdigest() if raw_secret else '',
'payload_sha256': hashlib.sha256(payload_bytes).hexdigest(),
'SourceIdentity': {
'type': _bounded_utf8(source_type, 32),
'file': _bounded_utf8(file_path, 128),
'line': _bounded_utf8(line_number, 16),
'commit': _bounded_utf8(commit_hash, 64),
'sha256': sha256_json(metadata),
},
}
marker_bytes = (json.dumps(marker, ensure_ascii=False, default=str) + '\n').encode('utf-8')
if len(marker_bytes) > line_limit:
marker = {
'finding_uid': _bounded_utf8(finding.get('finding_uid'), 256),
'DetectorName': _bounded_utf8(finding.get('DetectorName') or finding.get('DetectorType'), 32),
'finding_omitted': True,
'keycheck_uncheckable': True,
'secret_sha256': marker['secret_sha256'],
'payload_sha256': marker['payload_sha256'],
'source_identity_sha256': marker['SourceIdentity']['sha256'],
}
marker_bytes = (json.dumps(marker, ensure_ascii=False, default=str) + '\n').encode('utf-8')
if len(marker_bytes) > line_limit:
raise RuntimeError('bounded oversized-finding marker exceeds the keycheck input line limit')
return marker, True
def assign_finding_uids(result):
findings = result.get('findings') or []
scan_event_id = str(result.get('scan_event_id') or '')
if not scan_event_id:
raise ValueError('event-based finding identities require scan_event_id')
for index, finding in enumerate(findings, 1):
if not isinstance(finding, dict) or finding.get('finding_uid'):
continue
detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '')
raw_secret = finding_raw_secret_for_uid(finding)
secret_hash = hashlib.sha256(raw_secret.encode('utf-8', errors='replace')).hexdigest() if raw_secret else ''
fallback_hash = sha256_json(finding) if not secret_hash else ''
file_path, line_number, commit_hash = finding_location_for_uid(finding)
finding['finding_uid'] = hashlib.sha256('|'.join([
'truf-finding-v2',
scan_event_id,
str(index),
detector,
secret_hash or fallback_hash,
file_path,
line_number,
commit_hash,
]).encode('utf-8', errors='replace')).hexdigest()
def save_scan_result(result):
"""Persist findings like raw TruffleHog JSONL and keep errors separately."""
results_dir = get_results_dir()
if not results_dir:
return False
success = True
event_id = str(result.get('scan_event_id') or '')
if not event_id:
logger.error('Unable to persist scan result without scan_event_id')
return False
target = result.get('target', '')
scan_type = result.get('scan_type', '')
timestamp = result.get('timestamp', datetime.now().isoformat())
assign_finding_uids(result)
try:
foundry_candidates = write_foundry_keycheck_candidates_from_findings(copy.deepcopy(result))
if foundry_candidates:
logger.info(f"Queued {foundry_candidates} Azure Foundry keycheck candidate(s) from {scan_type}:{target}")
except Exception as e:
logger.warning(f"Unable to queue Azure Foundry keycheck candidate(s) for {scan_type}:{target}: {str(e)}")
success = False
projection_result = copy.deepcopy(result)
strip_nearby_context_for_persistence(projection_result)
projected_findings = []
oversized_count = 0
for finding in projection_result.get('findings') or []:
projected, oversized = finding_projection_payload(finding)
projected_findings.append(projected)
oversized_count += int(oversized)
projection_result['findings'] = projected_findings
if oversized_count:
projection_result.setdefault('warnings', []).append(
f'{oversized_count} oversized finding(s) were projected as uncheckable metadata markers; '
'PostgreSQL retains the authoritative findings'
)
projection_result['degraded'] = True
projection_result['oversized_findings_omitted'] = oversized_count
if projected_findings or projection_result.get('errors') or projection_result.get('warnings') or projection_result.get('skipped'):
success = append_rotating_jsonl_once(
os.path.join(results_dir, 'scan_results.jsonl'),
projection_result,
'scan_event_id',
event_id,
scan_config.scan_results_max_mb,
) and success
for finding in projected_findings:
success = append_rotating_jsonl_once(
os.path.join(results_dir, 'found_secrets.jsonl'),
finding,
'finding_uid',
finding.get('finding_uid'),
scan_config.found_secrets_max_mb,
) and success
if projection_result.get('errors'):
path = os.path.join(results_dir, 'scan_errors.log')
success = append_scan_errors_once(
path,
projection_result,
scan_config.scan_errors_max_mb,
scan_config.scan_errors_keep,
) and success
return success
# ======================
# DOCKER TOKEN MANAGEMENT
# ======================
@dataclass(frozen=True, repr=False)
class DockerAccount:
name: str
username: str
token: str
config_dir: str
def __repr__(self):
return f'DockerAccount(name={self.name!r})'
@dataclass(frozen=True, repr=False)
class DockerRegistryAuth:
token: str
account_name: str = ''
challenge: str = ''
class DockerTokenManager:
def __init__(self):
self.tokens = []
self.token_dirs = []
self.accounts = []
self.current_index = 0
self.request_index = 0
self.lock = threading.RLock()
self.config_key = None
self.cooldown_sec = 1800
self.cooldown_until = {}
self.cooldown_categories = {}
self.hub_tokens = {}
self.hub_token_locks = {}
self.invalid_accounts = set()
self.status_events = {}
self.explicit_pool = False
@staticmethod
def _account_name(entry, index):
return str(entry.get('name') or f'docker_{index + 1}').strip()
def setup_accounts(
self, entries, cooldown_sec=1800, explicit_pool=False,
create_config_dirs=True,
):
normalized = []
seen_names = set()
for index, entry in enumerate(entries or []):
if not isinstance(entry, dict):
continue
name = self._account_name(entry, index)
username = str(entry.get('username') or '').strip()
token = str(entry.get('token') or '').strip()
if not name or not username or not token or name in seen_names:
continue
seen_names.add(name)
normalized.append((name, username, token))
config_key = (bool(create_config_dirs),) + tuple(
(name, username, hashlib.sha256(token.encode('utf-8')).hexdigest())
for name, username, token in normalized
)
with self.lock:
self.cooldown_sec = max(60, min(86400, int(cooldown_sec or 1800)))
if (
config_key == self.config_key
and self.explicit_pool == bool(explicit_pool)
and (
not normalized or not create_config_dirs
or all(
os.path.isfile(os.path.join(account.config_dir, 'config.json'))
for account in self.accounts
)
)
):
return
self._cleanup_locked()
self.config_key = config_key
self.current_index = 0
self.request_index = 0
self.cooldown_until = {}
self.cooldown_categories = {}
self.hub_tokens = {}
self.hub_token_locks = {}
self.invalid_accounts = set()
self.status_events = {}
self.explicit_pool = bool(explicit_pool)
for name, username, token in normalized:
temp_dir = ''
if create_config_dirs:
temp_dir = create_docker_config_dir()
auth = base64.b64encode(f'{username}:{token}'.encode()).decode()
docker_auth = {'auth': auth}
config = {
'auths': {
'https://index.docker.io/v1/': docker_auth,
'index.docker.io': docker_auth,
'registry-1.docker.io': docker_auth,
'https://registry-1.docker.io': docker_auth,
'docker.io': docker_auth,
}
}
config_path = os.path.join(temp_dir, 'config.json')
with open(config_path, 'w') as f:
json.dump(config, f)
harden_private_file(config_path)
account = DockerAccount(name, username, token, temp_dir)
self.accounts.append(account)
self.tokens.append(f'{username}:{token}')
if temp_dir:
self.token_dirs.append(temp_dir)
logger.info('Configured Docker account: %s', name)
def setup_tokens(self, tokens_str=None, username=None, create_config_dirs=True):
"""Initialize Docker tokens from environment variable"""
username = (username or os.getenv('DOCKERHUB_USERNAME') or os.getenv('DOCKER_USERNAME') or '').strip()
tokens_str = tokens_str if tokens_str is not None else os.getenv('DOCKER_TOKENS', '')
if not tokens_str:
single_token = os.getenv('DOCKERHUB_TOKEN') or os.getenv('DOCKER_TOKEN')
if single_token:
tokens_str = single_token
entries = []
if tokens_str:
for index, raw_token in enumerate(tokens_str.replace('\n', ',').split(',')):
raw_token = raw_token.strip()
if not raw_token:
continue
if ':' in raw_token:
account_username, token = raw_token.split(':', 1)
elif username:
account_username, token = username, raw_token
else:
logger.warning('Docker token without username ignored. Use username:token or set DOCKERHUB_USERNAME.')
continue
account_username = str(account_username).strip()
token = str(token).strip()
if account_username and token:
entries.append({
'name': f'docker_{index + 1}',
'username': account_username,
'token': token,
})
self.setup_accounts(
entries, explicit_pool=False, create_config_dirs=create_config_dirs,
)
def has_accounts(self):
with self.lock:
return bool(self.accounts)
def account_count(self):
with self.lock:
return len(self.accounts)
def next_account(self, endpoint, excluded_names=None):
excluded_names = set(excluded_names or ())
with self.lock:
if not self.accounts:
return None
now = time.time()
for offset in range(len(self.accounts)):
position = (self.request_index + offset) % len(self.accounts)
account = self.accounts[position]
if account.name in excluded_names:
continue
if account.name in self.invalid_accounts:
continue
if float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now:
continue
self.request_index = (position + 1) % len(self.accounts)
return account
return None
def report_http_status(self, account, endpoint, status, response=None, category=None):
account_name = account.name if isinstance(account, DockerAccount) else str(account or '')
if not account_name:
return False
status = int(status or 0)
category = str(category or ('rate_limit' if status == 429 else 'auth_forbidden'))
with self.lock:
if account_name in self.invalid_accounts and category != 'auth_invalid':
return self._all_unavailable_locked(endpoint)
if category == 'auth_invalid':
self.invalid_accounts.add(account_name)
retry_at = float('inf')
reset_at = 'manual'
else:
delay = dockerhub_retry_after_seconds(response) if status == 429 else self.cooldown_sec
retry_at = time.time() + max(60, int(delay))
reset_at = datetime.fromtimestamp(retry_at, timezone.utc).isoformat(timespec='seconds')
self.cooldown_until[(account_name, endpoint)] = retry_at
self.cooldown_categories[(account_name, endpoint)] = category
if status == 401 or category == 'auth_invalid':
self.hub_tokens.pop(account_name, None)
self.status_events[(account_name, endpoint)] = {
'name': account_name,
'endpoint': endpoint,
'category': category,
'reset_at': reset_at,
'message': f'Docker {endpoint} HTTP {status}',
}
return self._all_unavailable_locked(endpoint)
def report_success(self, account, endpoint):
account_name = account.name if isinstance(account, DockerAccount) else str(account or '')
if not account_name:
return
with self.lock:
if account_name in self.invalid_accounts:
return
expires_at = float(self.cooldown_until.get((account_name, endpoint), 0) or 0)
if expires_at <= time.time():
self.cooldown_until.pop((account_name, endpoint), None)
self.cooldown_categories.pop((account_name, endpoint), None)
event_key = (account_name, endpoint)
if event_key not in self.status_events:
self.status_events[event_key] = {
'name': account_name,
'endpoint': endpoint,
'category': 'ok',
'reset_at': None,
'message': '',
}
def _all_unavailable_locked(self, endpoint):
if not self.accounts:
return False
now = time.time()
return all(
account.name in self.invalid_accounts
or float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now
for account in self.accounts
)
def all_unavailable(self, endpoint):
with self.lock:
return self._all_unavailable_locked(endpoint)
def rate_limit_contributes_to_exhaustion(self, endpoint):
with self.lock:
if not self._all_unavailable_locked(endpoint):
return False
now = time.time()
return any(
float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now
and self.cooldown_categories.get((account.name, endpoint)) == 'rate_limit'
for account in self.accounts
)
def uses_explicit_pool(self):
with self.lock:
return bool(self.explicit_pool)
def seconds_until_available(self, endpoint):
with self.lock:
now = time.time()
deadlines = [
float(self.cooldown_until.get((account.name, endpoint), 0) or 0)
for account in self.accounts
if float(self.cooldown_until.get((account.name, endpoint), 0) or 0) != float('inf')
and float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now
]
if not deadlines:
return self.cooldown_sec
return max(60, math.ceil(min(deadlines) - now))
def account_available(self, account_name, endpoint):
with self.lock:
return account_name not in self.invalid_accounts and float(
self.cooldown_until.get((account_name, endpoint), 0) or 0
) <= time.time()
def hub_token_lock(self, account_name):
with self.lock:
lock = self.hub_token_locks.get(account_name)
if lock is None:
lock = threading.Lock()
self.hub_token_locks[account_name] = lock
return lock
def restore_endpoint_cooldowns(self, endpoint_status):
if not isinstance(endpoint_status, dict):
return
with self.lock:
account_names = {account.name for account in self.accounts}
now = time.time()
for endpoint, accounts in endpoint_status.items():
if endpoint not in {'hub_search', 'hub_tags', 'registry'} or not isinstance(accounts, dict):
continue
for account_name, status in accounts.items():
if account_name not in account_names or not isinstance(status, dict):
continue
disabled_until = status.get('disabled_until')
if disabled_until == 'manual':
retry_at = float('inf')
else:
try:
retry_at = datetime.fromisoformat(
str(disabled_until).replace('Z', '+00:00')
).timestamp()
except (TypeError, ValueError, OverflowError):
continue
if retry_at <= now:
continue
key = (account_name, endpoint)
category = str(status.get('disabled_reason') or 'rate_limit')
if category == 'auth_invalid':
self.invalid_accounts.add(account_name)
if retry_at > float(self.cooldown_until.get(key, 0) or 0):
self.cooldown_until[key] = retry_at
self.cooldown_categories[key] = category
def cached_hub_token(self, account_name):
with self.lock:
token, expires_at = self.hub_tokens.get(account_name, ('', 0))
if token and float(expires_at or 0) > time.time() + 30:
return token
self.hub_tokens.pop(account_name, None)
return ''
def cache_hub_token(self, account_name, token, expires_in):
with self.lock:
lifetime = max(60, min(600, int(expires_in or 600)))
self.hub_tokens[account_name] = (token, time.time() + lifetime)
def invalidate_hub_token(self, account_name):
with self.lock:
self.hub_tokens.pop(account_name, None)
def drain_status_events(self):
with self.lock:
events = list(self.status_events.values())
self.status_events = {}
return events
def get_next_config(self):
"""Get the next Docker config directory in rotation"""
with self.lock:
if not self.accounts:
return None
for offset in range(len(self.accounts)):
position = (self.current_index + offset) % len(self.accounts)
account = self.accounts[position]
if account.name in self.invalid_accounts:
continue
self.current_index = (position + 1) % len(self.accounts)
return account.config_dir
return None
def _cleanup_locked(self):
for directory in self.token_dirs:
try:
cleanup_command_work_dir(directory)
except Exception as exc:
logger.error('Error cleaning Docker config: %s', str(exc))
self.tokens = []
self.token_dirs = []
self.accounts = []
def cleanup(self):
"""Clean up temporary Docker config directories"""
with self.lock:
self._cleanup_locked()
self.config_key = None
self.cooldown_until = {}
self.cooldown_categories = {}
self.hub_tokens = {}
self.hub_token_locks = {}
self.invalid_accounts = set()
self.status_events = {}
self.explicit_pool = False
docker_token_manager = DockerTokenManager()
def configure_docker_tokens(tokens_str=None, username=None):
"""Reconfigure Docker auth tokens after UI or CLI input."""
require_scanner_runtime_initialized()
docker_token_manager.setup_tokens(tokens_str, username)
def configure_docker_accounts(entries, cooldown_sec=1800):
require_scanner_runtime_initialized()
docker_token_manager.setup_accounts(
entries, cooldown_sec=cooldown_sec, explicit_pool=True,
)
def configure_docker_discovery_tokens(tokens_str=None, username=None):
docker_token_manager.setup_tokens(
tokens_str, username, create_config_dirs=False,
)
def configure_docker_discovery_accounts(entries, cooldown_sec=1800):
docker_token_manager.setup_accounts(
entries, cooldown_sec=cooldown_sec, explicit_pool=True,
create_config_dirs=False,
)
def restore_docker_endpoint_cooldowns(endpoint_status):
docker_token_manager.restore_endpoint_cooldowns(endpoint_status)
def drain_docker_auth_events():
return docker_token_manager.drain_status_events()
# ===================
# API FETCH FUNCTIONS
# ===================
def github_repo_to_target(item):
return {
'url': item.get('clone_url', ''),
'name': item.get('full_name', ''),
'created_at': item.get('created_at', ''),
'updated_at': item.get('updated_at', ''),
'pushed_at': item.get('pushed_at', ''),
}
def github_headers(token=None):
headers = {'Accept': 'application/vnd.github+json', 'User-Agent': 'GitSecretsScanner/2.0'}
if token:
headers['Authorization'] = f'Bearer {token}'
return headers
def gitlab_headers(token=None):
headers = {'User-Agent': 'GitSecretsScanner/2.0'}
if token:
headers['PRIVATE-TOKEN'] = token
return headers
def parse_github_repo_target(target):
if isinstance(target, dict):
repo = target.get('repo') or target.get('full_name') or ''
if not repo and target.get('repo_url'):
candidate = normalize_git_repo_candidate(target.get('repo_url'))
if candidate and candidate.get('provider') == 'github':
repo = candidate.get('repo_path') or ''
url = target.get('url') or (f'https://github.com/{repo}' if repo else '')
return repo.strip('/'), url
text = str(target or '').strip()
if text.startswith('{'):
try:
return parse_github_repo_target(json.loads(text))
except Exception:
pass
lowered = text.lower().rstrip('/')
if lowered.endswith('.git'):
lowered = lowered[:-4]
candidate = normalize_git_repo_candidate(text)
if candidate and candidate.get('provider') == 'github':
repo = candidate.get('repo_path') or ''
return repo, f'https://github.com/{repo}'
match = re.search(r'github\.com[:/]([^/\s]+/[^/\s]+)', lowered, re.IGNORECASE)
if match:
repo = match.group(1).strip('/')
return repo, f'https://github.com/{repo}'
if re.match(r'^[^/\s]+/[^/\s]+$', text):
repo = text.strip('/')
return repo, f'https://github.com/{repo}'
return '', text
def parse_gitlab_project_target(target):
if isinstance(target, dict):
project = target.get('project') or target.get('path_with_namespace') or target.get('repo') or ''
if not project and target.get('repo_url'):
candidate = normalize_git_repo_candidate(target.get('repo_url'))
if candidate and candidate.get('provider') == 'gitlab':
project = candidate.get('repo_path') or ''
url = target.get('url') or (f'https://gitlab.com/{project}' if project else '')
return project.strip('/'), url
text = str(target or '').strip()
if text.startswith('{'):
try:
return parse_gitlab_project_target(json.loads(text))
except Exception:
pass
lowered = text.lower().rstrip('/')
if lowered.endswith('.git'):
lowered = lowered[:-4]
candidate = normalize_git_repo_candidate(text)
if candidate and candidate.get('provider') == 'gitlab':
project = candidate.get('repo_path') or ''
return project, f'https://gitlab.com/{project}'
match = re.search(r'gitlab\.com[:/](.+)$', lowered, re.IGNORECASE)
if match:
project = match.group(1).strip('/')
return project, f'https://gitlab.com/{project}'
if '/' in text and '://' not in text:
project = text.strip('/')
return project, f'https://gitlab.com/{project}'
return '', text
def github_rate_limit_reset(response):
if not response:
return None
reset = response.headers.get('X-RateLimit-Reset')
if not reset:
return None
try:
return datetime.fromtimestamp(int(reset), timezone.utc).isoformat(timespec='seconds')
except (TypeError, ValueError):
return None
def page_is_known(targets, known_targets=None, normalize_target=None, known_target_lookup=None):
if not normalize_target or not targets:
return False
offered = [target for target in targets if target]
normalized = [normalize_target(target) for target in offered]
if known_target_lookup:
try:
known_targets = set(known_target_lookup(offered) or ())
except Exception as exc:
logger.warning(f'Known-target page lookup failed open: {str(exc)[:300]}')
return False
if not known_targets:
return False
return bool(normalized) and all(target in known_targets for target in normalized)
POSTMAN_COLLECTION_SUFFIX = 'postman_collection.json'
POSTMAN_ENVIRONMENT_SUFFIX = 'postman_environment.json'
API_ARTIFACT_PATTERNS = {
'collection': ('postman_collection.json',),
'environment': ('postman_environment.json',),
'postman': ('postman.json', '.postman.json'),
'insomnia': ('insomnia.json', '.insomnia.json'),
'bruno': ('.bru', 'bruno.json'),
'thunder_collection': ('thunder-collection.json',),
'thunder_environment': ('thunder-environment.json',),
'hoppscotch': ('hoppscotch.json',),
'generic': ('collection.json', 'environment.json'),
}
API_ARTIFACT_SEARCHES = {
'collection': ('filename:postman_collection.json {query}',),
'environment': ('filename:postman_environment.json {query}',),
'postman': ('filename:postman.json {query}', 'filename:.postman.json {query}'),
'insomnia': ('filename:insomnia.json {query}', 'filename:.insomnia.json {query}'),
'bruno': ('extension:bru {query}', 'filename:bruno.json {query}'),
'thunder_collection': ('filename:thunder-collection.json {query}', 'filename:thunder-collection_ {query}'),
'thunder_environment': ('filename:thunder-environment.json {query}', 'filename:thunder-environment_ {query}'),
'hoppscotch': ('filename:hoppscotch.json {query}',),
'generic': ('filename:collection.json postman {query}', 'filename:environment.json postman {query}'),
'signature': (
'"currentValue" "api_key" {query}',
'"pm.collectionVariables" {query}',
'"pm.environment.set" {query}',
'"openai.azure.com" "api-key" {query}',
'"services.ai.azure.com" "api-key" {query}',
'"generativelanguage.googleapis.com" "key" {query}',
),
}
POSTMAN_PLACEHOLDER_RE = re.compile(
r'^(?:\{\{[^}]+\}\}|<[^>]+>|your[_ -]?[a-z0-9_-]+|replace[_ -]?me|change[_ -]?me|changeme|example|dummy|test)$',
re.IGNORECASE,
)
POSTMAN_JSON_HARD_MAX_INPUT_BYTES = 16 * 1024 * 1024
POSTMAN_HARVEST_MAX_WARNINGS = 5
POSTMAN_HARVEST_WARNING_MAX_CHARS = 400
class PostmanCacheValidationError(ValueError):
pass
class PostmanCacheTooLarge(PostmanCacheValidationError):
pass
class PostmanCacheCapacityError(PostmanCacheValidationError):
pass
class _PostmanHarvestDeadlineReached(RuntimeError):
pass
def _postman_discovery_limits(max_artifacts=None, max_page_artifacts=None, max_total_bytes=None, max_elapsed_sec=None):
artifacts = int(
getattr(scan_config, 'postman_discovery_max_artifacts_per_cycle', 1000)
if max_artifacts is None else max_artifacts
)
page_artifacts = int(
getattr(scan_config, 'postman_discovery_max_artifacts_per_page', 100)
if max_page_artifacts is None else max_page_artifacts
)
total_bytes = int(
getattr(scan_config, 'postman_discovery_max_bytes_per_cycle', 1024 * 1024 * 1024)
if max_total_bytes is None else max_total_bytes
)
elapsed_sec = float(
getattr(scan_config, 'postman_discovery_max_elapsed_sec', 300.0)
if max_elapsed_sec is None else max_elapsed_sec
)
if artifacts <= 0 or page_artifacts <= 0 or total_bytes <= 0 or elapsed_sec <= 0 or not math.isfinite(elapsed_sec):
raise ValueError('Postman discovery limits must be finite and positive')
return artifacts, page_artifacts, total_bytes, elapsed_sec
class _PostmanDiscoveryBudget:
MAX_WARNINGS = 5
def __init__(self, source, max_artifacts=None, max_page_artifacts=None, max_total_bytes=None, max_elapsed_sec=None):
(
self.max_artifacts,
self.max_page_artifacts,
self.max_total_bytes,
elapsed_sec,
) = _postman_discovery_limits(max_artifacts, max_page_artifacts, max_total_bytes, max_elapsed_sec)
self.source = str(source or 'non-package')
self.elapsed_sec = elapsed_sec
self.deadline = time.monotonic() + elapsed_sec
self.artifacts = 0
self.total_bytes = 0
self.stop_reason = ''
self._warnings = set()
def warn(self, detail):
detail = str(detail or '')[:400]
if detail in self._warnings or len(self._warnings) >= self.MAX_WARNINGS:
return
self._warnings.add(detail)
logger.warning('Optional Postman %s discovery bounded: %s', self.source, detail)
def stop(self, detail):
if not self.stop_reason:
self.stop_reason = str(detail or 'bounded discovery limit reached')[:400]
self.warn(self.stop_reason)
return False
def check_deadline(self, context='during discovery'):
if self.stop_reason:
return False
if time.monotonic() >= self.deadline:
return self.stop(f'elapsed deadline of {self.elapsed_sec:g}s reached {context}')
return True
def request_timeout(self, configured_timeout):
if not self.check_deadline('before a network request'):
raise _PostmanHarvestDeadlineReached(self.stop_reason)
remaining = self.deadline - time.monotonic()
if remaining <= 0:
self.stop(f'elapsed deadline of {self.elapsed_sec:g}s reached before a network request')
raise _PostmanHarvestDeadlineReached(self.stop_reason)
return max(0.01, min(float(configured_timeout or remaining), remaining))
def admit(self, page_artifacts):
if not self.check_deadline():
return False, 'cycle'
if page_artifacts >= self.max_page_artifacts:
self.warn(f'per-page artifact limit of {self.max_page_artifacts} reached')
return False, 'page'
if self.artifacts >= self.max_artifacts:
self.stop(f'per-cycle artifact limit of {self.max_artifacts} reached')
return False, 'cycle'
self.artifacts += 1
return True, ''
def account_bytes(self, size):
size = max(0, int(size or 0))
if self.total_bytes + size > self.max_total_bytes:
return self.stop(
f'aggregate byte limit of {self.max_total_bytes} reached after {self.total_bytes} byte(s)'
)
self.total_bytes += size
return True
def exhausted(self):
if self.stop_reason:
return True
if self.artifacts >= self.max_artifacts:
return not self.stop(f'per-cycle artifact limit of {self.max_artifacts} reached')
if self.total_bytes >= self.max_total_bytes:
return not self.stop(f'aggregate byte limit of {self.max_total_bytes} reached')
return not self.check_deadline()
def _publish_postman_discovery_batch(prepared, cache_dir, budget):
if not prepared:
return [], False
published, deadline_reached = _publish_postman_cache_entries(
[item['entry'] for item in prepared],
cache_dir,
deadline=budget.deadline,
)
if deadline_reached:
budget.stop(f'elapsed deadline of {budget.elapsed_sec:g}s reached while scanning or publishing the cache')
return list(zip(prepared, published)), deadline_reached
def utc_now_for_postman():
return datetime.now(timezone.utc)
def parse_postman_time(value):
if not value:
return None
try:
parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00'))
if parsed.tzinfo is None:
parsed = parsed.replace(tzinfo=timezone.utc)
return parsed.astimezone(timezone.utc)
except ValueError:
return None
def postman_kind_for_path(path):
name = os.path.basename(str(path or '')).lower()
normalized = str(path or '').replace('\\', '/').lower()
for kind, suffixes in API_ARTIFACT_PATTERNS.items():
if any(name.endswith(suffix) or normalized.endswith('/bruno/' + suffix) for suffix in suffixes):
return kind
return 'artifact'
def normalize_postman_search_kinds(value):
if not value:
return ['collection', 'environment']
if isinstance(value, str):
parts = [item.strip().lower() for item in value.split(',')]
else:
parts = [str(item).strip().lower() for item in value]
aliases = {
'collections': 'collection', 'env': 'environment', 'environments': 'environment',
'api_artifacts': 'all', 'api-artifacts': 'all', 'thunder': 'thunder_collection',
}
kinds = []
for item in parts:
item = aliases.get(item, item)
if item == 'all':
for kind in API_ARTIFACT_SEARCHES:
if kind not in kinds:
kinds.append(kind)
continue
if item in API_ARTIFACT_SEARCHES and item not in kinds:
kinds.append(item)
return kinds or ['collection', 'environment']
def parse_postman_target(target):
if isinstance(target, dict):
return dict(target)
text = str(target or '').strip()
if not text:
return {}
if text.startswith('{'):
return json.loads(text)
if text.lower().startswith('file:'):
path = text[5:]
return {'source': 'local_file', 'kind': postman_kind_for_path(path), 'local_path': path, 'path': path}
if os.path.exists(text):
return {'source': 'local_file', 'kind': postman_kind_for_path(text), 'local_path': text, 'path': text}
return {'source': 'url', 'kind': postman_kind_for_path(text), 'url': text}
def postman_target_identity(target):
return semantic_postman_target_identity(target)
def get_postman_cache_dir(cache_dir=None):
path = cache_dir or getattr(scan_config, 'postman_cache_dir', None)
if not path:
raise PostmanCacheValidationError('configured Postman cache directory is required')
runtime_dir = getattr(scan_config, 'runtime_dir', None)
if not runtime_dir:
raise PostmanCacheValidationError('configured runtime directory is required for Postman cache containment')
runtime_root = canonical_path(runtime_dir)
cache_root = canonical_path(path)
try:
contained = os.path.commonpath((runtime_root, cache_root)) == runtime_root and cache_root != runtime_root
except ValueError:
contained = False
if not contained:
raise PostmanCacheValidationError('Postman cache must remain under the configured runtime directory')
return require_private_directory(cache_root, create=True)
def sha256_bytes(value):
return hashlib.sha256(value or b'').hexdigest()
def postman_cache_file_path(digest, cache_dir=None):
root = get_postman_cache_dir(cache_dir)
prefix = str(digest or '')[:2] or 'xx'
directory = os.path.join(root, prefix)
ensure_private_directory(directory, reject_reparse=True)
return os.path.join(directory, f'{digest}.json')
def postman_cache_usage(root, deadline=None):
usage = {'items': 0, 'files': 0, 'bytes': 0}
stack = [require_private_directory(root, create=False)]
while stack:
if deadline is not None and time.monotonic() >= deadline:
raise _PostmanHarvestDeadlineReached()
current = stack.pop()
with os.scandir(current) as entries:
for entry in entries:
if deadline is not None and time.monotonic() >= deadline:
raise _PostmanHarvestDeadlineReached()
if entry.is_symlink() or is_reparse_point(entry.path):
raise PostmanCacheCapacityError(f'Postman cache contains a linked entry: {entry.path}')
if entry.is_dir(follow_symlinks=False):
if not private_directory_ready(entry.path):
raise PostmanCacheCapacityError(f'Postman cache directory is not private: {entry.path}')
stack.append(entry.path)
continue
if not entry.is_file(follow_symlinks=False):
raise PostmanCacheCapacityError(f'Postman cache contains an unsupported entry: {entry.path}')
if entry.name == '.postman-cache.lock':
continue
details = entry.stat(follow_symlinks=False)
usage['files'] += 1
usage['bytes'] += int(details.st_size)
if not entry.name.endswith('.meta.json'):
usage['items'] += 1
return usage
def _postman_cache_limits(max_items=None, max_bytes=None, min_free_bytes=None):
values = {
'items': int(getattr(scan_config, 'postman_cache_max_items', 0) if max_items is None else max_items),
'bytes': int(getattr(scan_config, 'postman_cache_max_bytes', 0) if max_bytes is None else max_bytes),
'min_free_bytes': int(getattr(scan_config, 'postman_cache_min_free_bytes', 0) if min_free_bytes is None else min_free_bytes),
}
if values['items'] <= 0 or values['bytes'] <= 0 or values['min_free_bytes'] < 0:
raise PostmanCacheCapacityError('Postman cache aggregate limits must be finite positive values')
return values
def _write_private_cache_file(path, payload):
if os.path.lexists(path):
raise FileExistsError(path)
temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.tmp'
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0)
descriptor = os.open(temporary, flags, 0o600)
try:
os.close(descriptor)
descriptor = None
harden_private_file(temporary)
with open(temporary, 'wb') as handle:
handle.write(payload)
handle.flush()
os.fsync(handle.fileno())
if not private_file_ready(temporary):
raise PostmanCacheCapacityError(f'Postman cache temporary file is not private: {temporary}')
if os.path.lexists(path):
raise FileExistsError(path)
durable_replace(temporary, path)
if not private_file_ready(path):
raise PostmanCacheCapacityError(f'Postman cache publication is not private: {path}')
finally:
if descriptor is not None:
os.close(descriptor)
try:
if os.path.exists(temporary):
os.remove(temporary)
except OSError:
pass
def _bounded_json_bytes(value, max_bytes, *, indent=None, sort_keys=False, newline=False):
output = bytearray()
encoder = json.JSONEncoder(ensure_ascii=False, indent=indent, sort_keys=sort_keys)
for chunk in encoder.iterencode(value):
encoded = chunk.encode('utf-8')
if len(output) + len(encoded) + (1 if newline else 0) > max_bytes:
raise ValueError('serialized JSON exceeds its bounded byte limit')
output.extend(encoded)
if newline:
output.extend(b'\n')
return bytes(output)
def validate_postman_cache_artifact(target_data, max_artifact_size_mb=20, expected_size=None):
_raise_if_scan_slot_fatal()
if not isinstance(target_data, dict):
raise PostmanCacheValidationError('Postman cache metadata must be an object')
offered_path = target_data.get('cache_path') or target_data.get('local_path')
if not offered_path:
raise PostmanCacheValidationError('Postman target missing cached artifact')
cache_root = canonical_path(get_postman_cache_dir())
reject_reparse_components(offered_path)
cache_path = canonical_path(offered_path)
try:
contained = os.path.commonpath((cache_root, cache_path)) == cache_root and cache_path != cache_root
except ValueError:
contained = False
if not contained:
raise PostmanCacheValidationError('Postman cache path escapes the configured cache root')
reject_reparse_components(cache_path)
if is_reparse_point(cache_path) or not private_file_ready(cache_path):
raise PostmanCacheValidationError('Postman cached artifact is absent, linked, or not private')
declared_hash = str(target_data.get('sha256') or '').strip().lower()
if not re.fullmatch(r'[0-9a-f]{64}', declared_hash):
raise PostmanCacheValidationError('Postman target has no valid declared SHA-256')
if postman_target_identity(target_data) != f'postman:sha256:{declared_hash}':
raise PostmanCacheValidationError('Postman semantic identity does not match its declared SHA-256')
flags = os.O_RDONLY | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0)
descriptor = os.open(cache_path, flags)
try:
opened = os.fstat(descriptor)
if not stat.S_ISREG(opened.st_mode):
raise PostmanCacheValidationError('Postman cached artifact is not a regular file')
size = int(opened.st_size)
if size <= 0:
raise PostmanCacheValidationError('Postman cached artifact is empty')
declared_sizes = [expected_size]
declared_sizes.extend(target_data.get(name) for name in ('size', 'bytes'))
for declared_size in declared_sizes:
if declared_size in (None, ''):
continue
try:
parsed_size = int(declared_size)
except (TypeError, ValueError) as exc:
raise PostmanCacheValidationError('Postman cached artifact has invalid declared size') from exc
if parsed_size != size:
raise PostmanCacheValidationError('Postman cached artifact size mismatch')
max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024
if max_bytes <= 0:
raise PostmanCacheValidationError('Postman artifact limit must be finite and positive')
if size > max_bytes:
raise PostmanCacheTooLarge(f'artifact exceeds {max_artifact_size_mb} MB')
digest = hashlib.sha256()
while True:
_raise_if_scan_slot_fatal()
block = os.read(descriptor, 1024 * 1024)
if not block:
break
digest.update(block)
if digest.hexdigest() != declared_hash:
raise PostmanCacheValidationError('Postman cached artifact SHA-256 mismatch')
_raise_if_scan_slot_fatal()
current = os.stat(cache_path, follow_symlinks=False)
opened_identity = (opened.st_dev, opened.st_ino, opened.st_size, getattr(opened, 'st_mtime_ns', None))
current_identity = (current.st_dev, current.st_ino, current.st_size, getattr(current, 'st_mtime_ns', None))
if opened_identity != current_identity:
raise PostmanCacheValidationError('Postman cached artifact changed during validation')
finally:
os.close(descriptor)
return cache_path, size
def _prepare_postman_cache_entry(content, kind, origin, max_artifact_size_mb):
max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024
if max_bytes <= 0:
raise ValueError('Postman artifact limit must be finite and positive')
if isinstance(content, str) and len(content) > max_bytes:
raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB')
content_bytes = content.encode('utf-8', errors='replace') if isinstance(content, str) else bytes(content or b'')
if len(content_bytes) > max_bytes:
raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB')
if not content_bytes:
raise ValueError('Postman artifact is empty')
stripped = content_bytes.lstrip()
if stripped.startswith((b'{', b'[')):
if len(content_bytes) > POSTMAN_JSON_HARD_MAX_INPUT_BYTES:
raise ValueError('Postman JSON artifact exceeds the hard pre-parse byte limit')
try:
json.loads(content_bytes.decode('utf-8-sig'))
except (UnicodeDecodeError, ValueError, RecursionError) as exc:
raise ValueError('Postman JSON artifact is invalid') from exc
raw_content = content_bytes
digest = sha256_bytes(raw_content)
metadata = {
'sha256': digest,
'kind': kind,
'bytes': len(raw_content),
'origin': origin or {},
'cached_at': utc_now_for_postman().isoformat(timespec='seconds'),
}
try:
metadata_bytes = _bounded_json_bytes(metadata, max_bytes, indent=2, sort_keys=True, newline=True)
except ValueError as exc:
raise PostmanCacheCapacityError('Postman cache metadata exceeds the per-artifact byte limit') from exc
return {
'content': raw_content,
'digest': digest,
'metadata': metadata_bytes,
'size': len(raw_content),
}
def _publish_postman_cache_entries(
entries,
cache_dir=None,
cache_max_items=None,
cache_max_bytes=None,
cache_min_free_bytes=None,
deadline=None,
):
if not entries:
return [], False
root = get_postman_cache_dir(cache_dir)
limits = _postman_cache_limits(cache_max_items, cache_max_bytes, cache_min_free_bytes)
lock_path = os.path.join(root, '.postman-cache.lock')
lock_timeout = max(1.0, float(getattr(scan_config, 'postman_cache_lock_timeout_sec', 30)))
if deadline is not None:
remaining = deadline - time.monotonic()
if remaining <= 0:
return [], True
lock_timeout = min(lock_timeout, max(0.01, remaining))
try:
lock = acquire_file_lock(
lock_path,
timeout_sec=lock_timeout,
)
except TimeoutError:
if deadline is not None and time.monotonic() >= deadline:
return [], True
raise
try:
if deadline is not None and time.monotonic() >= deadline:
return [], True
try:
usage = (
postman_cache_usage(root)
if deadline is None
else postman_cache_usage(root, deadline=deadline)
)
except _PostmanHarvestDeadlineReached:
return [], True
plans = []
reserved_bytes = 0
free_bytes = None
seen = set()
for entry in entries:
if deadline is not None and time.monotonic() >= deadline:
return [], True
digest = entry['digest']
if digest in seen:
continue
seen.add(digest)
if sha256_bytes(entry['content']) != digest:
raise PostmanCacheValidationError('Postman cache entry SHA-256 changed before publication')
prefix_dir = os.path.join(root, digest[:2])
path = os.path.join(prefix_dir, f'{digest}.json')
meta_path = f'{path}.meta.json'
if os.path.lexists(prefix_dir):
reject_reparse_components(prefix_dir)
if not os.path.isdir(prefix_dir) or not private_directory_ready(prefix_dir):
raise PostmanCacheCapacityError(f'Postman cache prefix directory is not private: {prefix_dir}')
artifact_exists = os.path.lexists(path)
metadata_exists = os.path.lexists(meta_path)
for existing in (path, meta_path):
if os.path.lexists(existing):
reject_reparse_components(existing)
if not private_file_ready(existing):
raise PostmanCacheCapacityError(f'Postman cache file is not private: {existing}')
added_items = 0 if artifact_exists else 1
added_files = int(not artifact_exists) + int(not metadata_exists)
added_bytes = (
(0 if artifact_exists else entry['size'])
+ (0 if metadata_exists else len(entry['metadata']))
)
if usage['items'] + added_items > limits['items']:
raise PostmanCacheCapacityError(
f'Postman cache item capacity reached ({usage["items"]}/{limits["items"]})'
)
if usage['bytes'] + added_bytes > limits['bytes']:
raise PostmanCacheCapacityError(
f'Postman cache byte capacity reached ({usage["bytes"]}/{limits["bytes"]})'
)
if added_bytes and free_bytes is None:
free_bytes = int(shutil.disk_usage(root).free)
if added_bytes and free_bytes - reserved_bytes - added_bytes < limits['min_free_bytes']:
raise PostmanCacheCapacityError(
f'Postman cache free-space reserve would be crossed ({free_bytes} bytes free)'
)
usage['items'] += added_items
usage['files'] += added_files
usage['bytes'] += added_bytes
reserved_bytes += added_bytes
plans.append((entry, prefix_dir, path, meta_path, artifact_exists, metadata_exists))
published = []
for entry, prefix_dir, path, meta_path, artifact_exists, metadata_exists in plans:
if deadline is not None and time.monotonic() >= deadline:
return published, True
if not os.path.isdir(prefix_dir):
ensure_private_directory(prefix_dir, reject_reparse=True)
elif not private_directory_ready(prefix_dir):
raise PostmanCacheCapacityError(f'Postman cache prefix directory is not private: {prefix_dir}')
if not artifact_exists:
_write_private_cache_file(path, entry['content'])
if not metadata_exists:
_write_private_cache_file(meta_path, entry['metadata'])
published.append((path, entry['digest'], entry['size']))
return published, False
finally:
release_file_lock(lock, lock_path)
def write_postman_cache(
content,
kind='artifact',
origin=None,
cache_dir=None,
max_artifact_size_mb=20,
cache_max_items=None,
cache_max_bytes=None,
cache_min_free_bytes=None,
):
entry = _prepare_postman_cache_entry(content, kind, origin, max_artifact_size_mb)
published, deadline_reached = _publish_postman_cache_entries(
[entry],
cache_dir,
cache_max_items,
cache_max_bytes,
cache_min_free_bytes,
)
if deadline_reached or not published:
raise PostmanCacheCapacityError('Postman cache publication did not complete')
return published[0]
def postman_target_from_cached_artifact(source, kind, cache_path, digest, origin=None, **extra):
payload = {'source': source, 'kind': kind, 'cache_path': cache_path, 'sha256': digest, 'origin': origin or {}}
payload.update({key: value for key, value in extra.items() if value not in (None, '')})
return json.dumps(payload, separators=(',', ':'), ensure_ascii=False, sort_keys=True)
def cache_package_postman_artifact(file_path, package_source, package, relative_path, cache_dir=None, max_artifact_size_mb=20):
kind = postman_kind_for_path(file_path)
max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024
with open(file_path, 'rb') as f:
content = f.read(max_bytes + 1)
if max_bytes <= 0 or len(content) > max_bytes:
raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB')
origin = {
'package_source': package_source,
'package_name': package.get('name'),
'package_version': package.get('version'),
'package_artifact': package.get('artifact') or package.get('tarball'),
'path': relative_path,
}
cache_path, digest, size = write_postman_cache(content, kind, origin, cache_dir, max_artifact_size_mb)
return postman_target_from_cached_artifact(
f'{package_source}_package',
kind,
cache_path,
digest,
origin,
path=relative_path,
package_name=package.get('name'),
package_version=package.get('version'),
package_artifact=package.get('artifact') or package.get('tarball'),
size=size,
)
def _postman_package_harvest_limits(max_artifacts=None, max_total_bytes=None, max_elapsed_sec=None):
artifacts = int(
getattr(scan_config, 'postman_package_harvest_max_artifacts', 100)
if max_artifacts is None else max_artifacts
)
total_bytes = int(
getattr(scan_config, 'postman_package_harvest_max_bytes', 128 * 1024 * 1024)
if max_total_bytes is None else max_total_bytes
)
elapsed_sec = float(
getattr(scan_config, 'postman_package_harvest_max_elapsed_sec', 30.0)
if max_elapsed_sec is None else max_elapsed_sec
)
if artifacts <= 0 or total_bytes <= 0 or elapsed_sec <= 0 or not math.isfinite(elapsed_sec):
raise ValueError('Postman package harvest limits must be finite and positive')
return artifacts, total_bytes, elapsed_sec
def _add_postman_harvest_warning(warnings, detail, force=False):
message = f'Optional Postman package harvesting degraded: {detail}'[:POSTMAN_HARVEST_WARNING_MAX_CHARS]
if message in warnings:
return
if len(warnings) < POSTMAN_HARVEST_MAX_WARNINGS:
warnings.append(message)
logger.warning(message)
elif force:
warnings[-1] = message
logger.warning(message)
def _attach_postman_harvest_warnings(results, warnings):
if not warnings:
return results
merged = list(results.get('warnings') or [])
for warning in warnings:
if warning not in merged:
merged.append(warning)
results['warnings'] = merged
results['warning_classes'] = sorted(set(list(results.get('warning_classes') or []) + ['postman_package_harvest']))
results['degraded'] = True
return results
def _bounded_postman_walk(root_dir, deadline):
stack = [root_dir]
while stack:
_raise_if_scan_slot_fatal()
if time.monotonic() >= deadline:
yield None, (), ()
return
directory = stack.pop()
try:
entries = os.scandir(directory)
except OSError:
continue
try:
for entry in entries:
_raise_if_scan_slot_fatal()
if time.monotonic() >= deadline:
yield None, (), ()
return
try:
if entry.is_dir(follow_symlinks=False):
if not entry.is_symlink() and not is_reparse_point(entry.path):
stack.append(entry.path)
continue
except OSError:
continue
yield directory, (), (entry.name,)
finally:
entries.close()
def find_postman_artifacts(
root_dir,
package_source,
package,
cache_dir=None,
max_artifact_size_mb=20,
warnings=None,
max_artifacts=None,
max_total_bytes=None,
max_elapsed_sec=None,
):
targets = []
if not root_dir or not os.path.isdir(root_dir):
return targets
warning_sink = warnings if warnings is not None else []
artifact_limit, total_byte_limit, elapsed_limit = _postman_package_harvest_limits(
max_artifacts,
max_total_bytes,
max_elapsed_sec,
)
max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024
if max_bytes <= 0:
raise ValueError('Postman artifact limit must be finite and positive')
deadline = time.monotonic() + elapsed_limit
suffixes = tuple(suffix for values in API_ARTIFACT_PATTERNS.values() for suffix in values)
prepared = []
seen = set()
examined = 0
examined_bytes = 0
stopped = False
walk = _bounded_postman_walk(root_dir, deadline)
for directory, _, files in walk:
_raise_if_scan_slot_fatal()
if directory is None or time.monotonic() >= deadline:
_add_postman_harvest_warning(
warning_sink,
f'elapsed deadline of {elapsed_limit:g}s reached after examining {examined} artifact(s)',
force=True,
)
stopped = True
break
for name in files:
_raise_if_scan_slot_fatal()
if time.monotonic() >= deadline:
_add_postman_harvest_warning(
warning_sink,
f'elapsed deadline of {elapsed_limit:g}s reached after examining {examined} artifact(s)',
force=True,
)
stopped = True
break
lower_name = name.lower()
if not lower_name.endswith(suffixes):
continue
if examined >= artifact_limit:
_add_postman_harvest_warning(
warning_sink,
f'artifact limit of {artifact_limit} reached; remaining matching files were not examined',
force=True,
)
stopped = True
break
examined += 1
path = os.path.join(directory, name)
try:
details = os.stat(path, follow_symlinks=False)
if not stat.S_ISREG(details.st_mode) or os.path.islink(path) or is_reparse_point(path):
raise ValueError('artifact is not a regular unlinked file')
size = int(details.st_size)
relative_path = os.path.relpath(path, root_dir).replace('\\', '/')
if size > max_bytes:
_add_postman_harvest_warning(
warning_sink,
f'skipped {relative_path[:180]} because it exceeds the {max_artifact_size_mb} MB artifact limit',
)
continue
if examined_bytes + size > total_byte_limit:
_add_postman_harvest_warning(
warning_sink,
f'aggregate byte limit of {total_byte_limit} reached after {examined_bytes} byte(s)',
force=True,
)
stopped = True
break
examined_bytes += size
flags = os.O_RDONLY | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0)
descriptor = os.open(path, flags)
try:
opened = os.fstat(descriptor)
opened_identity = (opened.st_dev, opened.st_ino, opened.st_size, getattr(opened, 'st_mtime_ns', None))
expected_identity = (details.st_dev, details.st_ino, details.st_size, getattr(details, 'st_mtime_ns', None))
if not stat.S_ISREG(opened.st_mode) or opened_identity != expected_identity:
raise ValueError('artifact changed before harvesting')
content = bytearray()
while len(content) < size:
_raise_if_scan_slot_fatal()
if time.monotonic() >= deadline:
raise _PostmanHarvestDeadlineReached()
block = os.read(descriptor, min(1024 * 1024, size - len(content)))
if not block:
raise ValueError('artifact changed while harvesting')
content.extend(block)
if os.read(descriptor, 1):
raise ValueError('artifact grew while harvesting')
current = os.stat(path, follow_symlinks=False)
current_identity = (current.st_dev, current.st_ino, current.st_size, getattr(current, 'st_mtime_ns', None))
if current_identity != opened_identity:
raise ValueError('artifact changed while harvesting')
finally:
os.close(descriptor)
origin = {
'package_source': package_source,
'package_name': package.get('name'),
'package_version': package.get('version'),
'package_artifact': package.get('artifact') or package.get('tarball'),
'path': relative_path,
}
entry = _prepare_postman_cache_entry(bytes(content), postman_kind_for_path(path), origin, max_artifact_size_mb)
if entry['digest'] not in seen:
entry['origin'] = origin
entry['kind'] = postman_kind_for_path(path)
entry['relative_path'] = relative_path
prepared.append(entry)
seen.add(entry['digest'])
except _PostmanHarvestDeadlineReached:
_add_postman_harvest_warning(
warning_sink,
f'elapsed deadline of {elapsed_limit:g}s reached after examining {examined} artifact(s)',
force=True,
)
stopped = True
break
except PostmanCacheValidationError:
raise
except ScanSlotFatalError:
raise
except Exception as e:
relative_path = os.path.relpath(path, root_dir).replace('\\', '/')
_add_postman_harvest_warning(
warning_sink,
f'unable to harvest {relative_path[:180]}: {str(e)[:160]}',
)
if stopped:
break
walk.close()
if prepared and not (stopped and time.monotonic() >= deadline):
_raise_if_scan_slot_fatal()
published, deadline_reached = _publish_postman_cache_entries(prepared, cache_dir, deadline=deadline)
for entry, (cache_path, digest, size) in zip(prepared, published):
targets.append(postman_target_from_cached_artifact(
f'{package_source}_package',
entry['kind'],
cache_path,
digest,
entry['origin'],
path=entry['relative_path'],
package_name=package.get('name'),
package_version=package.get('version'),
package_artifact=package.get('artifact') or package.get('tarball'),
size=size,
))
if deadline_reached:
_add_postman_harvest_warning(
warning_sink,
f'elapsed deadline of {elapsed_limit:g}s reached while publishing {len(published)} artifact(s)',
force=True,
)
return targets
class GitHubTokenPool:
def __init__(self, token_entries=None, token=None, status=None, code_search_rpm_per_token=8, fallback_cooldown=1800):
entries = []
for index, entry in enumerate(token_entries or []):
if isinstance(entry, str):
item = {'name': f'github_{index + 1}', 'token': entry}
elif isinstance(entry, dict):
item = dict(entry)
item.setdefault('name', f'github_{index + 1}')
else:
continue
if item.get('token'):
entries.append(item)
if token and not any(item.get('token') == token for item in entries):
entries.append({'name': 'token', 'token': token})
self.entries = entries
self.status = status if isinstance(status, dict) else {}
self.index = 0
self.last_code_search_at = {}
self.code_search_interval = 60.0 / max(1, int(code_search_rpm_per_token or 8))
self.fallback_cooldown = int(fallback_cooldown or 1800)
def _status_for(self, entry):
return self.status.setdefault(entry.get('name'), {})
def _disabled_until(self, entry):
value = self._status_for(entry).get('disabled_until')
if value == 'manual':
return 'manual'
return parse_postman_time(value)
def _available_entries(self):
now = utc_now_for_postman()
available = []
for entry in self.entries:
disabled_until = self._disabled_until(entry)
if disabled_until == 'manual':
continue
if disabled_until and disabled_until > now:
continue
available.append(entry)
return available
def _mark_unavailable(self, entry, category, message, reset_at=None):
item = self._status_for(entry)
if category in ('auth_invalid', 'auth_forbidden'):
item['disabled_until'] = 'manual'
else:
item['disabled_until'] = reset_at or (utc_now_for_postman() + timedelta(seconds=self.fallback_cooldown)).isoformat(timespec='seconds')
item['disabled_reason'] = category
item['last_error'] = str(message or '')[:500]
item['last_failure_at'] = utc_now_for_postman().isoformat(timespec='seconds')
item['failures'] = int(item.get('failures', 0) or 0) + 1
def _next_entry(self):
available = self._available_entries()
if not available:
return None
for _ in range(len(self.entries)):
entry = self.entries[self.index % len(self.entries)]
self.index = (self.index + 1) % len(self.entries)
if entry in available:
return entry
return available[0]
def _sleep_for_resource(self, entry, resource, deadline=None):
if resource != 'code_search':
return
name = entry.get('name')
last = self.last_code_search_at.get(name)
now = time.monotonic()
if last is not None:
delay = self.code_search_interval - (now - last)
if delay > 0:
if deadline is not None and now + delay >= deadline:
remaining = max(0.0, deadline - now)
if remaining:
time.sleep(remaining)
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during code-search pacing')
time.sleep(delay)
self.last_code_search_at[name] = time.monotonic()
def _wait_until_any_available(self, deadline=None):
waits = []
now = utc_now_for_postman()
manual_count = 0
for entry in self.entries:
disabled_until = self._disabled_until(entry)
if disabled_until == 'manual':
manual_count += 1
if disabled_until and disabled_until != 'manual' and disabled_until > now:
waits.append((disabled_until - now).total_seconds())
if manual_count >= len(self.entries):
raise RateLimitError('postman', 'All GitHub tokens are manually disabled for Postman discovery', category='auth_invalid', retryable=False, auth_related=True)
wait_for = min(waits) if waits else self.fallback_cooldown
wait_for = max(1, min(wait_for, self.fallback_cooldown))
if deadline is not None and time.monotonic() + wait_for >= deadline:
remaining = max(0.0, deadline - time.monotonic())
logger.warning(f'All GitHub tokens unavailable for Postman discovery; deadline in {remaining:.1f}s')
if remaining:
time.sleep(remaining)
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached while all GitHub tokens were unavailable')
logger.warning(f'All GitHub tokens unavailable for Postman discovery; sleeping {wait_for:.0f}s')
time.sleep(wait_for)
def _token_core_status(self, entry, timeout=10):
request_headers = {
'Accept': 'application/vnd.github+json',
'User-Agent': 'GitSecretsScanner/2.0',
'Authorization': f"Bearer {entry.get('token')}",
}
try:
response = api_request('GET', 'https://api.github.com/user', headers=request_headers, timeout=timeout, max_retries=1)
except Exception:
return 'unknown'
try:
if response.status_code == 200:
return 'valid'
if response.status_code == 401:
return 'invalid'
if response.status_code in (403, 429) and response.headers.get('X-RateLimit-Remaining') == '0':
return 'rate_limited'
return f'http_{response.status_code}'
finally:
response.close()
def _mark_github_api_error(self, entry, api_error):
category = getattr(api_error, 'category', 'api')
if category not in ('rate_limit', 'secondary_rate_limit', 'auth_invalid', 'auth_forbidden'):
return False
# Code search may return auth-like 401/403 even when the token is valid for core GitHub API.
# Confirm against /user before permanently disabling the token as dead.
if category in ('auth_invalid', 'auth_forbidden'):
core_status = self._token_core_status(entry)
if core_status == 'valid':
self._mark_unavailable(entry, f'{category}_resource', str(api_error), api_error.reset_at)
return True
if core_status in ('unknown', 'rate_limited'):
self._mark_unavailable(entry, f'{category}_unconfirmed', str(api_error), api_error.reset_at)
return True
self._mark_unavailable(entry, category, str(api_error), api_error.reset_at)
return True
def request(self, method, url, params=None, headers=None, timeout=30, resource='core', deadline=None, use_proxy=None):
if not self.entries:
raise RateLimitError('postman', 'GitHub token is required for Postman GitHub code search', category='auth_invalid', retryable=False, auth_related=True)
attempts = 0
last_error = None
while attempts < max(1, len(self.entries) * 2):
if deadline is not None and time.monotonic() >= deadline:
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GitHub request')
entry = self._next_entry()
if not entry:
self._wait_until_any_available(deadline)
attempts += 1
continue
self._sleep_for_resource(entry, resource, deadline)
request_headers = {'Accept': 'application/vnd.github+json', 'User-Agent': 'GitSecretsScanner/2.0'}
request_headers.update(headers or {})
request_headers['Authorization'] = f"Bearer {entry.get('token')}"
try:
response = api_request(
method, url, headers=request_headers, params=params, timeout=timeout, deadline=deadline,
use_proxy=use_proxy,
)
if response.status_code in (401, 403, 429):
api_error = github_api_error(response)
response.close()
if self._mark_github_api_error(entry, api_error):
last_error = api_error
attempts += 1
continue
try:
response.raise_for_status()
except requests.exceptions.HTTPError as e:
raise github_api_error(e.response) from e
return response
except (requests.exceptions.RequestException, ApiRequestError) as e:
last_error = e
logger.warning(f'GitHub request failed for Postman discovery with token {entry.get("name")}: {str(e)[:300]}')
attempts += 1
delay = min(2, attempts)
if deadline is not None and time.monotonic() + delay >= deadline:
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during GitHub request retry') from e
time.sleep(delay)
continue
if last_error:
raise RateLimitError('postman', f'GitHub Postman discovery request failed after retries: {last_error}', category='network', retryable=True, auth_related=False)
raise RateLimitError('postman', 'All GitHub tokens are unavailable for Postman discovery', category='rate_limit', retryable=True, auth_related=True)
def latest_github_path_commit(repo, path, pool, request_timeout=20, deadline=None):
response = pool.request(
'GET',
f'https://api.github.com/repos/{repo}/commits',
params={'path': path, 'per_page': 1},
timeout=request_timeout,
resource='core',
deadline=deadline,
)
data = response.json()
if not data:
return None
commit = data[0].get('commit') or {}
committer = commit.get('committer') or {}
author = commit.get('author') or {}
return committer.get('date') or author.get('date')
def github_content_bytes(item, pool, request_timeout=20, max_artifact_size_mb=20, deadline=None):
response = pool.request(
'GET', item.get('url'), timeout=request_timeout, resource='core', deadline=deadline,
use_proxy=False,
)
data = response.json()
size = int(data.get('size') or 0)
max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024
if max_bytes and size > max_bytes:
raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB')
content = str(data.get('content') or '')
if str(data.get('encoding') or '').lower() == 'base64':
decoded = base64.b64decode(re.sub(r'\s+', '', content))
else:
decoded = content.encode('utf-8', errors='replace')
if max_bytes <= 0 or len(decoded) > max_bytes:
raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB')
return decoded
def github_postman_item_to_target(item, kind, cache_path, digest, size=None, commit_date=None):
repo = (item.get('repository') or {}).get('full_name') or ''
path = item.get('path') or ''
sha = item.get('sha') or digest or ''
origin = {
'provider': 'github',
'repo': repo,
'path': path,
'sha': sha,
'html_url': item.get('html_url'),
'api_url': item.get('url'),
'commit_date': commit_date,
}
return postman_target_from_cached_artifact(
'github_code',
kind,
cache_path,
digest,
origin,
repo=repo,
path=path,
sha=sha,
html_url=item.get('html_url'),
api_url=item.get('url'),
commit_date=commit_date,
size=size,
)
def fetch_github_postman_targets(query, pages=1, per_page=100, token_entries=None, token=None, search_kinds=None, cache_dir=None, max_file_age_days=365, max_artifact_size_mb=20, request_timeout=20, code_search_rpm_per_token=8, all_tokens_cooldown=1800, auth_status=None, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, discovery_max_artifacts=None, discovery_max_artifacts_per_page=None, discovery_max_bytes=None, discovery_max_elapsed_sec=None):
if not query:
return []
budget = _PostmanDiscoveryBudget(
'GitHub code-search',
discovery_max_artifacts,
discovery_max_artifacts_per_page,
discovery_max_bytes,
discovery_max_elapsed_sec,
)
pool = GitHubTokenPool(token_entries, token, auth_status, code_search_rpm_per_token, all_tokens_cooldown)
kinds = normalize_postman_search_kinds(search_kinds)
per_page = max(1, min(int(per_page or 100), 100))
max_pages = min(max(1, int(pages or 1)), max(1, (1000 + per_page - 1) // per_page))
cutoff = utc_now_for_postman() - timedelta(days=int(max_file_age_days or 0)) if int(max_file_age_days or 0) > 0 else None
targets = []
seen_targets = set()
seen_digests = set()
stop_cycle = False
for kind in kinds:
if stop_cycle or not budget.check_deadline():
break
templates = API_ARTIFACT_SEARCHES.get(kind) or API_ARTIFACT_SEARCHES['generic']
for template in templates:
if stop_cycle or not budget.check_deadline():
break
search_query = template.format(query=query).strip()
logger.info(f"Fetching GitHub API artifact {kind} targets for: {search_query!r}")
known_pages = 0
for page in range(1, max_pages + 1):
if not budget.check_deadline():
stop_cycle = True
break
try:
response = pool.request(
'GET',
'https://api.github.com/search/code',
params={'q': search_query, 'per_page': per_page, 'page': page, 'sort': 'indexed', 'order': 'desc'},
timeout=budget.request_timeout(request_timeout),
resource='code_search',
deadline=budget.deadline,
)
except _PostmanHarvestDeadlineReached as e:
budget.stop(str(e))
stop_cycle = True
break
except RateLimitError as e:
if getattr(e, 'category', '') == 'network':
logger.warning(f'GitHub API artifact {kind} network failure on page {page}: {str(e)[:300]}')
raise
payload = response.json()
if not isinstance(payload, dict) or 'items' not in payload or not isinstance(payload.get('items'), list):
raise ApiRequestError('invalid GitHub code search payload')
items = payload.get('items') or []
if not items:
logger.info(f'GitHub API artifact {kind} page {page}: no results')
break
page_targets = []
prepared = []
page_artifacts = 0
for item in items:
admitted, scope = budget.admit(page_artifacts)
if not admitted:
stop_cycle = scope == 'cycle'
break
page_artifacts += 1
repo = (item.get('repository') or {}).get('full_name') or ''
path = item.get('path') or ''
commit_date = None
if cutoff:
try:
commit_date = latest_github_path_commit(
repo, path, pool, budget.request_timeout(request_timeout), budget.deadline,
)
parsed_commit = parse_postman_time(commit_date)
if not parsed_commit or parsed_commit < cutoff:
continue
except _PostmanHarvestDeadlineReached as e:
budget.stop(str(e))
stop_cycle = True
break
except RateLimitError as e:
if getattr(e, 'category', '') in ('network', 'not_found'):
logger.warning(f'Skipping API artifact freshness check for {repo}:{path}: {str(e)[:300]}')
continue
raise
except Exception as e:
logger.warning(f'Unable to check API artifact freshness for {repo}:{path}: {str(e)}')
continue
try:
content = github_content_bytes(
item,
pool,
budget.request_timeout(request_timeout),
max_artifact_size_mb,
budget.deadline,
)
if not budget.account_bytes(len(content)):
stop_cycle = True
break
origin = {'provider': 'github', 'repo': repo, 'path': path, 'sha': item.get('sha'), 'html_url': item.get('html_url'), 'api_url': item.get('url'), 'commit_date': commit_date}
detected_kind = postman_kind_for_path(path)
entry = _prepare_postman_cache_entry(content, detected_kind or kind, origin, max_artifact_size_mb)
if entry['digest'] in seen_digests:
continue
seen_digests.add(entry['digest'])
prepared.append({
'entry': entry,
'item': item,
'kind': detected_kind or kind,
'commit_date': commit_date,
})
except _PostmanHarvestDeadlineReached as e:
budget.stop(str(e))
stop_cycle = True
break
except RateLimitError as e:
if getattr(e, 'category', '') in ('network', 'not_found'):
logger.warning(f'Skipping GitHub API artifact for {repo}:{path}: {str(e)[:300]}')
continue
raise
except Exception as e:
if isinstance(e, PostmanCacheCapacityError):
raise
logger.warning(f'Unable to cache GitHub API artifact {repo}:{path}: {str(e)}')
published, deadline_reached = _publish_postman_discovery_batch(prepared, cache_dir, budget)
for record, (cache_path, digest, size) in published:
target = github_postman_item_to_target(
record['item'], record['kind'], cache_path, digest, size, record['commit_date'],
)
identity = postman_target_identity(target)
if identity not in seen_targets:
targets.append(target)
page_targets.append(target)
seen_targets.add(identity)
logger.info(f'GitHub API artifact {kind} page {page}: fetched {len(items)}, queued candidates {len(page_targets)}')
if stop_on_seen_pages and page >= max(1, int(min_pages_before_stop or 1)):
if page_targets and page_is_known(
page_targets, known_targets, normalize_target, known_target_lookup,
):
known_pages += 1
logger.info(f'GitHub API artifact {kind} page {page}: all targets known ({known_pages}/{seen_page_threshold})')
if known_pages >= max(1, int(seen_page_threshold or 1)):
logger.info(f'Stopping API artifact {kind} pagination early after {known_pages} known page(s)')
break
else:
known_pages = 0
if deadline_reached or stop_cycle or budget.exhausted():
stop_cycle = True
break
return targets
def fetch_github_repo_items(query, pages, per_page=100, token=None, sort_by="updated", sort_order="desc", created_filter="any", raise_rate_limit=False, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None):
"""Fetch GitHub repositories with metadata for filtering."""
repos = []
headers = {'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/vnd.github.v3+json'}
if token:
headers['Authorization'] = f'Bearer {token}'
# Build search query with filters
search_query = query if query else "*"
# Add created date filter
if created_filter != "any":
date_filters = {
"today": "created:>{}".format((datetime.now() - timedelta(days=1)).strftime("%Y-%m-%d")),
"week": "created:>{}".format((datetime.now() - timedelta(weeks=1)).strftime("%Y-%m-%d")),
"month": "created:>{}".format((datetime.now() - timedelta(days=30)).strftime("%Y-%m-%d")),
"year": "created:>{}".format((datetime.now() - timedelta(days=365)).strftime("%Y-%m-%d"))
}
if created_filter in date_filters:
search_query += f" {date_filters[created_filter]}"
logger.info(f"Fetching GitHub repositories for: '{search_query}' sorted by {sort_by} ({sort_order})...")
seen_pages = 0
successful_pages = 0
for page in range(1, pages + 1):
url = "https://api.github.com/search/repositories"
params = {
'q': search_query,
'sort': sort_by,
'order': sort_order,
'per_page': per_page,
'page': page,
}
try:
response = api_request('GET', url, headers=headers, params=params, timeout=30)
response.raise_for_status()
# Handle rate limits
if response.status_code == 403 and 'X-RateLimit-Remaining' in response.headers:
if int(response.headers['X-RateLimit-Remaining']) == 0:
reset_time = datetime.fromtimestamp(int(response.headers['X-RateLimit-Reset']))
wait_seconds = (reset_time - datetime.now()).total_seconds() + 10
logger.warning(f"Rate limit exceeded. Resuming at {reset_time}. Waiting {wait_seconds:.0f} seconds...")
time.sleep(wait_seconds)
continue
data = response.json()
if not isinstance(data, dict) or 'items' not in data or not isinstance(data.get('items'), list):
raise ValueError('invalid GitHub repository search payload')
successful_pages += 1
if not data.get('items'):
logger.info(f"Page {page} returned no results. Stopping.")
break
page_repos = [github_repo_to_target(item) for item in data['items'] if item.get('clone_url')]
repos.extend(page_repos)
logger.info(f"Page {page}: Fetched {len(data['items'])} repositories")
if stop_on_seen_pages and page >= max(1, min_pages_before_stop):
if page_is_known(
[item.get('url') for item in page_repos], known_targets,
normalize_target, known_target_lookup,
):
seen_pages += 1
logger.info(f"Page {page}: all GitHub repositories are already queued/checked ({seen_pages}/{seen_page_threshold})")
if seen_pages >= max(1, seen_page_threshold):
logger.info(f"Stopping GitHub pagination early after {seen_pages} all-known page(s)")
break
else:
seen_pages = 0
except requests.exceptions.HTTPError as e:
api_error = github_api_error(e.response)
if raise_rate_limit:
raise api_error from e
logger.error(str(api_error))
raise api_error from e
except Exception as e:
if isinstance(e, ApiRequestError):
raise
logger.error(f"Error fetching page {page}: {str(e)}")
raise ApiRequestError(f'GitHub discovery payload failed: {e}') from e
return repos
def fetch_github_repos(query, pages, per_page=100, token=None, sort_by="updated", sort_order="desc", created_filter="any", raise_rate_limit=False, **kwargs):
"""Fetch GitHub repository clone URLs with pagination and authentication."""
return [
item['url'] for item in fetch_github_repo_items(
query, pages, per_page, token, sort_by, sort_order, created_filter, raise_rate_limit, **kwargs
)
if item.get('url')
]
GHARCHIVE_REPO_TERMS = (
'ai', 'agent', 'assistant', 'bot', 'chat', 'chatbot', 'gpt', 'llm', 'rag', 'mcp',
'model', 'inference', 'embedding', 'vector', 'semantic', 'prompt', 'workflow',
'copilot', 'codegen', 'langchain', 'llamaindex', 'litellm', 'ollama', 'vllm',
'claude', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq', 'grok', 'xai', 'openrouter', 'replicate',
'deepseek', 'zai', 'glm', 'zhipu', 'dashscope', 'bedrock', 'vertex', 'foundry',
'aiplatform', 'boto3', 'terraform', 'cloudbuild', 'service-account', 'credentials',
)
GHARCHIVE_PATH_TERMS = (
'.env', 'env.', 'secret', 'secrets', 'credential', 'credentials', 'apikey', 'api_key',
'token', 'tokens', 'key', 'keys', 'config', 'settings', 'appsettings', 'docker-compose',
'compose.yaml', 'compose.yml', '.github/workflows', 'workflow', 'deploy', 'deployment',
'kubernetes', 'k8s', 'helm', 'terraform', 'tfvars', 'notebook', '.ipynb', 'postman',
'collection.json', 'environment.json', 'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq',
'xai', 'grok', 'openrouter', 'replicate', 'deepseek', 'zai', 'glm', 'dashscope',
'aws_access_key_id', 'aws_secret_access_key', 'bedrock-runtime', 'boto3',
'google_application_credentials', 'service-account', 'service_account', 'credentials.json',
'application_default_credentials', 'vertexai', 'aiplatform', 'provider.tf', 'cloudbuild.yaml',
)
GHARCHIVE_FILE_FETCH_TERMS = (
'.env', 'env.', '.env.', '.env-', 'secret', 'secrets', 'credential', 'credentials',
'credentials.json', 'service-account', 'service_account', 'application_default_credentials',
'google_application_credentials', 'terraform.tfvars', '.tfvars', 'provider.tf',
'cloudbuild.yaml', 'cloudbuild.yml', '.github/workflows', 'docker-compose',
'compose.yml', 'compose.yaml', 'appsettings', '.ipynb', 'notebook',
'bedrock', 'vertex', 'aiplatform', 'boto3', 'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot',
'groq', 'xai', 'grok', 'openrouter', 'replicate', 'deepseek', 'zai', 'glm', 'dashscope',
)
GHARCHIVE_FILE_SKIP_SUFFIXES = (
'.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg', '.ico', '.pdf', '.zip', '.gz', '.tgz',
'.tar', '.rar', '.7z', '.bin', '.safetensors', '.pt', '.pth', '.onnx', '.parquet', '.arrow',
'.mp4', '.mov', '.avi', '.mp3', '.wav', '.lock', '.sum', '.min.js', '.map', '.pyc',
)
GHARCHIVE_FILE_SKIP_PARTS = (
'/__pycache__/', '/node_modules/', '/vendor/', '/.git/', '/dist/', '/build/', '/target/',
)
GITHUB_GIST_FILE_TERMS = (
'.env', 'env', 'secret', 'secrets', 'credential', 'credentials', 'apikey', 'api_key',
'token', 'tokens', 'key', 'keys', 'config', 'settings', 'appsettings', 'credentials.json',
'service-account', 'service_account', 'application_default_credentials', 'terraform',
'tfvars', 'provider.tf', 'docker-compose', 'compose.yml', 'compose.yaml', 'workflow',
'cloudbuild', 'ipynb', 'notebook', 'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq',
'xai', 'grok', 'openrouter', 'replicate', 'deepseek', 'zai', 'glm', 'dashscope',
'bedrock', 'vertex', 'aiplatform', 'postman',
)
GHARCHIVE_MESSAGE_TERMS = (
'api key', 'apikey', 'token', 'secret', 'credential', '.env', 'config', 'settings',
'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq', 'xai', 'grok', 'openrouter',
'replicate', 'deepseek', 'zai', 'glm', 'dashscope', 'bedrock', 'vertex',
'aws_access_key_id', 'aws_secret_access_key', 'google_application_credentials',
'service account', 'credentials.json', 'vertexai', 'aiplatform', 'bedrock-runtime',
'terraform', 'tfvars', 'cloudbuild',
)
GHARCHIVE_MAX_HOURS_BACK = 168
def _contains_term(text, terms):
text = str(text or '').lower()
return any(term in text for term in terms)
def github_archive_event_score(event):
repo = event.get('repo') if isinstance(event.get('repo'), dict) else {}
repo_name = str(repo.get('name') or '').lower()
score = 0
if _contains_term(repo_name.replace('-', ' ').replace('_', ' '), GHARCHIVE_REPO_TERMS):
score += 8
payload = event.get('payload') if isinstance(event.get('payload'), dict) else {}
ref = str(payload.get('ref') or '').lower()
if _contains_term(ref, GHARCHIVE_REPO_TERMS):
score += 3
commits = payload.get('commits') if isinstance(payload.get('commits'), list) else []
for commit in commits[:20]:
if not isinstance(commit, dict):
continue
message = str(commit.get('message') or '')
if _contains_term(message, GHARCHIVE_MESSAGE_TERMS):
score += 5
for key in ('added', 'modified', 'removed'):
paths = commit.get(key) if isinstance(commit.get(key), list) else []
for path in paths[:50]:
if _contains_term(path, GHARCHIVE_PATH_TERMS):
score += 4
if event.get('type') == 'PushEvent':
score += 1
return score
def github_archive_event_branch(event):
payload = event.get('payload') if isinstance(event.get('payload'), dict) else {}
ref = str(payload.get('ref') or '').strip()
if ref.startswith('refs/heads/'):
return ref[len('refs/heads/'):], ref
if payload.get('ref_type') == 'branch' and ref:
return ref, f'refs/heads/{ref}'
return '', ref
def github_archive_target_payload(repo_name, event, score, archive_hour):
payload = event.get('payload') if isinstance(event.get('payload'), dict) else {}
branch, ref = github_archive_event_branch(event)
data = {
'url': f'https://github.com/{repo_name}.git',
'repo': repo_name,
'source': 'gharchive',
'event_type': event.get('type') or '',
'archive_hour': archive_hour.isoformat(),
'score': int(score or 0),
'ref': ref,
'branch': branch,
'head_sha': payload.get('head') or payload.get('after') or '',
}
return json.dumps(data, separators=(',', ':'), ensure_ascii=False, sort_keys=True)
def gharchive_changed_paths(event):
payload = event.get('payload') if isinstance(event.get('payload'), dict) else {}
commits = payload.get('commits') if isinstance(payload.get('commits'), list) else []
for commit in commits[:20]:
if not isinstance(commit, dict):
continue
sha = commit.get('sha') or commit.get('id') or payload.get('head') or payload.get('after') or ''
for key in ('added', 'modified'):
paths = commit.get(key) if isinstance(commit.get(key), list) else []
for path in paths[:80]:
path = str(path or '').strip().replace('\\', '/')
if path:
yield sha, path
def gharchive_commit_urls(event, repo_name):
payload = event.get('payload') if isinstance(event.get('payload'), dict) else {}
commits = payload.get('commits') if isinstance(payload.get('commits'), list) else []
seen = set()
for commit in commits[:5]:
if not isinstance(commit, dict):
continue
sha = commit.get('sha') or commit.get('id') or ''
url = commit.get('url') or (f'https://api.github.com/repos/{repo_name}/commits/{sha}' if sha else '')
if sha and url and sha not in seen:
seen.add(sha)
yield sha, url
head = payload.get('head') or payload.get('after') or ''
if head and head not in seen:
yield head, f'https://api.github.com/repos/{repo_name}/commits/{head}'
def fetch_github_commit_files(event, repo_name, headers, request_timeout=20, deadline=None):
for sha, url in gharchive_commit_urls(event, repo_name):
if deadline is not None and time.monotonic() >= deadline:
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during commit lookup')
timeout = request_timeout
if deadline is not None:
timeout = max(0.01, min(float(request_timeout or 20), deadline - time.monotonic()))
try:
response = api_request(
'GET', url, headers=headers, timeout=timeout, max_retries=2,
retry_delay=1, deadline=deadline,
)
if response.status_code == 404:
continue
if response.status_code in (401, 403, 429):
raise github_api_error(response)
response.raise_for_status()
data = response.json()
except (requests.exceptions.RequestException, ApiRequestError) as e:
logger.warning(f'Unable to fetch commit files {repo_name}@{sha}: {str(e)[:300]}')
continue
files = data.get('files') if isinstance(data.get('files'), list) else []
for item in files[:100]:
if not isinstance(item, dict):
continue
status = str(item.get('status') or '')
if status not in ('added', 'modified', 'renamed'):
continue
path = item.get('filename') or item.get('previous_filename') or ''
raw_url = item.get('raw_url') or github_raw_url(repo_name, sha, path)
if path and raw_url:
yield sha, str(path).replace('\\', '/'), raw_url
def gharchive_path_interesting(path):
lowered = str(path or '').lower()
if not lowered or lowered.endswith(GHARCHIVE_FILE_SKIP_SUFFIXES):
return False
normalized = '/' + lowered.strip('/')
if any(part in normalized for part in GHARCHIVE_FILE_SKIP_PARTS):
return False
return _contains_term(lowered, GHARCHIVE_FILE_FETCH_TERMS)
def github_raw_url(repo_name, sha, path):
if not repo_name or not sha or not path:
return ''
return f'https://raw.githubusercontent.com/{repo_name}/{sha}/{quote(path)}'
def gharchive_file_target(source, kind, cache_path, digest, origin=None, **extra):
return postman_target_from_cached_artifact(source, kind, cache_path, digest, origin, **extra)
class GHArchiveBoundsError(ValueError):
pass
class GHArchiveCacheCapacityError(RuntimeError):
pass
def gharchive_item_lock_path(cache_dir, artifact_path):
digest = hashlib.sha256(canonical_path(artifact_path).encode('utf-8', errors='strict')).hexdigest()
return os.path.join(cache_dir, f'.item-lock-{int(digest[:8], 16) % 64:02d}.lock')
def _gharchive_limits():
return {
'max_items': max(1, int(scan_config.gharchive_cache_max_items)),
'max_bytes': max(1, int(scan_config.gharchive_cache_max_bytes)),
'min_free_bytes': max(0, int(scan_config.gharchive_cache_min_free_bytes)),
'download_max_bytes': max(1, int(scan_config.gharchive_download_max_bytes)),
'decompressed_max_bytes': max(1, int(scan_config.gharchive_decompressed_max_bytes)),
'max_events': max(1, int(scan_config.gharchive_max_events)),
'max_line_bytes': max(1, int(scan_config.gharchive_max_line_bytes)),
'lock_timeout_sec': max(1, int(scan_config.gharchive_cache_lock_timeout_sec)),
}
def require_gharchive_cache_dir(cache_dir=None):
cache_dir = canonical_path(cache_dir or scan_config.gharchive_cache_dir)
runtime_dir = canonical_path(scan_config.runtime_dir)
require_private_directory(runtime_dir, create=False)
try:
contained = os.path.commonpath((runtime_dir, cache_dir)) == runtime_dir
except ValueError:
contained = False
if not contained or cache_dir == runtime_dir:
raise RuntimeError('GHArchive cache must be a dedicated private directory under runtime_dir')
return require_private_directory(cache_dir, create=True)
def iter_gharchive_lines(path, decompressed_max_bytes=None, max_events=None, max_line_bytes=None, deadline=None):
_raise_if_scan_slot_fatal()
limits = _gharchive_limits()
decompressed_max_bytes = max(1, int(decompressed_max_bytes or limits['decompressed_max_bytes']))
max_events = max(1, int(max_events or limits['max_events']))
max_line_bytes = max(1, int(max_line_bytes or limits['max_line_bytes']))
total = 0
events = 0
with gzip.open(path, 'rb') as archive:
while True:
_raise_if_scan_slot_fatal()
if deadline is not None and time.monotonic() >= deadline:
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached while reading GHArchive data')
raw_line = archive.readline(max_line_bytes + 1)
if not raw_line:
break
if len(raw_line) > max_line_bytes:
raise GHArchiveBoundsError('GHArchive line exceeds configured byte limit')
total += len(raw_line)
if total > decompressed_max_bytes:
raise GHArchiveBoundsError('GHArchive decompressed bytes exceed configured limit')
events += 1
if events > max_events:
raise GHArchiveBoundsError('GHArchive event count exceeds configured limit')
yield raw_line
def validate_gharchive_gzip(path, decompressed_max_bytes=None, max_events=None, max_line_bytes=None, deadline=None):
for _ in iter_gharchive_lines(path, decompressed_max_bytes, max_events, max_line_bytes, deadline):
pass
return True
def gharchive_cache_usage(cache_dir):
usage = {'items': 0, 'files': 0, 'bytes': 0}
with os.scandir(cache_dir) as entries:
for entry in entries:
if entry.is_symlink() or is_reparse_point(entry.path) or not entry.is_file(follow_symlinks=False):
continue
size = entry.stat(follow_symlinks=False).st_size
usage['files'] += 1
usage['bytes'] += max(0, int(size))
if entry.name.endswith('.json.gz'):
usage['items'] += 1
return usage
def _gharchive_artifacts_oldest(cache_dir, protected):
candidates = []
with os.scandir(cache_dir) as entries:
for entry in entries:
canonical = canonical_path(entry.path)
if (
canonical in protected
or not entry.name.endswith('.json.gz')
or entry.is_symlink()
or is_reparse_point(entry.path)
or not entry.is_file(follow_symlinks=False)
):
continue
details = entry.stat(follow_symlinks=False)
candidates.append((details.st_mtime_ns, canonical))
return [path for _, path in sorted(candidates)]
def _gharchive_evict_for_capacity(cache_dir, required_items=0, required_bytes=0, protected=None):
limits = _gharchive_limits()
protected = {canonical_path(path) for path in (protected or set())}
while True:
usage = gharchive_cache_usage(cache_dir)
try:
free_bytes = shutil.disk_usage(cache_dir).free
except OSError as exc:
raise GHArchiveCacheCapacityError(f'unable to inspect GHArchive cache free space: {exc}') from exc
if (
usage['items'] + int(required_items) <= limits['max_items']
and usage['bytes'] + int(required_bytes) <= limits['max_bytes']
and free_bytes - int(required_bytes) >= limits['min_free_bytes']
):
return usage
evicted = False
for candidate in _gharchive_artifacts_oldest(cache_dir, protected):
item_lock = PrivateFileLock(gharchive_item_lock_path(cache_dir, candidate))
try:
item_lock.acquire()
except BlockingIOError:
continue
try:
if os.path.isfile(candidate):
reject_reparse_components(candidate)
os.remove(candidate)
evicted = True
break
finally:
item_lock.release()
if not evicted:
raise GHArchiveCacheCapacityError('GHArchive cache quota or free-space reserve cannot be satisfied without evicting an active entry')
def _acquire_cached_gharchive_hour(hour, cache_dir=None, request_timeout=120, retries=4, deadline=None):
cache_dir = require_gharchive_cache_dir(cache_dir)
limits = _gharchive_limits()
name = f'{hour:%Y-%m-%d-%H}.json.gz'
path = canonical_path(os.path.join(cache_dir, name))
global_lock_path = os.path.join(cache_dir, '.cache.lock')
if deadline is not None and time.monotonic() >= deadline:
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GHArchive cache access')
lock_timeout = limits['lock_timeout_sec']
if deadline is not None:
lock_timeout = min(lock_timeout, max(0.01, deadline - time.monotonic()))
try:
global_lock = acquire_file_lock(global_lock_path, timeout_sec=lock_timeout)
except TimeoutError as exc:
if deadline is not None and time.monotonic() >= deadline:
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached waiting for GHArchive cache lock') from exc
raise
item_lock = None
try:
item_lock_timeout = limits['lock_timeout_sec']
if deadline is not None:
remaining = deadline - time.monotonic()
if remaining <= 0:
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GHArchive item lock')
item_lock_timeout = min(item_lock_timeout, max(0.01, remaining))
try:
item_lock = acquire_file_lock(
gharchive_item_lock_path(cache_dir, path),
timeout_sec=item_lock_timeout,
)
except TimeoutError as exc:
if deadline is not None and time.monotonic() >= deadline:
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached waiting for GHArchive item lock') from exc
raise
for entry in os.scandir(cache_dir):
if entry.name.startswith(name + '.part.') and entry.is_file(follow_symlinks=False):
remove_file_quiet(entry.path)
if os.path.lexists(path):
reject_reparse_components(path)
if not private_file_ready(path):
raise RuntimeError(f'GHArchive cache file is not private: {path}')
try:
validate_gharchive_gzip(path, deadline=deadline)
os.utime(path, None)
return path, item_lock
except (OSError, EOFError, gzip.BadGzipFile, GHArchiveBoundsError):
remove_file_quiet(path)
_gharchive_evict_for_capacity(cache_dir, required_items=1, protected={path})
url = f'https://data.gharchive.org/{name}'
last_error = None
attempts = max(1, int(retries or 1))
for attempt in range(attempts):
_raise_if_scan_slot_fatal()
if deadline is not None and time.monotonic() >= deadline:
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GHArchive download')
temp_path = f'{path}.part.{os.getpid()}.{attempt}'
remove_file_quiet(temp_path)
response = None
try:
logger.info(f'Downloading GHArchive {name} attempt {attempt + 1}/{attempts}')
request_deadline = None if deadline is None else max(0.01, deadline - time.monotonic())
response = _direct_request(
'GET', url,
headers={'User-Agent': 'GitSecretsScanner/2.0'},
stream=True,
timeout=(
min(10, request_deadline) if request_deadline is not None else 10,
min(max(30, int(request_timeout or 120)), request_deadline) if request_deadline is not None else max(30, int(request_timeout or 120)),
),
)
if _scan_slot_fatal_event.is_set():
_raise_if_scan_slot_fatal()
if response.status_code == 404:
return None, item_lock
response.raise_for_status()
declared = response.headers.get('Content-Length')
if declared and int(declared) > limits['download_max_bytes']:
raise GHArchiveBoundsError('GHArchive compressed response exceeds configured byte limit')
total = 0
with open(temp_path, 'xb') as output:
for chunk in response.iter_content(chunk_size=1024 * 1024):
_raise_if_scan_slot_fatal()
if deadline is not None and time.monotonic() >= deadline:
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during GHArchive download')
if not chunk:
continue
total += len(chunk)
if total > limits['download_max_bytes']:
raise GHArchiveBoundsError('GHArchive compressed response exceeds configured byte limit')
_gharchive_evict_for_capacity(
cache_dir,
required_bytes=len(chunk),
protected={path, temp_path},
)
output.write(chunk)
output.flush()
os.fsync(output.fileno())
_raise_if_scan_slot_fatal()
harden_private_file(temp_path)
validate_gharchive_gzip(temp_path, deadline=deadline)
_raise_if_scan_slot_fatal()
os.replace(temp_path, path)
harden_private_file(path)
return path, item_lock
except _PostmanHarvestDeadlineReached:
remove_file_quiet(temp_path)
raise
except (
requests.RequestException, OSError, EOFError, ValueError, gzip.BadGzipFile,
GHArchiveBoundsError, GHArchiveCacheCapacityError,
urllib3_exceptions.ProtocolError, urllib3_exceptions.ReadTimeoutError,
) as exc:
last_error = exc
remove_file_quiet(temp_path)
if attempt + 1 < attempts:
delay = min(30, 2 ** attempt)
if deadline is not None and time.monotonic() + delay >= deadline:
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during GHArchive retry') from exc
_wait_or_raise_scan_slot_fatal(delay)
finally:
if response is not None:
response.close()
raise ApiRequestError(f'GHArchive download failed after {attempts} attempt(s): {name}: {last_error}')
except BaseException:
if item_lock is not None:
release_file_lock(item_lock, item_lock.path)
raise
finally:
release_file_lock(global_lock, global_lock_path)
def cached_gharchive_hour(hour, cache_dir=None, request_timeout=120, retries=4, deadline=None):
path, item_lock = _acquire_cached_gharchive_hour(hour, cache_dir, request_timeout, retries, deadline)
if item_lock is not None:
release_file_lock(item_lock, item_lock.path)
return path
@contextmanager
def cached_gharchive_hour_reader(hour, cache_dir=None, request_timeout=120, retries=4, deadline=None):
path = None
item_lock = None
try:
path, item_lock = _acquire_cached_gharchive_hour(hour, cache_dir, request_timeout, retries, deadline)
yield path
finally:
if item_lock is not None:
release_file_lock(item_lock, item_lock.path)
def fetch_github_archive_repos(hours_back=6, max_repos=200, event_types=None, request_timeout=120, archive_cache_dir=None):
"""Fetch recently active public GitHub repositories from GHArchive hourly dumps."""
event_types = {str(item).strip() for item in (event_types or []) if str(item).strip()}
if not event_types:
event_types = {'PushEvent', 'CreateEvent', 'PublicEvent'}
hours_back = min(GHARCHIVE_MAX_HOURS_BACK, max(1, int(hours_back or 1)))
max_repos = max(1, int(max_repos or 1))
now = datetime.now(timezone.utc).replace(minute=0, second=0, microsecond=0)
candidates = {}
order = 0
for offset in range(1, hours_back + 1):
hour = now - timedelta(hours=offset)
url = f'https://data.gharchive.org/{hour:%Y-%m-%d-%H}.json.gz'
logger.info(f'Fetching GHArchive hour {hour.isoformat()} from {url}')
with cached_gharchive_hour_reader(hour, archive_cache_dir, request_timeout) as archive_path:
if not archive_path:
logger.info(f'GHArchive hour unavailable yet: {url}')
continue
try:
for raw_line in iter_gharchive_lines(archive_path):
try:
event = json.loads(raw_line.decode('utf-8', errors='replace'))
except (ValueError, UnicodeDecodeError):
continue
if event.get('type') not in event_types:
continue
repo = event.get('repo') if isinstance(event.get('repo'), dict) else {}
name = str(repo.get('name') or '').strip()
if not name or '/' not in name:
continue
if not re.match(r'^[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$', name):
continue
key = name.lower()
order += 1
score = github_archive_event_score(event)
existing = candidates.get(key)
if existing and score > existing['score']:
candidates[key] = {
'name': name,
'score': score,
'order': existing['order'],
'event': event,
'archive_hour': hour,
}
elif not existing:
candidate = {
'name': name,
'score': score,
'order': order,
'event': event,
'archive_hour': hour,
}
if len(candidates) < max_repos:
candidates[key] = candidate
else:
worst_key = min(
candidates,
key=lambda value: (candidates[value]['score'], -candidates[value]['order']),
)
worst = candidates[worst_key]
if (score, -order) > (worst['score'], -worst['order']):
del candidates[worst_key]
candidates[key] = candidate
except (OSError, EOFError, gzip.BadGzipFile, GHArchiveBoundsError, urllib3_exceptions.ProtocolError) as e:
remove_file_quiet(archive_path)
raise ApiRequestError(f'Unable to read cached GHArchive gzip {url}: {e}') from e
ranked = sorted(candidates.values(), key=lambda item: (-item['score'], item['order']))
selected = ranked[:max_repos]
positive = sum(1 for item in selected if item['score'] > 0)
logger.info(f'GHArchive selected {len(selected)} repos from {len(candidates)} candidates; positive_score={positive}')
return [github_archive_target_payload(item['name'], item['event'], item['score'], item['archive_hour']) for item in selected]
def fetch_github_archive_file_targets(hours_back=6, max_files=300, event_types=None, request_timeout=120, cache_dir=None, max_file_size_mb=2, token=None, max_commit_lookups=200, archive_cache_dir=None, discovery_max_artifacts=None, discovery_max_artifacts_per_page=None, discovery_max_bytes=None, discovery_max_elapsed_sec=None):
event_types = {str(item).strip() for item in (event_types or []) if str(item).strip()} or {'PushEvent'}
hours_back = min(GHARCHIVE_MAX_HOURS_BACK, max(1, int(hours_back or 1)))
max_files = max(1, int(max_files or 1))
max_bytes = int(max_file_size_mb or 2) * 1024 * 1024
budget = _PostmanDiscoveryBudget(
'GHArchive-file',
discovery_max_artifacts,
discovery_max_artifacts_per_page,
discovery_max_bytes,
discovery_max_elapsed_sec,
)
now = datetime.now(timezone.utc).replace(minute=0, second=0, microsecond=0)
targets = []
seen = set()
seen_digests = set()
headers = github_headers(token)
commit_lookups = 0
stop_cycle = False
for offset in range(1, hours_back + 1):
if stop_cycle or not budget.check_deadline():
break
hour = now - timedelta(hours=offset)
url = f'https://data.gharchive.org/{hour:%Y-%m-%d-%H}.json.gz'
logger.info(f'Fetching GHArchive file candidates hour {hour.isoformat()} from {url}')
prepared = []
pending_keys = set()
page_artifacts = 0
stop_hour = False
try:
with cached_gharchive_hour_reader(
hour, archive_cache_dir, budget.request_timeout(request_timeout), deadline=budget.deadline,
) as archive_path:
if not archive_path:
continue
for raw_line in iter_gharchive_lines(archive_path, deadline=budget.deadline):
if not budget.check_deadline('while reading GHArchive events'):
stop_cycle = True
break
try:
event = json.loads(raw_line.decode('utf-8', errors='replace'))
except (ValueError, UnicodeDecodeError):
continue
if event.get('type') not in event_types:
continue
repo = event.get('repo') if isinstance(event.get('repo'), dict) else {}
repo_name = str(repo.get('name') or '').strip()
if not repo_name or '/' not in repo_name:
continue
path_items = [(sha, path, github_raw_url(repo_name, sha, path)) for sha, path in gharchive_changed_paths(event)]
if not path_items and github_archive_event_score(event) <= 0:
continue
if not path_items:
if commit_lookups >= int(max_commit_lookups or 0):
continue
commit_lookups += 1
try:
path_items = list(fetch_github_commit_files(
event,
repo_name,
headers,
budget.request_timeout(request_timeout),
budget.deadline,
))
except _PostmanHarvestDeadlineReached as e:
budget.stop(str(e))
stop_cycle = True
break
if not path_items and commit_lookups >= int(max_commit_lookups or 0):
logger.info(f'GHArchive file fetch reached commit lookup cap: {max_commit_lookups}')
for sha, path, raw_url in path_items:
if not budget.check_deadline('while examining GHArchive paths'):
stop_cycle = True
break
if len(targets) + len(prepared) >= max_files:
stop_cycle = True
break
if not gharchive_path_interesting(path):
continue
key = f'{repo_name.lower()}@{sha}:{path.lower()}'
if key in seen or key in pending_keys:
continue
if not raw_url:
continue
admitted, scope = budget.admit(page_artifacts)
if not admitted:
stop_cycle = scope == 'cycle'
stop_hour = True
break
page_artifacts += 1
pending_keys.add(key)
try:
raw = api_request(
'GET', raw_url, timeout=budget.request_timeout(request_timeout),
use_proxy=False,
max_retries=2, retry_delay=1,
retry_statuses={408, 500, 502, 503, 504},
deadline=budget.deadline,
)
if raw.status_code == 404:
continue
raw.raise_for_status()
content = raw.content
except _PostmanHarvestDeadlineReached as e:
budget.stop(str(e))
stop_cycle = True
break
except (requests.exceptions.RequestException, ApiRequestError) as e:
logger.warning(f'Unable to fetch GHArchive raw file {repo_name}:{path}: {str(e)[:300]}')
continue
if max_bytes and len(content) > max_bytes:
continue
if not budget.account_bytes(len(content)):
stop_cycle = True
break
origin = {
'provider': 'gharchive_file',
'repo': repo_name,
'path': path,
'sha': sha,
'raw_url': raw_url,
'archive_hour': hour.isoformat(),
'event_type': event.get('type') or '',
}
kind = postman_kind_for_path(path) or 'gharchive_file'
try:
entry = _prepare_postman_cache_entry(content, kind, origin, max_file_size_mb)
except Exception as e:
if isinstance(e, PostmanCacheCapacityError):
raise
logger.warning(f'Unable to cache GHArchive file {repo_name}:{path}: {str(e)[:300]}')
continue
if entry['digest'] in seen_digests:
continue
seen_digests.add(entry['digest'])
prepared.append({
'entry': entry,
'key': key,
'kind': kind,
'origin': origin,
'repo': repo_name,
'path': path,
'sha': sha,
'raw_url': raw_url,
})
if stop_cycle or stop_hour:
break
except _PostmanHarvestDeadlineReached as e:
budget.stop(str(e))
stop_cycle = True
except (OSError, EOFError, gzip.BadGzipFile, GHArchiveBoundsError, urllib3_exceptions.ProtocolError) as e:
if 'archive_path' in locals() and archive_path:
remove_file_quiet(archive_path)
raise ApiRequestError(f'Unable to read cached GHArchive gzip {url}: {e}') from e
published, deadline_reached = _publish_postman_discovery_batch(prepared, cache_dir, budget)
for record, (cache_path, digest, size) in published:
target = gharchive_file_target(
'github_archive_file', record['kind'], cache_path, digest, record['origin'],
repo=record['repo'], path=record['path'], sha=record['sha'],
raw_url=record['raw_url'], size=size,
)
targets.append(target)
seen.add(record['key'])
if deadline_reached or stop_cycle or budget.exhausted():
stop_cycle = True
break
logger.info(f'GHArchive file fetch produced {len(targets)} targets; commit_lookups={commit_lookups}')
return targets
def gist_file_interesting(filename, file_meta):
text = ' '.join([
str(filename or '').lower(),
str((file_meta or {}).get('type') or '').lower(),
str((file_meta or {}).get('language') or '').lower(),
])
return _contains_term(text, GITHUB_GIST_FILE_TERMS)
def fetch_github_gist_targets(pages=2, per_page=100, since=None, token=None, cache_dir=None, max_file_size_mb=2, request_timeout=20, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, discovery_max_artifacts=None, discovery_max_artifacts_per_page=None, discovery_max_bytes=None, discovery_max_elapsed_sec=None):
headers = github_headers(token)
per_page = max(1, min(int(per_page or 100), 100))
max_pages = max(1, int(pages or 1))
max_bytes = int(max_file_size_mb or 2) * 1024 * 1024
budget = _PostmanDiscoveryBudget(
'GitHub Gist',
discovery_max_artifacts,
discovery_max_artifacts_per_page,
discovery_max_bytes,
discovery_max_elapsed_sec,
)
targets = []
seen_candidates = set()
seen_targets = set()
seen_digests = set()
known_pages = 0
stop_cycle = False
params_base = {'per_page': per_page}
if since:
params_base['since'] = since
for page in range(1, max_pages + 1):
if stop_cycle or not budget.check_deadline():
break
params = dict(params_base)
params['page'] = page
try:
response = api_request(
'GET', 'https://api.github.com/gists/public', headers=headers, params=params,
timeout=budget.request_timeout(request_timeout),
deadline=budget.deadline,
)
response.raise_for_status()
except _PostmanHarvestDeadlineReached as e:
budget.stop(str(e))
break
except requests.exceptions.HTTPError as e:
raise github_api_error(e.response) from e
except (requests.exceptions.RequestException, ApiRequestError) as e:
logger.warning(f'GitHub Gists request failed on page {page}: {str(e)[:300]}')
break
gists = response.json() or []
if not gists:
break
page_targets = []
prepared = []
pending_candidates = set()
page_artifacts = 0
stop_page = False
for gist in gists:
if not budget.check_deadline('while examining a Gist page'):
stop_cycle = True
break
gist_id = str(gist.get('id') or '')
files = gist.get('files') if isinstance(gist.get('files'), dict) else {}
for filename, meta in files.items():
if not budget.check_deadline('while examining Gist files'):
stop_cycle = True
break
if not isinstance(meta, dict):
continue
size = int(meta.get('size') or 0)
raw_url = meta.get('raw_url') or ''
if not raw_url or (max_bytes and size > max_bytes):
continue
if not gist_file_interesting(filename, meta):
continue
identity = f'gist:{gist_id}:{filename}:{meta.get("raw_url")}'
if identity in seen_candidates or identity in pending_candidates:
continue
admitted, scope = budget.admit(page_artifacts)
if not admitted:
stop_cycle = scope == 'cycle'
stop_page = True
break
page_artifacts += 1
pending_candidates.add(identity)
try:
raw = api_request(
'GET', raw_url, headers=headers,
use_proxy=False,
timeout=budget.request_timeout(request_timeout),
deadline=budget.deadline,
)
raw.raise_for_status()
content = raw.content
except _PostmanHarvestDeadlineReached as e:
budget.stop(str(e))
stop_cycle = True
break
except (requests.exceptions.RequestException, ApiRequestError) as e:
logger.warning(f'Unable to fetch gist raw {gist_id}/{filename}: {str(e)[:300]}')
continue
if max_bytes and len(content) > max_bytes:
continue
if not budget.account_bytes(len(content)):
stop_cycle = True
break
origin = {
'provider': 'github_gist',
'gist_id': gist_id,
'filename': filename,
'raw_url': raw_url,
'html_url': gist.get('html_url'),
'created_at': gist.get('created_at'),
'updated_at': gist.get('updated_at'),
'owner': ((gist.get('owner') or {}).get('login') if isinstance(gist.get('owner'), dict) else ''),
}
kind = postman_kind_for_path(filename) or 'gist_file'
try:
entry = _prepare_postman_cache_entry(content, kind, origin, max_file_size_mb)
except Exception as e:
if isinstance(e, PostmanCacheCapacityError):
raise
logger.warning(f'Unable to cache gist {gist_id}/{filename}: {str(e)[:300]}')
continue
if entry['digest'] in seen_digests:
continue
seen_digests.add(entry['digest'])
prepared.append({
'entry': entry,
'candidate_identity': identity,
'kind': kind,
'origin': origin,
'gist_id': gist_id,
'filename': filename,
'raw_url': raw_url,
'html_url': gist.get('html_url'),
})
if stop_cycle or stop_page:
break
published, deadline_reached = _publish_postman_discovery_batch(prepared, cache_dir, budget)
for record, (cache_path, digest, cached_size) in published:
target = postman_target_from_cached_artifact(
'github_gist', record['kind'], cache_path, digest, record['origin'],
gist_id=record['gist_id'], filename=record['filename'],
raw_url=record['raw_url'], html_url=record['html_url'], size=cached_size,
)
target_identity = postman_target_identity(target)
seen_candidates.add(record['candidate_identity'])
if target_identity in seen_targets:
continue
targets.append(target)
page_targets.append(target)
seen_targets.add(target_identity)
logger.info(f'GitHub Gists page {page}: gists={len(gists)}, targets={len(page_targets)}')
if stop_on_seen_pages and page >= max(1, int(min_pages_before_stop or 1)):
if page_targets and page_is_known(
page_targets, known_targets, normalize_target, known_target_lookup,
):
known_pages += 1
if known_pages >= max(1, int(seen_page_threshold or 1)):
break
else:
known_pages = 0
if deadline_reached or stop_cycle or budget.exhausted():
stop_cycle = True
break
return targets
def gitlab_project_to_target(project):
return {
'url': project.get('http_url_to_repo', ''),
'name': project.get('path_with_namespace', ''),
'created_at': project.get('created_at', ''),
'updated_at': project.get('updated_at') or project.get('last_activity_at', ''),
'last_activity_at': project.get('last_activity_at', ''),
}
def gitlab_rate_limit_reset(response):
if not response:
return None
reset = response.headers.get('RateLimit-Reset') or response.headers.get('X-RateLimit-Reset')
if not reset:
retry_after = response.headers.get('Retry-After')
if retry_after:
try:
return (datetime.now(timezone.utc) + timedelta(seconds=int(retry_after))).isoformat(timespec='seconds')
except (TypeError, ValueError):
return None
return None
try:
if str(reset).isdigit():
return datetime.fromtimestamp(int(reset), timezone.utc).isoformat(timespec='seconds')
return str(reset)
except (TypeError, ValueError):
return None
def fetch_gitlab_repo_items(query, pages, per_page=100, token=None, sort_by="last_activity_at", sort_order="desc", visibility="public", last_activity_after=None, raise_rate_limit=False, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, request_attempts=1, retry_delay=0):
"""Fetch GitLab repositories with metadata for filtering."""
repos = []
headers = {}
if token:
headers['Authorization'] = f'Bearer {token}'
logger.info(f"Fetching GitLab repositories for: '{query if query else 'all repositories'}' sorted by {sort_by} ({sort_order})...")
page = 1
seen_pages = 0
successful_pages = 0
request_attempts = max(1, int(request_attempts or 1))
retry_delay = max(0, int(retry_delay or 0))
request_budget = 30 * request_attempts + retry_delay * (request_attempts - 1)
while page <= pages:
url = "https://gitlab.com/api/v4/projects"
params = {
'visibility': visibility,
'per_page': per_page,
'page': page,
'order_by': sort_by,
'sort': sort_order,
}
if query:
params['search'] = query
if last_activity_after:
params['last_activity_after'] = last_activity_after
try:
response = api_request(
'GET', url, headers=headers, params=params, timeout=30,
max_retries=request_attempts, retry_delay=retry_delay,
deadline=time.monotonic() + request_budget,
)
response.raise_for_status()
# Handle rate limits
if response.status_code == 429:
retry_after = int(response.headers.get('Retry-After', 60))
logger.warning(f"Rate limit exceeded. Waiting {retry_after} seconds...")
time.sleep(retry_after)
continue
data = response.json()
if not isinstance(data, list):
raise ValueError('invalid GitLab project search payload')
successful_pages += 1
if not data:
logger.info(f"Page {page} returned no results. Stopping.")
break
page_repos = [gitlab_project_to_target(project) for project in data if project.get('http_url_to_repo')]
repos.extend(page_repos)
logger.info(f"Page {page}: Fetched {len(data)} repositories")
if stop_on_seen_pages and page >= max(1, min_pages_before_stop):
if page_is_known(
[item.get('url') for item in page_repos], known_targets,
normalize_target, known_target_lookup,
):
seen_pages += 1
logger.info(f"Page {page}: all GitLab repositories are already queued/checked ({seen_pages}/{seen_page_threshold})")
if seen_pages >= max(1, seen_page_threshold):
logger.info(f"Stopping GitLab pagination early after {seen_pages} all-known page(s)")
break
else:
seen_pages = 0
page += 1
except requests.exceptions.HTTPError as e:
if e.response.status_code == 401 and not token:
logger.warning("GitLab API authentication would improve results. Consider adding a GitLab token.")
raise gitlab_api_error(e.response) from e
else:
api_error = gitlab_api_error(e.response)
if raise_rate_limit:
raise api_error from e
logger.error(str(api_error))
raise api_error from e
except Exception as e:
if isinstance(e, ApiRequestError):
raise GitLabDiscoveryTransportError(str(e)) from e
logger.error(f"Error fetching page {page}: {str(e)}")
raise ApiRequestError(f'GitLab discovery payload failed: {e}') from e
return repos
def fetch_gitlab_repos(query, pages, per_page=100, token=None, sort_by="last_activity_at", sort_order="desc", visibility="public", raise_rate_limit=False, **kwargs):
"""Fetch GitLab repository clone URLs with pagination and authentication."""
return [
item['url'] for item in fetch_gitlab_repo_items(
query, pages, per_page, token, sort_by, sort_order, visibility, raise_rate_limit=raise_rate_limit, **kwargs
)
if item.get('url')
]
def github_recent_query(query, since):
since_str = since.strftime("%Y-%m-%d")
query = (query or '').strip()
if query and ' in:' not in f' {query.lower()} ':
query = f'{query} in:name,description,readme'
return f"{query} updated:>={since_str}" if query else f"updated:>={since_str}"
def fetch_recent_github_repos(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, **kwargs):
"""Fetch GitHub repositories updated since a specific timestamp"""
return fetch_github_repos(github_recent_query(query, since), pages=pages, per_page=per_page, token=token, raise_rate_limit=raise_rate_limit, **kwargs)
def fetch_recent_github_repo_items(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, **kwargs):
"""Fetch recent GitHub repositories with metadata."""
return fetch_github_repo_items(github_recent_query(query, since), pages=pages, per_page=per_page, token=token, raise_rate_limit=raise_rate_limit, **kwargs)
def fetch_recent_gitlab_repos(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, visibility="public", **kwargs):
"""Fetch GitLab repositories updated since a specific timestamp"""
since_str = since.strftime("%Y-%m-%dT%H:%M:%SZ")
return [
item['url'] for item in fetch_gitlab_repo_items(
query, pages=pages, per_page=per_page, token=token,
sort_by="last_activity_at", sort_order="desc", visibility=visibility,
last_activity_after=since_str,
raise_rate_limit=raise_rate_limit,
**kwargs,
)
if item.get('url')
]
def fetch_recent_gitlab_repo_items(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, visibility="public", **kwargs):
"""Fetch recent GitLab repositories with metadata."""
since_str = since.strftime("%Y-%m-%dT%H:%M:%SZ")
return fetch_gitlab_repo_items(
query, pages=pages, per_page=per_page, token=token,
sort_by="last_activity_at", sort_order="desc", visibility=visibility,
last_activity_after=since_str,
raise_rate_limit=raise_rate_limit,
**kwargs,
)
DOCKERHUB_SEARCH_MAX_PAGES = 30
DOCKERHUB_TAG_RATE_LIMIT_MIN_BACKOFF_SEC = 60
DOCKERHUB_TAG_RATE_LIMIT_MAX_BACKOFF_SEC = 3600
DOCKER_REGISTRY_MANIFEST_MAX_BYTES = 8 * 1024 * 1024
DOCKER_REGISTRY_MAX_DESCRIPTORS = 1000
DOCKER_REGISTRY_MAX_LAYERS = 2048
DOCKER_REGISTRY_TOKEN_MAX_BYTES = 1024 * 1024
DOCKER_CONFIG_MEDIA_TYPES = frozenset((
'application/vnd.oci.image.config.v1+json',
'application/vnd.docker.container.image.v1+json',
))
DOCKER_LAYER_MEDIA_TYPES = frozenset((
'application/vnd.oci.image.layer.v1.tar',
'application/vnd.oci.image.layer.v1.tar+gzip',
'application/vnd.oci.image.layer.v1.tar+zstd',
'application/vnd.docker.image.rootfs.diff.tar',
'application/vnd.docker.image.rootfs.diff.tar.gzip',
))
DOCKER_LAYER_GZIP_MEDIA_TYPES = frozenset((
'application/vnd.oci.image.layer.v1.tar+gzip',
'application/vnd.docker.image.rootfs.diff.tar.gzip',
))
DOCKER_CONFIG_HISTORY_MAX_ENTRIES = DOCKER_REGISTRY_MAX_LAYERS * 2
DOCKER_CONFIG_HISTORY_COMMAND_MAX_CHARS = 64 * 1024
def docker_history_payload_class(created_by):
if not isinstance(created_by, str) or not created_by:
return 'unknown'
if len(created_by) > DOCKER_CONFIG_HISTORY_COMMAND_MAX_CHARS:
return 'unknown'
command = created_by.casefold()
if re.search(r'(^|#\(nop\)\s+)(copy|add)\s', command.strip()):
return 'copy_add'
if not re.search(r'(^|[\s#])run(\s|$)|/bin/(ba)?sh\s+-c', command):
return 'unknown'
if any(token in command for token in (
'/models/', '/model/', 'model_weights', 'checkpoint.', '.safetensors',
'.gguf', '.onnx', '.pt ', '.pth ', 'huggingface-cli download',
)):
return 'bulk_data'
if re.search(
r'\b(apt-get|apt|apk|yum|dnf|microdnf|pip|pip3|poetry|npm|pnpm|yarn|'
r'bundle|gem|cargo)\s+(install|add|sync)\b',
command,
):
return 'package_run'
if any(token in command for token in (
'/app', '/srv', '/workspace', '/opt/app', 'config', '.env',
'requirements.txt', 'package.json', 'pyproject.toml',
)):
return 'app_config_run'
return 'other_run'
def docker_config_payload_classes(config, layer_count):
try:
layer_count = int(layer_count)
except (TypeError, ValueError):
layer_count = -1
fallback = ['unknown'] * max(0, layer_count)
if (
not isinstance(config, dict)
or layer_count < 0
or layer_count > DOCKER_REGISTRY_MAX_LAYERS
):
return fallback
history = config.get('history')
if not isinstance(history, list) or len(history) > DOCKER_CONFIG_HISTORY_MAX_ENTRIES:
return fallback
rootfs = config.get('rootfs')
if rootfs is not None:
if not isinstance(rootfs, dict):
return fallback
diff_ids = rootfs.get('diff_ids')
if not isinstance(diff_ids, list) or len(diff_ids) != layer_count:
return fallback
commands = []
for entry in history:
if not isinstance(entry, dict):
return fallback
empty_layer = entry.get('empty_layer', False)
if not isinstance(empty_layer, bool):
return fallback
if empty_layer:
continue
created_by = entry.get('created_by')
if not isinstance(created_by, str):
return fallback
commands.append(created_by)
if len(commands) != layer_count:
return fallback
classes = [docker_history_payload_class(command) for command in commands]
if any(payload_class not in DOCKER_ADAPTIVE_PAYLOAD_CLASSES for payload_class in classes):
return fallback
return classes
class DockerRegistryResolutionError(ValueError):
pass
class DockerResolverLeaseLostError(RuntimeError):
pass
class DockerRemoteAccessError(DockerRegistryResolutionError):
def __init__(self, message, status='unknown', retry_at=None, remote_attempted=True):
super().__init__(message)
self.status = str(status or 'unknown')
self.retry_at = retry_at
self.remote_attempted = bool(remote_attempted)
class DockerContentTransferError(RuntimeError):
def __init__(self, error_code, message, retryable=True):
super().__init__(message)
self.error_code = str(error_code or 'transfer_failed')
self.retryable = bool(retryable)
self.source_failure = False
self.transfer_bytes = 0
self.duration_ms = 0
class DockerLayerInfrastructureError(DockerContentTransferError):
def __init__(
self, error_code, message, category='remote_transient', auth_related=False,
):
super().__init__(error_code, message, retryable=True)
self.category = str(category or 'remote_transient')
self.auth_related = bool(auth_related)
self.source_failure = True
class DockerContentScanError(RuntimeError):
def __init__(self, error_code, message, retryable=False):
super().__init__(message)
self.error_code = str(error_code or 'invalid_content')
self.retryable = bool(retryable)
@dataclass(frozen=True, repr=False)
class DockerBlobDownloadOutcome:
path: str
verified_bytes: int
transfer_bytes: int
duration_ms: int
bearer_auth: DockerRegistryAuth
@dataclass(frozen=True)
class DockerTagResolutionOutcome:
tags: tuple
status: str
remote_attempted: bool
retry_at: str = None
error: str = ''
selection_records: tuple = ()
candidate_records: tuple = ()
selector_version: str = ''
selector_hash: str = ''
candidate_distinct_graph_count: int = 0
fresh_graph_evidence: bool = False
cache_bypassed: bool = False
@property
def selections(self):
return self.selection_records
@property
def selector_sha256(self):
return self.selector_hash
DOCKER_TAG_CONCLUSIVE_STATUSES = frozenset({'ok', 'empty', 'unsupported', 'not_found'})
def docker_tag_resolution_is_conclusive(status):
return str(status or '') in DOCKER_TAG_CONCLUSIVE_STATUSES
def docker_images_per_repository_limit(value):
return validate_docker_images_per_repository(value)
def dockerhub_search_page_window(pages):
try:
requested = max(1, int(pages or 1))
except (TypeError, ValueError):
requested = 1
return requested, min(requested, DOCKERHUB_SEARCH_MAX_PAGES)
def fetch_dockerhub_search_page(
query, page, per_page=100, sort_by='updated_at', sort_order='desc',
request_timeout=15,
):
"""Fetch and validate one bounded Docker Hub repository-search page."""
try:
page = int(page)
per_page = max(1, int(per_page or 1))
except (TypeError, ValueError, OverflowError):
raise DockerHubDiscoveryTransportError(
'Docker Hub search page request is invalid',
category='invalid_payload', remote_attempted=False, retryable=False,
) from None
if page < 1 or page > DOCKERHUB_SEARCH_MAX_PAGES:
raise DockerHubDiscoveryTransportError(
'Docker Hub search page request is invalid',
category='invalid_payload', remote_attempted=False, retryable=False,
)
params = {
'query': query,
'page': page,
'page_size': per_page,
'sort': sort_by,
'order': sort_order,
}
started = time.perf_counter()
try:
response = dockerhub_search_response(
'https://hub.docker.com/v2/search/repositories', params,
request_timeout=request_timeout,
)
except DockerRemoteAccessError as error:
category = {
'rate_limited': 'rate_limit',
'auth_failed': 'auth_unavailable',
'remote_transient': 'remote_transient',
}.get(error.status, 'page_unavailable')
if error.retry_at and not error.remote_attempted:
category = 'provider_cooldown'
raise DockerHubDiscoveryTransportError(
f'Docker Hub search page {page} failed after bounded attempts',
category=category, retry_at=error.retry_at,
remote_attempted=error.remote_attempted,
) from None
except ApiRequestError:
raise DockerHubDiscoveryTransportError(
f'Docker Hub search page {page} failed after bounded attempts',
category='network', remote_attempted=True,
) from None
except Exception:
raise DockerHubDiscoveryTransportError(
f'Docker Hub search page {page} failed after bounded attempts',
category='page_unavailable', remote_attempted=True,
) from None
try:
response.raise_for_status()
except Exception:
raise DockerHubDiscoveryTransportError(
f'Docker Hub search page {page} failed after bounded attempts',
category='page_unavailable', remote_attempted=True,
) from None
try:
data = response.json()
except Exception:
raise DockerHubDiscoveryTransportError(
f'Docker Hub search page {page} returned an invalid payload',
category='invalid_payload', remote_attempted=True, retryable=False,
) from None
if not isinstance(data, dict) or not isinstance(data.get('results'), list):
raise DockerHubDiscoveryTransportError(
f'Docker Hub search page {page} returned an invalid payload',
category='invalid_payload', remote_attempted=True, retryable=False,
)
raw_count = data.get('count')
try:
total_count = int(raw_count)
except (TypeError, ValueError, OverflowError):
raise DockerHubDiscoveryTransportError(
f'Docker Hub search page {page} returned an invalid result count',
category='invalid_payload', remote_attempted=True, retryable=False,
) from None
if (
isinstance(raw_count, bool)
or (isinstance(raw_count, float) and not raw_count.is_integer())
or total_count < 0
):
raise DockerHubDiscoveryTransportError(
f'Docker Hub search page {page} returned an invalid result count',
category='invalid_payload', remote_attempted=True, retryable=False,
)
result_count = len(data['results'])
absolute_start = (page - 1) * per_page
if (
result_count > per_page
or total_count < absolute_start + result_count
or (result_count == 0 and total_count > absolute_start)
):
raise DockerHubDiscoveryTransportError(
f'Docker Hub search page {page} returned incoherent pagination evidence',
category='invalid_payload', remote_attempted=True, retryable=False,
)
repositories = []
seen_repo_names = set()
for repository in data['results']:
if not isinstance(repository, dict):
raise DockerHubDiscoveryTransportError(
f'Docker Hub search page {page} returned an invalid payload',
category='invalid_payload', remote_attempted=True, retryable=False,
)
repo_name = repository.get('repo_name')
if not isinstance(repo_name, str) or not repo_name.strip():
raise DockerHubDiscoveryTransportError(
f'Docker Hub search page {page} returned an invalid payload',
category='invalid_payload', remote_attempted=True, retryable=False,
)
repo_name = repo_name.strip()
if repo_name in seen_repo_names:
continue
seen_repo_names.add(repo_name)
safe_repository = {'repo_name': repo_name}
for field in ('last_updated', 'last_modified'):
if field in repository:
safe_repository[field] = repository[field]
repositories.append(safe_repository)
return {
'page': page,
'repositories': repositories,
'total_count': total_count,
'elapsed': time.perf_counter() - started,
}
def fetch_dockerhub_images(query, pages, per_page=100, sort_by="updated_at", sort_order="desc", fetch_workers=8, request_timeout=15, resolve_tags=True, tag_fetch_workers=None, tag_retry_count=2, tag_retry_delay=5, platform_filter_enabled=False, platform_os='linux', platform_arch='amd64', platform_candidate_tags=20, images_per_repository=1):
"""Fetch Docker Hub images with pagination and sorting"""
images = []
per_page = max(1, int(per_page or 1))
requested_pages, pages = dockerhub_search_page_window(pages)
if requested_pages > pages:
logger.info(
f'Docker Hub search is limited to {pages} accessible page(s); '
f'capping requested pages from {requested_pages}'
)
logger.info(f"Fetching Docker Hub images for: '{query}' sorted by {sort_by} ({sort_order})...")
def fetch_page(page):
started = time.perf_counter()
try:
return fetch_dockerhub_search_page(
query, page, per_page=per_page, sort_by=sort_by,
sort_order=sort_order, request_timeout=request_timeout,
), None
except DockerHubDiscoveryTransportError as error:
return {
'page': page,
'repositories': [],
'total_count': 0,
'elapsed': time.perf_counter() - started,
}, error
first_page, first_error = fetch_page(1)
if first_error is not None:
raise first_error
total_count = first_page['total_count']
expected_pages = min(
pages,
max(1, (total_count + per_page - 1) // per_page),
)
page_results = [(first_page, None)]
if expected_pages > 1:
max_workers = max(1, min(fetch_workers, expected_pages - 1))
with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
futures = [
executor.submit(fetch_page, page)
for page in range(2, expected_pages + 1)
]
page_results.extend(
future.result() for future in concurrent.futures.as_completed(futures)
)
repo_names = []
failed_pages = []
for page_result, error in sorted(
page_results, key=lambda item: item[0]['page'],
):
page = page_result['page']
elapsed = page_result['elapsed']
if error:
logger.error(
f"Page {page}: Docker Hub fetch failed after bounded attempts "
f"({elapsed:.1f}s)"
)
failed_pages.append(page)
continue
repositories = page_result['repositories']
if not repositories:
logger.info(
f"Page {page}: no results after {elapsed:.1f}s "
f"(total matches: {page_result['total_count']})"
)
continue
repo_names.extend(repository['repo_name'] for repository in repositories)
logger.info(f"Page {page}: fetched {len(repositories)} images in {elapsed:.1f}s")
if failed_pages:
raise DockerHubDiscoveryTransportError(
f'Docker Hub search pagination incomplete: {len(failed_pages)} '
f'of {expected_pages} expected page(s) failed'
)
if not resolve_tags:
return repo_names
tag_fetch_workers = tag_fetch_workers if tag_fetch_workers is not None else min(4, fetch_workers)
logger.info(f"Resolving Docker tags for {len(repo_names)} repositories with {tag_fetch_workers} worker(s)...")
max_workers = max(1, min(tag_fetch_workers, len(repo_names) or 1))
with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
futures = {
executor.submit(
fetch_dockerhub_tags, repo_name, None,
docker_images_per_repository_limit(images_per_repository),
tag_retry_count, tag_retry_delay,
platform_filter_enabled, platform_os, platform_arch, platform_candidate_tags,
True,
): repo_name
for repo_name in repo_names
}
for future in concurrent.futures.as_completed(futures):
repo_name = futures[future]
tags, status = future.result()
if tags:
images.extend(tags)
if status != 'ok':
images.append(repo_name)
elif not docker_tag_resolution_is_conclusive(status):
images.append(repo_name)
logger.info(f"Deferring Docker tag resolution for {repo_name}: metadata lookup unavailable")
else:
logger.info(f"Skipping {repo_name}: no tags found")
logger.info(f"Resolved {len(images)} tagged Docker images from {len(repo_names)} repositories")
return images
def parse_dockerhub_datetime(value):
if not value:
return None
value = value.rstrip('Z')
for date_format in ("%Y-%m-%dT%H:%M:%S.%f", "%Y-%m-%dT%H:%M:%S"):
try:
return datetime.strptime(value, date_format)
except ValueError:
continue
return None
def fetch_dockerhub_last_updated(repo_name):
if '/' in repo_name:
namespace, name = repo_name.split('/', 1)
else:
namespace, name = 'library', repo_name
url = f"https://hub.docker.com/v2/repositories/{namespace}/{name}/"
try:
response = api_request('GET', url, headers={'User-Agent': 'GitSecretsScanner/2.0'}, timeout=8)
response.raise_for_status()
data = response.json()
return parse_dockerhub_datetime(data.get('last_updated') or data.get('last_modified'))
except Exception as e:
if isinstance(e, ApiRequestError):
raise
logger.warning(f"Unable to fetch Docker Hub metadata for {repo_name}: {str(e)}")
return None
DOCKERHUB_TAG_CACHE_SCHEMA = """
CREATE TABLE IF NOT EXISTS dockerhub_tag_cache (
cache_key TEXT PRIMARY KEY,
repo_key TEXT NOT NULL,
status TEXT NOT NULL,
tags_json TEXT,
checked_at REAL NOT NULL,
expires_at REAL NOT NULL,
since_at REAL,
message TEXT
);
CREATE TABLE IF NOT EXISTS dockerhub_tag_cache_meta (
key TEXT PRIMARY KEY,
value TEXT,
expires_at REAL
);
"""
_dockerhub_tag_cache_init_lock = threading.Lock()
_dockerhub_tag_cache_write_lock = threading.Lock()
_dockerhub_tag_cache_initialized = set()
def dockerhub_repo_key(repo_name):
repo_name = str(repo_name or '').strip().lower()
if ':' in repo_name:
repo_name = repo_name.split(':', 1)[0]
if '/' not in repo_name:
repo_name = 'library/' + repo_name
return repo_name.strip('/')
def dockerhub_tag_cache_path():
return getattr(scan_config, 'dockerhub_tag_cache_path', '') or ''
def dockerhub_tag_cache_limits():
return {
'rows': max(1, int(getattr(scan_config, 'dockerhub_tag_cache_max_rows', 50000))),
'age': max(60, int(getattr(scan_config, 'dockerhub_tag_cache_max_age_sec', 7 * 86400))),
'bytes': max(4096, int(getattr(scan_config, 'dockerhub_tag_cache_max_bytes', 256 * 1024 * 1024))),
'min_free': max(0, int(getattr(scan_config, 'dockerhub_tag_cache_min_free_bytes', 512 * 1024 * 1024))),
}
def dockerhub_tag_cache_disk_bytes(path):
return sum(
os.path.getsize(candidate) for candidate in (path, path + '-wal', path + '-shm')
if os.path.isfile(candidate)
)
def _dockerhub_since_epoch(since):
if since is None:
return None
try:
if isinstance(since, datetime):
value = since
else:
value = datetime.fromisoformat(str(since).replace('Z', '+00:00'))
if value.tzinfo is None:
value = value.replace(tzinfo=timezone.utc)
return float(value.timestamp())
except (TypeError, ValueError, OverflowError):
return None
def maintain_dockerhub_tag_cache(conn, path):
limits = dockerhub_tag_cache_limits()
now = time.time()
conn.execute('DELETE FROM dockerhub_tag_cache WHERE expires_at <= ? OR checked_at < ?', (now, now - limits['age']))
conn.execute('DELETE FROM dockerhub_tag_cache_meta WHERE expires_at IS NOT NULL AND expires_at <= ?', (now,))
conn.execute(
'''DELETE FROM dockerhub_tag_cache WHERE cache_key NOT IN (
SELECT cache_key FROM dockerhub_tag_cache ORDER BY checked_at DESC LIMIT ?
)''',
(limits['rows'],),
)
conn.commit()
if dockerhub_tag_cache_disk_bytes(path) > limits['bytes']:
# The cache is disposable. Clearing it under SQLite's full auto-vacuum
# is safer than allowing stale pages to consume an unbounded volume.
conn.execute('DELETE FROM dockerhub_tag_cache')
conn.execute('DELETE FROM dockerhub_tag_cache_meta')
conn.commit()
try:
conn.execute('PRAGMA incremental_vacuum')
except sqlite3.DatabaseError:
pass
disk_bytes = dockerhub_tag_cache_disk_bytes(path)
parent = os.path.dirname(path) or os.getcwd()
if int(shutil.disk_usage(parent).free) - disk_bytes >= limits['min_free']:
try:
conn.execute('PRAGMA wal_checkpoint(TRUNCATE)')
conn.execute('PRAGMA journal_mode=DELETE')
conn.execute('VACUUM')
conn.execute('PRAGMA journal_mode=WAL')
except sqlite3.DatabaseError:
pass
return dockerhub_tag_cache_disk_bytes(path) <= limits['bytes']
def connect_dockerhub_tag_cache():
path = dockerhub_tag_cache_path()
if not path:
return None
parent = os.path.dirname(path)
if parent:
os.makedirs(parent, exist_ok=True)
with _dockerhub_tag_cache_init_lock:
if path not in _dockerhub_tag_cache_initialized:
new_database = not os.path.exists(path)
conn = sqlite3.connect(path, timeout=10)
try:
conn.execute('PRAGMA busy_timeout=10000')
if new_database:
conn.execute('PRAGMA auto_vacuum=FULL')
conn.execute('PRAGMA journal_mode=WAL')
conn.executescript(DOCKERHUB_TAG_CACHE_SCHEMA)
columns = {row[1] for row in conn.execute('PRAGMA table_info(dockerhub_tag_cache)').fetchall()}
if 'since_at' not in columns:
conn.execute('ALTER TABLE dockerhub_tag_cache ADD COLUMN since_at REAL')
conn.commit()
if not maintain_dockerhub_tag_cache(conn, path):
return None
finally:
conn.close()
_dockerhub_tag_cache_initialized.add(path)
conn = sqlite3.connect(path, timeout=10)
conn.execute('PRAGMA busy_timeout=10000')
return conn
def dockerhub_cache_key(repo_name, since, limit, platform_variant=''):
return f"{dockerhub_repo_key(repo_name)}|{int(limit or 1)}|{platform_variant}"
def get_dockerhub_tag_cache(repo_name, since, limit, platform_variant='', return_status=False):
conn = connect_dockerhub_tag_cache()
if not conn:
return None
try:
row = conn.execute(
'SELECT status, tags_json, expires_at, since_at FROM dockerhub_tag_cache WHERE cache_key = ?',
(dockerhub_cache_key(repo_name, since, limit, platform_variant),),
).fetchone()
now = time.time()
if not row or float(row[2] or 0) <= now:
return None
status, tags_json, _, stored_since = row
requested_since = _dockerhub_since_epoch(since)
if requested_since is not None and stored_since is None:
return None
if requested_since is not None and float(stored_since or 0) > requested_since:
return None
if status == 'ok':
records = json.loads(tags_json or '[]')
targets = []
for record in records:
if isinstance(record, str):
return None
if not isinstance(record, dict) or not record.get('name'):
continue
updated_at = record.get('updated_at')
if requested_since is not None and updated_at is not None and float(updated_at) < requested_since:
continue
target = record.get('target')
if not target:
return None
try:
targets.append(parse_docker_target(target)['target'])
except (TypeError, ValueError):
return None
if len(targets) >= max(1, int(limit or 1)):
break
if not targets:
return None
return (targets, status) if return_status else targets
return ([], status) if return_status else []
except Exception:
return None
finally:
conn.close()
def put_dockerhub_tag_cache(
repo_name, since, limit, status, tags=None, ttl=None, message='', platform_variant='', tag_records=None,
):
with _dockerhub_tag_cache_write_lock:
conn = connect_dockerhub_tag_cache()
if not conn:
return
path = dockerhub_tag_cache_path()
try:
now = time.time()
ttl = int(ttl if ttl is not None else getattr(scan_config, 'dockerhub_tag_cache_ttl_sec', 21600))
records = []
offered_records = list(tag_records or [])
for index, tag in enumerate(tags or []):
text = str(tag)
record = offered_records[index] if index < len(offered_records) and isinstance(offered_records[index], dict) else {}
name = str(record.get('name') or '')
if not name:
name = text.rsplit(':', 1)[1] if ':' in text and '@' not in text and not text.startswith('{') else text
target = record.get('target') or text
try:
target = parse_docker_target(target)['target']
except (TypeError, ValueError):
continue
records.append({'name': name, 'target': target, 'updated_at': record.get('updated_at')})
if status == 'ok' and not records:
return False
payload = json.dumps(records, ensure_ascii=True, sort_keys=True, separators=(',', ':'))
limits = dockerhub_tag_cache_limits()
projected_bytes = len(payload.encode('utf-8')) + len(str(message or '').encode('utf-8')) + 1024
parent = os.path.dirname(path) or os.getcwd()
if (
not maintain_dockerhub_tag_cache(conn, path)
or dockerhub_tag_cache_disk_bytes(path) + projected_bytes > limits['bytes']
or int(shutil.disk_usage(parent).free) - projected_bytes < limits['min_free']
):
return False
conn.execute(
'''INSERT OR REPLACE INTO dockerhub_tag_cache(
cache_key, repo_key, status, tags_json, checked_at, expires_at, since_at, message
) VALUES (?, ?, ?, ?, ?, ?, ?, ?)''',
(
dockerhub_cache_key(repo_name, since, limit, platform_variant),
dockerhub_repo_key(repo_name),
status,
payload,
now,
now + max(1, ttl),
_dockerhub_since_epoch(since),
str(message or '')[:500],
),
)
conn.commit()
maintain_dockerhub_tag_cache(conn, path)
return True
except Exception:
return False
finally:
conn.close()
def dockerhub_tag_rate_limit_state():
conn = connect_dockerhub_tag_cache()
if not conn:
return {'active': False, 'retry_at': None}
try:
row = conn.execute("SELECT expires_at FROM dockerhub_tag_cache_meta WHERE key = 'rate_limited'").fetchone()
expires_at = float(row[0] or 0) if row else 0
active = expires_at > time.time()
return {
'active': active,
'retry_at': (
datetime.fromtimestamp(expires_at, timezone.utc).isoformat(timespec='seconds')
if active else None
),
}
except Exception:
return {'active': False, 'retry_at': None}
finally:
conn.close()
def dockerhub_tags_rate_limited():
return dockerhub_tag_rate_limit_state()['active']
def dockerhub_retry_after_seconds(response=None):
fallback = int(getattr(scan_config, 'dockerhub_tag_rate_limit_cache_ttl_sec', 1800) or 1800)
seconds = None
headers = (getattr(response, 'headers', None) or {}) if response is not None else {}
retry_after = headers.get('Retry-After')
if retry_after:
try:
seconds = int(retry_after)
except (TypeError, ValueError):
try:
retry_at = parsedate_to_datetime(str(retry_after))
if retry_at.tzinfo is None:
retry_at = retry_at.replace(tzinfo=timezone.utc)
seconds = math.ceil((retry_at - datetime.now(timezone.utc)).total_seconds())
except (IndexError, TypeError, ValueError, OverflowError):
seconds = None
seconds = fallback if seconds is None else seconds
return max(
DOCKERHUB_TAG_RATE_LIMIT_MIN_BACKOFF_SEC,
min(DOCKERHUB_TAG_RATE_LIMIT_MAX_BACKOFF_SEC, int(seconds)),
)
def put_dockerhub_tags_rate_limit(response=None, retry_seconds=None, return_retry_at=False):
ttl = dockerhub_retry_after_seconds(response) if retry_seconds is None else max(
DOCKERHUB_TAG_RATE_LIMIT_MIN_BACKOFF_SEC,
min(DOCKERHUB_TAG_RATE_LIMIT_MAX_BACKOFF_SEC, int(retry_seconds)),
)
with _dockerhub_tag_cache_write_lock:
conn = connect_dockerhub_tag_cache()
if not conn:
return
try:
path = dockerhub_tag_cache_path()
limits = dockerhub_tag_cache_limits()
parent = os.path.dirname(path) or os.getcwd()
if (
dockerhub_tag_cache_disk_bytes(path) + 8192 > limits['bytes']
or int(shutil.disk_usage(parent).free) - 8192 < limits['min_free']
):
return False
expires_at = math.ceil(time.time() + ttl)
conn.execute(
'''INSERT INTO dockerhub_tag_cache_meta(key, value, expires_at)
VALUES ('rate_limited', '1', ?)
ON CONFLICT(key) DO UPDATE SET value = '1',
expires_at = MAX(dockerhub_tag_cache_meta.expires_at, excluded.expires_at)''',
(expires_at,),
)
conn.commit()
stored = conn.execute(
"SELECT expires_at FROM dockerhub_tag_cache_meta WHERE key = 'rate_limited'"
).fetchone()
maintain_dockerhub_tag_cache(conn, path)
if return_retry_at:
return datetime.fromtimestamp(float(stored[0]), timezone.utc).isoformat(timespec='seconds')
return True
except Exception:
return False
finally:
conn.close()
def put_dockerhub_exhausted_rate_limit(endpoint, response=None):
if endpoint == 'hub_search':
if not docker_token_manager.has_accounts():
if (
docker_token_manager.uses_explicit_pool()
or int(getattr(response, 'status_code', 0) or 0) != 429
):
return None
retry_seconds = dockerhub_retry_after_seconds(response)
else:
if not docker_token_manager.rate_limit_contributes_to_exhaustion(endpoint):
return None
retry_seconds = docker_token_manager.seconds_until_available(endpoint)
return datetime.fromtimestamp(
time.time() + retry_seconds, timezone.utc,
).isoformat(timespec='seconds')
if not docker_token_manager.has_accounts():
if docker_token_manager.uses_explicit_pool():
return None
if int(getattr(response, 'status_code', 0) or 0) != 429:
return None
return put_dockerhub_tags_rate_limit(response, return_retry_at=True)
if not docker_token_manager.rate_limit_contributes_to_exhaustion(endpoint):
return None
return put_dockerhub_tags_rate_limit(
response,
retry_seconds=docker_token_manager.seconds_until_available(endpoint),
return_retry_at=True,
)
def docker_tag_platform_support(tag, platform_os='linux', platform_arch='amd64'):
images = tag.get('images') if isinstance(tag, dict) else None
if not isinstance(images, list) or not images:
return None
platforms = []
for image in images:
if not isinstance(image, dict):
return None
image_os = str(image.get('os') or '').strip().lower()
image_arch = str(image.get('architecture') or '').strip().lower()
if not image_os or not image_arch or image_os == 'unknown' or image_arch == 'unknown':
return None
platforms.append((image_os, image_arch))
wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower())
return wanted in platforms
def _bounded_docker_registry_json(
response, label, max_bytes=DOCKER_REGISTRY_MANIFEST_MAX_BYTES, *,
deadline=None, return_raw=False,
):
max_bytes = max(1, int(max_bytes))
content_length = (getattr(response, 'headers', None) or {}).get('Content-Length')
if content_length is not None:
try:
content_length = int(content_length)
except (TypeError, ValueError) as exc:
raise DockerRegistryResolutionError(f'{label} has an invalid Content-Length') from exc
if content_length < 0:
raise DockerRegistryResolutionError(f'{label} has an invalid Content-Length')
if content_length > max_bytes:
raise DockerRegistryResolutionError(f'{label} exceeds the response size limit')
iterator = None
iter_content = getattr(response, 'iter_content', None)
if callable(iter_content):
try:
iterator = iter(iter_content(chunk_size=min(64 * 1024, max_bytes + 1)))
except TypeError:
if isinstance(response, requests.Response):
raise DockerRegistryResolutionError(f'{label} cannot be streamed safely')
except (requests.RequestException, OSError) as exc:
raise DockerRemoteAccessError(
f'{label} stream is unavailable', status='remote_transient',
remote_attempted=True,
) from exc
raw_content = None
if iterator is not None:
content = bytearray()
try:
for chunk in iterator:
if deadline is not None and time.monotonic() >= float(deadline):
raise DockerRemoteAccessError(
f'{label} deadline expired', status='remote_transient',
remote_attempted=True,
)
if not chunk:
continue
if not isinstance(chunk, (bytes, bytearray)):
raise DockerRegistryResolutionError(f'{label} returned invalid bytes')
if len(content) + len(chunk) > max_bytes:
raise DockerRegistryResolutionError(f'{label} exceeds the response size limit')
content.extend(chunk)
except (DockerRegistryResolutionError, DockerRemoteAccessError):
raise
except (requests.RequestException, OSError) as exc:
raise DockerRemoteAccessError(
f'{label} stream failed', status='remote_transient',
remote_attempted=True,
) from exc
if deadline is not None and time.monotonic() >= float(deadline):
raise DockerRemoteAccessError(
f'{label} deadline expired', status='remote_transient',
remote_attempted=True,
)
raw_content = bytes(content)
if content_length is not None and len(raw_content) != content_length:
raise DockerRegistryResolutionError(f'{label} Content-Length is inconsistent')
try:
payload = json.loads(raw_content.decode('utf-8'))
except (RecursionError, UnicodeDecodeError, TypeError, ValueError) as exc:
raise DockerRegistryResolutionError(f'{label} is not valid JSON') from exc
else:
# Lightweight response doubles used by unit tests may not implement streaming.
content = getattr(response, 'content', None)
if isinstance(content, (bytes, bytearray)):
raw_content = bytes(content)
if len(raw_content) > max_bytes:
raise DockerRegistryResolutionError(f'{label} exceeds the response size limit')
try:
payload = response.json()
except (RecursionError, TypeError, ValueError) as exc:
raise DockerRegistryResolutionError(f'{label} is not valid JSON') from exc
if deadline is not None and time.monotonic() >= float(deadline):
raise DockerRemoteAccessError(
f'{label} deadline expired', status='remote_transient',
remote_attempted=True,
)
if not isinstance(payload, dict):
raise DockerRegistryResolutionError(f'{label} must be a JSON object')
if raw_content is None:
encoded = json.dumps(payload, ensure_ascii=True, separators=(',', ':')).encode('utf-8')
if len(encoded) > max_bytes:
raise DockerRegistryResolutionError(f'{label} exceeds the response size limit')
if return_raw:
if raw_content is None:
raise DockerRegistryResolutionError(f'{label} raw bytes are unavailable')
return payload, raw_content
return payload
def _docker_access_exhausted(endpoint, response=None, remote_attempted=False):
retry_at = put_dockerhub_exhausted_rate_limit(endpoint, response)
if retry_at:
raise DockerRemoteAccessError(
f'Docker {endpoint} accounts are rate-limited',
status='rate_limited', retry_at=retry_at,
remote_attempted=remote_attempted,
)
raise DockerRemoteAccessError(
f'Docker {endpoint} authentication is unavailable',
status='auth_failed', remote_attempted=remote_attempted,
)
def _docker_hub_access_token(
account, force_refresh=False, endpoint='hub_tags', stale_token='',
):
with docker_token_manager.hub_token_lock(account.name):
if not docker_token_manager.account_available(account.name, endpoint):
raise DockerRemoteAccessError(
f'Docker {endpoint} authentication is unavailable',
status='auth_failed', remote_attempted=False,
)
cached = docker_token_manager.cached_hub_token(account.name)
if force_refresh:
if stale_token and cached and cached != stale_token:
return cached
elif cached:
return cached
docker_token_manager.invalidate_hub_token(account.name)
response = api_request(
'POST', 'https://hub.docker.com/v2/auth/token',
json={'identifier': account.username, 'secret': account.token},
headers={'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json'},
timeout=(5, 15), max_retries=1,
retry_statuses={408, 500, 502, 503, 504},
allow_redirects=False,
)
if response.status_code in (401, 403, 429):
category = 'rate_limit' if response.status_code == 429 else (
'auth_invalid' if response.status_code == 401 else 'auth_forbidden'
)
docker_token_manager.report_http_status(
account, endpoint, response.status_code, response, category,
)
raise DockerRemoteAccessError(
f'Docker Hub token endpoint returned HTTP {response.status_code}',
status='rate_limited' if response.status_code == 429 else 'auth_failed',
remote_attempted=True,
)
if response.status_code >= 400:
raise DockerRemoteAccessError(
f'Docker Hub token endpoint returned HTTP {response.status_code}',
remote_attempted=True,
)
try:
payload = _bounded_docker_registry_json(
response, 'Docker Hub token response', max_bytes=1024 * 1024,
)
except DockerRegistryResolutionError as exc:
raise DockerRemoteAccessError(
'Docker Hub returned an invalid token response', remote_attempted=True,
) from exc
token = payload.get('access_token') or payload.get('token')
if not isinstance(token, str) or not token or len(token) > 16384:
raise DockerRemoteAccessError(
'Docker Hub returned an invalid access token', remote_attempted=True,
)
docker_token_manager.cache_hub_token(
account.name, token, payload.get('expires_in') or 600,
)
docker_token_manager.report_success(account, endpoint)
return token
def dockerhub_search_response(url, params, request_timeout=15):
endpoint = 'hub_search'
def request(token='', attempts=1):
headers = {
'User-Agent': 'GitSecretsScanner/2.0',
'Accept': 'application/json',
}
if token:
headers['Authorization'] = f'Bearer {token}'
return api_request(
'GET', url, params=params, headers=headers,
timeout=(5, request_timeout), max_retries=attempts, retry_delay=1,
retry_statuses={408, 500, 502, 503, 504},
allow_redirects=False,
)
if not docker_token_manager.has_accounts():
if docker_token_manager.uses_explicit_pool():
_docker_access_exhausted(endpoint, remote_attempted=False)
response = request(attempts=2)
if response.status_code == 429:
_docker_access_exhausted(endpoint, response, remote_attempted=True)
return response
excluded = set()
last_response = None
remote_attempted = False
account = None
token = ''
refreshed_accounts = set()
page_attempts = 0
while page_attempts < 2:
if account is None:
account = docker_token_manager.next_account(endpoint, excluded)
if account is None:
break
try:
token = _docker_hub_access_token(account, endpoint=endpoint)
except (ApiRequestError, DockerRemoteAccessError):
remote_attempted = True
excluded.add(account.name)
account = None
continue
try:
response = request(token)
except ApiRequestError:
remote_attempted = True
page_attempts += 1
if page_attempts < 2:
_wait_or_raise_scan_slot_fatal(1)
continue
raise
page_attempts += 1
remote_attempted = True
last_response = response
if (
response.status_code == 401
and account.name not in refreshed_accounts
and page_attempts < 2
):
refreshed_accounts.add(account.name)
try:
token = _docker_hub_access_token(
account, force_refresh=True, endpoint=endpoint,
stale_token=token,
)
continue
except (ApiRequestError, DockerRemoteAccessError):
excluded.add(account.name)
account = None
continue
if response.status_code not in (401, 403, 429):
docker_token_manager.report_success(account, endpoint)
return response
category = 'rate_limit' if response.status_code == 429 else 'auth_forbidden'
docker_token_manager.report_http_status(
account, endpoint, response.status_code, response, category,
)
excluded.add(account.name)
account = None
token = ''
_docker_access_exhausted(
endpoint, last_response, remote_attempted=remote_attempted,
)
def dockerhub_tags_response(url, params):
endpoint = 'hub_tags'
if not docker_token_manager.has_accounts():
if docker_token_manager.uses_explicit_pool():
_docker_access_exhausted(endpoint, remote_attempted=False)
response = api_request(
'GET', url, params=params,
headers={'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json'},
timeout=8, max_retries=1,
retry_statuses={408, 500, 502, 503, 504},
allow_redirects=False,
)
if response.status_code == 429:
_docker_access_exhausted(endpoint, response, remote_attempted=True)
return response
excluded = set()
last_response = None
remote_attempted = False
for _ in range(docker_token_manager.account_count()):
account = docker_token_manager.next_account(endpoint, excluded)
if account is None:
break
excluded.add(account.name)
try:
token = _docker_hub_access_token(account)
except DockerRemoteAccessError:
remote_attempted = True
continue
def request():
return api_request(
'GET', url, params=params,
headers={
'User-Agent': 'GitSecretsScanner/2.0',
'Accept': 'application/json',
'Authorization': f'Bearer {token}',
},
timeout=8, max_retries=1,
retry_statuses={408, 500, 502, 503, 504},
allow_redirects=False,
)
response = request()
remote_attempted = True
last_response = response
if response.status_code == 401:
try:
token = _docker_hub_access_token(
account, force_refresh=True, stale_token=token,
)
response = request()
last_response = response
except DockerRemoteAccessError:
continue
if response.status_code not in (401, 403, 429):
docker_token_manager.report_success(account, endpoint)
return response
if response.status_code != 403:
category = 'rate_limit' if response.status_code == 429 else 'auth_forbidden'
docker_token_manager.report_http_status(
account, endpoint, response.status_code, response, category,
)
_docker_access_exhausted(
endpoint, last_response, remote_attempted=remote_attempted,
)
def docker_registry_bearer_token(
challenge, repo_key, excluded_accounts=None, *, deadline=None,
anonymous_only=False,
):
text = str(challenge or '').strip()
if not text.lower().startswith('bearer '):
raise DockerRegistryResolutionError('Docker registry did not provide a bearer challenge')
values = {
key.lower(): value
for key, value in re.findall(r'([A-Za-z][A-Za-z0-9_-]*)="([^"\\]*)"', text[7:])
}
realm = values.get('realm', '')
parsed = urlsplit(realm)
if (
parsed.scheme.lower() != 'https'
or (parsed.hostname or '').lower() != 'auth.docker.io'
or parsed.username is not None
or parsed.password is not None
or parsed.port not in (None, 443)
):
raise DockerRegistryResolutionError('Docker registry bearer realm is not trusted')
excluded = set(excluded_accounts or ())
authenticated = not anonymous_only and docker_token_manager.has_accounts()
if (
not authenticated and not anonymous_only
and docker_token_manager.uses_explicit_pool()
):
_docker_access_exhausted('registry', remote_attempted=False)
attempts = docker_token_manager.account_count() if authenticated else 1
last_response = None
remote_attempted = False
saw_auth_failure = False
saw_rate_limit = False
saw_target_forbidden = False
saw_invalid_response = False
for _ in range(max(1, attempts)):
if deadline is not None and time.monotonic() >= float(deadline):
raise DockerRemoteAccessError(
'Docker registry token deadline expired',
status='remote_transient', remote_attempted=remote_attempted,
)
account = docker_token_manager.next_account('registry', excluded) if authenticated else None
if authenticated and account is None:
break
if account is not None:
excluded.add(account.name)
try:
response = api_request(
'GET', realm,
params={
'service': 'registry.docker.io',
'scope': f'repository:{repo_key}:pull',
},
headers={
'User-Agent': 'GitSecretsScanner/2.0',
'Accept': 'application/json',
'Accept-Encoding': 'identity',
},
auth=(account.username, account.token) if account is not None else None,
timeout=(5, 15), max_retries=1,
retry_statuses={408, 500, 502, 503, 504},
allow_redirects=False, stream=True, deadline=deadline,
)
except ApiRequestError as exc:
raise DockerRemoteAccessError(
'Docker registry token endpoint is temporarily unavailable',
status='remote_transient', remote_attempted=True,
) from exc
remote_attempted = True
last_response = response
try:
status_code = int(response.status_code)
if status_code in (401, 403, 429):
if status_code == 429:
saw_rate_limit = True
if account is not None:
docker_token_manager.report_http_status(
account, 'registry', status_code, response, 'rate_limit',
)
elif status_code == 401:
saw_auth_failure = True
if account is not None:
docker_token_manager.report_http_status(
account, 'registry', status_code, response, 'auth_invalid',
)
else:
saw_target_forbidden = True
continue
if status_code in (408, 425) or status_code >= 500:
raise DockerRemoteAccessError(
'Docker registry token endpoint is temporarily unavailable',
status='remote_transient', remote_attempted=True,
)
if status_code >= 400:
raise DockerRemoteAccessError(
'Docker registry denied access to the requested target',
status='target_forbidden', remote_attempted=True,
)
try:
payload = _bounded_docker_registry_json(
response, 'Docker registry token response',
max_bytes=DOCKER_REGISTRY_TOKEN_MAX_BYTES, deadline=deadline,
)
except DockerRegistryResolutionError:
saw_invalid_response = True
continue
finally:
response.close()
token = payload.get('token') or payload.get('access_token')
if not isinstance(token, str) or not token or len(token) > 16384:
saw_invalid_response = True
continue
if account is not None:
docker_token_manager.report_success(account, 'registry')
return DockerRegistryAuth(
token=token,
account_name=account.name if account is not None else '',
challenge=text,
)
if saw_rate_limit:
retry_at = put_dockerhub_exhausted_rate_limit('registry', last_response)
raise DockerRemoteAccessError(
'Docker registry accounts are rate-limited', status='rate_limited',
retry_at=retry_at, remote_attempted=remote_attempted,
)
if saw_target_forbidden:
raise DockerRemoteAccessError(
'Docker registry denied access to the requested target',
status='target_forbidden', remote_attempted=remote_attempted,
)
if saw_invalid_response and not saw_auth_failure:
raise DockerRemoteAccessError(
'Docker registry returned an invalid token response',
status='remote_transient', remote_attempted=remote_attempted,
)
_docker_access_exhausted(
'registry', last_response, remote_attempted=remote_attempted,
)
def docker_registry_manifest(
repo_name, digest, bearer_auth=None, *, verify_content_digest=False,
return_raw=False, deadline=None, lease_renewal_callback=None,
anonymous_only=False,
):
repo_key = dockerhub_repo_key(repo_name)
digest = normalize_docker_digest(digest)
if not digest:
raise DockerRegistryResolutionError('Docker manifest digest is invalid')
url = f"https://registry-1.docker.io/v2/{quote(repo_key, safe='/')}/manifests/{digest}"
accept = ', '.join((
'application/vnd.oci.image.index.v1+json',
'application/vnd.docker.distribution.manifest.list.v2+json',
'application/vnd.oci.image.manifest.v1+json',
'application/vnd.docker.distribution.manifest.v2+json',
))
if isinstance(bearer_auth, str):
bearer_auth = DockerRegistryAuth(token=bearer_auth)
bearer_auth = bearer_auth or DockerRegistryAuth(token='')
def request(auth):
headers = {
'User-Agent': 'GitSecretsScanner/2.0',
'Accept': accept,
'Accept-Encoding': 'identity',
}
if auth.token:
headers['Authorization'] = f'Bearer {auth.token}'
return api_request(
'GET', url, headers=headers, timeout=(5, 15), max_retries=1,
retry_statuses={408, 500, 502, 503, 504},
allow_redirects=False, stream=True, deadline=deadline,
)
excluded = set()
attempts = 2 if anonymous_only else max(
2, docker_token_manager.account_count() + 1,
)
response = None
last_response = None
last_status = None
saw_bearer_unauthorized = False
for _ in range(attempts):
if lease_renewal_callback is not None:
if not callable(lease_renewal_callback):
raise ValueError('Docker resolver lease renewal callback is invalid')
if lease_renewal_callback() is False:
raise DockerResolverLeaseLostError('Docker resolver lease renewal was rejected')
if deadline is not None and time.monotonic() >= float(deadline):
raise DockerRemoteAccessError(
'Docker registry manifest deadline expired',
status='remote_transient', remote_attempted=response is not None,
)
response = request(bearer_auth)
last_response = response
last_status = int(response.status_code)
if response.status_code not in (401, 403, 429):
break
challenge = (
(getattr(response, 'headers', None) or {}).get('WWW-Authenticate')
or bearer_auth.challenge
)
status_code = int(response.status_code)
if status_code == 401 and bearer_auth.token:
saw_bearer_unauthorized = True
if bearer_auth.account_name:
if status_code == 429:
docker_token_manager.report_http_status(
bearer_auth.account_name, 'registry', 429, response, 'rate_limit',
)
excluded.add(bearer_auth.account_name)
if status_code == 403:
response.close()
raise DockerRemoteAccessError(
'Docker registry denied access to the requested manifest',
status='target_forbidden', remote_attempted=True,
)
if status_code == 429 and (
anonymous_only or not docker_token_manager.has_accounts()
):
try:
_docker_access_exhausted('registry', response, remote_attempted=True)
finally:
response.close()
response.close()
response = None
try:
if lease_renewal_callback is not None and lease_renewal_callback() is False:
raise DockerResolverLeaseLostError('Docker resolver lease renewal was rejected')
bearer_auth = docker_registry_bearer_token(
challenge, repo_key, excluded_accounts=excluded, deadline=deadline,
anonymous_only=anonymous_only,
)
except DockerRemoteAccessError as exc:
if saw_bearer_unauthorized and exc.status == 'auth_failed':
raise DockerRemoteAccessError(
'Docker registry denied access to the requested manifest',
status='target_forbidden', remote_attempted=True,
) from exc
raise
except DockerRegistryResolutionError as exc:
raise DockerRemoteAccessError(
'Docker registry authentication challenge is invalid',
status='auth_failed' if status_code == 401 else 'remote_transient',
remote_attempted=True,
) from exc
if response is None:
if last_status == 429:
retry_at = put_dockerhub_exhausted_rate_limit('registry', last_response)
raise DockerRemoteAccessError(
'Docker registry accounts are rate-limited', status='rate_limited',
retry_at=retry_at, remote_attempted=True,
)
if last_status == 401:
raise DockerRemoteAccessError(
'Docker registry denied access to the requested manifest'
if saw_bearer_unauthorized
else 'Docker registry authentication is unavailable',
status='target_forbidden' if saw_bearer_unauthorized else 'auth_failed',
remote_attempted=True,
)
raise DockerRemoteAccessError(
'Docker registry manifest request did not run',
status='remote_transient', remote_attempted=last_response is not None,
)
if response.status_code in (401, 429):
try:
_docker_access_exhausted('registry', response, remote_attempted=True)
finally:
response.close()
status_code = int(response.status_code)
if 300 <= status_code < 400:
response.close()
raise DockerRegistryResolutionError('Docker registry manifest redirect was rejected')
try:
if status_code == 404:
raise DockerRegistryResolutionError('Docker registry manifest was not found')
if status_code in (408, 425) or status_code >= 500:
raise DockerRemoteAccessError(
'Docker registry manifest endpoint is temporarily unavailable',
status='remote_transient', remote_attempted=True,
)
if status_code >= 400:
raise DockerRegistryResolutionError('Docker registry rejected the manifest request')
returned_digest = normalize_docker_digest(
(getattr(response, 'headers', None) or {}).get('Docker-Content-Digest')
)
if returned_digest and returned_digest != digest:
raise DockerRegistryResolutionError('Docker registry returned a different manifest digest')
if verify_content_digest or return_raw:
payload, raw_content = _bounded_docker_registry_json(
response, 'Docker manifest', deadline=deadline, return_raw=True,
)
else:
payload = _bounded_docker_registry_json(
response, 'Docker manifest', deadline=deadline,
)
raw_content = None
if verify_content_digest:
calculated = 'sha256:' + hashlib.sha256(raw_content).hexdigest()
if calculated != digest:
raise DockerRegistryResolutionError('Docker manifest payload digest is invalid')
if bearer_auth.account_name:
docker_token_manager.report_success(bearer_auth.account_name, 'registry')
if return_raw:
return payload, bearer_auth, raw_content
return payload, bearer_auth
finally:
response.close()
def resolve_docker_layer_graph(
repo_name, digest, platform_os='linux', platform_arch='amd64',
bearer_auth=None, *, deadline=None, include_descriptors=False,
lease_renewal_callback=None,
):
manifest_digest = normalize_docker_digest(digest)
if include_descriptors:
payload, bearer_auth, raw_content = docker_registry_manifest(
repo_name, manifest_digest, bearer_auth, deadline=deadline,
verify_content_digest=True, return_raw=True,
lease_renewal_callback=lease_renewal_callback,
)
else:
payload, bearer_auth = docker_registry_manifest(
repo_name, manifest_digest, bearer_auth, deadline=deadline,
lease_renewal_callback=lease_renewal_callback,
)
raw_content = None
descriptors = payload.get('manifests')
if descriptors is not None:
if not isinstance(descriptors, list) or len(descriptors) > DOCKER_REGISTRY_MAX_DESCRIPTORS:
raise DockerRegistryResolutionError('Docker manifest index has invalid descriptor bounds')
wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower())
descriptor = next((
item for item in descriptors
if isinstance(item, dict)
and isinstance(item.get('platform'), dict)
and (
str(item['platform'].get('os') or '').lower(),
str(item['platform'].get('architecture') or '').lower(),
) == wanted
), None)
if descriptor is None:
return None, bearer_auth
manifest_digest = normalize_docker_digest(descriptor.get('digest'))
if not manifest_digest:
raise DockerRegistryResolutionError('Docker platform descriptor has an invalid digest')
if include_descriptors:
payload, bearer_auth, raw_content = docker_registry_manifest(
repo_name, manifest_digest, bearer_auth, deadline=deadline,
verify_content_digest=True, return_raw=True,
lease_renewal_callback=lease_renewal_callback,
)
else:
payload, bearer_auth = docker_registry_manifest(
repo_name, manifest_digest, bearer_auth, deadline=deadline,
lease_renewal_callback=lease_renewal_callback,
)
layers = payload.get('layers')
if not isinstance(layers, list) or len(layers) > DOCKER_REGISTRY_MAX_LAYERS:
raise DockerRegistryResolutionError('Docker manifest has invalid layer bounds')
ordered_layers = []
layer_descriptors = []
for position, layer in enumerate(layers, 1):
layer_digest = normalize_docker_digest(layer.get('digest')) if isinstance(layer, dict) else ''
if not layer_digest:
raise DockerRegistryResolutionError('Docker manifest contains an invalid layer digest')
ordered_layers.append(layer_digest)
if include_descriptors:
layer_descriptors.append(_docker_content_descriptor(layer, 'layer', position))
graph = {
'manifest_digest': manifest_digest,
'layers': tuple(ordered_layers),
}
if include_descriptors:
config = _docker_content_descriptor(payload.get('config'), 'config', 0)
manifest_media_type = str(
payload.get('mediaType')
or 'application/vnd.docker.distribution.manifest.v2+json'
).strip().lower()
if not manifest_media_type or len(manifest_media_type) > 256:
raise DockerRegistryResolutionError('Docker manifest media type is invalid')
graph.update({
'manifest_media_type': manifest_media_type,
'manifest_size_bytes': len(raw_content),
'config_digest': config['digest'],
'layer_descriptors': tuple(layer_descriptors),
})
return graph, bearer_auth
def _dockerhub_manifest_target_parts(target):
parsed = parse_docker_target(target)
image = str(parsed['image']).lower()
image_name, manifest_digest = image.rsplit('@', 1)
manifest_digest = normalize_docker_digest(manifest_digest)
if not manifest_digest:
raise DockerRegistryResolutionError('Docker manifest target digest is invalid')
parts = image_name.split('/')
if len(parts) > 1 and ('.' in parts[0] or ':' in parts[0] or parts[0] == 'localhost'):
registry = parts.pop(0)
if registry not in ('docker.io', 'index.docker.io', 'registry-1.docker.io'):
raise DockerRegistryResolutionError(
'Docker layer scanning only supports Docker Hub targets'
)
if not parts or any(not part for part in parts):
raise DockerRegistryResolutionError('Docker Hub repository is invalid')
repository = '/'.join(parts)
registry_repository = repository if '/' in repository else f'library/{repository}'
return image, repository, registry_repository, manifest_digest
def _docker_content_descriptor(value, kind, position):
if not isinstance(value, dict):
raise DockerRegistryResolutionError(f'Docker {kind} descriptor is invalid')
digest = normalize_docker_digest(value.get('digest'))
size = value.get('size')
media_type = str(value.get('mediaType') or '').strip().lower()
if (
not digest
or isinstance(size, bool)
or not isinstance(size, int)
or size < 0
or size > 1024 * 1024 * 1024 * 1024
or not media_type
or len(media_type) > 256
):
raise DockerRegistryResolutionError(f'Docker {kind} descriptor has invalid bounds')
return {
'digest': digest,
'size': size,
'media_type': media_type,
}
def resolve_docker_content_manifest(
target, platform_os='linux', platform_arch='amd64', bearer_auth=None,
*, deadline=None, anonymous_only=False,
):
image, repository, registry_repository, manifest_digest = (
_dockerhub_manifest_target_parts(target)
)
target_manifest_digest = manifest_digest
payload, bearer_auth = docker_registry_manifest(
registry_repository, manifest_digest, bearer_auth,
verify_content_digest=True, deadline=deadline,
anonymous_only=anonymous_only,
)
descriptors = payload.get('manifests')
if descriptors is not None:
if not isinstance(descriptors, list) or len(descriptors) > DOCKER_REGISTRY_MAX_DESCRIPTORS:
raise DockerRegistryResolutionError('Docker manifest index has invalid descriptor bounds')
wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower())
child = next((
item for item in descriptors
if isinstance(item, dict)
and isinstance(item.get('platform'), dict)
and (
str(item['platform'].get('os') or '').lower(),
str(item['platform'].get('architecture') or '').lower(),
) == wanted
), None)
if child is None:
raise DockerRegistryResolutionError('Docker target platform manifest is unavailable')
manifest_digest = normalize_docker_digest(child.get('digest'))
if not manifest_digest:
raise DockerRegistryResolutionError('Docker platform descriptor digest is invalid')
payload, bearer_auth = docker_registry_manifest(
registry_repository, manifest_digest, bearer_auth,
verify_content_digest=True, deadline=deadline,
anonymous_only=anonymous_only,
)
config = _docker_content_descriptor(payload.get('config'), 'config', 0)
layers = payload.get('layers')
if not isinstance(layers, list) or len(layers) > DOCKER_REGISTRY_MAX_LAYERS:
raise DockerRegistryResolutionError('Docker manifest has invalid layer bounds')
normalized_layers = [
_docker_content_descriptor(layer, 'layer', position)
for position, layer in enumerate(layers, 1)
]
manifest_media_type = str(
payload.get('mediaType')
or 'application/vnd.docker.distribution.manifest.v2+json'
).strip().lower()
if not manifest_media_type or len(manifest_media_type) > 256:
raise DockerRegistryResolutionError('Docker manifest media type is invalid')
if deadline is not None and time.monotonic() >= float(deadline):
raise DockerRemoteAccessError(
'Docker registry manifest deadline expired',
status='remote_transient', remote_attempted=True,
)
return {
'version': 1,
'image': image,
'repository': registry_repository,
'manifest_digest': target_manifest_digest,
'platform_os': str(platform_os or 'linux').lower(),
'platform_arch': str(platform_arch or 'amd64').lower(),
'manifest_media_type': manifest_media_type,
'config': config,
'layers': normalized_layers,
}, bearer_auth
DOCKER_BLOB_REDIRECT_SUFFIXES = (
'.docker.com',
'.docker.io',
'.cloudfront.net',
'.cloudflarestorage.com',
'.amazonaws.com',
)
def _docker_blob_url_validation_error(url, *, registry_origin=False):
try:
parsed = urlsplit(str(url or ''))
hostname = (parsed.hostname or '').lower().rstrip('.')
port = parsed.port
except ValueError:
return 'invalid_url'
if (
parsed.scheme.lower() != 'https'
or not hostname
or parsed.username is not None
or parsed.password is not None
or port not in (None, 443)
or parsed.fragment
):
return 'invalid_url'
if registry_origin:
return '' if hostname == 'registry-1.docker.io' and not parsed.query else 'invalid_registry'
try:
address = ipaddress.ip_address(hostname)
except ValueError:
address = None
if address is not None and not address.is_global:
return 'non_global_address'
if not any(hostname.endswith(suffix) for suffix in DOCKER_BLOB_REDIRECT_SUFFIXES):
return 'untrusted_host'
try:
answers = socket.getaddrinfo(
hostname, 443, type=socket.SOCK_STREAM, proto=socket.IPPROTO_TCP,
)
except OSError:
return 'dns_unavailable'
if not answers:
return 'dns_unavailable'
for answer in answers:
try:
resolved = ipaddress.ip_address(str(answer[4][0]).split('%', 1)[0])
except (IndexError, TypeError, ValueError):
return 'invalid_dns_answer'
if not resolved.is_global:
return 'non_global_address'
return ''
def _docker_blob_url_allowed(url, *, registry_origin=False):
return not _docker_blob_url_validation_error(url, registry_origin=registry_origin)
def _require_docker_blob_url(url, *, registry_origin=False):
error = _docker_blob_url_validation_error(url, registry_origin=registry_origin)
if error == 'dns_unavailable':
raise DockerLayerInfrastructureError(
'remote_dns', 'Docker blob redirect DNS is temporarily unavailable',
category='remote_transient',
)
if error:
raise DockerContentTransferError(
'unsafe_redirect' if not registry_origin else 'unsafe_registry_url',
'Docker blob URL is not trusted', False,
)
def _docker_registry_blob_response(
repository, digest, bearer_auth, deadline, *, anonymous_only=False,
):
repo_key = dockerhub_repo_key(repository)
digest = normalize_docker_digest(digest)
if not digest:
raise DockerContentTransferError('invalid_descriptor', 'Docker blob digest is invalid', False)
url = f'https://registry-1.docker.io/v2/{quote(repo_key, safe="/")}/blobs/{digest}'
_require_docker_blob_url(url, registry_origin=True)
if isinstance(bearer_auth, str):
bearer_auth = DockerRegistryAuth(token=bearer_auth)
bearer_auth = bearer_auth or DockerRegistryAuth(token='')
excluded = set()
attempts = 2 if anonymous_only else max(
2, docker_token_manager.account_count() + 1,
)
response = None
saw_rate_limit = False
saw_bearer_unauthorized = False
for _ in range(attempts):
remaining = max(0.0, float(deadline) - time.monotonic())
if remaining <= 0:
raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired')
headers = {
'User-Agent': 'GitSecretsScanner/2.0',
'Accept': 'application/octet-stream',
'Accept-Encoding': 'identity',
}
if bearer_auth.token:
headers['Authorization'] = f'Bearer {bearer_auth.token}'
try:
response = api_request(
'GET', url, headers=headers, timeout=(5, min(30, remaining)),
use_proxy=False,
max_retries=1, retry_statuses={408, 500, 502, 503, 504},
allow_redirects=False, stream=True, deadline=deadline,
)
except ApiRequestError as exc:
if time.monotonic() >= float(deadline):
raise DockerContentTransferError(
'transfer_timeout', 'Docker blob deadline expired',
) from exc
raise DockerLayerInfrastructureError(
'remote_transient', 'Docker blob endpoint is temporarily unavailable',
category='remote_transient',
) from exc
if response.status_code not in (401, 429):
return response, bearer_auth
challenge = (
(getattr(response, 'headers', None) or {}).get('WWW-Authenticate')
or bearer_auth.challenge
)
if response.status_code == 401 and bearer_auth.token:
saw_bearer_unauthorized = True
if bearer_auth.account_name:
if response.status_code == 429:
saw_rate_limit = True
docker_token_manager.report_http_status(
bearer_auth.account_name, 'registry', 429, response, 'rate_limit',
)
excluded.add(bearer_auth.account_name)
elif response.status_code == 429:
saw_rate_limit = True
response.close()
response = None
try:
bearer_auth = docker_registry_bearer_token(
challenge, repo_key, excluded_accounts=excluded, deadline=deadline,
anonymous_only=anonymous_only,
)
except DockerRemoteAccessError as exc:
if exc.status == 'target_forbidden':
raise DockerContentTransferError(
'target_forbidden', 'Docker blob target is forbidden', False,
) from exc
if exc.status == 'rate_limited':
raise DockerLayerInfrastructureError(
'remote_rate_limit', 'Docker blob authorization is rate-limited',
category='docker_rate_limit',
) from exc
if exc.status == 'auth_failed':
if saw_bearer_unauthorized:
raise DockerContentTransferError(
'target_forbidden', 'Docker blob target is forbidden', False,
) from exc
raise DockerLayerInfrastructureError(
'remote_auth', 'Docker blob authorization is unavailable',
category='docker_auth', auth_related=True,
) from exc
raise DockerLayerInfrastructureError(
'remote_transient', 'Docker blob authorization is temporarily unavailable',
category='remote_transient',
) from exc
except DockerRegistryResolutionError as exc:
raise DockerLayerInfrastructureError(
'remote_auth', 'Docker blob authentication challenge is invalid',
category='docker_auth', auth_related=True,
) from exc
if response is not None:
response.close()
if saw_rate_limit:
raise DockerLayerInfrastructureError(
'remote_rate_limit', 'Docker blob authorization is rate-limited',
category='docker_rate_limit',
)
if saw_bearer_unauthorized:
raise DockerContentTransferError(
'target_forbidden', 'Docker blob target is forbidden', False,
)
raise DockerLayerInfrastructureError(
'remote_auth', 'Docker blob authorization is unavailable',
category='docker_auth', auth_related=True,
)
def stream_docker_registry_blob(
repository, descriptor, destination, bearer_auth=None, *, deadline,
min_free_bytes=0, redirect_limit=5, anonymous_only=False,
):
digest = normalize_docker_digest((descriptor or {}).get('digest'))
declared_bytes = (descriptor or {}).get('size')
kind = str((descriptor or {}).get('kind') or '')
media_type = str((descriptor or {}).get('media_type') or '').strip().lower()
if (
not digest
or isinstance(declared_bytes, bool)
or not isinstance(declared_bytes, int)
or declared_bytes < 0
or declared_bytes > 1024 * 1024 * 1024 * 1024
or kind not in ('config', 'layer')
or not media_type
):
raise DockerContentTransferError('invalid_descriptor', 'Docker blob descriptor is invalid', False)
supported_media = DOCKER_CONFIG_MEDIA_TYPES if kind == 'config' else DOCKER_LAYER_MEDIA_TYPES
if media_type not in supported_media:
raise DockerContentTransferError(
'unsupported_media_type', 'Docker blob media type is unsupported', False,
)
deadline = float(deadline)
if deadline <= time.monotonic():
raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired')
destination = os.path.abspath(destination)
try:
parent = require_private_directory(os.path.dirname(destination), create=False)
reject_reparse_components(parent)
except (OSError, ValueError) as exc:
raise DockerLayerInfrastructureError(
'private_storage', 'Docker blob private storage is unavailable',
category='source_resource',
) from exc
if os.path.lexists(destination):
raise DockerLayerInfrastructureError(
'destination_exists', 'Docker blob destination is not available',
category='source_resource',
)
required_free = max(0, int(min_free_bytes)) + declared_bytes
try:
free_bytes = shutil.disk_usage(parent).free
except OSError as exc:
raise DockerLayerInfrastructureError(
'disk_reserve', 'Docker blob free space cannot be verified',
category='source_resource',
) from exc
if free_bytes < required_free:
raise DockerLayerInfrastructureError(
'disk_reserve', 'Docker blob would violate the free-space reserve',
category='source_resource',
)
started = time.monotonic()
response = None
total = 0
temporary = (
f'{destination}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.partial'
)
published = False
try:
response, bearer_auth = _docker_registry_blob_response(
repository, digest, bearer_auth, deadline,
anonymous_only=anonymous_only,
)
redirects = 0
while response.status_code in (301, 302, 303, 307, 308):
location = (getattr(response, 'headers', None) or {}).get('Location')
current_url = str(getattr(response, 'url', '') or '')
response.close()
response = None
redirects += 1
if not location or redirects > max(0, min(5, int(redirect_limit))):
raise DockerContentTransferError('unsafe_redirect', 'Docker blob redirect limit exceeded', False)
next_url = urljoin(current_url, location)
_require_docker_blob_url(next_url)
remaining = max(0.0, deadline - time.monotonic())
if remaining <= 0:
raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired')
response = api_request(
'GET', next_url,
use_proxy=False,
headers={
'User-Agent': 'GitSecretsScanner/2.0',
'Accept': 'application/octet-stream',
'Accept-Encoding': 'identity',
},
timeout=(5, min(30, remaining)), max_retries=1,
retry_statuses={408, 500, 502, 503, 504},
allow_redirects=False, stream=True, deadline=deadline,
)
status_code = int(response.status_code)
if status_code == 401:
raise DockerContentTransferError(
'target_forbidden', 'Docker blob target is forbidden', False,
)
if status_code == 403:
raise DockerContentTransferError(
'target_forbidden', 'Docker blob target is forbidden', False,
)
if status_code == 404:
raise DockerContentTransferError(
'blob_not_found', 'Docker blob target is unavailable', False,
)
if status_code == 429:
raise DockerLayerInfrastructureError(
'remote_rate_limit', 'Docker blob endpoint is rate-limited',
category='docker_rate_limit',
)
if status_code in (408, 425) or status_code >= 500:
raise DockerLayerInfrastructureError(
'remote_transient', 'Docker blob endpoint is temporarily unavailable',
category='remote_transient',
)
if status_code >= 400:
raise DockerContentTransferError(
'target_rejected', 'Docker blob target was rejected', False,
)
content_encoding = str(
(getattr(response, 'headers', None) or {}).get('Content-Encoding') or ''
).strip().lower()
if content_encoding not in ('', 'identity'):
raise DockerContentTransferError(
'content_encoding', 'Docker blob response changed the content encoding',
)
content_length = (getattr(response, 'headers', None) or {}).get('Content-Length')
try:
content_length = int(content_length)
except (TypeError, ValueError) as exc:
raise DockerContentTransferError(
'size_mismatch', 'Docker blob response lacks an exact Content-Length', False,
) from exc
if content_length != declared_bytes:
raise DockerContentTransferError(
'size_mismatch', 'Docker blob Content-Length differs from its descriptor', False,
)
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0)
descriptor_fd = os.open(temporary, flags, 0o600)
os.close(descriptor_fd)
harden_private_file(temporary)
digest_hash = hashlib.sha256()
with open(temporary, 'wb', buffering=0) as output:
for chunk in response.iter_content(chunk_size=1024 * 1024):
_raise_if_scan_slot_fatal()
if time.monotonic() >= deadline:
raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired')
if not chunk:
continue
total += len(chunk)
if total > declared_bytes:
raise DockerContentTransferError(
'size_mismatch', 'Docker blob exceeded its declared size', False,
)
if shutil.disk_usage(parent).free < max(0, int(min_free_bytes)):
raise DockerLayerInfrastructureError(
'disk_reserve', 'Docker blob transfer reached the free-space reserve',
category='source_resource',
)
digest_hash.update(chunk)
output.write(chunk)
output.flush()
os.fsync(output.fileno())
if time.monotonic() >= deadline:
raise DockerContentTransferError(
'transfer_timeout', 'Docker blob deadline expired',
)
if total != declared_bytes:
raise DockerContentTransferError(
'size_mismatch', 'Docker blob byte count differs from its descriptor', False,
)
if f'sha256:{digest_hash.hexdigest()}' != digest:
raise DockerContentTransferError(
'digest_mismatch', 'Docker blob SHA-256 differs from its descriptor', False,
)
harden_private_file(temporary)
durable_replace(temporary, destination)
if not private_file_ready(destination):
raise DockerLayerInfrastructureError(
'private_file_lost', 'Docker blob lost its private file identity',
category='source_resource',
)
if time.monotonic() >= deadline:
raise DockerContentTransferError(
'transfer_timeout', 'Docker blob deadline expired',
)
published = True
duration_ms = max(0, int((time.monotonic() - started) * 1000))
return DockerBlobDownloadOutcome(
path=destination,
verified_bytes=total,
transfer_bytes=total,
duration_ms=duration_ms,
bearer_auth=bearer_auth,
)
except DockerContentTransferError as exc:
exc.transfer_bytes = min(declared_bytes, max(0, int(total)))
exc.duration_ms = max(0, int((time.monotonic() - started) * 1000))
raise
except (ApiRequestError, requests.RequestException) as exc:
if time.monotonic() >= deadline:
raise DockerContentTransferError(
'transfer_timeout', 'Docker blob deadline expired',
) from exc
raise DockerLayerInfrastructureError(
'remote_transient', 'Docker blob transfer is temporarily unavailable',
category='remote_transient',
) from exc
except OSError as exc:
raise DockerLayerInfrastructureError(
'private_storage', 'Docker blob private storage failed',
category='source_resource',
) from exc
finally:
if response is not None:
response.close()
if os.path.lexists(temporary):
durable_unlink(temporary)
if not published and os.path.lexists(destination):
durable_unlink(destination)
def fetch_docker_config_payload_classes(
resolved, bearer_auth=None, *, deadline, min_free_bytes=0,
):
layers = list((resolved or {}).get('layers') or ())
fallback = ['unknown'] * len(layers)
config = dict((resolved or {}).get('config') or {})
config.update({'kind': 'config', 'position': 0})
if (
config.get('media_type') not in DOCKER_CONFIG_MEDIA_TYPES
or not isinstance(config.get('size'), int)
or config['size'] < 0
or config['size'] > DOCKER_REGISTRY_MANIFEST_MAX_BYTES
):
return fallback, bearer_auth
work_root = None
destination = None
try:
work_root = tempfile.mkdtemp(prefix='docker-history-', dir=get_work_dir())
harden_private_directory(work_root)
write_temp_owner(work_root, ['docker-config-history'], os.getpid(), required=True)
destination = os.path.join(work_root, 'config.json')
outcome = stream_docker_registry_blob(
resolved['repository'], config, destination, bearer_auth,
deadline=float(deadline), min_free_bytes=max(0, int(min_free_bytes or 0)),
)
try:
parsed = validate_docker_content_artifact(destination, config)
except DockerContentScanError:
return fallback, outcome.bearer_auth
return docker_config_payload_classes(parsed, len(layers)), outcome.bearer_auth
except DockerContentScanError:
return fallback, bearer_auth
finally:
if destination and os.path.lexists(destination):
durable_unlink(destination)
if work_root:
cleanup_command_work_dir(work_root)
def _docker_depth_selection_evidence(
record, candidate_count, selector_version=DOCKER_DEPTH_SELECTOR_VERSION,
):
evidence = {
'schema': 1,
'type': 'docker-depth-selection-evidence-v1',
'selector_version': selector_version,
'selector_sha256': canonical_selector_hash(selector_version),
'candidate_distinct_graph_count': int(candidate_count),
'image_rank': int(record['image_rank']),
'selection_reason': str(record['selection_reason']),
'target': str(record['target']),
'repository': str(record['repository']),
'manifest_digest': str(record['manifest_digest']),
'manifest_media_type': str(record['manifest_media_type']),
'manifest_size_bytes': int(record['manifest_size_bytes']),
'config_digest': str(record['config_digest']),
'graph_sha256': str(record['graph_sha256']),
'layers': [dict(layer) for layer in record['layer_metadata']],
}
return {
**record,
'candidate_distinct_graph_count': int(candidate_count),
'selection_evidence_sha256': canonical_docker_depth_selection_evidence_hash(
evidence
),
}
def dockerhub_tag_digest(tag, platform_filter_enabled=False, platform_os='linux', platform_arch='amd64'):
if not isinstance(tag, dict):
return ''
images = tag.get('images') if isinstance(tag.get('images'), list) else []
wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower())
candidates = (
image.get('digest') for image in images
if isinstance(image, dict)
and (str(image.get('os') or '').lower(), str(image.get('architecture') or '').lower()) == wanted
)
platform_digest = next((
digest for digest in (normalize_docker_digest(value) for value in candidates) if digest
), '')
return platform_digest or normalize_docker_digest(tag.get('digest'))
def fetch_dockerhub_tags(
repo_name, since=None, limit=1, retry_count=2, retry_delay=5,
platform_filter_enabled=False, platform_os='linux', platform_arch='amd64',
platform_candidate_tags=20, return_status=False, *, return_outcome=False,
fresh_graph_evidence=False, lease_renewal_callback=None,
selector_version=DOCKER_DEPTH_SELECTOR_VERSION,
):
def output(
tags, status, remote_attempted=False, retry_at=None, error='',
selection_records=(), candidate_records=(), candidate_distinct_graph_count=0,
):
tags = list(tags or [])
if return_outcome:
return DockerTagResolutionOutcome(
tags=tuple(tags), status=str(status),
remote_attempted=bool(remote_attempted),
retry_at=retry_at, error=str(error or '')[:500],
selection_records=tuple(selection_records or ()),
candidate_records=tuple(candidate_records or selection_records or ()),
selector_version=selector_version,
selector_hash=canonical_selector_hash(selector_version),
candidate_distinct_graph_count=int(candidate_distinct_graph_count or 0),
fresh_graph_evidence=bool(
fresh_graph_evidence
and remote_attempted
and docker_tag_resolution_is_conclusive(status)
),
cache_bypassed=bool(fresh_graph_evidence),
)
return (tags, status) if return_status else tags
def cache_write(*values, **kwargs):
if not fresh_graph_evidence:
put_dockerhub_tag_cache(*values, **kwargs)
try:
retry_count = max(0, min(5, int(retry_count or 0)))
except (TypeError, ValueError):
retry_count = 2
try:
retry_delay = max(0, min(60, int(retry_delay or 0)))
except (TypeError, ValueError):
retry_delay = 5
limit = docker_images_per_repository_limit(limit)
repo_name = str(repo_name or '').strip()
if '@' in repo_name:
try:
return output([parse_docker_target(repo_name)['target']], 'ok')
except (TypeError, ValueError):
return output([], 'unknown')
if ':' in repo_name.rsplit('/', 1)[-1]:
return output([], 'unknown')
platform_variant = (
f'{selector_version}:{str(platform_os).lower()}/{str(platform_arch).lower()}:'
f'filter={int(bool(platform_filter_enabled))}:candidates={int(platform_candidate_tags or 0)}'
)
cached = None if fresh_graph_evidence else get_dockerhub_tag_cache(
repo_name, since, limit, platform_variant, return_status=True,
)
if cached is not None:
return output(cached[0], cached[1])
rate_limit_state = dockerhub_tag_rate_limit_state()
if rate_limit_state['active']:
logger.info(f"Docker Hub tag API is rate-limited; deferring tag fetch for {repo_name} from cache state")
return output(
[], 'global_cooldown', remote_attempted=False,
retry_at=rate_limit_state['retry_at'],
error='Docker Hub shared rate-limit cooldown is active',
)
if '/' in repo_name:
namespace, name = repo_name.split('/', 1)
else:
namespace, name = 'library', repo_name
url = (
f"https://hub.docker.com/v2/namespaces/{quote(namespace, safe='')}"
f"/repositories/{quote(name, safe='')}/tags"
)
last_error = None
for attempt in range(retry_count + 1):
_raise_if_scan_slot_fatal()
try:
if lease_renewal_callback is not None:
if not callable(lease_renewal_callback):
raise ValueError('Docker resolver lease renewal callback is invalid')
if lease_renewal_callback() is False:
raise DockerResolverLeaseLostError(
'Docker resolver lease renewal was rejected'
)
response = dockerhub_tags_response(
url, {
'page_size': max(
1, min(max(limit, int(platform_candidate_tags or 0)), 100),
),
},
)
if response.status_code == 404:
cache_write(
repo_name, since, limit, 'not_found', [],
getattr(scan_config, 'dockerhub_tag_negative_cache_ttl_sec', 3600),
'Docker Hub repository not found', platform_variant,
)
return output([], 'not_found', remote_attempted=True)
if response.status_code in (401, 403):
raise DockerRemoteAccessError(
f'Docker Hub tags endpoint returned HTTP {response.status_code}',
status='auth_failed', remote_attempted=True,
)
response.raise_for_status()
graph_candidates = []
supported_or_unknown = 0
unsupported = 0
unresolved_digest = 0
resolution_status = ''
resolution_retry_at = None
resolution_error = ''
tag_payload = _bounded_docker_registry_json(
response, 'Docker Hub tags response',
)
if not isinstance(tag_payload, dict) or not isinstance(tag_payload.get('results'), list):
raise DockerRegistryResolutionError('Docker Hub tags response is malformed')
registry_auth = DockerRegistryAuth(token='')
for source_index, tag in enumerate(tag_payload.get('results', [])):
_raise_if_scan_slot_fatal()
tag_name = tag.get('name')
if not tag_name:
continue
revision = str(tag.get('last_updated') or tag.get('tag_last_pushed') or '').strip()
tag_updated = parse_dockerhub_datetime(
revision
)
if since and tag_updated and tag_updated < since:
continue
platform_support = docker_tag_platform_support(tag, platform_os, platform_arch) if platform_filter_enabled else True
if platform_support is False:
unsupported += 1
logger.info(
f"Skipping Docker tag {repo_name}:{tag_name}: no {platform_os}/{platform_arch} image"
)
continue
supported_or_unknown += 1
tagged_image = f"{repo_name}:{tag_name}"
try:
validate_docker_image_reference(tagged_image, require_digest=False)
except ValueError:
logger.warning('Skipping invalid Docker Hub image reference for %s tag %s', repo_name, tag_name)
continue
digest = dockerhub_tag_digest(tag, platform_filter_enabled, platform_os, platform_arch)
if not digest:
unresolved_digest += 1
logger.warning('Deferring Docker tag without a valid content digest: %s', tagged_image)
continue
try:
graph_kwargs = (
{'include_descriptors': True} if fresh_graph_evidence else {}
)
if lease_renewal_callback is not None:
graph_kwargs['lease_renewal_callback'] = lease_renewal_callback
graph, registry_auth = resolve_docker_layer_graph(
repo_name, digest, platform_os, platform_arch, registry_auth,
**graph_kwargs,
)
except DockerResolverLeaseLostError:
raise
except ScanSlotFatalError:
raise
except DockerRemoteAccessError as exc:
unresolved_digest += 1
resolution_status = exc.status
resolution_retry_at = exc.retry_at
resolution_error = str(exc)
logger.warning(
'Deferring Docker manifest graph for %s: %s',
tagged_image, str(exc)[:300],
)
break
except Exception as exc:
unresolved_digest += 1
logger.warning(
'Deferring Docker manifest graph for %s: %s', tagged_image, str(exc)[:300],
)
if isinstance(exc, ApiRequestError) or 'rate-limit' in str(exc).lower():
break
continue
if graph is None:
unsupported += 1
continue
target = validate_docker_image_reference(
f"{repo_name}@{graph['manifest_digest']}"
)
graph_candidates.append({
'name': tag_name,
'target': target,
'repository': repo_name.lower(),
'manifest_digest': graph['manifest_digest'],
**({
'manifest_media_type': graph['manifest_media_type'],
'manifest_size_bytes': graph['manifest_size_bytes'],
'config_digest': graph['config_digest'],
'layer_descriptors': graph['layer_descriptors'],
} if fresh_graph_evidence else {}),
'layers': graph['layers'],
'source_index': source_index,
'updated_at': (
tag_updated.replace(tzinfo=timezone.utc).timestamp()
if tag_updated is not None and tag_updated.tzinfo is None
else tag_updated.timestamp() if tag_updated is not None else None
),
})
candidate_distinct_graph_count = len({
tuple(candidate['layers']) for candidate in graph_candidates
})
candidate_records = (
select_docker_layer_graphs(
graph_candidates,
min(100, candidate_distinct_graph_count),
replacement_pool=True,
)
if fresh_graph_evidence and candidate_distinct_graph_count
else ()
)
selected = (
candidate_records[:limit]
if fresh_graph_evidence
else select_docker_layer_graphs(graph_candidates, limit)
)
if fresh_graph_evidence:
candidate_records = [
_docker_depth_selection_evidence(
record, candidate_distinct_graph_count, selector_version,
)
for record in candidate_records
]
selected = candidate_records[:limit]
tags = [record['target'] for record in selected]
ttl = getattr(scan_config, 'dockerhub_tag_cache_ttl_sec', 21600) if tags else getattr(scan_config, 'dockerhub_tag_negative_cache_ttl_sec', 3600)
status = (
'partial' if tags and unresolved_digest
else resolution_status if resolution_status
else 'ok' if tags
else 'unknown' if unresolved_digest
else 'unsupported' if unsupported and not graph_candidates
else 'empty'
)
if status != 'unknown':
if status == 'ok':
cache_write(
repo_name, since, limit, status, tags, ttl,
platform_variant=platform_variant, tag_records=selected,
)
elif status not in ('partial',):
cache_write(
repo_name, since, limit, status, tags, ttl,
platform_variant=platform_variant,
)
return output(
tags, status, remote_attempted=True,
retry_at=resolution_retry_at, error=resolution_error,
selection_records=selected, candidate_records=candidate_records,
candidate_distinct_graph_count=candidate_distinct_graph_count,
)
except DockerResolverLeaseLostError:
raise
except ScanSlotFatalError:
raise
except DockerRemoteAccessError as exc:
logger.warning('Deferring Docker tag resolution for %s: %s', repo_name, str(exc))
return output(
[], exc.status, remote_attempted=exc.remote_attempted,
retry_at=exc.retry_at, error=str(exc),
)
except Exception as e:
last_error = str(e)
if '404' in last_error or 'not found' in last_error.lower():
cache_write(repo_name, since, limit, 'not_found', [], getattr(scan_config, 'dockerhub_tag_negative_cache_ttl_sec', 3600), last_error, platform_variant)
return output([], 'not_found', remote_attempted=True)
if attempt < retry_count and isinstance(e, ApiRequestError):
if lease_renewal_callback is not None and lease_renewal_callback() is False:
raise DockerResolverLeaseLostError(
'Docker resolver lease renewal was rejected'
)
_wait_or_raise_scan_slot_fatal(min(300, retry_delay * (attempt + 1)))
continue
break
logger.warning(f"Unable to fetch Docker Hub tags for {repo_name}: {last_error}")
return output(
[], 'unknown', remote_attempted=True,
error=last_error or 'Docker tag resolution failed',
)
def resolve_recent_dockerhub_image(
image, since, platform_filter_enabled=False, platform_os='linux',
platform_arch='amd64', platform_candidate_tags=20,
images_per_repository=1, resolve_tags=True,
):
repo_name = image.get('repo_name')
if not repo_name:
return [], 'missing_date'
last_updated = parse_dockerhub_datetime(
image.get('last_updated') or image.get('last_modified')
)
if last_updated and last_updated < since:
return [], 'old'
if last_updated is None:
last_updated = fetch_dockerhub_last_updated(repo_name)
if last_updated is None:
return [repo_name], 'recent'
if last_updated < since:
return [], 'old'
if not resolve_tags:
return [repo_name], 'recent'
tags, tag_status = fetch_dockerhub_tags(
repo_name, since=since,
limit=docker_images_per_repository_limit(images_per_repository),
platform_filter_enabled=platform_filter_enabled,
platform_os=platform_os,
platform_arch=platform_arch,
platform_candidate_tags=platform_candidate_tags,
return_status=True,
)
if tags:
if tag_status != 'ok':
tags.append(repo_name)
return tags, 'recent'
if not docker_tag_resolution_is_conclusive(tag_status):
return [repo_name], 'recent'
return [], 'missing_date'
def fetch_recent_dockerhub_images(
query, since, per_page=100, pages=1, platform_filter_enabled=False,
platform_os='linux', platform_arch='amd64', platform_candidate_tags=20,
images_per_repository=1, resolve_tags=True,
):
"""Fetch Docker Hub images updated since a specific timestamp"""
images = []
page = 1
per_page = max(1, int(per_page or 1))
requested_pages, pages = dockerhub_search_page_window(pages)
if requested_pages > pages:
logger.info(
f'Docker Hub search is limited to {pages} accessible page(s); '
f'capping requested pages from {requested_pages}'
)
logger.info(f"Fetching recent Docker Hub images updated since {since.strftime('%Y-%m-%d')}...")
expected_pages = pages
while page <= expected_pages:
try:
logger.info(f"Docker Hub query '{query}': fetching page {page}/{pages}...")
page_result = fetch_dockerhub_search_page(
query, page, per_page=per_page, request_timeout=30,
)
if page == 1:
expected_pages = min(
pages,
max(1, (page_result['total_count'] + per_page - 1) // per_page),
)
repositories = page_result['repositories']
if not repositories:
logger.info(
f"Docker Hub query '{query}' returned no results "
f"(total matches: {page_result['total_count']})."
)
break
logger.info(f"Page {page}: checking dates/tags for {len(repositories)} Docker Hub repositories...")
new_images = []
old_images = 0
missing_dates = 0
with concurrent.futures.ThreadPoolExecutor(max_workers=min(8, len(repositories))) as executor:
futures = [
executor.submit(
resolve_recent_dockerhub_image, image, since,
platform_filter_enabled, platform_os, platform_arch, platform_candidate_tags,
docker_images_per_repository_limit(images_per_repository),
resolve_tags,
)
for image in repositories
]
for future in concurrent.futures.as_completed(futures):
tags, status = future.result()
if status == 'recent':
new_images.extend(tags)
elif status == 'old':
old_images += 1
else:
missing_dates += 1
images.extend(new_images)
logger.info(f"Page {page}: fetched {len(new_images)} recent images, skipped {old_images} older images, skipped {missing_dates} without dates")
page += 1
except DockerHubDiscoveryTransportError:
raise
except Exception as e:
logger.error(f"Docker Hub recent discovery page {page} failed after bounded attempts")
raise DockerHubDiscoveryTransportError(
f'Docker Hub recent discovery page {page} failed after bounded attempts'
) from e
return images
def huggingface_space_to_target(space):
target = {
'url': space.get('id') or space.get('name') or '',
'name': space.get('id') or space.get('name') or '',
'created_at': space.get('createdAt') or space.get('created_at') or '',
'updated_at': space.get('lastModified') or space.get('updatedAt') or space.get('updated_at') or '',
}
for field in ('private', 'protected', 'gated', 'disabled'):
if field in space:
target[field] = space[field]
return target
def fetch_huggingface_spaces(pages=1, token=None, request_timeout=15, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, return_metadata=False, request_attempts=1, retry_delay=0):
spaces = []
headers = {'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json'}
if token:
headers['Authorization'] = f'Bearer {token}'
seen_pages = 0
request_attempts = max(1, int(request_attempts or 1))
retry_delay = max(0, int(retry_delay or 0))
request_budget = float(request_timeout) * request_attempts + retry_delay * (request_attempts - 1)
if return_metadata:
next_url = 'https://huggingface.co/api/spaces'
next_params = {'sort': 'lastModified', 'direction': '-1', 'limit': 100}
logger.info(f"Fetching newest-modified HuggingFace Spaces for {pages} page(s)...")
for page in range(max(1, int(pages or 1))):
try:
response = api_request(
'GET', next_url, headers=headers, params=next_params,
timeout=request_timeout,
max_retries=request_attempts, retry_delay=retry_delay,
deadline=time.monotonic() + request_budget,
)
if response.status_code >= 400:
status = response.status_code
category = 'auth_invalid' if status == 401 else 'auth_forbidden' if status == 403 else 'rate_limit' if status == 429 else 'server_error' if status >= 500 else 'api'
reset_at = retry_after_reset(response)
response.close()
raise RateLimitError(
'huggingface', f'HuggingFace discovery HTTP {status}',
reset_at=reset_at, category=category,
auth_related=status in (401, 403, 429),
)
response.raise_for_status()
data = response.json()
if not isinstance(data, list):
raise ValueError('invalid HuggingFace spaces payload')
page_spaces = [
huggingface_space_to_target(space) for space in data
if isinstance(space, dict) and space.get('id')
]
if not page_spaces:
logger.info(f"HuggingFace page {page}: no spaces returned. Stopping.")
break
spaces.extend(page_spaces)
logger.info(f"HuggingFace page {page}: fetched {len(page_spaces)} newest-modified spaces")
next_link = (getattr(response, 'links', {}) or {}).get('next') or {}
next_url = str(next_link.get('url') or '')
next_params = None
if not next_url:
break
except Exception as e:
if isinstance(e, (ApiRequestError, RateLimitError)):
raise
logger.error(f"Error fetching HuggingFace page {page}: {str(e)}")
raise ApiRequestError(f'HuggingFace discovery payload failed: {e}') from e
return spaces
logger.info(f"Fetching HuggingFace Spaces for {pages} page(s)...")
for page in range(max(1, int(pages or 1))):
url = 'https://huggingface.co/spaces-json'
params = {'p': page, 'withCount': 'false', 'sort': 'created'}
try:
response = api_request(
'GET', url, headers=headers, params=params, timeout=request_timeout,
max_retries=request_attempts, retry_delay=retry_delay,
deadline=time.monotonic() + request_budget,
)
if response.status_code >= 400:
status = response.status_code
category = 'auth_invalid' if status == 401 else 'auth_forbidden' if status == 403 else 'rate_limit' if status == 429 else 'server_error' if status >= 500 else 'api'
reset_at = retry_after_reset(response)
response.close()
raise RateLimitError(
'huggingface', f'HuggingFace discovery HTTP {status}',
reset_at=reset_at, category=category,
auth_related=status in (401, 403, 429),
)
response.raise_for_status()
data = response.json()
if not isinstance(data, dict) or 'spaces' not in data or not isinstance(data.get('spaces'), list):
raise ValueError('invalid HuggingFace spaces payload')
page_spaces = [huggingface_space_to_target(space) for space in data.get('spaces', []) if space.get('id')]
if not page_spaces:
logger.info(f"HuggingFace page {page}: no spaces returned. Stopping.")
break
spaces.extend(item['url'] for item in page_spaces if item.get('url'))
logger.info(f"HuggingFace page {page}: fetched {len(page_spaces)} spaces")
page_number = page + 1
if stop_on_seen_pages and page_number >= max(1, min_pages_before_stop):
if page_is_known(
[item.get('url') for item in page_spaces], known_targets,
normalize_target, known_target_lookup,
):
seen_pages += 1
logger.info(f"HuggingFace page {page}: all spaces are already queued/checked ({seen_pages}/{seen_page_threshold})")
if seen_pages >= max(1, seen_page_threshold):
logger.info(f"Stopping HuggingFace pagination early after {seen_pages} all-known page(s)")
break
else:
seen_pages = 0
except Exception as e:
if isinstance(e, (ApiRequestError, RateLimitError)):
raise
logger.error(f"Error fetching HuggingFace page {page}: {str(e)}")
raise ApiRequestError(f'HuggingFace discovery payload failed: {e}') from e
return spaces
# =====================
# NPM FETCH FUNCTIONS
# =====================
def parse_iso_datetime(value):
if not value:
return None
try:
parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00'))
if parsed.tzinfo is None:
parsed = parsed.replace(tzinfo=timezone.utc)
return parsed.astimezone(timezone.utc)
except ValueError:
return None
def npm_package_target(name, version, tarball_url, date=None):
return json.dumps({
'name': name,
'version': version,
'tarball': tarball_url,
'date': date or '',
}, separators=(',', ':'), ensure_ascii=False)
def parse_npm_target(target):
if isinstance(target, dict):
return target
target = str(target).strip()
if target.startswith('{'):
return json.loads(target)
package_id, tarball = target.split('|', 1)
name, version = package_id.rsplit('@', 1)
return {'name': name, 'version': version, 'tarball': tarball, 'date': ''}
def npm_package_id(target):
data = parse_npm_target(target)
return f"npm:{data.get('name')}@{data.get('version')}"
def select_npm_release_targets(name, data, max_versions=1, cutoff=None):
targets = []
version_times = data.get('time') or {}
for version, version_data in (data.get('versions') or {}).items():
tarball = (version_data.get('dist') or {}).get('tarball')
if not tarball:
continue
version_date = version_times.get(version) or ''
parsed_date = parse_iso_datetime(version_date)
if cutoff and (not parsed_date or parsed_date < cutoff):
continue
targets.append({
'name': name,
'version': version,
'tarball': tarball,
'date': version_date,
'parsed_date': parsed_date or datetime.min.replace(tzinfo=timezone.utc),
})
targets.sort(key=lambda item: item['parsed_date'], reverse=True)
return targets[:max(1, int(max_versions or 1))]
def fetch_npm_packages(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1, repo_candidate_callback=None):
"""Fetch npm package tarball targets for recent versions matching a query."""
if not query:
return []
targets = []
seen_packages = set()
seen_versions = set()
cutoff = None
if max_version_age_days and max_version_age_days > 0:
cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days)
logger.info(f"Fetching npm packages for query: '{query}'...")
for page in range(max(1, pages)):
params = {'text': query, 'size': per_page, 'from': page * per_page}
try:
response = api_request(
'GET',
'https://registry.npmjs.org/-/v1/search',
params=params,
headers={'User-Agent': 'GitSecretsScanner/2.0'},
timeout=request_timeout,
)
response.raise_for_status()
payload = response.json()
if not isinstance(payload, dict) or 'objects' not in payload or not isinstance(payload.get('objects'), list):
raise ValueError('invalid npm search payload')
objects = payload.get('objects', [])
if objects and not any(isinstance(item, dict) and isinstance(item.get('package'), dict) and item['package'].get('name') for item in objects):
raise ValueError('npm search payload contains no valid package entries')
if not objects:
logger.info(f"npm page {page + 1}: no results")
break
logger.info(f"npm page {page + 1}: fetched {len(objects)} packages")
for item in objects:
package = item.get('package', {})
name = package.get('name')
version = package.get('version')
if not name or not version:
continue
package_name_key = name.lower()
if package_name_key in seen_packages:
continue
metadata_url = f"https://registry.npmjs.org/{quote(name, safe='')}"
metadata = api_request(
'GET',
metadata_url,
headers={'User-Agent': 'GitSecretsScanner/2.0'},
timeout=request_timeout,
)
metadata.raise_for_status()
data = metadata.json()
selected_versions = select_npm_release_targets(name, data, versions_per_package, cutoff)
if repo_candidate_callback:
repo_candidates = []
for version_item in selected_versions or [{'version': version}]:
item_version = version_item.get('version') or version
for candidate in extract_npm_git_candidates(data, item_version):
repo_candidates.append({
'package_source': 'npm',
'name': name,
'version': item_version,
'repo_url': candidate['repo_url'],
'provider': candidate['provider'],
'evidence': candidate.get('evidence') or [],
'confidence': 'high',
})
repo_candidate_callback(repo_candidates)
for item in selected_versions:
package_key = f"{item['name']}@{item['version']}".lower()
if package_key in seen_versions:
continue
targets.append(npm_package_target(item['name'], item['version'], item['tarball'], item['date']))
seen_versions.add(package_key)
seen_packages.add(package_name_key)
except Exception as e:
if isinstance(e, ApiRequestError):
raise
logger.error(f"Error fetching npm page {page + 1}: {str(e)}")
raise ApiRequestError(f'npm discovery failed: {e}') from e
logger.info(f"npm query '{query}': prepared {len(targets)} package targets")
return targets
# ======================
# PYPI FETCH FUNCTIONS
# ======================
pypi_project_index_cache_path = None
def pypi_package_target(name, version, artifact_url, date=None, filename=None, packagetype=None, size=None):
return json.dumps({
'source': 'pypi',
'name': name,
'version': version,
'artifact': artifact_url,
'date': date or '',
'filename': filename or '',
'packagetype': packagetype or '',
'size': size or 0,
}, separators=(',', ':'), ensure_ascii=False)
def parse_pypi_target(target):
if isinstance(target, dict):
return target
target = str(target).strip()
if target.startswith('{'):
return json.loads(target)
package_id, artifact = target.split('|', 1)
name, version = package_id.rsplit('@', 1)
return {'source': 'pypi', 'name': name, 'version': version, 'artifact': artifact, 'date': ''}
def pypi_package_id(target):
data = parse_pypi_target(target)
return f"pypi:{data.get('name')}@{data.get('version')}"
# =============================
# PACKAGE -> GIT FETCH HELPERS
# =============================
GIT_PATH_STOP_SEGMENTS = {
'-', 'issues', 'issue', 'pull', 'pulls', 'merge_requests', 'merge_request',
'tree', 'blob', 'commit', 'commits', 'releases', 'tags', 'branches', 'wiki',
}
def canonical_git_path_parts(host, parts):
if host == 'github.com':
if len(parts) < 2:
return []
return parts[:2]
cleaned = []
for part in parts:
lowered = part.lower()
if lowered in GIT_PATH_STOP_SEGMENTS:
break
cleaned.append(part)
if len(cleaned) < 2:
return []
return cleaned
def normalize_git_repo_candidate(value):
if not value:
return None
raw = str(value).strip().strip('"\'')
if not raw:
return None
if raw.startswith('git+'):
raw = raw[4:]
if raw.startswith('github:'):
raw = 'https://github.com/' + raw.split(':', 1)[1]
elif raw.startswith('gitlab:'):
raw = 'https://gitlab.com/' + raw.split(':', 1)[1]
elif raw.startswith('git@github.com:'):
raw = 'https://github.com/' + raw.split(':', 1)[1]
elif raw.startswith('git@gitlab.com:'):
raw = 'https://gitlab.com/' + raw.split(':', 1)[1]
elif raw.startswith('git://'):
raw = 'https://' + raw[6:]
try:
parsed = urlsplit(raw)
except ValueError:
return None
if parsed.scheme not in ('http', 'https') or not parsed.netloc:
return None
host = (parsed.hostname or '').lower()
if host in ('www.github.com',):
host = 'github.com'
if host in ('www.gitlab.com',):
host = 'gitlab.com'
if host not in ('github.com', 'gitlab.com'):
return None
parts = [part for part in parsed.path.strip('/').split('/') if part]
repo_parts = canonical_git_path_parts(host, parts)
if not repo_parts:
return None
repo_parts[-1] = repo_parts[-1][:-4] if repo_parts[-1].endswith('.git') else repo_parts[-1]
if any(not part for part in repo_parts):
return None
provider = 'github' if host == 'github.com' else 'gitlab'
repo_path = '/'.join(repo_parts)
return {
'provider': provider,
'repo_url': f'https://{host}/{repo_path}.git',
'repo_path': repo_path,
}
def package_git_target(package_source, name, version, repo_url, provider, evidence=None, confidence='medium'):
return json.dumps({
'source': 'package_git',
'package_source': package_source,
'name': name,
'version': version or '',
'repo_url': repo_url,
'provider': provider,
'evidence': evidence or [],
'confidence': confidence,
}, separators=(',', ':'), ensure_ascii=False)
def parse_package_git_target(target):
if isinstance(target, dict):
return target
text = str(target).strip()
if text.startswith('{'):
return json.loads(text)
candidate = normalize_git_repo_candidate(text)
if not candidate:
raise ValueError(f'Unsupported package_git target: {text}')
return {
'source': 'package_git',
'package_source': 'custom',
'name': '',
'version': '',
'repo_url': candidate['repo_url'],
'provider': candidate['provider'],
'evidence': ['custom'],
'confidence': 'high',
}
def package_git_id(target):
data = parse_package_git_target(target)
return f"package_git:{data.get('provider')}:{data.get('repo_url')}".lower()
def collect_git_candidates(values):
candidates = []
seen = set()
for evidence, value in values:
candidate = normalize_git_repo_candidate(value)
if not candidate:
continue
key = candidate['repo_url'].lower()
if key in seen:
continue
seen.add(key)
candidate['evidence'] = [evidence]
candidates.append(candidate)
return candidates
GIT_URL_TEXT_RE = re.compile(
r'(?:https?://|git\+https?://|git://|git@)(?:github\.com[:/]|gitlab\.com[:/])'
r'[A-Za-z0-9_.-]+(?:/[A-Za-z0-9_.-]+){1,8}(?:\.git)?(?:/[A-Za-z0-9_.~/-]+)?',
re.IGNORECASE,
)
def git_candidates_from_text(label, text, max_urls=8):
if not text:
return []
values = []
seen = set()
for match in GIT_URL_TEXT_RE.finditer(str(text)):
raw = match.group(0).rstrip(').,;\'"<>')
if raw.startswith('git@github.com/'):
raw = raw.replace('git@github.com/', 'git@github.com:', 1)
if raw.startswith('git@gitlab.com/'):
raw = raw.replace('git@gitlab.com/', 'git@gitlab.com:', 1)
key = raw.lower()
if key in seen:
continue
seen.add(key)
values.append((label, raw))
if len(values) >= max_urls:
break
return collect_git_candidates(values)
def extract_npm_git_candidates(metadata, version=None):
values = []
repository = metadata.get('repository')
if isinstance(repository, dict):
values.append(('repository.url', repository.get('url')))
elif isinstance(repository, str):
values.append(('repository', repository))
bugs = metadata.get('bugs')
if isinstance(bugs, dict):
values.append(('bugs.url', bugs.get('url')))
values.append(('homepage', metadata.get('homepage')))
version_data = (metadata.get('versions') or {}).get(version or '', {})
version_repository = version_data.get('repository') if isinstance(version_data, dict) else None
if isinstance(version_repository, dict):
values.append(('version.repository.url', version_repository.get('url')))
elif isinstance(version_repository, str):
values.append(('version.repository', version_repository))
if isinstance(version_data, dict):
version_bugs = version_data.get('bugs')
if isinstance(version_bugs, dict):
values.append(('version.bugs.url', version_bugs.get('url')))
values.append(('version.homepage', version_data.get('homepage')))
candidates = collect_git_candidates(values)
candidates.extend(git_candidates_from_text('readme.github_url', metadata.get('readme')))
candidates.extend(git_candidates_from_text('description.github_url', metadata.get('description')))
deduped = []
seen = set()
for candidate in candidates:
key = candidate['repo_url'].lower()
if key in seen:
continue
seen.add(key)
deduped.append(candidate)
return deduped
def extract_pypi_git_candidates(metadata):
info = metadata.get('info') or {}
values = []
project_urls = info.get('project_urls') or {}
if isinstance(project_urls, dict):
for key, value in project_urls.items():
label = str(key).lower()
if any(item in label for item in ('source', 'repository', 'repo', 'code', 'homepage', 'home', 'bug', 'issue', 'tracker')):
values.append((f'project_urls.{key}', value))
values.append(('home_page', info.get('home_page')))
values.append(('project_url', info.get('project_url')))
candidates = collect_git_candidates(values)
candidates.extend(git_candidates_from_text('description.github_url', info.get('description')))
candidates.extend(git_candidates_from_text('summary.github_url', info.get('summary')))
deduped = []
seen = set()
for candidate in candidates:
key = candidate['repo_url'].lower()
if key in seen:
continue
seen.add(key)
deduped.append(candidate)
return deduped
def fetch_npm_package_git_repos(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1):
if not query:
return []
targets = []
seen_repos = set()
cutoff = None
if max_version_age_days and max_version_age_days > 0:
cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days)
logger.info(f"Fetching npm package git repos for query: '{query}'...")
for page in range(max(1, pages)):
params = {'text': query, 'size': per_page, 'from': page * per_page}
try:
response = api_request(
'GET',
'https://registry.npmjs.org/-/v1/search',
params=params,
headers={'User-Agent': 'GitSecretsScanner/2.0'},
timeout=request_timeout,
)
response.raise_for_status()
payload = response.json()
if not isinstance(payload, dict) or 'objects' not in payload or not isinstance(payload.get('objects'), list):
raise ValueError('invalid npm package_git search payload')
objects = payload.get('objects', [])
if objects and not any(isinstance(item, dict) and isinstance(item.get('package'), dict) and item['package'].get('name') for item in objects):
raise ValueError('npm package_git payload contains no valid package entries')
if not objects:
logger.info(f"npm package_git page {page + 1}: no results")
break
logger.info(f"npm package_git page {page + 1}: fetched {len(objects)} packages")
for item in objects:
package = item.get('package', {})
name = package.get('name')
if not name:
continue
metadata = api_request(
'GET',
f"https://registry.npmjs.org/{quote(name, safe='')}",
headers={'User-Agent': 'GitSecretsScanner/2.0'},
timeout=request_timeout,
)
metadata.raise_for_status()
data = metadata.json()
selected_versions = select_npm_release_targets(name, data, versions_per_package, cutoff) or [{'version': package.get('version') or ''}]
for version_item in selected_versions:
version = version_item.get('version') or package.get('version') or ''
for candidate in extract_npm_git_candidates(data, version):
key = candidate['repo_url'].lower()
if key in seen_repos:
continue
seen_repos.add(key)
targets.append(package_git_target('npm', name, version, candidate['repo_url'], candidate['provider'], candidate['evidence'], 'high'))
except Exception as e:
if isinstance(e, ApiRequestError):
raise
logger.error(f"Error fetching npm package_git page {page + 1}: {str(e)}")
raise ApiRequestError(f'npm package_git discovery failed: {e}') from e
logger.info(f"npm package_git query '{query}': prepared {len(targets)} git repo targets")
return targets
def fetch_pypi_package_git_repos(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1):
if not query:
return []
targets = []
seen_repos = set()
cutoff = None
if max_version_age_days and max_version_age_days > 0:
cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days)
logger.info(f"Fetching PyPI package git repos for query: '{query}'...")
for page in range(1, max(1, pages) + 1):
try:
names = fetch_pypi_package_names(query, page, per_page, request_timeout)
if not names:
logger.info(f"PyPI package_git page {page}: no results")
break
logger.info(f"PyPI package_git page {page}: fetched {len(names)} package names")
for name in names:
try:
metadata = api_request(
'GET',
f"https://pypi.org/pypi/{quote(name, safe='')}/json",
headers={'User-Agent': 'GitSecretsScanner/2.0'},
timeout=request_timeout,
)
metadata.raise_for_status()
data = metadata.json()
except requests.exceptions.HTTPError as e:
if e.response is not None and e.response.status_code == 404:
logger.info(f"Skipping PyPI package_git project {name}: metadata not found")
continue
raise ApiRequestError(f'PyPI package_git metadata failed for {name}: {e}') from e
except ApiRequestError:
raise
if not isinstance(data, dict) or not isinstance(data.get('info', {}), dict):
raise ApiRequestError(f'invalid PyPI package_git metadata payload for {name}')
release_files = select_pypi_release_files(data, cutoff, versions_per_package)
version = release_files[0]['version'] if release_files else (data.get('info') or {}).get('version') or ''
package_name = (data.get('info') or {}).get('name') or name
for candidate in extract_pypi_git_candidates(data):
key = candidate['repo_url'].lower()
if key in seen_repos:
continue
seen_repos.add(key)
targets.append(package_git_target('pypi', package_name, version, candidate['repo_url'], candidate['provider'], candidate['evidence'], 'medium'))
except Exception as e:
if isinstance(e, ApiRequestError):
raise
logger.error(f"Error fetching PyPI package_git page {page}: {str(e)}")
raise ApiRequestError(f'PyPI package_git discovery failed: {e}') from e
logger.info(f"PyPI package_git query '{query}': prepared {len(targets)} git repo targets")
return targets
class _PyPIProjectParser(HTMLParser):
def __init__(self, sink):
super().__init__(convert_charrefs=True)
self.sink = sink
self.in_anchor = False
self.parts = []
def handle_starttag(self, tag, attrs):
if tag.lower() == 'a':
self.in_anchor = True
self.parts = []
def handle_endtag(self, tag):
if tag.lower() == 'a':
name = ''.join(self.parts).strip()
if name:
self.sink(name)
self.in_anchor = False
self.parts = []
def handle_data(self, data):
if self.in_anchor:
text = str(data or '')
if sum(len(part) for part in self.parts) + len(text) <= 512:
self.parts.append(text)
def _pypi_index_path():
global pypi_project_index_cache_path
if pypi_project_index_cache_path:
return pypi_project_index_cache_path
state_dir = os.path.dirname(scan_limiter_db_path())
require_private_directory(state_dir, create=False)
pypi_project_index_cache_path = os.path.join(state_dir, 'pypi_project_index.sqlite3')
return pypi_project_index_cache_path
def load_pypi_project_index(request_timeout=20):
path = _pypi_index_path()
refresh_sec = max(3600, int(os.getenv('PYPI_PROJECT_INDEX_REFRESH_SEC', '86400')))
if not os.path.exists(path):
descriptor = os.open(
path, os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0), 0o600,
)
os.close(descriptor)
harden_private_file(path)
connection = sqlite3.connect(path, timeout=30)
try:
connection.execute('PRAGMA journal_mode=DELETE')
connection.execute('PRAGMA synchronous=FULL')
connection.executescript('''
CREATE TABLE IF NOT EXISTS pypi_projects (
normalized_name TEXT PRIMARY KEY,
name TEXT NOT NULL
);
CREATE TABLE IF NOT EXISTS pypi_index_meta (
id INTEGER PRIMARY KEY CHECK(id = 1),
refreshed_at REAL NOT NULL,
project_count INTEGER NOT NULL
);
''')
current = connection.execute(
'SELECT refreshed_at, project_count FROM pypi_index_meta WHERE id = 1'
).fetchone()
if current and time.time() - float(current[0]) < refresh_sec and int(current[1]) > 0:
return path
response = _direct_request(
'GET', 'https://pypi.org/simple/',
headers={'Accept': 'text/html', 'User-Agent': 'GitSecretsScanner/2.0'},
timeout=request_timeout,
stream=True,
)
response.raise_for_status()
connection.execute('BEGIN IMMEDIATE')
connection.execute('DELETE FROM pypi_projects')
batch = []
count = 0
def accept(name):
nonlocal count
normalized = re.sub(r'[-_.]+', '-', name).lower()
if not normalized or len(normalized) > 512:
return
batch.append((normalized, name[:512]))
if len(batch) >= 1000:
connection.executemany(
'INSERT OR IGNORE INTO pypi_projects(normalized_name, name) VALUES (?, ?)',
batch,
)
count += len(batch)
batch.clear()
parser = _PyPIProjectParser(accept)
decoder = codecs.getincrementaldecoder('utf-8')('strict')
response_bytes = 0
response_max_bytes = max(
1024 * 1024, int(os.getenv('PYPI_PROJECT_INDEX_MAX_BYTES', str(512 * 1024 * 1024))),
)
try:
for chunk in response.iter_content(chunk_size=256 * 1024):
_raise_if_scan_slot_fatal()
if chunk:
response_bytes += len(chunk)
if response_bytes > response_max_bytes:
raise ApiRequestError('PyPI simple index exceeds its streamed byte bound')
parser.feed(decoder.decode(chunk))
parser.feed(decoder.decode(b'', final=True))
parser.close()
finally:
response.close()
if batch:
connection.executemany(
'INSERT OR IGNORE INTO pypi_projects(normalized_name, name) VALUES (?, ?)',
batch,
)
count += len(batch)
actual = int(connection.execute('SELECT COUNT(*) FROM pypi_projects').fetchone()[0])
if actual <= 0:
raise ApiRequestError('PyPI simple index contains no valid project entries')
connection.execute(
'''INSERT INTO pypi_index_meta(id, refreshed_at, project_count) VALUES (1, ?, ?)
ON CONFLICT(id) DO UPDATE SET refreshed_at = excluded.refreshed_at,
project_count = excluded.project_count''',
(time.time(), actual),
)
connection.commit()
logger.info('Streamed %d PyPI project names into the on-disk index', actual)
return path
except Exception:
connection.rollback()
raise
finally:
connection.close()
if os.path.exists(path):
harden_private_file(path)
def pypi_name_rank(name, query):
normalized = name.lower()
query = query.lower()
parts = [part for part in re.split(r'[-_.]+', normalized) if part]
if normalized == query:
rank = 0
elif normalized.startswith(query):
rank = 1
elif query in parts:
rank = 2
else:
rank = 3
return rank, len(normalized), normalized
def fetch_pypi_package_names(query, page=1, per_page=50, request_timeout=20):
query = query.strip().lower()
if not query:
return []
tokens = [token for token in re.split(r'\s+', query) if token]
path = load_pypi_project_index(request_timeout)
page_size = max(1, int(per_page or 50))
requested_end = max(1, int(page)) * page_size
candidate_limit = min(100000, max(1000, requested_end * 20))
connection = sqlite3.connect(f'file:{path.replace(os.sep, "/")}?mode=ro', uri=True, timeout=30)
try:
clauses = ' AND '.join('normalized_name LIKE ?' for _ in tokens)
rows = connection.execute(
f'''SELECT name FROM pypi_projects WHERE {clauses}
ORDER BY normalized_name LIMIT ?''',
(*[f'%{token}%' for token in tokens], candidate_limit),
)
matches = [row[0] for row in rows]
finally:
connection.close()
matches.sort(key=lambda name: pypi_name_rank(name, query))
start = (max(1, int(page)) - 1) * page_size
return matches[start:start + page_size]
def select_pypi_release_files(data, cutoff=None, max_versions=1):
release_candidates = []
priority_by_type = {'sdist': 2, 'bdist_wheel': 1}
for version, files in (data.get('releases') or {}).items():
file_candidates = []
for file_info in files or []:
if file_info.get('yanked'):
continue
artifact_url = file_info.get('url')
if not artifact_url:
continue
uploaded = file_info.get('upload_time_iso_8601') or file_info.get('upload_time') or ''
parsed_date = parse_iso_datetime(uploaded)
if cutoff and (not parsed_date or parsed_date < cutoff):
continue
file_candidates.append({
'version': version,
'url': artifact_url,
'date': uploaded,
'parsed_date': parsed_date or datetime.min.replace(tzinfo=timezone.utc),
'filename': file_info.get('filename') or '',
'packagetype': file_info.get('packagetype') or '',
'size': file_info.get('size') or 0,
'priority': priority_by_type.get(file_info.get('packagetype'), 0),
})
if file_candidates:
release_date = max(item['parsed_date'] for item in file_candidates)
file_candidates.sort(key=lambda item: (item['priority'], item['parsed_date']), reverse=True)
release_candidates.append((release_date, file_candidates[0]))
if not release_candidates:
return []
release_candidates.sort(key=lambda item: item[0], reverse=True)
return [item[1] for item in release_candidates[:max(1, int(max_versions or 1))]]
def select_pypi_release_file(data, cutoff=None):
files = select_pypi_release_files(data, cutoff, 1)
return files[0] if files else None
def fetch_pypi_packages(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1, repo_candidate_callback=None):
"""Fetch PyPI package artifact targets for recent matching releases."""
if not query:
return []
targets = []
seen = set()
cutoff = None
if max_version_age_days and max_version_age_days > 0:
cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days)
logger.info(f"Fetching PyPI packages for query: '{query}'...")
for page in range(1, max(1, pages) + 1):
try:
names = fetch_pypi_package_names(query, page, per_page, request_timeout)
if not names:
logger.info(f"PyPI page {page}: no results")
break
logger.info(f"PyPI page {page}: fetched {len(names)} package names")
for name in names:
normalized_name = name.lower()
if normalized_name in seen:
continue
metadata_url = f"https://pypi.org/pypi/{quote(name, safe='')}/json"
try:
metadata = api_request(
'GET',
metadata_url,
headers={'User-Agent': 'GitSecretsScanner/2.0'},
timeout=request_timeout,
)
metadata.raise_for_status()
data = metadata.json()
except requests.exceptions.HTTPError as e:
if e.response is not None and e.response.status_code == 404:
logger.info(f"Skipping PyPI project {name}: metadata not found")
seen.add(normalized_name)
continue
logger.warning(f"Skipping PyPI project {name}: metadata fetch failed: {str(e)}")
raise ApiRequestError(f'PyPI metadata failed for {name}: {e}') from e
except ApiRequestError:
raise
except Exception as e:
raise ApiRequestError(f'PyPI metadata payload failed for {name}: {e}') from e
if not isinstance(data, dict) or not isinstance(data.get('info', {}), dict):
raise ApiRequestError(f'invalid PyPI metadata payload for {name}')
release_files = select_pypi_release_files(data, cutoff, versions_per_package)
if not release_files:
continue
if repo_candidate_callback:
package_name = data.get('info', {}).get('name') or name
repo_candidates = []
for release_file in release_files:
for candidate in extract_pypi_git_candidates(data):
repo_candidates.append({
'package_source': 'pypi',
'name': package_name,
'version': release_file['version'],
'repo_url': candidate['repo_url'],
'provider': candidate['provider'],
'evidence': candidate.get('evidence') or [],
'confidence': 'medium',
})
repo_candidate_callback(repo_candidates)
for release_file in release_files:
targets.append(pypi_package_target(
data.get('info', {}).get('name') or name,
release_file['version'],
release_file['url'],
release_file['date'],
release_file['filename'],
release_file['packagetype'],
release_file['size'],
))
seen.add(normalized_name)
except Exception as e:
if isinstance(e, ApiRequestError):
raise
logger.error(f"Error fetching PyPI page {page}: {str(e)}")
raise ApiRequestError(f'PyPI discovery failed: {e}') from e
logger.info(f"PyPI query '{query}': prepared {len(targets)} package targets")
return targets
# =====================
# SCANNING FUNCTIONS
# =====================
def check_dependencies():
"""Verify required dependencies are installed"""
probe_dir = None
try:
work_dir = get_work_dir()
probe_dir = tempfile.mkdtemp(prefix='trufflehog-probe-', dir=work_dir)
harden_private_directory(probe_dir)
command = [get_trufflehog_cmd(), '--version', '--no-update']
write_temp_owner(probe_dir, command, os.getpid())
env = os.environ.copy()
strip_supervisor_credentials(env)
env['PATH'] = os.pathsep.join([
os.path.expanduser('~/bin'),
os.path.expanduser('~/.local/bin'),
env.get('PATH', '')
])
prepend_client_git_environment(env)
env['TEMP'] = probe_dir
env['TMP'] = probe_dir
env['TMPDIR'] = probe_dir
require_trufflehog_launch_authority(command)
result = run_owned(
command,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
timeout=30,
env=env,
cwd=probe_dir,
creationflags=subprocess.CREATE_NO_WINDOW if os.name == 'nt' else 0,
)
if result.returncode != 0:
stderr = (result.stderr or b'') if isinstance(result.stderr, bytes) else str(result.stderr or '').encode('utf-8', errors='replace')
detail = stderr.decode('utf-8', errors='replace').strip()[:500]
raise RuntimeError(f"TruffleHog version probe exited with code {result.returncode}: {detail or 'no stderr'}")
logger.info("trufflehog is installed and working")
return True
except Exception as e:
if isinstance(e, subprocess.TimeoutExpired):
detail = f"timed out after {e.timeout}s"
else:
detail = f"{type(e).__name__}: {e}"
logger.error("TruffleHog dependency probe failed: %s", redact_scan_command_text([detail])[:500])
if isinstance(e, FileNotFoundError):
logger.error("TruffleHog executable was not found. Please install it.")
logger.error("Visit https://github.com/trufflesecurity/trufflehog for installation instructions.")
else:
logger.error("TruffleHog dependency probe failed closed; executable authority was not bypassed.")
return False
finally:
if probe_dir:
cleanup_command_work_dir(probe_dir)
def command_output_limits():
if _client_scan_policy.get() is None:
stdout_mb = int_setting(
os.getenv('TRUFFLEHOG_STDOUT_MAX_MB'),
getattr(scan_config, 'trufflehog_stdout_max_mb', 32),
)
stderr_mb = int_setting(
os.getenv('TRUFFLEHOG_STDERR_MAX_MB'),
getattr(scan_config, 'trufflehog_stderr_max_mb', 8),
)
else:
stdout_mb = int(_scan_policy_value('trufflehog_stdout_max_mb', 32))
stderr_mb = int(_scan_policy_value('trufflehog_stderr_max_mb', 8))
requested_stdout = max(1, int_setting(
stdout_mb, 32,
)) * 1024 * 1024
requested_stderr = max(1, int_setting(
stderr_mb, 8,
)) * 1024 * 1024
event_limit = max(1, int(_scan_policy_value(
'result_bundle_max_event_bytes', 64 * 1024 * 1024,
)))
reserve = min(32 * 1024 * 1024, max(64 * 1024, event_limit // 4))
output_budget = max(2, event_limit - reserve)
requested_total = requested_stdout + requested_stderr
if requested_total <= output_budget:
return requested_stdout, requested_stderr
stdout = max(1, (output_budget * requested_stdout) // requested_total)
stderr = max(1, output_budget - stdout)
return stdout, stderr
class CommandOutputLimitError(RuntimeError):
pass
class StreamedCommandOutput:
def __init__(self, stdout_file, stderr_file, returncode, max_stdout, max_stderr, synthetic_stderr=''):
self._stdout = stdout_file
self._stderr = stderr_file
self.returncode = int(returncode)
self.max_stdout = int(max_stdout)
self.max_stderr = int(max_stderr)
self.synthetic_stderr = str(synthetic_stderr or '')
@staticmethod
def _lines(handle, byte_limit, max_line_bytes, max_lines, redactions=()):
handle.seek(0)
consumed = 0
count = 0
while consumed < byte_limit and count < max_lines:
raw = handle.readline(min(max_line_bytes + 1, byte_limit - consumed + 1))
if not raw:
return
consumed += len(raw)
count += 1
if len(raw) > max_line_bytes and not raw.endswith((b'\n', b'\r')):
raise CommandOutputLimitError(f'command output line exceeded {max_line_bytes} bytes')
line = raw.decode('utf-8', errors='replace')
if redactions:
line = redact_secrets(line, redactions)
yield line
if handle.read(1):
raise CommandOutputLimitError('command output exceeded its line or byte bound')
def stdout_lines(self, max_line_bytes=16 * 1024 * 1024, max_lines=20000, redactions=()):
return self._lines(
self._stdout, self.max_stdout, max(1, int(max_line_bytes)),
max(1, int(max_lines)), redactions,
)
def stderr_lines(self, max_line_bytes=8192, max_lines=2000, redactions=()):
for line in self._lines(
self._stderr, self.max_stderr, max(1, int(max_line_bytes)),
max(1, int(max_lines)), redactions,
):
yield line
if self.synthetic_stderr:
yield self.synthetic_stderr.rstrip('\r\n') + '\n'
@staticmethod
def _raw_bytes(handle):
position = handle.tell()
try:
handle.seek(0)
return handle.read()
finally:
handle.seek(position)
def raw_stdout_bytes(self):
return self._raw_bytes(self._stdout)
def raw_stderr_bytes(self):
return self._raw_bytes(self._stderr)
@contextmanager
def streamed_output_from_text(stdout='', stderr='', returncode=0):
stdout_file = io.BytesIO(str(stdout).encode('utf-8'))
stderr_file = io.BytesIO(str(stderr).encode('utf-8'))
yield StreamedCommandOutput(
stdout_file, stderr_file, returncode,
max(1, len(stdout_file.getvalue())), max(1, len(stderr_file.getvalue())),
)
def _check_command_staging(roots, deadline, output_files=()):
"""Best-effort live staging watchdog, not a filesystem quota or atomic snapshot."""
exceeded = 'TruffleHog staging limit exceeded'
unavailable = 'Unable to monitor TruffleHog staging'
total_bytes = 0
entries = 0
try:
unique_roots = []
for root in sorted({os.path.normcase(os.path.abspath(root)) for root in roots}, key=len):
if not any(root == parent or root.startswith(os.path.join(parent, '')) for parent in unique_roots):
unique_roots.append(root)
# Check ancestors too: lstat on a child alone would follow a linked parent.
for root in unique_roots:
ancestor = root
while True:
if time.monotonic() >= deadline:
return unavailable
try:
info = os.lstat(ancestor)
except FileNotFoundError:
pass
else:
if stat.S_ISLNK(info.st_mode) or getattr(info, 'st_file_attributes', 0) & stat.FILE_ATTRIBUTE_REPARSE_POINT:
return unavailable
parent = os.path.dirname(ancestor)
if parent == ancestor:
break
ancestor = parent
# POSIX TemporaryFile output may be unlinked and thus absent from scandir.
for handle in output_files:
if time.monotonic() >= deadline:
return unavailable
info = os.fstat(handle.fileno())
if info.st_nlink == 0:
total_bytes += info.st_size
entries += 1
pending = list(unique_roots)
while pending:
if time.monotonic() >= deadline:
return unavailable
path = pending.pop()
try:
info = os.lstat(path)
if stat.S_ISLNK(info.st_mode) or getattr(info, 'st_file_attributes', 0) & stat.FILE_ATTRIBUTE_REPARSE_POINT:
return unavailable
if stat.S_ISREG(info.st_mode):
total_bytes += info.st_size
elif not stat.S_ISDIR(info.st_mode):
return unavailable
if total_bytes > 2 * 1024 ** 3:
return exceeded
if stat.S_ISDIR(info.st_mode):
with os.scandir(path) as children:
for child in children:
if time.monotonic() >= deadline:
return unavailable
entries += 1
if entries > 100000:
return exceeded
pending.append(child.path)
except FileNotFoundError:
continue
if time.monotonic() >= deadline:
return unavailable
if total_bytes > 2 * 1024 ** 3 or entries > 100000:
return exceeded
except (OSError, ValueError):
return unavailable
return ''
@contextmanager
def run_command_streamed(cmd, timeout_sec, env=None, *, deadline=None, staging_roots=None, native_git_clone=False):
"""Run one owned command; optional staging roots also monitor its private temp tree."""
if type(native_git_clone) is not bool:
raise ValueError('native_git_clone must be an explicit boolean')
if native_git_clone and not staging_roots:
raise ValueError('native Git clone requires private staging roots')
command_work_dir = None
process = None
scan_slot = None
owns_scan_slot = False
release_scan_slot = True
stdout_file = None
stderr_file = None
max_stdout, max_stderr = command_output_limits()
fatal_slot_error = None
propagating_fatal_error = None
returncode = -1
synthetic_stderr = ''
armed_owners = []
command_owner_published = False
requested_timeout = (
max(1, int(timeout_sec or 1))
if deadline is None
else max(0.001, float(timeout_sec or 0.001))
)
command_deadline = time.monotonic() + requested_timeout
if deadline is not None:
deadline = float(deadline)
if not math.isfinite(deadline):
raise ValueError('command deadline must be finite')
command_deadline = min(command_deadline, deadline)
try:
_raise_if_scan_slot_fatal()
env = dict(os.environ if env is None else env)
for key in list(env):
if key.lower() in ('http_proxy', 'https_proxy', 'all_proxy', 'no_proxy'):
del env[key]
env['NO_PROXY'] = '*'
strip_supervisor_credentials(env)
if os.name == 'nt':
env['PATH'] = os.pathsep.join([
os.path.expanduser('~/bin'), os.path.expanduser('~/.local/bin'), env.get('PATH', ''),
])
prepend_client_git_environment(env)
env['GIT_TERMINAL_PROMPT'] = '0'
env['GIT_ASKPASS'] = 'true'
min_free_gb = max(0.0, float(getattr(scan_config, 'min_free_gb', 0) or 0))
min_free_bytes = int(min_free_gb * 1024 * 1024 * 1024)
borrowed_scan_slot, scan_slot = scoped_scan_slot_lease()
if not borrowed_scan_slot:
scan_slot = acquire_scan_slot(
cmd, max(0.001, command_deadline - time.monotonic()),
)
owns_scan_slot = True
if scan_slot and not scan_slot.releasable:
raise RuntimeError('scan slot is fail-closed after unconfirmed child termination')
command_work_dir = create_command_work_dir()
shared_owners = _shared_staging_owners((command_work_dir, *(staging_roots or ())))
staging_roots = (command_work_dir, *staging_roots) if staging_roots is not None else None
env['TEMP'] = command_work_dir
env['TMP'] = command_work_dir
env['TMPDIR'] = command_work_dir
if env.get('TRUF_GIT_TOKEN'):
if os.name == 'nt':
askpass_path = os.path.join(command_work_dir, 'git-askpass.cmd')
with open(askpass_path, 'w', encoding='ascii') as askpass:
askpass.write('@echo off\r\n')
askpass.write('echo %~1 | findstr /I "username" >nul\r\n')
askpass.write('if %errorlevel%==0 (echo %TRUF_GIT_USERNAME%) else (echo %TRUF_GIT_TOKEN%)\r\n')
else:
askpass_path = os.path.join(command_work_dir, 'git-askpass.sh')
with open(askpass_path, 'x', encoding='ascii', newline='\n') as askpass:
askpass.write(
'#!/bin/sh\n'
'case "$1" in\n'
' *[Uu][Ss][Ee][Rr][Nn][Aa][Mm][Ee]*) printf \'%s\\n\' "$TRUF_GIT_USERNAME" ;;\n'
' *) printf \'%s\\n\' "$TRUF_GIT_TOKEN" ;;\n'
'esac\n'
)
harden_private_file(askpass_path)
if os.name != 'nt':
os.chmod(askpass_path, stat.S_IRWXU)
env['GIT_ASKPASS'] = askpass_path
creationflags = (
subprocess.CREATE_NEW_PROCESS_GROUP
| subprocess.CREATE_NO_WINDOW
) if os.name == 'nt' else 0
stdout_file = tempfile.TemporaryFile(mode='w+b', dir=command_work_dir)
stderr_file = tempfile.TemporaryFile(mode='w+b', dir=command_work_dir)
if native_git_clone:
require_git_clone_launch_authority(cmd)
if time.monotonic() >= command_deadline:
raise subprocess.TimeoutExpired(cmd, timeout_sec)
destination_parent = os.path.normcase(os.path.abspath(os.path.dirname(cmd[-1])))
if destination_parent not in {os.path.normcase(os.path.abspath(root)) for root in staging_roots}:
raise RuntimeError('Git clone destination is outside its private staging parent')
staging_error = _check_command_staging(staging_roots, command_deadline, (stdout_file, stderr_file))
if staging_error:
raise RuntimeError(staging_error)
else:
require_trufflehog_launch_authority(cmd)
if time.monotonic() >= command_deadline:
raise subprocess.TimeoutExpired(cmd, timeout_sec)
process_options = {
'stdout': stdout_file, 'stderr': stderr_file, 'env': env,
'cwd': command_work_dir, 'stdin': subprocess.DEVNULL,
'creationflags': creationflags,
}
if os.name == 'nt':
job_memory_limit_bytes = _scan_policy_value(
'trufflehog_job_memory_limit_bytes', 0,
)
if isinstance(job_memory_limit_bytes, bool) or not isinstance(job_memory_limit_bytes, int) or job_memory_limit_bytes <= 0:
raise RuntimeError('trufflehog_job_memory_limit_bytes must be a positive integer on Windows')
process_options['job_memory_limit_bytes'] = job_memory_limit_bytes
job_cpu_weight = int(_scan_policy_value('trufflehog_windows_job_cpu_weight', 0))
memory_priority = int(_scan_policy_value('trufflehog_windows_memory_priority', 0))
if job_cpu_weight < 0 or job_cpu_weight > 9:
raise RuntimeError('trufflehog_windows_job_cpu_weight must be between 0 and 9')
if memory_priority < 0 or memory_priority > 5:
raise RuntimeError('trufflehog_windows_memory_priority must be between 0 and 5')
process_options['job_cpu_weight'] = job_cpu_weight
process_options['process_memory_priority'] = memory_priority
for root, marker in shared_owners:
armed_owners.append((root, marker))
pending = dict(marker, child_pid=None, child_creation_time=None, child_executable=None)
atomic_write_private_json(os.path.join(root, TEMP_OWNER_FILE), pending)
if time.monotonic() >= command_deadline:
raise subprocess.TimeoutExpired(cmd, timeout_sec)
process = OwnedProcess(cmd, **process_options)
if not process.job_membership_verified:
raise RuntimeError('TruffleHog exact Job membership was not verified')
if scan_slot and not scan_slot.set_child_pid(process.pid):
raise RuntimeError('unable to publish TruffleHog child identity to the scan slot')
write_temp_owner(
command_work_dir, cmd, process.pid, required=True,
owner_identity=getattr(process, 'payload_identity', None),
)
command_owner_published = True
child_identity = getattr(process, 'payload_identity', None)
for root, marker in armed_owners:
if root == canonical_path(command_work_dir):
continue
if not isinstance(child_identity, dict) or any(not child_identity.get(field) for field in ('pid', 'creation_time', 'executable')):
raise RuntimeError('shared staging child identity is unavailable')
active = dict(marker, **{f'child_{field}': child_identity[field] for field in ('pid', 'creation_time', 'executable')})
atomic_write_private_json(os.path.join(root, TEMP_OWNER_FILE), active)
limit_error = ''
next_staging_check = 0.0
while True:
completed = process.poll() is not None
if completed and staging_roots is None:
break
_raise_if_scan_slot_fatal()
stdout_size = os.fstat(stdout_file.fileno()).st_size
stderr_size = os.fstat(stderr_file.fileno()).st_size
if stdout_size > max_stdout:
limit_error = f'TruffleHog stdout exceeded {max_stdout} bytes'
elif stderr_size > max_stderr:
limit_error = f'TruffleHog stderr exceeded {max_stderr} bytes'
elif min_free_bytes:
try:
free_bytes = shutil.disk_usage(command_work_dir).free
except OSError as exc:
limit_error = (
'Unable to monitor TruffleHog staging' if staging_roots is not None
else f'Unable to monitor configured TruffleHog work volume free space: {exc}'
)
else:
if free_bytes <= min_free_bytes:
limit_error = (
f'Not enough free space on configured TruffleHog work volume {command_work_dir}: '
f'{free_bytes / (1024 ** 3):.2f} GB free, minimum is {min_free_gb:.2f} GB'
)
if not limit_error and staging_roots is not None and (
completed or time.monotonic() >= next_staging_check
):
limit_error = _check_command_staging(
staging_roots, command_deadline, (stdout_file, stderr_file),
)
next_staging_check = time.monotonic() + 1.0
if limit_error:
process.kill()
try:
process.wait(timeout=10)
except subprocess.TimeoutExpired:
release_scan_slot = False
limit_error += '; process tree termination failed'
synthetic_stderr = f'Error: {limit_error}'
returncode = -1
break
if completed:
break
if time.monotonic() >= command_deadline:
raise subprocess.TimeoutExpired(cmd, timeout_sec)
if _scan_slot_fatal_event.wait(0.2):
_raise_if_scan_slot_fatal()
if not limit_error:
returncode = process.returncode
stdout_file.seek(0)
stderr_file.seek(0)
yield StreamedCommandOutput(
stdout_file, stderr_file, returncode, max_stdout, max_stderr, synthetic_stderr,
)
except subprocess.TimeoutExpired:
if process and process.poll() is None:
process.kill()
try:
process.wait(timeout=5)
except subprocess.TimeoutExpired:
release_scan_slot = False
if stdout_file is None or stderr_file is None:
raise
timeout_error = f'Command timed out after {timeout_sec} seconds'
if staging_roots is not None:
staging_error = _check_command_staging(
staging_roots, command_deadline, (stdout_file, stderr_file),
)
if staging_error:
timeout_error = f'Error: {staging_error}'
stdout_file.seek(0)
stderr_file.seek(0)
yield StreamedCommandOutput(
stdout_file, stderr_file, -1, max_stdout, max_stderr,
timeout_error,
)
except ScanSlotFatalError as exc:
propagating_fatal_error = exc
raise
finally:
if process:
try:
child_running = process.poll() is None
except Exception:
child_running = True
if child_running:
try:
process.kill()
process.wait(timeout=5)
except Exception:
release_scan_slot = False
if not release_scan_slot:
if scan_slot:
scan_slot.mark_non_releasable()
detail = 'FATAL: TruffleHog child termination was not confirmed; scan capacity remains fail-closed'
_set_scan_slot_fatal(detail)
fatal_slot_error = propagating_fatal_error or ScanSlotFatalError(detail)
restore_error = None
if release_scan_slot:
for root, marker in armed_owners:
if command_owner_published and root == canonical_path(command_work_dir):
continue
try:
atomic_write_private_json(os.path.join(root, TEMP_OWNER_FILE), marker)
except Exception as exc:
restore_error = exc
if scan_slot and owns_scan_slot and release_scan_slot:
scan_slot.release()
if stdout_file:
stdout_file.close()
if stderr_file:
stderr_file.close()
if release_scan_slot and fatal_slot_error is None and propagating_fatal_error is None:
cleanup_command_work_dir(command_work_dir)
if fatal_slot_error is not None:
raise fatal_slot_error
if restore_error is not None:
raise RuntimeError('Unable to restore shared staging ownership after child termination') from restore_error
def run_command(cmd, timeout_sec, env=None):
"""Bounded compatibility adapter; production parsers use run_command_streamed."""
try:
with run_command_streamed(cmd, timeout_sec, env) as output:
stdout = ''.join(output.stdout_lines(max_line_bytes=output.max_stdout, max_lines=20000))
stderr = ''.join(output.stderr_lines(max_line_bytes=output.max_stderr, max_lines=2000))
return stdout, stderr, output.returncode
except ScanSlotFatalError:
raise
except Exception as exc:
return '', f'Error running command: {exc}', -1
def _trufflehog_diagnostic_policy(line, source_type, returncode):
text = str(line or '').strip()
payload = None
try:
parsed = json.loads(text)
if isinstance(parsed, dict):
payload = parsed
except (TypeError, ValueError):
pass
if payload and payload.get('errors'):
causes = payload['errors']
limits = _trufflehog_diagnostic_limits()
if not isinstance(causes, list):
return 'error', 'trufflehog', True
if len(causes) > limits['errors'] or any(
isinstance(cause, str) and (
len(cause) > limits['line_chars']
or len(cause.encode('utf-8', errors='replace')) > limits['line_bytes']
) for cause in causes
):
return 'error', 'output_limit', False
envelope = dict(payload)
del envelope['errors']
envelope_policy = _trufflehog_diagnostic_policy(json.dumps(envelope), source_type, returncode)
if envelope_policy[1] == 'source_auth' and not (payload.get('error') or payload.get('message')):
envelope_policy = ('error', 'auth_or_permission', envelope_policy[2])
policies = []
for cause in causes:
if not isinstance(cause, str) or not cause.strip():
continue
cause_payload = {'level': 'error', 'error': cause}
policy = _trufflehog_diagnostic_policy(json.dumps(cause_payload), source_type, returncode)
cause_payload['msg'] = payload.get('msg') or ''
contextual_policy = _trufflehog_diagnostic_policy(json.dumps(cause_payload), source_type, returncode)
# Keep known warnings, but not at the expense of an independent fatal cause.
if contextual_policy[0] == 'warning' and (
policy[1] == 'trufflehog'
or (
contextual_policy[1] == 'detector_timeout' and policy[1] == 'timeout'
and cause.strip().lower() == 'context deadline exceeded'
)
):
policy = contextual_policy
if policy[1] == 'source_auth':
# A nested diagnostic can be detector verification, not the selected source credential.
policy = ('error', 'auth_or_permission', policy[2])
policies.append(policy)
# Envelope summaries are not additional causes, but explicit fatal details are.
if envelope_policy[0] == 'error' and (
envelope_policy[1] != 'trufflehog' or payload.get('error') or payload.get('message')
):
policies.append(envelope_policy)
fatal = [policy for policy in policies if policy[0] == 'error']
if not fatal:
if policies and all(policy[0] == 'warning' for policy in policies):
return 'warning', policies[0][1], all(policy[2] for policy in policies)
return 'error', 'trufflehog', True
retryable = all(policy[2] for policy in fatal)
classes = {policy[1] for policy in fatal}
for error_class in (
'memory_limit', 'source_configuration', 'output_limit', 'source_resource',
'source_auth', 'docker_registry_access', 'auth_or_permission',
):
if error_class in classes:
return 'error', error_class, retryable
return 'error', next(iter(classes)) if len(classes) == 1 else 'mixed', retryable
if (
source_type == 'docker' and payload
and payload.get('error') and payload.get('message')
and payload['error'] != payload['message']
):
# Independent detail channels must survive codec recovery; reuse fatal priority and bounds.
policy = _trufflehog_diagnostic_policy(json.dumps({
'level': 'error', 'msg': payload.get('msg'),
'errors': [str(payload['error']), str(payload['message'])],
}), source_type, returncode)
return policy if policy[0] == 'error' else ('error', 'trufflehog', True)
message = str((payload or {}).get('msg') or '')
detail = str((payload or {}).get('error') or (payload or {}).get('message') or '')
level = str((payload or {}).get('level') or '').lower()
message_lower = message.lower()
detail_lower = detail.lower()
combined = f'{message_lower} {detail_lower}' if payload else text.lower()
direct_kind = _client_remote_execution_kind.get()
if any(token in combined for token in ('virtualalloc', 'out of memory', 'cannot allocate memory')):
return 'error', 'memory_limit', False
if any(token in combined for token in ('unknown flag', 'unknown command', 'invalid detector', 'failed to load config')):
return 'error', 'source_configuration', False
if any(token in combined for token in (
'trufflehog stdout exceeded', 'trufflehog stderr exceeded',
)):
return 'error', 'output_limit', False
if any(token in combined for token in (
'no space left', 'not enough free space', 'disk quota',
'trufflehog work_dir', 'configured command work directory',
'temporary file', 'temporaryfile', 'disk fsync',
)):
return 'error', 'source_resource', True
if message_lower == 'a detector ignored the context timeout':
return 'warning', 'detector_timeout', False
if source_type == 'huggingface' and detail_lower == 'no repo found for repo':
return 'permanent', 'huggingface_no_repo', False
if source_type == 'docker' and 'no child with platform linux/amd64' in detail_lower:
return 'permanent', 'docker_no_linux_amd64', False
if any(token in combined for token in ('timed out', 'timeout', 'deadline exceeded')):
return 'error', 'timeout', True
if (
source_type == 'docker' and payload
and payload.get('msg') == 'error processing layer'
and payload.get('error') == 'unexpected EOF'
):
# Keep the persisted retry class; layer EOF alone does not prove a network cause.
return 'error', 'network', True
if any(token in combined for token in (
'connection reset', 'connection aborted', 'connection refused', 'could not resolve host',
'temporary failure', 'tls', 'ssl', 'proxy error', 'network is unreachable', 'unexpected eof',
)):
return 'error', 'network', True
if any(token in combined for token in (' 408', ' 429', ' 500', ' 502', ' 503', ' 504', 'too many requests', 'rate limit')):
return 'error', 'remote_transient', True
auth_error = any(token in combined for token in (
'authentication failed', 'unauthorized', 'invalid username or token', 'invalid api key',
'bad credentials',
))
if source_type == 'huggingface' and direct_kind == 'huggingface_space_v1' and (
auth_error or any(token in combined for token in (
'permission denied', 'repository not found', 'private repository',
'gated repo', ' 401', ' 403',
))
):
return 'permanent', 'huggingface_inaccessible', False
if source_type == 'docker' and direct_kind == 'docker_direct_v1' and (
auth_error or any(token in combined for token in (
'permission denied', 'pull access denied',
'requested access to the resource is denied', 'insufficient scope',
'manifest unknown', 'name unknown', 'repository does not exist',
' 401', ' 403', ' 404',
))
):
return 'permanent', 'docker_registry_access', False
if source_type == 'docker' and auth_error:
# A registry manifest may be private or stale; the rotating credential cannot be
# attributed from this target diagnostic, so it must not disable the whole pool.
return 'error', 'docker_registry_access', False
if auth_error:
return 'error', 'source_auth', True
if 'permission denied' in combined:
return 'error', 'auth_or_permission', True
if any(token in combined for token in ('killed by signal', 'terminated by signal', 'process terminated', 'segmentation fault')):
return 'error', 'command_exit', True
if (
source_type in ('docker', 'filesystem') and payload
and message == 'skipping file: size exceeds max allowed'
and not any(key in payload for key in ('error', 'errors', 'message'))
):
return 'warning' if returncode == 0 else 'error', 'archive_member_size', False
if returncode == 0:
if message_lower == 'error cleaning temporary artifacts':
return 'warning', 'cleanup', False
if message_lower == 'skipping result: invalid' and detail_lower == 'empty raw':
return 'warning', 'invalid_empty_result', False
if message_lower == 'non-critical error processing chunk':
return 'warning', 'chunk_processing', False
if source_type == 'git' and message_lower == 'error reading chunk' and detail_lower == 'brotli: excessive input':
return 'warning', 'chunk_read', False
if source_type in ('npm', 'pypi', 'postman', 'filesystem') and message_lower == 'error reading chunk' and any(
token in detail_lower for token in ('brotli:', 'flate: corrupt input', 'error identifying archive', 'invalid header')
):
return 'warning', 'chunk_read', False
if source_type == 'docker' and message_lower == 'error processing layer' and detail_lower == 'gzip: invalid header':
return 'warning', 'docker_layer_gzip', False
if payload and ('error' in level or 'error' in message_lower or payload.get('error')):
return 'error', 'trufflehog', True
if not payload and re.search(r'\b(error|failed|fatal|panic)\b', combined):
return 'error', 'command', True
# Only observed structured progress is exempt from retention, never error details.
if (
source_type in ('docker', 'filesystem') and payload
and payload.get('logger') == 'trufflehog'
and not any(key in payload for key in ('error', 'errors', 'message'))
and (
(level == 'info-0' and message in ('running source', 'finished scanning'))
or (level == 'info-2' and message in (
'trufflehog dev', 'starting scanner workers', 'starting detector workers',
'starting verificationOverlap workers', 'starting notifier workers', 'enumerating source',
))
or (source_type == 'docker' and level == 'info-2' and message in (
'scanning image', 'scanning image history', 'scanning image history entry',
'scanning image layers', 'scanning layer',
))
)
):
return 'routine', '', False
return 'info', '', False
def _trufflehog_diagnostic_limits():
return {
'lines': min(2000, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_lines', 2000), 2000))),
'line_chars': min(8192, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_line_chars', 8192), 8192))),
'line_bytes': min(8192, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_line_bytes', 8192), 8192))),
'errors': min(200, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_errors', 200), 200))),
'warnings': min(200, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_warnings', 200), 200))),
'unclassified': min(20, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_unclassified', 20), 20))),
}
def _iter_output_lines(value):
if isinstance(value, str):
yield from io.StringIO(value)
return
if isinstance(value, bytes):
for raw_line in io.BytesIO(value):
yield raw_line.decode('utf-8', errors='replace')
return
if value is None:
return
yield from value
def apply_trufflehog_diagnostics(
results, stderr, returncode, source_type, require_completion=False,
redactions=(),
):
errors = []
warnings = []
warning_classes = []
warning_retryability = []
permanent = []
unclassified = []
limits = _trufflehog_diagnostic_limits()
line_count = 0
timed_out_seen = False
finished_seen = False
output_limit = ''
captured_stdout = captured_stderr = None
captured_transformation = None
if isinstance(stderr, StreamedCommandOutput):
captured_stdout = stderr.raw_stdout_bytes()
captured_stderr = stderr.raw_stderr_bytes()
captured_transformation = (
'legacy E-frame classification parsed raw process stderr with any '
'pre-existing configured parser redactions; '
'canonical process material preserves the pre-parse captured bytes'
)
stderr = stderr.stderr_lines(redactions=redactions)
try:
for raw_line in _iter_output_lines(stderr):
if line_count >= limits['lines']:
output_limit = f'total line limit of {limits["lines"]} exceeded'
break
line_count += 1
if len(raw_line) > limits['line_chars']:
output_limit = f'line character limit of {limits["line_chars"]} exceeded'
break
if len(raw_line.encode('utf-8', errors='replace')) > limits['line_bytes']:
output_limit = f'line byte limit of {limits["line_bytes"]} exceeded'
break
line = raw_line.strip()
if not line:
continue
try:
payload = json.loads(line)
if isinstance(payload, dict) and str(payload.get('msg') or '').strip().lower() == 'finished scanning':
finished_seen = True
except (TypeError, ValueError):
pass
if 'timed out' in line.lower():
timed_out_seen = True
severity, error_class, retryable = _trufflehog_diagnostic_policy(line, source_type, returncode)
if severity == 'warning':
if len(warnings) + len(permanent) >= limits['warnings']:
output_limit = f'retained warning limit of {limits["warnings"]} exceeded'
break
warnings.append(line)
warning_retryability.append(bool(retryable))
if error_class:
warning_classes.append(error_class)
elif severity == 'permanent':
if len(warnings) + len(permanent) >= limits['warnings']:
output_limit = f'retained warning limit of {limits["warnings"]} exceeded'
break
permanent.append((line, error_class))
elif severity == 'error':
if len(errors) >= limits['errors']:
output_limit = f'retained error limit of {limits["errors"]} exceeded'
break
errors.append((line, error_class, retryable))
if error_class == 'memory_limit':
break
elif severity != 'routine': # Routine records still count against the hard limits above.
if len(unclassified) >= limits['unclassified']:
output_limit = f'retained unclassified limit of {limits["unclassified"]} exceeded'
break
unclassified.append(line)
except CommandOutputLimitError as exc:
output_limit = str(exc)
if output_limit:
synthetic = (
f'TruffleHog diagnostic output_limit reached: {output_limit}; '
'remaining diagnostic output was not retained'
)[:limits['line_chars']]
errors = errors[:max(0, limits['errors'] - 1)]
errors.append((synthetic, 'source_resource', True))
if permanent and not errors:
warnings.extend(line for line, _ in permanent)
warning_classes.extend(error_class for _, error_class in permanent if error_class)
classes = {error_class for _, error_class in permanent}
if classes in ({'huggingface_no_repo'}, {'huggingface_inaccessible'}):
results['skipped'] = 'HuggingFace Space repository is unavailable'
elif classes == {'docker_no_linux_amd64'}:
results['skipped'] = 'Docker image has no linux/amd64 manifest'
elif classes == {'docker_registry_access'}:
results['skipped'] = 'Docker image is unavailable to the worker'
else:
results['skipped'] = 'target is permanently unavailable'
results['error_class'] = next(iter(classes), 'permanent')
results['retryable'] = False
else:
warnings.extend(line for line, _ in permanent)
warning_classes.extend(error_class for _, error_class in permanent if error_class)
completion_required = (
source_type == 'docker' or bool(require_completion)
or (source_type == 'git' and 'chunk_read' in warning_classes)
)
if completion_required and not errors and not permanent and not results.get('skipped'):
if returncode != 0 and finished_seen:
errors.append((f'TruffleHog exited with code {returncode} after the completion marker', 'wrapper_exit', True))
elif returncode != 0 or not finished_seen:
errors.append((f'TruffleHog exited with code {returncode} before the completion marker', 'command_incomplete', True))
elif returncode != 0 and not errors and not permanent:
errors.append((f'TruffleHog exited with code {returncode} without a fatal diagnostic', 'command_exit', True))
elif returncode != 0 and warnings and not errors and not results.get('skipped'):
errors.append((f'TruffleHog exited with code {returncode} after non-fatal diagnostics', 'command_exit', True))
if errors:
results['errors'] = [line for line, _, _ in errors]
classes = [error_class for _, error_class, _ in errors if error_class]
results['error_class'] = classes[0] if len(set(classes)) <= 1 else 'mixed'
results['retryable'] = all(retryable for _, _, retryable in errors)
results['source_failure'] = any(error_class in classes for error_class in ('source_configuration', 'source_resource', 'source_auth'))
if results['source_failure']:
results['source_failure_category'] = 'source_auth' if 'source_auth' in classes else 'source_resource' if 'source_resource' in classes else 'source_configuration'
results['source_failure_auth_related'] = 'source_auth' in classes
if output_limit:
results['error_class'] = 'source_resource'
results['retryable'] = True
results['source_failure'] = True
results['source_failure_category'] = 'source_resource'
results['source_failure_auth_related'] = False
if warnings:
results['warnings'] = warnings
results['warning_classes'] = sorted(set(warning_classes))
results['degraded'] = not bool(results.get('skipped'))
# Nonfatal coverage warnings must not suppress retries of fatal errors.
if warning_retryability and not errors:
warnings_retryable = all(warning_retryability)
results['retryable'] = bool(
results.get('retryable', True)
) and warnings_retryable
scan_meta = results.setdefault('scan_meta', {})
scan_meta['trufflehog_returncode'] = returncode
scan_meta['trufflehog_finished'] = finished_seen
scan_meta['command_timed_out'] = returncode == -1 and timed_out_seen
scan_meta['diagnostic_lines_processed'] = line_count
if warning_retryability:
scan_meta['trufflehog_warnings_retryable'] = all(warning_retryability)
if output_limit:
scan_meta['diagnostic_output_limited'] = True
scan_meta['diagnostic_output_limit_reason'] = output_limit
if unclassified:
scan_meta['stderr_unclassified'] = unclassified
if results.get('errors') and captured_stderr is not None:
results['_diagnostic_raw_stdout_b64'] = base64.b64encode(
captured_stdout or b''
).decode('ascii')
results['_diagnostic_raw_stderr_b64'] = base64.b64encode(
captured_stderr
).decode('ascii')
results['_diagnostic_stderr_transformation'] = captured_transformation
return results
def convert_package_git_unavailable_to_skip(result):
errors = result.get('errors') or []
if not errors:
return result
for error in errors:
text = str(error).lower()
if not (
('repository not found' in text or 'project not found' in text)
and ('failed to clone' in text or 'remote:' in text or 'error preparing repo' in text)
):
return result
result['warnings'] = list(result.get('warnings') or []) + list(errors)
result['warning_classes'] = sorted(set(list(result.get('warning_classes') or []) + ['package_git_repo_unavailable']))
result['errors'] = []
result['skipped'] = 'package_git repository is unavailable or private'
result['error_class'] = 'package_git_repo_unavailable'
result['retryable'] = False
result['degraded'] = False
return result
def apply_result_error_scope(result):
errors = result.get('errors') or []
if not errors or result.get('error_class'):
return result
text = '\n'.join(str(error) for error in errors).lower()
if any(token in text for token in (
'no space left', 'not enough free space', 'disk quota',
'unable to create npm work dir', 'unable to create pypi work dir',
'unable to create postman work dir', 'unable to create github actions work dir',
'unable to create gitlab ci work dir',
)):
result['error_class'] = 'source_resource'
result['retryable'] = True
result['source_failure'] = True
result['source_failure_category'] = 'source_resource'
return result
def append_trufflehog_findings(results, stdout):
invalid = []
if _client_scan_policy.get() is None:
max_findings = max(1, int(os.getenv('TRUFFLEHOG_MAX_FINDINGS_PER_TARGET', '20000')))
else:
max_findings = max(1, int(_scan_policy_value(
'trufflehog_max_findings_per_target', 20000,
)))
try:
lines = _iter_output_lines(stdout)
for line in lines:
if not line.strip():
continue
try:
finding = json.loads(line)
if not isinstance(finding, dict):
raise ValueError('finding JSON is not an object')
if len(results.setdefault('findings', [])) >= max_findings:
results.setdefault('errors', []).append(f'TruffleHog findings exceeded {max_findings} per target')
results['error_class'] = 'output_limit'
results['retryable'] = False
break
results['findings'].append(finding)
except (json.JSONDecodeError, ValueError) as exc:
if len(invalid) < 5:
invalid.append(f'{str(exc)}: {line[:300]}')
except CommandOutputLimitError as exc:
invalid.append(str(exc))
if invalid:
results.setdefault('errors', []).append('Malformed TruffleHog JSON output: ' + '; '.join(invalid))
results['error_class'] = 'output_parse'
results['retryable'] = True
return results
def parse_git_scan_target(target):
text = str(target or '').strip()
if not text.startswith('{'):
return {'url': text, 'branch': '', 'metadata': {}}
try:
data = json.loads(text)
except (TypeError, ValueError):
return {'url': text, 'branch': '', 'metadata': {}}
if not isinstance(data, dict):
return {'url': text, 'branch': '', 'metadata': {}}
return {
'url': str(data.get('url') or data.get('repo_url') or text).strip(),
'branch': str(data.get('branch') or '').strip(),
'metadata': data,
}
def git_branch_ref(branch):
branch = str(branch or '')
resolution = {
'provider': 'github',
'repo_url': 'https://github.com/a/b.git',
'repo_path': 'a/b',
'branch': branch,
'ref': f'refs/heads/{branch}',
'head_sha': '0' * 40,
'ref_source': 'explicit',
}
validate_git_resolution(resolution)
return resolution['ref']
def normalize_git_scan_resolution_target(target, provider=None):
target_info = parse_git_scan_target(target)
raw_url = str(target_info['url'] or '').strip()
if not raw_url or re.search(r'[\x00-\x20\x7f]', raw_url):
raise ValueError('Git target URL is empty or contains control characters')
if re.search(r'%(?:2f|5c)', raw_url, flags=re.IGNORECASE):
raise ValueError('Git target URL contains an encoded path separator')
parse_url = raw_url[4:] if raw_url.startswith('git+') else raw_url
if parse_url.startswith(('github:', 'gitlab:')):
if any(marker in parse_url for marker in ('?', '#', '@')):
raise ValueError('Git target shorthand contains unsafe URL components')
else:
try:
parsed = urlsplit(parse_url)
port = parsed.port
except ValueError as exc:
raise ValueError('Git target URL is malformed') from exc
if (
parsed.scheme not in ('http', 'https', 'git') or not parsed.netloc
or parsed.username is not None or parsed.password is not None
or parsed.query or parsed.fragment or port is not None
):
raise ValueError('Git target URL contains unsupported or unsafe components')
normalized = normalize_git_repo_candidate(raw_url)
if not normalized:
raise ValueError('Git target is not a supported GitHub or GitLab repository')
expected_provider = str(provider or '').strip().lower()
if expected_provider and normalized['provider'] != expected_provider:
raise ValueError('Git target provider does not match the scan source')
metadata = target_info['metadata']
branch = str(target_info.get('branch') or '').strip()
raw_ref = str(metadata.get('ref') or '').strip() if isinstance(metadata, dict) else ''
ref_branch = ''
if raw_ref:
prefix = 'refs/heads/'
if not raw_ref.startswith(prefix):
raise ValueError('Git target ref must be a branch ref')
ref_branch = raw_ref[len(prefix):]
git_branch_ref(ref_branch)
if branch:
git_branch_ref(branch)
if branch and ref_branch and branch != ref_branch:
raise ValueError('Git target branch and ref hints conflict')
branch = branch or ref_branch
return normalized, branch
def git_ref_resolution_api_json(
provider, url, token, deadline, request_attempts, timeout_sec, max_response_bytes,
):
response = api_request(
'GET', url,
headers=github_headers(token) if provider == 'github' else gitlab_headers(token),
timeout=max(0.001, float(timeout_sec)), max_retries=max(1, int(request_attempts)),
retry_delay=1, deadline=deadline, stream=True, allow_redirects=False,
)
if 300 <= response.status_code < 400:
response.close()
raise ApiRequestError(f'{provider} ref resolution refused an HTTP redirect')
if response.status_code >= 400:
try:
error = github_api_error(response) if provider == 'github' else gitlab_api_error(response)
finally:
response.close()
raise error
payload = bounded_response_json(response, max_bytes=max_response_bytes)
if not isinstance(payload, dict):
raise ApiRequestError(f'{provider} ref resolution response is not an object')
return payload
def redacted_git_resolution_error(exc, token):
message = redact_secrets(str(exc), [token])
if isinstance(exc, RateLimitError):
return RateLimitError(
exc.source, message, reset_at=exc.reset_at, category=exc.category,
retryable=exc.retryable, auth_related=exc.auth_related,
)
if isinstance(exc, ApiRequestError):
return ApiRequestError(message)
return exc
def resolve_github_ref_head(
repo_path, token=None, ref_hint=None, *, request_attempts=2, timeout_sec=10,
deadline=None, max_response_bytes=1 << 20,
):
deadline = float(deadline) if deadline is not None else time.monotonic() + max(0.001, float(timeout_sec))
encoded_repo = quote(str(repo_path), safe='/')
branch = str(ref_hint or '')
ref_source = 'explicit' if branch else 'provider_default'
try:
if not branch:
payload = git_ref_resolution_api_json(
'github', f'https://api.github.com/repos/{encoded_repo}', token, deadline,
request_attempts, timeout_sec, max_response_bytes,
)
branch = str(payload.get('default_branch') or '')
git_branch_ref(branch)
ref = f'refs/heads/{branch}'
payload = git_ref_resolution_api_json(
'github', f'https://api.github.com/repos/{encoded_repo}/git/ref/{quote("heads/" + branch, safe="")}',
token, deadline, request_attempts, timeout_sec, max_response_bytes,
)
obj = payload.get('object')
if payload.get('ref') != ref or not isinstance(obj, dict) or obj.get('type') != 'commit':
raise ApiRequestError('GitHub ref resolution returned a mismatched commit ref')
resolved = {
'provider': 'github', 'repo_url': f'https://github.com/{repo_path}.git',
'repo_path': str(repo_path), 'branch': branch, 'ref': ref,
'head_sha': str(obj.get('sha') or '').lower(), 'ref_source': ref_source,
}
return validate_git_resolution(resolved)
except (RateLimitError, ApiRequestError) as exc:
sanitized = redacted_git_resolution_error(exc, token)
if sanitized is exc:
raise
raise sanitized from exc
def resolve_gitlab_ref_head(
repo_path, token=None, ref_hint=None, *, request_attempts=2, timeout_sec=10,
deadline=None, max_response_bytes=1 << 20,
):
deadline = float(deadline) if deadline is not None else time.monotonic() + max(0.001, float(timeout_sec))
project_url = f'https://gitlab.com/api/v4/projects/{quote(str(repo_path), safe="")}'
branch = str(ref_hint or '')
ref_source = 'explicit' if branch else 'provider_default'
try:
if not branch:
payload = git_ref_resolution_api_json(
'gitlab', project_url, token, deadline, request_attempts, timeout_sec,
max_response_bytes,
)
branch = str(payload.get('default_branch') or '')
git_branch_ref(branch)
payload = git_ref_resolution_api_json(
'gitlab', f'{project_url}/repository/branches/{quote(branch, safe="")}',
token, deadline, request_attempts, timeout_sec, max_response_bytes,
)
commit = payload.get('commit')
if payload.get('name') != branch or not isinstance(commit, dict):
raise ApiRequestError('GitLab ref resolution returned a mismatched branch')
resolved = {
'provider': 'gitlab', 'repo_url': f'https://gitlab.com/{repo_path}.git',
'repo_path': str(repo_path), 'branch': branch, 'ref': f'refs/heads/{branch}',
'head_sha': str(commit.get('id') or '').lower(), 'ref_source': ref_source,
}
return validate_git_resolution(resolved)
except (RateLimitError, ApiRequestError) as exc:
sanitized = redacted_git_resolution_error(exc, token)
if sanitized is exc:
raise
raise sanitized from exc
def resolve_git_scan_target(
target, provider, token=None, *, request_attempts=2, timeout_sec=10,
max_response_bytes=1 << 20,
):
normalized, branch = normalize_git_scan_resolution_target(target, provider)
resolver = resolve_github_ref_head if normalized['provider'] == 'github' else resolve_gitlab_ref_head
return resolver(
normalized['repo_path'], token, branch or None, request_attempts=request_attempts,
timeout_sec=timeout_sec, max_response_bytes=max_response_bytes,
)
def validate_bound_git_scan_plan(plan, target, provider=None):
if not isinstance(plan, dict):
raise ValueError('exact Git scan requires a bound plan object')
required = {
'version', 'provider', 'repo_url', 'repo_path', 'branch', 'ref', 'head_sha',
'ref_source', 'base_sha', 'mode', 'baseline_depth',
}
if set(plan) != required or plan.get('version') != 1:
raise ValueError('bound Git scan plan has an unsupported shape')
resolution = validate_git_resolution(plan)
normalized, _ = normalize_git_scan_resolution_target(target, provider or resolution['provider'])
if normalized['repo_url'] != resolution['repo_url'] or normalized['repo_path'] != resolution['repo_path']:
raise ValueError('bound Git scan plan repository conflicts with the target')
mode = str(plan.get('mode') or '')
base_sha = plan.get('base_sha')
if base_sha is not None:
base_sha = str(base_sha).lower()
if not re.fullmatch(r'[a-f0-9]{40}|[a-f0-9]{64}', base_sha):
raise ValueError('bound Git scan plan has an invalid base SHA')
try:
baseline_depth = int(plan.get('baseline_depth'))
except (TypeError, ValueError) as exc:
raise ValueError('bound Git scan plan has an invalid baseline depth') from exc
if not 1 <= baseline_depth <= 1000000:
raise ValueError('bound Git scan plan baseline depth is out of range')
if (
(mode == 'baseline' and base_sha is not None)
or (mode == 'delta' and (not base_sha or base_sha == resolution['head_sha']))
or (mode == 'noop' and base_sha != resolution['head_sha'])
or mode not in ('baseline', 'delta', 'noop')
):
raise ValueError('bound Git scan plan mode and base are inconsistent')
normalized_plan = {
'version': 1, **resolution, 'base_sha': base_sha,
'mode': mode, 'baseline_depth': baseline_depth,
}
if canonical_git_scan_plan_bytes(normalized_plan) != canonical_git_scan_plan_bytes(plan):
raise ValueError('bound Git scan plan is not normalized')
return normalized_plan, hashlib.sha256(canonical_git_scan_plan_bytes(plan)).hexdigest()
def git_delta_base_unavailable(result, base_sha):
diagnostics = '\n'.join(str(item) for item in (
list(result.get('errors') or []) + list(result.get('warnings') or [])
)).lower()
if not diagnostics:
return False
base_markers = (
'bad object', 'unknown revision', 'invalid object', 'object not found',
'reference not found', 'could not find commit', 'unable to resolve commit',
'invalid since commit', 'since-commit', 'since commit',
)
return any(marker in diagnostics for marker in base_markers) and (
str(base_sha or '').lower()[:12] in diagnostics
or 'since' in diagnostics
or 'commit' in diagnostics
or 'revision' in diagnostics
or 'object' in diagnostics
)
def git_checkout_recovery_allowed(result):
meta = result.get('scan_meta') or {}
errors = result.get('errors') or []
if (
os.name != 'nt' or len(errors) not in (1, 2) or result.get('error_class') != 'trufflehog'
or result.get('source_failure') or result.get('warnings') or result.get('degraded') or result.get('skipped')
or meta.get('trufflehog_returncode') != 1 or meta.get('trufflehog_finished') is not False
or meta.get('diagnostic_output_limited') or meta.get('command_timed_out')
):
return False
limits = _trufflehog_diagnostic_limits()
companion = None
if len(errors) == 2:
companion_line = errors[0]
if (
not isinstance(companion_line, str)
or len(companion_line) > limits['line_chars']
or len(companion_line.encode('utf-8', errors='replace')) > limits['line_bytes']
):
return False
try:
companion = json.loads(companion_line)
except (TypeError, ValueError):
return False
if (
not isinstance(companion, dict)
or set(companion) != {
'level', 'ts', 'logger', 'msg', 'subcommand', 'repo', 'path',
'args', 'error',
}
or companion.get('level') != 'info-0'
or companion.get('logger') != 'trufflehog'
or companion.get('msg') != 'git clone failed'
or companion.get('subcommand') != 'git clone'
or companion.get('args') != []
or any(not isinstance(companion.get(key), str) or not companion.get(key) for key in (
'ts', 'repo', 'path', 'error',
))
):
return False
line = errors[-1]
if not isinstance(line, str) or len(line) > limits['line_chars'] or len(line.encode('utf-8', errors='replace')) > limits['line_bytes']:
return False
try:
payload = json.loads(line)
except (TypeError, ValueError):
return False
if (
not isinstance(payload, dict) or payload.get('msg') != 'error running scan'
or payload.get('level') != 'error' or payload.get('errors')
):
return False
detail = payload.get('error')
if not isinstance(detail, str):
return False
if companion is not None and companion['error'] not in detail:
return False
detail = detail.lower()
if not all(marker in detail for marker in (
'error preparing repo', 'error executing git clone: exit status 128',
'clone succeeded, but checkout failed',
)):
return False
prefix, _, git_stderr = detail.partition('error executing git clone: exit status 128')
if re.search(r'\b(?:fatal|error):', prefix):
return False
quoted_path = r"(?:'(?:[^'\\\r\n]|\\.)+'|\"(?:[^\"\\\r\n]|\\.)+\")"
path_failure = False
for physical_line in git_stderr.lstrip(' ,').splitlines():
line_match = re.match(r'^(?:remote:\s*)?(?:fatal|error):\s*(.*)$', physical_line.strip())
if not line_match:
if re.search(r'\b(?:fatal|error):', physical_line):
return False
continue
cause = line_match.group(1)
if re.fullmatch(r'invalid path ' + quoted_path, cause):
path_failure = True
continue
long_path = re.fullmatch(r'(?:unable to create file |cannot create directory (?:at )?)(.+): filename too long', cause)
if long_path:
path = long_path.group(1)
if path.startswith(("'", '"')):
if not re.fullmatch(quoted_path, path):
return False
elif re.search(r'\b(?:fatal|error):', path):
return False
path_failure = True
elif cause != 'unable to checkout working tree':
return False
return path_failure
def scan_exact_git_plan(
target, plan, plan_sha256, timeout_sec, detectors, exclude_detectors,
no_verification, trufflehog_config, token, external_trufflehog_lifecycle,
):
deadline = time.monotonic() + max(0.001, float(timeout_sec or 1))
provider = plan['provider']
scan_url, secrets_to_redact = build_authenticated_git_url(plan['repo_url'], provider, token)
command_env = os.environ.copy()
_append_windows_git_longpaths(command_env, 'Exact Git')
if secrets_to_redact:
command_env['TRUF_GIT_TOKEN'] = token
command_env['TRUF_GIT_USERNAME'] = 'oauth2' if provider == 'gitlab' else 'x-access-token'
recovery_root = None
local_url = None
checkout_errors = []
findings = []
cleanup_safe = True
recovery = {'attempted': False, 'clone_succeeded': False, 'coverage_complete': False}
def remaining():
seconds = deadline - time.monotonic()
if seconds <= 0:
raise subprocess.TimeoutExpired('exact Git scan', timeout_sec)
return seconds
def run_mode(mode):
nonlocal recovery_root, local_url
remaining()
cmd = [
get_trufflehog_cmd(), 'git', local_url or scan_url, '--json', '--no-update',
'--branch', plan['head_sha'],
]
if external_trufflehog_lifecycle:
cmd.append('--local-dev')
append_trufflehog_scan_args(
cmd, detectors, exclude_detectors, no_verification, trufflehog_config,
)
if mode in ('baseline', 'baseline_reset'):
cmd.extend(['--max-depth', str(plan['baseline_depth'])])
elif mode == 'delta':
cmd.extend(['--since-commit', plan['base_sha']])
result = {'findings': findings, 'errors': []}
emit_client_scan_phase('scanning', {
'integrated_operation': 'git_acquisition_and_scan',
'execution_mode': mode,
})
with run_command_streamed(
cmd, remaining(), command_env, deadline=deadline,
staging_roots=(recovery_root,) if recovery_root else None,
) as output:
apply_trufflehog_diagnostics(
result, output,
output.returncode, 'git', require_completion=True,
redactions=secrets_to_redact,
)
append_trufflehog_findings(
result, output.stdout_lines(redactions=secrets_to_redact),
)
checkout_candidate = not recovery['attempted'] and git_checkout_recovery_allowed(result)
if checkout_candidate:
checkout_errors.extend(result['errors'])
if checkout_candidate:
remaining()
recovery['attempted'] = True
emit_client_scan_phase('cloning', {
'operation': 'git_clone_recovery',
})
recovery_root = create_command_work_dir()
destination = os.path.join(recovery_root, 'repo')
clone_cmd = [get_git_cmd(), 'clone', '--no-checkout', '--no-recurse-submodules', '--', scan_url, destination]
clone_result = {'errors': []}
with run_command_streamed(
clone_cmd, remaining(), command_env, deadline=deadline,
staging_roots=(recovery_root,), native_git_clone=True,
) as output:
apply_trufflehog_diagnostics(
clone_result, output, output.returncode, 'git',
redactions=secrets_to_redact,
)
try:
for _ in output.stdout_lines(max_line_bytes=8192, max_lines=2000, redactions=secrets_to_redact):
pass
except CommandOutputLimitError:
clone_result.setdefault('errors', []).append('Git clone output exceeded its diagnostic bounds')
clone_result.update(error_class='output_limit', retryable=False)
if clone_result.get('errors') or clone_result.get('skipped') or clone_result.get('degraded'):
return clone_result
recovery['clone_succeeded'] = True
# Native Windows TH expects the drive in the file URI authority, not /C:/.
from pathlib import Path
local_url = Path(destination).as_uri()
if os.name == 'nt':
local_url = local_url.replace('file:///', 'file://', 1)
return run_mode(mode)
return result
execution_mode = plan['mode']
continuity_reset = False
result = {'findings': [], 'errors': []}
if execution_mode == 'noop':
emit_client_scan_phase('scanning', {
'operation': 'exact_git_noop',
'execution_mode': 'noop',
})
result = {'findings': [], 'errors': []}
else:
try:
result = run_mode(execution_mode)
if (
execution_mode == 'delta' and (not recovery['attempted'] or recovery['clone_succeeded'])
and git_delta_base_unavailable(result, plan['base_sha'])
):
execution_mode = 'baseline_reset'
continuity_reset = True
result = run_mode(execution_mode)
result.setdefault('scan_meta', {})['git_continuity_reset_reason'] = 'covered base unavailable'
except ScanSlotFatalError:
cleanup_safe = False
raise
except subprocess.TimeoutExpired:
result.setdefault('errors', []).append('Exact Git scan exhausted its absolute deadline')
result.update(error_class='timeout', retryable=True)
except Exception as exc:
message = redact_secrets(str(exc), [token])
logger.error('Error scanning pinned Git repository %s: %s', plan['repo_url'], message)
result.setdefault('errors', []).append(f'Scan failed: {message}')
result.update(retryable=True, error_class='remote_transient')
finally:
if recovery_root and cleanup_safe:
try:
cleanup_command_work_dir(recovery_root)
except ScanSlotFatalError:
raise
except Exception as exc:
result.setdefault('errors', []).append('Git recovery cleanup failed: ' + redact_secrets(str(exc), [token]))
result.update(error_class='source_resource', retryable=True, source_failure=True,
source_failure_category='source_resource', source_failure_auth_related=False)
result['findings'] = findings
# Freeze scan coverage before optional filtering or candidate staging adds warnings.
success = not result.get('errors') and not result.get('skipped') and not result.get('degraded')
result['git_scan_plan'] = plan
result['git_scan_execution'] = {
'mode': execution_mode,
'pinned': True,
'success': bool(success),
'coverage_complete': bool(success),
'continuity_reset': continuity_reset,
'plan_sha256': plan_sha256,
}
result.setdefault('scan_meta', {})['exact_git_scope'] = {
'provider': plan['provider'], 'ref': plan['ref'], 'head_sha': plan['head_sha'],
'base_sha': plan['base_sha'], 'mode': execution_mode,
'baseline_depth': plan['baseline_depth'], 'ref_source': plan['ref_source'],
'pinned': True, 'continuity_reset': continuity_reset,
}
result = apply_finding_filters(result, target)
if execution_mode != 'noop' and time.monotonic() >= deadline:
if not result.get('errors'):
result.update(error_class='timeout', retryable=True)
result.setdefault('errors', []).append('Exact Git scan exceeded its absolute deadline including cleanup and filtering')
result.setdefault('scan_meta', {})['git_deadline_exceeded'] = True
result['git_scan_execution'].update(success=False, coverage_complete=False)
if result.get('errors'):
result['git_scan_execution'].update(success=False, coverage_complete=False)
if recovery['attempted']:
recovery['coverage_complete'] = result['git_scan_execution']['coverage_complete']
result.setdefault('scan_meta', {})['git_checkout_recovery'] = recovery
if checkout_errors and not result['git_scan_execution']['coverage_complete']:
result['errors'] = checkout_errors + list(result.get('errors') or [])
result.setdefault('error_class', 'trufflehog')
result.setdefault('retryable', True)
return result
def scan_git_repo(repo_url, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, provider=None, max_depth=None, max_commit_age_days=None, commit_lookup_pages=3, skip_if_commit_lookup_fails=True, external_trufflehog_lifecycle=False, git_plan=None):
"""Scan a single Git repository for secrets"""
target_info = parse_git_scan_target(repo_url)
original_target = repo_url
repo_url = target_info['url']
parsed_repo_url = urlsplit(repo_url)
if parsed_repo_url.username or parsed_repo_url.password:
return {"findings": [], "errors": ["Git target URL must not contain userinfo credentials"], "retryable": False, "error_class": "invalid_target"}
target_branch = target_info['branch']
target_metadata = target_info['metadata']
if git_plan is not None:
try:
plan, plan_sha256 = validate_bound_git_scan_plan(git_plan, original_target, provider)
except (TypeError, ValueError) as exc:
return {
'findings': [], 'errors': [f'Bound Git plan rejected: {exc}'],
'retryable': False, 'error_class': 'invalid_target',
}
return scan_exact_git_plan(
original_target, plan, plan_sha256, timeout_sec, detectors, exclude_detectors,
no_verification, trufflehog_config, token, external_trufflehog_lifecycle,
)
branch_label = f" branch={target_branch}" if target_branch else ''
logger.info(f"Scanning Git repository: {repo_url}{branch_label}")
emit_client_scan_phase('resolving', {
'operation': 'recent_commit_boundary',
'provider': str(provider or 'git'),
})
boundary = recent_commit_boundary(repo_url, provider, token, max_commit_age_days, commit_lookup_pages)
if boundary.get('error'):
category = str(boundary.get('error_category') or 'unknown')
auth_related = bool(boundary.get('auth_related'))
return {
"findings": [], "errors": [boundary.get('reason') or 'commit age lookup failed'],
"error_class": 'source_auth' if auth_related else 'remote_transient',
"retryable": True, "source_failure": True,
"source_failure_category": category,
"source_failure_auth_related": auth_related,
"scan_meta": {**boundary, 'target_metadata': target_metadata},
}
if boundary.get('skip'):
reason = boundary.get('reason', 'skipped by commit age filter')
logger.info(f"Skipping {repo_url}: {reason}")
if boundary.get('permanent') or skip_if_commit_lookup_fails:
return {"findings": [], "errors": [], "skipped": reason, "scan_meta": {**boundary, 'target_metadata': target_metadata}}
effective_provider, _ = get_git_provider_and_path(repo_url, provider)
scan_url, secrets_to_redact = build_authenticated_git_url(repo_url, effective_provider, token)
cmd = [get_trufflehog_cmd(), 'git', scan_url, '--json', '--no-update']
if external_trufflehog_lifecycle:
cmd.append('--local-dev')
append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config)
if max_depth:
cmd.extend(['--max-depth', str(max_depth)])
if target_branch:
cmd.extend(['--branch', target_branch])
if boundary.get('since_commit'):
cmd.extend(['--since-commit', boundary['since_commit']])
logger.info(
f"Scanning {repo_url} since commit {boundary['since_commit']} "
f"({boundary.get('recent_commit_count')} commits after {boundary.get('cutoff')})"
)
try:
command_env = os.environ.copy()
if secrets_to_redact:
command_env['TRUF_GIT_TOKEN'] = token
command_env['TRUF_GIT_USERNAME'] = 'oauth2' if effective_provider == 'gitlab' else 'x-access-token'
results = {"findings": [], "errors": []}
emit_client_scan_phase('scanning', {
'integrated_operation': 'git_acquisition_and_scan',
})
with run_command_streamed(cmd, timeout_sec, command_env) as output:
apply_trufflehog_diagnostics(
results, output,
output.returncode, 'git',
require_completion=external_trufflehog_lifecycle,
redactions=secrets_to_redact,
)
append_trufflehog_findings(
results, output.stdout_lines(redactions=secrets_to_redact),
)
if boundary.get('since_commit') or target_metadata or target_branch:
results.setdefault("scan_meta", {}).update({**boundary, 'branch': target_branch, 'target_metadata': target_metadata})
return apply_finding_filters(results, original_target)
except ScanSlotFatalError:
raise
except Exception as e:
logger.error(f"Error scanning repository {repo_url}: {str(e)}")
return {"findings": [], "errors": [f"Scan failed: {str(e)}"]}
DOCKER_ARCHIVE_MAX_DECODED_BYTES = 1 << 30
def _require_docker_archive_policy(limits):
# This independent decoded-stream ceiling is part of docker-layer-execution-v4.
if limits['archive_max_size_bytes'] > DOCKER_ARCHIVE_MAX_DECODED_BYTES:
raise DockerLayerInfrastructureError(
'archive_policy_incompatible', 'Docker member policy exceeds the decoded validation ceiling',
category='source_configuration',
)
def validate_docker_content_artifact(
path, descriptor, *, deadline=None, max_member_bytes=256 << 20,
max_decoded_bytes=DOCKER_ARCHIVE_MAX_DECODED_BYTES, max_members=100000,
):
import zlib
deadline = min(float(deadline) if deadline is not None else float('inf'), time.monotonic() + 30)
if not math.isfinite(deadline) or any(
isinstance(value, bool) or not isinstance(value, int) or not 0 < value <= bound
for value, bound in ((max_member_bytes, DOCKER_ARCHIVE_MAX_DECODED_BYTES),
(max_decoded_bytes, DOCKER_ARCHIVE_MAX_DECODED_BYTES), (max_members, 100000))
):
raise DockerLayerInfrastructureError(
'archive_validation_bounds', 'Docker archive validation bounds are invalid', category='source_configuration',
)
def check_deadline():
_raise_if_scan_slot_fatal()
if time.monotonic() >= deadline:
raise DockerContentScanError('archive_timeout', 'Docker archive validation deadline expired', True)
check_deadline()
kind = str((descriptor or {}).get('kind') or '')
media_type = str((descriptor or {}).get('media_type') or '').strip().lower()
expected_size = (descriptor or {}).get('size')
try:
actual_size = os.path.getsize(path)
except OSError as exc:
raise DockerLayerInfrastructureError(
'private_storage', 'Docker content artifact is unavailable',
category='source_resource',
) from exc
if actual_size != expected_size:
raise DockerContentScanError(
'size_mismatch', 'Docker content artifact size changed after verification',
)
if kind == 'config':
if media_type not in DOCKER_CONFIG_MEDIA_TYPES:
raise DockerContentScanError(
'unsupported_media_type', 'Docker configuration media type is unsupported',
)
if actual_size > 64 << 20:
raise DockerContentScanError('archive_limit', 'Docker configuration exceeds the validation bound')
try:
with open(path, 'rb') as config_file:
raw_config = config_file.read(actual_size + 1)
except OSError as exc:
raise DockerLayerInfrastructureError(
'private_storage', 'Docker configuration artifact cannot be read',
category='source_resource',
) from exc
try:
config = json.loads(raw_config.decode('utf-8'))
except (RecursionError, UnicodeDecodeError, TypeError, ValueError) as exc:
raise DockerContentScanError(
'invalid_config_json', 'Docker configuration is not valid UTF-8 JSON',
) from exc
if not isinstance(config, dict):
raise DockerContentScanError(
'invalid_config_json', 'Docker configuration JSON must be an object',
)
check_deadline()
return config
if kind != 'layer' or media_type not in DOCKER_LAYER_MEDIA_TYPES:
raise DockerContentScanError(
'unsupported_media_type', 'Docker layer media type is unsupported',
)
zstd = None
if media_type.endswith('+zstd'):
try:
import zstandard as zstd
except ImportError as exc:
raise DockerLayerInfrastructureError(
'archive_decoder_unavailable', 'Docker zstd validation capability is unavailable',
category='source_configuration',
) from exc
decoded_bytes = 0
members = 0
terminated = False
class TimedInput:
def read(self, size=-1):
check_deadline()
data = layer_file.read(min(size if size >= 0 else 65536, 65536))
check_deadline()
return data
class BoundedReader:
def read(self, size=-1):
nonlocal decoded_bytes
check_deadline()
data = decoder.read(min(size if size >= 0 else 65536, 65536, max_decoded_bytes - decoded_bytes + 1))
decoded_bytes += len(data)
if decoded_bytes > max_decoded_bytes:
raise DockerContentScanError('archive_limit', 'Docker archive decoded byte bound exceeded')
check_deadline()
return data
class BoundedTarInfo(tarfile.TarInfo):
@classmethod
def fromtarfile(cls, archive):
nonlocal terminated
try:
check_deadline()
member = super().fromtarfile(archive)
check_deadline()
return member
except tarfile.EOFHeaderError:
terminated = True
raise
except tarfile.HeaderError as exc:
# TarFile.next otherwise tolerates some corrupt headers after member one.
raise DockerContentScanError('invalid_layer_archive', 'Docker tar header is invalid') from exc
def _proc_member(self, archive):
nonlocal members
check_deadline()
members += 1
metadata = self.type in (tarfile.XHDTYPE, tarfile.XGLTYPE, tarfile.SOLARIS_XHDTYPE,
tarfile.GNUTYPE_LONGNAME, tarfile.GNUTYPE_LONGLINK)
if self.size < 0:
raise DockerContentScanError('invalid_layer_archive', 'Docker archive member size is negative')
if members > max_members or self.size > min(max_member_bytes, 1 << 20 if metadata else max_member_bytes):
raise DockerContentScanError('archive_limit', 'Docker archive member bound exceeded')
if self.type == tarfile.GNUTYPE_SPARSE:
raise DockerContentScanError('unsupported_layer_archive', 'Docker sparse tar validation is unsupported')
member = super()._proc_member(archive)
check_deadline()
return member
def _proc_pax(self, archive):
# Do not enter tarfile's unbounded hdrcharset/length regexes or sparse
# map parsers. Validate complete records before decoding/applying fields.
if self.size > 64 << 10:
raise DockerContentScanError('archive_limit', 'Docker PAX parse byte bound exceeded')
body = archive.fileobj.read(self._block(self.size))
if len(body) != self._block(self.size):
raise DockerContentScanError('invalid_layer_archive', 'Docker PAX header is truncated')
body = body[:self.size]
records = []
position = 0
while position < len(body):
check_deadline()
if len(records) >= min(max_members, 1024):
raise DockerContentScanError('archive_limit', 'Docker PAX record bound exceeded')
space = body.find(b' ', position, min(position + 9, len(body)))
if space < 0:
raise DockerContentScanError('archive_limit', 'Docker PAX record length field exceeds its bound')
digits = body[position:space]
if not digits or not digits.isdigit():
raise DockerContentScanError('invalid_layer_archive', 'Docker PAX record length is invalid')
end = position + int(digits)
if end > len(body) or end <= space + 3 or body[end - 1:end] != b'\n':
raise DockerContentScanError('invalid_layer_archive', 'Docker PAX record boundary is invalid')
equals = body.find(b'=', space + 1, min(end - 1, space + 258))
if equals <= space + 1:
raise DockerContentScanError('invalid_layer_archive', 'Docker PAX keyword is invalid or oversized')
key, value = body[space + 1:equals], body[equals + 1:end - 1]
if key.startswith(b'GNU.sparse.'):
raise DockerContentScanError('unsupported_layer_archive', 'Docker sparse PAX validation is unsupported')
converter = tarfile.PAX_NUMBER_FIELDS.get(key.decode('utf-8'))
if converter is not None:
if len(value) > (32 if converter is int else 64):
raise DockerContentScanError('archive_limit', 'Docker PAX numeric field exceeds its bound')
number = converter(value)
if converter is float and not math.isfinite(number):
raise DockerContentScanError('invalid_layer_archive', 'Docker PAX numeric field is not finite')
if key == b'size' and not 0 <= number <= max_member_bytes:
raise DockerContentScanError('archive_limit', 'Docker PAX member size exceeds its bound')
records.append((key, value))
position = end
check_deadline()
pax_headers = archive.pax_headers.copy()
charset = next((value.decode('utf-8') for key, value in records if key == b'hdrcharset'),
pax_headers.get('hdrcharset'))
encoding = archive.encoding if charset == 'BINARY' else 'utf-8'
for key, value in records:
check_deadline()
key = self._decode_pax_field(key, 'utf-8', 'utf-8', archive.errors)
if key in tarfile.PAX_NAME_FIELDS:
value = self._decode_pax_field(value, encoding, archive.encoding, archive.errors)
else:
value = self._decode_pax_field(value, 'utf-8', 'utf-8', archive.errors)
pax_headers[key] = value
if len(pax_headers) > 1024:
raise DockerContentScanError('archive_limit', 'Docker global PAX field bound exceeded')
if self.type == tarfile.XGLTYPE:
archive.pax_headers = pax_headers
try:
member = self.fromtarfile(archive)
except tarfile.HeaderError as exc:
raise DockerContentScanError('invalid_layer_archive', 'Docker PAX header has no following member') from exc
if self.type in (tarfile.XHDTYPE, tarfile.SOLARIS_XHDTYPE):
check_deadline()
member._apply_pax_info(pax_headers, archive.encoding, archive.errors)
member.offset = self.offset
if 'size' in pax_headers:
archive.offset = member.offset_data
if member.isreg() or member.type not in tarfile.SUPPORTED_TYPES:
archive.offset += member._block(member.size)
check_deadline()
return member
try:
with open(path, 'rb') as layer_file:
magic = layer_file.read(4)
layer_file.seek(0)
if zstd is not None:
# stream_reader can silently accept a truncated final frame. Check physical
# frame/block boundaries separately, then let the decoder verify checksums.
frames = 0
while layer_file.tell() < actual_size:
check_deadline()
frames += 1
start = layer_file.tell()
header = layer_file.read(18)
if frames > max_members:
raise DockerContentScanError('archive_limit', 'Docker zstd frame bound exceeded')
if len(header) >= 8 and 0x184d2a50 <= int.from_bytes(header[:4], 'little') <= 0x184d2a5f:
end = start + 8 + int.from_bytes(header[4:8], 'little')
if end > actual_size:
raise DockerContentScanError('invalid_layer_archive', 'Docker zstd skippable frame is truncated')
layer_file.seek(end)
continue
if header[:4] != b'\x28\xb5\x2f\xfd':
raise DockerContentScanError('invalid_layer_archive', 'Docker zstd frame header is invalid')
header_size = zstd.frame_header_size(header)
params = zstd.get_frame_parameters(header)
if params.window_size > max_decoded_bytes or (
params.content_size != zstd.CONTENTSIZE_UNKNOWN and params.content_size > max_decoded_bytes
):
raise DockerContentScanError('archive_limit', 'Docker zstd window bound exceeded')
layer_file.seek(start + header_size)
while True:
check_deadline()
block = layer_file.read(3)
if len(block) != 3:
raise DockerContentScanError('invalid_layer_archive', 'Docker zstd frame is truncated')
value = int.from_bytes(block, 'little')
block_type, block_size = (value >> 1) & 3, value >> 3
if block_type == 3 or block_size > 128 << 10:
raise DockerContentScanError('invalid_layer_archive', 'Docker zstd block is invalid')
end = layer_file.tell() + (1 if block_type == 1 else block_size)
if value & 1 and params.has_checksum:
end += 4
if end > actual_size:
raise DockerContentScanError('invalid_layer_archive', 'Docker zstd frame is truncated')
layer_file.seek(end)
if value & 1:
break
layer_file.seek(0)
decoder = zstd.ZstdDecompressor(max_window_size=max(1024, max_decoded_bytes)).stream_reader(
TimedInput(), read_across_frames=True, closefd=False,
)
elif media_type in DOCKER_LAYER_GZIP_MEDIA_TYPES:
if not magic.startswith(b'\x1f\x8b'):
raise DockerContentScanError('invalid_layer_archive', 'Docker layer does not match its gzip media type')
decoder = gzip.GzipFile(fileobj=TimedInput())
else:
decoder = layer_file
try:
with tarfile.open(fileobj=BoundedReader(), mode='r|', tarinfo=BoundedTarInfo) as archive:
for member in archive:
if member.size > max_member_bytes:
raise DockerContentScanError('archive_limit', 'Docker archive member bound exceeded')
if member.sparse is not None:
raise DockerContentScanError('unsupported_layer_archive', 'Docker sparse tar validation is unsupported')
if member.isfile():
with archive.extractfile(member) as body:
while body.read(65536):
check_deadline()
if not terminated or archive.fileobj.read(512) != b'\0' * 512:
raise DockerContentScanError('invalid_layer_archive', 'Docker tar end marker is missing')
while True:
padding = archive.fileobj.read(65536)
if not padding:
break
if padding.strip(b'\0'):
raise DockerContentScanError('invalid_layer_archive', 'Docker tar has trailing non-padding content')
if decoded_bytes % 512:
raise DockerContentScanError('invalid_layer_archive', 'Docker tar padding is truncated')
finally:
if decoder is not layer_file:
decoder.close()
except DockerContentScanError:
raise
except (tarfile.TarError, EOFError, gzip.BadGzipFile, zlib.error, ValueError, RecursionError) as exc:
raise DockerContentScanError(
'invalid_layer_archive', 'Docker layer is not a valid bounded tar archive',
) from exc
except OSError as exc:
raise DockerLayerInfrastructureError(
'private_storage', 'Docker layer artifact cannot be read',
category='source_resource',
) from exc
except Exception as exc:
if zstd is not None and isinstance(exc, zstd.ZstdError):
raise DockerContentScanError('invalid_layer_archive', 'Docker zstd stream is invalid') from exc
raise
def attach_docker_content_provenance(
findings, plan, descriptor, positions, private_blob_path,
):
location = (
f'docker://{plan["repository"]}@{plan["manifest_digest"]}/'
f'{descriptor["kind"]}/{descriptor["digest"]}'
)
private_blob_path = os.path.normcase(os.path.abspath(private_blob_path))
for finding in findings:
if not isinstance(finding, dict):
continue
source = finding.setdefault('SourceMetadata', {})
data = source.setdefault('Data', {}) if isinstance(source, dict) else {}
filesystem = data.get('Filesystem') if isinstance(data, dict) else None
if isinstance(filesystem, dict):
original = str(filesystem.get('file') or '')
if private_blob_path and private_blob_path in os.path.normcase(original):
original = original[len(private_blob_path):].lstrip('/\\:')
filesystem['file'] = location + (f'/{original}' if original else '')
if isinstance(data, dict):
docker_content = {
'image': plan['image'],
'manifest_digest': plan['manifest_digest'],
'blob_digest': descriptor['digest'],
'descriptor_kind': descriptor['kind'],
'positions': list(positions),
}
if plan['version'] == 2:
docker_content['payload_class'] = descriptor['payload_class']
data['DockerContent'] = docker_content
return findings
def _docker_content_error_code(value, fallback='scan_failed'):
text = re.sub(r'[^a-z0-9_]+', '_', str(value or '').strip().lower()).strip('_')
return text[:64] if re.fullmatch(r'[a-z][a-z0-9_]{0,63}', text) else fallback
def _scan_docker_content_file(
destination, descriptor, limits, deadline, detectors=None,
exclude_detectors=None, no_verification=False, trufflehog_config=None,
):
_require_docker_archive_policy(limits)
validate_docker_content_artifact(
destination, descriptor,
deadline=min(deadline, time.monotonic() + limits['archive_timeout_sec']),
max_member_bytes=limits['archive_max_size_bytes'],
)
remaining = deadline - time.monotonic()
if remaining <= 0:
raise DockerContentScanError('archive_timeout', 'Docker blob deadline expired before scanning', True)
command = [
get_trufflehog_cmd(), 'filesystem', destination, '--json', '--no-update',
'--log-level', '2',
'--archive-max-size', f'{limits["archive_max_size_bytes"]}B',
'--archive-max-depth', str(limits['archive_max_depth']),
'--archive-timeout', f'{limits["archive_timeout_sec"]}s',
'--concurrency', str(limits['filesystem_concurrency']),
]
append_trufflehog_scan_args(command, detectors, exclude_detectors, no_verification, trufflehog_config)
result = {'findings': [], 'errors': []}
with run_command_streamed(
command, remaining, os.environ.copy(), deadline=deadline,
staging_roots=(os.path.dirname(os.path.abspath(destination)),),
) as output:
apply_trufflehog_diagnostics(
result, output, output.returncode, 'filesystem', require_completion=True,
)
append_trufflehog_findings(result, output.stdout_lines())
# Only wrapper-owned diagnostics can attribute a watchdog stop to staging.
staging_error = output.synthetic_stderr.removeprefix('Error: ').split(';', 1)[0]
if staging_error in ('TruffleHog staging limit exceeded', 'Unable to monitor TruffleHog staging'):
monitor_failed = staging_error == 'Unable to monitor TruffleHog staging'
result.update(
errors=[staging_error],
error_class='source_resource' if monitor_failed else 'staging_limit',
retryable=monitor_failed,
source_failure=monitor_failed,
source_failure_auth_related=False,
)
result.pop('source_failure_category', None)
if monitor_failed:
result['source_failure_category'] = 'source_resource'
return result
if time.monotonic() >= deadline:
result['errors'].append('Docker blob scan exceeded its absolute deadline')
result['error_class'] = 'timeout'
result['retryable'] = True
return result
def scan_docker_layer_plan(
image_name, docker_layer_work, timeout_sec=600, detectors=None,
exclude_detectors=None, no_verification=False, trufflehog_config=None,
*, log_target=True,
):
try:
plan = validate_docker_layer_plan((docker_layer_work or {}).get('plan'))
plan_bytes = canonical_docker_layer_plan_bytes(plan)
plan_sha256 = hashlib.sha256(plan_bytes).hexdigest()
if plan_sha256 != str((docker_layer_work or {}).get('plan_sha256') or ''):
raise ValueError('Docker layer work plan hash is invalid')
if parse_docker_target(image_name)['image'].lower() != plan['image']:
raise ValueError('Docker layer work target does not match its plan')
except (TypeError, ValueError) as exc:
raise DockerLayerInfrastructureError(
'invalid_bound_plan', 'Docker layer bound plan is invalid',
category='source_configuration',
) from exc
_require_docker_archive_policy(plan['limits'])
leased_by_digest = {}
for descriptor in plan['descriptors']:
if descriptor['coverage_state'] == 'leased':
entry = leased_by_digest.setdefault(descriptor['digest'], {
'descriptor': descriptor, 'positions': [],
})
entry['positions'].append(descriptor['position'])
work_root = None
retain_work = False
created_paths = []
findings = []
finding_digests = {}
records = []
failures = []
bearer_auth = (docker_layer_work or {}).get('bearer_auth')
min_free_bytes = max(0, int((docker_layer_work or {}).get('min_free_bytes') or 0))
execution_deadline = time.monotonic() + max(0.001, float(timeout_sec or 0.001))
supplied_deadline = (docker_layer_work or {}).get('deadline')
if supplied_deadline is not None:
try:
supplied_deadline = float(supplied_deadline)
except (TypeError, ValueError) as exc:
raise DockerLayerInfrastructureError(
'invalid_deadline', 'Docker layer execution deadline is invalid',
category='source_configuration',
) from exc
if not math.isfinite(supplied_deadline):
raise DockerLayerInfrastructureError(
'invalid_deadline', 'Docker layer execution deadline is invalid',
category='source_configuration',
)
execution_deadline = min(execution_deadline, supplied_deadline)
try:
if leased_by_digest:
if time.monotonic() >= execution_deadline:
raise DockerLayerInfrastructureError(
'execution_deadline', 'Docker layer execution deadline expired before work began',
category='remote_transient',
)
try:
work_root = tempfile.mkdtemp(prefix='docker-layer-', dir=get_work_dir())
harden_private_directory(work_root)
write_temp_owner(work_root, ['docker-layer-content'], os.getpid(), required=True)
except ScanSlotFatalError:
raise
except (OSError, RuntimeError, ValueError) as exc:
raise DockerLayerInfrastructureError(
'private_storage', 'Docker layer private work storage is unavailable',
category='source_resource',
) from exc
for digest, entry in leased_by_digest.items():
descriptor = entry['descriptor']
blob_started = time.monotonic()
blob_deadline = min(
execution_deadline,
blob_started + plan['limits']['blob_timeout_sec'],
)
destination = os.path.join(work_root, f'blob-{len(created_paths):04d}')
verified_bytes = 0
transfer_bytes = 0
transfer_duration_ms = 0
scan_duration_ms = 0
blob_findings = []
error_code = ''
blob_retryable = True
try:
if not records:
emit_client_scan_phase('downloading', {
'operation': 'docker_blob_transfer',
})
else:
emit_client_scan_phase('downloading', {
'operation': 'additional_docker_blob_transfer',
})
outcome = stream_docker_registry_blob(
plan['repository'], descriptor, destination, bearer_auth,
deadline=blob_deadline, min_free_bytes=min_free_bytes,
)
bearer_auth = outcome.bearer_auth
created_paths.append(destination)
verified_bytes = outcome.verified_bytes
transfer_bytes = outcome.transfer_bytes
transfer_duration_ms = outcome.duration_ms
scan_started = time.monotonic()
emit_client_scan_phase('scanning', {
'operation': 'docker_blob_scan',
})
result = _scan_docker_content_file(
destination, descriptor, plan['limits'], blob_deadline,
detectors, exclude_detectors, no_verification, trufflehog_config,
)
scan_duration_ms = max(0, int((time.monotonic() - scan_started) * 1000))
if time.monotonic() >= blob_deadline and not result.get('errors'):
result['errors'] = ['Docker layer scan exceeded its absolute deadline']
result['error_class'] = 'timeout'
result['retryable'] = True
blob_findings = list(result.get('findings') or [])
attach_docker_content_provenance(
blob_findings, plan, descriptor, entry['positions'], destination,
)
finding_digests.update(
(id(finding), digest)
for finding in blob_findings if isinstance(finding, dict)
)
if result.get('source_failure'):
raise DockerLayerInfrastructureError(
_docker_content_error_code(
result.get('source_failure_category'), 'scanner_infrastructure',
),
'Docker layer scanner infrastructure is unavailable',
category=str(
result.get('source_failure_category') or 'source_resource'
),
auth_related=bool(result.get('source_failure_auth_related')),
)
if (
result.get('errors') or result.get('skipped')
or result.get('warnings') or result.get('degraded')
):
diagnostic_code = result.get('error_class')
if not diagnostic_code and result.get('warning_classes'):
diagnostic_code = result['warning_classes'][0]
error_code = _docker_content_error_code(
diagnostic_code,
'scan_incomplete' if result.get('warnings') else 'scan_failed',
)
blob_retryable = bool(
result.get('retryable', not bool(result.get('skipped')))
)
except DockerLayerInfrastructureError:
raise
except DockerContentTransferError as exc:
error_code = _docker_content_error_code(exc.error_code, 'transfer_failed')
blob_retryable = exc.retryable
transfer_bytes = max(transfer_bytes, int(exc.transfer_bytes or 0))
transfer_duration_ms = max(
transfer_duration_ms, int(exc.duration_ms or 0),
)
except DockerContentScanError as exc:
error_code = _docker_content_error_code(exc.error_code, 'invalid_content')
blob_retryable = exc.retryable
except ScanSlotFatalError:
retain_work = True
raise
except Exception as exc:
raise DockerLayerInfrastructureError(
'scanner_infrastructure', 'Docker layer scanner infrastructure failed',
category='source_resource',
) from exc
finally:
if not retain_work and destination in created_paths:
try:
durable_unlink(destination)
except OSError as exc:
raise DockerLayerInfrastructureError(
'private_cleanup', 'Docker layer artifact cleanup failed',
category='source_resource',
) from exc
created_paths.remove(destination)
terminal = (
not blob_retryable
or int(descriptor['attempt']) >= int(descriptor['max_attempts'])
)
status = (
'covered' if not error_code
else 'terminal_failed' if terminal
else 'retryable_failed'
)
records.append({
'digest': digest,
'lease_token': descriptor['lease_token'],
'status': status,
'verified_bytes': verified_bytes,
'transfer_bytes': transfer_bytes,
'transfer_duration_ms': transfer_duration_ms,
'scan_duration_ms': scan_duration_ms,
'finding_count': len(blob_findings),
'error_code': error_code or None,
})
findings.extend(blob_findings)
if error_code:
failures.append(f'Docker content {digest[:19]} failed: {error_code}')
except ScanSlotFatalError:
retain_work = True
raise
finally:
# Fatal process outcomes leave payloads and ownership evidence for the janitor.
if not retain_work:
for path in created_paths:
if os.path.lexists(path):
try:
durable_unlink(path)
except OSError as exc:
raise DockerLayerInfrastructureError(
'private_cleanup', 'Docker layer artifact cleanup failed',
category='source_resource',
) from exc
if work_root:
cleanup_command_work_dir(work_root)
records_by_digest = {item['digest']: item for item in records}
if set(records_by_digest) != set(leased_by_digest):
raise DockerLayerInfrastructureError(
'missing_execution', 'Docker layer execution metadata is incomplete',
category='source_resource',
)
effective_descriptors = []
for descriptor in plan['descriptors']:
effective_state = descriptor['coverage_state']
if effective_state == 'leased':
effective_state = records_by_digest[descriptor['digest']]['status']
effective_descriptors.append((descriptor, effective_state))
static_states = {item['coverage_state'] for item in plan['descriptors']}
if 'selected' in static_states:
failures.append('Docker content remains selected for the next durable checkpoint')
if 'shared_pending' in static_states:
failures.append('Docker content is pending under another fenced reservation')
result_states = {state for _, state in effective_descriptors}
has_retryable = bool(
result_states & {'selected', 'shared_pending', 'retryable_failed'}
)
has_terminal = 'terminal_failed' in result_states
if has_terminal:
failures.append('Docker content exhausted its bounded attempt budget')
result = {
'findings': findings,
'errors': failures,
'retryable': bool(has_retryable),
'error_class': ('docker_content_retry' if has_retryable else 'docker_content_terminal')
if failures else None,
'docker_layer_plan': plan,
'docker_layer_execution': {
'version': plan['version'],
'plan_sha256': plan_sha256,
'blobs': records,
},
'scan_meta': {
'docker_layer_scope': {
'manifest_digest': plan['manifest_digest'],
'coverage_complete': all(
state == 'covered' for _, state in effective_descriptors
),
'selected_descriptors': sum(
1 for item in plan['descriptors'] if item['selected']
),
'leased_blobs': len(leased_by_digest),
'covered_blobs': len({
item['digest'] for item, state in effective_descriptors
if state == 'covered'
}),
'newly_covered_blobs': sum(
1 for item in records if item['status'] == 'covered'
),
'globally_reused_blobs': len({
item['digest'] for item in plan['descriptors']
if item['coverage_state'] == 'covered'
}),
'retryable_failed_blobs': sum(
1 for item in records if item['status'] == 'retryable_failed'
),
'terminal_failed_blobs': sum(
1 for item in records if item['status'] == 'terminal_failed'
),
'pending_checkpoint_blobs': sum(
1 for item in plan['descriptors']
if item['coverage_state'] == 'selected'
),
'shared_pending_blobs': sum(
1 for item in plan['descriptors']
if item['coverage_state'] == 'shared_pending'
),
'selected_bytes': sum(
item['size'] for item in plan['descriptors'] if item['selected']
),
'covered_bytes': sum(
item['size'] for item, state in effective_descriptors
if state == 'covered'
),
'skipped_bytes': sum(
item['size'] for item, state in effective_descriptors
if state == 'skipped'
),
'shared_pending_bytes': sum(
item['size'] for item, state in effective_descriptors
if state == 'shared_pending'
),
'retryable_failed_bytes': sum(
item['size'] for item, state in effective_descriptors
if state == 'retryable_failed'
),
'terminal_failed_bytes': sum(
item['size'] for item, state in effective_descriptors
if state == 'terminal_failed'
),
'transfer_bytes': sum(item['transfer_bytes'] for item in records),
'transfer_duration_ms': sum(
item['transfer_duration_ms'] for item in records
),
'scan_duration_ms': sum(item['scan_duration_ms'] for item in records),
'timeout_blobs': sum(
1 for item in records
if item['error_code'] in ('timeout', 'transfer_timeout')
),
'skipped_descriptors': sum(
1 for item in plan['descriptors'] if item['coverage_state'] == 'skipped'
),
},
},
}
if not failures:
result.pop('error_class')
result.pop('retryable')
if 'skipped' in static_states:
result['degraded'] = True
result['warnings'] = ['Docker content plan intentionally skipped bounded descriptors']
result['warning_classes'] = ['docker_content_budget']
try:
result = apply_finding_filters(
result, plan['image'], log_target=log_target,
)
except Exception as exc:
raise DockerLayerInfrastructureError(
'result_filter', 'Docker layer result filtering failed',
category='source_configuration',
) from exc
filtered_counts = Counter(
finding_digests.get(id(finding))
for finding in result.get('findings', [])
if finding_digests.get(id(finding))
)
for record in result['docker_layer_execution']['blobs']:
record['finding_count'] = filtered_counts[record['digest']]
return result
def _docker_implicit_auth_present():
if os.environ.get('DOCKER_TOKEN') or os.environ.get('REGISTRY_AUTH_FILE') or docker_token_manager.has_accounts():
return True
homes = {os.environ.get('HOME'), os.environ.get('USERPROFILE'), os.path.expanduser('~')}
if os.environ.get('HOMEDRIVE') and os.environ.get('HOMEPATH'):
homes.add(os.environ['HOMEDRIVE'] + os.environ['HOMEPATH'])
paths = [os.path.join(home, '.docker', 'config.json') for home in homes if home and home != '~']
xdg = os.environ.get('XDG_RUNTIME_DIR', '')
if xdg and not os.path.isabs(xdg):
return True
paths.append(os.path.join(xdg, 'containers', 'auth.json'))
for path in paths:
try:
os.lstat(path)
return True
except (FileNotFoundError, NotADirectoryError):
continue
except OSError:
return True
return False
def _recover_docker_image_contents(
image_ref, deadline, config_dir, limits, min_free_bytes,
detectors, exclude_detectors, no_verification, trufflehog_config, *,
implicit_auth_unsupported=False, anonymous_public_client=False,
):
from scanner_db import validate_docker_layer_limits
result = {'findings': [], 'errors': []}
scope = {'coverage_complete': False, 'scanned_descriptors': 0, 'blob_transfer_attempted': False}
result['scan_meta'] = {'docker_full_recovery': scope}
phase = 'configuration'
diagnostic = {}
descriptor = None
work_root = None
retain_work = False
try:
limits = validate_docker_layer_limits(limits if limits is not None else {
'config_max_bytes': 1 << 20, 'layer_max_bytes': 256 << 20,
'image_max_bytes': 1 << 30, 'max_layers': 8,
'archive_max_size_bytes': 256 << 20, 'archive_max_depth': 4,
'archive_timeout_sec': 30, 'blob_timeout_sec': 600,
'filesystem_concurrency': 2, 'blob_max_attempts': 3,
})
_require_docker_archive_policy(limits)
phase = 'preflight'
if time.monotonic() >= deadline:
raise DockerContentScanError('timeout', 'Docker recovery deadline expired', True)
try:
_, _, repository, digest = _dockerhub_manifest_target_parts(image_ref)
except (TypeError, ValueError) as exc:
raise DockerContentScanError('unsupported_recovery_target', 'Docker recovery requires an immutable Docker Hub target') from exc
phase = 'authentication'
bearer_auth = None
if anonymous_public_client:
if config_dir:
raise DockerLayerInfrastructureError(
'recovery_auth_unavailable',
'Anonymous Docker recovery received credential configuration',
category='source_configuration',
)
elif config_dir:
# Only the already-managed credential pool is trusted. Never load an arbitrary
# Docker config or run its external credential helpers in the recovery path.
with docker_token_manager.lock:
matched = next((account for account in docker_token_manager.accounts if
os.path.normcase(os.path.abspath(account.config_dir)) == os.path.normcase(os.path.abspath(config_dir))), None)
if matched is None:
raise DockerLayerInfrastructureError(
'recovery_auth_unavailable', 'Docker recovery cannot use this credential configuration',
category='source_configuration',
)
excluded = {account.name for account in docker_token_manager.accounts if account.name != matched.name}
bearer_auth = docker_registry_bearer_token(
'Bearer realm="https://auth.docker.io/token",service="registry.docker.io"',
repository, excluded_accounts=excluded, deadline=deadline,
)
else:
# The native default keychain may use Docker/Podman configs without
# DOCKER_CONFIG. Inspect existence only; never read or invoke helpers.
if implicit_auth_unsupported or _docker_implicit_auth_present():
raise DockerLayerInfrastructureError(
'recovery_auth_unavailable', 'Docker recovery cannot map implicit credential identity',
category='source_configuration',
)
phase = 'manifest_resolution'
emit_client_scan_phase('resolving', {
'operation': 'docker_manifest_resolution_recovery',
})
resolved, bearer_auth = resolve_docker_content_manifest(
image_ref, bearer_auth=bearer_auth, deadline=deadline,
anonymous_only=anonymous_public_client,
)
phase = 'preflight'
if resolved['image'] != image_ref or resolved['manifest_digest'] != digest or resolved['repository'] != repository:
raise DockerContentScanError('recovery_identity_mismatch', 'Docker recovery manifest identity changed')
descriptors = [dict(resolved['config'], kind='config', position=0)] + [
dict(item, kind='layer', position=index) for index, item in enumerate(resolved['layers'], 1)
]
scope['descriptor_count'] = len(descriptors)
if len(resolved['layers']) > limits['max_layers']:
diagnostic.update(reason='count_bound', observed=len(resolved['layers']), limit=limits['max_layers'])
raise DockerContentScanError('recovery_budget', 'All Docker layers do not fit the recovery count bound')
unique = {}
for descriptor in descriptors:
kind, media, size = descriptor['kind'], descriptor.get('media_type'), descriptor['size']
allowed = DOCKER_CONFIG_MEDIA_TYPES if kind == 'config' else DOCKER_LAYER_MEDIA_TYPES
if not isinstance(media, str) or media not in allowed:
# Reporting only: these official types do not expand recovery support.
known_media = DOCKER_CONFIG_MEDIA_TYPES | DOCKER_LAYER_MEDIA_TYPES | {
'application/vnd.oci.image.layer.nondistributable.v1.tar',
'application/vnd.oci.image.layer.nondistributable.v1.tar+gzip',
'application/vnd.oci.image.layer.nondistributable.v1.tar+zstd',
'application/vnd.docker.image.rootfs.foreign.diff.tar',
'application/vnd.docker.image.rootfs.foreign.diff.tar.gzip',
'application/vnd.oci.image.manifest.v1+json',
'application/vnd.oci.image.index.v1+json',
'application/vnd.docker.distribution.manifest.v1+json',
'application/vnd.docker.distribution.manifest.v1+prettyjws',
'application/vnd.docker.distribution.manifest.v2+json',
'application/vnd.docker.distribution.manifest.list.v2+json',
}
diagnostic.update(
reason='media_type',
media_type=media if isinstance(media, str) and media in known_media else 'other',
)
raise DockerContentScanError('unsupported_media_type', 'Docker recovery descriptor is unsupported')
if not normalize_docker_digest(descriptor['digest']):
diagnostic['reason'] = 'integrity'
raise DockerContentScanError('unsupported_media_type', 'Docker recovery descriptor is unsupported')
if isinstance(size, bool) or not isinstance(size, int) or size < 0:
raise DockerContentScanError('invalid_descriptor', 'Docker recovery descriptor size is invalid')
byte_limit = limits['config_max_bytes' if kind == 'config' else 'layer_max_bytes']
if size > byte_limit:
diagnostic.update(reason='descriptor_byte_bound', observed=size, limit=byte_limit)
raise DockerContentScanError('recovery_budget', 'All Docker content does not fit the recovery byte bound')
entry = unique.setdefault(descriptor['digest'], {'descriptor': descriptor, 'positions': []})
previous = entry['descriptor']
if any(previous[key] != descriptor[key] for key in ('kind', 'size', 'media_type')):
raise DockerContentScanError('conflicting_descriptors', 'Docker recovery digest descriptors conflict')
entry['positions'].append(descriptor['position'])
descriptor = None
image_bytes = sum(entry['descriptor']['size'] for entry in unique.values())
if image_bytes > limits['image_max_bytes']:
diagnostic.update(reason='image_byte_bound', observed=image_bytes, limit=limits['image_max_bytes'])
raise DockerContentScanError('recovery_budget', 'All Docker content does not fit the cumulative recovery bound')
if time.monotonic() >= deadline:
raise DockerContentScanError('timeout', 'Docker recovery deadline expired after preflight', True)
phase = 'staging'
work_root = tempfile.mkdtemp(prefix='docker-layer-', dir=get_work_dir())
harden_private_directory(work_root)
write_temp_owner(work_root, ['docker-full-recovery'], os.getpid(), required=True)
for index, entry in enumerate(unique.values()):
descriptor = entry['descriptor']
phase = 'blob_transfer'
blob_deadline = min(deadline, time.monotonic() + limits['blob_timeout_sec'])
if time.monotonic() >= blob_deadline:
raise DockerContentScanError('timeout', 'Docker recovery deadline expired', True)
destination = os.path.join(work_root, f'blob-{index:04d}')
try:
blob_min_free_bytes = max(0, int(min_free_bytes))
scope['blob_transfer_attempted'] = True
emit_client_scan_phase('downloading', {
'operation': 'docker_blob_transfer_recovery',
'descriptor_index': index,
})
outcome = stream_docker_registry_blob(
repository, descriptor, destination, bearer_auth,
deadline=blob_deadline, min_free_bytes=blob_min_free_bytes,
anonymous_only=anonymous_public_client,
)
bearer_auth = outcome.bearer_auth
phase = 'blob_scan'
emit_client_scan_phase('scanning', {
'operation': 'docker_blob_scan_recovery',
'descriptor_index': index,
})
scanned = _scan_docker_content_file(
destination, descriptor, limits, blob_deadline,
detectors, exclude_detectors, no_verification, trufflehog_config,
)
result['findings'].extend(attach_docker_content_provenance(
scanned['findings'], resolved, descriptor, entry['positions'], destination,
))
if scanned.get('source_failure'):
raise DockerLayerInfrastructureError(
'recovery_scanner_unavailable', 'Docker recovery scanner is unavailable',
category=scanned.get('source_failure_category') or 'source_resource',
)
if any(scanned.get(key) for key in ('errors', 'warnings', 'degraded', 'skipped')):
raise DockerContentScanError(
'recovery_scan_incomplete', 'Docker recovery scanner did not cover a blob',
bool(scanned.get('retryable', False)),
)
scope['scanned_descriptors'] += len(entry['positions'])
except ScanSlotFatalError:
retain_work = True
raise
finally:
if not retain_work and os.path.lexists(destination):
previous_phase = phase
phase = 'cleanup'
durable_unlink(destination)
phase = previous_phase
descriptor = None
phase = 'completion'
if time.monotonic() >= deadline:
raise DockerContentScanError('timeout', 'Docker recovery deadline expired', True)
except ScanSlotFatalError:
retain_work = True
raise
except Exception as exc:
code = _docker_content_error_code(getattr(exc, 'error_code', None), 'recovery_infrastructure')
if isinstance(exc, DockerRemoteAccessError):
code = _docker_content_error_code(exc.status, 'recovery_remote')
elif isinstance(exc, DockerRegistryResolutionError):
code = 'recovery_manifest_invalid'
diagnostic.setdefault('reason', {
'recovery_identity_mismatch': 'integrity',
'invalid_descriptor': 'integrity',
'conflicting_descriptors': 'integrity',
'digest_mismatch': 'integrity',
'size_mismatch': 'integrity',
'invalid_layer_archive': 'integrity',
'invalid_config_json': 'integrity',
'recovery_manifest_invalid': 'manifest_invalid',
'timeout': 'timeout',
'transfer_timeout': 'timeout',
'archive_timeout': 'timeout',
'recovery_scan_incomplete': 'scan_incomplete',
}.get(code, 'configuration' if phase == 'configuration' else 'other'))
diagnostic['phase'] = phase
if descriptor is not None:
diagnostic.update(descriptor_kind=descriptor['kind'], descriptor_index=descriptor['position'])
scope['diagnostic'] = diagnostic
result['errors'].append(f'Docker full-image recovery incomplete: {code}')
result['error_class'] = code
result['retryable'] = bool(getattr(exc, 'retryable', True))
if isinstance(exc, DockerRegistryResolutionError):
result['retryable'] = (
isinstance(exc, DockerRemoteAccessError)
and exc.status not in ('target_forbidden', 'auth_failed')
)
if result['retryable']:
result['source_failure'] = True
result['source_failure_category'] = 'remote_auth' if exc.status == 'auth_failed' else 'remote_rate_limit' if exc.status == 'rate_limited' else 'remote_transient'
elif anonymous_public_client and exc.status == 'auth_failed':
result['error_class'] = 'docker_registry_access'
result.pop('source_failure', None)
result.pop('source_failure_category', None)
elif isinstance(exc, DockerLayerInfrastructureError) or not isinstance(exc, (DockerContentScanError, DockerContentTransferError)):
result['source_failure'] = True
result['source_failure_category'] = getattr(exc, 'category', 'source_configuration' if isinstance(exc, ValueError) else 'source_resource')
result['source_failure_auth_related'] = bool(getattr(exc, 'auth_related', False))
finally:
if work_root and not retain_work:
try:
cleanup_command_work_dir(work_root)
except ScanSlotFatalError:
raise
except (OSError, RuntimeError):
scope.setdefault('diagnostic', {'phase': 'cleanup', 'reason': 'cleanup'})
result['errors'].append('Docker full-image recovery private cleanup failed')
result.update(error_class='private_cleanup', retryable=True, source_failure=True,
source_failure_category='source_resource')
if not result['errors'] and time.monotonic() >= deadline:
scope['diagnostic'] = {'phase': 'completion', 'reason': 'timeout'}
result.update(errors=['Docker full-image recovery exceeded its absolute deadline'],
error_class='timeout', retryable=True)
scope['coverage_complete'] = not result['errors'] and scope['scanned_descriptors'] == scope.get('descriptor_count', 0) > 0
return result
def scan_docker_image(image_name, timeout_sec=1800, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, config_dir=None, trufflehog_concurrency=0, *, log_target=True, docker_recovery_limits=None, docker_recovery_min_free_bytes=20 << 30):
"""Scan a Docker image for secrets"""
started = time.monotonic()
try:
timeout_sec = float(timeout_sec)
if not math.isfinite(timeout_sec) or timeout_sec <= 0:
raise ValueError('invalid timeout')
except (TypeError, ValueError):
return {'findings': [], 'errors': ['Docker scan requires a positive finite time budget'],
'error_class': 'timeout', 'retryable': False}
deadline = started + timeout_sec
anonymous_public_client = (
_client_remote_execution_kind.get() == 'docker_direct_v1'
)
if anonymous_public_client and config_dir:
raise RuntimeError('remote Docker direct execution cannot use credentials')
try:
image_ref = (
parse_dockerhub_digest_target(image_name)['image']
if anonymous_public_client else parse_docker_target(image_name)['image']
)
except (TypeError, ValueError) as exc:
return {
"findings": [], "errors": [f"Docker image target rejected: {exc}"],
"error_class": "invalid_target", "retryable": False,
}
if log_target:
logger.info(f"Scanning Docker image: {image_ref}")
cmd = [get_trufflehog_cmd(), 'docker', '--image', image_ref, '--json', '--no-update', '--local-dev', '--log-level', '2']
concurrency = max(0, min(64, int(trufflehog_concurrency or 0)))
if concurrency:
cmd.extend(['--concurrency', str(concurrency)])
append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config)
env = os.environ.copy()
if config_dir:
env['DOCKER_CONFIG'] = config_dir
implicit_auth_unsupported = (
not anonymous_public_client
and not (config_dir or env.get('DOCKER_CONFIG'))
and _docker_implicit_auth_present()
)
results = {"findings": [], "errors": []}
emit_client_scan_phase('scanning', {
'integrated_operation': 'docker_pull_and_scan',
})
with run_command_streamed(cmd, max(0, deadline - time.monotonic()), env, deadline=deadline) as output:
apply_trufflehog_diagnostics(results, output, output.returncode, 'docker')
append_trufflehog_findings(results, output.stdout_lines())
for line in results.get('errors', []):
try:
diagnostic = json.loads(line)
except (TypeError, ValueError):
continue
if (
isinstance(diagnostic, dict) and diagnostic.get('msg') == 'error processing layer'
and diagnostic.get('error') == 'unexpected EOF'
):
results.setdefault('scan_meta', {})['docker_native_diagnostic'] = {
'phase': 'native_layer_processing', 'subcause': 'unexpected_eof_ambiguous',
'coverage_complete': False,
}
break
codec_warnings = []
for line in results.get('warnings', []):
try:
diagnostic = json.loads(line)
except (TypeError, ValueError):
continue
if isinstance(diagnostic, dict) and diagnostic.get('msg') == 'error processing layer' and diagnostic.get('error') == 'gzip: invalid header':
codec_warnings.append(line)
if codec_warnings:
if not results.get('errors') and not results.get('source_failure') and not results.get('skipped') and len(codec_warnings) == len(results.get('warnings', [])):
recovered = _recover_docker_image_contents(
image_ref, deadline, (
None if anonymous_public_client
else config_dir or env.get('DOCKER_CONFIG')
),
docker_recovery_limits, docker_recovery_min_free_bytes,
detectors, exclude_detectors, no_verification, trufflehog_config,
implicit_auth_unsupported=implicit_auth_unsupported,
anonymous_public_client=anonymous_public_client,
)
results['findings'].extend(recovered.pop('findings'))
results.setdefault('scan_meta', {}).update(recovered.pop('scan_meta'))
if not recovered['errors'] and results['scan_meta']['docker_full_recovery']['coverage_complete']:
for key in ('warnings', 'warning_classes', 'degraded', 'retryable'):
results.pop(key, None)
results['scan_meta']['docker_full_recovery']['recovered_codec'] = True
else:
if not recovered['errors']:
recovered.update(errors=['Docker full-image recovery coverage is incomplete'],
error_class='docker_recovery_incomplete', retryable=False)
results['scan_meta']['docker_full_recovery'].setdefault(
'diagnostic', {'phase': 'completion', 'reason': 'scan_incomplete'},
)
results.update(recovered)
else:
results['errors'].append('Docker codec recovery cannot clear unrelated scan diagnostics')
results.setdefault('error_class', 'docker_recovery_incomplete')
results.setdefault('scan_meta', {})['docker_full_recovery'] = {
'coverage_complete': False, 'blob_transfer_attempted': False,
'diagnostic': {'phase': 'native_diagnostics', 'reason': 'unrelated_diagnostics'},
}
results = apply_finding_filters(results, image_ref, log_target=log_target)
if time.monotonic() >= deadline:
had_errors = bool(results.get('errors'))
results.setdefault('errors', []).append('Docker scan exceeded its absolute deadline including cleanup and filtering')
if not had_errors:
results.update(error_class='timeout', retryable=True)
metadata = results.setdefault('scan_meta', {})
metadata['docker_deadline_exceeded'] = True
if 'docker_full_recovery' in metadata:
metadata['docker_full_recovery']['coverage_complete'] = False
metadata['docker_full_recovery'].setdefault('diagnostic', {'phase': 'completion', 'reason': 'timeout'})
return results
def _append_windows_git_longpaths(command_env, operation):
if os.name != 'nt':
return
config_count = command_env.get('GIT_CONFIG_COUNT') or '0'
if not re.fullmatch(r'[0-9]{1,3}', config_count) or int(config_count) > 255:
raise ValueError(f'{operation} GIT_CONFIG_COUNT must be an integer from 0 to 255')
config_count = int(config_count)
command_env[f'GIT_CONFIG_KEY_{config_count}'] = 'core.longpaths'
command_env[f'GIT_CONFIG_VALUE_{config_count}'] = 'true'
command_env['GIT_CONFIG_COUNT'] = str(config_count + 1)
def scan_huggingface_space(space_id, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None):
anonymous_public_client = (
_client_remote_execution_kind.get() == 'huggingface_space_v1'
)
if anonymous_public_client and token:
raise RuntimeError('remote HuggingFace direct execution cannot use a token')
logger.info(f"Scanning HuggingFace Space: {space_id}")
cmd = [get_trufflehog_cmd(), 'huggingface', '--space', space_id, '--json', '--no-update']
secrets_to_redact = []
command_env = os.environ.copy()
_append_windows_git_longpaths(command_env, 'HuggingFace')
if token:
command_env['HUGGINGFACE_TOKEN'] = token
command_env['HF_TOKEN'] = token
secrets_to_redact.append(token)
append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config)
results = {"findings": [], "errors": []}
emit_client_scan_phase('scanning', {
'integrated_operation': 'huggingface_clone_and_scan',
})
with run_command_streamed(cmd, timeout_sec, command_env) as output:
apply_trufflehog_diagnostics(
results, output, output.returncode, 'huggingface',
redactions=secrets_to_redact,
)
append_trufflehog_findings(
results, output.stdout_lines(redactions=secrets_to_redact),
)
return apply_finding_filters(results, space_id)
WINDOWS_RESERVED_NAMES = {
'con', 'prn', 'aux', 'nul',
*(f'com{index}' for index in range(1, 10)),
*(f'lpt{index}' for index in range(1, 10)),
}
def validate_archive_member_name(name):
normalized = str(name or '').replace('\\', '/')
if normalized.startswith('/') or '..' in normalized.split('/'):
raise ValueError(f'Unsafe archive member path: {name}')
for component in [part for part in normalized.split('/') if part]:
stem = component.split('.', 1)[0].lower()
if ':' in component or component.endswith((' ', '.')) or stem in WINDOWS_RESERVED_NAMES:
raise ValueError(f'Unsafe Windows archive member name: {name}')
return normalized
class LimitedReader:
def __init__(self, source, limit):
self.source = source
self.remaining = max(0, int(limit))
def read(self, size=-1):
if self.remaining <= 0:
raise ValueError('Archive decompressed stream exceeds safety budget')
if size is None or size < 0:
size = self.remaining + 1
data = self.source.read(min(size, self.remaining + 1))
self.remaining -= len(data)
if self.remaining < 0:
raise ValueError('Archive decompressed stream exceeds safety budget')
return data
def safe_extract_tar(tar_path, destination, max_files=100000, max_total_size_mb=1024, max_file_size_mb=512):
_raise_if_scan_slot_fatal()
destination_abs = os.path.abspath(destination)
max_total_bytes = int(max_total_size_mb or 0) * 1024 * 1024
max_file_bytes = int(max_file_size_mb or 0) * 1024 * 1024
stream_budget = max_total_bytes + (64 * 1024 * 1024)
raw_source = open(tar_path, 'rb')
magic = raw_source.read(6)
raw_source.seek(0)
if magic.startswith(b'\x1f\x8b'):
decompressed = gzip.GzipFile(fileobj=raw_source)
elif magic.startswith(b'BZh'):
decompressed = bz2.BZ2File(raw_source)
elif magic.startswith(b'\xfd7zXZ\x00'):
decompressed = lzma.LZMAFile(raw_source)
else:
decompressed = raw_source
try:
archive = tarfile.open(fileobj=LimitedReader(decompressed, stream_budget), mode='r|')
try:
total_size = 0
member_count = 0
for member in archive:
_raise_if_scan_slot_fatal()
if max_files and member_count >= int(max_files):
raise ValueError(f'Tar archive exceeds {max_files} members')
member_count += 1
validate_archive_member_name(member.name)
member_path = os.path.abspath(os.path.join(destination, member.name))
if not member_path.startswith(destination_abs + os.sep) and member_path != destination_abs:
raise ValueError(f"Unsafe tar member path: {member.name}")
if member.issym() or member.islnk() or member.isdev() or member.isfifo():
raise ValueError(f"Unsafe tar member type: {member.name}")
if member.isfile():
if max_file_bytes and member.size > max_file_bytes:
raise ValueError(f"Tar member exceeds {max_file_size_mb} MB: {member.name}")
total_size += max(0, int(member.size or 0))
if max_total_bytes and total_size > max_total_bytes:
raise ValueError(f"Tar archive exceeds {max_total_size_mb} MB extracted")
archive.extract(member, destination, filter='data')
_raise_if_scan_slot_fatal()
finally:
archive.close()
finally:
if decompressed is not raw_source:
decompressed.close()
raw_source.close()
def validate_zip_central_directory(zip_path, max_files, max_metadata_size_mb=16):
_raise_if_scan_slot_fatal()
with open(zip_path, 'rb') as source:
source.seek(0, os.SEEK_END)
size = source.tell()
source.seek(max(0, size - 65557))
tail = source.read()
offset = tail.rfind(b'PK\x05\x06')
if offset < 0 or offset + 22 > len(tail):
raise ValueError('Zip end-of-central-directory record is missing')
disk_number = int.from_bytes(tail[offset + 4:offset + 6], 'little')
central_disk = int.from_bytes(tail[offset + 6:offset + 8], 'little')
entry_count = int.from_bytes(tail[offset + 10:offset + 12], 'little')
central_size = int.from_bytes(tail[offset + 12:offset + 16], 'little')
central_offset = int.from_bytes(tail[offset + 16:offset + 20], 'little')
if disk_number or central_disk or entry_count == 0xFFFF or central_size == 0xFFFFFFFF or central_offset == 0xFFFFFFFF:
raise ValueError('Multi-disk and ZIP64 archives are not accepted')
if max_files and entry_count > int(max_files):
raise ValueError(f'Zip archive exceeds {max_files} members')
max_metadata_bytes = int(max_metadata_size_mb or 0) * 1024 * 1024
if max_metadata_bytes and central_size > max_metadata_bytes:
raise ValueError(f'Zip central directory exceeds {max_metadata_size_mb} MB')
if central_offset < 0 or central_size < 0 or central_offset + central_size > size:
raise ValueError('Zip central directory points outside the archive')
with open(zip_path, 'rb') as source:
source.seek(central_offset)
consumed = 0
for _ in range(entry_count):
_raise_if_scan_slot_fatal()
header = source.read(46)
if len(header) != 46 or header[:4] != b'PK\x01\x02':
raise ValueError('Invalid zip central-directory entry')
name_len = int.from_bytes(header[28:30], 'little')
extra_len = int.from_bytes(header[30:32], 'little')
comment_len = int.from_bytes(header[32:34], 'little')
variable_size = name_len + extra_len + comment_len
source.seek(variable_size, os.SEEK_CUR)
consumed += 46 + variable_size
if consumed > central_size:
raise ValueError('Zip central-directory size mismatch')
if consumed != central_size:
raise ValueError('Zip central-directory entry count mismatch')
return entry_count
def safe_extract_package_zip(zip_path, destination, max_files=100000, max_total_size_mb=1024, max_file_size_mb=512):
_raise_if_scan_slot_fatal()
destination_abs = os.path.abspath(destination)
max_total_bytes = int(max_total_size_mb or 0) * 1024 * 1024
max_file_bytes = int(max_file_size_mb or 0) * 1024 * 1024
validate_zip_central_directory(zip_path, max_files)
with zipfile.ZipFile(zip_path) as archive:
members = archive.infolist()
if max_files and len(members) > int(max_files):
raise ValueError(f'Zip archive exceeds {max_files} members')
total_size = 0
for member in members:
_raise_if_scan_slot_fatal()
validate_archive_member_name(member.filename)
member_path = os.path.abspath(os.path.join(destination, member.filename))
if not member_path.startswith(destination_abs + os.sep) and member_path != destination_abs:
raise ValueError(f"Unsafe zip member path: {member.filename}")
if member.is_dir():
continue
if max_file_bytes and member.file_size > max_file_bytes:
raise ValueError(f"Zip member exceeds {max_file_size_mb} MB: {member.filename}")
total_size += max(0, int(member.file_size or 0))
if max_total_bytes and total_size > max_total_bytes:
raise ValueError(f"Zip archive exceeds {max_total_size_mb} MB extracted")
for member in members:
_raise_if_scan_slot_fatal()
archive.extract(member, destination)
_raise_if_scan_slot_fatal()
def safe_extract_archive(archive_path, destination, max_total_size_mb=1024):
_raise_if_scan_slot_fatal()
if tarfile.is_tarfile(archive_path):
safe_extract_tar(archive_path, destination, max_total_size_mb=max_total_size_mb)
return
if zipfile.is_zipfile(archive_path):
safe_extract_package_zip(archive_path, destination, max_total_size_mb=max_total_size_mb)
return
raise ValueError("Unsupported package archive format")
def download_file(url, path, max_size_mb=50, timeout=60):
_raise_if_scan_slot_fatal()
max_bytes = max_size_mb * 1024 * 1024
with requests.Session() as session:
session.trust_env = False
with session.get(url, stream=True, headers={'User-Agent': 'GitSecretsScanner/2.0'}, timeout=timeout) as response:
_raise_if_scan_slot_fatal()
response.raise_for_status()
total = 0
with open(path, 'wb') as f:
for chunk in response.iter_content(chunk_size=1024 * 256):
_raise_if_scan_slot_fatal()
if not chunk:
continue
total += len(chunk)
if max_bytes and total > max_bytes:
raise ValueError(f"Artifact exceeds {max_size_mb} MB")
f.write(chunk)
_raise_if_scan_slot_fatal()
harden_private_file(path)
return total
def scan_npm_package(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, max_artifact_size_mb=50):
"""Download, extract, and scan an npm package tarball with TruffleHog filesystem."""
package = parse_npm_target(target)
package_id = npm_package_id(package)
logger.info(f"Scanning npm package: {package_id}")
_raise_if_scan_slot_fatal()
work_dir = create_command_work_dir()
_raise_if_scan_slot_fatal()
if not work_dir:
return {"findings": [], "errors": ["Unable to create npm work dir"]}
try:
tarball_path = os.path.join(work_dir, 'package.tgz')
extract_dir = os.path.join(work_dir, 'extract')
ensure_private_directory(extract_dir, reject_reparse=True)
downloaded = download_file(package['tarball'], tarball_path, max_artifact_size_mb, timeout=min(timeout_sec, 120))
_raise_if_scan_slot_fatal()
safe_extract_tar(tarball_path, extract_dir, max_total_size_mb=max(256, int(max_artifact_size_mb or 0) * 4))
_raise_if_scan_slot_fatal()
harden_private_tree(extract_dir)
_raise_if_scan_slot_fatal()
harvest_warnings = []
postman_targets = find_postman_artifacts(
extract_dir,
'npm',
package,
max_artifact_size_mb=max_artifact_size_mb,
warnings=harvest_warnings,
)
cmd = [get_trufflehog_cmd(), 'filesystem', extract_dir, '--json', '--no-update']
append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config)
results = {"findings": [], "errors": [], "package": package, "bytes": downloaded, "postman_targets": postman_targets}
with run_command_streamed(cmd, timeout_sec) as output:
apply_trufflehog_diagnostics(results, output, output.returncode, 'npm')
append_trufflehog_findings(results, output.stdout_lines())
attach_nearby_context(results)
_attach_postman_harvest_warnings(results, harvest_warnings)
return apply_finding_filters(results, package_id)
except ScanSlotFatalError:
raise
except Exception as e:
return {"findings": [], "errors": [f"npm scan failed: {str(e)}"], "package": package}
finally:
cleanup_command_work_dir(work_dir)
def scan_pypi_package(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, max_artifact_size_mb=50):
"""Download, extract, and scan a PyPI sdist/wheel with TruffleHog filesystem."""
package = parse_pypi_target(target)
package_id = pypi_package_id(package)
logger.info(f"Scanning PyPI package: {package_id}")
declared_size = int(package.get('size') or 0)
max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024
if max_bytes and declared_size > max_bytes:
return {
"findings": [],
"errors": [],
"skipped": f"artifact exceeds {max_artifact_size_mb} MB",
"package": package,
}
_raise_if_scan_slot_fatal()
work_dir = create_command_work_dir()
_raise_if_scan_slot_fatal()
if not work_dir:
return {"findings": [], "errors": ["Unable to create PyPI work dir"]}
try:
artifact_path = os.path.join(work_dir, 'package-artifact')
extract_dir = os.path.join(work_dir, 'extract')
ensure_private_directory(extract_dir, reject_reparse=True)
downloaded = download_file(package['artifact'], artifact_path, max_artifact_size_mb, timeout=min(timeout_sec, 120))
_raise_if_scan_slot_fatal()
safe_extract_archive(artifact_path, extract_dir, max_total_size_mb=max(256, int(max_artifact_size_mb or 0) * 4))
_raise_if_scan_slot_fatal()
harden_private_tree(extract_dir)
_raise_if_scan_slot_fatal()
harvest_warnings = []
postman_targets = find_postman_artifacts(
extract_dir,
'pypi',
package,
max_artifact_size_mb=max_artifact_size_mb,
warnings=harvest_warnings,
)
cmd = [get_trufflehog_cmd(), 'filesystem', extract_dir, '--json', '--no-update']
append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config)
results = {"findings": [], "errors": [], "package": package, "bytes": downloaded, "postman_targets": postman_targets}
with run_command_streamed(cmd, timeout_sec) as output:
apply_trufflehog_diagnostics(results, output, output.returncode, 'pypi')
append_trufflehog_findings(results, output.stdout_lines())
attach_nearby_context(results)
_attach_postman_harvest_warnings(results, harvest_warnings)
return apply_finding_filters(results, package_id)
except ScanSlotFatalError:
raise
except Exception as e:
return {"findings": [], "errors": [f"PyPI scan failed: {str(e)}"], "package": package}
finally:
cleanup_command_work_dir(work_dir)
def scan_package_git_repo(target, timeout_sec=900, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, max_depth=None, max_commit_age_days=None, commit_lookup_pages=3, skip_if_commit_lookup_fails=True):
data = parse_package_git_target(target)
repo_url = data.get('repo_url')
provider = data.get('provider')
if not repo_url:
return {"findings": [], "errors": ["package_git target missing repo_url"], "package": data}
logger.info(
f"Scanning package git repo: {repo_url} "
f"({data.get('package_source')}:{data.get('name')}@{data.get('version')})"
)
candidate = normalize_git_repo_candidate(repo_url) or {}
canonical_provider = candidate.get('provider') or provider
provider_token = token if canonical_provider == 'github' else None
preflight = preflight_package_git_repo(data, provider_token, min(int(timeout_sec or 30), 20))
if preflight and preflight.get('skip'):
reason = preflight.get('reason') or 'package_git repo unavailable'
logger.info(f"Skipping package git repo {repo_url}: {reason}")
return {"findings": [], "errors": [], "skipped": reason, "package": data, "scan_meta": {"preflight": preflight}}
scan_provider = (preflight or {}).get('provider') or canonical_provider
scan_token = None if preflight and preflight.get('retry_unauthenticated') else provider_token
result = scan_git_repo(
repo_url,
timeout_sec=timeout_sec,
detectors=detectors,
exclude_detectors=exclude_detectors,
no_verification=no_verification,
trufflehog_config=trufflehog_config,
token=scan_token,
provider=scan_provider,
max_depth=max_depth,
max_commit_age_days=max_commit_age_days,
commit_lookup_pages=commit_lookup_pages,
skip_if_commit_lookup_fails=skip_if_commit_lookup_fails,
)
convert_package_git_unavailable_to_skip(result)
result['package'] = data
return result
def preflight_package_git_repo(data, token=None, timeout=15):
repo_url = data.get('repo_url') or ''
provider = (data.get('provider') or '').lower()
candidate = normalize_git_repo_candidate(repo_url)
if not candidate:
return {'skip': True, 'reason': 'package_git repo URL is unsupported or invalid'}
provider = candidate.get('provider') or provider
repo_path = candidate.get('repo_path') or ''
try:
if provider == 'github':
response = api_request(
'GET',
f'https://api.github.com/repos/{repo_path}',
headers=github_headers(token),
timeout=timeout,
retry_statuses={500, 502, 503, 504},
)
if response.status_code == 200:
return {'skip': False, 'provider': provider, 'repo_path': repo_path}
message = response_message(response).lower()
if response.status_code == 404:
return {'skip': True, 'provider': provider, 'repo_path': repo_path, 'reason': 'package_git repo not found or private'}
if response.status_code in (401, 403) and 'rate limit' not in message:
return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code, 'retry_unauthenticated': True}
return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code}
if provider == 'gitlab':
response = api_request(
'GET',
f'https://gitlab.com/api/v4/projects/{quote(repo_path, safe="")}',
headers=gitlab_headers(token),
timeout=timeout,
retry_statuses={500, 502, 503, 504},
)
if response.status_code == 200:
return {'skip': False, 'provider': provider, 'repo_path': repo_path}
if response.status_code == 404:
return {'skip': True, 'provider': provider, 'repo_path': repo_path, 'reason': 'package_git project not found or private'}
if response.status_code in (401, 403):
return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code, 'retry_unauthenticated': True}
return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code}
except ScanSlotFatalError:
raise
except Exception as e:
logger.warning(f"Package git preflight failed for {repo_url}: {str(e)[:300]}")
return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': 'unknown'}
def postman_stage_filename(target_data):
kind = target_data.get('kind') or 'artifact'
digest = target_data.get('sha256') or target_data.get('sha') or 'postman'
suffix = POSTMAN_COLLECTION_SUFFIX if kind == 'collection' else POSTMAN_ENVIRONMENT_SUFFIX if kind == 'environment' else 'postman.json'
return f'{str(digest)[:16]}.{suffix}'
def is_postman_placeholder(value):
text = str(value or '').strip()
return bool(text and POSTMAN_PLACEHOLDER_RE.match(text))
def load_postman_context(cache_path, payload=None):
max_input_bytes = min(
POSTMAN_JSON_HARD_MAX_INPUT_BYTES,
max(1, int(getattr(scan_config, 'postman_context_max_input_bytes', POSTMAN_JSON_HARD_MAX_INPUT_BYTES))),
)
max_nodes = max(1, int(getattr(scan_config, 'postman_context_max_nodes', 100000)))
max_depth = max(1, int(getattr(scan_config, 'postman_context_max_depth', 64)))
max_scalar_bytes = max(1, int(getattr(scan_config, 'postman_context_max_scalar_bytes', 16 * 1024 * 1024)))
max_items = max(1, int(getattr(scan_config, 'postman_context_max_items', 50000)))
try:
if payload is None:
size = os.path.getsize(cache_path)
if size <= 0 or size > max_input_bytes:
raise PostmanCacheValidationError('Postman context input exceeds its bounded byte limit')
with open(cache_path, 'rb') as f:
payload = f.read(max_input_bytes + 1)
elif not isinstance(payload, bytes):
raise PostmanCacheValidationError('Postman context input must be bytes')
size = len(payload)
if size <= 0 or size > max_input_bytes:
raise PostmanCacheValidationError('Postman context input exceeds its bounded byte limit')
if len(payload) > max_input_bytes:
raise PostmanCacheValidationError('Postman context input exceeds its bounded byte limit')
data = json.loads(payload.decode('utf-8-sig'))
except PostmanCacheValidationError:
raise
except (OSError, UnicodeDecodeError, ValueError, RecursionError) as exc:
raise PostmanCacheValidationError('Postman context is not bounded valid JSON') from exc
contexts = []
traversed = 0
scalar_bytes = 0
def charge(*values):
nonlocal scalar_bytes
scalar_bytes += sum(len(str(value or '').encode('utf-8', errors='replace')) for value in values)
if scalar_bytes > max_scalar_bytes:
raise PostmanCacheValidationError('Postman context scalar byte limit exceeded')
def host_from_url(value):
try:
return urlsplit(str(value)).hostname or ''
except Exception:
return ''
def add_context(path, key, value, endpoint='', auth_type='', location='value'):
if value is None:
return
endpoint = str(endpoint or '')
text = str(value)
charge(path, key, text, endpoint, auth_type, location)
if len(contexts) >= max_items:
raise PostmanCacheValidationError('Postman context item limit exceeded')
contexts.append({
'path': path,
'key': str(key or ''),
'value': text,
'endpoint': endpoint,
'host': host_from_url(endpoint),
'auth_type': str(auth_type or ''),
'location': location,
})
stack = [(data, '$', '', '', 'value', 0)]
while stack:
value, path, endpoint, auth_type, inherited_location, depth = stack.pop()
traversed += 1
if traversed > max_nodes:
raise PostmanCacheValidationError('Postman context traversal item limit exceeded')
if depth > max_depth:
raise PostmanCacheValidationError('Postman context depth limit exceeded')
if len(path.encode('utf-8', errors='replace')) > 4096:
raise PostmanCacheValidationError('Postman context path limit exceeded')
if isinstance(value, dict):
local_endpoint = endpoint
url_value = value.get('url')
if isinstance(url_value, str):
local_endpoint = url_value
elif isinstance(url_value, dict) and url_value.get('raw'):
local_endpoint = str(url_value.get('raw'))
local_auth = auth_type
auth = value.get('auth')
if isinstance(auth, dict):
local_auth = str(auth.get('type') or local_auth or '')
if traversed + len(stack) + len(value) > max_nodes:
raise PostmanCacheValidationError('Postman context traversal item limit exceeded')
children = []
for key, item in value.items():
lower_key = str(key).lower()
location = 'value'
if lower_key in ('header', 'headers'):
location = 'header'
elif lower_key in ('query', 'queryparam', 'query_params'):
location = 'query_param'
elif lower_key in ('body', 'raw'):
location = 'body'
elif lower_key in ('variable', 'values'):
location = 'environment_variable'
child_path = f'{path}.{key}'
charge(key)
children.append((item, child_path, local_endpoint, local_auth, location, depth + 1))
stack.extend(reversed(children))
elif isinstance(value, list):
if traversed + len(stack) + len(value) > max_nodes:
raise PostmanCacheValidationError('Postman context traversal item limit exceeded')
stack.extend(
(value[index], f'{path}[{index}]', endpoint, auth_type, inherited_location, depth + 1)
for index in range(len(value) - 1, -1, -1)
)
else:
add_context(path, '', value, endpoint, auth_type, inherited_location)
return contexts
def provider_from_postman(detector_name='', host='', value=''):
detector = str(detector_name or '').lower()
if detector in ('openai', 'anthropic', 'github', 'gitlab', 'stripe', 'slack'):
return detector
text = ' '.join([str(host or '').lower(), str(value or '').lower()])
if 'api.openai.com' in text or 'sk-proj-' in text or re.search(r'\bsk-[A-Za-z0-9]{20,}', str(value or '')):
return 'openai'
if 'anthropic.com' in text or 'sk-ant-' in text:
return 'anthropic'
if 'generativelanguage.googleapis.com' in text or 'aiplatform.googleapis.com' in text:
return 'google'
if 'huggingface.co' in text or str(value or '').startswith('hf_'):
return 'huggingface'
if 'github.com' in text or str(value or '').startswith(GITHUB_TOKEN_PREFIXES):
return 'github'
if 'gitlab' in text or str(value or '').startswith(GITLAB_TOKEN_PREFIXES):
return 'gitlab'
if 'stripe.com' in text or str(value or '').startswith(('sk_live_', 'rk_live_')):
return 'stripe'
return ''
def credential_kind_from_postman(context, value):
key = str(context.get('key') or '').lower()
auth_type = str(context.get('auth_type') or '').lower()
location = str(context.get('location') or '').lower()
text = str(value or '').strip()
if is_postman_placeholder(text):
return 'placeholder'
if auth_type == 'bearer' or key == 'authorization' or text.lower().startswith('bearer '):
return 'jwt' if re.match(r'^(?:bearer\s+)?eyJ[A-Za-z0-9_-]+\.', text, re.IGNORECASE) else 'bearer_token'
if ('api' in key and 'key' in key) or key in ('x-api-key', 'apikey'):
return 'api_key'
if 'client_secret' in key or 'client-secret' in key:
return 'oauth_client_secret'
if 'password' in key:
return 'basic_auth_password' if auth_type == 'basic' else 'password'
if location == 'query_param' and ('token' in key or 'key' in key):
return 'api_key'
if re.match(r'^eyJ[A-Za-z0-9_-]+\.', text):
return 'jwt'
return 'unknown'
POSTMAN_GEMINI_KEY_RE = re.compile(r'(?:AIza[0-9A-Za-z\-_]{35}|AQ\.[0-9A-Za-z\-_]{50})')
POSTMAN_AZURE_OPENAI_KEY_RE = re.compile(r'\b[a-f0-9]{32}\b', re.IGNORECASE)
POSTMAN_AZURE_OPENAI_ENDPOINT_RE = re.compile(r'([a-z0-9-]+\.openai\.azure\.com)', re.IGNORECASE)
FOUNDRY_ENDPOINT_HOST_RE = r'[a-z0-9-]+(?:\.[a-z0-9-]+)*\.(?:models\.ai\.azure\.com|services\.ai\.azure\.com|inference\.ai\.azure\.com)'
POSTMAN_FOUNDRY_ENDPOINT_RE = re.compile(r'((?:https?://)?' + FOUNDRY_ENDPOINT_HOST_RE + r'(?:/[^\s:"\'<>\\]*)?)', re.IGNORECASE)
FOUNDRY_ASSIGNMENT_RE = re.compile(r'''(?ix)
(?:authorization|api[_-]?key|key|token|secret|credential|bearer)
[^\n:=]{0,80}
[:=]
\s*["']?(?:bearer\s+)?
([A-Za-z0-9_./+=\-]{20,512})
''')
NON_FOUNDRY_KEY_PREFIXES = (
'sk-', 'sk_', 'sk-or-', 'xai-', 'ghp_', 'gho_', 'ghu_', 'ghs_', 'ghr_', 'github_pat_',
'glpat-', 'glrt-', 'hf_', 'AIza', 'AQ.', 'zai-', 'gsk_', 'r8_', 'nvapi-',
)
def keycheck_output_path(service, filename):
root = getattr(scan_config, 'keycheck_dir', None) or os.path.join(get_results_dir() or os.getcwd(), 'keychecks')
directory = os.path.join(root, service)
ensure_private_directory(directory, reject_reparse=True)
return os.path.join(directory, filename)
def _candidate_limits():
return {
'artifact_items': max(1, int(getattr(scan_config, 'keycheck_candidate_artifact_max_items', 2000))),
'artifact_bytes': max(1, int(getattr(scan_config, 'keycheck_candidate_artifact_max_bytes', 2 * 1024 * 1024))),
'file_items': max(1, int(getattr(scan_config, 'keycheck_candidate_file_max_items', 100000))),
'file_bytes': max(1, int(getattr(scan_config, 'keycheck_candidate_file_max_bytes', 32 * 1024 * 1024))),
'line_bytes': max(1, int(getattr(scan_config, 'keycheck_candidate_line_max_bytes', 8192))),
}
class CandidateQueueCapacityError(RuntimeError):
pass
_CANDIDATE_CHECKED_LEDGERS = {
'gem.txt': 'geminiChecked.txt',
'azureOpenAI.txt': 'azureChecked.txt',
'azureFoundry.txt': 'azureChecked.txt',
}
def _candidate_identity(value):
return str(value or '').strip().split('\t', 1)[0].strip()
def _read_bounded_candidate_rows(path, limits, description):
if not os.path.exists(path):
return [], set(), 0, 0
reject_reparse_components(path)
if os.path.getsize(path) > limits['file_bytes']:
raise RuntimeError(f'{description} exceeds its aggregate byte bound: {path}')
rows = []
identities = set()
total_bytes = 0
with open(path, 'rb') as handle:
for raw_line in handle:
total_bytes += len(raw_line)
if total_bytes > limits['file_bytes']:
raise RuntimeError(f'{description} exceeds its aggregate byte bound: {path}')
if len(raw_line) > limits['line_bytes']:
raise RuntimeError(f'{description} line exceeds its byte bound: {path}')
text = raw_line.rstrip(b'\r\n').decode('utf-8', errors='strict')
if not text:
continue
rows.append(text)
if len(rows) > limits['file_items']:
raise RuntimeError(f'{description} exceeds its aggregate item bound: {path}')
identity = _candidate_identity(text)
if identity:
identities.add(identity)
return rows, identities, len(rows), total_bytes
def _candidate_additions(lines, existing_identities, limits, checked_identities=()):
seen = set(existing_identities)
seen.update(checked_identities)
additions = []
added_bytes = 0
for value in lines:
line = str(value or '').strip()
if not line or '\n' in line or '\r' in line:
continue
encoded = (line + '\n').encode('utf-8')
identity = _candidate_identity(line)
if not identity or len(encoded) > limits['line_bytes'] or identity in seen:
continue
additions.append(encoded)
added_bytes += len(encoded)
seen.add(identity)
return additions, added_bytes
def _candidate_batch_fits(existing_items, existing_bytes, additions, added_bytes, limits):
return (
existing_items + len(additions) <= limits['file_items']
and existing_bytes + added_bytes <= limits['file_bytes']
)
def _rewrite_candidate_rows_atomic(path, rows):
temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.tmp'
descriptor = None
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL
if hasattr(os, 'O_BINARY'):
flags |= os.O_BINARY
try:
descriptor = os.open(temporary, flags, 0o600)
os.close(descriptor)
descriptor = None
harden_private_file(temporary)
with open(temporary, 'wb') as handle:
for row in rows:
handle.write((row + '\n').encode('utf-8'))
handle.flush()
os.fsync(handle.fileno())
harden_private_file(temporary)
durable_replace(temporary, path)
harden_private_file(path)
finally:
if descriptor is not None:
os.close(descriptor)
try:
if os.path.exists(temporary):
os.remove(temporary)
except OSError:
pass
def _compact_checked_candidate_rows_unlocked(path, rows, existing_bytes, limits):
ledger_name = _CANDIDATE_CHECKED_LEDGERS.get(os.path.basename(path))
if not ledger_name:
return rows, set(), existing_bytes
checked_path = os.path.join(os.path.dirname(path), ledger_name)
checked_lock_path = f'{checked_path}.lock'
# Candidate locks are always outermost; checked-ledger writers never take them.
checked_lock = acquire_file_lock(checked_lock_path, stale_sec=120, timeout_sec=10)
try:
_, checked_identities, _, _ = _read_bounded_candidate_rows(
checked_path, limits, 'keycheck checked ledger',
)
finally:
release_file_lock(checked_lock, checked_lock_path)
retained = [row for row in rows if _candidate_identity(row) not in checked_identities]
if len(retained) != len(rows):
encoded_sizes = [len((row + '\n').encode('utf-8')) for row in retained]
retained_bytes = sum(encoded_sizes)
if retained_bytes > limits['file_bytes'] or any(
size > limits['line_bytes'] for size in encoded_sizes
):
raise RuntimeError(f'compacted keycheck candidate file would exceed its bounds: {path}')
_rewrite_candidate_rows_atomic(path, retained)
else:
retained_bytes = existing_bytes
return retained, checked_identities, retained_bytes
def _append_unique_lines_unlocked(path, lines, limits):
ensure_private_directory(os.path.dirname(path), reject_reparse=True)
rows, existing, existing_items, existing_bytes = _read_bounded_candidate_rows(
path, limits, 'keycheck candidate file',
)
additions, added_bytes = _candidate_additions(lines, existing, limits)
if not additions:
return 0
if not _candidate_batch_fits(existing_items, existing_bytes, additions, added_bytes, limits):
rows, checked, existing_bytes = _compact_checked_candidate_rows_unlocked(
path, rows, existing_bytes, limits,
)
existing = {_candidate_identity(row) for row in rows if _candidate_identity(row)}
existing_items = len(rows)
additions, added_bytes = _candidate_additions(lines, existing, limits, checked)
if not additions:
return 0
if not _candidate_batch_fits(existing_items, existing_bytes, additions, added_bytes, limits):
raise CandidateQueueCapacityError(
f'keycheck candidate queue has insufficient capacity for the complete offered batch: {path}'
)
with open(path, 'ab') as f:
f.write(b''.join(additions))
f.flush()
os.fsync(f.fileno())
harden_private_file(path)
return len(additions)
def append_unique_lines_locked(path, lines):
limits = _candidate_limits()
lock_path = f'{path}.lock'
lock = acquire_file_lock(lock_path, stale_sec=120, timeout_sec=10)
try:
return _append_unique_lines_unlocked(path, lines, limits)
finally:
release_file_lock(lock, lock_path)
def append_unique_line(path, line):
return bool(append_unique_lines_locked(path, [line]))
def append_unique_line_locked(path, line):
return bool(append_unique_lines_locked(path, [line]))
def _collect_candidate_line(batches, seen, budget, path, line):
limits = budget['limits']
text = str(line or '').strip()
encoded_size = len((text + '\n').encode('utf-8')) if text else 0
if not text or encoded_size > limits['line_bytes'] or text in seen.setdefault(path, set()):
return False
if budget['items'] >= limits['artifact_items'] or budget['bytes'] + encoded_size > limits['artifact_bytes']:
budget['truncated'] = True
return False
seen[path].add(text)
batches.setdefault(path, []).append(text)
budget['items'] += 1
budget['bytes'] += encoded_size
return True
def normalize_foundry_endpoint(value):
text = str(value or '').strip().strip('"\'`,;')
if not text:
return ''
split_text = text if re.match(r'(?i)^https?://', text) else 'https://' + text
try:
parsed = urlsplit(split_text)
host = parsed.netloc or parsed.path.split('/', 1)[0]
path = parsed.path if parsed.netloc else ('/' + parsed.path.split('/', 1)[1] if '/' in parsed.path else '')
except Exception:
host, path = re.sub(r'(?i)^https?://', '', text).split('/', 1)[0], ''
path = path.rstrip('.,;:)]}/')
terminal_routes = (
('/models/chat/completions', ''),
('/openai/v1/chat/completions', '/openai/v1'),
('/v1/chat/completions', '/v1'),
('/chat/completions', ''),
('/v1/models', '/v1'),
('/models', ''),
)
lower_path = path.lower()
for suffix, replacement in terminal_routes:
if lower_path.endswith(suffix):
path = path[:-len(suffix)] + replacement
break
return (host + path).strip('/').lower()
def dedupe_foundry_endpoints(endpoints):
normalized = []
for endpoint in endpoints or []:
endpoint = normalize_foundry_endpoint(endpoint)
if endpoint and endpoint not in normalized:
normalized.append(endpoint)
kept = []
for endpoint in sorted(normalized, key=len, reverse=True):
if any(other.startswith(endpoint + '/') for other in kept):
continue
kept.append(endpoint)
return list(reversed(kept))
def foundry_keyish(value):
text = re.sub(r'(?i)^bearer\s+', '', str(value or '').strip().strip('"\'`,;')).strip()
lower = text.lower()
if not (20 <= len(text) <= 512):
return False
if any(marker in lower for marker in ('http://', 'https://', '{{', '${', '<', 'azure.com')):
return False
if any(ch.isspace() for ch in text):
return False
if re.match(r'(?i)^(?:authorization|api[_-]?key|key|token|secret|credential|bearer)\s*[:=]', text):
return False
if text.startswith(NON_FOUNDRY_KEY_PREFIXES):
return False
return bool(re.search(r'[A-Za-z]', text) and re.search(r'[0-9]', text))
def is_foundry_detector(finding):
detector = str((finding or {}).get('DetectorName') or '').lower()
extra = (finding or {}).get('ExtraData') if isinstance((finding or {}).get('ExtraData'), dict) else {}
name = str(extra.get('name') or '').lower()
return detector.startswith('azurefoundry') or (detector == 'customregex' and name.startswith('azurefoundry'))
def foundry_finding_text(finding):
parts = []
for value in finding_raw_values(finding):
parts.append(value)
if is_foundry_detector(finding):
return '\n'.join(part for part in parts if part)
context = finding.get('ScannerContext') if isinstance(finding, dict) else None
if isinstance(context, dict):
parts.extend(str(context.get(item) or '') for item in ('nearby', 'file'))
postman_context = finding.get('PostmanContext') if isinstance(finding, dict) else None
if isinstance(postman_context, dict):
parts.extend(str(postman_context.get(item) or '') for item in ('endpoint', 'host', 'variable_name'))
extra = finding.get('ExtraData') if isinstance(finding, dict) else None
if isinstance(extra, dict):
parts.extend(str(value) for value in extra.values() if isinstance(value, str))
return '\n'.join(part for part in parts if part)
def foundry_candidate_keys(finding, text):
keys = []
if is_foundry_detector(finding):
for value in finding_raw_values(finding):
if foundry_keyish(value) and value not in keys:
keys.append(value.strip().strip('"\'`,;'))
for match in FOUNDRY_ASSIGNMENT_RE.findall(text or ''):
candidate = re.sub(r'(?i)^bearer\s+', '', str(match or '').strip().strip('"\'`,;')).strip()
if foundry_keyish(candidate) and candidate not in keys:
keys.append(candidate)
return keys[:5]
def foundry_candidate_origin(result, finding):
file_path, line_number = finding_source_location(finding)
location = f'{file_path}:{line_number}' if file_path and line_number else file_path or ''
target = result.get('target') or ''
scan_type = result.get('scan_type') or ''
return ':'.join(part for part in (scan_type, str(target), location) if part)
def write_foundry_keycheck_candidates_from_findings(result):
findings = result.get('findings') or []
if not findings:
return 0
azure_foundry_path = keycheck_output_path('azure', 'azureFoundry.txt')
batches = {}
seen = {}
budget = {'items': 0, 'bytes': 0, 'truncated': False, 'limits': _candidate_limits()}
for finding in findings:
if not is_foundry_detector(finding):
continue
text = foundry_finding_text(finding)
endpoints = []
for match in POSTMAN_FOUNDRY_ENDPOINT_RE.findall(text):
endpoint = normalize_foundry_endpoint(match)
if endpoint and endpoint not in endpoints:
endpoints.append(endpoint)
if not endpoints:
continue
endpoints = dedupe_foundry_endpoints(endpoints)
keys = foundry_candidate_keys(finding, text)
if not keys:
continue
origin = foundry_candidate_origin(result, finding)
for endpoint in endpoints[:5]:
for key in keys[:5]:
metadata = json.dumps({
'origin': origin,
'finding_uid': finding.get('finding_uid') or '',
}, ensure_ascii=False, separators=(',', ':'))
_collect_candidate_line(
batches, seen, budget, azure_foundry_path,
f'{endpoint}:{key}\t{metadata}',
)
if budget['truncated']:
break
if budget['truncated']:
break
if budget['truncated']:
break
if budget['truncated']:
logger.warning('Azure Foundry candidates reached the per-artifact bound; remaining values were capped')
return append_unique_lines_locked(azure_foundry_path, batches.get(azure_foundry_path, []))
def artifact_origin_label(target_data, cache_path):
origin = target_data.get('origin') if isinstance(target_data.get('origin'), dict) else {}
source = target_data.get('source') or origin.get('provider') or 'artifact'
repo = target_data.get('repo') or origin.get('repo') or ''
path = target_data.get('path') or origin.get('path') or os.path.basename(cache_path or '')
sha = target_data.get('sha') or origin.get('sha') or target_data.get('sha256') or ''
return f'{source}:{repo}:{path}:{sha}'
def context_values_for_pairing(contexts):
endpoints = []
values = []
for context in contexts or []:
value = str(context.get('value') or '').strip()
if not value or is_postman_placeholder(value):
continue
text = ' '.join([value, str(context.get('endpoint') or ''), str(context.get('host') or ''), str(context.get('key') or '')])
values.append((context, value, text))
for pattern in (POSTMAN_AZURE_OPENAI_ENDPOINT_RE, POSTMAN_FOUNDRY_ENDPOINT_RE):
for match in pattern.findall(text):
endpoint = normalize_foundry_endpoint(match) if pattern is POSTMAN_FOUNDRY_ENDPOINT_RE else str(match).lower().strip('/')
if endpoint not in endpoints:
endpoints.append(endpoint)
return endpoints, values
def write_structured_keycheck_candidates(cache_path, target_data):
contexts = load_postman_context(cache_path)
if not contexts:
return {}
context_value_by_path = {str(item.get('path') or ''): str(item.get('value') or '') for item in contexts}
endpoints, values = context_values_for_pairing(contexts)
origin = artifact_origin_label(target_data or {}, cache_path)
counts = {'gemini': 0, 'azure_openai': 0, 'azure_foundry': 0}
gemini_path = keycheck_output_path('gemini', 'gem.txt')
azure_openai_path = keycheck_output_path('azure', 'azureOpenAI.txt')
azure_foundry_path = keycheck_output_path('azure', 'azureFoundry.txt')
batches = {}
seen = {}
budget = {'items': 0, 'bytes': 0, 'truncated': False, 'limits': _candidate_limits()}
labels = {
gemini_path: 'gemini',
azure_openai_path: 'azure_openai',
azure_foundry_path: 'azure_foundry',
}
for context, value, text in values:
for key in POSTMAN_GEMINI_KEY_RE.findall(value):
_collect_candidate_line(
batches, seen, budget, gemini_path,
f'{key}\t{origin}\t{context.get("path") or ""}',
)
azure_key_match = POSTMAN_AZURE_OPENAI_KEY_RE.search(value)
if azure_key_match:
local_azure = [str(match).lower().strip('/') for match in POSTMAN_AZURE_OPENAI_ENDPOINT_RE.findall(text)]
azure_endpoints = local_azure or [endpoint for endpoint in endpoints if POSTMAN_AZURE_OPENAI_ENDPOINT_RE.fullmatch(endpoint)]
for endpoint in azure_endpoints[:5]:
_collect_candidate_line(
batches, seen, budget, azure_openai_path,
f'{endpoint}:{azure_key_match.group(0)}\t{origin}\t{context.get("path") or ""}',
)
sibling_name = ''
path = str(context.get('path') or '')
if path.endswith('.value'):
sibling_name = context_value_by_path.get(path[:-6] + '.key', '')
key_context = ' '.join([str(context.get('key') or ''), sibling_name]).lower()
foundry_value = re.sub(r'(?i)^bearer\s+', '', value.strip().strip('"\'`,;')).strip()
if foundry_keyish(foundry_value) and any(word in key_context for word in ('authorization', 'bearer', 'key', 'token', 'secret', 'api')):
foundry_endpoints = dedupe_foundry_endpoints(POSTMAN_FOUNDRY_ENDPOINT_RE.findall(text))
for endpoint in foundry_endpoints[:5]:
if endpoint:
_collect_candidate_line(
batches, seen, budget, azure_foundry_path,
f'{endpoint}:{foundry_value}\t{origin}\t{context.get("path") or ""}',
)
if budget['truncated']:
break
if budget['truncated']:
logger.warning('Structured keycheck candidates reached the per-artifact bound; remaining values were capped')
for path, lines in batches.items():
counts[labels[path]] = append_unique_lines_locked(path, lines)
return {key: value for key, value in counts.items() if value}
def _postman_substring_match(budget, needle, value):
if _context_budget_expired(budget):
return None
if budget['postman_comparisons'] >= budget['max_postman_comparisons']:
return None
budget['postman_comparisons'] += 1
return needle in value
def attach_postman_context(results, cache_path, budget=None):
budget = budget or context_enrichment_budget()
results.pop('structured_keycheck_pending', None)
try:
payload, complete, reason = _read_context_source(cache_path, budget)
if payload is None or not complete:
if reason in ('elapsed', 'source_bytes'):
dimension = 'elapsed-time' if reason == 'elapsed' else 'source-byte'
_add_context_warning(results, 'budget', f'{dimension} budget was exhausted; findings were retained')
else:
_add_context_warning(results, 'failure', 'Postman context could not be read; findings were retained')
return results
contexts = load_postman_context(cache_path, payload=payload)
except Exception:
logger.warning('Optional Postman context parsing failed; parsed findings were retained')
_add_context_warning(results, 'failure', 'Postman JSON context parsing failed; findings were retained')
return results
results['structured_keycheck_pending'] = True
if not contexts:
return results
try:
placeholder_count = 0
context_values = []
exact_values = {}
for index, context in enumerate(contexts):
if index % 256 == 0 and _context_budget_expired(budget):
_add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; findings were retained')
return results
if not isinstance(context, dict):
continue
value = str(context.get('value') or '')
context_values.append((context, value))
if value:
exact_values.setdefault(value, context)
placeholder_count += int(is_postman_placeholder(value))
results['postman_context'] = {
'context_count': len(contexts),
'placeholder_count': placeholder_count,
}
for finding in results.get('findings') or []:
if _context_budget_expired(budget):
_add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained')
return results
if not isinstance(finding, dict):
continue
if not _context_budget_claim_finding(budget, finding):
_add_context_warning(results, 'budget', 'finding-count budget was exhausted; remaining findings were retained')
return results
raw_values = finding_raw_values(finding)
matched = next((exact_values.get(raw) for raw in raw_values if raw and exact_values.get(raw)), None)
for raw in raw_values if not matched else ():
if not raw:
continue
for context, value in context_values:
comparison = _postman_substring_match(budget, raw, value)
if comparison is None:
_add_context_warning(results, 'budget', 'Postman comparison budget was exhausted; remaining findings were retained')
return results
if comparison:
matched = context
break
if matched:
break
if not matched and raw_values and raw_values[0][:8]:
prefix = raw_values[0][:8]
for context, value in context_values:
comparison = _postman_substring_match(budget, prefix, value)
if comparison is None:
_add_context_warning(results, 'budget', 'Postman comparison budget was exhausted; remaining findings were retained')
return results
if comparison:
matched = context
break
if not matched:
continue
raw_value = raw_values[0] if raw_values else matched.get('value')
kind = credential_kind_from_postman(matched, raw_value)
confidence = 'verified' if finding.get('Verified') else 'detector_match' if finding.get('DetectorName') else 'structured_complete' if kind != 'unknown' else 'context_only'
if kind == 'placeholder':
confidence = 'placeholder'
endpoint = sanitize_endpoint(matched.get('endpoint'))
host = sanitize_endpoint_host(endpoint) or sanitize_endpoint_host(matched.get('host'))
finding['PostmanContext'] = {
'provider': provider_from_postman(finding.get('DetectorName'), host, raw_value),
'credential_kind': kind,
'credential_confidence': confidence,
'context_location': matched.get('location'),
'variable_name': matched.get('key'),
'endpoint': endpoint,
'host': host,
'auth_type': matched.get('auth_type'),
'json_path': matched.get('path'),
'placeholder': is_postman_placeholder(matched.get('value')),
}
except Exception:
logger.warning('Optional Postman context matching failed; parsed findings were retained')
_add_context_warning(results, 'failure', 'Postman context matching failed; findings were retained')
return results
def scan_postman_target(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, max_artifact_size_mb=20, token=None):
data = parse_postman_target(target)
cache_path = data.get('cache_path') or data.get('local_path')
if not cache_path:
return {"findings": [], "errors": ["Postman target missing cached artifact"], "package": data.get('origin') or data}
try:
cache_path, declared_size = validate_postman_cache_artifact(data, max_artifact_size_mb)
except PostmanCacheTooLarge as exc:
return {"findings": [], "errors": [], "skipped": str(exc), "package": data.get('origin') or data}
except (OSError, PostmanCacheValidationError) as exc:
return {"findings": [], "errors": [f"Postman cache path rejected: {exc}"], "package": data.get('origin') or data}
_raise_if_scan_slot_fatal()
work_dir = create_command_work_dir()
_raise_if_scan_slot_fatal()
if not work_dir:
return {"findings": [], "errors": ["Unable to create Postman work dir"], "package": data.get('origin') or data}
try:
staged_path = os.path.join(work_dir, postman_stage_filename(data))
_raise_if_scan_slot_fatal()
shutil.copyfile(cache_path, staged_path)
_raise_if_scan_slot_fatal()
harden_private_file(staged_path)
cmd = [get_trufflehog_cmd(), 'filesystem', work_dir, '--json', '--no-update']
append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config)
results = {
"findings": [], "errors": [], "package": data.get('origin') or data,
"postman": data, "bytes": declared_size,
"postman_max_artifact_size_mb": int(max_artifact_size_mb or 0),
}
with run_command_streamed(cmd, timeout_sec) as output:
apply_trufflehog_diagnostics(results, output, output.returncode, 'postman')
append_trufflehog_findings(results, output.stdout_lines())
enrichment_budget = context_enrichment_budget()
attach_nearby_context(results, enrichment_budget)
artifact_path = str(data.get('path') or data.get('name') or data.get('url') or '').split('?', 1)[0].lower()
if str(data.get('kind') or '').lower() != 'bruno' or not artifact_path.endswith('.bru'):
attach_postman_context(results, staged_path, enrichment_budget)
return apply_finding_filters(results, cache_path)
except ScanSlotFatalError:
raise
except Exception as e:
return {"findings": [], "errors": [f"Postman scan failed: {str(e)}"], "package": data.get('origin') or data}
finally:
cleanup_command_work_dir(work_dir)
def safe_path_component(value, max_len=120):
text = str(value or '')[:max_len]
return re.sub(r'[^A-Za-z0-9_.-]+', '_', text).strip('._') or 'item'
def run_filesystem_scan(root_dir, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None):
cmd = [get_trufflehog_cmd(), 'filesystem', root_dir, '--json', '--no-update']
append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config)
results = {"findings": [], "errors": []}
with run_command_streamed(cmd, timeout_sec) as output:
apply_trufflehog_diagnostics(results, output, output.returncode, 'filesystem')
append_trufflehog_findings(results, output.stdout_lines())
attach_nearby_context(results)
return apply_finding_filters(results, root_dir)
CI_ARTIFACT_ALLOWED_SUFFIXES = {
'.env', '.log', '.txt', '.json', '.yaml', '.yml', '.xml', '.html', '.lcov', '.sarif',
'.tfstate', '.tfplan', '.py', '.js', '.jsx', '.ts', '.tsx', '.sh', '.toml', '.ini',
'.cfg', '.conf', '.properties', '.ipynb', '.md', '.out', '.err', '.csv', '.tsv',
}
CI_ARTIFACT_ALLOWED_NAMES = {'dockerfile', 'makefile', 'procfile'}
CI_ARTIFACT_SKIP_DIRS = {
'.git', '.hg', '.svn', 'node_modules', '__pycache__', '.venv', 'venv', 'env',
'.mypy_cache', '.pytest_cache', '.tox', '.gradle', '.idea', '.vscode',
}
def ci_artifact_file_allowed(name, allowed_suffixes=None):
if allowed_suffixes is None:
return True
normalized = name.replace('\\', '/').strip('/')
parts = [part.lower() for part in normalized.split('/') if part]
if any(part in CI_ARTIFACT_SKIP_DIRS for part in parts[:-1]):
return False
base = parts[-1] if parts else ''
if base in CI_ARTIFACT_ALLOWED_NAMES or base.startswith('.env'):
return True
return any(base.endswith(suffix) for suffix in allowed_suffixes)
def safe_extract_zip(zip_path, destination, max_file_size_mb=20, max_files=1000, allowed_suffixes=None, max_total_size_mb=250):
_raise_if_scan_slot_fatal()
extracted = 0
max_bytes = int(max_file_size_mb or 0) * 1024 * 1024
max_total_bytes = int(max_total_size_mb or 0) * 1024 * 1024
total_bytes = 0
validate_zip_central_directory(zip_path, max_files)
with zipfile.ZipFile(zip_path) as archive:
for info in archive.infolist():
_raise_if_scan_slot_fatal()
if info.is_dir():
continue
if max_files and extracted >= max_files:
raise ValueError(f'Zip archive exceeds {max_files} extracted files')
if max_bytes and info.file_size > max_bytes:
raise ValueError(f'Zip member exceeds {max_file_size_mb} MB: {info.filename}')
if max_total_bytes and total_bytes + max(0, int(info.file_size or 0)) > max_total_bytes:
raise ValueError(f'Zip archive exceeds {max_total_size_mb} MB extracted')
name = info.filename.replace('\\', '/')
validate_archive_member_name(name)
if not ci_artifact_file_allowed(name, allowed_suffixes):
continue
target_path = os.path.abspath(os.path.normpath(os.path.join(destination, name)))
if os.path.commonpath([os.path.abspath(destination), target_path]) != os.path.abspath(destination):
continue
os.makedirs(os.path.dirname(target_path), exist_ok=True)
with archive.open(info) as src, open(target_path, 'wb') as dst:
while True:
_raise_if_scan_slot_fatal()
chunk = src.read(256 * 1024)
if not chunk:
break
dst.write(chunk)
_raise_if_scan_slot_fatal()
extracted += 1
total_bytes += max(0, int(info.file_size or 0))
return extracted
def remove_file_quiet(path):
try:
reject_reparse_components(path)
os.remove(path)
except OSError:
pass
@dataclass(frozen=True)
class DownloadOutcome:
path: str
status_code: int
bytes_written: int
error: str = ''
@property
def ok(self):
return bool(self.path and not self.error)
def download_to_file(
url, destination, headers=None, timeout=20, max_size_mb=0,
max_size_bytes=None, byte_budget=None,
):
_raise_if_scan_slot_fatal()
destination = os.path.abspath(destination)
require_private_directory(os.path.dirname(destination), create=False)
reject_reparse_components(os.path.dirname(destination))
current_url = str(url)
current_headers = dict(headers or {})
configured_max = int(max_size_mb or 0) * 1024 * 1024
explicit_max = max(0, int(max_size_bytes or 0))
max_bytes = min(value for value in (configured_max, explicit_max) if value > 0) if configured_max and explicit_max else configured_max or explicit_max
if byte_budget is not None and int(byte_budget.get('remaining', 0)) <= 0:
return DownloadOutcome('', 0, 0, 'target byte budget exhausted')
for _ in range(6):
_raise_if_scan_slot_fatal()
try:
response = _direct_request(
'GET', current_url, headers=current_headers, timeout=timeout,
allow_redirects=False, stream=True,
)
except requests.RequestException as exc:
return DownloadOutcome('', 0, 0, str(exc))
if _scan_slot_fatal_event.is_set():
response.close()
_raise_if_scan_slot_fatal()
if response.status_code in (301, 302, 303, 307, 308) and response.headers.get('Location'):
next_url = urljoin(current_url, response.headers['Location'])
old_host = (urlsplit(current_url).hostname or '').lower()
parsed_next = urlsplit(next_url)
response.close()
if parsed_next.scheme != 'https':
return DownloadOutcome('', 0, 0, f'unsafe redirect scheme: {parsed_next.scheme}')
if (parsed_next.hostname or '').lower() != old_host:
current_headers = {
key: value for key, value in current_headers.items()
if key.lower() not in ('authorization', 'private-token', 'cookie', 'proxy-authorization')
}
current_url = next_url
continue
if response.status_code >= 400:
try:
prefix = bytearray()
for chunk in response.iter_content(chunk_size=300):
_raise_if_scan_slot_fatal()
prefix.extend(chunk[:max(0, 300 - len(prefix))])
if len(prefix) >= 300:
break
message = bytes(prefix).decode(response.encoding or 'utf-8', errors='replace')
finally:
response.close()
return DownloadOutcome('', response.status_code, 0, message)
content_length = response.headers.get('Content-Length')
if content_length:
try:
declared = int(content_length)
remaining = int(byte_budget.get('remaining', 0)) if byte_budget is not None else 0
if max_bytes and declared > max_bytes:
response.close()
return DownloadOutcome('', response.status_code, 0, f'response exceeds configured {max_bytes}-byte limit')
if byte_budget is not None and declared > remaining:
response.close()
return DownloadOutcome('', response.status_code, 0, 'target byte budget exhausted')
except ValueError:
pass
total = 0
temporary = f'{destination}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.partial'
descriptor = None
try:
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0)
descriptor = os.open(temporary, flags, 0o600)
os.close(descriptor)
descriptor = None
harden_private_file(temporary)
with open(temporary, 'wb', buffering=0) as output:
for chunk in response.iter_content(chunk_size=256 * 1024):
_raise_if_scan_slot_fatal()
if not chunk:
continue
if max_bytes and total + len(chunk) > max_bytes:
raise CommandOutputLimitError(f'object response exceeds configured {max_bytes}-byte limit')
if byte_budget is not None:
remaining = int(byte_budget.get('remaining', 0))
if len(chunk) > remaining:
byte_budget['remaining'] = 0
raise CommandOutputLimitError('target byte budget exhausted')
byte_budget['remaining'] = remaining - len(chunk)
output.write(chunk)
total += len(chunk)
output.flush()
os.fsync(output.fileno())
harden_private_file(temporary)
durable_replace(temporary, destination)
if not private_file_ready(destination):
raise OSError('download destination lost its private file identity')
return DownloadOutcome(destination, response.status_code, total)
except (CommandOutputLimitError, OSError, requests.RequestException) as exc:
return DownloadOutcome('', response.status_code, total, str(exc))
finally:
response.close()
if descriptor is not None:
os.close(descriptor)
if os.path.exists(temporary):
try:
os.remove(temporary)
except OSError:
pass
return DownloadOutcome('', 0, 0, 'too many redirects')
def github_api_get(url, token=None, timeout=20, stream=False):
response = api_request(
'GET', url, headers=github_headers(token), timeout=timeout, stream=stream,
)
if token and response.status_code in (401, 403):
response.close()
anonymous = api_request(
'GET', url, headers=github_headers(None), timeout=timeout, stream=stream,
)
if anonymous.status_code < 400:
return anonymous
response = anonymous
if response.status_code >= 400:
raise github_api_error(response)
return response
def bounded_response_json(response, max_bytes=8 * 1024 * 1024):
max_bytes = max(1, int(max_bytes))
declared = response.headers.get('Content-Length')
if declared:
try:
declared = int(declared)
except ValueError as exc:
response.close()
raise ApiRequestError('API JSON response has an invalid Content-Length') from exc
if declared > max_bytes:
response.close()
raise ApiRequestError(f'API JSON response exceeds {max_bytes} bytes')
payload = bytearray()
try:
for chunk in response.iter_content(chunk_size=64 * 1024):
if not chunk:
continue
if len(payload) + len(chunk) > max_bytes:
raise ApiRequestError(f'API JSON response exceeds {max_bytes} bytes')
payload.extend(chunk)
return json.loads(bytes(payload).decode('utf-8', errors='strict'))
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
raise ApiRequestError('API response is not bounded valid UTF-8 JSON') from exc
finally:
response.close()
def select_ci_runs(runs, runs_per_repo=5, lookback_days=30, failed_first=True):
cutoff = datetime.now(timezone.utc) - timedelta(days=int(lookback_days or 0)) if int(lookback_days or 0) > 0 else None
kept = []
for run in runs or []:
created = parse_postman_time(run.get('created_at'))
if cutoff and created and created < cutoff:
continue
kept.append(run)
def created_ts(item):
parsed = parse_postman_time(item.get('created_at'))
return parsed.timestamp() if parsed else 0
if failed_first:
kept.sort(key=lambda item: (0 if item.get('conclusion') not in ('success', None) else 1, -created_ts(item)))
else:
kept.sort(key=lambda item: -created_ts(item))
return kept[:max(1, int(runs_per_repo or 5))]
def scan_github_actions_repo(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, ci_runs_per_repo=5, ci_lookback_days=30, ci_max_log_archive_mb=50, ci_max_log_file_mb=20, ci_failed_first=True, ci_scan_artifacts=False, ci_max_artifacts_per_run=3, ci_max_artifact_archive_mb=50, ci_max_artifact_file_mb=10, ci_max_artifact_files=1000, ci_target_max_download_mb=500, fetch_timeout=20):
repo, repo_url = parse_github_repo_target(target)
if not repo:
return {"findings": [], "errors": ["Unable to parse GitHub repo target"]}
logger.info(f"Scanning GitHub Actions logs: {repo}")
_raise_if_scan_slot_fatal()
work_dir = create_command_work_dir()
_raise_if_scan_slot_fatal()
if not work_dir:
return {"findings": [], "errors": ["Unable to create GitHub Actions work dir"]}
try:
runs_url = f'https://api.github.com/repos/{repo}/actions/runs?per_page={max(1, min(100, int(ci_runs_per_repo or 5) * 3))}'
response = github_api_get(runs_url, token, fetch_timeout, stream=True)
payload = bounded_response_json(response)
if not isinstance(payload, dict) or 'workflow_runs' not in payload or not isinstance(payload.get('workflow_runs'), list):
raise ApiRequestError('invalid GitHub Actions runs payload')
runs = select_ci_runs(payload.get('workflow_runs') or [], ci_runs_per_repo, ci_lookback_days, ci_failed_first)
if not runs:
return {"findings": [], "errors": [], "skipped": "no recent workflow runs", "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url}}
max_archive_bytes = int(ci_max_log_archive_mb or 0) * 1024 * 1024
max_artifact_archive_bytes = int(ci_max_artifact_archive_mb or 0) * 1024 * 1024
extracted_total = 0
artifact_extracted_total = 0
download_failures = []
downloaded_total = 0
target_download_limit = int(ci_target_max_download_mb or 0) * 1024 * 1024
download_budget = {'remaining': target_download_limit} if target_download_limit else None
run_meta = []
for run in runs:
_raise_if_scan_slot_fatal()
if download_budget is not None and download_budget['remaining'] <= 0:
download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap')
break
run_id = run.get('id')
if not run_id:
continue
extracted = 0
artifacts_meta = []
logs_url = f'https://api.github.com/repos/{repo}/actions/runs/{run_id}/logs'
zip_path = os.path.join(work_dir, f'github_actions_{safe_path_component(repo)}_{run_id}.zip')
authenticated_status = None
try:
download = download_to_file(
logs_url, zip_path, github_headers(token), fetch_timeout, ci_max_log_archive_mb,
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
byte_budget=download_budget,
)
authenticated_status = download.status_code
except requests.RequestException as exc:
download = DownloadOutcome('', 0, 0, str(exc))
if not download.ok and token and download.status_code in (401, 403):
anonymous = download_to_file(
logs_url, zip_path, github_headers(None), fetch_timeout, ci_max_log_archive_mb,
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
byte_budget=download_budget,
)
if not anonymous.ok:
download = DownloadOutcome(
'', authenticated_status or anonymous.status_code, 0,
f'authenticated HTTP {authenticated_status}; anonymous HTTP '
f'{anonymous.status_code}: {anonymous.error}',
)
else:
download = anonymous
if download.ok:
downloaded_total += download.bytes_written
run_dir = os.path.join(work_dir, 'github_actions', safe_path_component(repo), str(run_id))
ensure_private_directory(run_dir, reject_reparse=True)
try:
extracted = safe_extract_zip(
zip_path, run_dir, ci_max_log_file_mb,
max_total_size_mb=ci_max_log_archive_mb,
)
harden_private_tree(run_dir)
except (zipfile.BadZipFile, ValueError) as exc:
logger.warning(f"GitHub Actions logs archive rejected for {repo} run {run_id}: {exc}")
download_failures.append(f'log archive rejected: {exc}')
extracted = 0
remove_file_quiet(zip_path)
else:
if download.status_code not in (404, 410):
download_failures.append(f'log download HTTP {download.status_code}: {download.error}')
run_dir = os.path.join(work_dir, 'github_actions', safe_path_component(repo), str(run_id))
ensure_private_directory(run_dir, reject_reparse=True)
if ci_scan_artifacts:
artifacts_url = f'https://api.github.com/repos/{repo}/actions/runs/{run_id}/artifacts?per_page={max(1, min(100, int(ci_max_artifacts_per_run or 3)))}'
try:
artifacts_response = github_api_get(
artifacts_url, token, fetch_timeout, stream=True,
)
artifact_payload = bounded_response_json(artifacts_response)
if not isinstance(artifact_payload, dict) or 'artifacts' not in artifact_payload or not isinstance(artifact_payload.get('artifacts'), list):
raise ApiRequestError('invalid GitHub Actions artifacts payload')
artifacts = artifact_payload.get('artifacts') or []
except ScanSlotFatalError:
raise
except Exception as exc:
logger.warning(f"Unable to list GitHub Actions artifacts for {repo} run {run_id}: {exc}")
download_failures.append(f'artifact listing failed: {exc}')
artifacts = []
for artifact in artifacts[:max(0, int(ci_max_artifacts_per_run or 3))]:
_raise_if_scan_slot_fatal()
if download_budget is not None and download_budget['remaining'] <= 0:
download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap')
break
if artifact.get('expired'):
continue
size = int(artifact.get('size_in_bytes') or 0)
if max_artifact_archive_bytes and size and size > max_artifact_archive_bytes:
download_failures.append(f'artifact declared size exceeds {ci_max_artifact_archive_mb} MB')
continue
download_url = artifact.get('archive_download_url')
if not download_url:
continue
artifact_id = artifact.get('id') or safe_path_component(artifact.get('name'))
artifact_zip = os.path.join(work_dir, f'github_actions_artifact_{safe_path_component(repo)}_{run_id}_{artifact_id}.zip')
download = download_to_file(
download_url, artifact_zip, github_headers(token), fetch_timeout, ci_max_artifact_archive_mb,
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
byte_budget=download_budget,
)
authenticated_status = download.status_code
if not download.ok and token and download.status_code in (401, 403):
anonymous = download_to_file(
download_url, artifact_zip, github_headers(None), fetch_timeout, ci_max_artifact_archive_mb,
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
byte_budget=download_budget,
)
if not anonymous.ok:
download = DownloadOutcome(
'', authenticated_status or anonymous.status_code, 0,
f'authenticated HTTP {authenticated_status}; anonymous HTTP '
f'{anonymous.status_code}: {anonymous.error}',
)
else:
download = anonymous
if not download.ok:
if download.status_code not in (404, 410):
logger.warning(f"GitHub Actions artifact download HTTP {download.status_code} for {repo} run {run_id}: {download.error}")
download_failures.append(f'artifact download HTTP {download.status_code}: {download.error}')
continue
downloaded_total += download.bytes_written
artifact_dir = os.path.join(run_dir, 'artifacts', safe_path_component(artifact.get('name') or artifact_id))
ensure_private_directory(artifact_dir, reject_reparse=True)
try:
artifact_files = safe_extract_zip(
artifact_zip,
artifact_dir,
ci_max_artifact_file_mb,
ci_max_artifact_files,
CI_ARTIFACT_ALLOWED_SUFFIXES,
max_total_size_mb=ci_max_artifact_archive_mb,
)
harden_private_tree(artifact_dir)
except (zipfile.BadZipFile, ValueError) as exc:
logger.warning(f"GitHub Actions artifact rejected for {repo} run {run_id}: {artifact.get('name')}: {exc}")
download_failures.append(f'artifact archive rejected: {exc}')
artifact_files = 0
remove_file_quiet(artifact_zip)
artifact_extracted_total += artifact_files
artifacts_meta.append({
'artifact_id': artifact.get('id'),
'name': artifact.get('name'),
'size_in_bytes': size,
'expired': artifact.get('expired'),
'extracted_files': artifact_files,
})
extracted_total += extracted
run_meta.append({
'run_id': run_id,
'run_number': run.get('run_number'),
'workflow_name': run.get('name'),
'status': run.get('status'),
'conclusion': run.get('conclusion'),
'created_at': run.get('created_at'),
'updated_at': run.get('updated_at'),
'extracted_files': extracted,
'artifacts': artifacts_meta,
})
if extracted_total == 0 and artifact_extracted_total == 0:
if download_failures:
text = '; '.join(download_failures[:5])
all_failures = '; '.join(download_failures)
auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any(
token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit')
)
return {
"findings": [], "errors": [text], "error_class": 'source_auth' if auth_failure else 'remote_transient',
"retryable": True, "source_failure": auth_failure,
"source_failure_category": 'auth_forbidden' if auth_failure else 'remote_transient',
"source_failure_auth_related": auth_failure,
"package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url, "runs": run_meta},
}
return {"findings": [], "errors": [], "skipped": "no downloadable workflow logs or artifacts", "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url, "runs": run_meta}}
_raise_if_scan_slot_fatal()
results = run_filesystem_scan(work_dir, timeout_sec, detectors, exclude_detectors, no_verification, trufflehog_config)
results['package'] = {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url, "runs": run_meta, "log_files": extracted_total, "artifact_files": artifact_extracted_total}
if download_failures:
text = '; '.join(download_failures[:5])
all_failures = '; '.join(download_failures)
auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any(
token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit')
)
results['errors'] = list(results.get('errors') or []) + [text]
results['error_class'] = 'source_auth' if auth_failure else 'remote_transient'
results['retryable'] = True
results['source_failure'] = auth_failure
results['source_failure_category'] = 'auth_forbidden' if auth_failure else 'remote_transient'
results['source_failure_auth_related'] = auth_failure
return results
except ScanSlotFatalError:
raise
except RateLimitError as exc:
category = getattr(exc, 'category', '')
if category == 'not_found':
return {"findings": [], "errors": [], "skipped": str(exc), "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url}}
return {
"findings": [], "errors": [str(exc)], "error_class": 'source_auth' if getattr(exc, 'auth_related', True) else 'remote_transient',
"retryable": True, "source_failure": True,
"source_failure_category": category or 'rate_limit',
"source_failure_auth_related": bool(getattr(exc, 'auth_related', True)),
"package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url},
**({'_diagnostic_http': exc.diagnostic_http} if exc.diagnostic_http else {}),
}
except Exception as exc:
return {
"findings": [], "errors": [f"GitHub Actions acquisition failed: {exc}"],
"error_class": "remote_transient", "retryable": True, "source_failure": True,
"source_failure_category": "network", "source_failure_auth_related": False,
"package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url},
}
finally:
cleanup_command_work_dir(work_dir)
def gitlab_api_get(url, token=None, timeout=20, stream=False):
response = api_request(
'GET', url, headers=gitlab_headers(token), timeout=timeout, stream=stream,
)
if token and response.status_code in (401, 403):
response.close()
anonymous = api_request(
'GET', url, headers=gitlab_headers(None), timeout=timeout, stream=stream,
)
if anonymous.status_code < 400:
return anonymous
response = anonymous
if response.status_code >= 400:
raise gitlab_api_error(response)
return response
def scan_gitlab_ci_project(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, ci_pipelines_per_project=5, ci_jobs_per_pipeline=20, ci_lookback_days=30, ci_max_trace_mb=20, ci_scan_artifacts=False, ci_max_artifacts_per_pipeline=5, ci_max_artifact_archive_mb=50, ci_max_artifact_file_mb=10, ci_max_artifact_files=1000, ci_target_max_download_mb=500, fetch_timeout=20):
project, project_url = parse_gitlab_project_target(target)
if not project:
return {"findings": [], "errors": ["Unable to parse GitLab project target"]}
logger.info(f"Scanning GitLab CI traces: {project}")
_raise_if_scan_slot_fatal()
work_dir = create_command_work_dir()
_raise_if_scan_slot_fatal()
if not work_dir:
return {"findings": [], "errors": ["Unable to create GitLab CI work dir"]}
try:
encoded = quote(project, safe='')
pipeline_url = f'https://gitlab.com/api/v4/projects/{encoded}/pipelines?per_page={max(1, min(100, int(ci_pipelines_per_project or 5)))}&order_by=updated_at&sort=desc'
response = gitlab_api_get(pipeline_url, token, fetch_timeout, stream=True)
pipelines = bounded_response_json(response) or []
if not isinstance(pipelines, list):
raise ApiRequestError('invalid GitLab pipelines payload')
cutoff = datetime.now(timezone.utc) - timedelta(days=int(ci_lookback_days or 0)) if int(ci_lookback_days or 0) > 0 else None
selected = []
for pipeline in pipelines:
updated = parse_postman_time(pipeline.get('updated_at') or pipeline.get('created_at'))
if cutoff and updated and updated < cutoff:
continue
selected.append(pipeline)
if len(selected) >= int(ci_pipelines_per_project or 5):
break
if not selected:
return {"findings": [], "errors": [], "skipped": "no recent pipelines", "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url}}
trace_dir = os.path.join(work_dir, 'gitlab_ci', safe_path_component(project))
ensure_private_directory(trace_dir, reject_reparse=True)
max_trace_bytes = int(ci_max_trace_mb or 0) * 1024 * 1024
max_artifact_archive_bytes = int(ci_max_artifact_archive_mb or 0) * 1024 * 1024
pipeline_meta = []
written = 0
artifact_extracted_total = 0
download_failures = []
downloaded_total = 0
target_download_limit = int(ci_target_max_download_mb or 0) * 1024 * 1024
download_budget = {'remaining': target_download_limit} if target_download_limit else None
for pipeline in selected:
_raise_if_scan_slot_fatal()
if download_budget is not None and download_budget['remaining'] <= 0:
download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap')
break
pipeline_id = pipeline.get('id')
if not pipeline_id:
continue
jobs_url = f'https://gitlab.com/api/v4/projects/{encoded}/pipelines/{pipeline_id}/jobs?per_page={max(1, min(100, int(ci_jobs_per_pipeline or 20)))}'
try:
jobs_response = gitlab_api_get(jobs_url, token, fetch_timeout, stream=True)
jobs = bounded_response_json(jobs_response) or []
if not isinstance(jobs, list):
raise ApiRequestError('invalid GitLab jobs payload')
except ScanSlotFatalError:
raise
except Exception as exc:
logger.warning(f"Unable to list GitLab CI jobs for {project} pipeline {pipeline_id}: {exc}")
download_failures.append(f'job listing failed: {exc}')
jobs = []
job_meta = []
artifacts_seen = 0
for job in jobs[:max(1, int(ci_jobs_per_pipeline or 20))]:
_raise_if_scan_slot_fatal()
if download_budget is not None and download_budget['remaining'] <= 0:
download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap')
break
job_id = job.get('id')
if not job_id:
continue
artifact_meta = None
trace_written = False
trace_url = f'https://gitlab.com/api/v4/projects/{encoded}/jobs/{job_id}/trace'
filename = f"pipeline_{pipeline_id}_job_{job_id}_{safe_path_component(job.get('name'))}.log"
trace_path = os.path.join(trace_dir, filename)
try:
trace_download = download_to_file(
trace_url, trace_path, gitlab_headers(token), fetch_timeout, ci_max_trace_mb,
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
byte_budget=download_budget,
)
except requests.RequestException as exc:
logger.warning(f"GitLab CI trace download failed for {project} job {job_id}: {exc}")
download_failures.append(f'trace download failed: {exc}')
trace_download = DownloadOutcome('', 0, 0, str(exc))
authenticated_status = trace_download.status_code
if not trace_download.ok and token and trace_download.status_code in (401, 403):
anonymous = download_to_file(
trace_url, trace_path, gitlab_headers(None), fetch_timeout, ci_max_trace_mb,
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
byte_budget=download_budget,
)
if not anonymous.ok:
trace_download = DownloadOutcome(
'', authenticated_status or anonymous.status_code, 0,
f'authenticated HTTP {authenticated_status}; anonymous HTTP '
f'{anonymous.status_code}: {anonymous.error}',
)
else:
trace_download = anonymous
if trace_download.ok:
downloaded_total += trace_download.bytes_written
written += 1
trace_written = True
elif trace_download.status_code not in (404, 410):
download_failures.append(
f'trace download HTTP {trace_download.status_code}: {trace_download.error}'
)
if ci_scan_artifacts and artifacts_seen < int(ci_max_artifacts_per_pipeline or 5):
if download_budget is not None and download_budget['remaining'] <= 0:
download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap')
job_meta.append({'job_id': job_id, 'name': job.get('name'), 'status': job.get('status'), 'stage': job.get('stage'), 'trace': trace_written, 'artifact': None})
break
artifact_file = job.get('artifacts_file') if isinstance(job.get('artifacts_file'), dict) else {}
artifact_size = int(artifact_file.get('size') or 0)
if artifact_file and max_artifact_archive_bytes and artifact_size and artifact_size > max_artifact_archive_bytes:
download_failures.append(f'artifact declared size exceeds {ci_max_artifact_archive_mb} MB')
artifact_file = {}
if artifact_file and (not max_artifact_archive_bytes or not artifact_size or artifact_size <= max_artifact_archive_bytes):
artifacts_url = f'https://gitlab.com/api/v4/projects/{encoded}/jobs/{job_id}/artifacts'
artifact_zip = os.path.join(work_dir, f'gitlab_ci_artifact_{safe_path_component(project)}_{job_id}.zip')
download = download_to_file(
artifacts_url, artifact_zip, gitlab_headers(token), fetch_timeout, ci_max_artifact_archive_mb,
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
byte_budget=download_budget,
)
authenticated_status = download.status_code
if not download.ok and token and download.status_code in (401, 403):
anonymous = download_to_file(
artifacts_url, artifact_zip, gitlab_headers(None), fetch_timeout, ci_max_artifact_archive_mb,
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
byte_budget=download_budget,
)
if not anonymous.ok:
download = DownloadOutcome(
'', authenticated_status or anonymous.status_code, 0,
f'authenticated HTTP {authenticated_status}; anonymous HTTP '
f'{anonymous.status_code}: {anonymous.error}',
)
else:
download = anonymous
if download.ok:
downloaded_total += download.bytes_written
artifact_dir = os.path.join(trace_dir, 'artifacts', f'pipeline_{pipeline_id}', f'job_{job_id}')
ensure_private_directory(artifact_dir, reject_reparse=True)
try:
artifact_files = safe_extract_zip(
artifact_zip,
artifact_dir,
ci_max_artifact_file_mb,
ci_max_artifact_files,
CI_ARTIFACT_ALLOWED_SUFFIXES,
max_total_size_mb=ci_max_artifact_archive_mb,
)
harden_private_tree(artifact_dir)
except (zipfile.BadZipFile, ValueError) as exc:
logger.warning(f"GitLab CI artifact rejected for {project} job {job_id}: {exc}")
download_failures.append(f'artifact archive rejected: {exc}')
artifact_files = 0
remove_file_quiet(artifact_zip)
artifact_extracted_total += artifact_files
artifacts_seen += 1
artifact_meta = {
'filename': artifact_file.get('filename'),
'size': artifact_size,
'extracted_files': artifact_files,
}
else:
if download.status_code not in (404, 410):
logger.warning(f"GitLab CI artifact download HTTP {download.status_code} for {project} job {job_id}: {download.error}")
download_failures.append(f'artifact download HTTP {download.status_code}: {download.error}')
job_meta.append({'job_id': job_id, 'name': job.get('name'), 'status': job.get('status'), 'stage': job.get('stage'), 'trace': trace_written, 'artifact': artifact_meta})
pipeline_meta.append({'pipeline_id': pipeline_id, 'status': pipeline.get('status'), 'ref': pipeline.get('ref'), 'updated_at': pipeline.get('updated_at'), 'jobs': job_meta})
if written == 0 and artifact_extracted_total == 0:
if download_failures:
text = '; '.join(download_failures[:5])
all_failures = '; '.join(download_failures)
auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any(
token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit')
)
return {
"findings": [], "errors": [text], "error_class": 'source_auth' if auth_failure else 'remote_transient',
"retryable": True, "source_failure": auth_failure,
"source_failure_category": 'auth_forbidden' if auth_failure else 'remote_transient',
"source_failure_auth_related": auth_failure,
"package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url, "pipelines": pipeline_meta},
}
return {"findings": [], "errors": [], "skipped": "no downloadable job traces or artifacts", "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url, "pipelines": pipeline_meta}}
_raise_if_scan_slot_fatal()
results = run_filesystem_scan(work_dir, timeout_sec, detectors, exclude_detectors, no_verification, trufflehog_config)
results['package'] = {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url, "pipelines": pipeline_meta, "trace_files": written, "artifact_files": artifact_extracted_total}
if download_failures:
text = '; '.join(download_failures[:5])
all_failures = '; '.join(download_failures)
auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any(
token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit')
)
results['errors'] = list(results.get('errors') or []) + [text]
results['error_class'] = 'source_auth' if auth_failure else 'remote_transient'
results['retryable'] = True
results['source_failure'] = auth_failure
results['source_failure_category'] = 'auth_forbidden' if auth_failure else 'remote_transient'
results['source_failure_auth_related'] = auth_failure
return results
except ScanSlotFatalError:
raise
except RateLimitError as exc:
category = getattr(exc, 'category', '')
if category == 'not_found':
return {"findings": [], "errors": [], "skipped": str(exc), "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url}}
return {
"findings": [], "errors": [str(exc)], "error_class": 'source_auth' if getattr(exc, 'auth_related', True) else 'remote_transient',
"retryable": True, "source_failure": True,
"source_failure_category": category or 'rate_limit',
"source_failure_auth_related": bool(getattr(exc, 'auth_related', True)),
"package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url},
**({'_diagnostic_http': exc.diagnostic_http} if exc.diagnostic_http else {}),
}
except Exception as exc:
return {
"findings": [], "errors": [f"GitLab CI acquisition failed: {exc}"],
"error_class": "remote_transient", "retryable": True, "source_failure": True,
"source_failure_category": "network", "source_failure_auth_related": False,
"package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url},
}
finally:
cleanup_command_work_dir(work_dir)
def send_webhook_notification(webhook_url, finding, target):
"""Send webhook notification for a finding"""
if not webhook_url:
return
try:
payload = {
"text": f"Secret Found in {target}",
"attachments": [{
"color": "danger",
"fields": [
{"title": "Detector", "value": finding.get('DetectorName', 'Unknown'), "short": True},
{"title": "Target", "value": target, "short": True},
{"title": "Verified", "value": str(finding.get('Verified', False)), "short": True}
]
}]
}
response = requests.post(webhook_url, json=payload, timeout=10)
response.raise_for_status()
logger.info(f"Webhook notification sent for {target}")
except Exception as e:
logger.error(f"Failed to send webhook notification: {str(e)}")
def _captured_http_failure(error):
current = error
seen = set()
for _index in range(8):
if current is None or id(current) in seen:
break
seen.add(id(current))
direct_status = getattr(current, 'status_code', None)
if type(direct_status) is int:
direct_body = getattr(current, 'body', b'')
result = {
'status_code': direct_status,
'content_type': getattr(current, 'content_type', None),
'request_id': getattr(current, 'request_id', None),
'operation': getattr(current, 'operation', 'provider-api'),
'headers_b64': base64.b64encode(json.dumps(
getattr(current, 'headers', None) or {},
ensure_ascii=True, sort_keys=True, separators=(',', ':'),
).encode('ascii')).decode('ascii'),
}
result['body_capture_truncated'] = bool(getattr(
current, 'body_capture_truncated', False
))
if direct_body is not None:
if isinstance(direct_body, str):
direct_body = direct_body.encode('utf-8')
if not isinstance(direct_body, bytes):
direct_body = bytes(direct_body or b'')
result['body_b64'] = base64.b64encode(direct_body).decode('ascii')
original_size = getattr(current, 'body_original_size', None)
stored_size = getattr(current, 'body_stored_size', None)
body_sha256 = getattr(current, 'body_sha256', None)
if original_size is None or stored_size is None or body_sha256 is None:
material = make_body_material(direct_body)
original_size = material.original_size
stored_size = material.stored_size
body_sha256 = material.sha256
result['body_capture_truncated'] = material.truncated
result.update({
'body_original_size': original_size,
'body_stored_size': stored_size,
'body_sha256': body_sha256,
})
return result
response = getattr(current, 'response', None)
status = getattr(response, 'status_code', None)
if type(status) is int:
body_material = getattr(response, '_truf_diagnostic_body_material', None)
if body_material is None and (
not getattr(response, 'raw', None)
or getattr(response, '_content_consumed', False)
):
try:
body = getattr(response, 'content', b'')
except RuntimeError:
body = None
if isinstance(body, str):
body = body.encode('utf-8')
if body is not None and not isinstance(body, bytes):
body = bytes(body)
if body is not None:
body_material = make_body_material(body)
headers = getattr(response, 'headers', {}) or {}
result = {
'status_code': status,
'content_type': str(headers.get('Content-Type') or '') or None,
'request_id': str(
headers.get('X-Request-Id') or headers.get('X-GitHub-Request-Id') or ''
) or None,
'operation': 'provider-request',
'headers_b64': base64.b64encode(json.dumps(
{str(key): str(value) for key, value in headers.items()},
ensure_ascii=True, sort_keys=True, separators=(',', ':'),
).encode('ascii')).decode('ascii'),
'body_capture_truncated': False,
}
if body_material is not None:
body = diagnostic_material_bytes(body_material)
result.update({
'body_b64': base64.b64encode(body).decode('ascii'),
'body_original_size': body_material.original_size,
'body_stored_size': body_material.stored_size,
'body_sha256': body_material.sha256,
'body_capture_truncated': body_material.truncated,
})
return result
current = getattr(current, '__cause__', None) or getattr(
current, '__context__', None
)
return None
def _captured_http_body_material(payload, capture):
material = make_body_material(payload)
metadata_fields = (
'body_original_size', 'body_stored_size', 'body_sha256',
)
if not any(name in capture for name in metadata_fields):
if capture.get('body_capture_truncated'):
raise ValueError('truncated diagnostic HTTP material lacks capture metadata')
return material
if (
not all(name in capture for name in metadata_fields)
or type(capture.get('body_capture_truncated')) is not bool
or isinstance(capture['body_original_size'], bool)
or not isinstance(capture['body_original_size'], int)
or isinstance(capture['body_stored_size'], bool)
or not isinstance(capture['body_stored_size'], int)
or capture['body_stored_size'] != len(payload)
or capture['body_original_size'] < capture['body_stored_size']
or capture['body_capture_truncated']
!= (capture['body_original_size'] > capture['body_stored_size'])
or not isinstance(capture['body_sha256'], str)
or re.fullmatch(r'[a-f0-9]{64}', capture['body_sha256']) is None
or (
not capture['body_capture_truncated']
and capture['body_sha256'] != hashlib.sha256(payload).hexdigest()
)
):
raise ValueError('captured diagnostic HTTP material metadata is invalid')
return replace(
material,
original_size=capture['body_original_size'],
stored_size=capture['body_stored_size'],
sha256=capture['body_sha256'],
truncated=capture['body_capture_truncated'],
)
def scan_target_result(target, scan_type, scan_event_id, scan_kwargs=None):
scan_kwargs = dict(scan_kwargs or {})
scan_started_at = datetime.now(timezone.utc).isoformat()
started = time.perf_counter()
try:
direct_kind = _client_remote_execution_kind.get()
expected_direct_kind = {
'docker': 'docker_direct_v1',
'huggingface': 'huggingface_space_v1',
}.get(scan_type)
if (
_client_scan_manifest.get() is not None
and expected_direct_kind is not None
and direct_kind != expected_direct_kind
):
raise RuntimeError(
'remote direct scan lacks its execution authority'
)
if direct_kind is not None and direct_kind != expected_direct_kind:
raise RuntimeError('remote direct execution platform changed')
if scan_type in ('git', 'github', 'github_archive', 'gitlab'):
provider = 'github' if scan_type == 'github_archive' else scan_type if scan_type in ('github', 'gitlab') else None
result = scan_git_repo(target, provider=provider, **scan_kwargs)
elif scan_type == 'docker':
docker_layer_work = scan_kwargs.pop('docker_layer_work', None)
if docker_layer_work is not None:
scan_kwargs.pop('trufflehog_concurrency', None)
scan_kwargs.pop('docker_recovery_limits', None)
scan_kwargs.pop('docker_recovery_min_free_bytes', None)
try:
result = scan_docker_layer_plan(
target, docker_layer_work, **scan_kwargs,
)
except (ScanSlotFatalError, DockerLayerInfrastructureError):
raise
except Exception as exc:
raise DockerLayerInfrastructureError(
'scanner_infrastructure',
'Docker layer scanner infrastructure failed',
category='source_resource',
) from exc
else:
result = scan_docker_image(
target,
config_dir=(
None if direct_kind == 'docker_direct_v1'
else docker_token_manager.get_next_config()
),
**scan_kwargs,
)
elif scan_type == 'huggingface':
result = scan_huggingface_space(target, **scan_kwargs)
elif scan_type == 'npm':
result = scan_npm_package(target, **scan_kwargs)
elif scan_type == 'pypi':
result = scan_pypi_package(target, **scan_kwargs)
elif scan_type == 'package_git':
result = scan_package_git_repo(target, **scan_kwargs)
elif scan_type in ('postman', 'github_gists', 'github_archive_files'):
result = scan_postman_target(target, **scan_kwargs)
elif scan_type == 'github_actions':
result = scan_github_actions_repo(target, **scan_kwargs)
elif scan_type == 'gitlab_ci':
result = scan_gitlab_ci_project(target, **scan_kwargs)
else:
result = {'findings': [], 'errors': [f'Unknown scan type: {scan_type}']}
except (ScanSlotFatalError, DockerLayerInfrastructureError):
raise
except Exception as exc:
result = {'findings': [], 'errors': [str(exc)]}
http_failure = _captured_http_failure(exc)
if http_failure is not None:
result['_diagnostic_http'] = http_failure
apply_result_error_scope(result)
result['target'] = target
result['scan_type'] = scan_type
result['scan_event_id'] = str(scan_event_id)
result['scan_started_at'] = scan_started_at
result['duration_sec'] = time.perf_counter() - started
result['timestamp'] = datetime.now(timezone.utc).isoformat()
assign_finding_uids(result)
return result
@dataclass(frozen=True)
class StagedResult:
target: str
scan_event_id: str
bundle_id: str
reservation_id: int
scan_event_hash: str
actual_bytes: int
relative_path: str
frame_count: int
finding_count: int
error_count: int
candidate_count: int
queue_status: str
source_failure: bool
source_failure_category: str
source_failure_auth_related: bool
first_error: str
def as_dict(self):
return dict(self.__dict__)
def stage_result_bundle(
result, reservation, bundle_root, scan_options, queue_disposition,
candidate_max_items=2000, candidate_max_bytes=2 * 1024 * 1024,
require_s_drive=False, fault=None, diagnostic_slot_id=0,
diagnostic_attempt=1,
):
reservation = reservation if isinstance(reservation, BundleReservation) else BundleReservation.from_mapping(reservation)
if not isinstance(result, dict):
raise ValueError('scan result must be an object')
canonical_reservation_target = normalize_target(
reservation.target, reservation.platform,
)
if not reservation.normalized_target:
reservation = replace(
reservation, normalized_target=canonical_reservation_target,
)
def optional_integer_identity(name, expected):
if name not in result or result[name] is None:
return
value = result[name]
if isinstance(value, bool):
raise ValueError('scan result identity does not match its reservation')
try:
value = int(value)
except (TypeError, ValueError, OverflowError):
raise ValueError('scan result identity does not match its reservation') from None
if value != int(expected):
raise ValueError('scan result identity does not match its reservation')
optional_integer_identity('reservation_id', reservation.reservation_id)
optional_integer_identity('result_reservation_id', reservation.reservation_id)
optional_integer_identity('queue_id', reservation.queue_id)
for name, expected in (
('bundle_id', reservation.bundle_id),
('source', reservation.source),
('platform', reservation.platform),
('query', reservation.query),
('normalized_target', reservation.normalized_target),
):
if name in result and result[name] is not None and str(result[name]) != str(expected):
raise ValueError('scan result identity does not match its reservation')
if (
str(result.get('scan_event_id') or '') != reservation.scan_event_id
or str(result.get('target') or '') != reservation.target
or str(result.get('scan_type') or '') != reservation.platform
or normalize_target(result.get('target'), reservation.platform)
!= canonical_reservation_target
or reservation.normalized_target != canonical_reservation_target
):
raise ValueError('scan result identity does not match its reservation')
strip_nearby_context_for_persistence(result)
findings = result.get('findings') or []
errors = result.get('errors') or []
candidates = []
candidate_bytes = 0
candidate_identities = set()
candidate_truncated = False
def add_candidate(candidate, attribution):
nonlocal candidate_bytes, candidate_truncated
identity = (
candidate.service, candidate.credential_hash,
'' if candidate.service == 'provider_resolver' else str(
(attribution or {}).get('finding_uid') or (attribution or {}).get('origin') or ''
),
)
if identity in candidate_identities:
return
frame = candidate.as_frame(attribution)
encoded_size = len(json.dumps(
frame, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str,
).encode('utf-8'))
if len(candidates) >= max(0, int(candidate_max_items)):
candidate_truncated = True
return
if candidate_bytes + encoded_size > max(0, int(candidate_max_bytes)):
candidate_truncated = True
return
candidate_identities.add(identity)
candidates.append(frame)
candidate_bytes += encoded_size
for finding in findings:
attribution = {
'finding_uid': str(finding.get('finding_uid') or ''),
'detector_name': str(finding.get('DetectorName') or finding.get('DetectorType') or ''),
}
try:
for candidate in extract_candidates(finding, attribution):
add_candidate(candidate, attribution)
except ValueError as exc:
result.setdefault('warnings', []).append(
f'Optional keycheck candidate was omitted: {type(exc).__name__}'
)
result['degraded'] = True
if result.get('structured_keycheck_pending') and isinstance(result.get('postman'), dict):
try:
postman_data = result['postman']
cache_path, _ = validate_postman_cache_artifact(
postman_data,
int(result.get('postman_max_artifact_size_mb') or 20),
expected_size=result.get('bytes'),
)
contexts = load_postman_context(cache_path)
origin = artifact_origin_label(postman_data, cache_path)
attribution = {'origin': origin}
for candidate in extract_structured_candidates({
'contexts': contexts, 'origin': origin,
}, attribution):
structured_origin = candidate.metadata.get('structured_origin') or origin
add_candidate(candidate, {'origin': structured_origin})
except Exception as exc:
result.setdefault('warnings', []).append(
f'Optional structured keycheck extraction failed: {type(exc).__name__}'
)
result['degraded'] = True
if candidate_truncated:
result.setdefault('warnings', []).append(
'Keycheck candidates reached their pre-reserved per-event bound; remaining candidates were omitted'
)
result['degraded'] = True
status = target_status(result)
first_error = ''
for error in errors:
first_error = next((line.strip() for line in str(error).splitlines() if line.strip()), '')
if first_error:
break
diagnostics = list(result.get('diagnostics') or ())
if errors and not diagnostics:
timestamp = str(result.get('timestamp') or result.get('scan_started_at') or '')
try:
occurred = datetime.fromisoformat(timestamp.replace('Z', '+00:00'))
except ValueError:
occurred = datetime.now(timezone.utc)
if occurred.tzinfo is None:
occurred = datetime.now(timezone.utc)
occurred_at = occurred.astimezone(timezone.utc).isoformat(
timespec='milliseconds'
).replace('+00:00', 'Z')
scan_meta = result.get('scan_meta') or {}
timed_out = bool(scan_meta.get('command_timed_out')) or result.get('error_class') == 'timeout'
category_name = str(
result.get('source_failure_category') or result.get('error_class') or ''
).lower()
category = {
'source_auth': DiagnosticCategory.AUTHORIZATION,
'auth_forbidden': DiagnosticCategory.AUTHORIZATION,
'rate_limit': DiagnosticCategory.RATE_LIMIT,
'not_found': DiagnosticCategory.NOT_FOUND,
'network': DiagnosticCategory.NETWORK,
'remote_transient': DiagnosticCategory.NETWORK,
'timeout': DiagnosticCategory.TIMEOUT,
'source_resource': DiagnosticCategory.STORAGE,
'storage': DiagnosticCategory.STORAGE,
}.get(category_name, DiagnosticCategory.SCANNER)
phase_name = str(scan_meta.get('timed_out_phase') or WorkerPhase.SCANNING.value)
try:
diagnostic_phase = WorkerPhase(phase_name)
except ValueError:
diagnostic_phase = WorkerPhase.SCANNING
scan_outcome = {
'clean': ScanOutcome.CLEAN,
'found': ScanOutcome.FOUND,
'degraded': ScanOutcome.DEGRADED,
'error': ScanOutcome.ERROR,
'skipped': ScanOutcome.SKIPPED,
}.get(status, ScanOutcome.ERROR)
try:
raw_stdout = base64.b64decode(
str(result['_diagnostic_raw_stdout_b64']).encode('ascii'),
validate=True,
) if '_diagnostic_raw_stdout_b64' in result else None
raw_stderr = base64.b64decode(
str(result['_diagnostic_raw_stderr_b64']).encode('ascii'),
validate=True,
) if '_diagnostic_raw_stderr_b64' in result else None
except (UnicodeEncodeError, ValueError, binascii.Error) as exc:
raise ValueError('captured diagnostic process material is invalid') from exc
stdout_limit = stderr_limit = 0
if raw_stdout is not None or raw_stderr is not None:
desired = [
min(len(raw_stdout), MAX_DIAGNOSTIC_LOG_BYTES)
if raw_stdout is not None else 0,
min(len(raw_stderr), MAX_DIAGNOSTIC_LOG_BYTES)
if raw_stderr is not None else 0,
]
total = sum(desired)
if total <= MAX_DIAGNOSTIC_LOG_BYTES:
stdout_limit, stderr_limit = desired
elif total:
stdout_limit = (
MAX_DIAGNOSTIC_LOG_BYTES * desired[0]
) // total
stderr_limit = MAX_DIAGNOSTIC_LOG_BYTES - stdout_limit
process = None
if raw_stdout is not None or raw_stderr is not None:
process = DiagnosticProcessContext(
name='trufflehog',
exit_code=(
int(scan_meta['trufflehog_returncode'])
if type(scan_meta.get('trufflehog_returncode')) is int else None
),
signal=None,
timed_out=timed_out,
stdout=(
make_log_material(raw_stdout, maximum=stdout_limit)
if raw_stdout is not None else None
),
stderr=(
make_log_material(raw_stderr, maximum=stderr_limit)
if raw_stderr is not None else None
),
)
http_value = result.get('_diagnostic_http')
http = None
if isinstance(http_value, dict) and type(http_value.get('status_code')) is int:
raw_body = None
raw_headers = None
if 'body_b64' in http_value:
try:
raw_body = base64.b64decode(
str(http_value['body_b64']).encode('ascii'),
validate=True,
)
except (UnicodeEncodeError, ValueError, binascii.Error) as exc:
raise ValueError('captured diagnostic HTTP material is invalid') from exc
if 'headers_b64' in http_value:
try:
raw_headers = base64.b64decode(
str(http_value['headers_b64']).encode('ascii'),
validate=True,
)
except (UnicodeEncodeError, ValueError, binascii.Error) as exc:
raise ValueError('captured diagnostic HTTP headers are invalid') from exc
http = DiagnosticHTTPContext(
operation=str(http_value.get('operation') or 'provider-request'),
status_code=http_value['status_code'],
content_type=http_value.get('content_type'),
request_id=http_value.get('request_id'),
body=(
_captured_http_body_material(raw_body, http_value)
if raw_body is not None else None
),
headers=(
make_body_material(raw_headers)
if raw_headers is not None else None
),
)
transformation = str(result.get('_diagnostic_stderr_transformation') or (
'HTTP status and bounded body evidence preserve captured values; parsed '
'response headers were serialized as a deterministic mapping because raw '
'wire order and casing are unavailable'
+ (
'; response body capture reached its explicit byte bound'
if http_value and http_value.get('body_capture_truncated') else ''
)
if http is not None else
'raw provider/process material was unavailable; canonical diagnostic '
'was projected from the existing legacy error representation'
))
fingerprint = hashlib.sha256(
('\n'.join(str(error) for error in errors)).encode('utf-8')
).hexdigest()
kind = (
DiagnosticKind.PROVIDER_HTTP if http is not None
else DiagnosticKind.SCANNER_PROCESS if process is not None
else DiagnosticKind.EXCEPTION
)
diagnostics.append(build_diagnostic_envelope(
occurrence_id=f'{reservation.scan_event_id}:scan-result',
reservation_id=reservation.reservation_id,
scan_event_id=reservation.scan_event_id,
slot_id=int(diagnostic_slot_id),
source=reservation.source,
phase=diagnostic_phase,
kind=kind,
category=(DiagnosticCategory.TIMEOUT if timed_out else category),
code=('scan.stage_timeout' if timed_out else 'scan.result_error'),
summary=(first_error or 'scan returned one or more errors')[:1000],
retryable=bool(result.get('retryable', False)),
attempt=max(1, int(diagnostic_attempt)),
assignment_outcome=AssignmentOutcome.ACCEPTED,
scan_outcome=scan_outcome,
occurred_at=occurred_at,
captured_at=occurred_at,
http=http,
process=process,
exception=DiagnosticExceptionContext(
type='truf.diagnostic.TechnicalTransformation',
message=transformation,
fingerprint=fingerprint,
),
))
metadata = {
key: value for key, value in result.items()
if key not in ('findings', 'errors', 'diagnostics')
and not key.startswith('_diagnostic_')
}
metadata.update({
'reservation_id': reservation.reservation_id,
'queue_id': reservation.queue_id,
'bundle_id': reservation.bundle_id,
'source': reservation.source,
'platform': reservation.platform,
'query': reservation.query,
'normalized_target': reservation.normalized_target,
'status': status,
'findings_count': len(findings),
'verified_findings_count': sum(1 for finding in findings if finding.get('Verified')),
'error_count': len(errors),
'first_error_summary': first_error[:500],
'scan_options': dict(scan_options or {}),
'derived_postman_targets': list(result.get('postman_targets') or ()),
**dict(queue_disposition or {}),
})
with ResultBundleWriter.open(
bundle_root, reservation, fault=fault, require_s_drive=require_s_drive,
) as writer:
for finding in findings:
writer.write_finding(finding)
for error in errors:
writer.write_error(error)
for diagnostic in diagnostics:
writer.write_diagnostic(diagnostic)
for candidate in candidates:
writer.write_candidate(candidate)
commit = writer.finish(metadata)
findings.clear()
errors.clear()
diagnostics.clear()
candidates.clear()
return StagedResult(
target=str(result.get('target') or ''),
scan_event_id=commit.scan_event_id,
bundle_id=commit.bundle_id,
reservation_id=commit.reservation_id,
scan_event_hash=commit.scan_event_hash,
actual_bytes=commit.actual_bytes,
relative_path=commit.relative_path,
frame_count=commit.frame_count,
finding_count=commit.finding_count,
error_count=commit.error_count,
candidate_count=commit.candidate_count,
queue_status=str(metadata.get('queue_status') or ''),
source_failure=bool(result.get('source_failure')),
source_failure_category=str(result.get('source_failure_category') or ''),
source_failure_auth_related=bool(result.get('source_failure_auth_related')),
first_error=first_error[:500],
)
class ResultSinkError(RuntimeError):
def __init__(self, failures, results):
self.failures = list(failures)
self.results = list(results)
targets = ', '.join(str(target) for target, _ in self.failures[:5])
super().__init__(f'result sink failed for {len(self.failures)} completed target(s): {targets}')
def scan_targets_batch(
targets,
scan_type,
progress_callback=None,
max_workers=4,
persist_results=True,
result_sink=None,
scan_slot_leases=None,
sink_within_scan_slot=False,
**kwargs,
):
"""Scan multiple targets with parallel processing and progress tracking"""
from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait
import threading
_raise_if_scan_slot_fatal()
results = []
total = len(targets) if hasattr(targets, '__len__') else None
completed = 0
results_lock = threading.Lock()
sink_failures = []
provided_leases = list(scan_slot_leases or [])
if provided_leases and len(provided_leases) != total:
raise ValueError('one pre-acquired scan slot lease is required for every target')
def apply_worker_sink(result):
if not sink_within_scan_slot or result_sink is None:
return result
try:
result_sink(result)
result['_result_sink_applied'] = True
except Exception as exc:
result['_result_sink_exception'] = exc
return result
def scan_single_target(target, scan_event_id, provided_lease=None):
"""Scan a single target"""
_raise_if_scan_slot_fatal()
scan_started_at = datetime.now(timezone.utc).isoformat()
started = time.perf_counter()
try:
with scan_slot_scope(
['scan-target', scan_type], kwargs.get('timeout_sec'), lease=provided_lease,
):
try:
_raise_if_scan_slot_fatal()
result = scan_target_result(target, scan_type, scan_event_id, kwargs)
return apply_worker_sink(result)
except ScanSlotFatalError:
raise
except Exception as e:
result = {
"target": target,
"scan_type": scan_type,
"scan_event_id": scan_event_id,
"scan_started_at": scan_started_at,
"duration_sec": time.perf_counter() - started,
"timestamp": datetime.now(timezone.utc).isoformat(),
"findings": [],
"errors": [str(e)]
}
return apply_worker_sink(apply_result_error_scope(result))
except ScanSlotFatalError:
raise
except Exception as e:
result = {
"target": target,
"scan_type": scan_type,
"scan_event_id": scan_event_id,
"scan_started_at": scan_started_at,
"duration_sec": time.perf_counter() - started,
"timestamp": datetime.now(timezone.utc).isoformat(),
"findings": [],
"errors": [str(e)]
}
return apply_worker_sink(apply_result_error_scope(result))
max_workers = max(1, int(max_workers or 1))
executor = ThreadPoolExecutor(max_workers=max_workers)
future_to_target = {}
target_iterator = iter(enumerate(targets))
def submit_next():
try:
index, target = next(target_iterator)
except StopIteration:
return False
provided_lease = provided_leases[index] if provided_leases else None
future_to_target[
executor.submit(scan_single_target, target, str(uuid.uuid4()), provided_lease)
] = target
return True
try:
for _ in range(max_workers):
_raise_if_scan_slot_fatal()
if not submit_next():
break
while future_to_target:
_raise_if_scan_slot_fatal()
done, _ = wait(tuple(future_to_target), timeout=0.2, return_when=FIRST_COMPLETED)
for future in done:
target = future_to_target.pop(future)
result = future.result()
with results_lock:
results.append(result)
persistence_error = result.pop('_result_sink_exception', None)
sink_applied = bool(result.pop('_result_sink_applied', False))
if result_sink is not None and not sink_applied and persistence_error is None:
try:
result_sink(result)
except Exception as exc:
persistence_error = exc
if persist_results:
try:
if not save_scan_result(result):
raise RuntimeError('save_scan_result returned false')
except Exception as exc:
persistence_error = persistence_error or exc
if persistence_error is not None:
result['persistence_failure'] = str(persistence_error)[:500]
sink_failures.append((target, persistence_error))
logger.error(f'Persistence failed for completed target {target}: {persistence_error}')
submit_next()
continue
with results_lock:
completed += 1
findings_count = len(result.get('findings', []))
errors_count = len(result.get('errors', []))
skipped = bool(result.get('skipped'))
logger.info(
f"Completed {completed}/{total}: {target} "
f"(findings={findings_count}, errors={errors_count}, skipped={skipped})"
)
if progress_callback:
progress_callback(completed, total, target)
submit_next()
except ScanSlotFatalError as exc:
_set_scan_slot_fatal(str(exc))
for future in future_to_target:
future.cancel()
wait(tuple(future_to_target), timeout=1.0)
try:
executor.shutdown(wait=False, cancel_futures=True)
except TypeError:
executor.shutdown(wait=False)
executor = None
raise
finally:
if executor is not None:
executor.shutdown(wait=True)
for lease in provided_leases:
if lease.heartbeat_thread is None:
lease.release()
if progress_callback:
progress_callback(
completed,
total,
"Scan completed" if not sink_failures else "Scan completed with persistence failures",
)
if sink_failures:
raise ResultSinkError(sink_failures, results)
return results
def extract_trufflehog_error_lines(output):
error_lines = []
for index, line in enumerate(_iter_output_lines(output), 1):
if index > 2000:
break
line = line.strip()
if not line:
continue
try:
payload = json.loads(line)
level = str(payload.get('level', '')).lower()
message = str(payload.get('msg', '')).lower()
if message == 'error cleaning temporary artifacts':
continue
if 'error' in level or 'error' in message or payload.get('error'):
error_lines.append(line)
except json.JSONDecodeError:
if 'error' in line.lower() or 'failed' in line.lower():
error_lines.append(line)
return error_lines