import os import sys import socket import sqlite3 import requests import json import re import subprocess import concurrent.futures import threading import time import multiprocessing import tempfile import shutil import tarfile import zipfile import gzip import bz2 import codecs import lzma import base64 import binascii import copy import hashlib import io import logging import math import stat import uuid import ctypes import ipaddress from email.utils import parsedate_to_datetime from html.parser import HTMLParser from collections import Counter from contextlib import contextmanager from contextvars import ContextVar from dataclasses import dataclass, replace from urllib.parse import parse_qsl, quote, urlencode, urljoin, urlsplit, urlunsplit from datetime import datetime, timedelta, timezone import urllib3.util.connection as urllib3_connection import urllib3.exceptions as urllib3_exceptions from paths import default_project_paths from scanner_db import ( DOCKER_ADAPTIVE_PAYLOAD_CLASSES, canonical_docker_layer_plan_bytes, canonical_git_scan_plan_bytes, normalize_target, sanitize_endpoint, sanitize_endpoint_host, sanitize_postman_context, target_status, validate_docker_layer_plan, validate_git_resolution, ) from owned_process import OwnedProcess, run_owned from process_identity import ( current_process_identity, exact_process_identity_state, open_process, serialize_process_identity, ) from janitor import JanitorBudget, bounded_remove_tree from runtime_security import ( atomic_write_private_json, canonical_path, durable_replace, durable_unlink, ensure_private_directory, harden_private_directory, harden_private_file, harden_private_tree, is_reparse_point, PrivateFileLock, private_directory_ready, private_file_ready, read_private_json, reject_reparse_components, require_private_directory, require_private_file, sha256_file, ) from target_identity import ( normalize_docker_digest, normalize_huggingface_space_id, parse_dockerhub_digest_target, parse_docker_target, postman_target_identity as semantic_postman_target_identity, validate_docker_image_reference, ) from docker_depth_experiment import ( DOCKER_DEPTH_SELECTOR_VERSION, canonical_docker_depth_selection_evidence_hash, canonical_selector_hash, select_docker_layer_graphs, validate_docker_images_per_repository, ) from lifecycle_authority import ( CHILD_KIND_ENV, REMOTE_WORKER_CODE_AUTHORITY_FILES, verify_code_manifest, require_active_supervisor_child, resolve_manifest_executable, strip_supervisor_credentials, ) from keycheck_candidates import extract_candidates, extract_structured_candidates from result_bundle import BundleReservation, ResultBundleWriter from worker_contracts import ( AssignmentOutcome, DiagnosticCategory, DiagnosticExceptionContext, DiagnosticHTTPContext, DiagnosticKind, DiagnosticProcessContext, MAX_DIAGNOSTIC_BODY_BYTES, MAX_DIAGNOSTIC_LOG_BYTES, ScanOutcome, WorkerPhase, build_diagnostic_envelope, diagnostic_material_bytes, make_body_material, make_log_material, ) def configure_requests_networking(): if os.getenv('SCANNER_FORCE_IPV4', '1').strip().lower() in ('0', 'false', 'no', 'off'): return urllib3_connection.allowed_gai_family = lambda: socket.AF_INET logger = logging.getLogger(__name__) _runtime_initialized = False _cleanup_registered = False _client_scan_manifest = ContextVar('client_scan_manifest', default=None) _client_scan_policy = ContextVar('client_scan_policy', default=None) _client_remote_execution_kind = ContextVar( 'client_remote_execution_kind', default=None, ) _client_scan_phase_callback = ContextVar( 'client_scan_phase_callback', default=None, ) @contextmanager def client_scan_launch_authority(manifest, expected_sha256=None): verified = verify_code_manifest( manifest, expected_sha256=expected_sha256, required_names=REMOTE_WORKER_CODE_AUTHORITY_FILES, external_names=(), ) token = _client_scan_manifest.set(verified) try: yield verified finally: _client_scan_manifest.reset(token) @contextmanager def client_scan_execution_policy(policy): if not isinstance(policy, dict): raise RuntimeError('remote scan execution policy is invalid') token = _client_scan_policy.set(dict(policy)) try: yield finally: _client_scan_policy.reset(token) @contextmanager def client_remote_execution_binding(planning_kind): if _client_scan_manifest.get() is None: raise RuntimeError('remote direct execution requires client launch authority') kind = str(planning_kind or '') if kind not in {'docker_direct_v1', 'huggingface_space_v1'}: raise RuntimeError('remote direct execution kind is invalid') token = _client_remote_execution_kind.set(kind) try: yield finally: _client_remote_execution_kind.reset(token) @contextmanager def client_scan_phase_events(callback): if callback is not None and not callable(callback): raise TypeError('scan phase callback must be callable') token = _client_scan_phase_callback.set(callback) try: yield finally: _client_scan_phase_callback.reset(token) def emit_client_scan_phase(phase, progress=None): callback = _client_scan_phase_callback.get() if callback is not None: callback(phase, dict(progress or {})) def _scan_policy_value(name, default): policy = _client_scan_policy.get() if policy is not None: if name not in policy: raise RuntimeError('remote scan execution policy is incomplete') return policy[name] return getattr(scan_config, name, default) def initialize_scanner_runtime(*, preflight_complete=False, register_cleanup=True): """Apply process-global scanner setup only after lifecycle preflight.""" global _runtime_initialized, _cleanup_registered if not preflight_complete: raise RuntimeError('scanner runtime initialization requires completed lifecycle preflight') if not _runtime_initialized: configure_requests_networking() logging.basicConfig( level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s', ) _runtime_initialized = True # Stale tree ownership belongs to the isolated janitor, never an atexit hook. _cleanup_registered = False def require_scanner_runtime_initialized(): if not _runtime_initialized: raise RuntimeError('scanner runtime is not initialized after lifecycle preflight') def bool_setting(value, default=False): if value is None: return default if isinstance(value, bool): return value return str(value).strip().lower() in ('1', 'true', 'yes', 'on') def int_setting(value, default): try: return int(value) except (TypeError, ValueError): return default def float_setting(value, default): try: return float(value) except (TypeError, ValueError): return default def csv_items(value): if not value: return [] if isinstance(value, str): return [item.strip() for item in value.split(',') if item.strip()] return [str(item).strip() for item in value if str(item).strip()] DEFAULT_DROP_DETECTORS = ( 'Privacy', 'URI', 'JDBC', 'Postgres', 'MongoDB', 'SQLServer', 'Box', 'ZohoCRM', 'Accuweather', 'Roaring', 'Flatio', 'LinkPreview', 'RailwayApp', ) class RateLimitError(Exception): def __init__( self, source, message, reset_at=None, category='rate_limit', retryable=True, auth_related=True, diagnostic_http=None, ): super().__init__(message) self.source = source self.reset_at = reset_at self.category = category self.retryable = retryable self.auth_related = auth_related self.diagnostic_http = ( dict(diagnostic_http) if isinstance(diagnostic_http, dict) else None ) def response_message(response): if response is None: return '' payload_bytes = bytearray() try: digest = hashlib.sha256() original_size = 0 for chunk in response.iter_content(chunk_size=4096): if not chunk: continue digest.update(chunk) original_size += len(chunk) remaining = MAX_DIAGNOSTIC_BODY_BYTES - len(payload_bytes) if remaining > 0: payload_bytes.extend(chunk[:remaining]) captured = bytes(payload_bytes) material = make_body_material(captured) if original_size > len(captured): material = replace( material, original_size=original_size, sha256=digest.hexdigest(), truncated=True, ) response._truf_diagnostic_body_material = material response._truf_diagnostic_body = captured response._truf_diagnostic_body_truncated = material.truncated payload = json.loads(captured.decode('utf-8', errors='strict')) if isinstance(payload, dict): message = str(payload.get('message') or payload.get('error') or payload) else: message = str(payload) return message.replace('\x00', '\\u0000') except Exception: captured = bytes(payload_bytes[:MAX_DIAGNOSTIC_BODY_BYTES]) response._truf_diagnostic_body = captured return captured[:500].decode( 'utf-8', errors='replace' ).replace('\x00', '\\u0000') def retry_after_reset(response): if response is None: return None retry_after = response.headers.get('Retry-After') if retry_after: try: return (datetime.now(timezone.utc) + timedelta(seconds=int(retry_after))).isoformat(timespec='seconds') except (TypeError, ValueError): return None return None def build_api_error(source, category, message, response=None, reset_at=None, retryable=True, auth_related=False): if reset_at is None: reset_at = retry_after_reset(response) diagnostic_http = None if response is not None and type(getattr(response, 'status_code', None)) is int: headers = getattr(response, 'headers', {}) or {} body_material = getattr(response, '_truf_diagnostic_body_material', None) if body_material is None and ( not getattr(response, 'raw', None) or getattr(response, '_content_consumed', False) ): captured = getattr(response, 'content', b'') body = captured if isinstance(captured, bytes) else str(captured).encode('utf-8') body_material = make_body_material(body) diagnostic_http = { 'operation': f'{source}-api', 'status_code': int(response.status_code), 'content_type': str(headers.get('Content-Type') or '') or None, 'request_id': str( headers.get('X-Request-Id') or headers.get('X-GitHub-Request-Id') or '' ) or None, 'headers_b64': base64.b64encode(json.dumps( {str(key): str(value) for key, value in headers.items()}, ensure_ascii=True, sort_keys=True, separators=(',', ':'), ).encode('ascii')).decode('ascii'), } if body_material is not None: body = diagnostic_material_bytes(body_material) diagnostic_http.update({ 'body_b64': base64.b64encode(body).decode('ascii'), 'body_original_size': body_material.original_size, 'body_stored_size': body_material.stored_size, 'body_sha256': body_material.sha256, 'body_capture_truncated': body_material.truncated, }) else: diagnostic_http['body_capture_truncated'] = False return RateLimitError( source, message, reset_at=reset_at, category=category, retryable=retryable, auth_related=auth_related, diagnostic_http=diagnostic_http, ) def github_api_error(response): status = response.status_code if response is not None else None message = response_message(response) lower_message = message.lower() reset_at = github_rate_limit_reset(response) or retry_after_reset(response) if status == 401: return build_api_error('github', 'auth_invalid', f'GitHub API authentication failed (HTTP 401): {message}', response, retryable=False, auth_related=True) if status == 403: remaining = response.headers.get('X-RateLimit-Remaining') if response is not None else None if remaining == '0': return build_api_error('github', 'rate_limit', f'GitHub API rate limit hit: {message}', response, reset_at, auth_related=True) if 'secondary rate limit' in lower_message or 'abuse' in lower_message: return build_api_error('github', 'secondary_rate_limit', f'GitHub secondary rate limit hit: {message}', response, reset_at, auth_related=True) return build_api_error('github', 'auth_forbidden', f'GitHub API access forbidden (HTTP 403): {message}', response, retryable=False, auth_related=True) if status == 429: return build_api_error('github', 'rate_limit', f'GitHub API returned HTTP 429: {message}', response, reset_at, auth_related=True) if status == 422: return build_api_error('github', 'query_invalid', f'GitHub search query invalid (HTTP 422): {message}', response, retryable=False, auth_related=False) if status == 404: return build_api_error('github', 'not_found', f'GitHub API resource not found (HTTP 404): {message}', response, retryable=False, auth_related=False) if status and status >= 500: return build_api_error('github', 'server_error', f'GitHub API server error (HTTP {status}): {message}', response, auth_related=False) return build_api_error('github', 'api', f'GitHub API error (HTTP {status}): {message}', response, auth_related=False) def gitlab_api_error(response): status = response.status_code if response is not None else None message = response_message(response) reset_at = gitlab_rate_limit_reset(response) or retry_after_reset(response) if status == 401: return build_api_error('gitlab', 'auth_invalid', f'GitLab API authentication failed (HTTP 401): {message}', response, retryable=False, auth_related=True) if status == 403: return build_api_error('gitlab', 'auth_forbidden', f'GitLab API access forbidden (HTTP 403): {message}', response, retryable=False, auth_related=True) if status == 429: return build_api_error('gitlab', 'rate_limit', f'GitLab API returned HTTP 429: {message}', response, reset_at, auth_related=True) if status == 404: return build_api_error('gitlab', 'not_found', f'GitLab API resource not found (HTTP 404): {message}', response, retryable=False, auth_related=False) if status and status >= 500: return build_api_error('gitlab', 'server_error', f'GitLab API server error (HTTP {status}): {message}', response, auth_related=False) return build_api_error('gitlab', 'api', f'GitLab API error (HTTP {status}): {message}', response, auth_related=False) # ===================== # GLOBAL CONFIGURATION # ===================== class ScanConfig: def __init__(self): defaults = default_project_paths() self.git_timeout = 900 self.docker_timeout = 1800 self.detectors = "" self.exclude_detectors = os.getenv("TRUFFLEHOG_EXCLUDE_DETECTORS", "github.v1,gitlab.v1,GitHubOauth2") self.no_verification = os.getenv("TRUFFLEHOG_NO_VERIFICATION", "0").strip().lower() in ("1", "true", "yes", "on") self.strict_git_provider_token_filter = os.getenv( "STRICT_GIT_PROVIDER_TOKEN_FILTER", "1", ).strip().lower() not in ("0", "false", "no", "off") self.drop_detectors = csv_items(os.getenv("SCANNER_DROP_DETECTORS")) self.webhook_url = None self.max_concurrent = min(10, max(1, multiprocessing.cpu_count() * 2)) self.trufflehog_path = os.getenv( "TRUFFLEHOG_PATH", defaults['trufflehog_path'] ) self.trufflehog_config = os.getenv("TRUFFLEHOG_CONFIG", "") self.trufflehog_job_memory_limit_bytes = int_setting( os.getenv("TRUFFLEHOG_JOB_MEMORY_LIMIT_BYTES"), 4096 * 1024 * 1024, ) self.trufflehog_windows_job_cpu_weight = int_setting( os.getenv("TRUFFLEHOG_WINDOWS_JOB_CPU_WEIGHT"), 0, ) self.trufflehog_windows_memory_priority = int_setting( os.getenv("TRUFFLEHOG_WINDOWS_MEMORY_PRIORITY"), 0, ) self.trufflehog_stdout_max_mb = int_setting(os.getenv('TRUFFLEHOG_STDOUT_MAX_MB'), 32) self.trufflehog_stderr_max_mb = int_setting(os.getenv('TRUFFLEHOG_STDERR_MAX_MB'), 8) self.trufflehog_max_findings_per_target = int_setting( os.getenv('TRUFFLEHOG_MAX_FINDINGS_PER_TARGET'), 20000, ) self.work_dir = os.getenv("TRUFFLEHOG_WORK_DIR", defaults['work_dir']) self.runtime_dir = defaults['runtime_dir'] self.results_dir = os.getenv("SCAN_RESULTS_DIR", defaults['results_dir']) self.result_spool_dir = os.getenv("SCANNER_RESULT_SPOOL_DIR", defaults['result_spool_dir']) self.result_spool_max_event_bytes = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MAX_EVENT_BYTES"), 192 * 1024 * 1024) self.result_spool_max_events = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MAX_EVENTS"), 10000) self.result_spool_max_total_bytes = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MAX_TOTAL_BYTES"), 2 * 1024 * 1024 * 1024) self.result_spool_min_free_bytes = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MIN_FREE_BYTES"), 1024 * 1024 * 1024) self.result_bundle_dir = os.getenv('SCANNER_RESULT_BUNDLE_DIR', defaults['result_bundle_dir']) self.result_bundle_max_event_bytes = int_setting( os.getenv('SCANNER_RESULT_BUNDLE_MAX_EVENT_BYTES'), 64 * 1024 * 1024, ) self.scan_outbox_max_pending_items = int_setting(os.getenv('SCAN_OUTBOX_MAX_PENDING_ITEMS'), 10000) self.scan_outbox_max_pending_bytes = int_setting(os.getenv('SCAN_OUTBOX_MAX_PENDING_BYTES'), 1024 * 1024 * 1024) self.scan_outbox_max_pending_age_sec = int_setting(os.getenv('SCAN_OUTBOX_MAX_PENDING_AGE_SEC'), 24 * 60 * 60) self.queue_dir = defaults['queue_dir'] self.keycheck_dir = defaults['keycheck_dir'] self.postman_cache_dir = defaults['postman_cache_dir'] self.postman_cache_max_items = int_setting(os.getenv('POSTMAN_CACHE_MAX_ITEMS'), 100000) self.postman_cache_max_bytes = int_setting(os.getenv('POSTMAN_CACHE_MAX_BYTES'), 20 * 1024 * 1024 * 1024) self.postman_cache_min_free_bytes = int_setting(os.getenv('POSTMAN_CACHE_MIN_FREE_BYTES'), 20 * 1024 * 1024 * 1024) self.postman_cache_lock_timeout_sec = int_setting(os.getenv('POSTMAN_CACHE_LOCK_TIMEOUT_SEC'), 30) self.postman_discovery_max_artifacts_per_cycle = int_setting(os.getenv('POSTMAN_DISCOVERY_MAX_ARTIFACTS_PER_CYCLE'), 1000) self.postman_discovery_max_artifacts_per_page = int_setting(os.getenv('POSTMAN_DISCOVERY_MAX_ARTIFACTS_PER_PAGE'), 100) self.postman_discovery_max_bytes_per_cycle = int_setting(os.getenv('POSTMAN_DISCOVERY_MAX_BYTES_PER_CYCLE'), 1024 * 1024 * 1024) self.postman_discovery_max_elapsed_sec = float_setting(os.getenv('POSTMAN_DISCOVERY_MAX_ELAPSED_SEC'), 300.0) self.postman_package_harvest_max_artifacts = int_setting(os.getenv('POSTMAN_PACKAGE_HARVEST_MAX_ARTIFACTS'), 100) self.postman_package_harvest_max_bytes = int_setting(os.getenv('POSTMAN_PACKAGE_HARVEST_MAX_BYTES'), 128 * 1024 * 1024) self.postman_package_harvest_max_elapsed_sec = float_setting(os.getenv('POSTMAN_PACKAGE_HARVEST_MAX_ELAPSED_SEC'), 30.0) self.postman_context_max_input_bytes = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_INPUT_BYTES'), 16 * 1024 * 1024) self.postman_context_max_nodes = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_NODES'), 100000) self.postman_context_max_depth = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_DEPTH'), 64) self.postman_context_max_scalar_bytes = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_SCALAR_BYTES'), 16 * 1024 * 1024) self.postman_context_max_items = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_ITEMS'), 50000) self.context_enrichment_max_source_bytes = int_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_SOURCE_BYTES'), 16 * 1024 * 1024) self.context_enrichment_max_findings = int_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_FINDINGS'), 2000) self.context_enrichment_max_postman_comparisons = int_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_POSTMAN_COMPARISONS'), 200000) self.context_enrichment_max_elapsed_sec = float_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_ELAPSED_SEC'), 5.0) self.trufflehog_diagnostic_max_lines = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_LINES'), 2000) self.trufflehog_diagnostic_max_line_chars = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_LINE_CHARS'), 8192) self.trufflehog_diagnostic_max_line_bytes = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_LINE_BYTES'), 8192) self.trufflehog_diagnostic_max_errors = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_ERRORS'), 200) self.trufflehog_diagnostic_max_warnings = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_WARNINGS'), 200) self.trufflehog_diagnostic_max_unclassified = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_UNCLASSIFIED'), 20) self.keycheck_input_max_line_bytes = int_setting(os.getenv('KEYCHECK_INPUT_MAX_LINE_BYTES'), 16 * 1024 * 1024) self.keycheck_candidate_artifact_max_items = int_setting(os.getenv('KEYCHECK_CANDIDATE_ARTIFACT_MAX_ITEMS'), 2000) self.keycheck_candidate_artifact_max_bytes = int_setting(os.getenv('KEYCHECK_CANDIDATE_ARTIFACT_MAX_BYTES'), 2 * 1024 * 1024) self.keycheck_candidate_file_max_items = int_setting(os.getenv('KEYCHECK_CANDIDATE_FILE_MAX_ITEMS'), 100000) self.keycheck_candidate_file_max_bytes = int_setting(os.getenv('KEYCHECK_CANDIDATE_FILE_MAX_BYTES'), 32 * 1024 * 1024) self.keycheck_candidate_line_max_bytes = int_setting(os.getenv('KEYCHECK_CANDIDATE_LINE_MAX_BYTES'), 8192) self.gharchive_cache_dir = defaults['gharchive_cache_dir'] self.gharchive_cache_max_items = int_setting(os.getenv('GHARCHIVE_CACHE_MAX_ITEMS'), 48) self.gharchive_cache_max_bytes = int_setting(os.getenv('GHARCHIVE_CACHE_MAX_BYTES'), 8 * 1024 * 1024 * 1024) self.gharchive_cache_min_free_bytes = int_setting(os.getenv('GHARCHIVE_CACHE_MIN_FREE_BYTES'), 5 * 1024 * 1024 * 1024) self.gharchive_download_max_bytes = int_setting(os.getenv('GHARCHIVE_DOWNLOAD_MAX_BYTES'), 512 * 1024 * 1024) self.gharchive_decompressed_max_bytes = int_setting(os.getenv('GHARCHIVE_DECOMPRESSED_MAX_BYTES'), 8 * 1024 * 1024 * 1024) self.gharchive_max_events = int_setting(os.getenv('GHARCHIVE_MAX_EVENTS'), 5000000) self.gharchive_max_line_bytes = int_setting(os.getenv('GHARCHIVE_MAX_LINE_BYTES'), 8 * 1024 * 1024) self.gharchive_cache_lock_timeout_sec = int_setting(os.getenv('GHARCHIVE_CACHE_LOCK_TIMEOUT_SEC'), 600) self.proxy_file = defaults['proxy_file'] self.api_proxy_enabled = bool_setting(os.getenv("SCANNER_API_PROXY_ENABLED"), False) self.api_proxy_file = os.getenv("SCANNER_API_PROXY_FILE", self.proxy_file) self.api_proxy_timeout = int_setting(os.getenv("SCANNER_API_PROXY_TIMEOUT"), 5) self.api_proxy_max_retries = int_setting(os.getenv("SCANNER_API_PROXY_MAX_RETRIES"), 100) self.api_proxy_retry_delay = int_setting(os.getenv("SCANNER_API_PROXY_RETRY_DELAY"), 5) self.download_proxy_enabled = bool_setting(os.getenv("SCANNER_DOWNLOAD_PROXY_ENABLED"), False) self.download_proxy_file = os.getenv("SCANNER_DOWNLOAD_PROXY_FILE", "") self.max_active_scans = int_setting(os.getenv("SCANNER_MAX_ACTIVE_SCANS"), 0) self.opportunistic_scan_slots = int_setting( os.getenv("SCANNER_OPPORTUNISTIC_SCAN_SLOTS"), 0, ) self.opportunistic_scan_sources = csv_items( os.getenv("SCANNER_OPPORTUNISTIC_SCAN_SOURCES") ) self.opportunistic_scan_reserve_overhead_bytes = int_setting( os.getenv("SCANNER_OPPORTUNISTIC_SCAN_RESERVE_OVERHEAD_BYTES"), 1024 * 1024 * 1024, ) self.opportunistic_scan_min_available_after_reserve_bytes = int_setting( os.getenv("SCANNER_OPPORTUNISTIC_SCAN_MIN_AVAILABLE_AFTER_RESERVE_BYTES"), 4 * 1024 * 1024 * 1024, ) self.opportunistic_scan_min_commit_after_reserve_bytes = int_setting( os.getenv("SCANNER_OPPORTUNISTIC_SCAN_MIN_COMMIT_AFTER_RESERVE_BYTES"), 6 * 1024 * 1024 * 1024, ) self.scan_limiter_db = os.getenv("SCANNER_SCAN_LIMITER_DB", os.path.join(defaults['state_dir'], 'scan_limiter.db')) self.dockerhub_tag_cache_path = os.getenv("DOCKERHUB_TAG_CACHE_PATH", os.path.join(defaults['state_dir'], 'dockerhub_tag_cache.sqlite')) self.dockerhub_tag_cache_ttl_sec = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_TTL_SEC"), 21600) self.dockerhub_tag_negative_cache_ttl_sec = int_setting(os.getenv("DOCKERHUB_TAG_NEGATIVE_CACHE_TTL_SEC"), 3600) self.dockerhub_tag_rate_limit_cache_ttl_sec = int_setting(os.getenv("DOCKERHUB_TAG_RATE_LIMIT_CACHE_TTL_SEC"), 1800) self.dockerhub_tag_cache_max_rows = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MAX_ROWS"), 50000) self.dockerhub_tag_cache_max_age_sec = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MAX_AGE_SEC"), 7 * 86400) self.dockerhub_tag_cache_max_bytes = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MAX_BYTES"), 256 * 1024 * 1024) self.dockerhub_tag_cache_min_free_bytes = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MIN_FREE_BYTES"), 512 * 1024 * 1024) self.scan_slot_wait_sec = float_setting(os.getenv("SCANNER_SCAN_SLOT_WAIT_SEC"), 0.5) self.scan_slot_wait_log_sec = int_setting(os.getenv("SCANNER_SCAN_SLOT_WAIT_LOG_SEC"), 30) self.scan_slot_stale_sec = int_setting(os.getenv("SCANNER_SCAN_SLOT_STALE_SEC"), 7200) self.low_space_cleanup_max_items = int_setting(os.getenv("SCANNER_LOW_SPACE_CLEANUP_MAX_ITEMS"), 50) self.min_free_gb = float(os.getenv("TRUFFLEHOG_MIN_FREE_GB", "5")) self.jsonl_rotation_enabled = bool_setting(os.getenv("SCANNER_JSONL_ROTATION_ENABLED"), False) self.found_secrets_max_mb = int_setting(os.getenv("SCANNER_FOUND_SECRETS_MAX_MB"), 512) self.scan_results_max_mb = int_setting(os.getenv("SCANNER_SCAN_RESULTS_MAX_MB"), 1024) self.scan_errors_max_mb = int_setting(os.getenv("SCANNER_SCAN_ERRORS_MAX_MB"), 64) self.scan_errors_keep = int_setting(os.getenv("SCANNER_SCAN_ERRORS_KEEP"), 5) self.jsonl_lock_stale_sec = int_setting(os.getenv("SCANNER_JSONL_LOCK_STALE_SEC"), 300) self.jsonl_max_segments = int_setting(os.getenv("SCANNER_JSONL_MAX_SEGMENTS"), 16) self.jsonl_ledger_max_rows = int_setting(os.getenv("SCANNER_JSONL_LEDGER_MAX_ROWS"), 1000000) self.jsonl_ledger_max_bytes = int_setting(os.getenv("SCANNER_JSONL_LEDGER_MAX_BYTES"), 512 * 1024 * 1024) self.jsonl_legacy_index_max_bytes = int_setting(os.getenv("SCANNER_JSONL_LEGACY_INDEX_MAX_BYTES"), 16 * 1024 * 1024) self.jsonl_tail_scan_max_bytes = int_setting(os.getenv("SCANNER_JSONL_TAIL_SCAN_MAX_BYTES"), 8 * 1024 * 1024) self.jsonl_torn_quarantine_max_bytes = int_setting(os.getenv("SCANNER_JSONL_TORN_QUARANTINE_MAX_BYTES"), 64 * 1024) scan_config = ScanConfig() pending_temp_dirs = set() pending_temp_lock = threading.Lock() TEMP_OWNER_FILE = '.scanner-owner.json' TEMP_OWNER_SCHEMA = 2 PENDING_TEMP_SCHEMA = 1 APPROVED_TEMP_PREFIXES = ('trufflehog-', 'trufflehog-run-', 'trufflehog-probe-', 'hg-run-', 'docker-config-', 'tmp-', 'worker-assignment-') class ApiRequestError(Exception): def __init__( self, message, *, response=None, operation='provider-api', capture_body=True, ): super().__init__(message) self.operation = str(operation) self.status_code = None self.content_type = None self.request_id = None self.body = None self.body_original_size = None self.body_stored_size = None self.body_sha256 = None self.body_capture_truncated = False self.headers = None if response is not None and type(getattr(response, 'status_code', None)) is int: self.status_code = int(response.status_code) headers = getattr(response, 'headers', {}) or {} self.headers = { str(key): str(value) for key, value in headers.items() } self.content_type = str(headers.get('Content-Type') or '') or None self.request_id = str( headers.get('X-Request-Id') or headers.get('X-GitHub-Request-Id') or '' ) or None body_material = getattr(response, '_truf_diagnostic_body_material', None) if body_material is not None: self.body = diagnostic_material_bytes(body_material) elif capture_body and ( not getattr(response, 'raw', None) or getattr(response, '_content_consumed', False) ): body = getattr(response, 'content', b'') body = body if isinstance(body, bytes) else str(body).encode('utf-8') body_material = make_body_material(body) self.body = diagnostic_material_bytes(body_material) if body_material is not None: self.body_original_size = body_material.original_size self.body_stored_size = body_material.stored_size self.body_sha256 = body_material.sha256 self.body_capture_truncated = body_material.truncated class GitLabDiscoveryTransportError(ApiRequestError): pass class DockerHubDiscoveryTransportError(ApiRequestError): def __init__( self, message, *, category='page_unavailable', retry_at=None, remote_attempted=True, retryable=True, ): super().__init__(message) self.category = str(category or 'page_unavailable') self.retry_at = retry_at self.remote_attempted = bool(remote_attempted) self.retryable = bool(retryable) API_RETRY_STATUSES = {408, 500, 502, 503, 504} API_RETRY_EXCEPTIONS = ( requests.exceptions.ProxyError, requests.exceptions.ConnectionError, requests.exceptions.ConnectTimeout, requests.exceptions.ReadTimeout, requests.exceptions.Timeout, requests.exceptions.SSLError, requests.exceptions.ChunkedEncodingError, ) _api_proxy_lock = threading.Lock() _api_proxy_cache_path = None _api_proxy_cache_mtime = None _api_proxy_cache_entries = [] _api_proxy_cache_index = 0 def _redacted_proxy_url(proxy_url): try: parsed = urlsplit(proxy_url) if '@' not in parsed.netloc: return proxy_url host = parsed.hostname or '' port = f':{parsed.port}' if parsed.port else '' return urlunsplit((parsed.scheme, f'***:***@{host}{port}', parsed.path, parsed.query, parsed.fragment)) except Exception: return '' def parse_proxy_line(line): line = str(line or '').strip() if not line or line.startswith('#'): return None if '://' in line: proxy_url = line else: parts = line.split(':', 3) if len(parts) == 2: host, port = parts proxy_url = f'http://{host}:{port}' elif len(parts) == 4: host, port, username, password = parts credentials = f'{quote(username, safe="")}:{quote(password, safe="")}' proxy_url = f'http://{credentials}@{host}:{port}' else: raise ValueError('expected host:port or host:port:username:password') return {'http': proxy_url, 'https': proxy_url} def load_proxy_entries(proxy_file): entries = [] if not proxy_file or not os.path.exists(proxy_file): return entries with open(proxy_file, 'r', encoding='utf-8') as f: for line_number, line in enumerate(f, 1): try: proxy = parse_proxy_line(line) except ValueError as e: logger.warning(f'Ignoring bad proxy line {proxy_file}:{line_number}: {e}') continue if proxy: entries.append(proxy) return entries def next_api_proxy(): global _api_proxy_cache_path, _api_proxy_cache_mtime, _api_proxy_cache_entries, _api_proxy_cache_index if not scan_config.api_proxy_enabled: return None proxy_file = scan_config.api_proxy_file or scan_config.proxy_file try: mtime = os.path.getmtime(proxy_file) if proxy_file else None except OSError: mtime = None with _api_proxy_lock: if proxy_file != _api_proxy_cache_path or mtime != _api_proxy_cache_mtime: _api_proxy_cache_path = proxy_file _api_proxy_cache_mtime = mtime _api_proxy_cache_entries = load_proxy_entries(proxy_file) _api_proxy_cache_index = 0 if _api_proxy_cache_entries: logger.info(f'Loaded {len(_api_proxy_cache_entries)} API proxy entry(ies) from {proxy_file}') if not _api_proxy_cache_entries: raise ApiRequestError(f'API proxy is enabled but no valid proxies are loaded from {proxy_file}') proxy = _api_proxy_cache_entries[_api_proxy_cache_index % len(_api_proxy_cache_entries)] _api_proxy_cache_index += 1 return proxy def _short_url(url): try: parsed = urlsplit(url) return urlunsplit((parsed.scheme, parsed.netloc, parsed.path, '', '')) except Exception: return str(url) def _log_api_retry(method, url, attempt, attempts, error, proxy): if attempt != 1 and attempt % 10 != 0 and attempt != attempts: return proxy_url = None if proxy: proxy_url = proxy.get('https') or proxy.get('http') proxy_part = f' via {_redacted_proxy_url(proxy_url)}' if proxy_url else '' logger.warning(f'API {method} {_short_url(url)} failed ({attempt}/{attempts}){proxy_part}: {str(error)[:300]}') def _direct_request(method, url, **kwargs): # None removes merged proxy routes; no_proxy also blocks environment rebuilds # on redirects. Keep Requests' certificate and streamed-response behavior. kwargs['proxies'] = {'http': None, 'https': None, 'all': None, 'no_proxy': '*'} return requests.request(method, url, **kwargs) def api_request( method, url, *, timeout=None, max_retries=None, retry_delay=None, retry_statuses=None, deadline=None, use_proxy=None, **kwargs, ): # False opts out; other values retain the configured discovery policy. use_proxy = use_proxy is not False and scan_config.api_proxy_enabled attempts = max(1, int(max_retries if max_retries is not None else (scan_config.api_proxy_max_retries if use_proxy else 1))) delay = max(0, int(retry_delay if retry_delay is not None else scan_config.api_proxy_retry_delay)) retry_statuses = set(API_RETRY_STATUSES if retry_statuses is None else retry_statuses) request_timeout = timeout if use_proxy and scan_config.api_proxy_timeout is not None: if isinstance(timeout, (tuple, list)): connect_timeout, read_timeout = timeout else: connect_timeout = read_timeout = timeout proxy_timeout = float(scan_config.api_proxy_timeout) # Proxy connection limits must not replace the caller's read budget. request_timeout = ( proxy_timeout if connect_timeout is None else min(proxy_timeout, float(connect_timeout)), proxy_timeout if timeout is None else read_timeout, ) last_error = None for attempt in range(1, attempts + 1): _raise_if_scan_slot_fatal() remaining = None if deadline is None else float(deadline) - time.monotonic() if remaining is not None and remaining <= 0: raise ApiRequestError(f'API request deadline expired before attempt {attempt}: {method} {_short_url(url)}') proxy = next_api_proxy() if use_proxy else None request_kwargs = dict(kwargs) if proxy: request_kwargs['proxies'] = proxy effective_timeout = request_timeout if remaining is not None and effective_timeout is None: effective_timeout = max(0.001, remaining) elif remaining is not None and isinstance(effective_timeout, (int, float)): effective_timeout = max(0.001, min(float(effective_timeout), remaining)) elif remaining is not None and isinstance(effective_timeout, (tuple, list)): effective_timeout = tuple( max(0.001, remaining if value is None else min(float(value), remaining)) for value in effective_timeout ) try: request = requests.request if use_proxy else _direct_request response = request(method, url, timeout=effective_timeout, **request_kwargs) if _scan_slot_fatal_event.is_set(): response.close() _raise_if_scan_slot_fatal() if deadline is not None and time.monotonic() >= float(deadline): response.close() raise ApiRequestError(f'API request deadline expired after response: {method} {_short_url(url)}') if response.status_code in retry_statuses: detail = '' if not request_kwargs.get('stream'): detail = response.text[:300] if response.text else '' last_error = f'HTTP {response.status_code}' + (f': {detail}' if detail else '') if attempt >= attempts: failure = ApiRequestError( f'API request failed after {attempts} attempt(s): ' f'{method} {_short_url(url)}: {last_error}', response=response, operation='provider-api-request', capture_body=not bool(request_kwargs.get('stream')), ) response.close() raise failure _log_api_retry(method, url, attempt, attempts, last_error, proxy) response.close() if deadline is not None and time.monotonic() + delay >= float(deadline): raise ApiRequestError(f'API request deadline expired during retry: {method} {_short_url(url)}') _wait_or_raise_scan_slot_fatal(delay) continue return response except API_RETRY_EXCEPTIONS as e: try: url_has_query = bool(urlsplit(str(url)).query) except ValueError: url_has_query = True safe_error = ( type(e).__name__ if request_kwargs.get('stream') or url_has_query else str(e)[:300] ) last_error = safe_error _log_api_retry(method, url, attempt, attempts, safe_error, proxy) if attempt >= attempts: raise ApiRequestError( f'API request failed after {attempts} attempt(s): ' f'{method} {_short_url(url)}: {safe_error}' ) from e if deadline is not None and time.monotonic() + delay >= float(deadline): raise ApiRequestError(f'API request deadline expired during retry: {method} {_short_url(url)}') from e _wait_or_raise_scan_slot_fatal(delay) raise ApiRequestError(f'API request failed after {attempts} attempt(s): {method} {_short_url(url)}: {last_error}') SCAN_SLOT_SCHEMA = """ CREATE TABLE IF NOT EXISTS scan_slots ( slot_id TEXT PRIMARY KEY, owner_pid INTEGER NOT NULL, owner_thread INTEGER NOT NULL, owner_source TEXT, owner_creation_time TEXT, owner_executable TEXT, child_pid INTEGER, child_creation_time TEXT, child_executable TEXT, slot_kind TEXT NOT NULL DEFAULT 'base' CHECK(slot_kind IN ('base', 'bonus')), command TEXT, acquired_at REAL NOT NULL, updated_at REAL NOT NULL ); CREATE TABLE IF NOT EXISTS scan_waiters ( waiter_id TEXT PRIMARY KEY, owner_pid INTEGER NOT NULL, owner_thread INTEGER NOT NULL, owner_source TEXT NOT NULL, owner_creation_time TEXT NOT NULL, owner_executable TEXT NOT NULL, enqueued_at REAL NOT NULL ); CREATE INDEX IF NOT EXISTS idx_scan_waiters_fair ON scan_waiters(owner_source, enqueued_at, waiter_id); CREATE TABLE IF NOT EXISTS scan_source_fairness ( owner_source TEXT PRIMARY KEY, last_granted_at REAL NOT NULL ); """ _scan_limiter_init_lock = threading.Lock() _scan_limiter_initialized_paths = set() _scan_slot_scope_local = threading.local() _scan_slot_fatal_event = threading.Event() _scan_slot_fatal_lock = threading.Lock() _scan_slot_fatal_detail = None class _ScanSlotScope: def __init__(self, lease): self.lease = lease class ScanSlotFatalError(RuntimeError): pass def _set_scan_slot_fatal(detail): global _scan_slot_fatal_detail bounded = str(detail or 'scan-slot durability failure')[:1000] with _scan_slot_fatal_lock: if not _scan_slot_fatal_event.is_set(): _scan_slot_fatal_detail = bounded _scan_slot_fatal_event.set() def _raise_if_scan_slot_fatal(): if not _scan_slot_fatal_event.is_set(): return with _scan_slot_fatal_lock: detail = _scan_slot_fatal_detail raise ScanSlotFatalError(detail or 'FATAL: scan-slot limiter is in an indeterminate state') def _wait_or_raise_scan_slot_fatal(delay): remaining = max(0.0, float(delay or 0)) while remaining > 0: _raise_if_scan_slot_fatal() interval = min(0.2, remaining) time.sleep(interval) remaining -= interval _raise_if_scan_slot_fatal() class ScanSlotLease: DB_RETRY_ATTEMPTS = 5 RELEASE_PENDING_DB_ATTEMPTS = 2 DB_RETRY_DELAY_SEC = 0.1 MIN_HEARTBEAT_INTERVAL_SEC = 5.0 RELEASE_PENDING_INTERVAL_SEC = 1.0 HEARTBEAT_JOIN_TIMEOUT_SEC = 2.0 def __init__(self, slot_id, db_path, owner_pid=None, owner_thread=None): self.slot_id = slot_id self.db_path = db_path self.owner_pid = int(os.getpid() if owner_pid is None else owner_pid) self.owner_thread = int(threading.get_ident() if owner_thread is None else owner_thread) self.child_pid = None self.heartbeat_stop = threading.Event() self.heartbeat_thread = None self._heartbeat_wake = threading.Event() self._state_lock = threading.Lock() self._db_lock = threading.Lock() self._release_call_lock = threading.Lock() self._released = False self._releasable = True self._release_requested = False self._release_pending = False self._heartbeat_interval = 30.0 self._release_pending_interval = self.RELEASE_PENDING_INTERVAL_SEC self._last_release_error_log_at = 0.0 @property def released(self): with self._state_lock: return self._released @property def releasable(self): with self._state_lock: return self._releasable @property def release_pending(self): with self._state_lock: return self._release_pending def _connect(self): return sqlite3.connect(self.db_path, timeout=30) def _close_connection(self, conn): try: conn.close() except Exception as exc: logger.warning('Unable to close scan-slot DB connection for %s: %s', self.slot_id, exc) def _execute_update_once(self, sql, params): conn = None try: conn = self._connect() conn.execute('PRAGMA busy_timeout=30000') cursor = conn.execute(sql, params) conn.commit() return cursor.rowcount != 0 finally: if conn is not None: self._close_connection(conn) def _delete_slot_once(self): conn = None try: conn = self._connect() conn.execute('PRAGMA busy_timeout=30000') cursor = conn.execute( '''DELETE FROM scan_slots WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''', (self.slot_id, self.owner_pid, self.owner_thread), ) if cursor.rowcount != 1: existing = conn.execute( 'SELECT owner_pid, owner_thread FROM scan_slots WHERE slot_id = ?', (self.slot_id,), ).fetchone() if existing is not None: raise RuntimeError( f'scan slot identity changed from pid/thread ' f'{self.owner_pid}/{self.owner_thread} to {existing[0]}/{existing[1]}' ) conn.commit() return True finally: if conn is not None: self._close_connection(conn) def _retry_db_operation(self, operation, attempts): attempts = max(1, int(attempts)) last_error = None for attempt in range(1, attempts + 1): try: with self._db_lock: return bool(operation()), None except (sqlite3.Error, OSError) as exc: last_error = exc except Exception as exc: last_error = exc break if attempt < attempts: time.sleep(max(0.0, float(self.DB_RETRY_DELAY_SEC)) * attempt) return False, last_error def update(self, sql, params, attempts=None, log_failure=True): success, last_error = self._retry_db_operation( lambda: self._execute_update_once(sql, params), self.DB_RETRY_ATTEMPTS if attempts is None else attempts, ) if last_error is not None and log_failure: logger.warning('Unable to update scan slot %s: %s', self.slot_id, last_error) return success def set_child_pid(self, child_pid): if not child_pid: return False child_pid = int(child_pid) with self._release_call_lock: with self._state_lock: if ( self._released or not self._releasable or self._release_requested or not self.slot_id or not self.db_path ): return False identity = capture_process_identity(child_pid) updated = self.update( '''UPDATE scan_slots SET child_pid = ?, child_creation_time = ?, child_executable = ?, updated_at = ? WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''', ( child_pid, identity.get('creation_time') if identity else None, identity.get('executable') if identity else None, time.time(), self.slot_id, self.owner_pid, self.owner_thread, ), ) if updated: with self._state_lock: self.child_pid = child_pid return updated def mark_non_releasable(self): with self._release_call_lock: with self._state_lock: if self._released: return self._releasable = False self._release_pending = False self._heartbeat_wake.set() def _heartbeat_update(self, attempts=None, log_failure=True): return self.update( '''UPDATE scan_slots SET updated_at = ? WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''', (time.time(), self.slot_id, self.owner_pid, self.owner_thread), attempts=attempts, log_failure=log_failure, ) def start_heartbeat(self): interval = max( float(self.MIN_HEARTBEAT_INTERVAL_SEC), float_setting(getattr(scan_config, 'scan_slot_heartbeat_sec', 30), 30), ) release_interval = max(0.01, min(interval, float(self.RELEASE_PENDING_INTERVAL_SEC))) with self._state_lock: if self._released: return False if self.heartbeat_thread is not None and self.heartbeat_thread.is_alive(): return True self._heartbeat_interval = interval self._release_pending_interval = release_interval thread = threading.Thread( target=self._heartbeat_loop, name=f'scan-slot-heartbeat-{self.slot_id[:8]}', daemon=True, ) self.heartbeat_thread = thread try: thread.start() except Exception: self.heartbeat_thread = None raise return True def transfer_to_current_thread(self): new_thread = int(threading.get_ident()) with self._release_call_lock: with self._state_lock: if self._released or self._release_requested or not self._releasable: raise RuntimeError(f'scan slot {self.slot_id} cannot be transferred after release started') if self.heartbeat_thread is not None: raise RuntimeError(f'scan slot {self.slot_id} cannot be transferred after heartbeat start') old_thread = self.owner_thread if old_thread == new_thread: return True updated = self.update( '''UPDATE scan_slots SET owner_thread = ?, updated_at = ? WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''', (new_thread, time.time(), self.slot_id, self.owner_pid, old_thread), ) if not updated: raise RuntimeError(f'scan slot {self.slot_id} ownership transfer was not confirmed') with self._state_lock: self.owner_thread = new_thread return True def _heartbeat_loop(self): while True: with self._state_lock: if self._released: return pending = self._release_pending and self._releasable interval = self._release_pending_interval if pending else self._heartbeat_interval self._heartbeat_wake.wait(interval) self._heartbeat_wake.clear() with self._state_lock: if self._released or self.heartbeat_stop.is_set(): return pending = self._release_pending and self._releasable if not pending: self._heartbeat_update() continue with self._release_call_lock: with self._state_lock: pending = self._release_pending and self._releasable and not self._released if not pending: continue success, last_error = self._retry_db_operation( self._delete_slot_once, self.RELEASE_PENDING_DB_ATTEMPTS, ) if success: completed = self._complete_release() if completed: logger.info('Released scan slot %s after background retry', self.slot_id) return self._log_pending_release_failure(last_error) # Keep the exact owner row fresh when DELETE itself is temporarily unavailable. self._heartbeat_update(attempts=1, log_failure=False) def _log_pending_release_failure(self, error): now = time.monotonic() with self._state_lock: if now - self._last_release_error_log_at < 30: return self._last_release_error_log_at = now logger.error('Scan slot %s release remains pending and fail-closed: %s', self.slot_id, error) def _complete_release(self): with self._state_lock: if self._released: return False self._released = True self._release_pending = False self.heartbeat_stop.set() self._heartbeat_wake.set() return True def _join_heartbeat(self): with self._state_lock: thread = self.heartbeat_thread if thread is None or thread is threading.current_thread() or not thread.is_alive(): return thread.join(timeout=max(0.0, float(self.HEARTBEAT_JOIN_TIMEOUT_SEC))) if thread.is_alive(): logger.warning('Scan-slot heartbeat did not stop promptly for %s', self.slot_id) def release(self): should_join = False success = False with self._release_call_lock: with self._state_lock: if self._released: success = True should_join = True elif ( not self._releasable or self._release_requested or not self.slot_id or not self.db_path ): return else: self._release_requested = True if not success: success, last_error = self._retry_db_operation( self._delete_slot_once, self.DB_RETRY_ATTEMPTS, ) if success: self._complete_release() should_join = True else: with self._state_lock: pending = not self._released and self._releasable if pending: self._release_pending = True self._last_release_error_log_at = time.monotonic() if pending: try: self.start_heartbeat() except Exception as exc: logger.error('Unable to start pending-release heartbeat for scan slot %s: %s', self.slot_id, exc) self._heartbeat_wake.set() logger.error( 'Unable to release scan slot %s after %d attempts; ' 'lease remains live and background retries will continue: %s', self.slot_id, self.DB_RETRY_ATTEMPTS, last_error, ) if should_join: self._join_heartbeat() def process_exists(pid): try: pid = int(pid) except (TypeError, ValueError): return False if pid <= 0: return False if pid == os.getpid(): return True if os.name == 'nt': try: import ctypes process_query_limited_information = 0x1000 handle = ctypes.windll.kernel32.OpenProcess(process_query_limited_information, False, pid) if handle: ctypes.windll.kernel32.CloseHandle(handle) return True return False except Exception: return True try: os.kill(pid, 0) return True except ProcessLookupError: return False except PermissionError: return True except Exception: return True def capture_process_identity(pid): try: process = open_process(int(pid)) except Exception: return None try: return { 'creation_time': str(process.identity.creation_time), 'executable': canonical_path(process.identity.executable), } finally: process.close() def exact_process_identity_live(pid, creation_time, executable): if not pid: return False if not creation_time or not executable: return None if process_exists(pid) else False try: process = open_process(int(pid)) except Exception: return None if process_exists(pid) else False try: return bool( process.is_running() and str(process.identity.creation_time) == str(creation_time) and canonical_path(process.identity.executable) == canonical_path(executable) ) finally: process.close() def scan_limiter_enabled(): return int_setting(getattr(scan_config, 'max_active_scans', 0), 0) > 0 def scan_limiter_db_path(): path = getattr(scan_config, 'scan_limiter_db', '') or '' if not path: defaults = default_project_paths() path = os.path.join(defaults['state_dir'], 'scan_limiter.db') return path def ensure_scan_limiter_db(path): parent = os.path.dirname(path) if parent: os.makedirs(parent, exist_ok=True) with _scan_limiter_init_lock: if path in _scan_limiter_initialized_paths: return conn = sqlite3.connect(path, timeout=30) try: conn.execute('PRAGMA busy_timeout=30000') conn.execute('PRAGMA journal_mode=WAL') conn.executescript(SCAN_SLOT_SCHEMA) conn.execute('BEGIN IMMEDIATE') existing = {row[1] for row in conn.execute('PRAGMA table_info(scan_slots)').fetchall()} if 'child_pid' not in existing: conn.execute('ALTER TABLE scan_slots ADD COLUMN child_pid INTEGER') for name, declaration in ( ('owner_creation_time', 'TEXT'), ('owner_executable', 'TEXT'), ('child_creation_time', 'TEXT'), ('child_executable', 'TEXT'), ): if name not in existing: conn.execute(f'ALTER TABLE scan_slots ADD COLUMN {name} {declaration}') if 'slot_kind' not in existing: conn.execute( "ALTER TABLE scan_slots ADD COLUMN slot_kind TEXT NOT NULL DEFAULT 'base'" ) conn.execute( "CREATE UNIQUE INDEX IF NOT EXISTS idx_scan_slots_single_bonus " "ON scan_slots(slot_kind) WHERE slot_kind = 'bonus'" ) conn.commit() finally: conn.close() _scan_limiter_initialized_paths.add(path) def connect_scan_limiter_db(path): ensure_scan_limiter_db(path) conn = sqlite3.connect(path, timeout=30) conn.execute('PRAGMA busy_timeout=30000') return conn def cleanup_stale_scan_slots(conn, now, stale_sec): columns = {row[1] for row in conn.execute('PRAGMA table_info(scan_slots)').fetchall()} child_expr = 'child_pid' if 'child_pid' in columns else 'NULL AS child_pid' owner_creation_expr = 'owner_creation_time' if 'owner_creation_time' in columns else 'NULL AS owner_creation_time' owner_executable_expr = 'owner_executable' if 'owner_executable' in columns else 'NULL AS owner_executable' child_creation_expr = 'child_creation_time' if 'child_creation_time' in columns else 'NULL AS child_creation_time' child_executable_expr = 'child_executable' if 'child_executable' in columns else 'NULL AS child_executable' rows = conn.execute( f'''SELECT slot_id, owner_pid, {owner_creation_expr}, {owner_executable_expr}, {child_expr}, {child_creation_expr}, {child_executable_expr}, acquired_at, updated_at FROM scan_slots''' ).fetchall() for slot_id, owner_pid, owner_creation, owner_executable, child_pid, child_creation, child_executable, acquired_at, updated_at in rows: heartbeat_age = now - float(updated_at or acquired_at or 0) owner_live = exact_process_identity_live(owner_pid, owner_creation, owner_executable) child_live = exact_process_identity_live(child_pid, child_creation, child_executable) if owner_live is False and child_live is False: conn.execute('DELETE FROM scan_slots WHERE slot_id = ?', (slot_id,)) elif heartbeat_age > stale_sec: logger.warning( 'Stale scan-slot heartbeat remains capacity-blocking: slot=%s owner_live=%s child_live=%s age=%.0fs', slot_id, owner_live, child_live, heartbeat_age, ) if 'scan_waiters' in { row[0] for row in conn.execute("SELECT name FROM sqlite_master WHERE type = 'table'").fetchall() }: waiters = conn.execute( '''SELECT waiter_id, owner_pid, owner_creation_time, owner_executable FROM scan_waiters''' ).fetchall() for waiter_id, owner_pid, owner_creation, owner_executable in waiters: if exact_process_identity_live(owner_pid, owner_creation, owner_executable) is False: conn.execute('DELETE FROM scan_waiters WHERE waiter_id = ?', (waiter_id,)) def redact_scan_command_text(cmd): text = ' '.join(str(part) for part in cmd) text = re.sub(r'(https?://)[^\s/@:]+:[^\s/@]+@', r'\1***:***@', text) text = re.sub(r'github_pat_[A-Za-z0-9_]+', 'github_pat_***', text) text = re.sub(r'gh[pousr]_[A-Za-z0-9_]+', 'ghp_***', text) text = re.sub(r'hf_[A-Za-z0-9]+', 'hf_***', text) return text[:1000] def windows_scan_capacity_snapshot(): if os.name != 'nt': raise OSError('opportunistic scan capacity is supported only on Windows') from ctypes import wintypes class PerformanceInformation(ctypes.Structure): _fields_ = [ ('cb', wintypes.DWORD), ('CommitTotal', ctypes.c_size_t), ('CommitLimit', ctypes.c_size_t), ('CommitPeak', ctypes.c_size_t), ('PhysicalTotal', ctypes.c_size_t), ('PhysicalAvailable', ctypes.c_size_t), ('SystemCache', ctypes.c_size_t), ('KernelTotal', ctypes.c_size_t), ('KernelPaged', ctypes.c_size_t), ('KernelNonpaged', ctypes.c_size_t), ('PageSize', ctypes.c_size_t), ('HandleCount', wintypes.DWORD), ('ProcessCount', wintypes.DWORD), ('ThreadCount', wintypes.DWORD), ] get_performance_info = ctypes.WinDLL('psapi', use_last_error=True).GetPerformanceInfo get_performance_info.argtypes = [ctypes.POINTER(PerformanceInformation), wintypes.DWORD] get_performance_info.restype = wintypes.BOOL info = PerformanceInformation() info.cb = ctypes.sizeof(info) if not get_performance_info(ctypes.byref(info), info.cb): raise ctypes.WinError(ctypes.get_last_error()) page_size = int(info.PageSize) if page_size <= 0 or int(info.CommitLimit) < int(info.CommitTotal): raise OSError('Windows returned invalid scan-capacity counters') return { 'available_physical_bytes': int(info.PhysicalAvailable) * page_size, 'commit_headroom_bytes': (int(info.CommitLimit) - int(info.CommitTotal)) * page_size, } def opportunistic_scan_slot_allowed(source): if max(0, min(1, int_setting(getattr(scan_config, 'opportunistic_scan_slots', 0), 0))) <= 0: return False eligible = { item.lower() for item in csv_items( getattr(scan_config, 'opportunistic_scan_sources', []) ) } if str(source or '').lower() not in eligible: return False try: job_limit = int(getattr(scan_config, 'trufflehog_job_memory_limit_bytes', 0)) overhead = max(0, int(getattr( scan_config, 'opportunistic_scan_reserve_overhead_bytes', 0, ))) reserve = job_limit + overhead if job_limit <= 0 or reserve <= 0: return False capacity = windows_scan_capacity_snapshot() available_after = int(capacity['available_physical_bytes']) - reserve commit_after = int(capacity['commit_headroom_bytes']) - reserve minimum_available = max(0, int(getattr( scan_config, 'opportunistic_scan_min_available_after_reserve_bytes', 0, ))) minimum_commit = max(0, int(getattr( scan_config, 'opportunistic_scan_min_commit_after_reserve_bytes', 0, ))) return available_after >= minimum_available and commit_after >= minimum_commit except (OSError, TypeError, ValueError, OverflowError, KeyError): return False def acquire_scan_slot(cmd, timeout_sec=None, wait=True, start_heartbeat=True): _raise_if_scan_slot_fatal() max_active = int_setting(getattr(scan_config, 'max_active_scans', 0), 0) if max_active <= 0: return None db_path = scan_limiter_db_path() wait_sec = max(0.1, float_setting(getattr(scan_config, 'scan_slot_wait_sec', 0.5), 0.5)) wait_log_sec = max(1, int_setting(getattr(scan_config, 'scan_slot_wait_log_sec', 30), 30)) stale_sec = max( int_setting(getattr(scan_config, 'scan_slot_stale_sec', 7200), 7200), int(timeout_sec or 0) + 300, ) source = os.getenv('SCANNER_SOURCE') or (cmd[1] if len(cmd) > 1 else 'unknown') command_text = redact_scan_command_text(cmd) owner_pid = os.getpid() owner_thread = threading.get_ident() slot_id = f'{owner_pid}-{owner_thread}-{uuid.uuid4().hex}' waiter_id = f'wait-{owner_pid}-{owner_thread}-{uuid.uuid4().hex}' owner_identity = current_process_identity() started_waiting = time.monotonic() enqueued_at = time.time() last_log_at = 0.0 waiter_registered = False while True: _raise_if_scan_slot_fatal() now = time.time() conn = None slot_committed = False try: conn = connect_scan_limiter_db(db_path) conn.execute('BEGIN IMMEDIATE') _raise_if_scan_slot_fatal() cleanup_stale_scan_slots(conn, now, stale_sec) conn.execute( '''INSERT OR IGNORE INTO scan_waiters( waiter_id, owner_pid, owner_thread, owner_source, owner_creation_time, owner_executable, enqueued_at ) VALUES (?, ?, ?, ?, ?, ?, ?)''', ( waiter_id, owner_pid, owner_thread, str(source), owner_identity.creation_time, canonical_path(owner_identity.executable), enqueued_at, ), ) waiter_registered = True base_active, bonus_active = conn.execute( "SELECT " "SUM(CASE WHEN slot_kind = 'base' THEN 1 ELSE 0 END), " "SUM(CASE WHEN slot_kind = 'bonus' THEN 1 ELSE 0 END) " "FROM scan_slots" ).fetchone() base_active = int(base_active or 0) bonus_active = int(bonus_active or 0) active = base_active + bonus_active next_waiter = conn.execute( '''SELECT w.waiter_id FROM scan_waiters w LEFT JOIN scan_source_fairness f ON f.owner_source = w.owner_source ORDER BY COALESCE(f.last_granted_at, 0), w.enqueued_at, w.waiter_id LIMIT 1''' ).fetchone() slot_kind = None if next_waiter and next_waiter[0] == waiter_id: if base_active < max_active: slot_kind = 'base' elif bonus_active < max(0, min(1, int_setting( getattr(scan_config, 'opportunistic_scan_slots', 0), 0, ))) and opportunistic_scan_slot_allowed(source): slot_kind = 'bonus' if slot_kind is not None: conn.execute( '''INSERT INTO scan_slots( slot_id, owner_pid, owner_thread, owner_source, owner_creation_time, owner_executable, slot_kind, command, acquired_at, updated_at ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', ( slot_id, owner_pid, owner_thread, str(source), owner_identity.creation_time, canonical_path(owner_identity.executable), slot_kind, command_text, now, now, ), ) conn.execute('DELETE FROM scan_waiters WHERE waiter_id = ?', (waiter_id,)) conn.execute( '''INSERT INTO scan_source_fairness(owner_source, last_granted_at) VALUES (?, ?) ON CONFLICT(owner_source) DO UPDATE SET last_granted_at = excluded.last_granted_at''', (str(source), now), ) conn.commit() slot_committed = True waiter_registered = False waited = time.monotonic() - started_waiting if waited >= wait_log_sec: hard_limit = max_active + max(0, min(1, int_setting( getattr(scan_config, 'opportunistic_scan_slots', 0), 0, ))) logger.info( f'Acquired {slot_kind} scan slot after waiting {waited:.0f}s ' f'({active + 1}/{hard_limit})' ) lease = ScanSlotLease(slot_id, db_path, owner_pid=owner_pid, owner_thread=owner_thread) if start_heartbeat: try: if not lease.start_heartbeat(): raise RuntimeError('scan-slot heartbeat did not start') except BaseException as start_error: logger.error( 'Scan-slot heartbeat failed to start for %s; synchronously removing the exact owner row: %s', slot_id, start_error, ) lease.release() if not lease.released: fatal_detail = ( 'FATAL: scan-slot heartbeat startup failed and exact-owner rollback ' f'could not be confirmed for slot {slot_id}' ) _set_scan_slot_fatal(fatal_detail) logger.critical( 'FATAL scan-slot acquisition rollback is unconfirmed for %s; capacity remains fail-closed', slot_id, ) raise ScanSlotFatalError(fatal_detail) from start_error raise return lease if not wait: conn.execute('DELETE FROM scan_waiters WHERE waiter_id = ?', (waiter_id,)) conn.commit() waiter_registered = False return None conn.commit() if time.monotonic() - last_log_at >= wait_log_sec: logger.info(f'Waiting for scan slot ({active}/{max_active} active)') last_log_at = time.monotonic() except sqlite3.OperationalError as e: if slot_committed: raise if not wait: return None if time.monotonic() - last_log_at >= wait_log_sec: logger.warning(f'Waiting for scan limiter DB lock: {str(e)}') last_log_at = time.monotonic() finally: if conn is not None: conn.close() if _scan_slot_fatal_event.wait(wait_sec): _raise_if_scan_slot_fatal() def acquire_scan_slot_leases(cmd, count, timeout_sec=None): count = max(0, int(count or 0)) if count <= 0 or not scan_limiter_enabled(): return [] leases = [] try: first = acquire_scan_slot(cmd, timeout_sec, wait=True, start_heartbeat=False) if first is not None: leases.append(first) while len(leases) < count: lease = acquire_scan_slot(cmd, timeout_sec, wait=False, start_heartbeat=False) if lease is None: break leases.append(lease) return leases except BaseException: for lease in leases: lease.release() raise @contextmanager def scan_slot_scope(cmd, timeout_sec=None, lease=None): """Own one physical lease through the target's durable bundle handoff.""" if getattr(_scan_slot_scope_local, 'scope', None) is not None: raise RuntimeError('scan slot scopes cannot be nested on one worker thread') if lease is None: lease = acquire_scan_slot(cmd, timeout_sec) else: try: lease.transfer_to_current_thread() if not lease.start_heartbeat(): raise RuntimeError('transferred scan-slot heartbeat did not start') except BaseException: lease.release() raise scope = _ScanSlotScope(lease) _scan_slot_scope_local.scope = scope try: yield lease finally: try: if lease and lease.releasable: lease.release() finally: if getattr(_scan_slot_scope_local, 'scope', None) is scope: del _scan_slot_scope_local.scope def scoped_scan_slot_lease(): scope = getattr(_scan_slot_scope_local, 'scope', None) return (scope is not None, scope.lease if scope is not None else None) def get_pending_temp_file(): work_dir = get_work_dir() if not work_dir: return None return os.path.join(work_dir, 'pending_cleanup.json') def get_pending_temp_lock_file(): path = get_pending_temp_file() return path + '.lock' if path else None def _load_persisted_pending_temp_dirs_unlocked(path): if not path or not os.path.exists(path): return set() try: value = read_private_json(path) except OSError: return set() if value.get('schema') != PENDING_TEMP_SCHEMA or not isinstance(value.get('paths'), list): return set() return {str(item) for item in value['paths'] if isinstance(item, str) and item.strip()} def _persist_pending_temp_dirs_unlocked(path, paths): normalized = sorted({str(item) for item in paths if str(item).strip()}) if normalized: atomic_write_private_json(path, {'schema': PENDING_TEMP_SCHEMA, 'paths': normalized}) elif os.path.exists(path): if not private_file_ready(path): raise OSError(f'refusing to remove non-private pending cleanup list: {path}') durable_unlink(path) def load_persisted_pending_temp_dirs(): path = get_pending_temp_file() lock_path = get_pending_temp_lock_file() if not path or not lock_path: return set() with PrivateFileLock(lock_path): return _load_persisted_pending_temp_dirs_unlocked(path) def persist_pending_temp_dirs(paths): path = get_pending_temp_file() lock_path = get_pending_temp_lock_file() if not path or not lock_path: return try: with PrivateFileLock(lock_path): _persist_pending_temp_dirs_unlocked(path, paths) except OSError: pass def redact_secrets(text, secrets): if not text: return text redacted = text for secret in secrets: if secret: redacted = redacted.replace(secret, '***REDACTED***') redacted = redacted.replace(quote(secret, safe=''), '***REDACTED***') return redacted def build_authenticated_git_url(repo_url, provider=None, token=None): if not token: return repo_url, [] parsed = urlsplit(repo_url) if parsed.scheme != 'https' or not parsed.netloc: return repo_url, [] hostname = (parsed.hostname or '').lower() detected_provider = 'gitlab' if hostname == 'gitlab.com' else 'github' if hostname == 'github.com' else None if provider and provider != detected_provider: return repo_url, [] provider = detected_provider if provider not in ('github', 'gitlab'): return repo_url, [] return repo_url, [token] def get_git_provider_and_path(repo_url, provider=None): parsed = urlsplit(repo_url) hostname = (parsed.hostname or '').lower() path = parsed.path.strip('/') if path.endswith('.git'): path = path[:-4] provider = provider or ('gitlab' if 'gitlab.' in hostname or hostname == 'gitlab.com' else 'github' if 'github.' in hostname or hostname == 'github.com' else None) return provider, path def recent_commit_boundary(repo_url, provider=None, token=None, max_age_days=None, lookup_pages=3): if not max_age_days or max_age_days <= 0: return {'since_commit': None, 'skip': False, 'reason': ''} provider, repo_path = get_git_provider_and_path(repo_url, provider) if provider not in ('github', 'gitlab') or not repo_path: return {'since_commit': None, 'skip': True, 'reason': 'unsupported provider for commit age lookup'} cutoff = datetime.now(timezone.utc) - timedelta(days=max_age_days) since = cutoff.isoformat().replace('+00:00', 'Z') headers = {'User-Agent': 'GitSecretsScanner/2.0'} if token: headers['Authorization'] = f'Bearer {token}' commits = [] lookup_cap_reached = False for page in range(1, max(1, lookup_pages) + 1): try: if provider == 'github': url = f'https://api.github.com/repos/{repo_path}/commits' params = {'since': since, 'per_page': 100, 'page': page} else: url = f'https://gitlab.com/api/v4/projects/{quote(repo_path, safe="")}/repository/commits' params = {'since': since, 'per_page': 100, 'page': page} response = api_request('GET', url, headers=headers, params=params, timeout=30) if token and response.status_code in (401, 403): anonymous = api_request( 'GET', url, headers={'User-Agent': 'GitSecretsScanner/2.0'}, params=params, timeout=30, ) if anonymous.status_code < 400: response = anonymous response.raise_for_status() page_commits = response.json() if not page_commits: break commits.extend(page_commits) if len(page_commits) < 100: break if page == max(1, lookup_pages): lookup_cap_reached = True except requests.exceptions.HTTPError as e: api_error = github_api_error(e.response) if provider == 'github' else gitlab_api_error(e.response) if api_error.category == 'not_found': return { 'since_commit': None, 'skip': True, 'permanent': True, 'reason': f'commit age lookup found no repository: {str(api_error)[:300]}', 'error_category': api_error.category, 'auth_related': False, } return { 'since_commit': None, 'skip': False, 'error': True, 'reason': f'commit age lookup failed ({api_error.category}): {str(api_error)[:300]}', 'error_category': api_error.category, 'auth_related': bool(getattr(api_error, 'auth_related', True)), } except Exception as e: if isinstance(e, ApiRequestError): return { 'since_commit': None, 'skip': False, 'error': True, 'reason': f'commit age lookup transport failed: {str(e)[:300]}', 'error_category': 'network', 'auth_related': False, } return { 'since_commit': None, 'skip': False, 'error': True, 'reason': f'commit age lookup failed: {str(e)[:300]}', 'error_category': 'unknown', 'auth_related': False, } if not commits: return { 'since_commit': None, 'skip': True, 'reason': f'no commits newer than {max_age_days} days' } if lookup_cap_reached: return { 'since_commit': None, 'skip': False, 'reason': 'commit lookup cap reached; scanning without since-commit boundary', 'recent_commit_count': len(commits), 'cutoff': since, } oldest = commits[-1] if provider == 'github': parents = oldest.get('parents') or [] parent_sha = parents[0].get('sha') if parents else None oldest_sha = oldest.get('sha') else: parents = oldest.get('parent_ids') or [] parent_sha = parents[0] if parents else None oldest_sha = oldest.get('id') return { 'since_commit': parent_sha, 'skip': False, 'reason': '', 'recent_commit_count': len(commits), 'cutoff': since, 'boundary_commit': oldest_sha, } def get_trufflehog_cmd(): """Return configured TruffleHog executable path.""" return scan_config.trufflehog_path or "trufflehog" def require_trufflehog_launch_authority(command=None): child_kind = str(os.getenv(CHILD_KIND_ENV) or 'scanner').strip().lower() if child_kind not in {'scanner', 'docker-shadow'}: raise RuntimeError('TruffleHog launch requires scanner or Docker shadow authority') manifest = _client_scan_manifest.get() metadata = {'code_manifest': manifest, 'authority': 'remote-worker'} if manifest else None if metadata is None: metadata = require_active_supervisor_child(child_kind=child_kind, require_dsn=True) manifest = metadata.get('code_manifest') or {} expected = (manifest.get('executables') or {}).get('trufflehog') or {} candidate = resolve_manifest_executable((command or [get_trufflehog_cmd()])[0]) if canonical_path(candidate) != canonical_path(expected.get('path') or ''): raise RuntimeError('TruffleHog command does not match immutable supervisor authority') values = list(command or []) if '--config' in values: try: policy_path = canonical_path(values[values.index('--config') + 1]) except (IndexError, TypeError, ValueError) as exc: raise RuntimeError('TruffleHog policy argument is incomplete') from exc assets = manifest.get('assets') or {} if policy_path not in {canonical_path(item.get('path') or '') for item in assets.values() if isinstance(item, dict)}: raise RuntimeError('TruffleHog policy does not match immutable supervisor authority') return metadata def get_git_cmd(): """Return manifested Git in runtime; uninitialized tests may resolve PATH.""" manifest = _client_scan_manifest.get() if manifest: return str(((manifest.get('executables') or {}).get('git') or {}).get('path') or '') if not _runtime_initialized: return shutil.which('git') or 'git' if str(os.getenv(CHILD_KIND_ENV) or 'scanner').strip().lower() != 'scanner': raise RuntimeError('Git clone launch requires scanner authority') manifest = _client_scan_manifest.get() if manifest: metadata = {'code_manifest': manifest, 'authority': 'remote-worker'} else: metadata = require_active_supervisor_child(child_kind='scanner', require_dsn=True) manifest = metadata.get('code_manifest') or {} expected = (manifest.get('executables') or {}).get('git') or {} path = expected.get('path') if not isinstance(path, str) or not os.path.isabs(path): raise RuntimeError('Git executable is absent from immutable supervisor authority') return path def prepend_client_git_environment(env): manifest = _client_scan_manifest.get() if manifest is None: return env path = str(((manifest.get('executables') or {}).get('git') or {}).get('path') or '') if not os.path.isabs(path): raise RuntimeError('remote worker Git executable is absent from immutable authority') directory = os.path.dirname(path) env['PATH'] = os.pathsep.join((directory, env.get('PATH', ''))) return env def require_git_clone_launch_authority(cmd): """Authorize only checkout-free HTTPS clones, never general Git commands.""" if ( not isinstance(cmd, (list, tuple)) or len(cmd) != 7 or any(not isinstance(value, str) or not value or any(ord(ch) < 32 or ord(ch) == 127 for ch in value) for value in cmd) or list(cmd[1:5]) != ['clone', '--no-checkout', '--no-recurse-submodules', '--'] ): raise RuntimeError('Git clone command does not match the allowed argv contract') source, destination = cmd[5:] try: parsed = urlsplit(source) valid_source = ( source.startswith('https://') and bool(parsed.hostname) and parsed.username is None and parsed.password is None and bool(parsed.path) and parsed.path.startswith('/') and not parsed.query and not parsed.fragment and not any(ch.isspace() for ch in source) and '\\' not in source and '%' not in parsed.netloc and parsed.port != 0 ) except ValueError: valid_source = False if not valid_source: raise RuntimeError('Git clone source must be credential-free absolute HTTPS without query or fragment') if ( not os.path.isabs(destination) or destination.startswith('-') or (os.name == 'nt' and not os.path.splitdrive(destination)[0]) ): raise RuntimeError('Git clone destination must be an absolute path') if not os.path.isabs(cmd[0]): raise RuntimeError('Git command does not match immutable supervisor authority') if str(os.getenv(CHILD_KIND_ENV) or 'scanner').strip().lower() != 'scanner': raise RuntimeError('Git clone launch requires scanner authority') manifest = _client_scan_manifest.get() if manifest: metadata = {'code_manifest': manifest, 'authority': 'remote-worker'} else: metadata = require_active_supervisor_child(child_kind='scanner', require_dsn=True) manifest = metadata.get('code_manifest') or {} expected = (manifest.get('executables') or {}).get('git') or {} # Authentication rehashes manifest contents, not just path/mtime identity. if cmd[0] != expected.get('path'): raise RuntimeError('Git command does not match immutable supervisor authority') return metadata def get_trufflehog_config(config_path=None): """Return configured TruffleHog custom detector config path, if any.""" return str(config_path if config_path is not None else getattr(scan_config, 'trufflehog_config', '') or '').strip() def append_trufflehog_scan_args(cmd, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None): """Append common TruffleHog scan flags in one place.""" config_path = get_trufflehog_config(trufflehog_config) if config_path: cmd.extend(['--config', config_path]) if detectors: cmd.extend(['--include-detectors', detectors]) if exclude_detectors: cmd.extend(['--exclude-detectors', exclude_detectors]) if no_verification: cmd.append('--no-verification') return cmd def get_work_dir(): """Prepare and return the directory used for TruffleHog temporary data.""" require_scanner_runtime_initialized() if not scan_config.work_dir: raise RuntimeError('TruffleHog work_dir is required') try: return require_private_directory(scan_config.work_dir, create=False) except OSError as exc: raise RuntimeError(f'Unable to use private TruffleHog work_dir {scan_config.work_dir}: {exc}') from exc def create_command_work_dir(): """Create an isolated temporary directory for one TruffleHog subprocess.""" work_dir = get_work_dir() ensure_work_dir_space() path = tempfile.mkdtemp(prefix='trufflehog-run-', dir=work_dir) try: harden_private_directory(path) if not write_temp_owner(path, ['scanner-workdir'], os.getpid(), required=True): raise RuntimeError(f'Unable to write required temp owner marker for {path}') return path except Exception: try_remove_tree(path, attempts=2, delay=0.2) raise def write_temp_owner(path, cmd=None, owner_pid=None, required=False, owner_identity=None): require_scanner_runtime_initialized() if not path: return try: owner_pid = int((owner_identity or {}).get('pid') if isinstance(owner_identity, dict) else owner_pid or os.getpid()) parent_identity = current_process_identity() if isinstance(owner_identity, dict): owner = dict(owner_identity) elif owner_pid == parent_identity.pid: owner = serialize_process_identity(parent_identity) else: with open_process(owner_pid) as retained: owner = serialize_process_identity(retained.identity) work_root = canonical_path(get_work_dir()) candidate = canonical_path(path) relative = os.path.relpath(candidate, work_root) if relative == '.' or relative.startswith('..' + os.sep) or os.path.isabs(relative): raise RuntimeError('temp owner marker path escapes configured work_dir') parent = serialize_process_identity(parent_identity) payload = { 'schema': TEMP_OWNER_SCHEMA, 'owner_pid': owner['pid'], 'owner_creation_time': owner['creation_time'], 'owner_executable': owner['executable'], 'parent_pid': parent['pid'], 'parent_creation_time': parent['creation_time'], 'parent_executable': parent['executable'], 'created_at': datetime.now(timezone.utc).isoformat(timespec='seconds'), 'root_kind': 'work', 'relative_path': relative.replace(os.sep, '/'), 'command': redact_command_args(cmd or [])[:64], } atomic_write_private_json(os.path.join(path, TEMP_OWNER_FILE), payload) return True except (OSError, ValueError) as e: if required: raise RuntimeError(f'Unable to write required temp owner marker for {path}: {e}') from e logger.warning(f'Unable to write temp owner marker for {path}: {str(e)[:200]}') return False def redact_command_args(args): sensitive_flags = {'--token', '--docker-token', '--password', '--api-key', '--secret'} redacted = [] hide_next = False for value in args or []: text = str(value) if hide_next: redacted.append('***REDACTED***') hide_next = False continue flag = text.split('=', 1)[0].lower() if flag in sensitive_flags: if '=' in text: redacted.append(text.split('=', 1)[0] + '=***REDACTED***') else: redacted.append(text) hide_next = True continue try: parsed = urlsplit(text) if parsed.scheme in ('http', 'https') and parsed.hostname and ('@' in parsed.netloc or parsed.password): netloc = parsed.hostname if parsed.port: netloc += f':{parsed.port}' else: netloc = parsed.netloc if parsed.scheme in ('http', 'https') and parsed.hostname: query = [] for key, query_value in parse_qsl(parsed.query, keep_blank_values=True): sensitive = any(part in key.lower() for part in ('password', 'passwd', 'pwd', 'token', 'secret', 'credential')) query.append((key, '***REDACTED***' if sensitive else query_value)) text = urlunsplit((parsed.scheme, netloc, parsed.path, urlencode(query), parsed.fragment)) except ValueError: pass redacted.append(text) return redacted def read_temp_owner(path): marker = os.path.join(path, TEMP_OWNER_FILE) try: if not private_file_ready(marker): return {} data = read_private_json(marker) return data if isinstance(data, dict) else {} except (OSError, ValueError): return {} def temp_dir_active(path): owner = read_temp_owner(path) if owner.get('schema') != TEMP_OWNER_SCHEMA: return True states = [ exact_process_identity_state( owner.get(f'{prefix}_pid'), owner.get(f'{prefix}_creation_time'), owner.get(f'{prefix}_executable'), ) for prefix in ('owner', 'parent') ] return any(state in ('alive', 'unknown') for state in states) def create_docker_config_dir(): work_dir = get_work_dir() docker_config_root = os.path.join(work_dir, 'docker-config') require_private_directory(docker_config_root, create=True) path = tempfile.mkdtemp(prefix='docker-config-', dir=docker_config_root) harden_private_directory(path) write_temp_owner(path, ['docker-auth-config'], os.getpid()) return path def force_remove_readonly(function, path, exc_info): try: reject_reparse_components(path) os.chmod(path, stat.S_IWRITE) function(path) except Exception: pass def try_remove_tree(path, attempts=1, delay=0.0): if not path: return True for attempt in range(attempts): try: budget = JanitorBudget( max_candidates=1, max_entries=10000, max_bytes=1024 * 1024 * 1024, max_seconds=5.0, max_depth=64, ) return bounded_remove_tree(path, budget) except FileNotFoundError: return True except KeyboardInterrupt: raise except Exception: if delay and attempt + 1 < attempts: time.sleep(delay) return False def _shared_staging_owners(roots): """Resolve only private trees already owned by this scanner; never adopt input data.""" work_root = canonical_path(get_work_dir()) current = serialize_process_identity(current_process_identity()) owners = {} for value in roots: root = canonical_path(require_private_directory(value, create=False)) if root == work_root or os.path.commonpath((root, work_root)) != work_root: raise RuntimeError('shared staging root escapes configured work_dir') while root != work_root and not os.path.lexists(os.path.join(root, TEMP_OWNER_FILE)): root = os.path.dirname(root) if root == work_root: raise RuntimeError('shared staging root has no authenticated owner') if root in owners: continue require_private_directory(root, create=False) marker = read_private_json(require_private_file(os.path.join(root, TEMP_OWNER_FILE)), max_bytes=65536) relative = os.path.relpath(root, work_root).replace(os.sep, '/') if marker.get('schema') != TEMP_OWNER_SCHEMA or marker.get('root_kind') != 'work' or marker.get('relative_path') != relative: raise RuntimeError('shared staging owner marker does not match its private root') if any(marker.get(f'{prefix}_{field}') != current[field] for prefix in ('owner', 'parent') for field in ('pid', 'creation_time', 'executable')): raise RuntimeError('shared staging root belongs to another owner') if any(f'child_{field}' in marker for field in ('pid', 'creation_time', 'executable')): if any(not marker.get(f'child_{field}') for field in ('pid', 'creation_time', 'executable')) or ( exact_process_identity_state(marker.get('child_pid'), marker.get('child_creation_time'), marker.get('child_executable')) != 'dead' ): raise RuntimeError('shared staging root has an unconfirmed child') for field in ('pid', 'creation_time', 'executable'): marker.pop(f'child_{field}', None) owners[root] = marker return list(owners.items()) def cleanup_command_work_dir(path): if not path: return marker_path = os.path.join(path, TEMP_OWNER_FILE) if os.path.lexists(marker_path): try: marker = read_private_json(require_private_file(marker_path), max_bytes=65536) if any(f'child_{field}' in marker for field in ('pid', 'creation_time', 'executable')): if any(not marker.get(f'child_{field}') for field in ('pid', 'creation_time', 'executable')) or ( exact_process_identity_state(marker.get('child_pid'), marker.get('child_creation_time'), marker.get('child_executable')) != 'dead' ): logger.warning('Retaining private command tree with an unconfirmed child') return except (OSError, TypeError, ValueError): logger.warning('Retaining private command tree with unreadable ownership evidence') return if try_remove_tree(path, attempts=2, delay=0.2): return logger.info('Bounded immediate cleanup deferred to authenticated janitor: %s', path) def approved_pending_temp_path(path, work_dir=None): """Validate scanner-owned placement without following an external path.""" if not path or not os.path.isabs(path): return False try: work_root = work_dir or get_work_dir() reject_reparse_components(work_root) reject_reparse_components(path) if is_reparse_point(path) or not os.path.isdir(path) or not private_directory_ready(path): return False work_root = canonical_path(work_root) candidate = canonical_path(path) if os.path.commonpath((work_root, candidate)) != work_root or candidate == work_root: return False relative = os.path.relpath(candidate, work_root) except (OSError, ValueError): return False parts = relative.split(os.sep) if len(parts) == 1: approved_name = parts[0].startswith(APPROVED_TEMP_PREFIXES) elif len(parts) == 2 and parts[0] == 'docker-config': approved_name = parts[1].startswith('docker-config-') elif len(parts) == 2 and parts[0] == 'hg': approved_name = parts[1].startswith('hg-run-') elif len(parts) == 2 and parts[0] == 'tmp': approved_name = parts[1].startswith(APPROVED_TEMP_PREFIXES) elif len(parts) == 3 and parts[:2] == ['tmp', 'docker-config']: approved_name = parts[2].startswith('docker-config-') else: approved_name = False if not approved_name: return False owner = read_temp_owner(candidate) owner_pid = owner.get('owner_pid') or owner.get('parent_pid') return bool(owner_pid) and not temp_dir_active(candidate) def cleanup_pending_command_work_dirs(max_items=None, attempts=1, delay=0.0, log_failures=False): if log_failures: logger.info('Source-side pending temp cleanup is retired; the authenticated janitor owns recovery') return 0 def cleanup_assignment_work_dir(max_items=256): """Bound one cooperative cleanup pass to this runner's private work root.""" work_root = get_work_dir() report = {'enumerated': 0, 'removed': 0, 'retained': 0} with os.scandir(work_root) as entries: for entry in entries: report['enumerated'] += 1 if report['enumerated'] > max(1, int(max_items)): report['retained'] += 1 break if ( entry.is_symlink() or not entry.is_dir(follow_symlinks=False) or not entry.name.startswith(APPROVED_TEMP_PREFIXES) ): continue if cleanup_command_work_dir(entry.path) is None and not os.path.exists(entry.path): report['removed'] += 1 else: report['retained'] += 1 return report def ensure_work_dir_space(): work_dir = get_work_dir() if not work_dir or scan_config.min_free_gb <= 0: return min_free_bytes = scan_config.min_free_gb * 1024 * 1024 * 1024 free_bytes = shutil.disk_usage(work_dir).free if free_bytes >= min_free_bytes: return raise RuntimeError( f"Not enough free space on {work_dir}: {free_bytes / (1024 ** 3):.2f} GB free, " f"minimum is {scan_config.min_free_gb:.2f} GB; admission is closed without cleanup" ) def cleanup_stale_temp_dirs(age_minutes=120, log=True, max_items=None): """Compatibility no-op; stale recovery is isolated in janitor.py.""" if log: logger.info('Source-side stale temp cleanup is retired; the authenticated janitor owns recovery') return 0 def get_results_dir(): """Prepare and return the directory used for persisted scan output.""" require_scanner_runtime_initialized() if not scan_config.results_dir: raise RuntimeError('scan results directory is required') try: return require_private_directory(scan_config.results_dir, create=False) except OSError as exc: raise RuntimeError(f'Unable to use private scan results directory {scan_config.results_dir}: {exc}') from exc def append_jsonl(path, payload): lock = None lock_path = f'{path}.lock' try: require_private_directory(os.path.dirname(os.path.abspath(path)), create=True) if os.path.lexists(path): reject_reparse_components(path) lock = acquire_file_lock(lock_path, timeout_sec=30) repair_jsonl_tail(path) serialized = (json.dumps(payload, ensure_ascii=False, default=str) + '\n').encode('utf-8') with open(path, 'ab') as f: f.write(serialized) f.flush() os.fsync(f.fileno()) harden_private_file(path) return True except OSError as e: if getattr(e, 'errno', None) == 28: logger.error(f"No space left while writing {path}. Result was not persisted.") else: logger.error(f"Unable to write {path}: {str(e)}") return False finally: if lock is not None: release_file_lock(lock, lock_path) def jsonl_manifest_path(path): base, ext = os.path.splitext(path) return f'{base}.manifest.json' def load_jsonl_manifest(path): manifest_path = jsonl_manifest_path(path) try: if os.path.getsize(manifest_path) > 1024 * 1024: raise ValueError(f'JSONL manifest exceeds its bounded size: {manifest_path}') with open(manifest_path, 'r', encoding='utf-8') as f: data = json.load(f) return data if isinstance(data, dict) else {} except FileNotFoundError: return {} def write_jsonl_manifest(path, manifest): manifest_path = jsonl_manifest_path(path) tmp_path = f'{manifest_path}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.tmp' with open(tmp_path, 'w', encoding='utf-8') as f: json.dump(manifest, f, ensure_ascii=False, indent=2, sort_keys=True) f.flush() os.fsync(f.fileno()) harden_private_file(tmp_path) os.replace(tmp_path, manifest_path) harden_private_file(manifest_path) def next_jsonl_segment_path(path, manifest): base, ext = os.path.splitext(path) seq = int(manifest.get('next_sequence') or 1) while True: segment = f'{base}.{seq:06d}{ext or ".jsonl"}' if not os.path.exists(segment): return segment, seq seq += 1 def acquire_file_lock(lock_path, stale_sec=300, timeout_sec=30): require_private_directory(os.path.dirname(os.path.abspath(lock_path)), create=True) reject_reparse_components(os.path.dirname(os.path.abspath(lock_path))) deadline = time.monotonic() + max(0.01, float(timeout_sec)) while True: lock = PrivateFileLock(lock_path) try: return lock.acquire() except BlockingIOError: if time.monotonic() >= deadline: raise TimeoutError(f'timed out acquiring lock {lock_path}') time.sleep(min(0.05, max(0.0, deadline - time.monotonic()))) def release_file_lock(lock, lock_path): try: lock.release() except (AttributeError, OSError): return class JsonlProjectionReconciliationRequired(RuntimeError): pass def jsonl_ledger_path(path): base, _ = os.path.splitext(path) return f'{base}.publication-ledger.sqlite3' def _projection_segment_sequence(path): parent = os.path.dirname(os.path.abspath(path)) base, extension = os.path.splitext(os.path.basename(path)) pattern = re.compile(rf'^{re.escape(base)}\.(\d{{6}}){re.escape(extension)}$') output = [] inspect_limit = max(2, int(getattr(scan_config, 'jsonl_max_segments', 16)) + 1) try: with os.scandir(parent) as entries: for entry in entries: match = pattern.fullmatch(entry.name) if not match: continue if entry.is_symlink() or is_reparse_point(entry.path) or not entry.is_file(follow_symlinks=False): raise JsonlProjectionReconciliationRequired(f'unsafe JSONL segment entry: {entry.path}') output.append((int(match.group(1)), entry.path)) if len(output) > inspect_limit: raise JsonlProjectionReconciliationRequired( f'JSONL physical segment count exceeds its bound for {path}; use offline reconciliation' ) except FileNotFoundError: return [] return sorted(output) def _write_torn_tail_quarantine(path, payload): limit = max(1, int(getattr(scan_config, 'jsonl_torn_quarantine_max_bytes', 64 * 1024))) sample = bytes(payload[:limit]) quarantine = f'{path}.torn-tail.bin' temporary = f'{quarantine}.{os.getpid()}.{threading.get_ident()}.tmp' with open(temporary, 'wb') as handle: handle.write(sample) handle.flush() os.fsync(handle.fileno()) harden_private_file(temporary) durable_replace(temporary, quarantine) harden_private_file(quarantine) def repair_jsonl_tail(path): """Quarantine and remove one bounded unterminated tail before appending.""" if not os.path.exists(path): return 0 reject_reparse_components(path) size = os.path.getsize(path) if size <= 0: return 0 scan_limit = max(1, int(getattr(scan_config, 'jsonl_tail_scan_max_bytes', 8 * 1024 * 1024))) with open(path, 'r+b') as handle: handle.seek(-1, os.SEEK_END) if handle.read(1) == b'\n': return 0 start = max(0, size - scan_limit) handle.seek(start) tail = handle.read(size - start) newline = tail.rfind(b'\n') if newline < 0 and start: raise JsonlProjectionReconciliationRequired( f'JSONL tail exceeds the bounded repair window for {path}; use offline reconciliation' ) truncate_at = start + newline + 1 if newline >= 0 else 0 torn = tail[newline + 1:] if newline >= 0 else tail _write_torn_tail_quarantine(path, torn) handle.truncate(truncate_at) handle.flush() os.fsync(handle.fileno()) logger.error('Quarantined and truncated %s torn byte(s) from %s', size - truncate_at, path) return size - truncate_at def _open_projection_ledger(path): ledger_path = jsonl_ledger_path(path) reject_reparse_components(os.path.dirname(os.path.abspath(ledger_path))) if os.path.lexists(ledger_path): reject_reparse_components(ledger_path) if not private_file_ready(ledger_path): raise JsonlProjectionReconciliationRequired(f'JSONL publication ledger is not private: {ledger_path}') connection = sqlite3.connect(ledger_path, timeout=30) try: connection.execute('PRAGMA busy_timeout=30000') connection.execute('PRAGMA journal_mode=DELETE') connection.execute('PRAGMA synchronous=FULL') connection.executescript(''' CREATE TABLE IF NOT EXISTS publication_identity ( identity_key TEXT NOT NULL, identity_value TEXT NOT NULL, payload_sha256 TEXT NOT NULL, state TEXT NOT NULL, file_name TEXT NOT NULL, byte_offset INTEGER NOT NULL, byte_length INTEGER NOT NULL, created_at REAL NOT NULL, updated_at REAL NOT NULL, PRIMARY KEY(identity_key, identity_value) ); CREATE TABLE IF NOT EXISTS publication_meta ( key TEXT PRIMARY KEY, value TEXT NOT NULL ); CREATE TABLE IF NOT EXISTS publication_identity_variant ( identity_key TEXT NOT NULL, identity_value TEXT NOT NULL, payload_sha256 TEXT NOT NULL, file_name TEXT NOT NULL, byte_offset INTEGER NOT NULL, byte_length INTEGER NOT NULL, created_at REAL NOT NULL, PRIMARY KEY(identity_key, identity_value, payload_sha256) ); CREATE TABLE IF NOT EXISTS reconciliation_issue ( id INTEGER PRIMARY KEY AUTOINCREMENT, identity_key TEXT NOT NULL, file_name TEXT NOT NULL, file_device TEXT NOT NULL, file_inode TEXT NOT NULL, file_size INTEGER NOT NULL, file_mtime_ns TEXT NOT NULL, byte_offset INTEGER NOT NULL, byte_length INTEGER NOT NULL, record_sha256 TEXT NOT NULL, classification TEXT NOT NULL, status TEXT NOT NULL, created_at REAL NOT NULL, resolved_at REAL, UNIQUE(identity_key, file_name, byte_offset, record_sha256) ); CREATE TABLE IF NOT EXISTS reconciliation_variant_issue ( id INTEGER PRIMARY KEY AUTOINCREMENT, identity_key TEXT NOT NULL, identity_sha256 TEXT NOT NULL, file_name TEXT NOT NULL, file_device TEXT NOT NULL, file_inode TEXT NOT NULL, file_size INTEGER NOT NULL, file_mtime_ns TEXT NOT NULL, byte_offset INTEGER NOT NULL, byte_length INTEGER NOT NULL, payload_sha256 TEXT NOT NULL, field_name_set_sha256 TEXT NOT NULL, status TEXT NOT NULL, created_at REAL NOT NULL, resolved_at REAL, UNIQUE(identity_key, identity_sha256, file_name, byte_offset, payload_sha256) ); CREATE INDEX IF NOT EXISTS idx_publication_identity_state_created ON publication_identity(state, created_at); CREATE INDEX IF NOT EXISTS idx_publication_identity_file_state ON publication_identity(file_name, state); CREATE INDEX IF NOT EXISTS idx_publication_identity_variant_identity ON publication_identity_variant(identity_key, identity_value); CREATE INDEX IF NOT EXISTS idx_reconciliation_issue_status ON reconciliation_issue(status, id); CREATE INDEX IF NOT EXISTS idx_reconciliation_variant_issue_status ON reconciliation_variant_issue(status, id); ''') connection.execute( '''INSERT OR IGNORE INTO publication_identity_variant ( identity_key, identity_value, payload_sha256, file_name, byte_offset, byte_length, created_at ) SELECT identity_key, identity_value, payload_sha256, file_name, byte_offset, byte_length, created_at FROM publication_identity WHERE state = 'appended' ''' ) row = connection.execute("SELECT value FROM publication_meta WHERE key = 'row_count'").fetchone() if row is None: count = int(connection.execute('SELECT COUNT(*) FROM publication_identity').fetchone()[0]) connection.execute( "INSERT INTO publication_meta(key, value) VALUES ('row_count', ?)", (str(count),), ) connection.commit() harden_private_file(ledger_path) return connection except BaseException: connection.close() raise def _ledger_row_count(connection): row = connection.execute("SELECT value FROM publication_meta WHERE key = 'row_count'").fetchone() return max(0, int(row[0] if row else 0)) def _set_ledger_row_count(connection, count): connection.execute( "INSERT OR REPLACE INTO publication_meta(key, value) VALUES ('row_count', ?)", (str(max(0, int(count))),), ) def _bounded_projection_bytes(path, offset, length): max_record = max( 1024 * 1024, int(getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)) + 1, ) if offset < 0 or length <= 0 or length > max_record: return b'' try: with open(path, 'rb') as handle: handle.seek(offset) return handle.read(length) except OSError: return b'' def _recover_prepared_publications(connection, path): rows = connection.execute( '''SELECT identity_key, identity_value, payload_sha256, file_name, byte_offset, byte_length FROM publication_identity WHERE state = 'prepared' ORDER BY created_at LIMIT 2''' ).fetchall() if len(rows) > 1: raise JsonlProjectionReconciliationRequired('publication ledger contains multiple unresolved append states') for identity_key, identity_value, digest, file_name, offset, length in rows: candidate = os.path.join(os.path.dirname(os.path.abspath(path)), os.path.basename(file_name)) payload = _bounded_projection_bytes(candidate, int(offset), int(length)) if payload.endswith(b'\n') and hashlib.sha256(payload).hexdigest() == digest: connection.execute( '''UPDATE publication_identity SET state = 'appended', updated_at = ? WHERE identity_key = ? AND identity_value = ? AND state = 'prepared' ''', (time.time(), identity_key, identity_value), ) connection.execute( '''INSERT OR IGNORE INTO publication_identity_variant ( identity_key, identity_value, payload_sha256, file_name, byte_offset, byte_length, created_at ) VALUES (?, ?, ?, ?, ?, ?, ?)''', (identity_key, identity_value, digest, file_name, offset, length, time.time()), ) else: connection.execute( 'DELETE FROM publication_identity WHERE identity_key = ? AND identity_value = ? AND state = ?', (identity_key, identity_value, 'prepared'), ) _set_ledger_row_count(connection, _ledger_row_count(connection) - 1) connection.commit() def _ensure_projection_ledger_bootstrapped(connection, path, identity_key): marker = f'bootstrapped:{identity_key}' if connection.execute('SELECT 1 FROM publication_meta WHERE key = ?', (marker,)).fetchone(): return candidates = projection_segment_paths(path) total_bytes = sum(os.path.getsize(candidate) for candidate in candidates) max_bytes = max(0, int(getattr(scan_config, 'jsonl_legacy_index_max_bytes', 16 * 1024 * 1024))) if total_bytes > max_bytes: raise JsonlProjectionReconciliationRequired( f'existing JSONL history for {path} is unindexed ({total_bytes} bytes); ' 'run the offline JSONL reconciliation procedure before publication' ) row_count = _ledger_row_count(connection) row_limit = max(1, int(getattr(scan_config, 'jsonl_ledger_max_rows', 1000000))) for candidate in candidates: offset = 0 with open(candidate, 'rb') as handle: for raw_line in handle: if not raw_line.endswith(b'\n'): raise JsonlProjectionReconciliationRequired(f'unterminated closed JSONL record in {candidate}') identity = _projection_identity_from_line(raw_line, identity_key, candidate) if identity: digest = hashlib.sha256(raw_line).hexdigest() existing = connection.execute( '''SELECT payload_sha256 FROM publication_identity WHERE identity_key = ? AND identity_value = ?''', (identity_key, identity), ).fetchone() if existing and existing[0] != digest: raise JsonlProjectionReconciliationRequired( 'conflicting identity in existing JSONL history: ' f'identity_key={identity_key} ' f'identity_sha256={hashlib.sha256(identity.encode("utf-8")).hexdigest()}' ) if not existing: if row_count >= row_limit: raise JsonlProjectionReconciliationRequired('existing JSONL identities exceed the ledger row bound') now = time.time() connection.execute( '''INSERT INTO publication_identity ( identity_key, identity_value, payload_sha256, state, file_name, byte_offset, byte_length, created_at, updated_at ) VALUES (?, ?, ?, 'appended', ?, ?, ?, ?, ?)''', ( identity_key, identity, digest, os.path.basename(candidate), offset, len(raw_line), now, now, ), ) connection.execute( '''INSERT INTO publication_identity_variant ( identity_key, identity_value, payload_sha256, file_name, byte_offset, byte_length, created_at ) VALUES (?, ?, ?, ?, ?, ?, ?)''', ( identity_key, identity, digest, os.path.basename(candidate), offset, len(raw_line), now, ), ) row_count += 1 offset += len(raw_line) _set_ledger_row_count(connection, row_count) connection.execute('INSERT INTO publication_meta(key, value) VALUES (?, ?)', (marker, '1')) connection.commit() def _projection_identity_from_line(raw_line, identity_key, candidate): if not raw_line.endswith(b'\n'): raise JsonlProjectionReconciliationRequired(f'unterminated projection record in {candidate}') if identity_key == 'error_row_id': try: identity, separator, _ = raw_line.partition(b'\t') if not separator or not identity: raise ValueError('missing error projection identity separator') return identity.decode('utf-8') except UnicodeDecodeError as exc: raise JsonlProjectionReconciliationRequired( f'invalid existing projection record in {candidate}' ) from exc except ValueError as exc: raise JsonlProjectionReconciliationRequired( f'invalid existing projection record in {candidate}' ) from exc try: payload = json.loads(raw_line.decode('utf-8')) except (UnicodeDecodeError, ValueError) as exc: raise JsonlProjectionReconciliationRequired( f'invalid existing JSONL record in {candidate}' ) from exc return str(payload.get(identity_key) or '') if isinstance(payload, dict) else '' def _bounded_projection_json_values(raw_line): body = raw_line[:-1] if raw_line.endswith(b'\n') else raw_line if not raw_line.endswith(b'\n'): return None, 'unterminated_record' try: text = body.decode('utf-8') except UnicodeDecodeError: return None, 'invalid_utf8' if text.startswith('\ufeff'): return None, 'utf8_bom_prefix' try: value = json.loads(text) return [{ 'value': value, 'relative_offset': 0, 'byte_length': len(raw_line), 'payload_sha256': hashlib.sha256(raw_line).hexdigest(), }], 'single_json' except ValueError: pass decoder = json.JSONDecoder() position = 0 parsed = [] while True: while position < len(text) and text[position].isspace(): position += 1 if position >= len(text): break start = position try: value, position = decoder.raw_decode(text, position) except json.JSONDecodeError: parsed = [] break parsed.append((start, position, value)) if len(parsed) > 1 and position >= len(text) and all(isinstance(item[2], dict) for item in parsed): values = [] for start, end, value in parsed: prefix_bytes = len(text[:start].encode('utf-8')) serialized = text[start:end].encode('utf-8') + b'\n' values.append({ 'value': value, 'relative_offset': prefix_bytes, 'byte_length': len(serialized), 'payload_sha256': hashlib.sha256(serialized).hexdigest(), }) return values, 'concatenated_json_objects' if re.match(r'^[0-9]+,\s*', text): return None, 'legacy_numeric_prefix_corrupt_json' return None, 'invalid_json' def _projection_field_name_set_sha256(value): paths = [] def visit(item, prefix=''): if isinstance(item, dict): for key in sorted(map(str, item.keys())): path = prefix + key paths.append(path) visit(item.get(key), path + '.') elif isinstance(item, list): paths.append(prefix + '[]') for child in item[:32]: visit(child, prefix + '[].') visit(value) return hashlib.sha256('\x00'.join(sorted(set(paths))).encode('utf-8')).hexdigest() def _error_projection_field_name_set_sha256(raw_line): try: text = raw_line[:-1].decode('utf-8') if raw_line.endswith(b'\n') else raw_line.decode('utf-8') tail = text.rsplit('\t', 1)[-1] value = json.loads(tail) except (UnicodeDecodeError, ValueError): value = {} return _projection_field_name_set_sha256(value) def _save_projection_reconciliation_progress(ledger, prefix, file_index, offset, indexed_rows): _set_ledger_row_count(ledger, indexed_rows) for key, value in ( (prefix + 'file_index', str(file_index)), (prefix + 'offset', str(offset)), ): ledger.execute('INSERT OR REPLACE INTO publication_meta(key, value) VALUES (?, ?)', (key, value)) def _record_projection_reconciliation_issue( ledger, identity_key, plan_entry, byte_offset, raw_line, classification, resolved=False, byte_length=None, record_sha256=None, ): if raw_line is not None: byte_length = len(raw_line) record_sha256 = hashlib.sha256(raw_line).hexdigest() byte_length = int(byte_length or 0) digest = str(record_sha256 or '').strip().lower() if byte_length <= 0 or not re.fullmatch(r'[a-f0-9]{64}', digest): raise ValueError('projection reconciliation issue metadata is invalid') now = time.time() ledger.execute( '''INSERT OR IGNORE INTO reconciliation_issue ( identity_key, file_name, file_device, file_inode, file_size, file_mtime_ns, byte_offset, byte_length, record_sha256, classification, status, created_at, resolved_at ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', ( identity_key, os.path.basename(plan_entry['path']), str(plan_entry['device']), str(plan_entry['inode']), int(plan_entry['size']), str(plan_entry['mtime_ns']), int(byte_offset), byte_length, digest, classification, 'resolved' if resolved else 'pending', now, now if resolved else None, ), ) if resolved: ledger.execute( '''UPDATE reconciliation_issue SET status = 'resolved', resolved_at = COALESCE(resolved_at, ?) WHERE identity_key = ? AND file_name = ? AND byte_offset = ? AND record_sha256 = ?''', (now, identity_key, os.path.basename(plan_entry['path']), int(byte_offset), digest), ) return ledger.execute( '''SELECT id, status FROM reconciliation_issue WHERE identity_key = ? AND file_name = ? AND byte_offset = ? AND record_sha256 = ?''', (identity_key, os.path.basename(plan_entry['path']), int(byte_offset), digest), ).fetchone() def _record_projection_variant_issue( ledger, identity_key, identity, plan_entry, byte_offset, byte_length, payload_sha256, field_name_set_sha256, resolved=False, ): identity_sha256 = hashlib.sha256(identity.encode('utf-8')).hexdigest() now = time.time() ledger.execute( '''INSERT OR IGNORE INTO reconciliation_variant_issue ( identity_key, identity_sha256, file_name, file_device, file_inode, file_size, file_mtime_ns, byte_offset, byte_length, payload_sha256, field_name_set_sha256, status, created_at, resolved_at ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', ( identity_key, identity_sha256, os.path.basename(plan_entry['path']), str(plan_entry['device']), str(plan_entry['inode']), int(plan_entry['size']), str(plan_entry['mtime_ns']), int(byte_offset), int(byte_length), payload_sha256, field_name_set_sha256, 'resolved' if resolved else 'pending', now, now if resolved else None, ), ) if resolved: ledger.execute( '''UPDATE reconciliation_variant_issue SET status = 'resolved', resolved_at = COALESCE(resolved_at, ?) WHERE identity_key = ? AND identity_sha256 = ? AND file_name = ? AND byte_offset = ? AND payload_sha256 = ?''', ( now, identity_key, identity_sha256, os.path.basename(plan_entry['path']), int(byte_offset), payload_sha256, ), ) return ledger.execute( '''SELECT id, status FROM reconciliation_variant_issue WHERE identity_key = ? AND identity_sha256 = ? AND file_name = ? AND byte_offset = ? AND payload_sha256 = ?''', ( identity_key, identity_sha256, os.path.basename(plan_entry['path']), int(byte_offset), payload_sha256, ), ).fetchone() def _stream_projection_record(handle, first_chunk, chunk_bytes=1024 * 1024): digest = hashlib.sha256() digest.update(first_chunk) total = len(first_chunk) newline_terminated = first_chunk.endswith(b'\n') while not newline_terminated: chunk = handle.readline(max(1, int(chunk_bytes))) if not chunk: break digest.update(chunk) total += len(chunk) newline_terminated = chunk.endswith(b'\n') return total, digest.hexdigest(), newline_terminated def _projection_reconciliation_plan(path): candidates = projection_segment_paths(path) if not candidates: _publish_empty_jsonl_generation(path) candidates = [os.path.abspath(path)] plan = [] for candidate in candidates: require_private_file(candidate) details = os.stat(candidate, follow_symlinks=False) plan.append({ 'path': os.path.abspath(candidate), 'device': int(getattr(details, 'st_dev', 0) or 0), 'inode': int(getattr(details, 'st_ino', 0) or 0), 'size': int(details.st_size), 'mtime_ns': int(getattr(details, 'st_mtime_ns', int(details.st_mtime * 1_000_000_000))), }) return plan def reconcile_projection_ledger_batch( path, identity_key, max_rows=10000, max_bytes=64 * 1024 * 1024, max_seconds=30.0, row_limit=None, ledger_byte_limit=None, max_record_bytes=None, ): """Build one bounded, resumable ledger batch without modifying JSONL history.""" if identity_key not in ('scan_event_id', 'finding_uid', 'error_row_id'): raise ValueError('unsupported projection reconciliation identity') path = os.path.abspath(path) require_private_directory(os.path.dirname(path), create=False) max_rows = max(1, int(max_rows)) max_bytes = max(1, int(max_bytes)) max_seconds = max(0.01, float(max_seconds)) row_limit = max(1, int(row_limit or getattr(scan_config, 'jsonl_ledger_max_rows', 1000000))) ledger_byte_limit = max( 1024 * 1024, int(ledger_byte_limit or getattr(scan_config, 'jsonl_ledger_max_bytes', 512 * 1024 * 1024)), ) max_record_bytes = max( 1024 * 1024, int(max_record_bytes or getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)), ) lock_path = f'{path}.lock' lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec) ledger = None try: prefix = f'offline-reconcile:{identity_key}:' marker = f'bootstrapped:{identity_key}' ledger = _open_projection_ledger(path) if ledger.execute('SELECT 1 FROM publication_meta WHERE key = ?', (marker,)).fetchone(): return { 'path': path, 'identity_key': identity_key, 'complete': True, 'batch_rows': 0, 'batch_bytes': 0, 'indexed_rows': _ledger_row_count(ledger), } plan = _projection_reconciliation_plan(path) plan_json = json.dumps(plan, ensure_ascii=True, sort_keys=True, separators=(',', ':')) existing_plan = ledger.execute( 'SELECT value FROM publication_meta WHERE key = ?', (prefix + 'plan',), ).fetchone() if existing_plan and existing_plan[0] != plan_json: raise JsonlProjectionReconciliationRequired( f'projection history changed during offline reconciliation: {path}' ) if not existing_plan: ledger.execute( 'INSERT INTO publication_meta(key, value) VALUES (?, ?)', (prefix + 'plan', plan_json), ) file_index_row = ledger.execute( 'SELECT value FROM publication_meta WHERE key = ?', (prefix + 'file_index',), ).fetchone() offset_row = ledger.execute( 'SELECT value FROM publication_meta WHERE key = ?', (prefix + 'offset',), ).fetchone() file_index = max(0, int(file_index_row[0] if file_index_row else 0)) offset = max(0, int(offset_row[0] if offset_row else 0)) indexed_rows = _ledger_row_count(ledger) batch_rows = 0 batch_bytes = 0 started = time.monotonic() while file_index < len(plan): candidate = plan[file_index]['path'] with open(candidate, 'rb') as handle: handle.seek(offset) while True: raw_line = handle.readline(max_record_bytes + 1) if not raw_line: file_index += 1 offset = 0 break if len(raw_line) > max_record_bytes: byte_length, record_sha256, newline_terminated = _stream_projection_record( handle, raw_line, ) classification = 'oversized_record' if newline_terminated else 'unterminated_record' issue = _record_projection_reconciliation_issue( ledger, identity_key, plan[file_index], offset, None, classification, byte_length=byte_length, record_sha256=record_sha256, ) _save_projection_reconciliation_progress( ledger, prefix, file_index, offset, indexed_rows, ) ledger.commit() if issue[1] != 'resolved': raise JsonlProjectionReconciliationRequired( 'projection reconciliation issue requires explicit review: ' f'id={issue[0]} file={os.path.basename(candidate)} offset={offset} ' f'length={byte_length} sha256={record_sha256} ' f'classification={classification}' ) offset += byte_length batch_rows += 1 batch_bytes += byte_length if ( batch_rows >= max_rows or batch_bytes >= max_bytes or time.monotonic() - started >= max_seconds ): break continue if identity_key == 'error_row_id': try: identity = _projection_identity_from_line(raw_line, identity_key, candidate) if len(identity) > 256 or any( character in identity for character in ('\x00', '\r', '\n') ): values = None classification = 'invalid_error_projection' else: values = [{ 'identity': identity, 'relative_offset': 0, 'byte_length': len(raw_line), 'payload_sha256': hashlib.sha256(raw_line).hexdigest(), 'field_name_set_sha256': _error_projection_field_name_set_sha256(raw_line), }] classification = 'error_projection' except JsonlProjectionReconciliationRequired: values = None classification = ( 'unterminated_record' if not raw_line.endswith(b'\n') else 'invalid_error_projection' ) else: parsed_values, classification = _bounded_projection_json_values(raw_line) values = None if parsed_values is None else [{ 'identity': ( str(item['value'].get(identity_key) or '') if isinstance(item['value'], dict) else '' ), 'relative_offset': item['relative_offset'], 'byte_length': item['byte_length'], 'payload_sha256': item['payload_sha256'], 'field_name_set_sha256': _projection_field_name_set_sha256(item['value']), } for item in parsed_values] if values is None: issue = _record_projection_reconciliation_issue( ledger, identity_key, plan[file_index], offset, raw_line, classification, ) _save_projection_reconciliation_progress( ledger, prefix, file_index, offset, indexed_rows, ) ledger.commit() if issue[1] != 'resolved': raise JsonlProjectionReconciliationRequired( 'projection reconciliation issue requires explicit review: ' f'id={issue[0]} file={os.path.basename(candidate)} offset={offset} ' f'length={len(raw_line)} sha256={hashlib.sha256(raw_line).hexdigest()} ' f'classification={classification}' ) offset += len(raw_line) batch_rows += 1 batch_bytes += len(raw_line) if ( batch_rows >= max_rows or batch_bytes >= max_bytes or time.monotonic() - started >= max_seconds ): break continue for value in values: identity = value['identity'] if not identity: continue if len(identity) > 256 or any(character in identity for character in ('\x00', '\r', '\n')): raise JsonlProjectionReconciliationRequired( f'invalid {identity_key} in existing projection history: {candidate}:{offset}' ) digest = value['payload_sha256'] existing = ledger.execute( '''SELECT payload_sha256, state FROM publication_identity WHERE identity_key = ? AND identity_value = ?''', (identity_key, identity), ).fetchone() variant = ledger.execute( '''SELECT 1 FROM publication_identity_variant WHERE identity_key = ? AND identity_value = ? AND payload_sha256 = ?''', (identity_key, identity, digest), ).fetchone() if variant: continue if existing: if existing[1] != 'appended': raise JsonlProjectionReconciliationRequired( 'publication ledger retained an unresolved historical identity state' ) variant_offset = offset + int(value['relative_offset']) issue = _record_projection_variant_issue( ledger, identity_key, identity, plan[file_index], variant_offset, int(value['byte_length']), digest, value['field_name_set_sha256'], ) _save_projection_reconciliation_progress( ledger, prefix, file_index, offset, indexed_rows, ) ledger.commit() raise JsonlProjectionReconciliationRequired( 'historical projection payload variant requires explicit review: ' f'id={issue[0]} file={os.path.basename(candidate)} ' f'offset={variant_offset} length={int(value["byte_length"])} ' f'identity_sha256={hashlib.sha256(identity.encode("utf-8")).hexdigest()} ' f'payload_sha256={digest}' ) if not existing: if indexed_rows >= row_limit: raise JsonlProjectionReconciliationRequired( f'existing JSONL identities exceed the ledger row bound: {row_limit}' ) now = time.time() ledger.execute( '''INSERT INTO publication_identity ( identity_key, identity_value, payload_sha256, state, file_name, byte_offset, byte_length, created_at, updated_at ) VALUES (?, ?, ?, 'appended', ?, ?, ?, ?, ?)''', ( identity_key, identity, digest, os.path.basename(candidate), offset + int(value['relative_offset']), int(value['byte_length']), now, now, ), ) ledger.execute( '''INSERT INTO publication_identity_variant ( identity_key, identity_value, payload_sha256, file_name, byte_offset, byte_length, created_at ) VALUES (?, ?, ?, ?, ?, ?, ?)''', ( identity_key, identity, digest, os.path.basename(candidate), offset + int(value['relative_offset']), int(value['byte_length']), now, ), ) indexed_rows += 1 offset += len(raw_line) batch_rows += 1 batch_bytes += len(raw_line) if ( batch_rows >= max_rows or batch_bytes >= max_bytes or time.monotonic() - started >= max_seconds ): break if batch_rows and ( batch_rows >= max_rows or batch_bytes >= max_bytes or time.monotonic() - started >= max_seconds ): break _save_projection_reconciliation_progress(ledger, prefix, file_index, offset, indexed_rows) complete = file_index >= len(plan) if complete: ledger.execute('INSERT OR REPLACE INTO publication_meta(key, value) VALUES (?, ?)', (marker, '1')) if os.path.getsize(jsonl_ledger_path(path)) > ledger_byte_limit: raise JsonlProjectionReconciliationRequired( f'JSONL identity ledger exceeds its {ledger_byte_limit} byte bound' ) ledger.commit() return { 'path': path, 'identity_key': identity_key, 'complete': complete, 'batch_rows': batch_rows, 'batch_bytes': batch_bytes, 'indexed_rows': indexed_rows, 'file_index': file_index, 'file_count': len(plan), 'byte_offset': offset, } except BaseException: if ledger is not None: ledger.rollback() raise finally: if ledger is not None: ledger.close() if os.path.exists(jsonl_ledger_path(path)): harden_private_file(jsonl_ledger_path(path)) release_file_lock(lock, lock_path) def _review_projection_issue_record( plan_entry, identity_key, byte_offset, expected_sha256, max_record_bytes, expected_length=None, expected_classification=None, ): if byte_offset >= int(plan_entry['size']): raise JsonlProjectionReconciliationRequired('projection issue offset is outside the immutable file') with open(plan_entry['path'], 'rb') as handle: if byte_offset: handle.seek(byte_offset - 1) if handle.read(1) != b'\n': raise JsonlProjectionReconciliationRequired('projection issue offset is not a record boundary') handle.seek(byte_offset) raw_line = handle.readline(max_record_bytes + 1) if not raw_line: raise JsonlProjectionReconciliationRequired('projection issue record is absent') if len(raw_line) > max_record_bytes: byte_length, actual_sha256, newline_terminated = _stream_projection_record(handle, raw_line) classification = 'oversized_record' if newline_terminated else 'unterminated_record' raw_line = None else: byte_length = len(raw_line) actual_sha256 = hashlib.sha256(raw_line).hexdigest() if identity_key == 'error_row_id': try: identity = _projection_identity_from_line(raw_line, identity_key, plan_entry['path']) if len(identity) > 256 or any(character in identity for character in ('\x00', '\r', '\n')): raise ValueError('error projection identity exceeds its bounded format') except (JsonlProjectionReconciliationRequired, ValueError): classification = ( 'unterminated_record' if not raw_line.endswith(b'\n') else 'invalid_error_projection' ) else: raise JsonlProjectionReconciliationRequired( 'reviewed projection issue is a parseable error record and must be indexed' ) else: parsed_values, classification = _bounded_projection_json_values(raw_line) if parsed_values is not None: raise JsonlProjectionReconciliationRequired( 'reviewed projection issue contains bounded parseable JSON and must be indexed' ) if actual_sha256 != expected_sha256: raise JsonlProjectionReconciliationRequired('projection issue record SHA-256 does not match review') if expected_length is not None and byte_length != int(expected_length): raise JsonlProjectionReconciliationRequired('projection issue record length does not match review') if expected_classification is not None and classification != str(expected_classification): raise JsonlProjectionReconciliationRequired('projection issue classification does not match review') return raw_line, byte_length, actual_sha256, classification def approve_projection_reconciliation_issues( path, identity_key, reviewed_issues, max_record_bytes=None, return_details=False, ): """Approve exact reviewed corrupt records without storing or changing their payloads.""" if identity_key not in ('scan_event_id', 'finding_uid', 'error_row_id'): raise ValueError('unsupported projection reconciliation identity') reviewed_issues = list(reviewed_issues or []) if not reviewed_issues or len(reviewed_issues) > 100000: raise ValueError('projection issue review count is outside its bound') path = os.path.abspath(path) max_record_bytes = max( 1024 * 1024, int(max_record_bytes or getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)), ) lock_path = f'{path}.lock' lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec) ledger = None try: plan = _projection_reconciliation_plan(path) plan_by_name = {os.path.basename(item['path']): (index, item) for index, item in enumerate(plan)} if len(plan_by_name) != len(plan): raise JsonlProjectionReconciliationRequired('projection history contains ambiguous basenames') normalized = [] seen = set() for value in reviewed_issues: requested_file = str(value.get('file') or value.get('basename') or '') physical_file = os.path.basename(requested_file) byte_offset = int(value.get('offset', value.get('byte_offset', -1))) expected_sha256 = str(value.get('sha256') or '').strip().lower() if ( not physical_file or physical_file != requested_file or physical_file not in plan_by_name or byte_offset < 0 or not re.fullmatch(r'[a-f0-9]{64}', expected_sha256) ): raise ValueError('projection issue review metadata is invalid') identity = (physical_file, byte_offset, expected_sha256) if identity in seen: raise ValueError('projection issue review contains duplicate metadata') seen.add(identity) normalized.append(( plan_by_name[physical_file][0], physical_file, byte_offset, expected_sha256, value.get('length', value.get('byte_length')), value.get('classification'), )) ledger = _open_projection_ledger(path) details = [] classifications = Counter() for sequence, (_, physical_file, byte_offset, expected_sha256, expected_length, expected_classification) in enumerate(sorted(normalized), 1): plan_entry = plan_by_name[physical_file][1] raw_line, byte_length, actual_sha256, classification = _review_projection_issue_record( plan_entry, identity_key, byte_offset, expected_sha256, max_record_bytes, expected_length=expected_length, expected_classification=expected_classification, ) issue = _record_projection_reconciliation_issue( ledger, identity_key, plan_entry, byte_offset, raw_line, classification, resolved=True, byte_length=byte_length, record_sha256=actual_sha256, ) classifications[classification] += 1 if return_details: details.append({ 'id': int(issue[0]), 'status': issue[1], 'file': physical_file, 'offset': byte_offset, 'length': byte_length, 'sha256': actual_sha256, 'classification': classification, }) if sequence % 100 == 0: ledger.commit() ledger.commit() report = { 'resolved_count': len(normalized), 'classifications': dict(sorted(classifications.items())), } if return_details: report['details'] = details return report finally: if ledger is not None: ledger.close() harden_private_file(jsonl_ledger_path(path)) release_file_lock(lock, lock_path) def approve_projection_reconciliation_issue( path, identity_key, physical_file, byte_offset, expected_sha256, max_record_bytes=None, ): report = approve_projection_reconciliation_issues( path, identity_key, [{ 'file': physical_file, 'offset': byte_offset, 'sha256': expected_sha256, }], max_record_bytes=max_record_bytes, return_details=True, ) return report['details'][0] def _apply_reviewed_projection_conflict_variant( ledger, handle, plan_entry, identity_key, reviewed, max_record_bytes, ): file_name, offset, length, identity_sha256, payload_sha256, field_set_sha256 = reviewed if offset >= int(plan_entry['size']): raise JsonlProjectionReconciliationRequired('projection conflict offset is outside the immutable file') if offset: handle.seek(offset - 1) if handle.read(1) != b'\n': raise JsonlProjectionReconciliationRequired( 'projection conflict offset is not a record boundary' ) handle.seek(offset) raw_line = handle.readline(max_record_bytes + 1) if not raw_line or len(raw_line) > max_record_bytes or len(raw_line) != length: raise JsonlProjectionReconciliationRequired( 'projection conflict record is absent or does not match its reviewed length' ) actual_payload_sha256 = hashlib.sha256(raw_line).hexdigest() if identity_key == 'error_row_id': identity = _projection_identity_from_line(raw_line, identity_key, plan_entry['path']) actual_field_set_sha256 = _error_projection_field_name_set_sha256(raw_line) else: parsed_values, _ = _bounded_projection_json_values(raw_line) if not parsed_values or len(parsed_values) != 1 or not isinstance(parsed_values[0]['value'], dict): raise JsonlProjectionReconciliationRequired( 'projection conflict record is not one bounded JSON identity record' ) identity = str(parsed_values[0]['value'].get(identity_key) or '') actual_field_set_sha256 = _projection_field_name_set_sha256(parsed_values[0]['value']) if ( not identity or hashlib.sha256(identity.encode('utf-8')).hexdigest() != identity_sha256 or actual_payload_sha256 != payload_sha256 or actual_field_set_sha256 != field_set_sha256 ): raise JsonlProjectionReconciliationRequired( 'projection conflict record does not match reviewed identity/payload/schema hashes' ) primary = ledger.execute( '''SELECT state FROM publication_identity WHERE identity_key = ? AND identity_value = ?''', (identity_key, identity), ).fetchone() if primary and primary[0] != 'appended': raise JsonlProjectionReconciliationRequired( 'projection conflict review primary identity is unresolved' ) if not primary: row_count = _ledger_row_count(ledger) row_limit = max(1, int(getattr(scan_config, 'jsonl_ledger_max_rows', 1000000))) if row_count >= row_limit: raise JsonlProjectionReconciliationRequired( f'JSONL identity ledger reached its {row_limit} row bound' ) now = time.time() ledger.execute( '''INSERT INTO publication_identity ( identity_key, identity_value, payload_sha256, state, file_name, byte_offset, byte_length, created_at, updated_at ) VALUES (?, ?, ?, 'appended', ?, ?, ?, ?, ?)''', ( identity_key, identity, payload_sha256, file_name, offset, length, now, now, ), ) _set_ledger_row_count(ledger, row_count + 1) ledger.execute( '''INSERT OR IGNORE INTO publication_identity_variant ( identity_key, identity_value, payload_sha256, file_name, byte_offset, byte_length, created_at ) VALUES (?, ?, ?, ?, ?, ?, ?)''', (identity_key, identity, payload_sha256, file_name, offset, length, time.time()), ) _record_projection_variant_issue( ledger, identity_key, identity, plan_entry, offset, length, payload_sha256, field_set_sha256, resolved=True, ) def approve_projection_conflict_variants( path, identity_key, reviewed_variants, max_record_bytes=None, commit_batch_size=250, progress_callback=None, ): if identity_key not in ('scan_event_id', 'finding_uid', 'error_row_id'): raise ValueError('unsupported projection conflict identity') reviewed_variants = list(reviewed_variants or []) if not reviewed_variants or len(reviewed_variants) > 100000: raise ValueError('projection conflict review count is outside its bound') path = os.path.abspath(path) max_record_bytes = max( 1024 * 1024, int(max_record_bytes or getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)), ) commit_batch_size = min(1000, max(1, int(commit_batch_size or 250))) lock_path = f'{path}.lock' lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec) ledger = None try: plan = _projection_reconciliation_plan(path) plan_by_name = {os.path.basename(item['path']): (index, item) for index, item in enumerate(plan)} if len(plan_by_name) != len(plan): raise JsonlProjectionReconciliationRequired('projection history contains ambiguous basenames') normalized = [] seen = set() for value in reviewed_variants: file_name = str(value.get('file') or '') offset = int(value.get('offset', -1)) length = int(value.get('length', 0)) identity_sha256 = str(value.get('identity_sha256') or '').lower() payload_sha256 = str(value.get('payload_sha256') or '').lower() field_set_sha256 = str(value.get('field_name_set_sha256') or '').lower() if ( not file_name or os.path.basename(file_name) != file_name or file_name not in plan_by_name or offset < 0 or length <= 0 or not re.fullmatch(r'[a-f0-9]{64}', identity_sha256) or not re.fullmatch(r'[a-f0-9]{64}', payload_sha256) or not re.fullmatch(r'[a-f0-9]{64}', field_set_sha256) or value.get('classification') != 'historical_payload_variant' ): raise ValueError('projection conflict review metadata is invalid') key = (file_name, offset, identity_sha256, payload_sha256) if key in seen: raise ValueError('projection conflict review contains duplicate metadata') seen.add(key) normalized.append(( plan_by_name[file_name][0], file_name, (file_name, offset, length, identity_sha256, payload_sha256, field_set_sha256), )) ledger = _open_projection_ledger(path) resolved_rows = ledger.execute( '''SELECT identity_sha256, file_name, byte_offset, byte_length, payload_sha256, field_name_set_sha256 FROM reconciliation_variant_issue WHERE identity_key = ? AND status = 'resolved' ''', (identity_key,), ).fetchall() resolved = { (row[1], int(row[2]), row[0], row[4]): (int(row[3]), row[5]) for row in resolved_rows } pending = [] already_resolved = 0 for plan_index, file_name, reviewed in sorted(normalized): key = (file_name, reviewed[1], reviewed[3], reviewed[4]) expected = resolved.get(key) if expected is not None: if expected != (reviewed[2], reviewed[5]): raise JsonlProjectionReconciliationRequired( 'resolved projection conflict metadata does not match this review manifest' ) already_resolved += 1 continue pending.append((plan_index, file_name, reviewed)) if progress_callback is not None: progress_callback({ 'reviewed': len(normalized), 'already_resolved': already_resolved, 'newly_resolved': 0, }) newly_resolved = 0 grouped = {} for plan_index, file_name, reviewed in pending: grouped.setdefault((plan_index, file_name), []).append(reviewed) for plan_index, file_name in sorted(grouped): plan_entry = plan_by_name[file_name][1] with open(plan_entry['path'], 'rb') as handle: for reviewed in sorted(grouped[(plan_index, file_name)], key=lambda item: item[1]): _apply_reviewed_projection_conflict_variant( ledger, handle, plan_entry, identity_key, reviewed, max_record_bytes, ) newly_resolved += 1 if newly_resolved % commit_batch_size == 0: ledger.commit() if progress_callback is not None: progress_callback({ 'reviewed': len(normalized), 'already_resolved': already_resolved, 'newly_resolved': newly_resolved, }) ledger.commit() if progress_callback is not None and newly_resolved % commit_batch_size: progress_callback({ 'reviewed': len(normalized), 'already_resolved': already_resolved, 'newly_resolved': newly_resolved, }) return { 'resolved_variant_count': len(normalized), 'already_resolved': already_resolved, 'newly_resolved': newly_resolved, } finally: if ledger is not None: ledger.close() harden_private_file(jsonl_ledger_path(path)) release_file_lock(lock, lock_path) def _files_share_prefix(segment_path, current_path, size): if size <= 0 or not os.path.isfile(current_path) or os.path.getsize(current_path) < size: return False remaining = size with open(segment_path, 'rb') as segment, open(current_path, 'rb') as current: while remaining: amount = min(1024 * 1024, remaining) left = segment.read(amount) right = current.read(amount) if left != right or not left: return False remaining -= len(left) return remaining == 0 def _publish_empty_jsonl_generation(path): temporary = ( f'{path}.{os.getpid()}.{threading.get_ident()}.' f'{uuid.uuid4().hex}.empty.tmp' ) try: with open(temporary, 'xb') as handle: handle.flush() os.fsync(handle.fileno()) harden_private_file(temporary) durable_replace(temporary, path) harden_private_file(path) finally: if os.path.exists(temporary): os.remove(temporary) def _remove_file_prefix(path, size): current_size = os.path.getsize(path) if size <= 0 or current_size < size: return if current_size == size: _publish_empty_jsonl_generation(path) return temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.prefix.tmp' with open(path, 'rb') as source, open(temporary, 'wb') as destination: source.seek(size) shutil.copyfileobj(source, destination, 1024 * 1024) destination.flush() os.fsync(destination.fileno()) harden_private_file(temporary) durable_replace(temporary, path) harden_private_file(path) def _segment_manifest_entry(path, sequence): return { 'name': os.path.basename(path), 'path': os.path.abspath(path), 'bytes': int(os.path.getsize(path)), 'closed_at': datetime.now(timezone.utc).isoformat(timespec='seconds'), 'sequence': int(sequence), } def _jsonl_file_generation(path): details = os.stat(path, follow_symlinks=False) return { 'device': int(getattr(details, 'st_dev', 0) or 0), 'size': int(details.st_size), 'mtime_ns': int(getattr(details, 'st_mtime_ns', int(details.st_mtime * 1_000_000_000))), 'inode': int(getattr(details, 'st_ino', 0) or 0), } def _manifest_skip_matches_generation(path, manifest, skip_bytes): signature = manifest.get('current_skip_signature') if isinstance(manifest, dict) else None if not isinstance(signature, dict) or not os.path.isfile(path): return False try: current = _jsonl_file_generation(path) return ( int(skip_bytes) <= current['size'] and all(int(current[key]) == int(signature.get(key, -1)) for key in ('device', 'size', 'mtime_ns', 'inode')) ) except (OSError, TypeError, ValueError): return False def reconcile_jsonl_segments(path): """Publish physical orphan segments before any active-file mutation.""" manifest = load_jsonl_manifest(path) physical = _projection_segment_sequence(path) existing = { str(item.get('name') or os.path.basename(str(item.get('path') or ''))): item for item in (manifest.get('segments') or []) if isinstance(item, dict) } normalized = [] missing = [] for sequence, segment_path in physical: item = existing.get(os.path.basename(segment_path)) if item is None: item = _segment_manifest_entry(segment_path, sequence) missing.append((sequence, segment_path)) else: item = dict(item, path=os.path.abspath(segment_path), name=os.path.basename(segment_path), sequence=sequence) normalized.append(item) skip_bytes = max(0, int(manifest.get('current_skip_bytes') or 0)) duplicate_path = None duplicate_size = 0 candidates = list(reversed(missing)) if skip_bytes and physical: candidates.insert(0, physical[-1]) for _, segment_path in candidates: size = os.path.getsize(segment_path) if _files_share_prefix(segment_path, path, size): duplicate_path = segment_path duplicate_size = size break changed = bool(missing) or normalized != (manifest.get('segments') or []) effective_skip = duplicate_size if duplicate_path else 0 if changed or skip_bytes or manifest.get('current_skip_signature'): next_sequence = max([sequence for sequence, _ in physical] or [0]) + 1 manifest.update({ 'current': os.path.basename(path), 'current_path': os.path.abspath(path), 'next_sequence': next_sequence, 'segments': normalized, 'current_skip_bytes': effective_skip, 'current_skip_signature': _jsonl_file_generation(path) if effective_skip and os.path.isfile(path) else None, 'updated_at': datetime.now(timezone.utc).isoformat(timespec='seconds'), }) write_jsonl_manifest(path, manifest) if duplicate_path: _remove_file_prefix(path, duplicate_size) manifest['current_skip_bytes'] = 0 manifest['current_skip_signature'] = None manifest['updated_at'] = datetime.now(timezone.utc).isoformat(timespec='seconds') write_jsonl_manifest(path, manifest) return manifest def _segment_consumed_by_registered_keychecks(segment_path): if os.path.basename(segment_path).lower().startswith('scan_results.'): return True keycheck_root = getattr(scan_config, 'keycheck_dir', '') or '' if not os.path.isdir(keycheck_root): return False states = [] consumer_directories = 0 with os.scandir(keycheck_root) as entries: for entry in entries: if not entry.is_dir(follow_symlinks=False): continue consumer_directories += 1 if consumer_directories > 128: return False state_path = os.path.join(entry.path, 'input_state.json') if not os.path.isfile(state_path): return False try: if os.path.getsize(state_path) > 1024 * 1024: return False with open(state_path, 'r', encoding='utf-8') as handle: states.append(json.load(handle)) except (OSError, ValueError): return False if not states or len(states) != consumer_directories: return False absolute = os.path.abspath(segment_path) size = os.path.getsize(segment_path) for state in states: files = state.get('files') if isinstance(state, dict) and isinstance(state.get('files'), dict) else {} record = files.get(absolute) or files.get(segment_path) if not isinstance(record, dict) or int(record.get('offset', 0) or 0) < size: return False return True def _prune_consumed_jsonl_segments(path, manifest, keep_below): physical = _projection_segment_sequence(path) removed = set() while len(physical) >= keep_below and physical: _, candidate = physical[0] if not _segment_consumed_by_registered_keychecks(candidate): break reject_reparse_components(candidate) os.remove(candidate) removed.add(os.path.basename(candidate)) physical.pop(0) if removed: manifest['segments'] = [ item for item in (manifest.get('segments') or []) if str(item.get('name') or os.path.basename(str(item.get('path') or ''))) not in removed ] manifest['updated_at'] = datetime.now(timezone.utc).isoformat(timespec='seconds') write_jsonl_manifest(path, manifest) return physical def rotate_jsonl_if_needed(path, max_bytes, ledger=None): if not max_bytes or max_bytes <= 0: return if not os.path.lexists(path): return reject_reparse_components(path) details = os.stat(path, follow_symlinks=False) if not stat.S_ISREG(details.st_mode) or details.st_size < max_bytes: return repair_jsonl_tail(path) manifest = reconcile_jsonl_segments(path) max_segments = max(1, int(getattr(scan_config, 'jsonl_max_segments', 16))) physical = _prune_consumed_jsonl_segments(path, manifest, max_segments) if len(physical) >= max_segments: raise JsonlProjectionReconciliationRequired( f'JSONL segment retention bound reached for {path}; registered consumers must catch up ' 'or offline reconciliation must retire acknowledged segments' ) segment_path, seq = next_jsonl_segment_path(path, manifest) size = os.path.getsize(path) temporary = f'{segment_path}.{os.getpid()}.{threading.get_ident()}.tmp' try: with open(path, 'rb') as source, open(temporary, 'xb') as destination: shutil.copyfileobj(source, destination, 1024 * 1024) destination.flush() os.fsync(destination.fileno()) harden_private_file(temporary) os.replace(temporary, segment_path) finally: if os.path.exists(temporary): os.remove(temporary) harden_private_file(segment_path) segments = manifest.get('segments') if isinstance(manifest.get('segments'), list) else [] segments.append(_segment_manifest_entry(segment_path, seq)) manifest.update({ 'current': os.path.basename(path), 'current_path': path, 'next_sequence': seq + 1, 'max_bytes': int(max_bytes), 'segments': segments, 'current_skip_bytes': int(size), 'current_skip_signature': _jsonl_file_generation(path), 'updated_at': datetime.now(timezone.utc).isoformat(timespec='seconds'), }) write_jsonl_manifest(path, manifest) if ledger is not None: ledger.execute( "UPDATE publication_identity SET file_name = ? WHERE file_name = ? AND state = 'appended'", (os.path.basename(segment_path), os.path.basename(path)), ) ledger.execute( "UPDATE publication_identity_variant SET file_name = ? WHERE file_name = ?", (os.path.basename(segment_path), os.path.basename(path)), ) ledger.commit() _publish_empty_jsonl_generation(path) manifest['current_skip_bytes'] = 0 manifest['current_skip_signature'] = None manifest['updated_at'] = datetime.now(timezone.utc).isoformat(timespec='seconds') write_jsonl_manifest(path, manifest) logger.info(f'Rotated JSONL {path} -> {segment_path} ({size} bytes)') def append_rotating_jsonl(path, payload, max_mb=None): if not scan_config.jsonl_rotation_enabled: return append_jsonl(path, payload) max_bytes = int(max_mb or 0) * 1024 * 1024 serialized = json.dumps(payload, ensure_ascii=False, default=str) + '\n' lock_path = f'{path}.lock' fd = None try: parent = os.path.dirname(path) if parent: require_private_directory(parent, create=True) fd = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec) repair_jsonl_tail(path) rotate_jsonl_if_needed(path, max_bytes) with open(path, 'ab') as f: f.write(serialized.encode('utf-8')) f.flush() os.fsync(f.fileno()) harden_private_file(path) return True except Exception as e: logger.error(f'Unable to rotating-write {path}: {str(e)}') return False finally: if fd is not None: release_file_lock(fd, lock_path) def projection_segment_paths(path): paths = [segment_path for _, segment_path in _projection_segment_sequence(path)] if os.path.isfile(path): paths.append(path) return paths def jsonl_projection_contains(path, identity_key, identity_value): expected = str(identity_value or '') if not expected: return False ledger_path = jsonl_ledger_path(path) if not os.path.isfile(ledger_path) or not private_file_ready(ledger_path): return False connection = sqlite3.connect(f'file:{ledger_path}?mode=ro', uri=True, timeout=5) try: row = connection.execute( '''SELECT state FROM publication_identity WHERE identity_key = ? AND identity_value = ?''', (identity_key, expected), ).fetchone() return bool(row and row[0] == 'appended') finally: connection.close() def _append_projection_once_locked( path, serialized, identity_key, identity_value, max_bytes, rotation_enabled=None, ): repair_jsonl_tail(path) reconcile_jsonl_segments(path) if len(identity_key) > 64 or len(identity_value) > 256 or any( character in identity_value for character in ('\x00', '\r', '\n') ): raise JsonlProjectionReconciliationRequired('projection identity exceeds its bounded format') max_record_bytes = max(1, int(getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024))) if len(serialized) > max_record_bytes: raise JsonlProjectionReconciliationRequired('projection record exceeds its per-record byte bound') ledger = _open_projection_ledger(path) try: _ensure_projection_ledger_bootstrapped(ledger, path, identity_key) _recover_prepared_publications(ledger, path) digest = hashlib.sha256(serialized).hexdigest() row = ledger.execute( '''SELECT payload_sha256, state FROM publication_identity WHERE identity_key = ? AND identity_value = ?''', (identity_key, identity_value), ).fetchone() if row: if row[1] != 'appended': raise JsonlProjectionReconciliationRequired('publication ledger retained an unresolved append state') variant = ledger.execute( '''SELECT 1 FROM publication_identity_variant WHERE identity_key = ? AND identity_value = ? AND payload_sha256 = ?''', (identity_key, identity_value, digest), ).fetchone() if variant: return True raise JsonlProjectionReconciliationRequired( f'{identity_key} {identity_value} has a conflicting projection payload' ) row_limit = max(1, int(getattr(scan_config, 'jsonl_ledger_max_rows', 1000000))) ledger_byte_limit = max(1024 * 1024, int(getattr(scan_config, 'jsonl_ledger_max_bytes', 512 * 1024 * 1024))) row_count = _ledger_row_count(ledger) if row_count >= row_limit: raise JsonlProjectionReconciliationRequired( f'JSONL identity ledger reached its {row_limit} row bound; run offline reconciliation' ) ledger_file = jsonl_ledger_path(path) if os.path.getsize(ledger_file) + 8192 > ledger_byte_limit: raise JsonlProjectionReconciliationRequired( f'JSONL identity ledger reached its {ledger_byte_limit} byte bound; run offline reconciliation' ) should_rotate = scan_config.jsonl_rotation_enabled if rotation_enabled is None else bool(rotation_enabled) if should_rotate: rotate_jsonl_if_needed(path, max_bytes, ledger=ledger) if not os.path.exists(path): with open(path, 'ab'): pass harden_private_file(path) offset = os.path.getsize(path) now = time.time() ledger.execute( '''INSERT INTO publication_identity ( identity_key, identity_value, payload_sha256, state, file_name, byte_offset, byte_length, created_at, updated_at ) VALUES (?, ?, ?, 'prepared', ?, ?, ?, ?, ?)''', ( identity_key, identity_value, digest, os.path.basename(path), offset, len(serialized), now, now, ), ) _set_ledger_row_count(ledger, row_count + 1) ledger.commit() with open(path, 'ab') as handle: handle.write(serialized) handle.flush() os.fsync(handle.fileno()) harden_private_file(path) ledger.execute( '''UPDATE publication_identity SET state = 'appended', updated_at = ? WHERE identity_key = ? AND identity_value = ? AND state = 'prepared' ''', (time.time(), identity_key, identity_value), ) ledger.execute( '''INSERT INTO publication_identity_variant ( identity_key, identity_value, payload_sha256, file_name, byte_offset, byte_length, created_at ) VALUES (?, ?, ?, ?, ?, ?, ?)''', ( identity_key, identity_value, digest, os.path.basename(path), offset, len(serialized), time.time(), ), ) ledger.commit() return True finally: ledger.close() if os.path.exists(jsonl_ledger_path(path)): harden_private_file(jsonl_ledger_path(path)) def append_rotating_jsonl_once(path, payload, identity_key, identity_value, max_mb=None): if not identity_value: logger.error(f'Unable to publish {path}: missing {identity_key}') return False max_bytes = int(max_mb or 0) * 1024 * 1024 lock_path = f'{path}.lock' lock = None try: require_private_directory(os.path.dirname(os.path.abspath(path)), create=True) lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec) serialized = (json.dumps(payload, ensure_ascii=False, default=str) + '\n').encode('utf-8') return _append_projection_once_locked( path, serialized, identity_key, str(identity_value), max_bytes, ) except Exception as exc: logger.error(f'Unable to idempotently publish {path}: {exc}') return False finally: if lock is not None: release_file_lock(lock, lock_path) def _rotate_bounded_text_log(path, keep, max_bytes=None): if not os.path.exists(path) or os.path.getsize(path) <= 0: return if max_bytes and os.path.getsize(path) > int(max_bytes): with open(path, 'r+b') as handle: handle.truncate(int(max_bytes)) handle.flush() os.fsync(handle.fileno()) segment, _ = next_jsonl_segment_path(path, {}) os.replace(path, segment) harden_private_file(segment) with open(path, 'ab'): pass harden_private_file(path) segments = [candidate for candidate in projection_segment_paths(path) if candidate != path] segments.sort(key=lambda candidate: os.path.getmtime(candidate), reverse=True) for candidate in segments[max(0, int(keep or 0)):]: reject_reparse_components(candidate) os.remove(candidate) def append_scan_errors_once(path, result, max_mb=None, keep=5): event_id = str(result.get('scan_event_id') or '') if not event_id: logger.error(f'Unable to publish {path}: missing scan_event_id') return False errors = list(result.get('errors') or []) if not errors: return True max_bytes = max(1, int(max_mb or 0) * 1024 * 1024) timestamp = result.get('timestamp') or result.get('scan_started_at') or event_id scan_type = result.get('scan_type', '') target = result.get('target', '') lock_path = f'{path}.lock' lock = None try: require_private_directory(os.path.dirname(os.path.abspath(path)), create=True) lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec) repair_jsonl_tail(path) for index, error in enumerate(errors, 1): row_id = f'{event_id}:{index}' line = f'{row_id}\t{timestamp}\t{scan_type}\t{target}\t{error}\n'.encode('utf-8', errors='replace') if len(line) > max_bytes: suffix = b'...[truncated]\n' line = line[:max(0, max_bytes - len(suffix))] + suffix if os.path.exists(path) and os.path.getsize(path) + len(line) > max_bytes: _rotate_bounded_text_log(path, keep, max_bytes) _append_projection_once_locked( path, line, 'error_row_id', row_id, 0, rotation_enabled=False, ) return True except Exception as exc: logger.error(f'Unable to idempotently publish {path}: {exc}') return False finally: if lock is not None: release_file_lock(lock, lock_path) def parse_finding_datetime(value): if not value: return None value = str(value).strip() for date_format in ("%Y-%m-%d %H:%M:%S %z", "%Y-%m-%dT%H:%M:%SZ", "%Y-%m-%dT%H:%M:%S%z"): try: return datetime.strptime(value, date_format) except ValueError: continue return None def get_finding_timestamp(finding): data = finding.get('SourceMetadata', {}).get('Data', {}) if not isinstance(data, dict): return None for source in data.values(): if isinstance(source, dict) and source.get('timestamp'): return parse_finding_datetime(source.get('timestamp')) return None def filter_findings_by_age(findings, max_age_days=None): if not max_age_days or max_age_days <= 0: return findings, 0, 0 cutoff = datetime.now().astimezone() - timedelta(days=max_age_days) kept = [] skipped_old = 0 skipped_missing = 0 for finding in findings: timestamp = get_finding_timestamp(finding) if timestamp is None: skipped_missing += 1 continue if timestamp >= cutoff: kept.append(finding) else: skipped_old += 1 return kept, skipped_old, skipped_missing def filter_dropped_detectors(findings): drop = { item.strip().lower() for item in csv_items(_scan_policy_value('drop_detectors', [])) if item.strip() } if not drop: return findings, 0, Counter() kept = [] counts = Counter() skipped = 0 for finding in findings or []: detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '') if detector.lower() in drop: skipped += 1 counts[detector or '(unknown)'] += 1 continue kept.append(finding) return kept, skipped, counts GITHUB_TOKEN_PREFIXES = ('ghp_', 'gho_', 'ghu_', 'ghs_', 'ghr_', 'github_pat_') GITLAB_TOKEN_PREFIXES = ('glpat-', 'gloas-', 'glcbt-', 'glimt-', 'glrt-', 'glft-', 'glsoat-') def finding_raw_values(finding): values = [] for key in ('Raw', 'RawV2'): value = finding.get(key) if value: values.append(str(value)) return values def is_known_provider_token_shape(detector_name, finding): values = finding_raw_values(finding) detector = str(detector_name or '').lower() if detector in ('github', 'githuboauth2'): return any(value.startswith(GITHUB_TOKEN_PREFIXES) for value in values) if detector == 'gitlab': return any(value.startswith(GITLAB_TOKEN_PREFIXES) for value in values) return True def filter_noisy_findings(findings): if not _scan_policy_value('strict_git_provider_token_filter', True): return findings, 0 kept = [] skipped = 0 for finding in findings: detector = finding.get('DetectorName') or '' detector_key = str(detector).lower() if detector_key in ('github', 'githuboauth2', 'gitlab') and not finding.get('Verified', False): if not is_known_provider_token_shape(detector, finding): skipped += 1 continue kept.append(finding) return kept, skipped CUSTOM_DETECTOR_NAME_ALIASES = { 'xaicontextafter': 'Xai', 'zaiglmcontextafter': 'ZaiGLM', } def normalize_custom_detector_names(findings): for finding in findings or []: detector = str(finding.get('DetectorName') or '') extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {} custom_name = str(extra.get('name') or '').strip() if detector.lower() == 'customregex' and custom_name: finding.setdefault('OriginalDetectorName', detector) finding['DetectorName'] = CUSTOM_DETECTOR_NAME_ALIASES.get( custom_name.lower(), custom_name, ) return findings def apply_finding_filters(results, target_label='target', *, log_target=True): emit_client_scan_phase('filtering') findings = results.get('findings') or [] normalize_custom_detector_names(findings) findings, dropped, dropped_counts = filter_dropped_detectors(findings) if dropped: results['findings'] = findings results['dropped_detectors_count'] = int(results.get('dropped_detectors_count', 0) or 0) + dropped results['dropped_detectors'] = dict(dropped_counts) target_context = f' for {target_label}' if log_target else '' logger.info( f"Dropped {dropped} configured noise detector finding(s)" f"{target_context}: {dict(dropped_counts)}" ) filtered, skipped = filter_noisy_findings(findings) if skipped: results['findings'] = filtered results['filtered_findings_count'] = int(results.get('filtered_findings_count', 0) or 0) + skipped target_context = f' for {target_label}' if log_target else '' logger.info( f"Filtered {skipped} noisy unverified Git provider finding(s)" f"{target_context}" ) return results def finding_source_location(finding): metadata = finding.get('SourceMetadata') or {} data = metadata.get('Data') if isinstance(metadata, dict) else {} if not isinstance(data, dict): return None, None for source in data.values(): if not isinstance(source, dict): continue file_path = source.get('file') or source.get('path') or source.get('File') line = source.get('line') or source.get('Line') if file_path: try: line = int(line) if line else None except (TypeError, ValueError): line = None return file_path, line return None, None def context_enrichment_budget(): return { 'started_at': time.monotonic(), 'max_source_bytes': max(0, int(getattr(scan_config, 'context_enrichment_max_source_bytes', 16 * 1024 * 1024))), 'max_findings': max(0, int(getattr(scan_config, 'context_enrichment_max_findings', 2000))), 'max_postman_comparisons': max(0, int(getattr(scan_config, 'context_enrichment_max_postman_comparisons', 200000))), 'max_elapsed_sec': max(0.0, float(getattr(scan_config, 'context_enrichment_max_elapsed_sec', 5.0))), 'source_bytes': 0, 'postman_comparisons': 0, 'finding_ids': set(), 'files': {}, } def _context_budget_expired(budget): return time.monotonic() - budget['started_at'] >= budget['max_elapsed_sec'] def _context_budget_claim_finding(budget, finding): identity = id(finding) if identity in budget['finding_ids']: return True if len(budget['finding_ids']) >= budget['max_findings']: return False budget['finding_ids'].add(identity) return True def _add_context_warning(results, warning_class, detail): warning_class = f'context_enrichment_{warning_class}' classes = list(results.get('warning_classes') or []) if warning_class not in classes: warnings = list(results.get('warnings') or []) warnings.append(f'Optional context enrichment degraded: {detail}'[:400]) results['warnings'] = warnings classes.append(warning_class) results['warning_classes'] = sorted(set(classes)) results['context_enrichment_degraded'] = True results['degraded'] = True def _read_context_source(file_path, budget): key = os.path.normcase(os.path.abspath(os.fspath(file_path))) if key in budget['files']: return budget['files'][key] if _context_budget_expired(budget): return None, False, 'elapsed' try: size = max(0, int(os.path.getsize(file_path))) remaining = max(0, budget['max_source_bytes'] - budget['source_bytes']) if remaining <= 0 and size: return None, False, 'source_bytes' amount = min(size, remaining) with open(file_path, 'rb') as handle: payload = handle.read(amount) budget['source_bytes'] += len(payload) complete = len(payload) == size reason = None if complete else 'source_bytes' if size > remaining else 'read' budget['files'][key] = (payload, complete, reason) return payload, complete, reason except (OSError, TypeError, ValueError): return None, False, 'read' def _nearby_context_from_lines(lines, file_path, line_number=None, radius=20, max_chars=12000): if not lines: return None if line_number and line_number > 0: start = max(0, line_number - radius - 1) requested_end = line_number + radius else: start = 0 requested_end = radius * 2 + 1 if start >= len(lines): return None end = min(len(lines), requested_end) return { 'file': file_path, 'line': line_number, 'start_line': start + 1, 'end_line': end, 'nearby': ''.join(lines[start:end])[:max_chars], } def read_nearby_context(file_path, line_number=None, radius=20, max_chars=12000): if not file_path or not os.path.exists(file_path): return None if line_number and line_number > 0: start = max(0, line_number - radius - 1) requested_end = line_number + radius else: start = 0 requested_end = radius * 2 + 1 selected = [] actual_end = 0 try: with open(file_path, 'r', encoding='utf-8', errors='replace') as f: for index in range(requested_end): line = f.readline(max_chars + 1) if not line: break if not line.endswith('\n') and len(line) > max_chars: while True: remainder = f.readline(64 * 1024) if not remainder or remainder.endswith('\n'): break actual_end = index + 1 if index >= start and sum(len(value) for value in selected) < max_chars: selected.append(line[:max_chars]) except OSError: return None if actual_end == 0: return None snippet = ''.join(selected)[:max_chars] return { 'file': file_path, 'line': line_number, 'start_line': start + 1, 'end_line': actual_end, 'nearby': snippet, } def attach_nearby_context(results, budget=None): budget = budget or context_enrichment_budget() grouped = {} finding_budget_exhausted = False try: for finding in results.get('findings') or []: if _context_budget_expired(budget): _add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained') return results if not isinstance(finding, dict): continue file_path, line_number = finding_source_location(finding) if not file_path: continue if not _context_budget_claim_finding(budget, finding): finding_budget_exhausted = True break key = os.path.normcase(os.path.abspath(os.fspath(file_path))) grouped.setdefault(key, {'path': file_path, 'findings': []})['findings'].append((finding, line_number)) for group in grouped.values(): if _context_budget_expired(budget): _add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained') return results payload, complete, reason = _read_context_source(group['path'], budget) if payload is None: if reason in ('elapsed', 'source_bytes'): dimension = 'elapsed-time' if reason == 'elapsed' else 'source-byte' _add_context_warning(results, 'budget', f'{dimension} budget was exhausted; remaining findings were retained') return results _add_context_warning(results, 'failure', 'a nearby source file could not be read; findings were retained') continue lines = payload.decode('utf-8', errors='replace').splitlines(keepends=True) for finding, line_number in group['findings']: if _context_budget_expired(budget): _add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained') return results context = _nearby_context_from_lines(lines, group['path'], line_number) if context: finding['ScannerContext'] = context if not complete: if reason == 'source_bytes': _add_context_warning(results, 'budget', 'source-byte budget was exhausted; remaining findings were retained') return results _add_context_warning(results, 'failure', 'a nearby source file changed during its bounded read; findings were retained') if finding_budget_exhausted: _add_context_warning(results, 'budget', 'finding-count budget was exhausted; remaining findings were retained') except Exception: logger.warning('Optional nearby context enrichment failed; parsed findings were retained') _add_context_warning(results, 'failure', 'nearby context parsing failed; findings were retained') return results QWEN_ROUTING_CONTEXT_RE = re.compile( r'(?:DASHSCOPE_API_KEY|QWEN_API_KEY|dashscope|qwen|model[_-]?studio|bailian)', re.IGNORECASE, ) DEEPSEEK_ROUTING_CONTEXT_RE = re.compile( r'(?:DEEPSEEK_API_KEY|deepseek|api\.deepseek\.com)', re.IGNORECASE, ) KIMI_ROUTING_CONTEXT_RE = re.compile( r'(?:MOONSHOT_API_KEY|KIMI_API_KEY|api\.moonshot\.(?:ai|cn)|platform\.kimi\.(?:ai|com))', re.IGNORECASE, ) ZAI_ROUTING_CONTEXT_RE = re.compile( r'(?:ZAI_API_KEY|GLM_API_KEY|ZHIPUAI_API_KEY|BIGMODEL_API_KEY|api\.z\.ai|' r'open\.bigmodel\.cn|zhipuai|chatglm)', re.IGNORECASE, ) QWEN_ROUTING_DETECTORS = {'qwendashscope', 'qwen_dashscope', 'dashscope', 'qwen'} QWEN_EXPLICIT_ROUTING_DETECTORS = {'qwendashscope', 'qwen_dashscope'} DEEPSEEK_ROUTING_DETECTORS = {'deepseek', 'deepseekapikey', 'deepseek_api_key'} DEEPSEEK_EXPLICIT_ROUTING_DETECTORS = {'deepseekapikey', 'deepseek_api_key'} KIMI_ROUTING_DETECTORS = {'kimimoonshot', 'moonshotai', 'moonshot', 'kimi'} KIMI_EXPLICIT_ROUTING_DETECTORS = {'kimimoonshot', 'moonshotai'} ZAI_ROUTING_DETECTORS = {'zaiglm'} ZAI_EXPLICIT_ROUTING_DETECTORS = {'zaiglm'} AMBIGUOUS_QWEN_DEEPSEEK_HINT = 'ambiguous_qwen_deepseek' AMBIGUOUS_GENERIC_SK_HINT = 'ambiguous_generic_sk' GENERIC_SK_PROVIDERS = {'qwen', 'deepseek', 'kimi', 'zai'} GENERIC_SK_PROVIDER_ORDER = ('deepseek', 'zai', 'qwen', 'kimi') GENERIC_SK_PROVIDER_HINTS = { *GENERIC_SK_PROVIDERS, AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT, } EXPLICIT_ASSIGNMENT_PROVIDER_HINT_SOURCE = 'explicit_assignment' GENERIC_SK_ROUTING_RE = re.compile(r'^sk-[A-Za-z0-9][A-Za-z0-9_-]{20,505}$') QWEN_SPECIFIC_ROUTING_RE = re.compile(r'^sk-sp-[A-Za-z0-9][A-Za-z0-9_-]{20,505}$') ZAI_SPECIFIC_ROUTING_RE = re.compile( r'^(?:zai-[A-Za-z0-9][A-Za-z0-9_-]{20,505}|' r'[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{20,505})$' ) def provider_routing_context_text(finding, max_context_chars=65536): if not isinstance(finding, dict): return '' context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {} parts = [] remaining = max(0, int(max_context_chars)) def add(value): nonlocal remaining if not value or remaining <= 0: return text = str(value)[:remaining] parts.append(text) remaining -= len(text) for key in ('nearby', 'file'): add(context.get(key)) metadata = finding.get('SourceMetadata') if isinstance(finding.get('SourceMetadata'), dict) else {} data = metadata.get('Data') if isinstance(metadata.get('Data'), dict) else {} for details in data.values(): if not isinstance(details, dict): continue for key in ('file', 'repository', 'repo', 'link', 'image'): add(details.get(key)) return '\n'.join(parts) def provider_routing_detector_names(finding): if not isinstance(finding, dict): return set() detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '').strip().lower() extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {} custom_name = str(extra.get('name') or '').strip().lower() return {name for name in (detector, custom_name) if name} def is_qwen_routing_detector(finding): return bool(provider_routing_detector_names(finding) & QWEN_ROUTING_DETECTORS) def is_explicit_qwen_routing_detector(finding): return bool(provider_routing_detector_names(finding) & QWEN_EXPLICIT_ROUTING_DETECTORS) def is_deepseek_routing_detector(finding): return bool(provider_routing_detector_names(finding) & DEEPSEEK_ROUTING_DETECTORS) def is_explicit_deepseek_routing_detector(finding): return bool(provider_routing_detector_names(finding) & DEEPSEEK_EXPLICIT_ROUTING_DETECTORS) def is_kimi_routing_detector(finding): return bool(provider_routing_detector_names(finding) & KIMI_ROUTING_DETECTORS) def is_explicit_kimi_routing_detector(finding): return bool(provider_routing_detector_names(finding) & KIMI_EXPLICIT_ROUTING_DETECTORS) def is_zai_routing_detector(finding): return bool(provider_routing_detector_names(finding) & ZAI_ROUTING_DETECTORS) def is_explicit_zai_routing_detector(finding): return bool(provider_routing_detector_names(finding) & ZAI_EXPLICIT_ROUTING_DETECTORS) def is_generic_sk_routing_detector(finding): return bool( is_qwen_routing_detector(finding) or is_deepseek_routing_detector(finding) or is_kimi_routing_detector(finding) or is_zai_routing_detector(finding) ) def provider_routing_hint_evidence(hint): if hint == AMBIGUOUS_QWEN_DEEPSEEK_HINT: return {'qwen', 'deepseek'} if hint == AMBIGUOUS_GENERIC_SK_HINT: return set(GENERIC_SK_PROVIDERS) return {hint} if hint in GENERIC_SK_PROVIDERS else set() def ambiguous_provider_routing_hint(evidence): evidence = set(evidence) if evidence == {'qwen', 'deepseek'}: return AMBIGUOUS_QWEN_DEEPSEEK_HINT return AMBIGUOUS_GENERIC_SK_HINT def explicit_provider_routing_evidence(finding): evidence = set() if is_explicit_qwen_routing_detector(finding): evidence.add('qwen') if is_explicit_deepseek_routing_detector(finding): evidence.add('deepseek') if is_explicit_kimi_routing_detector(finding): evidence.add('kimi') if is_explicit_zai_routing_detector(finding): evidence.add('zai') context = finding.get('ScannerContext') if isinstance(finding, dict) else None if isinstance(context, dict) and context.get('provider_hint_source') == EXPLICIT_ASSIGNMENT_PROVIDER_HINT_SOURCE: evidence.update(provider_routing_hint_evidence(context.get('provider_hint'))) return evidence def provider_routing_evidence(finding, max_context_chars=65536): if not isinstance(finding, dict): return set() context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {} text = provider_routing_context_text(finding, max_context_chars) evidence = explicit_provider_routing_evidence(finding) if QWEN_ROUTING_CONTEXT_RE.search(text): evidence.add('qwen') if DEEPSEEK_ROUTING_CONTEXT_RE.search(text): evidence.add('deepseek') if KIMI_ROUTING_CONTEXT_RE.search(text): evidence.add('kimi') if ZAI_ROUTING_CONTEXT_RE.search(text): evidence.add('zai') evidence.update(provider_routing_hint_evidence(context.get('provider_hint'))) raw_values = finding_raw_values(finding) if any(QWEN_SPECIFIC_ROUTING_RE.fullmatch(value) for value in raw_values): evidence.add('qwen') if any(ZAI_SPECIFIC_ROUTING_RE.fullmatch(value) for value in raw_values): evidence.add('zai') if ( not evidence and is_generic_sk_routing_detector(finding) and any(GENERIC_SK_ROUTING_RE.fullmatch(value) for value in raw_values) ): evidence.update(GENERIC_SK_PROVIDERS) return evidence def derive_provider_routing_hint( finding, max_context_chars=65536, evidence=None, explicit_evidence=None, ): if not isinstance(finding, dict): return '' context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {} evidence = set(evidence) if evidence is not None else provider_routing_evidence(finding, max_context_chars) explicit_evidence = ( set(explicit_evidence) if explicit_evidence is not None else explicit_provider_routing_evidence(finding) ) if len(explicit_evidence) > 1: provider_hint = ambiguous_provider_routing_hint(explicit_evidence) elif explicit_evidence: provider_hint = next(iter(explicit_evidence)) elif len(evidence) > 1: provider_hint = ambiguous_provider_routing_hint(evidence) elif evidence: provider_hint = next(iter(evidence)) else: provider_hint = '' persisted_context = dict(context) if provider_hint: persisted_context['provider_hint'] = provider_hint else: persisted_context.pop('provider_hint', None) if explicit_evidence: persisted_context['provider_hint_source'] = EXPLICIT_ASSIGNMENT_PROVIDER_HINT_SOURCE else: persisted_context.pop('provider_hint_source', None) if provider_hint in (AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT): persisted_context['provider_candidates'] = [ provider for provider in GENERIC_SK_PROVIDER_ORDER if provider in evidence ] else: persisted_context.pop('provider_candidates', None) if persisted_context or 'ScannerContext' in finding: finding['ScannerContext'] = persisted_context return provider_hint def strip_nearby_context_for_persistence(result): findings = result.get('findings') or [] evidence_by_value = {} explicit_evidence_by_value = {} for finding in findings: if not ( is_qwen_routing_detector(finding) or is_deepseek_routing_detector(finding) or is_kimi_routing_detector(finding) or is_zai_routing_detector(finding) ): continue evidence = provider_routing_evidence(finding) explicit_evidence = explicit_provider_routing_evidence(finding) for value in finding_raw_values(finding): evidence_by_value.setdefault(value, set()).update(evidence) explicit_evidence_by_value.setdefault(value, set()).update(explicit_evidence) for finding in findings: is_routed_detector = ( is_qwen_routing_detector(finding) or is_deepseek_routing_detector(finding) or is_kimi_routing_detector(finding) or is_zai_routing_detector(finding) ) evidence = provider_routing_evidence(finding) explicit_evidence = explicit_provider_routing_evidence(finding) if is_routed_detector: for value in finding_raw_values(finding): evidence.update(evidence_by_value.get(value, ())) explicit_evidence.update(explicit_evidence_by_value.get(value, ())) derive_provider_routing_hint( finding, evidence=evidence, explicit_evidence=explicit_evidence, ) for finding in findings: postman_context = finding.get('PostmanContext') if isinstance(finding, dict) else None if isinstance(postman_context, dict): finding['PostmanContext'] = sanitize_postman_context(postman_context) context = finding.get('ScannerContext') if isinstance(finding, dict) else None if isinstance(context, dict) and 'nearby' in context: finding['ScannerContext'] = {key: value for key, value in context.items() if key != 'nearby'} if is_foundry_detector(finding): text = foundry_finding_text(finding) endpoints = [normalize_foundry_endpoint(match) for match in POSTMAN_FOUNDRY_ENDPOINT_RE.findall(text)] keys = foundry_candidate_keys(finding, text) endpoint = next((item for item in endpoints if item), '') key = keys[0] if keys else '' if endpoint and key: finding['Raw'] = key finding['RawV2'] = f'{endpoint}:{key}' finding['Redacted'] = f'{endpoint}:***REDACTED***' return result def finding_raw_secret_for_uid(finding): for key in ('RawV2', 'Raw'): value = finding.get(key) if value: return str(value) structured = finding.get('StructuredData') if isinstance(structured, dict): for value in structured.values(): if isinstance(value, str) and value: return value return '' def finding_location_for_uid(finding): metadata = finding.get('SourceMetadata') if isinstance(finding, dict) else {} data = metadata.get('Data') if isinstance(metadata, dict) else {} if not isinstance(data, dict): return '', '', '' for details in data.values(): if not isinstance(details, dict): continue file_path = details.get('file') or details.get('path') or details.get('File') or '' line_number = details.get('line') or details.get('Line') or '' commit_hash = details.get('commit') or details.get('commitHash') or details.get('commit_hash') or '' return str(file_path or ''), str(line_number or ''), str(commit_hash or '') return '', '', '' def sha256_json(value): return hashlib.sha256(json.dumps(value, ensure_ascii=False, default=str, sort_keys=True).encode('utf-8', errors='replace')).hexdigest() def _bounded_utf8(value, max_bytes): text = ''.join(' ' if ord(character) < 32 else character for character in str(value or '')) return text.encode('utf-8', errors='replace')[:max_bytes].decode('utf-8', errors='ignore') def keycheck_input_line_limit(): return max(1024, int(getattr(scan_config, 'keycheck_input_max_line_bytes', 16 * 1024 * 1024))) def finding_projection_payload(finding): payload_bytes = json.dumps(finding, ensure_ascii=False, default=str).encode('utf-8') line_limit = keycheck_input_line_limit() if len(payload_bytes) + 1 <= line_limit: return finding, False raw_secret = finding_raw_secret_for_uid(finding) metadata = finding.get('SourceMetadata') if isinstance(finding.get('SourceMetadata'), dict) else {} data = metadata.get('Data') if isinstance(metadata.get('Data'), dict) else {} source_type = next(iter(data), '') file_path, line_number, commit_hash = finding_location_for_uid(finding) marker = { 'finding_uid': _bounded_utf8(finding.get('finding_uid'), 256), 'DetectorName': _bounded_utf8(finding.get('DetectorName') or finding.get('DetectorType'), 64), 'finding_omitted': True, 'keycheck_uncheckable': True, 'omission_reason': 'oversized_finding', 'secret_sha256': hashlib.sha256(raw_secret.encode('utf-8', errors='replace')).hexdigest() if raw_secret else '', 'payload_sha256': hashlib.sha256(payload_bytes).hexdigest(), 'SourceIdentity': { 'type': _bounded_utf8(source_type, 32), 'file': _bounded_utf8(file_path, 128), 'line': _bounded_utf8(line_number, 16), 'commit': _bounded_utf8(commit_hash, 64), 'sha256': sha256_json(metadata), }, } marker_bytes = (json.dumps(marker, ensure_ascii=False, default=str) + '\n').encode('utf-8') if len(marker_bytes) > line_limit: marker = { 'finding_uid': _bounded_utf8(finding.get('finding_uid'), 256), 'DetectorName': _bounded_utf8(finding.get('DetectorName') or finding.get('DetectorType'), 32), 'finding_omitted': True, 'keycheck_uncheckable': True, 'secret_sha256': marker['secret_sha256'], 'payload_sha256': marker['payload_sha256'], 'source_identity_sha256': marker['SourceIdentity']['sha256'], } marker_bytes = (json.dumps(marker, ensure_ascii=False, default=str) + '\n').encode('utf-8') if len(marker_bytes) > line_limit: raise RuntimeError('bounded oversized-finding marker exceeds the keycheck input line limit') return marker, True def assign_finding_uids(result): findings = result.get('findings') or [] scan_event_id = str(result.get('scan_event_id') or '') if not scan_event_id: raise ValueError('event-based finding identities require scan_event_id') for index, finding in enumerate(findings, 1): if not isinstance(finding, dict) or finding.get('finding_uid'): continue detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '') raw_secret = finding_raw_secret_for_uid(finding) secret_hash = hashlib.sha256(raw_secret.encode('utf-8', errors='replace')).hexdigest() if raw_secret else '' fallback_hash = sha256_json(finding) if not secret_hash else '' file_path, line_number, commit_hash = finding_location_for_uid(finding) finding['finding_uid'] = hashlib.sha256('|'.join([ 'truf-finding-v2', scan_event_id, str(index), detector, secret_hash or fallback_hash, file_path, line_number, commit_hash, ]).encode('utf-8', errors='replace')).hexdigest() def save_scan_result(result): """Persist findings like raw TruffleHog JSONL and keep errors separately.""" results_dir = get_results_dir() if not results_dir: return False success = True event_id = str(result.get('scan_event_id') or '') if not event_id: logger.error('Unable to persist scan result without scan_event_id') return False target = result.get('target', '') scan_type = result.get('scan_type', '') timestamp = result.get('timestamp', datetime.now().isoformat()) assign_finding_uids(result) try: foundry_candidates = write_foundry_keycheck_candidates_from_findings(copy.deepcopy(result)) if foundry_candidates: logger.info(f"Queued {foundry_candidates} Azure Foundry keycheck candidate(s) from {scan_type}:{target}") except Exception as e: logger.warning(f"Unable to queue Azure Foundry keycheck candidate(s) for {scan_type}:{target}: {str(e)}") success = False projection_result = copy.deepcopy(result) strip_nearby_context_for_persistence(projection_result) projected_findings = [] oversized_count = 0 for finding in projection_result.get('findings') or []: projected, oversized = finding_projection_payload(finding) projected_findings.append(projected) oversized_count += int(oversized) projection_result['findings'] = projected_findings if oversized_count: projection_result.setdefault('warnings', []).append( f'{oversized_count} oversized finding(s) were projected as uncheckable metadata markers; ' 'PostgreSQL retains the authoritative findings' ) projection_result['degraded'] = True projection_result['oversized_findings_omitted'] = oversized_count if projected_findings or projection_result.get('errors') or projection_result.get('warnings') or projection_result.get('skipped'): success = append_rotating_jsonl_once( os.path.join(results_dir, 'scan_results.jsonl'), projection_result, 'scan_event_id', event_id, scan_config.scan_results_max_mb, ) and success for finding in projected_findings: success = append_rotating_jsonl_once( os.path.join(results_dir, 'found_secrets.jsonl'), finding, 'finding_uid', finding.get('finding_uid'), scan_config.found_secrets_max_mb, ) and success if projection_result.get('errors'): path = os.path.join(results_dir, 'scan_errors.log') success = append_scan_errors_once( path, projection_result, scan_config.scan_errors_max_mb, scan_config.scan_errors_keep, ) and success return success # ====================== # DOCKER TOKEN MANAGEMENT # ====================== @dataclass(frozen=True, repr=False) class DockerAccount: name: str username: str token: str config_dir: str def __repr__(self): return f'DockerAccount(name={self.name!r})' @dataclass(frozen=True, repr=False) class DockerRegistryAuth: token: str account_name: str = '' challenge: str = '' class DockerTokenManager: def __init__(self): self.tokens = [] self.token_dirs = [] self.accounts = [] self.current_index = 0 self.request_index = 0 self.lock = threading.RLock() self.config_key = None self.cooldown_sec = 1800 self.cooldown_until = {} self.cooldown_categories = {} self.hub_tokens = {} self.hub_token_locks = {} self.invalid_accounts = set() self.status_events = {} self.explicit_pool = False @staticmethod def _account_name(entry, index): return str(entry.get('name') or f'docker_{index + 1}').strip() def setup_accounts( self, entries, cooldown_sec=1800, explicit_pool=False, create_config_dirs=True, ): normalized = [] seen_names = set() for index, entry in enumerate(entries or []): if not isinstance(entry, dict): continue name = self._account_name(entry, index) username = str(entry.get('username') or '').strip() token = str(entry.get('token') or '').strip() if not name or not username or not token or name in seen_names: continue seen_names.add(name) normalized.append((name, username, token)) config_key = (bool(create_config_dirs),) + tuple( (name, username, hashlib.sha256(token.encode('utf-8')).hexdigest()) for name, username, token in normalized ) with self.lock: self.cooldown_sec = max(60, min(86400, int(cooldown_sec or 1800))) if ( config_key == self.config_key and self.explicit_pool == bool(explicit_pool) and ( not normalized or not create_config_dirs or all( os.path.isfile(os.path.join(account.config_dir, 'config.json')) for account in self.accounts ) ) ): return self._cleanup_locked() self.config_key = config_key self.current_index = 0 self.request_index = 0 self.cooldown_until = {} self.cooldown_categories = {} self.hub_tokens = {} self.hub_token_locks = {} self.invalid_accounts = set() self.status_events = {} self.explicit_pool = bool(explicit_pool) for name, username, token in normalized: temp_dir = '' if create_config_dirs: temp_dir = create_docker_config_dir() auth = base64.b64encode(f'{username}:{token}'.encode()).decode() docker_auth = {'auth': auth} config = { 'auths': { 'https://index.docker.io/v1/': docker_auth, 'index.docker.io': docker_auth, 'registry-1.docker.io': docker_auth, 'https://registry-1.docker.io': docker_auth, 'docker.io': docker_auth, } } config_path = os.path.join(temp_dir, 'config.json') with open(config_path, 'w') as f: json.dump(config, f) harden_private_file(config_path) account = DockerAccount(name, username, token, temp_dir) self.accounts.append(account) self.tokens.append(f'{username}:{token}') if temp_dir: self.token_dirs.append(temp_dir) logger.info('Configured Docker account: %s', name) def setup_tokens(self, tokens_str=None, username=None, create_config_dirs=True): """Initialize Docker tokens from environment variable""" username = (username or os.getenv('DOCKERHUB_USERNAME') or os.getenv('DOCKER_USERNAME') or '').strip() tokens_str = tokens_str if tokens_str is not None else os.getenv('DOCKER_TOKENS', '') if not tokens_str: single_token = os.getenv('DOCKERHUB_TOKEN') or os.getenv('DOCKER_TOKEN') if single_token: tokens_str = single_token entries = [] if tokens_str: for index, raw_token in enumerate(tokens_str.replace('\n', ',').split(',')): raw_token = raw_token.strip() if not raw_token: continue if ':' in raw_token: account_username, token = raw_token.split(':', 1) elif username: account_username, token = username, raw_token else: logger.warning('Docker token without username ignored. Use username:token or set DOCKERHUB_USERNAME.') continue account_username = str(account_username).strip() token = str(token).strip() if account_username and token: entries.append({ 'name': f'docker_{index + 1}', 'username': account_username, 'token': token, }) self.setup_accounts( entries, explicit_pool=False, create_config_dirs=create_config_dirs, ) def has_accounts(self): with self.lock: return bool(self.accounts) def account_count(self): with self.lock: return len(self.accounts) def next_account(self, endpoint, excluded_names=None): excluded_names = set(excluded_names or ()) with self.lock: if not self.accounts: return None now = time.time() for offset in range(len(self.accounts)): position = (self.request_index + offset) % len(self.accounts) account = self.accounts[position] if account.name in excluded_names: continue if account.name in self.invalid_accounts: continue if float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now: continue self.request_index = (position + 1) % len(self.accounts) return account return None def report_http_status(self, account, endpoint, status, response=None, category=None): account_name = account.name if isinstance(account, DockerAccount) else str(account or '') if not account_name: return False status = int(status or 0) category = str(category or ('rate_limit' if status == 429 else 'auth_forbidden')) with self.lock: if account_name in self.invalid_accounts and category != 'auth_invalid': return self._all_unavailable_locked(endpoint) if category == 'auth_invalid': self.invalid_accounts.add(account_name) retry_at = float('inf') reset_at = 'manual' else: delay = dockerhub_retry_after_seconds(response) if status == 429 else self.cooldown_sec retry_at = time.time() + max(60, int(delay)) reset_at = datetime.fromtimestamp(retry_at, timezone.utc).isoformat(timespec='seconds') self.cooldown_until[(account_name, endpoint)] = retry_at self.cooldown_categories[(account_name, endpoint)] = category if status == 401 or category == 'auth_invalid': self.hub_tokens.pop(account_name, None) self.status_events[(account_name, endpoint)] = { 'name': account_name, 'endpoint': endpoint, 'category': category, 'reset_at': reset_at, 'message': f'Docker {endpoint} HTTP {status}', } return self._all_unavailable_locked(endpoint) def report_success(self, account, endpoint): account_name = account.name if isinstance(account, DockerAccount) else str(account or '') if not account_name: return with self.lock: if account_name in self.invalid_accounts: return expires_at = float(self.cooldown_until.get((account_name, endpoint), 0) or 0) if expires_at <= time.time(): self.cooldown_until.pop((account_name, endpoint), None) self.cooldown_categories.pop((account_name, endpoint), None) event_key = (account_name, endpoint) if event_key not in self.status_events: self.status_events[event_key] = { 'name': account_name, 'endpoint': endpoint, 'category': 'ok', 'reset_at': None, 'message': '', } def _all_unavailable_locked(self, endpoint): if not self.accounts: return False now = time.time() return all( account.name in self.invalid_accounts or float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now for account in self.accounts ) def all_unavailable(self, endpoint): with self.lock: return self._all_unavailable_locked(endpoint) def rate_limit_contributes_to_exhaustion(self, endpoint): with self.lock: if not self._all_unavailable_locked(endpoint): return False now = time.time() return any( float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now and self.cooldown_categories.get((account.name, endpoint)) == 'rate_limit' for account in self.accounts ) def uses_explicit_pool(self): with self.lock: return bool(self.explicit_pool) def seconds_until_available(self, endpoint): with self.lock: now = time.time() deadlines = [ float(self.cooldown_until.get((account.name, endpoint), 0) or 0) for account in self.accounts if float(self.cooldown_until.get((account.name, endpoint), 0) or 0) != float('inf') and float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now ] if not deadlines: return self.cooldown_sec return max(60, math.ceil(min(deadlines) - now)) def account_available(self, account_name, endpoint): with self.lock: return account_name not in self.invalid_accounts and float( self.cooldown_until.get((account_name, endpoint), 0) or 0 ) <= time.time() def hub_token_lock(self, account_name): with self.lock: lock = self.hub_token_locks.get(account_name) if lock is None: lock = threading.Lock() self.hub_token_locks[account_name] = lock return lock def restore_endpoint_cooldowns(self, endpoint_status): if not isinstance(endpoint_status, dict): return with self.lock: account_names = {account.name for account in self.accounts} now = time.time() for endpoint, accounts in endpoint_status.items(): if endpoint not in {'hub_search', 'hub_tags', 'registry'} or not isinstance(accounts, dict): continue for account_name, status in accounts.items(): if account_name not in account_names or not isinstance(status, dict): continue disabled_until = status.get('disabled_until') if disabled_until == 'manual': retry_at = float('inf') else: try: retry_at = datetime.fromisoformat( str(disabled_until).replace('Z', '+00:00') ).timestamp() except (TypeError, ValueError, OverflowError): continue if retry_at <= now: continue key = (account_name, endpoint) category = str(status.get('disabled_reason') or 'rate_limit') if category == 'auth_invalid': self.invalid_accounts.add(account_name) if retry_at > float(self.cooldown_until.get(key, 0) or 0): self.cooldown_until[key] = retry_at self.cooldown_categories[key] = category def cached_hub_token(self, account_name): with self.lock: token, expires_at = self.hub_tokens.get(account_name, ('', 0)) if token and float(expires_at or 0) > time.time() + 30: return token self.hub_tokens.pop(account_name, None) return '' def cache_hub_token(self, account_name, token, expires_in): with self.lock: lifetime = max(60, min(600, int(expires_in or 600))) self.hub_tokens[account_name] = (token, time.time() + lifetime) def invalidate_hub_token(self, account_name): with self.lock: self.hub_tokens.pop(account_name, None) def drain_status_events(self): with self.lock: events = list(self.status_events.values()) self.status_events = {} return events def get_next_config(self): """Get the next Docker config directory in rotation""" with self.lock: if not self.accounts: return None for offset in range(len(self.accounts)): position = (self.current_index + offset) % len(self.accounts) account = self.accounts[position] if account.name in self.invalid_accounts: continue self.current_index = (position + 1) % len(self.accounts) return account.config_dir return None def _cleanup_locked(self): for directory in self.token_dirs: try: cleanup_command_work_dir(directory) except Exception as exc: logger.error('Error cleaning Docker config: %s', str(exc)) self.tokens = [] self.token_dirs = [] self.accounts = [] def cleanup(self): """Clean up temporary Docker config directories""" with self.lock: self._cleanup_locked() self.config_key = None self.cooldown_until = {} self.cooldown_categories = {} self.hub_tokens = {} self.hub_token_locks = {} self.invalid_accounts = set() self.status_events = {} self.explicit_pool = False docker_token_manager = DockerTokenManager() def configure_docker_tokens(tokens_str=None, username=None): """Reconfigure Docker auth tokens after UI or CLI input.""" require_scanner_runtime_initialized() docker_token_manager.setup_tokens(tokens_str, username) def configure_docker_accounts(entries, cooldown_sec=1800): require_scanner_runtime_initialized() docker_token_manager.setup_accounts( entries, cooldown_sec=cooldown_sec, explicit_pool=True, ) def configure_docker_discovery_tokens(tokens_str=None, username=None): docker_token_manager.setup_tokens( tokens_str, username, create_config_dirs=False, ) def configure_docker_discovery_accounts(entries, cooldown_sec=1800): docker_token_manager.setup_accounts( entries, cooldown_sec=cooldown_sec, explicit_pool=True, create_config_dirs=False, ) def restore_docker_endpoint_cooldowns(endpoint_status): docker_token_manager.restore_endpoint_cooldowns(endpoint_status) def drain_docker_auth_events(): return docker_token_manager.drain_status_events() # =================== # API FETCH FUNCTIONS # =================== def github_repo_to_target(item): return { 'url': item.get('clone_url', ''), 'name': item.get('full_name', ''), 'created_at': item.get('created_at', ''), 'updated_at': item.get('updated_at', ''), 'pushed_at': item.get('pushed_at', ''), } def github_headers(token=None): headers = {'Accept': 'application/vnd.github+json', 'User-Agent': 'GitSecretsScanner/2.0'} if token: headers['Authorization'] = f'Bearer {token}' return headers def gitlab_headers(token=None): headers = {'User-Agent': 'GitSecretsScanner/2.0'} if token: headers['PRIVATE-TOKEN'] = token return headers def parse_github_repo_target(target): if isinstance(target, dict): repo = target.get('repo') or target.get('full_name') or '' if not repo and target.get('repo_url'): candidate = normalize_git_repo_candidate(target.get('repo_url')) if candidate and candidate.get('provider') == 'github': repo = candidate.get('repo_path') or '' url = target.get('url') or (f'https://github.com/{repo}' if repo else '') return repo.strip('/'), url text = str(target or '').strip() if text.startswith('{'): try: return parse_github_repo_target(json.loads(text)) except Exception: pass lowered = text.lower().rstrip('/') if lowered.endswith('.git'): lowered = lowered[:-4] candidate = normalize_git_repo_candidate(text) if candidate and candidate.get('provider') == 'github': repo = candidate.get('repo_path') or '' return repo, f'https://github.com/{repo}' match = re.search(r'github\.com[:/]([^/\s]+/[^/\s]+)', lowered, re.IGNORECASE) if match: repo = match.group(1).strip('/') return repo, f'https://github.com/{repo}' if re.match(r'^[^/\s]+/[^/\s]+$', text): repo = text.strip('/') return repo, f'https://github.com/{repo}' return '', text def parse_gitlab_project_target(target): if isinstance(target, dict): project = target.get('project') or target.get('path_with_namespace') or target.get('repo') or '' if not project and target.get('repo_url'): candidate = normalize_git_repo_candidate(target.get('repo_url')) if candidate and candidate.get('provider') == 'gitlab': project = candidate.get('repo_path') or '' url = target.get('url') or (f'https://gitlab.com/{project}' if project else '') return project.strip('/'), url text = str(target or '').strip() if text.startswith('{'): try: return parse_gitlab_project_target(json.loads(text)) except Exception: pass lowered = text.lower().rstrip('/') if lowered.endswith('.git'): lowered = lowered[:-4] candidate = normalize_git_repo_candidate(text) if candidate and candidate.get('provider') == 'gitlab': project = candidate.get('repo_path') or '' return project, f'https://gitlab.com/{project}' match = re.search(r'gitlab\.com[:/](.+)$', lowered, re.IGNORECASE) if match: project = match.group(1).strip('/') return project, f'https://gitlab.com/{project}' if '/' in text and '://' not in text: project = text.strip('/') return project, f'https://gitlab.com/{project}' return '', text def github_rate_limit_reset(response): if not response: return None reset = response.headers.get('X-RateLimit-Reset') if not reset: return None try: return datetime.fromtimestamp(int(reset), timezone.utc).isoformat(timespec='seconds') except (TypeError, ValueError): return None def page_is_known(targets, known_targets=None, normalize_target=None, known_target_lookup=None): if not normalize_target or not targets: return False offered = [target for target in targets if target] normalized = [normalize_target(target) for target in offered] if known_target_lookup: try: known_targets = set(known_target_lookup(offered) or ()) except Exception as exc: logger.warning(f'Known-target page lookup failed open: {str(exc)[:300]}') return False if not known_targets: return False return bool(normalized) and all(target in known_targets for target in normalized) POSTMAN_COLLECTION_SUFFIX = 'postman_collection.json' POSTMAN_ENVIRONMENT_SUFFIX = 'postman_environment.json' API_ARTIFACT_PATTERNS = { 'collection': ('postman_collection.json',), 'environment': ('postman_environment.json',), 'postman': ('postman.json', '.postman.json'), 'insomnia': ('insomnia.json', '.insomnia.json'), 'bruno': ('.bru', 'bruno.json'), 'thunder_collection': ('thunder-collection.json',), 'thunder_environment': ('thunder-environment.json',), 'hoppscotch': ('hoppscotch.json',), 'generic': ('collection.json', 'environment.json'), } API_ARTIFACT_SEARCHES = { 'collection': ('filename:postman_collection.json {query}',), 'environment': ('filename:postman_environment.json {query}',), 'postman': ('filename:postman.json {query}', 'filename:.postman.json {query}'), 'insomnia': ('filename:insomnia.json {query}', 'filename:.insomnia.json {query}'), 'bruno': ('extension:bru {query}', 'filename:bruno.json {query}'), 'thunder_collection': ('filename:thunder-collection.json {query}', 'filename:thunder-collection_ {query}'), 'thunder_environment': ('filename:thunder-environment.json {query}', 'filename:thunder-environment_ {query}'), 'hoppscotch': ('filename:hoppscotch.json {query}',), 'generic': ('filename:collection.json postman {query}', 'filename:environment.json postman {query}'), 'signature': ( '"currentValue" "api_key" {query}', '"pm.collectionVariables" {query}', '"pm.environment.set" {query}', '"openai.azure.com" "api-key" {query}', '"services.ai.azure.com" "api-key" {query}', '"generativelanguage.googleapis.com" "key" {query}', ), } POSTMAN_PLACEHOLDER_RE = re.compile( r'^(?:\{\{[^}]+\}\}|<[^>]+>|your[_ -]?[a-z0-9_-]+|replace[_ -]?me|change[_ -]?me|changeme|example|dummy|test)$', re.IGNORECASE, ) POSTMAN_JSON_HARD_MAX_INPUT_BYTES = 16 * 1024 * 1024 POSTMAN_HARVEST_MAX_WARNINGS = 5 POSTMAN_HARVEST_WARNING_MAX_CHARS = 400 class PostmanCacheValidationError(ValueError): pass class PostmanCacheTooLarge(PostmanCacheValidationError): pass class PostmanCacheCapacityError(PostmanCacheValidationError): pass class _PostmanHarvestDeadlineReached(RuntimeError): pass def _postman_discovery_limits(max_artifacts=None, max_page_artifacts=None, max_total_bytes=None, max_elapsed_sec=None): artifacts = int( getattr(scan_config, 'postman_discovery_max_artifacts_per_cycle', 1000) if max_artifacts is None else max_artifacts ) page_artifacts = int( getattr(scan_config, 'postman_discovery_max_artifacts_per_page', 100) if max_page_artifacts is None else max_page_artifacts ) total_bytes = int( getattr(scan_config, 'postman_discovery_max_bytes_per_cycle', 1024 * 1024 * 1024) if max_total_bytes is None else max_total_bytes ) elapsed_sec = float( getattr(scan_config, 'postman_discovery_max_elapsed_sec', 300.0) if max_elapsed_sec is None else max_elapsed_sec ) if artifacts <= 0 or page_artifacts <= 0 or total_bytes <= 0 or elapsed_sec <= 0 or not math.isfinite(elapsed_sec): raise ValueError('Postman discovery limits must be finite and positive') return artifacts, page_artifacts, total_bytes, elapsed_sec class _PostmanDiscoveryBudget: MAX_WARNINGS = 5 def __init__(self, source, max_artifacts=None, max_page_artifacts=None, max_total_bytes=None, max_elapsed_sec=None): ( self.max_artifacts, self.max_page_artifacts, self.max_total_bytes, elapsed_sec, ) = _postman_discovery_limits(max_artifacts, max_page_artifacts, max_total_bytes, max_elapsed_sec) self.source = str(source or 'non-package') self.elapsed_sec = elapsed_sec self.deadline = time.monotonic() + elapsed_sec self.artifacts = 0 self.total_bytes = 0 self.stop_reason = '' self._warnings = set() def warn(self, detail): detail = str(detail or '')[:400] if detail in self._warnings or len(self._warnings) >= self.MAX_WARNINGS: return self._warnings.add(detail) logger.warning('Optional Postman %s discovery bounded: %s', self.source, detail) def stop(self, detail): if not self.stop_reason: self.stop_reason = str(detail or 'bounded discovery limit reached')[:400] self.warn(self.stop_reason) return False def check_deadline(self, context='during discovery'): if self.stop_reason: return False if time.monotonic() >= self.deadline: return self.stop(f'elapsed deadline of {self.elapsed_sec:g}s reached {context}') return True def request_timeout(self, configured_timeout): if not self.check_deadline('before a network request'): raise _PostmanHarvestDeadlineReached(self.stop_reason) remaining = self.deadline - time.monotonic() if remaining <= 0: self.stop(f'elapsed deadline of {self.elapsed_sec:g}s reached before a network request') raise _PostmanHarvestDeadlineReached(self.stop_reason) return max(0.01, min(float(configured_timeout or remaining), remaining)) def admit(self, page_artifacts): if not self.check_deadline(): return False, 'cycle' if page_artifacts >= self.max_page_artifacts: self.warn(f'per-page artifact limit of {self.max_page_artifacts} reached') return False, 'page' if self.artifacts >= self.max_artifacts: self.stop(f'per-cycle artifact limit of {self.max_artifacts} reached') return False, 'cycle' self.artifacts += 1 return True, '' def account_bytes(self, size): size = max(0, int(size or 0)) if self.total_bytes + size > self.max_total_bytes: return self.stop( f'aggregate byte limit of {self.max_total_bytes} reached after {self.total_bytes} byte(s)' ) self.total_bytes += size return True def exhausted(self): if self.stop_reason: return True if self.artifacts >= self.max_artifacts: return not self.stop(f'per-cycle artifact limit of {self.max_artifacts} reached') if self.total_bytes >= self.max_total_bytes: return not self.stop(f'aggregate byte limit of {self.max_total_bytes} reached') return not self.check_deadline() def _publish_postman_discovery_batch(prepared, cache_dir, budget): if not prepared: return [], False published, deadline_reached = _publish_postman_cache_entries( [item['entry'] for item in prepared], cache_dir, deadline=budget.deadline, ) if deadline_reached: budget.stop(f'elapsed deadline of {budget.elapsed_sec:g}s reached while scanning or publishing the cache') return list(zip(prepared, published)), deadline_reached def utc_now_for_postman(): return datetime.now(timezone.utc) def parse_postman_time(value): if not value: return None try: parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00')) if parsed.tzinfo is None: parsed = parsed.replace(tzinfo=timezone.utc) return parsed.astimezone(timezone.utc) except ValueError: return None def postman_kind_for_path(path): name = os.path.basename(str(path or '')).lower() normalized = str(path or '').replace('\\', '/').lower() for kind, suffixes in API_ARTIFACT_PATTERNS.items(): if any(name.endswith(suffix) or normalized.endswith('/bruno/' + suffix) for suffix in suffixes): return kind return 'artifact' def normalize_postman_search_kinds(value): if not value: return ['collection', 'environment'] if isinstance(value, str): parts = [item.strip().lower() for item in value.split(',')] else: parts = [str(item).strip().lower() for item in value] aliases = { 'collections': 'collection', 'env': 'environment', 'environments': 'environment', 'api_artifacts': 'all', 'api-artifacts': 'all', 'thunder': 'thunder_collection', } kinds = [] for item in parts: item = aliases.get(item, item) if item == 'all': for kind in API_ARTIFACT_SEARCHES: if kind not in kinds: kinds.append(kind) continue if item in API_ARTIFACT_SEARCHES and item not in kinds: kinds.append(item) return kinds or ['collection', 'environment'] def parse_postman_target(target): if isinstance(target, dict): return dict(target) text = str(target or '').strip() if not text: return {} if text.startswith('{'): return json.loads(text) if text.lower().startswith('file:'): path = text[5:] return {'source': 'local_file', 'kind': postman_kind_for_path(path), 'local_path': path, 'path': path} if os.path.exists(text): return {'source': 'local_file', 'kind': postman_kind_for_path(text), 'local_path': text, 'path': text} return {'source': 'url', 'kind': postman_kind_for_path(text), 'url': text} def postman_target_identity(target): return semantic_postman_target_identity(target) def get_postman_cache_dir(cache_dir=None): path = cache_dir or getattr(scan_config, 'postman_cache_dir', None) if not path: raise PostmanCacheValidationError('configured Postman cache directory is required') runtime_dir = getattr(scan_config, 'runtime_dir', None) if not runtime_dir: raise PostmanCacheValidationError('configured runtime directory is required for Postman cache containment') runtime_root = canonical_path(runtime_dir) cache_root = canonical_path(path) try: contained = os.path.commonpath((runtime_root, cache_root)) == runtime_root and cache_root != runtime_root except ValueError: contained = False if not contained: raise PostmanCacheValidationError('Postman cache must remain under the configured runtime directory') return require_private_directory(cache_root, create=True) def sha256_bytes(value): return hashlib.sha256(value or b'').hexdigest() def postman_cache_file_path(digest, cache_dir=None): root = get_postman_cache_dir(cache_dir) prefix = str(digest or '')[:2] or 'xx' directory = os.path.join(root, prefix) ensure_private_directory(directory, reject_reparse=True) return os.path.join(directory, f'{digest}.json') def postman_cache_usage(root, deadline=None): usage = {'items': 0, 'files': 0, 'bytes': 0} stack = [require_private_directory(root, create=False)] while stack: if deadline is not None and time.monotonic() >= deadline: raise _PostmanHarvestDeadlineReached() current = stack.pop() with os.scandir(current) as entries: for entry in entries: if deadline is not None and time.monotonic() >= deadline: raise _PostmanHarvestDeadlineReached() if entry.is_symlink() or is_reparse_point(entry.path): raise PostmanCacheCapacityError(f'Postman cache contains a linked entry: {entry.path}') if entry.is_dir(follow_symlinks=False): if not private_directory_ready(entry.path): raise PostmanCacheCapacityError(f'Postman cache directory is not private: {entry.path}') stack.append(entry.path) continue if not entry.is_file(follow_symlinks=False): raise PostmanCacheCapacityError(f'Postman cache contains an unsupported entry: {entry.path}') if entry.name == '.postman-cache.lock': continue details = entry.stat(follow_symlinks=False) usage['files'] += 1 usage['bytes'] += int(details.st_size) if not entry.name.endswith('.meta.json'): usage['items'] += 1 return usage def _postman_cache_limits(max_items=None, max_bytes=None, min_free_bytes=None): values = { 'items': int(getattr(scan_config, 'postman_cache_max_items', 0) if max_items is None else max_items), 'bytes': int(getattr(scan_config, 'postman_cache_max_bytes', 0) if max_bytes is None else max_bytes), 'min_free_bytes': int(getattr(scan_config, 'postman_cache_min_free_bytes', 0) if min_free_bytes is None else min_free_bytes), } if values['items'] <= 0 or values['bytes'] <= 0 or values['min_free_bytes'] < 0: raise PostmanCacheCapacityError('Postman cache aggregate limits must be finite positive values') return values def _write_private_cache_file(path, payload): if os.path.lexists(path): raise FileExistsError(path) temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.tmp' flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0) descriptor = os.open(temporary, flags, 0o600) try: os.close(descriptor) descriptor = None harden_private_file(temporary) with open(temporary, 'wb') as handle: handle.write(payload) handle.flush() os.fsync(handle.fileno()) if not private_file_ready(temporary): raise PostmanCacheCapacityError(f'Postman cache temporary file is not private: {temporary}') if os.path.lexists(path): raise FileExistsError(path) durable_replace(temporary, path) if not private_file_ready(path): raise PostmanCacheCapacityError(f'Postman cache publication is not private: {path}') finally: if descriptor is not None: os.close(descriptor) try: if os.path.exists(temporary): os.remove(temporary) except OSError: pass def _bounded_json_bytes(value, max_bytes, *, indent=None, sort_keys=False, newline=False): output = bytearray() encoder = json.JSONEncoder(ensure_ascii=False, indent=indent, sort_keys=sort_keys) for chunk in encoder.iterencode(value): encoded = chunk.encode('utf-8') if len(output) + len(encoded) + (1 if newline else 0) > max_bytes: raise ValueError('serialized JSON exceeds its bounded byte limit') output.extend(encoded) if newline: output.extend(b'\n') return bytes(output) def validate_postman_cache_artifact(target_data, max_artifact_size_mb=20, expected_size=None): _raise_if_scan_slot_fatal() if not isinstance(target_data, dict): raise PostmanCacheValidationError('Postman cache metadata must be an object') offered_path = target_data.get('cache_path') or target_data.get('local_path') if not offered_path: raise PostmanCacheValidationError('Postman target missing cached artifact') cache_root = canonical_path(get_postman_cache_dir()) reject_reparse_components(offered_path) cache_path = canonical_path(offered_path) try: contained = os.path.commonpath((cache_root, cache_path)) == cache_root and cache_path != cache_root except ValueError: contained = False if not contained: raise PostmanCacheValidationError('Postman cache path escapes the configured cache root') reject_reparse_components(cache_path) if is_reparse_point(cache_path) or not private_file_ready(cache_path): raise PostmanCacheValidationError('Postman cached artifact is absent, linked, or not private') declared_hash = str(target_data.get('sha256') or '').strip().lower() if not re.fullmatch(r'[0-9a-f]{64}', declared_hash): raise PostmanCacheValidationError('Postman target has no valid declared SHA-256') if postman_target_identity(target_data) != f'postman:sha256:{declared_hash}': raise PostmanCacheValidationError('Postman semantic identity does not match its declared SHA-256') flags = os.O_RDONLY | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0) descriptor = os.open(cache_path, flags) try: opened = os.fstat(descriptor) if not stat.S_ISREG(opened.st_mode): raise PostmanCacheValidationError('Postman cached artifact is not a regular file') size = int(opened.st_size) if size <= 0: raise PostmanCacheValidationError('Postman cached artifact is empty') declared_sizes = [expected_size] declared_sizes.extend(target_data.get(name) for name in ('size', 'bytes')) for declared_size in declared_sizes: if declared_size in (None, ''): continue try: parsed_size = int(declared_size) except (TypeError, ValueError) as exc: raise PostmanCacheValidationError('Postman cached artifact has invalid declared size') from exc if parsed_size != size: raise PostmanCacheValidationError('Postman cached artifact size mismatch') max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024 if max_bytes <= 0: raise PostmanCacheValidationError('Postman artifact limit must be finite and positive') if size > max_bytes: raise PostmanCacheTooLarge(f'artifact exceeds {max_artifact_size_mb} MB') digest = hashlib.sha256() while True: _raise_if_scan_slot_fatal() block = os.read(descriptor, 1024 * 1024) if not block: break digest.update(block) if digest.hexdigest() != declared_hash: raise PostmanCacheValidationError('Postman cached artifact SHA-256 mismatch') _raise_if_scan_slot_fatal() current = os.stat(cache_path, follow_symlinks=False) opened_identity = (opened.st_dev, opened.st_ino, opened.st_size, getattr(opened, 'st_mtime_ns', None)) current_identity = (current.st_dev, current.st_ino, current.st_size, getattr(current, 'st_mtime_ns', None)) if opened_identity != current_identity: raise PostmanCacheValidationError('Postman cached artifact changed during validation') finally: os.close(descriptor) return cache_path, size def _prepare_postman_cache_entry(content, kind, origin, max_artifact_size_mb): max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024 if max_bytes <= 0: raise ValueError('Postman artifact limit must be finite and positive') if isinstance(content, str) and len(content) > max_bytes: raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB') content_bytes = content.encode('utf-8', errors='replace') if isinstance(content, str) else bytes(content or b'') if len(content_bytes) > max_bytes: raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB') if not content_bytes: raise ValueError('Postman artifact is empty') stripped = content_bytes.lstrip() if stripped.startswith((b'{', b'[')): if len(content_bytes) > POSTMAN_JSON_HARD_MAX_INPUT_BYTES: raise ValueError('Postman JSON artifact exceeds the hard pre-parse byte limit') try: json.loads(content_bytes.decode('utf-8-sig')) except (UnicodeDecodeError, ValueError, RecursionError) as exc: raise ValueError('Postman JSON artifact is invalid') from exc raw_content = content_bytes digest = sha256_bytes(raw_content) metadata = { 'sha256': digest, 'kind': kind, 'bytes': len(raw_content), 'origin': origin or {}, 'cached_at': utc_now_for_postman().isoformat(timespec='seconds'), } try: metadata_bytes = _bounded_json_bytes(metadata, max_bytes, indent=2, sort_keys=True, newline=True) except ValueError as exc: raise PostmanCacheCapacityError('Postman cache metadata exceeds the per-artifact byte limit') from exc return { 'content': raw_content, 'digest': digest, 'metadata': metadata_bytes, 'size': len(raw_content), } def _publish_postman_cache_entries( entries, cache_dir=None, cache_max_items=None, cache_max_bytes=None, cache_min_free_bytes=None, deadline=None, ): if not entries: return [], False root = get_postman_cache_dir(cache_dir) limits = _postman_cache_limits(cache_max_items, cache_max_bytes, cache_min_free_bytes) lock_path = os.path.join(root, '.postman-cache.lock') lock_timeout = max(1.0, float(getattr(scan_config, 'postman_cache_lock_timeout_sec', 30))) if deadline is not None: remaining = deadline - time.monotonic() if remaining <= 0: return [], True lock_timeout = min(lock_timeout, max(0.01, remaining)) try: lock = acquire_file_lock( lock_path, timeout_sec=lock_timeout, ) except TimeoutError: if deadline is not None and time.monotonic() >= deadline: return [], True raise try: if deadline is not None and time.monotonic() >= deadline: return [], True try: usage = ( postman_cache_usage(root) if deadline is None else postman_cache_usage(root, deadline=deadline) ) except _PostmanHarvestDeadlineReached: return [], True plans = [] reserved_bytes = 0 free_bytes = None seen = set() for entry in entries: if deadline is not None and time.monotonic() >= deadline: return [], True digest = entry['digest'] if digest in seen: continue seen.add(digest) if sha256_bytes(entry['content']) != digest: raise PostmanCacheValidationError('Postman cache entry SHA-256 changed before publication') prefix_dir = os.path.join(root, digest[:2]) path = os.path.join(prefix_dir, f'{digest}.json') meta_path = f'{path}.meta.json' if os.path.lexists(prefix_dir): reject_reparse_components(prefix_dir) if not os.path.isdir(prefix_dir) or not private_directory_ready(prefix_dir): raise PostmanCacheCapacityError(f'Postman cache prefix directory is not private: {prefix_dir}') artifact_exists = os.path.lexists(path) metadata_exists = os.path.lexists(meta_path) for existing in (path, meta_path): if os.path.lexists(existing): reject_reparse_components(existing) if not private_file_ready(existing): raise PostmanCacheCapacityError(f'Postman cache file is not private: {existing}') added_items = 0 if artifact_exists else 1 added_files = int(not artifact_exists) + int(not metadata_exists) added_bytes = ( (0 if artifact_exists else entry['size']) + (0 if metadata_exists else len(entry['metadata'])) ) if usage['items'] + added_items > limits['items']: raise PostmanCacheCapacityError( f'Postman cache item capacity reached ({usage["items"]}/{limits["items"]})' ) if usage['bytes'] + added_bytes > limits['bytes']: raise PostmanCacheCapacityError( f'Postman cache byte capacity reached ({usage["bytes"]}/{limits["bytes"]})' ) if added_bytes and free_bytes is None: free_bytes = int(shutil.disk_usage(root).free) if added_bytes and free_bytes - reserved_bytes - added_bytes < limits['min_free_bytes']: raise PostmanCacheCapacityError( f'Postman cache free-space reserve would be crossed ({free_bytes} bytes free)' ) usage['items'] += added_items usage['files'] += added_files usage['bytes'] += added_bytes reserved_bytes += added_bytes plans.append((entry, prefix_dir, path, meta_path, artifact_exists, metadata_exists)) published = [] for entry, prefix_dir, path, meta_path, artifact_exists, metadata_exists in plans: if deadline is not None and time.monotonic() >= deadline: return published, True if not os.path.isdir(prefix_dir): ensure_private_directory(prefix_dir, reject_reparse=True) elif not private_directory_ready(prefix_dir): raise PostmanCacheCapacityError(f'Postman cache prefix directory is not private: {prefix_dir}') if not artifact_exists: _write_private_cache_file(path, entry['content']) if not metadata_exists: _write_private_cache_file(meta_path, entry['metadata']) published.append((path, entry['digest'], entry['size'])) return published, False finally: release_file_lock(lock, lock_path) def write_postman_cache( content, kind='artifact', origin=None, cache_dir=None, max_artifact_size_mb=20, cache_max_items=None, cache_max_bytes=None, cache_min_free_bytes=None, ): entry = _prepare_postman_cache_entry(content, kind, origin, max_artifact_size_mb) published, deadline_reached = _publish_postman_cache_entries( [entry], cache_dir, cache_max_items, cache_max_bytes, cache_min_free_bytes, ) if deadline_reached or not published: raise PostmanCacheCapacityError('Postman cache publication did not complete') return published[0] def postman_target_from_cached_artifact(source, kind, cache_path, digest, origin=None, **extra): payload = {'source': source, 'kind': kind, 'cache_path': cache_path, 'sha256': digest, 'origin': origin or {}} payload.update({key: value for key, value in extra.items() if value not in (None, '')}) return json.dumps(payload, separators=(',', ':'), ensure_ascii=False, sort_keys=True) def cache_package_postman_artifact(file_path, package_source, package, relative_path, cache_dir=None, max_artifact_size_mb=20): kind = postman_kind_for_path(file_path) max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024 with open(file_path, 'rb') as f: content = f.read(max_bytes + 1) if max_bytes <= 0 or len(content) > max_bytes: raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB') origin = { 'package_source': package_source, 'package_name': package.get('name'), 'package_version': package.get('version'), 'package_artifact': package.get('artifact') or package.get('tarball'), 'path': relative_path, } cache_path, digest, size = write_postman_cache(content, kind, origin, cache_dir, max_artifact_size_mb) return postman_target_from_cached_artifact( f'{package_source}_package', kind, cache_path, digest, origin, path=relative_path, package_name=package.get('name'), package_version=package.get('version'), package_artifact=package.get('artifact') or package.get('tarball'), size=size, ) def _postman_package_harvest_limits(max_artifacts=None, max_total_bytes=None, max_elapsed_sec=None): artifacts = int( getattr(scan_config, 'postman_package_harvest_max_artifacts', 100) if max_artifacts is None else max_artifacts ) total_bytes = int( getattr(scan_config, 'postman_package_harvest_max_bytes', 128 * 1024 * 1024) if max_total_bytes is None else max_total_bytes ) elapsed_sec = float( getattr(scan_config, 'postman_package_harvest_max_elapsed_sec', 30.0) if max_elapsed_sec is None else max_elapsed_sec ) if artifacts <= 0 or total_bytes <= 0 or elapsed_sec <= 0 or not math.isfinite(elapsed_sec): raise ValueError('Postman package harvest limits must be finite and positive') return artifacts, total_bytes, elapsed_sec def _add_postman_harvest_warning(warnings, detail, force=False): message = f'Optional Postman package harvesting degraded: {detail}'[:POSTMAN_HARVEST_WARNING_MAX_CHARS] if message in warnings: return if len(warnings) < POSTMAN_HARVEST_MAX_WARNINGS: warnings.append(message) logger.warning(message) elif force: warnings[-1] = message logger.warning(message) def _attach_postman_harvest_warnings(results, warnings): if not warnings: return results merged = list(results.get('warnings') or []) for warning in warnings: if warning not in merged: merged.append(warning) results['warnings'] = merged results['warning_classes'] = sorted(set(list(results.get('warning_classes') or []) + ['postman_package_harvest'])) results['degraded'] = True return results def _bounded_postman_walk(root_dir, deadline): stack = [root_dir] while stack: _raise_if_scan_slot_fatal() if time.monotonic() >= deadline: yield None, (), () return directory = stack.pop() try: entries = os.scandir(directory) except OSError: continue try: for entry in entries: _raise_if_scan_slot_fatal() if time.monotonic() >= deadline: yield None, (), () return try: if entry.is_dir(follow_symlinks=False): if not entry.is_symlink() and not is_reparse_point(entry.path): stack.append(entry.path) continue except OSError: continue yield directory, (), (entry.name,) finally: entries.close() def find_postman_artifacts( root_dir, package_source, package, cache_dir=None, max_artifact_size_mb=20, warnings=None, max_artifacts=None, max_total_bytes=None, max_elapsed_sec=None, ): targets = [] if not root_dir or not os.path.isdir(root_dir): return targets warning_sink = warnings if warnings is not None else [] artifact_limit, total_byte_limit, elapsed_limit = _postman_package_harvest_limits( max_artifacts, max_total_bytes, max_elapsed_sec, ) max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024 if max_bytes <= 0: raise ValueError('Postman artifact limit must be finite and positive') deadline = time.monotonic() + elapsed_limit suffixes = tuple(suffix for values in API_ARTIFACT_PATTERNS.values() for suffix in values) prepared = [] seen = set() examined = 0 examined_bytes = 0 stopped = False walk = _bounded_postman_walk(root_dir, deadline) for directory, _, files in walk: _raise_if_scan_slot_fatal() if directory is None or time.monotonic() >= deadline: _add_postman_harvest_warning( warning_sink, f'elapsed deadline of {elapsed_limit:g}s reached after examining {examined} artifact(s)', force=True, ) stopped = True break for name in files: _raise_if_scan_slot_fatal() if time.monotonic() >= deadline: _add_postman_harvest_warning( warning_sink, f'elapsed deadline of {elapsed_limit:g}s reached after examining {examined} artifact(s)', force=True, ) stopped = True break lower_name = name.lower() if not lower_name.endswith(suffixes): continue if examined >= artifact_limit: _add_postman_harvest_warning( warning_sink, f'artifact limit of {artifact_limit} reached; remaining matching files were not examined', force=True, ) stopped = True break examined += 1 path = os.path.join(directory, name) try: details = os.stat(path, follow_symlinks=False) if not stat.S_ISREG(details.st_mode) or os.path.islink(path) or is_reparse_point(path): raise ValueError('artifact is not a regular unlinked file') size = int(details.st_size) relative_path = os.path.relpath(path, root_dir).replace('\\', '/') if size > max_bytes: _add_postman_harvest_warning( warning_sink, f'skipped {relative_path[:180]} because it exceeds the {max_artifact_size_mb} MB artifact limit', ) continue if examined_bytes + size > total_byte_limit: _add_postman_harvest_warning( warning_sink, f'aggregate byte limit of {total_byte_limit} reached after {examined_bytes} byte(s)', force=True, ) stopped = True break examined_bytes += size flags = os.O_RDONLY | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0) descriptor = os.open(path, flags) try: opened = os.fstat(descriptor) opened_identity = (opened.st_dev, opened.st_ino, opened.st_size, getattr(opened, 'st_mtime_ns', None)) expected_identity = (details.st_dev, details.st_ino, details.st_size, getattr(details, 'st_mtime_ns', None)) if not stat.S_ISREG(opened.st_mode) or opened_identity != expected_identity: raise ValueError('artifact changed before harvesting') content = bytearray() while len(content) < size: _raise_if_scan_slot_fatal() if time.monotonic() >= deadline: raise _PostmanHarvestDeadlineReached() block = os.read(descriptor, min(1024 * 1024, size - len(content))) if not block: raise ValueError('artifact changed while harvesting') content.extend(block) if os.read(descriptor, 1): raise ValueError('artifact grew while harvesting') current = os.stat(path, follow_symlinks=False) current_identity = (current.st_dev, current.st_ino, current.st_size, getattr(current, 'st_mtime_ns', None)) if current_identity != opened_identity: raise ValueError('artifact changed while harvesting') finally: os.close(descriptor) origin = { 'package_source': package_source, 'package_name': package.get('name'), 'package_version': package.get('version'), 'package_artifact': package.get('artifact') or package.get('tarball'), 'path': relative_path, } entry = _prepare_postman_cache_entry(bytes(content), postman_kind_for_path(path), origin, max_artifact_size_mb) if entry['digest'] not in seen: entry['origin'] = origin entry['kind'] = postman_kind_for_path(path) entry['relative_path'] = relative_path prepared.append(entry) seen.add(entry['digest']) except _PostmanHarvestDeadlineReached: _add_postman_harvest_warning( warning_sink, f'elapsed deadline of {elapsed_limit:g}s reached after examining {examined} artifact(s)', force=True, ) stopped = True break except PostmanCacheValidationError: raise except ScanSlotFatalError: raise except Exception as e: relative_path = os.path.relpath(path, root_dir).replace('\\', '/') _add_postman_harvest_warning( warning_sink, f'unable to harvest {relative_path[:180]}: {str(e)[:160]}', ) if stopped: break walk.close() if prepared and not (stopped and time.monotonic() >= deadline): _raise_if_scan_slot_fatal() published, deadline_reached = _publish_postman_cache_entries(prepared, cache_dir, deadline=deadline) for entry, (cache_path, digest, size) in zip(prepared, published): targets.append(postman_target_from_cached_artifact( f'{package_source}_package', entry['kind'], cache_path, digest, entry['origin'], path=entry['relative_path'], package_name=package.get('name'), package_version=package.get('version'), package_artifact=package.get('artifact') or package.get('tarball'), size=size, )) if deadline_reached: _add_postman_harvest_warning( warning_sink, f'elapsed deadline of {elapsed_limit:g}s reached while publishing {len(published)} artifact(s)', force=True, ) return targets class GitHubTokenPool: def __init__(self, token_entries=None, token=None, status=None, code_search_rpm_per_token=8, fallback_cooldown=1800): entries = [] for index, entry in enumerate(token_entries or []): if isinstance(entry, str): item = {'name': f'github_{index + 1}', 'token': entry} elif isinstance(entry, dict): item = dict(entry) item.setdefault('name', f'github_{index + 1}') else: continue if item.get('token'): entries.append(item) if token and not any(item.get('token') == token for item in entries): entries.append({'name': 'token', 'token': token}) self.entries = entries self.status = status if isinstance(status, dict) else {} self.index = 0 self.last_code_search_at = {} self.code_search_interval = 60.0 / max(1, int(code_search_rpm_per_token or 8)) self.fallback_cooldown = int(fallback_cooldown or 1800) def _status_for(self, entry): return self.status.setdefault(entry.get('name'), {}) def _disabled_until(self, entry): value = self._status_for(entry).get('disabled_until') if value == 'manual': return 'manual' return parse_postman_time(value) def _available_entries(self): now = utc_now_for_postman() available = [] for entry in self.entries: disabled_until = self._disabled_until(entry) if disabled_until == 'manual': continue if disabled_until and disabled_until > now: continue available.append(entry) return available def _mark_unavailable(self, entry, category, message, reset_at=None): item = self._status_for(entry) if category in ('auth_invalid', 'auth_forbidden'): item['disabled_until'] = 'manual' else: item['disabled_until'] = reset_at or (utc_now_for_postman() + timedelta(seconds=self.fallback_cooldown)).isoformat(timespec='seconds') item['disabled_reason'] = category item['last_error'] = str(message or '')[:500] item['last_failure_at'] = utc_now_for_postman().isoformat(timespec='seconds') item['failures'] = int(item.get('failures', 0) or 0) + 1 def _next_entry(self): available = self._available_entries() if not available: return None for _ in range(len(self.entries)): entry = self.entries[self.index % len(self.entries)] self.index = (self.index + 1) % len(self.entries) if entry in available: return entry return available[0] def _sleep_for_resource(self, entry, resource, deadline=None): if resource != 'code_search': return name = entry.get('name') last = self.last_code_search_at.get(name) now = time.monotonic() if last is not None: delay = self.code_search_interval - (now - last) if delay > 0: if deadline is not None and now + delay >= deadline: remaining = max(0.0, deadline - now) if remaining: time.sleep(remaining) raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during code-search pacing') time.sleep(delay) self.last_code_search_at[name] = time.monotonic() def _wait_until_any_available(self, deadline=None): waits = [] now = utc_now_for_postman() manual_count = 0 for entry in self.entries: disabled_until = self._disabled_until(entry) if disabled_until == 'manual': manual_count += 1 if disabled_until and disabled_until != 'manual' and disabled_until > now: waits.append((disabled_until - now).total_seconds()) if manual_count >= len(self.entries): raise RateLimitError('postman', 'All GitHub tokens are manually disabled for Postman discovery', category='auth_invalid', retryable=False, auth_related=True) wait_for = min(waits) if waits else self.fallback_cooldown wait_for = max(1, min(wait_for, self.fallback_cooldown)) if deadline is not None and time.monotonic() + wait_for >= deadline: remaining = max(0.0, deadline - time.monotonic()) logger.warning(f'All GitHub tokens unavailable for Postman discovery; deadline in {remaining:.1f}s') if remaining: time.sleep(remaining) raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached while all GitHub tokens were unavailable') logger.warning(f'All GitHub tokens unavailable for Postman discovery; sleeping {wait_for:.0f}s') time.sleep(wait_for) def _token_core_status(self, entry, timeout=10): request_headers = { 'Accept': 'application/vnd.github+json', 'User-Agent': 'GitSecretsScanner/2.0', 'Authorization': f"Bearer {entry.get('token')}", } try: response = api_request('GET', 'https://api.github.com/user', headers=request_headers, timeout=timeout, max_retries=1) except Exception: return 'unknown' try: if response.status_code == 200: return 'valid' if response.status_code == 401: return 'invalid' if response.status_code in (403, 429) and response.headers.get('X-RateLimit-Remaining') == '0': return 'rate_limited' return f'http_{response.status_code}' finally: response.close() def _mark_github_api_error(self, entry, api_error): category = getattr(api_error, 'category', 'api') if category not in ('rate_limit', 'secondary_rate_limit', 'auth_invalid', 'auth_forbidden'): return False # Code search may return auth-like 401/403 even when the token is valid for core GitHub API. # Confirm against /user before permanently disabling the token as dead. if category in ('auth_invalid', 'auth_forbidden'): core_status = self._token_core_status(entry) if core_status == 'valid': self._mark_unavailable(entry, f'{category}_resource', str(api_error), api_error.reset_at) return True if core_status in ('unknown', 'rate_limited'): self._mark_unavailable(entry, f'{category}_unconfirmed', str(api_error), api_error.reset_at) return True self._mark_unavailable(entry, category, str(api_error), api_error.reset_at) return True def request(self, method, url, params=None, headers=None, timeout=30, resource='core', deadline=None, use_proxy=None): if not self.entries: raise RateLimitError('postman', 'GitHub token is required for Postman GitHub code search', category='auth_invalid', retryable=False, auth_related=True) attempts = 0 last_error = None while attempts < max(1, len(self.entries) * 2): if deadline is not None and time.monotonic() >= deadline: raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GitHub request') entry = self._next_entry() if not entry: self._wait_until_any_available(deadline) attempts += 1 continue self._sleep_for_resource(entry, resource, deadline) request_headers = {'Accept': 'application/vnd.github+json', 'User-Agent': 'GitSecretsScanner/2.0'} request_headers.update(headers or {}) request_headers['Authorization'] = f"Bearer {entry.get('token')}" try: response = api_request( method, url, headers=request_headers, params=params, timeout=timeout, deadline=deadline, use_proxy=use_proxy, ) if response.status_code in (401, 403, 429): api_error = github_api_error(response) response.close() if self._mark_github_api_error(entry, api_error): last_error = api_error attempts += 1 continue try: response.raise_for_status() except requests.exceptions.HTTPError as e: raise github_api_error(e.response) from e return response except (requests.exceptions.RequestException, ApiRequestError) as e: last_error = e logger.warning(f'GitHub request failed for Postman discovery with token {entry.get("name")}: {str(e)[:300]}') attempts += 1 delay = min(2, attempts) if deadline is not None and time.monotonic() + delay >= deadline: raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during GitHub request retry') from e time.sleep(delay) continue if last_error: raise RateLimitError('postman', f'GitHub Postman discovery request failed after retries: {last_error}', category='network', retryable=True, auth_related=False) raise RateLimitError('postman', 'All GitHub tokens are unavailable for Postman discovery', category='rate_limit', retryable=True, auth_related=True) def latest_github_path_commit(repo, path, pool, request_timeout=20, deadline=None): response = pool.request( 'GET', f'https://api.github.com/repos/{repo}/commits', params={'path': path, 'per_page': 1}, timeout=request_timeout, resource='core', deadline=deadline, ) data = response.json() if not data: return None commit = data[0].get('commit') or {} committer = commit.get('committer') or {} author = commit.get('author') or {} return committer.get('date') or author.get('date') def github_content_bytes(item, pool, request_timeout=20, max_artifact_size_mb=20, deadline=None): response = pool.request( 'GET', item.get('url'), timeout=request_timeout, resource='core', deadline=deadline, use_proxy=False, ) data = response.json() size = int(data.get('size') or 0) max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024 if max_bytes and size > max_bytes: raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB') content = str(data.get('content') or '') if str(data.get('encoding') or '').lower() == 'base64': decoded = base64.b64decode(re.sub(r'\s+', '', content)) else: decoded = content.encode('utf-8', errors='replace') if max_bytes <= 0 or len(decoded) > max_bytes: raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB') return decoded def github_postman_item_to_target(item, kind, cache_path, digest, size=None, commit_date=None): repo = (item.get('repository') or {}).get('full_name') or '' path = item.get('path') or '' sha = item.get('sha') or digest or '' origin = { 'provider': 'github', 'repo': repo, 'path': path, 'sha': sha, 'html_url': item.get('html_url'), 'api_url': item.get('url'), 'commit_date': commit_date, } return postman_target_from_cached_artifact( 'github_code', kind, cache_path, digest, origin, repo=repo, path=path, sha=sha, html_url=item.get('html_url'), api_url=item.get('url'), commit_date=commit_date, size=size, ) def fetch_github_postman_targets(query, pages=1, per_page=100, token_entries=None, token=None, search_kinds=None, cache_dir=None, max_file_age_days=365, max_artifact_size_mb=20, request_timeout=20, code_search_rpm_per_token=8, all_tokens_cooldown=1800, auth_status=None, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, discovery_max_artifacts=None, discovery_max_artifacts_per_page=None, discovery_max_bytes=None, discovery_max_elapsed_sec=None): if not query: return [] budget = _PostmanDiscoveryBudget( 'GitHub code-search', discovery_max_artifacts, discovery_max_artifacts_per_page, discovery_max_bytes, discovery_max_elapsed_sec, ) pool = GitHubTokenPool(token_entries, token, auth_status, code_search_rpm_per_token, all_tokens_cooldown) kinds = normalize_postman_search_kinds(search_kinds) per_page = max(1, min(int(per_page or 100), 100)) max_pages = min(max(1, int(pages or 1)), max(1, (1000 + per_page - 1) // per_page)) cutoff = utc_now_for_postman() - timedelta(days=int(max_file_age_days or 0)) if int(max_file_age_days or 0) > 0 else None targets = [] seen_targets = set() seen_digests = set() stop_cycle = False for kind in kinds: if stop_cycle or not budget.check_deadline(): break templates = API_ARTIFACT_SEARCHES.get(kind) or API_ARTIFACT_SEARCHES['generic'] for template in templates: if stop_cycle or not budget.check_deadline(): break search_query = template.format(query=query).strip() logger.info(f"Fetching GitHub API artifact {kind} targets for: {search_query!r}") known_pages = 0 for page in range(1, max_pages + 1): if not budget.check_deadline(): stop_cycle = True break try: response = pool.request( 'GET', 'https://api.github.com/search/code', params={'q': search_query, 'per_page': per_page, 'page': page, 'sort': 'indexed', 'order': 'desc'}, timeout=budget.request_timeout(request_timeout), resource='code_search', deadline=budget.deadline, ) except _PostmanHarvestDeadlineReached as e: budget.stop(str(e)) stop_cycle = True break except RateLimitError as e: if getattr(e, 'category', '') == 'network': logger.warning(f'GitHub API artifact {kind} network failure on page {page}: {str(e)[:300]}') raise payload = response.json() if not isinstance(payload, dict) or 'items' not in payload or not isinstance(payload.get('items'), list): raise ApiRequestError('invalid GitHub code search payload') items = payload.get('items') or [] if not items: logger.info(f'GitHub API artifact {kind} page {page}: no results') break page_targets = [] prepared = [] page_artifacts = 0 for item in items: admitted, scope = budget.admit(page_artifacts) if not admitted: stop_cycle = scope == 'cycle' break page_artifacts += 1 repo = (item.get('repository') or {}).get('full_name') or '' path = item.get('path') or '' commit_date = None if cutoff: try: commit_date = latest_github_path_commit( repo, path, pool, budget.request_timeout(request_timeout), budget.deadline, ) parsed_commit = parse_postman_time(commit_date) if not parsed_commit or parsed_commit < cutoff: continue except _PostmanHarvestDeadlineReached as e: budget.stop(str(e)) stop_cycle = True break except RateLimitError as e: if getattr(e, 'category', '') in ('network', 'not_found'): logger.warning(f'Skipping API artifact freshness check for {repo}:{path}: {str(e)[:300]}') continue raise except Exception as e: logger.warning(f'Unable to check API artifact freshness for {repo}:{path}: {str(e)}') continue try: content = github_content_bytes( item, pool, budget.request_timeout(request_timeout), max_artifact_size_mb, budget.deadline, ) if not budget.account_bytes(len(content)): stop_cycle = True break origin = {'provider': 'github', 'repo': repo, 'path': path, 'sha': item.get('sha'), 'html_url': item.get('html_url'), 'api_url': item.get('url'), 'commit_date': commit_date} detected_kind = postman_kind_for_path(path) entry = _prepare_postman_cache_entry(content, detected_kind or kind, origin, max_artifact_size_mb) if entry['digest'] in seen_digests: continue seen_digests.add(entry['digest']) prepared.append({ 'entry': entry, 'item': item, 'kind': detected_kind or kind, 'commit_date': commit_date, }) except _PostmanHarvestDeadlineReached as e: budget.stop(str(e)) stop_cycle = True break except RateLimitError as e: if getattr(e, 'category', '') in ('network', 'not_found'): logger.warning(f'Skipping GitHub API artifact for {repo}:{path}: {str(e)[:300]}') continue raise except Exception as e: if isinstance(e, PostmanCacheCapacityError): raise logger.warning(f'Unable to cache GitHub API artifact {repo}:{path}: {str(e)}') published, deadline_reached = _publish_postman_discovery_batch(prepared, cache_dir, budget) for record, (cache_path, digest, size) in published: target = github_postman_item_to_target( record['item'], record['kind'], cache_path, digest, size, record['commit_date'], ) identity = postman_target_identity(target) if identity not in seen_targets: targets.append(target) page_targets.append(target) seen_targets.add(identity) logger.info(f'GitHub API artifact {kind} page {page}: fetched {len(items)}, queued candidates {len(page_targets)}') if stop_on_seen_pages and page >= max(1, int(min_pages_before_stop or 1)): if page_targets and page_is_known( page_targets, known_targets, normalize_target, known_target_lookup, ): known_pages += 1 logger.info(f'GitHub API artifact {kind} page {page}: all targets known ({known_pages}/{seen_page_threshold})') if known_pages >= max(1, int(seen_page_threshold or 1)): logger.info(f'Stopping API artifact {kind} pagination early after {known_pages} known page(s)') break else: known_pages = 0 if deadline_reached or stop_cycle or budget.exhausted(): stop_cycle = True break return targets def fetch_github_repo_items(query, pages, per_page=100, token=None, sort_by="updated", sort_order="desc", created_filter="any", raise_rate_limit=False, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None): """Fetch GitHub repositories with metadata for filtering.""" repos = [] headers = {'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/vnd.github.v3+json'} if token: headers['Authorization'] = f'Bearer {token}' # Build search query with filters search_query = query if query else "*" # Add created date filter if created_filter != "any": date_filters = { "today": "created:>{}".format((datetime.now() - timedelta(days=1)).strftime("%Y-%m-%d")), "week": "created:>{}".format((datetime.now() - timedelta(weeks=1)).strftime("%Y-%m-%d")), "month": "created:>{}".format((datetime.now() - timedelta(days=30)).strftime("%Y-%m-%d")), "year": "created:>{}".format((datetime.now() - timedelta(days=365)).strftime("%Y-%m-%d")) } if created_filter in date_filters: search_query += f" {date_filters[created_filter]}" logger.info(f"Fetching GitHub repositories for: '{search_query}' sorted by {sort_by} ({sort_order})...") seen_pages = 0 successful_pages = 0 for page in range(1, pages + 1): url = "https://api.github.com/search/repositories" params = { 'q': search_query, 'sort': sort_by, 'order': sort_order, 'per_page': per_page, 'page': page, } try: response = api_request('GET', url, headers=headers, params=params, timeout=30) response.raise_for_status() # Handle rate limits if response.status_code == 403 and 'X-RateLimit-Remaining' in response.headers: if int(response.headers['X-RateLimit-Remaining']) == 0: reset_time = datetime.fromtimestamp(int(response.headers['X-RateLimit-Reset'])) wait_seconds = (reset_time - datetime.now()).total_seconds() + 10 logger.warning(f"Rate limit exceeded. Resuming at {reset_time}. Waiting {wait_seconds:.0f} seconds...") time.sleep(wait_seconds) continue data = response.json() if not isinstance(data, dict) or 'items' not in data or not isinstance(data.get('items'), list): raise ValueError('invalid GitHub repository search payload') successful_pages += 1 if not data.get('items'): logger.info(f"Page {page} returned no results. Stopping.") break page_repos = [github_repo_to_target(item) for item in data['items'] if item.get('clone_url')] repos.extend(page_repos) logger.info(f"Page {page}: Fetched {len(data['items'])} repositories") if stop_on_seen_pages and page >= max(1, min_pages_before_stop): if page_is_known( [item.get('url') for item in page_repos], known_targets, normalize_target, known_target_lookup, ): seen_pages += 1 logger.info(f"Page {page}: all GitHub repositories are already queued/checked ({seen_pages}/{seen_page_threshold})") if seen_pages >= max(1, seen_page_threshold): logger.info(f"Stopping GitHub pagination early after {seen_pages} all-known page(s)") break else: seen_pages = 0 except requests.exceptions.HTTPError as e: api_error = github_api_error(e.response) if raise_rate_limit: raise api_error from e logger.error(str(api_error)) raise api_error from e except Exception as e: if isinstance(e, ApiRequestError): raise logger.error(f"Error fetching page {page}: {str(e)}") raise ApiRequestError(f'GitHub discovery payload failed: {e}') from e return repos def fetch_github_repos(query, pages, per_page=100, token=None, sort_by="updated", sort_order="desc", created_filter="any", raise_rate_limit=False, **kwargs): """Fetch GitHub repository clone URLs with pagination and authentication.""" return [ item['url'] for item in fetch_github_repo_items( query, pages, per_page, token, sort_by, sort_order, created_filter, raise_rate_limit, **kwargs ) if item.get('url') ] GHARCHIVE_REPO_TERMS = ( 'ai', 'agent', 'assistant', 'bot', 'chat', 'chatbot', 'gpt', 'llm', 'rag', 'mcp', 'model', 'inference', 'embedding', 'vector', 'semantic', 'prompt', 'workflow', 'copilot', 'codegen', 'langchain', 'llamaindex', 'litellm', 'ollama', 'vllm', 'claude', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq', 'grok', 'xai', 'openrouter', 'replicate', 'deepseek', 'zai', 'glm', 'zhipu', 'dashscope', 'bedrock', 'vertex', 'foundry', 'aiplatform', 'boto3', 'terraform', 'cloudbuild', 'service-account', 'credentials', ) GHARCHIVE_PATH_TERMS = ( '.env', 'env.', 'secret', 'secrets', 'credential', 'credentials', 'apikey', 'api_key', 'token', 'tokens', 'key', 'keys', 'config', 'settings', 'appsettings', 'docker-compose', 'compose.yaml', 'compose.yml', '.github/workflows', 'workflow', 'deploy', 'deployment', 'kubernetes', 'k8s', 'helm', 'terraform', 'tfvars', 'notebook', '.ipynb', 'postman', 'collection.json', 'environment.json', 'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq', 'xai', 'grok', 'openrouter', 'replicate', 'deepseek', 'zai', 'glm', 'dashscope', 'aws_access_key_id', 'aws_secret_access_key', 'bedrock-runtime', 'boto3', 'google_application_credentials', 'service-account', 'service_account', 'credentials.json', 'application_default_credentials', 'vertexai', 'aiplatform', 'provider.tf', 'cloudbuild.yaml', ) GHARCHIVE_FILE_FETCH_TERMS = ( '.env', 'env.', '.env.', '.env-', 'secret', 'secrets', 'credential', 'credentials', 'credentials.json', 'service-account', 'service_account', 'application_default_credentials', 'google_application_credentials', 'terraform.tfvars', '.tfvars', 'provider.tf', 'cloudbuild.yaml', 'cloudbuild.yml', '.github/workflows', 'docker-compose', 'compose.yml', 'compose.yaml', 'appsettings', '.ipynb', 'notebook', 'bedrock', 'vertex', 'aiplatform', 'boto3', 'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq', 'xai', 'grok', 'openrouter', 'replicate', 'deepseek', 'zai', 'glm', 'dashscope', ) GHARCHIVE_FILE_SKIP_SUFFIXES = ( '.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg', '.ico', '.pdf', '.zip', '.gz', '.tgz', '.tar', '.rar', '.7z', '.bin', '.safetensors', '.pt', '.pth', '.onnx', '.parquet', '.arrow', '.mp4', '.mov', '.avi', '.mp3', '.wav', '.lock', '.sum', '.min.js', '.map', '.pyc', ) GHARCHIVE_FILE_SKIP_PARTS = ( '/__pycache__/', '/node_modules/', '/vendor/', '/.git/', '/dist/', '/build/', '/target/', ) GITHUB_GIST_FILE_TERMS = ( '.env', 'env', 'secret', 'secrets', 'credential', 'credentials', 'apikey', 'api_key', 'token', 'tokens', 'key', 'keys', 'config', 'settings', 'appsettings', 'credentials.json', 'service-account', 'service_account', 'application_default_credentials', 'terraform', 'tfvars', 'provider.tf', 'docker-compose', 'compose.yml', 'compose.yaml', 'workflow', 'cloudbuild', 'ipynb', 'notebook', 'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq', 'xai', 'grok', 'openrouter', 'replicate', 'deepseek', 'zai', 'glm', 'dashscope', 'bedrock', 'vertex', 'aiplatform', 'postman', ) GHARCHIVE_MESSAGE_TERMS = ( 'api key', 'apikey', 'token', 'secret', 'credential', '.env', 'config', 'settings', 'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq', 'xai', 'grok', 'openrouter', 'replicate', 'deepseek', 'zai', 'glm', 'dashscope', 'bedrock', 'vertex', 'aws_access_key_id', 'aws_secret_access_key', 'google_application_credentials', 'service account', 'credentials.json', 'vertexai', 'aiplatform', 'bedrock-runtime', 'terraform', 'tfvars', 'cloudbuild', ) GHARCHIVE_MAX_HOURS_BACK = 168 def _contains_term(text, terms): text = str(text or '').lower() return any(term in text for term in terms) def github_archive_event_score(event): repo = event.get('repo') if isinstance(event.get('repo'), dict) else {} repo_name = str(repo.get('name') or '').lower() score = 0 if _contains_term(repo_name.replace('-', ' ').replace('_', ' '), GHARCHIVE_REPO_TERMS): score += 8 payload = event.get('payload') if isinstance(event.get('payload'), dict) else {} ref = str(payload.get('ref') or '').lower() if _contains_term(ref, GHARCHIVE_REPO_TERMS): score += 3 commits = payload.get('commits') if isinstance(payload.get('commits'), list) else [] for commit in commits[:20]: if not isinstance(commit, dict): continue message = str(commit.get('message') or '') if _contains_term(message, GHARCHIVE_MESSAGE_TERMS): score += 5 for key in ('added', 'modified', 'removed'): paths = commit.get(key) if isinstance(commit.get(key), list) else [] for path in paths[:50]: if _contains_term(path, GHARCHIVE_PATH_TERMS): score += 4 if event.get('type') == 'PushEvent': score += 1 return score def github_archive_event_branch(event): payload = event.get('payload') if isinstance(event.get('payload'), dict) else {} ref = str(payload.get('ref') or '').strip() if ref.startswith('refs/heads/'): return ref[len('refs/heads/'):], ref if payload.get('ref_type') == 'branch' and ref: return ref, f'refs/heads/{ref}' return '', ref def github_archive_target_payload(repo_name, event, score, archive_hour): payload = event.get('payload') if isinstance(event.get('payload'), dict) else {} branch, ref = github_archive_event_branch(event) data = { 'url': f'https://github.com/{repo_name}.git', 'repo': repo_name, 'source': 'gharchive', 'event_type': event.get('type') or '', 'archive_hour': archive_hour.isoformat(), 'score': int(score or 0), 'ref': ref, 'branch': branch, 'head_sha': payload.get('head') or payload.get('after') or '', } return json.dumps(data, separators=(',', ':'), ensure_ascii=False, sort_keys=True) def gharchive_changed_paths(event): payload = event.get('payload') if isinstance(event.get('payload'), dict) else {} commits = payload.get('commits') if isinstance(payload.get('commits'), list) else [] for commit in commits[:20]: if not isinstance(commit, dict): continue sha = commit.get('sha') or commit.get('id') or payload.get('head') or payload.get('after') or '' for key in ('added', 'modified'): paths = commit.get(key) if isinstance(commit.get(key), list) else [] for path in paths[:80]: path = str(path or '').strip().replace('\\', '/') if path: yield sha, path def gharchive_commit_urls(event, repo_name): payload = event.get('payload') if isinstance(event.get('payload'), dict) else {} commits = payload.get('commits') if isinstance(payload.get('commits'), list) else [] seen = set() for commit in commits[:5]: if not isinstance(commit, dict): continue sha = commit.get('sha') or commit.get('id') or '' url = commit.get('url') or (f'https://api.github.com/repos/{repo_name}/commits/{sha}' if sha else '') if sha and url and sha not in seen: seen.add(sha) yield sha, url head = payload.get('head') or payload.get('after') or '' if head and head not in seen: yield head, f'https://api.github.com/repos/{repo_name}/commits/{head}' def fetch_github_commit_files(event, repo_name, headers, request_timeout=20, deadline=None): for sha, url in gharchive_commit_urls(event, repo_name): if deadline is not None and time.monotonic() >= deadline: raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during commit lookup') timeout = request_timeout if deadline is not None: timeout = max(0.01, min(float(request_timeout or 20), deadline - time.monotonic())) try: response = api_request( 'GET', url, headers=headers, timeout=timeout, max_retries=2, retry_delay=1, deadline=deadline, ) if response.status_code == 404: continue if response.status_code in (401, 403, 429): raise github_api_error(response) response.raise_for_status() data = response.json() except (requests.exceptions.RequestException, ApiRequestError) as e: logger.warning(f'Unable to fetch commit files {repo_name}@{sha}: {str(e)[:300]}') continue files = data.get('files') if isinstance(data.get('files'), list) else [] for item in files[:100]: if not isinstance(item, dict): continue status = str(item.get('status') or '') if status not in ('added', 'modified', 'renamed'): continue path = item.get('filename') or item.get('previous_filename') or '' raw_url = item.get('raw_url') or github_raw_url(repo_name, sha, path) if path and raw_url: yield sha, str(path).replace('\\', '/'), raw_url def gharchive_path_interesting(path): lowered = str(path or '').lower() if not lowered or lowered.endswith(GHARCHIVE_FILE_SKIP_SUFFIXES): return False normalized = '/' + lowered.strip('/') if any(part in normalized for part in GHARCHIVE_FILE_SKIP_PARTS): return False return _contains_term(lowered, GHARCHIVE_FILE_FETCH_TERMS) def github_raw_url(repo_name, sha, path): if not repo_name or not sha or not path: return '' return f'https://raw.githubusercontent.com/{repo_name}/{sha}/{quote(path)}' def gharchive_file_target(source, kind, cache_path, digest, origin=None, **extra): return postman_target_from_cached_artifact(source, kind, cache_path, digest, origin, **extra) class GHArchiveBoundsError(ValueError): pass class GHArchiveCacheCapacityError(RuntimeError): pass def gharchive_item_lock_path(cache_dir, artifact_path): digest = hashlib.sha256(canonical_path(artifact_path).encode('utf-8', errors='strict')).hexdigest() return os.path.join(cache_dir, f'.item-lock-{int(digest[:8], 16) % 64:02d}.lock') def _gharchive_limits(): return { 'max_items': max(1, int(scan_config.gharchive_cache_max_items)), 'max_bytes': max(1, int(scan_config.gharchive_cache_max_bytes)), 'min_free_bytes': max(0, int(scan_config.gharchive_cache_min_free_bytes)), 'download_max_bytes': max(1, int(scan_config.gharchive_download_max_bytes)), 'decompressed_max_bytes': max(1, int(scan_config.gharchive_decompressed_max_bytes)), 'max_events': max(1, int(scan_config.gharchive_max_events)), 'max_line_bytes': max(1, int(scan_config.gharchive_max_line_bytes)), 'lock_timeout_sec': max(1, int(scan_config.gharchive_cache_lock_timeout_sec)), } def require_gharchive_cache_dir(cache_dir=None): cache_dir = canonical_path(cache_dir or scan_config.gharchive_cache_dir) runtime_dir = canonical_path(scan_config.runtime_dir) require_private_directory(runtime_dir, create=False) try: contained = os.path.commonpath((runtime_dir, cache_dir)) == runtime_dir except ValueError: contained = False if not contained or cache_dir == runtime_dir: raise RuntimeError('GHArchive cache must be a dedicated private directory under runtime_dir') return require_private_directory(cache_dir, create=True) def iter_gharchive_lines(path, decompressed_max_bytes=None, max_events=None, max_line_bytes=None, deadline=None): _raise_if_scan_slot_fatal() limits = _gharchive_limits() decompressed_max_bytes = max(1, int(decompressed_max_bytes or limits['decompressed_max_bytes'])) max_events = max(1, int(max_events or limits['max_events'])) max_line_bytes = max(1, int(max_line_bytes or limits['max_line_bytes'])) total = 0 events = 0 with gzip.open(path, 'rb') as archive: while True: _raise_if_scan_slot_fatal() if deadline is not None and time.monotonic() >= deadline: raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached while reading GHArchive data') raw_line = archive.readline(max_line_bytes + 1) if not raw_line: break if len(raw_line) > max_line_bytes: raise GHArchiveBoundsError('GHArchive line exceeds configured byte limit') total += len(raw_line) if total > decompressed_max_bytes: raise GHArchiveBoundsError('GHArchive decompressed bytes exceed configured limit') events += 1 if events > max_events: raise GHArchiveBoundsError('GHArchive event count exceeds configured limit') yield raw_line def validate_gharchive_gzip(path, decompressed_max_bytes=None, max_events=None, max_line_bytes=None, deadline=None): for _ in iter_gharchive_lines(path, decompressed_max_bytes, max_events, max_line_bytes, deadline): pass return True def gharchive_cache_usage(cache_dir): usage = {'items': 0, 'files': 0, 'bytes': 0} with os.scandir(cache_dir) as entries: for entry in entries: if entry.is_symlink() or is_reparse_point(entry.path) or not entry.is_file(follow_symlinks=False): continue size = entry.stat(follow_symlinks=False).st_size usage['files'] += 1 usage['bytes'] += max(0, int(size)) if entry.name.endswith('.json.gz'): usage['items'] += 1 return usage def _gharchive_artifacts_oldest(cache_dir, protected): candidates = [] with os.scandir(cache_dir) as entries: for entry in entries: canonical = canonical_path(entry.path) if ( canonical in protected or not entry.name.endswith('.json.gz') or entry.is_symlink() or is_reparse_point(entry.path) or not entry.is_file(follow_symlinks=False) ): continue details = entry.stat(follow_symlinks=False) candidates.append((details.st_mtime_ns, canonical)) return [path for _, path in sorted(candidates)] def _gharchive_evict_for_capacity(cache_dir, required_items=0, required_bytes=0, protected=None): limits = _gharchive_limits() protected = {canonical_path(path) for path in (protected or set())} while True: usage = gharchive_cache_usage(cache_dir) try: free_bytes = shutil.disk_usage(cache_dir).free except OSError as exc: raise GHArchiveCacheCapacityError(f'unable to inspect GHArchive cache free space: {exc}') from exc if ( usage['items'] + int(required_items) <= limits['max_items'] and usage['bytes'] + int(required_bytes) <= limits['max_bytes'] and free_bytes - int(required_bytes) >= limits['min_free_bytes'] ): return usage evicted = False for candidate in _gharchive_artifacts_oldest(cache_dir, protected): item_lock = PrivateFileLock(gharchive_item_lock_path(cache_dir, candidate)) try: item_lock.acquire() except BlockingIOError: continue try: if os.path.isfile(candidate): reject_reparse_components(candidate) os.remove(candidate) evicted = True break finally: item_lock.release() if not evicted: raise GHArchiveCacheCapacityError('GHArchive cache quota or free-space reserve cannot be satisfied without evicting an active entry') def _acquire_cached_gharchive_hour(hour, cache_dir=None, request_timeout=120, retries=4, deadline=None): cache_dir = require_gharchive_cache_dir(cache_dir) limits = _gharchive_limits() name = f'{hour:%Y-%m-%d-%H}.json.gz' path = canonical_path(os.path.join(cache_dir, name)) global_lock_path = os.path.join(cache_dir, '.cache.lock') if deadline is not None and time.monotonic() >= deadline: raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GHArchive cache access') lock_timeout = limits['lock_timeout_sec'] if deadline is not None: lock_timeout = min(lock_timeout, max(0.01, deadline - time.monotonic())) try: global_lock = acquire_file_lock(global_lock_path, timeout_sec=lock_timeout) except TimeoutError as exc: if deadline is not None and time.monotonic() >= deadline: raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached waiting for GHArchive cache lock') from exc raise item_lock = None try: item_lock_timeout = limits['lock_timeout_sec'] if deadline is not None: remaining = deadline - time.monotonic() if remaining <= 0: raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GHArchive item lock') item_lock_timeout = min(item_lock_timeout, max(0.01, remaining)) try: item_lock = acquire_file_lock( gharchive_item_lock_path(cache_dir, path), timeout_sec=item_lock_timeout, ) except TimeoutError as exc: if deadline is not None and time.monotonic() >= deadline: raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached waiting for GHArchive item lock') from exc raise for entry in os.scandir(cache_dir): if entry.name.startswith(name + '.part.') and entry.is_file(follow_symlinks=False): remove_file_quiet(entry.path) if os.path.lexists(path): reject_reparse_components(path) if not private_file_ready(path): raise RuntimeError(f'GHArchive cache file is not private: {path}') try: validate_gharchive_gzip(path, deadline=deadline) os.utime(path, None) return path, item_lock except (OSError, EOFError, gzip.BadGzipFile, GHArchiveBoundsError): remove_file_quiet(path) _gharchive_evict_for_capacity(cache_dir, required_items=1, protected={path}) url = f'https://data.gharchive.org/{name}' last_error = None attempts = max(1, int(retries or 1)) for attempt in range(attempts): _raise_if_scan_slot_fatal() if deadline is not None and time.monotonic() >= deadline: raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GHArchive download') temp_path = f'{path}.part.{os.getpid()}.{attempt}' remove_file_quiet(temp_path) response = None try: logger.info(f'Downloading GHArchive {name} attempt {attempt + 1}/{attempts}') request_deadline = None if deadline is None else max(0.01, deadline - time.monotonic()) response = _direct_request( 'GET', url, headers={'User-Agent': 'GitSecretsScanner/2.0'}, stream=True, timeout=( min(10, request_deadline) if request_deadline is not None else 10, min(max(30, int(request_timeout or 120)), request_deadline) if request_deadline is not None else max(30, int(request_timeout or 120)), ), ) if _scan_slot_fatal_event.is_set(): _raise_if_scan_slot_fatal() if response.status_code == 404: return None, item_lock response.raise_for_status() declared = response.headers.get('Content-Length') if declared and int(declared) > limits['download_max_bytes']: raise GHArchiveBoundsError('GHArchive compressed response exceeds configured byte limit') total = 0 with open(temp_path, 'xb') as output: for chunk in response.iter_content(chunk_size=1024 * 1024): _raise_if_scan_slot_fatal() if deadline is not None and time.monotonic() >= deadline: raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during GHArchive download') if not chunk: continue total += len(chunk) if total > limits['download_max_bytes']: raise GHArchiveBoundsError('GHArchive compressed response exceeds configured byte limit') _gharchive_evict_for_capacity( cache_dir, required_bytes=len(chunk), protected={path, temp_path}, ) output.write(chunk) output.flush() os.fsync(output.fileno()) _raise_if_scan_slot_fatal() harden_private_file(temp_path) validate_gharchive_gzip(temp_path, deadline=deadline) _raise_if_scan_slot_fatal() os.replace(temp_path, path) harden_private_file(path) return path, item_lock except _PostmanHarvestDeadlineReached: remove_file_quiet(temp_path) raise except ( requests.RequestException, OSError, EOFError, ValueError, gzip.BadGzipFile, GHArchiveBoundsError, GHArchiveCacheCapacityError, urllib3_exceptions.ProtocolError, urllib3_exceptions.ReadTimeoutError, ) as exc: last_error = exc remove_file_quiet(temp_path) if attempt + 1 < attempts: delay = min(30, 2 ** attempt) if deadline is not None and time.monotonic() + delay >= deadline: raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during GHArchive retry') from exc _wait_or_raise_scan_slot_fatal(delay) finally: if response is not None: response.close() raise ApiRequestError(f'GHArchive download failed after {attempts} attempt(s): {name}: {last_error}') except BaseException: if item_lock is not None: release_file_lock(item_lock, item_lock.path) raise finally: release_file_lock(global_lock, global_lock_path) def cached_gharchive_hour(hour, cache_dir=None, request_timeout=120, retries=4, deadline=None): path, item_lock = _acquire_cached_gharchive_hour(hour, cache_dir, request_timeout, retries, deadline) if item_lock is not None: release_file_lock(item_lock, item_lock.path) return path @contextmanager def cached_gharchive_hour_reader(hour, cache_dir=None, request_timeout=120, retries=4, deadline=None): path = None item_lock = None try: path, item_lock = _acquire_cached_gharchive_hour(hour, cache_dir, request_timeout, retries, deadline) yield path finally: if item_lock is not None: release_file_lock(item_lock, item_lock.path) def fetch_github_archive_repos(hours_back=6, max_repos=200, event_types=None, request_timeout=120, archive_cache_dir=None): """Fetch recently active public GitHub repositories from GHArchive hourly dumps.""" event_types = {str(item).strip() for item in (event_types or []) if str(item).strip()} if not event_types: event_types = {'PushEvent', 'CreateEvent', 'PublicEvent'} hours_back = min(GHARCHIVE_MAX_HOURS_BACK, max(1, int(hours_back or 1))) max_repos = max(1, int(max_repos or 1)) now = datetime.now(timezone.utc).replace(minute=0, second=0, microsecond=0) candidates = {} order = 0 for offset in range(1, hours_back + 1): hour = now - timedelta(hours=offset) url = f'https://data.gharchive.org/{hour:%Y-%m-%d-%H}.json.gz' logger.info(f'Fetching GHArchive hour {hour.isoformat()} from {url}') with cached_gharchive_hour_reader(hour, archive_cache_dir, request_timeout) as archive_path: if not archive_path: logger.info(f'GHArchive hour unavailable yet: {url}') continue try: for raw_line in iter_gharchive_lines(archive_path): try: event = json.loads(raw_line.decode('utf-8', errors='replace')) except (ValueError, UnicodeDecodeError): continue if event.get('type') not in event_types: continue repo = event.get('repo') if isinstance(event.get('repo'), dict) else {} name = str(repo.get('name') or '').strip() if not name or '/' not in name: continue if not re.match(r'^[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$', name): continue key = name.lower() order += 1 score = github_archive_event_score(event) existing = candidates.get(key) if existing and score > existing['score']: candidates[key] = { 'name': name, 'score': score, 'order': existing['order'], 'event': event, 'archive_hour': hour, } elif not existing: candidate = { 'name': name, 'score': score, 'order': order, 'event': event, 'archive_hour': hour, } if len(candidates) < max_repos: candidates[key] = candidate else: worst_key = min( candidates, key=lambda value: (candidates[value]['score'], -candidates[value]['order']), ) worst = candidates[worst_key] if (score, -order) > (worst['score'], -worst['order']): del candidates[worst_key] candidates[key] = candidate except (OSError, EOFError, gzip.BadGzipFile, GHArchiveBoundsError, urllib3_exceptions.ProtocolError) as e: remove_file_quiet(archive_path) raise ApiRequestError(f'Unable to read cached GHArchive gzip {url}: {e}') from e ranked = sorted(candidates.values(), key=lambda item: (-item['score'], item['order'])) selected = ranked[:max_repos] positive = sum(1 for item in selected if item['score'] > 0) logger.info(f'GHArchive selected {len(selected)} repos from {len(candidates)} candidates; positive_score={positive}') return [github_archive_target_payload(item['name'], item['event'], item['score'], item['archive_hour']) for item in selected] def fetch_github_archive_file_targets(hours_back=6, max_files=300, event_types=None, request_timeout=120, cache_dir=None, max_file_size_mb=2, token=None, max_commit_lookups=200, archive_cache_dir=None, discovery_max_artifacts=None, discovery_max_artifacts_per_page=None, discovery_max_bytes=None, discovery_max_elapsed_sec=None): event_types = {str(item).strip() for item in (event_types or []) if str(item).strip()} or {'PushEvent'} hours_back = min(GHARCHIVE_MAX_HOURS_BACK, max(1, int(hours_back or 1))) max_files = max(1, int(max_files or 1)) max_bytes = int(max_file_size_mb or 2) * 1024 * 1024 budget = _PostmanDiscoveryBudget( 'GHArchive-file', discovery_max_artifacts, discovery_max_artifacts_per_page, discovery_max_bytes, discovery_max_elapsed_sec, ) now = datetime.now(timezone.utc).replace(minute=0, second=0, microsecond=0) targets = [] seen = set() seen_digests = set() headers = github_headers(token) commit_lookups = 0 stop_cycle = False for offset in range(1, hours_back + 1): if stop_cycle or not budget.check_deadline(): break hour = now - timedelta(hours=offset) url = f'https://data.gharchive.org/{hour:%Y-%m-%d-%H}.json.gz' logger.info(f'Fetching GHArchive file candidates hour {hour.isoformat()} from {url}') prepared = [] pending_keys = set() page_artifacts = 0 stop_hour = False try: with cached_gharchive_hour_reader( hour, archive_cache_dir, budget.request_timeout(request_timeout), deadline=budget.deadline, ) as archive_path: if not archive_path: continue for raw_line in iter_gharchive_lines(archive_path, deadline=budget.deadline): if not budget.check_deadline('while reading GHArchive events'): stop_cycle = True break try: event = json.loads(raw_line.decode('utf-8', errors='replace')) except (ValueError, UnicodeDecodeError): continue if event.get('type') not in event_types: continue repo = event.get('repo') if isinstance(event.get('repo'), dict) else {} repo_name = str(repo.get('name') or '').strip() if not repo_name or '/' not in repo_name: continue path_items = [(sha, path, github_raw_url(repo_name, sha, path)) for sha, path in gharchive_changed_paths(event)] if not path_items and github_archive_event_score(event) <= 0: continue if not path_items: if commit_lookups >= int(max_commit_lookups or 0): continue commit_lookups += 1 try: path_items = list(fetch_github_commit_files( event, repo_name, headers, budget.request_timeout(request_timeout), budget.deadline, )) except _PostmanHarvestDeadlineReached as e: budget.stop(str(e)) stop_cycle = True break if not path_items and commit_lookups >= int(max_commit_lookups or 0): logger.info(f'GHArchive file fetch reached commit lookup cap: {max_commit_lookups}') for sha, path, raw_url in path_items: if not budget.check_deadline('while examining GHArchive paths'): stop_cycle = True break if len(targets) + len(prepared) >= max_files: stop_cycle = True break if not gharchive_path_interesting(path): continue key = f'{repo_name.lower()}@{sha}:{path.lower()}' if key in seen or key in pending_keys: continue if not raw_url: continue admitted, scope = budget.admit(page_artifacts) if not admitted: stop_cycle = scope == 'cycle' stop_hour = True break page_artifacts += 1 pending_keys.add(key) try: raw = api_request( 'GET', raw_url, timeout=budget.request_timeout(request_timeout), use_proxy=False, max_retries=2, retry_delay=1, retry_statuses={408, 500, 502, 503, 504}, deadline=budget.deadline, ) if raw.status_code == 404: continue raw.raise_for_status() content = raw.content except _PostmanHarvestDeadlineReached as e: budget.stop(str(e)) stop_cycle = True break except (requests.exceptions.RequestException, ApiRequestError) as e: logger.warning(f'Unable to fetch GHArchive raw file {repo_name}:{path}: {str(e)[:300]}') continue if max_bytes and len(content) > max_bytes: continue if not budget.account_bytes(len(content)): stop_cycle = True break origin = { 'provider': 'gharchive_file', 'repo': repo_name, 'path': path, 'sha': sha, 'raw_url': raw_url, 'archive_hour': hour.isoformat(), 'event_type': event.get('type') or '', } kind = postman_kind_for_path(path) or 'gharchive_file' try: entry = _prepare_postman_cache_entry(content, kind, origin, max_file_size_mb) except Exception as e: if isinstance(e, PostmanCacheCapacityError): raise logger.warning(f'Unable to cache GHArchive file {repo_name}:{path}: {str(e)[:300]}') continue if entry['digest'] in seen_digests: continue seen_digests.add(entry['digest']) prepared.append({ 'entry': entry, 'key': key, 'kind': kind, 'origin': origin, 'repo': repo_name, 'path': path, 'sha': sha, 'raw_url': raw_url, }) if stop_cycle or stop_hour: break except _PostmanHarvestDeadlineReached as e: budget.stop(str(e)) stop_cycle = True except (OSError, EOFError, gzip.BadGzipFile, GHArchiveBoundsError, urllib3_exceptions.ProtocolError) as e: if 'archive_path' in locals() and archive_path: remove_file_quiet(archive_path) raise ApiRequestError(f'Unable to read cached GHArchive gzip {url}: {e}') from e published, deadline_reached = _publish_postman_discovery_batch(prepared, cache_dir, budget) for record, (cache_path, digest, size) in published: target = gharchive_file_target( 'github_archive_file', record['kind'], cache_path, digest, record['origin'], repo=record['repo'], path=record['path'], sha=record['sha'], raw_url=record['raw_url'], size=size, ) targets.append(target) seen.add(record['key']) if deadline_reached or stop_cycle or budget.exhausted(): stop_cycle = True break logger.info(f'GHArchive file fetch produced {len(targets)} targets; commit_lookups={commit_lookups}') return targets def gist_file_interesting(filename, file_meta): text = ' '.join([ str(filename or '').lower(), str((file_meta or {}).get('type') or '').lower(), str((file_meta or {}).get('language') or '').lower(), ]) return _contains_term(text, GITHUB_GIST_FILE_TERMS) def fetch_github_gist_targets(pages=2, per_page=100, since=None, token=None, cache_dir=None, max_file_size_mb=2, request_timeout=20, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, discovery_max_artifacts=None, discovery_max_artifacts_per_page=None, discovery_max_bytes=None, discovery_max_elapsed_sec=None): headers = github_headers(token) per_page = max(1, min(int(per_page or 100), 100)) max_pages = max(1, int(pages or 1)) max_bytes = int(max_file_size_mb or 2) * 1024 * 1024 budget = _PostmanDiscoveryBudget( 'GitHub Gist', discovery_max_artifacts, discovery_max_artifacts_per_page, discovery_max_bytes, discovery_max_elapsed_sec, ) targets = [] seen_candidates = set() seen_targets = set() seen_digests = set() known_pages = 0 stop_cycle = False params_base = {'per_page': per_page} if since: params_base['since'] = since for page in range(1, max_pages + 1): if stop_cycle or not budget.check_deadline(): break params = dict(params_base) params['page'] = page try: response = api_request( 'GET', 'https://api.github.com/gists/public', headers=headers, params=params, timeout=budget.request_timeout(request_timeout), deadline=budget.deadline, ) response.raise_for_status() except _PostmanHarvestDeadlineReached as e: budget.stop(str(e)) break except requests.exceptions.HTTPError as e: raise github_api_error(e.response) from e except (requests.exceptions.RequestException, ApiRequestError) as e: logger.warning(f'GitHub Gists request failed on page {page}: {str(e)[:300]}') break gists = response.json() or [] if not gists: break page_targets = [] prepared = [] pending_candidates = set() page_artifacts = 0 stop_page = False for gist in gists: if not budget.check_deadline('while examining a Gist page'): stop_cycle = True break gist_id = str(gist.get('id') or '') files = gist.get('files') if isinstance(gist.get('files'), dict) else {} for filename, meta in files.items(): if not budget.check_deadline('while examining Gist files'): stop_cycle = True break if not isinstance(meta, dict): continue size = int(meta.get('size') or 0) raw_url = meta.get('raw_url') or '' if not raw_url or (max_bytes and size > max_bytes): continue if not gist_file_interesting(filename, meta): continue identity = f'gist:{gist_id}:{filename}:{meta.get("raw_url")}' if identity in seen_candidates or identity in pending_candidates: continue admitted, scope = budget.admit(page_artifacts) if not admitted: stop_cycle = scope == 'cycle' stop_page = True break page_artifacts += 1 pending_candidates.add(identity) try: raw = api_request( 'GET', raw_url, headers=headers, use_proxy=False, timeout=budget.request_timeout(request_timeout), deadline=budget.deadline, ) raw.raise_for_status() content = raw.content except _PostmanHarvestDeadlineReached as e: budget.stop(str(e)) stop_cycle = True break except (requests.exceptions.RequestException, ApiRequestError) as e: logger.warning(f'Unable to fetch gist raw {gist_id}/{filename}: {str(e)[:300]}') continue if max_bytes and len(content) > max_bytes: continue if not budget.account_bytes(len(content)): stop_cycle = True break origin = { 'provider': 'github_gist', 'gist_id': gist_id, 'filename': filename, 'raw_url': raw_url, 'html_url': gist.get('html_url'), 'created_at': gist.get('created_at'), 'updated_at': gist.get('updated_at'), 'owner': ((gist.get('owner') or {}).get('login') if isinstance(gist.get('owner'), dict) else ''), } kind = postman_kind_for_path(filename) or 'gist_file' try: entry = _prepare_postman_cache_entry(content, kind, origin, max_file_size_mb) except Exception as e: if isinstance(e, PostmanCacheCapacityError): raise logger.warning(f'Unable to cache gist {gist_id}/{filename}: {str(e)[:300]}') continue if entry['digest'] in seen_digests: continue seen_digests.add(entry['digest']) prepared.append({ 'entry': entry, 'candidate_identity': identity, 'kind': kind, 'origin': origin, 'gist_id': gist_id, 'filename': filename, 'raw_url': raw_url, 'html_url': gist.get('html_url'), }) if stop_cycle or stop_page: break published, deadline_reached = _publish_postman_discovery_batch(prepared, cache_dir, budget) for record, (cache_path, digest, cached_size) in published: target = postman_target_from_cached_artifact( 'github_gist', record['kind'], cache_path, digest, record['origin'], gist_id=record['gist_id'], filename=record['filename'], raw_url=record['raw_url'], html_url=record['html_url'], size=cached_size, ) target_identity = postman_target_identity(target) seen_candidates.add(record['candidate_identity']) if target_identity in seen_targets: continue targets.append(target) page_targets.append(target) seen_targets.add(target_identity) logger.info(f'GitHub Gists page {page}: gists={len(gists)}, targets={len(page_targets)}') if stop_on_seen_pages and page >= max(1, int(min_pages_before_stop or 1)): if page_targets and page_is_known( page_targets, known_targets, normalize_target, known_target_lookup, ): known_pages += 1 if known_pages >= max(1, int(seen_page_threshold or 1)): break else: known_pages = 0 if deadline_reached or stop_cycle or budget.exhausted(): stop_cycle = True break return targets def gitlab_project_to_target(project): return { 'url': project.get('http_url_to_repo', ''), 'name': project.get('path_with_namespace', ''), 'created_at': project.get('created_at', ''), 'updated_at': project.get('updated_at') or project.get('last_activity_at', ''), 'last_activity_at': project.get('last_activity_at', ''), } def gitlab_rate_limit_reset(response): if not response: return None reset = response.headers.get('RateLimit-Reset') or response.headers.get('X-RateLimit-Reset') if not reset: retry_after = response.headers.get('Retry-After') if retry_after: try: return (datetime.now(timezone.utc) + timedelta(seconds=int(retry_after))).isoformat(timespec='seconds') except (TypeError, ValueError): return None return None try: if str(reset).isdigit(): return datetime.fromtimestamp(int(reset), timezone.utc).isoformat(timespec='seconds') return str(reset) except (TypeError, ValueError): return None def fetch_gitlab_repo_items(query, pages, per_page=100, token=None, sort_by="last_activity_at", sort_order="desc", visibility="public", last_activity_after=None, raise_rate_limit=False, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, request_attempts=1, retry_delay=0): """Fetch GitLab repositories with metadata for filtering.""" repos = [] headers = {} if token: headers['Authorization'] = f'Bearer {token}' logger.info(f"Fetching GitLab repositories for: '{query if query else 'all repositories'}' sorted by {sort_by} ({sort_order})...") page = 1 seen_pages = 0 successful_pages = 0 request_attempts = max(1, int(request_attempts or 1)) retry_delay = max(0, int(retry_delay or 0)) request_budget = 30 * request_attempts + retry_delay * (request_attempts - 1) while page <= pages: url = "https://gitlab.com/api/v4/projects" params = { 'visibility': visibility, 'per_page': per_page, 'page': page, 'order_by': sort_by, 'sort': sort_order, } if query: params['search'] = query if last_activity_after: params['last_activity_after'] = last_activity_after try: response = api_request( 'GET', url, headers=headers, params=params, timeout=30, max_retries=request_attempts, retry_delay=retry_delay, deadline=time.monotonic() + request_budget, ) response.raise_for_status() # Handle rate limits if response.status_code == 429: retry_after = int(response.headers.get('Retry-After', 60)) logger.warning(f"Rate limit exceeded. Waiting {retry_after} seconds...") time.sleep(retry_after) continue data = response.json() if not isinstance(data, list): raise ValueError('invalid GitLab project search payload') successful_pages += 1 if not data: logger.info(f"Page {page} returned no results. Stopping.") break page_repos = [gitlab_project_to_target(project) for project in data if project.get('http_url_to_repo')] repos.extend(page_repos) logger.info(f"Page {page}: Fetched {len(data)} repositories") if stop_on_seen_pages and page >= max(1, min_pages_before_stop): if page_is_known( [item.get('url') for item in page_repos], known_targets, normalize_target, known_target_lookup, ): seen_pages += 1 logger.info(f"Page {page}: all GitLab repositories are already queued/checked ({seen_pages}/{seen_page_threshold})") if seen_pages >= max(1, seen_page_threshold): logger.info(f"Stopping GitLab pagination early after {seen_pages} all-known page(s)") break else: seen_pages = 0 page += 1 except requests.exceptions.HTTPError as e: if e.response.status_code == 401 and not token: logger.warning("GitLab API authentication would improve results. Consider adding a GitLab token.") raise gitlab_api_error(e.response) from e else: api_error = gitlab_api_error(e.response) if raise_rate_limit: raise api_error from e logger.error(str(api_error)) raise api_error from e except Exception as e: if isinstance(e, ApiRequestError): raise GitLabDiscoveryTransportError(str(e)) from e logger.error(f"Error fetching page {page}: {str(e)}") raise ApiRequestError(f'GitLab discovery payload failed: {e}') from e return repos def fetch_gitlab_repos(query, pages, per_page=100, token=None, sort_by="last_activity_at", sort_order="desc", visibility="public", raise_rate_limit=False, **kwargs): """Fetch GitLab repository clone URLs with pagination and authentication.""" return [ item['url'] for item in fetch_gitlab_repo_items( query, pages, per_page, token, sort_by, sort_order, visibility, raise_rate_limit=raise_rate_limit, **kwargs ) if item.get('url') ] def github_recent_query(query, since): since_str = since.strftime("%Y-%m-%d") query = (query or '').strip() if query and ' in:' not in f' {query.lower()} ': query = f'{query} in:name,description,readme' return f"{query} updated:>={since_str}" if query else f"updated:>={since_str}" def fetch_recent_github_repos(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, **kwargs): """Fetch GitHub repositories updated since a specific timestamp""" return fetch_github_repos(github_recent_query(query, since), pages=pages, per_page=per_page, token=token, raise_rate_limit=raise_rate_limit, **kwargs) def fetch_recent_github_repo_items(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, **kwargs): """Fetch recent GitHub repositories with metadata.""" return fetch_github_repo_items(github_recent_query(query, since), pages=pages, per_page=per_page, token=token, raise_rate_limit=raise_rate_limit, **kwargs) def fetch_recent_gitlab_repos(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, visibility="public", **kwargs): """Fetch GitLab repositories updated since a specific timestamp""" since_str = since.strftime("%Y-%m-%dT%H:%M:%SZ") return [ item['url'] for item in fetch_gitlab_repo_items( query, pages=pages, per_page=per_page, token=token, sort_by="last_activity_at", sort_order="desc", visibility=visibility, last_activity_after=since_str, raise_rate_limit=raise_rate_limit, **kwargs, ) if item.get('url') ] def fetch_recent_gitlab_repo_items(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, visibility="public", **kwargs): """Fetch recent GitLab repositories with metadata.""" since_str = since.strftime("%Y-%m-%dT%H:%M:%SZ") return fetch_gitlab_repo_items( query, pages=pages, per_page=per_page, token=token, sort_by="last_activity_at", sort_order="desc", visibility=visibility, last_activity_after=since_str, raise_rate_limit=raise_rate_limit, **kwargs, ) DOCKERHUB_SEARCH_MAX_PAGES = 30 DOCKERHUB_TAG_RATE_LIMIT_MIN_BACKOFF_SEC = 60 DOCKERHUB_TAG_RATE_LIMIT_MAX_BACKOFF_SEC = 3600 DOCKER_REGISTRY_MANIFEST_MAX_BYTES = 8 * 1024 * 1024 DOCKER_REGISTRY_MAX_DESCRIPTORS = 1000 DOCKER_REGISTRY_MAX_LAYERS = 2048 DOCKER_REGISTRY_TOKEN_MAX_BYTES = 1024 * 1024 DOCKER_CONFIG_MEDIA_TYPES = frozenset(( 'application/vnd.oci.image.config.v1+json', 'application/vnd.docker.container.image.v1+json', )) DOCKER_LAYER_MEDIA_TYPES = frozenset(( 'application/vnd.oci.image.layer.v1.tar', 'application/vnd.oci.image.layer.v1.tar+gzip', 'application/vnd.oci.image.layer.v1.tar+zstd', 'application/vnd.docker.image.rootfs.diff.tar', 'application/vnd.docker.image.rootfs.diff.tar.gzip', )) DOCKER_LAYER_GZIP_MEDIA_TYPES = frozenset(( 'application/vnd.oci.image.layer.v1.tar+gzip', 'application/vnd.docker.image.rootfs.diff.tar.gzip', )) DOCKER_CONFIG_HISTORY_MAX_ENTRIES = DOCKER_REGISTRY_MAX_LAYERS * 2 DOCKER_CONFIG_HISTORY_COMMAND_MAX_CHARS = 64 * 1024 def docker_history_payload_class(created_by): if not isinstance(created_by, str) or not created_by: return 'unknown' if len(created_by) > DOCKER_CONFIG_HISTORY_COMMAND_MAX_CHARS: return 'unknown' command = created_by.casefold() if re.search(r'(^|#\(nop\)\s+)(copy|add)\s', command.strip()): return 'copy_add' if not re.search(r'(^|[\s#])run(\s|$)|/bin/(ba)?sh\s+-c', command): return 'unknown' if any(token in command for token in ( '/models/', '/model/', 'model_weights', 'checkpoint.', '.safetensors', '.gguf', '.onnx', '.pt ', '.pth ', 'huggingface-cli download', )): return 'bulk_data' if re.search( r'\b(apt-get|apt|apk|yum|dnf|microdnf|pip|pip3|poetry|npm|pnpm|yarn|' r'bundle|gem|cargo)\s+(install|add|sync)\b', command, ): return 'package_run' if any(token in command for token in ( '/app', '/srv', '/workspace', '/opt/app', 'config', '.env', 'requirements.txt', 'package.json', 'pyproject.toml', )): return 'app_config_run' return 'other_run' def docker_config_payload_classes(config, layer_count): try: layer_count = int(layer_count) except (TypeError, ValueError): layer_count = -1 fallback = ['unknown'] * max(0, layer_count) if ( not isinstance(config, dict) or layer_count < 0 or layer_count > DOCKER_REGISTRY_MAX_LAYERS ): return fallback history = config.get('history') if not isinstance(history, list) or len(history) > DOCKER_CONFIG_HISTORY_MAX_ENTRIES: return fallback rootfs = config.get('rootfs') if rootfs is not None: if not isinstance(rootfs, dict): return fallback diff_ids = rootfs.get('diff_ids') if not isinstance(diff_ids, list) or len(diff_ids) != layer_count: return fallback commands = [] for entry in history: if not isinstance(entry, dict): return fallback empty_layer = entry.get('empty_layer', False) if not isinstance(empty_layer, bool): return fallback if empty_layer: continue created_by = entry.get('created_by') if not isinstance(created_by, str): return fallback commands.append(created_by) if len(commands) != layer_count: return fallback classes = [docker_history_payload_class(command) for command in commands] if any(payload_class not in DOCKER_ADAPTIVE_PAYLOAD_CLASSES for payload_class in classes): return fallback return classes class DockerRegistryResolutionError(ValueError): pass class DockerResolverLeaseLostError(RuntimeError): pass class DockerRemoteAccessError(DockerRegistryResolutionError): def __init__(self, message, status='unknown', retry_at=None, remote_attempted=True): super().__init__(message) self.status = str(status or 'unknown') self.retry_at = retry_at self.remote_attempted = bool(remote_attempted) class DockerContentTransferError(RuntimeError): def __init__(self, error_code, message, retryable=True): super().__init__(message) self.error_code = str(error_code or 'transfer_failed') self.retryable = bool(retryable) self.source_failure = False self.transfer_bytes = 0 self.duration_ms = 0 class DockerLayerInfrastructureError(DockerContentTransferError): def __init__( self, error_code, message, category='remote_transient', auth_related=False, ): super().__init__(error_code, message, retryable=True) self.category = str(category or 'remote_transient') self.auth_related = bool(auth_related) self.source_failure = True class DockerContentScanError(RuntimeError): def __init__(self, error_code, message, retryable=False): super().__init__(message) self.error_code = str(error_code or 'invalid_content') self.retryable = bool(retryable) @dataclass(frozen=True, repr=False) class DockerBlobDownloadOutcome: path: str verified_bytes: int transfer_bytes: int duration_ms: int bearer_auth: DockerRegistryAuth @dataclass(frozen=True) class DockerTagResolutionOutcome: tags: tuple status: str remote_attempted: bool retry_at: str = None error: str = '' selection_records: tuple = () candidate_records: tuple = () selector_version: str = '' selector_hash: str = '' candidate_distinct_graph_count: int = 0 fresh_graph_evidence: bool = False cache_bypassed: bool = False @property def selections(self): return self.selection_records @property def selector_sha256(self): return self.selector_hash DOCKER_TAG_CONCLUSIVE_STATUSES = frozenset({'ok', 'empty', 'unsupported', 'not_found'}) def docker_tag_resolution_is_conclusive(status): return str(status or '') in DOCKER_TAG_CONCLUSIVE_STATUSES def docker_images_per_repository_limit(value): return validate_docker_images_per_repository(value) def dockerhub_search_page_window(pages): try: requested = max(1, int(pages or 1)) except (TypeError, ValueError): requested = 1 return requested, min(requested, DOCKERHUB_SEARCH_MAX_PAGES) def fetch_dockerhub_search_page( query, page, per_page=100, sort_by='updated_at', sort_order='desc', request_timeout=15, ): """Fetch and validate one bounded Docker Hub repository-search page.""" try: page = int(page) per_page = max(1, int(per_page or 1)) except (TypeError, ValueError, OverflowError): raise DockerHubDiscoveryTransportError( 'Docker Hub search page request is invalid', category='invalid_payload', remote_attempted=False, retryable=False, ) from None if page < 1 or page > DOCKERHUB_SEARCH_MAX_PAGES: raise DockerHubDiscoveryTransportError( 'Docker Hub search page request is invalid', category='invalid_payload', remote_attempted=False, retryable=False, ) params = { 'query': query, 'page': page, 'page_size': per_page, 'sort': sort_by, 'order': sort_order, } started = time.perf_counter() try: response = dockerhub_search_response( 'https://hub.docker.com/v2/search/repositories', params, request_timeout=request_timeout, ) except DockerRemoteAccessError as error: category = { 'rate_limited': 'rate_limit', 'auth_failed': 'auth_unavailable', 'remote_transient': 'remote_transient', }.get(error.status, 'page_unavailable') if error.retry_at and not error.remote_attempted: category = 'provider_cooldown' raise DockerHubDiscoveryTransportError( f'Docker Hub search page {page} failed after bounded attempts', category=category, retry_at=error.retry_at, remote_attempted=error.remote_attempted, ) from None except ApiRequestError: raise DockerHubDiscoveryTransportError( f'Docker Hub search page {page} failed after bounded attempts', category='network', remote_attempted=True, ) from None except Exception: raise DockerHubDiscoveryTransportError( f'Docker Hub search page {page} failed after bounded attempts', category='page_unavailable', remote_attempted=True, ) from None try: response.raise_for_status() except Exception: raise DockerHubDiscoveryTransportError( f'Docker Hub search page {page} failed after bounded attempts', category='page_unavailable', remote_attempted=True, ) from None try: data = response.json() except Exception: raise DockerHubDiscoveryTransportError( f'Docker Hub search page {page} returned an invalid payload', category='invalid_payload', remote_attempted=True, retryable=False, ) from None if not isinstance(data, dict) or not isinstance(data.get('results'), list): raise DockerHubDiscoveryTransportError( f'Docker Hub search page {page} returned an invalid payload', category='invalid_payload', remote_attempted=True, retryable=False, ) raw_count = data.get('count') try: total_count = int(raw_count) except (TypeError, ValueError, OverflowError): raise DockerHubDiscoveryTransportError( f'Docker Hub search page {page} returned an invalid result count', category='invalid_payload', remote_attempted=True, retryable=False, ) from None if ( isinstance(raw_count, bool) or (isinstance(raw_count, float) and not raw_count.is_integer()) or total_count < 0 ): raise DockerHubDiscoveryTransportError( f'Docker Hub search page {page} returned an invalid result count', category='invalid_payload', remote_attempted=True, retryable=False, ) result_count = len(data['results']) absolute_start = (page - 1) * per_page if ( result_count > per_page or total_count < absolute_start + result_count or (result_count == 0 and total_count > absolute_start) ): raise DockerHubDiscoveryTransportError( f'Docker Hub search page {page} returned incoherent pagination evidence', category='invalid_payload', remote_attempted=True, retryable=False, ) repositories = [] seen_repo_names = set() for repository in data['results']: if not isinstance(repository, dict): raise DockerHubDiscoveryTransportError( f'Docker Hub search page {page} returned an invalid payload', category='invalid_payload', remote_attempted=True, retryable=False, ) repo_name = repository.get('repo_name') if not isinstance(repo_name, str) or not repo_name.strip(): raise DockerHubDiscoveryTransportError( f'Docker Hub search page {page} returned an invalid payload', category='invalid_payload', remote_attempted=True, retryable=False, ) repo_name = repo_name.strip() if repo_name in seen_repo_names: continue seen_repo_names.add(repo_name) safe_repository = {'repo_name': repo_name} for field in ('last_updated', 'last_modified'): if field in repository: safe_repository[field] = repository[field] repositories.append(safe_repository) return { 'page': page, 'repositories': repositories, 'total_count': total_count, 'elapsed': time.perf_counter() - started, } def fetch_dockerhub_images(query, pages, per_page=100, sort_by="updated_at", sort_order="desc", fetch_workers=8, request_timeout=15, resolve_tags=True, tag_fetch_workers=None, tag_retry_count=2, tag_retry_delay=5, platform_filter_enabled=False, platform_os='linux', platform_arch='amd64', platform_candidate_tags=20, images_per_repository=1): """Fetch Docker Hub images with pagination and sorting""" images = [] per_page = max(1, int(per_page or 1)) requested_pages, pages = dockerhub_search_page_window(pages) if requested_pages > pages: logger.info( f'Docker Hub search is limited to {pages} accessible page(s); ' f'capping requested pages from {requested_pages}' ) logger.info(f"Fetching Docker Hub images for: '{query}' sorted by {sort_by} ({sort_order})...") def fetch_page(page): started = time.perf_counter() try: return fetch_dockerhub_search_page( query, page, per_page=per_page, sort_by=sort_by, sort_order=sort_order, request_timeout=request_timeout, ), None except DockerHubDiscoveryTransportError as error: return { 'page': page, 'repositories': [], 'total_count': 0, 'elapsed': time.perf_counter() - started, }, error first_page, first_error = fetch_page(1) if first_error is not None: raise first_error total_count = first_page['total_count'] expected_pages = min( pages, max(1, (total_count + per_page - 1) // per_page), ) page_results = [(first_page, None)] if expected_pages > 1: max_workers = max(1, min(fetch_workers, expected_pages - 1)) with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor: futures = [ executor.submit(fetch_page, page) for page in range(2, expected_pages + 1) ] page_results.extend( future.result() for future in concurrent.futures.as_completed(futures) ) repo_names = [] failed_pages = [] for page_result, error in sorted( page_results, key=lambda item: item[0]['page'], ): page = page_result['page'] elapsed = page_result['elapsed'] if error: logger.error( f"Page {page}: Docker Hub fetch failed after bounded attempts " f"({elapsed:.1f}s)" ) failed_pages.append(page) continue repositories = page_result['repositories'] if not repositories: logger.info( f"Page {page}: no results after {elapsed:.1f}s " f"(total matches: {page_result['total_count']})" ) continue repo_names.extend(repository['repo_name'] for repository in repositories) logger.info(f"Page {page}: fetched {len(repositories)} images in {elapsed:.1f}s") if failed_pages: raise DockerHubDiscoveryTransportError( f'Docker Hub search pagination incomplete: {len(failed_pages)} ' f'of {expected_pages} expected page(s) failed' ) if not resolve_tags: return repo_names tag_fetch_workers = tag_fetch_workers if tag_fetch_workers is not None else min(4, fetch_workers) logger.info(f"Resolving Docker tags for {len(repo_names)} repositories with {tag_fetch_workers} worker(s)...") max_workers = max(1, min(tag_fetch_workers, len(repo_names) or 1)) with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor: futures = { executor.submit( fetch_dockerhub_tags, repo_name, None, docker_images_per_repository_limit(images_per_repository), tag_retry_count, tag_retry_delay, platform_filter_enabled, platform_os, platform_arch, platform_candidate_tags, True, ): repo_name for repo_name in repo_names } for future in concurrent.futures.as_completed(futures): repo_name = futures[future] tags, status = future.result() if tags: images.extend(tags) if status != 'ok': images.append(repo_name) elif not docker_tag_resolution_is_conclusive(status): images.append(repo_name) logger.info(f"Deferring Docker tag resolution for {repo_name}: metadata lookup unavailable") else: logger.info(f"Skipping {repo_name}: no tags found") logger.info(f"Resolved {len(images)} tagged Docker images from {len(repo_names)} repositories") return images def parse_dockerhub_datetime(value): if not value: return None value = value.rstrip('Z') for date_format in ("%Y-%m-%dT%H:%M:%S.%f", "%Y-%m-%dT%H:%M:%S"): try: return datetime.strptime(value, date_format) except ValueError: continue return None def fetch_dockerhub_last_updated(repo_name): if '/' in repo_name: namespace, name = repo_name.split('/', 1) else: namespace, name = 'library', repo_name url = f"https://hub.docker.com/v2/repositories/{namespace}/{name}/" try: response = api_request('GET', url, headers={'User-Agent': 'GitSecretsScanner/2.0'}, timeout=8) response.raise_for_status() data = response.json() return parse_dockerhub_datetime(data.get('last_updated') or data.get('last_modified')) except Exception as e: if isinstance(e, ApiRequestError): raise logger.warning(f"Unable to fetch Docker Hub metadata for {repo_name}: {str(e)}") return None DOCKERHUB_TAG_CACHE_SCHEMA = """ CREATE TABLE IF NOT EXISTS dockerhub_tag_cache ( cache_key TEXT PRIMARY KEY, repo_key TEXT NOT NULL, status TEXT NOT NULL, tags_json TEXT, checked_at REAL NOT NULL, expires_at REAL NOT NULL, since_at REAL, message TEXT ); CREATE TABLE IF NOT EXISTS dockerhub_tag_cache_meta ( key TEXT PRIMARY KEY, value TEXT, expires_at REAL ); """ _dockerhub_tag_cache_init_lock = threading.Lock() _dockerhub_tag_cache_write_lock = threading.Lock() _dockerhub_tag_cache_initialized = set() def dockerhub_repo_key(repo_name): repo_name = str(repo_name or '').strip().lower() if ':' in repo_name: repo_name = repo_name.split(':', 1)[0] if '/' not in repo_name: repo_name = 'library/' + repo_name return repo_name.strip('/') def dockerhub_tag_cache_path(): return getattr(scan_config, 'dockerhub_tag_cache_path', '') or '' def dockerhub_tag_cache_limits(): return { 'rows': max(1, int(getattr(scan_config, 'dockerhub_tag_cache_max_rows', 50000))), 'age': max(60, int(getattr(scan_config, 'dockerhub_tag_cache_max_age_sec', 7 * 86400))), 'bytes': max(4096, int(getattr(scan_config, 'dockerhub_tag_cache_max_bytes', 256 * 1024 * 1024))), 'min_free': max(0, int(getattr(scan_config, 'dockerhub_tag_cache_min_free_bytes', 512 * 1024 * 1024))), } def dockerhub_tag_cache_disk_bytes(path): return sum( os.path.getsize(candidate) for candidate in (path, path + '-wal', path + '-shm') if os.path.isfile(candidate) ) def _dockerhub_since_epoch(since): if since is None: return None try: if isinstance(since, datetime): value = since else: value = datetime.fromisoformat(str(since).replace('Z', '+00:00')) if value.tzinfo is None: value = value.replace(tzinfo=timezone.utc) return float(value.timestamp()) except (TypeError, ValueError, OverflowError): return None def maintain_dockerhub_tag_cache(conn, path): limits = dockerhub_tag_cache_limits() now = time.time() conn.execute('DELETE FROM dockerhub_tag_cache WHERE expires_at <= ? OR checked_at < ?', (now, now - limits['age'])) conn.execute('DELETE FROM dockerhub_tag_cache_meta WHERE expires_at IS NOT NULL AND expires_at <= ?', (now,)) conn.execute( '''DELETE FROM dockerhub_tag_cache WHERE cache_key NOT IN ( SELECT cache_key FROM dockerhub_tag_cache ORDER BY checked_at DESC LIMIT ? )''', (limits['rows'],), ) conn.commit() if dockerhub_tag_cache_disk_bytes(path) > limits['bytes']: # The cache is disposable. Clearing it under SQLite's full auto-vacuum # is safer than allowing stale pages to consume an unbounded volume. conn.execute('DELETE FROM dockerhub_tag_cache') conn.execute('DELETE FROM dockerhub_tag_cache_meta') conn.commit() try: conn.execute('PRAGMA incremental_vacuum') except sqlite3.DatabaseError: pass disk_bytes = dockerhub_tag_cache_disk_bytes(path) parent = os.path.dirname(path) or os.getcwd() if int(shutil.disk_usage(parent).free) - disk_bytes >= limits['min_free']: try: conn.execute('PRAGMA wal_checkpoint(TRUNCATE)') conn.execute('PRAGMA journal_mode=DELETE') conn.execute('VACUUM') conn.execute('PRAGMA journal_mode=WAL') except sqlite3.DatabaseError: pass return dockerhub_tag_cache_disk_bytes(path) <= limits['bytes'] def connect_dockerhub_tag_cache(): path = dockerhub_tag_cache_path() if not path: return None parent = os.path.dirname(path) if parent: os.makedirs(parent, exist_ok=True) with _dockerhub_tag_cache_init_lock: if path not in _dockerhub_tag_cache_initialized: new_database = not os.path.exists(path) conn = sqlite3.connect(path, timeout=10) try: conn.execute('PRAGMA busy_timeout=10000') if new_database: conn.execute('PRAGMA auto_vacuum=FULL') conn.execute('PRAGMA journal_mode=WAL') conn.executescript(DOCKERHUB_TAG_CACHE_SCHEMA) columns = {row[1] for row in conn.execute('PRAGMA table_info(dockerhub_tag_cache)').fetchall()} if 'since_at' not in columns: conn.execute('ALTER TABLE dockerhub_tag_cache ADD COLUMN since_at REAL') conn.commit() if not maintain_dockerhub_tag_cache(conn, path): return None finally: conn.close() _dockerhub_tag_cache_initialized.add(path) conn = sqlite3.connect(path, timeout=10) conn.execute('PRAGMA busy_timeout=10000') return conn def dockerhub_cache_key(repo_name, since, limit, platform_variant=''): return f"{dockerhub_repo_key(repo_name)}|{int(limit or 1)}|{platform_variant}" def get_dockerhub_tag_cache(repo_name, since, limit, platform_variant='', return_status=False): conn = connect_dockerhub_tag_cache() if not conn: return None try: row = conn.execute( 'SELECT status, tags_json, expires_at, since_at FROM dockerhub_tag_cache WHERE cache_key = ?', (dockerhub_cache_key(repo_name, since, limit, platform_variant),), ).fetchone() now = time.time() if not row or float(row[2] or 0) <= now: return None status, tags_json, _, stored_since = row requested_since = _dockerhub_since_epoch(since) if requested_since is not None and stored_since is None: return None if requested_since is not None and float(stored_since or 0) > requested_since: return None if status == 'ok': records = json.loads(tags_json or '[]') targets = [] for record in records: if isinstance(record, str): return None if not isinstance(record, dict) or not record.get('name'): continue updated_at = record.get('updated_at') if requested_since is not None and updated_at is not None and float(updated_at) < requested_since: continue target = record.get('target') if not target: return None try: targets.append(parse_docker_target(target)['target']) except (TypeError, ValueError): return None if len(targets) >= max(1, int(limit or 1)): break if not targets: return None return (targets, status) if return_status else targets return ([], status) if return_status else [] except Exception: return None finally: conn.close() def put_dockerhub_tag_cache( repo_name, since, limit, status, tags=None, ttl=None, message='', platform_variant='', tag_records=None, ): with _dockerhub_tag_cache_write_lock: conn = connect_dockerhub_tag_cache() if not conn: return path = dockerhub_tag_cache_path() try: now = time.time() ttl = int(ttl if ttl is not None else getattr(scan_config, 'dockerhub_tag_cache_ttl_sec', 21600)) records = [] offered_records = list(tag_records or []) for index, tag in enumerate(tags or []): text = str(tag) record = offered_records[index] if index < len(offered_records) and isinstance(offered_records[index], dict) else {} name = str(record.get('name') or '') if not name: name = text.rsplit(':', 1)[1] if ':' in text and '@' not in text and not text.startswith('{') else text target = record.get('target') or text try: target = parse_docker_target(target)['target'] except (TypeError, ValueError): continue records.append({'name': name, 'target': target, 'updated_at': record.get('updated_at')}) if status == 'ok' and not records: return False payload = json.dumps(records, ensure_ascii=True, sort_keys=True, separators=(',', ':')) limits = dockerhub_tag_cache_limits() projected_bytes = len(payload.encode('utf-8')) + len(str(message or '').encode('utf-8')) + 1024 parent = os.path.dirname(path) or os.getcwd() if ( not maintain_dockerhub_tag_cache(conn, path) or dockerhub_tag_cache_disk_bytes(path) + projected_bytes > limits['bytes'] or int(shutil.disk_usage(parent).free) - projected_bytes < limits['min_free'] ): return False conn.execute( '''INSERT OR REPLACE INTO dockerhub_tag_cache( cache_key, repo_key, status, tags_json, checked_at, expires_at, since_at, message ) VALUES (?, ?, ?, ?, ?, ?, ?, ?)''', ( dockerhub_cache_key(repo_name, since, limit, platform_variant), dockerhub_repo_key(repo_name), status, payload, now, now + max(1, ttl), _dockerhub_since_epoch(since), str(message or '')[:500], ), ) conn.commit() maintain_dockerhub_tag_cache(conn, path) return True except Exception: return False finally: conn.close() def dockerhub_tag_rate_limit_state(): conn = connect_dockerhub_tag_cache() if not conn: return {'active': False, 'retry_at': None} try: row = conn.execute("SELECT expires_at FROM dockerhub_tag_cache_meta WHERE key = 'rate_limited'").fetchone() expires_at = float(row[0] or 0) if row else 0 active = expires_at > time.time() return { 'active': active, 'retry_at': ( datetime.fromtimestamp(expires_at, timezone.utc).isoformat(timespec='seconds') if active else None ), } except Exception: return {'active': False, 'retry_at': None} finally: conn.close() def dockerhub_tags_rate_limited(): return dockerhub_tag_rate_limit_state()['active'] def dockerhub_retry_after_seconds(response=None): fallback = int(getattr(scan_config, 'dockerhub_tag_rate_limit_cache_ttl_sec', 1800) or 1800) seconds = None headers = (getattr(response, 'headers', None) or {}) if response is not None else {} retry_after = headers.get('Retry-After') if retry_after: try: seconds = int(retry_after) except (TypeError, ValueError): try: retry_at = parsedate_to_datetime(str(retry_after)) if retry_at.tzinfo is None: retry_at = retry_at.replace(tzinfo=timezone.utc) seconds = math.ceil((retry_at - datetime.now(timezone.utc)).total_seconds()) except (IndexError, TypeError, ValueError, OverflowError): seconds = None seconds = fallback if seconds is None else seconds return max( DOCKERHUB_TAG_RATE_LIMIT_MIN_BACKOFF_SEC, min(DOCKERHUB_TAG_RATE_LIMIT_MAX_BACKOFF_SEC, int(seconds)), ) def put_dockerhub_tags_rate_limit(response=None, retry_seconds=None, return_retry_at=False): ttl = dockerhub_retry_after_seconds(response) if retry_seconds is None else max( DOCKERHUB_TAG_RATE_LIMIT_MIN_BACKOFF_SEC, min(DOCKERHUB_TAG_RATE_LIMIT_MAX_BACKOFF_SEC, int(retry_seconds)), ) with _dockerhub_tag_cache_write_lock: conn = connect_dockerhub_tag_cache() if not conn: return try: path = dockerhub_tag_cache_path() limits = dockerhub_tag_cache_limits() parent = os.path.dirname(path) or os.getcwd() if ( dockerhub_tag_cache_disk_bytes(path) + 8192 > limits['bytes'] or int(shutil.disk_usage(parent).free) - 8192 < limits['min_free'] ): return False expires_at = math.ceil(time.time() + ttl) conn.execute( '''INSERT INTO dockerhub_tag_cache_meta(key, value, expires_at) VALUES ('rate_limited', '1', ?) ON CONFLICT(key) DO UPDATE SET value = '1', expires_at = MAX(dockerhub_tag_cache_meta.expires_at, excluded.expires_at)''', (expires_at,), ) conn.commit() stored = conn.execute( "SELECT expires_at FROM dockerhub_tag_cache_meta WHERE key = 'rate_limited'" ).fetchone() maintain_dockerhub_tag_cache(conn, path) if return_retry_at: return datetime.fromtimestamp(float(stored[0]), timezone.utc).isoformat(timespec='seconds') return True except Exception: return False finally: conn.close() def put_dockerhub_exhausted_rate_limit(endpoint, response=None): if endpoint == 'hub_search': if not docker_token_manager.has_accounts(): if ( docker_token_manager.uses_explicit_pool() or int(getattr(response, 'status_code', 0) or 0) != 429 ): return None retry_seconds = dockerhub_retry_after_seconds(response) else: if not docker_token_manager.rate_limit_contributes_to_exhaustion(endpoint): return None retry_seconds = docker_token_manager.seconds_until_available(endpoint) return datetime.fromtimestamp( time.time() + retry_seconds, timezone.utc, ).isoformat(timespec='seconds') if not docker_token_manager.has_accounts(): if docker_token_manager.uses_explicit_pool(): return None if int(getattr(response, 'status_code', 0) or 0) != 429: return None return put_dockerhub_tags_rate_limit(response, return_retry_at=True) if not docker_token_manager.rate_limit_contributes_to_exhaustion(endpoint): return None return put_dockerhub_tags_rate_limit( response, retry_seconds=docker_token_manager.seconds_until_available(endpoint), return_retry_at=True, ) def docker_tag_platform_support(tag, platform_os='linux', platform_arch='amd64'): images = tag.get('images') if isinstance(tag, dict) else None if not isinstance(images, list) or not images: return None platforms = [] for image in images: if not isinstance(image, dict): return None image_os = str(image.get('os') or '').strip().lower() image_arch = str(image.get('architecture') or '').strip().lower() if not image_os or not image_arch or image_os == 'unknown' or image_arch == 'unknown': return None platforms.append((image_os, image_arch)) wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower()) return wanted in platforms def _bounded_docker_registry_json( response, label, max_bytes=DOCKER_REGISTRY_MANIFEST_MAX_BYTES, *, deadline=None, return_raw=False, ): max_bytes = max(1, int(max_bytes)) content_length = (getattr(response, 'headers', None) or {}).get('Content-Length') if content_length is not None: try: content_length = int(content_length) except (TypeError, ValueError) as exc: raise DockerRegistryResolutionError(f'{label} has an invalid Content-Length') from exc if content_length < 0: raise DockerRegistryResolutionError(f'{label} has an invalid Content-Length') if content_length > max_bytes: raise DockerRegistryResolutionError(f'{label} exceeds the response size limit') iterator = None iter_content = getattr(response, 'iter_content', None) if callable(iter_content): try: iterator = iter(iter_content(chunk_size=min(64 * 1024, max_bytes + 1))) except TypeError: if isinstance(response, requests.Response): raise DockerRegistryResolutionError(f'{label} cannot be streamed safely') except (requests.RequestException, OSError) as exc: raise DockerRemoteAccessError( f'{label} stream is unavailable', status='remote_transient', remote_attempted=True, ) from exc raw_content = None if iterator is not None: content = bytearray() try: for chunk in iterator: if deadline is not None and time.monotonic() >= float(deadline): raise DockerRemoteAccessError( f'{label} deadline expired', status='remote_transient', remote_attempted=True, ) if not chunk: continue if not isinstance(chunk, (bytes, bytearray)): raise DockerRegistryResolutionError(f'{label} returned invalid bytes') if len(content) + len(chunk) > max_bytes: raise DockerRegistryResolutionError(f'{label} exceeds the response size limit') content.extend(chunk) except (DockerRegistryResolutionError, DockerRemoteAccessError): raise except (requests.RequestException, OSError) as exc: raise DockerRemoteAccessError( f'{label} stream failed', status='remote_transient', remote_attempted=True, ) from exc if deadline is not None and time.monotonic() >= float(deadline): raise DockerRemoteAccessError( f'{label} deadline expired', status='remote_transient', remote_attempted=True, ) raw_content = bytes(content) if content_length is not None and len(raw_content) != content_length: raise DockerRegistryResolutionError(f'{label} Content-Length is inconsistent') try: payload = json.loads(raw_content.decode('utf-8')) except (RecursionError, UnicodeDecodeError, TypeError, ValueError) as exc: raise DockerRegistryResolutionError(f'{label} is not valid JSON') from exc else: # Lightweight response doubles used by unit tests may not implement streaming. content = getattr(response, 'content', None) if isinstance(content, (bytes, bytearray)): raw_content = bytes(content) if len(raw_content) > max_bytes: raise DockerRegistryResolutionError(f'{label} exceeds the response size limit') try: payload = response.json() except (RecursionError, TypeError, ValueError) as exc: raise DockerRegistryResolutionError(f'{label} is not valid JSON') from exc if deadline is not None and time.monotonic() >= float(deadline): raise DockerRemoteAccessError( f'{label} deadline expired', status='remote_transient', remote_attempted=True, ) if not isinstance(payload, dict): raise DockerRegistryResolutionError(f'{label} must be a JSON object') if raw_content is None: encoded = json.dumps(payload, ensure_ascii=True, separators=(',', ':')).encode('utf-8') if len(encoded) > max_bytes: raise DockerRegistryResolutionError(f'{label} exceeds the response size limit') if return_raw: if raw_content is None: raise DockerRegistryResolutionError(f'{label} raw bytes are unavailable') return payload, raw_content return payload def _docker_access_exhausted(endpoint, response=None, remote_attempted=False): retry_at = put_dockerhub_exhausted_rate_limit(endpoint, response) if retry_at: raise DockerRemoteAccessError( f'Docker {endpoint} accounts are rate-limited', status='rate_limited', retry_at=retry_at, remote_attempted=remote_attempted, ) raise DockerRemoteAccessError( f'Docker {endpoint} authentication is unavailable', status='auth_failed', remote_attempted=remote_attempted, ) def _docker_hub_access_token( account, force_refresh=False, endpoint='hub_tags', stale_token='', ): with docker_token_manager.hub_token_lock(account.name): if not docker_token_manager.account_available(account.name, endpoint): raise DockerRemoteAccessError( f'Docker {endpoint} authentication is unavailable', status='auth_failed', remote_attempted=False, ) cached = docker_token_manager.cached_hub_token(account.name) if force_refresh: if stale_token and cached and cached != stale_token: return cached elif cached: return cached docker_token_manager.invalidate_hub_token(account.name) response = api_request( 'POST', 'https://hub.docker.com/v2/auth/token', json={'identifier': account.username, 'secret': account.token}, headers={'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json'}, timeout=(5, 15), max_retries=1, retry_statuses={408, 500, 502, 503, 504}, allow_redirects=False, ) if response.status_code in (401, 403, 429): category = 'rate_limit' if response.status_code == 429 else ( 'auth_invalid' if response.status_code == 401 else 'auth_forbidden' ) docker_token_manager.report_http_status( account, endpoint, response.status_code, response, category, ) raise DockerRemoteAccessError( f'Docker Hub token endpoint returned HTTP {response.status_code}', status='rate_limited' if response.status_code == 429 else 'auth_failed', remote_attempted=True, ) if response.status_code >= 400: raise DockerRemoteAccessError( f'Docker Hub token endpoint returned HTTP {response.status_code}', remote_attempted=True, ) try: payload = _bounded_docker_registry_json( response, 'Docker Hub token response', max_bytes=1024 * 1024, ) except DockerRegistryResolutionError as exc: raise DockerRemoteAccessError( 'Docker Hub returned an invalid token response', remote_attempted=True, ) from exc token = payload.get('access_token') or payload.get('token') if not isinstance(token, str) or not token or len(token) > 16384: raise DockerRemoteAccessError( 'Docker Hub returned an invalid access token', remote_attempted=True, ) docker_token_manager.cache_hub_token( account.name, token, payload.get('expires_in') or 600, ) docker_token_manager.report_success(account, endpoint) return token def dockerhub_search_response(url, params, request_timeout=15): endpoint = 'hub_search' def request(token='', attempts=1): headers = { 'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json', } if token: headers['Authorization'] = f'Bearer {token}' return api_request( 'GET', url, params=params, headers=headers, timeout=(5, request_timeout), max_retries=attempts, retry_delay=1, retry_statuses={408, 500, 502, 503, 504}, allow_redirects=False, ) if not docker_token_manager.has_accounts(): if docker_token_manager.uses_explicit_pool(): _docker_access_exhausted(endpoint, remote_attempted=False) response = request(attempts=2) if response.status_code == 429: _docker_access_exhausted(endpoint, response, remote_attempted=True) return response excluded = set() last_response = None remote_attempted = False account = None token = '' refreshed_accounts = set() page_attempts = 0 while page_attempts < 2: if account is None: account = docker_token_manager.next_account(endpoint, excluded) if account is None: break try: token = _docker_hub_access_token(account, endpoint=endpoint) except (ApiRequestError, DockerRemoteAccessError): remote_attempted = True excluded.add(account.name) account = None continue try: response = request(token) except ApiRequestError: remote_attempted = True page_attempts += 1 if page_attempts < 2: _wait_or_raise_scan_slot_fatal(1) continue raise page_attempts += 1 remote_attempted = True last_response = response if ( response.status_code == 401 and account.name not in refreshed_accounts and page_attempts < 2 ): refreshed_accounts.add(account.name) try: token = _docker_hub_access_token( account, force_refresh=True, endpoint=endpoint, stale_token=token, ) continue except (ApiRequestError, DockerRemoteAccessError): excluded.add(account.name) account = None continue if response.status_code not in (401, 403, 429): docker_token_manager.report_success(account, endpoint) return response category = 'rate_limit' if response.status_code == 429 else 'auth_forbidden' docker_token_manager.report_http_status( account, endpoint, response.status_code, response, category, ) excluded.add(account.name) account = None token = '' _docker_access_exhausted( endpoint, last_response, remote_attempted=remote_attempted, ) def dockerhub_tags_response(url, params): endpoint = 'hub_tags' if not docker_token_manager.has_accounts(): if docker_token_manager.uses_explicit_pool(): _docker_access_exhausted(endpoint, remote_attempted=False) response = api_request( 'GET', url, params=params, headers={'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json'}, timeout=8, max_retries=1, retry_statuses={408, 500, 502, 503, 504}, allow_redirects=False, ) if response.status_code == 429: _docker_access_exhausted(endpoint, response, remote_attempted=True) return response excluded = set() last_response = None remote_attempted = False for _ in range(docker_token_manager.account_count()): account = docker_token_manager.next_account(endpoint, excluded) if account is None: break excluded.add(account.name) try: token = _docker_hub_access_token(account) except DockerRemoteAccessError: remote_attempted = True continue def request(): return api_request( 'GET', url, params=params, headers={ 'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json', 'Authorization': f'Bearer {token}', }, timeout=8, max_retries=1, retry_statuses={408, 500, 502, 503, 504}, allow_redirects=False, ) response = request() remote_attempted = True last_response = response if response.status_code == 401: try: token = _docker_hub_access_token( account, force_refresh=True, stale_token=token, ) response = request() last_response = response except DockerRemoteAccessError: continue if response.status_code not in (401, 403, 429): docker_token_manager.report_success(account, endpoint) return response if response.status_code != 403: category = 'rate_limit' if response.status_code == 429 else 'auth_forbidden' docker_token_manager.report_http_status( account, endpoint, response.status_code, response, category, ) _docker_access_exhausted( endpoint, last_response, remote_attempted=remote_attempted, ) def docker_registry_bearer_token( challenge, repo_key, excluded_accounts=None, *, deadline=None, anonymous_only=False, ): text = str(challenge or '').strip() if not text.lower().startswith('bearer '): raise DockerRegistryResolutionError('Docker registry did not provide a bearer challenge') values = { key.lower(): value for key, value in re.findall(r'([A-Za-z][A-Za-z0-9_-]*)="([^"\\]*)"', text[7:]) } realm = values.get('realm', '') parsed = urlsplit(realm) if ( parsed.scheme.lower() != 'https' or (parsed.hostname or '').lower() != 'auth.docker.io' or parsed.username is not None or parsed.password is not None or parsed.port not in (None, 443) ): raise DockerRegistryResolutionError('Docker registry bearer realm is not trusted') excluded = set(excluded_accounts or ()) authenticated = not anonymous_only and docker_token_manager.has_accounts() if ( not authenticated and not anonymous_only and docker_token_manager.uses_explicit_pool() ): _docker_access_exhausted('registry', remote_attempted=False) attempts = docker_token_manager.account_count() if authenticated else 1 last_response = None remote_attempted = False saw_auth_failure = False saw_rate_limit = False saw_target_forbidden = False saw_invalid_response = False for _ in range(max(1, attempts)): if deadline is not None and time.monotonic() >= float(deadline): raise DockerRemoteAccessError( 'Docker registry token deadline expired', status='remote_transient', remote_attempted=remote_attempted, ) account = docker_token_manager.next_account('registry', excluded) if authenticated else None if authenticated and account is None: break if account is not None: excluded.add(account.name) try: response = api_request( 'GET', realm, params={ 'service': 'registry.docker.io', 'scope': f'repository:{repo_key}:pull', }, headers={ 'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json', 'Accept-Encoding': 'identity', }, auth=(account.username, account.token) if account is not None else None, timeout=(5, 15), max_retries=1, retry_statuses={408, 500, 502, 503, 504}, allow_redirects=False, stream=True, deadline=deadline, ) except ApiRequestError as exc: raise DockerRemoteAccessError( 'Docker registry token endpoint is temporarily unavailable', status='remote_transient', remote_attempted=True, ) from exc remote_attempted = True last_response = response try: status_code = int(response.status_code) if status_code in (401, 403, 429): if status_code == 429: saw_rate_limit = True if account is not None: docker_token_manager.report_http_status( account, 'registry', status_code, response, 'rate_limit', ) elif status_code == 401: saw_auth_failure = True if account is not None: docker_token_manager.report_http_status( account, 'registry', status_code, response, 'auth_invalid', ) else: saw_target_forbidden = True continue if status_code in (408, 425) or status_code >= 500: raise DockerRemoteAccessError( 'Docker registry token endpoint is temporarily unavailable', status='remote_transient', remote_attempted=True, ) if status_code >= 400: raise DockerRemoteAccessError( 'Docker registry denied access to the requested target', status='target_forbidden', remote_attempted=True, ) try: payload = _bounded_docker_registry_json( response, 'Docker registry token response', max_bytes=DOCKER_REGISTRY_TOKEN_MAX_BYTES, deadline=deadline, ) except DockerRegistryResolutionError: saw_invalid_response = True continue finally: response.close() token = payload.get('token') or payload.get('access_token') if not isinstance(token, str) or not token or len(token) > 16384: saw_invalid_response = True continue if account is not None: docker_token_manager.report_success(account, 'registry') return DockerRegistryAuth( token=token, account_name=account.name if account is not None else '', challenge=text, ) if saw_rate_limit: retry_at = put_dockerhub_exhausted_rate_limit('registry', last_response) raise DockerRemoteAccessError( 'Docker registry accounts are rate-limited', status='rate_limited', retry_at=retry_at, remote_attempted=remote_attempted, ) if saw_target_forbidden: raise DockerRemoteAccessError( 'Docker registry denied access to the requested target', status='target_forbidden', remote_attempted=remote_attempted, ) if saw_invalid_response and not saw_auth_failure: raise DockerRemoteAccessError( 'Docker registry returned an invalid token response', status='remote_transient', remote_attempted=remote_attempted, ) _docker_access_exhausted( 'registry', last_response, remote_attempted=remote_attempted, ) def docker_registry_manifest( repo_name, digest, bearer_auth=None, *, verify_content_digest=False, return_raw=False, deadline=None, lease_renewal_callback=None, anonymous_only=False, ): repo_key = dockerhub_repo_key(repo_name) digest = normalize_docker_digest(digest) if not digest: raise DockerRegistryResolutionError('Docker manifest digest is invalid') url = f"https://registry-1.docker.io/v2/{quote(repo_key, safe='/')}/manifests/{digest}" accept = ', '.join(( 'application/vnd.oci.image.index.v1+json', 'application/vnd.docker.distribution.manifest.list.v2+json', 'application/vnd.oci.image.manifest.v1+json', 'application/vnd.docker.distribution.manifest.v2+json', )) if isinstance(bearer_auth, str): bearer_auth = DockerRegistryAuth(token=bearer_auth) bearer_auth = bearer_auth or DockerRegistryAuth(token='') def request(auth): headers = { 'User-Agent': 'GitSecretsScanner/2.0', 'Accept': accept, 'Accept-Encoding': 'identity', } if auth.token: headers['Authorization'] = f'Bearer {auth.token}' return api_request( 'GET', url, headers=headers, timeout=(5, 15), max_retries=1, retry_statuses={408, 500, 502, 503, 504}, allow_redirects=False, stream=True, deadline=deadline, ) excluded = set() attempts = 2 if anonymous_only else max( 2, docker_token_manager.account_count() + 1, ) response = None last_response = None last_status = None saw_bearer_unauthorized = False for _ in range(attempts): if lease_renewal_callback is not None: if not callable(lease_renewal_callback): raise ValueError('Docker resolver lease renewal callback is invalid') if lease_renewal_callback() is False: raise DockerResolverLeaseLostError('Docker resolver lease renewal was rejected') if deadline is not None and time.monotonic() >= float(deadline): raise DockerRemoteAccessError( 'Docker registry manifest deadline expired', status='remote_transient', remote_attempted=response is not None, ) response = request(bearer_auth) last_response = response last_status = int(response.status_code) if response.status_code not in (401, 403, 429): break challenge = ( (getattr(response, 'headers', None) or {}).get('WWW-Authenticate') or bearer_auth.challenge ) status_code = int(response.status_code) if status_code == 401 and bearer_auth.token: saw_bearer_unauthorized = True if bearer_auth.account_name: if status_code == 429: docker_token_manager.report_http_status( bearer_auth.account_name, 'registry', 429, response, 'rate_limit', ) excluded.add(bearer_auth.account_name) if status_code == 403: response.close() raise DockerRemoteAccessError( 'Docker registry denied access to the requested manifest', status='target_forbidden', remote_attempted=True, ) if status_code == 429 and ( anonymous_only or not docker_token_manager.has_accounts() ): try: _docker_access_exhausted('registry', response, remote_attempted=True) finally: response.close() response.close() response = None try: if lease_renewal_callback is not None and lease_renewal_callback() is False: raise DockerResolverLeaseLostError('Docker resolver lease renewal was rejected') bearer_auth = docker_registry_bearer_token( challenge, repo_key, excluded_accounts=excluded, deadline=deadline, anonymous_only=anonymous_only, ) except DockerRemoteAccessError as exc: if saw_bearer_unauthorized and exc.status == 'auth_failed': raise DockerRemoteAccessError( 'Docker registry denied access to the requested manifest', status='target_forbidden', remote_attempted=True, ) from exc raise except DockerRegistryResolutionError as exc: raise DockerRemoteAccessError( 'Docker registry authentication challenge is invalid', status='auth_failed' if status_code == 401 else 'remote_transient', remote_attempted=True, ) from exc if response is None: if last_status == 429: retry_at = put_dockerhub_exhausted_rate_limit('registry', last_response) raise DockerRemoteAccessError( 'Docker registry accounts are rate-limited', status='rate_limited', retry_at=retry_at, remote_attempted=True, ) if last_status == 401: raise DockerRemoteAccessError( 'Docker registry denied access to the requested manifest' if saw_bearer_unauthorized else 'Docker registry authentication is unavailable', status='target_forbidden' if saw_bearer_unauthorized else 'auth_failed', remote_attempted=True, ) raise DockerRemoteAccessError( 'Docker registry manifest request did not run', status='remote_transient', remote_attempted=last_response is not None, ) if response.status_code in (401, 429): try: _docker_access_exhausted('registry', response, remote_attempted=True) finally: response.close() status_code = int(response.status_code) if 300 <= status_code < 400: response.close() raise DockerRegistryResolutionError('Docker registry manifest redirect was rejected') try: if status_code == 404: raise DockerRegistryResolutionError('Docker registry manifest was not found') if status_code in (408, 425) or status_code >= 500: raise DockerRemoteAccessError( 'Docker registry manifest endpoint is temporarily unavailable', status='remote_transient', remote_attempted=True, ) if status_code >= 400: raise DockerRegistryResolutionError('Docker registry rejected the manifest request') returned_digest = normalize_docker_digest( (getattr(response, 'headers', None) or {}).get('Docker-Content-Digest') ) if returned_digest and returned_digest != digest: raise DockerRegistryResolutionError('Docker registry returned a different manifest digest') if verify_content_digest or return_raw: payload, raw_content = _bounded_docker_registry_json( response, 'Docker manifest', deadline=deadline, return_raw=True, ) else: payload = _bounded_docker_registry_json( response, 'Docker manifest', deadline=deadline, ) raw_content = None if verify_content_digest: calculated = 'sha256:' + hashlib.sha256(raw_content).hexdigest() if calculated != digest: raise DockerRegistryResolutionError('Docker manifest payload digest is invalid') if bearer_auth.account_name: docker_token_manager.report_success(bearer_auth.account_name, 'registry') if return_raw: return payload, bearer_auth, raw_content return payload, bearer_auth finally: response.close() def resolve_docker_layer_graph( repo_name, digest, platform_os='linux', platform_arch='amd64', bearer_auth=None, *, deadline=None, include_descriptors=False, lease_renewal_callback=None, ): manifest_digest = normalize_docker_digest(digest) if include_descriptors: payload, bearer_auth, raw_content = docker_registry_manifest( repo_name, manifest_digest, bearer_auth, deadline=deadline, verify_content_digest=True, return_raw=True, lease_renewal_callback=lease_renewal_callback, ) else: payload, bearer_auth = docker_registry_manifest( repo_name, manifest_digest, bearer_auth, deadline=deadline, lease_renewal_callback=lease_renewal_callback, ) raw_content = None descriptors = payload.get('manifests') if descriptors is not None: if not isinstance(descriptors, list) or len(descriptors) > DOCKER_REGISTRY_MAX_DESCRIPTORS: raise DockerRegistryResolutionError('Docker manifest index has invalid descriptor bounds') wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower()) descriptor = next(( item for item in descriptors if isinstance(item, dict) and isinstance(item.get('platform'), dict) and ( str(item['platform'].get('os') or '').lower(), str(item['platform'].get('architecture') or '').lower(), ) == wanted ), None) if descriptor is None: return None, bearer_auth manifest_digest = normalize_docker_digest(descriptor.get('digest')) if not manifest_digest: raise DockerRegistryResolutionError('Docker platform descriptor has an invalid digest') if include_descriptors: payload, bearer_auth, raw_content = docker_registry_manifest( repo_name, manifest_digest, bearer_auth, deadline=deadline, verify_content_digest=True, return_raw=True, lease_renewal_callback=lease_renewal_callback, ) else: payload, bearer_auth = docker_registry_manifest( repo_name, manifest_digest, bearer_auth, deadline=deadline, lease_renewal_callback=lease_renewal_callback, ) layers = payload.get('layers') if not isinstance(layers, list) or len(layers) > DOCKER_REGISTRY_MAX_LAYERS: raise DockerRegistryResolutionError('Docker manifest has invalid layer bounds') ordered_layers = [] layer_descriptors = [] for position, layer in enumerate(layers, 1): layer_digest = normalize_docker_digest(layer.get('digest')) if isinstance(layer, dict) else '' if not layer_digest: raise DockerRegistryResolutionError('Docker manifest contains an invalid layer digest') ordered_layers.append(layer_digest) if include_descriptors: layer_descriptors.append(_docker_content_descriptor(layer, 'layer', position)) graph = { 'manifest_digest': manifest_digest, 'layers': tuple(ordered_layers), } if include_descriptors: config = _docker_content_descriptor(payload.get('config'), 'config', 0) manifest_media_type = str( payload.get('mediaType') or 'application/vnd.docker.distribution.manifest.v2+json' ).strip().lower() if not manifest_media_type or len(manifest_media_type) > 256: raise DockerRegistryResolutionError('Docker manifest media type is invalid') graph.update({ 'manifest_media_type': manifest_media_type, 'manifest_size_bytes': len(raw_content), 'config_digest': config['digest'], 'layer_descriptors': tuple(layer_descriptors), }) return graph, bearer_auth def _dockerhub_manifest_target_parts(target): parsed = parse_docker_target(target) image = str(parsed['image']).lower() image_name, manifest_digest = image.rsplit('@', 1) manifest_digest = normalize_docker_digest(manifest_digest) if not manifest_digest: raise DockerRegistryResolutionError('Docker manifest target digest is invalid') parts = image_name.split('/') if len(parts) > 1 and ('.' in parts[0] or ':' in parts[0] or parts[0] == 'localhost'): registry = parts.pop(0) if registry not in ('docker.io', 'index.docker.io', 'registry-1.docker.io'): raise DockerRegistryResolutionError( 'Docker layer scanning only supports Docker Hub targets' ) if not parts or any(not part for part in parts): raise DockerRegistryResolutionError('Docker Hub repository is invalid') repository = '/'.join(parts) registry_repository = repository if '/' in repository else f'library/{repository}' return image, repository, registry_repository, manifest_digest def _docker_content_descriptor(value, kind, position): if not isinstance(value, dict): raise DockerRegistryResolutionError(f'Docker {kind} descriptor is invalid') digest = normalize_docker_digest(value.get('digest')) size = value.get('size') media_type = str(value.get('mediaType') or '').strip().lower() if ( not digest or isinstance(size, bool) or not isinstance(size, int) or size < 0 or size > 1024 * 1024 * 1024 * 1024 or not media_type or len(media_type) > 256 ): raise DockerRegistryResolutionError(f'Docker {kind} descriptor has invalid bounds') return { 'digest': digest, 'size': size, 'media_type': media_type, } def resolve_docker_content_manifest( target, platform_os='linux', platform_arch='amd64', bearer_auth=None, *, deadline=None, anonymous_only=False, ): image, repository, registry_repository, manifest_digest = ( _dockerhub_manifest_target_parts(target) ) target_manifest_digest = manifest_digest payload, bearer_auth = docker_registry_manifest( registry_repository, manifest_digest, bearer_auth, verify_content_digest=True, deadline=deadline, anonymous_only=anonymous_only, ) descriptors = payload.get('manifests') if descriptors is not None: if not isinstance(descriptors, list) or len(descriptors) > DOCKER_REGISTRY_MAX_DESCRIPTORS: raise DockerRegistryResolutionError('Docker manifest index has invalid descriptor bounds') wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower()) child = next(( item for item in descriptors if isinstance(item, dict) and isinstance(item.get('platform'), dict) and ( str(item['platform'].get('os') or '').lower(), str(item['platform'].get('architecture') or '').lower(), ) == wanted ), None) if child is None: raise DockerRegistryResolutionError('Docker target platform manifest is unavailable') manifest_digest = normalize_docker_digest(child.get('digest')) if not manifest_digest: raise DockerRegistryResolutionError('Docker platform descriptor digest is invalid') payload, bearer_auth = docker_registry_manifest( registry_repository, manifest_digest, bearer_auth, verify_content_digest=True, deadline=deadline, anonymous_only=anonymous_only, ) config = _docker_content_descriptor(payload.get('config'), 'config', 0) layers = payload.get('layers') if not isinstance(layers, list) or len(layers) > DOCKER_REGISTRY_MAX_LAYERS: raise DockerRegistryResolutionError('Docker manifest has invalid layer bounds') normalized_layers = [ _docker_content_descriptor(layer, 'layer', position) for position, layer in enumerate(layers, 1) ] manifest_media_type = str( payload.get('mediaType') or 'application/vnd.docker.distribution.manifest.v2+json' ).strip().lower() if not manifest_media_type or len(manifest_media_type) > 256: raise DockerRegistryResolutionError('Docker manifest media type is invalid') if deadline is not None and time.monotonic() >= float(deadline): raise DockerRemoteAccessError( 'Docker registry manifest deadline expired', status='remote_transient', remote_attempted=True, ) return { 'version': 1, 'image': image, 'repository': registry_repository, 'manifest_digest': target_manifest_digest, 'platform_os': str(platform_os or 'linux').lower(), 'platform_arch': str(platform_arch or 'amd64').lower(), 'manifest_media_type': manifest_media_type, 'config': config, 'layers': normalized_layers, }, bearer_auth DOCKER_BLOB_REDIRECT_SUFFIXES = ( '.docker.com', '.docker.io', '.cloudfront.net', '.cloudflarestorage.com', '.amazonaws.com', ) def _docker_blob_url_validation_error(url, *, registry_origin=False): try: parsed = urlsplit(str(url or '')) hostname = (parsed.hostname or '').lower().rstrip('.') port = parsed.port except ValueError: return 'invalid_url' if ( parsed.scheme.lower() != 'https' or not hostname or parsed.username is not None or parsed.password is not None or port not in (None, 443) or parsed.fragment ): return 'invalid_url' if registry_origin: return '' if hostname == 'registry-1.docker.io' and not parsed.query else 'invalid_registry' try: address = ipaddress.ip_address(hostname) except ValueError: address = None if address is not None and not address.is_global: return 'non_global_address' if not any(hostname.endswith(suffix) for suffix in DOCKER_BLOB_REDIRECT_SUFFIXES): return 'untrusted_host' try: answers = socket.getaddrinfo( hostname, 443, type=socket.SOCK_STREAM, proto=socket.IPPROTO_TCP, ) except OSError: return 'dns_unavailable' if not answers: return 'dns_unavailable' for answer in answers: try: resolved = ipaddress.ip_address(str(answer[4][0]).split('%', 1)[0]) except (IndexError, TypeError, ValueError): return 'invalid_dns_answer' if not resolved.is_global: return 'non_global_address' return '' def _docker_blob_url_allowed(url, *, registry_origin=False): return not _docker_blob_url_validation_error(url, registry_origin=registry_origin) def _require_docker_blob_url(url, *, registry_origin=False): error = _docker_blob_url_validation_error(url, registry_origin=registry_origin) if error == 'dns_unavailable': raise DockerLayerInfrastructureError( 'remote_dns', 'Docker blob redirect DNS is temporarily unavailable', category='remote_transient', ) if error: raise DockerContentTransferError( 'unsafe_redirect' if not registry_origin else 'unsafe_registry_url', 'Docker blob URL is not trusted', False, ) def _docker_registry_blob_response( repository, digest, bearer_auth, deadline, *, anonymous_only=False, ): repo_key = dockerhub_repo_key(repository) digest = normalize_docker_digest(digest) if not digest: raise DockerContentTransferError('invalid_descriptor', 'Docker blob digest is invalid', False) url = f'https://registry-1.docker.io/v2/{quote(repo_key, safe="/")}/blobs/{digest}' _require_docker_blob_url(url, registry_origin=True) if isinstance(bearer_auth, str): bearer_auth = DockerRegistryAuth(token=bearer_auth) bearer_auth = bearer_auth or DockerRegistryAuth(token='') excluded = set() attempts = 2 if anonymous_only else max( 2, docker_token_manager.account_count() + 1, ) response = None saw_rate_limit = False saw_bearer_unauthorized = False for _ in range(attempts): remaining = max(0.0, float(deadline) - time.monotonic()) if remaining <= 0: raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired') headers = { 'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/octet-stream', 'Accept-Encoding': 'identity', } if bearer_auth.token: headers['Authorization'] = f'Bearer {bearer_auth.token}' try: response = api_request( 'GET', url, headers=headers, timeout=(5, min(30, remaining)), use_proxy=False, max_retries=1, retry_statuses={408, 500, 502, 503, 504}, allow_redirects=False, stream=True, deadline=deadline, ) except ApiRequestError as exc: if time.monotonic() >= float(deadline): raise DockerContentTransferError( 'transfer_timeout', 'Docker blob deadline expired', ) from exc raise DockerLayerInfrastructureError( 'remote_transient', 'Docker blob endpoint is temporarily unavailable', category='remote_transient', ) from exc if response.status_code not in (401, 429): return response, bearer_auth challenge = ( (getattr(response, 'headers', None) or {}).get('WWW-Authenticate') or bearer_auth.challenge ) if response.status_code == 401 and bearer_auth.token: saw_bearer_unauthorized = True if bearer_auth.account_name: if response.status_code == 429: saw_rate_limit = True docker_token_manager.report_http_status( bearer_auth.account_name, 'registry', 429, response, 'rate_limit', ) excluded.add(bearer_auth.account_name) elif response.status_code == 429: saw_rate_limit = True response.close() response = None try: bearer_auth = docker_registry_bearer_token( challenge, repo_key, excluded_accounts=excluded, deadline=deadline, anonymous_only=anonymous_only, ) except DockerRemoteAccessError as exc: if exc.status == 'target_forbidden': raise DockerContentTransferError( 'target_forbidden', 'Docker blob target is forbidden', False, ) from exc if exc.status == 'rate_limited': raise DockerLayerInfrastructureError( 'remote_rate_limit', 'Docker blob authorization is rate-limited', category='docker_rate_limit', ) from exc if exc.status == 'auth_failed': if saw_bearer_unauthorized: raise DockerContentTransferError( 'target_forbidden', 'Docker blob target is forbidden', False, ) from exc raise DockerLayerInfrastructureError( 'remote_auth', 'Docker blob authorization is unavailable', category='docker_auth', auth_related=True, ) from exc raise DockerLayerInfrastructureError( 'remote_transient', 'Docker blob authorization is temporarily unavailable', category='remote_transient', ) from exc except DockerRegistryResolutionError as exc: raise DockerLayerInfrastructureError( 'remote_auth', 'Docker blob authentication challenge is invalid', category='docker_auth', auth_related=True, ) from exc if response is not None: response.close() if saw_rate_limit: raise DockerLayerInfrastructureError( 'remote_rate_limit', 'Docker blob authorization is rate-limited', category='docker_rate_limit', ) if saw_bearer_unauthorized: raise DockerContentTransferError( 'target_forbidden', 'Docker blob target is forbidden', False, ) raise DockerLayerInfrastructureError( 'remote_auth', 'Docker blob authorization is unavailable', category='docker_auth', auth_related=True, ) def stream_docker_registry_blob( repository, descriptor, destination, bearer_auth=None, *, deadline, min_free_bytes=0, redirect_limit=5, anonymous_only=False, ): digest = normalize_docker_digest((descriptor or {}).get('digest')) declared_bytes = (descriptor or {}).get('size') kind = str((descriptor or {}).get('kind') or '') media_type = str((descriptor or {}).get('media_type') or '').strip().lower() if ( not digest or isinstance(declared_bytes, bool) or not isinstance(declared_bytes, int) or declared_bytes < 0 or declared_bytes > 1024 * 1024 * 1024 * 1024 or kind not in ('config', 'layer') or not media_type ): raise DockerContentTransferError('invalid_descriptor', 'Docker blob descriptor is invalid', False) supported_media = DOCKER_CONFIG_MEDIA_TYPES if kind == 'config' else DOCKER_LAYER_MEDIA_TYPES if media_type not in supported_media: raise DockerContentTransferError( 'unsupported_media_type', 'Docker blob media type is unsupported', False, ) deadline = float(deadline) if deadline <= time.monotonic(): raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired') destination = os.path.abspath(destination) try: parent = require_private_directory(os.path.dirname(destination), create=False) reject_reparse_components(parent) except (OSError, ValueError) as exc: raise DockerLayerInfrastructureError( 'private_storage', 'Docker blob private storage is unavailable', category='source_resource', ) from exc if os.path.lexists(destination): raise DockerLayerInfrastructureError( 'destination_exists', 'Docker blob destination is not available', category='source_resource', ) required_free = max(0, int(min_free_bytes)) + declared_bytes try: free_bytes = shutil.disk_usage(parent).free except OSError as exc: raise DockerLayerInfrastructureError( 'disk_reserve', 'Docker blob free space cannot be verified', category='source_resource', ) from exc if free_bytes < required_free: raise DockerLayerInfrastructureError( 'disk_reserve', 'Docker blob would violate the free-space reserve', category='source_resource', ) started = time.monotonic() response = None total = 0 temporary = ( f'{destination}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.partial' ) published = False try: response, bearer_auth = _docker_registry_blob_response( repository, digest, bearer_auth, deadline, anonymous_only=anonymous_only, ) redirects = 0 while response.status_code in (301, 302, 303, 307, 308): location = (getattr(response, 'headers', None) or {}).get('Location') current_url = str(getattr(response, 'url', '') or '') response.close() response = None redirects += 1 if not location or redirects > max(0, min(5, int(redirect_limit))): raise DockerContentTransferError('unsafe_redirect', 'Docker blob redirect limit exceeded', False) next_url = urljoin(current_url, location) _require_docker_blob_url(next_url) remaining = max(0.0, deadline - time.monotonic()) if remaining <= 0: raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired') response = api_request( 'GET', next_url, use_proxy=False, headers={ 'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/octet-stream', 'Accept-Encoding': 'identity', }, timeout=(5, min(30, remaining)), max_retries=1, retry_statuses={408, 500, 502, 503, 504}, allow_redirects=False, stream=True, deadline=deadline, ) status_code = int(response.status_code) if status_code == 401: raise DockerContentTransferError( 'target_forbidden', 'Docker blob target is forbidden', False, ) if status_code == 403: raise DockerContentTransferError( 'target_forbidden', 'Docker blob target is forbidden', False, ) if status_code == 404: raise DockerContentTransferError( 'blob_not_found', 'Docker blob target is unavailable', False, ) if status_code == 429: raise DockerLayerInfrastructureError( 'remote_rate_limit', 'Docker blob endpoint is rate-limited', category='docker_rate_limit', ) if status_code in (408, 425) or status_code >= 500: raise DockerLayerInfrastructureError( 'remote_transient', 'Docker blob endpoint is temporarily unavailable', category='remote_transient', ) if status_code >= 400: raise DockerContentTransferError( 'target_rejected', 'Docker blob target was rejected', False, ) content_encoding = str( (getattr(response, 'headers', None) or {}).get('Content-Encoding') or '' ).strip().lower() if content_encoding not in ('', 'identity'): raise DockerContentTransferError( 'content_encoding', 'Docker blob response changed the content encoding', ) content_length = (getattr(response, 'headers', None) or {}).get('Content-Length') try: content_length = int(content_length) except (TypeError, ValueError) as exc: raise DockerContentTransferError( 'size_mismatch', 'Docker blob response lacks an exact Content-Length', False, ) from exc if content_length != declared_bytes: raise DockerContentTransferError( 'size_mismatch', 'Docker blob Content-Length differs from its descriptor', False, ) flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0) descriptor_fd = os.open(temporary, flags, 0o600) os.close(descriptor_fd) harden_private_file(temporary) digest_hash = hashlib.sha256() with open(temporary, 'wb', buffering=0) as output: for chunk in response.iter_content(chunk_size=1024 * 1024): _raise_if_scan_slot_fatal() if time.monotonic() >= deadline: raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired') if not chunk: continue total += len(chunk) if total > declared_bytes: raise DockerContentTransferError( 'size_mismatch', 'Docker blob exceeded its declared size', False, ) if shutil.disk_usage(parent).free < max(0, int(min_free_bytes)): raise DockerLayerInfrastructureError( 'disk_reserve', 'Docker blob transfer reached the free-space reserve', category='source_resource', ) digest_hash.update(chunk) output.write(chunk) output.flush() os.fsync(output.fileno()) if time.monotonic() >= deadline: raise DockerContentTransferError( 'transfer_timeout', 'Docker blob deadline expired', ) if total != declared_bytes: raise DockerContentTransferError( 'size_mismatch', 'Docker blob byte count differs from its descriptor', False, ) if f'sha256:{digest_hash.hexdigest()}' != digest: raise DockerContentTransferError( 'digest_mismatch', 'Docker blob SHA-256 differs from its descriptor', False, ) harden_private_file(temporary) durable_replace(temporary, destination) if not private_file_ready(destination): raise DockerLayerInfrastructureError( 'private_file_lost', 'Docker blob lost its private file identity', category='source_resource', ) if time.monotonic() >= deadline: raise DockerContentTransferError( 'transfer_timeout', 'Docker blob deadline expired', ) published = True duration_ms = max(0, int((time.monotonic() - started) * 1000)) return DockerBlobDownloadOutcome( path=destination, verified_bytes=total, transfer_bytes=total, duration_ms=duration_ms, bearer_auth=bearer_auth, ) except DockerContentTransferError as exc: exc.transfer_bytes = min(declared_bytes, max(0, int(total))) exc.duration_ms = max(0, int((time.monotonic() - started) * 1000)) raise except (ApiRequestError, requests.RequestException) as exc: if time.monotonic() >= deadline: raise DockerContentTransferError( 'transfer_timeout', 'Docker blob deadline expired', ) from exc raise DockerLayerInfrastructureError( 'remote_transient', 'Docker blob transfer is temporarily unavailable', category='remote_transient', ) from exc except OSError as exc: raise DockerLayerInfrastructureError( 'private_storage', 'Docker blob private storage failed', category='source_resource', ) from exc finally: if response is not None: response.close() if os.path.lexists(temporary): durable_unlink(temporary) if not published and os.path.lexists(destination): durable_unlink(destination) def fetch_docker_config_payload_classes( resolved, bearer_auth=None, *, deadline, min_free_bytes=0, ): layers = list((resolved or {}).get('layers') or ()) fallback = ['unknown'] * len(layers) config = dict((resolved or {}).get('config') or {}) config.update({'kind': 'config', 'position': 0}) if ( config.get('media_type') not in DOCKER_CONFIG_MEDIA_TYPES or not isinstance(config.get('size'), int) or config['size'] < 0 or config['size'] > DOCKER_REGISTRY_MANIFEST_MAX_BYTES ): return fallback, bearer_auth work_root = None destination = None try: work_root = tempfile.mkdtemp(prefix='docker-history-', dir=get_work_dir()) harden_private_directory(work_root) write_temp_owner(work_root, ['docker-config-history'], os.getpid(), required=True) destination = os.path.join(work_root, 'config.json') outcome = stream_docker_registry_blob( resolved['repository'], config, destination, bearer_auth, deadline=float(deadline), min_free_bytes=max(0, int(min_free_bytes or 0)), ) try: parsed = validate_docker_content_artifact(destination, config) except DockerContentScanError: return fallback, outcome.bearer_auth return docker_config_payload_classes(parsed, len(layers)), outcome.bearer_auth except DockerContentScanError: return fallback, bearer_auth finally: if destination and os.path.lexists(destination): durable_unlink(destination) if work_root: cleanup_command_work_dir(work_root) def _docker_depth_selection_evidence( record, candidate_count, selector_version=DOCKER_DEPTH_SELECTOR_VERSION, ): evidence = { 'schema': 1, 'type': 'docker-depth-selection-evidence-v1', 'selector_version': selector_version, 'selector_sha256': canonical_selector_hash(selector_version), 'candidate_distinct_graph_count': int(candidate_count), 'image_rank': int(record['image_rank']), 'selection_reason': str(record['selection_reason']), 'target': str(record['target']), 'repository': str(record['repository']), 'manifest_digest': str(record['manifest_digest']), 'manifest_media_type': str(record['manifest_media_type']), 'manifest_size_bytes': int(record['manifest_size_bytes']), 'config_digest': str(record['config_digest']), 'graph_sha256': str(record['graph_sha256']), 'layers': [dict(layer) for layer in record['layer_metadata']], } return { **record, 'candidate_distinct_graph_count': int(candidate_count), 'selection_evidence_sha256': canonical_docker_depth_selection_evidence_hash( evidence ), } def dockerhub_tag_digest(tag, platform_filter_enabled=False, platform_os='linux', platform_arch='amd64'): if not isinstance(tag, dict): return '' images = tag.get('images') if isinstance(tag.get('images'), list) else [] wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower()) candidates = ( image.get('digest') for image in images if isinstance(image, dict) and (str(image.get('os') or '').lower(), str(image.get('architecture') or '').lower()) == wanted ) platform_digest = next(( digest for digest in (normalize_docker_digest(value) for value in candidates) if digest ), '') return platform_digest or normalize_docker_digest(tag.get('digest')) def fetch_dockerhub_tags( repo_name, since=None, limit=1, retry_count=2, retry_delay=5, platform_filter_enabled=False, platform_os='linux', platform_arch='amd64', platform_candidate_tags=20, return_status=False, *, return_outcome=False, fresh_graph_evidence=False, lease_renewal_callback=None, selector_version=DOCKER_DEPTH_SELECTOR_VERSION, ): def output( tags, status, remote_attempted=False, retry_at=None, error='', selection_records=(), candidate_records=(), candidate_distinct_graph_count=0, ): tags = list(tags or []) if return_outcome: return DockerTagResolutionOutcome( tags=tuple(tags), status=str(status), remote_attempted=bool(remote_attempted), retry_at=retry_at, error=str(error or '')[:500], selection_records=tuple(selection_records or ()), candidate_records=tuple(candidate_records or selection_records or ()), selector_version=selector_version, selector_hash=canonical_selector_hash(selector_version), candidate_distinct_graph_count=int(candidate_distinct_graph_count or 0), fresh_graph_evidence=bool( fresh_graph_evidence and remote_attempted and docker_tag_resolution_is_conclusive(status) ), cache_bypassed=bool(fresh_graph_evidence), ) return (tags, status) if return_status else tags def cache_write(*values, **kwargs): if not fresh_graph_evidence: put_dockerhub_tag_cache(*values, **kwargs) try: retry_count = max(0, min(5, int(retry_count or 0))) except (TypeError, ValueError): retry_count = 2 try: retry_delay = max(0, min(60, int(retry_delay or 0))) except (TypeError, ValueError): retry_delay = 5 limit = docker_images_per_repository_limit(limit) repo_name = str(repo_name or '').strip() if '@' in repo_name: try: return output([parse_docker_target(repo_name)['target']], 'ok') except (TypeError, ValueError): return output([], 'unknown') if ':' in repo_name.rsplit('/', 1)[-1]: return output([], 'unknown') platform_variant = ( f'{selector_version}:{str(platform_os).lower()}/{str(platform_arch).lower()}:' f'filter={int(bool(platform_filter_enabled))}:candidates={int(platform_candidate_tags or 0)}' ) cached = None if fresh_graph_evidence else get_dockerhub_tag_cache( repo_name, since, limit, platform_variant, return_status=True, ) if cached is not None: return output(cached[0], cached[1]) rate_limit_state = dockerhub_tag_rate_limit_state() if rate_limit_state['active']: logger.info(f"Docker Hub tag API is rate-limited; deferring tag fetch for {repo_name} from cache state") return output( [], 'global_cooldown', remote_attempted=False, retry_at=rate_limit_state['retry_at'], error='Docker Hub shared rate-limit cooldown is active', ) if '/' in repo_name: namespace, name = repo_name.split('/', 1) else: namespace, name = 'library', repo_name url = ( f"https://hub.docker.com/v2/namespaces/{quote(namespace, safe='')}" f"/repositories/{quote(name, safe='')}/tags" ) last_error = None for attempt in range(retry_count + 1): _raise_if_scan_slot_fatal() try: if lease_renewal_callback is not None: if not callable(lease_renewal_callback): raise ValueError('Docker resolver lease renewal callback is invalid') if lease_renewal_callback() is False: raise DockerResolverLeaseLostError( 'Docker resolver lease renewal was rejected' ) response = dockerhub_tags_response( url, { 'page_size': max( 1, min(max(limit, int(platform_candidate_tags or 0)), 100), ), }, ) if response.status_code == 404: cache_write( repo_name, since, limit, 'not_found', [], getattr(scan_config, 'dockerhub_tag_negative_cache_ttl_sec', 3600), 'Docker Hub repository not found', platform_variant, ) return output([], 'not_found', remote_attempted=True) if response.status_code in (401, 403): raise DockerRemoteAccessError( f'Docker Hub tags endpoint returned HTTP {response.status_code}', status='auth_failed', remote_attempted=True, ) response.raise_for_status() graph_candidates = [] supported_or_unknown = 0 unsupported = 0 unresolved_digest = 0 resolution_status = '' resolution_retry_at = None resolution_error = '' tag_payload = _bounded_docker_registry_json( response, 'Docker Hub tags response', ) if not isinstance(tag_payload, dict) or not isinstance(tag_payload.get('results'), list): raise DockerRegistryResolutionError('Docker Hub tags response is malformed') registry_auth = DockerRegistryAuth(token='') for source_index, tag in enumerate(tag_payload.get('results', [])): _raise_if_scan_slot_fatal() tag_name = tag.get('name') if not tag_name: continue revision = str(tag.get('last_updated') or tag.get('tag_last_pushed') or '').strip() tag_updated = parse_dockerhub_datetime( revision ) if since and tag_updated and tag_updated < since: continue platform_support = docker_tag_platform_support(tag, platform_os, platform_arch) if platform_filter_enabled else True if platform_support is False: unsupported += 1 logger.info( f"Skipping Docker tag {repo_name}:{tag_name}: no {platform_os}/{platform_arch} image" ) continue supported_or_unknown += 1 tagged_image = f"{repo_name}:{tag_name}" try: validate_docker_image_reference(tagged_image, require_digest=False) except ValueError: logger.warning('Skipping invalid Docker Hub image reference for %s tag %s', repo_name, tag_name) continue digest = dockerhub_tag_digest(tag, platform_filter_enabled, platform_os, platform_arch) if not digest: unresolved_digest += 1 logger.warning('Deferring Docker tag without a valid content digest: %s', tagged_image) continue try: graph_kwargs = ( {'include_descriptors': True} if fresh_graph_evidence else {} ) if lease_renewal_callback is not None: graph_kwargs['lease_renewal_callback'] = lease_renewal_callback graph, registry_auth = resolve_docker_layer_graph( repo_name, digest, platform_os, platform_arch, registry_auth, **graph_kwargs, ) except DockerResolverLeaseLostError: raise except ScanSlotFatalError: raise except DockerRemoteAccessError as exc: unresolved_digest += 1 resolution_status = exc.status resolution_retry_at = exc.retry_at resolution_error = str(exc) logger.warning( 'Deferring Docker manifest graph for %s: %s', tagged_image, str(exc)[:300], ) break except Exception as exc: unresolved_digest += 1 logger.warning( 'Deferring Docker manifest graph for %s: %s', tagged_image, str(exc)[:300], ) if isinstance(exc, ApiRequestError) or 'rate-limit' in str(exc).lower(): break continue if graph is None: unsupported += 1 continue target = validate_docker_image_reference( f"{repo_name}@{graph['manifest_digest']}" ) graph_candidates.append({ 'name': tag_name, 'target': target, 'repository': repo_name.lower(), 'manifest_digest': graph['manifest_digest'], **({ 'manifest_media_type': graph['manifest_media_type'], 'manifest_size_bytes': graph['manifest_size_bytes'], 'config_digest': graph['config_digest'], 'layer_descriptors': graph['layer_descriptors'], } if fresh_graph_evidence else {}), 'layers': graph['layers'], 'source_index': source_index, 'updated_at': ( tag_updated.replace(tzinfo=timezone.utc).timestamp() if tag_updated is not None and tag_updated.tzinfo is None else tag_updated.timestamp() if tag_updated is not None else None ), }) candidate_distinct_graph_count = len({ tuple(candidate['layers']) for candidate in graph_candidates }) candidate_records = ( select_docker_layer_graphs( graph_candidates, min(100, candidate_distinct_graph_count), replacement_pool=True, ) if fresh_graph_evidence and candidate_distinct_graph_count else () ) selected = ( candidate_records[:limit] if fresh_graph_evidence else select_docker_layer_graphs(graph_candidates, limit) ) if fresh_graph_evidence: candidate_records = [ _docker_depth_selection_evidence( record, candidate_distinct_graph_count, selector_version, ) for record in candidate_records ] selected = candidate_records[:limit] tags = [record['target'] for record in selected] ttl = getattr(scan_config, 'dockerhub_tag_cache_ttl_sec', 21600) if tags else getattr(scan_config, 'dockerhub_tag_negative_cache_ttl_sec', 3600) status = ( 'partial' if tags and unresolved_digest else resolution_status if resolution_status else 'ok' if tags else 'unknown' if unresolved_digest else 'unsupported' if unsupported and not graph_candidates else 'empty' ) if status != 'unknown': if status == 'ok': cache_write( repo_name, since, limit, status, tags, ttl, platform_variant=platform_variant, tag_records=selected, ) elif status not in ('partial',): cache_write( repo_name, since, limit, status, tags, ttl, platform_variant=platform_variant, ) return output( tags, status, remote_attempted=True, retry_at=resolution_retry_at, error=resolution_error, selection_records=selected, candidate_records=candidate_records, candidate_distinct_graph_count=candidate_distinct_graph_count, ) except DockerResolverLeaseLostError: raise except ScanSlotFatalError: raise except DockerRemoteAccessError as exc: logger.warning('Deferring Docker tag resolution for %s: %s', repo_name, str(exc)) return output( [], exc.status, remote_attempted=exc.remote_attempted, retry_at=exc.retry_at, error=str(exc), ) except Exception as e: last_error = str(e) if '404' in last_error or 'not found' in last_error.lower(): cache_write(repo_name, since, limit, 'not_found', [], getattr(scan_config, 'dockerhub_tag_negative_cache_ttl_sec', 3600), last_error, platform_variant) return output([], 'not_found', remote_attempted=True) if attempt < retry_count and isinstance(e, ApiRequestError): if lease_renewal_callback is not None and lease_renewal_callback() is False: raise DockerResolverLeaseLostError( 'Docker resolver lease renewal was rejected' ) _wait_or_raise_scan_slot_fatal(min(300, retry_delay * (attempt + 1))) continue break logger.warning(f"Unable to fetch Docker Hub tags for {repo_name}: {last_error}") return output( [], 'unknown', remote_attempted=True, error=last_error or 'Docker tag resolution failed', ) def resolve_recent_dockerhub_image( image, since, platform_filter_enabled=False, platform_os='linux', platform_arch='amd64', platform_candidate_tags=20, images_per_repository=1, resolve_tags=True, ): repo_name = image.get('repo_name') if not repo_name: return [], 'missing_date' last_updated = parse_dockerhub_datetime( image.get('last_updated') or image.get('last_modified') ) if last_updated and last_updated < since: return [], 'old' if last_updated is None: last_updated = fetch_dockerhub_last_updated(repo_name) if last_updated is None: return [repo_name], 'recent' if last_updated < since: return [], 'old' if not resolve_tags: return [repo_name], 'recent' tags, tag_status = fetch_dockerhub_tags( repo_name, since=since, limit=docker_images_per_repository_limit(images_per_repository), platform_filter_enabled=platform_filter_enabled, platform_os=platform_os, platform_arch=platform_arch, platform_candidate_tags=platform_candidate_tags, return_status=True, ) if tags: if tag_status != 'ok': tags.append(repo_name) return tags, 'recent' if not docker_tag_resolution_is_conclusive(tag_status): return [repo_name], 'recent' return [], 'missing_date' def fetch_recent_dockerhub_images( query, since, per_page=100, pages=1, platform_filter_enabled=False, platform_os='linux', platform_arch='amd64', platform_candidate_tags=20, images_per_repository=1, resolve_tags=True, ): """Fetch Docker Hub images updated since a specific timestamp""" images = [] page = 1 per_page = max(1, int(per_page or 1)) requested_pages, pages = dockerhub_search_page_window(pages) if requested_pages > pages: logger.info( f'Docker Hub search is limited to {pages} accessible page(s); ' f'capping requested pages from {requested_pages}' ) logger.info(f"Fetching recent Docker Hub images updated since {since.strftime('%Y-%m-%d')}...") expected_pages = pages while page <= expected_pages: try: logger.info(f"Docker Hub query '{query}': fetching page {page}/{pages}...") page_result = fetch_dockerhub_search_page( query, page, per_page=per_page, request_timeout=30, ) if page == 1: expected_pages = min( pages, max(1, (page_result['total_count'] + per_page - 1) // per_page), ) repositories = page_result['repositories'] if not repositories: logger.info( f"Docker Hub query '{query}' returned no results " f"(total matches: {page_result['total_count']})." ) break logger.info(f"Page {page}: checking dates/tags for {len(repositories)} Docker Hub repositories...") new_images = [] old_images = 0 missing_dates = 0 with concurrent.futures.ThreadPoolExecutor(max_workers=min(8, len(repositories))) as executor: futures = [ executor.submit( resolve_recent_dockerhub_image, image, since, platform_filter_enabled, platform_os, platform_arch, platform_candidate_tags, docker_images_per_repository_limit(images_per_repository), resolve_tags, ) for image in repositories ] for future in concurrent.futures.as_completed(futures): tags, status = future.result() if status == 'recent': new_images.extend(tags) elif status == 'old': old_images += 1 else: missing_dates += 1 images.extend(new_images) logger.info(f"Page {page}: fetched {len(new_images)} recent images, skipped {old_images} older images, skipped {missing_dates} without dates") page += 1 except DockerHubDiscoveryTransportError: raise except Exception as e: logger.error(f"Docker Hub recent discovery page {page} failed after bounded attempts") raise DockerHubDiscoveryTransportError( f'Docker Hub recent discovery page {page} failed after bounded attempts' ) from e return images def huggingface_space_to_target(space): target = { 'url': space.get('id') or space.get('name') or '', 'name': space.get('id') or space.get('name') or '', 'created_at': space.get('createdAt') or space.get('created_at') or '', 'updated_at': space.get('lastModified') or space.get('updatedAt') or space.get('updated_at') or '', } for field in ('private', 'protected', 'gated', 'disabled'): if field in space: target[field] = space[field] return target def fetch_huggingface_spaces(pages=1, token=None, request_timeout=15, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, return_metadata=False, request_attempts=1, retry_delay=0): spaces = [] headers = {'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json'} if token: headers['Authorization'] = f'Bearer {token}' seen_pages = 0 request_attempts = max(1, int(request_attempts or 1)) retry_delay = max(0, int(retry_delay or 0)) request_budget = float(request_timeout) * request_attempts + retry_delay * (request_attempts - 1) if return_metadata: next_url = 'https://huggingface.co/api/spaces' next_params = {'sort': 'lastModified', 'direction': '-1', 'limit': 100} logger.info(f"Fetching newest-modified HuggingFace Spaces for {pages} page(s)...") for page in range(max(1, int(pages or 1))): try: response = api_request( 'GET', next_url, headers=headers, params=next_params, timeout=request_timeout, max_retries=request_attempts, retry_delay=retry_delay, deadline=time.monotonic() + request_budget, ) if response.status_code >= 400: status = response.status_code category = 'auth_invalid' if status == 401 else 'auth_forbidden' if status == 403 else 'rate_limit' if status == 429 else 'server_error' if status >= 500 else 'api' reset_at = retry_after_reset(response) response.close() raise RateLimitError( 'huggingface', f'HuggingFace discovery HTTP {status}', reset_at=reset_at, category=category, auth_related=status in (401, 403, 429), ) response.raise_for_status() data = response.json() if not isinstance(data, list): raise ValueError('invalid HuggingFace spaces payload') page_spaces = [ huggingface_space_to_target(space) for space in data if isinstance(space, dict) and space.get('id') ] if not page_spaces: logger.info(f"HuggingFace page {page}: no spaces returned. Stopping.") break spaces.extend(page_spaces) logger.info(f"HuggingFace page {page}: fetched {len(page_spaces)} newest-modified spaces") next_link = (getattr(response, 'links', {}) or {}).get('next') or {} next_url = str(next_link.get('url') or '') next_params = None if not next_url: break except Exception as e: if isinstance(e, (ApiRequestError, RateLimitError)): raise logger.error(f"Error fetching HuggingFace page {page}: {str(e)}") raise ApiRequestError(f'HuggingFace discovery payload failed: {e}') from e return spaces logger.info(f"Fetching HuggingFace Spaces for {pages} page(s)...") for page in range(max(1, int(pages or 1))): url = 'https://huggingface.co/spaces-json' params = {'p': page, 'withCount': 'false', 'sort': 'created'} try: response = api_request( 'GET', url, headers=headers, params=params, timeout=request_timeout, max_retries=request_attempts, retry_delay=retry_delay, deadline=time.monotonic() + request_budget, ) if response.status_code >= 400: status = response.status_code category = 'auth_invalid' if status == 401 else 'auth_forbidden' if status == 403 else 'rate_limit' if status == 429 else 'server_error' if status >= 500 else 'api' reset_at = retry_after_reset(response) response.close() raise RateLimitError( 'huggingface', f'HuggingFace discovery HTTP {status}', reset_at=reset_at, category=category, auth_related=status in (401, 403, 429), ) response.raise_for_status() data = response.json() if not isinstance(data, dict) or 'spaces' not in data or not isinstance(data.get('spaces'), list): raise ValueError('invalid HuggingFace spaces payload') page_spaces = [huggingface_space_to_target(space) for space in data.get('spaces', []) if space.get('id')] if not page_spaces: logger.info(f"HuggingFace page {page}: no spaces returned. Stopping.") break spaces.extend(item['url'] for item in page_spaces if item.get('url')) logger.info(f"HuggingFace page {page}: fetched {len(page_spaces)} spaces") page_number = page + 1 if stop_on_seen_pages and page_number >= max(1, min_pages_before_stop): if page_is_known( [item.get('url') for item in page_spaces], known_targets, normalize_target, known_target_lookup, ): seen_pages += 1 logger.info(f"HuggingFace page {page}: all spaces are already queued/checked ({seen_pages}/{seen_page_threshold})") if seen_pages >= max(1, seen_page_threshold): logger.info(f"Stopping HuggingFace pagination early after {seen_pages} all-known page(s)") break else: seen_pages = 0 except Exception as e: if isinstance(e, (ApiRequestError, RateLimitError)): raise logger.error(f"Error fetching HuggingFace page {page}: {str(e)}") raise ApiRequestError(f'HuggingFace discovery payload failed: {e}') from e return spaces # ===================== # NPM FETCH FUNCTIONS # ===================== def parse_iso_datetime(value): if not value: return None try: parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00')) if parsed.tzinfo is None: parsed = parsed.replace(tzinfo=timezone.utc) return parsed.astimezone(timezone.utc) except ValueError: return None def npm_package_target(name, version, tarball_url, date=None): return json.dumps({ 'name': name, 'version': version, 'tarball': tarball_url, 'date': date or '', }, separators=(',', ':'), ensure_ascii=False) def parse_npm_target(target): if isinstance(target, dict): return target target = str(target).strip() if target.startswith('{'): return json.loads(target) package_id, tarball = target.split('|', 1) name, version = package_id.rsplit('@', 1) return {'name': name, 'version': version, 'tarball': tarball, 'date': ''} def npm_package_id(target): data = parse_npm_target(target) return f"npm:{data.get('name')}@{data.get('version')}" def select_npm_release_targets(name, data, max_versions=1, cutoff=None): targets = [] version_times = data.get('time') or {} for version, version_data in (data.get('versions') or {}).items(): tarball = (version_data.get('dist') or {}).get('tarball') if not tarball: continue version_date = version_times.get(version) or '' parsed_date = parse_iso_datetime(version_date) if cutoff and (not parsed_date or parsed_date < cutoff): continue targets.append({ 'name': name, 'version': version, 'tarball': tarball, 'date': version_date, 'parsed_date': parsed_date or datetime.min.replace(tzinfo=timezone.utc), }) targets.sort(key=lambda item: item['parsed_date'], reverse=True) return targets[:max(1, int(max_versions or 1))] def fetch_npm_packages(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1, repo_candidate_callback=None): """Fetch npm package tarball targets for recent versions matching a query.""" if not query: return [] targets = [] seen_packages = set() seen_versions = set() cutoff = None if max_version_age_days and max_version_age_days > 0: cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days) logger.info(f"Fetching npm packages for query: '{query}'...") for page in range(max(1, pages)): params = {'text': query, 'size': per_page, 'from': page * per_page} try: response = api_request( 'GET', 'https://registry.npmjs.org/-/v1/search', params=params, headers={'User-Agent': 'GitSecretsScanner/2.0'}, timeout=request_timeout, ) response.raise_for_status() payload = response.json() if not isinstance(payload, dict) or 'objects' not in payload or not isinstance(payload.get('objects'), list): raise ValueError('invalid npm search payload') objects = payload.get('objects', []) if objects and not any(isinstance(item, dict) and isinstance(item.get('package'), dict) and item['package'].get('name') for item in objects): raise ValueError('npm search payload contains no valid package entries') if not objects: logger.info(f"npm page {page + 1}: no results") break logger.info(f"npm page {page + 1}: fetched {len(objects)} packages") for item in objects: package = item.get('package', {}) name = package.get('name') version = package.get('version') if not name or not version: continue package_name_key = name.lower() if package_name_key in seen_packages: continue metadata_url = f"https://registry.npmjs.org/{quote(name, safe='')}" metadata = api_request( 'GET', metadata_url, headers={'User-Agent': 'GitSecretsScanner/2.0'}, timeout=request_timeout, ) metadata.raise_for_status() data = metadata.json() selected_versions = select_npm_release_targets(name, data, versions_per_package, cutoff) if repo_candidate_callback: repo_candidates = [] for version_item in selected_versions or [{'version': version}]: item_version = version_item.get('version') or version for candidate in extract_npm_git_candidates(data, item_version): repo_candidates.append({ 'package_source': 'npm', 'name': name, 'version': item_version, 'repo_url': candidate['repo_url'], 'provider': candidate['provider'], 'evidence': candidate.get('evidence') or [], 'confidence': 'high', }) repo_candidate_callback(repo_candidates) for item in selected_versions: package_key = f"{item['name']}@{item['version']}".lower() if package_key in seen_versions: continue targets.append(npm_package_target(item['name'], item['version'], item['tarball'], item['date'])) seen_versions.add(package_key) seen_packages.add(package_name_key) except Exception as e: if isinstance(e, ApiRequestError): raise logger.error(f"Error fetching npm page {page + 1}: {str(e)}") raise ApiRequestError(f'npm discovery failed: {e}') from e logger.info(f"npm query '{query}': prepared {len(targets)} package targets") return targets # ====================== # PYPI FETCH FUNCTIONS # ====================== pypi_project_index_cache_path = None def pypi_package_target(name, version, artifact_url, date=None, filename=None, packagetype=None, size=None): return json.dumps({ 'source': 'pypi', 'name': name, 'version': version, 'artifact': artifact_url, 'date': date or '', 'filename': filename or '', 'packagetype': packagetype or '', 'size': size or 0, }, separators=(',', ':'), ensure_ascii=False) def parse_pypi_target(target): if isinstance(target, dict): return target target = str(target).strip() if target.startswith('{'): return json.loads(target) package_id, artifact = target.split('|', 1) name, version = package_id.rsplit('@', 1) return {'source': 'pypi', 'name': name, 'version': version, 'artifact': artifact, 'date': ''} def pypi_package_id(target): data = parse_pypi_target(target) return f"pypi:{data.get('name')}@{data.get('version')}" # ============================= # PACKAGE -> GIT FETCH HELPERS # ============================= GIT_PATH_STOP_SEGMENTS = { '-', 'issues', 'issue', 'pull', 'pulls', 'merge_requests', 'merge_request', 'tree', 'blob', 'commit', 'commits', 'releases', 'tags', 'branches', 'wiki', } def canonical_git_path_parts(host, parts): if host == 'github.com': if len(parts) < 2: return [] return parts[:2] cleaned = [] for part in parts: lowered = part.lower() if lowered in GIT_PATH_STOP_SEGMENTS: break cleaned.append(part) if len(cleaned) < 2: return [] return cleaned def normalize_git_repo_candidate(value): if not value: return None raw = str(value).strip().strip('"\'') if not raw: return None if raw.startswith('git+'): raw = raw[4:] if raw.startswith('github:'): raw = 'https://github.com/' + raw.split(':', 1)[1] elif raw.startswith('gitlab:'): raw = 'https://gitlab.com/' + raw.split(':', 1)[1] elif raw.startswith('git@github.com:'): raw = 'https://github.com/' + raw.split(':', 1)[1] elif raw.startswith('git@gitlab.com:'): raw = 'https://gitlab.com/' + raw.split(':', 1)[1] elif raw.startswith('git://'): raw = 'https://' + raw[6:] try: parsed = urlsplit(raw) except ValueError: return None if parsed.scheme not in ('http', 'https') or not parsed.netloc: return None host = (parsed.hostname or '').lower() if host in ('www.github.com',): host = 'github.com' if host in ('www.gitlab.com',): host = 'gitlab.com' if host not in ('github.com', 'gitlab.com'): return None parts = [part for part in parsed.path.strip('/').split('/') if part] repo_parts = canonical_git_path_parts(host, parts) if not repo_parts: return None repo_parts[-1] = repo_parts[-1][:-4] if repo_parts[-1].endswith('.git') else repo_parts[-1] if any(not part for part in repo_parts): return None provider = 'github' if host == 'github.com' else 'gitlab' repo_path = '/'.join(repo_parts) return { 'provider': provider, 'repo_url': f'https://{host}/{repo_path}.git', 'repo_path': repo_path, } def package_git_target(package_source, name, version, repo_url, provider, evidence=None, confidence='medium'): return json.dumps({ 'source': 'package_git', 'package_source': package_source, 'name': name, 'version': version or '', 'repo_url': repo_url, 'provider': provider, 'evidence': evidence or [], 'confidence': confidence, }, separators=(',', ':'), ensure_ascii=False) def parse_package_git_target(target): if isinstance(target, dict): return target text = str(target).strip() if text.startswith('{'): return json.loads(text) candidate = normalize_git_repo_candidate(text) if not candidate: raise ValueError(f'Unsupported package_git target: {text}') return { 'source': 'package_git', 'package_source': 'custom', 'name': '', 'version': '', 'repo_url': candidate['repo_url'], 'provider': candidate['provider'], 'evidence': ['custom'], 'confidence': 'high', } def package_git_id(target): data = parse_package_git_target(target) return f"package_git:{data.get('provider')}:{data.get('repo_url')}".lower() def collect_git_candidates(values): candidates = [] seen = set() for evidence, value in values: candidate = normalize_git_repo_candidate(value) if not candidate: continue key = candidate['repo_url'].lower() if key in seen: continue seen.add(key) candidate['evidence'] = [evidence] candidates.append(candidate) return candidates GIT_URL_TEXT_RE = re.compile( r'(?:https?://|git\+https?://|git://|git@)(?:github\.com[:/]|gitlab\.com[:/])' r'[A-Za-z0-9_.-]+(?:/[A-Za-z0-9_.-]+){1,8}(?:\.git)?(?:/[A-Za-z0-9_.~/-]+)?', re.IGNORECASE, ) def git_candidates_from_text(label, text, max_urls=8): if not text: return [] values = [] seen = set() for match in GIT_URL_TEXT_RE.finditer(str(text)): raw = match.group(0).rstrip(').,;\'"<>') if raw.startswith('git@github.com/'): raw = raw.replace('git@github.com/', 'git@github.com:', 1) if raw.startswith('git@gitlab.com/'): raw = raw.replace('git@gitlab.com/', 'git@gitlab.com:', 1) key = raw.lower() if key in seen: continue seen.add(key) values.append((label, raw)) if len(values) >= max_urls: break return collect_git_candidates(values) def extract_npm_git_candidates(metadata, version=None): values = [] repository = metadata.get('repository') if isinstance(repository, dict): values.append(('repository.url', repository.get('url'))) elif isinstance(repository, str): values.append(('repository', repository)) bugs = metadata.get('bugs') if isinstance(bugs, dict): values.append(('bugs.url', bugs.get('url'))) values.append(('homepage', metadata.get('homepage'))) version_data = (metadata.get('versions') or {}).get(version or '', {}) version_repository = version_data.get('repository') if isinstance(version_data, dict) else None if isinstance(version_repository, dict): values.append(('version.repository.url', version_repository.get('url'))) elif isinstance(version_repository, str): values.append(('version.repository', version_repository)) if isinstance(version_data, dict): version_bugs = version_data.get('bugs') if isinstance(version_bugs, dict): values.append(('version.bugs.url', version_bugs.get('url'))) values.append(('version.homepage', version_data.get('homepage'))) candidates = collect_git_candidates(values) candidates.extend(git_candidates_from_text('readme.github_url', metadata.get('readme'))) candidates.extend(git_candidates_from_text('description.github_url', metadata.get('description'))) deduped = [] seen = set() for candidate in candidates: key = candidate['repo_url'].lower() if key in seen: continue seen.add(key) deduped.append(candidate) return deduped def extract_pypi_git_candidates(metadata): info = metadata.get('info') or {} values = [] project_urls = info.get('project_urls') or {} if isinstance(project_urls, dict): for key, value in project_urls.items(): label = str(key).lower() if any(item in label for item in ('source', 'repository', 'repo', 'code', 'homepage', 'home', 'bug', 'issue', 'tracker')): values.append((f'project_urls.{key}', value)) values.append(('home_page', info.get('home_page'))) values.append(('project_url', info.get('project_url'))) candidates = collect_git_candidates(values) candidates.extend(git_candidates_from_text('description.github_url', info.get('description'))) candidates.extend(git_candidates_from_text('summary.github_url', info.get('summary'))) deduped = [] seen = set() for candidate in candidates: key = candidate['repo_url'].lower() if key in seen: continue seen.add(key) deduped.append(candidate) return deduped def fetch_npm_package_git_repos(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1): if not query: return [] targets = [] seen_repos = set() cutoff = None if max_version_age_days and max_version_age_days > 0: cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days) logger.info(f"Fetching npm package git repos for query: '{query}'...") for page in range(max(1, pages)): params = {'text': query, 'size': per_page, 'from': page * per_page} try: response = api_request( 'GET', 'https://registry.npmjs.org/-/v1/search', params=params, headers={'User-Agent': 'GitSecretsScanner/2.0'}, timeout=request_timeout, ) response.raise_for_status() payload = response.json() if not isinstance(payload, dict) or 'objects' not in payload or not isinstance(payload.get('objects'), list): raise ValueError('invalid npm package_git search payload') objects = payload.get('objects', []) if objects and not any(isinstance(item, dict) and isinstance(item.get('package'), dict) and item['package'].get('name') for item in objects): raise ValueError('npm package_git payload contains no valid package entries') if not objects: logger.info(f"npm package_git page {page + 1}: no results") break logger.info(f"npm package_git page {page + 1}: fetched {len(objects)} packages") for item in objects: package = item.get('package', {}) name = package.get('name') if not name: continue metadata = api_request( 'GET', f"https://registry.npmjs.org/{quote(name, safe='')}", headers={'User-Agent': 'GitSecretsScanner/2.0'}, timeout=request_timeout, ) metadata.raise_for_status() data = metadata.json() selected_versions = select_npm_release_targets(name, data, versions_per_package, cutoff) or [{'version': package.get('version') or ''}] for version_item in selected_versions: version = version_item.get('version') or package.get('version') or '' for candidate in extract_npm_git_candidates(data, version): key = candidate['repo_url'].lower() if key in seen_repos: continue seen_repos.add(key) targets.append(package_git_target('npm', name, version, candidate['repo_url'], candidate['provider'], candidate['evidence'], 'high')) except Exception as e: if isinstance(e, ApiRequestError): raise logger.error(f"Error fetching npm package_git page {page + 1}: {str(e)}") raise ApiRequestError(f'npm package_git discovery failed: {e}') from e logger.info(f"npm package_git query '{query}': prepared {len(targets)} git repo targets") return targets def fetch_pypi_package_git_repos(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1): if not query: return [] targets = [] seen_repos = set() cutoff = None if max_version_age_days and max_version_age_days > 0: cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days) logger.info(f"Fetching PyPI package git repos for query: '{query}'...") for page in range(1, max(1, pages) + 1): try: names = fetch_pypi_package_names(query, page, per_page, request_timeout) if not names: logger.info(f"PyPI package_git page {page}: no results") break logger.info(f"PyPI package_git page {page}: fetched {len(names)} package names") for name in names: try: metadata = api_request( 'GET', f"https://pypi.org/pypi/{quote(name, safe='')}/json", headers={'User-Agent': 'GitSecretsScanner/2.0'}, timeout=request_timeout, ) metadata.raise_for_status() data = metadata.json() except requests.exceptions.HTTPError as e: if e.response is not None and e.response.status_code == 404: logger.info(f"Skipping PyPI package_git project {name}: metadata not found") continue raise ApiRequestError(f'PyPI package_git metadata failed for {name}: {e}') from e except ApiRequestError: raise if not isinstance(data, dict) or not isinstance(data.get('info', {}), dict): raise ApiRequestError(f'invalid PyPI package_git metadata payload for {name}') release_files = select_pypi_release_files(data, cutoff, versions_per_package) version = release_files[0]['version'] if release_files else (data.get('info') or {}).get('version') or '' package_name = (data.get('info') or {}).get('name') or name for candidate in extract_pypi_git_candidates(data): key = candidate['repo_url'].lower() if key in seen_repos: continue seen_repos.add(key) targets.append(package_git_target('pypi', package_name, version, candidate['repo_url'], candidate['provider'], candidate['evidence'], 'medium')) except Exception as e: if isinstance(e, ApiRequestError): raise logger.error(f"Error fetching PyPI package_git page {page}: {str(e)}") raise ApiRequestError(f'PyPI package_git discovery failed: {e}') from e logger.info(f"PyPI package_git query '{query}': prepared {len(targets)} git repo targets") return targets class _PyPIProjectParser(HTMLParser): def __init__(self, sink): super().__init__(convert_charrefs=True) self.sink = sink self.in_anchor = False self.parts = [] def handle_starttag(self, tag, attrs): if tag.lower() == 'a': self.in_anchor = True self.parts = [] def handle_endtag(self, tag): if tag.lower() == 'a': name = ''.join(self.parts).strip() if name: self.sink(name) self.in_anchor = False self.parts = [] def handle_data(self, data): if self.in_anchor: text = str(data or '') if sum(len(part) for part in self.parts) + len(text) <= 512: self.parts.append(text) def _pypi_index_path(): global pypi_project_index_cache_path if pypi_project_index_cache_path: return pypi_project_index_cache_path state_dir = os.path.dirname(scan_limiter_db_path()) require_private_directory(state_dir, create=False) pypi_project_index_cache_path = os.path.join(state_dir, 'pypi_project_index.sqlite3') return pypi_project_index_cache_path def load_pypi_project_index(request_timeout=20): path = _pypi_index_path() refresh_sec = max(3600, int(os.getenv('PYPI_PROJECT_INDEX_REFRESH_SEC', '86400'))) if not os.path.exists(path): descriptor = os.open( path, os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0), 0o600, ) os.close(descriptor) harden_private_file(path) connection = sqlite3.connect(path, timeout=30) try: connection.execute('PRAGMA journal_mode=DELETE') connection.execute('PRAGMA synchronous=FULL') connection.executescript(''' CREATE TABLE IF NOT EXISTS pypi_projects ( normalized_name TEXT PRIMARY KEY, name TEXT NOT NULL ); CREATE TABLE IF NOT EXISTS pypi_index_meta ( id INTEGER PRIMARY KEY CHECK(id = 1), refreshed_at REAL NOT NULL, project_count INTEGER NOT NULL ); ''') current = connection.execute( 'SELECT refreshed_at, project_count FROM pypi_index_meta WHERE id = 1' ).fetchone() if current and time.time() - float(current[0]) < refresh_sec and int(current[1]) > 0: return path response = _direct_request( 'GET', 'https://pypi.org/simple/', headers={'Accept': 'text/html', 'User-Agent': 'GitSecretsScanner/2.0'}, timeout=request_timeout, stream=True, ) response.raise_for_status() connection.execute('BEGIN IMMEDIATE') connection.execute('DELETE FROM pypi_projects') batch = [] count = 0 def accept(name): nonlocal count normalized = re.sub(r'[-_.]+', '-', name).lower() if not normalized or len(normalized) > 512: return batch.append((normalized, name[:512])) if len(batch) >= 1000: connection.executemany( 'INSERT OR IGNORE INTO pypi_projects(normalized_name, name) VALUES (?, ?)', batch, ) count += len(batch) batch.clear() parser = _PyPIProjectParser(accept) decoder = codecs.getincrementaldecoder('utf-8')('strict') response_bytes = 0 response_max_bytes = max( 1024 * 1024, int(os.getenv('PYPI_PROJECT_INDEX_MAX_BYTES', str(512 * 1024 * 1024))), ) try: for chunk in response.iter_content(chunk_size=256 * 1024): _raise_if_scan_slot_fatal() if chunk: response_bytes += len(chunk) if response_bytes > response_max_bytes: raise ApiRequestError('PyPI simple index exceeds its streamed byte bound') parser.feed(decoder.decode(chunk)) parser.feed(decoder.decode(b'', final=True)) parser.close() finally: response.close() if batch: connection.executemany( 'INSERT OR IGNORE INTO pypi_projects(normalized_name, name) VALUES (?, ?)', batch, ) count += len(batch) actual = int(connection.execute('SELECT COUNT(*) FROM pypi_projects').fetchone()[0]) if actual <= 0: raise ApiRequestError('PyPI simple index contains no valid project entries') connection.execute( '''INSERT INTO pypi_index_meta(id, refreshed_at, project_count) VALUES (1, ?, ?) ON CONFLICT(id) DO UPDATE SET refreshed_at = excluded.refreshed_at, project_count = excluded.project_count''', (time.time(), actual), ) connection.commit() logger.info('Streamed %d PyPI project names into the on-disk index', actual) return path except Exception: connection.rollback() raise finally: connection.close() if os.path.exists(path): harden_private_file(path) def pypi_name_rank(name, query): normalized = name.lower() query = query.lower() parts = [part for part in re.split(r'[-_.]+', normalized) if part] if normalized == query: rank = 0 elif normalized.startswith(query): rank = 1 elif query in parts: rank = 2 else: rank = 3 return rank, len(normalized), normalized def fetch_pypi_package_names(query, page=1, per_page=50, request_timeout=20): query = query.strip().lower() if not query: return [] tokens = [token for token in re.split(r'\s+', query) if token] path = load_pypi_project_index(request_timeout) page_size = max(1, int(per_page or 50)) requested_end = max(1, int(page)) * page_size candidate_limit = min(100000, max(1000, requested_end * 20)) connection = sqlite3.connect(f'file:{path.replace(os.sep, "/")}?mode=ro', uri=True, timeout=30) try: clauses = ' AND '.join('normalized_name LIKE ?' for _ in tokens) rows = connection.execute( f'''SELECT name FROM pypi_projects WHERE {clauses} ORDER BY normalized_name LIMIT ?''', (*[f'%{token}%' for token in tokens], candidate_limit), ) matches = [row[0] for row in rows] finally: connection.close() matches.sort(key=lambda name: pypi_name_rank(name, query)) start = (max(1, int(page)) - 1) * page_size return matches[start:start + page_size] def select_pypi_release_files(data, cutoff=None, max_versions=1): release_candidates = [] priority_by_type = {'sdist': 2, 'bdist_wheel': 1} for version, files in (data.get('releases') or {}).items(): file_candidates = [] for file_info in files or []: if file_info.get('yanked'): continue artifact_url = file_info.get('url') if not artifact_url: continue uploaded = file_info.get('upload_time_iso_8601') or file_info.get('upload_time') or '' parsed_date = parse_iso_datetime(uploaded) if cutoff and (not parsed_date or parsed_date < cutoff): continue file_candidates.append({ 'version': version, 'url': artifact_url, 'date': uploaded, 'parsed_date': parsed_date or datetime.min.replace(tzinfo=timezone.utc), 'filename': file_info.get('filename') or '', 'packagetype': file_info.get('packagetype') or '', 'size': file_info.get('size') or 0, 'priority': priority_by_type.get(file_info.get('packagetype'), 0), }) if file_candidates: release_date = max(item['parsed_date'] for item in file_candidates) file_candidates.sort(key=lambda item: (item['priority'], item['parsed_date']), reverse=True) release_candidates.append((release_date, file_candidates[0])) if not release_candidates: return [] release_candidates.sort(key=lambda item: item[0], reverse=True) return [item[1] for item in release_candidates[:max(1, int(max_versions or 1))]] def select_pypi_release_file(data, cutoff=None): files = select_pypi_release_files(data, cutoff, 1) return files[0] if files else None def fetch_pypi_packages(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1, repo_candidate_callback=None): """Fetch PyPI package artifact targets for recent matching releases.""" if not query: return [] targets = [] seen = set() cutoff = None if max_version_age_days and max_version_age_days > 0: cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days) logger.info(f"Fetching PyPI packages for query: '{query}'...") for page in range(1, max(1, pages) + 1): try: names = fetch_pypi_package_names(query, page, per_page, request_timeout) if not names: logger.info(f"PyPI page {page}: no results") break logger.info(f"PyPI page {page}: fetched {len(names)} package names") for name in names: normalized_name = name.lower() if normalized_name in seen: continue metadata_url = f"https://pypi.org/pypi/{quote(name, safe='')}/json" try: metadata = api_request( 'GET', metadata_url, headers={'User-Agent': 'GitSecretsScanner/2.0'}, timeout=request_timeout, ) metadata.raise_for_status() data = metadata.json() except requests.exceptions.HTTPError as e: if e.response is not None and e.response.status_code == 404: logger.info(f"Skipping PyPI project {name}: metadata not found") seen.add(normalized_name) continue logger.warning(f"Skipping PyPI project {name}: metadata fetch failed: {str(e)}") raise ApiRequestError(f'PyPI metadata failed for {name}: {e}') from e except ApiRequestError: raise except Exception as e: raise ApiRequestError(f'PyPI metadata payload failed for {name}: {e}') from e if not isinstance(data, dict) or not isinstance(data.get('info', {}), dict): raise ApiRequestError(f'invalid PyPI metadata payload for {name}') release_files = select_pypi_release_files(data, cutoff, versions_per_package) if not release_files: continue if repo_candidate_callback: package_name = data.get('info', {}).get('name') or name repo_candidates = [] for release_file in release_files: for candidate in extract_pypi_git_candidates(data): repo_candidates.append({ 'package_source': 'pypi', 'name': package_name, 'version': release_file['version'], 'repo_url': candidate['repo_url'], 'provider': candidate['provider'], 'evidence': candidate.get('evidence') or [], 'confidence': 'medium', }) repo_candidate_callback(repo_candidates) for release_file in release_files: targets.append(pypi_package_target( data.get('info', {}).get('name') or name, release_file['version'], release_file['url'], release_file['date'], release_file['filename'], release_file['packagetype'], release_file['size'], )) seen.add(normalized_name) except Exception as e: if isinstance(e, ApiRequestError): raise logger.error(f"Error fetching PyPI page {page}: {str(e)}") raise ApiRequestError(f'PyPI discovery failed: {e}') from e logger.info(f"PyPI query '{query}': prepared {len(targets)} package targets") return targets # ===================== # SCANNING FUNCTIONS # ===================== def check_dependencies(): """Verify required dependencies are installed""" probe_dir = None try: work_dir = get_work_dir() probe_dir = tempfile.mkdtemp(prefix='trufflehog-probe-', dir=work_dir) harden_private_directory(probe_dir) command = [get_trufflehog_cmd(), '--version', '--no-update'] write_temp_owner(probe_dir, command, os.getpid()) env = os.environ.copy() strip_supervisor_credentials(env) env['PATH'] = os.pathsep.join([ os.path.expanduser('~/bin'), os.path.expanduser('~/.local/bin'), env.get('PATH', '') ]) prepend_client_git_environment(env) env['TEMP'] = probe_dir env['TMP'] = probe_dir env['TMPDIR'] = probe_dir require_trufflehog_launch_authority(command) result = run_owned( command, stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=30, env=env, cwd=probe_dir, creationflags=subprocess.CREATE_NO_WINDOW if os.name == 'nt' else 0, ) if result.returncode != 0: stderr = (result.stderr or b'') if isinstance(result.stderr, bytes) else str(result.stderr or '').encode('utf-8', errors='replace') detail = stderr.decode('utf-8', errors='replace').strip()[:500] raise RuntimeError(f"TruffleHog version probe exited with code {result.returncode}: {detail or 'no stderr'}") logger.info("trufflehog is installed and working") return True except Exception as e: if isinstance(e, subprocess.TimeoutExpired): detail = f"timed out after {e.timeout}s" else: detail = f"{type(e).__name__}: {e}" logger.error("TruffleHog dependency probe failed: %s", redact_scan_command_text([detail])[:500]) if isinstance(e, FileNotFoundError): logger.error("TruffleHog executable was not found. Please install it.") logger.error("Visit https://github.com/trufflesecurity/trufflehog for installation instructions.") else: logger.error("TruffleHog dependency probe failed closed; executable authority was not bypassed.") return False finally: if probe_dir: cleanup_command_work_dir(probe_dir) def command_output_limits(): if _client_scan_policy.get() is None: stdout_mb = int_setting( os.getenv('TRUFFLEHOG_STDOUT_MAX_MB'), getattr(scan_config, 'trufflehog_stdout_max_mb', 32), ) stderr_mb = int_setting( os.getenv('TRUFFLEHOG_STDERR_MAX_MB'), getattr(scan_config, 'trufflehog_stderr_max_mb', 8), ) else: stdout_mb = int(_scan_policy_value('trufflehog_stdout_max_mb', 32)) stderr_mb = int(_scan_policy_value('trufflehog_stderr_max_mb', 8)) requested_stdout = max(1, int_setting( stdout_mb, 32, )) * 1024 * 1024 requested_stderr = max(1, int_setting( stderr_mb, 8, )) * 1024 * 1024 event_limit = max(1, int(_scan_policy_value( 'result_bundle_max_event_bytes', 64 * 1024 * 1024, ))) reserve = min(32 * 1024 * 1024, max(64 * 1024, event_limit // 4)) output_budget = max(2, event_limit - reserve) requested_total = requested_stdout + requested_stderr if requested_total <= output_budget: return requested_stdout, requested_stderr stdout = max(1, (output_budget * requested_stdout) // requested_total) stderr = max(1, output_budget - stdout) return stdout, stderr class CommandOutputLimitError(RuntimeError): pass class StreamedCommandOutput: def __init__(self, stdout_file, stderr_file, returncode, max_stdout, max_stderr, synthetic_stderr=''): self._stdout = stdout_file self._stderr = stderr_file self.returncode = int(returncode) self.max_stdout = int(max_stdout) self.max_stderr = int(max_stderr) self.synthetic_stderr = str(synthetic_stderr or '') @staticmethod def _lines(handle, byte_limit, max_line_bytes, max_lines, redactions=()): handle.seek(0) consumed = 0 count = 0 while consumed < byte_limit and count < max_lines: raw = handle.readline(min(max_line_bytes + 1, byte_limit - consumed + 1)) if not raw: return consumed += len(raw) count += 1 if len(raw) > max_line_bytes and not raw.endswith((b'\n', b'\r')): raise CommandOutputLimitError(f'command output line exceeded {max_line_bytes} bytes') line = raw.decode('utf-8', errors='replace') if redactions: line = redact_secrets(line, redactions) yield line if handle.read(1): raise CommandOutputLimitError('command output exceeded its line or byte bound') def stdout_lines(self, max_line_bytes=16 * 1024 * 1024, max_lines=20000, redactions=()): return self._lines( self._stdout, self.max_stdout, max(1, int(max_line_bytes)), max(1, int(max_lines)), redactions, ) def stderr_lines(self, max_line_bytes=8192, max_lines=2000, redactions=()): for line in self._lines( self._stderr, self.max_stderr, max(1, int(max_line_bytes)), max(1, int(max_lines)), redactions, ): yield line if self.synthetic_stderr: yield self.synthetic_stderr.rstrip('\r\n') + '\n' @staticmethod def _raw_bytes(handle): position = handle.tell() try: handle.seek(0) return handle.read() finally: handle.seek(position) def raw_stdout_bytes(self): return self._raw_bytes(self._stdout) def raw_stderr_bytes(self): return self._raw_bytes(self._stderr) @contextmanager def streamed_output_from_text(stdout='', stderr='', returncode=0): stdout_file = io.BytesIO(str(stdout).encode('utf-8')) stderr_file = io.BytesIO(str(stderr).encode('utf-8')) yield StreamedCommandOutput( stdout_file, stderr_file, returncode, max(1, len(stdout_file.getvalue())), max(1, len(stderr_file.getvalue())), ) def _check_command_staging(roots, deadline, output_files=()): """Best-effort live staging watchdog, not a filesystem quota or atomic snapshot.""" exceeded = 'TruffleHog staging limit exceeded' unavailable = 'Unable to monitor TruffleHog staging' total_bytes = 0 entries = 0 try: unique_roots = [] for root in sorted({os.path.normcase(os.path.abspath(root)) for root in roots}, key=len): if not any(root == parent or root.startswith(os.path.join(parent, '')) for parent in unique_roots): unique_roots.append(root) # Check ancestors too: lstat on a child alone would follow a linked parent. for root in unique_roots: ancestor = root while True: if time.monotonic() >= deadline: return unavailable try: info = os.lstat(ancestor) except FileNotFoundError: pass else: if stat.S_ISLNK(info.st_mode) or getattr(info, 'st_file_attributes', 0) & stat.FILE_ATTRIBUTE_REPARSE_POINT: return unavailable parent = os.path.dirname(ancestor) if parent == ancestor: break ancestor = parent # POSIX TemporaryFile output may be unlinked and thus absent from scandir. for handle in output_files: if time.monotonic() >= deadline: return unavailable info = os.fstat(handle.fileno()) if info.st_nlink == 0: total_bytes += info.st_size entries += 1 pending = list(unique_roots) while pending: if time.monotonic() >= deadline: return unavailable path = pending.pop() try: info = os.lstat(path) if stat.S_ISLNK(info.st_mode) or getattr(info, 'st_file_attributes', 0) & stat.FILE_ATTRIBUTE_REPARSE_POINT: return unavailable if stat.S_ISREG(info.st_mode): total_bytes += info.st_size elif not stat.S_ISDIR(info.st_mode): return unavailable if total_bytes > 2 * 1024 ** 3: return exceeded if stat.S_ISDIR(info.st_mode): with os.scandir(path) as children: for child in children: if time.monotonic() >= deadline: return unavailable entries += 1 if entries > 100000: return exceeded pending.append(child.path) except FileNotFoundError: continue if time.monotonic() >= deadline: return unavailable if total_bytes > 2 * 1024 ** 3 or entries > 100000: return exceeded except (OSError, ValueError): return unavailable return '' @contextmanager def run_command_streamed(cmd, timeout_sec, env=None, *, deadline=None, staging_roots=None, native_git_clone=False): """Run one owned command; optional staging roots also monitor its private temp tree.""" if type(native_git_clone) is not bool: raise ValueError('native_git_clone must be an explicit boolean') if native_git_clone and not staging_roots: raise ValueError('native Git clone requires private staging roots') command_work_dir = None process = None scan_slot = None owns_scan_slot = False release_scan_slot = True stdout_file = None stderr_file = None max_stdout, max_stderr = command_output_limits() fatal_slot_error = None propagating_fatal_error = None returncode = -1 synthetic_stderr = '' armed_owners = [] command_owner_published = False requested_timeout = ( max(1, int(timeout_sec or 1)) if deadline is None else max(0.001, float(timeout_sec or 0.001)) ) command_deadline = time.monotonic() + requested_timeout if deadline is not None: deadline = float(deadline) if not math.isfinite(deadline): raise ValueError('command deadline must be finite') command_deadline = min(command_deadline, deadline) try: _raise_if_scan_slot_fatal() env = dict(os.environ if env is None else env) for key in list(env): if key.lower() in ('http_proxy', 'https_proxy', 'all_proxy', 'no_proxy'): del env[key] env['NO_PROXY'] = '*' strip_supervisor_credentials(env) if os.name == 'nt': env['PATH'] = os.pathsep.join([ os.path.expanduser('~/bin'), os.path.expanduser('~/.local/bin'), env.get('PATH', ''), ]) prepend_client_git_environment(env) env['GIT_TERMINAL_PROMPT'] = '0' env['GIT_ASKPASS'] = 'true' min_free_gb = max(0.0, float(getattr(scan_config, 'min_free_gb', 0) or 0)) min_free_bytes = int(min_free_gb * 1024 * 1024 * 1024) borrowed_scan_slot, scan_slot = scoped_scan_slot_lease() if not borrowed_scan_slot: scan_slot = acquire_scan_slot( cmd, max(0.001, command_deadline - time.monotonic()), ) owns_scan_slot = True if scan_slot and not scan_slot.releasable: raise RuntimeError('scan slot is fail-closed after unconfirmed child termination') command_work_dir = create_command_work_dir() shared_owners = _shared_staging_owners((command_work_dir, *(staging_roots or ()))) staging_roots = (command_work_dir, *staging_roots) if staging_roots is not None else None env['TEMP'] = command_work_dir env['TMP'] = command_work_dir env['TMPDIR'] = command_work_dir if env.get('TRUF_GIT_TOKEN'): if os.name == 'nt': askpass_path = os.path.join(command_work_dir, 'git-askpass.cmd') with open(askpass_path, 'w', encoding='ascii') as askpass: askpass.write('@echo off\r\n') askpass.write('echo %~1 | findstr /I "username" >nul\r\n') askpass.write('if %errorlevel%==0 (echo %TRUF_GIT_USERNAME%) else (echo %TRUF_GIT_TOKEN%)\r\n') else: askpass_path = os.path.join(command_work_dir, 'git-askpass.sh') with open(askpass_path, 'x', encoding='ascii', newline='\n') as askpass: askpass.write( '#!/bin/sh\n' 'case "$1" in\n' ' *[Uu][Ss][Ee][Rr][Nn][Aa][Mm][Ee]*) printf \'%s\\n\' "$TRUF_GIT_USERNAME" ;;\n' ' *) printf \'%s\\n\' "$TRUF_GIT_TOKEN" ;;\n' 'esac\n' ) harden_private_file(askpass_path) if os.name != 'nt': os.chmod(askpass_path, stat.S_IRWXU) env['GIT_ASKPASS'] = askpass_path creationflags = ( subprocess.CREATE_NEW_PROCESS_GROUP | subprocess.CREATE_NO_WINDOW ) if os.name == 'nt' else 0 stdout_file = tempfile.TemporaryFile(mode='w+b', dir=command_work_dir) stderr_file = tempfile.TemporaryFile(mode='w+b', dir=command_work_dir) if native_git_clone: require_git_clone_launch_authority(cmd) if time.monotonic() >= command_deadline: raise subprocess.TimeoutExpired(cmd, timeout_sec) destination_parent = os.path.normcase(os.path.abspath(os.path.dirname(cmd[-1]))) if destination_parent not in {os.path.normcase(os.path.abspath(root)) for root in staging_roots}: raise RuntimeError('Git clone destination is outside its private staging parent') staging_error = _check_command_staging(staging_roots, command_deadline, (stdout_file, stderr_file)) if staging_error: raise RuntimeError(staging_error) else: require_trufflehog_launch_authority(cmd) if time.monotonic() >= command_deadline: raise subprocess.TimeoutExpired(cmd, timeout_sec) process_options = { 'stdout': stdout_file, 'stderr': stderr_file, 'env': env, 'cwd': command_work_dir, 'stdin': subprocess.DEVNULL, 'creationflags': creationflags, } if os.name == 'nt': job_memory_limit_bytes = _scan_policy_value( 'trufflehog_job_memory_limit_bytes', 0, ) if isinstance(job_memory_limit_bytes, bool) or not isinstance(job_memory_limit_bytes, int) or job_memory_limit_bytes <= 0: raise RuntimeError('trufflehog_job_memory_limit_bytes must be a positive integer on Windows') process_options['job_memory_limit_bytes'] = job_memory_limit_bytes job_cpu_weight = int(_scan_policy_value('trufflehog_windows_job_cpu_weight', 0)) memory_priority = int(_scan_policy_value('trufflehog_windows_memory_priority', 0)) if job_cpu_weight < 0 or job_cpu_weight > 9: raise RuntimeError('trufflehog_windows_job_cpu_weight must be between 0 and 9') if memory_priority < 0 or memory_priority > 5: raise RuntimeError('trufflehog_windows_memory_priority must be between 0 and 5') process_options['job_cpu_weight'] = job_cpu_weight process_options['process_memory_priority'] = memory_priority for root, marker in shared_owners: armed_owners.append((root, marker)) pending = dict(marker, child_pid=None, child_creation_time=None, child_executable=None) atomic_write_private_json(os.path.join(root, TEMP_OWNER_FILE), pending) if time.monotonic() >= command_deadline: raise subprocess.TimeoutExpired(cmd, timeout_sec) process = OwnedProcess(cmd, **process_options) if not process.job_membership_verified: raise RuntimeError('TruffleHog exact Job membership was not verified') if scan_slot and not scan_slot.set_child_pid(process.pid): raise RuntimeError('unable to publish TruffleHog child identity to the scan slot') write_temp_owner( command_work_dir, cmd, process.pid, required=True, owner_identity=getattr(process, 'payload_identity', None), ) command_owner_published = True child_identity = getattr(process, 'payload_identity', None) for root, marker in armed_owners: if root == canonical_path(command_work_dir): continue if not isinstance(child_identity, dict) or any(not child_identity.get(field) for field in ('pid', 'creation_time', 'executable')): raise RuntimeError('shared staging child identity is unavailable') active = dict(marker, **{f'child_{field}': child_identity[field] for field in ('pid', 'creation_time', 'executable')}) atomic_write_private_json(os.path.join(root, TEMP_OWNER_FILE), active) limit_error = '' next_staging_check = 0.0 while True: completed = process.poll() is not None if completed and staging_roots is None: break _raise_if_scan_slot_fatal() stdout_size = os.fstat(stdout_file.fileno()).st_size stderr_size = os.fstat(stderr_file.fileno()).st_size if stdout_size > max_stdout: limit_error = f'TruffleHog stdout exceeded {max_stdout} bytes' elif stderr_size > max_stderr: limit_error = f'TruffleHog stderr exceeded {max_stderr} bytes' elif min_free_bytes: try: free_bytes = shutil.disk_usage(command_work_dir).free except OSError as exc: limit_error = ( 'Unable to monitor TruffleHog staging' if staging_roots is not None else f'Unable to monitor configured TruffleHog work volume free space: {exc}' ) else: if free_bytes <= min_free_bytes: limit_error = ( f'Not enough free space on configured TruffleHog work volume {command_work_dir}: ' f'{free_bytes / (1024 ** 3):.2f} GB free, minimum is {min_free_gb:.2f} GB' ) if not limit_error and staging_roots is not None and ( completed or time.monotonic() >= next_staging_check ): limit_error = _check_command_staging( staging_roots, command_deadline, (stdout_file, stderr_file), ) next_staging_check = time.monotonic() + 1.0 if limit_error: process.kill() try: process.wait(timeout=10) except subprocess.TimeoutExpired: release_scan_slot = False limit_error += '; process tree termination failed' synthetic_stderr = f'Error: {limit_error}' returncode = -1 break if completed: break if time.monotonic() >= command_deadline: raise subprocess.TimeoutExpired(cmd, timeout_sec) if _scan_slot_fatal_event.wait(0.2): _raise_if_scan_slot_fatal() if not limit_error: returncode = process.returncode stdout_file.seek(0) stderr_file.seek(0) yield StreamedCommandOutput( stdout_file, stderr_file, returncode, max_stdout, max_stderr, synthetic_stderr, ) except subprocess.TimeoutExpired: if process and process.poll() is None: process.kill() try: process.wait(timeout=5) except subprocess.TimeoutExpired: release_scan_slot = False if stdout_file is None or stderr_file is None: raise timeout_error = f'Command timed out after {timeout_sec} seconds' if staging_roots is not None: staging_error = _check_command_staging( staging_roots, command_deadline, (stdout_file, stderr_file), ) if staging_error: timeout_error = f'Error: {staging_error}' stdout_file.seek(0) stderr_file.seek(0) yield StreamedCommandOutput( stdout_file, stderr_file, -1, max_stdout, max_stderr, timeout_error, ) except ScanSlotFatalError as exc: propagating_fatal_error = exc raise finally: if process: try: child_running = process.poll() is None except Exception: child_running = True if child_running: try: process.kill() process.wait(timeout=5) except Exception: release_scan_slot = False if not release_scan_slot: if scan_slot: scan_slot.mark_non_releasable() detail = 'FATAL: TruffleHog child termination was not confirmed; scan capacity remains fail-closed' _set_scan_slot_fatal(detail) fatal_slot_error = propagating_fatal_error or ScanSlotFatalError(detail) restore_error = None if release_scan_slot: for root, marker in armed_owners: if command_owner_published and root == canonical_path(command_work_dir): continue try: atomic_write_private_json(os.path.join(root, TEMP_OWNER_FILE), marker) except Exception as exc: restore_error = exc if scan_slot and owns_scan_slot and release_scan_slot: scan_slot.release() if stdout_file: stdout_file.close() if stderr_file: stderr_file.close() if release_scan_slot and fatal_slot_error is None and propagating_fatal_error is None: cleanup_command_work_dir(command_work_dir) if fatal_slot_error is not None: raise fatal_slot_error if restore_error is not None: raise RuntimeError('Unable to restore shared staging ownership after child termination') from restore_error def run_command(cmd, timeout_sec, env=None): """Bounded compatibility adapter; production parsers use run_command_streamed.""" try: with run_command_streamed(cmd, timeout_sec, env) as output: stdout = ''.join(output.stdout_lines(max_line_bytes=output.max_stdout, max_lines=20000)) stderr = ''.join(output.stderr_lines(max_line_bytes=output.max_stderr, max_lines=2000)) return stdout, stderr, output.returncode except ScanSlotFatalError: raise except Exception as exc: return '', f'Error running command: {exc}', -1 def _trufflehog_diagnostic_policy(line, source_type, returncode): text = str(line or '').strip() payload = None try: parsed = json.loads(text) if isinstance(parsed, dict): payload = parsed except (TypeError, ValueError): pass if payload and payload.get('errors'): causes = payload['errors'] limits = _trufflehog_diagnostic_limits() if not isinstance(causes, list): return 'error', 'trufflehog', True if len(causes) > limits['errors'] or any( isinstance(cause, str) and ( len(cause) > limits['line_chars'] or len(cause.encode('utf-8', errors='replace')) > limits['line_bytes'] ) for cause in causes ): return 'error', 'output_limit', False envelope = dict(payload) del envelope['errors'] envelope_policy = _trufflehog_diagnostic_policy(json.dumps(envelope), source_type, returncode) if envelope_policy[1] == 'source_auth' and not (payload.get('error') or payload.get('message')): envelope_policy = ('error', 'auth_or_permission', envelope_policy[2]) policies = [] for cause in causes: if not isinstance(cause, str) or not cause.strip(): continue cause_payload = {'level': 'error', 'error': cause} policy = _trufflehog_diagnostic_policy(json.dumps(cause_payload), source_type, returncode) cause_payload['msg'] = payload.get('msg') or '' contextual_policy = _trufflehog_diagnostic_policy(json.dumps(cause_payload), source_type, returncode) # Keep known warnings, but not at the expense of an independent fatal cause. if contextual_policy[0] == 'warning' and ( policy[1] == 'trufflehog' or ( contextual_policy[1] == 'detector_timeout' and policy[1] == 'timeout' and cause.strip().lower() == 'context deadline exceeded' ) ): policy = contextual_policy if policy[1] == 'source_auth': # A nested diagnostic can be detector verification, not the selected source credential. policy = ('error', 'auth_or_permission', policy[2]) policies.append(policy) # Envelope summaries are not additional causes, but explicit fatal details are. if envelope_policy[0] == 'error' and ( envelope_policy[1] != 'trufflehog' or payload.get('error') or payload.get('message') ): policies.append(envelope_policy) fatal = [policy for policy in policies if policy[0] == 'error'] if not fatal: if policies and all(policy[0] == 'warning' for policy in policies): return 'warning', policies[0][1], all(policy[2] for policy in policies) return 'error', 'trufflehog', True retryable = all(policy[2] for policy in fatal) classes = {policy[1] for policy in fatal} for error_class in ( 'memory_limit', 'source_configuration', 'output_limit', 'source_resource', 'source_auth', 'docker_registry_access', 'auth_or_permission', ): if error_class in classes: return 'error', error_class, retryable return 'error', next(iter(classes)) if len(classes) == 1 else 'mixed', retryable if ( source_type == 'docker' and payload and payload.get('error') and payload.get('message') and payload['error'] != payload['message'] ): # Independent detail channels must survive codec recovery; reuse fatal priority and bounds. policy = _trufflehog_diagnostic_policy(json.dumps({ 'level': 'error', 'msg': payload.get('msg'), 'errors': [str(payload['error']), str(payload['message'])], }), source_type, returncode) return policy if policy[0] == 'error' else ('error', 'trufflehog', True) message = str((payload or {}).get('msg') or '') detail = str((payload or {}).get('error') or (payload or {}).get('message') or '') level = str((payload or {}).get('level') or '').lower() message_lower = message.lower() detail_lower = detail.lower() combined = f'{message_lower} {detail_lower}' if payload else text.lower() direct_kind = _client_remote_execution_kind.get() if any(token in combined for token in ('virtualalloc', 'out of memory', 'cannot allocate memory')): return 'error', 'memory_limit', False if any(token in combined for token in ('unknown flag', 'unknown command', 'invalid detector', 'failed to load config')): return 'error', 'source_configuration', False if any(token in combined for token in ( 'trufflehog stdout exceeded', 'trufflehog stderr exceeded', )): return 'error', 'output_limit', False if any(token in combined for token in ( 'no space left', 'not enough free space', 'disk quota', 'trufflehog work_dir', 'configured command work directory', 'temporary file', 'temporaryfile', 'disk fsync', )): return 'error', 'source_resource', True if message_lower == 'a detector ignored the context timeout': return 'warning', 'detector_timeout', False if source_type == 'huggingface' and detail_lower == 'no repo found for repo': return 'permanent', 'huggingface_no_repo', False if source_type == 'docker' and 'no child with platform linux/amd64' in detail_lower: return 'permanent', 'docker_no_linux_amd64', False if any(token in combined for token in ('timed out', 'timeout', 'deadline exceeded')): return 'error', 'timeout', True if ( source_type == 'docker' and payload and payload.get('msg') == 'error processing layer' and payload.get('error') == 'unexpected EOF' ): # Keep the persisted retry class; layer EOF alone does not prove a network cause. return 'error', 'network', True if any(token in combined for token in ( 'connection reset', 'connection aborted', 'connection refused', 'could not resolve host', 'temporary failure', 'tls', 'ssl', 'proxy error', 'network is unreachable', 'unexpected eof', )): return 'error', 'network', True if any(token in combined for token in (' 408', ' 429', ' 500', ' 502', ' 503', ' 504', 'too many requests', 'rate limit')): return 'error', 'remote_transient', True auth_error = any(token in combined for token in ( 'authentication failed', 'unauthorized', 'invalid username or token', 'invalid api key', 'bad credentials', )) if source_type == 'huggingface' and direct_kind == 'huggingface_space_v1' and ( auth_error or any(token in combined for token in ( 'permission denied', 'repository not found', 'private repository', 'gated repo', ' 401', ' 403', )) ): return 'permanent', 'huggingface_inaccessible', False if source_type == 'docker' and direct_kind == 'docker_direct_v1' and ( auth_error or any(token in combined for token in ( 'permission denied', 'pull access denied', 'requested access to the resource is denied', 'insufficient scope', 'manifest unknown', 'name unknown', 'repository does not exist', ' 401', ' 403', ' 404', )) ): return 'permanent', 'docker_registry_access', False if source_type == 'docker' and auth_error: # A registry manifest may be private or stale; the rotating credential cannot be # attributed from this target diagnostic, so it must not disable the whole pool. return 'error', 'docker_registry_access', False if auth_error: return 'error', 'source_auth', True if 'permission denied' in combined: return 'error', 'auth_or_permission', True if any(token in combined for token in ('killed by signal', 'terminated by signal', 'process terminated', 'segmentation fault')): return 'error', 'command_exit', True if ( source_type in ('docker', 'filesystem') and payload and message == 'skipping file: size exceeds max allowed' and not any(key in payload for key in ('error', 'errors', 'message')) ): return 'warning' if returncode == 0 else 'error', 'archive_member_size', False if returncode == 0: if message_lower == 'error cleaning temporary artifacts': return 'warning', 'cleanup', False if message_lower == 'skipping result: invalid' and detail_lower == 'empty raw': return 'warning', 'invalid_empty_result', False if message_lower == 'non-critical error processing chunk': return 'warning', 'chunk_processing', False if source_type == 'git' and message_lower == 'error reading chunk' and detail_lower == 'brotli: excessive input': return 'warning', 'chunk_read', False if source_type in ('npm', 'pypi', 'postman', 'filesystem') and message_lower == 'error reading chunk' and any( token in detail_lower for token in ('brotli:', 'flate: corrupt input', 'error identifying archive', 'invalid header') ): return 'warning', 'chunk_read', False if source_type == 'docker' and message_lower == 'error processing layer' and detail_lower == 'gzip: invalid header': return 'warning', 'docker_layer_gzip', False if payload and ('error' in level or 'error' in message_lower or payload.get('error')): return 'error', 'trufflehog', True if not payload and re.search(r'\b(error|failed|fatal|panic)\b', combined): return 'error', 'command', True # Only observed structured progress is exempt from retention, never error details. if ( source_type in ('docker', 'filesystem') and payload and payload.get('logger') == 'trufflehog' and not any(key in payload for key in ('error', 'errors', 'message')) and ( (level == 'info-0' and message in ('running source', 'finished scanning')) or (level == 'info-2' and message in ( 'trufflehog dev', 'starting scanner workers', 'starting detector workers', 'starting verificationOverlap workers', 'starting notifier workers', 'enumerating source', )) or (source_type == 'docker' and level == 'info-2' and message in ( 'scanning image', 'scanning image history', 'scanning image history entry', 'scanning image layers', 'scanning layer', )) ) ): return 'routine', '', False return 'info', '', False def _trufflehog_diagnostic_limits(): return { 'lines': min(2000, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_lines', 2000), 2000))), 'line_chars': min(8192, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_line_chars', 8192), 8192))), 'line_bytes': min(8192, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_line_bytes', 8192), 8192))), 'errors': min(200, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_errors', 200), 200))), 'warnings': min(200, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_warnings', 200), 200))), 'unclassified': min(20, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_unclassified', 20), 20))), } def _iter_output_lines(value): if isinstance(value, str): yield from io.StringIO(value) return if isinstance(value, bytes): for raw_line in io.BytesIO(value): yield raw_line.decode('utf-8', errors='replace') return if value is None: return yield from value def apply_trufflehog_diagnostics( results, stderr, returncode, source_type, require_completion=False, redactions=(), ): errors = [] warnings = [] warning_classes = [] warning_retryability = [] permanent = [] unclassified = [] limits = _trufflehog_diagnostic_limits() line_count = 0 timed_out_seen = False finished_seen = False output_limit = '' captured_stdout = captured_stderr = None captured_transformation = None if isinstance(stderr, StreamedCommandOutput): captured_stdout = stderr.raw_stdout_bytes() captured_stderr = stderr.raw_stderr_bytes() captured_transformation = ( 'legacy E-frame classification parsed raw process stderr with any ' 'pre-existing configured parser redactions; ' 'canonical process material preserves the pre-parse captured bytes' ) stderr = stderr.stderr_lines(redactions=redactions) try: for raw_line in _iter_output_lines(stderr): if line_count >= limits['lines']: output_limit = f'total line limit of {limits["lines"]} exceeded' break line_count += 1 if len(raw_line) > limits['line_chars']: output_limit = f'line character limit of {limits["line_chars"]} exceeded' break if len(raw_line.encode('utf-8', errors='replace')) > limits['line_bytes']: output_limit = f'line byte limit of {limits["line_bytes"]} exceeded' break line = raw_line.strip() if not line: continue try: payload = json.loads(line) if isinstance(payload, dict) and str(payload.get('msg') or '').strip().lower() == 'finished scanning': finished_seen = True except (TypeError, ValueError): pass if 'timed out' in line.lower(): timed_out_seen = True severity, error_class, retryable = _trufflehog_diagnostic_policy(line, source_type, returncode) if severity == 'warning': if len(warnings) + len(permanent) >= limits['warnings']: output_limit = f'retained warning limit of {limits["warnings"]} exceeded' break warnings.append(line) warning_retryability.append(bool(retryable)) if error_class: warning_classes.append(error_class) elif severity == 'permanent': if len(warnings) + len(permanent) >= limits['warnings']: output_limit = f'retained warning limit of {limits["warnings"]} exceeded' break permanent.append((line, error_class)) elif severity == 'error': if len(errors) >= limits['errors']: output_limit = f'retained error limit of {limits["errors"]} exceeded' break errors.append((line, error_class, retryable)) if error_class == 'memory_limit': break elif severity != 'routine': # Routine records still count against the hard limits above. if len(unclassified) >= limits['unclassified']: output_limit = f'retained unclassified limit of {limits["unclassified"]} exceeded' break unclassified.append(line) except CommandOutputLimitError as exc: output_limit = str(exc) if output_limit: synthetic = ( f'TruffleHog diagnostic output_limit reached: {output_limit}; ' 'remaining diagnostic output was not retained' )[:limits['line_chars']] errors = errors[:max(0, limits['errors'] - 1)] errors.append((synthetic, 'source_resource', True)) if permanent and not errors: warnings.extend(line for line, _ in permanent) warning_classes.extend(error_class for _, error_class in permanent if error_class) classes = {error_class for _, error_class in permanent} if classes in ({'huggingface_no_repo'}, {'huggingface_inaccessible'}): results['skipped'] = 'HuggingFace Space repository is unavailable' elif classes == {'docker_no_linux_amd64'}: results['skipped'] = 'Docker image has no linux/amd64 manifest' elif classes == {'docker_registry_access'}: results['skipped'] = 'Docker image is unavailable to the worker' else: results['skipped'] = 'target is permanently unavailable' results['error_class'] = next(iter(classes), 'permanent') results['retryable'] = False else: warnings.extend(line for line, _ in permanent) warning_classes.extend(error_class for _, error_class in permanent if error_class) completion_required = ( source_type == 'docker' or bool(require_completion) or (source_type == 'git' and 'chunk_read' in warning_classes) ) if completion_required and not errors and not permanent and not results.get('skipped'): if returncode != 0 and finished_seen: errors.append((f'TruffleHog exited with code {returncode} after the completion marker', 'wrapper_exit', True)) elif returncode != 0 or not finished_seen: errors.append((f'TruffleHog exited with code {returncode} before the completion marker', 'command_incomplete', True)) elif returncode != 0 and not errors and not permanent: errors.append((f'TruffleHog exited with code {returncode} without a fatal diagnostic', 'command_exit', True)) elif returncode != 0 and warnings and not errors and not results.get('skipped'): errors.append((f'TruffleHog exited with code {returncode} after non-fatal diagnostics', 'command_exit', True)) if errors: results['errors'] = [line for line, _, _ in errors] classes = [error_class for _, error_class, _ in errors if error_class] results['error_class'] = classes[0] if len(set(classes)) <= 1 else 'mixed' results['retryable'] = all(retryable for _, _, retryable in errors) results['source_failure'] = any(error_class in classes for error_class in ('source_configuration', 'source_resource', 'source_auth')) if results['source_failure']: results['source_failure_category'] = 'source_auth' if 'source_auth' in classes else 'source_resource' if 'source_resource' in classes else 'source_configuration' results['source_failure_auth_related'] = 'source_auth' in classes if output_limit: results['error_class'] = 'source_resource' results['retryable'] = True results['source_failure'] = True results['source_failure_category'] = 'source_resource' results['source_failure_auth_related'] = False if warnings: results['warnings'] = warnings results['warning_classes'] = sorted(set(warning_classes)) results['degraded'] = not bool(results.get('skipped')) # Nonfatal coverage warnings must not suppress retries of fatal errors. if warning_retryability and not errors: warnings_retryable = all(warning_retryability) results['retryable'] = bool( results.get('retryable', True) ) and warnings_retryable scan_meta = results.setdefault('scan_meta', {}) scan_meta['trufflehog_returncode'] = returncode scan_meta['trufflehog_finished'] = finished_seen scan_meta['command_timed_out'] = returncode == -1 and timed_out_seen scan_meta['diagnostic_lines_processed'] = line_count if warning_retryability: scan_meta['trufflehog_warnings_retryable'] = all(warning_retryability) if output_limit: scan_meta['diagnostic_output_limited'] = True scan_meta['diagnostic_output_limit_reason'] = output_limit if unclassified: scan_meta['stderr_unclassified'] = unclassified if results.get('errors') and captured_stderr is not None: results['_diagnostic_raw_stdout_b64'] = base64.b64encode( captured_stdout or b'' ).decode('ascii') results['_diagnostic_raw_stderr_b64'] = base64.b64encode( captured_stderr ).decode('ascii') results['_diagnostic_stderr_transformation'] = captured_transformation return results def convert_package_git_unavailable_to_skip(result): errors = result.get('errors') or [] if not errors: return result for error in errors: text = str(error).lower() if not ( ('repository not found' in text or 'project not found' in text) and ('failed to clone' in text or 'remote:' in text or 'error preparing repo' in text) ): return result result['warnings'] = list(result.get('warnings') or []) + list(errors) result['warning_classes'] = sorted(set(list(result.get('warning_classes') or []) + ['package_git_repo_unavailable'])) result['errors'] = [] result['skipped'] = 'package_git repository is unavailable or private' result['error_class'] = 'package_git_repo_unavailable' result['retryable'] = False result['degraded'] = False return result def apply_result_error_scope(result): errors = result.get('errors') or [] if not errors or result.get('error_class'): return result text = '\n'.join(str(error) for error in errors).lower() if any(token in text for token in ( 'no space left', 'not enough free space', 'disk quota', 'unable to create npm work dir', 'unable to create pypi work dir', 'unable to create postman work dir', 'unable to create github actions work dir', 'unable to create gitlab ci work dir', )): result['error_class'] = 'source_resource' result['retryable'] = True result['source_failure'] = True result['source_failure_category'] = 'source_resource' return result def append_trufflehog_findings(results, stdout): invalid = [] if _client_scan_policy.get() is None: max_findings = max(1, int(os.getenv('TRUFFLEHOG_MAX_FINDINGS_PER_TARGET', '20000'))) else: max_findings = max(1, int(_scan_policy_value( 'trufflehog_max_findings_per_target', 20000, ))) try: lines = _iter_output_lines(stdout) for line in lines: if not line.strip(): continue try: finding = json.loads(line) if not isinstance(finding, dict): raise ValueError('finding JSON is not an object') if len(results.setdefault('findings', [])) >= max_findings: results.setdefault('errors', []).append(f'TruffleHog findings exceeded {max_findings} per target') results['error_class'] = 'output_limit' results['retryable'] = False break results['findings'].append(finding) except (json.JSONDecodeError, ValueError) as exc: if len(invalid) < 5: invalid.append(f'{str(exc)}: {line[:300]}') except CommandOutputLimitError as exc: invalid.append(str(exc)) if invalid: results.setdefault('errors', []).append('Malformed TruffleHog JSON output: ' + '; '.join(invalid)) results['error_class'] = 'output_parse' results['retryable'] = True return results def parse_git_scan_target(target): text = str(target or '').strip() if not text.startswith('{'): return {'url': text, 'branch': '', 'metadata': {}} try: data = json.loads(text) except (TypeError, ValueError): return {'url': text, 'branch': '', 'metadata': {}} if not isinstance(data, dict): return {'url': text, 'branch': '', 'metadata': {}} return { 'url': str(data.get('url') or data.get('repo_url') or text).strip(), 'branch': str(data.get('branch') or '').strip(), 'metadata': data, } def git_branch_ref(branch): branch = str(branch or '') resolution = { 'provider': 'github', 'repo_url': 'https://github.com/a/b.git', 'repo_path': 'a/b', 'branch': branch, 'ref': f'refs/heads/{branch}', 'head_sha': '0' * 40, 'ref_source': 'explicit', } validate_git_resolution(resolution) return resolution['ref'] def normalize_git_scan_resolution_target(target, provider=None): target_info = parse_git_scan_target(target) raw_url = str(target_info['url'] or '').strip() if not raw_url or re.search(r'[\x00-\x20\x7f]', raw_url): raise ValueError('Git target URL is empty or contains control characters') if re.search(r'%(?:2f|5c)', raw_url, flags=re.IGNORECASE): raise ValueError('Git target URL contains an encoded path separator') parse_url = raw_url[4:] if raw_url.startswith('git+') else raw_url if parse_url.startswith(('github:', 'gitlab:')): if any(marker in parse_url for marker in ('?', '#', '@')): raise ValueError('Git target shorthand contains unsafe URL components') else: try: parsed = urlsplit(parse_url) port = parsed.port except ValueError as exc: raise ValueError('Git target URL is malformed') from exc if ( parsed.scheme not in ('http', 'https', 'git') or not parsed.netloc or parsed.username is not None or parsed.password is not None or parsed.query or parsed.fragment or port is not None ): raise ValueError('Git target URL contains unsupported or unsafe components') normalized = normalize_git_repo_candidate(raw_url) if not normalized: raise ValueError('Git target is not a supported GitHub or GitLab repository') expected_provider = str(provider or '').strip().lower() if expected_provider and normalized['provider'] != expected_provider: raise ValueError('Git target provider does not match the scan source') metadata = target_info['metadata'] branch = str(target_info.get('branch') or '').strip() raw_ref = str(metadata.get('ref') or '').strip() if isinstance(metadata, dict) else '' ref_branch = '' if raw_ref: prefix = 'refs/heads/' if not raw_ref.startswith(prefix): raise ValueError('Git target ref must be a branch ref') ref_branch = raw_ref[len(prefix):] git_branch_ref(ref_branch) if branch: git_branch_ref(branch) if branch and ref_branch and branch != ref_branch: raise ValueError('Git target branch and ref hints conflict') branch = branch or ref_branch return normalized, branch def git_ref_resolution_api_json( provider, url, token, deadline, request_attempts, timeout_sec, max_response_bytes, ): response = api_request( 'GET', url, headers=github_headers(token) if provider == 'github' else gitlab_headers(token), timeout=max(0.001, float(timeout_sec)), max_retries=max(1, int(request_attempts)), retry_delay=1, deadline=deadline, stream=True, allow_redirects=False, ) if 300 <= response.status_code < 400: response.close() raise ApiRequestError(f'{provider} ref resolution refused an HTTP redirect') if response.status_code >= 400: try: error = github_api_error(response) if provider == 'github' else gitlab_api_error(response) finally: response.close() raise error payload = bounded_response_json(response, max_bytes=max_response_bytes) if not isinstance(payload, dict): raise ApiRequestError(f'{provider} ref resolution response is not an object') return payload def redacted_git_resolution_error(exc, token): message = redact_secrets(str(exc), [token]) if isinstance(exc, RateLimitError): return RateLimitError( exc.source, message, reset_at=exc.reset_at, category=exc.category, retryable=exc.retryable, auth_related=exc.auth_related, ) if isinstance(exc, ApiRequestError): return ApiRequestError(message) return exc def resolve_github_ref_head( repo_path, token=None, ref_hint=None, *, request_attempts=2, timeout_sec=10, deadline=None, max_response_bytes=1 << 20, ): deadline = float(deadline) if deadline is not None else time.monotonic() + max(0.001, float(timeout_sec)) encoded_repo = quote(str(repo_path), safe='/') branch = str(ref_hint or '') ref_source = 'explicit' if branch else 'provider_default' try: if not branch: payload = git_ref_resolution_api_json( 'github', f'https://api.github.com/repos/{encoded_repo}', token, deadline, request_attempts, timeout_sec, max_response_bytes, ) branch = str(payload.get('default_branch') or '') git_branch_ref(branch) ref = f'refs/heads/{branch}' payload = git_ref_resolution_api_json( 'github', f'https://api.github.com/repos/{encoded_repo}/git/ref/{quote("heads/" + branch, safe="")}', token, deadline, request_attempts, timeout_sec, max_response_bytes, ) obj = payload.get('object') if payload.get('ref') != ref or not isinstance(obj, dict) or obj.get('type') != 'commit': raise ApiRequestError('GitHub ref resolution returned a mismatched commit ref') resolved = { 'provider': 'github', 'repo_url': f'https://github.com/{repo_path}.git', 'repo_path': str(repo_path), 'branch': branch, 'ref': ref, 'head_sha': str(obj.get('sha') or '').lower(), 'ref_source': ref_source, } return validate_git_resolution(resolved) except (RateLimitError, ApiRequestError) as exc: sanitized = redacted_git_resolution_error(exc, token) if sanitized is exc: raise raise sanitized from exc def resolve_gitlab_ref_head( repo_path, token=None, ref_hint=None, *, request_attempts=2, timeout_sec=10, deadline=None, max_response_bytes=1 << 20, ): deadline = float(deadline) if deadline is not None else time.monotonic() + max(0.001, float(timeout_sec)) project_url = f'https://gitlab.com/api/v4/projects/{quote(str(repo_path), safe="")}' branch = str(ref_hint or '') ref_source = 'explicit' if branch else 'provider_default' try: if not branch: payload = git_ref_resolution_api_json( 'gitlab', project_url, token, deadline, request_attempts, timeout_sec, max_response_bytes, ) branch = str(payload.get('default_branch') or '') git_branch_ref(branch) payload = git_ref_resolution_api_json( 'gitlab', f'{project_url}/repository/branches/{quote(branch, safe="")}', token, deadline, request_attempts, timeout_sec, max_response_bytes, ) commit = payload.get('commit') if payload.get('name') != branch or not isinstance(commit, dict): raise ApiRequestError('GitLab ref resolution returned a mismatched branch') resolved = { 'provider': 'gitlab', 'repo_url': f'https://gitlab.com/{repo_path}.git', 'repo_path': str(repo_path), 'branch': branch, 'ref': f'refs/heads/{branch}', 'head_sha': str(commit.get('id') or '').lower(), 'ref_source': ref_source, } return validate_git_resolution(resolved) except (RateLimitError, ApiRequestError) as exc: sanitized = redacted_git_resolution_error(exc, token) if sanitized is exc: raise raise sanitized from exc def resolve_git_scan_target( target, provider, token=None, *, request_attempts=2, timeout_sec=10, max_response_bytes=1 << 20, ): normalized, branch = normalize_git_scan_resolution_target(target, provider) resolver = resolve_github_ref_head if normalized['provider'] == 'github' else resolve_gitlab_ref_head return resolver( normalized['repo_path'], token, branch or None, request_attempts=request_attempts, timeout_sec=timeout_sec, max_response_bytes=max_response_bytes, ) def validate_bound_git_scan_plan(plan, target, provider=None): if not isinstance(plan, dict): raise ValueError('exact Git scan requires a bound plan object') required = { 'version', 'provider', 'repo_url', 'repo_path', 'branch', 'ref', 'head_sha', 'ref_source', 'base_sha', 'mode', 'baseline_depth', } if set(plan) != required or plan.get('version') != 1: raise ValueError('bound Git scan plan has an unsupported shape') resolution = validate_git_resolution(plan) normalized, _ = normalize_git_scan_resolution_target(target, provider or resolution['provider']) if normalized['repo_url'] != resolution['repo_url'] or normalized['repo_path'] != resolution['repo_path']: raise ValueError('bound Git scan plan repository conflicts with the target') mode = str(plan.get('mode') or '') base_sha = plan.get('base_sha') if base_sha is not None: base_sha = str(base_sha).lower() if not re.fullmatch(r'[a-f0-9]{40}|[a-f0-9]{64}', base_sha): raise ValueError('bound Git scan plan has an invalid base SHA') try: baseline_depth = int(plan.get('baseline_depth')) except (TypeError, ValueError) as exc: raise ValueError('bound Git scan plan has an invalid baseline depth') from exc if not 1 <= baseline_depth <= 1000000: raise ValueError('bound Git scan plan baseline depth is out of range') if ( (mode == 'baseline' and base_sha is not None) or (mode == 'delta' and (not base_sha or base_sha == resolution['head_sha'])) or (mode == 'noop' and base_sha != resolution['head_sha']) or mode not in ('baseline', 'delta', 'noop') ): raise ValueError('bound Git scan plan mode and base are inconsistent') normalized_plan = { 'version': 1, **resolution, 'base_sha': base_sha, 'mode': mode, 'baseline_depth': baseline_depth, } if canonical_git_scan_plan_bytes(normalized_plan) != canonical_git_scan_plan_bytes(plan): raise ValueError('bound Git scan plan is not normalized') return normalized_plan, hashlib.sha256(canonical_git_scan_plan_bytes(plan)).hexdigest() def git_delta_base_unavailable(result, base_sha): diagnostics = '\n'.join(str(item) for item in ( list(result.get('errors') or []) + list(result.get('warnings') or []) )).lower() if not diagnostics: return False base_markers = ( 'bad object', 'unknown revision', 'invalid object', 'object not found', 'reference not found', 'could not find commit', 'unable to resolve commit', 'invalid since commit', 'since-commit', 'since commit', ) return any(marker in diagnostics for marker in base_markers) and ( str(base_sha or '').lower()[:12] in diagnostics or 'since' in diagnostics or 'commit' in diagnostics or 'revision' in diagnostics or 'object' in diagnostics ) def git_checkout_recovery_allowed(result): meta = result.get('scan_meta') or {} errors = result.get('errors') or [] if ( os.name != 'nt' or len(errors) not in (1, 2) or result.get('error_class') != 'trufflehog' or result.get('source_failure') or result.get('warnings') or result.get('degraded') or result.get('skipped') or meta.get('trufflehog_returncode') != 1 or meta.get('trufflehog_finished') is not False or meta.get('diagnostic_output_limited') or meta.get('command_timed_out') ): return False limits = _trufflehog_diagnostic_limits() companion = None if len(errors) == 2: companion_line = errors[0] if ( not isinstance(companion_line, str) or len(companion_line) > limits['line_chars'] or len(companion_line.encode('utf-8', errors='replace')) > limits['line_bytes'] ): return False try: companion = json.loads(companion_line) except (TypeError, ValueError): return False if ( not isinstance(companion, dict) or set(companion) != { 'level', 'ts', 'logger', 'msg', 'subcommand', 'repo', 'path', 'args', 'error', } or companion.get('level') != 'info-0' or companion.get('logger') != 'trufflehog' or companion.get('msg') != 'git clone failed' or companion.get('subcommand') != 'git clone' or companion.get('args') != [] or any(not isinstance(companion.get(key), str) or not companion.get(key) for key in ( 'ts', 'repo', 'path', 'error', )) ): return False line = errors[-1] if not isinstance(line, str) or len(line) > limits['line_chars'] or len(line.encode('utf-8', errors='replace')) > limits['line_bytes']: return False try: payload = json.loads(line) except (TypeError, ValueError): return False if ( not isinstance(payload, dict) or payload.get('msg') != 'error running scan' or payload.get('level') != 'error' or payload.get('errors') ): return False detail = payload.get('error') if not isinstance(detail, str): return False if companion is not None and companion['error'] not in detail: return False detail = detail.lower() if not all(marker in detail for marker in ( 'error preparing repo', 'error executing git clone: exit status 128', 'clone succeeded, but checkout failed', )): return False prefix, _, git_stderr = detail.partition('error executing git clone: exit status 128') if re.search(r'\b(?:fatal|error):', prefix): return False quoted_path = r"(?:'(?:[^'\\\r\n]|\\.)+'|\"(?:[^\"\\\r\n]|\\.)+\")" path_failure = False for physical_line in git_stderr.lstrip(' ,').splitlines(): line_match = re.match(r'^(?:remote:\s*)?(?:fatal|error):\s*(.*)$', physical_line.strip()) if not line_match: if re.search(r'\b(?:fatal|error):', physical_line): return False continue cause = line_match.group(1) if re.fullmatch(r'invalid path ' + quoted_path, cause): path_failure = True continue long_path = re.fullmatch(r'(?:unable to create file |cannot create directory (?:at )?)(.+): filename too long', cause) if long_path: path = long_path.group(1) if path.startswith(("'", '"')): if not re.fullmatch(quoted_path, path): return False elif re.search(r'\b(?:fatal|error):', path): return False path_failure = True elif cause != 'unable to checkout working tree': return False return path_failure def scan_exact_git_plan( target, plan, plan_sha256, timeout_sec, detectors, exclude_detectors, no_verification, trufflehog_config, token, external_trufflehog_lifecycle, ): deadline = time.monotonic() + max(0.001, float(timeout_sec or 1)) provider = plan['provider'] scan_url, secrets_to_redact = build_authenticated_git_url(plan['repo_url'], provider, token) command_env = os.environ.copy() _append_windows_git_longpaths(command_env, 'Exact Git') if secrets_to_redact: command_env['TRUF_GIT_TOKEN'] = token command_env['TRUF_GIT_USERNAME'] = 'oauth2' if provider == 'gitlab' else 'x-access-token' recovery_root = None local_url = None checkout_errors = [] findings = [] cleanup_safe = True recovery = {'attempted': False, 'clone_succeeded': False, 'coverage_complete': False} def remaining(): seconds = deadline - time.monotonic() if seconds <= 0: raise subprocess.TimeoutExpired('exact Git scan', timeout_sec) return seconds def run_mode(mode): nonlocal recovery_root, local_url remaining() cmd = [ get_trufflehog_cmd(), 'git', local_url or scan_url, '--json', '--no-update', '--branch', plan['head_sha'], ] if external_trufflehog_lifecycle: cmd.append('--local-dev') append_trufflehog_scan_args( cmd, detectors, exclude_detectors, no_verification, trufflehog_config, ) if mode in ('baseline', 'baseline_reset'): cmd.extend(['--max-depth', str(plan['baseline_depth'])]) elif mode == 'delta': cmd.extend(['--since-commit', plan['base_sha']]) result = {'findings': findings, 'errors': []} emit_client_scan_phase('scanning', { 'integrated_operation': 'git_acquisition_and_scan', 'execution_mode': mode, }) with run_command_streamed( cmd, remaining(), command_env, deadline=deadline, staging_roots=(recovery_root,) if recovery_root else None, ) as output: apply_trufflehog_diagnostics( result, output, output.returncode, 'git', require_completion=True, redactions=secrets_to_redact, ) append_trufflehog_findings( result, output.stdout_lines(redactions=secrets_to_redact), ) checkout_candidate = not recovery['attempted'] and git_checkout_recovery_allowed(result) if checkout_candidate: checkout_errors.extend(result['errors']) if checkout_candidate: remaining() recovery['attempted'] = True emit_client_scan_phase('cloning', { 'operation': 'git_clone_recovery', }) recovery_root = create_command_work_dir() destination = os.path.join(recovery_root, 'repo') clone_cmd = [get_git_cmd(), 'clone', '--no-checkout', '--no-recurse-submodules', '--', scan_url, destination] clone_result = {'errors': []} with run_command_streamed( clone_cmd, remaining(), command_env, deadline=deadline, staging_roots=(recovery_root,), native_git_clone=True, ) as output: apply_trufflehog_diagnostics( clone_result, output, output.returncode, 'git', redactions=secrets_to_redact, ) try: for _ in output.stdout_lines(max_line_bytes=8192, max_lines=2000, redactions=secrets_to_redact): pass except CommandOutputLimitError: clone_result.setdefault('errors', []).append('Git clone output exceeded its diagnostic bounds') clone_result.update(error_class='output_limit', retryable=False) if clone_result.get('errors') or clone_result.get('skipped') or clone_result.get('degraded'): return clone_result recovery['clone_succeeded'] = True # Native Windows TH expects the drive in the file URI authority, not /C:/. from pathlib import Path local_url = Path(destination).as_uri() if os.name == 'nt': local_url = local_url.replace('file:///', 'file://', 1) return run_mode(mode) return result execution_mode = plan['mode'] continuity_reset = False result = {'findings': [], 'errors': []} if execution_mode == 'noop': emit_client_scan_phase('scanning', { 'operation': 'exact_git_noop', 'execution_mode': 'noop', }) result = {'findings': [], 'errors': []} else: try: result = run_mode(execution_mode) if ( execution_mode == 'delta' and (not recovery['attempted'] or recovery['clone_succeeded']) and git_delta_base_unavailable(result, plan['base_sha']) ): execution_mode = 'baseline_reset' continuity_reset = True result = run_mode(execution_mode) result.setdefault('scan_meta', {})['git_continuity_reset_reason'] = 'covered base unavailable' except ScanSlotFatalError: cleanup_safe = False raise except subprocess.TimeoutExpired: result.setdefault('errors', []).append('Exact Git scan exhausted its absolute deadline') result.update(error_class='timeout', retryable=True) except Exception as exc: message = redact_secrets(str(exc), [token]) logger.error('Error scanning pinned Git repository %s: %s', plan['repo_url'], message) result.setdefault('errors', []).append(f'Scan failed: {message}') result.update(retryable=True, error_class='remote_transient') finally: if recovery_root and cleanup_safe: try: cleanup_command_work_dir(recovery_root) except ScanSlotFatalError: raise except Exception as exc: result.setdefault('errors', []).append('Git recovery cleanup failed: ' + redact_secrets(str(exc), [token])) result.update(error_class='source_resource', retryable=True, source_failure=True, source_failure_category='source_resource', source_failure_auth_related=False) result['findings'] = findings # Freeze scan coverage before optional filtering or candidate staging adds warnings. success = not result.get('errors') and not result.get('skipped') and not result.get('degraded') result['git_scan_plan'] = plan result['git_scan_execution'] = { 'mode': execution_mode, 'pinned': True, 'success': bool(success), 'coverage_complete': bool(success), 'continuity_reset': continuity_reset, 'plan_sha256': plan_sha256, } result.setdefault('scan_meta', {})['exact_git_scope'] = { 'provider': plan['provider'], 'ref': plan['ref'], 'head_sha': plan['head_sha'], 'base_sha': plan['base_sha'], 'mode': execution_mode, 'baseline_depth': plan['baseline_depth'], 'ref_source': plan['ref_source'], 'pinned': True, 'continuity_reset': continuity_reset, } result = apply_finding_filters(result, target) if execution_mode != 'noop' and time.monotonic() >= deadline: if not result.get('errors'): result.update(error_class='timeout', retryable=True) result.setdefault('errors', []).append('Exact Git scan exceeded its absolute deadline including cleanup and filtering') result.setdefault('scan_meta', {})['git_deadline_exceeded'] = True result['git_scan_execution'].update(success=False, coverage_complete=False) if result.get('errors'): result['git_scan_execution'].update(success=False, coverage_complete=False) if recovery['attempted']: recovery['coverage_complete'] = result['git_scan_execution']['coverage_complete'] result.setdefault('scan_meta', {})['git_checkout_recovery'] = recovery if checkout_errors and not result['git_scan_execution']['coverage_complete']: result['errors'] = checkout_errors + list(result.get('errors') or []) result.setdefault('error_class', 'trufflehog') result.setdefault('retryable', True) return result def scan_git_repo(repo_url, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, provider=None, max_depth=None, max_commit_age_days=None, commit_lookup_pages=3, skip_if_commit_lookup_fails=True, external_trufflehog_lifecycle=False, git_plan=None): """Scan a single Git repository for secrets""" target_info = parse_git_scan_target(repo_url) original_target = repo_url repo_url = target_info['url'] parsed_repo_url = urlsplit(repo_url) if parsed_repo_url.username or parsed_repo_url.password: return {"findings": [], "errors": ["Git target URL must not contain userinfo credentials"], "retryable": False, "error_class": "invalid_target"} target_branch = target_info['branch'] target_metadata = target_info['metadata'] if git_plan is not None: try: plan, plan_sha256 = validate_bound_git_scan_plan(git_plan, original_target, provider) except (TypeError, ValueError) as exc: return { 'findings': [], 'errors': [f'Bound Git plan rejected: {exc}'], 'retryable': False, 'error_class': 'invalid_target', } return scan_exact_git_plan( original_target, plan, plan_sha256, timeout_sec, detectors, exclude_detectors, no_verification, trufflehog_config, token, external_trufflehog_lifecycle, ) branch_label = f" branch={target_branch}" if target_branch else '' logger.info(f"Scanning Git repository: {repo_url}{branch_label}") emit_client_scan_phase('resolving', { 'operation': 'recent_commit_boundary', 'provider': str(provider or 'git'), }) boundary = recent_commit_boundary(repo_url, provider, token, max_commit_age_days, commit_lookup_pages) if boundary.get('error'): category = str(boundary.get('error_category') or 'unknown') auth_related = bool(boundary.get('auth_related')) return { "findings": [], "errors": [boundary.get('reason') or 'commit age lookup failed'], "error_class": 'source_auth' if auth_related else 'remote_transient', "retryable": True, "source_failure": True, "source_failure_category": category, "source_failure_auth_related": auth_related, "scan_meta": {**boundary, 'target_metadata': target_metadata}, } if boundary.get('skip'): reason = boundary.get('reason', 'skipped by commit age filter') logger.info(f"Skipping {repo_url}: {reason}") if boundary.get('permanent') or skip_if_commit_lookup_fails: return {"findings": [], "errors": [], "skipped": reason, "scan_meta": {**boundary, 'target_metadata': target_metadata}} effective_provider, _ = get_git_provider_and_path(repo_url, provider) scan_url, secrets_to_redact = build_authenticated_git_url(repo_url, effective_provider, token) cmd = [get_trufflehog_cmd(), 'git', scan_url, '--json', '--no-update'] if external_trufflehog_lifecycle: cmd.append('--local-dev') append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config) if max_depth: cmd.extend(['--max-depth', str(max_depth)]) if target_branch: cmd.extend(['--branch', target_branch]) if boundary.get('since_commit'): cmd.extend(['--since-commit', boundary['since_commit']]) logger.info( f"Scanning {repo_url} since commit {boundary['since_commit']} " f"({boundary.get('recent_commit_count')} commits after {boundary.get('cutoff')})" ) try: command_env = os.environ.copy() if secrets_to_redact: command_env['TRUF_GIT_TOKEN'] = token command_env['TRUF_GIT_USERNAME'] = 'oauth2' if effective_provider == 'gitlab' else 'x-access-token' results = {"findings": [], "errors": []} emit_client_scan_phase('scanning', { 'integrated_operation': 'git_acquisition_and_scan', }) with run_command_streamed(cmd, timeout_sec, command_env) as output: apply_trufflehog_diagnostics( results, output, output.returncode, 'git', require_completion=external_trufflehog_lifecycle, redactions=secrets_to_redact, ) append_trufflehog_findings( results, output.stdout_lines(redactions=secrets_to_redact), ) if boundary.get('since_commit') or target_metadata or target_branch: results.setdefault("scan_meta", {}).update({**boundary, 'branch': target_branch, 'target_metadata': target_metadata}) return apply_finding_filters(results, original_target) except ScanSlotFatalError: raise except Exception as e: logger.error(f"Error scanning repository {repo_url}: {str(e)}") return {"findings": [], "errors": [f"Scan failed: {str(e)}"]} DOCKER_ARCHIVE_MAX_DECODED_BYTES = 1 << 30 def _require_docker_archive_policy(limits): # This independent decoded-stream ceiling is part of docker-layer-execution-v4. if limits['archive_max_size_bytes'] > DOCKER_ARCHIVE_MAX_DECODED_BYTES: raise DockerLayerInfrastructureError( 'archive_policy_incompatible', 'Docker member policy exceeds the decoded validation ceiling', category='source_configuration', ) def validate_docker_content_artifact( path, descriptor, *, deadline=None, max_member_bytes=256 << 20, max_decoded_bytes=DOCKER_ARCHIVE_MAX_DECODED_BYTES, max_members=100000, ): import zlib deadline = min(float(deadline) if deadline is not None else float('inf'), time.monotonic() + 30) if not math.isfinite(deadline) or any( isinstance(value, bool) or not isinstance(value, int) or not 0 < value <= bound for value, bound in ((max_member_bytes, DOCKER_ARCHIVE_MAX_DECODED_BYTES), (max_decoded_bytes, DOCKER_ARCHIVE_MAX_DECODED_BYTES), (max_members, 100000)) ): raise DockerLayerInfrastructureError( 'archive_validation_bounds', 'Docker archive validation bounds are invalid', category='source_configuration', ) def check_deadline(): _raise_if_scan_slot_fatal() if time.monotonic() >= deadline: raise DockerContentScanError('archive_timeout', 'Docker archive validation deadline expired', True) check_deadline() kind = str((descriptor or {}).get('kind') or '') media_type = str((descriptor or {}).get('media_type') or '').strip().lower() expected_size = (descriptor or {}).get('size') try: actual_size = os.path.getsize(path) except OSError as exc: raise DockerLayerInfrastructureError( 'private_storage', 'Docker content artifact is unavailable', category='source_resource', ) from exc if actual_size != expected_size: raise DockerContentScanError( 'size_mismatch', 'Docker content artifact size changed after verification', ) if kind == 'config': if media_type not in DOCKER_CONFIG_MEDIA_TYPES: raise DockerContentScanError( 'unsupported_media_type', 'Docker configuration media type is unsupported', ) if actual_size > 64 << 20: raise DockerContentScanError('archive_limit', 'Docker configuration exceeds the validation bound') try: with open(path, 'rb') as config_file: raw_config = config_file.read(actual_size + 1) except OSError as exc: raise DockerLayerInfrastructureError( 'private_storage', 'Docker configuration artifact cannot be read', category='source_resource', ) from exc try: config = json.loads(raw_config.decode('utf-8')) except (RecursionError, UnicodeDecodeError, TypeError, ValueError) as exc: raise DockerContentScanError( 'invalid_config_json', 'Docker configuration is not valid UTF-8 JSON', ) from exc if not isinstance(config, dict): raise DockerContentScanError( 'invalid_config_json', 'Docker configuration JSON must be an object', ) check_deadline() return config if kind != 'layer' or media_type not in DOCKER_LAYER_MEDIA_TYPES: raise DockerContentScanError( 'unsupported_media_type', 'Docker layer media type is unsupported', ) zstd = None if media_type.endswith('+zstd'): try: import zstandard as zstd except ImportError as exc: raise DockerLayerInfrastructureError( 'archive_decoder_unavailable', 'Docker zstd validation capability is unavailable', category='source_configuration', ) from exc decoded_bytes = 0 members = 0 terminated = False class TimedInput: def read(self, size=-1): check_deadline() data = layer_file.read(min(size if size >= 0 else 65536, 65536)) check_deadline() return data class BoundedReader: def read(self, size=-1): nonlocal decoded_bytes check_deadline() data = decoder.read(min(size if size >= 0 else 65536, 65536, max_decoded_bytes - decoded_bytes + 1)) decoded_bytes += len(data) if decoded_bytes > max_decoded_bytes: raise DockerContentScanError('archive_limit', 'Docker archive decoded byte bound exceeded') check_deadline() return data class BoundedTarInfo(tarfile.TarInfo): @classmethod def fromtarfile(cls, archive): nonlocal terminated try: check_deadline() member = super().fromtarfile(archive) check_deadline() return member except tarfile.EOFHeaderError: terminated = True raise except tarfile.HeaderError as exc: # TarFile.next otherwise tolerates some corrupt headers after member one. raise DockerContentScanError('invalid_layer_archive', 'Docker tar header is invalid') from exc def _proc_member(self, archive): nonlocal members check_deadline() members += 1 metadata = self.type in (tarfile.XHDTYPE, tarfile.XGLTYPE, tarfile.SOLARIS_XHDTYPE, tarfile.GNUTYPE_LONGNAME, tarfile.GNUTYPE_LONGLINK) if self.size < 0: raise DockerContentScanError('invalid_layer_archive', 'Docker archive member size is negative') if members > max_members or self.size > min(max_member_bytes, 1 << 20 if metadata else max_member_bytes): raise DockerContentScanError('archive_limit', 'Docker archive member bound exceeded') if self.type == tarfile.GNUTYPE_SPARSE: raise DockerContentScanError('unsupported_layer_archive', 'Docker sparse tar validation is unsupported') member = super()._proc_member(archive) check_deadline() return member def _proc_pax(self, archive): # Do not enter tarfile's unbounded hdrcharset/length regexes or sparse # map parsers. Validate complete records before decoding/applying fields. if self.size > 64 << 10: raise DockerContentScanError('archive_limit', 'Docker PAX parse byte bound exceeded') body = archive.fileobj.read(self._block(self.size)) if len(body) != self._block(self.size): raise DockerContentScanError('invalid_layer_archive', 'Docker PAX header is truncated') body = body[:self.size] records = [] position = 0 while position < len(body): check_deadline() if len(records) >= min(max_members, 1024): raise DockerContentScanError('archive_limit', 'Docker PAX record bound exceeded') space = body.find(b' ', position, min(position + 9, len(body))) if space < 0: raise DockerContentScanError('archive_limit', 'Docker PAX record length field exceeds its bound') digits = body[position:space] if not digits or not digits.isdigit(): raise DockerContentScanError('invalid_layer_archive', 'Docker PAX record length is invalid') end = position + int(digits) if end > len(body) or end <= space + 3 or body[end - 1:end] != b'\n': raise DockerContentScanError('invalid_layer_archive', 'Docker PAX record boundary is invalid') equals = body.find(b'=', space + 1, min(end - 1, space + 258)) if equals <= space + 1: raise DockerContentScanError('invalid_layer_archive', 'Docker PAX keyword is invalid or oversized') key, value = body[space + 1:equals], body[equals + 1:end - 1] if key.startswith(b'GNU.sparse.'): raise DockerContentScanError('unsupported_layer_archive', 'Docker sparse PAX validation is unsupported') converter = tarfile.PAX_NUMBER_FIELDS.get(key.decode('utf-8')) if converter is not None: if len(value) > (32 if converter is int else 64): raise DockerContentScanError('archive_limit', 'Docker PAX numeric field exceeds its bound') number = converter(value) if converter is float and not math.isfinite(number): raise DockerContentScanError('invalid_layer_archive', 'Docker PAX numeric field is not finite') if key == b'size' and not 0 <= number <= max_member_bytes: raise DockerContentScanError('archive_limit', 'Docker PAX member size exceeds its bound') records.append((key, value)) position = end check_deadline() pax_headers = archive.pax_headers.copy() charset = next((value.decode('utf-8') for key, value in records if key == b'hdrcharset'), pax_headers.get('hdrcharset')) encoding = archive.encoding if charset == 'BINARY' else 'utf-8' for key, value in records: check_deadline() key = self._decode_pax_field(key, 'utf-8', 'utf-8', archive.errors) if key in tarfile.PAX_NAME_FIELDS: value = self._decode_pax_field(value, encoding, archive.encoding, archive.errors) else: value = self._decode_pax_field(value, 'utf-8', 'utf-8', archive.errors) pax_headers[key] = value if len(pax_headers) > 1024: raise DockerContentScanError('archive_limit', 'Docker global PAX field bound exceeded') if self.type == tarfile.XGLTYPE: archive.pax_headers = pax_headers try: member = self.fromtarfile(archive) except tarfile.HeaderError as exc: raise DockerContentScanError('invalid_layer_archive', 'Docker PAX header has no following member') from exc if self.type in (tarfile.XHDTYPE, tarfile.SOLARIS_XHDTYPE): check_deadline() member._apply_pax_info(pax_headers, archive.encoding, archive.errors) member.offset = self.offset if 'size' in pax_headers: archive.offset = member.offset_data if member.isreg() or member.type not in tarfile.SUPPORTED_TYPES: archive.offset += member._block(member.size) check_deadline() return member try: with open(path, 'rb') as layer_file: magic = layer_file.read(4) layer_file.seek(0) if zstd is not None: # stream_reader can silently accept a truncated final frame. Check physical # frame/block boundaries separately, then let the decoder verify checksums. frames = 0 while layer_file.tell() < actual_size: check_deadline() frames += 1 start = layer_file.tell() header = layer_file.read(18) if frames > max_members: raise DockerContentScanError('archive_limit', 'Docker zstd frame bound exceeded') if len(header) >= 8 and 0x184d2a50 <= int.from_bytes(header[:4], 'little') <= 0x184d2a5f: end = start + 8 + int.from_bytes(header[4:8], 'little') if end > actual_size: raise DockerContentScanError('invalid_layer_archive', 'Docker zstd skippable frame is truncated') layer_file.seek(end) continue if header[:4] != b'\x28\xb5\x2f\xfd': raise DockerContentScanError('invalid_layer_archive', 'Docker zstd frame header is invalid') header_size = zstd.frame_header_size(header) params = zstd.get_frame_parameters(header) if params.window_size > max_decoded_bytes or ( params.content_size != zstd.CONTENTSIZE_UNKNOWN and params.content_size > max_decoded_bytes ): raise DockerContentScanError('archive_limit', 'Docker zstd window bound exceeded') layer_file.seek(start + header_size) while True: check_deadline() block = layer_file.read(3) if len(block) != 3: raise DockerContentScanError('invalid_layer_archive', 'Docker zstd frame is truncated') value = int.from_bytes(block, 'little') block_type, block_size = (value >> 1) & 3, value >> 3 if block_type == 3 or block_size > 128 << 10: raise DockerContentScanError('invalid_layer_archive', 'Docker zstd block is invalid') end = layer_file.tell() + (1 if block_type == 1 else block_size) if value & 1 and params.has_checksum: end += 4 if end > actual_size: raise DockerContentScanError('invalid_layer_archive', 'Docker zstd frame is truncated') layer_file.seek(end) if value & 1: break layer_file.seek(0) decoder = zstd.ZstdDecompressor(max_window_size=max(1024, max_decoded_bytes)).stream_reader( TimedInput(), read_across_frames=True, closefd=False, ) elif media_type in DOCKER_LAYER_GZIP_MEDIA_TYPES: if not magic.startswith(b'\x1f\x8b'): raise DockerContentScanError('invalid_layer_archive', 'Docker layer does not match its gzip media type') decoder = gzip.GzipFile(fileobj=TimedInput()) else: decoder = layer_file try: with tarfile.open(fileobj=BoundedReader(), mode='r|', tarinfo=BoundedTarInfo) as archive: for member in archive: if member.size > max_member_bytes: raise DockerContentScanError('archive_limit', 'Docker archive member bound exceeded') if member.sparse is not None: raise DockerContentScanError('unsupported_layer_archive', 'Docker sparse tar validation is unsupported') if member.isfile(): with archive.extractfile(member) as body: while body.read(65536): check_deadline() if not terminated or archive.fileobj.read(512) != b'\0' * 512: raise DockerContentScanError('invalid_layer_archive', 'Docker tar end marker is missing') while True: padding = archive.fileobj.read(65536) if not padding: break if padding.strip(b'\0'): raise DockerContentScanError('invalid_layer_archive', 'Docker tar has trailing non-padding content') if decoded_bytes % 512: raise DockerContentScanError('invalid_layer_archive', 'Docker tar padding is truncated') finally: if decoder is not layer_file: decoder.close() except DockerContentScanError: raise except (tarfile.TarError, EOFError, gzip.BadGzipFile, zlib.error, ValueError, RecursionError) as exc: raise DockerContentScanError( 'invalid_layer_archive', 'Docker layer is not a valid bounded tar archive', ) from exc except OSError as exc: raise DockerLayerInfrastructureError( 'private_storage', 'Docker layer artifact cannot be read', category='source_resource', ) from exc except Exception as exc: if zstd is not None and isinstance(exc, zstd.ZstdError): raise DockerContentScanError('invalid_layer_archive', 'Docker zstd stream is invalid') from exc raise def attach_docker_content_provenance( findings, plan, descriptor, positions, private_blob_path, ): location = ( f'docker://{plan["repository"]}@{plan["manifest_digest"]}/' f'{descriptor["kind"]}/{descriptor["digest"]}' ) private_blob_path = os.path.normcase(os.path.abspath(private_blob_path)) for finding in findings: if not isinstance(finding, dict): continue source = finding.setdefault('SourceMetadata', {}) data = source.setdefault('Data', {}) if isinstance(source, dict) else {} filesystem = data.get('Filesystem') if isinstance(data, dict) else None if isinstance(filesystem, dict): original = str(filesystem.get('file') or '') if private_blob_path and private_blob_path in os.path.normcase(original): original = original[len(private_blob_path):].lstrip('/\\:') filesystem['file'] = location + (f'/{original}' if original else '') if isinstance(data, dict): docker_content = { 'image': plan['image'], 'manifest_digest': plan['manifest_digest'], 'blob_digest': descriptor['digest'], 'descriptor_kind': descriptor['kind'], 'positions': list(positions), } if plan['version'] == 2: docker_content['payload_class'] = descriptor['payload_class'] data['DockerContent'] = docker_content return findings def _docker_content_error_code(value, fallback='scan_failed'): text = re.sub(r'[^a-z0-9_]+', '_', str(value or '').strip().lower()).strip('_') return text[:64] if re.fullmatch(r'[a-z][a-z0-9_]{0,63}', text) else fallback def _scan_docker_content_file( destination, descriptor, limits, deadline, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, ): _require_docker_archive_policy(limits) validate_docker_content_artifact( destination, descriptor, deadline=min(deadline, time.monotonic() + limits['archive_timeout_sec']), max_member_bytes=limits['archive_max_size_bytes'], ) remaining = deadline - time.monotonic() if remaining <= 0: raise DockerContentScanError('archive_timeout', 'Docker blob deadline expired before scanning', True) command = [ get_trufflehog_cmd(), 'filesystem', destination, '--json', '--no-update', '--log-level', '2', '--archive-max-size', f'{limits["archive_max_size_bytes"]}B', '--archive-max-depth', str(limits['archive_max_depth']), '--archive-timeout', f'{limits["archive_timeout_sec"]}s', '--concurrency', str(limits['filesystem_concurrency']), ] append_trufflehog_scan_args(command, detectors, exclude_detectors, no_verification, trufflehog_config) result = {'findings': [], 'errors': []} with run_command_streamed( command, remaining, os.environ.copy(), deadline=deadline, staging_roots=(os.path.dirname(os.path.abspath(destination)),), ) as output: apply_trufflehog_diagnostics( result, output, output.returncode, 'filesystem', require_completion=True, ) append_trufflehog_findings(result, output.stdout_lines()) # Only wrapper-owned diagnostics can attribute a watchdog stop to staging. staging_error = output.synthetic_stderr.removeprefix('Error: ').split(';', 1)[0] if staging_error in ('TruffleHog staging limit exceeded', 'Unable to monitor TruffleHog staging'): monitor_failed = staging_error == 'Unable to monitor TruffleHog staging' result.update( errors=[staging_error], error_class='source_resource' if monitor_failed else 'staging_limit', retryable=monitor_failed, source_failure=monitor_failed, source_failure_auth_related=False, ) result.pop('source_failure_category', None) if monitor_failed: result['source_failure_category'] = 'source_resource' return result if time.monotonic() >= deadline: result['errors'].append('Docker blob scan exceeded its absolute deadline') result['error_class'] = 'timeout' result['retryable'] = True return result def scan_docker_layer_plan( image_name, docker_layer_work, timeout_sec=600, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, *, log_target=True, ): try: plan = validate_docker_layer_plan((docker_layer_work or {}).get('plan')) plan_bytes = canonical_docker_layer_plan_bytes(plan) plan_sha256 = hashlib.sha256(plan_bytes).hexdigest() if plan_sha256 != str((docker_layer_work or {}).get('plan_sha256') or ''): raise ValueError('Docker layer work plan hash is invalid') if parse_docker_target(image_name)['image'].lower() != plan['image']: raise ValueError('Docker layer work target does not match its plan') except (TypeError, ValueError) as exc: raise DockerLayerInfrastructureError( 'invalid_bound_plan', 'Docker layer bound plan is invalid', category='source_configuration', ) from exc _require_docker_archive_policy(plan['limits']) leased_by_digest = {} for descriptor in plan['descriptors']: if descriptor['coverage_state'] == 'leased': entry = leased_by_digest.setdefault(descriptor['digest'], { 'descriptor': descriptor, 'positions': [], }) entry['positions'].append(descriptor['position']) work_root = None retain_work = False created_paths = [] findings = [] finding_digests = {} records = [] failures = [] bearer_auth = (docker_layer_work or {}).get('bearer_auth') min_free_bytes = max(0, int((docker_layer_work or {}).get('min_free_bytes') or 0)) execution_deadline = time.monotonic() + max(0.001, float(timeout_sec or 0.001)) supplied_deadline = (docker_layer_work or {}).get('deadline') if supplied_deadline is not None: try: supplied_deadline = float(supplied_deadline) except (TypeError, ValueError) as exc: raise DockerLayerInfrastructureError( 'invalid_deadline', 'Docker layer execution deadline is invalid', category='source_configuration', ) from exc if not math.isfinite(supplied_deadline): raise DockerLayerInfrastructureError( 'invalid_deadline', 'Docker layer execution deadline is invalid', category='source_configuration', ) execution_deadline = min(execution_deadline, supplied_deadline) try: if leased_by_digest: if time.monotonic() >= execution_deadline: raise DockerLayerInfrastructureError( 'execution_deadline', 'Docker layer execution deadline expired before work began', category='remote_transient', ) try: work_root = tempfile.mkdtemp(prefix='docker-layer-', dir=get_work_dir()) harden_private_directory(work_root) write_temp_owner(work_root, ['docker-layer-content'], os.getpid(), required=True) except ScanSlotFatalError: raise except (OSError, RuntimeError, ValueError) as exc: raise DockerLayerInfrastructureError( 'private_storage', 'Docker layer private work storage is unavailable', category='source_resource', ) from exc for digest, entry in leased_by_digest.items(): descriptor = entry['descriptor'] blob_started = time.monotonic() blob_deadline = min( execution_deadline, blob_started + plan['limits']['blob_timeout_sec'], ) destination = os.path.join(work_root, f'blob-{len(created_paths):04d}') verified_bytes = 0 transfer_bytes = 0 transfer_duration_ms = 0 scan_duration_ms = 0 blob_findings = [] error_code = '' blob_retryable = True try: if not records: emit_client_scan_phase('downloading', { 'operation': 'docker_blob_transfer', }) else: emit_client_scan_phase('downloading', { 'operation': 'additional_docker_blob_transfer', }) outcome = stream_docker_registry_blob( plan['repository'], descriptor, destination, bearer_auth, deadline=blob_deadline, min_free_bytes=min_free_bytes, ) bearer_auth = outcome.bearer_auth created_paths.append(destination) verified_bytes = outcome.verified_bytes transfer_bytes = outcome.transfer_bytes transfer_duration_ms = outcome.duration_ms scan_started = time.monotonic() emit_client_scan_phase('scanning', { 'operation': 'docker_blob_scan', }) result = _scan_docker_content_file( destination, descriptor, plan['limits'], blob_deadline, detectors, exclude_detectors, no_verification, trufflehog_config, ) scan_duration_ms = max(0, int((time.monotonic() - scan_started) * 1000)) if time.monotonic() >= blob_deadline and not result.get('errors'): result['errors'] = ['Docker layer scan exceeded its absolute deadline'] result['error_class'] = 'timeout' result['retryable'] = True blob_findings = list(result.get('findings') or []) attach_docker_content_provenance( blob_findings, plan, descriptor, entry['positions'], destination, ) finding_digests.update( (id(finding), digest) for finding in blob_findings if isinstance(finding, dict) ) if result.get('source_failure'): raise DockerLayerInfrastructureError( _docker_content_error_code( result.get('source_failure_category'), 'scanner_infrastructure', ), 'Docker layer scanner infrastructure is unavailable', category=str( result.get('source_failure_category') or 'source_resource' ), auth_related=bool(result.get('source_failure_auth_related')), ) if ( result.get('errors') or result.get('skipped') or result.get('warnings') or result.get('degraded') ): diagnostic_code = result.get('error_class') if not diagnostic_code and result.get('warning_classes'): diagnostic_code = result['warning_classes'][0] error_code = _docker_content_error_code( diagnostic_code, 'scan_incomplete' if result.get('warnings') else 'scan_failed', ) blob_retryable = bool( result.get('retryable', not bool(result.get('skipped'))) ) except DockerLayerInfrastructureError: raise except DockerContentTransferError as exc: error_code = _docker_content_error_code(exc.error_code, 'transfer_failed') blob_retryable = exc.retryable transfer_bytes = max(transfer_bytes, int(exc.transfer_bytes or 0)) transfer_duration_ms = max( transfer_duration_ms, int(exc.duration_ms or 0), ) except DockerContentScanError as exc: error_code = _docker_content_error_code(exc.error_code, 'invalid_content') blob_retryable = exc.retryable except ScanSlotFatalError: retain_work = True raise except Exception as exc: raise DockerLayerInfrastructureError( 'scanner_infrastructure', 'Docker layer scanner infrastructure failed', category='source_resource', ) from exc finally: if not retain_work and destination in created_paths: try: durable_unlink(destination) except OSError as exc: raise DockerLayerInfrastructureError( 'private_cleanup', 'Docker layer artifact cleanup failed', category='source_resource', ) from exc created_paths.remove(destination) terminal = ( not blob_retryable or int(descriptor['attempt']) >= int(descriptor['max_attempts']) ) status = ( 'covered' if not error_code else 'terminal_failed' if terminal else 'retryable_failed' ) records.append({ 'digest': digest, 'lease_token': descriptor['lease_token'], 'status': status, 'verified_bytes': verified_bytes, 'transfer_bytes': transfer_bytes, 'transfer_duration_ms': transfer_duration_ms, 'scan_duration_ms': scan_duration_ms, 'finding_count': len(blob_findings), 'error_code': error_code or None, }) findings.extend(blob_findings) if error_code: failures.append(f'Docker content {digest[:19]} failed: {error_code}') except ScanSlotFatalError: retain_work = True raise finally: # Fatal process outcomes leave payloads and ownership evidence for the janitor. if not retain_work: for path in created_paths: if os.path.lexists(path): try: durable_unlink(path) except OSError as exc: raise DockerLayerInfrastructureError( 'private_cleanup', 'Docker layer artifact cleanup failed', category='source_resource', ) from exc if work_root: cleanup_command_work_dir(work_root) records_by_digest = {item['digest']: item for item in records} if set(records_by_digest) != set(leased_by_digest): raise DockerLayerInfrastructureError( 'missing_execution', 'Docker layer execution metadata is incomplete', category='source_resource', ) effective_descriptors = [] for descriptor in plan['descriptors']: effective_state = descriptor['coverage_state'] if effective_state == 'leased': effective_state = records_by_digest[descriptor['digest']]['status'] effective_descriptors.append((descriptor, effective_state)) static_states = {item['coverage_state'] for item in plan['descriptors']} if 'selected' in static_states: failures.append('Docker content remains selected for the next durable checkpoint') if 'shared_pending' in static_states: failures.append('Docker content is pending under another fenced reservation') result_states = {state for _, state in effective_descriptors} has_retryable = bool( result_states & {'selected', 'shared_pending', 'retryable_failed'} ) has_terminal = 'terminal_failed' in result_states if has_terminal: failures.append('Docker content exhausted its bounded attempt budget') result = { 'findings': findings, 'errors': failures, 'retryable': bool(has_retryable), 'error_class': ('docker_content_retry' if has_retryable else 'docker_content_terminal') if failures else None, 'docker_layer_plan': plan, 'docker_layer_execution': { 'version': plan['version'], 'plan_sha256': plan_sha256, 'blobs': records, }, 'scan_meta': { 'docker_layer_scope': { 'manifest_digest': plan['manifest_digest'], 'coverage_complete': all( state == 'covered' for _, state in effective_descriptors ), 'selected_descriptors': sum( 1 for item in plan['descriptors'] if item['selected'] ), 'leased_blobs': len(leased_by_digest), 'covered_blobs': len({ item['digest'] for item, state in effective_descriptors if state == 'covered' }), 'newly_covered_blobs': sum( 1 for item in records if item['status'] == 'covered' ), 'globally_reused_blobs': len({ item['digest'] for item in plan['descriptors'] if item['coverage_state'] == 'covered' }), 'retryable_failed_blobs': sum( 1 for item in records if item['status'] == 'retryable_failed' ), 'terminal_failed_blobs': sum( 1 for item in records if item['status'] == 'terminal_failed' ), 'pending_checkpoint_blobs': sum( 1 for item in plan['descriptors'] if item['coverage_state'] == 'selected' ), 'shared_pending_blobs': sum( 1 for item in plan['descriptors'] if item['coverage_state'] == 'shared_pending' ), 'selected_bytes': sum( item['size'] for item in plan['descriptors'] if item['selected'] ), 'covered_bytes': sum( item['size'] for item, state in effective_descriptors if state == 'covered' ), 'skipped_bytes': sum( item['size'] for item, state in effective_descriptors if state == 'skipped' ), 'shared_pending_bytes': sum( item['size'] for item, state in effective_descriptors if state == 'shared_pending' ), 'retryable_failed_bytes': sum( item['size'] for item, state in effective_descriptors if state == 'retryable_failed' ), 'terminal_failed_bytes': sum( item['size'] for item, state in effective_descriptors if state == 'terminal_failed' ), 'transfer_bytes': sum(item['transfer_bytes'] for item in records), 'transfer_duration_ms': sum( item['transfer_duration_ms'] for item in records ), 'scan_duration_ms': sum(item['scan_duration_ms'] for item in records), 'timeout_blobs': sum( 1 for item in records if item['error_code'] in ('timeout', 'transfer_timeout') ), 'skipped_descriptors': sum( 1 for item in plan['descriptors'] if item['coverage_state'] == 'skipped' ), }, }, } if not failures: result.pop('error_class') result.pop('retryable') if 'skipped' in static_states: result['degraded'] = True result['warnings'] = ['Docker content plan intentionally skipped bounded descriptors'] result['warning_classes'] = ['docker_content_budget'] try: result = apply_finding_filters( result, plan['image'], log_target=log_target, ) except Exception as exc: raise DockerLayerInfrastructureError( 'result_filter', 'Docker layer result filtering failed', category='source_configuration', ) from exc filtered_counts = Counter( finding_digests.get(id(finding)) for finding in result.get('findings', []) if finding_digests.get(id(finding)) ) for record in result['docker_layer_execution']['blobs']: record['finding_count'] = filtered_counts[record['digest']] return result def _docker_implicit_auth_present(): if os.environ.get('DOCKER_TOKEN') or os.environ.get('REGISTRY_AUTH_FILE') or docker_token_manager.has_accounts(): return True homes = {os.environ.get('HOME'), os.environ.get('USERPROFILE'), os.path.expanduser('~')} if os.environ.get('HOMEDRIVE') and os.environ.get('HOMEPATH'): homes.add(os.environ['HOMEDRIVE'] + os.environ['HOMEPATH']) paths = [os.path.join(home, '.docker', 'config.json') for home in homes if home and home != '~'] xdg = os.environ.get('XDG_RUNTIME_DIR', '') if xdg and not os.path.isabs(xdg): return True paths.append(os.path.join(xdg, 'containers', 'auth.json')) for path in paths: try: os.lstat(path) return True except (FileNotFoundError, NotADirectoryError): continue except OSError: return True return False def _recover_docker_image_contents( image_ref, deadline, config_dir, limits, min_free_bytes, detectors, exclude_detectors, no_verification, trufflehog_config, *, implicit_auth_unsupported=False, anonymous_public_client=False, ): from scanner_db import validate_docker_layer_limits result = {'findings': [], 'errors': []} scope = {'coverage_complete': False, 'scanned_descriptors': 0, 'blob_transfer_attempted': False} result['scan_meta'] = {'docker_full_recovery': scope} phase = 'configuration' diagnostic = {} descriptor = None work_root = None retain_work = False try: limits = validate_docker_layer_limits(limits if limits is not None else { 'config_max_bytes': 1 << 20, 'layer_max_bytes': 256 << 20, 'image_max_bytes': 1 << 30, 'max_layers': 8, 'archive_max_size_bytes': 256 << 20, 'archive_max_depth': 4, 'archive_timeout_sec': 30, 'blob_timeout_sec': 600, 'filesystem_concurrency': 2, 'blob_max_attempts': 3, }) _require_docker_archive_policy(limits) phase = 'preflight' if time.monotonic() >= deadline: raise DockerContentScanError('timeout', 'Docker recovery deadline expired', True) try: _, _, repository, digest = _dockerhub_manifest_target_parts(image_ref) except (TypeError, ValueError) as exc: raise DockerContentScanError('unsupported_recovery_target', 'Docker recovery requires an immutable Docker Hub target') from exc phase = 'authentication' bearer_auth = None if anonymous_public_client: if config_dir: raise DockerLayerInfrastructureError( 'recovery_auth_unavailable', 'Anonymous Docker recovery received credential configuration', category='source_configuration', ) elif config_dir: # Only the already-managed credential pool is trusted. Never load an arbitrary # Docker config or run its external credential helpers in the recovery path. with docker_token_manager.lock: matched = next((account for account in docker_token_manager.accounts if os.path.normcase(os.path.abspath(account.config_dir)) == os.path.normcase(os.path.abspath(config_dir))), None) if matched is None: raise DockerLayerInfrastructureError( 'recovery_auth_unavailable', 'Docker recovery cannot use this credential configuration', category='source_configuration', ) excluded = {account.name for account in docker_token_manager.accounts if account.name != matched.name} bearer_auth = docker_registry_bearer_token( 'Bearer realm="https://auth.docker.io/token",service="registry.docker.io"', repository, excluded_accounts=excluded, deadline=deadline, ) else: # The native default keychain may use Docker/Podman configs without # DOCKER_CONFIG. Inspect existence only; never read or invoke helpers. if implicit_auth_unsupported or _docker_implicit_auth_present(): raise DockerLayerInfrastructureError( 'recovery_auth_unavailable', 'Docker recovery cannot map implicit credential identity', category='source_configuration', ) phase = 'manifest_resolution' emit_client_scan_phase('resolving', { 'operation': 'docker_manifest_resolution_recovery', }) resolved, bearer_auth = resolve_docker_content_manifest( image_ref, bearer_auth=bearer_auth, deadline=deadline, anonymous_only=anonymous_public_client, ) phase = 'preflight' if resolved['image'] != image_ref or resolved['manifest_digest'] != digest or resolved['repository'] != repository: raise DockerContentScanError('recovery_identity_mismatch', 'Docker recovery manifest identity changed') descriptors = [dict(resolved['config'], kind='config', position=0)] + [ dict(item, kind='layer', position=index) for index, item in enumerate(resolved['layers'], 1) ] scope['descriptor_count'] = len(descriptors) if len(resolved['layers']) > limits['max_layers']: diagnostic.update(reason='count_bound', observed=len(resolved['layers']), limit=limits['max_layers']) raise DockerContentScanError('recovery_budget', 'All Docker layers do not fit the recovery count bound') unique = {} for descriptor in descriptors: kind, media, size = descriptor['kind'], descriptor.get('media_type'), descriptor['size'] allowed = DOCKER_CONFIG_MEDIA_TYPES if kind == 'config' else DOCKER_LAYER_MEDIA_TYPES if not isinstance(media, str) or media not in allowed: # Reporting only: these official types do not expand recovery support. known_media = DOCKER_CONFIG_MEDIA_TYPES | DOCKER_LAYER_MEDIA_TYPES | { 'application/vnd.oci.image.layer.nondistributable.v1.tar', 'application/vnd.oci.image.layer.nondistributable.v1.tar+gzip', 'application/vnd.oci.image.layer.nondistributable.v1.tar+zstd', 'application/vnd.docker.image.rootfs.foreign.diff.tar', 'application/vnd.docker.image.rootfs.foreign.diff.tar.gzip', 'application/vnd.oci.image.manifest.v1+json', 'application/vnd.oci.image.index.v1+json', 'application/vnd.docker.distribution.manifest.v1+json', 'application/vnd.docker.distribution.manifest.v1+prettyjws', 'application/vnd.docker.distribution.manifest.v2+json', 'application/vnd.docker.distribution.manifest.list.v2+json', } diagnostic.update( reason='media_type', media_type=media if isinstance(media, str) and media in known_media else 'other', ) raise DockerContentScanError('unsupported_media_type', 'Docker recovery descriptor is unsupported') if not normalize_docker_digest(descriptor['digest']): diagnostic['reason'] = 'integrity' raise DockerContentScanError('unsupported_media_type', 'Docker recovery descriptor is unsupported') if isinstance(size, bool) or not isinstance(size, int) or size < 0: raise DockerContentScanError('invalid_descriptor', 'Docker recovery descriptor size is invalid') byte_limit = limits['config_max_bytes' if kind == 'config' else 'layer_max_bytes'] if size > byte_limit: diagnostic.update(reason='descriptor_byte_bound', observed=size, limit=byte_limit) raise DockerContentScanError('recovery_budget', 'All Docker content does not fit the recovery byte bound') entry = unique.setdefault(descriptor['digest'], {'descriptor': descriptor, 'positions': []}) previous = entry['descriptor'] if any(previous[key] != descriptor[key] for key in ('kind', 'size', 'media_type')): raise DockerContentScanError('conflicting_descriptors', 'Docker recovery digest descriptors conflict') entry['positions'].append(descriptor['position']) descriptor = None image_bytes = sum(entry['descriptor']['size'] for entry in unique.values()) if image_bytes > limits['image_max_bytes']: diagnostic.update(reason='image_byte_bound', observed=image_bytes, limit=limits['image_max_bytes']) raise DockerContentScanError('recovery_budget', 'All Docker content does not fit the cumulative recovery bound') if time.monotonic() >= deadline: raise DockerContentScanError('timeout', 'Docker recovery deadline expired after preflight', True) phase = 'staging' work_root = tempfile.mkdtemp(prefix='docker-layer-', dir=get_work_dir()) harden_private_directory(work_root) write_temp_owner(work_root, ['docker-full-recovery'], os.getpid(), required=True) for index, entry in enumerate(unique.values()): descriptor = entry['descriptor'] phase = 'blob_transfer' blob_deadline = min(deadline, time.monotonic() + limits['blob_timeout_sec']) if time.monotonic() >= blob_deadline: raise DockerContentScanError('timeout', 'Docker recovery deadline expired', True) destination = os.path.join(work_root, f'blob-{index:04d}') try: blob_min_free_bytes = max(0, int(min_free_bytes)) scope['blob_transfer_attempted'] = True emit_client_scan_phase('downloading', { 'operation': 'docker_blob_transfer_recovery', 'descriptor_index': index, }) outcome = stream_docker_registry_blob( repository, descriptor, destination, bearer_auth, deadline=blob_deadline, min_free_bytes=blob_min_free_bytes, anonymous_only=anonymous_public_client, ) bearer_auth = outcome.bearer_auth phase = 'blob_scan' emit_client_scan_phase('scanning', { 'operation': 'docker_blob_scan_recovery', 'descriptor_index': index, }) scanned = _scan_docker_content_file( destination, descriptor, limits, blob_deadline, detectors, exclude_detectors, no_verification, trufflehog_config, ) result['findings'].extend(attach_docker_content_provenance( scanned['findings'], resolved, descriptor, entry['positions'], destination, )) if scanned.get('source_failure'): raise DockerLayerInfrastructureError( 'recovery_scanner_unavailable', 'Docker recovery scanner is unavailable', category=scanned.get('source_failure_category') or 'source_resource', ) if any(scanned.get(key) for key in ('errors', 'warnings', 'degraded', 'skipped')): raise DockerContentScanError( 'recovery_scan_incomplete', 'Docker recovery scanner did not cover a blob', bool(scanned.get('retryable', False)), ) scope['scanned_descriptors'] += len(entry['positions']) except ScanSlotFatalError: retain_work = True raise finally: if not retain_work and os.path.lexists(destination): previous_phase = phase phase = 'cleanup' durable_unlink(destination) phase = previous_phase descriptor = None phase = 'completion' if time.monotonic() >= deadline: raise DockerContentScanError('timeout', 'Docker recovery deadline expired', True) except ScanSlotFatalError: retain_work = True raise except Exception as exc: code = _docker_content_error_code(getattr(exc, 'error_code', None), 'recovery_infrastructure') if isinstance(exc, DockerRemoteAccessError): code = _docker_content_error_code(exc.status, 'recovery_remote') elif isinstance(exc, DockerRegistryResolutionError): code = 'recovery_manifest_invalid' diagnostic.setdefault('reason', { 'recovery_identity_mismatch': 'integrity', 'invalid_descriptor': 'integrity', 'conflicting_descriptors': 'integrity', 'digest_mismatch': 'integrity', 'size_mismatch': 'integrity', 'invalid_layer_archive': 'integrity', 'invalid_config_json': 'integrity', 'recovery_manifest_invalid': 'manifest_invalid', 'timeout': 'timeout', 'transfer_timeout': 'timeout', 'archive_timeout': 'timeout', 'recovery_scan_incomplete': 'scan_incomplete', }.get(code, 'configuration' if phase == 'configuration' else 'other')) diagnostic['phase'] = phase if descriptor is not None: diagnostic.update(descriptor_kind=descriptor['kind'], descriptor_index=descriptor['position']) scope['diagnostic'] = diagnostic result['errors'].append(f'Docker full-image recovery incomplete: {code}') result['error_class'] = code result['retryable'] = bool(getattr(exc, 'retryable', True)) if isinstance(exc, DockerRegistryResolutionError): result['retryable'] = ( isinstance(exc, DockerRemoteAccessError) and exc.status not in ('target_forbidden', 'auth_failed') ) if result['retryable']: result['source_failure'] = True result['source_failure_category'] = 'remote_auth' if exc.status == 'auth_failed' else 'remote_rate_limit' if exc.status == 'rate_limited' else 'remote_transient' elif anonymous_public_client and exc.status == 'auth_failed': result['error_class'] = 'docker_registry_access' result.pop('source_failure', None) result.pop('source_failure_category', None) elif isinstance(exc, DockerLayerInfrastructureError) or not isinstance(exc, (DockerContentScanError, DockerContentTransferError)): result['source_failure'] = True result['source_failure_category'] = getattr(exc, 'category', 'source_configuration' if isinstance(exc, ValueError) else 'source_resource') result['source_failure_auth_related'] = bool(getattr(exc, 'auth_related', False)) finally: if work_root and not retain_work: try: cleanup_command_work_dir(work_root) except ScanSlotFatalError: raise except (OSError, RuntimeError): scope.setdefault('diagnostic', {'phase': 'cleanup', 'reason': 'cleanup'}) result['errors'].append('Docker full-image recovery private cleanup failed') result.update(error_class='private_cleanup', retryable=True, source_failure=True, source_failure_category='source_resource') if not result['errors'] and time.monotonic() >= deadline: scope['diagnostic'] = {'phase': 'completion', 'reason': 'timeout'} result.update(errors=['Docker full-image recovery exceeded its absolute deadline'], error_class='timeout', retryable=True) scope['coverage_complete'] = not result['errors'] and scope['scanned_descriptors'] == scope.get('descriptor_count', 0) > 0 return result def scan_docker_image(image_name, timeout_sec=1800, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, config_dir=None, trufflehog_concurrency=0, *, log_target=True, docker_recovery_limits=None, docker_recovery_min_free_bytes=20 << 30): """Scan a Docker image for secrets""" started = time.monotonic() try: timeout_sec = float(timeout_sec) if not math.isfinite(timeout_sec) or timeout_sec <= 0: raise ValueError('invalid timeout') except (TypeError, ValueError): return {'findings': [], 'errors': ['Docker scan requires a positive finite time budget'], 'error_class': 'timeout', 'retryable': False} deadline = started + timeout_sec anonymous_public_client = ( _client_remote_execution_kind.get() == 'docker_direct_v1' ) if anonymous_public_client and config_dir: raise RuntimeError('remote Docker direct execution cannot use credentials') try: image_ref = ( parse_dockerhub_digest_target(image_name)['image'] if anonymous_public_client else parse_docker_target(image_name)['image'] ) except (TypeError, ValueError) as exc: return { "findings": [], "errors": [f"Docker image target rejected: {exc}"], "error_class": "invalid_target", "retryable": False, } if log_target: logger.info(f"Scanning Docker image: {image_ref}") cmd = [get_trufflehog_cmd(), 'docker', '--image', image_ref, '--json', '--no-update', '--local-dev', '--log-level', '2'] concurrency = max(0, min(64, int(trufflehog_concurrency or 0))) if concurrency: cmd.extend(['--concurrency', str(concurrency)]) append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config) env = os.environ.copy() if config_dir: env['DOCKER_CONFIG'] = config_dir implicit_auth_unsupported = ( not anonymous_public_client and not (config_dir or env.get('DOCKER_CONFIG')) and _docker_implicit_auth_present() ) results = {"findings": [], "errors": []} emit_client_scan_phase('scanning', { 'integrated_operation': 'docker_pull_and_scan', }) with run_command_streamed(cmd, max(0, deadline - time.monotonic()), env, deadline=deadline) as output: apply_trufflehog_diagnostics(results, output, output.returncode, 'docker') append_trufflehog_findings(results, output.stdout_lines()) for line in results.get('errors', []): try: diagnostic = json.loads(line) except (TypeError, ValueError): continue if ( isinstance(diagnostic, dict) and diagnostic.get('msg') == 'error processing layer' and diagnostic.get('error') == 'unexpected EOF' ): results.setdefault('scan_meta', {})['docker_native_diagnostic'] = { 'phase': 'native_layer_processing', 'subcause': 'unexpected_eof_ambiguous', 'coverage_complete': False, } break codec_warnings = [] for line in results.get('warnings', []): try: diagnostic = json.loads(line) except (TypeError, ValueError): continue if isinstance(diagnostic, dict) and diagnostic.get('msg') == 'error processing layer' and diagnostic.get('error') == 'gzip: invalid header': codec_warnings.append(line) if codec_warnings: if not results.get('errors') and not results.get('source_failure') and not results.get('skipped') and len(codec_warnings) == len(results.get('warnings', [])): recovered = _recover_docker_image_contents( image_ref, deadline, ( None if anonymous_public_client else config_dir or env.get('DOCKER_CONFIG') ), docker_recovery_limits, docker_recovery_min_free_bytes, detectors, exclude_detectors, no_verification, trufflehog_config, implicit_auth_unsupported=implicit_auth_unsupported, anonymous_public_client=anonymous_public_client, ) results['findings'].extend(recovered.pop('findings')) results.setdefault('scan_meta', {}).update(recovered.pop('scan_meta')) if not recovered['errors'] and results['scan_meta']['docker_full_recovery']['coverage_complete']: for key in ('warnings', 'warning_classes', 'degraded', 'retryable'): results.pop(key, None) results['scan_meta']['docker_full_recovery']['recovered_codec'] = True else: if not recovered['errors']: recovered.update(errors=['Docker full-image recovery coverage is incomplete'], error_class='docker_recovery_incomplete', retryable=False) results['scan_meta']['docker_full_recovery'].setdefault( 'diagnostic', {'phase': 'completion', 'reason': 'scan_incomplete'}, ) results.update(recovered) else: results['errors'].append('Docker codec recovery cannot clear unrelated scan diagnostics') results.setdefault('error_class', 'docker_recovery_incomplete') results.setdefault('scan_meta', {})['docker_full_recovery'] = { 'coverage_complete': False, 'blob_transfer_attempted': False, 'diagnostic': {'phase': 'native_diagnostics', 'reason': 'unrelated_diagnostics'}, } results = apply_finding_filters(results, image_ref, log_target=log_target) if time.monotonic() >= deadline: had_errors = bool(results.get('errors')) results.setdefault('errors', []).append('Docker scan exceeded its absolute deadline including cleanup and filtering') if not had_errors: results.update(error_class='timeout', retryable=True) metadata = results.setdefault('scan_meta', {}) metadata['docker_deadline_exceeded'] = True if 'docker_full_recovery' in metadata: metadata['docker_full_recovery']['coverage_complete'] = False metadata['docker_full_recovery'].setdefault('diagnostic', {'phase': 'completion', 'reason': 'timeout'}) return results def _append_windows_git_longpaths(command_env, operation): if os.name != 'nt': return config_count = command_env.get('GIT_CONFIG_COUNT') or '0' if not re.fullmatch(r'[0-9]{1,3}', config_count) or int(config_count) > 255: raise ValueError(f'{operation} GIT_CONFIG_COUNT must be an integer from 0 to 255') config_count = int(config_count) command_env[f'GIT_CONFIG_KEY_{config_count}'] = 'core.longpaths' command_env[f'GIT_CONFIG_VALUE_{config_count}'] = 'true' command_env['GIT_CONFIG_COUNT'] = str(config_count + 1) def scan_huggingface_space(space_id, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None): anonymous_public_client = ( _client_remote_execution_kind.get() == 'huggingface_space_v1' ) if anonymous_public_client and token: raise RuntimeError('remote HuggingFace direct execution cannot use a token') logger.info(f"Scanning HuggingFace Space: {space_id}") cmd = [get_trufflehog_cmd(), 'huggingface', '--space', space_id, '--json', '--no-update'] secrets_to_redact = [] command_env = os.environ.copy() _append_windows_git_longpaths(command_env, 'HuggingFace') if token: command_env['HUGGINGFACE_TOKEN'] = token command_env['HF_TOKEN'] = token secrets_to_redact.append(token) append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config) results = {"findings": [], "errors": []} emit_client_scan_phase('scanning', { 'integrated_operation': 'huggingface_clone_and_scan', }) with run_command_streamed(cmd, timeout_sec, command_env) as output: apply_trufflehog_diagnostics( results, output, output.returncode, 'huggingface', redactions=secrets_to_redact, ) append_trufflehog_findings( results, output.stdout_lines(redactions=secrets_to_redact), ) return apply_finding_filters(results, space_id) WINDOWS_RESERVED_NAMES = { 'con', 'prn', 'aux', 'nul', *(f'com{index}' for index in range(1, 10)), *(f'lpt{index}' for index in range(1, 10)), } def validate_archive_member_name(name): normalized = str(name or '').replace('\\', '/') if normalized.startswith('/') or '..' in normalized.split('/'): raise ValueError(f'Unsafe archive member path: {name}') for component in [part for part in normalized.split('/') if part]: stem = component.split('.', 1)[0].lower() if ':' in component or component.endswith((' ', '.')) or stem in WINDOWS_RESERVED_NAMES: raise ValueError(f'Unsafe Windows archive member name: {name}') return normalized class LimitedReader: def __init__(self, source, limit): self.source = source self.remaining = max(0, int(limit)) def read(self, size=-1): if self.remaining <= 0: raise ValueError('Archive decompressed stream exceeds safety budget') if size is None or size < 0: size = self.remaining + 1 data = self.source.read(min(size, self.remaining + 1)) self.remaining -= len(data) if self.remaining < 0: raise ValueError('Archive decompressed stream exceeds safety budget') return data def safe_extract_tar(tar_path, destination, max_files=100000, max_total_size_mb=1024, max_file_size_mb=512): _raise_if_scan_slot_fatal() destination_abs = os.path.abspath(destination) max_total_bytes = int(max_total_size_mb or 0) * 1024 * 1024 max_file_bytes = int(max_file_size_mb or 0) * 1024 * 1024 stream_budget = max_total_bytes + (64 * 1024 * 1024) raw_source = open(tar_path, 'rb') magic = raw_source.read(6) raw_source.seek(0) if magic.startswith(b'\x1f\x8b'): decompressed = gzip.GzipFile(fileobj=raw_source) elif magic.startswith(b'BZh'): decompressed = bz2.BZ2File(raw_source) elif magic.startswith(b'\xfd7zXZ\x00'): decompressed = lzma.LZMAFile(raw_source) else: decompressed = raw_source try: archive = tarfile.open(fileobj=LimitedReader(decompressed, stream_budget), mode='r|') try: total_size = 0 member_count = 0 for member in archive: _raise_if_scan_slot_fatal() if max_files and member_count >= int(max_files): raise ValueError(f'Tar archive exceeds {max_files} members') member_count += 1 validate_archive_member_name(member.name) member_path = os.path.abspath(os.path.join(destination, member.name)) if not member_path.startswith(destination_abs + os.sep) and member_path != destination_abs: raise ValueError(f"Unsafe tar member path: {member.name}") if member.issym() or member.islnk() or member.isdev() or member.isfifo(): raise ValueError(f"Unsafe tar member type: {member.name}") if member.isfile(): if max_file_bytes and member.size > max_file_bytes: raise ValueError(f"Tar member exceeds {max_file_size_mb} MB: {member.name}") total_size += max(0, int(member.size or 0)) if max_total_bytes and total_size > max_total_bytes: raise ValueError(f"Tar archive exceeds {max_total_size_mb} MB extracted") archive.extract(member, destination, filter='data') _raise_if_scan_slot_fatal() finally: archive.close() finally: if decompressed is not raw_source: decompressed.close() raw_source.close() def validate_zip_central_directory(zip_path, max_files, max_metadata_size_mb=16): _raise_if_scan_slot_fatal() with open(zip_path, 'rb') as source: source.seek(0, os.SEEK_END) size = source.tell() source.seek(max(0, size - 65557)) tail = source.read() offset = tail.rfind(b'PK\x05\x06') if offset < 0 or offset + 22 > len(tail): raise ValueError('Zip end-of-central-directory record is missing') disk_number = int.from_bytes(tail[offset + 4:offset + 6], 'little') central_disk = int.from_bytes(tail[offset + 6:offset + 8], 'little') entry_count = int.from_bytes(tail[offset + 10:offset + 12], 'little') central_size = int.from_bytes(tail[offset + 12:offset + 16], 'little') central_offset = int.from_bytes(tail[offset + 16:offset + 20], 'little') if disk_number or central_disk or entry_count == 0xFFFF or central_size == 0xFFFFFFFF or central_offset == 0xFFFFFFFF: raise ValueError('Multi-disk and ZIP64 archives are not accepted') if max_files and entry_count > int(max_files): raise ValueError(f'Zip archive exceeds {max_files} members') max_metadata_bytes = int(max_metadata_size_mb or 0) * 1024 * 1024 if max_metadata_bytes and central_size > max_metadata_bytes: raise ValueError(f'Zip central directory exceeds {max_metadata_size_mb} MB') if central_offset < 0 or central_size < 0 or central_offset + central_size > size: raise ValueError('Zip central directory points outside the archive') with open(zip_path, 'rb') as source: source.seek(central_offset) consumed = 0 for _ in range(entry_count): _raise_if_scan_slot_fatal() header = source.read(46) if len(header) != 46 or header[:4] != b'PK\x01\x02': raise ValueError('Invalid zip central-directory entry') name_len = int.from_bytes(header[28:30], 'little') extra_len = int.from_bytes(header[30:32], 'little') comment_len = int.from_bytes(header[32:34], 'little') variable_size = name_len + extra_len + comment_len source.seek(variable_size, os.SEEK_CUR) consumed += 46 + variable_size if consumed > central_size: raise ValueError('Zip central-directory size mismatch') if consumed != central_size: raise ValueError('Zip central-directory entry count mismatch') return entry_count def safe_extract_package_zip(zip_path, destination, max_files=100000, max_total_size_mb=1024, max_file_size_mb=512): _raise_if_scan_slot_fatal() destination_abs = os.path.abspath(destination) max_total_bytes = int(max_total_size_mb or 0) * 1024 * 1024 max_file_bytes = int(max_file_size_mb or 0) * 1024 * 1024 validate_zip_central_directory(zip_path, max_files) with zipfile.ZipFile(zip_path) as archive: members = archive.infolist() if max_files and len(members) > int(max_files): raise ValueError(f'Zip archive exceeds {max_files} members') total_size = 0 for member in members: _raise_if_scan_slot_fatal() validate_archive_member_name(member.filename) member_path = os.path.abspath(os.path.join(destination, member.filename)) if not member_path.startswith(destination_abs + os.sep) and member_path != destination_abs: raise ValueError(f"Unsafe zip member path: {member.filename}") if member.is_dir(): continue if max_file_bytes and member.file_size > max_file_bytes: raise ValueError(f"Zip member exceeds {max_file_size_mb} MB: {member.filename}") total_size += max(0, int(member.file_size or 0)) if max_total_bytes and total_size > max_total_bytes: raise ValueError(f"Zip archive exceeds {max_total_size_mb} MB extracted") for member in members: _raise_if_scan_slot_fatal() archive.extract(member, destination) _raise_if_scan_slot_fatal() def safe_extract_archive(archive_path, destination, max_total_size_mb=1024): _raise_if_scan_slot_fatal() if tarfile.is_tarfile(archive_path): safe_extract_tar(archive_path, destination, max_total_size_mb=max_total_size_mb) return if zipfile.is_zipfile(archive_path): safe_extract_package_zip(archive_path, destination, max_total_size_mb=max_total_size_mb) return raise ValueError("Unsupported package archive format") def download_file(url, path, max_size_mb=50, timeout=60): _raise_if_scan_slot_fatal() max_bytes = max_size_mb * 1024 * 1024 with requests.Session() as session: session.trust_env = False with session.get(url, stream=True, headers={'User-Agent': 'GitSecretsScanner/2.0'}, timeout=timeout) as response: _raise_if_scan_slot_fatal() response.raise_for_status() total = 0 with open(path, 'wb') as f: for chunk in response.iter_content(chunk_size=1024 * 256): _raise_if_scan_slot_fatal() if not chunk: continue total += len(chunk) if max_bytes and total > max_bytes: raise ValueError(f"Artifact exceeds {max_size_mb} MB") f.write(chunk) _raise_if_scan_slot_fatal() harden_private_file(path) return total def scan_npm_package(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, max_artifact_size_mb=50): """Download, extract, and scan an npm package tarball with TruffleHog filesystem.""" package = parse_npm_target(target) package_id = npm_package_id(package) logger.info(f"Scanning npm package: {package_id}") _raise_if_scan_slot_fatal() work_dir = create_command_work_dir() _raise_if_scan_slot_fatal() if not work_dir: return {"findings": [], "errors": ["Unable to create npm work dir"]} try: tarball_path = os.path.join(work_dir, 'package.tgz') extract_dir = os.path.join(work_dir, 'extract') ensure_private_directory(extract_dir, reject_reparse=True) downloaded = download_file(package['tarball'], tarball_path, max_artifact_size_mb, timeout=min(timeout_sec, 120)) _raise_if_scan_slot_fatal() safe_extract_tar(tarball_path, extract_dir, max_total_size_mb=max(256, int(max_artifact_size_mb or 0) * 4)) _raise_if_scan_slot_fatal() harden_private_tree(extract_dir) _raise_if_scan_slot_fatal() harvest_warnings = [] postman_targets = find_postman_artifacts( extract_dir, 'npm', package, max_artifact_size_mb=max_artifact_size_mb, warnings=harvest_warnings, ) cmd = [get_trufflehog_cmd(), 'filesystem', extract_dir, '--json', '--no-update'] append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config) results = {"findings": [], "errors": [], "package": package, "bytes": downloaded, "postman_targets": postman_targets} with run_command_streamed(cmd, timeout_sec) as output: apply_trufflehog_diagnostics(results, output, output.returncode, 'npm') append_trufflehog_findings(results, output.stdout_lines()) attach_nearby_context(results) _attach_postman_harvest_warnings(results, harvest_warnings) return apply_finding_filters(results, package_id) except ScanSlotFatalError: raise except Exception as e: return {"findings": [], "errors": [f"npm scan failed: {str(e)}"], "package": package} finally: cleanup_command_work_dir(work_dir) def scan_pypi_package(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, max_artifact_size_mb=50): """Download, extract, and scan a PyPI sdist/wheel with TruffleHog filesystem.""" package = parse_pypi_target(target) package_id = pypi_package_id(package) logger.info(f"Scanning PyPI package: {package_id}") declared_size = int(package.get('size') or 0) max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024 if max_bytes and declared_size > max_bytes: return { "findings": [], "errors": [], "skipped": f"artifact exceeds {max_artifact_size_mb} MB", "package": package, } _raise_if_scan_slot_fatal() work_dir = create_command_work_dir() _raise_if_scan_slot_fatal() if not work_dir: return {"findings": [], "errors": ["Unable to create PyPI work dir"]} try: artifact_path = os.path.join(work_dir, 'package-artifact') extract_dir = os.path.join(work_dir, 'extract') ensure_private_directory(extract_dir, reject_reparse=True) downloaded = download_file(package['artifact'], artifact_path, max_artifact_size_mb, timeout=min(timeout_sec, 120)) _raise_if_scan_slot_fatal() safe_extract_archive(artifact_path, extract_dir, max_total_size_mb=max(256, int(max_artifact_size_mb or 0) * 4)) _raise_if_scan_slot_fatal() harden_private_tree(extract_dir) _raise_if_scan_slot_fatal() harvest_warnings = [] postman_targets = find_postman_artifacts( extract_dir, 'pypi', package, max_artifact_size_mb=max_artifact_size_mb, warnings=harvest_warnings, ) cmd = [get_trufflehog_cmd(), 'filesystem', extract_dir, '--json', '--no-update'] append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config) results = {"findings": [], "errors": [], "package": package, "bytes": downloaded, "postman_targets": postman_targets} with run_command_streamed(cmd, timeout_sec) as output: apply_trufflehog_diagnostics(results, output, output.returncode, 'pypi') append_trufflehog_findings(results, output.stdout_lines()) attach_nearby_context(results) _attach_postman_harvest_warnings(results, harvest_warnings) return apply_finding_filters(results, package_id) except ScanSlotFatalError: raise except Exception as e: return {"findings": [], "errors": [f"PyPI scan failed: {str(e)}"], "package": package} finally: cleanup_command_work_dir(work_dir) def scan_package_git_repo(target, timeout_sec=900, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, max_depth=None, max_commit_age_days=None, commit_lookup_pages=3, skip_if_commit_lookup_fails=True): data = parse_package_git_target(target) repo_url = data.get('repo_url') provider = data.get('provider') if not repo_url: return {"findings": [], "errors": ["package_git target missing repo_url"], "package": data} logger.info( f"Scanning package git repo: {repo_url} " f"({data.get('package_source')}:{data.get('name')}@{data.get('version')})" ) candidate = normalize_git_repo_candidate(repo_url) or {} canonical_provider = candidate.get('provider') or provider provider_token = token if canonical_provider == 'github' else None preflight = preflight_package_git_repo(data, provider_token, min(int(timeout_sec or 30), 20)) if preflight and preflight.get('skip'): reason = preflight.get('reason') or 'package_git repo unavailable' logger.info(f"Skipping package git repo {repo_url}: {reason}") return {"findings": [], "errors": [], "skipped": reason, "package": data, "scan_meta": {"preflight": preflight}} scan_provider = (preflight or {}).get('provider') or canonical_provider scan_token = None if preflight and preflight.get('retry_unauthenticated') else provider_token result = scan_git_repo( repo_url, timeout_sec=timeout_sec, detectors=detectors, exclude_detectors=exclude_detectors, no_verification=no_verification, trufflehog_config=trufflehog_config, token=scan_token, provider=scan_provider, max_depth=max_depth, max_commit_age_days=max_commit_age_days, commit_lookup_pages=commit_lookup_pages, skip_if_commit_lookup_fails=skip_if_commit_lookup_fails, ) convert_package_git_unavailable_to_skip(result) result['package'] = data return result def preflight_package_git_repo(data, token=None, timeout=15): repo_url = data.get('repo_url') or '' provider = (data.get('provider') or '').lower() candidate = normalize_git_repo_candidate(repo_url) if not candidate: return {'skip': True, 'reason': 'package_git repo URL is unsupported or invalid'} provider = candidate.get('provider') or provider repo_path = candidate.get('repo_path') or '' try: if provider == 'github': response = api_request( 'GET', f'https://api.github.com/repos/{repo_path}', headers=github_headers(token), timeout=timeout, retry_statuses={500, 502, 503, 504}, ) if response.status_code == 200: return {'skip': False, 'provider': provider, 'repo_path': repo_path} message = response_message(response).lower() if response.status_code == 404: return {'skip': True, 'provider': provider, 'repo_path': repo_path, 'reason': 'package_git repo not found or private'} if response.status_code in (401, 403) and 'rate limit' not in message: return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code, 'retry_unauthenticated': True} return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code} if provider == 'gitlab': response = api_request( 'GET', f'https://gitlab.com/api/v4/projects/{quote(repo_path, safe="")}', headers=gitlab_headers(token), timeout=timeout, retry_statuses={500, 502, 503, 504}, ) if response.status_code == 200: return {'skip': False, 'provider': provider, 'repo_path': repo_path} if response.status_code == 404: return {'skip': True, 'provider': provider, 'repo_path': repo_path, 'reason': 'package_git project not found or private'} if response.status_code in (401, 403): return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code, 'retry_unauthenticated': True} return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code} except ScanSlotFatalError: raise except Exception as e: logger.warning(f"Package git preflight failed for {repo_url}: {str(e)[:300]}") return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': 'unknown'} def postman_stage_filename(target_data): kind = target_data.get('kind') or 'artifact' digest = target_data.get('sha256') or target_data.get('sha') or 'postman' suffix = POSTMAN_COLLECTION_SUFFIX if kind == 'collection' else POSTMAN_ENVIRONMENT_SUFFIX if kind == 'environment' else 'postman.json' return f'{str(digest)[:16]}.{suffix}' def is_postman_placeholder(value): text = str(value or '').strip() return bool(text and POSTMAN_PLACEHOLDER_RE.match(text)) def load_postman_context(cache_path, payload=None): max_input_bytes = min( POSTMAN_JSON_HARD_MAX_INPUT_BYTES, max(1, int(getattr(scan_config, 'postman_context_max_input_bytes', POSTMAN_JSON_HARD_MAX_INPUT_BYTES))), ) max_nodes = max(1, int(getattr(scan_config, 'postman_context_max_nodes', 100000))) max_depth = max(1, int(getattr(scan_config, 'postman_context_max_depth', 64))) max_scalar_bytes = max(1, int(getattr(scan_config, 'postman_context_max_scalar_bytes', 16 * 1024 * 1024))) max_items = max(1, int(getattr(scan_config, 'postman_context_max_items', 50000))) try: if payload is None: size = os.path.getsize(cache_path) if size <= 0 or size > max_input_bytes: raise PostmanCacheValidationError('Postman context input exceeds its bounded byte limit') with open(cache_path, 'rb') as f: payload = f.read(max_input_bytes + 1) elif not isinstance(payload, bytes): raise PostmanCacheValidationError('Postman context input must be bytes') size = len(payload) if size <= 0 or size > max_input_bytes: raise PostmanCacheValidationError('Postman context input exceeds its bounded byte limit') if len(payload) > max_input_bytes: raise PostmanCacheValidationError('Postman context input exceeds its bounded byte limit') data = json.loads(payload.decode('utf-8-sig')) except PostmanCacheValidationError: raise except (OSError, UnicodeDecodeError, ValueError, RecursionError) as exc: raise PostmanCacheValidationError('Postman context is not bounded valid JSON') from exc contexts = [] traversed = 0 scalar_bytes = 0 def charge(*values): nonlocal scalar_bytes scalar_bytes += sum(len(str(value or '').encode('utf-8', errors='replace')) for value in values) if scalar_bytes > max_scalar_bytes: raise PostmanCacheValidationError('Postman context scalar byte limit exceeded') def host_from_url(value): try: return urlsplit(str(value)).hostname or '' except Exception: return '' def add_context(path, key, value, endpoint='', auth_type='', location='value'): if value is None: return endpoint = str(endpoint or '') text = str(value) charge(path, key, text, endpoint, auth_type, location) if len(contexts) >= max_items: raise PostmanCacheValidationError('Postman context item limit exceeded') contexts.append({ 'path': path, 'key': str(key or ''), 'value': text, 'endpoint': endpoint, 'host': host_from_url(endpoint), 'auth_type': str(auth_type or ''), 'location': location, }) stack = [(data, '$', '', '', 'value', 0)] while stack: value, path, endpoint, auth_type, inherited_location, depth = stack.pop() traversed += 1 if traversed > max_nodes: raise PostmanCacheValidationError('Postman context traversal item limit exceeded') if depth > max_depth: raise PostmanCacheValidationError('Postman context depth limit exceeded') if len(path.encode('utf-8', errors='replace')) > 4096: raise PostmanCacheValidationError('Postman context path limit exceeded') if isinstance(value, dict): local_endpoint = endpoint url_value = value.get('url') if isinstance(url_value, str): local_endpoint = url_value elif isinstance(url_value, dict) and url_value.get('raw'): local_endpoint = str(url_value.get('raw')) local_auth = auth_type auth = value.get('auth') if isinstance(auth, dict): local_auth = str(auth.get('type') or local_auth or '') if traversed + len(stack) + len(value) > max_nodes: raise PostmanCacheValidationError('Postman context traversal item limit exceeded') children = [] for key, item in value.items(): lower_key = str(key).lower() location = 'value' if lower_key in ('header', 'headers'): location = 'header' elif lower_key in ('query', 'queryparam', 'query_params'): location = 'query_param' elif lower_key in ('body', 'raw'): location = 'body' elif lower_key in ('variable', 'values'): location = 'environment_variable' child_path = f'{path}.{key}' charge(key) children.append((item, child_path, local_endpoint, local_auth, location, depth + 1)) stack.extend(reversed(children)) elif isinstance(value, list): if traversed + len(stack) + len(value) > max_nodes: raise PostmanCacheValidationError('Postman context traversal item limit exceeded') stack.extend( (value[index], f'{path}[{index}]', endpoint, auth_type, inherited_location, depth + 1) for index in range(len(value) - 1, -1, -1) ) else: add_context(path, '', value, endpoint, auth_type, inherited_location) return contexts def provider_from_postman(detector_name='', host='', value=''): detector = str(detector_name or '').lower() if detector in ('openai', 'anthropic', 'github', 'gitlab', 'stripe', 'slack'): return detector text = ' '.join([str(host or '').lower(), str(value or '').lower()]) if 'api.openai.com' in text or 'sk-proj-' in text or re.search(r'\bsk-[A-Za-z0-9]{20,}', str(value or '')): return 'openai' if 'anthropic.com' in text or 'sk-ant-' in text: return 'anthropic' if 'generativelanguage.googleapis.com' in text or 'aiplatform.googleapis.com' in text: return 'google' if 'huggingface.co' in text or str(value or '').startswith('hf_'): return 'huggingface' if 'github.com' in text or str(value or '').startswith(GITHUB_TOKEN_PREFIXES): return 'github' if 'gitlab' in text or str(value or '').startswith(GITLAB_TOKEN_PREFIXES): return 'gitlab' if 'stripe.com' in text or str(value or '').startswith(('sk_live_', 'rk_live_')): return 'stripe' return '' def credential_kind_from_postman(context, value): key = str(context.get('key') or '').lower() auth_type = str(context.get('auth_type') or '').lower() location = str(context.get('location') or '').lower() text = str(value or '').strip() if is_postman_placeholder(text): return 'placeholder' if auth_type == 'bearer' or key == 'authorization' or text.lower().startswith('bearer '): return 'jwt' if re.match(r'^(?:bearer\s+)?eyJ[A-Za-z0-9_-]+\.', text, re.IGNORECASE) else 'bearer_token' if ('api' in key and 'key' in key) or key in ('x-api-key', 'apikey'): return 'api_key' if 'client_secret' in key or 'client-secret' in key: return 'oauth_client_secret' if 'password' in key: return 'basic_auth_password' if auth_type == 'basic' else 'password' if location == 'query_param' and ('token' in key or 'key' in key): return 'api_key' if re.match(r'^eyJ[A-Za-z0-9_-]+\.', text): return 'jwt' return 'unknown' POSTMAN_GEMINI_KEY_RE = re.compile(r'(?:AIza[0-9A-Za-z\-_]{35}|AQ\.[0-9A-Za-z\-_]{50})') POSTMAN_AZURE_OPENAI_KEY_RE = re.compile(r'\b[a-f0-9]{32}\b', re.IGNORECASE) POSTMAN_AZURE_OPENAI_ENDPOINT_RE = re.compile(r'([a-z0-9-]+\.openai\.azure\.com)', re.IGNORECASE) FOUNDRY_ENDPOINT_HOST_RE = r'[a-z0-9-]+(?:\.[a-z0-9-]+)*\.(?:models\.ai\.azure\.com|services\.ai\.azure\.com|inference\.ai\.azure\.com)' POSTMAN_FOUNDRY_ENDPOINT_RE = re.compile(r'((?:https?://)?' + FOUNDRY_ENDPOINT_HOST_RE + r'(?:/[^\s:"\'<>\\]*)?)', re.IGNORECASE) FOUNDRY_ASSIGNMENT_RE = re.compile(r'''(?ix) (?:authorization|api[_-]?key|key|token|secret|credential|bearer) [^\n:=]{0,80} [:=] \s*["']?(?:bearer\s+)? ([A-Za-z0-9_./+=\-]{20,512}) ''') NON_FOUNDRY_KEY_PREFIXES = ( 'sk-', 'sk_', 'sk-or-', 'xai-', 'ghp_', 'gho_', 'ghu_', 'ghs_', 'ghr_', 'github_pat_', 'glpat-', 'glrt-', 'hf_', 'AIza', 'AQ.', 'zai-', 'gsk_', 'r8_', 'nvapi-', ) def keycheck_output_path(service, filename): root = getattr(scan_config, 'keycheck_dir', None) or os.path.join(get_results_dir() or os.getcwd(), 'keychecks') directory = os.path.join(root, service) ensure_private_directory(directory, reject_reparse=True) return os.path.join(directory, filename) def _candidate_limits(): return { 'artifact_items': max(1, int(getattr(scan_config, 'keycheck_candidate_artifact_max_items', 2000))), 'artifact_bytes': max(1, int(getattr(scan_config, 'keycheck_candidate_artifact_max_bytes', 2 * 1024 * 1024))), 'file_items': max(1, int(getattr(scan_config, 'keycheck_candidate_file_max_items', 100000))), 'file_bytes': max(1, int(getattr(scan_config, 'keycheck_candidate_file_max_bytes', 32 * 1024 * 1024))), 'line_bytes': max(1, int(getattr(scan_config, 'keycheck_candidate_line_max_bytes', 8192))), } class CandidateQueueCapacityError(RuntimeError): pass _CANDIDATE_CHECKED_LEDGERS = { 'gem.txt': 'geminiChecked.txt', 'azureOpenAI.txt': 'azureChecked.txt', 'azureFoundry.txt': 'azureChecked.txt', } def _candidate_identity(value): return str(value or '').strip().split('\t', 1)[0].strip() def _read_bounded_candidate_rows(path, limits, description): if not os.path.exists(path): return [], set(), 0, 0 reject_reparse_components(path) if os.path.getsize(path) > limits['file_bytes']: raise RuntimeError(f'{description} exceeds its aggregate byte bound: {path}') rows = [] identities = set() total_bytes = 0 with open(path, 'rb') as handle: for raw_line in handle: total_bytes += len(raw_line) if total_bytes > limits['file_bytes']: raise RuntimeError(f'{description} exceeds its aggregate byte bound: {path}') if len(raw_line) > limits['line_bytes']: raise RuntimeError(f'{description} line exceeds its byte bound: {path}') text = raw_line.rstrip(b'\r\n').decode('utf-8', errors='strict') if not text: continue rows.append(text) if len(rows) > limits['file_items']: raise RuntimeError(f'{description} exceeds its aggregate item bound: {path}') identity = _candidate_identity(text) if identity: identities.add(identity) return rows, identities, len(rows), total_bytes def _candidate_additions(lines, existing_identities, limits, checked_identities=()): seen = set(existing_identities) seen.update(checked_identities) additions = [] added_bytes = 0 for value in lines: line = str(value or '').strip() if not line or '\n' in line or '\r' in line: continue encoded = (line + '\n').encode('utf-8') identity = _candidate_identity(line) if not identity or len(encoded) > limits['line_bytes'] or identity in seen: continue additions.append(encoded) added_bytes += len(encoded) seen.add(identity) return additions, added_bytes def _candidate_batch_fits(existing_items, existing_bytes, additions, added_bytes, limits): return ( existing_items + len(additions) <= limits['file_items'] and existing_bytes + added_bytes <= limits['file_bytes'] ) def _rewrite_candidate_rows_atomic(path, rows): temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.tmp' descriptor = None flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL if hasattr(os, 'O_BINARY'): flags |= os.O_BINARY try: descriptor = os.open(temporary, flags, 0o600) os.close(descriptor) descriptor = None harden_private_file(temporary) with open(temporary, 'wb') as handle: for row in rows: handle.write((row + '\n').encode('utf-8')) handle.flush() os.fsync(handle.fileno()) harden_private_file(temporary) durable_replace(temporary, path) harden_private_file(path) finally: if descriptor is not None: os.close(descriptor) try: if os.path.exists(temporary): os.remove(temporary) except OSError: pass def _compact_checked_candidate_rows_unlocked(path, rows, existing_bytes, limits): ledger_name = _CANDIDATE_CHECKED_LEDGERS.get(os.path.basename(path)) if not ledger_name: return rows, set(), existing_bytes checked_path = os.path.join(os.path.dirname(path), ledger_name) checked_lock_path = f'{checked_path}.lock' # Candidate locks are always outermost; checked-ledger writers never take them. checked_lock = acquire_file_lock(checked_lock_path, stale_sec=120, timeout_sec=10) try: _, checked_identities, _, _ = _read_bounded_candidate_rows( checked_path, limits, 'keycheck checked ledger', ) finally: release_file_lock(checked_lock, checked_lock_path) retained = [row for row in rows if _candidate_identity(row) not in checked_identities] if len(retained) != len(rows): encoded_sizes = [len((row + '\n').encode('utf-8')) for row in retained] retained_bytes = sum(encoded_sizes) if retained_bytes > limits['file_bytes'] or any( size > limits['line_bytes'] for size in encoded_sizes ): raise RuntimeError(f'compacted keycheck candidate file would exceed its bounds: {path}') _rewrite_candidate_rows_atomic(path, retained) else: retained_bytes = existing_bytes return retained, checked_identities, retained_bytes def _append_unique_lines_unlocked(path, lines, limits): ensure_private_directory(os.path.dirname(path), reject_reparse=True) rows, existing, existing_items, existing_bytes = _read_bounded_candidate_rows( path, limits, 'keycheck candidate file', ) additions, added_bytes = _candidate_additions(lines, existing, limits) if not additions: return 0 if not _candidate_batch_fits(existing_items, existing_bytes, additions, added_bytes, limits): rows, checked, existing_bytes = _compact_checked_candidate_rows_unlocked( path, rows, existing_bytes, limits, ) existing = {_candidate_identity(row) for row in rows if _candidate_identity(row)} existing_items = len(rows) additions, added_bytes = _candidate_additions(lines, existing, limits, checked) if not additions: return 0 if not _candidate_batch_fits(existing_items, existing_bytes, additions, added_bytes, limits): raise CandidateQueueCapacityError( f'keycheck candidate queue has insufficient capacity for the complete offered batch: {path}' ) with open(path, 'ab') as f: f.write(b''.join(additions)) f.flush() os.fsync(f.fileno()) harden_private_file(path) return len(additions) def append_unique_lines_locked(path, lines): limits = _candidate_limits() lock_path = f'{path}.lock' lock = acquire_file_lock(lock_path, stale_sec=120, timeout_sec=10) try: return _append_unique_lines_unlocked(path, lines, limits) finally: release_file_lock(lock, lock_path) def append_unique_line(path, line): return bool(append_unique_lines_locked(path, [line])) def append_unique_line_locked(path, line): return bool(append_unique_lines_locked(path, [line])) def _collect_candidate_line(batches, seen, budget, path, line): limits = budget['limits'] text = str(line or '').strip() encoded_size = len((text + '\n').encode('utf-8')) if text else 0 if not text or encoded_size > limits['line_bytes'] or text in seen.setdefault(path, set()): return False if budget['items'] >= limits['artifact_items'] or budget['bytes'] + encoded_size > limits['artifact_bytes']: budget['truncated'] = True return False seen[path].add(text) batches.setdefault(path, []).append(text) budget['items'] += 1 budget['bytes'] += encoded_size return True def normalize_foundry_endpoint(value): text = str(value or '').strip().strip('"\'`,;') if not text: return '' split_text = text if re.match(r'(?i)^https?://', text) else 'https://' + text try: parsed = urlsplit(split_text) host = parsed.netloc or parsed.path.split('/', 1)[0] path = parsed.path if parsed.netloc else ('/' + parsed.path.split('/', 1)[1] if '/' in parsed.path else '') except Exception: host, path = re.sub(r'(?i)^https?://', '', text).split('/', 1)[0], '' path = path.rstrip('.,;:)]}/') terminal_routes = ( ('/models/chat/completions', ''), ('/openai/v1/chat/completions', '/openai/v1'), ('/v1/chat/completions', '/v1'), ('/chat/completions', ''), ('/v1/models', '/v1'), ('/models', ''), ) lower_path = path.lower() for suffix, replacement in terminal_routes: if lower_path.endswith(suffix): path = path[:-len(suffix)] + replacement break return (host + path).strip('/').lower() def dedupe_foundry_endpoints(endpoints): normalized = [] for endpoint in endpoints or []: endpoint = normalize_foundry_endpoint(endpoint) if endpoint and endpoint not in normalized: normalized.append(endpoint) kept = [] for endpoint in sorted(normalized, key=len, reverse=True): if any(other.startswith(endpoint + '/') for other in kept): continue kept.append(endpoint) return list(reversed(kept)) def foundry_keyish(value): text = re.sub(r'(?i)^bearer\s+', '', str(value or '').strip().strip('"\'`,;')).strip() lower = text.lower() if not (20 <= len(text) <= 512): return False if any(marker in lower for marker in ('http://', 'https://', '{{', '${', '<', 'azure.com')): return False if any(ch.isspace() for ch in text): return False if re.match(r'(?i)^(?:authorization|api[_-]?key|key|token|secret|credential|bearer)\s*[:=]', text): return False if text.startswith(NON_FOUNDRY_KEY_PREFIXES): return False return bool(re.search(r'[A-Za-z]', text) and re.search(r'[0-9]', text)) def is_foundry_detector(finding): detector = str((finding or {}).get('DetectorName') or '').lower() extra = (finding or {}).get('ExtraData') if isinstance((finding or {}).get('ExtraData'), dict) else {} name = str(extra.get('name') or '').lower() return detector.startswith('azurefoundry') or (detector == 'customregex' and name.startswith('azurefoundry')) def foundry_finding_text(finding): parts = [] for value in finding_raw_values(finding): parts.append(value) if is_foundry_detector(finding): return '\n'.join(part for part in parts if part) context = finding.get('ScannerContext') if isinstance(finding, dict) else None if isinstance(context, dict): parts.extend(str(context.get(item) or '') for item in ('nearby', 'file')) postman_context = finding.get('PostmanContext') if isinstance(finding, dict) else None if isinstance(postman_context, dict): parts.extend(str(postman_context.get(item) or '') for item in ('endpoint', 'host', 'variable_name')) extra = finding.get('ExtraData') if isinstance(finding, dict) else None if isinstance(extra, dict): parts.extend(str(value) for value in extra.values() if isinstance(value, str)) return '\n'.join(part for part in parts if part) def foundry_candidate_keys(finding, text): keys = [] if is_foundry_detector(finding): for value in finding_raw_values(finding): if foundry_keyish(value) and value not in keys: keys.append(value.strip().strip('"\'`,;')) for match in FOUNDRY_ASSIGNMENT_RE.findall(text or ''): candidate = re.sub(r'(?i)^bearer\s+', '', str(match or '').strip().strip('"\'`,;')).strip() if foundry_keyish(candidate) and candidate not in keys: keys.append(candidate) return keys[:5] def foundry_candidate_origin(result, finding): file_path, line_number = finding_source_location(finding) location = f'{file_path}:{line_number}' if file_path and line_number else file_path or '' target = result.get('target') or '' scan_type = result.get('scan_type') or '' return ':'.join(part for part in (scan_type, str(target), location) if part) def write_foundry_keycheck_candidates_from_findings(result): findings = result.get('findings') or [] if not findings: return 0 azure_foundry_path = keycheck_output_path('azure', 'azureFoundry.txt') batches = {} seen = {} budget = {'items': 0, 'bytes': 0, 'truncated': False, 'limits': _candidate_limits()} for finding in findings: if not is_foundry_detector(finding): continue text = foundry_finding_text(finding) endpoints = [] for match in POSTMAN_FOUNDRY_ENDPOINT_RE.findall(text): endpoint = normalize_foundry_endpoint(match) if endpoint and endpoint not in endpoints: endpoints.append(endpoint) if not endpoints: continue endpoints = dedupe_foundry_endpoints(endpoints) keys = foundry_candidate_keys(finding, text) if not keys: continue origin = foundry_candidate_origin(result, finding) for endpoint in endpoints[:5]: for key in keys[:5]: metadata = json.dumps({ 'origin': origin, 'finding_uid': finding.get('finding_uid') or '', }, ensure_ascii=False, separators=(',', ':')) _collect_candidate_line( batches, seen, budget, azure_foundry_path, f'{endpoint}:{key}\t{metadata}', ) if budget['truncated']: break if budget['truncated']: break if budget['truncated']: break if budget['truncated']: logger.warning('Azure Foundry candidates reached the per-artifact bound; remaining values were capped') return append_unique_lines_locked(azure_foundry_path, batches.get(azure_foundry_path, [])) def artifact_origin_label(target_data, cache_path): origin = target_data.get('origin') if isinstance(target_data.get('origin'), dict) else {} source = target_data.get('source') or origin.get('provider') or 'artifact' repo = target_data.get('repo') or origin.get('repo') or '' path = target_data.get('path') or origin.get('path') or os.path.basename(cache_path or '') sha = target_data.get('sha') or origin.get('sha') or target_data.get('sha256') or '' return f'{source}:{repo}:{path}:{sha}' def context_values_for_pairing(contexts): endpoints = [] values = [] for context in contexts or []: value = str(context.get('value') or '').strip() if not value or is_postman_placeholder(value): continue text = ' '.join([value, str(context.get('endpoint') or ''), str(context.get('host') or ''), str(context.get('key') or '')]) values.append((context, value, text)) for pattern in (POSTMAN_AZURE_OPENAI_ENDPOINT_RE, POSTMAN_FOUNDRY_ENDPOINT_RE): for match in pattern.findall(text): endpoint = normalize_foundry_endpoint(match) if pattern is POSTMAN_FOUNDRY_ENDPOINT_RE else str(match).lower().strip('/') if endpoint not in endpoints: endpoints.append(endpoint) return endpoints, values def write_structured_keycheck_candidates(cache_path, target_data): contexts = load_postman_context(cache_path) if not contexts: return {} context_value_by_path = {str(item.get('path') or ''): str(item.get('value') or '') for item in contexts} endpoints, values = context_values_for_pairing(contexts) origin = artifact_origin_label(target_data or {}, cache_path) counts = {'gemini': 0, 'azure_openai': 0, 'azure_foundry': 0} gemini_path = keycheck_output_path('gemini', 'gem.txt') azure_openai_path = keycheck_output_path('azure', 'azureOpenAI.txt') azure_foundry_path = keycheck_output_path('azure', 'azureFoundry.txt') batches = {} seen = {} budget = {'items': 0, 'bytes': 0, 'truncated': False, 'limits': _candidate_limits()} labels = { gemini_path: 'gemini', azure_openai_path: 'azure_openai', azure_foundry_path: 'azure_foundry', } for context, value, text in values: for key in POSTMAN_GEMINI_KEY_RE.findall(value): _collect_candidate_line( batches, seen, budget, gemini_path, f'{key}\t{origin}\t{context.get("path") or ""}', ) azure_key_match = POSTMAN_AZURE_OPENAI_KEY_RE.search(value) if azure_key_match: local_azure = [str(match).lower().strip('/') for match in POSTMAN_AZURE_OPENAI_ENDPOINT_RE.findall(text)] azure_endpoints = local_azure or [endpoint for endpoint in endpoints if POSTMAN_AZURE_OPENAI_ENDPOINT_RE.fullmatch(endpoint)] for endpoint in azure_endpoints[:5]: _collect_candidate_line( batches, seen, budget, azure_openai_path, f'{endpoint}:{azure_key_match.group(0)}\t{origin}\t{context.get("path") or ""}', ) sibling_name = '' path = str(context.get('path') or '') if path.endswith('.value'): sibling_name = context_value_by_path.get(path[:-6] + '.key', '') key_context = ' '.join([str(context.get('key') or ''), sibling_name]).lower() foundry_value = re.sub(r'(?i)^bearer\s+', '', value.strip().strip('"\'`,;')).strip() if foundry_keyish(foundry_value) and any(word in key_context for word in ('authorization', 'bearer', 'key', 'token', 'secret', 'api')): foundry_endpoints = dedupe_foundry_endpoints(POSTMAN_FOUNDRY_ENDPOINT_RE.findall(text)) for endpoint in foundry_endpoints[:5]: if endpoint: _collect_candidate_line( batches, seen, budget, azure_foundry_path, f'{endpoint}:{foundry_value}\t{origin}\t{context.get("path") or ""}', ) if budget['truncated']: break if budget['truncated']: logger.warning('Structured keycheck candidates reached the per-artifact bound; remaining values were capped') for path, lines in batches.items(): counts[labels[path]] = append_unique_lines_locked(path, lines) return {key: value for key, value in counts.items() if value} def _postman_substring_match(budget, needle, value): if _context_budget_expired(budget): return None if budget['postman_comparisons'] >= budget['max_postman_comparisons']: return None budget['postman_comparisons'] += 1 return needle in value def attach_postman_context(results, cache_path, budget=None): budget = budget or context_enrichment_budget() results.pop('structured_keycheck_pending', None) try: payload, complete, reason = _read_context_source(cache_path, budget) if payload is None or not complete: if reason in ('elapsed', 'source_bytes'): dimension = 'elapsed-time' if reason == 'elapsed' else 'source-byte' _add_context_warning(results, 'budget', f'{dimension} budget was exhausted; findings were retained') else: _add_context_warning(results, 'failure', 'Postman context could not be read; findings were retained') return results contexts = load_postman_context(cache_path, payload=payload) except Exception: logger.warning('Optional Postman context parsing failed; parsed findings were retained') _add_context_warning(results, 'failure', 'Postman JSON context parsing failed; findings were retained') return results results['structured_keycheck_pending'] = True if not contexts: return results try: placeholder_count = 0 context_values = [] exact_values = {} for index, context in enumerate(contexts): if index % 256 == 0 and _context_budget_expired(budget): _add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; findings were retained') return results if not isinstance(context, dict): continue value = str(context.get('value') or '') context_values.append((context, value)) if value: exact_values.setdefault(value, context) placeholder_count += int(is_postman_placeholder(value)) results['postman_context'] = { 'context_count': len(contexts), 'placeholder_count': placeholder_count, } for finding in results.get('findings') or []: if _context_budget_expired(budget): _add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained') return results if not isinstance(finding, dict): continue if not _context_budget_claim_finding(budget, finding): _add_context_warning(results, 'budget', 'finding-count budget was exhausted; remaining findings were retained') return results raw_values = finding_raw_values(finding) matched = next((exact_values.get(raw) for raw in raw_values if raw and exact_values.get(raw)), None) for raw in raw_values if not matched else (): if not raw: continue for context, value in context_values: comparison = _postman_substring_match(budget, raw, value) if comparison is None: _add_context_warning(results, 'budget', 'Postman comparison budget was exhausted; remaining findings were retained') return results if comparison: matched = context break if matched: break if not matched and raw_values and raw_values[0][:8]: prefix = raw_values[0][:8] for context, value in context_values: comparison = _postman_substring_match(budget, prefix, value) if comparison is None: _add_context_warning(results, 'budget', 'Postman comparison budget was exhausted; remaining findings were retained') return results if comparison: matched = context break if not matched: continue raw_value = raw_values[0] if raw_values else matched.get('value') kind = credential_kind_from_postman(matched, raw_value) confidence = 'verified' if finding.get('Verified') else 'detector_match' if finding.get('DetectorName') else 'structured_complete' if kind != 'unknown' else 'context_only' if kind == 'placeholder': confidence = 'placeholder' endpoint = sanitize_endpoint(matched.get('endpoint')) host = sanitize_endpoint_host(endpoint) or sanitize_endpoint_host(matched.get('host')) finding['PostmanContext'] = { 'provider': provider_from_postman(finding.get('DetectorName'), host, raw_value), 'credential_kind': kind, 'credential_confidence': confidence, 'context_location': matched.get('location'), 'variable_name': matched.get('key'), 'endpoint': endpoint, 'host': host, 'auth_type': matched.get('auth_type'), 'json_path': matched.get('path'), 'placeholder': is_postman_placeholder(matched.get('value')), } except Exception: logger.warning('Optional Postman context matching failed; parsed findings were retained') _add_context_warning(results, 'failure', 'Postman context matching failed; findings were retained') return results def scan_postman_target(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, max_artifact_size_mb=20, token=None): data = parse_postman_target(target) cache_path = data.get('cache_path') or data.get('local_path') if not cache_path: return {"findings": [], "errors": ["Postman target missing cached artifact"], "package": data.get('origin') or data} try: cache_path, declared_size = validate_postman_cache_artifact(data, max_artifact_size_mb) except PostmanCacheTooLarge as exc: return {"findings": [], "errors": [], "skipped": str(exc), "package": data.get('origin') or data} except (OSError, PostmanCacheValidationError) as exc: return {"findings": [], "errors": [f"Postman cache path rejected: {exc}"], "package": data.get('origin') or data} _raise_if_scan_slot_fatal() work_dir = create_command_work_dir() _raise_if_scan_slot_fatal() if not work_dir: return {"findings": [], "errors": ["Unable to create Postman work dir"], "package": data.get('origin') or data} try: staged_path = os.path.join(work_dir, postman_stage_filename(data)) _raise_if_scan_slot_fatal() shutil.copyfile(cache_path, staged_path) _raise_if_scan_slot_fatal() harden_private_file(staged_path) cmd = [get_trufflehog_cmd(), 'filesystem', work_dir, '--json', '--no-update'] append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config) results = { "findings": [], "errors": [], "package": data.get('origin') or data, "postman": data, "bytes": declared_size, "postman_max_artifact_size_mb": int(max_artifact_size_mb or 0), } with run_command_streamed(cmd, timeout_sec) as output: apply_trufflehog_diagnostics(results, output, output.returncode, 'postman') append_trufflehog_findings(results, output.stdout_lines()) enrichment_budget = context_enrichment_budget() attach_nearby_context(results, enrichment_budget) artifact_path = str(data.get('path') or data.get('name') or data.get('url') or '').split('?', 1)[0].lower() if str(data.get('kind') or '').lower() != 'bruno' or not artifact_path.endswith('.bru'): attach_postman_context(results, staged_path, enrichment_budget) return apply_finding_filters(results, cache_path) except ScanSlotFatalError: raise except Exception as e: return {"findings": [], "errors": [f"Postman scan failed: {str(e)}"], "package": data.get('origin') or data} finally: cleanup_command_work_dir(work_dir) def safe_path_component(value, max_len=120): text = str(value or '')[:max_len] return re.sub(r'[^A-Za-z0-9_.-]+', '_', text).strip('._') or 'item' def run_filesystem_scan(root_dir, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None): cmd = [get_trufflehog_cmd(), 'filesystem', root_dir, '--json', '--no-update'] append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config) results = {"findings": [], "errors": []} with run_command_streamed(cmd, timeout_sec) as output: apply_trufflehog_diagnostics(results, output, output.returncode, 'filesystem') append_trufflehog_findings(results, output.stdout_lines()) attach_nearby_context(results) return apply_finding_filters(results, root_dir) CI_ARTIFACT_ALLOWED_SUFFIXES = { '.env', '.log', '.txt', '.json', '.yaml', '.yml', '.xml', '.html', '.lcov', '.sarif', '.tfstate', '.tfplan', '.py', '.js', '.jsx', '.ts', '.tsx', '.sh', '.toml', '.ini', '.cfg', '.conf', '.properties', '.ipynb', '.md', '.out', '.err', '.csv', '.tsv', } CI_ARTIFACT_ALLOWED_NAMES = {'dockerfile', 'makefile', 'procfile'} CI_ARTIFACT_SKIP_DIRS = { '.git', '.hg', '.svn', 'node_modules', '__pycache__', '.venv', 'venv', 'env', '.mypy_cache', '.pytest_cache', '.tox', '.gradle', '.idea', '.vscode', } def ci_artifact_file_allowed(name, allowed_suffixes=None): if allowed_suffixes is None: return True normalized = name.replace('\\', '/').strip('/') parts = [part.lower() for part in normalized.split('/') if part] if any(part in CI_ARTIFACT_SKIP_DIRS for part in parts[:-1]): return False base = parts[-1] if parts else '' if base in CI_ARTIFACT_ALLOWED_NAMES or base.startswith('.env'): return True return any(base.endswith(suffix) for suffix in allowed_suffixes) def safe_extract_zip(zip_path, destination, max_file_size_mb=20, max_files=1000, allowed_suffixes=None, max_total_size_mb=250): _raise_if_scan_slot_fatal() extracted = 0 max_bytes = int(max_file_size_mb or 0) * 1024 * 1024 max_total_bytes = int(max_total_size_mb or 0) * 1024 * 1024 total_bytes = 0 validate_zip_central_directory(zip_path, max_files) with zipfile.ZipFile(zip_path) as archive: for info in archive.infolist(): _raise_if_scan_slot_fatal() if info.is_dir(): continue if max_files and extracted >= max_files: raise ValueError(f'Zip archive exceeds {max_files} extracted files') if max_bytes and info.file_size > max_bytes: raise ValueError(f'Zip member exceeds {max_file_size_mb} MB: {info.filename}') if max_total_bytes and total_bytes + max(0, int(info.file_size or 0)) > max_total_bytes: raise ValueError(f'Zip archive exceeds {max_total_size_mb} MB extracted') name = info.filename.replace('\\', '/') validate_archive_member_name(name) if not ci_artifact_file_allowed(name, allowed_suffixes): continue target_path = os.path.abspath(os.path.normpath(os.path.join(destination, name))) if os.path.commonpath([os.path.abspath(destination), target_path]) != os.path.abspath(destination): continue os.makedirs(os.path.dirname(target_path), exist_ok=True) with archive.open(info) as src, open(target_path, 'wb') as dst: while True: _raise_if_scan_slot_fatal() chunk = src.read(256 * 1024) if not chunk: break dst.write(chunk) _raise_if_scan_slot_fatal() extracted += 1 total_bytes += max(0, int(info.file_size or 0)) return extracted def remove_file_quiet(path): try: reject_reparse_components(path) os.remove(path) except OSError: pass @dataclass(frozen=True) class DownloadOutcome: path: str status_code: int bytes_written: int error: str = '' @property def ok(self): return bool(self.path and not self.error) def download_to_file( url, destination, headers=None, timeout=20, max_size_mb=0, max_size_bytes=None, byte_budget=None, ): _raise_if_scan_slot_fatal() destination = os.path.abspath(destination) require_private_directory(os.path.dirname(destination), create=False) reject_reparse_components(os.path.dirname(destination)) current_url = str(url) current_headers = dict(headers or {}) configured_max = int(max_size_mb or 0) * 1024 * 1024 explicit_max = max(0, int(max_size_bytes or 0)) max_bytes = min(value for value in (configured_max, explicit_max) if value > 0) if configured_max and explicit_max else configured_max or explicit_max if byte_budget is not None and int(byte_budget.get('remaining', 0)) <= 0: return DownloadOutcome('', 0, 0, 'target byte budget exhausted') for _ in range(6): _raise_if_scan_slot_fatal() try: response = _direct_request( 'GET', current_url, headers=current_headers, timeout=timeout, allow_redirects=False, stream=True, ) except requests.RequestException as exc: return DownloadOutcome('', 0, 0, str(exc)) if _scan_slot_fatal_event.is_set(): response.close() _raise_if_scan_slot_fatal() if response.status_code in (301, 302, 303, 307, 308) and response.headers.get('Location'): next_url = urljoin(current_url, response.headers['Location']) old_host = (urlsplit(current_url).hostname or '').lower() parsed_next = urlsplit(next_url) response.close() if parsed_next.scheme != 'https': return DownloadOutcome('', 0, 0, f'unsafe redirect scheme: {parsed_next.scheme}') if (parsed_next.hostname or '').lower() != old_host: current_headers = { key: value for key, value in current_headers.items() if key.lower() not in ('authorization', 'private-token', 'cookie', 'proxy-authorization') } current_url = next_url continue if response.status_code >= 400: try: prefix = bytearray() for chunk in response.iter_content(chunk_size=300): _raise_if_scan_slot_fatal() prefix.extend(chunk[:max(0, 300 - len(prefix))]) if len(prefix) >= 300: break message = bytes(prefix).decode(response.encoding or 'utf-8', errors='replace') finally: response.close() return DownloadOutcome('', response.status_code, 0, message) content_length = response.headers.get('Content-Length') if content_length: try: declared = int(content_length) remaining = int(byte_budget.get('remaining', 0)) if byte_budget is not None else 0 if max_bytes and declared > max_bytes: response.close() return DownloadOutcome('', response.status_code, 0, f'response exceeds configured {max_bytes}-byte limit') if byte_budget is not None and declared > remaining: response.close() return DownloadOutcome('', response.status_code, 0, 'target byte budget exhausted') except ValueError: pass total = 0 temporary = f'{destination}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.partial' descriptor = None try: flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0) descriptor = os.open(temporary, flags, 0o600) os.close(descriptor) descriptor = None harden_private_file(temporary) with open(temporary, 'wb', buffering=0) as output: for chunk in response.iter_content(chunk_size=256 * 1024): _raise_if_scan_slot_fatal() if not chunk: continue if max_bytes and total + len(chunk) > max_bytes: raise CommandOutputLimitError(f'object response exceeds configured {max_bytes}-byte limit') if byte_budget is not None: remaining = int(byte_budget.get('remaining', 0)) if len(chunk) > remaining: byte_budget['remaining'] = 0 raise CommandOutputLimitError('target byte budget exhausted') byte_budget['remaining'] = remaining - len(chunk) output.write(chunk) total += len(chunk) output.flush() os.fsync(output.fileno()) harden_private_file(temporary) durable_replace(temporary, destination) if not private_file_ready(destination): raise OSError('download destination lost its private file identity') return DownloadOutcome(destination, response.status_code, total) except (CommandOutputLimitError, OSError, requests.RequestException) as exc: return DownloadOutcome('', response.status_code, total, str(exc)) finally: response.close() if descriptor is not None: os.close(descriptor) if os.path.exists(temporary): try: os.remove(temporary) except OSError: pass return DownloadOutcome('', 0, 0, 'too many redirects') def github_api_get(url, token=None, timeout=20, stream=False): response = api_request( 'GET', url, headers=github_headers(token), timeout=timeout, stream=stream, ) if token and response.status_code in (401, 403): response.close() anonymous = api_request( 'GET', url, headers=github_headers(None), timeout=timeout, stream=stream, ) if anonymous.status_code < 400: return anonymous response = anonymous if response.status_code >= 400: raise github_api_error(response) return response def bounded_response_json(response, max_bytes=8 * 1024 * 1024): max_bytes = max(1, int(max_bytes)) declared = response.headers.get('Content-Length') if declared: try: declared = int(declared) except ValueError as exc: response.close() raise ApiRequestError('API JSON response has an invalid Content-Length') from exc if declared > max_bytes: response.close() raise ApiRequestError(f'API JSON response exceeds {max_bytes} bytes') payload = bytearray() try: for chunk in response.iter_content(chunk_size=64 * 1024): if not chunk: continue if len(payload) + len(chunk) > max_bytes: raise ApiRequestError(f'API JSON response exceeds {max_bytes} bytes') payload.extend(chunk) return json.loads(bytes(payload).decode('utf-8', errors='strict')) except (UnicodeDecodeError, json.JSONDecodeError) as exc: raise ApiRequestError('API response is not bounded valid UTF-8 JSON') from exc finally: response.close() def select_ci_runs(runs, runs_per_repo=5, lookback_days=30, failed_first=True): cutoff = datetime.now(timezone.utc) - timedelta(days=int(lookback_days or 0)) if int(lookback_days or 0) > 0 else None kept = [] for run in runs or []: created = parse_postman_time(run.get('created_at')) if cutoff and created and created < cutoff: continue kept.append(run) def created_ts(item): parsed = parse_postman_time(item.get('created_at')) return parsed.timestamp() if parsed else 0 if failed_first: kept.sort(key=lambda item: (0 if item.get('conclusion') not in ('success', None) else 1, -created_ts(item))) else: kept.sort(key=lambda item: -created_ts(item)) return kept[:max(1, int(runs_per_repo or 5))] def scan_github_actions_repo(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, ci_runs_per_repo=5, ci_lookback_days=30, ci_max_log_archive_mb=50, ci_max_log_file_mb=20, ci_failed_first=True, ci_scan_artifacts=False, ci_max_artifacts_per_run=3, ci_max_artifact_archive_mb=50, ci_max_artifact_file_mb=10, ci_max_artifact_files=1000, ci_target_max_download_mb=500, fetch_timeout=20): repo, repo_url = parse_github_repo_target(target) if not repo: return {"findings": [], "errors": ["Unable to parse GitHub repo target"]} logger.info(f"Scanning GitHub Actions logs: {repo}") _raise_if_scan_slot_fatal() work_dir = create_command_work_dir() _raise_if_scan_slot_fatal() if not work_dir: return {"findings": [], "errors": ["Unable to create GitHub Actions work dir"]} try: runs_url = f'https://api.github.com/repos/{repo}/actions/runs?per_page={max(1, min(100, int(ci_runs_per_repo or 5) * 3))}' response = github_api_get(runs_url, token, fetch_timeout, stream=True) payload = bounded_response_json(response) if not isinstance(payload, dict) or 'workflow_runs' not in payload or not isinstance(payload.get('workflow_runs'), list): raise ApiRequestError('invalid GitHub Actions runs payload') runs = select_ci_runs(payload.get('workflow_runs') or [], ci_runs_per_repo, ci_lookback_days, ci_failed_first) if not runs: return {"findings": [], "errors": [], "skipped": "no recent workflow runs", "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url}} max_archive_bytes = int(ci_max_log_archive_mb or 0) * 1024 * 1024 max_artifact_archive_bytes = int(ci_max_artifact_archive_mb or 0) * 1024 * 1024 extracted_total = 0 artifact_extracted_total = 0 download_failures = [] downloaded_total = 0 target_download_limit = int(ci_target_max_download_mb or 0) * 1024 * 1024 download_budget = {'remaining': target_download_limit} if target_download_limit else None run_meta = [] for run in runs: _raise_if_scan_slot_fatal() if download_budget is not None and download_budget['remaining'] <= 0: download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap') break run_id = run.get('id') if not run_id: continue extracted = 0 artifacts_meta = [] logs_url = f'https://api.github.com/repos/{repo}/actions/runs/{run_id}/logs' zip_path = os.path.join(work_dir, f'github_actions_{safe_path_component(repo)}_{run_id}.zip') authenticated_status = None try: download = download_to_file( logs_url, zip_path, github_headers(token), fetch_timeout, ci_max_log_archive_mb, max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, byte_budget=download_budget, ) authenticated_status = download.status_code except requests.RequestException as exc: download = DownloadOutcome('', 0, 0, str(exc)) if not download.ok and token and download.status_code in (401, 403): anonymous = download_to_file( logs_url, zip_path, github_headers(None), fetch_timeout, ci_max_log_archive_mb, max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, byte_budget=download_budget, ) if not anonymous.ok: download = DownloadOutcome( '', authenticated_status or anonymous.status_code, 0, f'authenticated HTTP {authenticated_status}; anonymous HTTP ' f'{anonymous.status_code}: {anonymous.error}', ) else: download = anonymous if download.ok: downloaded_total += download.bytes_written run_dir = os.path.join(work_dir, 'github_actions', safe_path_component(repo), str(run_id)) ensure_private_directory(run_dir, reject_reparse=True) try: extracted = safe_extract_zip( zip_path, run_dir, ci_max_log_file_mb, max_total_size_mb=ci_max_log_archive_mb, ) harden_private_tree(run_dir) except (zipfile.BadZipFile, ValueError) as exc: logger.warning(f"GitHub Actions logs archive rejected for {repo} run {run_id}: {exc}") download_failures.append(f'log archive rejected: {exc}') extracted = 0 remove_file_quiet(zip_path) else: if download.status_code not in (404, 410): download_failures.append(f'log download HTTP {download.status_code}: {download.error}') run_dir = os.path.join(work_dir, 'github_actions', safe_path_component(repo), str(run_id)) ensure_private_directory(run_dir, reject_reparse=True) if ci_scan_artifacts: artifacts_url = f'https://api.github.com/repos/{repo}/actions/runs/{run_id}/artifacts?per_page={max(1, min(100, int(ci_max_artifacts_per_run or 3)))}' try: artifacts_response = github_api_get( artifacts_url, token, fetch_timeout, stream=True, ) artifact_payload = bounded_response_json(artifacts_response) if not isinstance(artifact_payload, dict) or 'artifacts' not in artifact_payload or not isinstance(artifact_payload.get('artifacts'), list): raise ApiRequestError('invalid GitHub Actions artifacts payload') artifacts = artifact_payload.get('artifacts') or [] except ScanSlotFatalError: raise except Exception as exc: logger.warning(f"Unable to list GitHub Actions artifacts for {repo} run {run_id}: {exc}") download_failures.append(f'artifact listing failed: {exc}') artifacts = [] for artifact in artifacts[:max(0, int(ci_max_artifacts_per_run or 3))]: _raise_if_scan_slot_fatal() if download_budget is not None and download_budget['remaining'] <= 0: download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap') break if artifact.get('expired'): continue size = int(artifact.get('size_in_bytes') or 0) if max_artifact_archive_bytes and size and size > max_artifact_archive_bytes: download_failures.append(f'artifact declared size exceeds {ci_max_artifact_archive_mb} MB') continue download_url = artifact.get('archive_download_url') if not download_url: continue artifact_id = artifact.get('id') or safe_path_component(artifact.get('name')) artifact_zip = os.path.join(work_dir, f'github_actions_artifact_{safe_path_component(repo)}_{run_id}_{artifact_id}.zip') download = download_to_file( download_url, artifact_zip, github_headers(token), fetch_timeout, ci_max_artifact_archive_mb, max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, byte_budget=download_budget, ) authenticated_status = download.status_code if not download.ok and token and download.status_code in (401, 403): anonymous = download_to_file( download_url, artifact_zip, github_headers(None), fetch_timeout, ci_max_artifact_archive_mb, max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, byte_budget=download_budget, ) if not anonymous.ok: download = DownloadOutcome( '', authenticated_status or anonymous.status_code, 0, f'authenticated HTTP {authenticated_status}; anonymous HTTP ' f'{anonymous.status_code}: {anonymous.error}', ) else: download = anonymous if not download.ok: if download.status_code not in (404, 410): logger.warning(f"GitHub Actions artifact download HTTP {download.status_code} for {repo} run {run_id}: {download.error}") download_failures.append(f'artifact download HTTP {download.status_code}: {download.error}') continue downloaded_total += download.bytes_written artifact_dir = os.path.join(run_dir, 'artifacts', safe_path_component(artifact.get('name') or artifact_id)) ensure_private_directory(artifact_dir, reject_reparse=True) try: artifact_files = safe_extract_zip( artifact_zip, artifact_dir, ci_max_artifact_file_mb, ci_max_artifact_files, CI_ARTIFACT_ALLOWED_SUFFIXES, max_total_size_mb=ci_max_artifact_archive_mb, ) harden_private_tree(artifact_dir) except (zipfile.BadZipFile, ValueError) as exc: logger.warning(f"GitHub Actions artifact rejected for {repo} run {run_id}: {artifact.get('name')}: {exc}") download_failures.append(f'artifact archive rejected: {exc}') artifact_files = 0 remove_file_quiet(artifact_zip) artifact_extracted_total += artifact_files artifacts_meta.append({ 'artifact_id': artifact.get('id'), 'name': artifact.get('name'), 'size_in_bytes': size, 'expired': artifact.get('expired'), 'extracted_files': artifact_files, }) extracted_total += extracted run_meta.append({ 'run_id': run_id, 'run_number': run.get('run_number'), 'workflow_name': run.get('name'), 'status': run.get('status'), 'conclusion': run.get('conclusion'), 'created_at': run.get('created_at'), 'updated_at': run.get('updated_at'), 'extracted_files': extracted, 'artifacts': artifacts_meta, }) if extracted_total == 0 and artifact_extracted_total == 0: if download_failures: text = '; '.join(download_failures[:5]) all_failures = '; '.join(download_failures) auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any( token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit') ) return { "findings": [], "errors": [text], "error_class": 'source_auth' if auth_failure else 'remote_transient', "retryable": True, "source_failure": auth_failure, "source_failure_category": 'auth_forbidden' if auth_failure else 'remote_transient', "source_failure_auth_related": auth_failure, "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url, "runs": run_meta}, } return {"findings": [], "errors": [], "skipped": "no downloadable workflow logs or artifacts", "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url, "runs": run_meta}} _raise_if_scan_slot_fatal() results = run_filesystem_scan(work_dir, timeout_sec, detectors, exclude_detectors, no_verification, trufflehog_config) results['package'] = {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url, "runs": run_meta, "log_files": extracted_total, "artifact_files": artifact_extracted_total} if download_failures: text = '; '.join(download_failures[:5]) all_failures = '; '.join(download_failures) auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any( token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit') ) results['errors'] = list(results.get('errors') or []) + [text] results['error_class'] = 'source_auth' if auth_failure else 'remote_transient' results['retryable'] = True results['source_failure'] = auth_failure results['source_failure_category'] = 'auth_forbidden' if auth_failure else 'remote_transient' results['source_failure_auth_related'] = auth_failure return results except ScanSlotFatalError: raise except RateLimitError as exc: category = getattr(exc, 'category', '') if category == 'not_found': return {"findings": [], "errors": [], "skipped": str(exc), "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url}} return { "findings": [], "errors": [str(exc)], "error_class": 'source_auth' if getattr(exc, 'auth_related', True) else 'remote_transient', "retryable": True, "source_failure": True, "source_failure_category": category or 'rate_limit', "source_failure_auth_related": bool(getattr(exc, 'auth_related', True)), "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url}, **({'_diagnostic_http': exc.diagnostic_http} if exc.diagnostic_http else {}), } except Exception as exc: return { "findings": [], "errors": [f"GitHub Actions acquisition failed: {exc}"], "error_class": "remote_transient", "retryable": True, "source_failure": True, "source_failure_category": "network", "source_failure_auth_related": False, "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url}, } finally: cleanup_command_work_dir(work_dir) def gitlab_api_get(url, token=None, timeout=20, stream=False): response = api_request( 'GET', url, headers=gitlab_headers(token), timeout=timeout, stream=stream, ) if token and response.status_code in (401, 403): response.close() anonymous = api_request( 'GET', url, headers=gitlab_headers(None), timeout=timeout, stream=stream, ) if anonymous.status_code < 400: return anonymous response = anonymous if response.status_code >= 400: raise gitlab_api_error(response) return response def scan_gitlab_ci_project(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, ci_pipelines_per_project=5, ci_jobs_per_pipeline=20, ci_lookback_days=30, ci_max_trace_mb=20, ci_scan_artifacts=False, ci_max_artifacts_per_pipeline=5, ci_max_artifact_archive_mb=50, ci_max_artifact_file_mb=10, ci_max_artifact_files=1000, ci_target_max_download_mb=500, fetch_timeout=20): project, project_url = parse_gitlab_project_target(target) if not project: return {"findings": [], "errors": ["Unable to parse GitLab project target"]} logger.info(f"Scanning GitLab CI traces: {project}") _raise_if_scan_slot_fatal() work_dir = create_command_work_dir() _raise_if_scan_slot_fatal() if not work_dir: return {"findings": [], "errors": ["Unable to create GitLab CI work dir"]} try: encoded = quote(project, safe='') pipeline_url = f'https://gitlab.com/api/v4/projects/{encoded}/pipelines?per_page={max(1, min(100, int(ci_pipelines_per_project or 5)))}&order_by=updated_at&sort=desc' response = gitlab_api_get(pipeline_url, token, fetch_timeout, stream=True) pipelines = bounded_response_json(response) or [] if not isinstance(pipelines, list): raise ApiRequestError('invalid GitLab pipelines payload') cutoff = datetime.now(timezone.utc) - timedelta(days=int(ci_lookback_days or 0)) if int(ci_lookback_days or 0) > 0 else None selected = [] for pipeline in pipelines: updated = parse_postman_time(pipeline.get('updated_at') or pipeline.get('created_at')) if cutoff and updated and updated < cutoff: continue selected.append(pipeline) if len(selected) >= int(ci_pipelines_per_project or 5): break if not selected: return {"findings": [], "errors": [], "skipped": "no recent pipelines", "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url}} trace_dir = os.path.join(work_dir, 'gitlab_ci', safe_path_component(project)) ensure_private_directory(trace_dir, reject_reparse=True) max_trace_bytes = int(ci_max_trace_mb or 0) * 1024 * 1024 max_artifact_archive_bytes = int(ci_max_artifact_archive_mb or 0) * 1024 * 1024 pipeline_meta = [] written = 0 artifact_extracted_total = 0 download_failures = [] downloaded_total = 0 target_download_limit = int(ci_target_max_download_mb or 0) * 1024 * 1024 download_budget = {'remaining': target_download_limit} if target_download_limit else None for pipeline in selected: _raise_if_scan_slot_fatal() if download_budget is not None and download_budget['remaining'] <= 0: download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap') break pipeline_id = pipeline.get('id') if not pipeline_id: continue jobs_url = f'https://gitlab.com/api/v4/projects/{encoded}/pipelines/{pipeline_id}/jobs?per_page={max(1, min(100, int(ci_jobs_per_pipeline or 20)))}' try: jobs_response = gitlab_api_get(jobs_url, token, fetch_timeout, stream=True) jobs = bounded_response_json(jobs_response) or [] if not isinstance(jobs, list): raise ApiRequestError('invalid GitLab jobs payload') except ScanSlotFatalError: raise except Exception as exc: logger.warning(f"Unable to list GitLab CI jobs for {project} pipeline {pipeline_id}: {exc}") download_failures.append(f'job listing failed: {exc}') jobs = [] job_meta = [] artifacts_seen = 0 for job in jobs[:max(1, int(ci_jobs_per_pipeline or 20))]: _raise_if_scan_slot_fatal() if download_budget is not None and download_budget['remaining'] <= 0: download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap') break job_id = job.get('id') if not job_id: continue artifact_meta = None trace_written = False trace_url = f'https://gitlab.com/api/v4/projects/{encoded}/jobs/{job_id}/trace' filename = f"pipeline_{pipeline_id}_job_{job_id}_{safe_path_component(job.get('name'))}.log" trace_path = os.path.join(trace_dir, filename) try: trace_download = download_to_file( trace_url, trace_path, gitlab_headers(token), fetch_timeout, ci_max_trace_mb, max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, byte_budget=download_budget, ) except requests.RequestException as exc: logger.warning(f"GitLab CI trace download failed for {project} job {job_id}: {exc}") download_failures.append(f'trace download failed: {exc}') trace_download = DownloadOutcome('', 0, 0, str(exc)) authenticated_status = trace_download.status_code if not trace_download.ok and token and trace_download.status_code in (401, 403): anonymous = download_to_file( trace_url, trace_path, gitlab_headers(None), fetch_timeout, ci_max_trace_mb, max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, byte_budget=download_budget, ) if not anonymous.ok: trace_download = DownloadOutcome( '', authenticated_status or anonymous.status_code, 0, f'authenticated HTTP {authenticated_status}; anonymous HTTP ' f'{anonymous.status_code}: {anonymous.error}', ) else: trace_download = anonymous if trace_download.ok: downloaded_total += trace_download.bytes_written written += 1 trace_written = True elif trace_download.status_code not in (404, 410): download_failures.append( f'trace download HTTP {trace_download.status_code}: {trace_download.error}' ) if ci_scan_artifacts and artifacts_seen < int(ci_max_artifacts_per_pipeline or 5): if download_budget is not None and download_budget['remaining'] <= 0: download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap') job_meta.append({'job_id': job_id, 'name': job.get('name'), 'status': job.get('status'), 'stage': job.get('stage'), 'trace': trace_written, 'artifact': None}) break artifact_file = job.get('artifacts_file') if isinstance(job.get('artifacts_file'), dict) else {} artifact_size = int(artifact_file.get('size') or 0) if artifact_file and max_artifact_archive_bytes and artifact_size and artifact_size > max_artifact_archive_bytes: download_failures.append(f'artifact declared size exceeds {ci_max_artifact_archive_mb} MB') artifact_file = {} if artifact_file and (not max_artifact_archive_bytes or not artifact_size or artifact_size <= max_artifact_archive_bytes): artifacts_url = f'https://gitlab.com/api/v4/projects/{encoded}/jobs/{job_id}/artifacts' artifact_zip = os.path.join(work_dir, f'gitlab_ci_artifact_{safe_path_component(project)}_{job_id}.zip') download = download_to_file( artifacts_url, artifact_zip, gitlab_headers(token), fetch_timeout, ci_max_artifact_archive_mb, max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, byte_budget=download_budget, ) authenticated_status = download.status_code if not download.ok and token and download.status_code in (401, 403): anonymous = download_to_file( artifacts_url, artifact_zip, gitlab_headers(None), fetch_timeout, ci_max_artifact_archive_mb, max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, byte_budget=download_budget, ) if not anonymous.ok: download = DownloadOutcome( '', authenticated_status or anonymous.status_code, 0, f'authenticated HTTP {authenticated_status}; anonymous HTTP ' f'{anonymous.status_code}: {anonymous.error}', ) else: download = anonymous if download.ok: downloaded_total += download.bytes_written artifact_dir = os.path.join(trace_dir, 'artifacts', f'pipeline_{pipeline_id}', f'job_{job_id}') ensure_private_directory(artifact_dir, reject_reparse=True) try: artifact_files = safe_extract_zip( artifact_zip, artifact_dir, ci_max_artifact_file_mb, ci_max_artifact_files, CI_ARTIFACT_ALLOWED_SUFFIXES, max_total_size_mb=ci_max_artifact_archive_mb, ) harden_private_tree(artifact_dir) except (zipfile.BadZipFile, ValueError) as exc: logger.warning(f"GitLab CI artifact rejected for {project} job {job_id}: {exc}") download_failures.append(f'artifact archive rejected: {exc}') artifact_files = 0 remove_file_quiet(artifact_zip) artifact_extracted_total += artifact_files artifacts_seen += 1 artifact_meta = { 'filename': artifact_file.get('filename'), 'size': artifact_size, 'extracted_files': artifact_files, } else: if download.status_code not in (404, 410): logger.warning(f"GitLab CI artifact download HTTP {download.status_code} for {project} job {job_id}: {download.error}") download_failures.append(f'artifact download HTTP {download.status_code}: {download.error}') job_meta.append({'job_id': job_id, 'name': job.get('name'), 'status': job.get('status'), 'stage': job.get('stage'), 'trace': trace_written, 'artifact': artifact_meta}) pipeline_meta.append({'pipeline_id': pipeline_id, 'status': pipeline.get('status'), 'ref': pipeline.get('ref'), 'updated_at': pipeline.get('updated_at'), 'jobs': job_meta}) if written == 0 and artifact_extracted_total == 0: if download_failures: text = '; '.join(download_failures[:5]) all_failures = '; '.join(download_failures) auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any( token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit') ) return { "findings": [], "errors": [text], "error_class": 'source_auth' if auth_failure else 'remote_transient', "retryable": True, "source_failure": auth_failure, "source_failure_category": 'auth_forbidden' if auth_failure else 'remote_transient', "source_failure_auth_related": auth_failure, "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url, "pipelines": pipeline_meta}, } return {"findings": [], "errors": [], "skipped": "no downloadable job traces or artifacts", "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url, "pipelines": pipeline_meta}} _raise_if_scan_slot_fatal() results = run_filesystem_scan(work_dir, timeout_sec, detectors, exclude_detectors, no_verification, trufflehog_config) results['package'] = {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url, "pipelines": pipeline_meta, "trace_files": written, "artifact_files": artifact_extracted_total} if download_failures: text = '; '.join(download_failures[:5]) all_failures = '; '.join(download_failures) auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any( token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit') ) results['errors'] = list(results.get('errors') or []) + [text] results['error_class'] = 'source_auth' if auth_failure else 'remote_transient' results['retryable'] = True results['source_failure'] = auth_failure results['source_failure_category'] = 'auth_forbidden' if auth_failure else 'remote_transient' results['source_failure_auth_related'] = auth_failure return results except ScanSlotFatalError: raise except RateLimitError as exc: category = getattr(exc, 'category', '') if category == 'not_found': return {"findings": [], "errors": [], "skipped": str(exc), "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url}} return { "findings": [], "errors": [str(exc)], "error_class": 'source_auth' if getattr(exc, 'auth_related', True) else 'remote_transient', "retryable": True, "source_failure": True, "source_failure_category": category or 'rate_limit', "source_failure_auth_related": bool(getattr(exc, 'auth_related', True)), "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url}, **({'_diagnostic_http': exc.diagnostic_http} if exc.diagnostic_http else {}), } except Exception as exc: return { "findings": [], "errors": [f"GitLab CI acquisition failed: {exc}"], "error_class": "remote_transient", "retryable": True, "source_failure": True, "source_failure_category": "network", "source_failure_auth_related": False, "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url}, } finally: cleanup_command_work_dir(work_dir) def send_webhook_notification(webhook_url, finding, target): """Send webhook notification for a finding""" if not webhook_url: return try: payload = { "text": f"Secret Found in {target}", "attachments": [{ "color": "danger", "fields": [ {"title": "Detector", "value": finding.get('DetectorName', 'Unknown'), "short": True}, {"title": "Target", "value": target, "short": True}, {"title": "Verified", "value": str(finding.get('Verified', False)), "short": True} ] }] } response = requests.post(webhook_url, json=payload, timeout=10) response.raise_for_status() logger.info(f"Webhook notification sent for {target}") except Exception as e: logger.error(f"Failed to send webhook notification: {str(e)}") def _captured_http_failure(error): current = error seen = set() for _index in range(8): if current is None or id(current) in seen: break seen.add(id(current)) direct_status = getattr(current, 'status_code', None) if type(direct_status) is int: direct_body = getattr(current, 'body', b'') result = { 'status_code': direct_status, 'content_type': getattr(current, 'content_type', None), 'request_id': getattr(current, 'request_id', None), 'operation': getattr(current, 'operation', 'provider-api'), 'headers_b64': base64.b64encode(json.dumps( getattr(current, 'headers', None) or {}, ensure_ascii=True, sort_keys=True, separators=(',', ':'), ).encode('ascii')).decode('ascii'), } result['body_capture_truncated'] = bool(getattr( current, 'body_capture_truncated', False )) if direct_body is not None: if isinstance(direct_body, str): direct_body = direct_body.encode('utf-8') if not isinstance(direct_body, bytes): direct_body = bytes(direct_body or b'') result['body_b64'] = base64.b64encode(direct_body).decode('ascii') original_size = getattr(current, 'body_original_size', None) stored_size = getattr(current, 'body_stored_size', None) body_sha256 = getattr(current, 'body_sha256', None) if original_size is None or stored_size is None or body_sha256 is None: material = make_body_material(direct_body) original_size = material.original_size stored_size = material.stored_size body_sha256 = material.sha256 result['body_capture_truncated'] = material.truncated result.update({ 'body_original_size': original_size, 'body_stored_size': stored_size, 'body_sha256': body_sha256, }) return result response = getattr(current, 'response', None) status = getattr(response, 'status_code', None) if type(status) is int: body_material = getattr(response, '_truf_diagnostic_body_material', None) if body_material is None and ( not getattr(response, 'raw', None) or getattr(response, '_content_consumed', False) ): try: body = getattr(response, 'content', b'') except RuntimeError: body = None if isinstance(body, str): body = body.encode('utf-8') if body is not None and not isinstance(body, bytes): body = bytes(body) if body is not None: body_material = make_body_material(body) headers = getattr(response, 'headers', {}) or {} result = { 'status_code': status, 'content_type': str(headers.get('Content-Type') or '') or None, 'request_id': str( headers.get('X-Request-Id') or headers.get('X-GitHub-Request-Id') or '' ) or None, 'operation': 'provider-request', 'headers_b64': base64.b64encode(json.dumps( {str(key): str(value) for key, value in headers.items()}, ensure_ascii=True, sort_keys=True, separators=(',', ':'), ).encode('ascii')).decode('ascii'), 'body_capture_truncated': False, } if body_material is not None: body = diagnostic_material_bytes(body_material) result.update({ 'body_b64': base64.b64encode(body).decode('ascii'), 'body_original_size': body_material.original_size, 'body_stored_size': body_material.stored_size, 'body_sha256': body_material.sha256, 'body_capture_truncated': body_material.truncated, }) return result current = getattr(current, '__cause__', None) or getattr( current, '__context__', None ) return None def _captured_http_body_material(payload, capture): material = make_body_material(payload) metadata_fields = ( 'body_original_size', 'body_stored_size', 'body_sha256', ) if not any(name in capture for name in metadata_fields): if capture.get('body_capture_truncated'): raise ValueError('truncated diagnostic HTTP material lacks capture metadata') return material if ( not all(name in capture for name in metadata_fields) or type(capture.get('body_capture_truncated')) is not bool or isinstance(capture['body_original_size'], bool) or not isinstance(capture['body_original_size'], int) or isinstance(capture['body_stored_size'], bool) or not isinstance(capture['body_stored_size'], int) or capture['body_stored_size'] != len(payload) or capture['body_original_size'] < capture['body_stored_size'] or capture['body_capture_truncated'] != (capture['body_original_size'] > capture['body_stored_size']) or not isinstance(capture['body_sha256'], str) or re.fullmatch(r'[a-f0-9]{64}', capture['body_sha256']) is None or ( not capture['body_capture_truncated'] and capture['body_sha256'] != hashlib.sha256(payload).hexdigest() ) ): raise ValueError('captured diagnostic HTTP material metadata is invalid') return replace( material, original_size=capture['body_original_size'], stored_size=capture['body_stored_size'], sha256=capture['body_sha256'], truncated=capture['body_capture_truncated'], ) def scan_target_result(target, scan_type, scan_event_id, scan_kwargs=None): scan_kwargs = dict(scan_kwargs or {}) scan_started_at = datetime.now(timezone.utc).isoformat() started = time.perf_counter() try: direct_kind = _client_remote_execution_kind.get() expected_direct_kind = { 'docker': 'docker_direct_v1', 'huggingface': 'huggingface_space_v1', }.get(scan_type) if ( _client_scan_manifest.get() is not None and expected_direct_kind is not None and direct_kind != expected_direct_kind ): raise RuntimeError( 'remote direct scan lacks its execution authority' ) if direct_kind is not None and direct_kind != expected_direct_kind: raise RuntimeError('remote direct execution platform changed') if scan_type in ('git', 'github', 'github_archive', 'gitlab'): provider = 'github' if scan_type == 'github_archive' else scan_type if scan_type in ('github', 'gitlab') else None result = scan_git_repo(target, provider=provider, **scan_kwargs) elif scan_type == 'docker': docker_layer_work = scan_kwargs.pop('docker_layer_work', None) if docker_layer_work is not None: scan_kwargs.pop('trufflehog_concurrency', None) scan_kwargs.pop('docker_recovery_limits', None) scan_kwargs.pop('docker_recovery_min_free_bytes', None) try: result = scan_docker_layer_plan( target, docker_layer_work, **scan_kwargs, ) except (ScanSlotFatalError, DockerLayerInfrastructureError): raise except Exception as exc: raise DockerLayerInfrastructureError( 'scanner_infrastructure', 'Docker layer scanner infrastructure failed', category='source_resource', ) from exc else: result = scan_docker_image( target, config_dir=( None if direct_kind == 'docker_direct_v1' else docker_token_manager.get_next_config() ), **scan_kwargs, ) elif scan_type == 'huggingface': result = scan_huggingface_space(target, **scan_kwargs) elif scan_type == 'npm': result = scan_npm_package(target, **scan_kwargs) elif scan_type == 'pypi': result = scan_pypi_package(target, **scan_kwargs) elif scan_type == 'package_git': result = scan_package_git_repo(target, **scan_kwargs) elif scan_type in ('postman', 'github_gists', 'github_archive_files'): result = scan_postman_target(target, **scan_kwargs) elif scan_type == 'github_actions': result = scan_github_actions_repo(target, **scan_kwargs) elif scan_type == 'gitlab_ci': result = scan_gitlab_ci_project(target, **scan_kwargs) else: result = {'findings': [], 'errors': [f'Unknown scan type: {scan_type}']} except (ScanSlotFatalError, DockerLayerInfrastructureError): raise except Exception as exc: result = {'findings': [], 'errors': [str(exc)]} http_failure = _captured_http_failure(exc) if http_failure is not None: result['_diagnostic_http'] = http_failure apply_result_error_scope(result) result['target'] = target result['scan_type'] = scan_type result['scan_event_id'] = str(scan_event_id) result['scan_started_at'] = scan_started_at result['duration_sec'] = time.perf_counter() - started result['timestamp'] = datetime.now(timezone.utc).isoformat() assign_finding_uids(result) return result @dataclass(frozen=True) class StagedResult: target: str scan_event_id: str bundle_id: str reservation_id: int scan_event_hash: str actual_bytes: int relative_path: str frame_count: int finding_count: int error_count: int candidate_count: int queue_status: str source_failure: bool source_failure_category: str source_failure_auth_related: bool first_error: str def as_dict(self): return dict(self.__dict__) def stage_result_bundle( result, reservation, bundle_root, scan_options, queue_disposition, candidate_max_items=2000, candidate_max_bytes=2 * 1024 * 1024, require_s_drive=False, fault=None, diagnostic_slot_id=0, diagnostic_attempt=1, ): reservation = reservation if isinstance(reservation, BundleReservation) else BundleReservation.from_mapping(reservation) if not isinstance(result, dict): raise ValueError('scan result must be an object') canonical_reservation_target = normalize_target( reservation.target, reservation.platform, ) if not reservation.normalized_target: reservation = replace( reservation, normalized_target=canonical_reservation_target, ) def optional_integer_identity(name, expected): if name not in result or result[name] is None: return value = result[name] if isinstance(value, bool): raise ValueError('scan result identity does not match its reservation') try: value = int(value) except (TypeError, ValueError, OverflowError): raise ValueError('scan result identity does not match its reservation') from None if value != int(expected): raise ValueError('scan result identity does not match its reservation') optional_integer_identity('reservation_id', reservation.reservation_id) optional_integer_identity('result_reservation_id', reservation.reservation_id) optional_integer_identity('queue_id', reservation.queue_id) for name, expected in ( ('bundle_id', reservation.bundle_id), ('source', reservation.source), ('platform', reservation.platform), ('query', reservation.query), ('normalized_target', reservation.normalized_target), ): if name in result and result[name] is not None and str(result[name]) != str(expected): raise ValueError('scan result identity does not match its reservation') if ( str(result.get('scan_event_id') or '') != reservation.scan_event_id or str(result.get('target') or '') != reservation.target or str(result.get('scan_type') or '') != reservation.platform or normalize_target(result.get('target'), reservation.platform) != canonical_reservation_target or reservation.normalized_target != canonical_reservation_target ): raise ValueError('scan result identity does not match its reservation') strip_nearby_context_for_persistence(result) findings = result.get('findings') or [] errors = result.get('errors') or [] candidates = [] candidate_bytes = 0 candidate_identities = set() candidate_truncated = False def add_candidate(candidate, attribution): nonlocal candidate_bytes, candidate_truncated identity = ( candidate.service, candidate.credential_hash, '' if candidate.service == 'provider_resolver' else str( (attribution or {}).get('finding_uid') or (attribution or {}).get('origin') or '' ), ) if identity in candidate_identities: return frame = candidate.as_frame(attribution) encoded_size = len(json.dumps( frame, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str, ).encode('utf-8')) if len(candidates) >= max(0, int(candidate_max_items)): candidate_truncated = True return if candidate_bytes + encoded_size > max(0, int(candidate_max_bytes)): candidate_truncated = True return candidate_identities.add(identity) candidates.append(frame) candidate_bytes += encoded_size for finding in findings: attribution = { 'finding_uid': str(finding.get('finding_uid') or ''), 'detector_name': str(finding.get('DetectorName') or finding.get('DetectorType') or ''), } try: for candidate in extract_candidates(finding, attribution): add_candidate(candidate, attribution) except ValueError as exc: result.setdefault('warnings', []).append( f'Optional keycheck candidate was omitted: {type(exc).__name__}' ) result['degraded'] = True if result.get('structured_keycheck_pending') and isinstance(result.get('postman'), dict): try: postman_data = result['postman'] cache_path, _ = validate_postman_cache_artifact( postman_data, int(result.get('postman_max_artifact_size_mb') or 20), expected_size=result.get('bytes'), ) contexts = load_postman_context(cache_path) origin = artifact_origin_label(postman_data, cache_path) attribution = {'origin': origin} for candidate in extract_structured_candidates({ 'contexts': contexts, 'origin': origin, }, attribution): structured_origin = candidate.metadata.get('structured_origin') or origin add_candidate(candidate, {'origin': structured_origin}) except Exception as exc: result.setdefault('warnings', []).append( f'Optional structured keycheck extraction failed: {type(exc).__name__}' ) result['degraded'] = True if candidate_truncated: result.setdefault('warnings', []).append( 'Keycheck candidates reached their pre-reserved per-event bound; remaining candidates were omitted' ) result['degraded'] = True status = target_status(result) first_error = '' for error in errors: first_error = next((line.strip() for line in str(error).splitlines() if line.strip()), '') if first_error: break diagnostics = list(result.get('diagnostics') or ()) if errors and not diagnostics: timestamp = str(result.get('timestamp') or result.get('scan_started_at') or '') try: occurred = datetime.fromisoformat(timestamp.replace('Z', '+00:00')) except ValueError: occurred = datetime.now(timezone.utc) if occurred.tzinfo is None: occurred = datetime.now(timezone.utc) occurred_at = occurred.astimezone(timezone.utc).isoformat( timespec='milliseconds' ).replace('+00:00', 'Z') scan_meta = result.get('scan_meta') or {} timed_out = bool(scan_meta.get('command_timed_out')) or result.get('error_class') == 'timeout' category_name = str( result.get('source_failure_category') or result.get('error_class') or '' ).lower() category = { 'source_auth': DiagnosticCategory.AUTHORIZATION, 'auth_forbidden': DiagnosticCategory.AUTHORIZATION, 'rate_limit': DiagnosticCategory.RATE_LIMIT, 'not_found': DiagnosticCategory.NOT_FOUND, 'network': DiagnosticCategory.NETWORK, 'remote_transient': DiagnosticCategory.NETWORK, 'timeout': DiagnosticCategory.TIMEOUT, 'source_resource': DiagnosticCategory.STORAGE, 'storage': DiagnosticCategory.STORAGE, }.get(category_name, DiagnosticCategory.SCANNER) phase_name = str(scan_meta.get('timed_out_phase') or WorkerPhase.SCANNING.value) try: diagnostic_phase = WorkerPhase(phase_name) except ValueError: diagnostic_phase = WorkerPhase.SCANNING scan_outcome = { 'clean': ScanOutcome.CLEAN, 'found': ScanOutcome.FOUND, 'degraded': ScanOutcome.DEGRADED, 'error': ScanOutcome.ERROR, 'skipped': ScanOutcome.SKIPPED, }.get(status, ScanOutcome.ERROR) try: raw_stdout = base64.b64decode( str(result['_diagnostic_raw_stdout_b64']).encode('ascii'), validate=True, ) if '_diagnostic_raw_stdout_b64' in result else None raw_stderr = base64.b64decode( str(result['_diagnostic_raw_stderr_b64']).encode('ascii'), validate=True, ) if '_diagnostic_raw_stderr_b64' in result else None except (UnicodeEncodeError, ValueError, binascii.Error) as exc: raise ValueError('captured diagnostic process material is invalid') from exc stdout_limit = stderr_limit = 0 if raw_stdout is not None or raw_stderr is not None: desired = [ min(len(raw_stdout), MAX_DIAGNOSTIC_LOG_BYTES) if raw_stdout is not None else 0, min(len(raw_stderr), MAX_DIAGNOSTIC_LOG_BYTES) if raw_stderr is not None else 0, ] total = sum(desired) if total <= MAX_DIAGNOSTIC_LOG_BYTES: stdout_limit, stderr_limit = desired elif total: stdout_limit = ( MAX_DIAGNOSTIC_LOG_BYTES * desired[0] ) // total stderr_limit = MAX_DIAGNOSTIC_LOG_BYTES - stdout_limit process = None if raw_stdout is not None or raw_stderr is not None: process = DiagnosticProcessContext( name='trufflehog', exit_code=( int(scan_meta['trufflehog_returncode']) if type(scan_meta.get('trufflehog_returncode')) is int else None ), signal=None, timed_out=timed_out, stdout=( make_log_material(raw_stdout, maximum=stdout_limit) if raw_stdout is not None else None ), stderr=( make_log_material(raw_stderr, maximum=stderr_limit) if raw_stderr is not None else None ), ) http_value = result.get('_diagnostic_http') http = None if isinstance(http_value, dict) and type(http_value.get('status_code')) is int: raw_body = None raw_headers = None if 'body_b64' in http_value: try: raw_body = base64.b64decode( str(http_value['body_b64']).encode('ascii'), validate=True, ) except (UnicodeEncodeError, ValueError, binascii.Error) as exc: raise ValueError('captured diagnostic HTTP material is invalid') from exc if 'headers_b64' in http_value: try: raw_headers = base64.b64decode( str(http_value['headers_b64']).encode('ascii'), validate=True, ) except (UnicodeEncodeError, ValueError, binascii.Error) as exc: raise ValueError('captured diagnostic HTTP headers are invalid') from exc http = DiagnosticHTTPContext( operation=str(http_value.get('operation') or 'provider-request'), status_code=http_value['status_code'], content_type=http_value.get('content_type'), request_id=http_value.get('request_id'), body=( _captured_http_body_material(raw_body, http_value) if raw_body is not None else None ), headers=( make_body_material(raw_headers) if raw_headers is not None else None ), ) transformation = str(result.get('_diagnostic_stderr_transformation') or ( 'HTTP status and bounded body evidence preserve captured values; parsed ' 'response headers were serialized as a deterministic mapping because raw ' 'wire order and casing are unavailable' + ( '; response body capture reached its explicit byte bound' if http_value and http_value.get('body_capture_truncated') else '' ) if http is not None else 'raw provider/process material was unavailable; canonical diagnostic ' 'was projected from the existing legacy error representation' )) fingerprint = hashlib.sha256( ('\n'.join(str(error) for error in errors)).encode('utf-8') ).hexdigest() kind = ( DiagnosticKind.PROVIDER_HTTP if http is not None else DiagnosticKind.SCANNER_PROCESS if process is not None else DiagnosticKind.EXCEPTION ) diagnostics.append(build_diagnostic_envelope( occurrence_id=f'{reservation.scan_event_id}:scan-result', reservation_id=reservation.reservation_id, scan_event_id=reservation.scan_event_id, slot_id=int(diagnostic_slot_id), source=reservation.source, phase=diagnostic_phase, kind=kind, category=(DiagnosticCategory.TIMEOUT if timed_out else category), code=('scan.stage_timeout' if timed_out else 'scan.result_error'), summary=(first_error or 'scan returned one or more errors')[:1000], retryable=bool(result.get('retryable', False)), attempt=max(1, int(diagnostic_attempt)), assignment_outcome=AssignmentOutcome.ACCEPTED, scan_outcome=scan_outcome, occurred_at=occurred_at, captured_at=occurred_at, http=http, process=process, exception=DiagnosticExceptionContext( type='truf.diagnostic.TechnicalTransformation', message=transformation, fingerprint=fingerprint, ), )) metadata = { key: value for key, value in result.items() if key not in ('findings', 'errors', 'diagnostics') and not key.startswith('_diagnostic_') } metadata.update({ 'reservation_id': reservation.reservation_id, 'queue_id': reservation.queue_id, 'bundle_id': reservation.bundle_id, 'source': reservation.source, 'platform': reservation.platform, 'query': reservation.query, 'normalized_target': reservation.normalized_target, 'status': status, 'findings_count': len(findings), 'verified_findings_count': sum(1 for finding in findings if finding.get('Verified')), 'error_count': len(errors), 'first_error_summary': first_error[:500], 'scan_options': dict(scan_options or {}), 'derived_postman_targets': list(result.get('postman_targets') or ()), **dict(queue_disposition or {}), }) with ResultBundleWriter.open( bundle_root, reservation, fault=fault, require_s_drive=require_s_drive, ) as writer: for finding in findings: writer.write_finding(finding) for error in errors: writer.write_error(error) for diagnostic in diagnostics: writer.write_diagnostic(diagnostic) for candidate in candidates: writer.write_candidate(candidate) commit = writer.finish(metadata) findings.clear() errors.clear() diagnostics.clear() candidates.clear() return StagedResult( target=str(result.get('target') or ''), scan_event_id=commit.scan_event_id, bundle_id=commit.bundle_id, reservation_id=commit.reservation_id, scan_event_hash=commit.scan_event_hash, actual_bytes=commit.actual_bytes, relative_path=commit.relative_path, frame_count=commit.frame_count, finding_count=commit.finding_count, error_count=commit.error_count, candidate_count=commit.candidate_count, queue_status=str(metadata.get('queue_status') or ''), source_failure=bool(result.get('source_failure')), source_failure_category=str(result.get('source_failure_category') or ''), source_failure_auth_related=bool(result.get('source_failure_auth_related')), first_error=first_error[:500], ) class ResultSinkError(RuntimeError): def __init__(self, failures, results): self.failures = list(failures) self.results = list(results) targets = ', '.join(str(target) for target, _ in self.failures[:5]) super().__init__(f'result sink failed for {len(self.failures)} completed target(s): {targets}') def scan_targets_batch( targets, scan_type, progress_callback=None, max_workers=4, persist_results=True, result_sink=None, scan_slot_leases=None, sink_within_scan_slot=False, **kwargs, ): """Scan multiple targets with parallel processing and progress tracking""" from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait import threading _raise_if_scan_slot_fatal() results = [] total = len(targets) if hasattr(targets, '__len__') else None completed = 0 results_lock = threading.Lock() sink_failures = [] provided_leases = list(scan_slot_leases or []) if provided_leases and len(provided_leases) != total: raise ValueError('one pre-acquired scan slot lease is required for every target') def apply_worker_sink(result): if not sink_within_scan_slot or result_sink is None: return result try: result_sink(result) result['_result_sink_applied'] = True except Exception as exc: result['_result_sink_exception'] = exc return result def scan_single_target(target, scan_event_id, provided_lease=None): """Scan a single target""" _raise_if_scan_slot_fatal() scan_started_at = datetime.now(timezone.utc).isoformat() started = time.perf_counter() try: with scan_slot_scope( ['scan-target', scan_type], kwargs.get('timeout_sec'), lease=provided_lease, ): try: _raise_if_scan_slot_fatal() result = scan_target_result(target, scan_type, scan_event_id, kwargs) return apply_worker_sink(result) except ScanSlotFatalError: raise except Exception as e: result = { "target": target, "scan_type": scan_type, "scan_event_id": scan_event_id, "scan_started_at": scan_started_at, "duration_sec": time.perf_counter() - started, "timestamp": datetime.now(timezone.utc).isoformat(), "findings": [], "errors": [str(e)] } return apply_worker_sink(apply_result_error_scope(result)) except ScanSlotFatalError: raise except Exception as e: result = { "target": target, "scan_type": scan_type, "scan_event_id": scan_event_id, "scan_started_at": scan_started_at, "duration_sec": time.perf_counter() - started, "timestamp": datetime.now(timezone.utc).isoformat(), "findings": [], "errors": [str(e)] } return apply_worker_sink(apply_result_error_scope(result)) max_workers = max(1, int(max_workers or 1)) executor = ThreadPoolExecutor(max_workers=max_workers) future_to_target = {} target_iterator = iter(enumerate(targets)) def submit_next(): try: index, target = next(target_iterator) except StopIteration: return False provided_lease = provided_leases[index] if provided_leases else None future_to_target[ executor.submit(scan_single_target, target, str(uuid.uuid4()), provided_lease) ] = target return True try: for _ in range(max_workers): _raise_if_scan_slot_fatal() if not submit_next(): break while future_to_target: _raise_if_scan_slot_fatal() done, _ = wait(tuple(future_to_target), timeout=0.2, return_when=FIRST_COMPLETED) for future in done: target = future_to_target.pop(future) result = future.result() with results_lock: results.append(result) persistence_error = result.pop('_result_sink_exception', None) sink_applied = bool(result.pop('_result_sink_applied', False)) if result_sink is not None and not sink_applied and persistence_error is None: try: result_sink(result) except Exception as exc: persistence_error = exc if persist_results: try: if not save_scan_result(result): raise RuntimeError('save_scan_result returned false') except Exception as exc: persistence_error = persistence_error or exc if persistence_error is not None: result['persistence_failure'] = str(persistence_error)[:500] sink_failures.append((target, persistence_error)) logger.error(f'Persistence failed for completed target {target}: {persistence_error}') submit_next() continue with results_lock: completed += 1 findings_count = len(result.get('findings', [])) errors_count = len(result.get('errors', [])) skipped = bool(result.get('skipped')) logger.info( f"Completed {completed}/{total}: {target} " f"(findings={findings_count}, errors={errors_count}, skipped={skipped})" ) if progress_callback: progress_callback(completed, total, target) submit_next() except ScanSlotFatalError as exc: _set_scan_slot_fatal(str(exc)) for future in future_to_target: future.cancel() wait(tuple(future_to_target), timeout=1.0) try: executor.shutdown(wait=False, cancel_futures=True) except TypeError: executor.shutdown(wait=False) executor = None raise finally: if executor is not None: executor.shutdown(wait=True) for lease in provided_leases: if lease.heartbeat_thread is None: lease.release() if progress_callback: progress_callback( completed, total, "Scan completed" if not sink_failures else "Scan completed with persistence failures", ) if sink_failures: raise ResultSinkError(sink_failures, results) return results def extract_trufflehog_error_lines(output): error_lines = [] for index, line in enumerate(_iter_output_lines(output), 1): if index > 2000: break line = line.strip() if not line: continue try: payload = json.loads(line) level = str(payload.get('level', '')).lower() message = str(payload.get('msg', '')).lower() if message == 'error cleaning temporary artifacts': continue if 'error' in level or 'error' in message or payload.get('error'): error_lines.append(line) except json.JSONDecodeError: if 'error' in line.lower() or 'failed' in line.lower(): error_lines.append(line) return error_lines