17046 lines
756 KiB
Python
17046 lines
756 KiB
Python
import os
|
|
import sys
|
|
import socket
|
|
import sqlite3
|
|
import requests
|
|
import json
|
|
import re
|
|
import subprocess
|
|
import concurrent.futures
|
|
import threading
|
|
import time
|
|
import multiprocessing
|
|
import tempfile
|
|
import shutil
|
|
import tarfile
|
|
import zipfile
|
|
import gzip
|
|
import bz2
|
|
import codecs
|
|
import lzma
|
|
import base64
|
|
import binascii
|
|
import copy
|
|
import hashlib
|
|
import io
|
|
import logging
|
|
import math
|
|
import stat
|
|
import uuid
|
|
import ctypes
|
|
import ipaddress
|
|
from email.utils import parsedate_to_datetime
|
|
from html.parser import HTMLParser
|
|
from collections import Counter
|
|
from contextlib import contextmanager
|
|
from contextvars import ContextVar
|
|
from dataclasses import dataclass, replace
|
|
from urllib.parse import parse_qsl, quote, urlencode, urljoin, urlsplit, urlunsplit
|
|
from datetime import datetime, timedelta, timezone
|
|
|
|
import urllib3.util.connection as urllib3_connection
|
|
import urllib3.exceptions as urllib3_exceptions
|
|
|
|
from paths import default_project_paths
|
|
from scanner_db import (
|
|
DOCKER_ADAPTIVE_PAYLOAD_CLASSES,
|
|
canonical_docker_layer_plan_bytes,
|
|
canonical_git_scan_plan_bytes,
|
|
normalize_target,
|
|
sanitize_endpoint,
|
|
sanitize_endpoint_host,
|
|
sanitize_postman_context,
|
|
target_status,
|
|
validate_docker_layer_plan,
|
|
validate_git_resolution,
|
|
)
|
|
from owned_process import OwnedProcess, run_owned
|
|
from process_identity import (
|
|
current_process_identity,
|
|
exact_process_identity_state,
|
|
open_process,
|
|
serialize_process_identity,
|
|
)
|
|
from janitor import JanitorBudget, bounded_remove_tree
|
|
from runtime_security import (
|
|
atomic_write_private_json,
|
|
canonical_path,
|
|
durable_replace,
|
|
durable_unlink,
|
|
ensure_private_directory,
|
|
harden_private_directory,
|
|
harden_private_file,
|
|
harden_private_tree,
|
|
is_reparse_point,
|
|
PrivateFileLock,
|
|
private_directory_ready,
|
|
private_file_ready,
|
|
read_private_json,
|
|
reject_reparse_components,
|
|
require_private_directory,
|
|
require_private_file,
|
|
sha256_file,
|
|
)
|
|
from target_identity import (
|
|
normalize_docker_digest,
|
|
normalize_huggingface_space_id,
|
|
parse_dockerhub_digest_target,
|
|
parse_docker_target,
|
|
postman_target_identity as semantic_postman_target_identity,
|
|
validate_docker_image_reference,
|
|
)
|
|
from docker_depth_experiment import (
|
|
DOCKER_DEPTH_SELECTOR_VERSION,
|
|
canonical_docker_depth_selection_evidence_hash,
|
|
canonical_selector_hash,
|
|
select_docker_layer_graphs,
|
|
validate_docker_images_per_repository,
|
|
)
|
|
from lifecycle_authority import (
|
|
CHILD_KIND_ENV,
|
|
REMOTE_WORKER_CODE_AUTHORITY_FILES,
|
|
verify_code_manifest,
|
|
require_active_supervisor_child,
|
|
resolve_manifest_executable,
|
|
strip_supervisor_credentials,
|
|
)
|
|
from keycheck_candidates import extract_candidates, extract_structured_candidates
|
|
from result_bundle import BundleReservation, ResultBundleWriter
|
|
from worker_contracts import (
|
|
AssignmentOutcome,
|
|
DiagnosticCategory,
|
|
DiagnosticExceptionContext,
|
|
DiagnosticHTTPContext,
|
|
DiagnosticKind,
|
|
DiagnosticProcessContext,
|
|
MAX_DIAGNOSTIC_BODY_BYTES,
|
|
MAX_DIAGNOSTIC_LOG_BYTES,
|
|
ScanOutcome,
|
|
WorkerPhase,
|
|
build_diagnostic_envelope,
|
|
diagnostic_material_bytes,
|
|
make_body_material,
|
|
make_log_material,
|
|
)
|
|
|
|
|
|
def configure_requests_networking():
|
|
if os.getenv('SCANNER_FORCE_IPV4', '1').strip().lower() in ('0', 'false', 'no', 'off'):
|
|
return
|
|
urllib3_connection.allowed_gai_family = lambda: socket.AF_INET
|
|
|
|
|
|
logger = logging.getLogger(__name__)
|
|
_runtime_initialized = False
|
|
_cleanup_registered = False
|
|
_client_scan_manifest = ContextVar('client_scan_manifest', default=None)
|
|
_client_scan_policy = ContextVar('client_scan_policy', default=None)
|
|
_client_remote_execution_kind = ContextVar(
|
|
'client_remote_execution_kind', default=None,
|
|
)
|
|
_client_scan_phase_callback = ContextVar(
|
|
'client_scan_phase_callback', default=None,
|
|
)
|
|
|
|
|
|
@contextmanager
|
|
def client_scan_launch_authority(manifest, expected_sha256=None):
|
|
verified = verify_code_manifest(
|
|
manifest,
|
|
expected_sha256=expected_sha256,
|
|
required_names=REMOTE_WORKER_CODE_AUTHORITY_FILES,
|
|
external_names=(),
|
|
)
|
|
token = _client_scan_manifest.set(verified)
|
|
try:
|
|
yield verified
|
|
finally:
|
|
_client_scan_manifest.reset(token)
|
|
|
|
|
|
@contextmanager
|
|
def client_scan_execution_policy(policy):
|
|
if not isinstance(policy, dict):
|
|
raise RuntimeError('remote scan execution policy is invalid')
|
|
token = _client_scan_policy.set(dict(policy))
|
|
try:
|
|
yield
|
|
finally:
|
|
_client_scan_policy.reset(token)
|
|
|
|
|
|
@contextmanager
|
|
def client_remote_execution_binding(planning_kind):
|
|
if _client_scan_manifest.get() is None:
|
|
raise RuntimeError('remote direct execution requires client launch authority')
|
|
kind = str(planning_kind or '')
|
|
if kind not in {'docker_direct_v1', 'huggingface_space_v1'}:
|
|
raise RuntimeError('remote direct execution kind is invalid')
|
|
token = _client_remote_execution_kind.set(kind)
|
|
try:
|
|
yield
|
|
finally:
|
|
_client_remote_execution_kind.reset(token)
|
|
|
|
|
|
@contextmanager
|
|
def client_scan_phase_events(callback):
|
|
if callback is not None and not callable(callback):
|
|
raise TypeError('scan phase callback must be callable')
|
|
token = _client_scan_phase_callback.set(callback)
|
|
try:
|
|
yield
|
|
finally:
|
|
_client_scan_phase_callback.reset(token)
|
|
|
|
|
|
def emit_client_scan_phase(phase, progress=None):
|
|
callback = _client_scan_phase_callback.get()
|
|
if callback is not None:
|
|
callback(phase, dict(progress or {}))
|
|
|
|
|
|
def _scan_policy_value(name, default):
|
|
policy = _client_scan_policy.get()
|
|
if policy is not None:
|
|
if name not in policy:
|
|
raise RuntimeError('remote scan execution policy is incomplete')
|
|
return policy[name]
|
|
return getattr(scan_config, name, default)
|
|
|
|
|
|
def initialize_scanner_runtime(*, preflight_complete=False, register_cleanup=True):
|
|
"""Apply process-global scanner setup only after lifecycle preflight."""
|
|
global _runtime_initialized, _cleanup_registered
|
|
if not preflight_complete:
|
|
raise RuntimeError('scanner runtime initialization requires completed lifecycle preflight')
|
|
if not _runtime_initialized:
|
|
configure_requests_networking()
|
|
logging.basicConfig(
|
|
level=logging.INFO,
|
|
format='%(asctime)s - %(levelname)s - %(message)s',
|
|
)
|
|
_runtime_initialized = True
|
|
# Stale tree ownership belongs to the isolated janitor, never an atexit hook.
|
|
_cleanup_registered = False
|
|
|
|
|
|
def require_scanner_runtime_initialized():
|
|
if not _runtime_initialized:
|
|
raise RuntimeError('scanner runtime is not initialized after lifecycle preflight')
|
|
|
|
|
|
def bool_setting(value, default=False):
|
|
if value is None:
|
|
return default
|
|
if isinstance(value, bool):
|
|
return value
|
|
return str(value).strip().lower() in ('1', 'true', 'yes', 'on')
|
|
|
|
|
|
def int_setting(value, default):
|
|
try:
|
|
return int(value)
|
|
except (TypeError, ValueError):
|
|
return default
|
|
|
|
|
|
def float_setting(value, default):
|
|
try:
|
|
return float(value)
|
|
except (TypeError, ValueError):
|
|
return default
|
|
|
|
|
|
def csv_items(value):
|
|
if not value:
|
|
return []
|
|
if isinstance(value, str):
|
|
return [item.strip() for item in value.split(',') if item.strip()]
|
|
return [str(item).strip() for item in value if str(item).strip()]
|
|
|
|
|
|
DEFAULT_DROP_DETECTORS = (
|
|
'Privacy',
|
|
'URI',
|
|
'JDBC',
|
|
'Postgres',
|
|
'MongoDB',
|
|
'SQLServer',
|
|
'Box',
|
|
'ZohoCRM',
|
|
'Accuweather',
|
|
'Roaring',
|
|
'Flatio',
|
|
'LinkPreview',
|
|
'RailwayApp',
|
|
)
|
|
|
|
|
|
class RateLimitError(Exception):
|
|
def __init__(
|
|
self, source, message, reset_at=None, category='rate_limit',
|
|
retryable=True, auth_related=True, diagnostic_http=None,
|
|
):
|
|
super().__init__(message)
|
|
self.source = source
|
|
self.reset_at = reset_at
|
|
self.category = category
|
|
self.retryable = retryable
|
|
self.auth_related = auth_related
|
|
self.diagnostic_http = (
|
|
dict(diagnostic_http) if isinstance(diagnostic_http, dict) else None
|
|
)
|
|
|
|
|
|
def response_message(response):
|
|
if response is None:
|
|
return ''
|
|
payload_bytes = bytearray()
|
|
try:
|
|
digest = hashlib.sha256()
|
|
original_size = 0
|
|
for chunk in response.iter_content(chunk_size=4096):
|
|
if not chunk:
|
|
continue
|
|
digest.update(chunk)
|
|
original_size += len(chunk)
|
|
remaining = MAX_DIAGNOSTIC_BODY_BYTES - len(payload_bytes)
|
|
if remaining > 0:
|
|
payload_bytes.extend(chunk[:remaining])
|
|
captured = bytes(payload_bytes)
|
|
material = make_body_material(captured)
|
|
if original_size > len(captured):
|
|
material = replace(
|
|
material,
|
|
original_size=original_size,
|
|
sha256=digest.hexdigest(),
|
|
truncated=True,
|
|
)
|
|
response._truf_diagnostic_body_material = material
|
|
response._truf_diagnostic_body = captured
|
|
response._truf_diagnostic_body_truncated = material.truncated
|
|
payload = json.loads(captured.decode('utf-8', errors='strict'))
|
|
if isinstance(payload, dict):
|
|
message = str(payload.get('message') or payload.get('error') or payload)
|
|
else:
|
|
message = str(payload)
|
|
return message.replace('\x00', '\\u0000')
|
|
except Exception:
|
|
captured = bytes(payload_bytes[:MAX_DIAGNOSTIC_BODY_BYTES])
|
|
response._truf_diagnostic_body = captured
|
|
return captured[:500].decode(
|
|
'utf-8', errors='replace'
|
|
).replace('\x00', '\\u0000')
|
|
|
|
def retry_after_reset(response):
|
|
if response is None:
|
|
return None
|
|
retry_after = response.headers.get('Retry-After')
|
|
if retry_after:
|
|
try:
|
|
return (datetime.now(timezone.utc) + timedelta(seconds=int(retry_after))).isoformat(timespec='seconds')
|
|
except (TypeError, ValueError):
|
|
return None
|
|
return None
|
|
|
|
def build_api_error(source, category, message, response=None, reset_at=None, retryable=True, auth_related=False):
|
|
if reset_at is None:
|
|
reset_at = retry_after_reset(response)
|
|
diagnostic_http = None
|
|
if response is not None and type(getattr(response, 'status_code', None)) is int:
|
|
headers = getattr(response, 'headers', {}) or {}
|
|
body_material = getattr(response, '_truf_diagnostic_body_material', None)
|
|
if body_material is None and (
|
|
not getattr(response, 'raw', None)
|
|
or getattr(response, '_content_consumed', False)
|
|
):
|
|
captured = getattr(response, 'content', b'')
|
|
body = captured if isinstance(captured, bytes) else str(captured).encode('utf-8')
|
|
body_material = make_body_material(body)
|
|
diagnostic_http = {
|
|
'operation': f'{source}-api',
|
|
'status_code': int(response.status_code),
|
|
'content_type': str(headers.get('Content-Type') or '') or None,
|
|
'request_id': str(
|
|
headers.get('X-Request-Id') or headers.get('X-GitHub-Request-Id') or ''
|
|
) or None,
|
|
'headers_b64': base64.b64encode(json.dumps(
|
|
{str(key): str(value) for key, value in headers.items()},
|
|
ensure_ascii=True, sort_keys=True, separators=(',', ':'),
|
|
).encode('ascii')).decode('ascii'),
|
|
}
|
|
if body_material is not None:
|
|
body = diagnostic_material_bytes(body_material)
|
|
diagnostic_http.update({
|
|
'body_b64': base64.b64encode(body).decode('ascii'),
|
|
'body_original_size': body_material.original_size,
|
|
'body_stored_size': body_material.stored_size,
|
|
'body_sha256': body_material.sha256,
|
|
'body_capture_truncated': body_material.truncated,
|
|
})
|
|
else:
|
|
diagnostic_http['body_capture_truncated'] = False
|
|
return RateLimitError(
|
|
source,
|
|
message,
|
|
reset_at=reset_at,
|
|
category=category,
|
|
retryable=retryable,
|
|
auth_related=auth_related,
|
|
diagnostic_http=diagnostic_http,
|
|
)
|
|
|
|
def github_api_error(response):
|
|
status = response.status_code if response is not None else None
|
|
message = response_message(response)
|
|
lower_message = message.lower()
|
|
reset_at = github_rate_limit_reset(response) or retry_after_reset(response)
|
|
if status == 401:
|
|
return build_api_error('github', 'auth_invalid', f'GitHub API authentication failed (HTTP 401): {message}', response, retryable=False, auth_related=True)
|
|
if status == 403:
|
|
remaining = response.headers.get('X-RateLimit-Remaining') if response is not None else None
|
|
if remaining == '0':
|
|
return build_api_error('github', 'rate_limit', f'GitHub API rate limit hit: {message}', response, reset_at, auth_related=True)
|
|
if 'secondary rate limit' in lower_message or 'abuse' in lower_message:
|
|
return build_api_error('github', 'secondary_rate_limit', f'GitHub secondary rate limit hit: {message}', response, reset_at, auth_related=True)
|
|
return build_api_error('github', 'auth_forbidden', f'GitHub API access forbidden (HTTP 403): {message}', response, retryable=False, auth_related=True)
|
|
if status == 429:
|
|
return build_api_error('github', 'rate_limit', f'GitHub API returned HTTP 429: {message}', response, reset_at, auth_related=True)
|
|
if status == 422:
|
|
return build_api_error('github', 'query_invalid', f'GitHub search query invalid (HTTP 422): {message}', response, retryable=False, auth_related=False)
|
|
if status == 404:
|
|
return build_api_error('github', 'not_found', f'GitHub API resource not found (HTTP 404): {message}', response, retryable=False, auth_related=False)
|
|
if status and status >= 500:
|
|
return build_api_error('github', 'server_error', f'GitHub API server error (HTTP {status}): {message}', response, auth_related=False)
|
|
return build_api_error('github', 'api', f'GitHub API error (HTTP {status}): {message}', response, auth_related=False)
|
|
|
|
def gitlab_api_error(response):
|
|
status = response.status_code if response is not None else None
|
|
message = response_message(response)
|
|
reset_at = gitlab_rate_limit_reset(response) or retry_after_reset(response)
|
|
if status == 401:
|
|
return build_api_error('gitlab', 'auth_invalid', f'GitLab API authentication failed (HTTP 401): {message}', response, retryable=False, auth_related=True)
|
|
if status == 403:
|
|
return build_api_error('gitlab', 'auth_forbidden', f'GitLab API access forbidden (HTTP 403): {message}', response, retryable=False, auth_related=True)
|
|
if status == 429:
|
|
return build_api_error('gitlab', 'rate_limit', f'GitLab API returned HTTP 429: {message}', response, reset_at, auth_related=True)
|
|
if status == 404:
|
|
return build_api_error('gitlab', 'not_found', f'GitLab API resource not found (HTTP 404): {message}', response, retryable=False, auth_related=False)
|
|
if status and status >= 500:
|
|
return build_api_error('gitlab', 'server_error', f'GitLab API server error (HTTP {status}): {message}', response, auth_related=False)
|
|
return build_api_error('gitlab', 'api', f'GitLab API error (HTTP {status}): {message}', response, auth_related=False)
|
|
|
|
# =====================
|
|
# GLOBAL CONFIGURATION
|
|
# =====================
|
|
class ScanConfig:
|
|
def __init__(self):
|
|
defaults = default_project_paths()
|
|
self.git_timeout = 900
|
|
self.docker_timeout = 1800
|
|
self.detectors = ""
|
|
self.exclude_detectors = os.getenv("TRUFFLEHOG_EXCLUDE_DETECTORS", "github.v1,gitlab.v1,GitHubOauth2")
|
|
self.no_verification = os.getenv("TRUFFLEHOG_NO_VERIFICATION", "0").strip().lower() in ("1", "true", "yes", "on")
|
|
self.strict_git_provider_token_filter = os.getenv(
|
|
"STRICT_GIT_PROVIDER_TOKEN_FILTER",
|
|
"1",
|
|
).strip().lower() not in ("0", "false", "no", "off")
|
|
self.drop_detectors = csv_items(os.getenv("SCANNER_DROP_DETECTORS"))
|
|
self.webhook_url = None
|
|
self.max_concurrent = min(10, max(1, multiprocessing.cpu_count() * 2))
|
|
self.trufflehog_path = os.getenv(
|
|
"TRUFFLEHOG_PATH",
|
|
defaults['trufflehog_path']
|
|
)
|
|
self.trufflehog_config = os.getenv("TRUFFLEHOG_CONFIG", "")
|
|
self.trufflehog_job_memory_limit_bytes = int_setting(
|
|
os.getenv("TRUFFLEHOG_JOB_MEMORY_LIMIT_BYTES"),
|
|
4096 * 1024 * 1024,
|
|
)
|
|
self.trufflehog_windows_job_cpu_weight = int_setting(
|
|
os.getenv("TRUFFLEHOG_WINDOWS_JOB_CPU_WEIGHT"), 0,
|
|
)
|
|
self.trufflehog_windows_memory_priority = int_setting(
|
|
os.getenv("TRUFFLEHOG_WINDOWS_MEMORY_PRIORITY"), 0,
|
|
)
|
|
self.trufflehog_stdout_max_mb = int_setting(os.getenv('TRUFFLEHOG_STDOUT_MAX_MB'), 32)
|
|
self.trufflehog_stderr_max_mb = int_setting(os.getenv('TRUFFLEHOG_STDERR_MAX_MB'), 8)
|
|
self.trufflehog_max_findings_per_target = int_setting(
|
|
os.getenv('TRUFFLEHOG_MAX_FINDINGS_PER_TARGET'), 20000,
|
|
)
|
|
self.work_dir = os.getenv("TRUFFLEHOG_WORK_DIR", defaults['work_dir'])
|
|
self.runtime_dir = defaults['runtime_dir']
|
|
self.results_dir = os.getenv("SCAN_RESULTS_DIR", defaults['results_dir'])
|
|
self.result_spool_dir = os.getenv("SCANNER_RESULT_SPOOL_DIR", defaults['result_spool_dir'])
|
|
self.result_spool_max_event_bytes = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MAX_EVENT_BYTES"), 192 * 1024 * 1024)
|
|
self.result_spool_max_events = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MAX_EVENTS"), 10000)
|
|
self.result_spool_max_total_bytes = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MAX_TOTAL_BYTES"), 2 * 1024 * 1024 * 1024)
|
|
self.result_spool_min_free_bytes = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MIN_FREE_BYTES"), 1024 * 1024 * 1024)
|
|
self.result_bundle_dir = os.getenv('SCANNER_RESULT_BUNDLE_DIR', defaults['result_bundle_dir'])
|
|
self.result_bundle_max_event_bytes = int_setting(
|
|
os.getenv('SCANNER_RESULT_BUNDLE_MAX_EVENT_BYTES'), 64 * 1024 * 1024,
|
|
)
|
|
self.scan_outbox_max_pending_items = int_setting(os.getenv('SCAN_OUTBOX_MAX_PENDING_ITEMS'), 10000)
|
|
self.scan_outbox_max_pending_bytes = int_setting(os.getenv('SCAN_OUTBOX_MAX_PENDING_BYTES'), 1024 * 1024 * 1024)
|
|
self.scan_outbox_max_pending_age_sec = int_setting(os.getenv('SCAN_OUTBOX_MAX_PENDING_AGE_SEC'), 24 * 60 * 60)
|
|
self.queue_dir = defaults['queue_dir']
|
|
self.keycheck_dir = defaults['keycheck_dir']
|
|
self.postman_cache_dir = defaults['postman_cache_dir']
|
|
self.postman_cache_max_items = int_setting(os.getenv('POSTMAN_CACHE_MAX_ITEMS'), 100000)
|
|
self.postman_cache_max_bytes = int_setting(os.getenv('POSTMAN_CACHE_MAX_BYTES'), 20 * 1024 * 1024 * 1024)
|
|
self.postman_cache_min_free_bytes = int_setting(os.getenv('POSTMAN_CACHE_MIN_FREE_BYTES'), 20 * 1024 * 1024 * 1024)
|
|
self.postman_cache_lock_timeout_sec = int_setting(os.getenv('POSTMAN_CACHE_LOCK_TIMEOUT_SEC'), 30)
|
|
self.postman_discovery_max_artifacts_per_cycle = int_setting(os.getenv('POSTMAN_DISCOVERY_MAX_ARTIFACTS_PER_CYCLE'), 1000)
|
|
self.postman_discovery_max_artifacts_per_page = int_setting(os.getenv('POSTMAN_DISCOVERY_MAX_ARTIFACTS_PER_PAGE'), 100)
|
|
self.postman_discovery_max_bytes_per_cycle = int_setting(os.getenv('POSTMAN_DISCOVERY_MAX_BYTES_PER_CYCLE'), 1024 * 1024 * 1024)
|
|
self.postman_discovery_max_elapsed_sec = float_setting(os.getenv('POSTMAN_DISCOVERY_MAX_ELAPSED_SEC'), 300.0)
|
|
self.postman_package_harvest_max_artifacts = int_setting(os.getenv('POSTMAN_PACKAGE_HARVEST_MAX_ARTIFACTS'), 100)
|
|
self.postman_package_harvest_max_bytes = int_setting(os.getenv('POSTMAN_PACKAGE_HARVEST_MAX_BYTES'), 128 * 1024 * 1024)
|
|
self.postman_package_harvest_max_elapsed_sec = float_setting(os.getenv('POSTMAN_PACKAGE_HARVEST_MAX_ELAPSED_SEC'), 30.0)
|
|
self.postman_context_max_input_bytes = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_INPUT_BYTES'), 16 * 1024 * 1024)
|
|
self.postman_context_max_nodes = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_NODES'), 100000)
|
|
self.postman_context_max_depth = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_DEPTH'), 64)
|
|
self.postman_context_max_scalar_bytes = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_SCALAR_BYTES'), 16 * 1024 * 1024)
|
|
self.postman_context_max_items = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_ITEMS'), 50000)
|
|
self.context_enrichment_max_source_bytes = int_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_SOURCE_BYTES'), 16 * 1024 * 1024)
|
|
self.context_enrichment_max_findings = int_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_FINDINGS'), 2000)
|
|
self.context_enrichment_max_postman_comparisons = int_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_POSTMAN_COMPARISONS'), 200000)
|
|
self.context_enrichment_max_elapsed_sec = float_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_ELAPSED_SEC'), 5.0)
|
|
self.trufflehog_diagnostic_max_lines = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_LINES'), 2000)
|
|
self.trufflehog_diagnostic_max_line_chars = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_LINE_CHARS'), 8192)
|
|
self.trufflehog_diagnostic_max_line_bytes = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_LINE_BYTES'), 8192)
|
|
self.trufflehog_diagnostic_max_errors = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_ERRORS'), 200)
|
|
self.trufflehog_diagnostic_max_warnings = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_WARNINGS'), 200)
|
|
self.trufflehog_diagnostic_max_unclassified = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_UNCLASSIFIED'), 20)
|
|
self.keycheck_input_max_line_bytes = int_setting(os.getenv('KEYCHECK_INPUT_MAX_LINE_BYTES'), 16 * 1024 * 1024)
|
|
self.keycheck_candidate_artifact_max_items = int_setting(os.getenv('KEYCHECK_CANDIDATE_ARTIFACT_MAX_ITEMS'), 2000)
|
|
self.keycheck_candidate_artifact_max_bytes = int_setting(os.getenv('KEYCHECK_CANDIDATE_ARTIFACT_MAX_BYTES'), 2 * 1024 * 1024)
|
|
self.keycheck_candidate_file_max_items = int_setting(os.getenv('KEYCHECK_CANDIDATE_FILE_MAX_ITEMS'), 100000)
|
|
self.keycheck_candidate_file_max_bytes = int_setting(os.getenv('KEYCHECK_CANDIDATE_FILE_MAX_BYTES'), 32 * 1024 * 1024)
|
|
self.keycheck_candidate_line_max_bytes = int_setting(os.getenv('KEYCHECK_CANDIDATE_LINE_MAX_BYTES'), 8192)
|
|
self.gharchive_cache_dir = defaults['gharchive_cache_dir']
|
|
self.gharchive_cache_max_items = int_setting(os.getenv('GHARCHIVE_CACHE_MAX_ITEMS'), 48)
|
|
self.gharchive_cache_max_bytes = int_setting(os.getenv('GHARCHIVE_CACHE_MAX_BYTES'), 8 * 1024 * 1024 * 1024)
|
|
self.gharchive_cache_min_free_bytes = int_setting(os.getenv('GHARCHIVE_CACHE_MIN_FREE_BYTES'), 5 * 1024 * 1024 * 1024)
|
|
self.gharchive_download_max_bytes = int_setting(os.getenv('GHARCHIVE_DOWNLOAD_MAX_BYTES'), 512 * 1024 * 1024)
|
|
self.gharchive_decompressed_max_bytes = int_setting(os.getenv('GHARCHIVE_DECOMPRESSED_MAX_BYTES'), 8 * 1024 * 1024 * 1024)
|
|
self.gharchive_max_events = int_setting(os.getenv('GHARCHIVE_MAX_EVENTS'), 5000000)
|
|
self.gharchive_max_line_bytes = int_setting(os.getenv('GHARCHIVE_MAX_LINE_BYTES'), 8 * 1024 * 1024)
|
|
self.gharchive_cache_lock_timeout_sec = int_setting(os.getenv('GHARCHIVE_CACHE_LOCK_TIMEOUT_SEC'), 600)
|
|
self.proxy_file = defaults['proxy_file']
|
|
self.api_proxy_enabled = bool_setting(os.getenv("SCANNER_API_PROXY_ENABLED"), False)
|
|
self.api_proxy_file = os.getenv("SCANNER_API_PROXY_FILE", self.proxy_file)
|
|
self.api_proxy_timeout = int_setting(os.getenv("SCANNER_API_PROXY_TIMEOUT"), 5)
|
|
self.api_proxy_max_retries = int_setting(os.getenv("SCANNER_API_PROXY_MAX_RETRIES"), 100)
|
|
self.api_proxy_retry_delay = int_setting(os.getenv("SCANNER_API_PROXY_RETRY_DELAY"), 5)
|
|
self.download_proxy_enabled = bool_setting(os.getenv("SCANNER_DOWNLOAD_PROXY_ENABLED"), False)
|
|
self.download_proxy_file = os.getenv("SCANNER_DOWNLOAD_PROXY_FILE", "")
|
|
self.max_active_scans = int_setting(os.getenv("SCANNER_MAX_ACTIVE_SCANS"), 0)
|
|
self.opportunistic_scan_slots = int_setting(
|
|
os.getenv("SCANNER_OPPORTUNISTIC_SCAN_SLOTS"), 0,
|
|
)
|
|
self.opportunistic_scan_sources = csv_items(
|
|
os.getenv("SCANNER_OPPORTUNISTIC_SCAN_SOURCES")
|
|
)
|
|
self.opportunistic_scan_reserve_overhead_bytes = int_setting(
|
|
os.getenv("SCANNER_OPPORTUNISTIC_SCAN_RESERVE_OVERHEAD_BYTES"),
|
|
1024 * 1024 * 1024,
|
|
)
|
|
self.opportunistic_scan_min_available_after_reserve_bytes = int_setting(
|
|
os.getenv("SCANNER_OPPORTUNISTIC_SCAN_MIN_AVAILABLE_AFTER_RESERVE_BYTES"),
|
|
4 * 1024 * 1024 * 1024,
|
|
)
|
|
self.opportunistic_scan_min_commit_after_reserve_bytes = int_setting(
|
|
os.getenv("SCANNER_OPPORTUNISTIC_SCAN_MIN_COMMIT_AFTER_RESERVE_BYTES"),
|
|
6 * 1024 * 1024 * 1024,
|
|
)
|
|
self.scan_limiter_db = os.getenv("SCANNER_SCAN_LIMITER_DB", os.path.join(defaults['state_dir'], 'scan_limiter.db'))
|
|
self.dockerhub_tag_cache_path = os.getenv("DOCKERHUB_TAG_CACHE_PATH", os.path.join(defaults['state_dir'], 'dockerhub_tag_cache.sqlite'))
|
|
self.dockerhub_tag_cache_ttl_sec = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_TTL_SEC"), 21600)
|
|
self.dockerhub_tag_negative_cache_ttl_sec = int_setting(os.getenv("DOCKERHUB_TAG_NEGATIVE_CACHE_TTL_SEC"), 3600)
|
|
self.dockerhub_tag_rate_limit_cache_ttl_sec = int_setting(os.getenv("DOCKERHUB_TAG_RATE_LIMIT_CACHE_TTL_SEC"), 1800)
|
|
self.dockerhub_tag_cache_max_rows = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MAX_ROWS"), 50000)
|
|
self.dockerhub_tag_cache_max_age_sec = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MAX_AGE_SEC"), 7 * 86400)
|
|
self.dockerhub_tag_cache_max_bytes = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MAX_BYTES"), 256 * 1024 * 1024)
|
|
self.dockerhub_tag_cache_min_free_bytes = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MIN_FREE_BYTES"), 512 * 1024 * 1024)
|
|
self.scan_slot_wait_sec = float_setting(os.getenv("SCANNER_SCAN_SLOT_WAIT_SEC"), 0.5)
|
|
self.scan_slot_wait_log_sec = int_setting(os.getenv("SCANNER_SCAN_SLOT_WAIT_LOG_SEC"), 30)
|
|
self.scan_slot_stale_sec = int_setting(os.getenv("SCANNER_SCAN_SLOT_STALE_SEC"), 7200)
|
|
self.low_space_cleanup_max_items = int_setting(os.getenv("SCANNER_LOW_SPACE_CLEANUP_MAX_ITEMS"), 50)
|
|
self.min_free_gb = float(os.getenv("TRUFFLEHOG_MIN_FREE_GB", "5"))
|
|
self.jsonl_rotation_enabled = bool_setting(os.getenv("SCANNER_JSONL_ROTATION_ENABLED"), False)
|
|
self.found_secrets_max_mb = int_setting(os.getenv("SCANNER_FOUND_SECRETS_MAX_MB"), 512)
|
|
self.scan_results_max_mb = int_setting(os.getenv("SCANNER_SCAN_RESULTS_MAX_MB"), 1024)
|
|
self.scan_errors_max_mb = int_setting(os.getenv("SCANNER_SCAN_ERRORS_MAX_MB"), 64)
|
|
self.scan_errors_keep = int_setting(os.getenv("SCANNER_SCAN_ERRORS_KEEP"), 5)
|
|
self.jsonl_lock_stale_sec = int_setting(os.getenv("SCANNER_JSONL_LOCK_STALE_SEC"), 300)
|
|
self.jsonl_max_segments = int_setting(os.getenv("SCANNER_JSONL_MAX_SEGMENTS"), 16)
|
|
self.jsonl_ledger_max_rows = int_setting(os.getenv("SCANNER_JSONL_LEDGER_MAX_ROWS"), 1000000)
|
|
self.jsonl_ledger_max_bytes = int_setting(os.getenv("SCANNER_JSONL_LEDGER_MAX_BYTES"), 512 * 1024 * 1024)
|
|
self.jsonl_legacy_index_max_bytes = int_setting(os.getenv("SCANNER_JSONL_LEGACY_INDEX_MAX_BYTES"), 16 * 1024 * 1024)
|
|
self.jsonl_tail_scan_max_bytes = int_setting(os.getenv("SCANNER_JSONL_TAIL_SCAN_MAX_BYTES"), 8 * 1024 * 1024)
|
|
self.jsonl_torn_quarantine_max_bytes = int_setting(os.getenv("SCANNER_JSONL_TORN_QUARANTINE_MAX_BYTES"), 64 * 1024)
|
|
|
|
scan_config = ScanConfig()
|
|
pending_temp_dirs = set()
|
|
pending_temp_lock = threading.Lock()
|
|
TEMP_OWNER_FILE = '.scanner-owner.json'
|
|
TEMP_OWNER_SCHEMA = 2
|
|
PENDING_TEMP_SCHEMA = 1
|
|
APPROVED_TEMP_PREFIXES = ('trufflehog-', 'trufflehog-run-', 'trufflehog-probe-', 'hg-run-', 'docker-config-', 'tmp-', 'worker-assignment-')
|
|
|
|
|
|
class ApiRequestError(Exception):
|
|
def __init__(
|
|
self, message, *, response=None, operation='provider-api',
|
|
capture_body=True,
|
|
):
|
|
super().__init__(message)
|
|
self.operation = str(operation)
|
|
self.status_code = None
|
|
self.content_type = None
|
|
self.request_id = None
|
|
self.body = None
|
|
self.body_original_size = None
|
|
self.body_stored_size = None
|
|
self.body_sha256 = None
|
|
self.body_capture_truncated = False
|
|
self.headers = None
|
|
if response is not None and type(getattr(response, 'status_code', None)) is int:
|
|
self.status_code = int(response.status_code)
|
|
headers = getattr(response, 'headers', {}) or {}
|
|
self.headers = {
|
|
str(key): str(value) for key, value in headers.items()
|
|
}
|
|
self.content_type = str(headers.get('Content-Type') or '') or None
|
|
self.request_id = str(
|
|
headers.get('X-Request-Id') or headers.get('X-GitHub-Request-Id') or ''
|
|
) or None
|
|
body_material = getattr(response, '_truf_diagnostic_body_material', None)
|
|
if body_material is not None:
|
|
self.body = diagnostic_material_bytes(body_material)
|
|
elif capture_body and (
|
|
not getattr(response, 'raw', None)
|
|
or getattr(response, '_content_consumed', False)
|
|
):
|
|
body = getattr(response, 'content', b'')
|
|
body = body if isinstance(body, bytes) else str(body).encode('utf-8')
|
|
body_material = make_body_material(body)
|
|
self.body = diagnostic_material_bytes(body_material)
|
|
if body_material is not None:
|
|
self.body_original_size = body_material.original_size
|
|
self.body_stored_size = body_material.stored_size
|
|
self.body_sha256 = body_material.sha256
|
|
self.body_capture_truncated = body_material.truncated
|
|
|
|
|
|
class GitLabDiscoveryTransportError(ApiRequestError):
|
|
pass
|
|
|
|
|
|
class DockerHubDiscoveryTransportError(ApiRequestError):
|
|
def __init__(
|
|
self, message, *, category='page_unavailable', retry_at=None,
|
|
remote_attempted=True, retryable=True,
|
|
):
|
|
super().__init__(message)
|
|
self.category = str(category or 'page_unavailable')
|
|
self.retry_at = retry_at
|
|
self.remote_attempted = bool(remote_attempted)
|
|
self.retryable = bool(retryable)
|
|
|
|
|
|
API_RETRY_STATUSES = {408, 500, 502, 503, 504}
|
|
API_RETRY_EXCEPTIONS = (
|
|
requests.exceptions.ProxyError,
|
|
requests.exceptions.ConnectionError,
|
|
requests.exceptions.ConnectTimeout,
|
|
requests.exceptions.ReadTimeout,
|
|
requests.exceptions.Timeout,
|
|
requests.exceptions.SSLError,
|
|
requests.exceptions.ChunkedEncodingError,
|
|
)
|
|
_api_proxy_lock = threading.Lock()
|
|
_api_proxy_cache_path = None
|
|
_api_proxy_cache_mtime = None
|
|
_api_proxy_cache_entries = []
|
|
_api_proxy_cache_index = 0
|
|
|
|
|
|
def _redacted_proxy_url(proxy_url):
|
|
try:
|
|
parsed = urlsplit(proxy_url)
|
|
if '@' not in parsed.netloc:
|
|
return proxy_url
|
|
host = parsed.hostname or ''
|
|
port = f':{parsed.port}' if parsed.port else ''
|
|
return urlunsplit((parsed.scheme, f'***:***@{host}{port}', parsed.path, parsed.query, parsed.fragment))
|
|
except Exception:
|
|
return '<proxy>'
|
|
|
|
|
|
def parse_proxy_line(line):
|
|
line = str(line or '').strip()
|
|
if not line or line.startswith('#'):
|
|
return None
|
|
if '://' in line:
|
|
proxy_url = line
|
|
else:
|
|
parts = line.split(':', 3)
|
|
if len(parts) == 2:
|
|
host, port = parts
|
|
proxy_url = f'http://{host}:{port}'
|
|
elif len(parts) == 4:
|
|
host, port, username, password = parts
|
|
credentials = f'{quote(username, safe="")}:{quote(password, safe="")}'
|
|
proxy_url = f'http://{credentials}@{host}:{port}'
|
|
else:
|
|
raise ValueError('expected host:port or host:port:username:password')
|
|
return {'http': proxy_url, 'https': proxy_url}
|
|
|
|
|
|
def load_proxy_entries(proxy_file):
|
|
entries = []
|
|
if not proxy_file or not os.path.exists(proxy_file):
|
|
return entries
|
|
with open(proxy_file, 'r', encoding='utf-8') as f:
|
|
for line_number, line in enumerate(f, 1):
|
|
try:
|
|
proxy = parse_proxy_line(line)
|
|
except ValueError as e:
|
|
logger.warning(f'Ignoring bad proxy line {proxy_file}:{line_number}: {e}')
|
|
continue
|
|
if proxy:
|
|
entries.append(proxy)
|
|
return entries
|
|
|
|
|
|
def next_api_proxy():
|
|
global _api_proxy_cache_path, _api_proxy_cache_mtime, _api_proxy_cache_entries, _api_proxy_cache_index
|
|
if not scan_config.api_proxy_enabled:
|
|
return None
|
|
|
|
proxy_file = scan_config.api_proxy_file or scan_config.proxy_file
|
|
try:
|
|
mtime = os.path.getmtime(proxy_file) if proxy_file else None
|
|
except OSError:
|
|
mtime = None
|
|
|
|
with _api_proxy_lock:
|
|
if proxy_file != _api_proxy_cache_path or mtime != _api_proxy_cache_mtime:
|
|
_api_proxy_cache_path = proxy_file
|
|
_api_proxy_cache_mtime = mtime
|
|
_api_proxy_cache_entries = load_proxy_entries(proxy_file)
|
|
_api_proxy_cache_index = 0
|
|
if _api_proxy_cache_entries:
|
|
logger.info(f'Loaded {len(_api_proxy_cache_entries)} API proxy entry(ies) from {proxy_file}')
|
|
if not _api_proxy_cache_entries:
|
|
raise ApiRequestError(f'API proxy is enabled but no valid proxies are loaded from {proxy_file}')
|
|
proxy = _api_proxy_cache_entries[_api_proxy_cache_index % len(_api_proxy_cache_entries)]
|
|
_api_proxy_cache_index += 1
|
|
return proxy
|
|
|
|
|
|
def _short_url(url):
|
|
try:
|
|
parsed = urlsplit(url)
|
|
return urlunsplit((parsed.scheme, parsed.netloc, parsed.path, '', ''))
|
|
except Exception:
|
|
return str(url)
|
|
|
|
|
|
def _log_api_retry(method, url, attempt, attempts, error, proxy):
|
|
if attempt != 1 and attempt % 10 != 0 and attempt != attempts:
|
|
return
|
|
proxy_url = None
|
|
if proxy:
|
|
proxy_url = proxy.get('https') or proxy.get('http')
|
|
proxy_part = f' via {_redacted_proxy_url(proxy_url)}' if proxy_url else ''
|
|
logger.warning(f'API {method} {_short_url(url)} failed ({attempt}/{attempts}){proxy_part}: {str(error)[:300]}')
|
|
|
|
|
|
def _direct_request(method, url, **kwargs):
|
|
# None removes merged proxy routes; no_proxy also blocks environment rebuilds
|
|
# on redirects. Keep Requests' certificate and streamed-response behavior.
|
|
kwargs['proxies'] = {'http': None, 'https': None, 'all': None, 'no_proxy': '*'}
|
|
return requests.request(method, url, **kwargs)
|
|
|
|
|
|
def api_request(
|
|
method, url, *, timeout=None, max_retries=None, retry_delay=None,
|
|
retry_statuses=None, deadline=None, use_proxy=None, **kwargs,
|
|
):
|
|
# False opts out; other values retain the configured discovery policy.
|
|
use_proxy = use_proxy is not False and scan_config.api_proxy_enabled
|
|
attempts = max(1, int(max_retries if max_retries is not None else (scan_config.api_proxy_max_retries if use_proxy else 1)))
|
|
delay = max(0, int(retry_delay if retry_delay is not None else scan_config.api_proxy_retry_delay))
|
|
retry_statuses = set(API_RETRY_STATUSES if retry_statuses is None else retry_statuses)
|
|
request_timeout = timeout
|
|
if use_proxy and scan_config.api_proxy_timeout is not None:
|
|
if isinstance(timeout, (tuple, list)):
|
|
connect_timeout, read_timeout = timeout
|
|
else:
|
|
connect_timeout = read_timeout = timeout
|
|
proxy_timeout = float(scan_config.api_proxy_timeout)
|
|
# Proxy connection limits must not replace the caller's read budget.
|
|
request_timeout = (
|
|
proxy_timeout if connect_timeout is None else min(proxy_timeout, float(connect_timeout)),
|
|
proxy_timeout if timeout is None else read_timeout,
|
|
)
|
|
|
|
last_error = None
|
|
for attempt in range(1, attempts + 1):
|
|
_raise_if_scan_slot_fatal()
|
|
remaining = None if deadline is None else float(deadline) - time.monotonic()
|
|
if remaining is not None and remaining <= 0:
|
|
raise ApiRequestError(f'API request deadline expired before attempt {attempt}: {method} {_short_url(url)}')
|
|
proxy = next_api_proxy() if use_proxy else None
|
|
request_kwargs = dict(kwargs)
|
|
if proxy:
|
|
request_kwargs['proxies'] = proxy
|
|
effective_timeout = request_timeout
|
|
if remaining is not None and effective_timeout is None:
|
|
effective_timeout = max(0.001, remaining)
|
|
elif remaining is not None and isinstance(effective_timeout, (int, float)):
|
|
effective_timeout = max(0.001, min(float(effective_timeout), remaining))
|
|
elif remaining is not None and isinstance(effective_timeout, (tuple, list)):
|
|
effective_timeout = tuple(
|
|
max(0.001, remaining if value is None else min(float(value), remaining))
|
|
for value in effective_timeout
|
|
)
|
|
try:
|
|
request = requests.request if use_proxy else _direct_request
|
|
response = request(method, url, timeout=effective_timeout, **request_kwargs)
|
|
if _scan_slot_fatal_event.is_set():
|
|
response.close()
|
|
_raise_if_scan_slot_fatal()
|
|
if deadline is not None and time.monotonic() >= float(deadline):
|
|
response.close()
|
|
raise ApiRequestError(f'API request deadline expired after response: {method} {_short_url(url)}')
|
|
if response.status_code in retry_statuses:
|
|
detail = ''
|
|
if not request_kwargs.get('stream'):
|
|
detail = response.text[:300] if response.text else ''
|
|
last_error = f'HTTP {response.status_code}' + (f': {detail}' if detail else '')
|
|
if attempt >= attempts:
|
|
failure = ApiRequestError(
|
|
f'API request failed after {attempts} attempt(s): '
|
|
f'{method} {_short_url(url)}: {last_error}',
|
|
response=response,
|
|
operation='provider-api-request',
|
|
capture_body=not bool(request_kwargs.get('stream')),
|
|
)
|
|
response.close()
|
|
raise failure
|
|
_log_api_retry(method, url, attempt, attempts, last_error, proxy)
|
|
response.close()
|
|
if deadline is not None and time.monotonic() + delay >= float(deadline):
|
|
raise ApiRequestError(f'API request deadline expired during retry: {method} {_short_url(url)}')
|
|
_wait_or_raise_scan_slot_fatal(delay)
|
|
continue
|
|
return response
|
|
except API_RETRY_EXCEPTIONS as e:
|
|
try:
|
|
url_has_query = bool(urlsplit(str(url)).query)
|
|
except ValueError:
|
|
url_has_query = True
|
|
safe_error = (
|
|
type(e).__name__
|
|
if request_kwargs.get('stream') or url_has_query
|
|
else str(e)[:300]
|
|
)
|
|
last_error = safe_error
|
|
_log_api_retry(method, url, attempt, attempts, safe_error, proxy)
|
|
if attempt >= attempts:
|
|
raise ApiRequestError(
|
|
f'API request failed after {attempts} attempt(s): '
|
|
f'{method} {_short_url(url)}: {safe_error}'
|
|
) from e
|
|
if deadline is not None and time.monotonic() + delay >= float(deadline):
|
|
raise ApiRequestError(f'API request deadline expired during retry: {method} {_short_url(url)}') from e
|
|
_wait_or_raise_scan_slot_fatal(delay)
|
|
raise ApiRequestError(f'API request failed after {attempts} attempt(s): {method} {_short_url(url)}: {last_error}')
|
|
|
|
|
|
SCAN_SLOT_SCHEMA = """
|
|
CREATE TABLE IF NOT EXISTS scan_slots (
|
|
slot_id TEXT PRIMARY KEY,
|
|
owner_pid INTEGER NOT NULL,
|
|
owner_thread INTEGER NOT NULL,
|
|
owner_source TEXT,
|
|
owner_creation_time TEXT,
|
|
owner_executable TEXT,
|
|
child_pid INTEGER,
|
|
child_creation_time TEXT,
|
|
child_executable TEXT,
|
|
slot_kind TEXT NOT NULL DEFAULT 'base' CHECK(slot_kind IN ('base', 'bonus')),
|
|
command TEXT,
|
|
acquired_at REAL NOT NULL,
|
|
updated_at REAL NOT NULL
|
|
);
|
|
CREATE TABLE IF NOT EXISTS scan_waiters (
|
|
waiter_id TEXT PRIMARY KEY,
|
|
owner_pid INTEGER NOT NULL,
|
|
owner_thread INTEGER NOT NULL,
|
|
owner_source TEXT NOT NULL,
|
|
owner_creation_time TEXT NOT NULL,
|
|
owner_executable TEXT NOT NULL,
|
|
enqueued_at REAL NOT NULL
|
|
);
|
|
CREATE INDEX IF NOT EXISTS idx_scan_waiters_fair
|
|
ON scan_waiters(owner_source, enqueued_at, waiter_id);
|
|
CREATE TABLE IF NOT EXISTS scan_source_fairness (
|
|
owner_source TEXT PRIMARY KEY,
|
|
last_granted_at REAL NOT NULL
|
|
);
|
|
"""
|
|
_scan_limiter_init_lock = threading.Lock()
|
|
_scan_limiter_initialized_paths = set()
|
|
_scan_slot_scope_local = threading.local()
|
|
_scan_slot_fatal_event = threading.Event()
|
|
_scan_slot_fatal_lock = threading.Lock()
|
|
_scan_slot_fatal_detail = None
|
|
|
|
|
|
class _ScanSlotScope:
|
|
def __init__(self, lease):
|
|
self.lease = lease
|
|
|
|
|
|
class ScanSlotFatalError(RuntimeError):
|
|
pass
|
|
|
|
|
|
def _set_scan_slot_fatal(detail):
|
|
global _scan_slot_fatal_detail
|
|
bounded = str(detail or 'scan-slot durability failure')[:1000]
|
|
with _scan_slot_fatal_lock:
|
|
if not _scan_slot_fatal_event.is_set():
|
|
_scan_slot_fatal_detail = bounded
|
|
_scan_slot_fatal_event.set()
|
|
|
|
|
|
def _raise_if_scan_slot_fatal():
|
|
if not _scan_slot_fatal_event.is_set():
|
|
return
|
|
with _scan_slot_fatal_lock:
|
|
detail = _scan_slot_fatal_detail
|
|
raise ScanSlotFatalError(detail or 'FATAL: scan-slot limiter is in an indeterminate state')
|
|
|
|
|
|
def _wait_or_raise_scan_slot_fatal(delay):
|
|
remaining = max(0.0, float(delay or 0))
|
|
while remaining > 0:
|
|
_raise_if_scan_slot_fatal()
|
|
interval = min(0.2, remaining)
|
|
time.sleep(interval)
|
|
remaining -= interval
|
|
_raise_if_scan_slot_fatal()
|
|
|
|
|
|
class ScanSlotLease:
|
|
DB_RETRY_ATTEMPTS = 5
|
|
RELEASE_PENDING_DB_ATTEMPTS = 2
|
|
DB_RETRY_DELAY_SEC = 0.1
|
|
MIN_HEARTBEAT_INTERVAL_SEC = 5.0
|
|
RELEASE_PENDING_INTERVAL_SEC = 1.0
|
|
HEARTBEAT_JOIN_TIMEOUT_SEC = 2.0
|
|
|
|
def __init__(self, slot_id, db_path, owner_pid=None, owner_thread=None):
|
|
self.slot_id = slot_id
|
|
self.db_path = db_path
|
|
self.owner_pid = int(os.getpid() if owner_pid is None else owner_pid)
|
|
self.owner_thread = int(threading.get_ident() if owner_thread is None else owner_thread)
|
|
self.child_pid = None
|
|
self.heartbeat_stop = threading.Event()
|
|
self.heartbeat_thread = None
|
|
self._heartbeat_wake = threading.Event()
|
|
self._state_lock = threading.Lock()
|
|
self._db_lock = threading.Lock()
|
|
self._release_call_lock = threading.Lock()
|
|
self._released = False
|
|
self._releasable = True
|
|
self._release_requested = False
|
|
self._release_pending = False
|
|
self._heartbeat_interval = 30.0
|
|
self._release_pending_interval = self.RELEASE_PENDING_INTERVAL_SEC
|
|
self._last_release_error_log_at = 0.0
|
|
|
|
@property
|
|
def released(self):
|
|
with self._state_lock:
|
|
return self._released
|
|
|
|
@property
|
|
def releasable(self):
|
|
with self._state_lock:
|
|
return self._releasable
|
|
|
|
@property
|
|
def release_pending(self):
|
|
with self._state_lock:
|
|
return self._release_pending
|
|
|
|
def _connect(self):
|
|
return sqlite3.connect(self.db_path, timeout=30)
|
|
|
|
def _close_connection(self, conn):
|
|
try:
|
|
conn.close()
|
|
except Exception as exc:
|
|
logger.warning('Unable to close scan-slot DB connection for %s: %s', self.slot_id, exc)
|
|
|
|
def _execute_update_once(self, sql, params):
|
|
conn = None
|
|
try:
|
|
conn = self._connect()
|
|
conn.execute('PRAGMA busy_timeout=30000')
|
|
cursor = conn.execute(sql, params)
|
|
conn.commit()
|
|
return cursor.rowcount != 0
|
|
finally:
|
|
if conn is not None:
|
|
self._close_connection(conn)
|
|
|
|
def _delete_slot_once(self):
|
|
conn = None
|
|
try:
|
|
conn = self._connect()
|
|
conn.execute('PRAGMA busy_timeout=30000')
|
|
cursor = conn.execute(
|
|
'''DELETE FROM scan_slots
|
|
WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''',
|
|
(self.slot_id, self.owner_pid, self.owner_thread),
|
|
)
|
|
if cursor.rowcount != 1:
|
|
existing = conn.execute(
|
|
'SELECT owner_pid, owner_thread FROM scan_slots WHERE slot_id = ?',
|
|
(self.slot_id,),
|
|
).fetchone()
|
|
if existing is not None:
|
|
raise RuntimeError(
|
|
f'scan slot identity changed from pid/thread '
|
|
f'{self.owner_pid}/{self.owner_thread} to {existing[0]}/{existing[1]}'
|
|
)
|
|
conn.commit()
|
|
return True
|
|
finally:
|
|
if conn is not None:
|
|
self._close_connection(conn)
|
|
|
|
def _retry_db_operation(self, operation, attempts):
|
|
attempts = max(1, int(attempts))
|
|
last_error = None
|
|
for attempt in range(1, attempts + 1):
|
|
try:
|
|
with self._db_lock:
|
|
return bool(operation()), None
|
|
except (sqlite3.Error, OSError) as exc:
|
|
last_error = exc
|
|
except Exception as exc:
|
|
last_error = exc
|
|
break
|
|
if attempt < attempts:
|
|
time.sleep(max(0.0, float(self.DB_RETRY_DELAY_SEC)) * attempt)
|
|
return False, last_error
|
|
|
|
def update(self, sql, params, attempts=None, log_failure=True):
|
|
success, last_error = self._retry_db_operation(
|
|
lambda: self._execute_update_once(sql, params),
|
|
self.DB_RETRY_ATTEMPTS if attempts is None else attempts,
|
|
)
|
|
if last_error is not None and log_failure:
|
|
logger.warning('Unable to update scan slot %s: %s', self.slot_id, last_error)
|
|
return success
|
|
|
|
def set_child_pid(self, child_pid):
|
|
if not child_pid:
|
|
return False
|
|
child_pid = int(child_pid)
|
|
with self._release_call_lock:
|
|
with self._state_lock:
|
|
if (
|
|
self._released or not self._releasable or self._release_requested
|
|
or not self.slot_id or not self.db_path
|
|
):
|
|
return False
|
|
identity = capture_process_identity(child_pid)
|
|
updated = self.update(
|
|
'''UPDATE scan_slots SET child_pid = ?, child_creation_time = ?, child_executable = ?, updated_at = ?
|
|
WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''',
|
|
(
|
|
child_pid,
|
|
identity.get('creation_time') if identity else None,
|
|
identity.get('executable') if identity else None,
|
|
time.time(),
|
|
self.slot_id,
|
|
self.owner_pid,
|
|
self.owner_thread,
|
|
),
|
|
)
|
|
if updated:
|
|
with self._state_lock:
|
|
self.child_pid = child_pid
|
|
return updated
|
|
|
|
def mark_non_releasable(self):
|
|
with self._release_call_lock:
|
|
with self._state_lock:
|
|
if self._released:
|
|
return
|
|
self._releasable = False
|
|
self._release_pending = False
|
|
self._heartbeat_wake.set()
|
|
|
|
def _heartbeat_update(self, attempts=None, log_failure=True):
|
|
return self.update(
|
|
'''UPDATE scan_slots SET updated_at = ?
|
|
WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''',
|
|
(time.time(), self.slot_id, self.owner_pid, self.owner_thread),
|
|
attempts=attempts,
|
|
log_failure=log_failure,
|
|
)
|
|
|
|
def start_heartbeat(self):
|
|
interval = max(
|
|
float(self.MIN_HEARTBEAT_INTERVAL_SEC),
|
|
float_setting(getattr(scan_config, 'scan_slot_heartbeat_sec', 30), 30),
|
|
)
|
|
release_interval = max(0.01, min(interval, float(self.RELEASE_PENDING_INTERVAL_SEC)))
|
|
with self._state_lock:
|
|
if self._released:
|
|
return False
|
|
if self.heartbeat_thread is not None and self.heartbeat_thread.is_alive():
|
|
return True
|
|
self._heartbeat_interval = interval
|
|
self._release_pending_interval = release_interval
|
|
thread = threading.Thread(
|
|
target=self._heartbeat_loop,
|
|
name=f'scan-slot-heartbeat-{self.slot_id[:8]}',
|
|
daemon=True,
|
|
)
|
|
self.heartbeat_thread = thread
|
|
try:
|
|
thread.start()
|
|
except Exception:
|
|
self.heartbeat_thread = None
|
|
raise
|
|
return True
|
|
|
|
def transfer_to_current_thread(self):
|
|
new_thread = int(threading.get_ident())
|
|
with self._release_call_lock:
|
|
with self._state_lock:
|
|
if self._released or self._release_requested or not self._releasable:
|
|
raise RuntimeError(f'scan slot {self.slot_id} cannot be transferred after release started')
|
|
if self.heartbeat_thread is not None:
|
|
raise RuntimeError(f'scan slot {self.slot_id} cannot be transferred after heartbeat start')
|
|
old_thread = self.owner_thread
|
|
if old_thread == new_thread:
|
|
return True
|
|
updated = self.update(
|
|
'''UPDATE scan_slots SET owner_thread = ?, updated_at = ?
|
|
WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''',
|
|
(new_thread, time.time(), self.slot_id, self.owner_pid, old_thread),
|
|
)
|
|
if not updated:
|
|
raise RuntimeError(f'scan slot {self.slot_id} ownership transfer was not confirmed')
|
|
with self._state_lock:
|
|
self.owner_thread = new_thread
|
|
return True
|
|
|
|
def _heartbeat_loop(self):
|
|
while True:
|
|
with self._state_lock:
|
|
if self._released:
|
|
return
|
|
pending = self._release_pending and self._releasable
|
|
interval = self._release_pending_interval if pending else self._heartbeat_interval
|
|
|
|
self._heartbeat_wake.wait(interval)
|
|
self._heartbeat_wake.clear()
|
|
|
|
with self._state_lock:
|
|
if self._released or self.heartbeat_stop.is_set():
|
|
return
|
|
pending = self._release_pending and self._releasable
|
|
|
|
if not pending:
|
|
self._heartbeat_update()
|
|
continue
|
|
|
|
with self._release_call_lock:
|
|
with self._state_lock:
|
|
pending = self._release_pending and self._releasable and not self._released
|
|
if not pending:
|
|
continue
|
|
success, last_error = self._retry_db_operation(
|
|
self._delete_slot_once,
|
|
self.RELEASE_PENDING_DB_ATTEMPTS,
|
|
)
|
|
if success:
|
|
completed = self._complete_release()
|
|
if completed:
|
|
logger.info('Released scan slot %s after background retry', self.slot_id)
|
|
return
|
|
|
|
self._log_pending_release_failure(last_error)
|
|
# Keep the exact owner row fresh when DELETE itself is temporarily unavailable.
|
|
self._heartbeat_update(attempts=1, log_failure=False)
|
|
|
|
def _log_pending_release_failure(self, error):
|
|
now = time.monotonic()
|
|
with self._state_lock:
|
|
if now - self._last_release_error_log_at < 30:
|
|
return
|
|
self._last_release_error_log_at = now
|
|
logger.error('Scan slot %s release remains pending and fail-closed: %s', self.slot_id, error)
|
|
|
|
def _complete_release(self):
|
|
with self._state_lock:
|
|
if self._released:
|
|
return False
|
|
self._released = True
|
|
self._release_pending = False
|
|
self.heartbeat_stop.set()
|
|
self._heartbeat_wake.set()
|
|
return True
|
|
|
|
def _join_heartbeat(self):
|
|
with self._state_lock:
|
|
thread = self.heartbeat_thread
|
|
if thread is None or thread is threading.current_thread() or not thread.is_alive():
|
|
return
|
|
thread.join(timeout=max(0.0, float(self.HEARTBEAT_JOIN_TIMEOUT_SEC)))
|
|
if thread.is_alive():
|
|
logger.warning('Scan-slot heartbeat did not stop promptly for %s', self.slot_id)
|
|
|
|
def release(self):
|
|
should_join = False
|
|
success = False
|
|
with self._release_call_lock:
|
|
with self._state_lock:
|
|
if self._released:
|
|
success = True
|
|
should_join = True
|
|
elif (
|
|
not self._releasable or self._release_requested
|
|
or not self.slot_id or not self.db_path
|
|
):
|
|
return
|
|
else:
|
|
self._release_requested = True
|
|
|
|
if not success:
|
|
success, last_error = self._retry_db_operation(
|
|
self._delete_slot_once,
|
|
self.DB_RETRY_ATTEMPTS,
|
|
)
|
|
if success:
|
|
self._complete_release()
|
|
should_join = True
|
|
else:
|
|
with self._state_lock:
|
|
pending = not self._released and self._releasable
|
|
if pending:
|
|
self._release_pending = True
|
|
self._last_release_error_log_at = time.monotonic()
|
|
if pending:
|
|
try:
|
|
self.start_heartbeat()
|
|
except Exception as exc:
|
|
logger.error('Unable to start pending-release heartbeat for scan slot %s: %s', self.slot_id, exc)
|
|
self._heartbeat_wake.set()
|
|
logger.error(
|
|
'Unable to release scan slot %s after %d attempts; '
|
|
'lease remains live and background retries will continue: %s',
|
|
self.slot_id, self.DB_RETRY_ATTEMPTS, last_error,
|
|
)
|
|
|
|
if should_join:
|
|
self._join_heartbeat()
|
|
|
|
|
|
def process_exists(pid):
|
|
try:
|
|
pid = int(pid)
|
|
except (TypeError, ValueError):
|
|
return False
|
|
if pid <= 0:
|
|
return False
|
|
if pid == os.getpid():
|
|
return True
|
|
if os.name == 'nt':
|
|
try:
|
|
import ctypes
|
|
process_query_limited_information = 0x1000
|
|
handle = ctypes.windll.kernel32.OpenProcess(process_query_limited_information, False, pid)
|
|
if handle:
|
|
ctypes.windll.kernel32.CloseHandle(handle)
|
|
return True
|
|
return False
|
|
except Exception:
|
|
return True
|
|
try:
|
|
os.kill(pid, 0)
|
|
return True
|
|
except ProcessLookupError:
|
|
return False
|
|
except PermissionError:
|
|
return True
|
|
except Exception:
|
|
return True
|
|
|
|
|
|
def capture_process_identity(pid):
|
|
try:
|
|
process = open_process(int(pid))
|
|
except Exception:
|
|
return None
|
|
try:
|
|
return {
|
|
'creation_time': str(process.identity.creation_time),
|
|
'executable': canonical_path(process.identity.executable),
|
|
}
|
|
finally:
|
|
process.close()
|
|
|
|
|
|
def exact_process_identity_live(pid, creation_time, executable):
|
|
if not pid:
|
|
return False
|
|
if not creation_time or not executable:
|
|
return None if process_exists(pid) else False
|
|
try:
|
|
process = open_process(int(pid))
|
|
except Exception:
|
|
return None if process_exists(pid) else False
|
|
try:
|
|
return bool(
|
|
process.is_running()
|
|
and str(process.identity.creation_time) == str(creation_time)
|
|
and canonical_path(process.identity.executable) == canonical_path(executable)
|
|
)
|
|
finally:
|
|
process.close()
|
|
|
|
|
|
def scan_limiter_enabled():
|
|
return int_setting(getattr(scan_config, 'max_active_scans', 0), 0) > 0
|
|
|
|
|
|
def scan_limiter_db_path():
|
|
path = getattr(scan_config, 'scan_limiter_db', '') or ''
|
|
if not path:
|
|
defaults = default_project_paths()
|
|
path = os.path.join(defaults['state_dir'], 'scan_limiter.db')
|
|
return path
|
|
|
|
|
|
def ensure_scan_limiter_db(path):
|
|
parent = os.path.dirname(path)
|
|
if parent:
|
|
os.makedirs(parent, exist_ok=True)
|
|
with _scan_limiter_init_lock:
|
|
if path in _scan_limiter_initialized_paths:
|
|
return
|
|
conn = sqlite3.connect(path, timeout=30)
|
|
try:
|
|
conn.execute('PRAGMA busy_timeout=30000')
|
|
conn.execute('PRAGMA journal_mode=WAL')
|
|
conn.executescript(SCAN_SLOT_SCHEMA)
|
|
conn.execute('BEGIN IMMEDIATE')
|
|
existing = {row[1] for row in conn.execute('PRAGMA table_info(scan_slots)').fetchall()}
|
|
if 'child_pid' not in existing:
|
|
conn.execute('ALTER TABLE scan_slots ADD COLUMN child_pid INTEGER')
|
|
for name, declaration in (
|
|
('owner_creation_time', 'TEXT'),
|
|
('owner_executable', 'TEXT'),
|
|
('child_creation_time', 'TEXT'),
|
|
('child_executable', 'TEXT'),
|
|
):
|
|
if name not in existing:
|
|
conn.execute(f'ALTER TABLE scan_slots ADD COLUMN {name} {declaration}')
|
|
if 'slot_kind' not in existing:
|
|
conn.execute(
|
|
"ALTER TABLE scan_slots ADD COLUMN slot_kind TEXT NOT NULL DEFAULT 'base'"
|
|
)
|
|
conn.execute(
|
|
"CREATE UNIQUE INDEX IF NOT EXISTS idx_scan_slots_single_bonus "
|
|
"ON scan_slots(slot_kind) WHERE slot_kind = 'bonus'"
|
|
)
|
|
conn.commit()
|
|
finally:
|
|
conn.close()
|
|
_scan_limiter_initialized_paths.add(path)
|
|
|
|
|
|
def connect_scan_limiter_db(path):
|
|
ensure_scan_limiter_db(path)
|
|
conn = sqlite3.connect(path, timeout=30)
|
|
conn.execute('PRAGMA busy_timeout=30000')
|
|
return conn
|
|
|
|
|
|
def cleanup_stale_scan_slots(conn, now, stale_sec):
|
|
columns = {row[1] for row in conn.execute('PRAGMA table_info(scan_slots)').fetchall()}
|
|
child_expr = 'child_pid' if 'child_pid' in columns else 'NULL AS child_pid'
|
|
owner_creation_expr = 'owner_creation_time' if 'owner_creation_time' in columns else 'NULL AS owner_creation_time'
|
|
owner_executable_expr = 'owner_executable' if 'owner_executable' in columns else 'NULL AS owner_executable'
|
|
child_creation_expr = 'child_creation_time' if 'child_creation_time' in columns else 'NULL AS child_creation_time'
|
|
child_executable_expr = 'child_executable' if 'child_executable' in columns else 'NULL AS child_executable'
|
|
rows = conn.execute(
|
|
f'''SELECT slot_id, owner_pid, {owner_creation_expr}, {owner_executable_expr},
|
|
{child_expr}, {child_creation_expr}, {child_executable_expr}, acquired_at, updated_at
|
|
FROM scan_slots'''
|
|
).fetchall()
|
|
for slot_id, owner_pid, owner_creation, owner_executable, child_pid, child_creation, child_executable, acquired_at, updated_at in rows:
|
|
heartbeat_age = now - float(updated_at or acquired_at or 0)
|
|
owner_live = exact_process_identity_live(owner_pid, owner_creation, owner_executable)
|
|
child_live = exact_process_identity_live(child_pid, child_creation, child_executable)
|
|
if owner_live is False and child_live is False:
|
|
conn.execute('DELETE FROM scan_slots WHERE slot_id = ?', (slot_id,))
|
|
elif heartbeat_age > stale_sec:
|
|
logger.warning(
|
|
'Stale scan-slot heartbeat remains capacity-blocking: slot=%s owner_live=%s child_live=%s age=%.0fs',
|
|
slot_id, owner_live, child_live, heartbeat_age,
|
|
)
|
|
|
|
if 'scan_waiters' in {
|
|
row[0] for row in conn.execute("SELECT name FROM sqlite_master WHERE type = 'table'").fetchall()
|
|
}:
|
|
waiters = conn.execute(
|
|
'''SELECT waiter_id, owner_pid, owner_creation_time, owner_executable
|
|
FROM scan_waiters'''
|
|
).fetchall()
|
|
for waiter_id, owner_pid, owner_creation, owner_executable in waiters:
|
|
if exact_process_identity_live(owner_pid, owner_creation, owner_executable) is False:
|
|
conn.execute('DELETE FROM scan_waiters WHERE waiter_id = ?', (waiter_id,))
|
|
|
|
|
|
def redact_scan_command_text(cmd):
|
|
text = ' '.join(str(part) for part in cmd)
|
|
text = re.sub(r'(https?://)[^\s/@:]+:[^\s/@]+@', r'\1***:***@', text)
|
|
text = re.sub(r'github_pat_[A-Za-z0-9_]+', 'github_pat_***', text)
|
|
text = re.sub(r'gh[pousr]_[A-Za-z0-9_]+', 'ghp_***', text)
|
|
text = re.sub(r'hf_[A-Za-z0-9]+', 'hf_***', text)
|
|
return text[:1000]
|
|
|
|
|
|
def windows_scan_capacity_snapshot():
|
|
if os.name != 'nt':
|
|
raise OSError('opportunistic scan capacity is supported only on Windows')
|
|
|
|
from ctypes import wintypes
|
|
|
|
class PerformanceInformation(ctypes.Structure):
|
|
_fields_ = [
|
|
('cb', wintypes.DWORD),
|
|
('CommitTotal', ctypes.c_size_t),
|
|
('CommitLimit', ctypes.c_size_t),
|
|
('CommitPeak', ctypes.c_size_t),
|
|
('PhysicalTotal', ctypes.c_size_t),
|
|
('PhysicalAvailable', ctypes.c_size_t),
|
|
('SystemCache', ctypes.c_size_t),
|
|
('KernelTotal', ctypes.c_size_t),
|
|
('KernelPaged', ctypes.c_size_t),
|
|
('KernelNonpaged', ctypes.c_size_t),
|
|
('PageSize', ctypes.c_size_t),
|
|
('HandleCount', wintypes.DWORD),
|
|
('ProcessCount', wintypes.DWORD),
|
|
('ThreadCount', wintypes.DWORD),
|
|
]
|
|
|
|
get_performance_info = ctypes.WinDLL('psapi', use_last_error=True).GetPerformanceInfo
|
|
get_performance_info.argtypes = [ctypes.POINTER(PerformanceInformation), wintypes.DWORD]
|
|
get_performance_info.restype = wintypes.BOOL
|
|
info = PerformanceInformation()
|
|
info.cb = ctypes.sizeof(info)
|
|
if not get_performance_info(ctypes.byref(info), info.cb):
|
|
raise ctypes.WinError(ctypes.get_last_error())
|
|
page_size = int(info.PageSize)
|
|
if page_size <= 0 or int(info.CommitLimit) < int(info.CommitTotal):
|
|
raise OSError('Windows returned invalid scan-capacity counters')
|
|
return {
|
|
'available_physical_bytes': int(info.PhysicalAvailable) * page_size,
|
|
'commit_headroom_bytes': (int(info.CommitLimit) - int(info.CommitTotal)) * page_size,
|
|
}
|
|
|
|
|
|
def opportunistic_scan_slot_allowed(source):
|
|
if max(0, min(1, int_setting(getattr(scan_config, 'opportunistic_scan_slots', 0), 0))) <= 0:
|
|
return False
|
|
eligible = {
|
|
item.lower() for item in csv_items(
|
|
getattr(scan_config, 'opportunistic_scan_sources', [])
|
|
)
|
|
}
|
|
if str(source or '').lower() not in eligible:
|
|
return False
|
|
try:
|
|
job_limit = int(getattr(scan_config, 'trufflehog_job_memory_limit_bytes', 0))
|
|
overhead = max(0, int(getattr(
|
|
scan_config, 'opportunistic_scan_reserve_overhead_bytes', 0,
|
|
)))
|
|
reserve = job_limit + overhead
|
|
if job_limit <= 0 or reserve <= 0:
|
|
return False
|
|
capacity = windows_scan_capacity_snapshot()
|
|
available_after = int(capacity['available_physical_bytes']) - reserve
|
|
commit_after = int(capacity['commit_headroom_bytes']) - reserve
|
|
minimum_available = max(0, int(getattr(
|
|
scan_config, 'opportunistic_scan_min_available_after_reserve_bytes', 0,
|
|
)))
|
|
minimum_commit = max(0, int(getattr(
|
|
scan_config, 'opportunistic_scan_min_commit_after_reserve_bytes', 0,
|
|
)))
|
|
return available_after >= minimum_available and commit_after >= minimum_commit
|
|
except (OSError, TypeError, ValueError, OverflowError, KeyError):
|
|
return False
|
|
|
|
|
|
def acquire_scan_slot(cmd, timeout_sec=None, wait=True, start_heartbeat=True):
|
|
_raise_if_scan_slot_fatal()
|
|
max_active = int_setting(getattr(scan_config, 'max_active_scans', 0), 0)
|
|
if max_active <= 0:
|
|
return None
|
|
|
|
db_path = scan_limiter_db_path()
|
|
wait_sec = max(0.1, float_setting(getattr(scan_config, 'scan_slot_wait_sec', 0.5), 0.5))
|
|
wait_log_sec = max(1, int_setting(getattr(scan_config, 'scan_slot_wait_log_sec', 30), 30))
|
|
stale_sec = max(
|
|
int_setting(getattr(scan_config, 'scan_slot_stale_sec', 7200), 7200),
|
|
int(timeout_sec or 0) + 300,
|
|
)
|
|
source = os.getenv('SCANNER_SOURCE') or (cmd[1] if len(cmd) > 1 else 'unknown')
|
|
command_text = redact_scan_command_text(cmd)
|
|
owner_pid = os.getpid()
|
|
owner_thread = threading.get_ident()
|
|
slot_id = f'{owner_pid}-{owner_thread}-{uuid.uuid4().hex}'
|
|
waiter_id = f'wait-{owner_pid}-{owner_thread}-{uuid.uuid4().hex}'
|
|
owner_identity = current_process_identity()
|
|
started_waiting = time.monotonic()
|
|
enqueued_at = time.time()
|
|
last_log_at = 0.0
|
|
waiter_registered = False
|
|
|
|
while True:
|
|
_raise_if_scan_slot_fatal()
|
|
now = time.time()
|
|
conn = None
|
|
slot_committed = False
|
|
try:
|
|
conn = connect_scan_limiter_db(db_path)
|
|
conn.execute('BEGIN IMMEDIATE')
|
|
_raise_if_scan_slot_fatal()
|
|
cleanup_stale_scan_slots(conn, now, stale_sec)
|
|
conn.execute(
|
|
'''INSERT OR IGNORE INTO scan_waiters(
|
|
waiter_id, owner_pid, owner_thread, owner_source,
|
|
owner_creation_time, owner_executable, enqueued_at
|
|
) VALUES (?, ?, ?, ?, ?, ?, ?)''',
|
|
(
|
|
waiter_id, owner_pid, owner_thread, str(source),
|
|
owner_identity.creation_time, canonical_path(owner_identity.executable),
|
|
enqueued_at,
|
|
),
|
|
)
|
|
waiter_registered = True
|
|
base_active, bonus_active = conn.execute(
|
|
"SELECT "
|
|
"SUM(CASE WHEN slot_kind = 'base' THEN 1 ELSE 0 END), "
|
|
"SUM(CASE WHEN slot_kind = 'bonus' THEN 1 ELSE 0 END) "
|
|
"FROM scan_slots"
|
|
).fetchone()
|
|
base_active = int(base_active or 0)
|
|
bonus_active = int(bonus_active or 0)
|
|
active = base_active + bonus_active
|
|
next_waiter = conn.execute(
|
|
'''SELECT w.waiter_id
|
|
FROM scan_waiters w
|
|
LEFT JOIN scan_source_fairness f ON f.owner_source = w.owner_source
|
|
ORDER BY COALESCE(f.last_granted_at, 0), w.enqueued_at, w.waiter_id
|
|
LIMIT 1'''
|
|
).fetchone()
|
|
slot_kind = None
|
|
if next_waiter and next_waiter[0] == waiter_id:
|
|
if base_active < max_active:
|
|
slot_kind = 'base'
|
|
elif bonus_active < max(0, min(1, int_setting(
|
|
getattr(scan_config, 'opportunistic_scan_slots', 0), 0,
|
|
))) and opportunistic_scan_slot_allowed(source):
|
|
slot_kind = 'bonus'
|
|
if slot_kind is not None:
|
|
conn.execute(
|
|
'''INSERT INTO scan_slots(
|
|
slot_id, owner_pid, owner_thread, owner_source, owner_creation_time,
|
|
owner_executable, slot_kind, command, acquired_at, updated_at
|
|
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''',
|
|
(
|
|
slot_id, owner_pid, owner_thread, str(source),
|
|
owner_identity.creation_time, canonical_path(owner_identity.executable),
|
|
slot_kind, command_text, now, now,
|
|
),
|
|
)
|
|
conn.execute('DELETE FROM scan_waiters WHERE waiter_id = ?', (waiter_id,))
|
|
conn.execute(
|
|
'''INSERT INTO scan_source_fairness(owner_source, last_granted_at)
|
|
VALUES (?, ?)
|
|
ON CONFLICT(owner_source) DO UPDATE SET last_granted_at = excluded.last_granted_at''',
|
|
(str(source), now),
|
|
)
|
|
conn.commit()
|
|
slot_committed = True
|
|
waiter_registered = False
|
|
waited = time.monotonic() - started_waiting
|
|
if waited >= wait_log_sec:
|
|
hard_limit = max_active + max(0, min(1, int_setting(
|
|
getattr(scan_config, 'opportunistic_scan_slots', 0), 0,
|
|
)))
|
|
logger.info(
|
|
f'Acquired {slot_kind} scan slot after waiting {waited:.0f}s '
|
|
f'({active + 1}/{hard_limit})'
|
|
)
|
|
lease = ScanSlotLease(slot_id, db_path, owner_pid=owner_pid, owner_thread=owner_thread)
|
|
if start_heartbeat:
|
|
try:
|
|
if not lease.start_heartbeat():
|
|
raise RuntimeError('scan-slot heartbeat did not start')
|
|
except BaseException as start_error:
|
|
logger.error(
|
|
'Scan-slot heartbeat failed to start for %s; synchronously removing the exact owner row: %s',
|
|
slot_id, start_error,
|
|
)
|
|
lease.release()
|
|
if not lease.released:
|
|
fatal_detail = (
|
|
'FATAL: scan-slot heartbeat startup failed and exact-owner rollback '
|
|
f'could not be confirmed for slot {slot_id}'
|
|
)
|
|
_set_scan_slot_fatal(fatal_detail)
|
|
logger.critical(
|
|
'FATAL scan-slot acquisition rollback is unconfirmed for %s; capacity remains fail-closed',
|
|
slot_id,
|
|
)
|
|
raise ScanSlotFatalError(fatal_detail) from start_error
|
|
raise
|
|
return lease
|
|
if not wait:
|
|
conn.execute('DELETE FROM scan_waiters WHERE waiter_id = ?', (waiter_id,))
|
|
conn.commit()
|
|
waiter_registered = False
|
|
return None
|
|
conn.commit()
|
|
if time.monotonic() - last_log_at >= wait_log_sec:
|
|
logger.info(f'Waiting for scan slot ({active}/{max_active} active)')
|
|
last_log_at = time.monotonic()
|
|
except sqlite3.OperationalError as e:
|
|
if slot_committed:
|
|
raise
|
|
if not wait:
|
|
return None
|
|
if time.monotonic() - last_log_at >= wait_log_sec:
|
|
logger.warning(f'Waiting for scan limiter DB lock: {str(e)}')
|
|
last_log_at = time.monotonic()
|
|
finally:
|
|
if conn is not None:
|
|
conn.close()
|
|
if _scan_slot_fatal_event.wait(wait_sec):
|
|
_raise_if_scan_slot_fatal()
|
|
|
|
|
|
def acquire_scan_slot_leases(cmd, count, timeout_sec=None):
|
|
count = max(0, int(count or 0))
|
|
if count <= 0 or not scan_limiter_enabled():
|
|
return []
|
|
leases = []
|
|
try:
|
|
first = acquire_scan_slot(cmd, timeout_sec, wait=True, start_heartbeat=False)
|
|
if first is not None:
|
|
leases.append(first)
|
|
while len(leases) < count:
|
|
lease = acquire_scan_slot(cmd, timeout_sec, wait=False, start_heartbeat=False)
|
|
if lease is None:
|
|
break
|
|
leases.append(lease)
|
|
return leases
|
|
except BaseException:
|
|
for lease in leases:
|
|
lease.release()
|
|
raise
|
|
|
|
|
|
@contextmanager
|
|
def scan_slot_scope(cmd, timeout_sec=None, lease=None):
|
|
"""Own one physical lease through the target's durable bundle handoff."""
|
|
if getattr(_scan_slot_scope_local, 'scope', None) is not None:
|
|
raise RuntimeError('scan slot scopes cannot be nested on one worker thread')
|
|
if lease is None:
|
|
lease = acquire_scan_slot(cmd, timeout_sec)
|
|
else:
|
|
try:
|
|
lease.transfer_to_current_thread()
|
|
if not lease.start_heartbeat():
|
|
raise RuntimeError('transferred scan-slot heartbeat did not start')
|
|
except BaseException:
|
|
lease.release()
|
|
raise
|
|
scope = _ScanSlotScope(lease)
|
|
_scan_slot_scope_local.scope = scope
|
|
try:
|
|
yield lease
|
|
finally:
|
|
try:
|
|
if lease and lease.releasable:
|
|
lease.release()
|
|
finally:
|
|
if getattr(_scan_slot_scope_local, 'scope', None) is scope:
|
|
del _scan_slot_scope_local.scope
|
|
|
|
|
|
def scoped_scan_slot_lease():
|
|
scope = getattr(_scan_slot_scope_local, 'scope', None)
|
|
return (scope is not None, scope.lease if scope is not None else None)
|
|
|
|
def get_pending_temp_file():
|
|
work_dir = get_work_dir()
|
|
if not work_dir:
|
|
return None
|
|
return os.path.join(work_dir, 'pending_cleanup.json')
|
|
|
|
|
|
def get_pending_temp_lock_file():
|
|
path = get_pending_temp_file()
|
|
return path + '.lock' if path else None
|
|
|
|
|
|
def _load_persisted_pending_temp_dirs_unlocked(path):
|
|
if not path or not os.path.exists(path):
|
|
return set()
|
|
try:
|
|
value = read_private_json(path)
|
|
except OSError:
|
|
return set()
|
|
if value.get('schema') != PENDING_TEMP_SCHEMA or not isinstance(value.get('paths'), list):
|
|
return set()
|
|
return {str(item) for item in value['paths'] if isinstance(item, str) and item.strip()}
|
|
|
|
|
|
def _persist_pending_temp_dirs_unlocked(path, paths):
|
|
normalized = sorted({str(item) for item in paths if str(item).strip()})
|
|
if normalized:
|
|
atomic_write_private_json(path, {'schema': PENDING_TEMP_SCHEMA, 'paths': normalized})
|
|
elif os.path.exists(path):
|
|
if not private_file_ready(path):
|
|
raise OSError(f'refusing to remove non-private pending cleanup list: {path}')
|
|
durable_unlink(path)
|
|
|
|
def load_persisted_pending_temp_dirs():
|
|
path = get_pending_temp_file()
|
|
lock_path = get_pending_temp_lock_file()
|
|
if not path or not lock_path:
|
|
return set()
|
|
with PrivateFileLock(lock_path):
|
|
return _load_persisted_pending_temp_dirs_unlocked(path)
|
|
|
|
def persist_pending_temp_dirs(paths):
|
|
path = get_pending_temp_file()
|
|
lock_path = get_pending_temp_lock_file()
|
|
if not path or not lock_path:
|
|
return
|
|
try:
|
|
with PrivateFileLock(lock_path):
|
|
_persist_pending_temp_dirs_unlocked(path, paths)
|
|
except OSError:
|
|
pass
|
|
|
|
def redact_secrets(text, secrets):
|
|
if not text:
|
|
return text
|
|
|
|
redacted = text
|
|
for secret in secrets:
|
|
if secret:
|
|
redacted = redacted.replace(secret, '***REDACTED***')
|
|
redacted = redacted.replace(quote(secret, safe=''), '***REDACTED***')
|
|
return redacted
|
|
|
|
def build_authenticated_git_url(repo_url, provider=None, token=None):
|
|
if not token:
|
|
return repo_url, []
|
|
|
|
parsed = urlsplit(repo_url)
|
|
if parsed.scheme != 'https' or not parsed.netloc:
|
|
return repo_url, []
|
|
|
|
hostname = (parsed.hostname or '').lower()
|
|
detected_provider = 'gitlab' if hostname == 'gitlab.com' else 'github' if hostname == 'github.com' else None
|
|
if provider and provider != detected_provider:
|
|
return repo_url, []
|
|
provider = detected_provider
|
|
if provider not in ('github', 'gitlab'):
|
|
return repo_url, []
|
|
|
|
return repo_url, [token]
|
|
|
|
def get_git_provider_and_path(repo_url, provider=None):
|
|
parsed = urlsplit(repo_url)
|
|
hostname = (parsed.hostname or '').lower()
|
|
path = parsed.path.strip('/')
|
|
if path.endswith('.git'):
|
|
path = path[:-4]
|
|
provider = provider or ('gitlab' if 'gitlab.' in hostname or hostname == 'gitlab.com' else 'github' if 'github.' in hostname or hostname == 'github.com' else None)
|
|
return provider, path
|
|
|
|
def recent_commit_boundary(repo_url, provider=None, token=None, max_age_days=None, lookup_pages=3):
|
|
if not max_age_days or max_age_days <= 0:
|
|
return {'since_commit': None, 'skip': False, 'reason': ''}
|
|
|
|
provider, repo_path = get_git_provider_and_path(repo_url, provider)
|
|
if provider not in ('github', 'gitlab') or not repo_path:
|
|
return {'since_commit': None, 'skip': True, 'reason': 'unsupported provider for commit age lookup'}
|
|
|
|
cutoff = datetime.now(timezone.utc) - timedelta(days=max_age_days)
|
|
since = cutoff.isoformat().replace('+00:00', 'Z')
|
|
headers = {'User-Agent': 'GitSecretsScanner/2.0'}
|
|
if token:
|
|
headers['Authorization'] = f'Bearer {token}'
|
|
|
|
commits = []
|
|
lookup_cap_reached = False
|
|
for page in range(1, max(1, lookup_pages) + 1):
|
|
try:
|
|
if provider == 'github':
|
|
url = f'https://api.github.com/repos/{repo_path}/commits'
|
|
params = {'since': since, 'per_page': 100, 'page': page}
|
|
else:
|
|
url = f'https://gitlab.com/api/v4/projects/{quote(repo_path, safe="")}/repository/commits'
|
|
params = {'since': since, 'per_page': 100, 'page': page}
|
|
response = api_request('GET', url, headers=headers, params=params, timeout=30)
|
|
if token and response.status_code in (401, 403):
|
|
anonymous = api_request(
|
|
'GET', url, headers={'User-Agent': 'GitSecretsScanner/2.0'},
|
|
params=params, timeout=30,
|
|
)
|
|
if anonymous.status_code < 400:
|
|
response = anonymous
|
|
response.raise_for_status()
|
|
page_commits = response.json()
|
|
if not page_commits:
|
|
break
|
|
commits.extend(page_commits)
|
|
if len(page_commits) < 100:
|
|
break
|
|
if page == max(1, lookup_pages):
|
|
lookup_cap_reached = True
|
|
except requests.exceptions.HTTPError as e:
|
|
api_error = github_api_error(e.response) if provider == 'github' else gitlab_api_error(e.response)
|
|
if api_error.category == 'not_found':
|
|
return {
|
|
'since_commit': None,
|
|
'skip': True,
|
|
'permanent': True,
|
|
'reason': f'commit age lookup found no repository: {str(api_error)[:300]}',
|
|
'error_category': api_error.category,
|
|
'auth_related': False,
|
|
}
|
|
return {
|
|
'since_commit': None,
|
|
'skip': False,
|
|
'error': True,
|
|
'reason': f'commit age lookup failed ({api_error.category}): {str(api_error)[:300]}',
|
|
'error_category': api_error.category,
|
|
'auth_related': bool(getattr(api_error, 'auth_related', True)),
|
|
}
|
|
except Exception as e:
|
|
if isinstance(e, ApiRequestError):
|
|
return {
|
|
'since_commit': None, 'skip': False, 'error': True,
|
|
'reason': f'commit age lookup transport failed: {str(e)[:300]}',
|
|
'error_category': 'network', 'auth_related': False,
|
|
}
|
|
return {
|
|
'since_commit': None,
|
|
'skip': False,
|
|
'error': True,
|
|
'reason': f'commit age lookup failed: {str(e)[:300]}',
|
|
'error_category': 'unknown',
|
|
'auth_related': False,
|
|
}
|
|
|
|
if not commits:
|
|
return {
|
|
'since_commit': None,
|
|
'skip': True,
|
|
'reason': f'no commits newer than {max_age_days} days'
|
|
}
|
|
if lookup_cap_reached:
|
|
return {
|
|
'since_commit': None,
|
|
'skip': False,
|
|
'reason': 'commit lookup cap reached; scanning without since-commit boundary',
|
|
'recent_commit_count': len(commits),
|
|
'cutoff': since,
|
|
}
|
|
|
|
oldest = commits[-1]
|
|
if provider == 'github':
|
|
parents = oldest.get('parents') or []
|
|
parent_sha = parents[0].get('sha') if parents else None
|
|
oldest_sha = oldest.get('sha')
|
|
else:
|
|
parents = oldest.get('parent_ids') or []
|
|
parent_sha = parents[0] if parents else None
|
|
oldest_sha = oldest.get('id')
|
|
|
|
return {
|
|
'since_commit': parent_sha,
|
|
'skip': False,
|
|
'reason': '',
|
|
'recent_commit_count': len(commits),
|
|
'cutoff': since,
|
|
'boundary_commit': oldest_sha,
|
|
}
|
|
|
|
def get_trufflehog_cmd():
|
|
"""Return configured TruffleHog executable path."""
|
|
return scan_config.trufflehog_path or "trufflehog"
|
|
|
|
|
|
def require_trufflehog_launch_authority(command=None):
|
|
child_kind = str(os.getenv(CHILD_KIND_ENV) or 'scanner').strip().lower()
|
|
if child_kind not in {'scanner', 'docker-shadow'}:
|
|
raise RuntimeError('TruffleHog launch requires scanner or Docker shadow authority')
|
|
manifest = _client_scan_manifest.get()
|
|
metadata = {'code_manifest': manifest, 'authority': 'remote-worker'} if manifest else None
|
|
if metadata is None:
|
|
metadata = require_active_supervisor_child(child_kind=child_kind, require_dsn=True)
|
|
manifest = metadata.get('code_manifest') or {}
|
|
expected = (manifest.get('executables') or {}).get('trufflehog') or {}
|
|
candidate = resolve_manifest_executable((command or [get_trufflehog_cmd()])[0])
|
|
if canonical_path(candidate) != canonical_path(expected.get('path') or ''):
|
|
raise RuntimeError('TruffleHog command does not match immutable supervisor authority')
|
|
values = list(command or [])
|
|
if '--config' in values:
|
|
try:
|
|
policy_path = canonical_path(values[values.index('--config') + 1])
|
|
except (IndexError, TypeError, ValueError) as exc:
|
|
raise RuntimeError('TruffleHog policy argument is incomplete') from exc
|
|
assets = manifest.get('assets') or {}
|
|
if policy_path not in {canonical_path(item.get('path') or '') for item in assets.values() if isinstance(item, dict)}:
|
|
raise RuntimeError('TruffleHog policy does not match immutable supervisor authority')
|
|
return metadata
|
|
|
|
def get_git_cmd():
|
|
"""Return manifested Git in runtime; uninitialized tests may resolve PATH."""
|
|
manifest = _client_scan_manifest.get()
|
|
if manifest:
|
|
return str(((manifest.get('executables') or {}).get('git') or {}).get('path') or '')
|
|
if not _runtime_initialized:
|
|
return shutil.which('git') or 'git'
|
|
if str(os.getenv(CHILD_KIND_ENV) or 'scanner').strip().lower() != 'scanner':
|
|
raise RuntimeError('Git clone launch requires scanner authority')
|
|
manifest = _client_scan_manifest.get()
|
|
if manifest:
|
|
metadata = {'code_manifest': manifest, 'authority': 'remote-worker'}
|
|
else:
|
|
metadata = require_active_supervisor_child(child_kind='scanner', require_dsn=True)
|
|
manifest = metadata.get('code_manifest') or {}
|
|
expected = (manifest.get('executables') or {}).get('git') or {}
|
|
path = expected.get('path')
|
|
if not isinstance(path, str) or not os.path.isabs(path):
|
|
raise RuntimeError('Git executable is absent from immutable supervisor authority')
|
|
return path
|
|
|
|
|
|
def prepend_client_git_environment(env):
|
|
manifest = _client_scan_manifest.get()
|
|
if manifest is None:
|
|
return env
|
|
path = str(((manifest.get('executables') or {}).get('git') or {}).get('path') or '')
|
|
if not os.path.isabs(path):
|
|
raise RuntimeError('remote worker Git executable is absent from immutable authority')
|
|
directory = os.path.dirname(path)
|
|
env['PATH'] = os.pathsep.join((directory, env.get('PATH', '')))
|
|
return env
|
|
|
|
|
|
def require_git_clone_launch_authority(cmd):
|
|
"""Authorize only checkout-free HTTPS clones, never general Git commands."""
|
|
if (
|
|
not isinstance(cmd, (list, tuple)) or len(cmd) != 7
|
|
or any(not isinstance(value, str) or not value or any(ord(ch) < 32 or ord(ch) == 127 for ch in value) for value in cmd)
|
|
or list(cmd[1:5]) != ['clone', '--no-checkout', '--no-recurse-submodules', '--']
|
|
):
|
|
raise RuntimeError('Git clone command does not match the allowed argv contract')
|
|
source, destination = cmd[5:]
|
|
try:
|
|
parsed = urlsplit(source)
|
|
valid_source = (
|
|
source.startswith('https://') and bool(parsed.hostname)
|
|
and parsed.username is None and parsed.password is None
|
|
and bool(parsed.path) and parsed.path.startswith('/')
|
|
and not parsed.query and not parsed.fragment
|
|
and not any(ch.isspace() for ch in source) and '\\' not in source
|
|
and '%' not in parsed.netloc and parsed.port != 0
|
|
)
|
|
except ValueError:
|
|
valid_source = False
|
|
if not valid_source:
|
|
raise RuntimeError('Git clone source must be credential-free absolute HTTPS without query or fragment')
|
|
if (
|
|
not os.path.isabs(destination) or destination.startswith('-')
|
|
or (os.name == 'nt' and not os.path.splitdrive(destination)[0])
|
|
):
|
|
raise RuntimeError('Git clone destination must be an absolute path')
|
|
if not os.path.isabs(cmd[0]):
|
|
raise RuntimeError('Git command does not match immutable supervisor authority')
|
|
if str(os.getenv(CHILD_KIND_ENV) or 'scanner').strip().lower() != 'scanner':
|
|
raise RuntimeError('Git clone launch requires scanner authority')
|
|
manifest = _client_scan_manifest.get()
|
|
if manifest:
|
|
metadata = {'code_manifest': manifest, 'authority': 'remote-worker'}
|
|
else:
|
|
metadata = require_active_supervisor_child(child_kind='scanner', require_dsn=True)
|
|
manifest = metadata.get('code_manifest') or {}
|
|
expected = (manifest.get('executables') or {}).get('git') or {}
|
|
# Authentication rehashes manifest contents, not just path/mtime identity.
|
|
if cmd[0] != expected.get('path'):
|
|
raise RuntimeError('Git command does not match immutable supervisor authority')
|
|
return metadata
|
|
|
|
|
|
def get_trufflehog_config(config_path=None):
|
|
"""Return configured TruffleHog custom detector config path, if any."""
|
|
return str(config_path if config_path is not None else getattr(scan_config, 'trufflehog_config', '') or '').strip()
|
|
|
|
def append_trufflehog_scan_args(cmd, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None):
|
|
"""Append common TruffleHog scan flags in one place."""
|
|
config_path = get_trufflehog_config(trufflehog_config)
|
|
if config_path:
|
|
cmd.extend(['--config', config_path])
|
|
if detectors:
|
|
cmd.extend(['--include-detectors', detectors])
|
|
if exclude_detectors:
|
|
cmd.extend(['--exclude-detectors', exclude_detectors])
|
|
if no_verification:
|
|
cmd.append('--no-verification')
|
|
return cmd
|
|
|
|
def get_work_dir():
|
|
"""Prepare and return the directory used for TruffleHog temporary data."""
|
|
require_scanner_runtime_initialized()
|
|
if not scan_config.work_dir:
|
|
raise RuntimeError('TruffleHog work_dir is required')
|
|
try:
|
|
return require_private_directory(scan_config.work_dir, create=False)
|
|
except OSError as exc:
|
|
raise RuntimeError(f'Unable to use private TruffleHog work_dir {scan_config.work_dir}: {exc}') from exc
|
|
|
|
def create_command_work_dir():
|
|
"""Create an isolated temporary directory for one TruffleHog subprocess."""
|
|
work_dir = get_work_dir()
|
|
ensure_work_dir_space()
|
|
path = tempfile.mkdtemp(prefix='trufflehog-run-', dir=work_dir)
|
|
try:
|
|
harden_private_directory(path)
|
|
if not write_temp_owner(path, ['scanner-workdir'], os.getpid(), required=True):
|
|
raise RuntimeError(f'Unable to write required temp owner marker for {path}')
|
|
return path
|
|
except Exception:
|
|
try_remove_tree(path, attempts=2, delay=0.2)
|
|
raise
|
|
|
|
|
|
def write_temp_owner(path, cmd=None, owner_pid=None, required=False, owner_identity=None):
|
|
require_scanner_runtime_initialized()
|
|
if not path:
|
|
return
|
|
try:
|
|
owner_pid = int((owner_identity or {}).get('pid') if isinstance(owner_identity, dict) else owner_pid or os.getpid())
|
|
parent_identity = current_process_identity()
|
|
if isinstance(owner_identity, dict):
|
|
owner = dict(owner_identity)
|
|
elif owner_pid == parent_identity.pid:
|
|
owner = serialize_process_identity(parent_identity)
|
|
else:
|
|
with open_process(owner_pid) as retained:
|
|
owner = serialize_process_identity(retained.identity)
|
|
work_root = canonical_path(get_work_dir())
|
|
candidate = canonical_path(path)
|
|
relative = os.path.relpath(candidate, work_root)
|
|
if relative == '.' or relative.startswith('..' + os.sep) or os.path.isabs(relative):
|
|
raise RuntimeError('temp owner marker path escapes configured work_dir')
|
|
parent = serialize_process_identity(parent_identity)
|
|
payload = {
|
|
'schema': TEMP_OWNER_SCHEMA,
|
|
'owner_pid': owner['pid'],
|
|
'owner_creation_time': owner['creation_time'],
|
|
'owner_executable': owner['executable'],
|
|
'parent_pid': parent['pid'],
|
|
'parent_creation_time': parent['creation_time'],
|
|
'parent_executable': parent['executable'],
|
|
'created_at': datetime.now(timezone.utc).isoformat(timespec='seconds'),
|
|
'root_kind': 'work',
|
|
'relative_path': relative.replace(os.sep, '/'),
|
|
'command': redact_command_args(cmd or [])[:64],
|
|
}
|
|
atomic_write_private_json(os.path.join(path, TEMP_OWNER_FILE), payload)
|
|
return True
|
|
except (OSError, ValueError) as e:
|
|
if required:
|
|
raise RuntimeError(f'Unable to write required temp owner marker for {path}: {e}') from e
|
|
logger.warning(f'Unable to write temp owner marker for {path}: {str(e)[:200]}')
|
|
return False
|
|
|
|
|
|
def redact_command_args(args):
|
|
sensitive_flags = {'--token', '--docker-token', '--password', '--api-key', '--secret'}
|
|
redacted = []
|
|
hide_next = False
|
|
for value in args or []:
|
|
text = str(value)
|
|
if hide_next:
|
|
redacted.append('***REDACTED***')
|
|
hide_next = False
|
|
continue
|
|
flag = text.split('=', 1)[0].lower()
|
|
if flag in sensitive_flags:
|
|
if '=' in text:
|
|
redacted.append(text.split('=', 1)[0] + '=***REDACTED***')
|
|
else:
|
|
redacted.append(text)
|
|
hide_next = True
|
|
continue
|
|
try:
|
|
parsed = urlsplit(text)
|
|
if parsed.scheme in ('http', 'https') and parsed.hostname and ('@' in parsed.netloc or parsed.password):
|
|
netloc = parsed.hostname
|
|
if parsed.port:
|
|
netloc += f':{parsed.port}'
|
|
else:
|
|
netloc = parsed.netloc
|
|
if parsed.scheme in ('http', 'https') and parsed.hostname:
|
|
query = []
|
|
for key, query_value in parse_qsl(parsed.query, keep_blank_values=True):
|
|
sensitive = any(part in key.lower() for part in ('password', 'passwd', 'pwd', 'token', 'secret', 'credential'))
|
|
query.append((key, '***REDACTED***' if sensitive else query_value))
|
|
text = urlunsplit((parsed.scheme, netloc, parsed.path, urlencode(query), parsed.fragment))
|
|
except ValueError:
|
|
pass
|
|
redacted.append(text)
|
|
return redacted
|
|
|
|
|
|
def read_temp_owner(path):
|
|
marker = os.path.join(path, TEMP_OWNER_FILE)
|
|
try:
|
|
if not private_file_ready(marker):
|
|
return {}
|
|
data = read_private_json(marker)
|
|
return data if isinstance(data, dict) else {}
|
|
except (OSError, ValueError):
|
|
return {}
|
|
|
|
|
|
def temp_dir_active(path):
|
|
owner = read_temp_owner(path)
|
|
if owner.get('schema') != TEMP_OWNER_SCHEMA:
|
|
return True
|
|
states = [
|
|
exact_process_identity_state(
|
|
owner.get(f'{prefix}_pid'),
|
|
owner.get(f'{prefix}_creation_time'),
|
|
owner.get(f'{prefix}_executable'),
|
|
)
|
|
for prefix in ('owner', 'parent')
|
|
]
|
|
return any(state in ('alive', 'unknown') for state in states)
|
|
|
|
|
|
def create_docker_config_dir():
|
|
work_dir = get_work_dir()
|
|
docker_config_root = os.path.join(work_dir, 'docker-config')
|
|
require_private_directory(docker_config_root, create=True)
|
|
path = tempfile.mkdtemp(prefix='docker-config-', dir=docker_config_root)
|
|
harden_private_directory(path)
|
|
write_temp_owner(path, ['docker-auth-config'], os.getpid())
|
|
return path
|
|
|
|
def force_remove_readonly(function, path, exc_info):
|
|
try:
|
|
reject_reparse_components(path)
|
|
os.chmod(path, stat.S_IWRITE)
|
|
function(path)
|
|
except Exception:
|
|
pass
|
|
|
|
def try_remove_tree(path, attempts=1, delay=0.0):
|
|
if not path:
|
|
return True
|
|
|
|
for attempt in range(attempts):
|
|
try:
|
|
budget = JanitorBudget(
|
|
max_candidates=1,
|
|
max_entries=10000,
|
|
max_bytes=1024 * 1024 * 1024,
|
|
max_seconds=5.0,
|
|
max_depth=64,
|
|
)
|
|
return bounded_remove_tree(path, budget)
|
|
except FileNotFoundError:
|
|
return True
|
|
except KeyboardInterrupt:
|
|
raise
|
|
except Exception:
|
|
if delay and attempt + 1 < attempts:
|
|
time.sleep(delay)
|
|
return False
|
|
|
|
def _shared_staging_owners(roots):
|
|
"""Resolve only private trees already owned by this scanner; never adopt input data."""
|
|
work_root = canonical_path(get_work_dir())
|
|
current = serialize_process_identity(current_process_identity())
|
|
owners = {}
|
|
for value in roots:
|
|
root = canonical_path(require_private_directory(value, create=False))
|
|
if root == work_root or os.path.commonpath((root, work_root)) != work_root:
|
|
raise RuntimeError('shared staging root escapes configured work_dir')
|
|
while root != work_root and not os.path.lexists(os.path.join(root, TEMP_OWNER_FILE)):
|
|
root = os.path.dirname(root)
|
|
if root == work_root:
|
|
raise RuntimeError('shared staging root has no authenticated owner')
|
|
if root in owners:
|
|
continue
|
|
require_private_directory(root, create=False)
|
|
marker = read_private_json(require_private_file(os.path.join(root, TEMP_OWNER_FILE)), max_bytes=65536)
|
|
relative = os.path.relpath(root, work_root).replace(os.sep, '/')
|
|
if marker.get('schema') != TEMP_OWNER_SCHEMA or marker.get('root_kind') != 'work' or marker.get('relative_path') != relative:
|
|
raise RuntimeError('shared staging owner marker does not match its private root')
|
|
if any(marker.get(f'{prefix}_{field}') != current[field]
|
|
for prefix in ('owner', 'parent') for field in ('pid', 'creation_time', 'executable')):
|
|
raise RuntimeError('shared staging root belongs to another owner')
|
|
if any(f'child_{field}' in marker for field in ('pid', 'creation_time', 'executable')):
|
|
if any(not marker.get(f'child_{field}') for field in ('pid', 'creation_time', 'executable')) or (
|
|
exact_process_identity_state(marker.get('child_pid'), marker.get('child_creation_time'),
|
|
marker.get('child_executable')) != 'dead'
|
|
):
|
|
raise RuntimeError('shared staging root has an unconfirmed child')
|
|
for field in ('pid', 'creation_time', 'executable'):
|
|
marker.pop(f'child_{field}', None)
|
|
owners[root] = marker
|
|
return list(owners.items())
|
|
|
|
|
|
def cleanup_command_work_dir(path):
|
|
if not path:
|
|
return
|
|
|
|
marker_path = os.path.join(path, TEMP_OWNER_FILE)
|
|
if os.path.lexists(marker_path):
|
|
try:
|
|
marker = read_private_json(require_private_file(marker_path), max_bytes=65536)
|
|
if any(f'child_{field}' in marker for field in ('pid', 'creation_time', 'executable')):
|
|
if any(not marker.get(f'child_{field}') for field in ('pid', 'creation_time', 'executable')) or (
|
|
exact_process_identity_state(marker.get('child_pid'), marker.get('child_creation_time'),
|
|
marker.get('child_executable')) != 'dead'
|
|
):
|
|
logger.warning('Retaining private command tree with an unconfirmed child')
|
|
return
|
|
except (OSError, TypeError, ValueError):
|
|
logger.warning('Retaining private command tree with unreadable ownership evidence')
|
|
return
|
|
if try_remove_tree(path, attempts=2, delay=0.2):
|
|
return
|
|
|
|
logger.info('Bounded immediate cleanup deferred to authenticated janitor: %s', path)
|
|
|
|
|
|
def approved_pending_temp_path(path, work_dir=None):
|
|
"""Validate scanner-owned placement without following an external path."""
|
|
if not path or not os.path.isabs(path):
|
|
return False
|
|
try:
|
|
work_root = work_dir or get_work_dir()
|
|
reject_reparse_components(work_root)
|
|
reject_reparse_components(path)
|
|
if is_reparse_point(path) or not os.path.isdir(path) or not private_directory_ready(path):
|
|
return False
|
|
work_root = canonical_path(work_root)
|
|
candidate = canonical_path(path)
|
|
if os.path.commonpath((work_root, candidate)) != work_root or candidate == work_root:
|
|
return False
|
|
relative = os.path.relpath(candidate, work_root)
|
|
except (OSError, ValueError):
|
|
return False
|
|
parts = relative.split(os.sep)
|
|
if len(parts) == 1:
|
|
approved_name = parts[0].startswith(APPROVED_TEMP_PREFIXES)
|
|
elif len(parts) == 2 and parts[0] == 'docker-config':
|
|
approved_name = parts[1].startswith('docker-config-')
|
|
elif len(parts) == 2 and parts[0] == 'hg':
|
|
approved_name = parts[1].startswith('hg-run-')
|
|
elif len(parts) == 2 and parts[0] == 'tmp':
|
|
approved_name = parts[1].startswith(APPROVED_TEMP_PREFIXES)
|
|
elif len(parts) == 3 and parts[:2] == ['tmp', 'docker-config']:
|
|
approved_name = parts[2].startswith('docker-config-')
|
|
else:
|
|
approved_name = False
|
|
if not approved_name:
|
|
return False
|
|
owner = read_temp_owner(candidate)
|
|
owner_pid = owner.get('owner_pid') or owner.get('parent_pid')
|
|
return bool(owner_pid) and not temp_dir_active(candidate)
|
|
|
|
def cleanup_pending_command_work_dirs(max_items=None, attempts=1, delay=0.0, log_failures=False):
|
|
if log_failures:
|
|
logger.info('Source-side pending temp cleanup is retired; the authenticated janitor owns recovery')
|
|
return 0
|
|
|
|
|
|
def cleanup_assignment_work_dir(max_items=256):
|
|
"""Bound one cooperative cleanup pass to this runner's private work root."""
|
|
work_root = get_work_dir()
|
|
report = {'enumerated': 0, 'removed': 0, 'retained': 0}
|
|
with os.scandir(work_root) as entries:
|
|
for entry in entries:
|
|
report['enumerated'] += 1
|
|
if report['enumerated'] > max(1, int(max_items)):
|
|
report['retained'] += 1
|
|
break
|
|
if (
|
|
entry.is_symlink()
|
|
or not entry.is_dir(follow_symlinks=False)
|
|
or not entry.name.startswith(APPROVED_TEMP_PREFIXES)
|
|
):
|
|
continue
|
|
if cleanup_command_work_dir(entry.path) is None and not os.path.exists(entry.path):
|
|
report['removed'] += 1
|
|
else:
|
|
report['retained'] += 1
|
|
return report
|
|
|
|
def ensure_work_dir_space():
|
|
work_dir = get_work_dir()
|
|
if not work_dir or scan_config.min_free_gb <= 0:
|
|
return
|
|
|
|
min_free_bytes = scan_config.min_free_gb * 1024 * 1024 * 1024
|
|
free_bytes = shutil.disk_usage(work_dir).free
|
|
if free_bytes >= min_free_bytes:
|
|
return
|
|
|
|
raise RuntimeError(
|
|
f"Not enough free space on {work_dir}: {free_bytes / (1024 ** 3):.2f} GB free, "
|
|
f"minimum is {scan_config.min_free_gb:.2f} GB; admission is closed without cleanup"
|
|
)
|
|
|
|
def cleanup_stale_temp_dirs(age_minutes=120, log=True, max_items=None):
|
|
"""Compatibility no-op; stale recovery is isolated in janitor.py."""
|
|
if log:
|
|
logger.info('Source-side stale temp cleanup is retired; the authenticated janitor owns recovery')
|
|
return 0
|
|
|
|
def get_results_dir():
|
|
"""Prepare and return the directory used for persisted scan output."""
|
|
require_scanner_runtime_initialized()
|
|
if not scan_config.results_dir:
|
|
raise RuntimeError('scan results directory is required')
|
|
try:
|
|
return require_private_directory(scan_config.results_dir, create=False)
|
|
except OSError as exc:
|
|
raise RuntimeError(f'Unable to use private scan results directory {scan_config.results_dir}: {exc}') from exc
|
|
|
|
def append_jsonl(path, payload):
|
|
lock = None
|
|
lock_path = f'{path}.lock'
|
|
try:
|
|
require_private_directory(os.path.dirname(os.path.abspath(path)), create=True)
|
|
if os.path.lexists(path):
|
|
reject_reparse_components(path)
|
|
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
|
repair_jsonl_tail(path)
|
|
serialized = (json.dumps(payload, ensure_ascii=False, default=str) + '\n').encode('utf-8')
|
|
with open(path, 'ab') as f:
|
|
f.write(serialized)
|
|
f.flush()
|
|
os.fsync(f.fileno())
|
|
harden_private_file(path)
|
|
return True
|
|
except OSError as e:
|
|
if getattr(e, 'errno', None) == 28:
|
|
logger.error(f"No space left while writing {path}. Result was not persisted.")
|
|
else:
|
|
logger.error(f"Unable to write {path}: {str(e)}")
|
|
return False
|
|
finally:
|
|
if lock is not None:
|
|
release_file_lock(lock, lock_path)
|
|
|
|
|
|
def jsonl_manifest_path(path):
|
|
base, ext = os.path.splitext(path)
|
|
return f'{base}.manifest.json'
|
|
|
|
|
|
def load_jsonl_manifest(path):
|
|
manifest_path = jsonl_manifest_path(path)
|
|
try:
|
|
if os.path.getsize(manifest_path) > 1024 * 1024:
|
|
raise ValueError(f'JSONL manifest exceeds its bounded size: {manifest_path}')
|
|
with open(manifest_path, 'r', encoding='utf-8') as f:
|
|
data = json.load(f)
|
|
return data if isinstance(data, dict) else {}
|
|
except FileNotFoundError:
|
|
return {}
|
|
|
|
|
|
def write_jsonl_manifest(path, manifest):
|
|
manifest_path = jsonl_manifest_path(path)
|
|
tmp_path = f'{manifest_path}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.tmp'
|
|
with open(tmp_path, 'w', encoding='utf-8') as f:
|
|
json.dump(manifest, f, ensure_ascii=False, indent=2, sort_keys=True)
|
|
f.flush()
|
|
os.fsync(f.fileno())
|
|
harden_private_file(tmp_path)
|
|
os.replace(tmp_path, manifest_path)
|
|
harden_private_file(manifest_path)
|
|
|
|
|
|
def next_jsonl_segment_path(path, manifest):
|
|
base, ext = os.path.splitext(path)
|
|
seq = int(manifest.get('next_sequence') or 1)
|
|
while True:
|
|
segment = f'{base}.{seq:06d}{ext or ".jsonl"}'
|
|
if not os.path.exists(segment):
|
|
return segment, seq
|
|
seq += 1
|
|
|
|
|
|
def acquire_file_lock(lock_path, stale_sec=300, timeout_sec=30):
|
|
require_private_directory(os.path.dirname(os.path.abspath(lock_path)), create=True)
|
|
reject_reparse_components(os.path.dirname(os.path.abspath(lock_path)))
|
|
deadline = time.monotonic() + max(0.01, float(timeout_sec))
|
|
while True:
|
|
lock = PrivateFileLock(lock_path)
|
|
try:
|
|
return lock.acquire()
|
|
except BlockingIOError:
|
|
if time.monotonic() >= deadline:
|
|
raise TimeoutError(f'timed out acquiring lock {lock_path}')
|
|
time.sleep(min(0.05, max(0.0, deadline - time.monotonic())))
|
|
|
|
|
|
def release_file_lock(lock, lock_path):
|
|
try:
|
|
lock.release()
|
|
except (AttributeError, OSError):
|
|
return
|
|
|
|
|
|
class JsonlProjectionReconciliationRequired(RuntimeError):
|
|
pass
|
|
|
|
|
|
def jsonl_ledger_path(path):
|
|
base, _ = os.path.splitext(path)
|
|
return f'{base}.publication-ledger.sqlite3'
|
|
|
|
|
|
def _projection_segment_sequence(path):
|
|
parent = os.path.dirname(os.path.abspath(path))
|
|
base, extension = os.path.splitext(os.path.basename(path))
|
|
pattern = re.compile(rf'^{re.escape(base)}\.(\d{{6}}){re.escape(extension)}$')
|
|
output = []
|
|
inspect_limit = max(2, int(getattr(scan_config, 'jsonl_max_segments', 16)) + 1)
|
|
try:
|
|
with os.scandir(parent) as entries:
|
|
for entry in entries:
|
|
match = pattern.fullmatch(entry.name)
|
|
if not match:
|
|
continue
|
|
if entry.is_symlink() or is_reparse_point(entry.path) or not entry.is_file(follow_symlinks=False):
|
|
raise JsonlProjectionReconciliationRequired(f'unsafe JSONL segment entry: {entry.path}')
|
|
output.append((int(match.group(1)), entry.path))
|
|
if len(output) > inspect_limit:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'JSONL physical segment count exceeds its bound for {path}; use offline reconciliation'
|
|
)
|
|
except FileNotFoundError:
|
|
return []
|
|
return sorted(output)
|
|
|
|
|
|
def _write_torn_tail_quarantine(path, payload):
|
|
limit = max(1, int(getattr(scan_config, 'jsonl_torn_quarantine_max_bytes', 64 * 1024)))
|
|
sample = bytes(payload[:limit])
|
|
quarantine = f'{path}.torn-tail.bin'
|
|
temporary = f'{quarantine}.{os.getpid()}.{threading.get_ident()}.tmp'
|
|
with open(temporary, 'wb') as handle:
|
|
handle.write(sample)
|
|
handle.flush()
|
|
os.fsync(handle.fileno())
|
|
harden_private_file(temporary)
|
|
durable_replace(temporary, quarantine)
|
|
harden_private_file(quarantine)
|
|
|
|
|
|
def repair_jsonl_tail(path):
|
|
"""Quarantine and remove one bounded unterminated tail before appending."""
|
|
if not os.path.exists(path):
|
|
return 0
|
|
reject_reparse_components(path)
|
|
size = os.path.getsize(path)
|
|
if size <= 0:
|
|
return 0
|
|
scan_limit = max(1, int(getattr(scan_config, 'jsonl_tail_scan_max_bytes', 8 * 1024 * 1024)))
|
|
with open(path, 'r+b') as handle:
|
|
handle.seek(-1, os.SEEK_END)
|
|
if handle.read(1) == b'\n':
|
|
return 0
|
|
start = max(0, size - scan_limit)
|
|
handle.seek(start)
|
|
tail = handle.read(size - start)
|
|
newline = tail.rfind(b'\n')
|
|
if newline < 0 and start:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'JSONL tail exceeds the bounded repair window for {path}; use offline reconciliation'
|
|
)
|
|
truncate_at = start + newline + 1 if newline >= 0 else 0
|
|
torn = tail[newline + 1:] if newline >= 0 else tail
|
|
_write_torn_tail_quarantine(path, torn)
|
|
handle.truncate(truncate_at)
|
|
handle.flush()
|
|
os.fsync(handle.fileno())
|
|
logger.error('Quarantined and truncated %s torn byte(s) from %s', size - truncate_at, path)
|
|
return size - truncate_at
|
|
|
|
|
|
def _open_projection_ledger(path):
|
|
ledger_path = jsonl_ledger_path(path)
|
|
reject_reparse_components(os.path.dirname(os.path.abspath(ledger_path)))
|
|
if os.path.lexists(ledger_path):
|
|
reject_reparse_components(ledger_path)
|
|
if not private_file_ready(ledger_path):
|
|
raise JsonlProjectionReconciliationRequired(f'JSONL publication ledger is not private: {ledger_path}')
|
|
connection = sqlite3.connect(ledger_path, timeout=30)
|
|
try:
|
|
connection.execute('PRAGMA busy_timeout=30000')
|
|
connection.execute('PRAGMA journal_mode=DELETE')
|
|
connection.execute('PRAGMA synchronous=FULL')
|
|
connection.executescript('''
|
|
CREATE TABLE IF NOT EXISTS publication_identity (
|
|
identity_key TEXT NOT NULL,
|
|
identity_value TEXT NOT NULL,
|
|
payload_sha256 TEXT NOT NULL,
|
|
state TEXT NOT NULL,
|
|
file_name TEXT NOT NULL,
|
|
byte_offset INTEGER NOT NULL,
|
|
byte_length INTEGER NOT NULL,
|
|
created_at REAL NOT NULL,
|
|
updated_at REAL NOT NULL,
|
|
PRIMARY KEY(identity_key, identity_value)
|
|
);
|
|
CREATE TABLE IF NOT EXISTS publication_meta (
|
|
key TEXT PRIMARY KEY,
|
|
value TEXT NOT NULL
|
|
);
|
|
CREATE TABLE IF NOT EXISTS publication_identity_variant (
|
|
identity_key TEXT NOT NULL,
|
|
identity_value TEXT NOT NULL,
|
|
payload_sha256 TEXT NOT NULL,
|
|
file_name TEXT NOT NULL,
|
|
byte_offset INTEGER NOT NULL,
|
|
byte_length INTEGER NOT NULL,
|
|
created_at REAL NOT NULL,
|
|
PRIMARY KEY(identity_key, identity_value, payload_sha256)
|
|
);
|
|
CREATE TABLE IF NOT EXISTS reconciliation_issue (
|
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
identity_key TEXT NOT NULL,
|
|
file_name TEXT NOT NULL,
|
|
file_device TEXT NOT NULL,
|
|
file_inode TEXT NOT NULL,
|
|
file_size INTEGER NOT NULL,
|
|
file_mtime_ns TEXT NOT NULL,
|
|
byte_offset INTEGER NOT NULL,
|
|
byte_length INTEGER NOT NULL,
|
|
record_sha256 TEXT NOT NULL,
|
|
classification TEXT NOT NULL,
|
|
status TEXT NOT NULL,
|
|
created_at REAL NOT NULL,
|
|
resolved_at REAL,
|
|
UNIQUE(identity_key, file_name, byte_offset, record_sha256)
|
|
);
|
|
CREATE TABLE IF NOT EXISTS reconciliation_variant_issue (
|
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
identity_key TEXT NOT NULL,
|
|
identity_sha256 TEXT NOT NULL,
|
|
file_name TEXT NOT NULL,
|
|
file_device TEXT NOT NULL,
|
|
file_inode TEXT NOT NULL,
|
|
file_size INTEGER NOT NULL,
|
|
file_mtime_ns TEXT NOT NULL,
|
|
byte_offset INTEGER NOT NULL,
|
|
byte_length INTEGER NOT NULL,
|
|
payload_sha256 TEXT NOT NULL,
|
|
field_name_set_sha256 TEXT NOT NULL,
|
|
status TEXT NOT NULL,
|
|
created_at REAL NOT NULL,
|
|
resolved_at REAL,
|
|
UNIQUE(identity_key, identity_sha256, file_name, byte_offset, payload_sha256)
|
|
);
|
|
CREATE INDEX IF NOT EXISTS idx_publication_identity_state_created
|
|
ON publication_identity(state, created_at);
|
|
CREATE INDEX IF NOT EXISTS idx_publication_identity_file_state
|
|
ON publication_identity(file_name, state);
|
|
CREATE INDEX IF NOT EXISTS idx_publication_identity_variant_identity
|
|
ON publication_identity_variant(identity_key, identity_value);
|
|
CREATE INDEX IF NOT EXISTS idx_reconciliation_issue_status
|
|
ON reconciliation_issue(status, id);
|
|
CREATE INDEX IF NOT EXISTS idx_reconciliation_variant_issue_status
|
|
ON reconciliation_variant_issue(status, id);
|
|
''')
|
|
connection.execute(
|
|
'''INSERT OR IGNORE INTO publication_identity_variant (
|
|
identity_key, identity_value, payload_sha256, file_name,
|
|
byte_offset, byte_length, created_at
|
|
)
|
|
SELECT identity_key, identity_value, payload_sha256, file_name,
|
|
byte_offset, byte_length, created_at
|
|
FROM publication_identity WHERE state = 'appended' '''
|
|
)
|
|
row = connection.execute("SELECT value FROM publication_meta WHERE key = 'row_count'").fetchone()
|
|
if row is None:
|
|
count = int(connection.execute('SELECT COUNT(*) FROM publication_identity').fetchone()[0])
|
|
connection.execute(
|
|
"INSERT INTO publication_meta(key, value) VALUES ('row_count', ?)",
|
|
(str(count),),
|
|
)
|
|
connection.commit()
|
|
harden_private_file(ledger_path)
|
|
return connection
|
|
except BaseException:
|
|
connection.close()
|
|
raise
|
|
|
|
|
|
def _ledger_row_count(connection):
|
|
row = connection.execute("SELECT value FROM publication_meta WHERE key = 'row_count'").fetchone()
|
|
return max(0, int(row[0] if row else 0))
|
|
|
|
|
|
def _set_ledger_row_count(connection, count):
|
|
connection.execute(
|
|
"INSERT OR REPLACE INTO publication_meta(key, value) VALUES ('row_count', ?)",
|
|
(str(max(0, int(count))),),
|
|
)
|
|
|
|
|
|
def _bounded_projection_bytes(path, offset, length):
|
|
max_record = max(
|
|
1024 * 1024,
|
|
int(getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)) + 1,
|
|
)
|
|
if offset < 0 or length <= 0 or length > max_record:
|
|
return b''
|
|
try:
|
|
with open(path, 'rb') as handle:
|
|
handle.seek(offset)
|
|
return handle.read(length)
|
|
except OSError:
|
|
return b''
|
|
|
|
|
|
def _recover_prepared_publications(connection, path):
|
|
rows = connection.execute(
|
|
'''SELECT identity_key, identity_value, payload_sha256, file_name, byte_offset, byte_length
|
|
FROM publication_identity WHERE state = 'prepared' ORDER BY created_at LIMIT 2'''
|
|
).fetchall()
|
|
if len(rows) > 1:
|
|
raise JsonlProjectionReconciliationRequired('publication ledger contains multiple unresolved append states')
|
|
for identity_key, identity_value, digest, file_name, offset, length in rows:
|
|
candidate = os.path.join(os.path.dirname(os.path.abspath(path)), os.path.basename(file_name))
|
|
payload = _bounded_projection_bytes(candidate, int(offset), int(length))
|
|
if payload.endswith(b'\n') and hashlib.sha256(payload).hexdigest() == digest:
|
|
connection.execute(
|
|
'''UPDATE publication_identity SET state = 'appended', updated_at = ?
|
|
WHERE identity_key = ? AND identity_value = ? AND state = 'prepared' ''',
|
|
(time.time(), identity_key, identity_value),
|
|
)
|
|
connection.execute(
|
|
'''INSERT OR IGNORE INTO publication_identity_variant (
|
|
identity_key, identity_value, payload_sha256, file_name,
|
|
byte_offset, byte_length, created_at
|
|
) VALUES (?, ?, ?, ?, ?, ?, ?)''',
|
|
(identity_key, identity_value, digest, file_name, offset, length, time.time()),
|
|
)
|
|
else:
|
|
connection.execute(
|
|
'DELETE FROM publication_identity WHERE identity_key = ? AND identity_value = ? AND state = ?',
|
|
(identity_key, identity_value, 'prepared'),
|
|
)
|
|
_set_ledger_row_count(connection, _ledger_row_count(connection) - 1)
|
|
connection.commit()
|
|
|
|
|
|
def _ensure_projection_ledger_bootstrapped(connection, path, identity_key):
|
|
marker = f'bootstrapped:{identity_key}'
|
|
if connection.execute('SELECT 1 FROM publication_meta WHERE key = ?', (marker,)).fetchone():
|
|
return
|
|
candidates = projection_segment_paths(path)
|
|
total_bytes = sum(os.path.getsize(candidate) for candidate in candidates)
|
|
max_bytes = max(0, int(getattr(scan_config, 'jsonl_legacy_index_max_bytes', 16 * 1024 * 1024)))
|
|
if total_bytes > max_bytes:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'existing JSONL history for {path} is unindexed ({total_bytes} bytes); '
|
|
'run the offline JSONL reconciliation procedure before publication'
|
|
)
|
|
row_count = _ledger_row_count(connection)
|
|
row_limit = max(1, int(getattr(scan_config, 'jsonl_ledger_max_rows', 1000000)))
|
|
for candidate in candidates:
|
|
offset = 0
|
|
with open(candidate, 'rb') as handle:
|
|
for raw_line in handle:
|
|
if not raw_line.endswith(b'\n'):
|
|
raise JsonlProjectionReconciliationRequired(f'unterminated closed JSONL record in {candidate}')
|
|
identity = _projection_identity_from_line(raw_line, identity_key, candidate)
|
|
if identity:
|
|
digest = hashlib.sha256(raw_line).hexdigest()
|
|
existing = connection.execute(
|
|
'''SELECT payload_sha256 FROM publication_identity
|
|
WHERE identity_key = ? AND identity_value = ?''',
|
|
(identity_key, identity),
|
|
).fetchone()
|
|
if existing and existing[0] != digest:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
'conflicting identity in existing JSONL history: '
|
|
f'identity_key={identity_key} '
|
|
f'identity_sha256={hashlib.sha256(identity.encode("utf-8")).hexdigest()}'
|
|
)
|
|
if not existing:
|
|
if row_count >= row_limit:
|
|
raise JsonlProjectionReconciliationRequired('existing JSONL identities exceed the ledger row bound')
|
|
now = time.time()
|
|
connection.execute(
|
|
'''INSERT INTO publication_identity (
|
|
identity_key, identity_value, payload_sha256, state, file_name,
|
|
byte_offset, byte_length, created_at, updated_at
|
|
) VALUES (?, ?, ?, 'appended', ?, ?, ?, ?, ?)''',
|
|
(
|
|
identity_key, identity, digest, os.path.basename(candidate),
|
|
offset, len(raw_line), now, now,
|
|
),
|
|
)
|
|
connection.execute(
|
|
'''INSERT INTO publication_identity_variant (
|
|
identity_key, identity_value, payload_sha256, file_name,
|
|
byte_offset, byte_length, created_at
|
|
) VALUES (?, ?, ?, ?, ?, ?, ?)''',
|
|
(
|
|
identity_key, identity, digest, os.path.basename(candidate),
|
|
offset, len(raw_line), now,
|
|
),
|
|
)
|
|
row_count += 1
|
|
offset += len(raw_line)
|
|
_set_ledger_row_count(connection, row_count)
|
|
connection.execute('INSERT INTO publication_meta(key, value) VALUES (?, ?)', (marker, '1'))
|
|
connection.commit()
|
|
|
|
|
|
def _projection_identity_from_line(raw_line, identity_key, candidate):
|
|
if not raw_line.endswith(b'\n'):
|
|
raise JsonlProjectionReconciliationRequired(f'unterminated projection record in {candidate}')
|
|
if identity_key == 'error_row_id':
|
|
try:
|
|
identity, separator, _ = raw_line.partition(b'\t')
|
|
if not separator or not identity:
|
|
raise ValueError('missing error projection identity separator')
|
|
return identity.decode('utf-8')
|
|
except UnicodeDecodeError as exc:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'invalid existing projection record in {candidate}'
|
|
) from exc
|
|
except ValueError as exc:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'invalid existing projection record in {candidate}'
|
|
) from exc
|
|
try:
|
|
payload = json.loads(raw_line.decode('utf-8'))
|
|
except (UnicodeDecodeError, ValueError) as exc:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'invalid existing JSONL record in {candidate}'
|
|
) from exc
|
|
return str(payload.get(identity_key) or '') if isinstance(payload, dict) else ''
|
|
|
|
|
|
def _bounded_projection_json_values(raw_line):
|
|
body = raw_line[:-1] if raw_line.endswith(b'\n') else raw_line
|
|
if not raw_line.endswith(b'\n'):
|
|
return None, 'unterminated_record'
|
|
try:
|
|
text = body.decode('utf-8')
|
|
except UnicodeDecodeError:
|
|
return None, 'invalid_utf8'
|
|
if text.startswith('\ufeff'):
|
|
return None, 'utf8_bom_prefix'
|
|
try:
|
|
value = json.loads(text)
|
|
return [{
|
|
'value': value,
|
|
'relative_offset': 0,
|
|
'byte_length': len(raw_line),
|
|
'payload_sha256': hashlib.sha256(raw_line).hexdigest(),
|
|
}], 'single_json'
|
|
except ValueError:
|
|
pass
|
|
decoder = json.JSONDecoder()
|
|
position = 0
|
|
parsed = []
|
|
while True:
|
|
while position < len(text) and text[position].isspace():
|
|
position += 1
|
|
if position >= len(text):
|
|
break
|
|
start = position
|
|
try:
|
|
value, position = decoder.raw_decode(text, position)
|
|
except json.JSONDecodeError:
|
|
parsed = []
|
|
break
|
|
parsed.append((start, position, value))
|
|
if len(parsed) > 1 and position >= len(text) and all(isinstance(item[2], dict) for item in parsed):
|
|
values = []
|
|
for start, end, value in parsed:
|
|
prefix_bytes = len(text[:start].encode('utf-8'))
|
|
serialized = text[start:end].encode('utf-8') + b'\n'
|
|
values.append({
|
|
'value': value,
|
|
'relative_offset': prefix_bytes,
|
|
'byte_length': len(serialized),
|
|
'payload_sha256': hashlib.sha256(serialized).hexdigest(),
|
|
})
|
|
return values, 'concatenated_json_objects'
|
|
if re.match(r'^[0-9]+,\s*', text):
|
|
return None, 'legacy_numeric_prefix_corrupt_json'
|
|
return None, 'invalid_json'
|
|
|
|
|
|
def _projection_field_name_set_sha256(value):
|
|
paths = []
|
|
|
|
def visit(item, prefix=''):
|
|
if isinstance(item, dict):
|
|
for key in sorted(map(str, item.keys())):
|
|
path = prefix + key
|
|
paths.append(path)
|
|
visit(item.get(key), path + '.')
|
|
elif isinstance(item, list):
|
|
paths.append(prefix + '[]')
|
|
for child in item[:32]:
|
|
visit(child, prefix + '[].')
|
|
|
|
visit(value)
|
|
return hashlib.sha256('\x00'.join(sorted(set(paths))).encode('utf-8')).hexdigest()
|
|
|
|
|
|
def _error_projection_field_name_set_sha256(raw_line):
|
|
try:
|
|
text = raw_line[:-1].decode('utf-8') if raw_line.endswith(b'\n') else raw_line.decode('utf-8')
|
|
tail = text.rsplit('\t', 1)[-1]
|
|
value = json.loads(tail)
|
|
except (UnicodeDecodeError, ValueError):
|
|
value = {}
|
|
return _projection_field_name_set_sha256(value)
|
|
|
|
|
|
def _save_projection_reconciliation_progress(ledger, prefix, file_index, offset, indexed_rows):
|
|
_set_ledger_row_count(ledger, indexed_rows)
|
|
for key, value in (
|
|
(prefix + 'file_index', str(file_index)),
|
|
(prefix + 'offset', str(offset)),
|
|
):
|
|
ledger.execute('INSERT OR REPLACE INTO publication_meta(key, value) VALUES (?, ?)', (key, value))
|
|
|
|
|
|
def _record_projection_reconciliation_issue(
|
|
ledger,
|
|
identity_key,
|
|
plan_entry,
|
|
byte_offset,
|
|
raw_line,
|
|
classification,
|
|
resolved=False,
|
|
byte_length=None,
|
|
record_sha256=None,
|
|
):
|
|
if raw_line is not None:
|
|
byte_length = len(raw_line)
|
|
record_sha256 = hashlib.sha256(raw_line).hexdigest()
|
|
byte_length = int(byte_length or 0)
|
|
digest = str(record_sha256 or '').strip().lower()
|
|
if byte_length <= 0 or not re.fullmatch(r'[a-f0-9]{64}', digest):
|
|
raise ValueError('projection reconciliation issue metadata is invalid')
|
|
now = time.time()
|
|
ledger.execute(
|
|
'''INSERT OR IGNORE INTO reconciliation_issue (
|
|
identity_key, file_name, file_device, file_inode, file_size, file_mtime_ns,
|
|
byte_offset, byte_length, record_sha256, classification, status, created_at, resolved_at
|
|
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''',
|
|
(
|
|
identity_key,
|
|
os.path.basename(plan_entry['path']),
|
|
str(plan_entry['device']),
|
|
str(plan_entry['inode']),
|
|
int(plan_entry['size']),
|
|
str(plan_entry['mtime_ns']),
|
|
int(byte_offset),
|
|
byte_length,
|
|
digest,
|
|
classification,
|
|
'resolved' if resolved else 'pending',
|
|
now,
|
|
now if resolved else None,
|
|
),
|
|
)
|
|
if resolved:
|
|
ledger.execute(
|
|
'''UPDATE reconciliation_issue SET status = 'resolved', resolved_at = COALESCE(resolved_at, ?)
|
|
WHERE identity_key = ? AND file_name = ? AND byte_offset = ? AND record_sha256 = ?''',
|
|
(now, identity_key, os.path.basename(plan_entry['path']), int(byte_offset), digest),
|
|
)
|
|
return ledger.execute(
|
|
'''SELECT id, status FROM reconciliation_issue
|
|
WHERE identity_key = ? AND file_name = ? AND byte_offset = ? AND record_sha256 = ?''',
|
|
(identity_key, os.path.basename(plan_entry['path']), int(byte_offset), digest),
|
|
).fetchone()
|
|
|
|
|
|
def _record_projection_variant_issue(
|
|
ledger,
|
|
identity_key,
|
|
identity,
|
|
plan_entry,
|
|
byte_offset,
|
|
byte_length,
|
|
payload_sha256,
|
|
field_name_set_sha256,
|
|
resolved=False,
|
|
):
|
|
identity_sha256 = hashlib.sha256(identity.encode('utf-8')).hexdigest()
|
|
now = time.time()
|
|
ledger.execute(
|
|
'''INSERT OR IGNORE INTO reconciliation_variant_issue (
|
|
identity_key, identity_sha256, file_name, file_device, file_inode,
|
|
file_size, file_mtime_ns, byte_offset, byte_length, payload_sha256,
|
|
field_name_set_sha256, status, created_at, resolved_at
|
|
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''',
|
|
(
|
|
identity_key,
|
|
identity_sha256,
|
|
os.path.basename(plan_entry['path']),
|
|
str(plan_entry['device']),
|
|
str(plan_entry['inode']),
|
|
int(plan_entry['size']),
|
|
str(plan_entry['mtime_ns']),
|
|
int(byte_offset),
|
|
int(byte_length),
|
|
payload_sha256,
|
|
field_name_set_sha256,
|
|
'resolved' if resolved else 'pending',
|
|
now,
|
|
now if resolved else None,
|
|
),
|
|
)
|
|
if resolved:
|
|
ledger.execute(
|
|
'''UPDATE reconciliation_variant_issue
|
|
SET status = 'resolved', resolved_at = COALESCE(resolved_at, ?)
|
|
WHERE identity_key = ? AND identity_sha256 = ? AND file_name = ?
|
|
AND byte_offset = ? AND payload_sha256 = ?''',
|
|
(
|
|
now, identity_key, identity_sha256, os.path.basename(plan_entry['path']),
|
|
int(byte_offset), payload_sha256,
|
|
),
|
|
)
|
|
return ledger.execute(
|
|
'''SELECT id, status FROM reconciliation_variant_issue
|
|
WHERE identity_key = ? AND identity_sha256 = ? AND file_name = ?
|
|
AND byte_offset = ? AND payload_sha256 = ?''',
|
|
(
|
|
identity_key, identity_sha256, os.path.basename(plan_entry['path']),
|
|
int(byte_offset), payload_sha256,
|
|
),
|
|
).fetchone()
|
|
|
|
|
|
def _stream_projection_record(handle, first_chunk, chunk_bytes=1024 * 1024):
|
|
digest = hashlib.sha256()
|
|
digest.update(first_chunk)
|
|
total = len(first_chunk)
|
|
newline_terminated = first_chunk.endswith(b'\n')
|
|
while not newline_terminated:
|
|
chunk = handle.readline(max(1, int(chunk_bytes)))
|
|
if not chunk:
|
|
break
|
|
digest.update(chunk)
|
|
total += len(chunk)
|
|
newline_terminated = chunk.endswith(b'\n')
|
|
return total, digest.hexdigest(), newline_terminated
|
|
|
|
|
|
def _projection_reconciliation_plan(path):
|
|
candidates = projection_segment_paths(path)
|
|
if not candidates:
|
|
_publish_empty_jsonl_generation(path)
|
|
candidates = [os.path.abspath(path)]
|
|
plan = []
|
|
for candidate in candidates:
|
|
require_private_file(candidate)
|
|
details = os.stat(candidate, follow_symlinks=False)
|
|
plan.append({
|
|
'path': os.path.abspath(candidate),
|
|
'device': int(getattr(details, 'st_dev', 0) or 0),
|
|
'inode': int(getattr(details, 'st_ino', 0) or 0),
|
|
'size': int(details.st_size),
|
|
'mtime_ns': int(getattr(details, 'st_mtime_ns', int(details.st_mtime * 1_000_000_000))),
|
|
})
|
|
return plan
|
|
|
|
|
|
def reconcile_projection_ledger_batch(
|
|
path,
|
|
identity_key,
|
|
max_rows=10000,
|
|
max_bytes=64 * 1024 * 1024,
|
|
max_seconds=30.0,
|
|
row_limit=None,
|
|
ledger_byte_limit=None,
|
|
max_record_bytes=None,
|
|
):
|
|
"""Build one bounded, resumable ledger batch without modifying JSONL history."""
|
|
if identity_key not in ('scan_event_id', 'finding_uid', 'error_row_id'):
|
|
raise ValueError('unsupported projection reconciliation identity')
|
|
path = os.path.abspath(path)
|
|
require_private_directory(os.path.dirname(path), create=False)
|
|
max_rows = max(1, int(max_rows))
|
|
max_bytes = max(1, int(max_bytes))
|
|
max_seconds = max(0.01, float(max_seconds))
|
|
row_limit = max(1, int(row_limit or getattr(scan_config, 'jsonl_ledger_max_rows', 1000000)))
|
|
ledger_byte_limit = max(
|
|
1024 * 1024,
|
|
int(ledger_byte_limit or getattr(scan_config, 'jsonl_ledger_max_bytes', 512 * 1024 * 1024)),
|
|
)
|
|
max_record_bytes = max(
|
|
1024 * 1024,
|
|
int(max_record_bytes or getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)),
|
|
)
|
|
lock_path = f'{path}.lock'
|
|
lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec)
|
|
ledger = None
|
|
try:
|
|
prefix = f'offline-reconcile:{identity_key}:'
|
|
marker = f'bootstrapped:{identity_key}'
|
|
ledger = _open_projection_ledger(path)
|
|
if ledger.execute('SELECT 1 FROM publication_meta WHERE key = ?', (marker,)).fetchone():
|
|
return {
|
|
'path': path, 'identity_key': identity_key, 'complete': True,
|
|
'batch_rows': 0, 'batch_bytes': 0, 'indexed_rows': _ledger_row_count(ledger),
|
|
}
|
|
plan = _projection_reconciliation_plan(path)
|
|
plan_json = json.dumps(plan, ensure_ascii=True, sort_keys=True, separators=(',', ':'))
|
|
existing_plan = ledger.execute(
|
|
'SELECT value FROM publication_meta WHERE key = ?', (prefix + 'plan',),
|
|
).fetchone()
|
|
if existing_plan and existing_plan[0] != plan_json:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'projection history changed during offline reconciliation: {path}'
|
|
)
|
|
if not existing_plan:
|
|
ledger.execute(
|
|
'INSERT INTO publication_meta(key, value) VALUES (?, ?)',
|
|
(prefix + 'plan', plan_json),
|
|
)
|
|
file_index_row = ledger.execute(
|
|
'SELECT value FROM publication_meta WHERE key = ?', (prefix + 'file_index',),
|
|
).fetchone()
|
|
offset_row = ledger.execute(
|
|
'SELECT value FROM publication_meta WHERE key = ?', (prefix + 'offset',),
|
|
).fetchone()
|
|
file_index = max(0, int(file_index_row[0] if file_index_row else 0))
|
|
offset = max(0, int(offset_row[0] if offset_row else 0))
|
|
indexed_rows = _ledger_row_count(ledger)
|
|
batch_rows = 0
|
|
batch_bytes = 0
|
|
started = time.monotonic()
|
|
while file_index < len(plan):
|
|
candidate = plan[file_index]['path']
|
|
with open(candidate, 'rb') as handle:
|
|
handle.seek(offset)
|
|
while True:
|
|
raw_line = handle.readline(max_record_bytes + 1)
|
|
if not raw_line:
|
|
file_index += 1
|
|
offset = 0
|
|
break
|
|
if len(raw_line) > max_record_bytes:
|
|
byte_length, record_sha256, newline_terminated = _stream_projection_record(
|
|
handle, raw_line,
|
|
)
|
|
classification = 'oversized_record' if newline_terminated else 'unterminated_record'
|
|
issue = _record_projection_reconciliation_issue(
|
|
ledger,
|
|
identity_key,
|
|
plan[file_index],
|
|
offset,
|
|
None,
|
|
classification,
|
|
byte_length=byte_length,
|
|
record_sha256=record_sha256,
|
|
)
|
|
_save_projection_reconciliation_progress(
|
|
ledger, prefix, file_index, offset, indexed_rows,
|
|
)
|
|
ledger.commit()
|
|
if issue[1] != 'resolved':
|
|
raise JsonlProjectionReconciliationRequired(
|
|
'projection reconciliation issue requires explicit review: '
|
|
f'id={issue[0]} file={os.path.basename(candidate)} offset={offset} '
|
|
f'length={byte_length} sha256={record_sha256} '
|
|
f'classification={classification}'
|
|
)
|
|
offset += byte_length
|
|
batch_rows += 1
|
|
batch_bytes += byte_length
|
|
if (
|
|
batch_rows >= max_rows
|
|
or batch_bytes >= max_bytes
|
|
or time.monotonic() - started >= max_seconds
|
|
):
|
|
break
|
|
continue
|
|
if identity_key == 'error_row_id':
|
|
try:
|
|
identity = _projection_identity_from_line(raw_line, identity_key, candidate)
|
|
if len(identity) > 256 or any(
|
|
character in identity for character in ('\x00', '\r', '\n')
|
|
):
|
|
values = None
|
|
classification = 'invalid_error_projection'
|
|
else:
|
|
values = [{
|
|
'identity': identity,
|
|
'relative_offset': 0,
|
|
'byte_length': len(raw_line),
|
|
'payload_sha256': hashlib.sha256(raw_line).hexdigest(),
|
|
'field_name_set_sha256': _error_projection_field_name_set_sha256(raw_line),
|
|
}]
|
|
classification = 'error_projection'
|
|
except JsonlProjectionReconciliationRequired:
|
|
values = None
|
|
classification = (
|
|
'unterminated_record' if not raw_line.endswith(b'\n')
|
|
else 'invalid_error_projection'
|
|
)
|
|
else:
|
|
parsed_values, classification = _bounded_projection_json_values(raw_line)
|
|
values = None if parsed_values is None else [{
|
|
'identity': (
|
|
str(item['value'].get(identity_key) or '')
|
|
if isinstance(item['value'], dict) else ''
|
|
),
|
|
'relative_offset': item['relative_offset'],
|
|
'byte_length': item['byte_length'],
|
|
'payload_sha256': item['payload_sha256'],
|
|
'field_name_set_sha256': _projection_field_name_set_sha256(item['value']),
|
|
} for item in parsed_values]
|
|
if values is None:
|
|
issue = _record_projection_reconciliation_issue(
|
|
ledger,
|
|
identity_key,
|
|
plan[file_index],
|
|
offset,
|
|
raw_line,
|
|
classification,
|
|
)
|
|
_save_projection_reconciliation_progress(
|
|
ledger, prefix, file_index, offset, indexed_rows,
|
|
)
|
|
ledger.commit()
|
|
if issue[1] != 'resolved':
|
|
raise JsonlProjectionReconciliationRequired(
|
|
'projection reconciliation issue requires explicit review: '
|
|
f'id={issue[0]} file={os.path.basename(candidate)} offset={offset} '
|
|
f'length={len(raw_line)} sha256={hashlib.sha256(raw_line).hexdigest()} '
|
|
f'classification={classification}'
|
|
)
|
|
offset += len(raw_line)
|
|
batch_rows += 1
|
|
batch_bytes += len(raw_line)
|
|
if (
|
|
batch_rows >= max_rows
|
|
or batch_bytes >= max_bytes
|
|
or time.monotonic() - started >= max_seconds
|
|
):
|
|
break
|
|
continue
|
|
for value in values:
|
|
identity = value['identity']
|
|
if not identity:
|
|
continue
|
|
if len(identity) > 256 or any(character in identity for character in ('\x00', '\r', '\n')):
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'invalid {identity_key} in existing projection history: {candidate}:{offset}'
|
|
)
|
|
digest = value['payload_sha256']
|
|
existing = ledger.execute(
|
|
'''SELECT payload_sha256, state FROM publication_identity
|
|
WHERE identity_key = ? AND identity_value = ?''',
|
|
(identity_key, identity),
|
|
).fetchone()
|
|
variant = ledger.execute(
|
|
'''SELECT 1 FROM publication_identity_variant
|
|
WHERE identity_key = ? AND identity_value = ? AND payload_sha256 = ?''',
|
|
(identity_key, identity, digest),
|
|
).fetchone()
|
|
if variant:
|
|
continue
|
|
if existing:
|
|
if existing[1] != 'appended':
|
|
raise JsonlProjectionReconciliationRequired(
|
|
'publication ledger retained an unresolved historical identity state'
|
|
)
|
|
variant_offset = offset + int(value['relative_offset'])
|
|
issue = _record_projection_variant_issue(
|
|
ledger,
|
|
identity_key,
|
|
identity,
|
|
plan[file_index],
|
|
variant_offset,
|
|
int(value['byte_length']),
|
|
digest,
|
|
value['field_name_set_sha256'],
|
|
)
|
|
_save_projection_reconciliation_progress(
|
|
ledger, prefix, file_index, offset, indexed_rows,
|
|
)
|
|
ledger.commit()
|
|
raise JsonlProjectionReconciliationRequired(
|
|
'historical projection payload variant requires explicit review: '
|
|
f'id={issue[0]} file={os.path.basename(candidate)} '
|
|
f'offset={variant_offset} length={int(value["byte_length"])} '
|
|
f'identity_sha256={hashlib.sha256(identity.encode("utf-8")).hexdigest()} '
|
|
f'payload_sha256={digest}'
|
|
)
|
|
if not existing:
|
|
if indexed_rows >= row_limit:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'existing JSONL identities exceed the ledger row bound: {row_limit}'
|
|
)
|
|
now = time.time()
|
|
ledger.execute(
|
|
'''INSERT INTO publication_identity (
|
|
identity_key, identity_value, payload_sha256, state, file_name,
|
|
byte_offset, byte_length, created_at, updated_at
|
|
) VALUES (?, ?, ?, 'appended', ?, ?, ?, ?, ?)''',
|
|
(
|
|
identity_key, identity, digest, os.path.basename(candidate),
|
|
offset + int(value['relative_offset']), int(value['byte_length']), now, now,
|
|
),
|
|
)
|
|
ledger.execute(
|
|
'''INSERT INTO publication_identity_variant (
|
|
identity_key, identity_value, payload_sha256, file_name,
|
|
byte_offset, byte_length, created_at
|
|
) VALUES (?, ?, ?, ?, ?, ?, ?)''',
|
|
(
|
|
identity_key, identity, digest, os.path.basename(candidate),
|
|
offset + int(value['relative_offset']), int(value['byte_length']), now,
|
|
),
|
|
)
|
|
indexed_rows += 1
|
|
offset += len(raw_line)
|
|
batch_rows += 1
|
|
batch_bytes += len(raw_line)
|
|
if (
|
|
batch_rows >= max_rows
|
|
or batch_bytes >= max_bytes
|
|
or time.monotonic() - started >= max_seconds
|
|
):
|
|
break
|
|
if batch_rows and (
|
|
batch_rows >= max_rows
|
|
or batch_bytes >= max_bytes
|
|
or time.monotonic() - started >= max_seconds
|
|
):
|
|
break
|
|
_save_projection_reconciliation_progress(ledger, prefix, file_index, offset, indexed_rows)
|
|
complete = file_index >= len(plan)
|
|
if complete:
|
|
ledger.execute('INSERT OR REPLACE INTO publication_meta(key, value) VALUES (?, ?)', (marker, '1'))
|
|
if os.path.getsize(jsonl_ledger_path(path)) > ledger_byte_limit:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'JSONL identity ledger exceeds its {ledger_byte_limit} byte bound'
|
|
)
|
|
ledger.commit()
|
|
return {
|
|
'path': path,
|
|
'identity_key': identity_key,
|
|
'complete': complete,
|
|
'batch_rows': batch_rows,
|
|
'batch_bytes': batch_bytes,
|
|
'indexed_rows': indexed_rows,
|
|
'file_index': file_index,
|
|
'file_count': len(plan),
|
|
'byte_offset': offset,
|
|
}
|
|
except BaseException:
|
|
if ledger is not None:
|
|
ledger.rollback()
|
|
raise
|
|
finally:
|
|
if ledger is not None:
|
|
ledger.close()
|
|
if os.path.exists(jsonl_ledger_path(path)):
|
|
harden_private_file(jsonl_ledger_path(path))
|
|
release_file_lock(lock, lock_path)
|
|
|
|
|
|
def _review_projection_issue_record(
|
|
plan_entry,
|
|
identity_key,
|
|
byte_offset,
|
|
expected_sha256,
|
|
max_record_bytes,
|
|
expected_length=None,
|
|
expected_classification=None,
|
|
):
|
|
if byte_offset >= int(plan_entry['size']):
|
|
raise JsonlProjectionReconciliationRequired('projection issue offset is outside the immutable file')
|
|
with open(plan_entry['path'], 'rb') as handle:
|
|
if byte_offset:
|
|
handle.seek(byte_offset - 1)
|
|
if handle.read(1) != b'\n':
|
|
raise JsonlProjectionReconciliationRequired('projection issue offset is not a record boundary')
|
|
handle.seek(byte_offset)
|
|
raw_line = handle.readline(max_record_bytes + 1)
|
|
if not raw_line:
|
|
raise JsonlProjectionReconciliationRequired('projection issue record is absent')
|
|
if len(raw_line) > max_record_bytes:
|
|
byte_length, actual_sha256, newline_terminated = _stream_projection_record(handle, raw_line)
|
|
classification = 'oversized_record' if newline_terminated else 'unterminated_record'
|
|
raw_line = None
|
|
else:
|
|
byte_length = len(raw_line)
|
|
actual_sha256 = hashlib.sha256(raw_line).hexdigest()
|
|
if identity_key == 'error_row_id':
|
|
try:
|
|
identity = _projection_identity_from_line(raw_line, identity_key, plan_entry['path'])
|
|
if len(identity) > 256 or any(character in identity for character in ('\x00', '\r', '\n')):
|
|
raise ValueError('error projection identity exceeds its bounded format')
|
|
except (JsonlProjectionReconciliationRequired, ValueError):
|
|
classification = (
|
|
'unterminated_record' if not raw_line.endswith(b'\n')
|
|
else 'invalid_error_projection'
|
|
)
|
|
else:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
'reviewed projection issue is a parseable error record and must be indexed'
|
|
)
|
|
else:
|
|
parsed_values, classification = _bounded_projection_json_values(raw_line)
|
|
if parsed_values is not None:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
'reviewed projection issue contains bounded parseable JSON and must be indexed'
|
|
)
|
|
if actual_sha256 != expected_sha256:
|
|
raise JsonlProjectionReconciliationRequired('projection issue record SHA-256 does not match review')
|
|
if expected_length is not None and byte_length != int(expected_length):
|
|
raise JsonlProjectionReconciliationRequired('projection issue record length does not match review')
|
|
if expected_classification is not None and classification != str(expected_classification):
|
|
raise JsonlProjectionReconciliationRequired('projection issue classification does not match review')
|
|
return raw_line, byte_length, actual_sha256, classification
|
|
|
|
|
|
def approve_projection_reconciliation_issues(
|
|
path, identity_key, reviewed_issues, max_record_bytes=None, return_details=False,
|
|
):
|
|
"""Approve exact reviewed corrupt records without storing or changing their payloads."""
|
|
if identity_key not in ('scan_event_id', 'finding_uid', 'error_row_id'):
|
|
raise ValueError('unsupported projection reconciliation identity')
|
|
reviewed_issues = list(reviewed_issues or [])
|
|
if not reviewed_issues or len(reviewed_issues) > 100000:
|
|
raise ValueError('projection issue review count is outside its bound')
|
|
path = os.path.abspath(path)
|
|
max_record_bytes = max(
|
|
1024 * 1024,
|
|
int(max_record_bytes or getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)),
|
|
)
|
|
lock_path = f'{path}.lock'
|
|
lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec)
|
|
ledger = None
|
|
try:
|
|
plan = _projection_reconciliation_plan(path)
|
|
plan_by_name = {os.path.basename(item['path']): (index, item) for index, item in enumerate(plan)}
|
|
if len(plan_by_name) != len(plan):
|
|
raise JsonlProjectionReconciliationRequired('projection history contains ambiguous basenames')
|
|
normalized = []
|
|
seen = set()
|
|
for value in reviewed_issues:
|
|
requested_file = str(value.get('file') or value.get('basename') or '')
|
|
physical_file = os.path.basename(requested_file)
|
|
byte_offset = int(value.get('offset', value.get('byte_offset', -1)))
|
|
expected_sha256 = str(value.get('sha256') or '').strip().lower()
|
|
if (
|
|
not physical_file or physical_file != requested_file
|
|
or physical_file not in plan_by_name
|
|
or byte_offset < 0
|
|
or not re.fullmatch(r'[a-f0-9]{64}', expected_sha256)
|
|
):
|
|
raise ValueError('projection issue review metadata is invalid')
|
|
identity = (physical_file, byte_offset, expected_sha256)
|
|
if identity in seen:
|
|
raise ValueError('projection issue review contains duplicate metadata')
|
|
seen.add(identity)
|
|
normalized.append((
|
|
plan_by_name[physical_file][0],
|
|
physical_file,
|
|
byte_offset,
|
|
expected_sha256,
|
|
value.get('length', value.get('byte_length')),
|
|
value.get('classification'),
|
|
))
|
|
ledger = _open_projection_ledger(path)
|
|
details = []
|
|
classifications = Counter()
|
|
for sequence, (_, physical_file, byte_offset, expected_sha256, expected_length, expected_classification) in enumerate(sorted(normalized), 1):
|
|
plan_entry = plan_by_name[physical_file][1]
|
|
raw_line, byte_length, actual_sha256, classification = _review_projection_issue_record(
|
|
plan_entry,
|
|
identity_key,
|
|
byte_offset,
|
|
expected_sha256,
|
|
max_record_bytes,
|
|
expected_length=expected_length,
|
|
expected_classification=expected_classification,
|
|
)
|
|
issue = _record_projection_reconciliation_issue(
|
|
ledger,
|
|
identity_key,
|
|
plan_entry,
|
|
byte_offset,
|
|
raw_line,
|
|
classification,
|
|
resolved=True,
|
|
byte_length=byte_length,
|
|
record_sha256=actual_sha256,
|
|
)
|
|
classifications[classification] += 1
|
|
if return_details:
|
|
details.append({
|
|
'id': int(issue[0]), 'status': issue[1], 'file': physical_file,
|
|
'offset': byte_offset, 'length': byte_length,
|
|
'sha256': actual_sha256, 'classification': classification,
|
|
})
|
|
if sequence % 100 == 0:
|
|
ledger.commit()
|
|
ledger.commit()
|
|
report = {
|
|
'resolved_count': len(normalized),
|
|
'classifications': dict(sorted(classifications.items())),
|
|
}
|
|
if return_details:
|
|
report['details'] = details
|
|
return report
|
|
finally:
|
|
if ledger is not None:
|
|
ledger.close()
|
|
harden_private_file(jsonl_ledger_path(path))
|
|
release_file_lock(lock, lock_path)
|
|
|
|
|
|
def approve_projection_reconciliation_issue(
|
|
path, identity_key, physical_file, byte_offset, expected_sha256, max_record_bytes=None,
|
|
):
|
|
report = approve_projection_reconciliation_issues(
|
|
path,
|
|
identity_key,
|
|
[{
|
|
'file': physical_file,
|
|
'offset': byte_offset,
|
|
'sha256': expected_sha256,
|
|
}],
|
|
max_record_bytes=max_record_bytes,
|
|
return_details=True,
|
|
)
|
|
return report['details'][0]
|
|
|
|
|
|
def _apply_reviewed_projection_conflict_variant(
|
|
ledger,
|
|
handle,
|
|
plan_entry,
|
|
identity_key,
|
|
reviewed,
|
|
max_record_bytes,
|
|
):
|
|
file_name, offset, length, identity_sha256, payload_sha256, field_set_sha256 = reviewed
|
|
if offset >= int(plan_entry['size']):
|
|
raise JsonlProjectionReconciliationRequired('projection conflict offset is outside the immutable file')
|
|
if offset:
|
|
handle.seek(offset - 1)
|
|
if handle.read(1) != b'\n':
|
|
raise JsonlProjectionReconciliationRequired(
|
|
'projection conflict offset is not a record boundary'
|
|
)
|
|
handle.seek(offset)
|
|
raw_line = handle.readline(max_record_bytes + 1)
|
|
if not raw_line or len(raw_line) > max_record_bytes or len(raw_line) != length:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
'projection conflict record is absent or does not match its reviewed length'
|
|
)
|
|
actual_payload_sha256 = hashlib.sha256(raw_line).hexdigest()
|
|
if identity_key == 'error_row_id':
|
|
identity = _projection_identity_from_line(raw_line, identity_key, plan_entry['path'])
|
|
actual_field_set_sha256 = _error_projection_field_name_set_sha256(raw_line)
|
|
else:
|
|
parsed_values, _ = _bounded_projection_json_values(raw_line)
|
|
if not parsed_values or len(parsed_values) != 1 or not isinstance(parsed_values[0]['value'], dict):
|
|
raise JsonlProjectionReconciliationRequired(
|
|
'projection conflict record is not one bounded JSON identity record'
|
|
)
|
|
identity = str(parsed_values[0]['value'].get(identity_key) or '')
|
|
actual_field_set_sha256 = _projection_field_name_set_sha256(parsed_values[0]['value'])
|
|
if (
|
|
not identity
|
|
or hashlib.sha256(identity.encode('utf-8')).hexdigest() != identity_sha256
|
|
or actual_payload_sha256 != payload_sha256
|
|
or actual_field_set_sha256 != field_set_sha256
|
|
):
|
|
raise JsonlProjectionReconciliationRequired(
|
|
'projection conflict record does not match reviewed identity/payload/schema hashes'
|
|
)
|
|
primary = ledger.execute(
|
|
'''SELECT state FROM publication_identity
|
|
WHERE identity_key = ? AND identity_value = ?''',
|
|
(identity_key, identity),
|
|
).fetchone()
|
|
if primary and primary[0] != 'appended':
|
|
raise JsonlProjectionReconciliationRequired(
|
|
'projection conflict review primary identity is unresolved'
|
|
)
|
|
if not primary:
|
|
row_count = _ledger_row_count(ledger)
|
|
row_limit = max(1, int(getattr(scan_config, 'jsonl_ledger_max_rows', 1000000)))
|
|
if row_count >= row_limit:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'JSONL identity ledger reached its {row_limit} row bound'
|
|
)
|
|
now = time.time()
|
|
ledger.execute(
|
|
'''INSERT INTO publication_identity (
|
|
identity_key, identity_value, payload_sha256, state, file_name,
|
|
byte_offset, byte_length, created_at, updated_at
|
|
) VALUES (?, ?, ?, 'appended', ?, ?, ?, ?, ?)''',
|
|
(
|
|
identity_key, identity, payload_sha256, file_name,
|
|
offset, length, now, now,
|
|
),
|
|
)
|
|
_set_ledger_row_count(ledger, row_count + 1)
|
|
ledger.execute(
|
|
'''INSERT OR IGNORE INTO publication_identity_variant (
|
|
identity_key, identity_value, payload_sha256, file_name,
|
|
byte_offset, byte_length, created_at
|
|
) VALUES (?, ?, ?, ?, ?, ?, ?)''',
|
|
(identity_key, identity, payload_sha256, file_name, offset, length, time.time()),
|
|
)
|
|
_record_projection_variant_issue(
|
|
ledger,
|
|
identity_key,
|
|
identity,
|
|
plan_entry,
|
|
offset,
|
|
length,
|
|
payload_sha256,
|
|
field_set_sha256,
|
|
resolved=True,
|
|
)
|
|
|
|
|
|
def approve_projection_conflict_variants(
|
|
path,
|
|
identity_key,
|
|
reviewed_variants,
|
|
max_record_bytes=None,
|
|
commit_batch_size=250,
|
|
progress_callback=None,
|
|
):
|
|
if identity_key not in ('scan_event_id', 'finding_uid', 'error_row_id'):
|
|
raise ValueError('unsupported projection conflict identity')
|
|
reviewed_variants = list(reviewed_variants or [])
|
|
if not reviewed_variants or len(reviewed_variants) > 100000:
|
|
raise ValueError('projection conflict review count is outside its bound')
|
|
path = os.path.abspath(path)
|
|
max_record_bytes = max(
|
|
1024 * 1024,
|
|
int(max_record_bytes or getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)),
|
|
)
|
|
commit_batch_size = min(1000, max(1, int(commit_batch_size or 250)))
|
|
lock_path = f'{path}.lock'
|
|
lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec)
|
|
ledger = None
|
|
try:
|
|
plan = _projection_reconciliation_plan(path)
|
|
plan_by_name = {os.path.basename(item['path']): (index, item) for index, item in enumerate(plan)}
|
|
if len(plan_by_name) != len(plan):
|
|
raise JsonlProjectionReconciliationRequired('projection history contains ambiguous basenames')
|
|
normalized = []
|
|
seen = set()
|
|
for value in reviewed_variants:
|
|
file_name = str(value.get('file') or '')
|
|
offset = int(value.get('offset', -1))
|
|
length = int(value.get('length', 0))
|
|
identity_sha256 = str(value.get('identity_sha256') or '').lower()
|
|
payload_sha256 = str(value.get('payload_sha256') or '').lower()
|
|
field_set_sha256 = str(value.get('field_name_set_sha256') or '').lower()
|
|
if (
|
|
not file_name or os.path.basename(file_name) != file_name
|
|
or file_name not in plan_by_name or offset < 0 or length <= 0
|
|
or not re.fullmatch(r'[a-f0-9]{64}', identity_sha256)
|
|
or not re.fullmatch(r'[a-f0-9]{64}', payload_sha256)
|
|
or not re.fullmatch(r'[a-f0-9]{64}', field_set_sha256)
|
|
or value.get('classification') != 'historical_payload_variant'
|
|
):
|
|
raise ValueError('projection conflict review metadata is invalid')
|
|
key = (file_name, offset, identity_sha256, payload_sha256)
|
|
if key in seen:
|
|
raise ValueError('projection conflict review contains duplicate metadata')
|
|
seen.add(key)
|
|
normalized.append((
|
|
plan_by_name[file_name][0], file_name,
|
|
(file_name, offset, length, identity_sha256, payload_sha256, field_set_sha256),
|
|
))
|
|
ledger = _open_projection_ledger(path)
|
|
resolved_rows = ledger.execute(
|
|
'''SELECT identity_sha256, file_name, byte_offset, byte_length,
|
|
payload_sha256, field_name_set_sha256
|
|
FROM reconciliation_variant_issue
|
|
WHERE identity_key = ? AND status = 'resolved' ''',
|
|
(identity_key,),
|
|
).fetchall()
|
|
resolved = {
|
|
(row[1], int(row[2]), row[0], row[4]): (int(row[3]), row[5])
|
|
for row in resolved_rows
|
|
}
|
|
pending = []
|
|
already_resolved = 0
|
|
for plan_index, file_name, reviewed in sorted(normalized):
|
|
key = (file_name, reviewed[1], reviewed[3], reviewed[4])
|
|
expected = resolved.get(key)
|
|
if expected is not None:
|
|
if expected != (reviewed[2], reviewed[5]):
|
|
raise JsonlProjectionReconciliationRequired(
|
|
'resolved projection conflict metadata does not match this review manifest'
|
|
)
|
|
already_resolved += 1
|
|
continue
|
|
pending.append((plan_index, file_name, reviewed))
|
|
if progress_callback is not None:
|
|
progress_callback({
|
|
'reviewed': len(normalized),
|
|
'already_resolved': already_resolved,
|
|
'newly_resolved': 0,
|
|
})
|
|
newly_resolved = 0
|
|
grouped = {}
|
|
for plan_index, file_name, reviewed in pending:
|
|
grouped.setdefault((plan_index, file_name), []).append(reviewed)
|
|
for plan_index, file_name in sorted(grouped):
|
|
plan_entry = plan_by_name[file_name][1]
|
|
with open(plan_entry['path'], 'rb') as handle:
|
|
for reviewed in sorted(grouped[(plan_index, file_name)], key=lambda item: item[1]):
|
|
_apply_reviewed_projection_conflict_variant(
|
|
ledger, handle, plan_entry, identity_key, reviewed, max_record_bytes,
|
|
)
|
|
newly_resolved += 1
|
|
if newly_resolved % commit_batch_size == 0:
|
|
ledger.commit()
|
|
if progress_callback is not None:
|
|
progress_callback({
|
|
'reviewed': len(normalized),
|
|
'already_resolved': already_resolved,
|
|
'newly_resolved': newly_resolved,
|
|
})
|
|
ledger.commit()
|
|
if progress_callback is not None and newly_resolved % commit_batch_size:
|
|
progress_callback({
|
|
'reviewed': len(normalized),
|
|
'already_resolved': already_resolved,
|
|
'newly_resolved': newly_resolved,
|
|
})
|
|
return {
|
|
'resolved_variant_count': len(normalized),
|
|
'already_resolved': already_resolved,
|
|
'newly_resolved': newly_resolved,
|
|
}
|
|
finally:
|
|
if ledger is not None:
|
|
ledger.close()
|
|
harden_private_file(jsonl_ledger_path(path))
|
|
release_file_lock(lock, lock_path)
|
|
|
|
|
|
def _files_share_prefix(segment_path, current_path, size):
|
|
if size <= 0 or not os.path.isfile(current_path) or os.path.getsize(current_path) < size:
|
|
return False
|
|
remaining = size
|
|
with open(segment_path, 'rb') as segment, open(current_path, 'rb') as current:
|
|
while remaining:
|
|
amount = min(1024 * 1024, remaining)
|
|
left = segment.read(amount)
|
|
right = current.read(amount)
|
|
if left != right or not left:
|
|
return False
|
|
remaining -= len(left)
|
|
return remaining == 0
|
|
|
|
|
|
def _publish_empty_jsonl_generation(path):
|
|
temporary = (
|
|
f'{path}.{os.getpid()}.{threading.get_ident()}.'
|
|
f'{uuid.uuid4().hex}.empty.tmp'
|
|
)
|
|
try:
|
|
with open(temporary, 'xb') as handle:
|
|
handle.flush()
|
|
os.fsync(handle.fileno())
|
|
harden_private_file(temporary)
|
|
durable_replace(temporary, path)
|
|
harden_private_file(path)
|
|
finally:
|
|
if os.path.exists(temporary):
|
|
os.remove(temporary)
|
|
|
|
|
|
def _remove_file_prefix(path, size):
|
|
current_size = os.path.getsize(path)
|
|
if size <= 0 or current_size < size:
|
|
return
|
|
if current_size == size:
|
|
_publish_empty_jsonl_generation(path)
|
|
return
|
|
temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.prefix.tmp'
|
|
with open(path, 'rb') as source, open(temporary, 'wb') as destination:
|
|
source.seek(size)
|
|
shutil.copyfileobj(source, destination, 1024 * 1024)
|
|
destination.flush()
|
|
os.fsync(destination.fileno())
|
|
harden_private_file(temporary)
|
|
durable_replace(temporary, path)
|
|
harden_private_file(path)
|
|
|
|
|
|
def _segment_manifest_entry(path, sequence):
|
|
return {
|
|
'name': os.path.basename(path),
|
|
'path': os.path.abspath(path),
|
|
'bytes': int(os.path.getsize(path)),
|
|
'closed_at': datetime.now(timezone.utc).isoformat(timespec='seconds'),
|
|
'sequence': int(sequence),
|
|
}
|
|
|
|
|
|
def _jsonl_file_generation(path):
|
|
details = os.stat(path, follow_symlinks=False)
|
|
return {
|
|
'device': int(getattr(details, 'st_dev', 0) or 0),
|
|
'size': int(details.st_size),
|
|
'mtime_ns': int(getattr(details, 'st_mtime_ns', int(details.st_mtime * 1_000_000_000))),
|
|
'inode': int(getattr(details, 'st_ino', 0) or 0),
|
|
}
|
|
|
|
|
|
def _manifest_skip_matches_generation(path, manifest, skip_bytes):
|
|
signature = manifest.get('current_skip_signature') if isinstance(manifest, dict) else None
|
|
if not isinstance(signature, dict) or not os.path.isfile(path):
|
|
return False
|
|
try:
|
|
current = _jsonl_file_generation(path)
|
|
return (
|
|
int(skip_bytes) <= current['size']
|
|
and all(int(current[key]) == int(signature.get(key, -1)) for key in ('device', 'size', 'mtime_ns', 'inode'))
|
|
)
|
|
except (OSError, TypeError, ValueError):
|
|
return False
|
|
|
|
|
|
def reconcile_jsonl_segments(path):
|
|
"""Publish physical orphan segments before any active-file mutation."""
|
|
manifest = load_jsonl_manifest(path)
|
|
physical = _projection_segment_sequence(path)
|
|
existing = {
|
|
str(item.get('name') or os.path.basename(str(item.get('path') or ''))): item
|
|
for item in (manifest.get('segments') or []) if isinstance(item, dict)
|
|
}
|
|
normalized = []
|
|
missing = []
|
|
for sequence, segment_path in physical:
|
|
item = existing.get(os.path.basename(segment_path))
|
|
if item is None:
|
|
item = _segment_manifest_entry(segment_path, sequence)
|
|
missing.append((sequence, segment_path))
|
|
else:
|
|
item = dict(item, path=os.path.abspath(segment_path), name=os.path.basename(segment_path), sequence=sequence)
|
|
normalized.append(item)
|
|
|
|
skip_bytes = max(0, int(manifest.get('current_skip_bytes') or 0))
|
|
duplicate_path = None
|
|
duplicate_size = 0
|
|
candidates = list(reversed(missing))
|
|
if skip_bytes and physical:
|
|
candidates.insert(0, physical[-1])
|
|
for _, segment_path in candidates:
|
|
size = os.path.getsize(segment_path)
|
|
if _files_share_prefix(segment_path, path, size):
|
|
duplicate_path = segment_path
|
|
duplicate_size = size
|
|
break
|
|
|
|
changed = bool(missing) or normalized != (manifest.get('segments') or [])
|
|
effective_skip = duplicate_size if duplicate_path else 0
|
|
if changed or skip_bytes or manifest.get('current_skip_signature'):
|
|
next_sequence = max([sequence for sequence, _ in physical] or [0]) + 1
|
|
manifest.update({
|
|
'current': os.path.basename(path),
|
|
'current_path': os.path.abspath(path),
|
|
'next_sequence': next_sequence,
|
|
'segments': normalized,
|
|
'current_skip_bytes': effective_skip,
|
|
'current_skip_signature': _jsonl_file_generation(path) if effective_skip and os.path.isfile(path) else None,
|
|
'updated_at': datetime.now(timezone.utc).isoformat(timespec='seconds'),
|
|
})
|
|
write_jsonl_manifest(path, manifest)
|
|
if duplicate_path:
|
|
_remove_file_prefix(path, duplicate_size)
|
|
manifest['current_skip_bytes'] = 0
|
|
manifest['current_skip_signature'] = None
|
|
manifest['updated_at'] = datetime.now(timezone.utc).isoformat(timespec='seconds')
|
|
write_jsonl_manifest(path, manifest)
|
|
return manifest
|
|
|
|
|
|
def _segment_consumed_by_registered_keychecks(segment_path):
|
|
if os.path.basename(segment_path).lower().startswith('scan_results.'):
|
|
return True
|
|
keycheck_root = getattr(scan_config, 'keycheck_dir', '') or ''
|
|
if not os.path.isdir(keycheck_root):
|
|
return False
|
|
states = []
|
|
consumer_directories = 0
|
|
with os.scandir(keycheck_root) as entries:
|
|
for entry in entries:
|
|
if not entry.is_dir(follow_symlinks=False):
|
|
continue
|
|
consumer_directories += 1
|
|
if consumer_directories > 128:
|
|
return False
|
|
state_path = os.path.join(entry.path, 'input_state.json')
|
|
if not os.path.isfile(state_path):
|
|
return False
|
|
try:
|
|
if os.path.getsize(state_path) > 1024 * 1024:
|
|
return False
|
|
with open(state_path, 'r', encoding='utf-8') as handle:
|
|
states.append(json.load(handle))
|
|
except (OSError, ValueError):
|
|
return False
|
|
if not states or len(states) != consumer_directories:
|
|
return False
|
|
absolute = os.path.abspath(segment_path)
|
|
size = os.path.getsize(segment_path)
|
|
for state in states:
|
|
files = state.get('files') if isinstance(state, dict) and isinstance(state.get('files'), dict) else {}
|
|
record = files.get(absolute) or files.get(segment_path)
|
|
if not isinstance(record, dict) or int(record.get('offset', 0) or 0) < size:
|
|
return False
|
|
return True
|
|
|
|
|
|
def _prune_consumed_jsonl_segments(path, manifest, keep_below):
|
|
physical = _projection_segment_sequence(path)
|
|
removed = set()
|
|
while len(physical) >= keep_below and physical:
|
|
_, candidate = physical[0]
|
|
if not _segment_consumed_by_registered_keychecks(candidate):
|
|
break
|
|
reject_reparse_components(candidate)
|
|
os.remove(candidate)
|
|
removed.add(os.path.basename(candidate))
|
|
physical.pop(0)
|
|
if removed:
|
|
manifest['segments'] = [
|
|
item for item in (manifest.get('segments') or [])
|
|
if str(item.get('name') or os.path.basename(str(item.get('path') or ''))) not in removed
|
|
]
|
|
manifest['updated_at'] = datetime.now(timezone.utc).isoformat(timespec='seconds')
|
|
write_jsonl_manifest(path, manifest)
|
|
return physical
|
|
|
|
|
|
def rotate_jsonl_if_needed(path, max_bytes, ledger=None):
|
|
if not max_bytes or max_bytes <= 0:
|
|
return
|
|
if not os.path.lexists(path):
|
|
return
|
|
reject_reparse_components(path)
|
|
details = os.stat(path, follow_symlinks=False)
|
|
if not stat.S_ISREG(details.st_mode) or details.st_size < max_bytes:
|
|
return
|
|
repair_jsonl_tail(path)
|
|
manifest = reconcile_jsonl_segments(path)
|
|
max_segments = max(1, int(getattr(scan_config, 'jsonl_max_segments', 16)))
|
|
physical = _prune_consumed_jsonl_segments(path, manifest, max_segments)
|
|
if len(physical) >= max_segments:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'JSONL segment retention bound reached for {path}; registered consumers must catch up '
|
|
'or offline reconciliation must retire acknowledged segments'
|
|
)
|
|
segment_path, seq = next_jsonl_segment_path(path, manifest)
|
|
size = os.path.getsize(path)
|
|
temporary = f'{segment_path}.{os.getpid()}.{threading.get_ident()}.tmp'
|
|
try:
|
|
with open(path, 'rb') as source, open(temporary, 'xb') as destination:
|
|
shutil.copyfileobj(source, destination, 1024 * 1024)
|
|
destination.flush()
|
|
os.fsync(destination.fileno())
|
|
harden_private_file(temporary)
|
|
os.replace(temporary, segment_path)
|
|
finally:
|
|
if os.path.exists(temporary):
|
|
os.remove(temporary)
|
|
harden_private_file(segment_path)
|
|
segments = manifest.get('segments') if isinstance(manifest.get('segments'), list) else []
|
|
segments.append(_segment_manifest_entry(segment_path, seq))
|
|
manifest.update({
|
|
'current': os.path.basename(path),
|
|
'current_path': path,
|
|
'next_sequence': seq + 1,
|
|
'max_bytes': int(max_bytes),
|
|
'segments': segments,
|
|
'current_skip_bytes': int(size),
|
|
'current_skip_signature': _jsonl_file_generation(path),
|
|
'updated_at': datetime.now(timezone.utc).isoformat(timespec='seconds'),
|
|
})
|
|
write_jsonl_manifest(path, manifest)
|
|
if ledger is not None:
|
|
ledger.execute(
|
|
"UPDATE publication_identity SET file_name = ? WHERE file_name = ? AND state = 'appended'",
|
|
(os.path.basename(segment_path), os.path.basename(path)),
|
|
)
|
|
ledger.execute(
|
|
"UPDATE publication_identity_variant SET file_name = ? WHERE file_name = ?",
|
|
(os.path.basename(segment_path), os.path.basename(path)),
|
|
)
|
|
ledger.commit()
|
|
_publish_empty_jsonl_generation(path)
|
|
manifest['current_skip_bytes'] = 0
|
|
manifest['current_skip_signature'] = None
|
|
manifest['updated_at'] = datetime.now(timezone.utc).isoformat(timespec='seconds')
|
|
write_jsonl_manifest(path, manifest)
|
|
logger.info(f'Rotated JSONL {path} -> {segment_path} ({size} bytes)')
|
|
|
|
|
|
def append_rotating_jsonl(path, payload, max_mb=None):
|
|
if not scan_config.jsonl_rotation_enabled:
|
|
return append_jsonl(path, payload)
|
|
max_bytes = int(max_mb or 0) * 1024 * 1024
|
|
serialized = json.dumps(payload, ensure_ascii=False, default=str) + '\n'
|
|
lock_path = f'{path}.lock'
|
|
fd = None
|
|
try:
|
|
parent = os.path.dirname(path)
|
|
if parent:
|
|
require_private_directory(parent, create=True)
|
|
fd = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec)
|
|
repair_jsonl_tail(path)
|
|
rotate_jsonl_if_needed(path, max_bytes)
|
|
with open(path, 'ab') as f:
|
|
f.write(serialized.encode('utf-8'))
|
|
f.flush()
|
|
os.fsync(f.fileno())
|
|
harden_private_file(path)
|
|
return True
|
|
except Exception as e:
|
|
logger.error(f'Unable to rotating-write {path}: {str(e)}')
|
|
return False
|
|
finally:
|
|
if fd is not None:
|
|
release_file_lock(fd, lock_path)
|
|
|
|
|
|
def projection_segment_paths(path):
|
|
paths = [segment_path for _, segment_path in _projection_segment_sequence(path)]
|
|
if os.path.isfile(path):
|
|
paths.append(path)
|
|
return paths
|
|
|
|
|
|
def jsonl_projection_contains(path, identity_key, identity_value):
|
|
expected = str(identity_value or '')
|
|
if not expected:
|
|
return False
|
|
ledger_path = jsonl_ledger_path(path)
|
|
if not os.path.isfile(ledger_path) or not private_file_ready(ledger_path):
|
|
return False
|
|
connection = sqlite3.connect(f'file:{ledger_path}?mode=ro', uri=True, timeout=5)
|
|
try:
|
|
row = connection.execute(
|
|
'''SELECT state FROM publication_identity
|
|
WHERE identity_key = ? AND identity_value = ?''',
|
|
(identity_key, expected),
|
|
).fetchone()
|
|
return bool(row and row[0] == 'appended')
|
|
finally:
|
|
connection.close()
|
|
|
|
|
|
def _append_projection_once_locked(
|
|
path, serialized, identity_key, identity_value, max_bytes, rotation_enabled=None,
|
|
):
|
|
repair_jsonl_tail(path)
|
|
reconcile_jsonl_segments(path)
|
|
if len(identity_key) > 64 or len(identity_value) > 256 or any(
|
|
character in identity_value for character in ('\x00', '\r', '\n')
|
|
):
|
|
raise JsonlProjectionReconciliationRequired('projection identity exceeds its bounded format')
|
|
max_record_bytes = max(1, int(getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)))
|
|
if len(serialized) > max_record_bytes:
|
|
raise JsonlProjectionReconciliationRequired('projection record exceeds its per-record byte bound')
|
|
ledger = _open_projection_ledger(path)
|
|
try:
|
|
_ensure_projection_ledger_bootstrapped(ledger, path, identity_key)
|
|
_recover_prepared_publications(ledger, path)
|
|
digest = hashlib.sha256(serialized).hexdigest()
|
|
row = ledger.execute(
|
|
'''SELECT payload_sha256, state FROM publication_identity
|
|
WHERE identity_key = ? AND identity_value = ?''',
|
|
(identity_key, identity_value),
|
|
).fetchone()
|
|
if row:
|
|
if row[1] != 'appended':
|
|
raise JsonlProjectionReconciliationRequired('publication ledger retained an unresolved append state')
|
|
variant = ledger.execute(
|
|
'''SELECT 1 FROM publication_identity_variant
|
|
WHERE identity_key = ? AND identity_value = ? AND payload_sha256 = ?''',
|
|
(identity_key, identity_value, digest),
|
|
).fetchone()
|
|
if variant:
|
|
return True
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'{identity_key} {identity_value} has a conflicting projection payload'
|
|
)
|
|
|
|
row_limit = max(1, int(getattr(scan_config, 'jsonl_ledger_max_rows', 1000000)))
|
|
ledger_byte_limit = max(1024 * 1024, int(getattr(scan_config, 'jsonl_ledger_max_bytes', 512 * 1024 * 1024)))
|
|
row_count = _ledger_row_count(ledger)
|
|
if row_count >= row_limit:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'JSONL identity ledger reached its {row_limit} row bound; run offline reconciliation'
|
|
)
|
|
ledger_file = jsonl_ledger_path(path)
|
|
if os.path.getsize(ledger_file) + 8192 > ledger_byte_limit:
|
|
raise JsonlProjectionReconciliationRequired(
|
|
f'JSONL identity ledger reached its {ledger_byte_limit} byte bound; run offline reconciliation'
|
|
)
|
|
should_rotate = scan_config.jsonl_rotation_enabled if rotation_enabled is None else bool(rotation_enabled)
|
|
if should_rotate:
|
|
rotate_jsonl_if_needed(path, max_bytes, ledger=ledger)
|
|
if not os.path.exists(path):
|
|
with open(path, 'ab'):
|
|
pass
|
|
harden_private_file(path)
|
|
offset = os.path.getsize(path)
|
|
now = time.time()
|
|
ledger.execute(
|
|
'''INSERT INTO publication_identity (
|
|
identity_key, identity_value, payload_sha256, state, file_name,
|
|
byte_offset, byte_length, created_at, updated_at
|
|
) VALUES (?, ?, ?, 'prepared', ?, ?, ?, ?, ?)''',
|
|
(
|
|
identity_key, identity_value, digest, os.path.basename(path),
|
|
offset, len(serialized), now, now,
|
|
),
|
|
)
|
|
_set_ledger_row_count(ledger, row_count + 1)
|
|
ledger.commit()
|
|
with open(path, 'ab') as handle:
|
|
handle.write(serialized)
|
|
handle.flush()
|
|
os.fsync(handle.fileno())
|
|
harden_private_file(path)
|
|
ledger.execute(
|
|
'''UPDATE publication_identity SET state = 'appended', updated_at = ?
|
|
WHERE identity_key = ? AND identity_value = ? AND state = 'prepared' ''',
|
|
(time.time(), identity_key, identity_value),
|
|
)
|
|
ledger.execute(
|
|
'''INSERT INTO publication_identity_variant (
|
|
identity_key, identity_value, payload_sha256, file_name,
|
|
byte_offset, byte_length, created_at
|
|
) VALUES (?, ?, ?, ?, ?, ?, ?)''',
|
|
(
|
|
identity_key, identity_value, digest, os.path.basename(path),
|
|
offset, len(serialized), time.time(),
|
|
),
|
|
)
|
|
ledger.commit()
|
|
return True
|
|
finally:
|
|
ledger.close()
|
|
if os.path.exists(jsonl_ledger_path(path)):
|
|
harden_private_file(jsonl_ledger_path(path))
|
|
|
|
|
|
def append_rotating_jsonl_once(path, payload, identity_key, identity_value, max_mb=None):
|
|
if not identity_value:
|
|
logger.error(f'Unable to publish {path}: missing {identity_key}')
|
|
return False
|
|
max_bytes = int(max_mb or 0) * 1024 * 1024
|
|
lock_path = f'{path}.lock'
|
|
lock = None
|
|
try:
|
|
require_private_directory(os.path.dirname(os.path.abspath(path)), create=True)
|
|
lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec)
|
|
serialized = (json.dumps(payload, ensure_ascii=False, default=str) + '\n').encode('utf-8')
|
|
return _append_projection_once_locked(
|
|
path, serialized, identity_key, str(identity_value), max_bytes,
|
|
)
|
|
except Exception as exc:
|
|
logger.error(f'Unable to idempotently publish {path}: {exc}')
|
|
return False
|
|
finally:
|
|
if lock is not None:
|
|
release_file_lock(lock, lock_path)
|
|
|
|
|
|
def _rotate_bounded_text_log(path, keep, max_bytes=None):
|
|
if not os.path.exists(path) or os.path.getsize(path) <= 0:
|
|
return
|
|
if max_bytes and os.path.getsize(path) > int(max_bytes):
|
|
with open(path, 'r+b') as handle:
|
|
handle.truncate(int(max_bytes))
|
|
handle.flush()
|
|
os.fsync(handle.fileno())
|
|
segment, _ = next_jsonl_segment_path(path, {})
|
|
os.replace(path, segment)
|
|
harden_private_file(segment)
|
|
with open(path, 'ab'):
|
|
pass
|
|
harden_private_file(path)
|
|
segments = [candidate for candidate in projection_segment_paths(path) if candidate != path]
|
|
segments.sort(key=lambda candidate: os.path.getmtime(candidate), reverse=True)
|
|
for candidate in segments[max(0, int(keep or 0)):]:
|
|
reject_reparse_components(candidate)
|
|
os.remove(candidate)
|
|
|
|
|
|
def append_scan_errors_once(path, result, max_mb=None, keep=5):
|
|
event_id = str(result.get('scan_event_id') or '')
|
|
if not event_id:
|
|
logger.error(f'Unable to publish {path}: missing scan_event_id')
|
|
return False
|
|
errors = list(result.get('errors') or [])
|
|
if not errors:
|
|
return True
|
|
max_bytes = max(1, int(max_mb or 0) * 1024 * 1024)
|
|
timestamp = result.get('timestamp') or result.get('scan_started_at') or event_id
|
|
scan_type = result.get('scan_type', '')
|
|
target = result.get('target', '')
|
|
lock_path = f'{path}.lock'
|
|
lock = None
|
|
try:
|
|
require_private_directory(os.path.dirname(os.path.abspath(path)), create=True)
|
|
lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec)
|
|
repair_jsonl_tail(path)
|
|
for index, error in enumerate(errors, 1):
|
|
row_id = f'{event_id}:{index}'
|
|
line = f'{row_id}\t{timestamp}\t{scan_type}\t{target}\t{error}\n'.encode('utf-8', errors='replace')
|
|
if len(line) > max_bytes:
|
|
suffix = b'...[truncated]\n'
|
|
line = line[:max(0, max_bytes - len(suffix))] + suffix
|
|
if os.path.exists(path) and os.path.getsize(path) + len(line) > max_bytes:
|
|
_rotate_bounded_text_log(path, keep, max_bytes)
|
|
_append_projection_once_locked(
|
|
path, line, 'error_row_id', row_id, 0, rotation_enabled=False,
|
|
)
|
|
return True
|
|
except Exception as exc:
|
|
logger.error(f'Unable to idempotently publish {path}: {exc}')
|
|
return False
|
|
finally:
|
|
if lock is not None:
|
|
release_file_lock(lock, lock_path)
|
|
|
|
def parse_finding_datetime(value):
|
|
if not value:
|
|
return None
|
|
value = str(value).strip()
|
|
for date_format in ("%Y-%m-%d %H:%M:%S %z", "%Y-%m-%dT%H:%M:%SZ", "%Y-%m-%dT%H:%M:%S%z"):
|
|
try:
|
|
return datetime.strptime(value, date_format)
|
|
except ValueError:
|
|
continue
|
|
return None
|
|
|
|
def get_finding_timestamp(finding):
|
|
data = finding.get('SourceMetadata', {}).get('Data', {})
|
|
if not isinstance(data, dict):
|
|
return None
|
|
for source in data.values():
|
|
if isinstance(source, dict) and source.get('timestamp'):
|
|
return parse_finding_datetime(source.get('timestamp'))
|
|
return None
|
|
|
|
def filter_findings_by_age(findings, max_age_days=None):
|
|
if not max_age_days or max_age_days <= 0:
|
|
return findings, 0, 0
|
|
|
|
cutoff = datetime.now().astimezone() - timedelta(days=max_age_days)
|
|
kept = []
|
|
skipped_old = 0
|
|
skipped_missing = 0
|
|
|
|
for finding in findings:
|
|
timestamp = get_finding_timestamp(finding)
|
|
if timestamp is None:
|
|
skipped_missing += 1
|
|
continue
|
|
if timestamp >= cutoff:
|
|
kept.append(finding)
|
|
else:
|
|
skipped_old += 1
|
|
|
|
return kept, skipped_old, skipped_missing
|
|
|
|
|
|
def filter_dropped_detectors(findings):
|
|
drop = {
|
|
item.strip().lower()
|
|
for item in csv_items(_scan_policy_value('drop_detectors', []))
|
|
if item.strip()
|
|
}
|
|
if not drop:
|
|
return findings, 0, Counter()
|
|
kept = []
|
|
counts = Counter()
|
|
skipped = 0
|
|
for finding in findings or []:
|
|
detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '')
|
|
if detector.lower() in drop:
|
|
skipped += 1
|
|
counts[detector or '(unknown)'] += 1
|
|
continue
|
|
kept.append(finding)
|
|
return kept, skipped, counts
|
|
|
|
GITHUB_TOKEN_PREFIXES = ('ghp_', 'gho_', 'ghu_', 'ghs_', 'ghr_', 'github_pat_')
|
|
GITLAB_TOKEN_PREFIXES = ('glpat-', 'gloas-', 'glcbt-', 'glimt-', 'glrt-', 'glft-', 'glsoat-')
|
|
|
|
def finding_raw_values(finding):
|
|
values = []
|
|
for key in ('Raw', 'RawV2'):
|
|
value = finding.get(key)
|
|
if value:
|
|
values.append(str(value))
|
|
return values
|
|
|
|
def is_known_provider_token_shape(detector_name, finding):
|
|
values = finding_raw_values(finding)
|
|
detector = str(detector_name or '').lower()
|
|
if detector in ('github', 'githuboauth2'):
|
|
return any(value.startswith(GITHUB_TOKEN_PREFIXES) for value in values)
|
|
if detector == 'gitlab':
|
|
return any(value.startswith(GITLAB_TOKEN_PREFIXES) for value in values)
|
|
return True
|
|
|
|
def filter_noisy_findings(findings):
|
|
if not _scan_policy_value('strict_git_provider_token_filter', True):
|
|
return findings, 0
|
|
|
|
kept = []
|
|
skipped = 0
|
|
for finding in findings:
|
|
detector = finding.get('DetectorName') or ''
|
|
detector_key = str(detector).lower()
|
|
if detector_key in ('github', 'githuboauth2', 'gitlab') and not finding.get('Verified', False):
|
|
if not is_known_provider_token_shape(detector, finding):
|
|
skipped += 1
|
|
continue
|
|
kept.append(finding)
|
|
return kept, skipped
|
|
|
|
CUSTOM_DETECTOR_NAME_ALIASES = {
|
|
'xaicontextafter': 'Xai',
|
|
'zaiglmcontextafter': 'ZaiGLM',
|
|
}
|
|
|
|
|
|
def normalize_custom_detector_names(findings):
|
|
for finding in findings or []:
|
|
detector = str(finding.get('DetectorName') or '')
|
|
extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {}
|
|
custom_name = str(extra.get('name') or '').strip()
|
|
if detector.lower() == 'customregex' and custom_name:
|
|
finding.setdefault('OriginalDetectorName', detector)
|
|
finding['DetectorName'] = CUSTOM_DETECTOR_NAME_ALIASES.get(
|
|
custom_name.lower(), custom_name,
|
|
)
|
|
return findings
|
|
|
|
def apply_finding_filters(results, target_label='target', *, log_target=True):
|
|
emit_client_scan_phase('filtering')
|
|
findings = results.get('findings') or []
|
|
normalize_custom_detector_names(findings)
|
|
findings, dropped, dropped_counts = filter_dropped_detectors(findings)
|
|
if dropped:
|
|
results['findings'] = findings
|
|
results['dropped_detectors_count'] = int(results.get('dropped_detectors_count', 0) or 0) + dropped
|
|
results['dropped_detectors'] = dict(dropped_counts)
|
|
target_context = f' for {target_label}' if log_target else ''
|
|
logger.info(
|
|
f"Dropped {dropped} configured noise detector finding(s)"
|
|
f"{target_context}: {dict(dropped_counts)}"
|
|
)
|
|
filtered, skipped = filter_noisy_findings(findings)
|
|
if skipped:
|
|
results['findings'] = filtered
|
|
results['filtered_findings_count'] = int(results.get('filtered_findings_count', 0) or 0) + skipped
|
|
target_context = f' for {target_label}' if log_target else ''
|
|
logger.info(
|
|
f"Filtered {skipped} noisy unverified Git provider finding(s)"
|
|
f"{target_context}"
|
|
)
|
|
return results
|
|
|
|
def finding_source_location(finding):
|
|
metadata = finding.get('SourceMetadata') or {}
|
|
data = metadata.get('Data') if isinstance(metadata, dict) else {}
|
|
if not isinstance(data, dict):
|
|
return None, None
|
|
for source in data.values():
|
|
if not isinstance(source, dict):
|
|
continue
|
|
file_path = source.get('file') or source.get('path') or source.get('File')
|
|
line = source.get('line') or source.get('Line')
|
|
if file_path:
|
|
try:
|
|
line = int(line) if line else None
|
|
except (TypeError, ValueError):
|
|
line = None
|
|
return file_path, line
|
|
return None, None
|
|
|
|
|
|
def context_enrichment_budget():
|
|
return {
|
|
'started_at': time.monotonic(),
|
|
'max_source_bytes': max(0, int(getattr(scan_config, 'context_enrichment_max_source_bytes', 16 * 1024 * 1024))),
|
|
'max_findings': max(0, int(getattr(scan_config, 'context_enrichment_max_findings', 2000))),
|
|
'max_postman_comparisons': max(0, int(getattr(scan_config, 'context_enrichment_max_postman_comparisons', 200000))),
|
|
'max_elapsed_sec': max(0.0, float(getattr(scan_config, 'context_enrichment_max_elapsed_sec', 5.0))),
|
|
'source_bytes': 0,
|
|
'postman_comparisons': 0,
|
|
'finding_ids': set(),
|
|
'files': {},
|
|
}
|
|
|
|
|
|
def _context_budget_expired(budget):
|
|
return time.monotonic() - budget['started_at'] >= budget['max_elapsed_sec']
|
|
|
|
|
|
def _context_budget_claim_finding(budget, finding):
|
|
identity = id(finding)
|
|
if identity in budget['finding_ids']:
|
|
return True
|
|
if len(budget['finding_ids']) >= budget['max_findings']:
|
|
return False
|
|
budget['finding_ids'].add(identity)
|
|
return True
|
|
|
|
|
|
def _add_context_warning(results, warning_class, detail):
|
|
warning_class = f'context_enrichment_{warning_class}'
|
|
classes = list(results.get('warning_classes') or [])
|
|
if warning_class not in classes:
|
|
warnings = list(results.get('warnings') or [])
|
|
warnings.append(f'Optional context enrichment degraded: {detail}'[:400])
|
|
results['warnings'] = warnings
|
|
classes.append(warning_class)
|
|
results['warning_classes'] = sorted(set(classes))
|
|
results['context_enrichment_degraded'] = True
|
|
results['degraded'] = True
|
|
|
|
|
|
def _read_context_source(file_path, budget):
|
|
key = os.path.normcase(os.path.abspath(os.fspath(file_path)))
|
|
if key in budget['files']:
|
|
return budget['files'][key]
|
|
if _context_budget_expired(budget):
|
|
return None, False, 'elapsed'
|
|
try:
|
|
size = max(0, int(os.path.getsize(file_path)))
|
|
remaining = max(0, budget['max_source_bytes'] - budget['source_bytes'])
|
|
if remaining <= 0 and size:
|
|
return None, False, 'source_bytes'
|
|
amount = min(size, remaining)
|
|
with open(file_path, 'rb') as handle:
|
|
payload = handle.read(amount)
|
|
budget['source_bytes'] += len(payload)
|
|
complete = len(payload) == size
|
|
reason = None if complete else 'source_bytes' if size > remaining else 'read'
|
|
budget['files'][key] = (payload, complete, reason)
|
|
return payload, complete, reason
|
|
except (OSError, TypeError, ValueError):
|
|
return None, False, 'read'
|
|
|
|
|
|
def _nearby_context_from_lines(lines, file_path, line_number=None, radius=20, max_chars=12000):
|
|
if not lines:
|
|
return None
|
|
if line_number and line_number > 0:
|
|
start = max(0, line_number - radius - 1)
|
|
requested_end = line_number + radius
|
|
else:
|
|
start = 0
|
|
requested_end = radius * 2 + 1
|
|
if start >= len(lines):
|
|
return None
|
|
end = min(len(lines), requested_end)
|
|
return {
|
|
'file': file_path,
|
|
'line': line_number,
|
|
'start_line': start + 1,
|
|
'end_line': end,
|
|
'nearby': ''.join(lines[start:end])[:max_chars],
|
|
}
|
|
|
|
|
|
def read_nearby_context(file_path, line_number=None, radius=20, max_chars=12000):
|
|
if not file_path or not os.path.exists(file_path):
|
|
return None
|
|
if line_number and line_number > 0:
|
|
start = max(0, line_number - radius - 1)
|
|
requested_end = line_number + radius
|
|
else:
|
|
start = 0
|
|
requested_end = radius * 2 + 1
|
|
selected = []
|
|
actual_end = 0
|
|
try:
|
|
with open(file_path, 'r', encoding='utf-8', errors='replace') as f:
|
|
for index in range(requested_end):
|
|
line = f.readline(max_chars + 1)
|
|
if not line:
|
|
break
|
|
if not line.endswith('\n') and len(line) > max_chars:
|
|
while True:
|
|
remainder = f.readline(64 * 1024)
|
|
if not remainder or remainder.endswith('\n'):
|
|
break
|
|
actual_end = index + 1
|
|
if index >= start and sum(len(value) for value in selected) < max_chars:
|
|
selected.append(line[:max_chars])
|
|
except OSError:
|
|
return None
|
|
if actual_end == 0:
|
|
return None
|
|
snippet = ''.join(selected)[:max_chars]
|
|
return {
|
|
'file': file_path,
|
|
'line': line_number,
|
|
'start_line': start + 1,
|
|
'end_line': actual_end,
|
|
'nearby': snippet,
|
|
}
|
|
|
|
def attach_nearby_context(results, budget=None):
|
|
budget = budget or context_enrichment_budget()
|
|
grouped = {}
|
|
finding_budget_exhausted = False
|
|
try:
|
|
for finding in results.get('findings') or []:
|
|
if _context_budget_expired(budget):
|
|
_add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained')
|
|
return results
|
|
if not isinstance(finding, dict):
|
|
continue
|
|
file_path, line_number = finding_source_location(finding)
|
|
if not file_path:
|
|
continue
|
|
if not _context_budget_claim_finding(budget, finding):
|
|
finding_budget_exhausted = True
|
|
break
|
|
key = os.path.normcase(os.path.abspath(os.fspath(file_path)))
|
|
grouped.setdefault(key, {'path': file_path, 'findings': []})['findings'].append((finding, line_number))
|
|
|
|
for group in grouped.values():
|
|
if _context_budget_expired(budget):
|
|
_add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained')
|
|
return results
|
|
payload, complete, reason = _read_context_source(group['path'], budget)
|
|
if payload is None:
|
|
if reason in ('elapsed', 'source_bytes'):
|
|
dimension = 'elapsed-time' if reason == 'elapsed' else 'source-byte'
|
|
_add_context_warning(results, 'budget', f'{dimension} budget was exhausted; remaining findings were retained')
|
|
return results
|
|
_add_context_warning(results, 'failure', 'a nearby source file could not be read; findings were retained')
|
|
continue
|
|
lines = payload.decode('utf-8', errors='replace').splitlines(keepends=True)
|
|
for finding, line_number in group['findings']:
|
|
if _context_budget_expired(budget):
|
|
_add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained')
|
|
return results
|
|
context = _nearby_context_from_lines(lines, group['path'], line_number)
|
|
if context:
|
|
finding['ScannerContext'] = context
|
|
if not complete:
|
|
if reason == 'source_bytes':
|
|
_add_context_warning(results, 'budget', 'source-byte budget was exhausted; remaining findings were retained')
|
|
return results
|
|
_add_context_warning(results, 'failure', 'a nearby source file changed during its bounded read; findings were retained')
|
|
if finding_budget_exhausted:
|
|
_add_context_warning(results, 'budget', 'finding-count budget was exhausted; remaining findings were retained')
|
|
except Exception:
|
|
logger.warning('Optional nearby context enrichment failed; parsed findings were retained')
|
|
_add_context_warning(results, 'failure', 'nearby context parsing failed; findings were retained')
|
|
return results
|
|
|
|
|
|
QWEN_ROUTING_CONTEXT_RE = re.compile(
|
|
r'(?:DASHSCOPE_API_KEY|QWEN_API_KEY|dashscope|qwen|model[_-]?studio|bailian)',
|
|
re.IGNORECASE,
|
|
)
|
|
DEEPSEEK_ROUTING_CONTEXT_RE = re.compile(
|
|
r'(?:DEEPSEEK_API_KEY|deepseek|api\.deepseek\.com)',
|
|
re.IGNORECASE,
|
|
)
|
|
KIMI_ROUTING_CONTEXT_RE = re.compile(
|
|
r'(?:MOONSHOT_API_KEY|KIMI_API_KEY|api\.moonshot\.(?:ai|cn)|platform\.kimi\.(?:ai|com))',
|
|
re.IGNORECASE,
|
|
)
|
|
ZAI_ROUTING_CONTEXT_RE = re.compile(
|
|
r'(?:ZAI_API_KEY|GLM_API_KEY|ZHIPUAI_API_KEY|BIGMODEL_API_KEY|api\.z\.ai|'
|
|
r'open\.bigmodel\.cn|zhipuai|chatglm)',
|
|
re.IGNORECASE,
|
|
)
|
|
QWEN_ROUTING_DETECTORS = {'qwendashscope', 'qwen_dashscope', 'dashscope', 'qwen'}
|
|
QWEN_EXPLICIT_ROUTING_DETECTORS = {'qwendashscope', 'qwen_dashscope'}
|
|
DEEPSEEK_ROUTING_DETECTORS = {'deepseek', 'deepseekapikey', 'deepseek_api_key'}
|
|
DEEPSEEK_EXPLICIT_ROUTING_DETECTORS = {'deepseekapikey', 'deepseek_api_key'}
|
|
KIMI_ROUTING_DETECTORS = {'kimimoonshot', 'moonshotai', 'moonshot', 'kimi'}
|
|
KIMI_EXPLICIT_ROUTING_DETECTORS = {'kimimoonshot', 'moonshotai'}
|
|
ZAI_ROUTING_DETECTORS = {'zaiglm'}
|
|
ZAI_EXPLICIT_ROUTING_DETECTORS = {'zaiglm'}
|
|
AMBIGUOUS_QWEN_DEEPSEEK_HINT = 'ambiguous_qwen_deepseek'
|
|
AMBIGUOUS_GENERIC_SK_HINT = 'ambiguous_generic_sk'
|
|
GENERIC_SK_PROVIDERS = {'qwen', 'deepseek', 'kimi', 'zai'}
|
|
GENERIC_SK_PROVIDER_ORDER = ('deepseek', 'zai', 'qwen', 'kimi')
|
|
GENERIC_SK_PROVIDER_HINTS = {
|
|
*GENERIC_SK_PROVIDERS, AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT,
|
|
}
|
|
EXPLICIT_ASSIGNMENT_PROVIDER_HINT_SOURCE = 'explicit_assignment'
|
|
GENERIC_SK_ROUTING_RE = re.compile(r'^sk-[A-Za-z0-9][A-Za-z0-9_-]{20,505}$')
|
|
QWEN_SPECIFIC_ROUTING_RE = re.compile(r'^sk-sp-[A-Za-z0-9][A-Za-z0-9_-]{20,505}$')
|
|
ZAI_SPECIFIC_ROUTING_RE = re.compile(
|
|
r'^(?:zai-[A-Za-z0-9][A-Za-z0-9_-]{20,505}|'
|
|
r'[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{20,505})$'
|
|
)
|
|
|
|
|
|
def provider_routing_context_text(finding, max_context_chars=65536):
|
|
if not isinstance(finding, dict):
|
|
return ''
|
|
context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {}
|
|
parts = []
|
|
remaining = max(0, int(max_context_chars))
|
|
|
|
def add(value):
|
|
nonlocal remaining
|
|
if not value or remaining <= 0:
|
|
return
|
|
text = str(value)[:remaining]
|
|
parts.append(text)
|
|
remaining -= len(text)
|
|
|
|
for key in ('nearby', 'file'):
|
|
add(context.get(key))
|
|
metadata = finding.get('SourceMetadata') if isinstance(finding.get('SourceMetadata'), dict) else {}
|
|
data = metadata.get('Data') if isinstance(metadata.get('Data'), dict) else {}
|
|
for details in data.values():
|
|
if not isinstance(details, dict):
|
|
continue
|
|
for key in ('file', 'repository', 'repo', 'link', 'image'):
|
|
add(details.get(key))
|
|
return '\n'.join(parts)
|
|
|
|
|
|
def provider_routing_detector_names(finding):
|
|
if not isinstance(finding, dict):
|
|
return set()
|
|
detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '').strip().lower()
|
|
extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {}
|
|
custom_name = str(extra.get('name') or '').strip().lower()
|
|
return {name for name in (detector, custom_name) if name}
|
|
|
|
|
|
def is_qwen_routing_detector(finding):
|
|
return bool(provider_routing_detector_names(finding) & QWEN_ROUTING_DETECTORS)
|
|
|
|
|
|
def is_explicit_qwen_routing_detector(finding):
|
|
return bool(provider_routing_detector_names(finding) & QWEN_EXPLICIT_ROUTING_DETECTORS)
|
|
|
|
|
|
def is_deepseek_routing_detector(finding):
|
|
return bool(provider_routing_detector_names(finding) & DEEPSEEK_ROUTING_DETECTORS)
|
|
|
|
|
|
def is_explicit_deepseek_routing_detector(finding):
|
|
return bool(provider_routing_detector_names(finding) & DEEPSEEK_EXPLICIT_ROUTING_DETECTORS)
|
|
|
|
|
|
def is_kimi_routing_detector(finding):
|
|
return bool(provider_routing_detector_names(finding) & KIMI_ROUTING_DETECTORS)
|
|
|
|
|
|
def is_explicit_kimi_routing_detector(finding):
|
|
return bool(provider_routing_detector_names(finding) & KIMI_EXPLICIT_ROUTING_DETECTORS)
|
|
|
|
|
|
def is_zai_routing_detector(finding):
|
|
return bool(provider_routing_detector_names(finding) & ZAI_ROUTING_DETECTORS)
|
|
|
|
|
|
def is_explicit_zai_routing_detector(finding):
|
|
return bool(provider_routing_detector_names(finding) & ZAI_EXPLICIT_ROUTING_DETECTORS)
|
|
|
|
|
|
def is_generic_sk_routing_detector(finding):
|
|
return bool(
|
|
is_qwen_routing_detector(finding)
|
|
or is_deepseek_routing_detector(finding)
|
|
or is_kimi_routing_detector(finding)
|
|
or is_zai_routing_detector(finding)
|
|
)
|
|
|
|
|
|
def provider_routing_hint_evidence(hint):
|
|
if hint == AMBIGUOUS_QWEN_DEEPSEEK_HINT:
|
|
return {'qwen', 'deepseek'}
|
|
if hint == AMBIGUOUS_GENERIC_SK_HINT:
|
|
return set(GENERIC_SK_PROVIDERS)
|
|
return {hint} if hint in GENERIC_SK_PROVIDERS else set()
|
|
|
|
|
|
def ambiguous_provider_routing_hint(evidence):
|
|
evidence = set(evidence)
|
|
if evidence == {'qwen', 'deepseek'}:
|
|
return AMBIGUOUS_QWEN_DEEPSEEK_HINT
|
|
return AMBIGUOUS_GENERIC_SK_HINT
|
|
|
|
|
|
def explicit_provider_routing_evidence(finding):
|
|
evidence = set()
|
|
if is_explicit_qwen_routing_detector(finding):
|
|
evidence.add('qwen')
|
|
if is_explicit_deepseek_routing_detector(finding):
|
|
evidence.add('deepseek')
|
|
if is_explicit_kimi_routing_detector(finding):
|
|
evidence.add('kimi')
|
|
if is_explicit_zai_routing_detector(finding):
|
|
evidence.add('zai')
|
|
context = finding.get('ScannerContext') if isinstance(finding, dict) else None
|
|
if isinstance(context, dict) and context.get('provider_hint_source') == EXPLICIT_ASSIGNMENT_PROVIDER_HINT_SOURCE:
|
|
evidence.update(provider_routing_hint_evidence(context.get('provider_hint')))
|
|
return evidence
|
|
|
|
|
|
def provider_routing_evidence(finding, max_context_chars=65536):
|
|
if not isinstance(finding, dict):
|
|
return set()
|
|
context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {}
|
|
text = provider_routing_context_text(finding, max_context_chars)
|
|
evidence = explicit_provider_routing_evidence(finding)
|
|
if QWEN_ROUTING_CONTEXT_RE.search(text):
|
|
evidence.add('qwen')
|
|
if DEEPSEEK_ROUTING_CONTEXT_RE.search(text):
|
|
evidence.add('deepseek')
|
|
if KIMI_ROUTING_CONTEXT_RE.search(text):
|
|
evidence.add('kimi')
|
|
if ZAI_ROUTING_CONTEXT_RE.search(text):
|
|
evidence.add('zai')
|
|
evidence.update(provider_routing_hint_evidence(context.get('provider_hint')))
|
|
raw_values = finding_raw_values(finding)
|
|
if any(QWEN_SPECIFIC_ROUTING_RE.fullmatch(value) for value in raw_values):
|
|
evidence.add('qwen')
|
|
if any(ZAI_SPECIFIC_ROUTING_RE.fullmatch(value) for value in raw_values):
|
|
evidence.add('zai')
|
|
if (
|
|
not evidence
|
|
and is_generic_sk_routing_detector(finding)
|
|
and any(GENERIC_SK_ROUTING_RE.fullmatch(value) for value in raw_values)
|
|
):
|
|
evidence.update(GENERIC_SK_PROVIDERS)
|
|
return evidence
|
|
|
|
|
|
def derive_provider_routing_hint(
|
|
finding, max_context_chars=65536, evidence=None, explicit_evidence=None,
|
|
):
|
|
if not isinstance(finding, dict):
|
|
return ''
|
|
context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {}
|
|
evidence = set(evidence) if evidence is not None else provider_routing_evidence(finding, max_context_chars)
|
|
explicit_evidence = (
|
|
set(explicit_evidence)
|
|
if explicit_evidence is not None
|
|
else explicit_provider_routing_evidence(finding)
|
|
)
|
|
if len(explicit_evidence) > 1:
|
|
provider_hint = ambiguous_provider_routing_hint(explicit_evidence)
|
|
elif explicit_evidence:
|
|
provider_hint = next(iter(explicit_evidence))
|
|
elif len(evidence) > 1:
|
|
provider_hint = ambiguous_provider_routing_hint(evidence)
|
|
elif evidence:
|
|
provider_hint = next(iter(evidence))
|
|
else:
|
|
provider_hint = ''
|
|
|
|
persisted_context = dict(context)
|
|
if provider_hint:
|
|
persisted_context['provider_hint'] = provider_hint
|
|
else:
|
|
persisted_context.pop('provider_hint', None)
|
|
if explicit_evidence:
|
|
persisted_context['provider_hint_source'] = EXPLICIT_ASSIGNMENT_PROVIDER_HINT_SOURCE
|
|
else:
|
|
persisted_context.pop('provider_hint_source', None)
|
|
if provider_hint in (AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT):
|
|
persisted_context['provider_candidates'] = [
|
|
provider for provider in GENERIC_SK_PROVIDER_ORDER if provider in evidence
|
|
]
|
|
else:
|
|
persisted_context.pop('provider_candidates', None)
|
|
if persisted_context or 'ScannerContext' in finding:
|
|
finding['ScannerContext'] = persisted_context
|
|
return provider_hint
|
|
|
|
|
|
def strip_nearby_context_for_persistence(result):
|
|
findings = result.get('findings') or []
|
|
evidence_by_value = {}
|
|
explicit_evidence_by_value = {}
|
|
for finding in findings:
|
|
if not (
|
|
is_qwen_routing_detector(finding)
|
|
or is_deepseek_routing_detector(finding)
|
|
or is_kimi_routing_detector(finding)
|
|
or is_zai_routing_detector(finding)
|
|
):
|
|
continue
|
|
evidence = provider_routing_evidence(finding)
|
|
explicit_evidence = explicit_provider_routing_evidence(finding)
|
|
for value in finding_raw_values(finding):
|
|
evidence_by_value.setdefault(value, set()).update(evidence)
|
|
explicit_evidence_by_value.setdefault(value, set()).update(explicit_evidence)
|
|
|
|
for finding in findings:
|
|
is_routed_detector = (
|
|
is_qwen_routing_detector(finding)
|
|
or is_deepseek_routing_detector(finding)
|
|
or is_kimi_routing_detector(finding)
|
|
or is_zai_routing_detector(finding)
|
|
)
|
|
evidence = provider_routing_evidence(finding)
|
|
explicit_evidence = explicit_provider_routing_evidence(finding)
|
|
if is_routed_detector:
|
|
for value in finding_raw_values(finding):
|
|
evidence.update(evidence_by_value.get(value, ()))
|
|
explicit_evidence.update(explicit_evidence_by_value.get(value, ()))
|
|
derive_provider_routing_hint(
|
|
finding, evidence=evidence, explicit_evidence=explicit_evidence,
|
|
)
|
|
|
|
for finding in findings:
|
|
postman_context = finding.get('PostmanContext') if isinstance(finding, dict) else None
|
|
if isinstance(postman_context, dict):
|
|
finding['PostmanContext'] = sanitize_postman_context(postman_context)
|
|
context = finding.get('ScannerContext') if isinstance(finding, dict) else None
|
|
if isinstance(context, dict) and 'nearby' in context:
|
|
finding['ScannerContext'] = {key: value for key, value in context.items() if key != 'nearby'}
|
|
if is_foundry_detector(finding):
|
|
text = foundry_finding_text(finding)
|
|
endpoints = [normalize_foundry_endpoint(match) for match in POSTMAN_FOUNDRY_ENDPOINT_RE.findall(text)]
|
|
keys = foundry_candidate_keys(finding, text)
|
|
endpoint = next((item for item in endpoints if item), '')
|
|
key = keys[0] if keys else ''
|
|
if endpoint and key:
|
|
finding['Raw'] = key
|
|
finding['RawV2'] = f'{endpoint}:{key}'
|
|
finding['Redacted'] = f'{endpoint}:***REDACTED***'
|
|
return result
|
|
|
|
|
|
def finding_raw_secret_for_uid(finding):
|
|
for key in ('RawV2', 'Raw'):
|
|
value = finding.get(key)
|
|
if value:
|
|
return str(value)
|
|
structured = finding.get('StructuredData')
|
|
if isinstance(structured, dict):
|
|
for value in structured.values():
|
|
if isinstance(value, str) and value:
|
|
return value
|
|
return ''
|
|
|
|
|
|
def finding_location_for_uid(finding):
|
|
metadata = finding.get('SourceMetadata') if isinstance(finding, dict) else {}
|
|
data = metadata.get('Data') if isinstance(metadata, dict) else {}
|
|
if not isinstance(data, dict):
|
|
return '', '', ''
|
|
for details in data.values():
|
|
if not isinstance(details, dict):
|
|
continue
|
|
file_path = details.get('file') or details.get('path') or details.get('File') or ''
|
|
line_number = details.get('line') or details.get('Line') or ''
|
|
commit_hash = details.get('commit') or details.get('commitHash') or details.get('commit_hash') or ''
|
|
return str(file_path or ''), str(line_number or ''), str(commit_hash or '')
|
|
return '', '', ''
|
|
|
|
|
|
def sha256_json(value):
|
|
return hashlib.sha256(json.dumps(value, ensure_ascii=False, default=str, sort_keys=True).encode('utf-8', errors='replace')).hexdigest()
|
|
|
|
|
|
def _bounded_utf8(value, max_bytes):
|
|
text = ''.join(' ' if ord(character) < 32 else character for character in str(value or ''))
|
|
return text.encode('utf-8', errors='replace')[:max_bytes].decode('utf-8', errors='ignore')
|
|
|
|
|
|
def keycheck_input_line_limit():
|
|
return max(1024, int(getattr(scan_config, 'keycheck_input_max_line_bytes', 16 * 1024 * 1024)))
|
|
|
|
|
|
def finding_projection_payload(finding):
|
|
payload_bytes = json.dumps(finding, ensure_ascii=False, default=str).encode('utf-8')
|
|
line_limit = keycheck_input_line_limit()
|
|
if len(payload_bytes) + 1 <= line_limit:
|
|
return finding, False
|
|
|
|
raw_secret = finding_raw_secret_for_uid(finding)
|
|
metadata = finding.get('SourceMetadata') if isinstance(finding.get('SourceMetadata'), dict) else {}
|
|
data = metadata.get('Data') if isinstance(metadata.get('Data'), dict) else {}
|
|
source_type = next(iter(data), '')
|
|
file_path, line_number, commit_hash = finding_location_for_uid(finding)
|
|
marker = {
|
|
'finding_uid': _bounded_utf8(finding.get('finding_uid'), 256),
|
|
'DetectorName': _bounded_utf8(finding.get('DetectorName') or finding.get('DetectorType'), 64),
|
|
'finding_omitted': True,
|
|
'keycheck_uncheckable': True,
|
|
'omission_reason': 'oversized_finding',
|
|
'secret_sha256': hashlib.sha256(raw_secret.encode('utf-8', errors='replace')).hexdigest() if raw_secret else '',
|
|
'payload_sha256': hashlib.sha256(payload_bytes).hexdigest(),
|
|
'SourceIdentity': {
|
|
'type': _bounded_utf8(source_type, 32),
|
|
'file': _bounded_utf8(file_path, 128),
|
|
'line': _bounded_utf8(line_number, 16),
|
|
'commit': _bounded_utf8(commit_hash, 64),
|
|
'sha256': sha256_json(metadata),
|
|
},
|
|
}
|
|
marker_bytes = (json.dumps(marker, ensure_ascii=False, default=str) + '\n').encode('utf-8')
|
|
if len(marker_bytes) > line_limit:
|
|
marker = {
|
|
'finding_uid': _bounded_utf8(finding.get('finding_uid'), 256),
|
|
'DetectorName': _bounded_utf8(finding.get('DetectorName') or finding.get('DetectorType'), 32),
|
|
'finding_omitted': True,
|
|
'keycheck_uncheckable': True,
|
|
'secret_sha256': marker['secret_sha256'],
|
|
'payload_sha256': marker['payload_sha256'],
|
|
'source_identity_sha256': marker['SourceIdentity']['sha256'],
|
|
}
|
|
marker_bytes = (json.dumps(marker, ensure_ascii=False, default=str) + '\n').encode('utf-8')
|
|
if len(marker_bytes) > line_limit:
|
|
raise RuntimeError('bounded oversized-finding marker exceeds the keycheck input line limit')
|
|
return marker, True
|
|
|
|
|
|
def assign_finding_uids(result):
|
|
findings = result.get('findings') or []
|
|
scan_event_id = str(result.get('scan_event_id') or '')
|
|
if not scan_event_id:
|
|
raise ValueError('event-based finding identities require scan_event_id')
|
|
for index, finding in enumerate(findings, 1):
|
|
if not isinstance(finding, dict) or finding.get('finding_uid'):
|
|
continue
|
|
detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '')
|
|
raw_secret = finding_raw_secret_for_uid(finding)
|
|
secret_hash = hashlib.sha256(raw_secret.encode('utf-8', errors='replace')).hexdigest() if raw_secret else ''
|
|
fallback_hash = sha256_json(finding) if not secret_hash else ''
|
|
file_path, line_number, commit_hash = finding_location_for_uid(finding)
|
|
finding['finding_uid'] = hashlib.sha256('|'.join([
|
|
'truf-finding-v2',
|
|
scan_event_id,
|
|
str(index),
|
|
detector,
|
|
secret_hash or fallback_hash,
|
|
file_path,
|
|
line_number,
|
|
commit_hash,
|
|
]).encode('utf-8', errors='replace')).hexdigest()
|
|
|
|
|
|
def save_scan_result(result):
|
|
"""Persist findings like raw TruffleHog JSONL and keep errors separately."""
|
|
results_dir = get_results_dir()
|
|
if not results_dir:
|
|
return False
|
|
success = True
|
|
|
|
event_id = str(result.get('scan_event_id') or '')
|
|
if not event_id:
|
|
logger.error('Unable to persist scan result without scan_event_id')
|
|
return False
|
|
target = result.get('target', '')
|
|
scan_type = result.get('scan_type', '')
|
|
timestamp = result.get('timestamp', datetime.now().isoformat())
|
|
assign_finding_uids(result)
|
|
try:
|
|
foundry_candidates = write_foundry_keycheck_candidates_from_findings(copy.deepcopy(result))
|
|
if foundry_candidates:
|
|
logger.info(f"Queued {foundry_candidates} Azure Foundry keycheck candidate(s) from {scan_type}:{target}")
|
|
except Exception as e:
|
|
logger.warning(f"Unable to queue Azure Foundry keycheck candidate(s) for {scan_type}:{target}: {str(e)}")
|
|
success = False
|
|
projection_result = copy.deepcopy(result)
|
|
strip_nearby_context_for_persistence(projection_result)
|
|
projected_findings = []
|
|
oversized_count = 0
|
|
for finding in projection_result.get('findings') or []:
|
|
projected, oversized = finding_projection_payload(finding)
|
|
projected_findings.append(projected)
|
|
oversized_count += int(oversized)
|
|
projection_result['findings'] = projected_findings
|
|
if oversized_count:
|
|
projection_result.setdefault('warnings', []).append(
|
|
f'{oversized_count} oversized finding(s) were projected as uncheckable metadata markers; '
|
|
'PostgreSQL retains the authoritative findings'
|
|
)
|
|
projection_result['degraded'] = True
|
|
projection_result['oversized_findings_omitted'] = oversized_count
|
|
|
|
if projected_findings or projection_result.get('errors') or projection_result.get('warnings') or projection_result.get('skipped'):
|
|
success = append_rotating_jsonl_once(
|
|
os.path.join(results_dir, 'scan_results.jsonl'),
|
|
projection_result,
|
|
'scan_event_id',
|
|
event_id,
|
|
scan_config.scan_results_max_mb,
|
|
) and success
|
|
|
|
for finding in projected_findings:
|
|
success = append_rotating_jsonl_once(
|
|
os.path.join(results_dir, 'found_secrets.jsonl'),
|
|
finding,
|
|
'finding_uid',
|
|
finding.get('finding_uid'),
|
|
scan_config.found_secrets_max_mb,
|
|
) and success
|
|
|
|
if projection_result.get('errors'):
|
|
path = os.path.join(results_dir, 'scan_errors.log')
|
|
success = append_scan_errors_once(
|
|
path,
|
|
projection_result,
|
|
scan_config.scan_errors_max_mb,
|
|
scan_config.scan_errors_keep,
|
|
) and success
|
|
return success
|
|
|
|
# ======================
|
|
# DOCKER TOKEN MANAGEMENT
|
|
# ======================
|
|
@dataclass(frozen=True, repr=False)
|
|
class DockerAccount:
|
|
name: str
|
|
username: str
|
|
token: str
|
|
config_dir: str
|
|
|
|
def __repr__(self):
|
|
return f'DockerAccount(name={self.name!r})'
|
|
|
|
|
|
@dataclass(frozen=True, repr=False)
|
|
class DockerRegistryAuth:
|
|
token: str
|
|
account_name: str = ''
|
|
challenge: str = ''
|
|
|
|
|
|
class DockerTokenManager:
|
|
def __init__(self):
|
|
self.tokens = []
|
|
self.token_dirs = []
|
|
self.accounts = []
|
|
self.current_index = 0
|
|
self.request_index = 0
|
|
self.lock = threading.RLock()
|
|
self.config_key = None
|
|
self.cooldown_sec = 1800
|
|
self.cooldown_until = {}
|
|
self.cooldown_categories = {}
|
|
self.hub_tokens = {}
|
|
self.hub_token_locks = {}
|
|
self.invalid_accounts = set()
|
|
self.status_events = {}
|
|
self.explicit_pool = False
|
|
|
|
@staticmethod
|
|
def _account_name(entry, index):
|
|
return str(entry.get('name') or f'docker_{index + 1}').strip()
|
|
|
|
def setup_accounts(
|
|
self, entries, cooldown_sec=1800, explicit_pool=False,
|
|
create_config_dirs=True,
|
|
):
|
|
normalized = []
|
|
seen_names = set()
|
|
for index, entry in enumerate(entries or []):
|
|
if not isinstance(entry, dict):
|
|
continue
|
|
name = self._account_name(entry, index)
|
|
username = str(entry.get('username') or '').strip()
|
|
token = str(entry.get('token') or '').strip()
|
|
if not name or not username or not token or name in seen_names:
|
|
continue
|
|
seen_names.add(name)
|
|
normalized.append((name, username, token))
|
|
config_key = (bool(create_config_dirs),) + tuple(
|
|
(name, username, hashlib.sha256(token.encode('utf-8')).hexdigest())
|
|
for name, username, token in normalized
|
|
)
|
|
|
|
with self.lock:
|
|
self.cooldown_sec = max(60, min(86400, int(cooldown_sec or 1800)))
|
|
if (
|
|
config_key == self.config_key
|
|
and self.explicit_pool == bool(explicit_pool)
|
|
and (
|
|
not normalized or not create_config_dirs
|
|
or all(
|
|
os.path.isfile(os.path.join(account.config_dir, 'config.json'))
|
|
for account in self.accounts
|
|
)
|
|
)
|
|
):
|
|
return
|
|
self._cleanup_locked()
|
|
self.config_key = config_key
|
|
self.current_index = 0
|
|
self.request_index = 0
|
|
self.cooldown_until = {}
|
|
self.cooldown_categories = {}
|
|
self.hub_tokens = {}
|
|
self.hub_token_locks = {}
|
|
self.invalid_accounts = set()
|
|
self.status_events = {}
|
|
self.explicit_pool = bool(explicit_pool)
|
|
|
|
for name, username, token in normalized:
|
|
temp_dir = ''
|
|
if create_config_dirs:
|
|
temp_dir = create_docker_config_dir()
|
|
auth = base64.b64encode(f'{username}:{token}'.encode()).decode()
|
|
docker_auth = {'auth': auth}
|
|
config = {
|
|
'auths': {
|
|
'https://index.docker.io/v1/': docker_auth,
|
|
'index.docker.io': docker_auth,
|
|
'registry-1.docker.io': docker_auth,
|
|
'https://registry-1.docker.io': docker_auth,
|
|
'docker.io': docker_auth,
|
|
}
|
|
}
|
|
config_path = os.path.join(temp_dir, 'config.json')
|
|
with open(config_path, 'w') as f:
|
|
json.dump(config, f)
|
|
harden_private_file(config_path)
|
|
account = DockerAccount(name, username, token, temp_dir)
|
|
self.accounts.append(account)
|
|
self.tokens.append(f'{username}:{token}')
|
|
if temp_dir:
|
|
self.token_dirs.append(temp_dir)
|
|
logger.info('Configured Docker account: %s', name)
|
|
|
|
def setup_tokens(self, tokens_str=None, username=None, create_config_dirs=True):
|
|
"""Initialize Docker tokens from environment variable"""
|
|
username = (username or os.getenv('DOCKERHUB_USERNAME') or os.getenv('DOCKER_USERNAME') or '').strip()
|
|
tokens_str = tokens_str if tokens_str is not None else os.getenv('DOCKER_TOKENS', '')
|
|
|
|
if not tokens_str:
|
|
single_token = os.getenv('DOCKERHUB_TOKEN') or os.getenv('DOCKER_TOKEN')
|
|
if single_token:
|
|
tokens_str = single_token
|
|
|
|
entries = []
|
|
if tokens_str:
|
|
for index, raw_token in enumerate(tokens_str.replace('\n', ',').split(',')):
|
|
raw_token = raw_token.strip()
|
|
if not raw_token:
|
|
continue
|
|
if ':' in raw_token:
|
|
account_username, token = raw_token.split(':', 1)
|
|
elif username:
|
|
account_username, token = username, raw_token
|
|
else:
|
|
logger.warning('Docker token without username ignored. Use username:token or set DOCKERHUB_USERNAME.')
|
|
continue
|
|
account_username = str(account_username).strip()
|
|
token = str(token).strip()
|
|
if account_username and token:
|
|
entries.append({
|
|
'name': f'docker_{index + 1}',
|
|
'username': account_username,
|
|
'token': token,
|
|
})
|
|
self.setup_accounts(
|
|
entries, explicit_pool=False, create_config_dirs=create_config_dirs,
|
|
)
|
|
|
|
def has_accounts(self):
|
|
with self.lock:
|
|
return bool(self.accounts)
|
|
|
|
def account_count(self):
|
|
with self.lock:
|
|
return len(self.accounts)
|
|
|
|
def next_account(self, endpoint, excluded_names=None):
|
|
excluded_names = set(excluded_names or ())
|
|
with self.lock:
|
|
if not self.accounts:
|
|
return None
|
|
now = time.time()
|
|
for offset in range(len(self.accounts)):
|
|
position = (self.request_index + offset) % len(self.accounts)
|
|
account = self.accounts[position]
|
|
if account.name in excluded_names:
|
|
continue
|
|
if account.name in self.invalid_accounts:
|
|
continue
|
|
if float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now:
|
|
continue
|
|
self.request_index = (position + 1) % len(self.accounts)
|
|
return account
|
|
return None
|
|
|
|
def report_http_status(self, account, endpoint, status, response=None, category=None):
|
|
account_name = account.name if isinstance(account, DockerAccount) else str(account or '')
|
|
if not account_name:
|
|
return False
|
|
status = int(status or 0)
|
|
category = str(category or ('rate_limit' if status == 429 else 'auth_forbidden'))
|
|
with self.lock:
|
|
if account_name in self.invalid_accounts and category != 'auth_invalid':
|
|
return self._all_unavailable_locked(endpoint)
|
|
if category == 'auth_invalid':
|
|
self.invalid_accounts.add(account_name)
|
|
retry_at = float('inf')
|
|
reset_at = 'manual'
|
|
else:
|
|
delay = dockerhub_retry_after_seconds(response) if status == 429 else self.cooldown_sec
|
|
retry_at = time.time() + max(60, int(delay))
|
|
reset_at = datetime.fromtimestamp(retry_at, timezone.utc).isoformat(timespec='seconds')
|
|
self.cooldown_until[(account_name, endpoint)] = retry_at
|
|
self.cooldown_categories[(account_name, endpoint)] = category
|
|
if status == 401 or category == 'auth_invalid':
|
|
self.hub_tokens.pop(account_name, None)
|
|
self.status_events[(account_name, endpoint)] = {
|
|
'name': account_name,
|
|
'endpoint': endpoint,
|
|
'category': category,
|
|
'reset_at': reset_at,
|
|
'message': f'Docker {endpoint} HTTP {status}',
|
|
}
|
|
return self._all_unavailable_locked(endpoint)
|
|
|
|
def report_success(self, account, endpoint):
|
|
account_name = account.name if isinstance(account, DockerAccount) else str(account or '')
|
|
if not account_name:
|
|
return
|
|
with self.lock:
|
|
if account_name in self.invalid_accounts:
|
|
return
|
|
expires_at = float(self.cooldown_until.get((account_name, endpoint), 0) or 0)
|
|
if expires_at <= time.time():
|
|
self.cooldown_until.pop((account_name, endpoint), None)
|
|
self.cooldown_categories.pop((account_name, endpoint), None)
|
|
event_key = (account_name, endpoint)
|
|
if event_key not in self.status_events:
|
|
self.status_events[event_key] = {
|
|
'name': account_name,
|
|
'endpoint': endpoint,
|
|
'category': 'ok',
|
|
'reset_at': None,
|
|
'message': '',
|
|
}
|
|
|
|
def _all_unavailable_locked(self, endpoint):
|
|
if not self.accounts:
|
|
return False
|
|
now = time.time()
|
|
return all(
|
|
account.name in self.invalid_accounts
|
|
or float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now
|
|
for account in self.accounts
|
|
)
|
|
|
|
def all_unavailable(self, endpoint):
|
|
with self.lock:
|
|
return self._all_unavailable_locked(endpoint)
|
|
|
|
def rate_limit_contributes_to_exhaustion(self, endpoint):
|
|
with self.lock:
|
|
if not self._all_unavailable_locked(endpoint):
|
|
return False
|
|
now = time.time()
|
|
return any(
|
|
float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now
|
|
and self.cooldown_categories.get((account.name, endpoint)) == 'rate_limit'
|
|
for account in self.accounts
|
|
)
|
|
|
|
def uses_explicit_pool(self):
|
|
with self.lock:
|
|
return bool(self.explicit_pool)
|
|
|
|
def seconds_until_available(self, endpoint):
|
|
with self.lock:
|
|
now = time.time()
|
|
deadlines = [
|
|
float(self.cooldown_until.get((account.name, endpoint), 0) or 0)
|
|
for account in self.accounts
|
|
if float(self.cooldown_until.get((account.name, endpoint), 0) or 0) != float('inf')
|
|
and float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now
|
|
]
|
|
if not deadlines:
|
|
return self.cooldown_sec
|
|
return max(60, math.ceil(min(deadlines) - now))
|
|
|
|
def account_available(self, account_name, endpoint):
|
|
with self.lock:
|
|
return account_name not in self.invalid_accounts and float(
|
|
self.cooldown_until.get((account_name, endpoint), 0) or 0
|
|
) <= time.time()
|
|
|
|
def hub_token_lock(self, account_name):
|
|
with self.lock:
|
|
lock = self.hub_token_locks.get(account_name)
|
|
if lock is None:
|
|
lock = threading.Lock()
|
|
self.hub_token_locks[account_name] = lock
|
|
return lock
|
|
|
|
def restore_endpoint_cooldowns(self, endpoint_status):
|
|
if not isinstance(endpoint_status, dict):
|
|
return
|
|
with self.lock:
|
|
account_names = {account.name for account in self.accounts}
|
|
now = time.time()
|
|
for endpoint, accounts in endpoint_status.items():
|
|
if endpoint not in {'hub_search', 'hub_tags', 'registry'} or not isinstance(accounts, dict):
|
|
continue
|
|
for account_name, status in accounts.items():
|
|
if account_name not in account_names or not isinstance(status, dict):
|
|
continue
|
|
disabled_until = status.get('disabled_until')
|
|
if disabled_until == 'manual':
|
|
retry_at = float('inf')
|
|
else:
|
|
try:
|
|
retry_at = datetime.fromisoformat(
|
|
str(disabled_until).replace('Z', '+00:00')
|
|
).timestamp()
|
|
except (TypeError, ValueError, OverflowError):
|
|
continue
|
|
if retry_at <= now:
|
|
continue
|
|
key = (account_name, endpoint)
|
|
category = str(status.get('disabled_reason') or 'rate_limit')
|
|
if category == 'auth_invalid':
|
|
self.invalid_accounts.add(account_name)
|
|
if retry_at > float(self.cooldown_until.get(key, 0) or 0):
|
|
self.cooldown_until[key] = retry_at
|
|
self.cooldown_categories[key] = category
|
|
|
|
def cached_hub_token(self, account_name):
|
|
with self.lock:
|
|
token, expires_at = self.hub_tokens.get(account_name, ('', 0))
|
|
if token and float(expires_at or 0) > time.time() + 30:
|
|
return token
|
|
self.hub_tokens.pop(account_name, None)
|
|
return ''
|
|
|
|
def cache_hub_token(self, account_name, token, expires_in):
|
|
with self.lock:
|
|
lifetime = max(60, min(600, int(expires_in or 600)))
|
|
self.hub_tokens[account_name] = (token, time.time() + lifetime)
|
|
|
|
def invalidate_hub_token(self, account_name):
|
|
with self.lock:
|
|
self.hub_tokens.pop(account_name, None)
|
|
|
|
def drain_status_events(self):
|
|
with self.lock:
|
|
events = list(self.status_events.values())
|
|
self.status_events = {}
|
|
return events
|
|
|
|
def get_next_config(self):
|
|
"""Get the next Docker config directory in rotation"""
|
|
with self.lock:
|
|
if not self.accounts:
|
|
return None
|
|
for offset in range(len(self.accounts)):
|
|
position = (self.current_index + offset) % len(self.accounts)
|
|
account = self.accounts[position]
|
|
if account.name in self.invalid_accounts:
|
|
continue
|
|
self.current_index = (position + 1) % len(self.accounts)
|
|
return account.config_dir
|
|
return None
|
|
|
|
def _cleanup_locked(self):
|
|
for directory in self.token_dirs:
|
|
try:
|
|
cleanup_command_work_dir(directory)
|
|
except Exception as exc:
|
|
logger.error('Error cleaning Docker config: %s', str(exc))
|
|
self.tokens = []
|
|
self.token_dirs = []
|
|
self.accounts = []
|
|
|
|
def cleanup(self):
|
|
"""Clean up temporary Docker config directories"""
|
|
with self.lock:
|
|
self._cleanup_locked()
|
|
self.config_key = None
|
|
self.cooldown_until = {}
|
|
self.cooldown_categories = {}
|
|
self.hub_tokens = {}
|
|
self.hub_token_locks = {}
|
|
self.invalid_accounts = set()
|
|
self.status_events = {}
|
|
self.explicit_pool = False
|
|
|
|
docker_token_manager = DockerTokenManager()
|
|
|
|
def configure_docker_tokens(tokens_str=None, username=None):
|
|
"""Reconfigure Docker auth tokens after UI or CLI input."""
|
|
require_scanner_runtime_initialized()
|
|
docker_token_manager.setup_tokens(tokens_str, username)
|
|
|
|
|
|
def configure_docker_accounts(entries, cooldown_sec=1800):
|
|
require_scanner_runtime_initialized()
|
|
docker_token_manager.setup_accounts(
|
|
entries, cooldown_sec=cooldown_sec, explicit_pool=True,
|
|
)
|
|
|
|
|
|
def configure_docker_discovery_tokens(tokens_str=None, username=None):
|
|
docker_token_manager.setup_tokens(
|
|
tokens_str, username, create_config_dirs=False,
|
|
)
|
|
|
|
|
|
def configure_docker_discovery_accounts(entries, cooldown_sec=1800):
|
|
docker_token_manager.setup_accounts(
|
|
entries, cooldown_sec=cooldown_sec, explicit_pool=True,
|
|
create_config_dirs=False,
|
|
)
|
|
|
|
|
|
def restore_docker_endpoint_cooldowns(endpoint_status):
|
|
docker_token_manager.restore_endpoint_cooldowns(endpoint_status)
|
|
|
|
|
|
def drain_docker_auth_events():
|
|
return docker_token_manager.drain_status_events()
|
|
|
|
# ===================
|
|
# API FETCH FUNCTIONS
|
|
# ===================
|
|
def github_repo_to_target(item):
|
|
return {
|
|
'url': item.get('clone_url', ''),
|
|
'name': item.get('full_name', ''),
|
|
'created_at': item.get('created_at', ''),
|
|
'updated_at': item.get('updated_at', ''),
|
|
'pushed_at': item.get('pushed_at', ''),
|
|
}
|
|
|
|
|
|
def github_headers(token=None):
|
|
headers = {'Accept': 'application/vnd.github+json', 'User-Agent': 'GitSecretsScanner/2.0'}
|
|
if token:
|
|
headers['Authorization'] = f'Bearer {token}'
|
|
return headers
|
|
|
|
|
|
def gitlab_headers(token=None):
|
|
headers = {'User-Agent': 'GitSecretsScanner/2.0'}
|
|
if token:
|
|
headers['PRIVATE-TOKEN'] = token
|
|
return headers
|
|
|
|
|
|
def parse_github_repo_target(target):
|
|
if isinstance(target, dict):
|
|
repo = target.get('repo') or target.get('full_name') or ''
|
|
if not repo and target.get('repo_url'):
|
|
candidate = normalize_git_repo_candidate(target.get('repo_url'))
|
|
if candidate and candidate.get('provider') == 'github':
|
|
repo = candidate.get('repo_path') or ''
|
|
url = target.get('url') or (f'https://github.com/{repo}' if repo else '')
|
|
return repo.strip('/'), url
|
|
text = str(target or '').strip()
|
|
if text.startswith('{'):
|
|
try:
|
|
return parse_github_repo_target(json.loads(text))
|
|
except Exception:
|
|
pass
|
|
lowered = text.lower().rstrip('/')
|
|
if lowered.endswith('.git'):
|
|
lowered = lowered[:-4]
|
|
candidate = normalize_git_repo_candidate(text)
|
|
if candidate and candidate.get('provider') == 'github':
|
|
repo = candidate.get('repo_path') or ''
|
|
return repo, f'https://github.com/{repo}'
|
|
match = re.search(r'github\.com[:/]([^/\s]+/[^/\s]+)', lowered, re.IGNORECASE)
|
|
if match:
|
|
repo = match.group(1).strip('/')
|
|
return repo, f'https://github.com/{repo}'
|
|
if re.match(r'^[^/\s]+/[^/\s]+$', text):
|
|
repo = text.strip('/')
|
|
return repo, f'https://github.com/{repo}'
|
|
return '', text
|
|
|
|
|
|
def parse_gitlab_project_target(target):
|
|
if isinstance(target, dict):
|
|
project = target.get('project') or target.get('path_with_namespace') or target.get('repo') or ''
|
|
if not project and target.get('repo_url'):
|
|
candidate = normalize_git_repo_candidate(target.get('repo_url'))
|
|
if candidate and candidate.get('provider') == 'gitlab':
|
|
project = candidate.get('repo_path') or ''
|
|
url = target.get('url') or (f'https://gitlab.com/{project}' if project else '')
|
|
return project.strip('/'), url
|
|
text = str(target or '').strip()
|
|
if text.startswith('{'):
|
|
try:
|
|
return parse_gitlab_project_target(json.loads(text))
|
|
except Exception:
|
|
pass
|
|
lowered = text.lower().rstrip('/')
|
|
if lowered.endswith('.git'):
|
|
lowered = lowered[:-4]
|
|
candidate = normalize_git_repo_candidate(text)
|
|
if candidate and candidate.get('provider') == 'gitlab':
|
|
project = candidate.get('repo_path') or ''
|
|
return project, f'https://gitlab.com/{project}'
|
|
match = re.search(r'gitlab\.com[:/](.+)$', lowered, re.IGNORECASE)
|
|
if match:
|
|
project = match.group(1).strip('/')
|
|
return project, f'https://gitlab.com/{project}'
|
|
if '/' in text and '://' not in text:
|
|
project = text.strip('/')
|
|
return project, f'https://gitlab.com/{project}'
|
|
return '', text
|
|
|
|
def github_rate_limit_reset(response):
|
|
if not response:
|
|
return None
|
|
reset = response.headers.get('X-RateLimit-Reset')
|
|
if not reset:
|
|
return None
|
|
try:
|
|
return datetime.fromtimestamp(int(reset), timezone.utc).isoformat(timespec='seconds')
|
|
except (TypeError, ValueError):
|
|
return None
|
|
|
|
def page_is_known(targets, known_targets=None, normalize_target=None, known_target_lookup=None):
|
|
if not normalize_target or not targets:
|
|
return False
|
|
offered = [target for target in targets if target]
|
|
normalized = [normalize_target(target) for target in offered]
|
|
if known_target_lookup:
|
|
try:
|
|
known_targets = set(known_target_lookup(offered) or ())
|
|
except Exception as exc:
|
|
logger.warning(f'Known-target page lookup failed open: {str(exc)[:300]}')
|
|
return False
|
|
if not known_targets:
|
|
return False
|
|
return bool(normalized) and all(target in known_targets for target in normalized)
|
|
|
|
|
|
POSTMAN_COLLECTION_SUFFIX = 'postman_collection.json'
|
|
POSTMAN_ENVIRONMENT_SUFFIX = 'postman_environment.json'
|
|
API_ARTIFACT_PATTERNS = {
|
|
'collection': ('postman_collection.json',),
|
|
'environment': ('postman_environment.json',),
|
|
'postman': ('postman.json', '.postman.json'),
|
|
'insomnia': ('insomnia.json', '.insomnia.json'),
|
|
'bruno': ('.bru', 'bruno.json'),
|
|
'thunder_collection': ('thunder-collection.json',),
|
|
'thunder_environment': ('thunder-environment.json',),
|
|
'hoppscotch': ('hoppscotch.json',),
|
|
'generic': ('collection.json', 'environment.json'),
|
|
}
|
|
API_ARTIFACT_SEARCHES = {
|
|
'collection': ('filename:postman_collection.json {query}',),
|
|
'environment': ('filename:postman_environment.json {query}',),
|
|
'postman': ('filename:postman.json {query}', 'filename:.postman.json {query}'),
|
|
'insomnia': ('filename:insomnia.json {query}', 'filename:.insomnia.json {query}'),
|
|
'bruno': ('extension:bru {query}', 'filename:bruno.json {query}'),
|
|
'thunder_collection': ('filename:thunder-collection.json {query}', 'filename:thunder-collection_ {query}'),
|
|
'thunder_environment': ('filename:thunder-environment.json {query}', 'filename:thunder-environment_ {query}'),
|
|
'hoppscotch': ('filename:hoppscotch.json {query}',),
|
|
'generic': ('filename:collection.json postman {query}', 'filename:environment.json postman {query}'),
|
|
'signature': (
|
|
'"currentValue" "api_key" {query}',
|
|
'"pm.collectionVariables" {query}',
|
|
'"pm.environment.set" {query}',
|
|
'"openai.azure.com" "api-key" {query}',
|
|
'"services.ai.azure.com" "api-key" {query}',
|
|
'"generativelanguage.googleapis.com" "key" {query}',
|
|
),
|
|
}
|
|
POSTMAN_PLACEHOLDER_RE = re.compile(
|
|
r'^(?:\{\{[^}]+\}\}|<[^>]+>|your[_ -]?[a-z0-9_-]+|replace[_ -]?me|change[_ -]?me|changeme|example|dummy|test)$',
|
|
re.IGNORECASE,
|
|
)
|
|
POSTMAN_JSON_HARD_MAX_INPUT_BYTES = 16 * 1024 * 1024
|
|
POSTMAN_HARVEST_MAX_WARNINGS = 5
|
|
POSTMAN_HARVEST_WARNING_MAX_CHARS = 400
|
|
|
|
|
|
class PostmanCacheValidationError(ValueError):
|
|
pass
|
|
|
|
|
|
class PostmanCacheTooLarge(PostmanCacheValidationError):
|
|
pass
|
|
|
|
|
|
class PostmanCacheCapacityError(PostmanCacheValidationError):
|
|
pass
|
|
|
|
|
|
class _PostmanHarvestDeadlineReached(RuntimeError):
|
|
pass
|
|
|
|
|
|
def _postman_discovery_limits(max_artifacts=None, max_page_artifacts=None, max_total_bytes=None, max_elapsed_sec=None):
|
|
artifacts = int(
|
|
getattr(scan_config, 'postman_discovery_max_artifacts_per_cycle', 1000)
|
|
if max_artifacts is None else max_artifacts
|
|
)
|
|
page_artifacts = int(
|
|
getattr(scan_config, 'postman_discovery_max_artifacts_per_page', 100)
|
|
if max_page_artifacts is None else max_page_artifacts
|
|
)
|
|
total_bytes = int(
|
|
getattr(scan_config, 'postman_discovery_max_bytes_per_cycle', 1024 * 1024 * 1024)
|
|
if max_total_bytes is None else max_total_bytes
|
|
)
|
|
elapsed_sec = float(
|
|
getattr(scan_config, 'postman_discovery_max_elapsed_sec', 300.0)
|
|
if max_elapsed_sec is None else max_elapsed_sec
|
|
)
|
|
if artifacts <= 0 or page_artifacts <= 0 or total_bytes <= 0 or elapsed_sec <= 0 or not math.isfinite(elapsed_sec):
|
|
raise ValueError('Postman discovery limits must be finite and positive')
|
|
return artifacts, page_artifacts, total_bytes, elapsed_sec
|
|
|
|
|
|
class _PostmanDiscoveryBudget:
|
|
MAX_WARNINGS = 5
|
|
|
|
def __init__(self, source, max_artifacts=None, max_page_artifacts=None, max_total_bytes=None, max_elapsed_sec=None):
|
|
(
|
|
self.max_artifacts,
|
|
self.max_page_artifacts,
|
|
self.max_total_bytes,
|
|
elapsed_sec,
|
|
) = _postman_discovery_limits(max_artifacts, max_page_artifacts, max_total_bytes, max_elapsed_sec)
|
|
self.source = str(source or 'non-package')
|
|
self.elapsed_sec = elapsed_sec
|
|
self.deadline = time.monotonic() + elapsed_sec
|
|
self.artifacts = 0
|
|
self.total_bytes = 0
|
|
self.stop_reason = ''
|
|
self._warnings = set()
|
|
|
|
def warn(self, detail):
|
|
detail = str(detail or '')[:400]
|
|
if detail in self._warnings or len(self._warnings) >= self.MAX_WARNINGS:
|
|
return
|
|
self._warnings.add(detail)
|
|
logger.warning('Optional Postman %s discovery bounded: %s', self.source, detail)
|
|
|
|
def stop(self, detail):
|
|
if not self.stop_reason:
|
|
self.stop_reason = str(detail or 'bounded discovery limit reached')[:400]
|
|
self.warn(self.stop_reason)
|
|
return False
|
|
|
|
def check_deadline(self, context='during discovery'):
|
|
if self.stop_reason:
|
|
return False
|
|
if time.monotonic() >= self.deadline:
|
|
return self.stop(f'elapsed deadline of {self.elapsed_sec:g}s reached {context}')
|
|
return True
|
|
|
|
def request_timeout(self, configured_timeout):
|
|
if not self.check_deadline('before a network request'):
|
|
raise _PostmanHarvestDeadlineReached(self.stop_reason)
|
|
remaining = self.deadline - time.monotonic()
|
|
if remaining <= 0:
|
|
self.stop(f'elapsed deadline of {self.elapsed_sec:g}s reached before a network request')
|
|
raise _PostmanHarvestDeadlineReached(self.stop_reason)
|
|
return max(0.01, min(float(configured_timeout or remaining), remaining))
|
|
|
|
def admit(self, page_artifacts):
|
|
if not self.check_deadline():
|
|
return False, 'cycle'
|
|
if page_artifacts >= self.max_page_artifacts:
|
|
self.warn(f'per-page artifact limit of {self.max_page_artifacts} reached')
|
|
return False, 'page'
|
|
if self.artifacts >= self.max_artifacts:
|
|
self.stop(f'per-cycle artifact limit of {self.max_artifacts} reached')
|
|
return False, 'cycle'
|
|
self.artifacts += 1
|
|
return True, ''
|
|
|
|
def account_bytes(self, size):
|
|
size = max(0, int(size or 0))
|
|
if self.total_bytes + size > self.max_total_bytes:
|
|
return self.stop(
|
|
f'aggregate byte limit of {self.max_total_bytes} reached after {self.total_bytes} byte(s)'
|
|
)
|
|
self.total_bytes += size
|
|
return True
|
|
|
|
def exhausted(self):
|
|
if self.stop_reason:
|
|
return True
|
|
if self.artifacts >= self.max_artifacts:
|
|
return not self.stop(f'per-cycle artifact limit of {self.max_artifacts} reached')
|
|
if self.total_bytes >= self.max_total_bytes:
|
|
return not self.stop(f'aggregate byte limit of {self.max_total_bytes} reached')
|
|
return not self.check_deadline()
|
|
|
|
|
|
def _publish_postman_discovery_batch(prepared, cache_dir, budget):
|
|
if not prepared:
|
|
return [], False
|
|
published, deadline_reached = _publish_postman_cache_entries(
|
|
[item['entry'] for item in prepared],
|
|
cache_dir,
|
|
deadline=budget.deadline,
|
|
)
|
|
if deadline_reached:
|
|
budget.stop(f'elapsed deadline of {budget.elapsed_sec:g}s reached while scanning or publishing the cache')
|
|
return list(zip(prepared, published)), deadline_reached
|
|
|
|
|
|
def utc_now_for_postman():
|
|
return datetime.now(timezone.utc)
|
|
|
|
|
|
def parse_postman_time(value):
|
|
if not value:
|
|
return None
|
|
try:
|
|
parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00'))
|
|
if parsed.tzinfo is None:
|
|
parsed = parsed.replace(tzinfo=timezone.utc)
|
|
return parsed.astimezone(timezone.utc)
|
|
except ValueError:
|
|
return None
|
|
|
|
|
|
def postman_kind_for_path(path):
|
|
name = os.path.basename(str(path or '')).lower()
|
|
normalized = str(path or '').replace('\\', '/').lower()
|
|
for kind, suffixes in API_ARTIFACT_PATTERNS.items():
|
|
if any(name.endswith(suffix) or normalized.endswith('/bruno/' + suffix) for suffix in suffixes):
|
|
return kind
|
|
return 'artifact'
|
|
|
|
|
|
def normalize_postman_search_kinds(value):
|
|
if not value:
|
|
return ['collection', 'environment']
|
|
if isinstance(value, str):
|
|
parts = [item.strip().lower() for item in value.split(',')]
|
|
else:
|
|
parts = [str(item).strip().lower() for item in value]
|
|
aliases = {
|
|
'collections': 'collection', 'env': 'environment', 'environments': 'environment',
|
|
'api_artifacts': 'all', 'api-artifacts': 'all', 'thunder': 'thunder_collection',
|
|
}
|
|
kinds = []
|
|
for item in parts:
|
|
item = aliases.get(item, item)
|
|
if item == 'all':
|
|
for kind in API_ARTIFACT_SEARCHES:
|
|
if kind not in kinds:
|
|
kinds.append(kind)
|
|
continue
|
|
if item in API_ARTIFACT_SEARCHES and item not in kinds:
|
|
kinds.append(item)
|
|
return kinds or ['collection', 'environment']
|
|
|
|
|
|
def parse_postman_target(target):
|
|
if isinstance(target, dict):
|
|
return dict(target)
|
|
text = str(target or '').strip()
|
|
if not text:
|
|
return {}
|
|
if text.startswith('{'):
|
|
return json.loads(text)
|
|
if text.lower().startswith('file:'):
|
|
path = text[5:]
|
|
return {'source': 'local_file', 'kind': postman_kind_for_path(path), 'local_path': path, 'path': path}
|
|
if os.path.exists(text):
|
|
return {'source': 'local_file', 'kind': postman_kind_for_path(text), 'local_path': text, 'path': text}
|
|
return {'source': 'url', 'kind': postman_kind_for_path(text), 'url': text}
|
|
|
|
|
|
def postman_target_identity(target):
|
|
return semantic_postman_target_identity(target)
|
|
|
|
|
|
def get_postman_cache_dir(cache_dir=None):
|
|
path = cache_dir or getattr(scan_config, 'postman_cache_dir', None)
|
|
if not path:
|
|
raise PostmanCacheValidationError('configured Postman cache directory is required')
|
|
runtime_dir = getattr(scan_config, 'runtime_dir', None)
|
|
if not runtime_dir:
|
|
raise PostmanCacheValidationError('configured runtime directory is required for Postman cache containment')
|
|
runtime_root = canonical_path(runtime_dir)
|
|
cache_root = canonical_path(path)
|
|
try:
|
|
contained = os.path.commonpath((runtime_root, cache_root)) == runtime_root and cache_root != runtime_root
|
|
except ValueError:
|
|
contained = False
|
|
if not contained:
|
|
raise PostmanCacheValidationError('Postman cache must remain under the configured runtime directory')
|
|
return require_private_directory(cache_root, create=True)
|
|
|
|
|
|
def sha256_bytes(value):
|
|
return hashlib.sha256(value or b'').hexdigest()
|
|
|
|
|
|
def postman_cache_file_path(digest, cache_dir=None):
|
|
root = get_postman_cache_dir(cache_dir)
|
|
prefix = str(digest or '')[:2] or 'xx'
|
|
directory = os.path.join(root, prefix)
|
|
ensure_private_directory(directory, reject_reparse=True)
|
|
return os.path.join(directory, f'{digest}.json')
|
|
|
|
|
|
def postman_cache_usage(root, deadline=None):
|
|
usage = {'items': 0, 'files': 0, 'bytes': 0}
|
|
stack = [require_private_directory(root, create=False)]
|
|
while stack:
|
|
if deadline is not None and time.monotonic() >= deadline:
|
|
raise _PostmanHarvestDeadlineReached()
|
|
current = stack.pop()
|
|
with os.scandir(current) as entries:
|
|
for entry in entries:
|
|
if deadline is not None and time.monotonic() >= deadline:
|
|
raise _PostmanHarvestDeadlineReached()
|
|
if entry.is_symlink() or is_reparse_point(entry.path):
|
|
raise PostmanCacheCapacityError(f'Postman cache contains a linked entry: {entry.path}')
|
|
if entry.is_dir(follow_symlinks=False):
|
|
if not private_directory_ready(entry.path):
|
|
raise PostmanCacheCapacityError(f'Postman cache directory is not private: {entry.path}')
|
|
stack.append(entry.path)
|
|
continue
|
|
if not entry.is_file(follow_symlinks=False):
|
|
raise PostmanCacheCapacityError(f'Postman cache contains an unsupported entry: {entry.path}')
|
|
if entry.name == '.postman-cache.lock':
|
|
continue
|
|
details = entry.stat(follow_symlinks=False)
|
|
usage['files'] += 1
|
|
usage['bytes'] += int(details.st_size)
|
|
if not entry.name.endswith('.meta.json'):
|
|
usage['items'] += 1
|
|
return usage
|
|
|
|
|
|
def _postman_cache_limits(max_items=None, max_bytes=None, min_free_bytes=None):
|
|
values = {
|
|
'items': int(getattr(scan_config, 'postman_cache_max_items', 0) if max_items is None else max_items),
|
|
'bytes': int(getattr(scan_config, 'postman_cache_max_bytes', 0) if max_bytes is None else max_bytes),
|
|
'min_free_bytes': int(getattr(scan_config, 'postman_cache_min_free_bytes', 0) if min_free_bytes is None else min_free_bytes),
|
|
}
|
|
if values['items'] <= 0 or values['bytes'] <= 0 or values['min_free_bytes'] < 0:
|
|
raise PostmanCacheCapacityError('Postman cache aggregate limits must be finite positive values')
|
|
return values
|
|
|
|
|
|
def _write_private_cache_file(path, payload):
|
|
if os.path.lexists(path):
|
|
raise FileExistsError(path)
|
|
temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.tmp'
|
|
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0)
|
|
descriptor = os.open(temporary, flags, 0o600)
|
|
try:
|
|
os.close(descriptor)
|
|
descriptor = None
|
|
harden_private_file(temporary)
|
|
with open(temporary, 'wb') as handle:
|
|
handle.write(payload)
|
|
handle.flush()
|
|
os.fsync(handle.fileno())
|
|
if not private_file_ready(temporary):
|
|
raise PostmanCacheCapacityError(f'Postman cache temporary file is not private: {temporary}')
|
|
if os.path.lexists(path):
|
|
raise FileExistsError(path)
|
|
durable_replace(temporary, path)
|
|
if not private_file_ready(path):
|
|
raise PostmanCacheCapacityError(f'Postman cache publication is not private: {path}')
|
|
finally:
|
|
if descriptor is not None:
|
|
os.close(descriptor)
|
|
try:
|
|
if os.path.exists(temporary):
|
|
os.remove(temporary)
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
def _bounded_json_bytes(value, max_bytes, *, indent=None, sort_keys=False, newline=False):
|
|
output = bytearray()
|
|
encoder = json.JSONEncoder(ensure_ascii=False, indent=indent, sort_keys=sort_keys)
|
|
for chunk in encoder.iterencode(value):
|
|
encoded = chunk.encode('utf-8')
|
|
if len(output) + len(encoded) + (1 if newline else 0) > max_bytes:
|
|
raise ValueError('serialized JSON exceeds its bounded byte limit')
|
|
output.extend(encoded)
|
|
if newline:
|
|
output.extend(b'\n')
|
|
return bytes(output)
|
|
|
|
|
|
def validate_postman_cache_artifact(target_data, max_artifact_size_mb=20, expected_size=None):
|
|
_raise_if_scan_slot_fatal()
|
|
if not isinstance(target_data, dict):
|
|
raise PostmanCacheValidationError('Postman cache metadata must be an object')
|
|
offered_path = target_data.get('cache_path') or target_data.get('local_path')
|
|
if not offered_path:
|
|
raise PostmanCacheValidationError('Postman target missing cached artifact')
|
|
cache_root = canonical_path(get_postman_cache_dir())
|
|
reject_reparse_components(offered_path)
|
|
cache_path = canonical_path(offered_path)
|
|
try:
|
|
contained = os.path.commonpath((cache_root, cache_path)) == cache_root and cache_path != cache_root
|
|
except ValueError:
|
|
contained = False
|
|
if not contained:
|
|
raise PostmanCacheValidationError('Postman cache path escapes the configured cache root')
|
|
reject_reparse_components(cache_path)
|
|
if is_reparse_point(cache_path) or not private_file_ready(cache_path):
|
|
raise PostmanCacheValidationError('Postman cached artifact is absent, linked, or not private')
|
|
declared_hash = str(target_data.get('sha256') or '').strip().lower()
|
|
if not re.fullmatch(r'[0-9a-f]{64}', declared_hash):
|
|
raise PostmanCacheValidationError('Postman target has no valid declared SHA-256')
|
|
if postman_target_identity(target_data) != f'postman:sha256:{declared_hash}':
|
|
raise PostmanCacheValidationError('Postman semantic identity does not match its declared SHA-256')
|
|
|
|
flags = os.O_RDONLY | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0)
|
|
descriptor = os.open(cache_path, flags)
|
|
try:
|
|
opened = os.fstat(descriptor)
|
|
if not stat.S_ISREG(opened.st_mode):
|
|
raise PostmanCacheValidationError('Postman cached artifact is not a regular file')
|
|
size = int(opened.st_size)
|
|
if size <= 0:
|
|
raise PostmanCacheValidationError('Postman cached artifact is empty')
|
|
declared_sizes = [expected_size]
|
|
declared_sizes.extend(target_data.get(name) for name in ('size', 'bytes'))
|
|
for declared_size in declared_sizes:
|
|
if declared_size in (None, ''):
|
|
continue
|
|
try:
|
|
parsed_size = int(declared_size)
|
|
except (TypeError, ValueError) as exc:
|
|
raise PostmanCacheValidationError('Postman cached artifact has invalid declared size') from exc
|
|
if parsed_size != size:
|
|
raise PostmanCacheValidationError('Postman cached artifact size mismatch')
|
|
max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024
|
|
if max_bytes <= 0:
|
|
raise PostmanCacheValidationError('Postman artifact limit must be finite and positive')
|
|
if size > max_bytes:
|
|
raise PostmanCacheTooLarge(f'artifact exceeds {max_artifact_size_mb} MB')
|
|
digest = hashlib.sha256()
|
|
while True:
|
|
_raise_if_scan_slot_fatal()
|
|
block = os.read(descriptor, 1024 * 1024)
|
|
if not block:
|
|
break
|
|
digest.update(block)
|
|
if digest.hexdigest() != declared_hash:
|
|
raise PostmanCacheValidationError('Postman cached artifact SHA-256 mismatch')
|
|
_raise_if_scan_slot_fatal()
|
|
current = os.stat(cache_path, follow_symlinks=False)
|
|
opened_identity = (opened.st_dev, opened.st_ino, opened.st_size, getattr(opened, 'st_mtime_ns', None))
|
|
current_identity = (current.st_dev, current.st_ino, current.st_size, getattr(current, 'st_mtime_ns', None))
|
|
if opened_identity != current_identity:
|
|
raise PostmanCacheValidationError('Postman cached artifact changed during validation')
|
|
finally:
|
|
os.close(descriptor)
|
|
return cache_path, size
|
|
|
|
|
|
def _prepare_postman_cache_entry(content, kind, origin, max_artifact_size_mb):
|
|
max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024
|
|
if max_bytes <= 0:
|
|
raise ValueError('Postman artifact limit must be finite and positive')
|
|
if isinstance(content, str) and len(content) > max_bytes:
|
|
raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB')
|
|
content_bytes = content.encode('utf-8', errors='replace') if isinstance(content, str) else bytes(content or b'')
|
|
if len(content_bytes) > max_bytes:
|
|
raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB')
|
|
if not content_bytes:
|
|
raise ValueError('Postman artifact is empty')
|
|
stripped = content_bytes.lstrip()
|
|
if stripped.startswith((b'{', b'[')):
|
|
if len(content_bytes) > POSTMAN_JSON_HARD_MAX_INPUT_BYTES:
|
|
raise ValueError('Postman JSON artifact exceeds the hard pre-parse byte limit')
|
|
try:
|
|
json.loads(content_bytes.decode('utf-8-sig'))
|
|
except (UnicodeDecodeError, ValueError, RecursionError) as exc:
|
|
raise ValueError('Postman JSON artifact is invalid') from exc
|
|
raw_content = content_bytes
|
|
digest = sha256_bytes(raw_content)
|
|
metadata = {
|
|
'sha256': digest,
|
|
'kind': kind,
|
|
'bytes': len(raw_content),
|
|
'origin': origin or {},
|
|
'cached_at': utc_now_for_postman().isoformat(timespec='seconds'),
|
|
}
|
|
try:
|
|
metadata_bytes = _bounded_json_bytes(metadata, max_bytes, indent=2, sort_keys=True, newline=True)
|
|
except ValueError as exc:
|
|
raise PostmanCacheCapacityError('Postman cache metadata exceeds the per-artifact byte limit') from exc
|
|
return {
|
|
'content': raw_content,
|
|
'digest': digest,
|
|
'metadata': metadata_bytes,
|
|
'size': len(raw_content),
|
|
}
|
|
|
|
|
|
def _publish_postman_cache_entries(
|
|
entries,
|
|
cache_dir=None,
|
|
cache_max_items=None,
|
|
cache_max_bytes=None,
|
|
cache_min_free_bytes=None,
|
|
deadline=None,
|
|
):
|
|
if not entries:
|
|
return [], False
|
|
root = get_postman_cache_dir(cache_dir)
|
|
limits = _postman_cache_limits(cache_max_items, cache_max_bytes, cache_min_free_bytes)
|
|
lock_path = os.path.join(root, '.postman-cache.lock')
|
|
lock_timeout = max(1.0, float(getattr(scan_config, 'postman_cache_lock_timeout_sec', 30)))
|
|
if deadline is not None:
|
|
remaining = deadline - time.monotonic()
|
|
if remaining <= 0:
|
|
return [], True
|
|
lock_timeout = min(lock_timeout, max(0.01, remaining))
|
|
try:
|
|
lock = acquire_file_lock(
|
|
lock_path,
|
|
timeout_sec=lock_timeout,
|
|
)
|
|
except TimeoutError:
|
|
if deadline is not None and time.monotonic() >= deadline:
|
|
return [], True
|
|
raise
|
|
try:
|
|
if deadline is not None and time.monotonic() >= deadline:
|
|
return [], True
|
|
try:
|
|
usage = (
|
|
postman_cache_usage(root)
|
|
if deadline is None
|
|
else postman_cache_usage(root, deadline=deadline)
|
|
)
|
|
except _PostmanHarvestDeadlineReached:
|
|
return [], True
|
|
|
|
plans = []
|
|
reserved_bytes = 0
|
|
free_bytes = None
|
|
seen = set()
|
|
for entry in entries:
|
|
if deadline is not None and time.monotonic() >= deadline:
|
|
return [], True
|
|
digest = entry['digest']
|
|
if digest in seen:
|
|
continue
|
|
seen.add(digest)
|
|
if sha256_bytes(entry['content']) != digest:
|
|
raise PostmanCacheValidationError('Postman cache entry SHA-256 changed before publication')
|
|
|
|
prefix_dir = os.path.join(root, digest[:2])
|
|
path = os.path.join(prefix_dir, f'{digest}.json')
|
|
meta_path = f'{path}.meta.json'
|
|
if os.path.lexists(prefix_dir):
|
|
reject_reparse_components(prefix_dir)
|
|
if not os.path.isdir(prefix_dir) or not private_directory_ready(prefix_dir):
|
|
raise PostmanCacheCapacityError(f'Postman cache prefix directory is not private: {prefix_dir}')
|
|
artifact_exists = os.path.lexists(path)
|
|
metadata_exists = os.path.lexists(meta_path)
|
|
for existing in (path, meta_path):
|
|
if os.path.lexists(existing):
|
|
reject_reparse_components(existing)
|
|
if not private_file_ready(existing):
|
|
raise PostmanCacheCapacityError(f'Postman cache file is not private: {existing}')
|
|
|
|
added_items = 0 if artifact_exists else 1
|
|
added_files = int(not artifact_exists) + int(not metadata_exists)
|
|
added_bytes = (
|
|
(0 if artifact_exists else entry['size'])
|
|
+ (0 if metadata_exists else len(entry['metadata']))
|
|
)
|
|
if usage['items'] + added_items > limits['items']:
|
|
raise PostmanCacheCapacityError(
|
|
f'Postman cache item capacity reached ({usage["items"]}/{limits["items"]})'
|
|
)
|
|
if usage['bytes'] + added_bytes > limits['bytes']:
|
|
raise PostmanCacheCapacityError(
|
|
f'Postman cache byte capacity reached ({usage["bytes"]}/{limits["bytes"]})'
|
|
)
|
|
if added_bytes and free_bytes is None:
|
|
free_bytes = int(shutil.disk_usage(root).free)
|
|
if added_bytes and free_bytes - reserved_bytes - added_bytes < limits['min_free_bytes']:
|
|
raise PostmanCacheCapacityError(
|
|
f'Postman cache free-space reserve would be crossed ({free_bytes} bytes free)'
|
|
)
|
|
usage['items'] += added_items
|
|
usage['files'] += added_files
|
|
usage['bytes'] += added_bytes
|
|
reserved_bytes += added_bytes
|
|
plans.append((entry, prefix_dir, path, meta_path, artifact_exists, metadata_exists))
|
|
|
|
published = []
|
|
for entry, prefix_dir, path, meta_path, artifact_exists, metadata_exists in plans:
|
|
if deadline is not None and time.monotonic() >= deadline:
|
|
return published, True
|
|
if not os.path.isdir(prefix_dir):
|
|
ensure_private_directory(prefix_dir, reject_reparse=True)
|
|
elif not private_directory_ready(prefix_dir):
|
|
raise PostmanCacheCapacityError(f'Postman cache prefix directory is not private: {prefix_dir}')
|
|
if not artifact_exists:
|
|
_write_private_cache_file(path, entry['content'])
|
|
if not metadata_exists:
|
|
_write_private_cache_file(meta_path, entry['metadata'])
|
|
published.append((path, entry['digest'], entry['size']))
|
|
return published, False
|
|
finally:
|
|
release_file_lock(lock, lock_path)
|
|
|
|
|
|
def write_postman_cache(
|
|
content,
|
|
kind='artifact',
|
|
origin=None,
|
|
cache_dir=None,
|
|
max_artifact_size_mb=20,
|
|
cache_max_items=None,
|
|
cache_max_bytes=None,
|
|
cache_min_free_bytes=None,
|
|
):
|
|
entry = _prepare_postman_cache_entry(content, kind, origin, max_artifact_size_mb)
|
|
published, deadline_reached = _publish_postman_cache_entries(
|
|
[entry],
|
|
cache_dir,
|
|
cache_max_items,
|
|
cache_max_bytes,
|
|
cache_min_free_bytes,
|
|
)
|
|
if deadline_reached or not published:
|
|
raise PostmanCacheCapacityError('Postman cache publication did not complete')
|
|
return published[0]
|
|
|
|
|
|
def postman_target_from_cached_artifact(source, kind, cache_path, digest, origin=None, **extra):
|
|
payload = {'source': source, 'kind': kind, 'cache_path': cache_path, 'sha256': digest, 'origin': origin or {}}
|
|
payload.update({key: value for key, value in extra.items() if value not in (None, '')})
|
|
return json.dumps(payload, separators=(',', ':'), ensure_ascii=False, sort_keys=True)
|
|
|
|
|
|
def cache_package_postman_artifact(file_path, package_source, package, relative_path, cache_dir=None, max_artifact_size_mb=20):
|
|
kind = postman_kind_for_path(file_path)
|
|
max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024
|
|
with open(file_path, 'rb') as f:
|
|
content = f.read(max_bytes + 1)
|
|
if max_bytes <= 0 or len(content) > max_bytes:
|
|
raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB')
|
|
origin = {
|
|
'package_source': package_source,
|
|
'package_name': package.get('name'),
|
|
'package_version': package.get('version'),
|
|
'package_artifact': package.get('artifact') or package.get('tarball'),
|
|
'path': relative_path,
|
|
}
|
|
cache_path, digest, size = write_postman_cache(content, kind, origin, cache_dir, max_artifact_size_mb)
|
|
return postman_target_from_cached_artifact(
|
|
f'{package_source}_package',
|
|
kind,
|
|
cache_path,
|
|
digest,
|
|
origin,
|
|
path=relative_path,
|
|
package_name=package.get('name'),
|
|
package_version=package.get('version'),
|
|
package_artifact=package.get('artifact') or package.get('tarball'),
|
|
size=size,
|
|
)
|
|
|
|
|
|
def _postman_package_harvest_limits(max_artifacts=None, max_total_bytes=None, max_elapsed_sec=None):
|
|
artifacts = int(
|
|
getattr(scan_config, 'postman_package_harvest_max_artifacts', 100)
|
|
if max_artifacts is None else max_artifacts
|
|
)
|
|
total_bytes = int(
|
|
getattr(scan_config, 'postman_package_harvest_max_bytes', 128 * 1024 * 1024)
|
|
if max_total_bytes is None else max_total_bytes
|
|
)
|
|
elapsed_sec = float(
|
|
getattr(scan_config, 'postman_package_harvest_max_elapsed_sec', 30.0)
|
|
if max_elapsed_sec is None else max_elapsed_sec
|
|
)
|
|
if artifacts <= 0 or total_bytes <= 0 or elapsed_sec <= 0 or not math.isfinite(elapsed_sec):
|
|
raise ValueError('Postman package harvest limits must be finite and positive')
|
|
return artifacts, total_bytes, elapsed_sec
|
|
|
|
|
|
def _add_postman_harvest_warning(warnings, detail, force=False):
|
|
message = f'Optional Postman package harvesting degraded: {detail}'[:POSTMAN_HARVEST_WARNING_MAX_CHARS]
|
|
if message in warnings:
|
|
return
|
|
if len(warnings) < POSTMAN_HARVEST_MAX_WARNINGS:
|
|
warnings.append(message)
|
|
logger.warning(message)
|
|
elif force:
|
|
warnings[-1] = message
|
|
logger.warning(message)
|
|
|
|
|
|
def _attach_postman_harvest_warnings(results, warnings):
|
|
if not warnings:
|
|
return results
|
|
merged = list(results.get('warnings') or [])
|
|
for warning in warnings:
|
|
if warning not in merged:
|
|
merged.append(warning)
|
|
results['warnings'] = merged
|
|
results['warning_classes'] = sorted(set(list(results.get('warning_classes') or []) + ['postman_package_harvest']))
|
|
results['degraded'] = True
|
|
return results
|
|
|
|
|
|
def _bounded_postman_walk(root_dir, deadline):
|
|
stack = [root_dir]
|
|
while stack:
|
|
_raise_if_scan_slot_fatal()
|
|
if time.monotonic() >= deadline:
|
|
yield None, (), ()
|
|
return
|
|
directory = stack.pop()
|
|
try:
|
|
entries = os.scandir(directory)
|
|
except OSError:
|
|
continue
|
|
try:
|
|
for entry in entries:
|
|
_raise_if_scan_slot_fatal()
|
|
if time.monotonic() >= deadline:
|
|
yield None, (), ()
|
|
return
|
|
try:
|
|
if entry.is_dir(follow_symlinks=False):
|
|
if not entry.is_symlink() and not is_reparse_point(entry.path):
|
|
stack.append(entry.path)
|
|
continue
|
|
except OSError:
|
|
continue
|
|
yield directory, (), (entry.name,)
|
|
finally:
|
|
entries.close()
|
|
|
|
|
|
def find_postman_artifacts(
|
|
root_dir,
|
|
package_source,
|
|
package,
|
|
cache_dir=None,
|
|
max_artifact_size_mb=20,
|
|
warnings=None,
|
|
max_artifacts=None,
|
|
max_total_bytes=None,
|
|
max_elapsed_sec=None,
|
|
):
|
|
targets = []
|
|
if not root_dir or not os.path.isdir(root_dir):
|
|
return targets
|
|
warning_sink = warnings if warnings is not None else []
|
|
artifact_limit, total_byte_limit, elapsed_limit = _postman_package_harvest_limits(
|
|
max_artifacts,
|
|
max_total_bytes,
|
|
max_elapsed_sec,
|
|
)
|
|
max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024
|
|
if max_bytes <= 0:
|
|
raise ValueError('Postman artifact limit must be finite and positive')
|
|
deadline = time.monotonic() + elapsed_limit
|
|
suffixes = tuple(suffix for values in API_ARTIFACT_PATTERNS.values() for suffix in values)
|
|
prepared = []
|
|
seen = set()
|
|
examined = 0
|
|
examined_bytes = 0
|
|
stopped = False
|
|
|
|
walk = _bounded_postman_walk(root_dir, deadline)
|
|
for directory, _, files in walk:
|
|
_raise_if_scan_slot_fatal()
|
|
if directory is None or time.monotonic() >= deadline:
|
|
_add_postman_harvest_warning(
|
|
warning_sink,
|
|
f'elapsed deadline of {elapsed_limit:g}s reached after examining {examined} artifact(s)',
|
|
force=True,
|
|
)
|
|
stopped = True
|
|
break
|
|
for name in files:
|
|
_raise_if_scan_slot_fatal()
|
|
if time.monotonic() >= deadline:
|
|
_add_postman_harvest_warning(
|
|
warning_sink,
|
|
f'elapsed deadline of {elapsed_limit:g}s reached after examining {examined} artifact(s)',
|
|
force=True,
|
|
)
|
|
stopped = True
|
|
break
|
|
lower_name = name.lower()
|
|
if not lower_name.endswith(suffixes):
|
|
continue
|
|
if examined >= artifact_limit:
|
|
_add_postman_harvest_warning(
|
|
warning_sink,
|
|
f'artifact limit of {artifact_limit} reached; remaining matching files were not examined',
|
|
force=True,
|
|
)
|
|
stopped = True
|
|
break
|
|
examined += 1
|
|
path = os.path.join(directory, name)
|
|
try:
|
|
details = os.stat(path, follow_symlinks=False)
|
|
if not stat.S_ISREG(details.st_mode) or os.path.islink(path) or is_reparse_point(path):
|
|
raise ValueError('artifact is not a regular unlinked file')
|
|
size = int(details.st_size)
|
|
relative_path = os.path.relpath(path, root_dir).replace('\\', '/')
|
|
if size > max_bytes:
|
|
_add_postman_harvest_warning(
|
|
warning_sink,
|
|
f'skipped {relative_path[:180]} because it exceeds the {max_artifact_size_mb} MB artifact limit',
|
|
)
|
|
continue
|
|
if examined_bytes + size > total_byte_limit:
|
|
_add_postman_harvest_warning(
|
|
warning_sink,
|
|
f'aggregate byte limit of {total_byte_limit} reached after {examined_bytes} byte(s)',
|
|
force=True,
|
|
)
|
|
stopped = True
|
|
break
|
|
examined_bytes += size
|
|
|
|
flags = os.O_RDONLY | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0)
|
|
descriptor = os.open(path, flags)
|
|
try:
|
|
opened = os.fstat(descriptor)
|
|
opened_identity = (opened.st_dev, opened.st_ino, opened.st_size, getattr(opened, 'st_mtime_ns', None))
|
|
expected_identity = (details.st_dev, details.st_ino, details.st_size, getattr(details, 'st_mtime_ns', None))
|
|
if not stat.S_ISREG(opened.st_mode) or opened_identity != expected_identity:
|
|
raise ValueError('artifact changed before harvesting')
|
|
content = bytearray()
|
|
while len(content) < size:
|
|
_raise_if_scan_slot_fatal()
|
|
if time.monotonic() >= deadline:
|
|
raise _PostmanHarvestDeadlineReached()
|
|
block = os.read(descriptor, min(1024 * 1024, size - len(content)))
|
|
if not block:
|
|
raise ValueError('artifact changed while harvesting')
|
|
content.extend(block)
|
|
if os.read(descriptor, 1):
|
|
raise ValueError('artifact grew while harvesting')
|
|
current = os.stat(path, follow_symlinks=False)
|
|
current_identity = (current.st_dev, current.st_ino, current.st_size, getattr(current, 'st_mtime_ns', None))
|
|
if current_identity != opened_identity:
|
|
raise ValueError('artifact changed while harvesting')
|
|
finally:
|
|
os.close(descriptor)
|
|
|
|
origin = {
|
|
'package_source': package_source,
|
|
'package_name': package.get('name'),
|
|
'package_version': package.get('version'),
|
|
'package_artifact': package.get('artifact') or package.get('tarball'),
|
|
'path': relative_path,
|
|
}
|
|
entry = _prepare_postman_cache_entry(bytes(content), postman_kind_for_path(path), origin, max_artifact_size_mb)
|
|
if entry['digest'] not in seen:
|
|
entry['origin'] = origin
|
|
entry['kind'] = postman_kind_for_path(path)
|
|
entry['relative_path'] = relative_path
|
|
prepared.append(entry)
|
|
seen.add(entry['digest'])
|
|
except _PostmanHarvestDeadlineReached:
|
|
_add_postman_harvest_warning(
|
|
warning_sink,
|
|
f'elapsed deadline of {elapsed_limit:g}s reached after examining {examined} artifact(s)',
|
|
force=True,
|
|
)
|
|
stopped = True
|
|
break
|
|
except PostmanCacheValidationError:
|
|
raise
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except Exception as e:
|
|
relative_path = os.path.relpath(path, root_dir).replace('\\', '/')
|
|
_add_postman_harvest_warning(
|
|
warning_sink,
|
|
f'unable to harvest {relative_path[:180]}: {str(e)[:160]}',
|
|
)
|
|
if stopped:
|
|
break
|
|
walk.close()
|
|
|
|
if prepared and not (stopped and time.monotonic() >= deadline):
|
|
_raise_if_scan_slot_fatal()
|
|
published, deadline_reached = _publish_postman_cache_entries(prepared, cache_dir, deadline=deadline)
|
|
for entry, (cache_path, digest, size) in zip(prepared, published):
|
|
targets.append(postman_target_from_cached_artifact(
|
|
f'{package_source}_package',
|
|
entry['kind'],
|
|
cache_path,
|
|
digest,
|
|
entry['origin'],
|
|
path=entry['relative_path'],
|
|
package_name=package.get('name'),
|
|
package_version=package.get('version'),
|
|
package_artifact=package.get('artifact') or package.get('tarball'),
|
|
size=size,
|
|
))
|
|
if deadline_reached:
|
|
_add_postman_harvest_warning(
|
|
warning_sink,
|
|
f'elapsed deadline of {elapsed_limit:g}s reached while publishing {len(published)} artifact(s)',
|
|
force=True,
|
|
)
|
|
return targets
|
|
|
|
|
|
class GitHubTokenPool:
|
|
def __init__(self, token_entries=None, token=None, status=None, code_search_rpm_per_token=8, fallback_cooldown=1800):
|
|
entries = []
|
|
for index, entry in enumerate(token_entries or []):
|
|
if isinstance(entry, str):
|
|
item = {'name': f'github_{index + 1}', 'token': entry}
|
|
elif isinstance(entry, dict):
|
|
item = dict(entry)
|
|
item.setdefault('name', f'github_{index + 1}')
|
|
else:
|
|
continue
|
|
if item.get('token'):
|
|
entries.append(item)
|
|
if token and not any(item.get('token') == token for item in entries):
|
|
entries.append({'name': 'token', 'token': token})
|
|
self.entries = entries
|
|
self.status = status if isinstance(status, dict) else {}
|
|
self.index = 0
|
|
self.last_code_search_at = {}
|
|
self.code_search_interval = 60.0 / max(1, int(code_search_rpm_per_token or 8))
|
|
self.fallback_cooldown = int(fallback_cooldown or 1800)
|
|
|
|
def _status_for(self, entry):
|
|
return self.status.setdefault(entry.get('name'), {})
|
|
|
|
def _disabled_until(self, entry):
|
|
value = self._status_for(entry).get('disabled_until')
|
|
if value == 'manual':
|
|
return 'manual'
|
|
return parse_postman_time(value)
|
|
|
|
def _available_entries(self):
|
|
now = utc_now_for_postman()
|
|
available = []
|
|
for entry in self.entries:
|
|
disabled_until = self._disabled_until(entry)
|
|
if disabled_until == 'manual':
|
|
continue
|
|
if disabled_until and disabled_until > now:
|
|
continue
|
|
available.append(entry)
|
|
return available
|
|
|
|
def _mark_unavailable(self, entry, category, message, reset_at=None):
|
|
item = self._status_for(entry)
|
|
if category in ('auth_invalid', 'auth_forbidden'):
|
|
item['disabled_until'] = 'manual'
|
|
else:
|
|
item['disabled_until'] = reset_at or (utc_now_for_postman() + timedelta(seconds=self.fallback_cooldown)).isoformat(timespec='seconds')
|
|
item['disabled_reason'] = category
|
|
item['last_error'] = str(message or '')[:500]
|
|
item['last_failure_at'] = utc_now_for_postman().isoformat(timespec='seconds')
|
|
item['failures'] = int(item.get('failures', 0) or 0) + 1
|
|
|
|
def _next_entry(self):
|
|
available = self._available_entries()
|
|
if not available:
|
|
return None
|
|
for _ in range(len(self.entries)):
|
|
entry = self.entries[self.index % len(self.entries)]
|
|
self.index = (self.index + 1) % len(self.entries)
|
|
if entry in available:
|
|
return entry
|
|
return available[0]
|
|
|
|
def _sleep_for_resource(self, entry, resource, deadline=None):
|
|
if resource != 'code_search':
|
|
return
|
|
name = entry.get('name')
|
|
last = self.last_code_search_at.get(name)
|
|
now = time.monotonic()
|
|
if last is not None:
|
|
delay = self.code_search_interval - (now - last)
|
|
if delay > 0:
|
|
if deadline is not None and now + delay >= deadline:
|
|
remaining = max(0.0, deadline - now)
|
|
if remaining:
|
|
time.sleep(remaining)
|
|
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during code-search pacing')
|
|
time.sleep(delay)
|
|
self.last_code_search_at[name] = time.monotonic()
|
|
|
|
def _wait_until_any_available(self, deadline=None):
|
|
waits = []
|
|
now = utc_now_for_postman()
|
|
manual_count = 0
|
|
for entry in self.entries:
|
|
disabled_until = self._disabled_until(entry)
|
|
if disabled_until == 'manual':
|
|
manual_count += 1
|
|
if disabled_until and disabled_until != 'manual' and disabled_until > now:
|
|
waits.append((disabled_until - now).total_seconds())
|
|
if manual_count >= len(self.entries):
|
|
raise RateLimitError('postman', 'All GitHub tokens are manually disabled for Postman discovery', category='auth_invalid', retryable=False, auth_related=True)
|
|
wait_for = min(waits) if waits else self.fallback_cooldown
|
|
wait_for = max(1, min(wait_for, self.fallback_cooldown))
|
|
if deadline is not None and time.monotonic() + wait_for >= deadline:
|
|
remaining = max(0.0, deadline - time.monotonic())
|
|
logger.warning(f'All GitHub tokens unavailable for Postman discovery; deadline in {remaining:.1f}s')
|
|
if remaining:
|
|
time.sleep(remaining)
|
|
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached while all GitHub tokens were unavailable')
|
|
logger.warning(f'All GitHub tokens unavailable for Postman discovery; sleeping {wait_for:.0f}s')
|
|
time.sleep(wait_for)
|
|
|
|
def _token_core_status(self, entry, timeout=10):
|
|
request_headers = {
|
|
'Accept': 'application/vnd.github+json',
|
|
'User-Agent': 'GitSecretsScanner/2.0',
|
|
'Authorization': f"Bearer {entry.get('token')}",
|
|
}
|
|
try:
|
|
response = api_request('GET', 'https://api.github.com/user', headers=request_headers, timeout=timeout, max_retries=1)
|
|
except Exception:
|
|
return 'unknown'
|
|
try:
|
|
if response.status_code == 200:
|
|
return 'valid'
|
|
if response.status_code == 401:
|
|
return 'invalid'
|
|
if response.status_code in (403, 429) and response.headers.get('X-RateLimit-Remaining') == '0':
|
|
return 'rate_limited'
|
|
return f'http_{response.status_code}'
|
|
finally:
|
|
response.close()
|
|
|
|
def _mark_github_api_error(self, entry, api_error):
|
|
category = getattr(api_error, 'category', 'api')
|
|
if category not in ('rate_limit', 'secondary_rate_limit', 'auth_invalid', 'auth_forbidden'):
|
|
return False
|
|
# Code search may return auth-like 401/403 even when the token is valid for core GitHub API.
|
|
# Confirm against /user before permanently disabling the token as dead.
|
|
if category in ('auth_invalid', 'auth_forbidden'):
|
|
core_status = self._token_core_status(entry)
|
|
if core_status == 'valid':
|
|
self._mark_unavailable(entry, f'{category}_resource', str(api_error), api_error.reset_at)
|
|
return True
|
|
if core_status in ('unknown', 'rate_limited'):
|
|
self._mark_unavailable(entry, f'{category}_unconfirmed', str(api_error), api_error.reset_at)
|
|
return True
|
|
self._mark_unavailable(entry, category, str(api_error), api_error.reset_at)
|
|
return True
|
|
|
|
def request(self, method, url, params=None, headers=None, timeout=30, resource='core', deadline=None, use_proxy=None):
|
|
if not self.entries:
|
|
raise RateLimitError('postman', 'GitHub token is required for Postman GitHub code search', category='auth_invalid', retryable=False, auth_related=True)
|
|
attempts = 0
|
|
last_error = None
|
|
while attempts < max(1, len(self.entries) * 2):
|
|
if deadline is not None and time.monotonic() >= deadline:
|
|
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GitHub request')
|
|
entry = self._next_entry()
|
|
if not entry:
|
|
self._wait_until_any_available(deadline)
|
|
attempts += 1
|
|
continue
|
|
self._sleep_for_resource(entry, resource, deadline)
|
|
request_headers = {'Accept': 'application/vnd.github+json', 'User-Agent': 'GitSecretsScanner/2.0'}
|
|
request_headers.update(headers or {})
|
|
request_headers['Authorization'] = f"Bearer {entry.get('token')}"
|
|
try:
|
|
response = api_request(
|
|
method, url, headers=request_headers, params=params, timeout=timeout, deadline=deadline,
|
|
use_proxy=use_proxy,
|
|
)
|
|
if response.status_code in (401, 403, 429):
|
|
api_error = github_api_error(response)
|
|
response.close()
|
|
if self._mark_github_api_error(entry, api_error):
|
|
last_error = api_error
|
|
attempts += 1
|
|
continue
|
|
try:
|
|
response.raise_for_status()
|
|
except requests.exceptions.HTTPError as e:
|
|
raise github_api_error(e.response) from e
|
|
return response
|
|
except (requests.exceptions.RequestException, ApiRequestError) as e:
|
|
last_error = e
|
|
logger.warning(f'GitHub request failed for Postman discovery with token {entry.get("name")}: {str(e)[:300]}')
|
|
attempts += 1
|
|
delay = min(2, attempts)
|
|
if deadline is not None and time.monotonic() + delay >= deadline:
|
|
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during GitHub request retry') from e
|
|
time.sleep(delay)
|
|
continue
|
|
if last_error:
|
|
raise RateLimitError('postman', f'GitHub Postman discovery request failed after retries: {last_error}', category='network', retryable=True, auth_related=False)
|
|
raise RateLimitError('postman', 'All GitHub tokens are unavailable for Postman discovery', category='rate_limit', retryable=True, auth_related=True)
|
|
|
|
|
|
def latest_github_path_commit(repo, path, pool, request_timeout=20, deadline=None):
|
|
response = pool.request(
|
|
'GET',
|
|
f'https://api.github.com/repos/{repo}/commits',
|
|
params={'path': path, 'per_page': 1},
|
|
timeout=request_timeout,
|
|
resource='core',
|
|
deadline=deadline,
|
|
)
|
|
data = response.json()
|
|
if not data:
|
|
return None
|
|
commit = data[0].get('commit') or {}
|
|
committer = commit.get('committer') or {}
|
|
author = commit.get('author') or {}
|
|
return committer.get('date') or author.get('date')
|
|
|
|
|
|
def github_content_bytes(item, pool, request_timeout=20, max_artifact_size_mb=20, deadline=None):
|
|
response = pool.request(
|
|
'GET', item.get('url'), timeout=request_timeout, resource='core', deadline=deadline,
|
|
use_proxy=False,
|
|
)
|
|
data = response.json()
|
|
size = int(data.get('size') or 0)
|
|
max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024
|
|
if max_bytes and size > max_bytes:
|
|
raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB')
|
|
content = str(data.get('content') or '')
|
|
if str(data.get('encoding') or '').lower() == 'base64':
|
|
decoded = base64.b64decode(re.sub(r'\s+', '', content))
|
|
else:
|
|
decoded = content.encode('utf-8', errors='replace')
|
|
if max_bytes <= 0 or len(decoded) > max_bytes:
|
|
raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB')
|
|
return decoded
|
|
|
|
|
|
def github_postman_item_to_target(item, kind, cache_path, digest, size=None, commit_date=None):
|
|
repo = (item.get('repository') or {}).get('full_name') or ''
|
|
path = item.get('path') or ''
|
|
sha = item.get('sha') or digest or ''
|
|
origin = {
|
|
'provider': 'github',
|
|
'repo': repo,
|
|
'path': path,
|
|
'sha': sha,
|
|
'html_url': item.get('html_url'),
|
|
'api_url': item.get('url'),
|
|
'commit_date': commit_date,
|
|
}
|
|
return postman_target_from_cached_artifact(
|
|
'github_code',
|
|
kind,
|
|
cache_path,
|
|
digest,
|
|
origin,
|
|
repo=repo,
|
|
path=path,
|
|
sha=sha,
|
|
html_url=item.get('html_url'),
|
|
api_url=item.get('url'),
|
|
commit_date=commit_date,
|
|
size=size,
|
|
)
|
|
|
|
|
|
def fetch_github_postman_targets(query, pages=1, per_page=100, token_entries=None, token=None, search_kinds=None, cache_dir=None, max_file_age_days=365, max_artifact_size_mb=20, request_timeout=20, code_search_rpm_per_token=8, all_tokens_cooldown=1800, auth_status=None, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, discovery_max_artifacts=None, discovery_max_artifacts_per_page=None, discovery_max_bytes=None, discovery_max_elapsed_sec=None):
|
|
if not query:
|
|
return []
|
|
budget = _PostmanDiscoveryBudget(
|
|
'GitHub code-search',
|
|
discovery_max_artifacts,
|
|
discovery_max_artifacts_per_page,
|
|
discovery_max_bytes,
|
|
discovery_max_elapsed_sec,
|
|
)
|
|
pool = GitHubTokenPool(token_entries, token, auth_status, code_search_rpm_per_token, all_tokens_cooldown)
|
|
kinds = normalize_postman_search_kinds(search_kinds)
|
|
per_page = max(1, min(int(per_page or 100), 100))
|
|
max_pages = min(max(1, int(pages or 1)), max(1, (1000 + per_page - 1) // per_page))
|
|
cutoff = utc_now_for_postman() - timedelta(days=int(max_file_age_days or 0)) if int(max_file_age_days or 0) > 0 else None
|
|
targets = []
|
|
seen_targets = set()
|
|
seen_digests = set()
|
|
stop_cycle = False
|
|
for kind in kinds:
|
|
if stop_cycle or not budget.check_deadline():
|
|
break
|
|
templates = API_ARTIFACT_SEARCHES.get(kind) or API_ARTIFACT_SEARCHES['generic']
|
|
for template in templates:
|
|
if stop_cycle or not budget.check_deadline():
|
|
break
|
|
search_query = template.format(query=query).strip()
|
|
logger.info(f"Fetching GitHub API artifact {kind} targets for: {search_query!r}")
|
|
known_pages = 0
|
|
for page in range(1, max_pages + 1):
|
|
if not budget.check_deadline():
|
|
stop_cycle = True
|
|
break
|
|
try:
|
|
response = pool.request(
|
|
'GET',
|
|
'https://api.github.com/search/code',
|
|
params={'q': search_query, 'per_page': per_page, 'page': page, 'sort': 'indexed', 'order': 'desc'},
|
|
timeout=budget.request_timeout(request_timeout),
|
|
resource='code_search',
|
|
deadline=budget.deadline,
|
|
)
|
|
except _PostmanHarvestDeadlineReached as e:
|
|
budget.stop(str(e))
|
|
stop_cycle = True
|
|
break
|
|
except RateLimitError as e:
|
|
if getattr(e, 'category', '') == 'network':
|
|
logger.warning(f'GitHub API artifact {kind} network failure on page {page}: {str(e)[:300]}')
|
|
raise
|
|
payload = response.json()
|
|
if not isinstance(payload, dict) or 'items' not in payload or not isinstance(payload.get('items'), list):
|
|
raise ApiRequestError('invalid GitHub code search payload')
|
|
items = payload.get('items') or []
|
|
if not items:
|
|
logger.info(f'GitHub API artifact {kind} page {page}: no results')
|
|
break
|
|
page_targets = []
|
|
prepared = []
|
|
page_artifacts = 0
|
|
for item in items:
|
|
admitted, scope = budget.admit(page_artifacts)
|
|
if not admitted:
|
|
stop_cycle = scope == 'cycle'
|
|
break
|
|
page_artifacts += 1
|
|
repo = (item.get('repository') or {}).get('full_name') or ''
|
|
path = item.get('path') or ''
|
|
commit_date = None
|
|
if cutoff:
|
|
try:
|
|
commit_date = latest_github_path_commit(
|
|
repo, path, pool, budget.request_timeout(request_timeout), budget.deadline,
|
|
)
|
|
parsed_commit = parse_postman_time(commit_date)
|
|
if not parsed_commit or parsed_commit < cutoff:
|
|
continue
|
|
except _PostmanHarvestDeadlineReached as e:
|
|
budget.stop(str(e))
|
|
stop_cycle = True
|
|
break
|
|
except RateLimitError as e:
|
|
if getattr(e, 'category', '') in ('network', 'not_found'):
|
|
logger.warning(f'Skipping API artifact freshness check for {repo}:{path}: {str(e)[:300]}')
|
|
continue
|
|
raise
|
|
except Exception as e:
|
|
logger.warning(f'Unable to check API artifact freshness for {repo}:{path}: {str(e)}')
|
|
continue
|
|
try:
|
|
content = github_content_bytes(
|
|
item,
|
|
pool,
|
|
budget.request_timeout(request_timeout),
|
|
max_artifact_size_mb,
|
|
budget.deadline,
|
|
)
|
|
if not budget.account_bytes(len(content)):
|
|
stop_cycle = True
|
|
break
|
|
origin = {'provider': 'github', 'repo': repo, 'path': path, 'sha': item.get('sha'), 'html_url': item.get('html_url'), 'api_url': item.get('url'), 'commit_date': commit_date}
|
|
detected_kind = postman_kind_for_path(path)
|
|
entry = _prepare_postman_cache_entry(content, detected_kind or kind, origin, max_artifact_size_mb)
|
|
if entry['digest'] in seen_digests:
|
|
continue
|
|
seen_digests.add(entry['digest'])
|
|
prepared.append({
|
|
'entry': entry,
|
|
'item': item,
|
|
'kind': detected_kind or kind,
|
|
'commit_date': commit_date,
|
|
})
|
|
except _PostmanHarvestDeadlineReached as e:
|
|
budget.stop(str(e))
|
|
stop_cycle = True
|
|
break
|
|
except RateLimitError as e:
|
|
if getattr(e, 'category', '') in ('network', 'not_found'):
|
|
logger.warning(f'Skipping GitHub API artifact for {repo}:{path}: {str(e)[:300]}')
|
|
continue
|
|
raise
|
|
except Exception as e:
|
|
if isinstance(e, PostmanCacheCapacityError):
|
|
raise
|
|
logger.warning(f'Unable to cache GitHub API artifact {repo}:{path}: {str(e)}')
|
|
published, deadline_reached = _publish_postman_discovery_batch(prepared, cache_dir, budget)
|
|
for record, (cache_path, digest, size) in published:
|
|
target = github_postman_item_to_target(
|
|
record['item'], record['kind'], cache_path, digest, size, record['commit_date'],
|
|
)
|
|
identity = postman_target_identity(target)
|
|
if identity not in seen_targets:
|
|
targets.append(target)
|
|
page_targets.append(target)
|
|
seen_targets.add(identity)
|
|
logger.info(f'GitHub API artifact {kind} page {page}: fetched {len(items)}, queued candidates {len(page_targets)}')
|
|
if stop_on_seen_pages and page >= max(1, int(min_pages_before_stop or 1)):
|
|
if page_targets and page_is_known(
|
|
page_targets, known_targets, normalize_target, known_target_lookup,
|
|
):
|
|
known_pages += 1
|
|
logger.info(f'GitHub API artifact {kind} page {page}: all targets known ({known_pages}/{seen_page_threshold})')
|
|
if known_pages >= max(1, int(seen_page_threshold or 1)):
|
|
logger.info(f'Stopping API artifact {kind} pagination early after {known_pages} known page(s)')
|
|
break
|
|
else:
|
|
known_pages = 0
|
|
if deadline_reached or stop_cycle or budget.exhausted():
|
|
stop_cycle = True
|
|
break
|
|
return targets
|
|
|
|
|
|
def fetch_github_repo_items(query, pages, per_page=100, token=None, sort_by="updated", sort_order="desc", created_filter="any", raise_rate_limit=False, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None):
|
|
"""Fetch GitHub repositories with metadata for filtering."""
|
|
repos = []
|
|
headers = {'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/vnd.github.v3+json'}
|
|
if token:
|
|
headers['Authorization'] = f'Bearer {token}'
|
|
|
|
# Build search query with filters
|
|
search_query = query if query else "*"
|
|
|
|
# Add created date filter
|
|
if created_filter != "any":
|
|
date_filters = {
|
|
"today": "created:>{}".format((datetime.now() - timedelta(days=1)).strftime("%Y-%m-%d")),
|
|
"week": "created:>{}".format((datetime.now() - timedelta(weeks=1)).strftime("%Y-%m-%d")),
|
|
"month": "created:>{}".format((datetime.now() - timedelta(days=30)).strftime("%Y-%m-%d")),
|
|
"year": "created:>{}".format((datetime.now() - timedelta(days=365)).strftime("%Y-%m-%d"))
|
|
}
|
|
if created_filter in date_filters:
|
|
search_query += f" {date_filters[created_filter]}"
|
|
|
|
logger.info(f"Fetching GitHub repositories for: '{search_query}' sorted by {sort_by} ({sort_order})...")
|
|
|
|
seen_pages = 0
|
|
successful_pages = 0
|
|
for page in range(1, pages + 1):
|
|
url = "https://api.github.com/search/repositories"
|
|
params = {
|
|
'q': search_query,
|
|
'sort': sort_by,
|
|
'order': sort_order,
|
|
'per_page': per_page,
|
|
'page': page,
|
|
}
|
|
|
|
try:
|
|
response = api_request('GET', url, headers=headers, params=params, timeout=30)
|
|
response.raise_for_status()
|
|
|
|
# Handle rate limits
|
|
if response.status_code == 403 and 'X-RateLimit-Remaining' in response.headers:
|
|
if int(response.headers['X-RateLimit-Remaining']) == 0:
|
|
reset_time = datetime.fromtimestamp(int(response.headers['X-RateLimit-Reset']))
|
|
wait_seconds = (reset_time - datetime.now()).total_seconds() + 10
|
|
logger.warning(f"Rate limit exceeded. Resuming at {reset_time}. Waiting {wait_seconds:.0f} seconds...")
|
|
time.sleep(wait_seconds)
|
|
continue
|
|
|
|
data = response.json()
|
|
if not isinstance(data, dict) or 'items' not in data or not isinstance(data.get('items'), list):
|
|
raise ValueError('invalid GitHub repository search payload')
|
|
successful_pages += 1
|
|
if not data.get('items'):
|
|
logger.info(f"Page {page} returned no results. Stopping.")
|
|
break
|
|
|
|
page_repos = [github_repo_to_target(item) for item in data['items'] if item.get('clone_url')]
|
|
repos.extend(page_repos)
|
|
logger.info(f"Page {page}: Fetched {len(data['items'])} repositories")
|
|
if stop_on_seen_pages and page >= max(1, min_pages_before_stop):
|
|
if page_is_known(
|
|
[item.get('url') for item in page_repos], known_targets,
|
|
normalize_target, known_target_lookup,
|
|
):
|
|
seen_pages += 1
|
|
logger.info(f"Page {page}: all GitHub repositories are already queued/checked ({seen_pages}/{seen_page_threshold})")
|
|
if seen_pages >= max(1, seen_page_threshold):
|
|
logger.info(f"Stopping GitHub pagination early after {seen_pages} all-known page(s)")
|
|
break
|
|
else:
|
|
seen_pages = 0
|
|
|
|
except requests.exceptions.HTTPError as e:
|
|
api_error = github_api_error(e.response)
|
|
if raise_rate_limit:
|
|
raise api_error from e
|
|
logger.error(str(api_error))
|
|
raise api_error from e
|
|
except Exception as e:
|
|
if isinstance(e, ApiRequestError):
|
|
raise
|
|
logger.error(f"Error fetching page {page}: {str(e)}")
|
|
raise ApiRequestError(f'GitHub discovery payload failed: {e}') from e
|
|
|
|
return repos
|
|
|
|
def fetch_github_repos(query, pages, per_page=100, token=None, sort_by="updated", sort_order="desc", created_filter="any", raise_rate_limit=False, **kwargs):
|
|
"""Fetch GitHub repository clone URLs with pagination and authentication."""
|
|
return [
|
|
item['url'] for item in fetch_github_repo_items(
|
|
query, pages, per_page, token, sort_by, sort_order, created_filter, raise_rate_limit, **kwargs
|
|
)
|
|
if item.get('url')
|
|
]
|
|
|
|
|
|
GHARCHIVE_REPO_TERMS = (
|
|
'ai', 'agent', 'assistant', 'bot', 'chat', 'chatbot', 'gpt', 'llm', 'rag', 'mcp',
|
|
'model', 'inference', 'embedding', 'vector', 'semantic', 'prompt', 'workflow',
|
|
'copilot', 'codegen', 'langchain', 'llamaindex', 'litellm', 'ollama', 'vllm',
|
|
'claude', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq', 'grok', 'xai', 'openrouter', 'replicate',
|
|
'deepseek', 'zai', 'glm', 'zhipu', 'dashscope', 'bedrock', 'vertex', 'foundry',
|
|
'aiplatform', 'boto3', 'terraform', 'cloudbuild', 'service-account', 'credentials',
|
|
)
|
|
GHARCHIVE_PATH_TERMS = (
|
|
'.env', 'env.', 'secret', 'secrets', 'credential', 'credentials', 'apikey', 'api_key',
|
|
'token', 'tokens', 'key', 'keys', 'config', 'settings', 'appsettings', 'docker-compose',
|
|
'compose.yaml', 'compose.yml', '.github/workflows', 'workflow', 'deploy', 'deployment',
|
|
'kubernetes', 'k8s', 'helm', 'terraform', 'tfvars', 'notebook', '.ipynb', 'postman',
|
|
'collection.json', 'environment.json', 'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq',
|
|
'xai', 'grok', 'openrouter', 'replicate', 'deepseek', 'zai', 'glm', 'dashscope',
|
|
'aws_access_key_id', 'aws_secret_access_key', 'bedrock-runtime', 'boto3',
|
|
'google_application_credentials', 'service-account', 'service_account', 'credentials.json',
|
|
'application_default_credentials', 'vertexai', 'aiplatform', 'provider.tf', 'cloudbuild.yaml',
|
|
)
|
|
GHARCHIVE_FILE_FETCH_TERMS = (
|
|
'.env', 'env.', '.env.', '.env-', 'secret', 'secrets', 'credential', 'credentials',
|
|
'credentials.json', 'service-account', 'service_account', 'application_default_credentials',
|
|
'google_application_credentials', 'terraform.tfvars', '.tfvars', 'provider.tf',
|
|
'cloudbuild.yaml', 'cloudbuild.yml', '.github/workflows', 'docker-compose',
|
|
'compose.yml', 'compose.yaml', 'appsettings', '.ipynb', 'notebook',
|
|
'bedrock', 'vertex', 'aiplatform', 'boto3', 'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot',
|
|
'groq', 'xai', 'grok', 'openrouter', 'replicate', 'deepseek', 'zai', 'glm', 'dashscope',
|
|
)
|
|
GHARCHIVE_FILE_SKIP_SUFFIXES = (
|
|
'.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg', '.ico', '.pdf', '.zip', '.gz', '.tgz',
|
|
'.tar', '.rar', '.7z', '.bin', '.safetensors', '.pt', '.pth', '.onnx', '.parquet', '.arrow',
|
|
'.mp4', '.mov', '.avi', '.mp3', '.wav', '.lock', '.sum', '.min.js', '.map', '.pyc',
|
|
)
|
|
GHARCHIVE_FILE_SKIP_PARTS = (
|
|
'/__pycache__/', '/node_modules/', '/vendor/', '/.git/', '/dist/', '/build/', '/target/',
|
|
)
|
|
GITHUB_GIST_FILE_TERMS = (
|
|
'.env', 'env', 'secret', 'secrets', 'credential', 'credentials', 'apikey', 'api_key',
|
|
'token', 'tokens', 'key', 'keys', 'config', 'settings', 'appsettings', 'credentials.json',
|
|
'service-account', 'service_account', 'application_default_credentials', 'terraform',
|
|
'tfvars', 'provider.tf', 'docker-compose', 'compose.yml', 'compose.yaml', 'workflow',
|
|
'cloudbuild', 'ipynb', 'notebook', 'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq',
|
|
'xai', 'grok', 'openrouter', 'replicate', 'deepseek', 'zai', 'glm', 'dashscope',
|
|
'bedrock', 'vertex', 'aiplatform', 'postman',
|
|
)
|
|
GHARCHIVE_MESSAGE_TERMS = (
|
|
'api key', 'apikey', 'token', 'secret', 'credential', '.env', 'config', 'settings',
|
|
'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq', 'xai', 'grok', 'openrouter',
|
|
'replicate', 'deepseek', 'zai', 'glm', 'dashscope', 'bedrock', 'vertex',
|
|
'aws_access_key_id', 'aws_secret_access_key', 'google_application_credentials',
|
|
'service account', 'credentials.json', 'vertexai', 'aiplatform', 'bedrock-runtime',
|
|
'terraform', 'tfvars', 'cloudbuild',
|
|
)
|
|
GHARCHIVE_MAX_HOURS_BACK = 168
|
|
|
|
|
|
def _contains_term(text, terms):
|
|
text = str(text or '').lower()
|
|
return any(term in text for term in terms)
|
|
|
|
|
|
def github_archive_event_score(event):
|
|
repo = event.get('repo') if isinstance(event.get('repo'), dict) else {}
|
|
repo_name = str(repo.get('name') or '').lower()
|
|
score = 0
|
|
if _contains_term(repo_name.replace('-', ' ').replace('_', ' '), GHARCHIVE_REPO_TERMS):
|
|
score += 8
|
|
payload = event.get('payload') if isinstance(event.get('payload'), dict) else {}
|
|
ref = str(payload.get('ref') or '').lower()
|
|
if _contains_term(ref, GHARCHIVE_REPO_TERMS):
|
|
score += 3
|
|
commits = payload.get('commits') if isinstance(payload.get('commits'), list) else []
|
|
for commit in commits[:20]:
|
|
if not isinstance(commit, dict):
|
|
continue
|
|
message = str(commit.get('message') or '')
|
|
if _contains_term(message, GHARCHIVE_MESSAGE_TERMS):
|
|
score += 5
|
|
for key in ('added', 'modified', 'removed'):
|
|
paths = commit.get(key) if isinstance(commit.get(key), list) else []
|
|
for path in paths[:50]:
|
|
if _contains_term(path, GHARCHIVE_PATH_TERMS):
|
|
score += 4
|
|
if event.get('type') == 'PushEvent':
|
|
score += 1
|
|
return score
|
|
|
|
|
|
def github_archive_event_branch(event):
|
|
payload = event.get('payload') if isinstance(event.get('payload'), dict) else {}
|
|
ref = str(payload.get('ref') or '').strip()
|
|
if ref.startswith('refs/heads/'):
|
|
return ref[len('refs/heads/'):], ref
|
|
if payload.get('ref_type') == 'branch' and ref:
|
|
return ref, f'refs/heads/{ref}'
|
|
return '', ref
|
|
|
|
|
|
def github_archive_target_payload(repo_name, event, score, archive_hour):
|
|
payload = event.get('payload') if isinstance(event.get('payload'), dict) else {}
|
|
branch, ref = github_archive_event_branch(event)
|
|
data = {
|
|
'url': f'https://github.com/{repo_name}.git',
|
|
'repo': repo_name,
|
|
'source': 'gharchive',
|
|
'event_type': event.get('type') or '',
|
|
'archive_hour': archive_hour.isoformat(),
|
|
'score': int(score or 0),
|
|
'ref': ref,
|
|
'branch': branch,
|
|
'head_sha': payload.get('head') or payload.get('after') or '',
|
|
}
|
|
return json.dumps(data, separators=(',', ':'), ensure_ascii=False, sort_keys=True)
|
|
|
|
|
|
def gharchive_changed_paths(event):
|
|
payload = event.get('payload') if isinstance(event.get('payload'), dict) else {}
|
|
commits = payload.get('commits') if isinstance(payload.get('commits'), list) else []
|
|
for commit in commits[:20]:
|
|
if not isinstance(commit, dict):
|
|
continue
|
|
sha = commit.get('sha') or commit.get('id') or payload.get('head') or payload.get('after') or ''
|
|
for key in ('added', 'modified'):
|
|
paths = commit.get(key) if isinstance(commit.get(key), list) else []
|
|
for path in paths[:80]:
|
|
path = str(path or '').strip().replace('\\', '/')
|
|
if path:
|
|
yield sha, path
|
|
|
|
|
|
def gharchive_commit_urls(event, repo_name):
|
|
payload = event.get('payload') if isinstance(event.get('payload'), dict) else {}
|
|
commits = payload.get('commits') if isinstance(payload.get('commits'), list) else []
|
|
seen = set()
|
|
for commit in commits[:5]:
|
|
if not isinstance(commit, dict):
|
|
continue
|
|
sha = commit.get('sha') or commit.get('id') or ''
|
|
url = commit.get('url') or (f'https://api.github.com/repos/{repo_name}/commits/{sha}' if sha else '')
|
|
if sha and url and sha not in seen:
|
|
seen.add(sha)
|
|
yield sha, url
|
|
head = payload.get('head') or payload.get('after') or ''
|
|
if head and head not in seen:
|
|
yield head, f'https://api.github.com/repos/{repo_name}/commits/{head}'
|
|
|
|
|
|
def fetch_github_commit_files(event, repo_name, headers, request_timeout=20, deadline=None):
|
|
for sha, url in gharchive_commit_urls(event, repo_name):
|
|
if deadline is not None and time.monotonic() >= deadline:
|
|
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during commit lookup')
|
|
timeout = request_timeout
|
|
if deadline is not None:
|
|
timeout = max(0.01, min(float(request_timeout or 20), deadline - time.monotonic()))
|
|
try:
|
|
response = api_request(
|
|
'GET', url, headers=headers, timeout=timeout, max_retries=2,
|
|
retry_delay=1, deadline=deadline,
|
|
)
|
|
if response.status_code == 404:
|
|
continue
|
|
if response.status_code in (401, 403, 429):
|
|
raise github_api_error(response)
|
|
response.raise_for_status()
|
|
data = response.json()
|
|
except (requests.exceptions.RequestException, ApiRequestError) as e:
|
|
logger.warning(f'Unable to fetch commit files {repo_name}@{sha}: {str(e)[:300]}')
|
|
continue
|
|
files = data.get('files') if isinstance(data.get('files'), list) else []
|
|
for item in files[:100]:
|
|
if not isinstance(item, dict):
|
|
continue
|
|
status = str(item.get('status') or '')
|
|
if status not in ('added', 'modified', 'renamed'):
|
|
continue
|
|
path = item.get('filename') or item.get('previous_filename') or ''
|
|
raw_url = item.get('raw_url') or github_raw_url(repo_name, sha, path)
|
|
if path and raw_url:
|
|
yield sha, str(path).replace('\\', '/'), raw_url
|
|
|
|
|
|
def gharchive_path_interesting(path):
|
|
lowered = str(path or '').lower()
|
|
if not lowered or lowered.endswith(GHARCHIVE_FILE_SKIP_SUFFIXES):
|
|
return False
|
|
normalized = '/' + lowered.strip('/')
|
|
if any(part in normalized for part in GHARCHIVE_FILE_SKIP_PARTS):
|
|
return False
|
|
return _contains_term(lowered, GHARCHIVE_FILE_FETCH_TERMS)
|
|
|
|
|
|
def github_raw_url(repo_name, sha, path):
|
|
if not repo_name or not sha or not path:
|
|
return ''
|
|
return f'https://raw.githubusercontent.com/{repo_name}/{sha}/{quote(path)}'
|
|
|
|
|
|
def gharchive_file_target(source, kind, cache_path, digest, origin=None, **extra):
|
|
return postman_target_from_cached_artifact(source, kind, cache_path, digest, origin, **extra)
|
|
|
|
|
|
class GHArchiveBoundsError(ValueError):
|
|
pass
|
|
|
|
|
|
class GHArchiveCacheCapacityError(RuntimeError):
|
|
pass
|
|
|
|
|
|
def gharchive_item_lock_path(cache_dir, artifact_path):
|
|
digest = hashlib.sha256(canonical_path(artifact_path).encode('utf-8', errors='strict')).hexdigest()
|
|
return os.path.join(cache_dir, f'.item-lock-{int(digest[:8], 16) % 64:02d}.lock')
|
|
|
|
|
|
def _gharchive_limits():
|
|
return {
|
|
'max_items': max(1, int(scan_config.gharchive_cache_max_items)),
|
|
'max_bytes': max(1, int(scan_config.gharchive_cache_max_bytes)),
|
|
'min_free_bytes': max(0, int(scan_config.gharchive_cache_min_free_bytes)),
|
|
'download_max_bytes': max(1, int(scan_config.gharchive_download_max_bytes)),
|
|
'decompressed_max_bytes': max(1, int(scan_config.gharchive_decompressed_max_bytes)),
|
|
'max_events': max(1, int(scan_config.gharchive_max_events)),
|
|
'max_line_bytes': max(1, int(scan_config.gharchive_max_line_bytes)),
|
|
'lock_timeout_sec': max(1, int(scan_config.gharchive_cache_lock_timeout_sec)),
|
|
}
|
|
|
|
|
|
def require_gharchive_cache_dir(cache_dir=None):
|
|
cache_dir = canonical_path(cache_dir or scan_config.gharchive_cache_dir)
|
|
runtime_dir = canonical_path(scan_config.runtime_dir)
|
|
require_private_directory(runtime_dir, create=False)
|
|
try:
|
|
contained = os.path.commonpath((runtime_dir, cache_dir)) == runtime_dir
|
|
except ValueError:
|
|
contained = False
|
|
if not contained or cache_dir == runtime_dir:
|
|
raise RuntimeError('GHArchive cache must be a dedicated private directory under runtime_dir')
|
|
return require_private_directory(cache_dir, create=True)
|
|
|
|
|
|
def iter_gharchive_lines(path, decompressed_max_bytes=None, max_events=None, max_line_bytes=None, deadline=None):
|
|
_raise_if_scan_slot_fatal()
|
|
limits = _gharchive_limits()
|
|
decompressed_max_bytes = max(1, int(decompressed_max_bytes or limits['decompressed_max_bytes']))
|
|
max_events = max(1, int(max_events or limits['max_events']))
|
|
max_line_bytes = max(1, int(max_line_bytes or limits['max_line_bytes']))
|
|
total = 0
|
|
events = 0
|
|
with gzip.open(path, 'rb') as archive:
|
|
while True:
|
|
_raise_if_scan_slot_fatal()
|
|
if deadline is not None and time.monotonic() >= deadline:
|
|
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached while reading GHArchive data')
|
|
raw_line = archive.readline(max_line_bytes + 1)
|
|
if not raw_line:
|
|
break
|
|
if len(raw_line) > max_line_bytes:
|
|
raise GHArchiveBoundsError('GHArchive line exceeds configured byte limit')
|
|
total += len(raw_line)
|
|
if total > decompressed_max_bytes:
|
|
raise GHArchiveBoundsError('GHArchive decompressed bytes exceed configured limit')
|
|
events += 1
|
|
if events > max_events:
|
|
raise GHArchiveBoundsError('GHArchive event count exceeds configured limit')
|
|
yield raw_line
|
|
|
|
|
|
def validate_gharchive_gzip(path, decompressed_max_bytes=None, max_events=None, max_line_bytes=None, deadline=None):
|
|
for _ in iter_gharchive_lines(path, decompressed_max_bytes, max_events, max_line_bytes, deadline):
|
|
pass
|
|
return True
|
|
|
|
|
|
def gharchive_cache_usage(cache_dir):
|
|
usage = {'items': 0, 'files': 0, 'bytes': 0}
|
|
with os.scandir(cache_dir) as entries:
|
|
for entry in entries:
|
|
if entry.is_symlink() or is_reparse_point(entry.path) or not entry.is_file(follow_symlinks=False):
|
|
continue
|
|
size = entry.stat(follow_symlinks=False).st_size
|
|
usage['files'] += 1
|
|
usage['bytes'] += max(0, int(size))
|
|
if entry.name.endswith('.json.gz'):
|
|
usage['items'] += 1
|
|
return usage
|
|
|
|
|
|
def _gharchive_artifacts_oldest(cache_dir, protected):
|
|
candidates = []
|
|
with os.scandir(cache_dir) as entries:
|
|
for entry in entries:
|
|
canonical = canonical_path(entry.path)
|
|
if (
|
|
canonical in protected
|
|
or not entry.name.endswith('.json.gz')
|
|
or entry.is_symlink()
|
|
or is_reparse_point(entry.path)
|
|
or not entry.is_file(follow_symlinks=False)
|
|
):
|
|
continue
|
|
details = entry.stat(follow_symlinks=False)
|
|
candidates.append((details.st_mtime_ns, canonical))
|
|
return [path for _, path in sorted(candidates)]
|
|
|
|
|
|
def _gharchive_evict_for_capacity(cache_dir, required_items=0, required_bytes=0, protected=None):
|
|
limits = _gharchive_limits()
|
|
protected = {canonical_path(path) for path in (protected or set())}
|
|
while True:
|
|
usage = gharchive_cache_usage(cache_dir)
|
|
try:
|
|
free_bytes = shutil.disk_usage(cache_dir).free
|
|
except OSError as exc:
|
|
raise GHArchiveCacheCapacityError(f'unable to inspect GHArchive cache free space: {exc}') from exc
|
|
if (
|
|
usage['items'] + int(required_items) <= limits['max_items']
|
|
and usage['bytes'] + int(required_bytes) <= limits['max_bytes']
|
|
and free_bytes - int(required_bytes) >= limits['min_free_bytes']
|
|
):
|
|
return usage
|
|
evicted = False
|
|
for candidate in _gharchive_artifacts_oldest(cache_dir, protected):
|
|
item_lock = PrivateFileLock(gharchive_item_lock_path(cache_dir, candidate))
|
|
try:
|
|
item_lock.acquire()
|
|
except BlockingIOError:
|
|
continue
|
|
try:
|
|
if os.path.isfile(candidate):
|
|
reject_reparse_components(candidate)
|
|
os.remove(candidate)
|
|
evicted = True
|
|
break
|
|
finally:
|
|
item_lock.release()
|
|
if not evicted:
|
|
raise GHArchiveCacheCapacityError('GHArchive cache quota or free-space reserve cannot be satisfied without evicting an active entry')
|
|
|
|
|
|
def _acquire_cached_gharchive_hour(hour, cache_dir=None, request_timeout=120, retries=4, deadline=None):
|
|
cache_dir = require_gharchive_cache_dir(cache_dir)
|
|
limits = _gharchive_limits()
|
|
name = f'{hour:%Y-%m-%d-%H}.json.gz'
|
|
path = canonical_path(os.path.join(cache_dir, name))
|
|
global_lock_path = os.path.join(cache_dir, '.cache.lock')
|
|
if deadline is not None and time.monotonic() >= deadline:
|
|
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GHArchive cache access')
|
|
lock_timeout = limits['lock_timeout_sec']
|
|
if deadline is not None:
|
|
lock_timeout = min(lock_timeout, max(0.01, deadline - time.monotonic()))
|
|
try:
|
|
global_lock = acquire_file_lock(global_lock_path, timeout_sec=lock_timeout)
|
|
except TimeoutError as exc:
|
|
if deadline is not None and time.monotonic() >= deadline:
|
|
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached waiting for GHArchive cache lock') from exc
|
|
raise
|
|
item_lock = None
|
|
try:
|
|
item_lock_timeout = limits['lock_timeout_sec']
|
|
if deadline is not None:
|
|
remaining = deadline - time.monotonic()
|
|
if remaining <= 0:
|
|
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GHArchive item lock')
|
|
item_lock_timeout = min(item_lock_timeout, max(0.01, remaining))
|
|
try:
|
|
item_lock = acquire_file_lock(
|
|
gharchive_item_lock_path(cache_dir, path),
|
|
timeout_sec=item_lock_timeout,
|
|
)
|
|
except TimeoutError as exc:
|
|
if deadline is not None and time.monotonic() >= deadline:
|
|
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached waiting for GHArchive item lock') from exc
|
|
raise
|
|
for entry in os.scandir(cache_dir):
|
|
if entry.name.startswith(name + '.part.') and entry.is_file(follow_symlinks=False):
|
|
remove_file_quiet(entry.path)
|
|
if os.path.lexists(path):
|
|
reject_reparse_components(path)
|
|
if not private_file_ready(path):
|
|
raise RuntimeError(f'GHArchive cache file is not private: {path}')
|
|
try:
|
|
validate_gharchive_gzip(path, deadline=deadline)
|
|
os.utime(path, None)
|
|
return path, item_lock
|
|
except (OSError, EOFError, gzip.BadGzipFile, GHArchiveBoundsError):
|
|
remove_file_quiet(path)
|
|
|
|
_gharchive_evict_for_capacity(cache_dir, required_items=1, protected={path})
|
|
url = f'https://data.gharchive.org/{name}'
|
|
last_error = None
|
|
attempts = max(1, int(retries or 1))
|
|
for attempt in range(attempts):
|
|
_raise_if_scan_slot_fatal()
|
|
if deadline is not None and time.monotonic() >= deadline:
|
|
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GHArchive download')
|
|
temp_path = f'{path}.part.{os.getpid()}.{attempt}'
|
|
remove_file_quiet(temp_path)
|
|
response = None
|
|
try:
|
|
logger.info(f'Downloading GHArchive {name} attempt {attempt + 1}/{attempts}')
|
|
request_deadline = None if deadline is None else max(0.01, deadline - time.monotonic())
|
|
response = _direct_request(
|
|
'GET', url,
|
|
headers={'User-Agent': 'GitSecretsScanner/2.0'},
|
|
stream=True,
|
|
timeout=(
|
|
min(10, request_deadline) if request_deadline is not None else 10,
|
|
min(max(30, int(request_timeout or 120)), request_deadline) if request_deadline is not None else max(30, int(request_timeout or 120)),
|
|
),
|
|
)
|
|
if _scan_slot_fatal_event.is_set():
|
|
_raise_if_scan_slot_fatal()
|
|
if response.status_code == 404:
|
|
return None, item_lock
|
|
response.raise_for_status()
|
|
declared = response.headers.get('Content-Length')
|
|
if declared and int(declared) > limits['download_max_bytes']:
|
|
raise GHArchiveBoundsError('GHArchive compressed response exceeds configured byte limit')
|
|
total = 0
|
|
with open(temp_path, 'xb') as output:
|
|
for chunk in response.iter_content(chunk_size=1024 * 1024):
|
|
_raise_if_scan_slot_fatal()
|
|
if deadline is not None and time.monotonic() >= deadline:
|
|
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during GHArchive download')
|
|
if not chunk:
|
|
continue
|
|
total += len(chunk)
|
|
if total > limits['download_max_bytes']:
|
|
raise GHArchiveBoundsError('GHArchive compressed response exceeds configured byte limit')
|
|
_gharchive_evict_for_capacity(
|
|
cache_dir,
|
|
required_bytes=len(chunk),
|
|
protected={path, temp_path},
|
|
)
|
|
output.write(chunk)
|
|
output.flush()
|
|
os.fsync(output.fileno())
|
|
_raise_if_scan_slot_fatal()
|
|
harden_private_file(temp_path)
|
|
validate_gharchive_gzip(temp_path, deadline=deadline)
|
|
_raise_if_scan_slot_fatal()
|
|
os.replace(temp_path, path)
|
|
harden_private_file(path)
|
|
return path, item_lock
|
|
except _PostmanHarvestDeadlineReached:
|
|
remove_file_quiet(temp_path)
|
|
raise
|
|
except (
|
|
requests.RequestException, OSError, EOFError, ValueError, gzip.BadGzipFile,
|
|
GHArchiveBoundsError, GHArchiveCacheCapacityError,
|
|
urllib3_exceptions.ProtocolError, urllib3_exceptions.ReadTimeoutError,
|
|
) as exc:
|
|
last_error = exc
|
|
remove_file_quiet(temp_path)
|
|
if attempt + 1 < attempts:
|
|
delay = min(30, 2 ** attempt)
|
|
if deadline is not None and time.monotonic() + delay >= deadline:
|
|
raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during GHArchive retry') from exc
|
|
_wait_or_raise_scan_slot_fatal(delay)
|
|
finally:
|
|
if response is not None:
|
|
response.close()
|
|
raise ApiRequestError(f'GHArchive download failed after {attempts} attempt(s): {name}: {last_error}')
|
|
except BaseException:
|
|
if item_lock is not None:
|
|
release_file_lock(item_lock, item_lock.path)
|
|
raise
|
|
finally:
|
|
release_file_lock(global_lock, global_lock_path)
|
|
|
|
|
|
def cached_gharchive_hour(hour, cache_dir=None, request_timeout=120, retries=4, deadline=None):
|
|
path, item_lock = _acquire_cached_gharchive_hour(hour, cache_dir, request_timeout, retries, deadline)
|
|
if item_lock is not None:
|
|
release_file_lock(item_lock, item_lock.path)
|
|
return path
|
|
|
|
|
|
@contextmanager
|
|
def cached_gharchive_hour_reader(hour, cache_dir=None, request_timeout=120, retries=4, deadline=None):
|
|
path = None
|
|
item_lock = None
|
|
try:
|
|
path, item_lock = _acquire_cached_gharchive_hour(hour, cache_dir, request_timeout, retries, deadline)
|
|
yield path
|
|
finally:
|
|
if item_lock is not None:
|
|
release_file_lock(item_lock, item_lock.path)
|
|
|
|
|
|
def fetch_github_archive_repos(hours_back=6, max_repos=200, event_types=None, request_timeout=120, archive_cache_dir=None):
|
|
"""Fetch recently active public GitHub repositories from GHArchive hourly dumps."""
|
|
event_types = {str(item).strip() for item in (event_types or []) if str(item).strip()}
|
|
if not event_types:
|
|
event_types = {'PushEvent', 'CreateEvent', 'PublicEvent'}
|
|
hours_back = min(GHARCHIVE_MAX_HOURS_BACK, max(1, int(hours_back or 1)))
|
|
max_repos = max(1, int(max_repos or 1))
|
|
now = datetime.now(timezone.utc).replace(minute=0, second=0, microsecond=0)
|
|
candidates = {}
|
|
order = 0
|
|
for offset in range(1, hours_back + 1):
|
|
hour = now - timedelta(hours=offset)
|
|
url = f'https://data.gharchive.org/{hour:%Y-%m-%d-%H}.json.gz'
|
|
logger.info(f'Fetching GHArchive hour {hour.isoformat()} from {url}')
|
|
with cached_gharchive_hour_reader(hour, archive_cache_dir, request_timeout) as archive_path:
|
|
if not archive_path:
|
|
logger.info(f'GHArchive hour unavailable yet: {url}')
|
|
continue
|
|
try:
|
|
for raw_line in iter_gharchive_lines(archive_path):
|
|
try:
|
|
event = json.loads(raw_line.decode('utf-8', errors='replace'))
|
|
except (ValueError, UnicodeDecodeError):
|
|
continue
|
|
if event.get('type') not in event_types:
|
|
continue
|
|
repo = event.get('repo') if isinstance(event.get('repo'), dict) else {}
|
|
name = str(repo.get('name') or '').strip()
|
|
if not name or '/' not in name:
|
|
continue
|
|
if not re.match(r'^[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$', name):
|
|
continue
|
|
key = name.lower()
|
|
order += 1
|
|
score = github_archive_event_score(event)
|
|
existing = candidates.get(key)
|
|
if existing and score > existing['score']:
|
|
candidates[key] = {
|
|
'name': name,
|
|
'score': score,
|
|
'order': existing['order'],
|
|
'event': event,
|
|
'archive_hour': hour,
|
|
}
|
|
elif not existing:
|
|
candidate = {
|
|
'name': name,
|
|
'score': score,
|
|
'order': order,
|
|
'event': event,
|
|
'archive_hour': hour,
|
|
}
|
|
if len(candidates) < max_repos:
|
|
candidates[key] = candidate
|
|
else:
|
|
worst_key = min(
|
|
candidates,
|
|
key=lambda value: (candidates[value]['score'], -candidates[value]['order']),
|
|
)
|
|
worst = candidates[worst_key]
|
|
if (score, -order) > (worst['score'], -worst['order']):
|
|
del candidates[worst_key]
|
|
candidates[key] = candidate
|
|
except (OSError, EOFError, gzip.BadGzipFile, GHArchiveBoundsError, urllib3_exceptions.ProtocolError) as e:
|
|
remove_file_quiet(archive_path)
|
|
raise ApiRequestError(f'Unable to read cached GHArchive gzip {url}: {e}') from e
|
|
ranked = sorted(candidates.values(), key=lambda item: (-item['score'], item['order']))
|
|
selected = ranked[:max_repos]
|
|
positive = sum(1 for item in selected if item['score'] > 0)
|
|
logger.info(f'GHArchive selected {len(selected)} repos from {len(candidates)} candidates; positive_score={positive}')
|
|
return [github_archive_target_payload(item['name'], item['event'], item['score'], item['archive_hour']) for item in selected]
|
|
|
|
|
|
def fetch_github_archive_file_targets(hours_back=6, max_files=300, event_types=None, request_timeout=120, cache_dir=None, max_file_size_mb=2, token=None, max_commit_lookups=200, archive_cache_dir=None, discovery_max_artifacts=None, discovery_max_artifacts_per_page=None, discovery_max_bytes=None, discovery_max_elapsed_sec=None):
|
|
event_types = {str(item).strip() for item in (event_types or []) if str(item).strip()} or {'PushEvent'}
|
|
hours_back = min(GHARCHIVE_MAX_HOURS_BACK, max(1, int(hours_back or 1)))
|
|
max_files = max(1, int(max_files or 1))
|
|
max_bytes = int(max_file_size_mb or 2) * 1024 * 1024
|
|
budget = _PostmanDiscoveryBudget(
|
|
'GHArchive-file',
|
|
discovery_max_artifacts,
|
|
discovery_max_artifacts_per_page,
|
|
discovery_max_bytes,
|
|
discovery_max_elapsed_sec,
|
|
)
|
|
now = datetime.now(timezone.utc).replace(minute=0, second=0, microsecond=0)
|
|
targets = []
|
|
seen = set()
|
|
seen_digests = set()
|
|
headers = github_headers(token)
|
|
commit_lookups = 0
|
|
stop_cycle = False
|
|
for offset in range(1, hours_back + 1):
|
|
if stop_cycle or not budget.check_deadline():
|
|
break
|
|
hour = now - timedelta(hours=offset)
|
|
url = f'https://data.gharchive.org/{hour:%Y-%m-%d-%H}.json.gz'
|
|
logger.info(f'Fetching GHArchive file candidates hour {hour.isoformat()} from {url}')
|
|
prepared = []
|
|
pending_keys = set()
|
|
page_artifacts = 0
|
|
stop_hour = False
|
|
try:
|
|
with cached_gharchive_hour_reader(
|
|
hour, archive_cache_dir, budget.request_timeout(request_timeout), deadline=budget.deadline,
|
|
) as archive_path:
|
|
if not archive_path:
|
|
continue
|
|
for raw_line in iter_gharchive_lines(archive_path, deadline=budget.deadline):
|
|
if not budget.check_deadline('while reading GHArchive events'):
|
|
stop_cycle = True
|
|
break
|
|
try:
|
|
event = json.loads(raw_line.decode('utf-8', errors='replace'))
|
|
except (ValueError, UnicodeDecodeError):
|
|
continue
|
|
if event.get('type') not in event_types:
|
|
continue
|
|
repo = event.get('repo') if isinstance(event.get('repo'), dict) else {}
|
|
repo_name = str(repo.get('name') or '').strip()
|
|
if not repo_name or '/' not in repo_name:
|
|
continue
|
|
path_items = [(sha, path, github_raw_url(repo_name, sha, path)) for sha, path in gharchive_changed_paths(event)]
|
|
if not path_items and github_archive_event_score(event) <= 0:
|
|
continue
|
|
if not path_items:
|
|
if commit_lookups >= int(max_commit_lookups or 0):
|
|
continue
|
|
commit_lookups += 1
|
|
try:
|
|
path_items = list(fetch_github_commit_files(
|
|
event,
|
|
repo_name,
|
|
headers,
|
|
budget.request_timeout(request_timeout),
|
|
budget.deadline,
|
|
))
|
|
except _PostmanHarvestDeadlineReached as e:
|
|
budget.stop(str(e))
|
|
stop_cycle = True
|
|
break
|
|
if not path_items and commit_lookups >= int(max_commit_lookups or 0):
|
|
logger.info(f'GHArchive file fetch reached commit lookup cap: {max_commit_lookups}')
|
|
for sha, path, raw_url in path_items:
|
|
if not budget.check_deadline('while examining GHArchive paths'):
|
|
stop_cycle = True
|
|
break
|
|
if len(targets) + len(prepared) >= max_files:
|
|
stop_cycle = True
|
|
break
|
|
if not gharchive_path_interesting(path):
|
|
continue
|
|
key = f'{repo_name.lower()}@{sha}:{path.lower()}'
|
|
if key in seen or key in pending_keys:
|
|
continue
|
|
if not raw_url:
|
|
continue
|
|
admitted, scope = budget.admit(page_artifacts)
|
|
if not admitted:
|
|
stop_cycle = scope == 'cycle'
|
|
stop_hour = True
|
|
break
|
|
page_artifacts += 1
|
|
pending_keys.add(key)
|
|
try:
|
|
raw = api_request(
|
|
'GET', raw_url, timeout=budget.request_timeout(request_timeout),
|
|
use_proxy=False,
|
|
max_retries=2, retry_delay=1,
|
|
retry_statuses={408, 500, 502, 503, 504},
|
|
deadline=budget.deadline,
|
|
)
|
|
if raw.status_code == 404:
|
|
continue
|
|
raw.raise_for_status()
|
|
content = raw.content
|
|
except _PostmanHarvestDeadlineReached as e:
|
|
budget.stop(str(e))
|
|
stop_cycle = True
|
|
break
|
|
except (requests.exceptions.RequestException, ApiRequestError) as e:
|
|
logger.warning(f'Unable to fetch GHArchive raw file {repo_name}:{path}: {str(e)[:300]}')
|
|
continue
|
|
if max_bytes and len(content) > max_bytes:
|
|
continue
|
|
if not budget.account_bytes(len(content)):
|
|
stop_cycle = True
|
|
break
|
|
origin = {
|
|
'provider': 'gharchive_file',
|
|
'repo': repo_name,
|
|
'path': path,
|
|
'sha': sha,
|
|
'raw_url': raw_url,
|
|
'archive_hour': hour.isoformat(),
|
|
'event_type': event.get('type') or '',
|
|
}
|
|
kind = postman_kind_for_path(path) or 'gharchive_file'
|
|
try:
|
|
entry = _prepare_postman_cache_entry(content, kind, origin, max_file_size_mb)
|
|
except Exception as e:
|
|
if isinstance(e, PostmanCacheCapacityError):
|
|
raise
|
|
logger.warning(f'Unable to cache GHArchive file {repo_name}:{path}: {str(e)[:300]}')
|
|
continue
|
|
if entry['digest'] in seen_digests:
|
|
continue
|
|
seen_digests.add(entry['digest'])
|
|
prepared.append({
|
|
'entry': entry,
|
|
'key': key,
|
|
'kind': kind,
|
|
'origin': origin,
|
|
'repo': repo_name,
|
|
'path': path,
|
|
'sha': sha,
|
|
'raw_url': raw_url,
|
|
})
|
|
if stop_cycle or stop_hour:
|
|
break
|
|
except _PostmanHarvestDeadlineReached as e:
|
|
budget.stop(str(e))
|
|
stop_cycle = True
|
|
except (OSError, EOFError, gzip.BadGzipFile, GHArchiveBoundsError, urllib3_exceptions.ProtocolError) as e:
|
|
if 'archive_path' in locals() and archive_path:
|
|
remove_file_quiet(archive_path)
|
|
raise ApiRequestError(f'Unable to read cached GHArchive gzip {url}: {e}') from e
|
|
|
|
published, deadline_reached = _publish_postman_discovery_batch(prepared, cache_dir, budget)
|
|
for record, (cache_path, digest, size) in published:
|
|
target = gharchive_file_target(
|
|
'github_archive_file', record['kind'], cache_path, digest, record['origin'],
|
|
repo=record['repo'], path=record['path'], sha=record['sha'],
|
|
raw_url=record['raw_url'], size=size,
|
|
)
|
|
targets.append(target)
|
|
seen.add(record['key'])
|
|
if deadline_reached or stop_cycle or budget.exhausted():
|
|
stop_cycle = True
|
|
break
|
|
logger.info(f'GHArchive file fetch produced {len(targets)} targets; commit_lookups={commit_lookups}')
|
|
return targets
|
|
|
|
|
|
def gist_file_interesting(filename, file_meta):
|
|
text = ' '.join([
|
|
str(filename or '').lower(),
|
|
str((file_meta or {}).get('type') or '').lower(),
|
|
str((file_meta or {}).get('language') or '').lower(),
|
|
])
|
|
return _contains_term(text, GITHUB_GIST_FILE_TERMS)
|
|
|
|
|
|
def fetch_github_gist_targets(pages=2, per_page=100, since=None, token=None, cache_dir=None, max_file_size_mb=2, request_timeout=20, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, discovery_max_artifacts=None, discovery_max_artifacts_per_page=None, discovery_max_bytes=None, discovery_max_elapsed_sec=None):
|
|
headers = github_headers(token)
|
|
per_page = max(1, min(int(per_page or 100), 100))
|
|
max_pages = max(1, int(pages or 1))
|
|
max_bytes = int(max_file_size_mb or 2) * 1024 * 1024
|
|
budget = _PostmanDiscoveryBudget(
|
|
'GitHub Gist',
|
|
discovery_max_artifacts,
|
|
discovery_max_artifacts_per_page,
|
|
discovery_max_bytes,
|
|
discovery_max_elapsed_sec,
|
|
)
|
|
targets = []
|
|
seen_candidates = set()
|
|
seen_targets = set()
|
|
seen_digests = set()
|
|
known_pages = 0
|
|
stop_cycle = False
|
|
params_base = {'per_page': per_page}
|
|
if since:
|
|
params_base['since'] = since
|
|
for page in range(1, max_pages + 1):
|
|
if stop_cycle or not budget.check_deadline():
|
|
break
|
|
params = dict(params_base)
|
|
params['page'] = page
|
|
try:
|
|
response = api_request(
|
|
'GET', 'https://api.github.com/gists/public', headers=headers, params=params,
|
|
timeout=budget.request_timeout(request_timeout),
|
|
deadline=budget.deadline,
|
|
)
|
|
response.raise_for_status()
|
|
except _PostmanHarvestDeadlineReached as e:
|
|
budget.stop(str(e))
|
|
break
|
|
except requests.exceptions.HTTPError as e:
|
|
raise github_api_error(e.response) from e
|
|
except (requests.exceptions.RequestException, ApiRequestError) as e:
|
|
logger.warning(f'GitHub Gists request failed on page {page}: {str(e)[:300]}')
|
|
break
|
|
gists = response.json() or []
|
|
if not gists:
|
|
break
|
|
page_targets = []
|
|
prepared = []
|
|
pending_candidates = set()
|
|
page_artifacts = 0
|
|
stop_page = False
|
|
for gist in gists:
|
|
if not budget.check_deadline('while examining a Gist page'):
|
|
stop_cycle = True
|
|
break
|
|
gist_id = str(gist.get('id') or '')
|
|
files = gist.get('files') if isinstance(gist.get('files'), dict) else {}
|
|
for filename, meta in files.items():
|
|
if not budget.check_deadline('while examining Gist files'):
|
|
stop_cycle = True
|
|
break
|
|
if not isinstance(meta, dict):
|
|
continue
|
|
size = int(meta.get('size') or 0)
|
|
raw_url = meta.get('raw_url') or ''
|
|
if not raw_url or (max_bytes and size > max_bytes):
|
|
continue
|
|
if not gist_file_interesting(filename, meta):
|
|
continue
|
|
identity = f'gist:{gist_id}:{filename}:{meta.get("raw_url")}'
|
|
if identity in seen_candidates or identity in pending_candidates:
|
|
continue
|
|
admitted, scope = budget.admit(page_artifacts)
|
|
if not admitted:
|
|
stop_cycle = scope == 'cycle'
|
|
stop_page = True
|
|
break
|
|
page_artifacts += 1
|
|
pending_candidates.add(identity)
|
|
try:
|
|
raw = api_request(
|
|
'GET', raw_url, headers=headers,
|
|
use_proxy=False,
|
|
timeout=budget.request_timeout(request_timeout),
|
|
deadline=budget.deadline,
|
|
)
|
|
raw.raise_for_status()
|
|
content = raw.content
|
|
except _PostmanHarvestDeadlineReached as e:
|
|
budget.stop(str(e))
|
|
stop_cycle = True
|
|
break
|
|
except (requests.exceptions.RequestException, ApiRequestError) as e:
|
|
logger.warning(f'Unable to fetch gist raw {gist_id}/{filename}: {str(e)[:300]}')
|
|
continue
|
|
if max_bytes and len(content) > max_bytes:
|
|
continue
|
|
if not budget.account_bytes(len(content)):
|
|
stop_cycle = True
|
|
break
|
|
origin = {
|
|
'provider': 'github_gist',
|
|
'gist_id': gist_id,
|
|
'filename': filename,
|
|
'raw_url': raw_url,
|
|
'html_url': gist.get('html_url'),
|
|
'created_at': gist.get('created_at'),
|
|
'updated_at': gist.get('updated_at'),
|
|
'owner': ((gist.get('owner') or {}).get('login') if isinstance(gist.get('owner'), dict) else ''),
|
|
}
|
|
kind = postman_kind_for_path(filename) or 'gist_file'
|
|
try:
|
|
entry = _prepare_postman_cache_entry(content, kind, origin, max_file_size_mb)
|
|
except Exception as e:
|
|
if isinstance(e, PostmanCacheCapacityError):
|
|
raise
|
|
logger.warning(f'Unable to cache gist {gist_id}/{filename}: {str(e)[:300]}')
|
|
continue
|
|
if entry['digest'] in seen_digests:
|
|
continue
|
|
seen_digests.add(entry['digest'])
|
|
prepared.append({
|
|
'entry': entry,
|
|
'candidate_identity': identity,
|
|
'kind': kind,
|
|
'origin': origin,
|
|
'gist_id': gist_id,
|
|
'filename': filename,
|
|
'raw_url': raw_url,
|
|
'html_url': gist.get('html_url'),
|
|
})
|
|
if stop_cycle or stop_page:
|
|
break
|
|
|
|
published, deadline_reached = _publish_postman_discovery_batch(prepared, cache_dir, budget)
|
|
for record, (cache_path, digest, cached_size) in published:
|
|
target = postman_target_from_cached_artifact(
|
|
'github_gist', record['kind'], cache_path, digest, record['origin'],
|
|
gist_id=record['gist_id'], filename=record['filename'],
|
|
raw_url=record['raw_url'], html_url=record['html_url'], size=cached_size,
|
|
)
|
|
target_identity = postman_target_identity(target)
|
|
seen_candidates.add(record['candidate_identity'])
|
|
if target_identity in seen_targets:
|
|
continue
|
|
targets.append(target)
|
|
page_targets.append(target)
|
|
seen_targets.add(target_identity)
|
|
logger.info(f'GitHub Gists page {page}: gists={len(gists)}, targets={len(page_targets)}')
|
|
if stop_on_seen_pages and page >= max(1, int(min_pages_before_stop or 1)):
|
|
if page_targets and page_is_known(
|
|
page_targets, known_targets, normalize_target, known_target_lookup,
|
|
):
|
|
known_pages += 1
|
|
if known_pages >= max(1, int(seen_page_threshold or 1)):
|
|
break
|
|
else:
|
|
known_pages = 0
|
|
if deadline_reached or stop_cycle or budget.exhausted():
|
|
stop_cycle = True
|
|
break
|
|
return targets
|
|
|
|
def gitlab_project_to_target(project):
|
|
return {
|
|
'url': project.get('http_url_to_repo', ''),
|
|
'name': project.get('path_with_namespace', ''),
|
|
'created_at': project.get('created_at', ''),
|
|
'updated_at': project.get('updated_at') or project.get('last_activity_at', ''),
|
|
'last_activity_at': project.get('last_activity_at', ''),
|
|
}
|
|
|
|
def gitlab_rate_limit_reset(response):
|
|
if not response:
|
|
return None
|
|
reset = response.headers.get('RateLimit-Reset') or response.headers.get('X-RateLimit-Reset')
|
|
if not reset:
|
|
retry_after = response.headers.get('Retry-After')
|
|
if retry_after:
|
|
try:
|
|
return (datetime.now(timezone.utc) + timedelta(seconds=int(retry_after))).isoformat(timespec='seconds')
|
|
except (TypeError, ValueError):
|
|
return None
|
|
return None
|
|
try:
|
|
if str(reset).isdigit():
|
|
return datetime.fromtimestamp(int(reset), timezone.utc).isoformat(timespec='seconds')
|
|
return str(reset)
|
|
except (TypeError, ValueError):
|
|
return None
|
|
|
|
def fetch_gitlab_repo_items(query, pages, per_page=100, token=None, sort_by="last_activity_at", sort_order="desc", visibility="public", last_activity_after=None, raise_rate_limit=False, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, request_attempts=1, retry_delay=0):
|
|
"""Fetch GitLab repositories with metadata for filtering."""
|
|
repos = []
|
|
headers = {}
|
|
if token:
|
|
headers['Authorization'] = f'Bearer {token}'
|
|
|
|
logger.info(f"Fetching GitLab repositories for: '{query if query else 'all repositories'}' sorted by {sort_by} ({sort_order})...")
|
|
page = 1
|
|
|
|
seen_pages = 0
|
|
successful_pages = 0
|
|
request_attempts = max(1, int(request_attempts or 1))
|
|
retry_delay = max(0, int(retry_delay or 0))
|
|
request_budget = 30 * request_attempts + retry_delay * (request_attempts - 1)
|
|
while page <= pages:
|
|
url = "https://gitlab.com/api/v4/projects"
|
|
params = {
|
|
'visibility': visibility,
|
|
'per_page': per_page,
|
|
'page': page,
|
|
'order_by': sort_by,
|
|
'sort': sort_order,
|
|
}
|
|
if query:
|
|
params['search'] = query
|
|
if last_activity_after:
|
|
params['last_activity_after'] = last_activity_after
|
|
|
|
try:
|
|
response = api_request(
|
|
'GET', url, headers=headers, params=params, timeout=30,
|
|
max_retries=request_attempts, retry_delay=retry_delay,
|
|
deadline=time.monotonic() + request_budget,
|
|
)
|
|
response.raise_for_status()
|
|
|
|
# Handle rate limits
|
|
if response.status_code == 429:
|
|
retry_after = int(response.headers.get('Retry-After', 60))
|
|
logger.warning(f"Rate limit exceeded. Waiting {retry_after} seconds...")
|
|
time.sleep(retry_after)
|
|
continue
|
|
|
|
data = response.json()
|
|
if not isinstance(data, list):
|
|
raise ValueError('invalid GitLab project search payload')
|
|
successful_pages += 1
|
|
if not data:
|
|
logger.info(f"Page {page} returned no results. Stopping.")
|
|
break
|
|
|
|
page_repos = [gitlab_project_to_target(project) for project in data if project.get('http_url_to_repo')]
|
|
repos.extend(page_repos)
|
|
logger.info(f"Page {page}: Fetched {len(data)} repositories")
|
|
if stop_on_seen_pages and page >= max(1, min_pages_before_stop):
|
|
if page_is_known(
|
|
[item.get('url') for item in page_repos], known_targets,
|
|
normalize_target, known_target_lookup,
|
|
):
|
|
seen_pages += 1
|
|
logger.info(f"Page {page}: all GitLab repositories are already queued/checked ({seen_pages}/{seen_page_threshold})")
|
|
if seen_pages >= max(1, seen_page_threshold):
|
|
logger.info(f"Stopping GitLab pagination early after {seen_pages} all-known page(s)")
|
|
break
|
|
else:
|
|
seen_pages = 0
|
|
page += 1
|
|
|
|
except requests.exceptions.HTTPError as e:
|
|
if e.response.status_code == 401 and not token:
|
|
logger.warning("GitLab API authentication would improve results. Consider adding a GitLab token.")
|
|
raise gitlab_api_error(e.response) from e
|
|
else:
|
|
api_error = gitlab_api_error(e.response)
|
|
if raise_rate_limit:
|
|
raise api_error from e
|
|
logger.error(str(api_error))
|
|
raise api_error from e
|
|
except Exception as e:
|
|
if isinstance(e, ApiRequestError):
|
|
raise GitLabDiscoveryTransportError(str(e)) from e
|
|
logger.error(f"Error fetching page {page}: {str(e)}")
|
|
raise ApiRequestError(f'GitLab discovery payload failed: {e}') from e
|
|
|
|
return repos
|
|
|
|
def fetch_gitlab_repos(query, pages, per_page=100, token=None, sort_by="last_activity_at", sort_order="desc", visibility="public", raise_rate_limit=False, **kwargs):
|
|
"""Fetch GitLab repository clone URLs with pagination and authentication."""
|
|
return [
|
|
item['url'] for item in fetch_gitlab_repo_items(
|
|
query, pages, per_page, token, sort_by, sort_order, visibility, raise_rate_limit=raise_rate_limit, **kwargs
|
|
)
|
|
if item.get('url')
|
|
]
|
|
|
|
def github_recent_query(query, since):
|
|
since_str = since.strftime("%Y-%m-%d")
|
|
query = (query or '').strip()
|
|
if query and ' in:' not in f' {query.lower()} ':
|
|
query = f'{query} in:name,description,readme'
|
|
return f"{query} updated:>={since_str}" if query else f"updated:>={since_str}"
|
|
|
|
def fetch_recent_github_repos(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, **kwargs):
|
|
"""Fetch GitHub repositories updated since a specific timestamp"""
|
|
return fetch_github_repos(github_recent_query(query, since), pages=pages, per_page=per_page, token=token, raise_rate_limit=raise_rate_limit, **kwargs)
|
|
|
|
def fetch_recent_github_repo_items(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, **kwargs):
|
|
"""Fetch recent GitHub repositories with metadata."""
|
|
return fetch_github_repo_items(github_recent_query(query, since), pages=pages, per_page=per_page, token=token, raise_rate_limit=raise_rate_limit, **kwargs)
|
|
|
|
def fetch_recent_gitlab_repos(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, visibility="public", **kwargs):
|
|
"""Fetch GitLab repositories updated since a specific timestamp"""
|
|
since_str = since.strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
return [
|
|
item['url'] for item in fetch_gitlab_repo_items(
|
|
query, pages=pages, per_page=per_page, token=token,
|
|
sort_by="last_activity_at", sort_order="desc", visibility=visibility,
|
|
last_activity_after=since_str,
|
|
raise_rate_limit=raise_rate_limit,
|
|
**kwargs,
|
|
)
|
|
if item.get('url')
|
|
]
|
|
|
|
def fetch_recent_gitlab_repo_items(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, visibility="public", **kwargs):
|
|
"""Fetch recent GitLab repositories with metadata."""
|
|
since_str = since.strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
return fetch_gitlab_repo_items(
|
|
query, pages=pages, per_page=per_page, token=token,
|
|
sort_by="last_activity_at", sort_order="desc", visibility=visibility,
|
|
last_activity_after=since_str,
|
|
raise_rate_limit=raise_rate_limit,
|
|
**kwargs,
|
|
)
|
|
|
|
DOCKERHUB_SEARCH_MAX_PAGES = 30
|
|
DOCKERHUB_TAG_RATE_LIMIT_MIN_BACKOFF_SEC = 60
|
|
DOCKERHUB_TAG_RATE_LIMIT_MAX_BACKOFF_SEC = 3600
|
|
DOCKER_REGISTRY_MANIFEST_MAX_BYTES = 8 * 1024 * 1024
|
|
DOCKER_REGISTRY_MAX_DESCRIPTORS = 1000
|
|
DOCKER_REGISTRY_MAX_LAYERS = 2048
|
|
DOCKER_REGISTRY_TOKEN_MAX_BYTES = 1024 * 1024
|
|
DOCKER_CONFIG_MEDIA_TYPES = frozenset((
|
|
'application/vnd.oci.image.config.v1+json',
|
|
'application/vnd.docker.container.image.v1+json',
|
|
))
|
|
DOCKER_LAYER_MEDIA_TYPES = frozenset((
|
|
'application/vnd.oci.image.layer.v1.tar',
|
|
'application/vnd.oci.image.layer.v1.tar+gzip',
|
|
'application/vnd.oci.image.layer.v1.tar+zstd',
|
|
'application/vnd.docker.image.rootfs.diff.tar',
|
|
'application/vnd.docker.image.rootfs.diff.tar.gzip',
|
|
))
|
|
DOCKER_LAYER_GZIP_MEDIA_TYPES = frozenset((
|
|
'application/vnd.oci.image.layer.v1.tar+gzip',
|
|
'application/vnd.docker.image.rootfs.diff.tar.gzip',
|
|
))
|
|
DOCKER_CONFIG_HISTORY_MAX_ENTRIES = DOCKER_REGISTRY_MAX_LAYERS * 2
|
|
DOCKER_CONFIG_HISTORY_COMMAND_MAX_CHARS = 64 * 1024
|
|
|
|
|
|
def docker_history_payload_class(created_by):
|
|
if not isinstance(created_by, str) or not created_by:
|
|
return 'unknown'
|
|
if len(created_by) > DOCKER_CONFIG_HISTORY_COMMAND_MAX_CHARS:
|
|
return 'unknown'
|
|
command = created_by.casefold()
|
|
if re.search(r'(^|#\(nop\)\s+)(copy|add)\s', command.strip()):
|
|
return 'copy_add'
|
|
if not re.search(r'(^|[\s#])run(\s|$)|/bin/(ba)?sh\s+-c', command):
|
|
return 'unknown'
|
|
if any(token in command for token in (
|
|
'/models/', '/model/', 'model_weights', 'checkpoint.', '.safetensors',
|
|
'.gguf', '.onnx', '.pt ', '.pth ', 'huggingface-cli download',
|
|
)):
|
|
return 'bulk_data'
|
|
if re.search(
|
|
r'\b(apt-get|apt|apk|yum|dnf|microdnf|pip|pip3|poetry|npm|pnpm|yarn|'
|
|
r'bundle|gem|cargo)\s+(install|add|sync)\b',
|
|
command,
|
|
):
|
|
return 'package_run'
|
|
if any(token in command for token in (
|
|
'/app', '/srv', '/workspace', '/opt/app', 'config', '.env',
|
|
'requirements.txt', 'package.json', 'pyproject.toml',
|
|
)):
|
|
return 'app_config_run'
|
|
return 'other_run'
|
|
|
|
|
|
def docker_config_payload_classes(config, layer_count):
|
|
try:
|
|
layer_count = int(layer_count)
|
|
except (TypeError, ValueError):
|
|
layer_count = -1
|
|
fallback = ['unknown'] * max(0, layer_count)
|
|
if (
|
|
not isinstance(config, dict)
|
|
or layer_count < 0
|
|
or layer_count > DOCKER_REGISTRY_MAX_LAYERS
|
|
):
|
|
return fallback
|
|
history = config.get('history')
|
|
if not isinstance(history, list) or len(history) > DOCKER_CONFIG_HISTORY_MAX_ENTRIES:
|
|
return fallback
|
|
rootfs = config.get('rootfs')
|
|
if rootfs is not None:
|
|
if not isinstance(rootfs, dict):
|
|
return fallback
|
|
diff_ids = rootfs.get('diff_ids')
|
|
if not isinstance(diff_ids, list) or len(diff_ids) != layer_count:
|
|
return fallback
|
|
commands = []
|
|
for entry in history:
|
|
if not isinstance(entry, dict):
|
|
return fallback
|
|
empty_layer = entry.get('empty_layer', False)
|
|
if not isinstance(empty_layer, bool):
|
|
return fallback
|
|
if empty_layer:
|
|
continue
|
|
created_by = entry.get('created_by')
|
|
if not isinstance(created_by, str):
|
|
return fallback
|
|
commands.append(created_by)
|
|
if len(commands) != layer_count:
|
|
return fallback
|
|
classes = [docker_history_payload_class(command) for command in commands]
|
|
if any(payload_class not in DOCKER_ADAPTIVE_PAYLOAD_CLASSES for payload_class in classes):
|
|
return fallback
|
|
return classes
|
|
|
|
|
|
class DockerRegistryResolutionError(ValueError):
|
|
pass
|
|
|
|
|
|
class DockerResolverLeaseLostError(RuntimeError):
|
|
pass
|
|
|
|
|
|
class DockerRemoteAccessError(DockerRegistryResolutionError):
|
|
def __init__(self, message, status='unknown', retry_at=None, remote_attempted=True):
|
|
super().__init__(message)
|
|
self.status = str(status or 'unknown')
|
|
self.retry_at = retry_at
|
|
self.remote_attempted = bool(remote_attempted)
|
|
|
|
|
|
class DockerContentTransferError(RuntimeError):
|
|
def __init__(self, error_code, message, retryable=True):
|
|
super().__init__(message)
|
|
self.error_code = str(error_code or 'transfer_failed')
|
|
self.retryable = bool(retryable)
|
|
self.source_failure = False
|
|
self.transfer_bytes = 0
|
|
self.duration_ms = 0
|
|
|
|
|
|
class DockerLayerInfrastructureError(DockerContentTransferError):
|
|
def __init__(
|
|
self, error_code, message, category='remote_transient', auth_related=False,
|
|
):
|
|
super().__init__(error_code, message, retryable=True)
|
|
self.category = str(category or 'remote_transient')
|
|
self.auth_related = bool(auth_related)
|
|
self.source_failure = True
|
|
|
|
|
|
class DockerContentScanError(RuntimeError):
|
|
def __init__(self, error_code, message, retryable=False):
|
|
super().__init__(message)
|
|
self.error_code = str(error_code or 'invalid_content')
|
|
self.retryable = bool(retryable)
|
|
|
|
|
|
@dataclass(frozen=True, repr=False)
|
|
class DockerBlobDownloadOutcome:
|
|
path: str
|
|
verified_bytes: int
|
|
transfer_bytes: int
|
|
duration_ms: int
|
|
bearer_auth: DockerRegistryAuth
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class DockerTagResolutionOutcome:
|
|
tags: tuple
|
|
status: str
|
|
remote_attempted: bool
|
|
retry_at: str = None
|
|
error: str = ''
|
|
selection_records: tuple = ()
|
|
candidate_records: tuple = ()
|
|
selector_version: str = ''
|
|
selector_hash: str = ''
|
|
candidate_distinct_graph_count: int = 0
|
|
fresh_graph_evidence: bool = False
|
|
cache_bypassed: bool = False
|
|
|
|
@property
|
|
def selections(self):
|
|
return self.selection_records
|
|
|
|
@property
|
|
def selector_sha256(self):
|
|
return self.selector_hash
|
|
|
|
|
|
DOCKER_TAG_CONCLUSIVE_STATUSES = frozenset({'ok', 'empty', 'unsupported', 'not_found'})
|
|
|
|
|
|
def docker_tag_resolution_is_conclusive(status):
|
|
return str(status or '') in DOCKER_TAG_CONCLUSIVE_STATUSES
|
|
|
|
|
|
def docker_images_per_repository_limit(value):
|
|
return validate_docker_images_per_repository(value)
|
|
|
|
|
|
def dockerhub_search_page_window(pages):
|
|
try:
|
|
requested = max(1, int(pages or 1))
|
|
except (TypeError, ValueError):
|
|
requested = 1
|
|
return requested, min(requested, DOCKERHUB_SEARCH_MAX_PAGES)
|
|
|
|
|
|
def fetch_dockerhub_search_page(
|
|
query, page, per_page=100, sort_by='updated_at', sort_order='desc',
|
|
request_timeout=15,
|
|
):
|
|
"""Fetch and validate one bounded Docker Hub repository-search page."""
|
|
try:
|
|
page = int(page)
|
|
per_page = max(1, int(per_page or 1))
|
|
except (TypeError, ValueError, OverflowError):
|
|
raise DockerHubDiscoveryTransportError(
|
|
'Docker Hub search page request is invalid',
|
|
category='invalid_payload', remote_attempted=False, retryable=False,
|
|
) from None
|
|
if page < 1 or page > DOCKERHUB_SEARCH_MAX_PAGES:
|
|
raise DockerHubDiscoveryTransportError(
|
|
'Docker Hub search page request is invalid',
|
|
category='invalid_payload', remote_attempted=False, retryable=False,
|
|
)
|
|
|
|
params = {
|
|
'query': query,
|
|
'page': page,
|
|
'page_size': per_page,
|
|
'sort': sort_by,
|
|
'order': sort_order,
|
|
}
|
|
started = time.perf_counter()
|
|
try:
|
|
response = dockerhub_search_response(
|
|
'https://hub.docker.com/v2/search/repositories', params,
|
|
request_timeout=request_timeout,
|
|
)
|
|
except DockerRemoteAccessError as error:
|
|
category = {
|
|
'rate_limited': 'rate_limit',
|
|
'auth_failed': 'auth_unavailable',
|
|
'remote_transient': 'remote_transient',
|
|
}.get(error.status, 'page_unavailable')
|
|
if error.retry_at and not error.remote_attempted:
|
|
category = 'provider_cooldown'
|
|
raise DockerHubDiscoveryTransportError(
|
|
f'Docker Hub search page {page} failed after bounded attempts',
|
|
category=category, retry_at=error.retry_at,
|
|
remote_attempted=error.remote_attempted,
|
|
) from None
|
|
except ApiRequestError:
|
|
raise DockerHubDiscoveryTransportError(
|
|
f'Docker Hub search page {page} failed after bounded attempts',
|
|
category='network', remote_attempted=True,
|
|
) from None
|
|
except Exception:
|
|
raise DockerHubDiscoveryTransportError(
|
|
f'Docker Hub search page {page} failed after bounded attempts',
|
|
category='page_unavailable', remote_attempted=True,
|
|
) from None
|
|
|
|
try:
|
|
response.raise_for_status()
|
|
except Exception:
|
|
raise DockerHubDiscoveryTransportError(
|
|
f'Docker Hub search page {page} failed after bounded attempts',
|
|
category='page_unavailable', remote_attempted=True,
|
|
) from None
|
|
|
|
try:
|
|
data = response.json()
|
|
except Exception:
|
|
raise DockerHubDiscoveryTransportError(
|
|
f'Docker Hub search page {page} returned an invalid payload',
|
|
category='invalid_payload', remote_attempted=True, retryable=False,
|
|
) from None
|
|
|
|
if not isinstance(data, dict) or not isinstance(data.get('results'), list):
|
|
raise DockerHubDiscoveryTransportError(
|
|
f'Docker Hub search page {page} returned an invalid payload',
|
|
category='invalid_payload', remote_attempted=True, retryable=False,
|
|
)
|
|
|
|
raw_count = data.get('count')
|
|
try:
|
|
total_count = int(raw_count)
|
|
except (TypeError, ValueError, OverflowError):
|
|
raise DockerHubDiscoveryTransportError(
|
|
f'Docker Hub search page {page} returned an invalid result count',
|
|
category='invalid_payload', remote_attempted=True, retryable=False,
|
|
) from None
|
|
if (
|
|
isinstance(raw_count, bool)
|
|
or (isinstance(raw_count, float) and not raw_count.is_integer())
|
|
or total_count < 0
|
|
):
|
|
raise DockerHubDiscoveryTransportError(
|
|
f'Docker Hub search page {page} returned an invalid result count',
|
|
category='invalid_payload', remote_attempted=True, retryable=False,
|
|
)
|
|
|
|
result_count = len(data['results'])
|
|
absolute_start = (page - 1) * per_page
|
|
if (
|
|
result_count > per_page
|
|
or total_count < absolute_start + result_count
|
|
or (result_count == 0 and total_count > absolute_start)
|
|
):
|
|
raise DockerHubDiscoveryTransportError(
|
|
f'Docker Hub search page {page} returned incoherent pagination evidence',
|
|
category='invalid_payload', remote_attempted=True, retryable=False,
|
|
)
|
|
|
|
repositories = []
|
|
seen_repo_names = set()
|
|
for repository in data['results']:
|
|
if not isinstance(repository, dict):
|
|
raise DockerHubDiscoveryTransportError(
|
|
f'Docker Hub search page {page} returned an invalid payload',
|
|
category='invalid_payload', remote_attempted=True, retryable=False,
|
|
)
|
|
repo_name = repository.get('repo_name')
|
|
if not isinstance(repo_name, str) or not repo_name.strip():
|
|
raise DockerHubDiscoveryTransportError(
|
|
f'Docker Hub search page {page} returned an invalid payload',
|
|
category='invalid_payload', remote_attempted=True, retryable=False,
|
|
)
|
|
repo_name = repo_name.strip()
|
|
if repo_name in seen_repo_names:
|
|
continue
|
|
seen_repo_names.add(repo_name)
|
|
safe_repository = {'repo_name': repo_name}
|
|
for field in ('last_updated', 'last_modified'):
|
|
if field in repository:
|
|
safe_repository[field] = repository[field]
|
|
repositories.append(safe_repository)
|
|
|
|
return {
|
|
'page': page,
|
|
'repositories': repositories,
|
|
'total_count': total_count,
|
|
'elapsed': time.perf_counter() - started,
|
|
}
|
|
|
|
|
|
def fetch_dockerhub_images(query, pages, per_page=100, sort_by="updated_at", sort_order="desc", fetch_workers=8, request_timeout=15, resolve_tags=True, tag_fetch_workers=None, tag_retry_count=2, tag_retry_delay=5, platform_filter_enabled=False, platform_os='linux', platform_arch='amd64', platform_candidate_tags=20, images_per_repository=1):
|
|
"""Fetch Docker Hub images with pagination and sorting"""
|
|
images = []
|
|
per_page = max(1, int(per_page or 1))
|
|
requested_pages, pages = dockerhub_search_page_window(pages)
|
|
if requested_pages > pages:
|
|
logger.info(
|
|
f'Docker Hub search is limited to {pages} accessible page(s); '
|
|
f'capping requested pages from {requested_pages}'
|
|
)
|
|
logger.info(f"Fetching Docker Hub images for: '{query}' sorted by {sort_by} ({sort_order})...")
|
|
|
|
def fetch_page(page):
|
|
started = time.perf_counter()
|
|
try:
|
|
return fetch_dockerhub_search_page(
|
|
query, page, per_page=per_page, sort_by=sort_by,
|
|
sort_order=sort_order, request_timeout=request_timeout,
|
|
), None
|
|
except DockerHubDiscoveryTransportError as error:
|
|
return {
|
|
'page': page,
|
|
'repositories': [],
|
|
'total_count': 0,
|
|
'elapsed': time.perf_counter() - started,
|
|
}, error
|
|
|
|
first_page, first_error = fetch_page(1)
|
|
if first_error is not None:
|
|
raise first_error
|
|
|
|
total_count = first_page['total_count']
|
|
expected_pages = min(
|
|
pages,
|
|
max(1, (total_count + per_page - 1) // per_page),
|
|
)
|
|
page_results = [(first_page, None)]
|
|
if expected_pages > 1:
|
|
max_workers = max(1, min(fetch_workers, expected_pages - 1))
|
|
with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
|
|
futures = [
|
|
executor.submit(fetch_page, page)
|
|
for page in range(2, expected_pages + 1)
|
|
]
|
|
page_results.extend(
|
|
future.result() for future in concurrent.futures.as_completed(futures)
|
|
)
|
|
|
|
repo_names = []
|
|
failed_pages = []
|
|
for page_result, error in sorted(
|
|
page_results, key=lambda item: item[0]['page'],
|
|
):
|
|
page = page_result['page']
|
|
elapsed = page_result['elapsed']
|
|
if error:
|
|
logger.error(
|
|
f"Page {page}: Docker Hub fetch failed after bounded attempts "
|
|
f"({elapsed:.1f}s)"
|
|
)
|
|
failed_pages.append(page)
|
|
continue
|
|
repositories = page_result['repositories']
|
|
if not repositories:
|
|
logger.info(
|
|
f"Page {page}: no results after {elapsed:.1f}s "
|
|
f"(total matches: {page_result['total_count']})"
|
|
)
|
|
continue
|
|
|
|
repo_names.extend(repository['repo_name'] for repository in repositories)
|
|
logger.info(f"Page {page}: fetched {len(repositories)} images in {elapsed:.1f}s")
|
|
|
|
if failed_pages:
|
|
raise DockerHubDiscoveryTransportError(
|
|
f'Docker Hub search pagination incomplete: {len(failed_pages)} '
|
|
f'of {expected_pages} expected page(s) failed'
|
|
)
|
|
|
|
if not resolve_tags:
|
|
return repo_names
|
|
|
|
tag_fetch_workers = tag_fetch_workers if tag_fetch_workers is not None else min(4, fetch_workers)
|
|
logger.info(f"Resolving Docker tags for {len(repo_names)} repositories with {tag_fetch_workers} worker(s)...")
|
|
max_workers = max(1, min(tag_fetch_workers, len(repo_names) or 1))
|
|
with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
|
|
futures = {
|
|
executor.submit(
|
|
fetch_dockerhub_tags, repo_name, None,
|
|
docker_images_per_repository_limit(images_per_repository),
|
|
tag_retry_count, tag_retry_delay,
|
|
platform_filter_enabled, platform_os, platform_arch, platform_candidate_tags,
|
|
True,
|
|
): repo_name
|
|
for repo_name in repo_names
|
|
}
|
|
for future in concurrent.futures.as_completed(futures):
|
|
repo_name = futures[future]
|
|
tags, status = future.result()
|
|
if tags:
|
|
images.extend(tags)
|
|
if status != 'ok':
|
|
images.append(repo_name)
|
|
elif not docker_tag_resolution_is_conclusive(status):
|
|
images.append(repo_name)
|
|
logger.info(f"Deferring Docker tag resolution for {repo_name}: metadata lookup unavailable")
|
|
else:
|
|
logger.info(f"Skipping {repo_name}: no tags found")
|
|
|
|
logger.info(f"Resolved {len(images)} tagged Docker images from {len(repo_names)} repositories")
|
|
|
|
return images
|
|
|
|
def parse_dockerhub_datetime(value):
|
|
if not value:
|
|
return None
|
|
|
|
value = value.rstrip('Z')
|
|
for date_format in ("%Y-%m-%dT%H:%M:%S.%f", "%Y-%m-%dT%H:%M:%S"):
|
|
try:
|
|
return datetime.strptime(value, date_format)
|
|
except ValueError:
|
|
continue
|
|
return None
|
|
|
|
def fetch_dockerhub_last_updated(repo_name):
|
|
if '/' in repo_name:
|
|
namespace, name = repo_name.split('/', 1)
|
|
else:
|
|
namespace, name = 'library', repo_name
|
|
|
|
url = f"https://hub.docker.com/v2/repositories/{namespace}/{name}/"
|
|
try:
|
|
response = api_request('GET', url, headers={'User-Agent': 'GitSecretsScanner/2.0'}, timeout=8)
|
|
response.raise_for_status()
|
|
data = response.json()
|
|
return parse_dockerhub_datetime(data.get('last_updated') or data.get('last_modified'))
|
|
except Exception as e:
|
|
if isinstance(e, ApiRequestError):
|
|
raise
|
|
logger.warning(f"Unable to fetch Docker Hub metadata for {repo_name}: {str(e)}")
|
|
return None
|
|
|
|
|
|
DOCKERHUB_TAG_CACHE_SCHEMA = """
|
|
CREATE TABLE IF NOT EXISTS dockerhub_tag_cache (
|
|
cache_key TEXT PRIMARY KEY,
|
|
repo_key TEXT NOT NULL,
|
|
status TEXT NOT NULL,
|
|
tags_json TEXT,
|
|
checked_at REAL NOT NULL,
|
|
expires_at REAL NOT NULL,
|
|
since_at REAL,
|
|
message TEXT
|
|
);
|
|
CREATE TABLE IF NOT EXISTS dockerhub_tag_cache_meta (
|
|
key TEXT PRIMARY KEY,
|
|
value TEXT,
|
|
expires_at REAL
|
|
);
|
|
"""
|
|
_dockerhub_tag_cache_init_lock = threading.Lock()
|
|
_dockerhub_tag_cache_write_lock = threading.Lock()
|
|
_dockerhub_tag_cache_initialized = set()
|
|
|
|
|
|
def dockerhub_repo_key(repo_name):
|
|
repo_name = str(repo_name or '').strip().lower()
|
|
if ':' in repo_name:
|
|
repo_name = repo_name.split(':', 1)[0]
|
|
if '/' not in repo_name:
|
|
repo_name = 'library/' + repo_name
|
|
return repo_name.strip('/')
|
|
|
|
|
|
def dockerhub_tag_cache_path():
|
|
return getattr(scan_config, 'dockerhub_tag_cache_path', '') or ''
|
|
|
|
|
|
def dockerhub_tag_cache_limits():
|
|
return {
|
|
'rows': max(1, int(getattr(scan_config, 'dockerhub_tag_cache_max_rows', 50000))),
|
|
'age': max(60, int(getattr(scan_config, 'dockerhub_tag_cache_max_age_sec', 7 * 86400))),
|
|
'bytes': max(4096, int(getattr(scan_config, 'dockerhub_tag_cache_max_bytes', 256 * 1024 * 1024))),
|
|
'min_free': max(0, int(getattr(scan_config, 'dockerhub_tag_cache_min_free_bytes', 512 * 1024 * 1024))),
|
|
}
|
|
|
|
|
|
def dockerhub_tag_cache_disk_bytes(path):
|
|
return sum(
|
|
os.path.getsize(candidate) for candidate in (path, path + '-wal', path + '-shm')
|
|
if os.path.isfile(candidate)
|
|
)
|
|
|
|
|
|
def _dockerhub_since_epoch(since):
|
|
if since is None:
|
|
return None
|
|
try:
|
|
if isinstance(since, datetime):
|
|
value = since
|
|
else:
|
|
value = datetime.fromisoformat(str(since).replace('Z', '+00:00'))
|
|
if value.tzinfo is None:
|
|
value = value.replace(tzinfo=timezone.utc)
|
|
return float(value.timestamp())
|
|
except (TypeError, ValueError, OverflowError):
|
|
return None
|
|
|
|
|
|
def maintain_dockerhub_tag_cache(conn, path):
|
|
limits = dockerhub_tag_cache_limits()
|
|
now = time.time()
|
|
conn.execute('DELETE FROM dockerhub_tag_cache WHERE expires_at <= ? OR checked_at < ?', (now, now - limits['age']))
|
|
conn.execute('DELETE FROM dockerhub_tag_cache_meta WHERE expires_at IS NOT NULL AND expires_at <= ?', (now,))
|
|
conn.execute(
|
|
'''DELETE FROM dockerhub_tag_cache WHERE cache_key NOT IN (
|
|
SELECT cache_key FROM dockerhub_tag_cache ORDER BY checked_at DESC LIMIT ?
|
|
)''',
|
|
(limits['rows'],),
|
|
)
|
|
conn.commit()
|
|
if dockerhub_tag_cache_disk_bytes(path) > limits['bytes']:
|
|
# The cache is disposable. Clearing it under SQLite's full auto-vacuum
|
|
# is safer than allowing stale pages to consume an unbounded volume.
|
|
conn.execute('DELETE FROM dockerhub_tag_cache')
|
|
conn.execute('DELETE FROM dockerhub_tag_cache_meta')
|
|
conn.commit()
|
|
try:
|
|
conn.execute('PRAGMA incremental_vacuum')
|
|
except sqlite3.DatabaseError:
|
|
pass
|
|
disk_bytes = dockerhub_tag_cache_disk_bytes(path)
|
|
parent = os.path.dirname(path) or os.getcwd()
|
|
if int(shutil.disk_usage(parent).free) - disk_bytes >= limits['min_free']:
|
|
try:
|
|
conn.execute('PRAGMA wal_checkpoint(TRUNCATE)')
|
|
conn.execute('PRAGMA journal_mode=DELETE')
|
|
conn.execute('VACUUM')
|
|
conn.execute('PRAGMA journal_mode=WAL')
|
|
except sqlite3.DatabaseError:
|
|
pass
|
|
return dockerhub_tag_cache_disk_bytes(path) <= limits['bytes']
|
|
|
|
|
|
def connect_dockerhub_tag_cache():
|
|
path = dockerhub_tag_cache_path()
|
|
if not path:
|
|
return None
|
|
parent = os.path.dirname(path)
|
|
if parent:
|
|
os.makedirs(parent, exist_ok=True)
|
|
with _dockerhub_tag_cache_init_lock:
|
|
if path not in _dockerhub_tag_cache_initialized:
|
|
new_database = not os.path.exists(path)
|
|
conn = sqlite3.connect(path, timeout=10)
|
|
try:
|
|
conn.execute('PRAGMA busy_timeout=10000')
|
|
if new_database:
|
|
conn.execute('PRAGMA auto_vacuum=FULL')
|
|
conn.execute('PRAGMA journal_mode=WAL')
|
|
conn.executescript(DOCKERHUB_TAG_CACHE_SCHEMA)
|
|
columns = {row[1] for row in conn.execute('PRAGMA table_info(dockerhub_tag_cache)').fetchall()}
|
|
if 'since_at' not in columns:
|
|
conn.execute('ALTER TABLE dockerhub_tag_cache ADD COLUMN since_at REAL')
|
|
conn.commit()
|
|
if not maintain_dockerhub_tag_cache(conn, path):
|
|
return None
|
|
finally:
|
|
conn.close()
|
|
_dockerhub_tag_cache_initialized.add(path)
|
|
conn = sqlite3.connect(path, timeout=10)
|
|
conn.execute('PRAGMA busy_timeout=10000')
|
|
return conn
|
|
|
|
|
|
def dockerhub_cache_key(repo_name, since, limit, platform_variant=''):
|
|
return f"{dockerhub_repo_key(repo_name)}|{int(limit or 1)}|{platform_variant}"
|
|
|
|
|
|
def get_dockerhub_tag_cache(repo_name, since, limit, platform_variant='', return_status=False):
|
|
conn = connect_dockerhub_tag_cache()
|
|
if not conn:
|
|
return None
|
|
try:
|
|
row = conn.execute(
|
|
'SELECT status, tags_json, expires_at, since_at FROM dockerhub_tag_cache WHERE cache_key = ?',
|
|
(dockerhub_cache_key(repo_name, since, limit, platform_variant),),
|
|
).fetchone()
|
|
now = time.time()
|
|
if not row or float(row[2] or 0) <= now:
|
|
return None
|
|
status, tags_json, _, stored_since = row
|
|
requested_since = _dockerhub_since_epoch(since)
|
|
if requested_since is not None and stored_since is None:
|
|
return None
|
|
if requested_since is not None and float(stored_since or 0) > requested_since:
|
|
return None
|
|
if status == 'ok':
|
|
records = json.loads(tags_json or '[]')
|
|
targets = []
|
|
for record in records:
|
|
if isinstance(record, str):
|
|
return None
|
|
if not isinstance(record, dict) or not record.get('name'):
|
|
continue
|
|
updated_at = record.get('updated_at')
|
|
if requested_since is not None and updated_at is not None and float(updated_at) < requested_since:
|
|
continue
|
|
target = record.get('target')
|
|
if not target:
|
|
return None
|
|
try:
|
|
targets.append(parse_docker_target(target)['target'])
|
|
except (TypeError, ValueError):
|
|
return None
|
|
if len(targets) >= max(1, int(limit or 1)):
|
|
break
|
|
if not targets:
|
|
return None
|
|
return (targets, status) if return_status else targets
|
|
return ([], status) if return_status else []
|
|
except Exception:
|
|
return None
|
|
finally:
|
|
conn.close()
|
|
|
|
|
|
def put_dockerhub_tag_cache(
|
|
repo_name, since, limit, status, tags=None, ttl=None, message='', platform_variant='', tag_records=None,
|
|
):
|
|
with _dockerhub_tag_cache_write_lock:
|
|
conn = connect_dockerhub_tag_cache()
|
|
if not conn:
|
|
return
|
|
path = dockerhub_tag_cache_path()
|
|
try:
|
|
now = time.time()
|
|
ttl = int(ttl if ttl is not None else getattr(scan_config, 'dockerhub_tag_cache_ttl_sec', 21600))
|
|
records = []
|
|
offered_records = list(tag_records or [])
|
|
for index, tag in enumerate(tags or []):
|
|
text = str(tag)
|
|
record = offered_records[index] if index < len(offered_records) and isinstance(offered_records[index], dict) else {}
|
|
name = str(record.get('name') or '')
|
|
if not name:
|
|
name = text.rsplit(':', 1)[1] if ':' in text and '@' not in text and not text.startswith('{') else text
|
|
target = record.get('target') or text
|
|
try:
|
|
target = parse_docker_target(target)['target']
|
|
except (TypeError, ValueError):
|
|
continue
|
|
records.append({'name': name, 'target': target, 'updated_at': record.get('updated_at')})
|
|
if status == 'ok' and not records:
|
|
return False
|
|
payload = json.dumps(records, ensure_ascii=True, sort_keys=True, separators=(',', ':'))
|
|
limits = dockerhub_tag_cache_limits()
|
|
projected_bytes = len(payload.encode('utf-8')) + len(str(message or '').encode('utf-8')) + 1024
|
|
parent = os.path.dirname(path) or os.getcwd()
|
|
if (
|
|
not maintain_dockerhub_tag_cache(conn, path)
|
|
or dockerhub_tag_cache_disk_bytes(path) + projected_bytes > limits['bytes']
|
|
or int(shutil.disk_usage(parent).free) - projected_bytes < limits['min_free']
|
|
):
|
|
return False
|
|
conn.execute(
|
|
'''INSERT OR REPLACE INTO dockerhub_tag_cache(
|
|
cache_key, repo_key, status, tags_json, checked_at, expires_at, since_at, message
|
|
) VALUES (?, ?, ?, ?, ?, ?, ?, ?)''',
|
|
(
|
|
dockerhub_cache_key(repo_name, since, limit, platform_variant),
|
|
dockerhub_repo_key(repo_name),
|
|
status,
|
|
payload,
|
|
now,
|
|
now + max(1, ttl),
|
|
_dockerhub_since_epoch(since),
|
|
str(message or '')[:500],
|
|
),
|
|
)
|
|
conn.commit()
|
|
maintain_dockerhub_tag_cache(conn, path)
|
|
return True
|
|
except Exception:
|
|
return False
|
|
finally:
|
|
conn.close()
|
|
|
|
|
|
def dockerhub_tag_rate_limit_state():
|
|
conn = connect_dockerhub_tag_cache()
|
|
if not conn:
|
|
return {'active': False, 'retry_at': None}
|
|
try:
|
|
row = conn.execute("SELECT expires_at FROM dockerhub_tag_cache_meta WHERE key = 'rate_limited'").fetchone()
|
|
expires_at = float(row[0] or 0) if row else 0
|
|
active = expires_at > time.time()
|
|
return {
|
|
'active': active,
|
|
'retry_at': (
|
|
datetime.fromtimestamp(expires_at, timezone.utc).isoformat(timespec='seconds')
|
|
if active else None
|
|
),
|
|
}
|
|
except Exception:
|
|
return {'active': False, 'retry_at': None}
|
|
finally:
|
|
conn.close()
|
|
|
|
|
|
def dockerhub_tags_rate_limited():
|
|
return dockerhub_tag_rate_limit_state()['active']
|
|
|
|
|
|
def dockerhub_retry_after_seconds(response=None):
|
|
fallback = int(getattr(scan_config, 'dockerhub_tag_rate_limit_cache_ttl_sec', 1800) or 1800)
|
|
seconds = None
|
|
headers = (getattr(response, 'headers', None) or {}) if response is not None else {}
|
|
retry_after = headers.get('Retry-After')
|
|
if retry_after:
|
|
try:
|
|
seconds = int(retry_after)
|
|
except (TypeError, ValueError):
|
|
try:
|
|
retry_at = parsedate_to_datetime(str(retry_after))
|
|
if retry_at.tzinfo is None:
|
|
retry_at = retry_at.replace(tzinfo=timezone.utc)
|
|
seconds = math.ceil((retry_at - datetime.now(timezone.utc)).total_seconds())
|
|
except (IndexError, TypeError, ValueError, OverflowError):
|
|
seconds = None
|
|
seconds = fallback if seconds is None else seconds
|
|
return max(
|
|
DOCKERHUB_TAG_RATE_LIMIT_MIN_BACKOFF_SEC,
|
|
min(DOCKERHUB_TAG_RATE_LIMIT_MAX_BACKOFF_SEC, int(seconds)),
|
|
)
|
|
|
|
|
|
def put_dockerhub_tags_rate_limit(response=None, retry_seconds=None, return_retry_at=False):
|
|
ttl = dockerhub_retry_after_seconds(response) if retry_seconds is None else max(
|
|
DOCKERHUB_TAG_RATE_LIMIT_MIN_BACKOFF_SEC,
|
|
min(DOCKERHUB_TAG_RATE_LIMIT_MAX_BACKOFF_SEC, int(retry_seconds)),
|
|
)
|
|
with _dockerhub_tag_cache_write_lock:
|
|
conn = connect_dockerhub_tag_cache()
|
|
if not conn:
|
|
return
|
|
try:
|
|
path = dockerhub_tag_cache_path()
|
|
limits = dockerhub_tag_cache_limits()
|
|
parent = os.path.dirname(path) or os.getcwd()
|
|
if (
|
|
dockerhub_tag_cache_disk_bytes(path) + 8192 > limits['bytes']
|
|
or int(shutil.disk_usage(parent).free) - 8192 < limits['min_free']
|
|
):
|
|
return False
|
|
expires_at = math.ceil(time.time() + ttl)
|
|
conn.execute(
|
|
'''INSERT INTO dockerhub_tag_cache_meta(key, value, expires_at)
|
|
VALUES ('rate_limited', '1', ?)
|
|
ON CONFLICT(key) DO UPDATE SET value = '1',
|
|
expires_at = MAX(dockerhub_tag_cache_meta.expires_at, excluded.expires_at)''',
|
|
(expires_at,),
|
|
)
|
|
conn.commit()
|
|
stored = conn.execute(
|
|
"SELECT expires_at FROM dockerhub_tag_cache_meta WHERE key = 'rate_limited'"
|
|
).fetchone()
|
|
maintain_dockerhub_tag_cache(conn, path)
|
|
if return_retry_at:
|
|
return datetime.fromtimestamp(float(stored[0]), timezone.utc).isoformat(timespec='seconds')
|
|
return True
|
|
except Exception:
|
|
return False
|
|
finally:
|
|
conn.close()
|
|
|
|
|
|
def put_dockerhub_exhausted_rate_limit(endpoint, response=None):
|
|
if endpoint == 'hub_search':
|
|
if not docker_token_manager.has_accounts():
|
|
if (
|
|
docker_token_manager.uses_explicit_pool()
|
|
or int(getattr(response, 'status_code', 0) or 0) != 429
|
|
):
|
|
return None
|
|
retry_seconds = dockerhub_retry_after_seconds(response)
|
|
else:
|
|
if not docker_token_manager.rate_limit_contributes_to_exhaustion(endpoint):
|
|
return None
|
|
retry_seconds = docker_token_manager.seconds_until_available(endpoint)
|
|
return datetime.fromtimestamp(
|
|
time.time() + retry_seconds, timezone.utc,
|
|
).isoformat(timespec='seconds')
|
|
if not docker_token_manager.has_accounts():
|
|
if docker_token_manager.uses_explicit_pool():
|
|
return None
|
|
if int(getattr(response, 'status_code', 0) or 0) != 429:
|
|
return None
|
|
return put_dockerhub_tags_rate_limit(response, return_retry_at=True)
|
|
if not docker_token_manager.rate_limit_contributes_to_exhaustion(endpoint):
|
|
return None
|
|
return put_dockerhub_tags_rate_limit(
|
|
response,
|
|
retry_seconds=docker_token_manager.seconds_until_available(endpoint),
|
|
return_retry_at=True,
|
|
)
|
|
|
|
|
|
def docker_tag_platform_support(tag, platform_os='linux', platform_arch='amd64'):
|
|
images = tag.get('images') if isinstance(tag, dict) else None
|
|
if not isinstance(images, list) or not images:
|
|
return None
|
|
platforms = []
|
|
for image in images:
|
|
if not isinstance(image, dict):
|
|
return None
|
|
image_os = str(image.get('os') or '').strip().lower()
|
|
image_arch = str(image.get('architecture') or '').strip().lower()
|
|
if not image_os or not image_arch or image_os == 'unknown' or image_arch == 'unknown':
|
|
return None
|
|
platforms.append((image_os, image_arch))
|
|
wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower())
|
|
return wanted in platforms
|
|
|
|
|
|
def _bounded_docker_registry_json(
|
|
response, label, max_bytes=DOCKER_REGISTRY_MANIFEST_MAX_BYTES, *,
|
|
deadline=None, return_raw=False,
|
|
):
|
|
max_bytes = max(1, int(max_bytes))
|
|
content_length = (getattr(response, 'headers', None) or {}).get('Content-Length')
|
|
if content_length is not None:
|
|
try:
|
|
content_length = int(content_length)
|
|
except (TypeError, ValueError) as exc:
|
|
raise DockerRegistryResolutionError(f'{label} has an invalid Content-Length') from exc
|
|
if content_length < 0:
|
|
raise DockerRegistryResolutionError(f'{label} has an invalid Content-Length')
|
|
if content_length > max_bytes:
|
|
raise DockerRegistryResolutionError(f'{label} exceeds the response size limit')
|
|
|
|
iterator = None
|
|
iter_content = getattr(response, 'iter_content', None)
|
|
if callable(iter_content):
|
|
try:
|
|
iterator = iter(iter_content(chunk_size=min(64 * 1024, max_bytes + 1)))
|
|
except TypeError:
|
|
if isinstance(response, requests.Response):
|
|
raise DockerRegistryResolutionError(f'{label} cannot be streamed safely')
|
|
except (requests.RequestException, OSError) as exc:
|
|
raise DockerRemoteAccessError(
|
|
f'{label} stream is unavailable', status='remote_transient',
|
|
remote_attempted=True,
|
|
) from exc
|
|
|
|
raw_content = None
|
|
if iterator is not None:
|
|
content = bytearray()
|
|
try:
|
|
for chunk in iterator:
|
|
if deadline is not None and time.monotonic() >= float(deadline):
|
|
raise DockerRemoteAccessError(
|
|
f'{label} deadline expired', status='remote_transient',
|
|
remote_attempted=True,
|
|
)
|
|
if not chunk:
|
|
continue
|
|
if not isinstance(chunk, (bytes, bytearray)):
|
|
raise DockerRegistryResolutionError(f'{label} returned invalid bytes')
|
|
if len(content) + len(chunk) > max_bytes:
|
|
raise DockerRegistryResolutionError(f'{label} exceeds the response size limit')
|
|
content.extend(chunk)
|
|
except (DockerRegistryResolutionError, DockerRemoteAccessError):
|
|
raise
|
|
except (requests.RequestException, OSError) as exc:
|
|
raise DockerRemoteAccessError(
|
|
f'{label} stream failed', status='remote_transient',
|
|
remote_attempted=True,
|
|
) from exc
|
|
if deadline is not None and time.monotonic() >= float(deadline):
|
|
raise DockerRemoteAccessError(
|
|
f'{label} deadline expired', status='remote_transient',
|
|
remote_attempted=True,
|
|
)
|
|
raw_content = bytes(content)
|
|
if content_length is not None and len(raw_content) != content_length:
|
|
raise DockerRegistryResolutionError(f'{label} Content-Length is inconsistent')
|
|
try:
|
|
payload = json.loads(raw_content.decode('utf-8'))
|
|
except (RecursionError, UnicodeDecodeError, TypeError, ValueError) as exc:
|
|
raise DockerRegistryResolutionError(f'{label} is not valid JSON') from exc
|
|
else:
|
|
# Lightweight response doubles used by unit tests may not implement streaming.
|
|
content = getattr(response, 'content', None)
|
|
if isinstance(content, (bytes, bytearray)):
|
|
raw_content = bytes(content)
|
|
if len(raw_content) > max_bytes:
|
|
raise DockerRegistryResolutionError(f'{label} exceeds the response size limit')
|
|
try:
|
|
payload = response.json()
|
|
except (RecursionError, TypeError, ValueError) as exc:
|
|
raise DockerRegistryResolutionError(f'{label} is not valid JSON') from exc
|
|
if deadline is not None and time.monotonic() >= float(deadline):
|
|
raise DockerRemoteAccessError(
|
|
f'{label} deadline expired', status='remote_transient',
|
|
remote_attempted=True,
|
|
)
|
|
if not isinstance(payload, dict):
|
|
raise DockerRegistryResolutionError(f'{label} must be a JSON object')
|
|
if raw_content is None:
|
|
encoded = json.dumps(payload, ensure_ascii=True, separators=(',', ':')).encode('utf-8')
|
|
if len(encoded) > max_bytes:
|
|
raise DockerRegistryResolutionError(f'{label} exceeds the response size limit')
|
|
if return_raw:
|
|
if raw_content is None:
|
|
raise DockerRegistryResolutionError(f'{label} raw bytes are unavailable')
|
|
return payload, raw_content
|
|
return payload
|
|
|
|
|
|
def _docker_access_exhausted(endpoint, response=None, remote_attempted=False):
|
|
retry_at = put_dockerhub_exhausted_rate_limit(endpoint, response)
|
|
if retry_at:
|
|
raise DockerRemoteAccessError(
|
|
f'Docker {endpoint} accounts are rate-limited',
|
|
status='rate_limited', retry_at=retry_at,
|
|
remote_attempted=remote_attempted,
|
|
)
|
|
raise DockerRemoteAccessError(
|
|
f'Docker {endpoint} authentication is unavailable',
|
|
status='auth_failed', remote_attempted=remote_attempted,
|
|
)
|
|
|
|
|
|
def _docker_hub_access_token(
|
|
account, force_refresh=False, endpoint='hub_tags', stale_token='',
|
|
):
|
|
with docker_token_manager.hub_token_lock(account.name):
|
|
if not docker_token_manager.account_available(account.name, endpoint):
|
|
raise DockerRemoteAccessError(
|
|
f'Docker {endpoint} authentication is unavailable',
|
|
status='auth_failed', remote_attempted=False,
|
|
)
|
|
cached = docker_token_manager.cached_hub_token(account.name)
|
|
if force_refresh:
|
|
if stale_token and cached and cached != stale_token:
|
|
return cached
|
|
elif cached:
|
|
return cached
|
|
docker_token_manager.invalidate_hub_token(account.name)
|
|
response = api_request(
|
|
'POST', 'https://hub.docker.com/v2/auth/token',
|
|
json={'identifier': account.username, 'secret': account.token},
|
|
headers={'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json'},
|
|
timeout=(5, 15), max_retries=1,
|
|
retry_statuses={408, 500, 502, 503, 504},
|
|
allow_redirects=False,
|
|
)
|
|
if response.status_code in (401, 403, 429):
|
|
category = 'rate_limit' if response.status_code == 429 else (
|
|
'auth_invalid' if response.status_code == 401 else 'auth_forbidden'
|
|
)
|
|
docker_token_manager.report_http_status(
|
|
account, endpoint, response.status_code, response, category,
|
|
)
|
|
raise DockerRemoteAccessError(
|
|
f'Docker Hub token endpoint returned HTTP {response.status_code}',
|
|
status='rate_limited' if response.status_code == 429 else 'auth_failed',
|
|
remote_attempted=True,
|
|
)
|
|
if response.status_code >= 400:
|
|
raise DockerRemoteAccessError(
|
|
f'Docker Hub token endpoint returned HTTP {response.status_code}',
|
|
remote_attempted=True,
|
|
)
|
|
try:
|
|
payload = _bounded_docker_registry_json(
|
|
response, 'Docker Hub token response', max_bytes=1024 * 1024,
|
|
)
|
|
except DockerRegistryResolutionError as exc:
|
|
raise DockerRemoteAccessError(
|
|
'Docker Hub returned an invalid token response', remote_attempted=True,
|
|
) from exc
|
|
token = payload.get('access_token') or payload.get('token')
|
|
if not isinstance(token, str) or not token or len(token) > 16384:
|
|
raise DockerRemoteAccessError(
|
|
'Docker Hub returned an invalid access token', remote_attempted=True,
|
|
)
|
|
docker_token_manager.cache_hub_token(
|
|
account.name, token, payload.get('expires_in') or 600,
|
|
)
|
|
docker_token_manager.report_success(account, endpoint)
|
|
return token
|
|
|
|
|
|
def dockerhub_search_response(url, params, request_timeout=15):
|
|
endpoint = 'hub_search'
|
|
|
|
def request(token='', attempts=1):
|
|
headers = {
|
|
'User-Agent': 'GitSecretsScanner/2.0',
|
|
'Accept': 'application/json',
|
|
}
|
|
if token:
|
|
headers['Authorization'] = f'Bearer {token}'
|
|
return api_request(
|
|
'GET', url, params=params, headers=headers,
|
|
timeout=(5, request_timeout), max_retries=attempts, retry_delay=1,
|
|
retry_statuses={408, 500, 502, 503, 504},
|
|
allow_redirects=False,
|
|
)
|
|
|
|
if not docker_token_manager.has_accounts():
|
|
if docker_token_manager.uses_explicit_pool():
|
|
_docker_access_exhausted(endpoint, remote_attempted=False)
|
|
response = request(attempts=2)
|
|
if response.status_code == 429:
|
|
_docker_access_exhausted(endpoint, response, remote_attempted=True)
|
|
return response
|
|
|
|
excluded = set()
|
|
last_response = None
|
|
remote_attempted = False
|
|
account = None
|
|
token = ''
|
|
refreshed_accounts = set()
|
|
page_attempts = 0
|
|
while page_attempts < 2:
|
|
if account is None:
|
|
account = docker_token_manager.next_account(endpoint, excluded)
|
|
if account is None:
|
|
break
|
|
try:
|
|
token = _docker_hub_access_token(account, endpoint=endpoint)
|
|
except (ApiRequestError, DockerRemoteAccessError):
|
|
remote_attempted = True
|
|
excluded.add(account.name)
|
|
account = None
|
|
continue
|
|
|
|
try:
|
|
response = request(token)
|
|
except ApiRequestError:
|
|
remote_attempted = True
|
|
page_attempts += 1
|
|
if page_attempts < 2:
|
|
_wait_or_raise_scan_slot_fatal(1)
|
|
continue
|
|
raise
|
|
|
|
page_attempts += 1
|
|
remote_attempted = True
|
|
last_response = response
|
|
if (
|
|
response.status_code == 401
|
|
and account.name not in refreshed_accounts
|
|
and page_attempts < 2
|
|
):
|
|
refreshed_accounts.add(account.name)
|
|
try:
|
|
token = _docker_hub_access_token(
|
|
account, force_refresh=True, endpoint=endpoint,
|
|
stale_token=token,
|
|
)
|
|
continue
|
|
except (ApiRequestError, DockerRemoteAccessError):
|
|
excluded.add(account.name)
|
|
account = None
|
|
continue
|
|
if response.status_code not in (401, 403, 429):
|
|
docker_token_manager.report_success(account, endpoint)
|
|
return response
|
|
category = 'rate_limit' if response.status_code == 429 else 'auth_forbidden'
|
|
docker_token_manager.report_http_status(
|
|
account, endpoint, response.status_code, response, category,
|
|
)
|
|
excluded.add(account.name)
|
|
account = None
|
|
token = ''
|
|
_docker_access_exhausted(
|
|
endpoint, last_response, remote_attempted=remote_attempted,
|
|
)
|
|
|
|
|
|
def dockerhub_tags_response(url, params):
|
|
endpoint = 'hub_tags'
|
|
if not docker_token_manager.has_accounts():
|
|
if docker_token_manager.uses_explicit_pool():
|
|
_docker_access_exhausted(endpoint, remote_attempted=False)
|
|
response = api_request(
|
|
'GET', url, params=params,
|
|
headers={'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json'},
|
|
timeout=8, max_retries=1,
|
|
retry_statuses={408, 500, 502, 503, 504},
|
|
allow_redirects=False,
|
|
)
|
|
if response.status_code == 429:
|
|
_docker_access_exhausted(endpoint, response, remote_attempted=True)
|
|
return response
|
|
|
|
excluded = set()
|
|
last_response = None
|
|
remote_attempted = False
|
|
for _ in range(docker_token_manager.account_count()):
|
|
account = docker_token_manager.next_account(endpoint, excluded)
|
|
if account is None:
|
|
break
|
|
excluded.add(account.name)
|
|
try:
|
|
token = _docker_hub_access_token(account)
|
|
except DockerRemoteAccessError:
|
|
remote_attempted = True
|
|
continue
|
|
|
|
def request():
|
|
return api_request(
|
|
'GET', url, params=params,
|
|
headers={
|
|
'User-Agent': 'GitSecretsScanner/2.0',
|
|
'Accept': 'application/json',
|
|
'Authorization': f'Bearer {token}',
|
|
},
|
|
timeout=8, max_retries=1,
|
|
retry_statuses={408, 500, 502, 503, 504},
|
|
allow_redirects=False,
|
|
)
|
|
|
|
response = request()
|
|
remote_attempted = True
|
|
last_response = response
|
|
if response.status_code == 401:
|
|
try:
|
|
token = _docker_hub_access_token(
|
|
account, force_refresh=True, stale_token=token,
|
|
)
|
|
response = request()
|
|
last_response = response
|
|
except DockerRemoteAccessError:
|
|
continue
|
|
if response.status_code not in (401, 403, 429):
|
|
docker_token_manager.report_success(account, endpoint)
|
|
return response
|
|
if response.status_code != 403:
|
|
category = 'rate_limit' if response.status_code == 429 else 'auth_forbidden'
|
|
docker_token_manager.report_http_status(
|
|
account, endpoint, response.status_code, response, category,
|
|
)
|
|
_docker_access_exhausted(
|
|
endpoint, last_response, remote_attempted=remote_attempted,
|
|
)
|
|
|
|
|
|
def docker_registry_bearer_token(
|
|
challenge, repo_key, excluded_accounts=None, *, deadline=None,
|
|
anonymous_only=False,
|
|
):
|
|
text = str(challenge or '').strip()
|
|
if not text.lower().startswith('bearer '):
|
|
raise DockerRegistryResolutionError('Docker registry did not provide a bearer challenge')
|
|
values = {
|
|
key.lower(): value
|
|
for key, value in re.findall(r'([A-Za-z][A-Za-z0-9_-]*)="([^"\\]*)"', text[7:])
|
|
}
|
|
realm = values.get('realm', '')
|
|
parsed = urlsplit(realm)
|
|
if (
|
|
parsed.scheme.lower() != 'https'
|
|
or (parsed.hostname or '').lower() != 'auth.docker.io'
|
|
or parsed.username is not None
|
|
or parsed.password is not None
|
|
or parsed.port not in (None, 443)
|
|
):
|
|
raise DockerRegistryResolutionError('Docker registry bearer realm is not trusted')
|
|
excluded = set(excluded_accounts or ())
|
|
authenticated = not anonymous_only and docker_token_manager.has_accounts()
|
|
if (
|
|
not authenticated and not anonymous_only
|
|
and docker_token_manager.uses_explicit_pool()
|
|
):
|
|
_docker_access_exhausted('registry', remote_attempted=False)
|
|
attempts = docker_token_manager.account_count() if authenticated else 1
|
|
last_response = None
|
|
remote_attempted = False
|
|
saw_auth_failure = False
|
|
saw_rate_limit = False
|
|
saw_target_forbidden = False
|
|
saw_invalid_response = False
|
|
for _ in range(max(1, attempts)):
|
|
if deadline is not None and time.monotonic() >= float(deadline):
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry token deadline expired',
|
|
status='remote_transient', remote_attempted=remote_attempted,
|
|
)
|
|
account = docker_token_manager.next_account('registry', excluded) if authenticated else None
|
|
if authenticated and account is None:
|
|
break
|
|
if account is not None:
|
|
excluded.add(account.name)
|
|
try:
|
|
response = api_request(
|
|
'GET', realm,
|
|
params={
|
|
'service': 'registry.docker.io',
|
|
'scope': f'repository:{repo_key}:pull',
|
|
},
|
|
headers={
|
|
'User-Agent': 'GitSecretsScanner/2.0',
|
|
'Accept': 'application/json',
|
|
'Accept-Encoding': 'identity',
|
|
},
|
|
auth=(account.username, account.token) if account is not None else None,
|
|
timeout=(5, 15), max_retries=1,
|
|
retry_statuses={408, 500, 502, 503, 504},
|
|
allow_redirects=False, stream=True, deadline=deadline,
|
|
)
|
|
except ApiRequestError as exc:
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry token endpoint is temporarily unavailable',
|
|
status='remote_transient', remote_attempted=True,
|
|
) from exc
|
|
remote_attempted = True
|
|
last_response = response
|
|
try:
|
|
status_code = int(response.status_code)
|
|
if status_code in (401, 403, 429):
|
|
if status_code == 429:
|
|
saw_rate_limit = True
|
|
if account is not None:
|
|
docker_token_manager.report_http_status(
|
|
account, 'registry', status_code, response, 'rate_limit',
|
|
)
|
|
elif status_code == 401:
|
|
saw_auth_failure = True
|
|
if account is not None:
|
|
docker_token_manager.report_http_status(
|
|
account, 'registry', status_code, response, 'auth_invalid',
|
|
)
|
|
else:
|
|
saw_target_forbidden = True
|
|
continue
|
|
if status_code in (408, 425) or status_code >= 500:
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry token endpoint is temporarily unavailable',
|
|
status='remote_transient', remote_attempted=True,
|
|
)
|
|
if status_code >= 400:
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry denied access to the requested target',
|
|
status='target_forbidden', remote_attempted=True,
|
|
)
|
|
try:
|
|
payload = _bounded_docker_registry_json(
|
|
response, 'Docker registry token response',
|
|
max_bytes=DOCKER_REGISTRY_TOKEN_MAX_BYTES, deadline=deadline,
|
|
)
|
|
except DockerRegistryResolutionError:
|
|
saw_invalid_response = True
|
|
continue
|
|
finally:
|
|
response.close()
|
|
token = payload.get('token') or payload.get('access_token')
|
|
if not isinstance(token, str) or not token or len(token) > 16384:
|
|
saw_invalid_response = True
|
|
continue
|
|
if account is not None:
|
|
docker_token_manager.report_success(account, 'registry')
|
|
return DockerRegistryAuth(
|
|
token=token,
|
|
account_name=account.name if account is not None else '',
|
|
challenge=text,
|
|
)
|
|
if saw_rate_limit:
|
|
retry_at = put_dockerhub_exhausted_rate_limit('registry', last_response)
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry accounts are rate-limited', status='rate_limited',
|
|
retry_at=retry_at, remote_attempted=remote_attempted,
|
|
)
|
|
if saw_target_forbidden:
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry denied access to the requested target',
|
|
status='target_forbidden', remote_attempted=remote_attempted,
|
|
)
|
|
if saw_invalid_response and not saw_auth_failure:
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry returned an invalid token response',
|
|
status='remote_transient', remote_attempted=remote_attempted,
|
|
)
|
|
_docker_access_exhausted(
|
|
'registry', last_response, remote_attempted=remote_attempted,
|
|
)
|
|
|
|
|
|
def docker_registry_manifest(
|
|
repo_name, digest, bearer_auth=None, *, verify_content_digest=False,
|
|
return_raw=False, deadline=None, lease_renewal_callback=None,
|
|
anonymous_only=False,
|
|
):
|
|
repo_key = dockerhub_repo_key(repo_name)
|
|
digest = normalize_docker_digest(digest)
|
|
if not digest:
|
|
raise DockerRegistryResolutionError('Docker manifest digest is invalid')
|
|
url = f"https://registry-1.docker.io/v2/{quote(repo_key, safe='/')}/manifests/{digest}"
|
|
accept = ', '.join((
|
|
'application/vnd.oci.image.index.v1+json',
|
|
'application/vnd.docker.distribution.manifest.list.v2+json',
|
|
'application/vnd.oci.image.manifest.v1+json',
|
|
'application/vnd.docker.distribution.manifest.v2+json',
|
|
))
|
|
|
|
if isinstance(bearer_auth, str):
|
|
bearer_auth = DockerRegistryAuth(token=bearer_auth)
|
|
bearer_auth = bearer_auth or DockerRegistryAuth(token='')
|
|
|
|
def request(auth):
|
|
headers = {
|
|
'User-Agent': 'GitSecretsScanner/2.0',
|
|
'Accept': accept,
|
|
'Accept-Encoding': 'identity',
|
|
}
|
|
if auth.token:
|
|
headers['Authorization'] = f'Bearer {auth.token}'
|
|
return api_request(
|
|
'GET', url, headers=headers, timeout=(5, 15), max_retries=1,
|
|
retry_statuses={408, 500, 502, 503, 504},
|
|
allow_redirects=False, stream=True, deadline=deadline,
|
|
)
|
|
|
|
excluded = set()
|
|
attempts = 2 if anonymous_only else max(
|
|
2, docker_token_manager.account_count() + 1,
|
|
)
|
|
response = None
|
|
last_response = None
|
|
last_status = None
|
|
saw_bearer_unauthorized = False
|
|
for _ in range(attempts):
|
|
if lease_renewal_callback is not None:
|
|
if not callable(lease_renewal_callback):
|
|
raise ValueError('Docker resolver lease renewal callback is invalid')
|
|
if lease_renewal_callback() is False:
|
|
raise DockerResolverLeaseLostError('Docker resolver lease renewal was rejected')
|
|
if deadline is not None and time.monotonic() >= float(deadline):
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry manifest deadline expired',
|
|
status='remote_transient', remote_attempted=response is not None,
|
|
)
|
|
response = request(bearer_auth)
|
|
last_response = response
|
|
last_status = int(response.status_code)
|
|
if response.status_code not in (401, 403, 429):
|
|
break
|
|
challenge = (
|
|
(getattr(response, 'headers', None) or {}).get('WWW-Authenticate')
|
|
or bearer_auth.challenge
|
|
)
|
|
status_code = int(response.status_code)
|
|
if status_code == 401 and bearer_auth.token:
|
|
saw_bearer_unauthorized = True
|
|
if bearer_auth.account_name:
|
|
if status_code == 429:
|
|
docker_token_manager.report_http_status(
|
|
bearer_auth.account_name, 'registry', 429, response, 'rate_limit',
|
|
)
|
|
excluded.add(bearer_auth.account_name)
|
|
if status_code == 403:
|
|
response.close()
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry denied access to the requested manifest',
|
|
status='target_forbidden', remote_attempted=True,
|
|
)
|
|
if status_code == 429 and (
|
|
anonymous_only or not docker_token_manager.has_accounts()
|
|
):
|
|
try:
|
|
_docker_access_exhausted('registry', response, remote_attempted=True)
|
|
finally:
|
|
response.close()
|
|
response.close()
|
|
response = None
|
|
try:
|
|
if lease_renewal_callback is not None and lease_renewal_callback() is False:
|
|
raise DockerResolverLeaseLostError('Docker resolver lease renewal was rejected')
|
|
bearer_auth = docker_registry_bearer_token(
|
|
challenge, repo_key, excluded_accounts=excluded, deadline=deadline,
|
|
anonymous_only=anonymous_only,
|
|
)
|
|
except DockerRemoteAccessError as exc:
|
|
if saw_bearer_unauthorized and exc.status == 'auth_failed':
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry denied access to the requested manifest',
|
|
status='target_forbidden', remote_attempted=True,
|
|
) from exc
|
|
raise
|
|
except DockerRegistryResolutionError as exc:
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry authentication challenge is invalid',
|
|
status='auth_failed' if status_code == 401 else 'remote_transient',
|
|
remote_attempted=True,
|
|
) from exc
|
|
if response is None:
|
|
if last_status == 429:
|
|
retry_at = put_dockerhub_exhausted_rate_limit('registry', last_response)
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry accounts are rate-limited', status='rate_limited',
|
|
retry_at=retry_at, remote_attempted=True,
|
|
)
|
|
if last_status == 401:
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry denied access to the requested manifest'
|
|
if saw_bearer_unauthorized
|
|
else 'Docker registry authentication is unavailable',
|
|
status='target_forbidden' if saw_bearer_unauthorized else 'auth_failed',
|
|
remote_attempted=True,
|
|
)
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry manifest request did not run',
|
|
status='remote_transient', remote_attempted=last_response is not None,
|
|
)
|
|
if response.status_code in (401, 429):
|
|
try:
|
|
_docker_access_exhausted('registry', response, remote_attempted=True)
|
|
finally:
|
|
response.close()
|
|
status_code = int(response.status_code)
|
|
if 300 <= status_code < 400:
|
|
response.close()
|
|
raise DockerRegistryResolutionError('Docker registry manifest redirect was rejected')
|
|
try:
|
|
if status_code == 404:
|
|
raise DockerRegistryResolutionError('Docker registry manifest was not found')
|
|
if status_code in (408, 425) or status_code >= 500:
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry manifest endpoint is temporarily unavailable',
|
|
status='remote_transient', remote_attempted=True,
|
|
)
|
|
if status_code >= 400:
|
|
raise DockerRegistryResolutionError('Docker registry rejected the manifest request')
|
|
returned_digest = normalize_docker_digest(
|
|
(getattr(response, 'headers', None) or {}).get('Docker-Content-Digest')
|
|
)
|
|
if returned_digest and returned_digest != digest:
|
|
raise DockerRegistryResolutionError('Docker registry returned a different manifest digest')
|
|
if verify_content_digest or return_raw:
|
|
payload, raw_content = _bounded_docker_registry_json(
|
|
response, 'Docker manifest', deadline=deadline, return_raw=True,
|
|
)
|
|
else:
|
|
payload = _bounded_docker_registry_json(
|
|
response, 'Docker manifest', deadline=deadline,
|
|
)
|
|
raw_content = None
|
|
if verify_content_digest:
|
|
calculated = 'sha256:' + hashlib.sha256(raw_content).hexdigest()
|
|
if calculated != digest:
|
|
raise DockerRegistryResolutionError('Docker manifest payload digest is invalid')
|
|
if bearer_auth.account_name:
|
|
docker_token_manager.report_success(bearer_auth.account_name, 'registry')
|
|
if return_raw:
|
|
return payload, bearer_auth, raw_content
|
|
return payload, bearer_auth
|
|
finally:
|
|
response.close()
|
|
|
|
|
|
def resolve_docker_layer_graph(
|
|
repo_name, digest, platform_os='linux', platform_arch='amd64',
|
|
bearer_auth=None, *, deadline=None, include_descriptors=False,
|
|
lease_renewal_callback=None,
|
|
):
|
|
manifest_digest = normalize_docker_digest(digest)
|
|
if include_descriptors:
|
|
payload, bearer_auth, raw_content = docker_registry_manifest(
|
|
repo_name, manifest_digest, bearer_auth, deadline=deadline,
|
|
verify_content_digest=True, return_raw=True,
|
|
lease_renewal_callback=lease_renewal_callback,
|
|
)
|
|
else:
|
|
payload, bearer_auth = docker_registry_manifest(
|
|
repo_name, manifest_digest, bearer_auth, deadline=deadline,
|
|
lease_renewal_callback=lease_renewal_callback,
|
|
)
|
|
raw_content = None
|
|
descriptors = payload.get('manifests')
|
|
if descriptors is not None:
|
|
if not isinstance(descriptors, list) or len(descriptors) > DOCKER_REGISTRY_MAX_DESCRIPTORS:
|
|
raise DockerRegistryResolutionError('Docker manifest index has invalid descriptor bounds')
|
|
wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower())
|
|
descriptor = next((
|
|
item for item in descriptors
|
|
if isinstance(item, dict)
|
|
and isinstance(item.get('platform'), dict)
|
|
and (
|
|
str(item['platform'].get('os') or '').lower(),
|
|
str(item['platform'].get('architecture') or '').lower(),
|
|
) == wanted
|
|
), None)
|
|
if descriptor is None:
|
|
return None, bearer_auth
|
|
manifest_digest = normalize_docker_digest(descriptor.get('digest'))
|
|
if not manifest_digest:
|
|
raise DockerRegistryResolutionError('Docker platform descriptor has an invalid digest')
|
|
if include_descriptors:
|
|
payload, bearer_auth, raw_content = docker_registry_manifest(
|
|
repo_name, manifest_digest, bearer_auth, deadline=deadline,
|
|
verify_content_digest=True, return_raw=True,
|
|
lease_renewal_callback=lease_renewal_callback,
|
|
)
|
|
else:
|
|
payload, bearer_auth = docker_registry_manifest(
|
|
repo_name, manifest_digest, bearer_auth, deadline=deadline,
|
|
lease_renewal_callback=lease_renewal_callback,
|
|
)
|
|
|
|
layers = payload.get('layers')
|
|
if not isinstance(layers, list) or len(layers) > DOCKER_REGISTRY_MAX_LAYERS:
|
|
raise DockerRegistryResolutionError('Docker manifest has invalid layer bounds')
|
|
ordered_layers = []
|
|
layer_descriptors = []
|
|
for position, layer in enumerate(layers, 1):
|
|
layer_digest = normalize_docker_digest(layer.get('digest')) if isinstance(layer, dict) else ''
|
|
if not layer_digest:
|
|
raise DockerRegistryResolutionError('Docker manifest contains an invalid layer digest')
|
|
ordered_layers.append(layer_digest)
|
|
if include_descriptors:
|
|
layer_descriptors.append(_docker_content_descriptor(layer, 'layer', position))
|
|
graph = {
|
|
'manifest_digest': manifest_digest,
|
|
'layers': tuple(ordered_layers),
|
|
}
|
|
if include_descriptors:
|
|
config = _docker_content_descriptor(payload.get('config'), 'config', 0)
|
|
manifest_media_type = str(
|
|
payload.get('mediaType')
|
|
or 'application/vnd.docker.distribution.manifest.v2+json'
|
|
).strip().lower()
|
|
if not manifest_media_type or len(manifest_media_type) > 256:
|
|
raise DockerRegistryResolutionError('Docker manifest media type is invalid')
|
|
graph.update({
|
|
'manifest_media_type': manifest_media_type,
|
|
'manifest_size_bytes': len(raw_content),
|
|
'config_digest': config['digest'],
|
|
'layer_descriptors': tuple(layer_descriptors),
|
|
})
|
|
return graph, bearer_auth
|
|
|
|
|
|
def _dockerhub_manifest_target_parts(target):
|
|
parsed = parse_docker_target(target)
|
|
image = str(parsed['image']).lower()
|
|
image_name, manifest_digest = image.rsplit('@', 1)
|
|
manifest_digest = normalize_docker_digest(manifest_digest)
|
|
if not manifest_digest:
|
|
raise DockerRegistryResolutionError('Docker manifest target digest is invalid')
|
|
parts = image_name.split('/')
|
|
if len(parts) > 1 and ('.' in parts[0] or ':' in parts[0] or parts[0] == 'localhost'):
|
|
registry = parts.pop(0)
|
|
if registry not in ('docker.io', 'index.docker.io', 'registry-1.docker.io'):
|
|
raise DockerRegistryResolutionError(
|
|
'Docker layer scanning only supports Docker Hub targets'
|
|
)
|
|
if not parts or any(not part for part in parts):
|
|
raise DockerRegistryResolutionError('Docker Hub repository is invalid')
|
|
repository = '/'.join(parts)
|
|
registry_repository = repository if '/' in repository else f'library/{repository}'
|
|
return image, repository, registry_repository, manifest_digest
|
|
|
|
|
|
def _docker_content_descriptor(value, kind, position):
|
|
if not isinstance(value, dict):
|
|
raise DockerRegistryResolutionError(f'Docker {kind} descriptor is invalid')
|
|
digest = normalize_docker_digest(value.get('digest'))
|
|
size = value.get('size')
|
|
media_type = str(value.get('mediaType') or '').strip().lower()
|
|
if (
|
|
not digest
|
|
or isinstance(size, bool)
|
|
or not isinstance(size, int)
|
|
or size < 0
|
|
or size > 1024 * 1024 * 1024 * 1024
|
|
or not media_type
|
|
or len(media_type) > 256
|
|
):
|
|
raise DockerRegistryResolutionError(f'Docker {kind} descriptor has invalid bounds')
|
|
return {
|
|
'digest': digest,
|
|
'size': size,
|
|
'media_type': media_type,
|
|
}
|
|
|
|
|
|
def resolve_docker_content_manifest(
|
|
target, platform_os='linux', platform_arch='amd64', bearer_auth=None,
|
|
*, deadline=None, anonymous_only=False,
|
|
):
|
|
image, repository, registry_repository, manifest_digest = (
|
|
_dockerhub_manifest_target_parts(target)
|
|
)
|
|
target_manifest_digest = manifest_digest
|
|
payload, bearer_auth = docker_registry_manifest(
|
|
registry_repository, manifest_digest, bearer_auth,
|
|
verify_content_digest=True, deadline=deadline,
|
|
anonymous_only=anonymous_only,
|
|
)
|
|
descriptors = payload.get('manifests')
|
|
if descriptors is not None:
|
|
if not isinstance(descriptors, list) or len(descriptors) > DOCKER_REGISTRY_MAX_DESCRIPTORS:
|
|
raise DockerRegistryResolutionError('Docker manifest index has invalid descriptor bounds')
|
|
wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower())
|
|
child = next((
|
|
item for item in descriptors
|
|
if isinstance(item, dict)
|
|
and isinstance(item.get('platform'), dict)
|
|
and (
|
|
str(item['platform'].get('os') or '').lower(),
|
|
str(item['platform'].get('architecture') or '').lower(),
|
|
) == wanted
|
|
), None)
|
|
if child is None:
|
|
raise DockerRegistryResolutionError('Docker target platform manifest is unavailable')
|
|
manifest_digest = normalize_docker_digest(child.get('digest'))
|
|
if not manifest_digest:
|
|
raise DockerRegistryResolutionError('Docker platform descriptor digest is invalid')
|
|
payload, bearer_auth = docker_registry_manifest(
|
|
registry_repository, manifest_digest, bearer_auth,
|
|
verify_content_digest=True, deadline=deadline,
|
|
anonymous_only=anonymous_only,
|
|
)
|
|
|
|
config = _docker_content_descriptor(payload.get('config'), 'config', 0)
|
|
layers = payload.get('layers')
|
|
if not isinstance(layers, list) or len(layers) > DOCKER_REGISTRY_MAX_LAYERS:
|
|
raise DockerRegistryResolutionError('Docker manifest has invalid layer bounds')
|
|
normalized_layers = [
|
|
_docker_content_descriptor(layer, 'layer', position)
|
|
for position, layer in enumerate(layers, 1)
|
|
]
|
|
manifest_media_type = str(
|
|
payload.get('mediaType')
|
|
or 'application/vnd.docker.distribution.manifest.v2+json'
|
|
).strip().lower()
|
|
if not manifest_media_type or len(manifest_media_type) > 256:
|
|
raise DockerRegistryResolutionError('Docker manifest media type is invalid')
|
|
if deadline is not None and time.monotonic() >= float(deadline):
|
|
raise DockerRemoteAccessError(
|
|
'Docker registry manifest deadline expired',
|
|
status='remote_transient', remote_attempted=True,
|
|
)
|
|
return {
|
|
'version': 1,
|
|
'image': image,
|
|
'repository': registry_repository,
|
|
'manifest_digest': target_manifest_digest,
|
|
'platform_os': str(platform_os or 'linux').lower(),
|
|
'platform_arch': str(platform_arch or 'amd64').lower(),
|
|
'manifest_media_type': manifest_media_type,
|
|
'config': config,
|
|
'layers': normalized_layers,
|
|
}, bearer_auth
|
|
|
|
|
|
DOCKER_BLOB_REDIRECT_SUFFIXES = (
|
|
'.docker.com',
|
|
'.docker.io',
|
|
'.cloudfront.net',
|
|
'.cloudflarestorage.com',
|
|
'.amazonaws.com',
|
|
)
|
|
|
|
|
|
def _docker_blob_url_validation_error(url, *, registry_origin=False):
|
|
try:
|
|
parsed = urlsplit(str(url or ''))
|
|
hostname = (parsed.hostname or '').lower().rstrip('.')
|
|
port = parsed.port
|
|
except ValueError:
|
|
return 'invalid_url'
|
|
if (
|
|
parsed.scheme.lower() != 'https'
|
|
or not hostname
|
|
or parsed.username is not None
|
|
or parsed.password is not None
|
|
or port not in (None, 443)
|
|
or parsed.fragment
|
|
):
|
|
return 'invalid_url'
|
|
if registry_origin:
|
|
return '' if hostname == 'registry-1.docker.io' and not parsed.query else 'invalid_registry'
|
|
try:
|
|
address = ipaddress.ip_address(hostname)
|
|
except ValueError:
|
|
address = None
|
|
if address is not None and not address.is_global:
|
|
return 'non_global_address'
|
|
if not any(hostname.endswith(suffix) for suffix in DOCKER_BLOB_REDIRECT_SUFFIXES):
|
|
return 'untrusted_host'
|
|
try:
|
|
answers = socket.getaddrinfo(
|
|
hostname, 443, type=socket.SOCK_STREAM, proto=socket.IPPROTO_TCP,
|
|
)
|
|
except OSError:
|
|
return 'dns_unavailable'
|
|
if not answers:
|
|
return 'dns_unavailable'
|
|
for answer in answers:
|
|
try:
|
|
resolved = ipaddress.ip_address(str(answer[4][0]).split('%', 1)[0])
|
|
except (IndexError, TypeError, ValueError):
|
|
return 'invalid_dns_answer'
|
|
if not resolved.is_global:
|
|
return 'non_global_address'
|
|
return ''
|
|
|
|
|
|
def _docker_blob_url_allowed(url, *, registry_origin=False):
|
|
return not _docker_blob_url_validation_error(url, registry_origin=registry_origin)
|
|
|
|
|
|
def _require_docker_blob_url(url, *, registry_origin=False):
|
|
error = _docker_blob_url_validation_error(url, registry_origin=registry_origin)
|
|
if error == 'dns_unavailable':
|
|
raise DockerLayerInfrastructureError(
|
|
'remote_dns', 'Docker blob redirect DNS is temporarily unavailable',
|
|
category='remote_transient',
|
|
)
|
|
if error:
|
|
raise DockerContentTransferError(
|
|
'unsafe_redirect' if not registry_origin else 'unsafe_registry_url',
|
|
'Docker blob URL is not trusted', False,
|
|
)
|
|
|
|
|
|
def _docker_registry_blob_response(
|
|
repository, digest, bearer_auth, deadline, *, anonymous_only=False,
|
|
):
|
|
repo_key = dockerhub_repo_key(repository)
|
|
digest = normalize_docker_digest(digest)
|
|
if not digest:
|
|
raise DockerContentTransferError('invalid_descriptor', 'Docker blob digest is invalid', False)
|
|
url = f'https://registry-1.docker.io/v2/{quote(repo_key, safe="/")}/blobs/{digest}'
|
|
_require_docker_blob_url(url, registry_origin=True)
|
|
if isinstance(bearer_auth, str):
|
|
bearer_auth = DockerRegistryAuth(token=bearer_auth)
|
|
bearer_auth = bearer_auth or DockerRegistryAuth(token='')
|
|
excluded = set()
|
|
attempts = 2 if anonymous_only else max(
|
|
2, docker_token_manager.account_count() + 1,
|
|
)
|
|
response = None
|
|
saw_rate_limit = False
|
|
saw_bearer_unauthorized = False
|
|
for _ in range(attempts):
|
|
remaining = max(0.0, float(deadline) - time.monotonic())
|
|
if remaining <= 0:
|
|
raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired')
|
|
headers = {
|
|
'User-Agent': 'GitSecretsScanner/2.0',
|
|
'Accept': 'application/octet-stream',
|
|
'Accept-Encoding': 'identity',
|
|
}
|
|
if bearer_auth.token:
|
|
headers['Authorization'] = f'Bearer {bearer_auth.token}'
|
|
try:
|
|
response = api_request(
|
|
'GET', url, headers=headers, timeout=(5, min(30, remaining)),
|
|
use_proxy=False,
|
|
max_retries=1, retry_statuses={408, 500, 502, 503, 504},
|
|
allow_redirects=False, stream=True, deadline=deadline,
|
|
)
|
|
except ApiRequestError as exc:
|
|
if time.monotonic() >= float(deadline):
|
|
raise DockerContentTransferError(
|
|
'transfer_timeout', 'Docker blob deadline expired',
|
|
) from exc
|
|
raise DockerLayerInfrastructureError(
|
|
'remote_transient', 'Docker blob endpoint is temporarily unavailable',
|
|
category='remote_transient',
|
|
) from exc
|
|
if response.status_code not in (401, 429):
|
|
return response, bearer_auth
|
|
challenge = (
|
|
(getattr(response, 'headers', None) or {}).get('WWW-Authenticate')
|
|
or bearer_auth.challenge
|
|
)
|
|
if response.status_code == 401 and bearer_auth.token:
|
|
saw_bearer_unauthorized = True
|
|
if bearer_auth.account_name:
|
|
if response.status_code == 429:
|
|
saw_rate_limit = True
|
|
docker_token_manager.report_http_status(
|
|
bearer_auth.account_name, 'registry', 429, response, 'rate_limit',
|
|
)
|
|
excluded.add(bearer_auth.account_name)
|
|
elif response.status_code == 429:
|
|
saw_rate_limit = True
|
|
response.close()
|
|
response = None
|
|
try:
|
|
bearer_auth = docker_registry_bearer_token(
|
|
challenge, repo_key, excluded_accounts=excluded, deadline=deadline,
|
|
anonymous_only=anonymous_only,
|
|
)
|
|
except DockerRemoteAccessError as exc:
|
|
if exc.status == 'target_forbidden':
|
|
raise DockerContentTransferError(
|
|
'target_forbidden', 'Docker blob target is forbidden', False,
|
|
) from exc
|
|
if exc.status == 'rate_limited':
|
|
raise DockerLayerInfrastructureError(
|
|
'remote_rate_limit', 'Docker blob authorization is rate-limited',
|
|
category='docker_rate_limit',
|
|
) from exc
|
|
if exc.status == 'auth_failed':
|
|
if saw_bearer_unauthorized:
|
|
raise DockerContentTransferError(
|
|
'target_forbidden', 'Docker blob target is forbidden', False,
|
|
) from exc
|
|
raise DockerLayerInfrastructureError(
|
|
'remote_auth', 'Docker blob authorization is unavailable',
|
|
category='docker_auth', auth_related=True,
|
|
) from exc
|
|
raise DockerLayerInfrastructureError(
|
|
'remote_transient', 'Docker blob authorization is temporarily unavailable',
|
|
category='remote_transient',
|
|
) from exc
|
|
except DockerRegistryResolutionError as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'remote_auth', 'Docker blob authentication challenge is invalid',
|
|
category='docker_auth', auth_related=True,
|
|
) from exc
|
|
if response is not None:
|
|
response.close()
|
|
if saw_rate_limit:
|
|
raise DockerLayerInfrastructureError(
|
|
'remote_rate_limit', 'Docker blob authorization is rate-limited',
|
|
category='docker_rate_limit',
|
|
)
|
|
if saw_bearer_unauthorized:
|
|
raise DockerContentTransferError(
|
|
'target_forbidden', 'Docker blob target is forbidden', False,
|
|
)
|
|
raise DockerLayerInfrastructureError(
|
|
'remote_auth', 'Docker blob authorization is unavailable',
|
|
category='docker_auth', auth_related=True,
|
|
)
|
|
|
|
|
|
def stream_docker_registry_blob(
|
|
repository, descriptor, destination, bearer_auth=None, *, deadline,
|
|
min_free_bytes=0, redirect_limit=5, anonymous_only=False,
|
|
):
|
|
digest = normalize_docker_digest((descriptor or {}).get('digest'))
|
|
declared_bytes = (descriptor or {}).get('size')
|
|
kind = str((descriptor or {}).get('kind') or '')
|
|
media_type = str((descriptor or {}).get('media_type') or '').strip().lower()
|
|
if (
|
|
not digest
|
|
or isinstance(declared_bytes, bool)
|
|
or not isinstance(declared_bytes, int)
|
|
or declared_bytes < 0
|
|
or declared_bytes > 1024 * 1024 * 1024 * 1024
|
|
or kind not in ('config', 'layer')
|
|
or not media_type
|
|
):
|
|
raise DockerContentTransferError('invalid_descriptor', 'Docker blob descriptor is invalid', False)
|
|
supported_media = DOCKER_CONFIG_MEDIA_TYPES if kind == 'config' else DOCKER_LAYER_MEDIA_TYPES
|
|
if media_type not in supported_media:
|
|
raise DockerContentTransferError(
|
|
'unsupported_media_type', 'Docker blob media type is unsupported', False,
|
|
)
|
|
deadline = float(deadline)
|
|
if deadline <= time.monotonic():
|
|
raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired')
|
|
destination = os.path.abspath(destination)
|
|
try:
|
|
parent = require_private_directory(os.path.dirname(destination), create=False)
|
|
reject_reparse_components(parent)
|
|
except (OSError, ValueError) as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'private_storage', 'Docker blob private storage is unavailable',
|
|
category='source_resource',
|
|
) from exc
|
|
if os.path.lexists(destination):
|
|
raise DockerLayerInfrastructureError(
|
|
'destination_exists', 'Docker blob destination is not available',
|
|
category='source_resource',
|
|
)
|
|
required_free = max(0, int(min_free_bytes)) + declared_bytes
|
|
try:
|
|
free_bytes = shutil.disk_usage(parent).free
|
|
except OSError as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'disk_reserve', 'Docker blob free space cannot be verified',
|
|
category='source_resource',
|
|
) from exc
|
|
if free_bytes < required_free:
|
|
raise DockerLayerInfrastructureError(
|
|
'disk_reserve', 'Docker blob would violate the free-space reserve',
|
|
category='source_resource',
|
|
)
|
|
|
|
started = time.monotonic()
|
|
response = None
|
|
total = 0
|
|
temporary = (
|
|
f'{destination}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.partial'
|
|
)
|
|
published = False
|
|
try:
|
|
response, bearer_auth = _docker_registry_blob_response(
|
|
repository, digest, bearer_auth, deadline,
|
|
anonymous_only=anonymous_only,
|
|
)
|
|
redirects = 0
|
|
while response.status_code in (301, 302, 303, 307, 308):
|
|
location = (getattr(response, 'headers', None) or {}).get('Location')
|
|
current_url = str(getattr(response, 'url', '') or '')
|
|
response.close()
|
|
response = None
|
|
redirects += 1
|
|
if not location or redirects > max(0, min(5, int(redirect_limit))):
|
|
raise DockerContentTransferError('unsafe_redirect', 'Docker blob redirect limit exceeded', False)
|
|
next_url = urljoin(current_url, location)
|
|
_require_docker_blob_url(next_url)
|
|
remaining = max(0.0, deadline - time.monotonic())
|
|
if remaining <= 0:
|
|
raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired')
|
|
response = api_request(
|
|
'GET', next_url,
|
|
use_proxy=False,
|
|
headers={
|
|
'User-Agent': 'GitSecretsScanner/2.0',
|
|
'Accept': 'application/octet-stream',
|
|
'Accept-Encoding': 'identity',
|
|
},
|
|
timeout=(5, min(30, remaining)), max_retries=1,
|
|
retry_statuses={408, 500, 502, 503, 504},
|
|
allow_redirects=False, stream=True, deadline=deadline,
|
|
)
|
|
status_code = int(response.status_code)
|
|
if status_code == 401:
|
|
raise DockerContentTransferError(
|
|
'target_forbidden', 'Docker blob target is forbidden', False,
|
|
)
|
|
if status_code == 403:
|
|
raise DockerContentTransferError(
|
|
'target_forbidden', 'Docker blob target is forbidden', False,
|
|
)
|
|
if status_code == 404:
|
|
raise DockerContentTransferError(
|
|
'blob_not_found', 'Docker blob target is unavailable', False,
|
|
)
|
|
if status_code == 429:
|
|
raise DockerLayerInfrastructureError(
|
|
'remote_rate_limit', 'Docker blob endpoint is rate-limited',
|
|
category='docker_rate_limit',
|
|
)
|
|
if status_code in (408, 425) or status_code >= 500:
|
|
raise DockerLayerInfrastructureError(
|
|
'remote_transient', 'Docker blob endpoint is temporarily unavailable',
|
|
category='remote_transient',
|
|
)
|
|
if status_code >= 400:
|
|
raise DockerContentTransferError(
|
|
'target_rejected', 'Docker blob target was rejected', False,
|
|
)
|
|
content_encoding = str(
|
|
(getattr(response, 'headers', None) or {}).get('Content-Encoding') or ''
|
|
).strip().lower()
|
|
if content_encoding not in ('', 'identity'):
|
|
raise DockerContentTransferError(
|
|
'content_encoding', 'Docker blob response changed the content encoding',
|
|
)
|
|
content_length = (getattr(response, 'headers', None) or {}).get('Content-Length')
|
|
try:
|
|
content_length = int(content_length)
|
|
except (TypeError, ValueError) as exc:
|
|
raise DockerContentTransferError(
|
|
'size_mismatch', 'Docker blob response lacks an exact Content-Length', False,
|
|
) from exc
|
|
if content_length != declared_bytes:
|
|
raise DockerContentTransferError(
|
|
'size_mismatch', 'Docker blob Content-Length differs from its descriptor', False,
|
|
)
|
|
|
|
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0)
|
|
descriptor_fd = os.open(temporary, flags, 0o600)
|
|
os.close(descriptor_fd)
|
|
harden_private_file(temporary)
|
|
digest_hash = hashlib.sha256()
|
|
with open(temporary, 'wb', buffering=0) as output:
|
|
for chunk in response.iter_content(chunk_size=1024 * 1024):
|
|
_raise_if_scan_slot_fatal()
|
|
if time.monotonic() >= deadline:
|
|
raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired')
|
|
if not chunk:
|
|
continue
|
|
total += len(chunk)
|
|
if total > declared_bytes:
|
|
raise DockerContentTransferError(
|
|
'size_mismatch', 'Docker blob exceeded its declared size', False,
|
|
)
|
|
if shutil.disk_usage(parent).free < max(0, int(min_free_bytes)):
|
|
raise DockerLayerInfrastructureError(
|
|
'disk_reserve', 'Docker blob transfer reached the free-space reserve',
|
|
category='source_resource',
|
|
)
|
|
digest_hash.update(chunk)
|
|
output.write(chunk)
|
|
output.flush()
|
|
os.fsync(output.fileno())
|
|
if time.monotonic() >= deadline:
|
|
raise DockerContentTransferError(
|
|
'transfer_timeout', 'Docker blob deadline expired',
|
|
)
|
|
if total != declared_bytes:
|
|
raise DockerContentTransferError(
|
|
'size_mismatch', 'Docker blob byte count differs from its descriptor', False,
|
|
)
|
|
if f'sha256:{digest_hash.hexdigest()}' != digest:
|
|
raise DockerContentTransferError(
|
|
'digest_mismatch', 'Docker blob SHA-256 differs from its descriptor', False,
|
|
)
|
|
harden_private_file(temporary)
|
|
durable_replace(temporary, destination)
|
|
if not private_file_ready(destination):
|
|
raise DockerLayerInfrastructureError(
|
|
'private_file_lost', 'Docker blob lost its private file identity',
|
|
category='source_resource',
|
|
)
|
|
if time.monotonic() >= deadline:
|
|
raise DockerContentTransferError(
|
|
'transfer_timeout', 'Docker blob deadline expired',
|
|
)
|
|
published = True
|
|
duration_ms = max(0, int((time.monotonic() - started) * 1000))
|
|
return DockerBlobDownloadOutcome(
|
|
path=destination,
|
|
verified_bytes=total,
|
|
transfer_bytes=total,
|
|
duration_ms=duration_ms,
|
|
bearer_auth=bearer_auth,
|
|
)
|
|
except DockerContentTransferError as exc:
|
|
exc.transfer_bytes = min(declared_bytes, max(0, int(total)))
|
|
exc.duration_ms = max(0, int((time.monotonic() - started) * 1000))
|
|
raise
|
|
except (ApiRequestError, requests.RequestException) as exc:
|
|
if time.monotonic() >= deadline:
|
|
raise DockerContentTransferError(
|
|
'transfer_timeout', 'Docker blob deadline expired',
|
|
) from exc
|
|
raise DockerLayerInfrastructureError(
|
|
'remote_transient', 'Docker blob transfer is temporarily unavailable',
|
|
category='remote_transient',
|
|
) from exc
|
|
except OSError as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'private_storage', 'Docker blob private storage failed',
|
|
category='source_resource',
|
|
) from exc
|
|
finally:
|
|
if response is not None:
|
|
response.close()
|
|
if os.path.lexists(temporary):
|
|
durable_unlink(temporary)
|
|
if not published and os.path.lexists(destination):
|
|
durable_unlink(destination)
|
|
|
|
|
|
def fetch_docker_config_payload_classes(
|
|
resolved, bearer_auth=None, *, deadline, min_free_bytes=0,
|
|
):
|
|
layers = list((resolved or {}).get('layers') or ())
|
|
fallback = ['unknown'] * len(layers)
|
|
config = dict((resolved or {}).get('config') or {})
|
|
config.update({'kind': 'config', 'position': 0})
|
|
if (
|
|
config.get('media_type') not in DOCKER_CONFIG_MEDIA_TYPES
|
|
or not isinstance(config.get('size'), int)
|
|
or config['size'] < 0
|
|
or config['size'] > DOCKER_REGISTRY_MANIFEST_MAX_BYTES
|
|
):
|
|
return fallback, bearer_auth
|
|
work_root = None
|
|
destination = None
|
|
try:
|
|
work_root = tempfile.mkdtemp(prefix='docker-history-', dir=get_work_dir())
|
|
harden_private_directory(work_root)
|
|
write_temp_owner(work_root, ['docker-config-history'], os.getpid(), required=True)
|
|
destination = os.path.join(work_root, 'config.json')
|
|
outcome = stream_docker_registry_blob(
|
|
resolved['repository'], config, destination, bearer_auth,
|
|
deadline=float(deadline), min_free_bytes=max(0, int(min_free_bytes or 0)),
|
|
)
|
|
try:
|
|
parsed = validate_docker_content_artifact(destination, config)
|
|
except DockerContentScanError:
|
|
return fallback, outcome.bearer_auth
|
|
return docker_config_payload_classes(parsed, len(layers)), outcome.bearer_auth
|
|
except DockerContentScanError:
|
|
return fallback, bearer_auth
|
|
finally:
|
|
if destination and os.path.lexists(destination):
|
|
durable_unlink(destination)
|
|
if work_root:
|
|
cleanup_command_work_dir(work_root)
|
|
|
|
|
|
def _docker_depth_selection_evidence(
|
|
record, candidate_count, selector_version=DOCKER_DEPTH_SELECTOR_VERSION,
|
|
):
|
|
evidence = {
|
|
'schema': 1,
|
|
'type': 'docker-depth-selection-evidence-v1',
|
|
'selector_version': selector_version,
|
|
'selector_sha256': canonical_selector_hash(selector_version),
|
|
'candidate_distinct_graph_count': int(candidate_count),
|
|
'image_rank': int(record['image_rank']),
|
|
'selection_reason': str(record['selection_reason']),
|
|
'target': str(record['target']),
|
|
'repository': str(record['repository']),
|
|
'manifest_digest': str(record['manifest_digest']),
|
|
'manifest_media_type': str(record['manifest_media_type']),
|
|
'manifest_size_bytes': int(record['manifest_size_bytes']),
|
|
'config_digest': str(record['config_digest']),
|
|
'graph_sha256': str(record['graph_sha256']),
|
|
'layers': [dict(layer) for layer in record['layer_metadata']],
|
|
}
|
|
return {
|
|
**record,
|
|
'candidate_distinct_graph_count': int(candidate_count),
|
|
'selection_evidence_sha256': canonical_docker_depth_selection_evidence_hash(
|
|
evidence
|
|
),
|
|
}
|
|
|
|
|
|
def dockerhub_tag_digest(tag, platform_filter_enabled=False, platform_os='linux', platform_arch='amd64'):
|
|
if not isinstance(tag, dict):
|
|
return ''
|
|
images = tag.get('images') if isinstance(tag.get('images'), list) else []
|
|
wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower())
|
|
candidates = (
|
|
image.get('digest') for image in images
|
|
if isinstance(image, dict)
|
|
and (str(image.get('os') or '').lower(), str(image.get('architecture') or '').lower()) == wanted
|
|
)
|
|
platform_digest = next((
|
|
digest for digest in (normalize_docker_digest(value) for value in candidates) if digest
|
|
), '')
|
|
return platform_digest or normalize_docker_digest(tag.get('digest'))
|
|
|
|
|
|
def fetch_dockerhub_tags(
|
|
repo_name, since=None, limit=1, retry_count=2, retry_delay=5,
|
|
platform_filter_enabled=False, platform_os='linux', platform_arch='amd64',
|
|
platform_candidate_tags=20, return_status=False, *, return_outcome=False,
|
|
fresh_graph_evidence=False, lease_renewal_callback=None,
|
|
selector_version=DOCKER_DEPTH_SELECTOR_VERSION,
|
|
):
|
|
def output(
|
|
tags, status, remote_attempted=False, retry_at=None, error='',
|
|
selection_records=(), candidate_records=(), candidate_distinct_graph_count=0,
|
|
):
|
|
tags = list(tags or [])
|
|
if return_outcome:
|
|
return DockerTagResolutionOutcome(
|
|
tags=tuple(tags), status=str(status),
|
|
remote_attempted=bool(remote_attempted),
|
|
retry_at=retry_at, error=str(error or '')[:500],
|
|
selection_records=tuple(selection_records or ()),
|
|
candidate_records=tuple(candidate_records or selection_records or ()),
|
|
selector_version=selector_version,
|
|
selector_hash=canonical_selector_hash(selector_version),
|
|
candidate_distinct_graph_count=int(candidate_distinct_graph_count or 0),
|
|
fresh_graph_evidence=bool(
|
|
fresh_graph_evidence
|
|
and remote_attempted
|
|
and docker_tag_resolution_is_conclusive(status)
|
|
),
|
|
cache_bypassed=bool(fresh_graph_evidence),
|
|
)
|
|
return (tags, status) if return_status else tags
|
|
|
|
def cache_write(*values, **kwargs):
|
|
if not fresh_graph_evidence:
|
|
put_dockerhub_tag_cache(*values, **kwargs)
|
|
|
|
try:
|
|
retry_count = max(0, min(5, int(retry_count or 0)))
|
|
except (TypeError, ValueError):
|
|
retry_count = 2
|
|
try:
|
|
retry_delay = max(0, min(60, int(retry_delay or 0)))
|
|
except (TypeError, ValueError):
|
|
retry_delay = 5
|
|
|
|
limit = docker_images_per_repository_limit(limit)
|
|
repo_name = str(repo_name or '').strip()
|
|
if '@' in repo_name:
|
|
try:
|
|
return output([parse_docker_target(repo_name)['target']], 'ok')
|
|
except (TypeError, ValueError):
|
|
return output([], 'unknown')
|
|
if ':' in repo_name.rsplit('/', 1)[-1]:
|
|
return output([], 'unknown')
|
|
|
|
platform_variant = (
|
|
f'{selector_version}:{str(platform_os).lower()}/{str(platform_arch).lower()}:'
|
|
f'filter={int(bool(platform_filter_enabled))}:candidates={int(platform_candidate_tags or 0)}'
|
|
)
|
|
cached = None if fresh_graph_evidence else get_dockerhub_tag_cache(
|
|
repo_name, since, limit, platform_variant, return_status=True,
|
|
)
|
|
if cached is not None:
|
|
return output(cached[0], cached[1])
|
|
|
|
rate_limit_state = dockerhub_tag_rate_limit_state()
|
|
if rate_limit_state['active']:
|
|
logger.info(f"Docker Hub tag API is rate-limited; deferring tag fetch for {repo_name} from cache state")
|
|
return output(
|
|
[], 'global_cooldown', remote_attempted=False,
|
|
retry_at=rate_limit_state['retry_at'],
|
|
error='Docker Hub shared rate-limit cooldown is active',
|
|
)
|
|
|
|
if '/' in repo_name:
|
|
namespace, name = repo_name.split('/', 1)
|
|
else:
|
|
namespace, name = 'library', repo_name
|
|
|
|
url = (
|
|
f"https://hub.docker.com/v2/namespaces/{quote(namespace, safe='')}"
|
|
f"/repositories/{quote(name, safe='')}/tags"
|
|
)
|
|
last_error = None
|
|
for attempt in range(retry_count + 1):
|
|
_raise_if_scan_slot_fatal()
|
|
try:
|
|
if lease_renewal_callback is not None:
|
|
if not callable(lease_renewal_callback):
|
|
raise ValueError('Docker resolver lease renewal callback is invalid')
|
|
if lease_renewal_callback() is False:
|
|
raise DockerResolverLeaseLostError(
|
|
'Docker resolver lease renewal was rejected'
|
|
)
|
|
response = dockerhub_tags_response(
|
|
url, {
|
|
'page_size': max(
|
|
1, min(max(limit, int(platform_candidate_tags or 0)), 100),
|
|
),
|
|
},
|
|
)
|
|
if response.status_code == 404:
|
|
cache_write(
|
|
repo_name, since, limit, 'not_found', [],
|
|
getattr(scan_config, 'dockerhub_tag_negative_cache_ttl_sec', 3600),
|
|
'Docker Hub repository not found', platform_variant,
|
|
)
|
|
return output([], 'not_found', remote_attempted=True)
|
|
if response.status_code in (401, 403):
|
|
raise DockerRemoteAccessError(
|
|
f'Docker Hub tags endpoint returned HTTP {response.status_code}',
|
|
status='auth_failed', remote_attempted=True,
|
|
)
|
|
response.raise_for_status()
|
|
graph_candidates = []
|
|
supported_or_unknown = 0
|
|
unsupported = 0
|
|
unresolved_digest = 0
|
|
resolution_status = ''
|
|
resolution_retry_at = None
|
|
resolution_error = ''
|
|
tag_payload = _bounded_docker_registry_json(
|
|
response, 'Docker Hub tags response',
|
|
)
|
|
if not isinstance(tag_payload, dict) or not isinstance(tag_payload.get('results'), list):
|
|
raise DockerRegistryResolutionError('Docker Hub tags response is malformed')
|
|
registry_auth = DockerRegistryAuth(token='')
|
|
for source_index, tag in enumerate(tag_payload.get('results', [])):
|
|
_raise_if_scan_slot_fatal()
|
|
tag_name = tag.get('name')
|
|
if not tag_name:
|
|
continue
|
|
|
|
revision = str(tag.get('last_updated') or tag.get('tag_last_pushed') or '').strip()
|
|
tag_updated = parse_dockerhub_datetime(
|
|
revision
|
|
)
|
|
if since and tag_updated and tag_updated < since:
|
|
continue
|
|
platform_support = docker_tag_platform_support(tag, platform_os, platform_arch) if platform_filter_enabled else True
|
|
if platform_support is False:
|
|
unsupported += 1
|
|
logger.info(
|
|
f"Skipping Docker tag {repo_name}:{tag_name}: no {platform_os}/{platform_arch} image"
|
|
)
|
|
continue
|
|
supported_or_unknown += 1
|
|
tagged_image = f"{repo_name}:{tag_name}"
|
|
try:
|
|
validate_docker_image_reference(tagged_image, require_digest=False)
|
|
except ValueError:
|
|
logger.warning('Skipping invalid Docker Hub image reference for %s tag %s', repo_name, tag_name)
|
|
continue
|
|
digest = dockerhub_tag_digest(tag, platform_filter_enabled, platform_os, platform_arch)
|
|
if not digest:
|
|
unresolved_digest += 1
|
|
logger.warning('Deferring Docker tag without a valid content digest: %s', tagged_image)
|
|
continue
|
|
try:
|
|
graph_kwargs = (
|
|
{'include_descriptors': True} if fresh_graph_evidence else {}
|
|
)
|
|
if lease_renewal_callback is not None:
|
|
graph_kwargs['lease_renewal_callback'] = lease_renewal_callback
|
|
graph, registry_auth = resolve_docker_layer_graph(
|
|
repo_name, digest, platform_os, platform_arch, registry_auth,
|
|
**graph_kwargs,
|
|
)
|
|
except DockerResolverLeaseLostError:
|
|
raise
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except DockerRemoteAccessError as exc:
|
|
unresolved_digest += 1
|
|
resolution_status = exc.status
|
|
resolution_retry_at = exc.retry_at
|
|
resolution_error = str(exc)
|
|
logger.warning(
|
|
'Deferring Docker manifest graph for %s: %s',
|
|
tagged_image, str(exc)[:300],
|
|
)
|
|
break
|
|
except Exception as exc:
|
|
unresolved_digest += 1
|
|
logger.warning(
|
|
'Deferring Docker manifest graph for %s: %s', tagged_image, str(exc)[:300],
|
|
)
|
|
if isinstance(exc, ApiRequestError) or 'rate-limit' in str(exc).lower():
|
|
break
|
|
continue
|
|
if graph is None:
|
|
unsupported += 1
|
|
continue
|
|
target = validate_docker_image_reference(
|
|
f"{repo_name}@{graph['manifest_digest']}"
|
|
)
|
|
graph_candidates.append({
|
|
'name': tag_name,
|
|
'target': target,
|
|
'repository': repo_name.lower(),
|
|
'manifest_digest': graph['manifest_digest'],
|
|
**({
|
|
'manifest_media_type': graph['manifest_media_type'],
|
|
'manifest_size_bytes': graph['manifest_size_bytes'],
|
|
'config_digest': graph['config_digest'],
|
|
'layer_descriptors': graph['layer_descriptors'],
|
|
} if fresh_graph_evidence else {}),
|
|
'layers': graph['layers'],
|
|
'source_index': source_index,
|
|
'updated_at': (
|
|
tag_updated.replace(tzinfo=timezone.utc).timestamp()
|
|
if tag_updated is not None and tag_updated.tzinfo is None
|
|
else tag_updated.timestamp() if tag_updated is not None else None
|
|
),
|
|
})
|
|
candidate_distinct_graph_count = len({
|
|
tuple(candidate['layers']) for candidate in graph_candidates
|
|
})
|
|
candidate_records = (
|
|
select_docker_layer_graphs(
|
|
graph_candidates,
|
|
min(100, candidate_distinct_graph_count),
|
|
replacement_pool=True,
|
|
)
|
|
if fresh_graph_evidence and candidate_distinct_graph_count
|
|
else ()
|
|
)
|
|
selected = (
|
|
candidate_records[:limit]
|
|
if fresh_graph_evidence
|
|
else select_docker_layer_graphs(graph_candidates, limit)
|
|
)
|
|
if fresh_graph_evidence:
|
|
candidate_records = [
|
|
_docker_depth_selection_evidence(
|
|
record, candidate_distinct_graph_count, selector_version,
|
|
)
|
|
for record in candidate_records
|
|
]
|
|
selected = candidate_records[:limit]
|
|
tags = [record['target'] for record in selected]
|
|
ttl = getattr(scan_config, 'dockerhub_tag_cache_ttl_sec', 21600) if tags else getattr(scan_config, 'dockerhub_tag_negative_cache_ttl_sec', 3600)
|
|
status = (
|
|
'partial' if tags and unresolved_digest
|
|
else resolution_status if resolution_status
|
|
else 'ok' if tags
|
|
else 'unknown' if unresolved_digest
|
|
else 'unsupported' if unsupported and not graph_candidates
|
|
else 'empty'
|
|
)
|
|
if status != 'unknown':
|
|
if status == 'ok':
|
|
cache_write(
|
|
repo_name, since, limit, status, tags, ttl,
|
|
platform_variant=platform_variant, tag_records=selected,
|
|
)
|
|
elif status not in ('partial',):
|
|
cache_write(
|
|
repo_name, since, limit, status, tags, ttl,
|
|
platform_variant=platform_variant,
|
|
)
|
|
return output(
|
|
tags, status, remote_attempted=True,
|
|
retry_at=resolution_retry_at, error=resolution_error,
|
|
selection_records=selected, candidate_records=candidate_records,
|
|
candidate_distinct_graph_count=candidate_distinct_graph_count,
|
|
)
|
|
except DockerResolverLeaseLostError:
|
|
raise
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except DockerRemoteAccessError as exc:
|
|
logger.warning('Deferring Docker tag resolution for %s: %s', repo_name, str(exc))
|
|
return output(
|
|
[], exc.status, remote_attempted=exc.remote_attempted,
|
|
retry_at=exc.retry_at, error=str(exc),
|
|
)
|
|
except Exception as e:
|
|
last_error = str(e)
|
|
if '404' in last_error or 'not found' in last_error.lower():
|
|
cache_write(repo_name, since, limit, 'not_found', [], getattr(scan_config, 'dockerhub_tag_negative_cache_ttl_sec', 3600), last_error, platform_variant)
|
|
return output([], 'not_found', remote_attempted=True)
|
|
if attempt < retry_count and isinstance(e, ApiRequestError):
|
|
if lease_renewal_callback is not None and lease_renewal_callback() is False:
|
|
raise DockerResolverLeaseLostError(
|
|
'Docker resolver lease renewal was rejected'
|
|
)
|
|
_wait_or_raise_scan_slot_fatal(min(300, retry_delay * (attempt + 1)))
|
|
continue
|
|
break
|
|
logger.warning(f"Unable to fetch Docker Hub tags for {repo_name}: {last_error}")
|
|
return output(
|
|
[], 'unknown', remote_attempted=True,
|
|
error=last_error or 'Docker tag resolution failed',
|
|
)
|
|
|
|
def resolve_recent_dockerhub_image(
|
|
image, since, platform_filter_enabled=False, platform_os='linux',
|
|
platform_arch='amd64', platform_candidate_tags=20,
|
|
images_per_repository=1, resolve_tags=True,
|
|
):
|
|
repo_name = image.get('repo_name')
|
|
if not repo_name:
|
|
return [], 'missing_date'
|
|
|
|
last_updated = parse_dockerhub_datetime(
|
|
image.get('last_updated') or image.get('last_modified')
|
|
)
|
|
|
|
if last_updated and last_updated < since:
|
|
return [], 'old'
|
|
|
|
if last_updated is None:
|
|
last_updated = fetch_dockerhub_last_updated(repo_name)
|
|
if last_updated is None:
|
|
return [repo_name], 'recent'
|
|
if last_updated < since:
|
|
return [], 'old'
|
|
|
|
if not resolve_tags:
|
|
return [repo_name], 'recent'
|
|
|
|
tags, tag_status = fetch_dockerhub_tags(
|
|
repo_name, since=since,
|
|
limit=docker_images_per_repository_limit(images_per_repository),
|
|
platform_filter_enabled=platform_filter_enabled,
|
|
platform_os=platform_os,
|
|
platform_arch=platform_arch,
|
|
platform_candidate_tags=platform_candidate_tags,
|
|
return_status=True,
|
|
)
|
|
if tags:
|
|
if tag_status != 'ok':
|
|
tags.append(repo_name)
|
|
return tags, 'recent'
|
|
|
|
if not docker_tag_resolution_is_conclusive(tag_status):
|
|
return [repo_name], 'recent'
|
|
return [], 'missing_date'
|
|
|
|
def fetch_recent_dockerhub_images(
|
|
query, since, per_page=100, pages=1, platform_filter_enabled=False,
|
|
platform_os='linux', platform_arch='amd64', platform_candidate_tags=20,
|
|
images_per_repository=1, resolve_tags=True,
|
|
):
|
|
"""Fetch Docker Hub images updated since a specific timestamp"""
|
|
images = []
|
|
page = 1
|
|
per_page = max(1, int(per_page or 1))
|
|
requested_pages, pages = dockerhub_search_page_window(pages)
|
|
if requested_pages > pages:
|
|
logger.info(
|
|
f'Docker Hub search is limited to {pages} accessible page(s); '
|
|
f'capping requested pages from {requested_pages}'
|
|
)
|
|
|
|
logger.info(f"Fetching recent Docker Hub images updated since {since.strftime('%Y-%m-%d')}...")
|
|
|
|
expected_pages = pages
|
|
while page <= expected_pages:
|
|
try:
|
|
logger.info(f"Docker Hub query '{query}': fetching page {page}/{pages}...")
|
|
page_result = fetch_dockerhub_search_page(
|
|
query, page, per_page=per_page, request_timeout=30,
|
|
)
|
|
if page == 1:
|
|
expected_pages = min(
|
|
pages,
|
|
max(1, (page_result['total_count'] + per_page - 1) // per_page),
|
|
)
|
|
repositories = page_result['repositories']
|
|
|
|
if not repositories:
|
|
logger.info(
|
|
f"Docker Hub query '{query}' returned no results "
|
|
f"(total matches: {page_result['total_count']})."
|
|
)
|
|
break
|
|
|
|
logger.info(f"Page {page}: checking dates/tags for {len(repositories)} Docker Hub repositories...")
|
|
new_images = []
|
|
old_images = 0
|
|
missing_dates = 0
|
|
with concurrent.futures.ThreadPoolExecutor(max_workers=min(8, len(repositories))) as executor:
|
|
futures = [
|
|
executor.submit(
|
|
resolve_recent_dockerhub_image, image, since,
|
|
platform_filter_enabled, platform_os, platform_arch, platform_candidate_tags,
|
|
docker_images_per_repository_limit(images_per_repository),
|
|
resolve_tags,
|
|
)
|
|
for image in repositories
|
|
]
|
|
for future in concurrent.futures.as_completed(futures):
|
|
tags, status = future.result()
|
|
if status == 'recent':
|
|
new_images.extend(tags)
|
|
elif status == 'old':
|
|
old_images += 1
|
|
else:
|
|
missing_dates += 1
|
|
|
|
images.extend(new_images)
|
|
logger.info(f"Page {page}: fetched {len(new_images)} recent images, skipped {old_images} older images, skipped {missing_dates} without dates")
|
|
page += 1
|
|
|
|
except DockerHubDiscoveryTransportError:
|
|
raise
|
|
except Exception as e:
|
|
logger.error(f"Docker Hub recent discovery page {page} failed after bounded attempts")
|
|
raise DockerHubDiscoveryTransportError(
|
|
f'Docker Hub recent discovery page {page} failed after bounded attempts'
|
|
) from e
|
|
|
|
return images
|
|
|
|
|
|
def huggingface_space_to_target(space):
|
|
target = {
|
|
'url': space.get('id') or space.get('name') or '',
|
|
'name': space.get('id') or space.get('name') or '',
|
|
'created_at': space.get('createdAt') or space.get('created_at') or '',
|
|
'updated_at': space.get('lastModified') or space.get('updatedAt') or space.get('updated_at') or '',
|
|
}
|
|
for field in ('private', 'protected', 'gated', 'disabled'):
|
|
if field in space:
|
|
target[field] = space[field]
|
|
return target
|
|
|
|
|
|
def fetch_huggingface_spaces(pages=1, token=None, request_timeout=15, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, return_metadata=False, request_attempts=1, retry_delay=0):
|
|
spaces = []
|
|
headers = {'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json'}
|
|
if token:
|
|
headers['Authorization'] = f'Bearer {token}'
|
|
seen_pages = 0
|
|
request_attempts = max(1, int(request_attempts or 1))
|
|
retry_delay = max(0, int(retry_delay or 0))
|
|
request_budget = float(request_timeout) * request_attempts + retry_delay * (request_attempts - 1)
|
|
|
|
if return_metadata:
|
|
next_url = 'https://huggingface.co/api/spaces'
|
|
next_params = {'sort': 'lastModified', 'direction': '-1', 'limit': 100}
|
|
logger.info(f"Fetching newest-modified HuggingFace Spaces for {pages} page(s)...")
|
|
for page in range(max(1, int(pages or 1))):
|
|
try:
|
|
response = api_request(
|
|
'GET', next_url, headers=headers, params=next_params,
|
|
timeout=request_timeout,
|
|
max_retries=request_attempts, retry_delay=retry_delay,
|
|
deadline=time.monotonic() + request_budget,
|
|
)
|
|
if response.status_code >= 400:
|
|
status = response.status_code
|
|
category = 'auth_invalid' if status == 401 else 'auth_forbidden' if status == 403 else 'rate_limit' if status == 429 else 'server_error' if status >= 500 else 'api'
|
|
reset_at = retry_after_reset(response)
|
|
response.close()
|
|
raise RateLimitError(
|
|
'huggingface', f'HuggingFace discovery HTTP {status}',
|
|
reset_at=reset_at, category=category,
|
|
auth_related=status in (401, 403, 429),
|
|
)
|
|
response.raise_for_status()
|
|
data = response.json()
|
|
if not isinstance(data, list):
|
|
raise ValueError('invalid HuggingFace spaces payload')
|
|
page_spaces = [
|
|
huggingface_space_to_target(space) for space in data
|
|
if isinstance(space, dict) and space.get('id')
|
|
]
|
|
if not page_spaces:
|
|
logger.info(f"HuggingFace page {page}: no spaces returned. Stopping.")
|
|
break
|
|
spaces.extend(page_spaces)
|
|
logger.info(f"HuggingFace page {page}: fetched {len(page_spaces)} newest-modified spaces")
|
|
next_link = (getattr(response, 'links', {}) or {}).get('next') or {}
|
|
next_url = str(next_link.get('url') or '')
|
|
next_params = None
|
|
if not next_url:
|
|
break
|
|
except Exception as e:
|
|
if isinstance(e, (ApiRequestError, RateLimitError)):
|
|
raise
|
|
logger.error(f"Error fetching HuggingFace page {page}: {str(e)}")
|
|
raise ApiRequestError(f'HuggingFace discovery payload failed: {e}') from e
|
|
return spaces
|
|
|
|
logger.info(f"Fetching HuggingFace Spaces for {pages} page(s)...")
|
|
for page in range(max(1, int(pages or 1))):
|
|
url = 'https://huggingface.co/spaces-json'
|
|
params = {'p': page, 'withCount': 'false', 'sort': 'created'}
|
|
try:
|
|
response = api_request(
|
|
'GET', url, headers=headers, params=params, timeout=request_timeout,
|
|
max_retries=request_attempts, retry_delay=retry_delay,
|
|
deadline=time.monotonic() + request_budget,
|
|
)
|
|
if response.status_code >= 400:
|
|
status = response.status_code
|
|
category = 'auth_invalid' if status == 401 else 'auth_forbidden' if status == 403 else 'rate_limit' if status == 429 else 'server_error' if status >= 500 else 'api'
|
|
reset_at = retry_after_reset(response)
|
|
response.close()
|
|
raise RateLimitError(
|
|
'huggingface', f'HuggingFace discovery HTTP {status}',
|
|
reset_at=reset_at, category=category,
|
|
auth_related=status in (401, 403, 429),
|
|
)
|
|
response.raise_for_status()
|
|
data = response.json()
|
|
if not isinstance(data, dict) or 'spaces' not in data or not isinstance(data.get('spaces'), list):
|
|
raise ValueError('invalid HuggingFace spaces payload')
|
|
page_spaces = [huggingface_space_to_target(space) for space in data.get('spaces', []) if space.get('id')]
|
|
if not page_spaces:
|
|
logger.info(f"HuggingFace page {page}: no spaces returned. Stopping.")
|
|
break
|
|
spaces.extend(item['url'] for item in page_spaces if item.get('url'))
|
|
logger.info(f"HuggingFace page {page}: fetched {len(page_spaces)} spaces")
|
|
page_number = page + 1
|
|
if stop_on_seen_pages and page_number >= max(1, min_pages_before_stop):
|
|
if page_is_known(
|
|
[item.get('url') for item in page_spaces], known_targets,
|
|
normalize_target, known_target_lookup,
|
|
):
|
|
seen_pages += 1
|
|
logger.info(f"HuggingFace page {page}: all spaces are already queued/checked ({seen_pages}/{seen_page_threshold})")
|
|
if seen_pages >= max(1, seen_page_threshold):
|
|
logger.info(f"Stopping HuggingFace pagination early after {seen_pages} all-known page(s)")
|
|
break
|
|
else:
|
|
seen_pages = 0
|
|
except Exception as e:
|
|
if isinstance(e, (ApiRequestError, RateLimitError)):
|
|
raise
|
|
logger.error(f"Error fetching HuggingFace page {page}: {str(e)}")
|
|
raise ApiRequestError(f'HuggingFace discovery payload failed: {e}') from e
|
|
|
|
return spaces
|
|
|
|
# =====================
|
|
# NPM FETCH FUNCTIONS
|
|
# =====================
|
|
def parse_iso_datetime(value):
|
|
if not value:
|
|
return None
|
|
try:
|
|
parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00'))
|
|
if parsed.tzinfo is None:
|
|
parsed = parsed.replace(tzinfo=timezone.utc)
|
|
return parsed.astimezone(timezone.utc)
|
|
except ValueError:
|
|
return None
|
|
|
|
def npm_package_target(name, version, tarball_url, date=None):
|
|
return json.dumps({
|
|
'name': name,
|
|
'version': version,
|
|
'tarball': tarball_url,
|
|
'date': date or '',
|
|
}, separators=(',', ':'), ensure_ascii=False)
|
|
|
|
def parse_npm_target(target):
|
|
if isinstance(target, dict):
|
|
return target
|
|
target = str(target).strip()
|
|
if target.startswith('{'):
|
|
return json.loads(target)
|
|
package_id, tarball = target.split('|', 1)
|
|
name, version = package_id.rsplit('@', 1)
|
|
return {'name': name, 'version': version, 'tarball': tarball, 'date': ''}
|
|
|
|
def npm_package_id(target):
|
|
data = parse_npm_target(target)
|
|
return f"npm:{data.get('name')}@{data.get('version')}"
|
|
|
|
def select_npm_release_targets(name, data, max_versions=1, cutoff=None):
|
|
targets = []
|
|
version_times = data.get('time') or {}
|
|
for version, version_data in (data.get('versions') or {}).items():
|
|
tarball = (version_data.get('dist') or {}).get('tarball')
|
|
if not tarball:
|
|
continue
|
|
version_date = version_times.get(version) or ''
|
|
parsed_date = parse_iso_datetime(version_date)
|
|
if cutoff and (not parsed_date or parsed_date < cutoff):
|
|
continue
|
|
targets.append({
|
|
'name': name,
|
|
'version': version,
|
|
'tarball': tarball,
|
|
'date': version_date,
|
|
'parsed_date': parsed_date or datetime.min.replace(tzinfo=timezone.utc),
|
|
})
|
|
targets.sort(key=lambda item: item['parsed_date'], reverse=True)
|
|
return targets[:max(1, int(max_versions or 1))]
|
|
|
|
def fetch_npm_packages(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1, repo_candidate_callback=None):
|
|
"""Fetch npm package tarball targets for recent versions matching a query."""
|
|
if not query:
|
|
return []
|
|
|
|
targets = []
|
|
seen_packages = set()
|
|
seen_versions = set()
|
|
cutoff = None
|
|
if max_version_age_days and max_version_age_days > 0:
|
|
cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days)
|
|
|
|
logger.info(f"Fetching npm packages for query: '{query}'...")
|
|
for page in range(max(1, pages)):
|
|
params = {'text': query, 'size': per_page, 'from': page * per_page}
|
|
try:
|
|
response = api_request(
|
|
'GET',
|
|
'https://registry.npmjs.org/-/v1/search',
|
|
params=params,
|
|
headers={'User-Agent': 'GitSecretsScanner/2.0'},
|
|
timeout=request_timeout,
|
|
)
|
|
response.raise_for_status()
|
|
payload = response.json()
|
|
if not isinstance(payload, dict) or 'objects' not in payload or not isinstance(payload.get('objects'), list):
|
|
raise ValueError('invalid npm search payload')
|
|
objects = payload.get('objects', [])
|
|
if objects and not any(isinstance(item, dict) and isinstance(item.get('package'), dict) and item['package'].get('name') for item in objects):
|
|
raise ValueError('npm search payload contains no valid package entries')
|
|
if not objects:
|
|
logger.info(f"npm page {page + 1}: no results")
|
|
break
|
|
|
|
logger.info(f"npm page {page + 1}: fetched {len(objects)} packages")
|
|
for item in objects:
|
|
package = item.get('package', {})
|
|
name = package.get('name')
|
|
version = package.get('version')
|
|
if not name or not version:
|
|
continue
|
|
package_name_key = name.lower()
|
|
if package_name_key in seen_packages:
|
|
continue
|
|
|
|
metadata_url = f"https://registry.npmjs.org/{quote(name, safe='')}"
|
|
metadata = api_request(
|
|
'GET',
|
|
metadata_url,
|
|
headers={'User-Agent': 'GitSecretsScanner/2.0'},
|
|
timeout=request_timeout,
|
|
)
|
|
metadata.raise_for_status()
|
|
data = metadata.json()
|
|
selected_versions = select_npm_release_targets(name, data, versions_per_package, cutoff)
|
|
if repo_candidate_callback:
|
|
repo_candidates = []
|
|
for version_item in selected_versions or [{'version': version}]:
|
|
item_version = version_item.get('version') or version
|
|
for candidate in extract_npm_git_candidates(data, item_version):
|
|
repo_candidates.append({
|
|
'package_source': 'npm',
|
|
'name': name,
|
|
'version': item_version,
|
|
'repo_url': candidate['repo_url'],
|
|
'provider': candidate['provider'],
|
|
'evidence': candidate.get('evidence') or [],
|
|
'confidence': 'high',
|
|
})
|
|
repo_candidate_callback(repo_candidates)
|
|
for item in selected_versions:
|
|
package_key = f"{item['name']}@{item['version']}".lower()
|
|
if package_key in seen_versions:
|
|
continue
|
|
targets.append(npm_package_target(item['name'], item['version'], item['tarball'], item['date']))
|
|
seen_versions.add(package_key)
|
|
seen_packages.add(package_name_key)
|
|
except Exception as e:
|
|
if isinstance(e, ApiRequestError):
|
|
raise
|
|
logger.error(f"Error fetching npm page {page + 1}: {str(e)}")
|
|
raise ApiRequestError(f'npm discovery failed: {e}') from e
|
|
|
|
logger.info(f"npm query '{query}': prepared {len(targets)} package targets")
|
|
return targets
|
|
|
|
# ======================
|
|
# PYPI FETCH FUNCTIONS
|
|
# ======================
|
|
pypi_project_index_cache_path = None
|
|
|
|
def pypi_package_target(name, version, artifact_url, date=None, filename=None, packagetype=None, size=None):
|
|
return json.dumps({
|
|
'source': 'pypi',
|
|
'name': name,
|
|
'version': version,
|
|
'artifact': artifact_url,
|
|
'date': date or '',
|
|
'filename': filename or '',
|
|
'packagetype': packagetype or '',
|
|
'size': size or 0,
|
|
}, separators=(',', ':'), ensure_ascii=False)
|
|
|
|
def parse_pypi_target(target):
|
|
if isinstance(target, dict):
|
|
return target
|
|
target = str(target).strip()
|
|
if target.startswith('{'):
|
|
return json.loads(target)
|
|
package_id, artifact = target.split('|', 1)
|
|
name, version = package_id.rsplit('@', 1)
|
|
return {'source': 'pypi', 'name': name, 'version': version, 'artifact': artifact, 'date': ''}
|
|
|
|
def pypi_package_id(target):
|
|
data = parse_pypi_target(target)
|
|
return f"pypi:{data.get('name')}@{data.get('version')}"
|
|
|
|
# =============================
|
|
# PACKAGE -> GIT FETCH HELPERS
|
|
# =============================
|
|
GIT_PATH_STOP_SEGMENTS = {
|
|
'-', 'issues', 'issue', 'pull', 'pulls', 'merge_requests', 'merge_request',
|
|
'tree', 'blob', 'commit', 'commits', 'releases', 'tags', 'branches', 'wiki',
|
|
}
|
|
|
|
|
|
def canonical_git_path_parts(host, parts):
|
|
if host == 'github.com':
|
|
if len(parts) < 2:
|
|
return []
|
|
return parts[:2]
|
|
|
|
cleaned = []
|
|
for part in parts:
|
|
lowered = part.lower()
|
|
if lowered in GIT_PATH_STOP_SEGMENTS:
|
|
break
|
|
cleaned.append(part)
|
|
if len(cleaned) < 2:
|
|
return []
|
|
return cleaned
|
|
|
|
|
|
def normalize_git_repo_candidate(value):
|
|
if not value:
|
|
return None
|
|
raw = str(value).strip().strip('"\'')
|
|
if not raw:
|
|
return None
|
|
|
|
if raw.startswith('git+'):
|
|
raw = raw[4:]
|
|
if raw.startswith('github:'):
|
|
raw = 'https://github.com/' + raw.split(':', 1)[1]
|
|
elif raw.startswith('gitlab:'):
|
|
raw = 'https://gitlab.com/' + raw.split(':', 1)[1]
|
|
elif raw.startswith('git@github.com:'):
|
|
raw = 'https://github.com/' + raw.split(':', 1)[1]
|
|
elif raw.startswith('git@gitlab.com:'):
|
|
raw = 'https://gitlab.com/' + raw.split(':', 1)[1]
|
|
elif raw.startswith('git://'):
|
|
raw = 'https://' + raw[6:]
|
|
|
|
try:
|
|
parsed = urlsplit(raw)
|
|
except ValueError:
|
|
return None
|
|
if parsed.scheme not in ('http', 'https') or not parsed.netloc:
|
|
return None
|
|
host = (parsed.hostname or '').lower()
|
|
if host in ('www.github.com',):
|
|
host = 'github.com'
|
|
if host in ('www.gitlab.com',):
|
|
host = 'gitlab.com'
|
|
if host not in ('github.com', 'gitlab.com'):
|
|
return None
|
|
|
|
parts = [part for part in parsed.path.strip('/').split('/') if part]
|
|
repo_parts = canonical_git_path_parts(host, parts)
|
|
if not repo_parts:
|
|
return None
|
|
repo_parts[-1] = repo_parts[-1][:-4] if repo_parts[-1].endswith('.git') else repo_parts[-1]
|
|
if any(not part for part in repo_parts):
|
|
return None
|
|
provider = 'github' if host == 'github.com' else 'gitlab'
|
|
repo_path = '/'.join(repo_parts)
|
|
return {
|
|
'provider': provider,
|
|
'repo_url': f'https://{host}/{repo_path}.git',
|
|
'repo_path': repo_path,
|
|
}
|
|
|
|
def package_git_target(package_source, name, version, repo_url, provider, evidence=None, confidence='medium'):
|
|
return json.dumps({
|
|
'source': 'package_git',
|
|
'package_source': package_source,
|
|
'name': name,
|
|
'version': version or '',
|
|
'repo_url': repo_url,
|
|
'provider': provider,
|
|
'evidence': evidence or [],
|
|
'confidence': confidence,
|
|
}, separators=(',', ':'), ensure_ascii=False)
|
|
|
|
def parse_package_git_target(target):
|
|
if isinstance(target, dict):
|
|
return target
|
|
text = str(target).strip()
|
|
if text.startswith('{'):
|
|
return json.loads(text)
|
|
candidate = normalize_git_repo_candidate(text)
|
|
if not candidate:
|
|
raise ValueError(f'Unsupported package_git target: {text}')
|
|
return {
|
|
'source': 'package_git',
|
|
'package_source': 'custom',
|
|
'name': '',
|
|
'version': '',
|
|
'repo_url': candidate['repo_url'],
|
|
'provider': candidate['provider'],
|
|
'evidence': ['custom'],
|
|
'confidence': 'high',
|
|
}
|
|
|
|
def package_git_id(target):
|
|
data = parse_package_git_target(target)
|
|
return f"package_git:{data.get('provider')}:{data.get('repo_url')}".lower()
|
|
|
|
def collect_git_candidates(values):
|
|
candidates = []
|
|
seen = set()
|
|
for evidence, value in values:
|
|
candidate = normalize_git_repo_candidate(value)
|
|
if not candidate:
|
|
continue
|
|
key = candidate['repo_url'].lower()
|
|
if key in seen:
|
|
continue
|
|
seen.add(key)
|
|
candidate['evidence'] = [evidence]
|
|
candidates.append(candidate)
|
|
return candidates
|
|
|
|
|
|
GIT_URL_TEXT_RE = re.compile(
|
|
r'(?:https?://|git\+https?://|git://|git@)(?:github\.com[:/]|gitlab\.com[:/])'
|
|
r'[A-Za-z0-9_.-]+(?:/[A-Za-z0-9_.-]+){1,8}(?:\.git)?(?:/[A-Za-z0-9_.~/-]+)?',
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
|
|
def git_candidates_from_text(label, text, max_urls=8):
|
|
if not text:
|
|
return []
|
|
values = []
|
|
seen = set()
|
|
for match in GIT_URL_TEXT_RE.finditer(str(text)):
|
|
raw = match.group(0).rstrip(').,;\'"<>')
|
|
if raw.startswith('git@github.com/'):
|
|
raw = raw.replace('git@github.com/', 'git@github.com:', 1)
|
|
if raw.startswith('git@gitlab.com/'):
|
|
raw = raw.replace('git@gitlab.com/', 'git@gitlab.com:', 1)
|
|
key = raw.lower()
|
|
if key in seen:
|
|
continue
|
|
seen.add(key)
|
|
values.append((label, raw))
|
|
if len(values) >= max_urls:
|
|
break
|
|
return collect_git_candidates(values)
|
|
|
|
def extract_npm_git_candidates(metadata, version=None):
|
|
values = []
|
|
repository = metadata.get('repository')
|
|
if isinstance(repository, dict):
|
|
values.append(('repository.url', repository.get('url')))
|
|
elif isinstance(repository, str):
|
|
values.append(('repository', repository))
|
|
bugs = metadata.get('bugs')
|
|
if isinstance(bugs, dict):
|
|
values.append(('bugs.url', bugs.get('url')))
|
|
values.append(('homepage', metadata.get('homepage')))
|
|
|
|
version_data = (metadata.get('versions') or {}).get(version or '', {})
|
|
version_repository = version_data.get('repository') if isinstance(version_data, dict) else None
|
|
if isinstance(version_repository, dict):
|
|
values.append(('version.repository.url', version_repository.get('url')))
|
|
elif isinstance(version_repository, str):
|
|
values.append(('version.repository', version_repository))
|
|
if isinstance(version_data, dict):
|
|
version_bugs = version_data.get('bugs')
|
|
if isinstance(version_bugs, dict):
|
|
values.append(('version.bugs.url', version_bugs.get('url')))
|
|
values.append(('version.homepage', version_data.get('homepage')))
|
|
candidates = collect_git_candidates(values)
|
|
candidates.extend(git_candidates_from_text('readme.github_url', metadata.get('readme')))
|
|
candidates.extend(git_candidates_from_text('description.github_url', metadata.get('description')))
|
|
deduped = []
|
|
seen = set()
|
|
for candidate in candidates:
|
|
key = candidate['repo_url'].lower()
|
|
if key in seen:
|
|
continue
|
|
seen.add(key)
|
|
deduped.append(candidate)
|
|
return deduped
|
|
|
|
def extract_pypi_git_candidates(metadata):
|
|
info = metadata.get('info') or {}
|
|
values = []
|
|
project_urls = info.get('project_urls') or {}
|
|
if isinstance(project_urls, dict):
|
|
for key, value in project_urls.items():
|
|
label = str(key).lower()
|
|
if any(item in label for item in ('source', 'repository', 'repo', 'code', 'homepage', 'home', 'bug', 'issue', 'tracker')):
|
|
values.append((f'project_urls.{key}', value))
|
|
values.append(('home_page', info.get('home_page')))
|
|
values.append(('project_url', info.get('project_url')))
|
|
candidates = collect_git_candidates(values)
|
|
candidates.extend(git_candidates_from_text('description.github_url', info.get('description')))
|
|
candidates.extend(git_candidates_from_text('summary.github_url', info.get('summary')))
|
|
deduped = []
|
|
seen = set()
|
|
for candidate in candidates:
|
|
key = candidate['repo_url'].lower()
|
|
if key in seen:
|
|
continue
|
|
seen.add(key)
|
|
deduped.append(candidate)
|
|
return deduped
|
|
|
|
def fetch_npm_package_git_repos(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1):
|
|
if not query:
|
|
return []
|
|
targets = []
|
|
seen_repos = set()
|
|
cutoff = None
|
|
if max_version_age_days and max_version_age_days > 0:
|
|
cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days)
|
|
|
|
logger.info(f"Fetching npm package git repos for query: '{query}'...")
|
|
for page in range(max(1, pages)):
|
|
params = {'text': query, 'size': per_page, 'from': page * per_page}
|
|
try:
|
|
response = api_request(
|
|
'GET',
|
|
'https://registry.npmjs.org/-/v1/search',
|
|
params=params,
|
|
headers={'User-Agent': 'GitSecretsScanner/2.0'},
|
|
timeout=request_timeout,
|
|
)
|
|
response.raise_for_status()
|
|
payload = response.json()
|
|
if not isinstance(payload, dict) or 'objects' not in payload or not isinstance(payload.get('objects'), list):
|
|
raise ValueError('invalid npm package_git search payload')
|
|
objects = payload.get('objects', [])
|
|
if objects and not any(isinstance(item, dict) and isinstance(item.get('package'), dict) and item['package'].get('name') for item in objects):
|
|
raise ValueError('npm package_git payload contains no valid package entries')
|
|
if not objects:
|
|
logger.info(f"npm package_git page {page + 1}: no results")
|
|
break
|
|
logger.info(f"npm package_git page {page + 1}: fetched {len(objects)} packages")
|
|
for item in objects:
|
|
package = item.get('package', {})
|
|
name = package.get('name')
|
|
if not name:
|
|
continue
|
|
metadata = api_request(
|
|
'GET',
|
|
f"https://registry.npmjs.org/{quote(name, safe='')}",
|
|
headers={'User-Agent': 'GitSecretsScanner/2.0'},
|
|
timeout=request_timeout,
|
|
)
|
|
metadata.raise_for_status()
|
|
data = metadata.json()
|
|
selected_versions = select_npm_release_targets(name, data, versions_per_package, cutoff) or [{'version': package.get('version') or ''}]
|
|
for version_item in selected_versions:
|
|
version = version_item.get('version') or package.get('version') or ''
|
|
for candidate in extract_npm_git_candidates(data, version):
|
|
key = candidate['repo_url'].lower()
|
|
if key in seen_repos:
|
|
continue
|
|
seen_repos.add(key)
|
|
targets.append(package_git_target('npm', name, version, candidate['repo_url'], candidate['provider'], candidate['evidence'], 'high'))
|
|
except Exception as e:
|
|
if isinstance(e, ApiRequestError):
|
|
raise
|
|
logger.error(f"Error fetching npm package_git page {page + 1}: {str(e)}")
|
|
raise ApiRequestError(f'npm package_git discovery failed: {e}') from e
|
|
logger.info(f"npm package_git query '{query}': prepared {len(targets)} git repo targets")
|
|
return targets
|
|
|
|
def fetch_pypi_package_git_repos(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1):
|
|
if not query:
|
|
return []
|
|
targets = []
|
|
seen_repos = set()
|
|
cutoff = None
|
|
if max_version_age_days and max_version_age_days > 0:
|
|
cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days)
|
|
|
|
logger.info(f"Fetching PyPI package git repos for query: '{query}'...")
|
|
for page in range(1, max(1, pages) + 1):
|
|
try:
|
|
names = fetch_pypi_package_names(query, page, per_page, request_timeout)
|
|
if not names:
|
|
logger.info(f"PyPI package_git page {page}: no results")
|
|
break
|
|
logger.info(f"PyPI package_git page {page}: fetched {len(names)} package names")
|
|
for name in names:
|
|
try:
|
|
metadata = api_request(
|
|
'GET',
|
|
f"https://pypi.org/pypi/{quote(name, safe='')}/json",
|
|
headers={'User-Agent': 'GitSecretsScanner/2.0'},
|
|
timeout=request_timeout,
|
|
)
|
|
metadata.raise_for_status()
|
|
data = metadata.json()
|
|
except requests.exceptions.HTTPError as e:
|
|
if e.response is not None and e.response.status_code == 404:
|
|
logger.info(f"Skipping PyPI package_git project {name}: metadata not found")
|
|
continue
|
|
raise ApiRequestError(f'PyPI package_git metadata failed for {name}: {e}') from e
|
|
except ApiRequestError:
|
|
raise
|
|
if not isinstance(data, dict) or not isinstance(data.get('info', {}), dict):
|
|
raise ApiRequestError(f'invalid PyPI package_git metadata payload for {name}')
|
|
release_files = select_pypi_release_files(data, cutoff, versions_per_package)
|
|
version = release_files[0]['version'] if release_files else (data.get('info') or {}).get('version') or ''
|
|
package_name = (data.get('info') or {}).get('name') or name
|
|
for candidate in extract_pypi_git_candidates(data):
|
|
key = candidate['repo_url'].lower()
|
|
if key in seen_repos:
|
|
continue
|
|
seen_repos.add(key)
|
|
targets.append(package_git_target('pypi', package_name, version, candidate['repo_url'], candidate['provider'], candidate['evidence'], 'medium'))
|
|
except Exception as e:
|
|
if isinstance(e, ApiRequestError):
|
|
raise
|
|
logger.error(f"Error fetching PyPI package_git page {page}: {str(e)}")
|
|
raise ApiRequestError(f'PyPI package_git discovery failed: {e}') from e
|
|
logger.info(f"PyPI package_git query '{query}': prepared {len(targets)} git repo targets")
|
|
return targets
|
|
|
|
class _PyPIProjectParser(HTMLParser):
|
|
def __init__(self, sink):
|
|
super().__init__(convert_charrefs=True)
|
|
self.sink = sink
|
|
self.in_anchor = False
|
|
self.parts = []
|
|
|
|
def handle_starttag(self, tag, attrs):
|
|
if tag.lower() == 'a':
|
|
self.in_anchor = True
|
|
self.parts = []
|
|
|
|
def handle_endtag(self, tag):
|
|
if tag.lower() == 'a':
|
|
name = ''.join(self.parts).strip()
|
|
if name:
|
|
self.sink(name)
|
|
self.in_anchor = False
|
|
self.parts = []
|
|
|
|
def handle_data(self, data):
|
|
if self.in_anchor:
|
|
text = str(data or '')
|
|
if sum(len(part) for part in self.parts) + len(text) <= 512:
|
|
self.parts.append(text)
|
|
|
|
|
|
def _pypi_index_path():
|
|
global pypi_project_index_cache_path
|
|
if pypi_project_index_cache_path:
|
|
return pypi_project_index_cache_path
|
|
state_dir = os.path.dirname(scan_limiter_db_path())
|
|
require_private_directory(state_dir, create=False)
|
|
pypi_project_index_cache_path = os.path.join(state_dir, 'pypi_project_index.sqlite3')
|
|
return pypi_project_index_cache_path
|
|
|
|
|
|
def load_pypi_project_index(request_timeout=20):
|
|
path = _pypi_index_path()
|
|
refresh_sec = max(3600, int(os.getenv('PYPI_PROJECT_INDEX_REFRESH_SEC', '86400')))
|
|
if not os.path.exists(path):
|
|
descriptor = os.open(
|
|
path, os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0), 0o600,
|
|
)
|
|
os.close(descriptor)
|
|
harden_private_file(path)
|
|
connection = sqlite3.connect(path, timeout=30)
|
|
try:
|
|
connection.execute('PRAGMA journal_mode=DELETE')
|
|
connection.execute('PRAGMA synchronous=FULL')
|
|
connection.executescript('''
|
|
CREATE TABLE IF NOT EXISTS pypi_projects (
|
|
normalized_name TEXT PRIMARY KEY,
|
|
name TEXT NOT NULL
|
|
);
|
|
CREATE TABLE IF NOT EXISTS pypi_index_meta (
|
|
id INTEGER PRIMARY KEY CHECK(id = 1),
|
|
refreshed_at REAL NOT NULL,
|
|
project_count INTEGER NOT NULL
|
|
);
|
|
''')
|
|
current = connection.execute(
|
|
'SELECT refreshed_at, project_count FROM pypi_index_meta WHERE id = 1'
|
|
).fetchone()
|
|
if current and time.time() - float(current[0]) < refresh_sec and int(current[1]) > 0:
|
|
return path
|
|
|
|
response = _direct_request(
|
|
'GET', 'https://pypi.org/simple/',
|
|
headers={'Accept': 'text/html', 'User-Agent': 'GitSecretsScanner/2.0'},
|
|
timeout=request_timeout,
|
|
stream=True,
|
|
)
|
|
response.raise_for_status()
|
|
connection.execute('BEGIN IMMEDIATE')
|
|
connection.execute('DELETE FROM pypi_projects')
|
|
batch = []
|
|
count = 0
|
|
|
|
def accept(name):
|
|
nonlocal count
|
|
normalized = re.sub(r'[-_.]+', '-', name).lower()
|
|
if not normalized or len(normalized) > 512:
|
|
return
|
|
batch.append((normalized, name[:512]))
|
|
if len(batch) >= 1000:
|
|
connection.executemany(
|
|
'INSERT OR IGNORE INTO pypi_projects(normalized_name, name) VALUES (?, ?)',
|
|
batch,
|
|
)
|
|
count += len(batch)
|
|
batch.clear()
|
|
|
|
parser = _PyPIProjectParser(accept)
|
|
decoder = codecs.getincrementaldecoder('utf-8')('strict')
|
|
response_bytes = 0
|
|
response_max_bytes = max(
|
|
1024 * 1024, int(os.getenv('PYPI_PROJECT_INDEX_MAX_BYTES', str(512 * 1024 * 1024))),
|
|
)
|
|
try:
|
|
for chunk in response.iter_content(chunk_size=256 * 1024):
|
|
_raise_if_scan_slot_fatal()
|
|
if chunk:
|
|
response_bytes += len(chunk)
|
|
if response_bytes > response_max_bytes:
|
|
raise ApiRequestError('PyPI simple index exceeds its streamed byte bound')
|
|
parser.feed(decoder.decode(chunk))
|
|
parser.feed(decoder.decode(b'', final=True))
|
|
parser.close()
|
|
finally:
|
|
response.close()
|
|
if batch:
|
|
connection.executemany(
|
|
'INSERT OR IGNORE INTO pypi_projects(normalized_name, name) VALUES (?, ?)',
|
|
batch,
|
|
)
|
|
count += len(batch)
|
|
actual = int(connection.execute('SELECT COUNT(*) FROM pypi_projects').fetchone()[0])
|
|
if actual <= 0:
|
|
raise ApiRequestError('PyPI simple index contains no valid project entries')
|
|
connection.execute(
|
|
'''INSERT INTO pypi_index_meta(id, refreshed_at, project_count) VALUES (1, ?, ?)
|
|
ON CONFLICT(id) DO UPDATE SET refreshed_at = excluded.refreshed_at,
|
|
project_count = excluded.project_count''',
|
|
(time.time(), actual),
|
|
)
|
|
connection.commit()
|
|
logger.info('Streamed %d PyPI project names into the on-disk index', actual)
|
|
return path
|
|
except Exception:
|
|
connection.rollback()
|
|
raise
|
|
finally:
|
|
connection.close()
|
|
if os.path.exists(path):
|
|
harden_private_file(path)
|
|
|
|
def pypi_name_rank(name, query):
|
|
normalized = name.lower()
|
|
query = query.lower()
|
|
parts = [part for part in re.split(r'[-_.]+', normalized) if part]
|
|
if normalized == query:
|
|
rank = 0
|
|
elif normalized.startswith(query):
|
|
rank = 1
|
|
elif query in parts:
|
|
rank = 2
|
|
else:
|
|
rank = 3
|
|
return rank, len(normalized), normalized
|
|
|
|
def fetch_pypi_package_names(query, page=1, per_page=50, request_timeout=20):
|
|
query = query.strip().lower()
|
|
if not query:
|
|
return []
|
|
|
|
tokens = [token for token in re.split(r'\s+', query) if token]
|
|
path = load_pypi_project_index(request_timeout)
|
|
page_size = max(1, int(per_page or 50))
|
|
requested_end = max(1, int(page)) * page_size
|
|
candidate_limit = min(100000, max(1000, requested_end * 20))
|
|
connection = sqlite3.connect(f'file:{path.replace(os.sep, "/")}?mode=ro', uri=True, timeout=30)
|
|
try:
|
|
clauses = ' AND '.join('normalized_name LIKE ?' for _ in tokens)
|
|
rows = connection.execute(
|
|
f'''SELECT name FROM pypi_projects WHERE {clauses}
|
|
ORDER BY normalized_name LIMIT ?''',
|
|
(*[f'%{token}%' for token in tokens], candidate_limit),
|
|
)
|
|
matches = [row[0] for row in rows]
|
|
finally:
|
|
connection.close()
|
|
matches.sort(key=lambda name: pypi_name_rank(name, query))
|
|
|
|
start = (max(1, int(page)) - 1) * page_size
|
|
return matches[start:start + page_size]
|
|
|
|
def select_pypi_release_files(data, cutoff=None, max_versions=1):
|
|
release_candidates = []
|
|
priority_by_type = {'sdist': 2, 'bdist_wheel': 1}
|
|
|
|
for version, files in (data.get('releases') or {}).items():
|
|
file_candidates = []
|
|
for file_info in files or []:
|
|
if file_info.get('yanked'):
|
|
continue
|
|
artifact_url = file_info.get('url')
|
|
if not artifact_url:
|
|
continue
|
|
uploaded = file_info.get('upload_time_iso_8601') or file_info.get('upload_time') or ''
|
|
parsed_date = parse_iso_datetime(uploaded)
|
|
if cutoff and (not parsed_date or parsed_date < cutoff):
|
|
continue
|
|
file_candidates.append({
|
|
'version': version,
|
|
'url': artifact_url,
|
|
'date': uploaded,
|
|
'parsed_date': parsed_date or datetime.min.replace(tzinfo=timezone.utc),
|
|
'filename': file_info.get('filename') or '',
|
|
'packagetype': file_info.get('packagetype') or '',
|
|
'size': file_info.get('size') or 0,
|
|
'priority': priority_by_type.get(file_info.get('packagetype'), 0),
|
|
})
|
|
|
|
if file_candidates:
|
|
release_date = max(item['parsed_date'] for item in file_candidates)
|
|
file_candidates.sort(key=lambda item: (item['priority'], item['parsed_date']), reverse=True)
|
|
release_candidates.append((release_date, file_candidates[0]))
|
|
|
|
if not release_candidates:
|
|
return []
|
|
release_candidates.sort(key=lambda item: item[0], reverse=True)
|
|
return [item[1] for item in release_candidates[:max(1, int(max_versions or 1))]]
|
|
|
|
def select_pypi_release_file(data, cutoff=None):
|
|
files = select_pypi_release_files(data, cutoff, 1)
|
|
return files[0] if files else None
|
|
|
|
def fetch_pypi_packages(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1, repo_candidate_callback=None):
|
|
"""Fetch PyPI package artifact targets for recent matching releases."""
|
|
if not query:
|
|
return []
|
|
|
|
targets = []
|
|
seen = set()
|
|
cutoff = None
|
|
if max_version_age_days and max_version_age_days > 0:
|
|
cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days)
|
|
|
|
logger.info(f"Fetching PyPI packages for query: '{query}'...")
|
|
for page in range(1, max(1, pages) + 1):
|
|
try:
|
|
names = fetch_pypi_package_names(query, page, per_page, request_timeout)
|
|
if not names:
|
|
logger.info(f"PyPI page {page}: no results")
|
|
break
|
|
|
|
logger.info(f"PyPI page {page}: fetched {len(names)} package names")
|
|
|
|
for name in names:
|
|
normalized_name = name.lower()
|
|
if normalized_name in seen:
|
|
continue
|
|
|
|
metadata_url = f"https://pypi.org/pypi/{quote(name, safe='')}/json"
|
|
try:
|
|
metadata = api_request(
|
|
'GET',
|
|
metadata_url,
|
|
headers={'User-Agent': 'GitSecretsScanner/2.0'},
|
|
timeout=request_timeout,
|
|
)
|
|
metadata.raise_for_status()
|
|
data = metadata.json()
|
|
except requests.exceptions.HTTPError as e:
|
|
if e.response is not None and e.response.status_code == 404:
|
|
logger.info(f"Skipping PyPI project {name}: metadata not found")
|
|
seen.add(normalized_name)
|
|
continue
|
|
logger.warning(f"Skipping PyPI project {name}: metadata fetch failed: {str(e)}")
|
|
raise ApiRequestError(f'PyPI metadata failed for {name}: {e}') from e
|
|
except ApiRequestError:
|
|
raise
|
|
except Exception as e:
|
|
raise ApiRequestError(f'PyPI metadata payload failed for {name}: {e}') from e
|
|
if not isinstance(data, dict) or not isinstance(data.get('info', {}), dict):
|
|
raise ApiRequestError(f'invalid PyPI metadata payload for {name}')
|
|
|
|
release_files = select_pypi_release_files(data, cutoff, versions_per_package)
|
|
if not release_files:
|
|
continue
|
|
|
|
if repo_candidate_callback:
|
|
package_name = data.get('info', {}).get('name') or name
|
|
repo_candidates = []
|
|
for release_file in release_files:
|
|
for candidate in extract_pypi_git_candidates(data):
|
|
repo_candidates.append({
|
|
'package_source': 'pypi',
|
|
'name': package_name,
|
|
'version': release_file['version'],
|
|
'repo_url': candidate['repo_url'],
|
|
'provider': candidate['provider'],
|
|
'evidence': candidate.get('evidence') or [],
|
|
'confidence': 'medium',
|
|
})
|
|
repo_candidate_callback(repo_candidates)
|
|
|
|
for release_file in release_files:
|
|
targets.append(pypi_package_target(
|
|
data.get('info', {}).get('name') or name,
|
|
release_file['version'],
|
|
release_file['url'],
|
|
release_file['date'],
|
|
release_file['filename'],
|
|
release_file['packagetype'],
|
|
release_file['size'],
|
|
))
|
|
seen.add(normalized_name)
|
|
except Exception as e:
|
|
if isinstance(e, ApiRequestError):
|
|
raise
|
|
logger.error(f"Error fetching PyPI page {page}: {str(e)}")
|
|
raise ApiRequestError(f'PyPI discovery failed: {e}') from e
|
|
|
|
logger.info(f"PyPI query '{query}': prepared {len(targets)} package targets")
|
|
return targets
|
|
|
|
# =====================
|
|
# SCANNING FUNCTIONS
|
|
# =====================
|
|
def check_dependencies():
|
|
"""Verify required dependencies are installed"""
|
|
probe_dir = None
|
|
try:
|
|
work_dir = get_work_dir()
|
|
probe_dir = tempfile.mkdtemp(prefix='trufflehog-probe-', dir=work_dir)
|
|
harden_private_directory(probe_dir)
|
|
command = [get_trufflehog_cmd(), '--version', '--no-update']
|
|
write_temp_owner(probe_dir, command, os.getpid())
|
|
env = os.environ.copy()
|
|
strip_supervisor_credentials(env)
|
|
env['PATH'] = os.pathsep.join([
|
|
os.path.expanduser('~/bin'),
|
|
os.path.expanduser('~/.local/bin'),
|
|
env.get('PATH', '')
|
|
])
|
|
prepend_client_git_environment(env)
|
|
env['TEMP'] = probe_dir
|
|
env['TMP'] = probe_dir
|
|
env['TMPDIR'] = probe_dir
|
|
require_trufflehog_launch_authority(command)
|
|
result = run_owned(
|
|
command,
|
|
stdout=subprocess.PIPE,
|
|
stderr=subprocess.PIPE,
|
|
timeout=30,
|
|
env=env,
|
|
cwd=probe_dir,
|
|
creationflags=subprocess.CREATE_NO_WINDOW if os.name == 'nt' else 0,
|
|
)
|
|
if result.returncode != 0:
|
|
stderr = (result.stderr or b'') if isinstance(result.stderr, bytes) else str(result.stderr or '').encode('utf-8', errors='replace')
|
|
detail = stderr.decode('utf-8', errors='replace').strip()[:500]
|
|
raise RuntimeError(f"TruffleHog version probe exited with code {result.returncode}: {detail or 'no stderr'}")
|
|
logger.info("trufflehog is installed and working")
|
|
return True
|
|
except Exception as e:
|
|
if isinstance(e, subprocess.TimeoutExpired):
|
|
detail = f"timed out after {e.timeout}s"
|
|
else:
|
|
detail = f"{type(e).__name__}: {e}"
|
|
logger.error("TruffleHog dependency probe failed: %s", redact_scan_command_text([detail])[:500])
|
|
if isinstance(e, FileNotFoundError):
|
|
logger.error("TruffleHog executable was not found. Please install it.")
|
|
logger.error("Visit https://github.com/trufflesecurity/trufflehog for installation instructions.")
|
|
else:
|
|
logger.error("TruffleHog dependency probe failed closed; executable authority was not bypassed.")
|
|
return False
|
|
finally:
|
|
if probe_dir:
|
|
cleanup_command_work_dir(probe_dir)
|
|
|
|
|
|
def command_output_limits():
|
|
if _client_scan_policy.get() is None:
|
|
stdout_mb = int_setting(
|
|
os.getenv('TRUFFLEHOG_STDOUT_MAX_MB'),
|
|
getattr(scan_config, 'trufflehog_stdout_max_mb', 32),
|
|
)
|
|
stderr_mb = int_setting(
|
|
os.getenv('TRUFFLEHOG_STDERR_MAX_MB'),
|
|
getattr(scan_config, 'trufflehog_stderr_max_mb', 8),
|
|
)
|
|
else:
|
|
stdout_mb = int(_scan_policy_value('trufflehog_stdout_max_mb', 32))
|
|
stderr_mb = int(_scan_policy_value('trufflehog_stderr_max_mb', 8))
|
|
requested_stdout = max(1, int_setting(
|
|
stdout_mb, 32,
|
|
)) * 1024 * 1024
|
|
requested_stderr = max(1, int_setting(
|
|
stderr_mb, 8,
|
|
)) * 1024 * 1024
|
|
event_limit = max(1, int(_scan_policy_value(
|
|
'result_bundle_max_event_bytes', 64 * 1024 * 1024,
|
|
)))
|
|
reserve = min(32 * 1024 * 1024, max(64 * 1024, event_limit // 4))
|
|
output_budget = max(2, event_limit - reserve)
|
|
requested_total = requested_stdout + requested_stderr
|
|
if requested_total <= output_budget:
|
|
return requested_stdout, requested_stderr
|
|
stdout = max(1, (output_budget * requested_stdout) // requested_total)
|
|
stderr = max(1, output_budget - stdout)
|
|
return stdout, stderr
|
|
|
|
|
|
class CommandOutputLimitError(RuntimeError):
|
|
pass
|
|
|
|
|
|
class StreamedCommandOutput:
|
|
def __init__(self, stdout_file, stderr_file, returncode, max_stdout, max_stderr, synthetic_stderr=''):
|
|
self._stdout = stdout_file
|
|
self._stderr = stderr_file
|
|
self.returncode = int(returncode)
|
|
self.max_stdout = int(max_stdout)
|
|
self.max_stderr = int(max_stderr)
|
|
self.synthetic_stderr = str(synthetic_stderr or '')
|
|
|
|
@staticmethod
|
|
def _lines(handle, byte_limit, max_line_bytes, max_lines, redactions=()):
|
|
handle.seek(0)
|
|
consumed = 0
|
|
count = 0
|
|
while consumed < byte_limit and count < max_lines:
|
|
raw = handle.readline(min(max_line_bytes + 1, byte_limit - consumed + 1))
|
|
if not raw:
|
|
return
|
|
consumed += len(raw)
|
|
count += 1
|
|
if len(raw) > max_line_bytes and not raw.endswith((b'\n', b'\r')):
|
|
raise CommandOutputLimitError(f'command output line exceeded {max_line_bytes} bytes')
|
|
line = raw.decode('utf-8', errors='replace')
|
|
if redactions:
|
|
line = redact_secrets(line, redactions)
|
|
yield line
|
|
if handle.read(1):
|
|
raise CommandOutputLimitError('command output exceeded its line or byte bound')
|
|
|
|
def stdout_lines(self, max_line_bytes=16 * 1024 * 1024, max_lines=20000, redactions=()):
|
|
return self._lines(
|
|
self._stdout, self.max_stdout, max(1, int(max_line_bytes)),
|
|
max(1, int(max_lines)), redactions,
|
|
)
|
|
|
|
def stderr_lines(self, max_line_bytes=8192, max_lines=2000, redactions=()):
|
|
for line in self._lines(
|
|
self._stderr, self.max_stderr, max(1, int(max_line_bytes)),
|
|
max(1, int(max_lines)), redactions,
|
|
):
|
|
yield line
|
|
if self.synthetic_stderr:
|
|
yield self.synthetic_stderr.rstrip('\r\n') + '\n'
|
|
|
|
@staticmethod
|
|
def _raw_bytes(handle):
|
|
position = handle.tell()
|
|
try:
|
|
handle.seek(0)
|
|
return handle.read()
|
|
finally:
|
|
handle.seek(position)
|
|
|
|
def raw_stdout_bytes(self):
|
|
return self._raw_bytes(self._stdout)
|
|
|
|
def raw_stderr_bytes(self):
|
|
return self._raw_bytes(self._stderr)
|
|
|
|
|
|
@contextmanager
|
|
def streamed_output_from_text(stdout='', stderr='', returncode=0):
|
|
stdout_file = io.BytesIO(str(stdout).encode('utf-8'))
|
|
stderr_file = io.BytesIO(str(stderr).encode('utf-8'))
|
|
yield StreamedCommandOutput(
|
|
stdout_file, stderr_file, returncode,
|
|
max(1, len(stdout_file.getvalue())), max(1, len(stderr_file.getvalue())),
|
|
)
|
|
|
|
|
|
def _check_command_staging(roots, deadline, output_files=()):
|
|
"""Best-effort live staging watchdog, not a filesystem quota or atomic snapshot."""
|
|
exceeded = 'TruffleHog staging limit exceeded'
|
|
unavailable = 'Unable to monitor TruffleHog staging'
|
|
total_bytes = 0
|
|
entries = 0
|
|
try:
|
|
unique_roots = []
|
|
for root in sorted({os.path.normcase(os.path.abspath(root)) for root in roots}, key=len):
|
|
if not any(root == parent or root.startswith(os.path.join(parent, '')) for parent in unique_roots):
|
|
unique_roots.append(root)
|
|
# Check ancestors too: lstat on a child alone would follow a linked parent.
|
|
for root in unique_roots:
|
|
ancestor = root
|
|
while True:
|
|
if time.monotonic() >= deadline:
|
|
return unavailable
|
|
try:
|
|
info = os.lstat(ancestor)
|
|
except FileNotFoundError:
|
|
pass
|
|
else:
|
|
if stat.S_ISLNK(info.st_mode) or getattr(info, 'st_file_attributes', 0) & stat.FILE_ATTRIBUTE_REPARSE_POINT:
|
|
return unavailable
|
|
parent = os.path.dirname(ancestor)
|
|
if parent == ancestor:
|
|
break
|
|
ancestor = parent
|
|
# POSIX TemporaryFile output may be unlinked and thus absent from scandir.
|
|
for handle in output_files:
|
|
if time.monotonic() >= deadline:
|
|
return unavailable
|
|
info = os.fstat(handle.fileno())
|
|
if info.st_nlink == 0:
|
|
total_bytes += info.st_size
|
|
entries += 1
|
|
pending = list(unique_roots)
|
|
while pending:
|
|
if time.monotonic() >= deadline:
|
|
return unavailable
|
|
path = pending.pop()
|
|
try:
|
|
info = os.lstat(path)
|
|
if stat.S_ISLNK(info.st_mode) or getattr(info, 'st_file_attributes', 0) & stat.FILE_ATTRIBUTE_REPARSE_POINT:
|
|
return unavailable
|
|
if stat.S_ISREG(info.st_mode):
|
|
total_bytes += info.st_size
|
|
elif not stat.S_ISDIR(info.st_mode):
|
|
return unavailable
|
|
if total_bytes > 2 * 1024 ** 3:
|
|
return exceeded
|
|
if stat.S_ISDIR(info.st_mode):
|
|
with os.scandir(path) as children:
|
|
for child in children:
|
|
if time.monotonic() >= deadline:
|
|
return unavailable
|
|
entries += 1
|
|
if entries > 100000:
|
|
return exceeded
|
|
pending.append(child.path)
|
|
except FileNotFoundError:
|
|
continue
|
|
if time.monotonic() >= deadline:
|
|
return unavailable
|
|
if total_bytes > 2 * 1024 ** 3 or entries > 100000:
|
|
return exceeded
|
|
except (OSError, ValueError):
|
|
return unavailable
|
|
return ''
|
|
|
|
|
|
@contextmanager
|
|
def run_command_streamed(cmd, timeout_sec, env=None, *, deadline=None, staging_roots=None, native_git_clone=False):
|
|
"""Run one owned command; optional staging roots also monitor its private temp tree."""
|
|
if type(native_git_clone) is not bool:
|
|
raise ValueError('native_git_clone must be an explicit boolean')
|
|
if native_git_clone and not staging_roots:
|
|
raise ValueError('native Git clone requires private staging roots')
|
|
command_work_dir = None
|
|
process = None
|
|
scan_slot = None
|
|
owns_scan_slot = False
|
|
release_scan_slot = True
|
|
stdout_file = None
|
|
stderr_file = None
|
|
max_stdout, max_stderr = command_output_limits()
|
|
fatal_slot_error = None
|
|
propagating_fatal_error = None
|
|
returncode = -1
|
|
synthetic_stderr = ''
|
|
armed_owners = []
|
|
command_owner_published = False
|
|
requested_timeout = (
|
|
max(1, int(timeout_sec or 1))
|
|
if deadline is None
|
|
else max(0.001, float(timeout_sec or 0.001))
|
|
)
|
|
command_deadline = time.monotonic() + requested_timeout
|
|
if deadline is not None:
|
|
deadline = float(deadline)
|
|
if not math.isfinite(deadline):
|
|
raise ValueError('command deadline must be finite')
|
|
command_deadline = min(command_deadline, deadline)
|
|
try:
|
|
_raise_if_scan_slot_fatal()
|
|
env = dict(os.environ if env is None else env)
|
|
for key in list(env):
|
|
if key.lower() in ('http_proxy', 'https_proxy', 'all_proxy', 'no_proxy'):
|
|
del env[key]
|
|
env['NO_PROXY'] = '*'
|
|
strip_supervisor_credentials(env)
|
|
if os.name == 'nt':
|
|
env['PATH'] = os.pathsep.join([
|
|
os.path.expanduser('~/bin'), os.path.expanduser('~/.local/bin'), env.get('PATH', ''),
|
|
])
|
|
prepend_client_git_environment(env)
|
|
env['GIT_TERMINAL_PROMPT'] = '0'
|
|
env['GIT_ASKPASS'] = 'true'
|
|
min_free_gb = max(0.0, float(getattr(scan_config, 'min_free_gb', 0) or 0))
|
|
min_free_bytes = int(min_free_gb * 1024 * 1024 * 1024)
|
|
|
|
borrowed_scan_slot, scan_slot = scoped_scan_slot_lease()
|
|
if not borrowed_scan_slot:
|
|
scan_slot = acquire_scan_slot(
|
|
cmd, max(0.001, command_deadline - time.monotonic()),
|
|
)
|
|
owns_scan_slot = True
|
|
if scan_slot and not scan_slot.releasable:
|
|
raise RuntimeError('scan slot is fail-closed after unconfirmed child termination')
|
|
|
|
command_work_dir = create_command_work_dir()
|
|
shared_owners = _shared_staging_owners((command_work_dir, *(staging_roots or ())))
|
|
staging_roots = (command_work_dir, *staging_roots) if staging_roots is not None else None
|
|
env['TEMP'] = command_work_dir
|
|
env['TMP'] = command_work_dir
|
|
env['TMPDIR'] = command_work_dir
|
|
if env.get('TRUF_GIT_TOKEN'):
|
|
if os.name == 'nt':
|
|
askpass_path = os.path.join(command_work_dir, 'git-askpass.cmd')
|
|
with open(askpass_path, 'w', encoding='ascii') as askpass:
|
|
askpass.write('@echo off\r\n')
|
|
askpass.write('echo %~1 | findstr /I "username" >nul\r\n')
|
|
askpass.write('if %errorlevel%==0 (echo %TRUF_GIT_USERNAME%) else (echo %TRUF_GIT_TOKEN%)\r\n')
|
|
else:
|
|
askpass_path = os.path.join(command_work_dir, 'git-askpass.sh')
|
|
with open(askpass_path, 'x', encoding='ascii', newline='\n') as askpass:
|
|
askpass.write(
|
|
'#!/bin/sh\n'
|
|
'case "$1" in\n'
|
|
' *[Uu][Ss][Ee][Rr][Nn][Aa][Mm][Ee]*) printf \'%s\\n\' "$TRUF_GIT_USERNAME" ;;\n'
|
|
' *) printf \'%s\\n\' "$TRUF_GIT_TOKEN" ;;\n'
|
|
'esac\n'
|
|
)
|
|
harden_private_file(askpass_path)
|
|
if os.name != 'nt':
|
|
os.chmod(askpass_path, stat.S_IRWXU)
|
|
env['GIT_ASKPASS'] = askpass_path
|
|
|
|
creationflags = (
|
|
subprocess.CREATE_NEW_PROCESS_GROUP
|
|
| subprocess.CREATE_NO_WINDOW
|
|
) if os.name == 'nt' else 0
|
|
stdout_file = tempfile.TemporaryFile(mode='w+b', dir=command_work_dir)
|
|
stderr_file = tempfile.TemporaryFile(mode='w+b', dir=command_work_dir)
|
|
if native_git_clone:
|
|
require_git_clone_launch_authority(cmd)
|
|
if time.monotonic() >= command_deadline:
|
|
raise subprocess.TimeoutExpired(cmd, timeout_sec)
|
|
destination_parent = os.path.normcase(os.path.abspath(os.path.dirname(cmd[-1])))
|
|
if destination_parent not in {os.path.normcase(os.path.abspath(root)) for root in staging_roots}:
|
|
raise RuntimeError('Git clone destination is outside its private staging parent')
|
|
staging_error = _check_command_staging(staging_roots, command_deadline, (stdout_file, stderr_file))
|
|
if staging_error:
|
|
raise RuntimeError(staging_error)
|
|
else:
|
|
require_trufflehog_launch_authority(cmd)
|
|
if time.monotonic() >= command_deadline:
|
|
raise subprocess.TimeoutExpired(cmd, timeout_sec)
|
|
process_options = {
|
|
'stdout': stdout_file, 'stderr': stderr_file, 'env': env,
|
|
'cwd': command_work_dir, 'stdin': subprocess.DEVNULL,
|
|
'creationflags': creationflags,
|
|
}
|
|
if os.name == 'nt':
|
|
job_memory_limit_bytes = _scan_policy_value(
|
|
'trufflehog_job_memory_limit_bytes', 0,
|
|
)
|
|
if isinstance(job_memory_limit_bytes, bool) or not isinstance(job_memory_limit_bytes, int) or job_memory_limit_bytes <= 0:
|
|
raise RuntimeError('trufflehog_job_memory_limit_bytes must be a positive integer on Windows')
|
|
process_options['job_memory_limit_bytes'] = job_memory_limit_bytes
|
|
job_cpu_weight = int(_scan_policy_value('trufflehog_windows_job_cpu_weight', 0))
|
|
memory_priority = int(_scan_policy_value('trufflehog_windows_memory_priority', 0))
|
|
if job_cpu_weight < 0 or job_cpu_weight > 9:
|
|
raise RuntimeError('trufflehog_windows_job_cpu_weight must be between 0 and 9')
|
|
if memory_priority < 0 or memory_priority > 5:
|
|
raise RuntimeError('trufflehog_windows_memory_priority must be between 0 and 5')
|
|
process_options['job_cpu_weight'] = job_cpu_weight
|
|
process_options['process_memory_priority'] = memory_priority
|
|
for root, marker in shared_owners:
|
|
armed_owners.append((root, marker))
|
|
pending = dict(marker, child_pid=None, child_creation_time=None, child_executable=None)
|
|
atomic_write_private_json(os.path.join(root, TEMP_OWNER_FILE), pending)
|
|
if time.monotonic() >= command_deadline:
|
|
raise subprocess.TimeoutExpired(cmd, timeout_sec)
|
|
process = OwnedProcess(cmd, **process_options)
|
|
if not process.job_membership_verified:
|
|
raise RuntimeError('TruffleHog exact Job membership was not verified')
|
|
if scan_slot and not scan_slot.set_child_pid(process.pid):
|
|
raise RuntimeError('unable to publish TruffleHog child identity to the scan slot')
|
|
write_temp_owner(
|
|
command_work_dir, cmd, process.pid, required=True,
|
|
owner_identity=getattr(process, 'payload_identity', None),
|
|
)
|
|
command_owner_published = True
|
|
child_identity = getattr(process, 'payload_identity', None)
|
|
for root, marker in armed_owners:
|
|
if root == canonical_path(command_work_dir):
|
|
continue
|
|
if not isinstance(child_identity, dict) or any(not child_identity.get(field) for field in ('pid', 'creation_time', 'executable')):
|
|
raise RuntimeError('shared staging child identity is unavailable')
|
|
active = dict(marker, **{f'child_{field}': child_identity[field] for field in ('pid', 'creation_time', 'executable')})
|
|
atomic_write_private_json(os.path.join(root, TEMP_OWNER_FILE), active)
|
|
limit_error = ''
|
|
next_staging_check = 0.0
|
|
while True:
|
|
completed = process.poll() is not None
|
|
if completed and staging_roots is None:
|
|
break
|
|
_raise_if_scan_slot_fatal()
|
|
stdout_size = os.fstat(stdout_file.fileno()).st_size
|
|
stderr_size = os.fstat(stderr_file.fileno()).st_size
|
|
if stdout_size > max_stdout:
|
|
limit_error = f'TruffleHog stdout exceeded {max_stdout} bytes'
|
|
elif stderr_size > max_stderr:
|
|
limit_error = f'TruffleHog stderr exceeded {max_stderr} bytes'
|
|
elif min_free_bytes:
|
|
try:
|
|
free_bytes = shutil.disk_usage(command_work_dir).free
|
|
except OSError as exc:
|
|
limit_error = (
|
|
'Unable to monitor TruffleHog staging' if staging_roots is not None
|
|
else f'Unable to monitor configured TruffleHog work volume free space: {exc}'
|
|
)
|
|
else:
|
|
if free_bytes <= min_free_bytes:
|
|
limit_error = (
|
|
f'Not enough free space on configured TruffleHog work volume {command_work_dir}: '
|
|
f'{free_bytes / (1024 ** 3):.2f} GB free, minimum is {min_free_gb:.2f} GB'
|
|
)
|
|
if not limit_error and staging_roots is not None and (
|
|
completed or time.monotonic() >= next_staging_check
|
|
):
|
|
limit_error = _check_command_staging(
|
|
staging_roots, command_deadline, (stdout_file, stderr_file),
|
|
)
|
|
next_staging_check = time.monotonic() + 1.0
|
|
if limit_error:
|
|
process.kill()
|
|
try:
|
|
process.wait(timeout=10)
|
|
except subprocess.TimeoutExpired:
|
|
release_scan_slot = False
|
|
limit_error += '; process tree termination failed'
|
|
synthetic_stderr = f'Error: {limit_error}'
|
|
returncode = -1
|
|
break
|
|
if completed:
|
|
break
|
|
if time.monotonic() >= command_deadline:
|
|
raise subprocess.TimeoutExpired(cmd, timeout_sec)
|
|
if _scan_slot_fatal_event.wait(0.2):
|
|
_raise_if_scan_slot_fatal()
|
|
if not limit_error:
|
|
returncode = process.returncode
|
|
stdout_file.seek(0)
|
|
stderr_file.seek(0)
|
|
yield StreamedCommandOutput(
|
|
stdout_file, stderr_file, returncode, max_stdout, max_stderr, synthetic_stderr,
|
|
)
|
|
except subprocess.TimeoutExpired:
|
|
if process and process.poll() is None:
|
|
process.kill()
|
|
try:
|
|
process.wait(timeout=5)
|
|
except subprocess.TimeoutExpired:
|
|
release_scan_slot = False
|
|
if stdout_file is None or stderr_file is None:
|
|
raise
|
|
timeout_error = f'Command timed out after {timeout_sec} seconds'
|
|
if staging_roots is not None:
|
|
staging_error = _check_command_staging(
|
|
staging_roots, command_deadline, (stdout_file, stderr_file),
|
|
)
|
|
if staging_error:
|
|
timeout_error = f'Error: {staging_error}'
|
|
stdout_file.seek(0)
|
|
stderr_file.seek(0)
|
|
yield StreamedCommandOutput(
|
|
stdout_file, stderr_file, -1, max_stdout, max_stderr,
|
|
timeout_error,
|
|
)
|
|
except ScanSlotFatalError as exc:
|
|
propagating_fatal_error = exc
|
|
raise
|
|
finally:
|
|
if process:
|
|
try:
|
|
child_running = process.poll() is None
|
|
except Exception:
|
|
child_running = True
|
|
if child_running:
|
|
try:
|
|
process.kill()
|
|
process.wait(timeout=5)
|
|
except Exception:
|
|
release_scan_slot = False
|
|
if not release_scan_slot:
|
|
if scan_slot:
|
|
scan_slot.mark_non_releasable()
|
|
detail = 'FATAL: TruffleHog child termination was not confirmed; scan capacity remains fail-closed'
|
|
_set_scan_slot_fatal(detail)
|
|
fatal_slot_error = propagating_fatal_error or ScanSlotFatalError(detail)
|
|
restore_error = None
|
|
if release_scan_slot:
|
|
for root, marker in armed_owners:
|
|
if command_owner_published and root == canonical_path(command_work_dir):
|
|
continue
|
|
try:
|
|
atomic_write_private_json(os.path.join(root, TEMP_OWNER_FILE), marker)
|
|
except Exception as exc:
|
|
restore_error = exc
|
|
if scan_slot and owns_scan_slot and release_scan_slot:
|
|
scan_slot.release()
|
|
if stdout_file:
|
|
stdout_file.close()
|
|
if stderr_file:
|
|
stderr_file.close()
|
|
if release_scan_slot and fatal_slot_error is None and propagating_fatal_error is None:
|
|
cleanup_command_work_dir(command_work_dir)
|
|
if fatal_slot_error is not None:
|
|
raise fatal_slot_error
|
|
if restore_error is not None:
|
|
raise RuntimeError('Unable to restore shared staging ownership after child termination') from restore_error
|
|
|
|
|
|
def run_command(cmd, timeout_sec, env=None):
|
|
"""Bounded compatibility adapter; production parsers use run_command_streamed."""
|
|
try:
|
|
with run_command_streamed(cmd, timeout_sec, env) as output:
|
|
stdout = ''.join(output.stdout_lines(max_line_bytes=output.max_stdout, max_lines=20000))
|
|
stderr = ''.join(output.stderr_lines(max_line_bytes=output.max_stderr, max_lines=2000))
|
|
return stdout, stderr, output.returncode
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except Exception as exc:
|
|
return '', f'Error running command: {exc}', -1
|
|
|
|
|
|
def _trufflehog_diagnostic_policy(line, source_type, returncode):
|
|
text = str(line or '').strip()
|
|
payload = None
|
|
try:
|
|
parsed = json.loads(text)
|
|
if isinstance(parsed, dict):
|
|
payload = parsed
|
|
except (TypeError, ValueError):
|
|
pass
|
|
|
|
if payload and payload.get('errors'):
|
|
causes = payload['errors']
|
|
limits = _trufflehog_diagnostic_limits()
|
|
if not isinstance(causes, list):
|
|
return 'error', 'trufflehog', True
|
|
if len(causes) > limits['errors'] or any(
|
|
isinstance(cause, str) and (
|
|
len(cause) > limits['line_chars']
|
|
or len(cause.encode('utf-8', errors='replace')) > limits['line_bytes']
|
|
) for cause in causes
|
|
):
|
|
return 'error', 'output_limit', False
|
|
envelope = dict(payload)
|
|
del envelope['errors']
|
|
envelope_policy = _trufflehog_diagnostic_policy(json.dumps(envelope), source_type, returncode)
|
|
if envelope_policy[1] == 'source_auth' and not (payload.get('error') or payload.get('message')):
|
|
envelope_policy = ('error', 'auth_or_permission', envelope_policy[2])
|
|
policies = []
|
|
for cause in causes:
|
|
if not isinstance(cause, str) or not cause.strip():
|
|
continue
|
|
cause_payload = {'level': 'error', 'error': cause}
|
|
policy = _trufflehog_diagnostic_policy(json.dumps(cause_payload), source_type, returncode)
|
|
cause_payload['msg'] = payload.get('msg') or ''
|
|
contextual_policy = _trufflehog_diagnostic_policy(json.dumps(cause_payload), source_type, returncode)
|
|
# Keep known warnings, but not at the expense of an independent fatal cause.
|
|
if contextual_policy[0] == 'warning' and (
|
|
policy[1] == 'trufflehog'
|
|
or (
|
|
contextual_policy[1] == 'detector_timeout' and policy[1] == 'timeout'
|
|
and cause.strip().lower() == 'context deadline exceeded'
|
|
)
|
|
):
|
|
policy = contextual_policy
|
|
if policy[1] == 'source_auth':
|
|
# A nested diagnostic can be detector verification, not the selected source credential.
|
|
policy = ('error', 'auth_or_permission', policy[2])
|
|
policies.append(policy)
|
|
# Envelope summaries are not additional causes, but explicit fatal details are.
|
|
if envelope_policy[0] == 'error' and (
|
|
envelope_policy[1] != 'trufflehog' or payload.get('error') or payload.get('message')
|
|
):
|
|
policies.append(envelope_policy)
|
|
fatal = [policy for policy in policies if policy[0] == 'error']
|
|
if not fatal:
|
|
if policies and all(policy[0] == 'warning' for policy in policies):
|
|
return 'warning', policies[0][1], all(policy[2] for policy in policies)
|
|
return 'error', 'trufflehog', True
|
|
retryable = all(policy[2] for policy in fatal)
|
|
classes = {policy[1] for policy in fatal}
|
|
for error_class in (
|
|
'memory_limit', 'source_configuration', 'output_limit', 'source_resource',
|
|
'source_auth', 'docker_registry_access', 'auth_or_permission',
|
|
):
|
|
if error_class in classes:
|
|
return 'error', error_class, retryable
|
|
return 'error', next(iter(classes)) if len(classes) == 1 else 'mixed', retryable
|
|
|
|
if (
|
|
source_type == 'docker' and payload
|
|
and payload.get('error') and payload.get('message')
|
|
and payload['error'] != payload['message']
|
|
):
|
|
# Independent detail channels must survive codec recovery; reuse fatal priority and bounds.
|
|
policy = _trufflehog_diagnostic_policy(json.dumps({
|
|
'level': 'error', 'msg': payload.get('msg'),
|
|
'errors': [str(payload['error']), str(payload['message'])],
|
|
}), source_type, returncode)
|
|
return policy if policy[0] == 'error' else ('error', 'trufflehog', True)
|
|
|
|
message = str((payload or {}).get('msg') or '')
|
|
detail = str((payload or {}).get('error') or (payload or {}).get('message') or '')
|
|
level = str((payload or {}).get('level') or '').lower()
|
|
message_lower = message.lower()
|
|
detail_lower = detail.lower()
|
|
combined = f'{message_lower} {detail_lower}' if payload else text.lower()
|
|
direct_kind = _client_remote_execution_kind.get()
|
|
|
|
if any(token in combined for token in ('virtualalloc', 'out of memory', 'cannot allocate memory')):
|
|
return 'error', 'memory_limit', False
|
|
if any(token in combined for token in ('unknown flag', 'unknown command', 'invalid detector', 'failed to load config')):
|
|
return 'error', 'source_configuration', False
|
|
if any(token in combined for token in (
|
|
'trufflehog stdout exceeded', 'trufflehog stderr exceeded',
|
|
)):
|
|
return 'error', 'output_limit', False
|
|
if any(token in combined for token in (
|
|
'no space left', 'not enough free space', 'disk quota',
|
|
'trufflehog work_dir', 'configured command work directory',
|
|
'temporary file', 'temporaryfile', 'disk fsync',
|
|
)):
|
|
return 'error', 'source_resource', True
|
|
if message_lower == 'a detector ignored the context timeout':
|
|
return 'warning', 'detector_timeout', False
|
|
if source_type == 'huggingface' and detail_lower == 'no repo found for repo':
|
|
return 'permanent', 'huggingface_no_repo', False
|
|
if source_type == 'docker' and 'no child with platform linux/amd64' in detail_lower:
|
|
return 'permanent', 'docker_no_linux_amd64', False
|
|
|
|
if any(token in combined for token in ('timed out', 'timeout', 'deadline exceeded')):
|
|
return 'error', 'timeout', True
|
|
if (
|
|
source_type == 'docker' and payload
|
|
and payload.get('msg') == 'error processing layer'
|
|
and payload.get('error') == 'unexpected EOF'
|
|
):
|
|
# Keep the persisted retry class; layer EOF alone does not prove a network cause.
|
|
return 'error', 'network', True
|
|
if any(token in combined for token in (
|
|
'connection reset', 'connection aborted', 'connection refused', 'could not resolve host',
|
|
'temporary failure', 'tls', 'ssl', 'proxy error', 'network is unreachable', 'unexpected eof',
|
|
)):
|
|
return 'error', 'network', True
|
|
if any(token in combined for token in (' 408', ' 429', ' 500', ' 502', ' 503', ' 504', 'too many requests', 'rate limit')):
|
|
return 'error', 'remote_transient', True
|
|
auth_error = any(token in combined for token in (
|
|
'authentication failed', 'unauthorized', 'invalid username or token', 'invalid api key',
|
|
'bad credentials',
|
|
))
|
|
if source_type == 'huggingface' and direct_kind == 'huggingface_space_v1' and (
|
|
auth_error or any(token in combined for token in (
|
|
'permission denied', 'repository not found', 'private repository',
|
|
'gated repo', ' 401', ' 403',
|
|
))
|
|
):
|
|
return 'permanent', 'huggingface_inaccessible', False
|
|
if source_type == 'docker' and direct_kind == 'docker_direct_v1' and (
|
|
auth_error or any(token in combined for token in (
|
|
'permission denied', 'pull access denied',
|
|
'requested access to the resource is denied', 'insufficient scope',
|
|
'manifest unknown', 'name unknown', 'repository does not exist',
|
|
' 401', ' 403', ' 404',
|
|
))
|
|
):
|
|
return 'permanent', 'docker_registry_access', False
|
|
if source_type == 'docker' and auth_error:
|
|
# A registry manifest may be private or stale; the rotating credential cannot be
|
|
# attributed from this target diagnostic, so it must not disable the whole pool.
|
|
return 'error', 'docker_registry_access', False
|
|
if auth_error:
|
|
return 'error', 'source_auth', True
|
|
if 'permission denied' in combined:
|
|
return 'error', 'auth_or_permission', True
|
|
if any(token in combined for token in ('killed by signal', 'terminated by signal', 'process terminated', 'segmentation fault')):
|
|
return 'error', 'command_exit', True
|
|
|
|
if (
|
|
source_type in ('docker', 'filesystem') and payload
|
|
and message == 'skipping file: size exceeds max allowed'
|
|
and not any(key in payload for key in ('error', 'errors', 'message'))
|
|
):
|
|
return 'warning' if returncode == 0 else 'error', 'archive_member_size', False
|
|
|
|
if returncode == 0:
|
|
if message_lower == 'error cleaning temporary artifacts':
|
|
return 'warning', 'cleanup', False
|
|
if message_lower == 'skipping result: invalid' and detail_lower == 'empty raw':
|
|
return 'warning', 'invalid_empty_result', False
|
|
if message_lower == 'non-critical error processing chunk':
|
|
return 'warning', 'chunk_processing', False
|
|
if source_type == 'git' and message_lower == 'error reading chunk' and detail_lower == 'brotli: excessive input':
|
|
return 'warning', 'chunk_read', False
|
|
if source_type in ('npm', 'pypi', 'postman', 'filesystem') and message_lower == 'error reading chunk' and any(
|
|
token in detail_lower for token in ('brotli:', 'flate: corrupt input', 'error identifying archive', 'invalid header')
|
|
):
|
|
return 'warning', 'chunk_read', False
|
|
if source_type == 'docker' and message_lower == 'error processing layer' and detail_lower == 'gzip: invalid header':
|
|
return 'warning', 'docker_layer_gzip', False
|
|
|
|
if payload and ('error' in level or 'error' in message_lower or payload.get('error')):
|
|
return 'error', 'trufflehog', True
|
|
if not payload and re.search(r'\b(error|failed|fatal|panic)\b', combined):
|
|
return 'error', 'command', True
|
|
# Only observed structured progress is exempt from retention, never error details.
|
|
if (
|
|
source_type in ('docker', 'filesystem') and payload
|
|
and payload.get('logger') == 'trufflehog'
|
|
and not any(key in payload for key in ('error', 'errors', 'message'))
|
|
and (
|
|
(level == 'info-0' and message in ('running source', 'finished scanning'))
|
|
or (level == 'info-2' and message in (
|
|
'trufflehog dev', 'starting scanner workers', 'starting detector workers',
|
|
'starting verificationOverlap workers', 'starting notifier workers', 'enumerating source',
|
|
))
|
|
or (source_type == 'docker' and level == 'info-2' and message in (
|
|
'scanning image', 'scanning image history', 'scanning image history entry',
|
|
'scanning image layers', 'scanning layer',
|
|
))
|
|
)
|
|
):
|
|
return 'routine', '', False
|
|
return 'info', '', False
|
|
|
|
|
|
def _trufflehog_diagnostic_limits():
|
|
return {
|
|
'lines': min(2000, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_lines', 2000), 2000))),
|
|
'line_chars': min(8192, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_line_chars', 8192), 8192))),
|
|
'line_bytes': min(8192, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_line_bytes', 8192), 8192))),
|
|
'errors': min(200, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_errors', 200), 200))),
|
|
'warnings': min(200, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_warnings', 200), 200))),
|
|
'unclassified': min(20, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_unclassified', 20), 20))),
|
|
}
|
|
|
|
|
|
def _iter_output_lines(value):
|
|
if isinstance(value, str):
|
|
yield from io.StringIO(value)
|
|
return
|
|
if isinstance(value, bytes):
|
|
for raw_line in io.BytesIO(value):
|
|
yield raw_line.decode('utf-8', errors='replace')
|
|
return
|
|
if value is None:
|
|
return
|
|
yield from value
|
|
|
|
|
|
def apply_trufflehog_diagnostics(
|
|
results, stderr, returncode, source_type, require_completion=False,
|
|
redactions=(),
|
|
):
|
|
errors = []
|
|
warnings = []
|
|
warning_classes = []
|
|
warning_retryability = []
|
|
permanent = []
|
|
unclassified = []
|
|
limits = _trufflehog_diagnostic_limits()
|
|
line_count = 0
|
|
timed_out_seen = False
|
|
finished_seen = False
|
|
output_limit = ''
|
|
|
|
captured_stdout = captured_stderr = None
|
|
captured_transformation = None
|
|
if isinstance(stderr, StreamedCommandOutput):
|
|
captured_stdout = stderr.raw_stdout_bytes()
|
|
captured_stderr = stderr.raw_stderr_bytes()
|
|
captured_transformation = (
|
|
'legacy E-frame classification parsed raw process stderr with any '
|
|
'pre-existing configured parser redactions; '
|
|
'canonical process material preserves the pre-parse captured bytes'
|
|
)
|
|
stderr = stderr.stderr_lines(redactions=redactions)
|
|
|
|
try:
|
|
for raw_line in _iter_output_lines(stderr):
|
|
if line_count >= limits['lines']:
|
|
output_limit = f'total line limit of {limits["lines"]} exceeded'
|
|
break
|
|
line_count += 1
|
|
if len(raw_line) > limits['line_chars']:
|
|
output_limit = f'line character limit of {limits["line_chars"]} exceeded'
|
|
break
|
|
if len(raw_line.encode('utf-8', errors='replace')) > limits['line_bytes']:
|
|
output_limit = f'line byte limit of {limits["line_bytes"]} exceeded'
|
|
break
|
|
line = raw_line.strip()
|
|
if not line:
|
|
continue
|
|
try:
|
|
payload = json.loads(line)
|
|
if isinstance(payload, dict) and str(payload.get('msg') or '').strip().lower() == 'finished scanning':
|
|
finished_seen = True
|
|
except (TypeError, ValueError):
|
|
pass
|
|
if 'timed out' in line.lower():
|
|
timed_out_seen = True
|
|
severity, error_class, retryable = _trufflehog_diagnostic_policy(line, source_type, returncode)
|
|
if severity == 'warning':
|
|
if len(warnings) + len(permanent) >= limits['warnings']:
|
|
output_limit = f'retained warning limit of {limits["warnings"]} exceeded'
|
|
break
|
|
warnings.append(line)
|
|
warning_retryability.append(bool(retryable))
|
|
if error_class:
|
|
warning_classes.append(error_class)
|
|
elif severity == 'permanent':
|
|
if len(warnings) + len(permanent) >= limits['warnings']:
|
|
output_limit = f'retained warning limit of {limits["warnings"]} exceeded'
|
|
break
|
|
permanent.append((line, error_class))
|
|
elif severity == 'error':
|
|
if len(errors) >= limits['errors']:
|
|
output_limit = f'retained error limit of {limits["errors"]} exceeded'
|
|
break
|
|
errors.append((line, error_class, retryable))
|
|
if error_class == 'memory_limit':
|
|
break
|
|
elif severity != 'routine': # Routine records still count against the hard limits above.
|
|
if len(unclassified) >= limits['unclassified']:
|
|
output_limit = f'retained unclassified limit of {limits["unclassified"]} exceeded'
|
|
break
|
|
unclassified.append(line)
|
|
except CommandOutputLimitError as exc:
|
|
output_limit = str(exc)
|
|
|
|
if output_limit:
|
|
synthetic = (
|
|
f'TruffleHog diagnostic output_limit reached: {output_limit}; '
|
|
'remaining diagnostic output was not retained'
|
|
)[:limits['line_chars']]
|
|
errors = errors[:max(0, limits['errors'] - 1)]
|
|
errors.append((synthetic, 'source_resource', True))
|
|
|
|
if permanent and not errors:
|
|
warnings.extend(line for line, _ in permanent)
|
|
warning_classes.extend(error_class for _, error_class in permanent if error_class)
|
|
classes = {error_class for _, error_class in permanent}
|
|
if classes in ({'huggingface_no_repo'}, {'huggingface_inaccessible'}):
|
|
results['skipped'] = 'HuggingFace Space repository is unavailable'
|
|
elif classes == {'docker_no_linux_amd64'}:
|
|
results['skipped'] = 'Docker image has no linux/amd64 manifest'
|
|
elif classes == {'docker_registry_access'}:
|
|
results['skipped'] = 'Docker image is unavailable to the worker'
|
|
else:
|
|
results['skipped'] = 'target is permanently unavailable'
|
|
results['error_class'] = next(iter(classes), 'permanent')
|
|
results['retryable'] = False
|
|
else:
|
|
warnings.extend(line for line, _ in permanent)
|
|
warning_classes.extend(error_class for _, error_class in permanent if error_class)
|
|
|
|
completion_required = (
|
|
source_type == 'docker' or bool(require_completion)
|
|
or (source_type == 'git' and 'chunk_read' in warning_classes)
|
|
)
|
|
if completion_required and not errors and not permanent and not results.get('skipped'):
|
|
if returncode != 0 and finished_seen:
|
|
errors.append((f'TruffleHog exited with code {returncode} after the completion marker', 'wrapper_exit', True))
|
|
elif returncode != 0 or not finished_seen:
|
|
errors.append((f'TruffleHog exited with code {returncode} before the completion marker', 'command_incomplete', True))
|
|
elif returncode != 0 and not errors and not permanent:
|
|
errors.append((f'TruffleHog exited with code {returncode} without a fatal diagnostic', 'command_exit', True))
|
|
elif returncode != 0 and warnings and not errors and not results.get('skipped'):
|
|
errors.append((f'TruffleHog exited with code {returncode} after non-fatal diagnostics', 'command_exit', True))
|
|
|
|
if errors:
|
|
results['errors'] = [line for line, _, _ in errors]
|
|
classes = [error_class for _, error_class, _ in errors if error_class]
|
|
results['error_class'] = classes[0] if len(set(classes)) <= 1 else 'mixed'
|
|
results['retryable'] = all(retryable for _, _, retryable in errors)
|
|
results['source_failure'] = any(error_class in classes for error_class in ('source_configuration', 'source_resource', 'source_auth'))
|
|
if results['source_failure']:
|
|
results['source_failure_category'] = 'source_auth' if 'source_auth' in classes else 'source_resource' if 'source_resource' in classes else 'source_configuration'
|
|
results['source_failure_auth_related'] = 'source_auth' in classes
|
|
if output_limit:
|
|
results['error_class'] = 'source_resource'
|
|
results['retryable'] = True
|
|
results['source_failure'] = True
|
|
results['source_failure_category'] = 'source_resource'
|
|
results['source_failure_auth_related'] = False
|
|
if warnings:
|
|
results['warnings'] = warnings
|
|
results['warning_classes'] = sorted(set(warning_classes))
|
|
results['degraded'] = not bool(results.get('skipped'))
|
|
# Nonfatal coverage warnings must not suppress retries of fatal errors.
|
|
if warning_retryability and not errors:
|
|
warnings_retryable = all(warning_retryability)
|
|
results['retryable'] = bool(
|
|
results.get('retryable', True)
|
|
) and warnings_retryable
|
|
|
|
scan_meta = results.setdefault('scan_meta', {})
|
|
scan_meta['trufflehog_returncode'] = returncode
|
|
scan_meta['trufflehog_finished'] = finished_seen
|
|
scan_meta['command_timed_out'] = returncode == -1 and timed_out_seen
|
|
scan_meta['diagnostic_lines_processed'] = line_count
|
|
if warning_retryability:
|
|
scan_meta['trufflehog_warnings_retryable'] = all(warning_retryability)
|
|
if output_limit:
|
|
scan_meta['diagnostic_output_limited'] = True
|
|
scan_meta['diagnostic_output_limit_reason'] = output_limit
|
|
if unclassified:
|
|
scan_meta['stderr_unclassified'] = unclassified
|
|
if results.get('errors') and captured_stderr is not None:
|
|
results['_diagnostic_raw_stdout_b64'] = base64.b64encode(
|
|
captured_stdout or b''
|
|
).decode('ascii')
|
|
results['_diagnostic_raw_stderr_b64'] = base64.b64encode(
|
|
captured_stderr
|
|
).decode('ascii')
|
|
results['_diagnostic_stderr_transformation'] = captured_transformation
|
|
return results
|
|
|
|
|
|
def convert_package_git_unavailable_to_skip(result):
|
|
errors = result.get('errors') or []
|
|
if not errors:
|
|
return result
|
|
for error in errors:
|
|
text = str(error).lower()
|
|
if not (
|
|
('repository not found' in text or 'project not found' in text)
|
|
and ('failed to clone' in text or 'remote:' in text or 'error preparing repo' in text)
|
|
):
|
|
return result
|
|
result['warnings'] = list(result.get('warnings') or []) + list(errors)
|
|
result['warning_classes'] = sorted(set(list(result.get('warning_classes') or []) + ['package_git_repo_unavailable']))
|
|
result['errors'] = []
|
|
result['skipped'] = 'package_git repository is unavailable or private'
|
|
result['error_class'] = 'package_git_repo_unavailable'
|
|
result['retryable'] = False
|
|
result['degraded'] = False
|
|
return result
|
|
|
|
|
|
def apply_result_error_scope(result):
|
|
errors = result.get('errors') or []
|
|
if not errors or result.get('error_class'):
|
|
return result
|
|
text = '\n'.join(str(error) for error in errors).lower()
|
|
if any(token in text for token in (
|
|
'no space left', 'not enough free space', 'disk quota',
|
|
'unable to create npm work dir', 'unable to create pypi work dir',
|
|
'unable to create postman work dir', 'unable to create github actions work dir',
|
|
'unable to create gitlab ci work dir',
|
|
)):
|
|
result['error_class'] = 'source_resource'
|
|
result['retryable'] = True
|
|
result['source_failure'] = True
|
|
result['source_failure_category'] = 'source_resource'
|
|
return result
|
|
|
|
|
|
def append_trufflehog_findings(results, stdout):
|
|
invalid = []
|
|
if _client_scan_policy.get() is None:
|
|
max_findings = max(1, int(os.getenv('TRUFFLEHOG_MAX_FINDINGS_PER_TARGET', '20000')))
|
|
else:
|
|
max_findings = max(1, int(_scan_policy_value(
|
|
'trufflehog_max_findings_per_target', 20000,
|
|
)))
|
|
try:
|
|
lines = _iter_output_lines(stdout)
|
|
for line in lines:
|
|
if not line.strip():
|
|
continue
|
|
try:
|
|
finding = json.loads(line)
|
|
if not isinstance(finding, dict):
|
|
raise ValueError('finding JSON is not an object')
|
|
if len(results.setdefault('findings', [])) >= max_findings:
|
|
results.setdefault('errors', []).append(f'TruffleHog findings exceeded {max_findings} per target')
|
|
results['error_class'] = 'output_limit'
|
|
results['retryable'] = False
|
|
break
|
|
results['findings'].append(finding)
|
|
except (json.JSONDecodeError, ValueError) as exc:
|
|
if len(invalid) < 5:
|
|
invalid.append(f'{str(exc)}: {line[:300]}')
|
|
except CommandOutputLimitError as exc:
|
|
invalid.append(str(exc))
|
|
if invalid:
|
|
results.setdefault('errors', []).append('Malformed TruffleHog JSON output: ' + '; '.join(invalid))
|
|
results['error_class'] = 'output_parse'
|
|
results['retryable'] = True
|
|
return results
|
|
|
|
def parse_git_scan_target(target):
|
|
text = str(target or '').strip()
|
|
if not text.startswith('{'):
|
|
return {'url': text, 'branch': '', 'metadata': {}}
|
|
try:
|
|
data = json.loads(text)
|
|
except (TypeError, ValueError):
|
|
return {'url': text, 'branch': '', 'metadata': {}}
|
|
if not isinstance(data, dict):
|
|
return {'url': text, 'branch': '', 'metadata': {}}
|
|
return {
|
|
'url': str(data.get('url') or data.get('repo_url') or text).strip(),
|
|
'branch': str(data.get('branch') or '').strip(),
|
|
'metadata': data,
|
|
}
|
|
|
|
|
|
def git_branch_ref(branch):
|
|
branch = str(branch or '')
|
|
resolution = {
|
|
'provider': 'github',
|
|
'repo_url': 'https://github.com/a/b.git',
|
|
'repo_path': 'a/b',
|
|
'branch': branch,
|
|
'ref': f'refs/heads/{branch}',
|
|
'head_sha': '0' * 40,
|
|
'ref_source': 'explicit',
|
|
}
|
|
validate_git_resolution(resolution)
|
|
return resolution['ref']
|
|
|
|
|
|
def normalize_git_scan_resolution_target(target, provider=None):
|
|
target_info = parse_git_scan_target(target)
|
|
raw_url = str(target_info['url'] or '').strip()
|
|
if not raw_url or re.search(r'[\x00-\x20\x7f]', raw_url):
|
|
raise ValueError('Git target URL is empty or contains control characters')
|
|
if re.search(r'%(?:2f|5c)', raw_url, flags=re.IGNORECASE):
|
|
raise ValueError('Git target URL contains an encoded path separator')
|
|
parse_url = raw_url[4:] if raw_url.startswith('git+') else raw_url
|
|
if parse_url.startswith(('github:', 'gitlab:')):
|
|
if any(marker in parse_url for marker in ('?', '#', '@')):
|
|
raise ValueError('Git target shorthand contains unsafe URL components')
|
|
else:
|
|
try:
|
|
parsed = urlsplit(parse_url)
|
|
port = parsed.port
|
|
except ValueError as exc:
|
|
raise ValueError('Git target URL is malformed') from exc
|
|
if (
|
|
parsed.scheme not in ('http', 'https', 'git') or not parsed.netloc
|
|
or parsed.username is not None or parsed.password is not None
|
|
or parsed.query or parsed.fragment or port is not None
|
|
):
|
|
raise ValueError('Git target URL contains unsupported or unsafe components')
|
|
normalized = normalize_git_repo_candidate(raw_url)
|
|
if not normalized:
|
|
raise ValueError('Git target is not a supported GitHub or GitLab repository')
|
|
expected_provider = str(provider or '').strip().lower()
|
|
if expected_provider and normalized['provider'] != expected_provider:
|
|
raise ValueError('Git target provider does not match the scan source')
|
|
|
|
metadata = target_info['metadata']
|
|
branch = str(target_info.get('branch') or '').strip()
|
|
raw_ref = str(metadata.get('ref') or '').strip() if isinstance(metadata, dict) else ''
|
|
ref_branch = ''
|
|
if raw_ref:
|
|
prefix = 'refs/heads/'
|
|
if not raw_ref.startswith(prefix):
|
|
raise ValueError('Git target ref must be a branch ref')
|
|
ref_branch = raw_ref[len(prefix):]
|
|
git_branch_ref(ref_branch)
|
|
if branch:
|
|
git_branch_ref(branch)
|
|
if branch and ref_branch and branch != ref_branch:
|
|
raise ValueError('Git target branch and ref hints conflict')
|
|
branch = branch or ref_branch
|
|
return normalized, branch
|
|
|
|
|
|
def git_ref_resolution_api_json(
|
|
provider, url, token, deadline, request_attempts, timeout_sec, max_response_bytes,
|
|
):
|
|
response = api_request(
|
|
'GET', url,
|
|
headers=github_headers(token) if provider == 'github' else gitlab_headers(token),
|
|
timeout=max(0.001, float(timeout_sec)), max_retries=max(1, int(request_attempts)),
|
|
retry_delay=1, deadline=deadline, stream=True, allow_redirects=False,
|
|
)
|
|
if 300 <= response.status_code < 400:
|
|
response.close()
|
|
raise ApiRequestError(f'{provider} ref resolution refused an HTTP redirect')
|
|
if response.status_code >= 400:
|
|
try:
|
|
error = github_api_error(response) if provider == 'github' else gitlab_api_error(response)
|
|
finally:
|
|
response.close()
|
|
raise error
|
|
payload = bounded_response_json(response, max_bytes=max_response_bytes)
|
|
if not isinstance(payload, dict):
|
|
raise ApiRequestError(f'{provider} ref resolution response is not an object')
|
|
return payload
|
|
|
|
|
|
def redacted_git_resolution_error(exc, token):
|
|
message = redact_secrets(str(exc), [token])
|
|
if isinstance(exc, RateLimitError):
|
|
return RateLimitError(
|
|
exc.source, message, reset_at=exc.reset_at, category=exc.category,
|
|
retryable=exc.retryable, auth_related=exc.auth_related,
|
|
)
|
|
if isinstance(exc, ApiRequestError):
|
|
return ApiRequestError(message)
|
|
return exc
|
|
|
|
|
|
def resolve_github_ref_head(
|
|
repo_path, token=None, ref_hint=None, *, request_attempts=2, timeout_sec=10,
|
|
deadline=None, max_response_bytes=1 << 20,
|
|
):
|
|
deadline = float(deadline) if deadline is not None else time.monotonic() + max(0.001, float(timeout_sec))
|
|
encoded_repo = quote(str(repo_path), safe='/')
|
|
branch = str(ref_hint or '')
|
|
ref_source = 'explicit' if branch else 'provider_default'
|
|
try:
|
|
if not branch:
|
|
payload = git_ref_resolution_api_json(
|
|
'github', f'https://api.github.com/repos/{encoded_repo}', token, deadline,
|
|
request_attempts, timeout_sec, max_response_bytes,
|
|
)
|
|
branch = str(payload.get('default_branch') or '')
|
|
git_branch_ref(branch)
|
|
ref = f'refs/heads/{branch}'
|
|
payload = git_ref_resolution_api_json(
|
|
'github', f'https://api.github.com/repos/{encoded_repo}/git/ref/{quote("heads/" + branch, safe="")}',
|
|
token, deadline, request_attempts, timeout_sec, max_response_bytes,
|
|
)
|
|
obj = payload.get('object')
|
|
if payload.get('ref') != ref or not isinstance(obj, dict) or obj.get('type') != 'commit':
|
|
raise ApiRequestError('GitHub ref resolution returned a mismatched commit ref')
|
|
resolved = {
|
|
'provider': 'github', 'repo_url': f'https://github.com/{repo_path}.git',
|
|
'repo_path': str(repo_path), 'branch': branch, 'ref': ref,
|
|
'head_sha': str(obj.get('sha') or '').lower(), 'ref_source': ref_source,
|
|
}
|
|
return validate_git_resolution(resolved)
|
|
except (RateLimitError, ApiRequestError) as exc:
|
|
sanitized = redacted_git_resolution_error(exc, token)
|
|
if sanitized is exc:
|
|
raise
|
|
raise sanitized from exc
|
|
|
|
|
|
def resolve_gitlab_ref_head(
|
|
repo_path, token=None, ref_hint=None, *, request_attempts=2, timeout_sec=10,
|
|
deadline=None, max_response_bytes=1 << 20,
|
|
):
|
|
deadline = float(deadline) if deadline is not None else time.monotonic() + max(0.001, float(timeout_sec))
|
|
project_url = f'https://gitlab.com/api/v4/projects/{quote(str(repo_path), safe="")}'
|
|
branch = str(ref_hint or '')
|
|
ref_source = 'explicit' if branch else 'provider_default'
|
|
try:
|
|
if not branch:
|
|
payload = git_ref_resolution_api_json(
|
|
'gitlab', project_url, token, deadline, request_attempts, timeout_sec,
|
|
max_response_bytes,
|
|
)
|
|
branch = str(payload.get('default_branch') or '')
|
|
git_branch_ref(branch)
|
|
payload = git_ref_resolution_api_json(
|
|
'gitlab', f'{project_url}/repository/branches/{quote(branch, safe="")}',
|
|
token, deadline, request_attempts, timeout_sec, max_response_bytes,
|
|
)
|
|
commit = payload.get('commit')
|
|
if payload.get('name') != branch or not isinstance(commit, dict):
|
|
raise ApiRequestError('GitLab ref resolution returned a mismatched branch')
|
|
resolved = {
|
|
'provider': 'gitlab', 'repo_url': f'https://gitlab.com/{repo_path}.git',
|
|
'repo_path': str(repo_path), 'branch': branch, 'ref': f'refs/heads/{branch}',
|
|
'head_sha': str(commit.get('id') or '').lower(), 'ref_source': ref_source,
|
|
}
|
|
return validate_git_resolution(resolved)
|
|
except (RateLimitError, ApiRequestError) as exc:
|
|
sanitized = redacted_git_resolution_error(exc, token)
|
|
if sanitized is exc:
|
|
raise
|
|
raise sanitized from exc
|
|
|
|
|
|
def resolve_git_scan_target(
|
|
target, provider, token=None, *, request_attempts=2, timeout_sec=10,
|
|
max_response_bytes=1 << 20,
|
|
):
|
|
normalized, branch = normalize_git_scan_resolution_target(target, provider)
|
|
resolver = resolve_github_ref_head if normalized['provider'] == 'github' else resolve_gitlab_ref_head
|
|
return resolver(
|
|
normalized['repo_path'], token, branch or None, request_attempts=request_attempts,
|
|
timeout_sec=timeout_sec, max_response_bytes=max_response_bytes,
|
|
)
|
|
|
|
|
|
def validate_bound_git_scan_plan(plan, target, provider=None):
|
|
if not isinstance(plan, dict):
|
|
raise ValueError('exact Git scan requires a bound plan object')
|
|
required = {
|
|
'version', 'provider', 'repo_url', 'repo_path', 'branch', 'ref', 'head_sha',
|
|
'ref_source', 'base_sha', 'mode', 'baseline_depth',
|
|
}
|
|
if set(plan) != required or plan.get('version') != 1:
|
|
raise ValueError('bound Git scan plan has an unsupported shape')
|
|
resolution = validate_git_resolution(plan)
|
|
normalized, _ = normalize_git_scan_resolution_target(target, provider or resolution['provider'])
|
|
if normalized['repo_url'] != resolution['repo_url'] or normalized['repo_path'] != resolution['repo_path']:
|
|
raise ValueError('bound Git scan plan repository conflicts with the target')
|
|
mode = str(plan.get('mode') or '')
|
|
base_sha = plan.get('base_sha')
|
|
if base_sha is not None:
|
|
base_sha = str(base_sha).lower()
|
|
if not re.fullmatch(r'[a-f0-9]{40}|[a-f0-9]{64}', base_sha):
|
|
raise ValueError('bound Git scan plan has an invalid base SHA')
|
|
try:
|
|
baseline_depth = int(plan.get('baseline_depth'))
|
|
except (TypeError, ValueError) as exc:
|
|
raise ValueError('bound Git scan plan has an invalid baseline depth') from exc
|
|
if not 1 <= baseline_depth <= 1000000:
|
|
raise ValueError('bound Git scan plan baseline depth is out of range')
|
|
if (
|
|
(mode == 'baseline' and base_sha is not None)
|
|
or (mode == 'delta' and (not base_sha or base_sha == resolution['head_sha']))
|
|
or (mode == 'noop' and base_sha != resolution['head_sha'])
|
|
or mode not in ('baseline', 'delta', 'noop')
|
|
):
|
|
raise ValueError('bound Git scan plan mode and base are inconsistent')
|
|
normalized_plan = {
|
|
'version': 1, **resolution, 'base_sha': base_sha,
|
|
'mode': mode, 'baseline_depth': baseline_depth,
|
|
}
|
|
if canonical_git_scan_plan_bytes(normalized_plan) != canonical_git_scan_plan_bytes(plan):
|
|
raise ValueError('bound Git scan plan is not normalized')
|
|
return normalized_plan, hashlib.sha256(canonical_git_scan_plan_bytes(plan)).hexdigest()
|
|
|
|
|
|
def git_delta_base_unavailable(result, base_sha):
|
|
diagnostics = '\n'.join(str(item) for item in (
|
|
list(result.get('errors') or []) + list(result.get('warnings') or [])
|
|
)).lower()
|
|
if not diagnostics:
|
|
return False
|
|
base_markers = (
|
|
'bad object', 'unknown revision', 'invalid object', 'object not found',
|
|
'reference not found', 'could not find commit', 'unable to resolve commit',
|
|
'invalid since commit', 'since-commit', 'since commit',
|
|
)
|
|
return any(marker in diagnostics for marker in base_markers) and (
|
|
str(base_sha or '').lower()[:12] in diagnostics
|
|
or 'since' in diagnostics
|
|
or 'commit' in diagnostics
|
|
or 'revision' in diagnostics
|
|
or 'object' in diagnostics
|
|
)
|
|
|
|
|
|
def git_checkout_recovery_allowed(result):
|
|
meta = result.get('scan_meta') or {}
|
|
errors = result.get('errors') or []
|
|
if (
|
|
os.name != 'nt' or len(errors) not in (1, 2) or result.get('error_class') != 'trufflehog'
|
|
or result.get('source_failure') or result.get('warnings') or result.get('degraded') or result.get('skipped')
|
|
or meta.get('trufflehog_returncode') != 1 or meta.get('trufflehog_finished') is not False
|
|
or meta.get('diagnostic_output_limited') or meta.get('command_timed_out')
|
|
):
|
|
return False
|
|
limits = _trufflehog_diagnostic_limits()
|
|
companion = None
|
|
if len(errors) == 2:
|
|
companion_line = errors[0]
|
|
if (
|
|
not isinstance(companion_line, str)
|
|
or len(companion_line) > limits['line_chars']
|
|
or len(companion_line.encode('utf-8', errors='replace')) > limits['line_bytes']
|
|
):
|
|
return False
|
|
try:
|
|
companion = json.loads(companion_line)
|
|
except (TypeError, ValueError):
|
|
return False
|
|
if (
|
|
not isinstance(companion, dict)
|
|
or set(companion) != {
|
|
'level', 'ts', 'logger', 'msg', 'subcommand', 'repo', 'path',
|
|
'args', 'error',
|
|
}
|
|
or companion.get('level') != 'info-0'
|
|
or companion.get('logger') != 'trufflehog'
|
|
or companion.get('msg') != 'git clone failed'
|
|
or companion.get('subcommand') != 'git clone'
|
|
or companion.get('args') != []
|
|
or any(not isinstance(companion.get(key), str) or not companion.get(key) for key in (
|
|
'ts', 'repo', 'path', 'error',
|
|
))
|
|
):
|
|
return False
|
|
line = errors[-1]
|
|
if not isinstance(line, str) or len(line) > limits['line_chars'] or len(line.encode('utf-8', errors='replace')) > limits['line_bytes']:
|
|
return False
|
|
try:
|
|
payload = json.loads(line)
|
|
except (TypeError, ValueError):
|
|
return False
|
|
if (
|
|
not isinstance(payload, dict) or payload.get('msg') != 'error running scan'
|
|
or payload.get('level') != 'error' or payload.get('errors')
|
|
):
|
|
return False
|
|
detail = payload.get('error')
|
|
if not isinstance(detail, str):
|
|
return False
|
|
if companion is not None and companion['error'] not in detail:
|
|
return False
|
|
detail = detail.lower()
|
|
if not all(marker in detail for marker in (
|
|
'error preparing repo', 'error executing git clone: exit status 128',
|
|
'clone succeeded, but checkout failed',
|
|
)):
|
|
return False
|
|
prefix, _, git_stderr = detail.partition('error executing git clone: exit status 128')
|
|
if re.search(r'\b(?:fatal|error):', prefix):
|
|
return False
|
|
quoted_path = r"(?:'(?:[^'\\\r\n]|\\.)+'|\"(?:[^\"\\\r\n]|\\.)+\")"
|
|
path_failure = False
|
|
for physical_line in git_stderr.lstrip(' ,').splitlines():
|
|
line_match = re.match(r'^(?:remote:\s*)?(?:fatal|error):\s*(.*)$', physical_line.strip())
|
|
if not line_match:
|
|
if re.search(r'\b(?:fatal|error):', physical_line):
|
|
return False
|
|
continue
|
|
cause = line_match.group(1)
|
|
if re.fullmatch(r'invalid path ' + quoted_path, cause):
|
|
path_failure = True
|
|
continue
|
|
long_path = re.fullmatch(r'(?:unable to create file |cannot create directory (?:at )?)(.+): filename too long', cause)
|
|
if long_path:
|
|
path = long_path.group(1)
|
|
if path.startswith(("'", '"')):
|
|
if not re.fullmatch(quoted_path, path):
|
|
return False
|
|
elif re.search(r'\b(?:fatal|error):', path):
|
|
return False
|
|
path_failure = True
|
|
elif cause != 'unable to checkout working tree':
|
|
return False
|
|
return path_failure
|
|
|
|
|
|
def scan_exact_git_plan(
|
|
target, plan, plan_sha256, timeout_sec, detectors, exclude_detectors,
|
|
no_verification, trufflehog_config, token, external_trufflehog_lifecycle,
|
|
):
|
|
deadline = time.monotonic() + max(0.001, float(timeout_sec or 1))
|
|
provider = plan['provider']
|
|
scan_url, secrets_to_redact = build_authenticated_git_url(plan['repo_url'], provider, token)
|
|
command_env = os.environ.copy()
|
|
_append_windows_git_longpaths(command_env, 'Exact Git')
|
|
if secrets_to_redact:
|
|
command_env['TRUF_GIT_TOKEN'] = token
|
|
command_env['TRUF_GIT_USERNAME'] = 'oauth2' if provider == 'gitlab' else 'x-access-token'
|
|
recovery_root = None
|
|
local_url = None
|
|
checkout_errors = []
|
|
findings = []
|
|
cleanup_safe = True
|
|
recovery = {'attempted': False, 'clone_succeeded': False, 'coverage_complete': False}
|
|
|
|
def remaining():
|
|
seconds = deadline - time.monotonic()
|
|
if seconds <= 0:
|
|
raise subprocess.TimeoutExpired('exact Git scan', timeout_sec)
|
|
return seconds
|
|
|
|
def run_mode(mode):
|
|
nonlocal recovery_root, local_url
|
|
remaining()
|
|
cmd = [
|
|
get_trufflehog_cmd(), 'git', local_url or scan_url, '--json', '--no-update',
|
|
'--branch', plan['head_sha'],
|
|
]
|
|
if external_trufflehog_lifecycle:
|
|
cmd.append('--local-dev')
|
|
append_trufflehog_scan_args(
|
|
cmd, detectors, exclude_detectors, no_verification, trufflehog_config,
|
|
)
|
|
if mode in ('baseline', 'baseline_reset'):
|
|
cmd.extend(['--max-depth', str(plan['baseline_depth'])])
|
|
elif mode == 'delta':
|
|
cmd.extend(['--since-commit', plan['base_sha']])
|
|
result = {'findings': findings, 'errors': []}
|
|
emit_client_scan_phase('scanning', {
|
|
'integrated_operation': 'git_acquisition_and_scan',
|
|
'execution_mode': mode,
|
|
})
|
|
with run_command_streamed(
|
|
cmd, remaining(), command_env, deadline=deadline,
|
|
staging_roots=(recovery_root,) if recovery_root else None,
|
|
) as output:
|
|
apply_trufflehog_diagnostics(
|
|
result, output,
|
|
output.returncode, 'git', require_completion=True,
|
|
redactions=secrets_to_redact,
|
|
)
|
|
append_trufflehog_findings(
|
|
result, output.stdout_lines(redactions=secrets_to_redact),
|
|
)
|
|
checkout_candidate = not recovery['attempted'] and git_checkout_recovery_allowed(result)
|
|
if checkout_candidate:
|
|
checkout_errors.extend(result['errors'])
|
|
if checkout_candidate:
|
|
remaining()
|
|
recovery['attempted'] = True
|
|
emit_client_scan_phase('cloning', {
|
|
'operation': 'git_clone_recovery',
|
|
})
|
|
recovery_root = create_command_work_dir()
|
|
destination = os.path.join(recovery_root, 'repo')
|
|
clone_cmd = [get_git_cmd(), 'clone', '--no-checkout', '--no-recurse-submodules', '--', scan_url, destination]
|
|
clone_result = {'errors': []}
|
|
with run_command_streamed(
|
|
clone_cmd, remaining(), command_env, deadline=deadline,
|
|
staging_roots=(recovery_root,), native_git_clone=True,
|
|
) as output:
|
|
apply_trufflehog_diagnostics(
|
|
clone_result, output, output.returncode, 'git',
|
|
redactions=secrets_to_redact,
|
|
)
|
|
try:
|
|
for _ in output.stdout_lines(max_line_bytes=8192, max_lines=2000, redactions=secrets_to_redact):
|
|
pass
|
|
except CommandOutputLimitError:
|
|
clone_result.setdefault('errors', []).append('Git clone output exceeded its diagnostic bounds')
|
|
clone_result.update(error_class='output_limit', retryable=False)
|
|
if clone_result.get('errors') or clone_result.get('skipped') or clone_result.get('degraded'):
|
|
return clone_result
|
|
recovery['clone_succeeded'] = True
|
|
# Native Windows TH expects the drive in the file URI authority, not /C:/.
|
|
from pathlib import Path
|
|
local_url = Path(destination).as_uri()
|
|
if os.name == 'nt':
|
|
local_url = local_url.replace('file:///', 'file://', 1)
|
|
return run_mode(mode)
|
|
return result
|
|
|
|
execution_mode = plan['mode']
|
|
continuity_reset = False
|
|
result = {'findings': [], 'errors': []}
|
|
if execution_mode == 'noop':
|
|
emit_client_scan_phase('scanning', {
|
|
'operation': 'exact_git_noop',
|
|
'execution_mode': 'noop',
|
|
})
|
|
result = {'findings': [], 'errors': []}
|
|
else:
|
|
try:
|
|
result = run_mode(execution_mode)
|
|
if (
|
|
execution_mode == 'delta' and (not recovery['attempted'] or recovery['clone_succeeded'])
|
|
and git_delta_base_unavailable(result, plan['base_sha'])
|
|
):
|
|
execution_mode = 'baseline_reset'
|
|
continuity_reset = True
|
|
result = run_mode(execution_mode)
|
|
result.setdefault('scan_meta', {})['git_continuity_reset_reason'] = 'covered base unavailable'
|
|
except ScanSlotFatalError:
|
|
cleanup_safe = False
|
|
raise
|
|
except subprocess.TimeoutExpired:
|
|
result.setdefault('errors', []).append('Exact Git scan exhausted its absolute deadline')
|
|
result.update(error_class='timeout', retryable=True)
|
|
except Exception as exc:
|
|
message = redact_secrets(str(exc), [token])
|
|
logger.error('Error scanning pinned Git repository %s: %s', plan['repo_url'], message)
|
|
result.setdefault('errors', []).append(f'Scan failed: {message}')
|
|
result.update(retryable=True, error_class='remote_transient')
|
|
finally:
|
|
if recovery_root and cleanup_safe:
|
|
try:
|
|
cleanup_command_work_dir(recovery_root)
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except Exception as exc:
|
|
result.setdefault('errors', []).append('Git recovery cleanup failed: ' + redact_secrets(str(exc), [token]))
|
|
result.update(error_class='source_resource', retryable=True, source_failure=True,
|
|
source_failure_category='source_resource', source_failure_auth_related=False)
|
|
|
|
result['findings'] = findings
|
|
# Freeze scan coverage before optional filtering or candidate staging adds warnings.
|
|
success = not result.get('errors') and not result.get('skipped') and not result.get('degraded')
|
|
result['git_scan_plan'] = plan
|
|
result['git_scan_execution'] = {
|
|
'mode': execution_mode,
|
|
'pinned': True,
|
|
'success': bool(success),
|
|
'coverage_complete': bool(success),
|
|
'continuity_reset': continuity_reset,
|
|
'plan_sha256': plan_sha256,
|
|
}
|
|
result.setdefault('scan_meta', {})['exact_git_scope'] = {
|
|
'provider': plan['provider'], 'ref': plan['ref'], 'head_sha': plan['head_sha'],
|
|
'base_sha': plan['base_sha'], 'mode': execution_mode,
|
|
'baseline_depth': plan['baseline_depth'], 'ref_source': plan['ref_source'],
|
|
'pinned': True, 'continuity_reset': continuity_reset,
|
|
}
|
|
result = apply_finding_filters(result, target)
|
|
if execution_mode != 'noop' and time.monotonic() >= deadline:
|
|
if not result.get('errors'):
|
|
result.update(error_class='timeout', retryable=True)
|
|
result.setdefault('errors', []).append('Exact Git scan exceeded its absolute deadline including cleanup and filtering')
|
|
result.setdefault('scan_meta', {})['git_deadline_exceeded'] = True
|
|
result['git_scan_execution'].update(success=False, coverage_complete=False)
|
|
if result.get('errors'):
|
|
result['git_scan_execution'].update(success=False, coverage_complete=False)
|
|
if recovery['attempted']:
|
|
recovery['coverage_complete'] = result['git_scan_execution']['coverage_complete']
|
|
result.setdefault('scan_meta', {})['git_checkout_recovery'] = recovery
|
|
if checkout_errors and not result['git_scan_execution']['coverage_complete']:
|
|
result['errors'] = checkout_errors + list(result.get('errors') or [])
|
|
result.setdefault('error_class', 'trufflehog')
|
|
result.setdefault('retryable', True)
|
|
return result
|
|
|
|
|
|
def scan_git_repo(repo_url, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, provider=None, max_depth=None, max_commit_age_days=None, commit_lookup_pages=3, skip_if_commit_lookup_fails=True, external_trufflehog_lifecycle=False, git_plan=None):
|
|
"""Scan a single Git repository for secrets"""
|
|
target_info = parse_git_scan_target(repo_url)
|
|
original_target = repo_url
|
|
repo_url = target_info['url']
|
|
parsed_repo_url = urlsplit(repo_url)
|
|
if parsed_repo_url.username or parsed_repo_url.password:
|
|
return {"findings": [], "errors": ["Git target URL must not contain userinfo credentials"], "retryable": False, "error_class": "invalid_target"}
|
|
target_branch = target_info['branch']
|
|
target_metadata = target_info['metadata']
|
|
if git_plan is not None:
|
|
try:
|
|
plan, plan_sha256 = validate_bound_git_scan_plan(git_plan, original_target, provider)
|
|
except (TypeError, ValueError) as exc:
|
|
return {
|
|
'findings': [], 'errors': [f'Bound Git plan rejected: {exc}'],
|
|
'retryable': False, 'error_class': 'invalid_target',
|
|
}
|
|
return scan_exact_git_plan(
|
|
original_target, plan, plan_sha256, timeout_sec, detectors, exclude_detectors,
|
|
no_verification, trufflehog_config, token, external_trufflehog_lifecycle,
|
|
)
|
|
branch_label = f" branch={target_branch}" if target_branch else ''
|
|
logger.info(f"Scanning Git repository: {repo_url}{branch_label}")
|
|
|
|
emit_client_scan_phase('resolving', {
|
|
'operation': 'recent_commit_boundary',
|
|
'provider': str(provider or 'git'),
|
|
})
|
|
boundary = recent_commit_boundary(repo_url, provider, token, max_commit_age_days, commit_lookup_pages)
|
|
if boundary.get('error'):
|
|
category = str(boundary.get('error_category') or 'unknown')
|
|
auth_related = bool(boundary.get('auth_related'))
|
|
return {
|
|
"findings": [], "errors": [boundary.get('reason') or 'commit age lookup failed'],
|
|
"error_class": 'source_auth' if auth_related else 'remote_transient',
|
|
"retryable": True, "source_failure": True,
|
|
"source_failure_category": category,
|
|
"source_failure_auth_related": auth_related,
|
|
"scan_meta": {**boundary, 'target_metadata': target_metadata},
|
|
}
|
|
if boundary.get('skip'):
|
|
reason = boundary.get('reason', 'skipped by commit age filter')
|
|
logger.info(f"Skipping {repo_url}: {reason}")
|
|
if boundary.get('permanent') or skip_if_commit_lookup_fails:
|
|
return {"findings": [], "errors": [], "skipped": reason, "scan_meta": {**boundary, 'target_metadata': target_metadata}}
|
|
|
|
effective_provider, _ = get_git_provider_and_path(repo_url, provider)
|
|
scan_url, secrets_to_redact = build_authenticated_git_url(repo_url, effective_provider, token)
|
|
cmd = [get_trufflehog_cmd(), 'git', scan_url, '--json', '--no-update']
|
|
if external_trufflehog_lifecycle:
|
|
cmd.append('--local-dev')
|
|
append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config)
|
|
if max_depth:
|
|
cmd.extend(['--max-depth', str(max_depth)])
|
|
if target_branch:
|
|
cmd.extend(['--branch', target_branch])
|
|
if boundary.get('since_commit'):
|
|
cmd.extend(['--since-commit', boundary['since_commit']])
|
|
logger.info(
|
|
f"Scanning {repo_url} since commit {boundary['since_commit']} "
|
|
f"({boundary.get('recent_commit_count')} commits after {boundary.get('cutoff')})"
|
|
)
|
|
|
|
try:
|
|
command_env = os.environ.copy()
|
|
if secrets_to_redact:
|
|
command_env['TRUF_GIT_TOKEN'] = token
|
|
command_env['TRUF_GIT_USERNAME'] = 'oauth2' if effective_provider == 'gitlab' else 'x-access-token'
|
|
results = {"findings": [], "errors": []}
|
|
emit_client_scan_phase('scanning', {
|
|
'integrated_operation': 'git_acquisition_and_scan',
|
|
})
|
|
with run_command_streamed(cmd, timeout_sec, command_env) as output:
|
|
apply_trufflehog_diagnostics(
|
|
results, output,
|
|
output.returncode, 'git',
|
|
require_completion=external_trufflehog_lifecycle,
|
|
redactions=secrets_to_redact,
|
|
)
|
|
append_trufflehog_findings(
|
|
results, output.stdout_lines(redactions=secrets_to_redact),
|
|
)
|
|
|
|
if boundary.get('since_commit') or target_metadata or target_branch:
|
|
results.setdefault("scan_meta", {}).update({**boundary, 'branch': target_branch, 'target_metadata': target_metadata})
|
|
|
|
return apply_finding_filters(results, original_target)
|
|
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except Exception as e:
|
|
logger.error(f"Error scanning repository {repo_url}: {str(e)}")
|
|
return {"findings": [], "errors": [f"Scan failed: {str(e)}"]}
|
|
|
|
|
|
DOCKER_ARCHIVE_MAX_DECODED_BYTES = 1 << 30
|
|
|
|
|
|
def _require_docker_archive_policy(limits):
|
|
# This independent decoded-stream ceiling is part of docker-layer-execution-v4.
|
|
if limits['archive_max_size_bytes'] > DOCKER_ARCHIVE_MAX_DECODED_BYTES:
|
|
raise DockerLayerInfrastructureError(
|
|
'archive_policy_incompatible', 'Docker member policy exceeds the decoded validation ceiling',
|
|
category='source_configuration',
|
|
)
|
|
|
|
|
|
def validate_docker_content_artifact(
|
|
path, descriptor, *, deadline=None, max_member_bytes=256 << 20,
|
|
max_decoded_bytes=DOCKER_ARCHIVE_MAX_DECODED_BYTES, max_members=100000,
|
|
):
|
|
import zlib
|
|
|
|
deadline = min(float(deadline) if deadline is not None else float('inf'), time.monotonic() + 30)
|
|
if not math.isfinite(deadline) or any(
|
|
isinstance(value, bool) or not isinstance(value, int) or not 0 < value <= bound
|
|
for value, bound in ((max_member_bytes, DOCKER_ARCHIVE_MAX_DECODED_BYTES),
|
|
(max_decoded_bytes, DOCKER_ARCHIVE_MAX_DECODED_BYTES), (max_members, 100000))
|
|
):
|
|
raise DockerLayerInfrastructureError(
|
|
'archive_validation_bounds', 'Docker archive validation bounds are invalid', category='source_configuration',
|
|
)
|
|
|
|
def check_deadline():
|
|
_raise_if_scan_slot_fatal()
|
|
if time.monotonic() >= deadline:
|
|
raise DockerContentScanError('archive_timeout', 'Docker archive validation deadline expired', True)
|
|
|
|
check_deadline()
|
|
kind = str((descriptor or {}).get('kind') or '')
|
|
media_type = str((descriptor or {}).get('media_type') or '').strip().lower()
|
|
expected_size = (descriptor or {}).get('size')
|
|
try:
|
|
actual_size = os.path.getsize(path)
|
|
except OSError as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'private_storage', 'Docker content artifact is unavailable',
|
|
category='source_resource',
|
|
) from exc
|
|
if actual_size != expected_size:
|
|
raise DockerContentScanError(
|
|
'size_mismatch', 'Docker content artifact size changed after verification',
|
|
)
|
|
|
|
if kind == 'config':
|
|
if media_type not in DOCKER_CONFIG_MEDIA_TYPES:
|
|
raise DockerContentScanError(
|
|
'unsupported_media_type', 'Docker configuration media type is unsupported',
|
|
)
|
|
if actual_size > 64 << 20:
|
|
raise DockerContentScanError('archive_limit', 'Docker configuration exceeds the validation bound')
|
|
try:
|
|
with open(path, 'rb') as config_file:
|
|
raw_config = config_file.read(actual_size + 1)
|
|
except OSError as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'private_storage', 'Docker configuration artifact cannot be read',
|
|
category='source_resource',
|
|
) from exc
|
|
try:
|
|
config = json.loads(raw_config.decode('utf-8'))
|
|
except (RecursionError, UnicodeDecodeError, TypeError, ValueError) as exc:
|
|
raise DockerContentScanError(
|
|
'invalid_config_json', 'Docker configuration is not valid UTF-8 JSON',
|
|
) from exc
|
|
if not isinstance(config, dict):
|
|
raise DockerContentScanError(
|
|
'invalid_config_json', 'Docker configuration JSON must be an object',
|
|
)
|
|
check_deadline()
|
|
return config
|
|
|
|
if kind != 'layer' or media_type not in DOCKER_LAYER_MEDIA_TYPES:
|
|
raise DockerContentScanError(
|
|
'unsupported_media_type', 'Docker layer media type is unsupported',
|
|
)
|
|
zstd = None
|
|
if media_type.endswith('+zstd'):
|
|
try:
|
|
import zstandard as zstd
|
|
except ImportError as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'archive_decoder_unavailable', 'Docker zstd validation capability is unavailable',
|
|
category='source_configuration',
|
|
) from exc
|
|
decoded_bytes = 0
|
|
members = 0
|
|
terminated = False
|
|
|
|
class TimedInput:
|
|
def read(self, size=-1):
|
|
check_deadline()
|
|
data = layer_file.read(min(size if size >= 0 else 65536, 65536))
|
|
check_deadline()
|
|
return data
|
|
|
|
class BoundedReader:
|
|
def read(self, size=-1):
|
|
nonlocal decoded_bytes
|
|
check_deadline()
|
|
data = decoder.read(min(size if size >= 0 else 65536, 65536, max_decoded_bytes - decoded_bytes + 1))
|
|
decoded_bytes += len(data)
|
|
if decoded_bytes > max_decoded_bytes:
|
|
raise DockerContentScanError('archive_limit', 'Docker archive decoded byte bound exceeded')
|
|
check_deadline()
|
|
return data
|
|
|
|
class BoundedTarInfo(tarfile.TarInfo):
|
|
@classmethod
|
|
def fromtarfile(cls, archive):
|
|
nonlocal terminated
|
|
try:
|
|
check_deadline()
|
|
member = super().fromtarfile(archive)
|
|
check_deadline()
|
|
return member
|
|
except tarfile.EOFHeaderError:
|
|
terminated = True
|
|
raise
|
|
except tarfile.HeaderError as exc:
|
|
# TarFile.next otherwise tolerates some corrupt headers after member one.
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker tar header is invalid') from exc
|
|
|
|
def _proc_member(self, archive):
|
|
nonlocal members
|
|
check_deadline()
|
|
members += 1
|
|
metadata = self.type in (tarfile.XHDTYPE, tarfile.XGLTYPE, tarfile.SOLARIS_XHDTYPE,
|
|
tarfile.GNUTYPE_LONGNAME, tarfile.GNUTYPE_LONGLINK)
|
|
if self.size < 0:
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker archive member size is negative')
|
|
if members > max_members or self.size > min(max_member_bytes, 1 << 20 if metadata else max_member_bytes):
|
|
raise DockerContentScanError('archive_limit', 'Docker archive member bound exceeded')
|
|
if self.type == tarfile.GNUTYPE_SPARSE:
|
|
raise DockerContentScanError('unsupported_layer_archive', 'Docker sparse tar validation is unsupported')
|
|
member = super()._proc_member(archive)
|
|
check_deadline()
|
|
return member
|
|
|
|
def _proc_pax(self, archive):
|
|
# Do not enter tarfile's unbounded hdrcharset/length regexes or sparse
|
|
# map parsers. Validate complete records before decoding/applying fields.
|
|
if self.size > 64 << 10:
|
|
raise DockerContentScanError('archive_limit', 'Docker PAX parse byte bound exceeded')
|
|
body = archive.fileobj.read(self._block(self.size))
|
|
if len(body) != self._block(self.size):
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker PAX header is truncated')
|
|
body = body[:self.size]
|
|
records = []
|
|
position = 0
|
|
while position < len(body):
|
|
check_deadline()
|
|
if len(records) >= min(max_members, 1024):
|
|
raise DockerContentScanError('archive_limit', 'Docker PAX record bound exceeded')
|
|
space = body.find(b' ', position, min(position + 9, len(body)))
|
|
if space < 0:
|
|
raise DockerContentScanError('archive_limit', 'Docker PAX record length field exceeds its bound')
|
|
digits = body[position:space]
|
|
if not digits or not digits.isdigit():
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker PAX record length is invalid')
|
|
end = position + int(digits)
|
|
if end > len(body) or end <= space + 3 or body[end - 1:end] != b'\n':
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker PAX record boundary is invalid')
|
|
equals = body.find(b'=', space + 1, min(end - 1, space + 258))
|
|
if equals <= space + 1:
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker PAX keyword is invalid or oversized')
|
|
key, value = body[space + 1:equals], body[equals + 1:end - 1]
|
|
if key.startswith(b'GNU.sparse.'):
|
|
raise DockerContentScanError('unsupported_layer_archive', 'Docker sparse PAX validation is unsupported')
|
|
converter = tarfile.PAX_NUMBER_FIELDS.get(key.decode('utf-8'))
|
|
if converter is not None:
|
|
if len(value) > (32 if converter is int else 64):
|
|
raise DockerContentScanError('archive_limit', 'Docker PAX numeric field exceeds its bound')
|
|
number = converter(value)
|
|
if converter is float and not math.isfinite(number):
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker PAX numeric field is not finite')
|
|
if key == b'size' and not 0 <= number <= max_member_bytes:
|
|
raise DockerContentScanError('archive_limit', 'Docker PAX member size exceeds its bound')
|
|
records.append((key, value))
|
|
position = end
|
|
check_deadline()
|
|
pax_headers = archive.pax_headers.copy()
|
|
charset = next((value.decode('utf-8') for key, value in records if key == b'hdrcharset'),
|
|
pax_headers.get('hdrcharset'))
|
|
encoding = archive.encoding if charset == 'BINARY' else 'utf-8'
|
|
for key, value in records:
|
|
check_deadline()
|
|
key = self._decode_pax_field(key, 'utf-8', 'utf-8', archive.errors)
|
|
if key in tarfile.PAX_NAME_FIELDS:
|
|
value = self._decode_pax_field(value, encoding, archive.encoding, archive.errors)
|
|
else:
|
|
value = self._decode_pax_field(value, 'utf-8', 'utf-8', archive.errors)
|
|
pax_headers[key] = value
|
|
if len(pax_headers) > 1024:
|
|
raise DockerContentScanError('archive_limit', 'Docker global PAX field bound exceeded')
|
|
if self.type == tarfile.XGLTYPE:
|
|
archive.pax_headers = pax_headers
|
|
try:
|
|
member = self.fromtarfile(archive)
|
|
except tarfile.HeaderError as exc:
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker PAX header has no following member') from exc
|
|
if self.type in (tarfile.XHDTYPE, tarfile.SOLARIS_XHDTYPE):
|
|
check_deadline()
|
|
member._apply_pax_info(pax_headers, archive.encoding, archive.errors)
|
|
member.offset = self.offset
|
|
if 'size' in pax_headers:
|
|
archive.offset = member.offset_data
|
|
if member.isreg() or member.type not in tarfile.SUPPORTED_TYPES:
|
|
archive.offset += member._block(member.size)
|
|
check_deadline()
|
|
return member
|
|
|
|
try:
|
|
with open(path, 'rb') as layer_file:
|
|
magic = layer_file.read(4)
|
|
layer_file.seek(0)
|
|
if zstd is not None:
|
|
# stream_reader can silently accept a truncated final frame. Check physical
|
|
# frame/block boundaries separately, then let the decoder verify checksums.
|
|
frames = 0
|
|
while layer_file.tell() < actual_size:
|
|
check_deadline()
|
|
frames += 1
|
|
start = layer_file.tell()
|
|
header = layer_file.read(18)
|
|
if frames > max_members:
|
|
raise DockerContentScanError('archive_limit', 'Docker zstd frame bound exceeded')
|
|
if len(header) >= 8 and 0x184d2a50 <= int.from_bytes(header[:4], 'little') <= 0x184d2a5f:
|
|
end = start + 8 + int.from_bytes(header[4:8], 'little')
|
|
if end > actual_size:
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker zstd skippable frame is truncated')
|
|
layer_file.seek(end)
|
|
continue
|
|
if header[:4] != b'\x28\xb5\x2f\xfd':
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker zstd frame header is invalid')
|
|
header_size = zstd.frame_header_size(header)
|
|
params = zstd.get_frame_parameters(header)
|
|
if params.window_size > max_decoded_bytes or (
|
|
params.content_size != zstd.CONTENTSIZE_UNKNOWN and params.content_size > max_decoded_bytes
|
|
):
|
|
raise DockerContentScanError('archive_limit', 'Docker zstd window bound exceeded')
|
|
layer_file.seek(start + header_size)
|
|
while True:
|
|
check_deadline()
|
|
block = layer_file.read(3)
|
|
if len(block) != 3:
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker zstd frame is truncated')
|
|
value = int.from_bytes(block, 'little')
|
|
block_type, block_size = (value >> 1) & 3, value >> 3
|
|
if block_type == 3 or block_size > 128 << 10:
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker zstd block is invalid')
|
|
end = layer_file.tell() + (1 if block_type == 1 else block_size)
|
|
if value & 1 and params.has_checksum:
|
|
end += 4
|
|
if end > actual_size:
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker zstd frame is truncated')
|
|
layer_file.seek(end)
|
|
if value & 1:
|
|
break
|
|
layer_file.seek(0)
|
|
decoder = zstd.ZstdDecompressor(max_window_size=max(1024, max_decoded_bytes)).stream_reader(
|
|
TimedInput(), read_across_frames=True, closefd=False,
|
|
)
|
|
elif media_type in DOCKER_LAYER_GZIP_MEDIA_TYPES:
|
|
if not magic.startswith(b'\x1f\x8b'):
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker layer does not match its gzip media type')
|
|
decoder = gzip.GzipFile(fileobj=TimedInput())
|
|
else:
|
|
decoder = layer_file
|
|
try:
|
|
with tarfile.open(fileobj=BoundedReader(), mode='r|', tarinfo=BoundedTarInfo) as archive:
|
|
for member in archive:
|
|
if member.size > max_member_bytes:
|
|
raise DockerContentScanError('archive_limit', 'Docker archive member bound exceeded')
|
|
if member.sparse is not None:
|
|
raise DockerContentScanError('unsupported_layer_archive', 'Docker sparse tar validation is unsupported')
|
|
if member.isfile():
|
|
with archive.extractfile(member) as body:
|
|
while body.read(65536):
|
|
check_deadline()
|
|
if not terminated or archive.fileobj.read(512) != b'\0' * 512:
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker tar end marker is missing')
|
|
while True:
|
|
padding = archive.fileobj.read(65536)
|
|
if not padding:
|
|
break
|
|
if padding.strip(b'\0'):
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker tar has trailing non-padding content')
|
|
if decoded_bytes % 512:
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker tar padding is truncated')
|
|
finally:
|
|
if decoder is not layer_file:
|
|
decoder.close()
|
|
except DockerContentScanError:
|
|
raise
|
|
except (tarfile.TarError, EOFError, gzip.BadGzipFile, zlib.error, ValueError, RecursionError) as exc:
|
|
raise DockerContentScanError(
|
|
'invalid_layer_archive', 'Docker layer is not a valid bounded tar archive',
|
|
) from exc
|
|
except OSError as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'private_storage', 'Docker layer artifact cannot be read',
|
|
category='source_resource',
|
|
) from exc
|
|
except Exception as exc:
|
|
if zstd is not None and isinstance(exc, zstd.ZstdError):
|
|
raise DockerContentScanError('invalid_layer_archive', 'Docker zstd stream is invalid') from exc
|
|
raise
|
|
|
|
|
|
def attach_docker_content_provenance(
|
|
findings, plan, descriptor, positions, private_blob_path,
|
|
):
|
|
location = (
|
|
f'docker://{plan["repository"]}@{plan["manifest_digest"]}/'
|
|
f'{descriptor["kind"]}/{descriptor["digest"]}'
|
|
)
|
|
private_blob_path = os.path.normcase(os.path.abspath(private_blob_path))
|
|
for finding in findings:
|
|
if not isinstance(finding, dict):
|
|
continue
|
|
source = finding.setdefault('SourceMetadata', {})
|
|
data = source.setdefault('Data', {}) if isinstance(source, dict) else {}
|
|
filesystem = data.get('Filesystem') if isinstance(data, dict) else None
|
|
if isinstance(filesystem, dict):
|
|
original = str(filesystem.get('file') or '')
|
|
if private_blob_path and private_blob_path in os.path.normcase(original):
|
|
original = original[len(private_blob_path):].lstrip('/\\:')
|
|
filesystem['file'] = location + (f'/{original}' if original else '')
|
|
if isinstance(data, dict):
|
|
docker_content = {
|
|
'image': plan['image'],
|
|
'manifest_digest': plan['manifest_digest'],
|
|
'blob_digest': descriptor['digest'],
|
|
'descriptor_kind': descriptor['kind'],
|
|
'positions': list(positions),
|
|
}
|
|
if plan['version'] == 2:
|
|
docker_content['payload_class'] = descriptor['payload_class']
|
|
data['DockerContent'] = docker_content
|
|
return findings
|
|
|
|
|
|
def _docker_content_error_code(value, fallback='scan_failed'):
|
|
text = re.sub(r'[^a-z0-9_]+', '_', str(value or '').strip().lower()).strip('_')
|
|
return text[:64] if re.fullmatch(r'[a-z][a-z0-9_]{0,63}', text) else fallback
|
|
|
|
|
|
def _scan_docker_content_file(
|
|
destination, descriptor, limits, deadline, detectors=None,
|
|
exclude_detectors=None, no_verification=False, trufflehog_config=None,
|
|
):
|
|
_require_docker_archive_policy(limits)
|
|
validate_docker_content_artifact(
|
|
destination, descriptor,
|
|
deadline=min(deadline, time.monotonic() + limits['archive_timeout_sec']),
|
|
max_member_bytes=limits['archive_max_size_bytes'],
|
|
)
|
|
remaining = deadline - time.monotonic()
|
|
if remaining <= 0:
|
|
raise DockerContentScanError('archive_timeout', 'Docker blob deadline expired before scanning', True)
|
|
command = [
|
|
get_trufflehog_cmd(), 'filesystem', destination, '--json', '--no-update',
|
|
'--log-level', '2',
|
|
'--archive-max-size', f'{limits["archive_max_size_bytes"]}B',
|
|
'--archive-max-depth', str(limits['archive_max_depth']),
|
|
'--archive-timeout', f'{limits["archive_timeout_sec"]}s',
|
|
'--concurrency', str(limits['filesystem_concurrency']),
|
|
]
|
|
append_trufflehog_scan_args(command, detectors, exclude_detectors, no_verification, trufflehog_config)
|
|
result = {'findings': [], 'errors': []}
|
|
with run_command_streamed(
|
|
command, remaining, os.environ.copy(), deadline=deadline,
|
|
staging_roots=(os.path.dirname(os.path.abspath(destination)),),
|
|
) as output:
|
|
apply_trufflehog_diagnostics(
|
|
result, output, output.returncode, 'filesystem', require_completion=True,
|
|
)
|
|
append_trufflehog_findings(result, output.stdout_lines())
|
|
# Only wrapper-owned diagnostics can attribute a watchdog stop to staging.
|
|
staging_error = output.synthetic_stderr.removeprefix('Error: ').split(';', 1)[0]
|
|
if staging_error in ('TruffleHog staging limit exceeded', 'Unable to monitor TruffleHog staging'):
|
|
monitor_failed = staging_error == 'Unable to monitor TruffleHog staging'
|
|
result.update(
|
|
errors=[staging_error],
|
|
error_class='source_resource' if monitor_failed else 'staging_limit',
|
|
retryable=monitor_failed,
|
|
source_failure=monitor_failed,
|
|
source_failure_auth_related=False,
|
|
)
|
|
result.pop('source_failure_category', None)
|
|
if monitor_failed:
|
|
result['source_failure_category'] = 'source_resource'
|
|
return result
|
|
if time.monotonic() >= deadline:
|
|
result['errors'].append('Docker blob scan exceeded its absolute deadline')
|
|
result['error_class'] = 'timeout'
|
|
result['retryable'] = True
|
|
return result
|
|
|
|
|
|
def scan_docker_layer_plan(
|
|
image_name, docker_layer_work, timeout_sec=600, detectors=None,
|
|
exclude_detectors=None, no_verification=False, trufflehog_config=None,
|
|
*, log_target=True,
|
|
):
|
|
try:
|
|
plan = validate_docker_layer_plan((docker_layer_work or {}).get('plan'))
|
|
plan_bytes = canonical_docker_layer_plan_bytes(plan)
|
|
plan_sha256 = hashlib.sha256(plan_bytes).hexdigest()
|
|
if plan_sha256 != str((docker_layer_work or {}).get('plan_sha256') or ''):
|
|
raise ValueError('Docker layer work plan hash is invalid')
|
|
if parse_docker_target(image_name)['image'].lower() != plan['image']:
|
|
raise ValueError('Docker layer work target does not match its plan')
|
|
except (TypeError, ValueError) as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'invalid_bound_plan', 'Docker layer bound plan is invalid',
|
|
category='source_configuration',
|
|
) from exc
|
|
|
|
_require_docker_archive_policy(plan['limits'])
|
|
leased_by_digest = {}
|
|
for descriptor in plan['descriptors']:
|
|
if descriptor['coverage_state'] == 'leased':
|
|
entry = leased_by_digest.setdefault(descriptor['digest'], {
|
|
'descriptor': descriptor, 'positions': [],
|
|
})
|
|
entry['positions'].append(descriptor['position'])
|
|
|
|
work_root = None
|
|
retain_work = False
|
|
created_paths = []
|
|
findings = []
|
|
finding_digests = {}
|
|
records = []
|
|
failures = []
|
|
bearer_auth = (docker_layer_work or {}).get('bearer_auth')
|
|
min_free_bytes = max(0, int((docker_layer_work or {}).get('min_free_bytes') or 0))
|
|
execution_deadline = time.monotonic() + max(0.001, float(timeout_sec or 0.001))
|
|
supplied_deadline = (docker_layer_work or {}).get('deadline')
|
|
if supplied_deadline is not None:
|
|
try:
|
|
supplied_deadline = float(supplied_deadline)
|
|
except (TypeError, ValueError) as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'invalid_deadline', 'Docker layer execution deadline is invalid',
|
|
category='source_configuration',
|
|
) from exc
|
|
if not math.isfinite(supplied_deadline):
|
|
raise DockerLayerInfrastructureError(
|
|
'invalid_deadline', 'Docker layer execution deadline is invalid',
|
|
category='source_configuration',
|
|
)
|
|
execution_deadline = min(execution_deadline, supplied_deadline)
|
|
try:
|
|
if leased_by_digest:
|
|
if time.monotonic() >= execution_deadline:
|
|
raise DockerLayerInfrastructureError(
|
|
'execution_deadline', 'Docker layer execution deadline expired before work began',
|
|
category='remote_transient',
|
|
)
|
|
try:
|
|
work_root = tempfile.mkdtemp(prefix='docker-layer-', dir=get_work_dir())
|
|
harden_private_directory(work_root)
|
|
write_temp_owner(work_root, ['docker-layer-content'], os.getpid(), required=True)
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except (OSError, RuntimeError, ValueError) as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'private_storage', 'Docker layer private work storage is unavailable',
|
|
category='source_resource',
|
|
) from exc
|
|
for digest, entry in leased_by_digest.items():
|
|
descriptor = entry['descriptor']
|
|
blob_started = time.monotonic()
|
|
blob_deadline = min(
|
|
execution_deadline,
|
|
blob_started + plan['limits']['blob_timeout_sec'],
|
|
)
|
|
destination = os.path.join(work_root, f'blob-{len(created_paths):04d}')
|
|
verified_bytes = 0
|
|
transfer_bytes = 0
|
|
transfer_duration_ms = 0
|
|
scan_duration_ms = 0
|
|
blob_findings = []
|
|
error_code = ''
|
|
blob_retryable = True
|
|
try:
|
|
if not records:
|
|
emit_client_scan_phase('downloading', {
|
|
'operation': 'docker_blob_transfer',
|
|
})
|
|
else:
|
|
emit_client_scan_phase('downloading', {
|
|
'operation': 'additional_docker_blob_transfer',
|
|
})
|
|
outcome = stream_docker_registry_blob(
|
|
plan['repository'], descriptor, destination, bearer_auth,
|
|
deadline=blob_deadline, min_free_bytes=min_free_bytes,
|
|
)
|
|
bearer_auth = outcome.bearer_auth
|
|
created_paths.append(destination)
|
|
verified_bytes = outcome.verified_bytes
|
|
transfer_bytes = outcome.transfer_bytes
|
|
transfer_duration_ms = outcome.duration_ms
|
|
scan_started = time.monotonic()
|
|
emit_client_scan_phase('scanning', {
|
|
'operation': 'docker_blob_scan',
|
|
})
|
|
result = _scan_docker_content_file(
|
|
destination, descriptor, plan['limits'], blob_deadline,
|
|
detectors, exclude_detectors, no_verification, trufflehog_config,
|
|
)
|
|
scan_duration_ms = max(0, int((time.monotonic() - scan_started) * 1000))
|
|
if time.monotonic() >= blob_deadline and not result.get('errors'):
|
|
result['errors'] = ['Docker layer scan exceeded its absolute deadline']
|
|
result['error_class'] = 'timeout'
|
|
result['retryable'] = True
|
|
blob_findings = list(result.get('findings') or [])
|
|
attach_docker_content_provenance(
|
|
blob_findings, plan, descriptor, entry['positions'], destination,
|
|
)
|
|
finding_digests.update(
|
|
(id(finding), digest)
|
|
for finding in blob_findings if isinstance(finding, dict)
|
|
)
|
|
if result.get('source_failure'):
|
|
raise DockerLayerInfrastructureError(
|
|
_docker_content_error_code(
|
|
result.get('source_failure_category'), 'scanner_infrastructure',
|
|
),
|
|
'Docker layer scanner infrastructure is unavailable',
|
|
category=str(
|
|
result.get('source_failure_category') or 'source_resource'
|
|
),
|
|
auth_related=bool(result.get('source_failure_auth_related')),
|
|
)
|
|
if (
|
|
result.get('errors') or result.get('skipped')
|
|
or result.get('warnings') or result.get('degraded')
|
|
):
|
|
diagnostic_code = result.get('error_class')
|
|
if not diagnostic_code and result.get('warning_classes'):
|
|
diagnostic_code = result['warning_classes'][0]
|
|
error_code = _docker_content_error_code(
|
|
diagnostic_code,
|
|
'scan_incomplete' if result.get('warnings') else 'scan_failed',
|
|
)
|
|
blob_retryable = bool(
|
|
result.get('retryable', not bool(result.get('skipped')))
|
|
)
|
|
except DockerLayerInfrastructureError:
|
|
raise
|
|
except DockerContentTransferError as exc:
|
|
error_code = _docker_content_error_code(exc.error_code, 'transfer_failed')
|
|
blob_retryable = exc.retryable
|
|
transfer_bytes = max(transfer_bytes, int(exc.transfer_bytes or 0))
|
|
transfer_duration_ms = max(
|
|
transfer_duration_ms, int(exc.duration_ms or 0),
|
|
)
|
|
except DockerContentScanError as exc:
|
|
error_code = _docker_content_error_code(exc.error_code, 'invalid_content')
|
|
blob_retryable = exc.retryable
|
|
except ScanSlotFatalError:
|
|
retain_work = True
|
|
raise
|
|
except Exception as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'scanner_infrastructure', 'Docker layer scanner infrastructure failed',
|
|
category='source_resource',
|
|
) from exc
|
|
finally:
|
|
if not retain_work and destination in created_paths:
|
|
try:
|
|
durable_unlink(destination)
|
|
except OSError as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'private_cleanup', 'Docker layer artifact cleanup failed',
|
|
category='source_resource',
|
|
) from exc
|
|
created_paths.remove(destination)
|
|
|
|
terminal = (
|
|
not blob_retryable
|
|
or int(descriptor['attempt']) >= int(descriptor['max_attempts'])
|
|
)
|
|
status = (
|
|
'covered' if not error_code
|
|
else 'terminal_failed' if terminal
|
|
else 'retryable_failed'
|
|
)
|
|
records.append({
|
|
'digest': digest,
|
|
'lease_token': descriptor['lease_token'],
|
|
'status': status,
|
|
'verified_bytes': verified_bytes,
|
|
'transfer_bytes': transfer_bytes,
|
|
'transfer_duration_ms': transfer_duration_ms,
|
|
'scan_duration_ms': scan_duration_ms,
|
|
'finding_count': len(blob_findings),
|
|
'error_code': error_code or None,
|
|
})
|
|
findings.extend(blob_findings)
|
|
if error_code:
|
|
failures.append(f'Docker content {digest[:19]} failed: {error_code}')
|
|
except ScanSlotFatalError:
|
|
retain_work = True
|
|
raise
|
|
finally:
|
|
# Fatal process outcomes leave payloads and ownership evidence for the janitor.
|
|
if not retain_work:
|
|
for path in created_paths:
|
|
if os.path.lexists(path):
|
|
try:
|
|
durable_unlink(path)
|
|
except OSError as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'private_cleanup', 'Docker layer artifact cleanup failed',
|
|
category='source_resource',
|
|
) from exc
|
|
if work_root:
|
|
cleanup_command_work_dir(work_root)
|
|
|
|
records_by_digest = {item['digest']: item for item in records}
|
|
if set(records_by_digest) != set(leased_by_digest):
|
|
raise DockerLayerInfrastructureError(
|
|
'missing_execution', 'Docker layer execution metadata is incomplete',
|
|
category='source_resource',
|
|
)
|
|
effective_descriptors = []
|
|
for descriptor in plan['descriptors']:
|
|
effective_state = descriptor['coverage_state']
|
|
if effective_state == 'leased':
|
|
effective_state = records_by_digest[descriptor['digest']]['status']
|
|
effective_descriptors.append((descriptor, effective_state))
|
|
|
|
static_states = {item['coverage_state'] for item in plan['descriptors']}
|
|
if 'selected' in static_states:
|
|
failures.append('Docker content remains selected for the next durable checkpoint')
|
|
if 'shared_pending' in static_states:
|
|
failures.append('Docker content is pending under another fenced reservation')
|
|
result_states = {state for _, state in effective_descriptors}
|
|
has_retryable = bool(
|
|
result_states & {'selected', 'shared_pending', 'retryable_failed'}
|
|
)
|
|
has_terminal = 'terminal_failed' in result_states
|
|
if has_terminal:
|
|
failures.append('Docker content exhausted its bounded attempt budget')
|
|
result = {
|
|
'findings': findings,
|
|
'errors': failures,
|
|
'retryable': bool(has_retryable),
|
|
'error_class': ('docker_content_retry' if has_retryable else 'docker_content_terminal')
|
|
if failures else None,
|
|
'docker_layer_plan': plan,
|
|
'docker_layer_execution': {
|
|
'version': plan['version'],
|
|
'plan_sha256': plan_sha256,
|
|
'blobs': records,
|
|
},
|
|
'scan_meta': {
|
|
'docker_layer_scope': {
|
|
'manifest_digest': plan['manifest_digest'],
|
|
'coverage_complete': all(
|
|
state == 'covered' for _, state in effective_descriptors
|
|
),
|
|
'selected_descriptors': sum(
|
|
1 for item in plan['descriptors'] if item['selected']
|
|
),
|
|
'leased_blobs': len(leased_by_digest),
|
|
'covered_blobs': len({
|
|
item['digest'] for item, state in effective_descriptors
|
|
if state == 'covered'
|
|
}),
|
|
'newly_covered_blobs': sum(
|
|
1 for item in records if item['status'] == 'covered'
|
|
),
|
|
'globally_reused_blobs': len({
|
|
item['digest'] for item in plan['descriptors']
|
|
if item['coverage_state'] == 'covered'
|
|
}),
|
|
'retryable_failed_blobs': sum(
|
|
1 for item in records if item['status'] == 'retryable_failed'
|
|
),
|
|
'terminal_failed_blobs': sum(
|
|
1 for item in records if item['status'] == 'terminal_failed'
|
|
),
|
|
'pending_checkpoint_blobs': sum(
|
|
1 for item in plan['descriptors']
|
|
if item['coverage_state'] == 'selected'
|
|
),
|
|
'shared_pending_blobs': sum(
|
|
1 for item in plan['descriptors']
|
|
if item['coverage_state'] == 'shared_pending'
|
|
),
|
|
'selected_bytes': sum(
|
|
item['size'] for item in plan['descriptors'] if item['selected']
|
|
),
|
|
'covered_bytes': sum(
|
|
item['size'] for item, state in effective_descriptors
|
|
if state == 'covered'
|
|
),
|
|
'skipped_bytes': sum(
|
|
item['size'] for item, state in effective_descriptors
|
|
if state == 'skipped'
|
|
),
|
|
'shared_pending_bytes': sum(
|
|
item['size'] for item, state in effective_descriptors
|
|
if state == 'shared_pending'
|
|
),
|
|
'retryable_failed_bytes': sum(
|
|
item['size'] for item, state in effective_descriptors
|
|
if state == 'retryable_failed'
|
|
),
|
|
'terminal_failed_bytes': sum(
|
|
item['size'] for item, state in effective_descriptors
|
|
if state == 'terminal_failed'
|
|
),
|
|
'transfer_bytes': sum(item['transfer_bytes'] for item in records),
|
|
'transfer_duration_ms': sum(
|
|
item['transfer_duration_ms'] for item in records
|
|
),
|
|
'scan_duration_ms': sum(item['scan_duration_ms'] for item in records),
|
|
'timeout_blobs': sum(
|
|
1 for item in records
|
|
if item['error_code'] in ('timeout', 'transfer_timeout')
|
|
),
|
|
'skipped_descriptors': sum(
|
|
1 for item in plan['descriptors'] if item['coverage_state'] == 'skipped'
|
|
),
|
|
},
|
|
},
|
|
}
|
|
if not failures:
|
|
result.pop('error_class')
|
|
result.pop('retryable')
|
|
if 'skipped' in static_states:
|
|
result['degraded'] = True
|
|
result['warnings'] = ['Docker content plan intentionally skipped bounded descriptors']
|
|
result['warning_classes'] = ['docker_content_budget']
|
|
try:
|
|
result = apply_finding_filters(
|
|
result, plan['image'], log_target=log_target,
|
|
)
|
|
except Exception as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'result_filter', 'Docker layer result filtering failed',
|
|
category='source_configuration',
|
|
) from exc
|
|
filtered_counts = Counter(
|
|
finding_digests.get(id(finding))
|
|
for finding in result.get('findings', [])
|
|
if finding_digests.get(id(finding))
|
|
)
|
|
for record in result['docker_layer_execution']['blobs']:
|
|
record['finding_count'] = filtered_counts[record['digest']]
|
|
return result
|
|
|
|
|
|
def _docker_implicit_auth_present():
|
|
if os.environ.get('DOCKER_TOKEN') or os.environ.get('REGISTRY_AUTH_FILE') or docker_token_manager.has_accounts():
|
|
return True
|
|
homes = {os.environ.get('HOME'), os.environ.get('USERPROFILE'), os.path.expanduser('~')}
|
|
if os.environ.get('HOMEDRIVE') and os.environ.get('HOMEPATH'):
|
|
homes.add(os.environ['HOMEDRIVE'] + os.environ['HOMEPATH'])
|
|
paths = [os.path.join(home, '.docker', 'config.json') for home in homes if home and home != '~']
|
|
xdg = os.environ.get('XDG_RUNTIME_DIR', '')
|
|
if xdg and not os.path.isabs(xdg):
|
|
return True
|
|
paths.append(os.path.join(xdg, 'containers', 'auth.json'))
|
|
for path in paths:
|
|
try:
|
|
os.lstat(path)
|
|
return True
|
|
except (FileNotFoundError, NotADirectoryError):
|
|
continue
|
|
except OSError:
|
|
return True
|
|
return False
|
|
|
|
|
|
def _recover_docker_image_contents(
|
|
image_ref, deadline, config_dir, limits, min_free_bytes,
|
|
detectors, exclude_detectors, no_verification, trufflehog_config, *,
|
|
implicit_auth_unsupported=False, anonymous_public_client=False,
|
|
):
|
|
from scanner_db import validate_docker_layer_limits
|
|
|
|
result = {'findings': [], 'errors': []}
|
|
scope = {'coverage_complete': False, 'scanned_descriptors': 0, 'blob_transfer_attempted': False}
|
|
result['scan_meta'] = {'docker_full_recovery': scope}
|
|
phase = 'configuration'
|
|
diagnostic = {}
|
|
descriptor = None
|
|
work_root = None
|
|
retain_work = False
|
|
try:
|
|
limits = validate_docker_layer_limits(limits if limits is not None else {
|
|
'config_max_bytes': 1 << 20, 'layer_max_bytes': 256 << 20,
|
|
'image_max_bytes': 1 << 30, 'max_layers': 8,
|
|
'archive_max_size_bytes': 256 << 20, 'archive_max_depth': 4,
|
|
'archive_timeout_sec': 30, 'blob_timeout_sec': 600,
|
|
'filesystem_concurrency': 2, 'blob_max_attempts': 3,
|
|
})
|
|
_require_docker_archive_policy(limits)
|
|
phase = 'preflight'
|
|
if time.monotonic() >= deadline:
|
|
raise DockerContentScanError('timeout', 'Docker recovery deadline expired', True)
|
|
try:
|
|
_, _, repository, digest = _dockerhub_manifest_target_parts(image_ref)
|
|
except (TypeError, ValueError) as exc:
|
|
raise DockerContentScanError('unsupported_recovery_target', 'Docker recovery requires an immutable Docker Hub target') from exc
|
|
phase = 'authentication'
|
|
bearer_auth = None
|
|
if anonymous_public_client:
|
|
if config_dir:
|
|
raise DockerLayerInfrastructureError(
|
|
'recovery_auth_unavailable',
|
|
'Anonymous Docker recovery received credential configuration',
|
|
category='source_configuration',
|
|
)
|
|
elif config_dir:
|
|
# Only the already-managed credential pool is trusted. Never load an arbitrary
|
|
# Docker config or run its external credential helpers in the recovery path.
|
|
with docker_token_manager.lock:
|
|
matched = next((account for account in docker_token_manager.accounts if
|
|
os.path.normcase(os.path.abspath(account.config_dir)) == os.path.normcase(os.path.abspath(config_dir))), None)
|
|
if matched is None:
|
|
raise DockerLayerInfrastructureError(
|
|
'recovery_auth_unavailable', 'Docker recovery cannot use this credential configuration',
|
|
category='source_configuration',
|
|
)
|
|
excluded = {account.name for account in docker_token_manager.accounts if account.name != matched.name}
|
|
bearer_auth = docker_registry_bearer_token(
|
|
'Bearer realm="https://auth.docker.io/token",service="registry.docker.io"',
|
|
repository, excluded_accounts=excluded, deadline=deadline,
|
|
)
|
|
else:
|
|
# The native default keychain may use Docker/Podman configs without
|
|
# DOCKER_CONFIG. Inspect existence only; never read or invoke helpers.
|
|
if implicit_auth_unsupported or _docker_implicit_auth_present():
|
|
raise DockerLayerInfrastructureError(
|
|
'recovery_auth_unavailable', 'Docker recovery cannot map implicit credential identity',
|
|
category='source_configuration',
|
|
)
|
|
phase = 'manifest_resolution'
|
|
emit_client_scan_phase('resolving', {
|
|
'operation': 'docker_manifest_resolution_recovery',
|
|
})
|
|
resolved, bearer_auth = resolve_docker_content_manifest(
|
|
image_ref, bearer_auth=bearer_auth, deadline=deadline,
|
|
anonymous_only=anonymous_public_client,
|
|
)
|
|
phase = 'preflight'
|
|
if resolved['image'] != image_ref or resolved['manifest_digest'] != digest or resolved['repository'] != repository:
|
|
raise DockerContentScanError('recovery_identity_mismatch', 'Docker recovery manifest identity changed')
|
|
descriptors = [dict(resolved['config'], kind='config', position=0)] + [
|
|
dict(item, kind='layer', position=index) for index, item in enumerate(resolved['layers'], 1)
|
|
]
|
|
scope['descriptor_count'] = len(descriptors)
|
|
if len(resolved['layers']) > limits['max_layers']:
|
|
diagnostic.update(reason='count_bound', observed=len(resolved['layers']), limit=limits['max_layers'])
|
|
raise DockerContentScanError('recovery_budget', 'All Docker layers do not fit the recovery count bound')
|
|
unique = {}
|
|
for descriptor in descriptors:
|
|
kind, media, size = descriptor['kind'], descriptor.get('media_type'), descriptor['size']
|
|
allowed = DOCKER_CONFIG_MEDIA_TYPES if kind == 'config' else DOCKER_LAYER_MEDIA_TYPES
|
|
if not isinstance(media, str) or media not in allowed:
|
|
# Reporting only: these official types do not expand recovery support.
|
|
known_media = DOCKER_CONFIG_MEDIA_TYPES | DOCKER_LAYER_MEDIA_TYPES | {
|
|
'application/vnd.oci.image.layer.nondistributable.v1.tar',
|
|
'application/vnd.oci.image.layer.nondistributable.v1.tar+gzip',
|
|
'application/vnd.oci.image.layer.nondistributable.v1.tar+zstd',
|
|
'application/vnd.docker.image.rootfs.foreign.diff.tar',
|
|
'application/vnd.docker.image.rootfs.foreign.diff.tar.gzip',
|
|
'application/vnd.oci.image.manifest.v1+json',
|
|
'application/vnd.oci.image.index.v1+json',
|
|
'application/vnd.docker.distribution.manifest.v1+json',
|
|
'application/vnd.docker.distribution.manifest.v1+prettyjws',
|
|
'application/vnd.docker.distribution.manifest.v2+json',
|
|
'application/vnd.docker.distribution.manifest.list.v2+json',
|
|
}
|
|
diagnostic.update(
|
|
reason='media_type',
|
|
media_type=media if isinstance(media, str) and media in known_media else 'other',
|
|
)
|
|
raise DockerContentScanError('unsupported_media_type', 'Docker recovery descriptor is unsupported')
|
|
if not normalize_docker_digest(descriptor['digest']):
|
|
diagnostic['reason'] = 'integrity'
|
|
raise DockerContentScanError('unsupported_media_type', 'Docker recovery descriptor is unsupported')
|
|
if isinstance(size, bool) or not isinstance(size, int) or size < 0:
|
|
raise DockerContentScanError('invalid_descriptor', 'Docker recovery descriptor size is invalid')
|
|
byte_limit = limits['config_max_bytes' if kind == 'config' else 'layer_max_bytes']
|
|
if size > byte_limit:
|
|
diagnostic.update(reason='descriptor_byte_bound', observed=size, limit=byte_limit)
|
|
raise DockerContentScanError('recovery_budget', 'All Docker content does not fit the recovery byte bound')
|
|
entry = unique.setdefault(descriptor['digest'], {'descriptor': descriptor, 'positions': []})
|
|
previous = entry['descriptor']
|
|
if any(previous[key] != descriptor[key] for key in ('kind', 'size', 'media_type')):
|
|
raise DockerContentScanError('conflicting_descriptors', 'Docker recovery digest descriptors conflict')
|
|
entry['positions'].append(descriptor['position'])
|
|
descriptor = None
|
|
image_bytes = sum(entry['descriptor']['size'] for entry in unique.values())
|
|
if image_bytes > limits['image_max_bytes']:
|
|
diagnostic.update(reason='image_byte_bound', observed=image_bytes, limit=limits['image_max_bytes'])
|
|
raise DockerContentScanError('recovery_budget', 'All Docker content does not fit the cumulative recovery bound')
|
|
if time.monotonic() >= deadline:
|
|
raise DockerContentScanError('timeout', 'Docker recovery deadline expired after preflight', True)
|
|
phase = 'staging'
|
|
work_root = tempfile.mkdtemp(prefix='docker-layer-', dir=get_work_dir())
|
|
harden_private_directory(work_root)
|
|
write_temp_owner(work_root, ['docker-full-recovery'], os.getpid(), required=True)
|
|
for index, entry in enumerate(unique.values()):
|
|
descriptor = entry['descriptor']
|
|
phase = 'blob_transfer'
|
|
blob_deadline = min(deadline, time.monotonic() + limits['blob_timeout_sec'])
|
|
if time.monotonic() >= blob_deadline:
|
|
raise DockerContentScanError('timeout', 'Docker recovery deadline expired', True)
|
|
destination = os.path.join(work_root, f'blob-{index:04d}')
|
|
try:
|
|
blob_min_free_bytes = max(0, int(min_free_bytes))
|
|
scope['blob_transfer_attempted'] = True
|
|
emit_client_scan_phase('downloading', {
|
|
'operation': 'docker_blob_transfer_recovery',
|
|
'descriptor_index': index,
|
|
})
|
|
outcome = stream_docker_registry_blob(
|
|
repository, descriptor, destination, bearer_auth,
|
|
deadline=blob_deadline, min_free_bytes=blob_min_free_bytes,
|
|
anonymous_only=anonymous_public_client,
|
|
)
|
|
bearer_auth = outcome.bearer_auth
|
|
phase = 'blob_scan'
|
|
emit_client_scan_phase('scanning', {
|
|
'operation': 'docker_blob_scan_recovery',
|
|
'descriptor_index': index,
|
|
})
|
|
scanned = _scan_docker_content_file(
|
|
destination, descriptor, limits, blob_deadline,
|
|
detectors, exclude_detectors, no_verification, trufflehog_config,
|
|
)
|
|
result['findings'].extend(attach_docker_content_provenance(
|
|
scanned['findings'], resolved, descriptor, entry['positions'], destination,
|
|
))
|
|
if scanned.get('source_failure'):
|
|
raise DockerLayerInfrastructureError(
|
|
'recovery_scanner_unavailable', 'Docker recovery scanner is unavailable',
|
|
category=scanned.get('source_failure_category') or 'source_resource',
|
|
)
|
|
if any(scanned.get(key) for key in ('errors', 'warnings', 'degraded', 'skipped')):
|
|
raise DockerContentScanError(
|
|
'recovery_scan_incomplete', 'Docker recovery scanner did not cover a blob',
|
|
bool(scanned.get('retryable', False)),
|
|
)
|
|
scope['scanned_descriptors'] += len(entry['positions'])
|
|
except ScanSlotFatalError:
|
|
retain_work = True
|
|
raise
|
|
finally:
|
|
if not retain_work and os.path.lexists(destination):
|
|
previous_phase = phase
|
|
phase = 'cleanup'
|
|
durable_unlink(destination)
|
|
phase = previous_phase
|
|
descriptor = None
|
|
phase = 'completion'
|
|
if time.monotonic() >= deadline:
|
|
raise DockerContentScanError('timeout', 'Docker recovery deadline expired', True)
|
|
except ScanSlotFatalError:
|
|
retain_work = True
|
|
raise
|
|
except Exception as exc:
|
|
code = _docker_content_error_code(getattr(exc, 'error_code', None), 'recovery_infrastructure')
|
|
if isinstance(exc, DockerRemoteAccessError):
|
|
code = _docker_content_error_code(exc.status, 'recovery_remote')
|
|
elif isinstance(exc, DockerRegistryResolutionError):
|
|
code = 'recovery_manifest_invalid'
|
|
diagnostic.setdefault('reason', {
|
|
'recovery_identity_mismatch': 'integrity',
|
|
'invalid_descriptor': 'integrity',
|
|
'conflicting_descriptors': 'integrity',
|
|
'digest_mismatch': 'integrity',
|
|
'size_mismatch': 'integrity',
|
|
'invalid_layer_archive': 'integrity',
|
|
'invalid_config_json': 'integrity',
|
|
'recovery_manifest_invalid': 'manifest_invalid',
|
|
'timeout': 'timeout',
|
|
'transfer_timeout': 'timeout',
|
|
'archive_timeout': 'timeout',
|
|
'recovery_scan_incomplete': 'scan_incomplete',
|
|
}.get(code, 'configuration' if phase == 'configuration' else 'other'))
|
|
diagnostic['phase'] = phase
|
|
if descriptor is not None:
|
|
diagnostic.update(descriptor_kind=descriptor['kind'], descriptor_index=descriptor['position'])
|
|
scope['diagnostic'] = diagnostic
|
|
result['errors'].append(f'Docker full-image recovery incomplete: {code}')
|
|
result['error_class'] = code
|
|
result['retryable'] = bool(getattr(exc, 'retryable', True))
|
|
if isinstance(exc, DockerRegistryResolutionError):
|
|
result['retryable'] = (
|
|
isinstance(exc, DockerRemoteAccessError)
|
|
and exc.status not in ('target_forbidden', 'auth_failed')
|
|
)
|
|
if result['retryable']:
|
|
result['source_failure'] = True
|
|
result['source_failure_category'] = 'remote_auth' if exc.status == 'auth_failed' else 'remote_rate_limit' if exc.status == 'rate_limited' else 'remote_transient'
|
|
elif anonymous_public_client and exc.status == 'auth_failed':
|
|
result['error_class'] = 'docker_registry_access'
|
|
result.pop('source_failure', None)
|
|
result.pop('source_failure_category', None)
|
|
elif isinstance(exc, DockerLayerInfrastructureError) or not isinstance(exc, (DockerContentScanError, DockerContentTransferError)):
|
|
result['source_failure'] = True
|
|
result['source_failure_category'] = getattr(exc, 'category', 'source_configuration' if isinstance(exc, ValueError) else 'source_resource')
|
|
result['source_failure_auth_related'] = bool(getattr(exc, 'auth_related', False))
|
|
finally:
|
|
if work_root and not retain_work:
|
|
try:
|
|
cleanup_command_work_dir(work_root)
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except (OSError, RuntimeError):
|
|
scope.setdefault('diagnostic', {'phase': 'cleanup', 'reason': 'cleanup'})
|
|
result['errors'].append('Docker full-image recovery private cleanup failed')
|
|
result.update(error_class='private_cleanup', retryable=True, source_failure=True,
|
|
source_failure_category='source_resource')
|
|
if not result['errors'] and time.monotonic() >= deadline:
|
|
scope['diagnostic'] = {'phase': 'completion', 'reason': 'timeout'}
|
|
result.update(errors=['Docker full-image recovery exceeded its absolute deadline'],
|
|
error_class='timeout', retryable=True)
|
|
scope['coverage_complete'] = not result['errors'] and scope['scanned_descriptors'] == scope.get('descriptor_count', 0) > 0
|
|
return result
|
|
|
|
|
|
def scan_docker_image(image_name, timeout_sec=1800, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, config_dir=None, trufflehog_concurrency=0, *, log_target=True, docker_recovery_limits=None, docker_recovery_min_free_bytes=20 << 30):
|
|
"""Scan a Docker image for secrets"""
|
|
started = time.monotonic()
|
|
try:
|
|
timeout_sec = float(timeout_sec)
|
|
if not math.isfinite(timeout_sec) or timeout_sec <= 0:
|
|
raise ValueError('invalid timeout')
|
|
except (TypeError, ValueError):
|
|
return {'findings': [], 'errors': ['Docker scan requires a positive finite time budget'],
|
|
'error_class': 'timeout', 'retryable': False}
|
|
deadline = started + timeout_sec
|
|
anonymous_public_client = (
|
|
_client_remote_execution_kind.get() == 'docker_direct_v1'
|
|
)
|
|
if anonymous_public_client and config_dir:
|
|
raise RuntimeError('remote Docker direct execution cannot use credentials')
|
|
try:
|
|
image_ref = (
|
|
parse_dockerhub_digest_target(image_name)['image']
|
|
if anonymous_public_client else parse_docker_target(image_name)['image']
|
|
)
|
|
except (TypeError, ValueError) as exc:
|
|
return {
|
|
"findings": [], "errors": [f"Docker image target rejected: {exc}"],
|
|
"error_class": "invalid_target", "retryable": False,
|
|
}
|
|
if log_target:
|
|
logger.info(f"Scanning Docker image: {image_ref}")
|
|
|
|
cmd = [get_trufflehog_cmd(), 'docker', '--image', image_ref, '--json', '--no-update', '--local-dev', '--log-level', '2']
|
|
concurrency = max(0, min(64, int(trufflehog_concurrency or 0)))
|
|
if concurrency:
|
|
cmd.extend(['--concurrency', str(concurrency)])
|
|
append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config)
|
|
|
|
env = os.environ.copy()
|
|
if config_dir:
|
|
env['DOCKER_CONFIG'] = config_dir
|
|
implicit_auth_unsupported = (
|
|
not anonymous_public_client
|
|
and not (config_dir or env.get('DOCKER_CONFIG'))
|
|
and _docker_implicit_auth_present()
|
|
)
|
|
|
|
results = {"findings": [], "errors": []}
|
|
emit_client_scan_phase('scanning', {
|
|
'integrated_operation': 'docker_pull_and_scan',
|
|
})
|
|
with run_command_streamed(cmd, max(0, deadline - time.monotonic()), env, deadline=deadline) as output:
|
|
apply_trufflehog_diagnostics(results, output, output.returncode, 'docker')
|
|
append_trufflehog_findings(results, output.stdout_lines())
|
|
|
|
for line in results.get('errors', []):
|
|
try:
|
|
diagnostic = json.loads(line)
|
|
except (TypeError, ValueError):
|
|
continue
|
|
if (
|
|
isinstance(diagnostic, dict) and diagnostic.get('msg') == 'error processing layer'
|
|
and diagnostic.get('error') == 'unexpected EOF'
|
|
):
|
|
results.setdefault('scan_meta', {})['docker_native_diagnostic'] = {
|
|
'phase': 'native_layer_processing', 'subcause': 'unexpected_eof_ambiguous',
|
|
'coverage_complete': False,
|
|
}
|
|
break
|
|
|
|
codec_warnings = []
|
|
for line in results.get('warnings', []):
|
|
try:
|
|
diagnostic = json.loads(line)
|
|
except (TypeError, ValueError):
|
|
continue
|
|
if isinstance(diagnostic, dict) and diagnostic.get('msg') == 'error processing layer' and diagnostic.get('error') == 'gzip: invalid header':
|
|
codec_warnings.append(line)
|
|
if codec_warnings:
|
|
if not results.get('errors') and not results.get('source_failure') and not results.get('skipped') and len(codec_warnings) == len(results.get('warnings', [])):
|
|
recovered = _recover_docker_image_contents(
|
|
image_ref, deadline, (
|
|
None if anonymous_public_client
|
|
else config_dir or env.get('DOCKER_CONFIG')
|
|
),
|
|
docker_recovery_limits, docker_recovery_min_free_bytes,
|
|
detectors, exclude_detectors, no_verification, trufflehog_config,
|
|
implicit_auth_unsupported=implicit_auth_unsupported,
|
|
anonymous_public_client=anonymous_public_client,
|
|
)
|
|
results['findings'].extend(recovered.pop('findings'))
|
|
results.setdefault('scan_meta', {}).update(recovered.pop('scan_meta'))
|
|
if not recovered['errors'] and results['scan_meta']['docker_full_recovery']['coverage_complete']:
|
|
for key in ('warnings', 'warning_classes', 'degraded', 'retryable'):
|
|
results.pop(key, None)
|
|
results['scan_meta']['docker_full_recovery']['recovered_codec'] = True
|
|
else:
|
|
if not recovered['errors']:
|
|
recovered.update(errors=['Docker full-image recovery coverage is incomplete'],
|
|
error_class='docker_recovery_incomplete', retryable=False)
|
|
results['scan_meta']['docker_full_recovery'].setdefault(
|
|
'diagnostic', {'phase': 'completion', 'reason': 'scan_incomplete'},
|
|
)
|
|
results.update(recovered)
|
|
else:
|
|
results['errors'].append('Docker codec recovery cannot clear unrelated scan diagnostics')
|
|
results.setdefault('error_class', 'docker_recovery_incomplete')
|
|
results.setdefault('scan_meta', {})['docker_full_recovery'] = {
|
|
'coverage_complete': False, 'blob_transfer_attempted': False,
|
|
'diagnostic': {'phase': 'native_diagnostics', 'reason': 'unrelated_diagnostics'},
|
|
}
|
|
results = apply_finding_filters(results, image_ref, log_target=log_target)
|
|
if time.monotonic() >= deadline:
|
|
had_errors = bool(results.get('errors'))
|
|
results.setdefault('errors', []).append('Docker scan exceeded its absolute deadline including cleanup and filtering')
|
|
if not had_errors:
|
|
results.update(error_class='timeout', retryable=True)
|
|
metadata = results.setdefault('scan_meta', {})
|
|
metadata['docker_deadline_exceeded'] = True
|
|
if 'docker_full_recovery' in metadata:
|
|
metadata['docker_full_recovery']['coverage_complete'] = False
|
|
metadata['docker_full_recovery'].setdefault('diagnostic', {'phase': 'completion', 'reason': 'timeout'})
|
|
return results
|
|
|
|
|
|
def _append_windows_git_longpaths(command_env, operation):
|
|
if os.name != 'nt':
|
|
return
|
|
config_count = command_env.get('GIT_CONFIG_COUNT') or '0'
|
|
if not re.fullmatch(r'[0-9]{1,3}', config_count) or int(config_count) > 255:
|
|
raise ValueError(f'{operation} GIT_CONFIG_COUNT must be an integer from 0 to 255')
|
|
config_count = int(config_count)
|
|
command_env[f'GIT_CONFIG_KEY_{config_count}'] = 'core.longpaths'
|
|
command_env[f'GIT_CONFIG_VALUE_{config_count}'] = 'true'
|
|
command_env['GIT_CONFIG_COUNT'] = str(config_count + 1)
|
|
|
|
|
|
def scan_huggingface_space(space_id, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None):
|
|
anonymous_public_client = (
|
|
_client_remote_execution_kind.get() == 'huggingface_space_v1'
|
|
)
|
|
if anonymous_public_client and token:
|
|
raise RuntimeError('remote HuggingFace direct execution cannot use a token')
|
|
logger.info(f"Scanning HuggingFace Space: {space_id}")
|
|
|
|
cmd = [get_trufflehog_cmd(), 'huggingface', '--space', space_id, '--json', '--no-update']
|
|
secrets_to_redact = []
|
|
command_env = os.environ.copy()
|
|
_append_windows_git_longpaths(command_env, 'HuggingFace')
|
|
if token:
|
|
command_env['HUGGINGFACE_TOKEN'] = token
|
|
command_env['HF_TOKEN'] = token
|
|
secrets_to_redact.append(token)
|
|
append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config)
|
|
|
|
results = {"findings": [], "errors": []}
|
|
emit_client_scan_phase('scanning', {
|
|
'integrated_operation': 'huggingface_clone_and_scan',
|
|
})
|
|
with run_command_streamed(cmd, timeout_sec, command_env) as output:
|
|
apply_trufflehog_diagnostics(
|
|
results, output, output.returncode, 'huggingface',
|
|
redactions=secrets_to_redact,
|
|
)
|
|
append_trufflehog_findings(
|
|
results, output.stdout_lines(redactions=secrets_to_redact),
|
|
)
|
|
|
|
return apply_finding_filters(results, space_id)
|
|
|
|
WINDOWS_RESERVED_NAMES = {
|
|
'con', 'prn', 'aux', 'nul',
|
|
*(f'com{index}' for index in range(1, 10)),
|
|
*(f'lpt{index}' for index in range(1, 10)),
|
|
}
|
|
|
|
|
|
def validate_archive_member_name(name):
|
|
normalized = str(name or '').replace('\\', '/')
|
|
if normalized.startswith('/') or '..' in normalized.split('/'):
|
|
raise ValueError(f'Unsafe archive member path: {name}')
|
|
for component in [part for part in normalized.split('/') if part]:
|
|
stem = component.split('.', 1)[0].lower()
|
|
if ':' in component or component.endswith((' ', '.')) or stem in WINDOWS_RESERVED_NAMES:
|
|
raise ValueError(f'Unsafe Windows archive member name: {name}')
|
|
return normalized
|
|
|
|
|
|
class LimitedReader:
|
|
def __init__(self, source, limit):
|
|
self.source = source
|
|
self.remaining = max(0, int(limit))
|
|
|
|
def read(self, size=-1):
|
|
if self.remaining <= 0:
|
|
raise ValueError('Archive decompressed stream exceeds safety budget')
|
|
if size is None or size < 0:
|
|
size = self.remaining + 1
|
|
data = self.source.read(min(size, self.remaining + 1))
|
|
self.remaining -= len(data)
|
|
if self.remaining < 0:
|
|
raise ValueError('Archive decompressed stream exceeds safety budget')
|
|
return data
|
|
|
|
|
|
def safe_extract_tar(tar_path, destination, max_files=100000, max_total_size_mb=1024, max_file_size_mb=512):
|
|
_raise_if_scan_slot_fatal()
|
|
destination_abs = os.path.abspath(destination)
|
|
max_total_bytes = int(max_total_size_mb or 0) * 1024 * 1024
|
|
max_file_bytes = int(max_file_size_mb or 0) * 1024 * 1024
|
|
stream_budget = max_total_bytes + (64 * 1024 * 1024)
|
|
raw_source = open(tar_path, 'rb')
|
|
magic = raw_source.read(6)
|
|
raw_source.seek(0)
|
|
if magic.startswith(b'\x1f\x8b'):
|
|
decompressed = gzip.GzipFile(fileobj=raw_source)
|
|
elif magic.startswith(b'BZh'):
|
|
decompressed = bz2.BZ2File(raw_source)
|
|
elif magic.startswith(b'\xfd7zXZ\x00'):
|
|
decompressed = lzma.LZMAFile(raw_source)
|
|
else:
|
|
decompressed = raw_source
|
|
try:
|
|
archive = tarfile.open(fileobj=LimitedReader(decompressed, stream_budget), mode='r|')
|
|
try:
|
|
total_size = 0
|
|
member_count = 0
|
|
for member in archive:
|
|
_raise_if_scan_slot_fatal()
|
|
if max_files and member_count >= int(max_files):
|
|
raise ValueError(f'Tar archive exceeds {max_files} members')
|
|
member_count += 1
|
|
validate_archive_member_name(member.name)
|
|
member_path = os.path.abspath(os.path.join(destination, member.name))
|
|
if not member_path.startswith(destination_abs + os.sep) and member_path != destination_abs:
|
|
raise ValueError(f"Unsafe tar member path: {member.name}")
|
|
if member.issym() or member.islnk() or member.isdev() or member.isfifo():
|
|
raise ValueError(f"Unsafe tar member type: {member.name}")
|
|
if member.isfile():
|
|
if max_file_bytes and member.size > max_file_bytes:
|
|
raise ValueError(f"Tar member exceeds {max_file_size_mb} MB: {member.name}")
|
|
total_size += max(0, int(member.size or 0))
|
|
if max_total_bytes and total_size > max_total_bytes:
|
|
raise ValueError(f"Tar archive exceeds {max_total_size_mb} MB extracted")
|
|
archive.extract(member, destination, filter='data')
|
|
_raise_if_scan_slot_fatal()
|
|
finally:
|
|
archive.close()
|
|
finally:
|
|
if decompressed is not raw_source:
|
|
decompressed.close()
|
|
raw_source.close()
|
|
|
|
|
|
def validate_zip_central_directory(zip_path, max_files, max_metadata_size_mb=16):
|
|
_raise_if_scan_slot_fatal()
|
|
with open(zip_path, 'rb') as source:
|
|
source.seek(0, os.SEEK_END)
|
|
size = source.tell()
|
|
source.seek(max(0, size - 65557))
|
|
tail = source.read()
|
|
offset = tail.rfind(b'PK\x05\x06')
|
|
if offset < 0 or offset + 22 > len(tail):
|
|
raise ValueError('Zip end-of-central-directory record is missing')
|
|
disk_number = int.from_bytes(tail[offset + 4:offset + 6], 'little')
|
|
central_disk = int.from_bytes(tail[offset + 6:offset + 8], 'little')
|
|
entry_count = int.from_bytes(tail[offset + 10:offset + 12], 'little')
|
|
central_size = int.from_bytes(tail[offset + 12:offset + 16], 'little')
|
|
central_offset = int.from_bytes(tail[offset + 16:offset + 20], 'little')
|
|
if disk_number or central_disk or entry_count == 0xFFFF or central_size == 0xFFFFFFFF or central_offset == 0xFFFFFFFF:
|
|
raise ValueError('Multi-disk and ZIP64 archives are not accepted')
|
|
if max_files and entry_count > int(max_files):
|
|
raise ValueError(f'Zip archive exceeds {max_files} members')
|
|
max_metadata_bytes = int(max_metadata_size_mb or 0) * 1024 * 1024
|
|
if max_metadata_bytes and central_size > max_metadata_bytes:
|
|
raise ValueError(f'Zip central directory exceeds {max_metadata_size_mb} MB')
|
|
if central_offset < 0 or central_size < 0 or central_offset + central_size > size:
|
|
raise ValueError('Zip central directory points outside the archive')
|
|
with open(zip_path, 'rb') as source:
|
|
source.seek(central_offset)
|
|
consumed = 0
|
|
for _ in range(entry_count):
|
|
_raise_if_scan_slot_fatal()
|
|
header = source.read(46)
|
|
if len(header) != 46 or header[:4] != b'PK\x01\x02':
|
|
raise ValueError('Invalid zip central-directory entry')
|
|
name_len = int.from_bytes(header[28:30], 'little')
|
|
extra_len = int.from_bytes(header[30:32], 'little')
|
|
comment_len = int.from_bytes(header[32:34], 'little')
|
|
variable_size = name_len + extra_len + comment_len
|
|
source.seek(variable_size, os.SEEK_CUR)
|
|
consumed += 46 + variable_size
|
|
if consumed > central_size:
|
|
raise ValueError('Zip central-directory size mismatch')
|
|
if consumed != central_size:
|
|
raise ValueError('Zip central-directory entry count mismatch')
|
|
return entry_count
|
|
|
|
|
|
def safe_extract_package_zip(zip_path, destination, max_files=100000, max_total_size_mb=1024, max_file_size_mb=512):
|
|
_raise_if_scan_slot_fatal()
|
|
destination_abs = os.path.abspath(destination)
|
|
max_total_bytes = int(max_total_size_mb or 0) * 1024 * 1024
|
|
max_file_bytes = int(max_file_size_mb or 0) * 1024 * 1024
|
|
validate_zip_central_directory(zip_path, max_files)
|
|
with zipfile.ZipFile(zip_path) as archive:
|
|
members = archive.infolist()
|
|
if max_files and len(members) > int(max_files):
|
|
raise ValueError(f'Zip archive exceeds {max_files} members')
|
|
total_size = 0
|
|
for member in members:
|
|
_raise_if_scan_slot_fatal()
|
|
validate_archive_member_name(member.filename)
|
|
member_path = os.path.abspath(os.path.join(destination, member.filename))
|
|
if not member_path.startswith(destination_abs + os.sep) and member_path != destination_abs:
|
|
raise ValueError(f"Unsafe zip member path: {member.filename}")
|
|
if member.is_dir():
|
|
continue
|
|
if max_file_bytes and member.file_size > max_file_bytes:
|
|
raise ValueError(f"Zip member exceeds {max_file_size_mb} MB: {member.filename}")
|
|
total_size += max(0, int(member.file_size or 0))
|
|
if max_total_bytes and total_size > max_total_bytes:
|
|
raise ValueError(f"Zip archive exceeds {max_total_size_mb} MB extracted")
|
|
for member in members:
|
|
_raise_if_scan_slot_fatal()
|
|
archive.extract(member, destination)
|
|
_raise_if_scan_slot_fatal()
|
|
|
|
def safe_extract_archive(archive_path, destination, max_total_size_mb=1024):
|
|
_raise_if_scan_slot_fatal()
|
|
if tarfile.is_tarfile(archive_path):
|
|
safe_extract_tar(archive_path, destination, max_total_size_mb=max_total_size_mb)
|
|
return
|
|
if zipfile.is_zipfile(archive_path):
|
|
safe_extract_package_zip(archive_path, destination, max_total_size_mb=max_total_size_mb)
|
|
return
|
|
raise ValueError("Unsupported package archive format")
|
|
|
|
def download_file(url, path, max_size_mb=50, timeout=60):
|
|
_raise_if_scan_slot_fatal()
|
|
max_bytes = max_size_mb * 1024 * 1024
|
|
with requests.Session() as session:
|
|
session.trust_env = False
|
|
with session.get(url, stream=True, headers={'User-Agent': 'GitSecretsScanner/2.0'}, timeout=timeout) as response:
|
|
_raise_if_scan_slot_fatal()
|
|
response.raise_for_status()
|
|
total = 0
|
|
with open(path, 'wb') as f:
|
|
for chunk in response.iter_content(chunk_size=1024 * 256):
|
|
_raise_if_scan_slot_fatal()
|
|
if not chunk:
|
|
continue
|
|
total += len(chunk)
|
|
if max_bytes and total > max_bytes:
|
|
raise ValueError(f"Artifact exceeds {max_size_mb} MB")
|
|
f.write(chunk)
|
|
_raise_if_scan_slot_fatal()
|
|
harden_private_file(path)
|
|
return total
|
|
|
|
def scan_npm_package(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, max_artifact_size_mb=50):
|
|
"""Download, extract, and scan an npm package tarball with TruffleHog filesystem."""
|
|
package = parse_npm_target(target)
|
|
package_id = npm_package_id(package)
|
|
logger.info(f"Scanning npm package: {package_id}")
|
|
|
|
_raise_if_scan_slot_fatal()
|
|
work_dir = create_command_work_dir()
|
|
_raise_if_scan_slot_fatal()
|
|
if not work_dir:
|
|
return {"findings": [], "errors": ["Unable to create npm work dir"]}
|
|
|
|
try:
|
|
tarball_path = os.path.join(work_dir, 'package.tgz')
|
|
extract_dir = os.path.join(work_dir, 'extract')
|
|
ensure_private_directory(extract_dir, reject_reparse=True)
|
|
downloaded = download_file(package['tarball'], tarball_path, max_artifact_size_mb, timeout=min(timeout_sec, 120))
|
|
_raise_if_scan_slot_fatal()
|
|
safe_extract_tar(tarball_path, extract_dir, max_total_size_mb=max(256, int(max_artifact_size_mb or 0) * 4))
|
|
_raise_if_scan_slot_fatal()
|
|
harden_private_tree(extract_dir)
|
|
_raise_if_scan_slot_fatal()
|
|
harvest_warnings = []
|
|
postman_targets = find_postman_artifacts(
|
|
extract_dir,
|
|
'npm',
|
|
package,
|
|
max_artifact_size_mb=max_artifact_size_mb,
|
|
warnings=harvest_warnings,
|
|
)
|
|
|
|
cmd = [get_trufflehog_cmd(), 'filesystem', extract_dir, '--json', '--no-update']
|
|
append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config)
|
|
results = {"findings": [], "errors": [], "package": package, "bytes": downloaded, "postman_targets": postman_targets}
|
|
with run_command_streamed(cmd, timeout_sec) as output:
|
|
apply_trufflehog_diagnostics(results, output, output.returncode, 'npm')
|
|
append_trufflehog_findings(results, output.stdout_lines())
|
|
attach_nearby_context(results)
|
|
_attach_postman_harvest_warnings(results, harvest_warnings)
|
|
return apply_finding_filters(results, package_id)
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except Exception as e:
|
|
return {"findings": [], "errors": [f"npm scan failed: {str(e)}"], "package": package}
|
|
finally:
|
|
cleanup_command_work_dir(work_dir)
|
|
|
|
def scan_pypi_package(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, max_artifact_size_mb=50):
|
|
"""Download, extract, and scan a PyPI sdist/wheel with TruffleHog filesystem."""
|
|
package = parse_pypi_target(target)
|
|
package_id = pypi_package_id(package)
|
|
logger.info(f"Scanning PyPI package: {package_id}")
|
|
|
|
declared_size = int(package.get('size') or 0)
|
|
max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024
|
|
if max_bytes and declared_size > max_bytes:
|
|
return {
|
|
"findings": [],
|
|
"errors": [],
|
|
"skipped": f"artifact exceeds {max_artifact_size_mb} MB",
|
|
"package": package,
|
|
}
|
|
|
|
_raise_if_scan_slot_fatal()
|
|
work_dir = create_command_work_dir()
|
|
_raise_if_scan_slot_fatal()
|
|
if not work_dir:
|
|
return {"findings": [], "errors": ["Unable to create PyPI work dir"]}
|
|
|
|
try:
|
|
artifact_path = os.path.join(work_dir, 'package-artifact')
|
|
extract_dir = os.path.join(work_dir, 'extract')
|
|
ensure_private_directory(extract_dir, reject_reparse=True)
|
|
downloaded = download_file(package['artifact'], artifact_path, max_artifact_size_mb, timeout=min(timeout_sec, 120))
|
|
_raise_if_scan_slot_fatal()
|
|
safe_extract_archive(artifact_path, extract_dir, max_total_size_mb=max(256, int(max_artifact_size_mb or 0) * 4))
|
|
_raise_if_scan_slot_fatal()
|
|
harden_private_tree(extract_dir)
|
|
_raise_if_scan_slot_fatal()
|
|
harvest_warnings = []
|
|
postman_targets = find_postman_artifacts(
|
|
extract_dir,
|
|
'pypi',
|
|
package,
|
|
max_artifact_size_mb=max_artifact_size_mb,
|
|
warnings=harvest_warnings,
|
|
)
|
|
|
|
cmd = [get_trufflehog_cmd(), 'filesystem', extract_dir, '--json', '--no-update']
|
|
append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config)
|
|
results = {"findings": [], "errors": [], "package": package, "bytes": downloaded, "postman_targets": postman_targets}
|
|
with run_command_streamed(cmd, timeout_sec) as output:
|
|
apply_trufflehog_diagnostics(results, output, output.returncode, 'pypi')
|
|
append_trufflehog_findings(results, output.stdout_lines())
|
|
attach_nearby_context(results)
|
|
_attach_postman_harvest_warnings(results, harvest_warnings)
|
|
return apply_finding_filters(results, package_id)
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except Exception as e:
|
|
return {"findings": [], "errors": [f"PyPI scan failed: {str(e)}"], "package": package}
|
|
finally:
|
|
cleanup_command_work_dir(work_dir)
|
|
|
|
def scan_package_git_repo(target, timeout_sec=900, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, max_depth=None, max_commit_age_days=None, commit_lookup_pages=3, skip_if_commit_lookup_fails=True):
|
|
data = parse_package_git_target(target)
|
|
repo_url = data.get('repo_url')
|
|
provider = data.get('provider')
|
|
if not repo_url:
|
|
return {"findings": [], "errors": ["package_git target missing repo_url"], "package": data}
|
|
logger.info(
|
|
f"Scanning package git repo: {repo_url} "
|
|
f"({data.get('package_source')}:{data.get('name')}@{data.get('version')})"
|
|
)
|
|
candidate = normalize_git_repo_candidate(repo_url) or {}
|
|
canonical_provider = candidate.get('provider') or provider
|
|
provider_token = token if canonical_provider == 'github' else None
|
|
preflight = preflight_package_git_repo(data, provider_token, min(int(timeout_sec or 30), 20))
|
|
if preflight and preflight.get('skip'):
|
|
reason = preflight.get('reason') or 'package_git repo unavailable'
|
|
logger.info(f"Skipping package git repo {repo_url}: {reason}")
|
|
return {"findings": [], "errors": [], "skipped": reason, "package": data, "scan_meta": {"preflight": preflight}}
|
|
scan_provider = (preflight or {}).get('provider') or canonical_provider
|
|
scan_token = None if preflight and preflight.get('retry_unauthenticated') else provider_token
|
|
result = scan_git_repo(
|
|
repo_url,
|
|
timeout_sec=timeout_sec,
|
|
detectors=detectors,
|
|
exclude_detectors=exclude_detectors,
|
|
no_verification=no_verification,
|
|
trufflehog_config=trufflehog_config,
|
|
token=scan_token,
|
|
provider=scan_provider,
|
|
max_depth=max_depth,
|
|
max_commit_age_days=max_commit_age_days,
|
|
commit_lookup_pages=commit_lookup_pages,
|
|
skip_if_commit_lookup_fails=skip_if_commit_lookup_fails,
|
|
)
|
|
convert_package_git_unavailable_to_skip(result)
|
|
result['package'] = data
|
|
return result
|
|
|
|
|
|
def preflight_package_git_repo(data, token=None, timeout=15):
|
|
repo_url = data.get('repo_url') or ''
|
|
provider = (data.get('provider') or '').lower()
|
|
candidate = normalize_git_repo_candidate(repo_url)
|
|
if not candidate:
|
|
return {'skip': True, 'reason': 'package_git repo URL is unsupported or invalid'}
|
|
provider = candidate.get('provider') or provider
|
|
repo_path = candidate.get('repo_path') or ''
|
|
try:
|
|
if provider == 'github':
|
|
response = api_request(
|
|
'GET',
|
|
f'https://api.github.com/repos/{repo_path}',
|
|
headers=github_headers(token),
|
|
timeout=timeout,
|
|
retry_statuses={500, 502, 503, 504},
|
|
)
|
|
if response.status_code == 200:
|
|
return {'skip': False, 'provider': provider, 'repo_path': repo_path}
|
|
message = response_message(response).lower()
|
|
if response.status_code == 404:
|
|
return {'skip': True, 'provider': provider, 'repo_path': repo_path, 'reason': 'package_git repo not found or private'}
|
|
if response.status_code in (401, 403) and 'rate limit' not in message:
|
|
return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code, 'retry_unauthenticated': True}
|
|
return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code}
|
|
if provider == 'gitlab':
|
|
response = api_request(
|
|
'GET',
|
|
f'https://gitlab.com/api/v4/projects/{quote(repo_path, safe="")}',
|
|
headers=gitlab_headers(token),
|
|
timeout=timeout,
|
|
retry_statuses={500, 502, 503, 504},
|
|
)
|
|
if response.status_code == 200:
|
|
return {'skip': False, 'provider': provider, 'repo_path': repo_path}
|
|
if response.status_code == 404:
|
|
return {'skip': True, 'provider': provider, 'repo_path': repo_path, 'reason': 'package_git project not found or private'}
|
|
if response.status_code in (401, 403):
|
|
return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code, 'retry_unauthenticated': True}
|
|
return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code}
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except Exception as e:
|
|
logger.warning(f"Package git preflight failed for {repo_url}: {str(e)[:300]}")
|
|
return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': 'unknown'}
|
|
|
|
|
|
def postman_stage_filename(target_data):
|
|
kind = target_data.get('kind') or 'artifact'
|
|
digest = target_data.get('sha256') or target_data.get('sha') or 'postman'
|
|
suffix = POSTMAN_COLLECTION_SUFFIX if kind == 'collection' else POSTMAN_ENVIRONMENT_SUFFIX if kind == 'environment' else 'postman.json'
|
|
return f'{str(digest)[:16]}.{suffix}'
|
|
|
|
|
|
def is_postman_placeholder(value):
|
|
text = str(value or '').strip()
|
|
return bool(text and POSTMAN_PLACEHOLDER_RE.match(text))
|
|
|
|
|
|
def load_postman_context(cache_path, payload=None):
|
|
max_input_bytes = min(
|
|
POSTMAN_JSON_HARD_MAX_INPUT_BYTES,
|
|
max(1, int(getattr(scan_config, 'postman_context_max_input_bytes', POSTMAN_JSON_HARD_MAX_INPUT_BYTES))),
|
|
)
|
|
max_nodes = max(1, int(getattr(scan_config, 'postman_context_max_nodes', 100000)))
|
|
max_depth = max(1, int(getattr(scan_config, 'postman_context_max_depth', 64)))
|
|
max_scalar_bytes = max(1, int(getattr(scan_config, 'postman_context_max_scalar_bytes', 16 * 1024 * 1024)))
|
|
max_items = max(1, int(getattr(scan_config, 'postman_context_max_items', 50000)))
|
|
try:
|
|
if payload is None:
|
|
size = os.path.getsize(cache_path)
|
|
if size <= 0 or size > max_input_bytes:
|
|
raise PostmanCacheValidationError('Postman context input exceeds its bounded byte limit')
|
|
with open(cache_path, 'rb') as f:
|
|
payload = f.read(max_input_bytes + 1)
|
|
elif not isinstance(payload, bytes):
|
|
raise PostmanCacheValidationError('Postman context input must be bytes')
|
|
size = len(payload)
|
|
if size <= 0 or size > max_input_bytes:
|
|
raise PostmanCacheValidationError('Postman context input exceeds its bounded byte limit')
|
|
if len(payload) > max_input_bytes:
|
|
raise PostmanCacheValidationError('Postman context input exceeds its bounded byte limit')
|
|
data = json.loads(payload.decode('utf-8-sig'))
|
|
except PostmanCacheValidationError:
|
|
raise
|
|
except (OSError, UnicodeDecodeError, ValueError, RecursionError) as exc:
|
|
raise PostmanCacheValidationError('Postman context is not bounded valid JSON') from exc
|
|
contexts = []
|
|
traversed = 0
|
|
scalar_bytes = 0
|
|
|
|
def charge(*values):
|
|
nonlocal scalar_bytes
|
|
scalar_bytes += sum(len(str(value or '').encode('utf-8', errors='replace')) for value in values)
|
|
if scalar_bytes > max_scalar_bytes:
|
|
raise PostmanCacheValidationError('Postman context scalar byte limit exceeded')
|
|
|
|
def host_from_url(value):
|
|
try:
|
|
return urlsplit(str(value)).hostname or ''
|
|
except Exception:
|
|
return ''
|
|
|
|
def add_context(path, key, value, endpoint='', auth_type='', location='value'):
|
|
if value is None:
|
|
return
|
|
endpoint = str(endpoint or '')
|
|
text = str(value)
|
|
charge(path, key, text, endpoint, auth_type, location)
|
|
if len(contexts) >= max_items:
|
|
raise PostmanCacheValidationError('Postman context item limit exceeded')
|
|
contexts.append({
|
|
'path': path,
|
|
'key': str(key or ''),
|
|
'value': text,
|
|
'endpoint': endpoint,
|
|
'host': host_from_url(endpoint),
|
|
'auth_type': str(auth_type or ''),
|
|
'location': location,
|
|
})
|
|
|
|
stack = [(data, '$', '', '', 'value', 0)]
|
|
while stack:
|
|
value, path, endpoint, auth_type, inherited_location, depth = stack.pop()
|
|
traversed += 1
|
|
if traversed > max_nodes:
|
|
raise PostmanCacheValidationError('Postman context traversal item limit exceeded')
|
|
if depth > max_depth:
|
|
raise PostmanCacheValidationError('Postman context depth limit exceeded')
|
|
if len(path.encode('utf-8', errors='replace')) > 4096:
|
|
raise PostmanCacheValidationError('Postman context path limit exceeded')
|
|
if isinstance(value, dict):
|
|
local_endpoint = endpoint
|
|
url_value = value.get('url')
|
|
if isinstance(url_value, str):
|
|
local_endpoint = url_value
|
|
elif isinstance(url_value, dict) and url_value.get('raw'):
|
|
local_endpoint = str(url_value.get('raw'))
|
|
local_auth = auth_type
|
|
auth = value.get('auth')
|
|
if isinstance(auth, dict):
|
|
local_auth = str(auth.get('type') or local_auth or '')
|
|
if traversed + len(stack) + len(value) > max_nodes:
|
|
raise PostmanCacheValidationError('Postman context traversal item limit exceeded')
|
|
children = []
|
|
for key, item in value.items():
|
|
lower_key = str(key).lower()
|
|
location = 'value'
|
|
if lower_key in ('header', 'headers'):
|
|
location = 'header'
|
|
elif lower_key in ('query', 'queryparam', 'query_params'):
|
|
location = 'query_param'
|
|
elif lower_key in ('body', 'raw'):
|
|
location = 'body'
|
|
elif lower_key in ('variable', 'values'):
|
|
location = 'environment_variable'
|
|
child_path = f'{path}.{key}'
|
|
charge(key)
|
|
children.append((item, child_path, local_endpoint, local_auth, location, depth + 1))
|
|
stack.extend(reversed(children))
|
|
elif isinstance(value, list):
|
|
if traversed + len(stack) + len(value) > max_nodes:
|
|
raise PostmanCacheValidationError('Postman context traversal item limit exceeded')
|
|
stack.extend(
|
|
(value[index], f'{path}[{index}]', endpoint, auth_type, inherited_location, depth + 1)
|
|
for index in range(len(value) - 1, -1, -1)
|
|
)
|
|
else:
|
|
add_context(path, '', value, endpoint, auth_type, inherited_location)
|
|
return contexts
|
|
|
|
|
|
def provider_from_postman(detector_name='', host='', value=''):
|
|
detector = str(detector_name or '').lower()
|
|
if detector in ('openai', 'anthropic', 'github', 'gitlab', 'stripe', 'slack'):
|
|
return detector
|
|
text = ' '.join([str(host or '').lower(), str(value or '').lower()])
|
|
if 'api.openai.com' in text or 'sk-proj-' in text or re.search(r'\bsk-[A-Za-z0-9]{20,}', str(value or '')):
|
|
return 'openai'
|
|
if 'anthropic.com' in text or 'sk-ant-' in text:
|
|
return 'anthropic'
|
|
if 'generativelanguage.googleapis.com' in text or 'aiplatform.googleapis.com' in text:
|
|
return 'google'
|
|
if 'huggingface.co' in text or str(value or '').startswith('hf_'):
|
|
return 'huggingface'
|
|
if 'github.com' in text or str(value or '').startswith(GITHUB_TOKEN_PREFIXES):
|
|
return 'github'
|
|
if 'gitlab' in text or str(value or '').startswith(GITLAB_TOKEN_PREFIXES):
|
|
return 'gitlab'
|
|
if 'stripe.com' in text or str(value or '').startswith(('sk_live_', 'rk_live_')):
|
|
return 'stripe'
|
|
return ''
|
|
|
|
|
|
def credential_kind_from_postman(context, value):
|
|
key = str(context.get('key') or '').lower()
|
|
auth_type = str(context.get('auth_type') or '').lower()
|
|
location = str(context.get('location') or '').lower()
|
|
text = str(value or '').strip()
|
|
if is_postman_placeholder(text):
|
|
return 'placeholder'
|
|
if auth_type == 'bearer' or key == 'authorization' or text.lower().startswith('bearer '):
|
|
return 'jwt' if re.match(r'^(?:bearer\s+)?eyJ[A-Za-z0-9_-]+\.', text, re.IGNORECASE) else 'bearer_token'
|
|
if ('api' in key and 'key' in key) or key in ('x-api-key', 'apikey'):
|
|
return 'api_key'
|
|
if 'client_secret' in key or 'client-secret' in key:
|
|
return 'oauth_client_secret'
|
|
if 'password' in key:
|
|
return 'basic_auth_password' if auth_type == 'basic' else 'password'
|
|
if location == 'query_param' and ('token' in key or 'key' in key):
|
|
return 'api_key'
|
|
if re.match(r'^eyJ[A-Za-z0-9_-]+\.', text):
|
|
return 'jwt'
|
|
return 'unknown'
|
|
|
|
|
|
POSTMAN_GEMINI_KEY_RE = re.compile(r'(?:AIza[0-9A-Za-z\-_]{35}|AQ\.[0-9A-Za-z\-_]{50})')
|
|
POSTMAN_AZURE_OPENAI_KEY_RE = re.compile(r'\b[a-f0-9]{32}\b', re.IGNORECASE)
|
|
POSTMAN_AZURE_OPENAI_ENDPOINT_RE = re.compile(r'([a-z0-9-]+\.openai\.azure\.com)', re.IGNORECASE)
|
|
FOUNDRY_ENDPOINT_HOST_RE = r'[a-z0-9-]+(?:\.[a-z0-9-]+)*\.(?:models\.ai\.azure\.com|services\.ai\.azure\.com|inference\.ai\.azure\.com)'
|
|
POSTMAN_FOUNDRY_ENDPOINT_RE = re.compile(r'((?:https?://)?' + FOUNDRY_ENDPOINT_HOST_RE + r'(?:/[^\s:"\'<>\\]*)?)', re.IGNORECASE)
|
|
FOUNDRY_ASSIGNMENT_RE = re.compile(r'''(?ix)
|
|
(?:authorization|api[_-]?key|key|token|secret|credential|bearer)
|
|
[^\n:=]{0,80}
|
|
[:=]
|
|
\s*["']?(?:bearer\s+)?
|
|
([A-Za-z0-9_./+=\-]{20,512})
|
|
''')
|
|
NON_FOUNDRY_KEY_PREFIXES = (
|
|
'sk-', 'sk_', 'sk-or-', 'xai-', 'ghp_', 'gho_', 'ghu_', 'ghs_', 'ghr_', 'github_pat_',
|
|
'glpat-', 'glrt-', 'hf_', 'AIza', 'AQ.', 'zai-', 'gsk_', 'r8_', 'nvapi-',
|
|
)
|
|
|
|
|
|
def keycheck_output_path(service, filename):
|
|
root = getattr(scan_config, 'keycheck_dir', None) or os.path.join(get_results_dir() or os.getcwd(), 'keychecks')
|
|
directory = os.path.join(root, service)
|
|
ensure_private_directory(directory, reject_reparse=True)
|
|
return os.path.join(directory, filename)
|
|
|
|
|
|
def _candidate_limits():
|
|
return {
|
|
'artifact_items': max(1, int(getattr(scan_config, 'keycheck_candidate_artifact_max_items', 2000))),
|
|
'artifact_bytes': max(1, int(getattr(scan_config, 'keycheck_candidate_artifact_max_bytes', 2 * 1024 * 1024))),
|
|
'file_items': max(1, int(getattr(scan_config, 'keycheck_candidate_file_max_items', 100000))),
|
|
'file_bytes': max(1, int(getattr(scan_config, 'keycheck_candidate_file_max_bytes', 32 * 1024 * 1024))),
|
|
'line_bytes': max(1, int(getattr(scan_config, 'keycheck_candidate_line_max_bytes', 8192))),
|
|
}
|
|
|
|
|
|
class CandidateQueueCapacityError(RuntimeError):
|
|
pass
|
|
|
|
|
|
_CANDIDATE_CHECKED_LEDGERS = {
|
|
'gem.txt': 'geminiChecked.txt',
|
|
'azureOpenAI.txt': 'azureChecked.txt',
|
|
'azureFoundry.txt': 'azureChecked.txt',
|
|
}
|
|
|
|
|
|
def _candidate_identity(value):
|
|
return str(value or '').strip().split('\t', 1)[0].strip()
|
|
|
|
|
|
def _read_bounded_candidate_rows(path, limits, description):
|
|
if not os.path.exists(path):
|
|
return [], set(), 0, 0
|
|
reject_reparse_components(path)
|
|
if os.path.getsize(path) > limits['file_bytes']:
|
|
raise RuntimeError(f'{description} exceeds its aggregate byte bound: {path}')
|
|
rows = []
|
|
identities = set()
|
|
total_bytes = 0
|
|
with open(path, 'rb') as handle:
|
|
for raw_line in handle:
|
|
total_bytes += len(raw_line)
|
|
if total_bytes > limits['file_bytes']:
|
|
raise RuntimeError(f'{description} exceeds its aggregate byte bound: {path}')
|
|
if len(raw_line) > limits['line_bytes']:
|
|
raise RuntimeError(f'{description} line exceeds its byte bound: {path}')
|
|
text = raw_line.rstrip(b'\r\n').decode('utf-8', errors='strict')
|
|
if not text:
|
|
continue
|
|
rows.append(text)
|
|
if len(rows) > limits['file_items']:
|
|
raise RuntimeError(f'{description} exceeds its aggregate item bound: {path}')
|
|
identity = _candidate_identity(text)
|
|
if identity:
|
|
identities.add(identity)
|
|
return rows, identities, len(rows), total_bytes
|
|
|
|
|
|
def _candidate_additions(lines, existing_identities, limits, checked_identities=()):
|
|
seen = set(existing_identities)
|
|
seen.update(checked_identities)
|
|
additions = []
|
|
added_bytes = 0
|
|
for value in lines:
|
|
line = str(value or '').strip()
|
|
if not line or '\n' in line or '\r' in line:
|
|
continue
|
|
encoded = (line + '\n').encode('utf-8')
|
|
identity = _candidate_identity(line)
|
|
if not identity or len(encoded) > limits['line_bytes'] or identity in seen:
|
|
continue
|
|
additions.append(encoded)
|
|
added_bytes += len(encoded)
|
|
seen.add(identity)
|
|
return additions, added_bytes
|
|
|
|
|
|
def _candidate_batch_fits(existing_items, existing_bytes, additions, added_bytes, limits):
|
|
return (
|
|
existing_items + len(additions) <= limits['file_items']
|
|
and existing_bytes + added_bytes <= limits['file_bytes']
|
|
)
|
|
|
|
|
|
def _rewrite_candidate_rows_atomic(path, rows):
|
|
temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.tmp'
|
|
descriptor = None
|
|
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL
|
|
if hasattr(os, 'O_BINARY'):
|
|
flags |= os.O_BINARY
|
|
try:
|
|
descriptor = os.open(temporary, flags, 0o600)
|
|
os.close(descriptor)
|
|
descriptor = None
|
|
harden_private_file(temporary)
|
|
with open(temporary, 'wb') as handle:
|
|
for row in rows:
|
|
handle.write((row + '\n').encode('utf-8'))
|
|
handle.flush()
|
|
os.fsync(handle.fileno())
|
|
harden_private_file(temporary)
|
|
durable_replace(temporary, path)
|
|
harden_private_file(path)
|
|
finally:
|
|
if descriptor is not None:
|
|
os.close(descriptor)
|
|
try:
|
|
if os.path.exists(temporary):
|
|
os.remove(temporary)
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
def _compact_checked_candidate_rows_unlocked(path, rows, existing_bytes, limits):
|
|
ledger_name = _CANDIDATE_CHECKED_LEDGERS.get(os.path.basename(path))
|
|
if not ledger_name:
|
|
return rows, set(), existing_bytes
|
|
checked_path = os.path.join(os.path.dirname(path), ledger_name)
|
|
checked_lock_path = f'{checked_path}.lock'
|
|
# Candidate locks are always outermost; checked-ledger writers never take them.
|
|
checked_lock = acquire_file_lock(checked_lock_path, stale_sec=120, timeout_sec=10)
|
|
try:
|
|
_, checked_identities, _, _ = _read_bounded_candidate_rows(
|
|
checked_path, limits, 'keycheck checked ledger',
|
|
)
|
|
finally:
|
|
release_file_lock(checked_lock, checked_lock_path)
|
|
retained = [row for row in rows if _candidate_identity(row) not in checked_identities]
|
|
if len(retained) != len(rows):
|
|
encoded_sizes = [len((row + '\n').encode('utf-8')) for row in retained]
|
|
retained_bytes = sum(encoded_sizes)
|
|
if retained_bytes > limits['file_bytes'] or any(
|
|
size > limits['line_bytes'] for size in encoded_sizes
|
|
):
|
|
raise RuntimeError(f'compacted keycheck candidate file would exceed its bounds: {path}')
|
|
_rewrite_candidate_rows_atomic(path, retained)
|
|
else:
|
|
retained_bytes = existing_bytes
|
|
return retained, checked_identities, retained_bytes
|
|
|
|
|
|
def _append_unique_lines_unlocked(path, lines, limits):
|
|
ensure_private_directory(os.path.dirname(path), reject_reparse=True)
|
|
rows, existing, existing_items, existing_bytes = _read_bounded_candidate_rows(
|
|
path, limits, 'keycheck candidate file',
|
|
)
|
|
additions, added_bytes = _candidate_additions(lines, existing, limits)
|
|
if not additions:
|
|
return 0
|
|
if not _candidate_batch_fits(existing_items, existing_bytes, additions, added_bytes, limits):
|
|
rows, checked, existing_bytes = _compact_checked_candidate_rows_unlocked(
|
|
path, rows, existing_bytes, limits,
|
|
)
|
|
existing = {_candidate_identity(row) for row in rows if _candidate_identity(row)}
|
|
existing_items = len(rows)
|
|
additions, added_bytes = _candidate_additions(lines, existing, limits, checked)
|
|
if not additions:
|
|
return 0
|
|
if not _candidate_batch_fits(existing_items, existing_bytes, additions, added_bytes, limits):
|
|
raise CandidateQueueCapacityError(
|
|
f'keycheck candidate queue has insufficient capacity for the complete offered batch: {path}'
|
|
)
|
|
with open(path, 'ab') as f:
|
|
f.write(b''.join(additions))
|
|
f.flush()
|
|
os.fsync(f.fileno())
|
|
harden_private_file(path)
|
|
return len(additions)
|
|
|
|
|
|
def append_unique_lines_locked(path, lines):
|
|
limits = _candidate_limits()
|
|
lock_path = f'{path}.lock'
|
|
lock = acquire_file_lock(lock_path, stale_sec=120, timeout_sec=10)
|
|
try:
|
|
return _append_unique_lines_unlocked(path, lines, limits)
|
|
finally:
|
|
release_file_lock(lock, lock_path)
|
|
|
|
|
|
def append_unique_line(path, line):
|
|
return bool(append_unique_lines_locked(path, [line]))
|
|
|
|
|
|
def append_unique_line_locked(path, line):
|
|
return bool(append_unique_lines_locked(path, [line]))
|
|
|
|
|
|
def _collect_candidate_line(batches, seen, budget, path, line):
|
|
limits = budget['limits']
|
|
text = str(line or '').strip()
|
|
encoded_size = len((text + '\n').encode('utf-8')) if text else 0
|
|
if not text or encoded_size > limits['line_bytes'] or text in seen.setdefault(path, set()):
|
|
return False
|
|
if budget['items'] >= limits['artifact_items'] or budget['bytes'] + encoded_size > limits['artifact_bytes']:
|
|
budget['truncated'] = True
|
|
return False
|
|
seen[path].add(text)
|
|
batches.setdefault(path, []).append(text)
|
|
budget['items'] += 1
|
|
budget['bytes'] += encoded_size
|
|
return True
|
|
|
|
|
|
def normalize_foundry_endpoint(value):
|
|
text = str(value or '').strip().strip('"\'`,;')
|
|
if not text:
|
|
return ''
|
|
split_text = text if re.match(r'(?i)^https?://', text) else 'https://' + text
|
|
try:
|
|
parsed = urlsplit(split_text)
|
|
host = parsed.netloc or parsed.path.split('/', 1)[0]
|
|
path = parsed.path if parsed.netloc else ('/' + parsed.path.split('/', 1)[1] if '/' in parsed.path else '')
|
|
except Exception:
|
|
host, path = re.sub(r'(?i)^https?://', '', text).split('/', 1)[0], ''
|
|
path = path.rstrip('.,;:)]}/')
|
|
terminal_routes = (
|
|
('/models/chat/completions', ''),
|
|
('/openai/v1/chat/completions', '/openai/v1'),
|
|
('/v1/chat/completions', '/v1'),
|
|
('/chat/completions', ''),
|
|
('/v1/models', '/v1'),
|
|
('/models', ''),
|
|
)
|
|
lower_path = path.lower()
|
|
for suffix, replacement in terminal_routes:
|
|
if lower_path.endswith(suffix):
|
|
path = path[:-len(suffix)] + replacement
|
|
break
|
|
return (host + path).strip('/').lower()
|
|
|
|
|
|
def dedupe_foundry_endpoints(endpoints):
|
|
normalized = []
|
|
for endpoint in endpoints or []:
|
|
endpoint = normalize_foundry_endpoint(endpoint)
|
|
if endpoint and endpoint not in normalized:
|
|
normalized.append(endpoint)
|
|
kept = []
|
|
for endpoint in sorted(normalized, key=len, reverse=True):
|
|
if any(other.startswith(endpoint + '/') for other in kept):
|
|
continue
|
|
kept.append(endpoint)
|
|
return list(reversed(kept))
|
|
|
|
|
|
def foundry_keyish(value):
|
|
text = re.sub(r'(?i)^bearer\s+', '', str(value or '').strip().strip('"\'`,;')).strip()
|
|
lower = text.lower()
|
|
if not (20 <= len(text) <= 512):
|
|
return False
|
|
if any(marker in lower for marker in ('http://', 'https://', '{{', '${', '<', 'azure.com')):
|
|
return False
|
|
if any(ch.isspace() for ch in text):
|
|
return False
|
|
if re.match(r'(?i)^(?:authorization|api[_-]?key|key|token|secret|credential|bearer)\s*[:=]', text):
|
|
return False
|
|
if text.startswith(NON_FOUNDRY_KEY_PREFIXES):
|
|
return False
|
|
return bool(re.search(r'[A-Za-z]', text) and re.search(r'[0-9]', text))
|
|
|
|
|
|
def is_foundry_detector(finding):
|
|
detector = str((finding or {}).get('DetectorName') or '').lower()
|
|
extra = (finding or {}).get('ExtraData') if isinstance((finding or {}).get('ExtraData'), dict) else {}
|
|
name = str(extra.get('name') or '').lower()
|
|
return detector.startswith('azurefoundry') or (detector == 'customregex' and name.startswith('azurefoundry'))
|
|
|
|
|
|
def foundry_finding_text(finding):
|
|
parts = []
|
|
for value in finding_raw_values(finding):
|
|
parts.append(value)
|
|
if is_foundry_detector(finding):
|
|
return '\n'.join(part for part in parts if part)
|
|
context = finding.get('ScannerContext') if isinstance(finding, dict) else None
|
|
if isinstance(context, dict):
|
|
parts.extend(str(context.get(item) or '') for item in ('nearby', 'file'))
|
|
postman_context = finding.get('PostmanContext') if isinstance(finding, dict) else None
|
|
if isinstance(postman_context, dict):
|
|
parts.extend(str(postman_context.get(item) or '') for item in ('endpoint', 'host', 'variable_name'))
|
|
extra = finding.get('ExtraData') if isinstance(finding, dict) else None
|
|
if isinstance(extra, dict):
|
|
parts.extend(str(value) for value in extra.values() if isinstance(value, str))
|
|
return '\n'.join(part for part in parts if part)
|
|
|
|
|
|
def foundry_candidate_keys(finding, text):
|
|
keys = []
|
|
if is_foundry_detector(finding):
|
|
for value in finding_raw_values(finding):
|
|
if foundry_keyish(value) and value not in keys:
|
|
keys.append(value.strip().strip('"\'`,;'))
|
|
for match in FOUNDRY_ASSIGNMENT_RE.findall(text or ''):
|
|
candidate = re.sub(r'(?i)^bearer\s+', '', str(match or '').strip().strip('"\'`,;')).strip()
|
|
if foundry_keyish(candidate) and candidate not in keys:
|
|
keys.append(candidate)
|
|
return keys[:5]
|
|
|
|
|
|
def foundry_candidate_origin(result, finding):
|
|
file_path, line_number = finding_source_location(finding)
|
|
location = f'{file_path}:{line_number}' if file_path and line_number else file_path or ''
|
|
target = result.get('target') or ''
|
|
scan_type = result.get('scan_type') or ''
|
|
return ':'.join(part for part in (scan_type, str(target), location) if part)
|
|
|
|
|
|
def write_foundry_keycheck_candidates_from_findings(result):
|
|
findings = result.get('findings') or []
|
|
if not findings:
|
|
return 0
|
|
azure_foundry_path = keycheck_output_path('azure', 'azureFoundry.txt')
|
|
batches = {}
|
|
seen = {}
|
|
budget = {'items': 0, 'bytes': 0, 'truncated': False, 'limits': _candidate_limits()}
|
|
for finding in findings:
|
|
if not is_foundry_detector(finding):
|
|
continue
|
|
text = foundry_finding_text(finding)
|
|
endpoints = []
|
|
for match in POSTMAN_FOUNDRY_ENDPOINT_RE.findall(text):
|
|
endpoint = normalize_foundry_endpoint(match)
|
|
if endpoint and endpoint not in endpoints:
|
|
endpoints.append(endpoint)
|
|
if not endpoints:
|
|
continue
|
|
endpoints = dedupe_foundry_endpoints(endpoints)
|
|
keys = foundry_candidate_keys(finding, text)
|
|
if not keys:
|
|
continue
|
|
origin = foundry_candidate_origin(result, finding)
|
|
for endpoint in endpoints[:5]:
|
|
for key in keys[:5]:
|
|
metadata = json.dumps({
|
|
'origin': origin,
|
|
'finding_uid': finding.get('finding_uid') or '',
|
|
}, ensure_ascii=False, separators=(',', ':'))
|
|
_collect_candidate_line(
|
|
batches, seen, budget, azure_foundry_path,
|
|
f'{endpoint}:{key}\t{metadata}',
|
|
)
|
|
if budget['truncated']:
|
|
break
|
|
if budget['truncated']:
|
|
break
|
|
if budget['truncated']:
|
|
break
|
|
if budget['truncated']:
|
|
logger.warning('Azure Foundry candidates reached the per-artifact bound; remaining values were capped')
|
|
return append_unique_lines_locked(azure_foundry_path, batches.get(azure_foundry_path, []))
|
|
|
|
|
|
def artifact_origin_label(target_data, cache_path):
|
|
origin = target_data.get('origin') if isinstance(target_data.get('origin'), dict) else {}
|
|
source = target_data.get('source') or origin.get('provider') or 'artifact'
|
|
repo = target_data.get('repo') or origin.get('repo') or ''
|
|
path = target_data.get('path') or origin.get('path') or os.path.basename(cache_path or '')
|
|
sha = target_data.get('sha') or origin.get('sha') or target_data.get('sha256') or ''
|
|
return f'{source}:{repo}:{path}:{sha}'
|
|
|
|
|
|
def context_values_for_pairing(contexts):
|
|
endpoints = []
|
|
values = []
|
|
for context in contexts or []:
|
|
value = str(context.get('value') or '').strip()
|
|
if not value or is_postman_placeholder(value):
|
|
continue
|
|
text = ' '.join([value, str(context.get('endpoint') or ''), str(context.get('host') or ''), str(context.get('key') or '')])
|
|
values.append((context, value, text))
|
|
for pattern in (POSTMAN_AZURE_OPENAI_ENDPOINT_RE, POSTMAN_FOUNDRY_ENDPOINT_RE):
|
|
for match in pattern.findall(text):
|
|
endpoint = normalize_foundry_endpoint(match) if pattern is POSTMAN_FOUNDRY_ENDPOINT_RE else str(match).lower().strip('/')
|
|
if endpoint not in endpoints:
|
|
endpoints.append(endpoint)
|
|
return endpoints, values
|
|
|
|
|
|
def write_structured_keycheck_candidates(cache_path, target_data):
|
|
contexts = load_postman_context(cache_path)
|
|
if not contexts:
|
|
return {}
|
|
context_value_by_path = {str(item.get('path') or ''): str(item.get('value') or '') for item in contexts}
|
|
endpoints, values = context_values_for_pairing(contexts)
|
|
origin = artifact_origin_label(target_data or {}, cache_path)
|
|
counts = {'gemini': 0, 'azure_openai': 0, 'azure_foundry': 0}
|
|
|
|
gemini_path = keycheck_output_path('gemini', 'gem.txt')
|
|
azure_openai_path = keycheck_output_path('azure', 'azureOpenAI.txt')
|
|
azure_foundry_path = keycheck_output_path('azure', 'azureFoundry.txt')
|
|
batches = {}
|
|
seen = {}
|
|
budget = {'items': 0, 'bytes': 0, 'truncated': False, 'limits': _candidate_limits()}
|
|
labels = {
|
|
gemini_path: 'gemini',
|
|
azure_openai_path: 'azure_openai',
|
|
azure_foundry_path: 'azure_foundry',
|
|
}
|
|
|
|
for context, value, text in values:
|
|
for key in POSTMAN_GEMINI_KEY_RE.findall(value):
|
|
_collect_candidate_line(
|
|
batches, seen, budget, gemini_path,
|
|
f'{key}\t{origin}\t{context.get("path") or ""}',
|
|
)
|
|
|
|
azure_key_match = POSTMAN_AZURE_OPENAI_KEY_RE.search(value)
|
|
if azure_key_match:
|
|
local_azure = [str(match).lower().strip('/') for match in POSTMAN_AZURE_OPENAI_ENDPOINT_RE.findall(text)]
|
|
azure_endpoints = local_azure or [endpoint for endpoint in endpoints if POSTMAN_AZURE_OPENAI_ENDPOINT_RE.fullmatch(endpoint)]
|
|
for endpoint in azure_endpoints[:5]:
|
|
_collect_candidate_line(
|
|
batches, seen, budget, azure_openai_path,
|
|
f'{endpoint}:{azure_key_match.group(0)}\t{origin}\t{context.get("path") or ""}',
|
|
)
|
|
|
|
sibling_name = ''
|
|
path = str(context.get('path') or '')
|
|
if path.endswith('.value'):
|
|
sibling_name = context_value_by_path.get(path[:-6] + '.key', '')
|
|
key_context = ' '.join([str(context.get('key') or ''), sibling_name]).lower()
|
|
foundry_value = re.sub(r'(?i)^bearer\s+', '', value.strip().strip('"\'`,;')).strip()
|
|
if foundry_keyish(foundry_value) and any(word in key_context for word in ('authorization', 'bearer', 'key', 'token', 'secret', 'api')):
|
|
foundry_endpoints = dedupe_foundry_endpoints(POSTMAN_FOUNDRY_ENDPOINT_RE.findall(text))
|
|
for endpoint in foundry_endpoints[:5]:
|
|
if endpoint:
|
|
_collect_candidate_line(
|
|
batches, seen, budget, azure_foundry_path,
|
|
f'{endpoint}:{foundry_value}\t{origin}\t{context.get("path") or ""}',
|
|
)
|
|
if budget['truncated']:
|
|
break
|
|
if budget['truncated']:
|
|
logger.warning('Structured keycheck candidates reached the per-artifact bound; remaining values were capped')
|
|
for path, lines in batches.items():
|
|
counts[labels[path]] = append_unique_lines_locked(path, lines)
|
|
return {key: value for key, value in counts.items() if value}
|
|
|
|
|
|
def _postman_substring_match(budget, needle, value):
|
|
if _context_budget_expired(budget):
|
|
return None
|
|
if budget['postman_comparisons'] >= budget['max_postman_comparisons']:
|
|
return None
|
|
budget['postman_comparisons'] += 1
|
|
return needle in value
|
|
|
|
|
|
def attach_postman_context(results, cache_path, budget=None):
|
|
budget = budget or context_enrichment_budget()
|
|
results.pop('structured_keycheck_pending', None)
|
|
try:
|
|
payload, complete, reason = _read_context_source(cache_path, budget)
|
|
if payload is None or not complete:
|
|
if reason in ('elapsed', 'source_bytes'):
|
|
dimension = 'elapsed-time' if reason == 'elapsed' else 'source-byte'
|
|
_add_context_warning(results, 'budget', f'{dimension} budget was exhausted; findings were retained')
|
|
else:
|
|
_add_context_warning(results, 'failure', 'Postman context could not be read; findings were retained')
|
|
return results
|
|
contexts = load_postman_context(cache_path, payload=payload)
|
|
except Exception:
|
|
logger.warning('Optional Postman context parsing failed; parsed findings were retained')
|
|
_add_context_warning(results, 'failure', 'Postman JSON context parsing failed; findings were retained')
|
|
return results
|
|
|
|
results['structured_keycheck_pending'] = True
|
|
if not contexts:
|
|
return results
|
|
|
|
try:
|
|
placeholder_count = 0
|
|
context_values = []
|
|
exact_values = {}
|
|
for index, context in enumerate(contexts):
|
|
if index % 256 == 0 and _context_budget_expired(budget):
|
|
_add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; findings were retained')
|
|
return results
|
|
if not isinstance(context, dict):
|
|
continue
|
|
value = str(context.get('value') or '')
|
|
context_values.append((context, value))
|
|
if value:
|
|
exact_values.setdefault(value, context)
|
|
placeholder_count += int(is_postman_placeholder(value))
|
|
results['postman_context'] = {
|
|
'context_count': len(contexts),
|
|
'placeholder_count': placeholder_count,
|
|
}
|
|
|
|
for finding in results.get('findings') or []:
|
|
if _context_budget_expired(budget):
|
|
_add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained')
|
|
return results
|
|
if not isinstance(finding, dict):
|
|
continue
|
|
if not _context_budget_claim_finding(budget, finding):
|
|
_add_context_warning(results, 'budget', 'finding-count budget was exhausted; remaining findings were retained')
|
|
return results
|
|
raw_values = finding_raw_values(finding)
|
|
matched = next((exact_values.get(raw) for raw in raw_values if raw and exact_values.get(raw)), None)
|
|
for raw in raw_values if not matched else ():
|
|
if not raw:
|
|
continue
|
|
for context, value in context_values:
|
|
comparison = _postman_substring_match(budget, raw, value)
|
|
if comparison is None:
|
|
_add_context_warning(results, 'budget', 'Postman comparison budget was exhausted; remaining findings were retained')
|
|
return results
|
|
if comparison:
|
|
matched = context
|
|
break
|
|
if matched:
|
|
break
|
|
if not matched and raw_values and raw_values[0][:8]:
|
|
prefix = raw_values[0][:8]
|
|
for context, value in context_values:
|
|
comparison = _postman_substring_match(budget, prefix, value)
|
|
if comparison is None:
|
|
_add_context_warning(results, 'budget', 'Postman comparison budget was exhausted; remaining findings were retained')
|
|
return results
|
|
if comparison:
|
|
matched = context
|
|
break
|
|
if not matched:
|
|
continue
|
|
raw_value = raw_values[0] if raw_values else matched.get('value')
|
|
kind = credential_kind_from_postman(matched, raw_value)
|
|
confidence = 'verified' if finding.get('Verified') else 'detector_match' if finding.get('DetectorName') else 'structured_complete' if kind != 'unknown' else 'context_only'
|
|
if kind == 'placeholder':
|
|
confidence = 'placeholder'
|
|
endpoint = sanitize_endpoint(matched.get('endpoint'))
|
|
host = sanitize_endpoint_host(endpoint) or sanitize_endpoint_host(matched.get('host'))
|
|
finding['PostmanContext'] = {
|
|
'provider': provider_from_postman(finding.get('DetectorName'), host, raw_value),
|
|
'credential_kind': kind,
|
|
'credential_confidence': confidence,
|
|
'context_location': matched.get('location'),
|
|
'variable_name': matched.get('key'),
|
|
'endpoint': endpoint,
|
|
'host': host,
|
|
'auth_type': matched.get('auth_type'),
|
|
'json_path': matched.get('path'),
|
|
'placeholder': is_postman_placeholder(matched.get('value')),
|
|
}
|
|
except Exception:
|
|
logger.warning('Optional Postman context matching failed; parsed findings were retained')
|
|
_add_context_warning(results, 'failure', 'Postman context matching failed; findings were retained')
|
|
return results
|
|
|
|
|
|
def scan_postman_target(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, max_artifact_size_mb=20, token=None):
|
|
data = parse_postman_target(target)
|
|
cache_path = data.get('cache_path') or data.get('local_path')
|
|
if not cache_path:
|
|
return {"findings": [], "errors": ["Postman target missing cached artifact"], "package": data.get('origin') or data}
|
|
try:
|
|
cache_path, declared_size = validate_postman_cache_artifact(data, max_artifact_size_mb)
|
|
except PostmanCacheTooLarge as exc:
|
|
return {"findings": [], "errors": [], "skipped": str(exc), "package": data.get('origin') or data}
|
|
except (OSError, PostmanCacheValidationError) as exc:
|
|
return {"findings": [], "errors": [f"Postman cache path rejected: {exc}"], "package": data.get('origin') or data}
|
|
_raise_if_scan_slot_fatal()
|
|
work_dir = create_command_work_dir()
|
|
_raise_if_scan_slot_fatal()
|
|
if not work_dir:
|
|
return {"findings": [], "errors": ["Unable to create Postman work dir"], "package": data.get('origin') or data}
|
|
try:
|
|
staged_path = os.path.join(work_dir, postman_stage_filename(data))
|
|
_raise_if_scan_slot_fatal()
|
|
shutil.copyfile(cache_path, staged_path)
|
|
_raise_if_scan_slot_fatal()
|
|
harden_private_file(staged_path)
|
|
cmd = [get_trufflehog_cmd(), 'filesystem', work_dir, '--json', '--no-update']
|
|
append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config)
|
|
results = {
|
|
"findings": [], "errors": [], "package": data.get('origin') or data,
|
|
"postman": data, "bytes": declared_size,
|
|
"postman_max_artifact_size_mb": int(max_artifact_size_mb or 0),
|
|
}
|
|
with run_command_streamed(cmd, timeout_sec) as output:
|
|
apply_trufflehog_diagnostics(results, output, output.returncode, 'postman')
|
|
append_trufflehog_findings(results, output.stdout_lines())
|
|
enrichment_budget = context_enrichment_budget()
|
|
attach_nearby_context(results, enrichment_budget)
|
|
artifact_path = str(data.get('path') or data.get('name') or data.get('url') or '').split('?', 1)[0].lower()
|
|
if str(data.get('kind') or '').lower() != 'bruno' or not artifact_path.endswith('.bru'):
|
|
attach_postman_context(results, staged_path, enrichment_budget)
|
|
return apply_finding_filters(results, cache_path)
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except Exception as e:
|
|
return {"findings": [], "errors": [f"Postman scan failed: {str(e)}"], "package": data.get('origin') or data}
|
|
finally:
|
|
cleanup_command_work_dir(work_dir)
|
|
|
|
|
|
def safe_path_component(value, max_len=120):
|
|
text = str(value or '')[:max_len]
|
|
return re.sub(r'[^A-Za-z0-9_.-]+', '_', text).strip('._') or 'item'
|
|
|
|
|
|
def run_filesystem_scan(root_dir, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None):
|
|
cmd = [get_trufflehog_cmd(), 'filesystem', root_dir, '--json', '--no-update']
|
|
append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config)
|
|
results = {"findings": [], "errors": []}
|
|
with run_command_streamed(cmd, timeout_sec) as output:
|
|
apply_trufflehog_diagnostics(results, output, output.returncode, 'filesystem')
|
|
append_trufflehog_findings(results, output.stdout_lines())
|
|
attach_nearby_context(results)
|
|
return apply_finding_filters(results, root_dir)
|
|
|
|
|
|
CI_ARTIFACT_ALLOWED_SUFFIXES = {
|
|
'.env', '.log', '.txt', '.json', '.yaml', '.yml', '.xml', '.html', '.lcov', '.sarif',
|
|
'.tfstate', '.tfplan', '.py', '.js', '.jsx', '.ts', '.tsx', '.sh', '.toml', '.ini',
|
|
'.cfg', '.conf', '.properties', '.ipynb', '.md', '.out', '.err', '.csv', '.tsv',
|
|
}
|
|
CI_ARTIFACT_ALLOWED_NAMES = {'dockerfile', 'makefile', 'procfile'}
|
|
CI_ARTIFACT_SKIP_DIRS = {
|
|
'.git', '.hg', '.svn', 'node_modules', '__pycache__', '.venv', 'venv', 'env',
|
|
'.mypy_cache', '.pytest_cache', '.tox', '.gradle', '.idea', '.vscode',
|
|
}
|
|
|
|
|
|
def ci_artifact_file_allowed(name, allowed_suffixes=None):
|
|
if allowed_suffixes is None:
|
|
return True
|
|
normalized = name.replace('\\', '/').strip('/')
|
|
parts = [part.lower() for part in normalized.split('/') if part]
|
|
if any(part in CI_ARTIFACT_SKIP_DIRS for part in parts[:-1]):
|
|
return False
|
|
base = parts[-1] if parts else ''
|
|
if base in CI_ARTIFACT_ALLOWED_NAMES or base.startswith('.env'):
|
|
return True
|
|
return any(base.endswith(suffix) for suffix in allowed_suffixes)
|
|
|
|
|
|
def safe_extract_zip(zip_path, destination, max_file_size_mb=20, max_files=1000, allowed_suffixes=None, max_total_size_mb=250):
|
|
_raise_if_scan_slot_fatal()
|
|
extracted = 0
|
|
max_bytes = int(max_file_size_mb or 0) * 1024 * 1024
|
|
max_total_bytes = int(max_total_size_mb or 0) * 1024 * 1024
|
|
total_bytes = 0
|
|
validate_zip_central_directory(zip_path, max_files)
|
|
with zipfile.ZipFile(zip_path) as archive:
|
|
for info in archive.infolist():
|
|
_raise_if_scan_slot_fatal()
|
|
if info.is_dir():
|
|
continue
|
|
if max_files and extracted >= max_files:
|
|
raise ValueError(f'Zip archive exceeds {max_files} extracted files')
|
|
if max_bytes and info.file_size > max_bytes:
|
|
raise ValueError(f'Zip member exceeds {max_file_size_mb} MB: {info.filename}')
|
|
if max_total_bytes and total_bytes + max(0, int(info.file_size or 0)) > max_total_bytes:
|
|
raise ValueError(f'Zip archive exceeds {max_total_size_mb} MB extracted')
|
|
name = info.filename.replace('\\', '/')
|
|
validate_archive_member_name(name)
|
|
if not ci_artifact_file_allowed(name, allowed_suffixes):
|
|
continue
|
|
target_path = os.path.abspath(os.path.normpath(os.path.join(destination, name)))
|
|
if os.path.commonpath([os.path.abspath(destination), target_path]) != os.path.abspath(destination):
|
|
continue
|
|
os.makedirs(os.path.dirname(target_path), exist_ok=True)
|
|
with archive.open(info) as src, open(target_path, 'wb') as dst:
|
|
while True:
|
|
_raise_if_scan_slot_fatal()
|
|
chunk = src.read(256 * 1024)
|
|
if not chunk:
|
|
break
|
|
dst.write(chunk)
|
|
_raise_if_scan_slot_fatal()
|
|
extracted += 1
|
|
total_bytes += max(0, int(info.file_size or 0))
|
|
return extracted
|
|
|
|
|
|
def remove_file_quiet(path):
|
|
try:
|
|
reject_reparse_components(path)
|
|
os.remove(path)
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class DownloadOutcome:
|
|
path: str
|
|
status_code: int
|
|
bytes_written: int
|
|
error: str = ''
|
|
|
|
@property
|
|
def ok(self):
|
|
return bool(self.path and not self.error)
|
|
|
|
|
|
def download_to_file(
|
|
url, destination, headers=None, timeout=20, max_size_mb=0,
|
|
max_size_bytes=None, byte_budget=None,
|
|
):
|
|
_raise_if_scan_slot_fatal()
|
|
destination = os.path.abspath(destination)
|
|
require_private_directory(os.path.dirname(destination), create=False)
|
|
reject_reparse_components(os.path.dirname(destination))
|
|
current_url = str(url)
|
|
current_headers = dict(headers or {})
|
|
configured_max = int(max_size_mb or 0) * 1024 * 1024
|
|
explicit_max = max(0, int(max_size_bytes or 0))
|
|
max_bytes = min(value for value in (configured_max, explicit_max) if value > 0) if configured_max and explicit_max else configured_max or explicit_max
|
|
if byte_budget is not None and int(byte_budget.get('remaining', 0)) <= 0:
|
|
return DownloadOutcome('', 0, 0, 'target byte budget exhausted')
|
|
for _ in range(6):
|
|
_raise_if_scan_slot_fatal()
|
|
try:
|
|
response = _direct_request(
|
|
'GET', current_url, headers=current_headers, timeout=timeout,
|
|
allow_redirects=False, stream=True,
|
|
)
|
|
except requests.RequestException as exc:
|
|
return DownloadOutcome('', 0, 0, str(exc))
|
|
if _scan_slot_fatal_event.is_set():
|
|
response.close()
|
|
_raise_if_scan_slot_fatal()
|
|
if response.status_code in (301, 302, 303, 307, 308) and response.headers.get('Location'):
|
|
next_url = urljoin(current_url, response.headers['Location'])
|
|
old_host = (urlsplit(current_url).hostname or '').lower()
|
|
parsed_next = urlsplit(next_url)
|
|
response.close()
|
|
if parsed_next.scheme != 'https':
|
|
return DownloadOutcome('', 0, 0, f'unsafe redirect scheme: {parsed_next.scheme}')
|
|
if (parsed_next.hostname or '').lower() != old_host:
|
|
current_headers = {
|
|
key: value for key, value in current_headers.items()
|
|
if key.lower() not in ('authorization', 'private-token', 'cookie', 'proxy-authorization')
|
|
}
|
|
current_url = next_url
|
|
continue
|
|
if response.status_code >= 400:
|
|
try:
|
|
prefix = bytearray()
|
|
for chunk in response.iter_content(chunk_size=300):
|
|
_raise_if_scan_slot_fatal()
|
|
prefix.extend(chunk[:max(0, 300 - len(prefix))])
|
|
if len(prefix) >= 300:
|
|
break
|
|
message = bytes(prefix).decode(response.encoding or 'utf-8', errors='replace')
|
|
finally:
|
|
response.close()
|
|
return DownloadOutcome('', response.status_code, 0, message)
|
|
content_length = response.headers.get('Content-Length')
|
|
if content_length:
|
|
try:
|
|
declared = int(content_length)
|
|
remaining = int(byte_budget.get('remaining', 0)) if byte_budget is not None else 0
|
|
if max_bytes and declared > max_bytes:
|
|
response.close()
|
|
return DownloadOutcome('', response.status_code, 0, f'response exceeds configured {max_bytes}-byte limit')
|
|
if byte_budget is not None and declared > remaining:
|
|
response.close()
|
|
return DownloadOutcome('', response.status_code, 0, 'target byte budget exhausted')
|
|
except ValueError:
|
|
pass
|
|
total = 0
|
|
temporary = f'{destination}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.partial'
|
|
descriptor = None
|
|
try:
|
|
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0)
|
|
descriptor = os.open(temporary, flags, 0o600)
|
|
os.close(descriptor)
|
|
descriptor = None
|
|
harden_private_file(temporary)
|
|
with open(temporary, 'wb', buffering=0) as output:
|
|
for chunk in response.iter_content(chunk_size=256 * 1024):
|
|
_raise_if_scan_slot_fatal()
|
|
if not chunk:
|
|
continue
|
|
if max_bytes and total + len(chunk) > max_bytes:
|
|
raise CommandOutputLimitError(f'object response exceeds configured {max_bytes}-byte limit')
|
|
if byte_budget is not None:
|
|
remaining = int(byte_budget.get('remaining', 0))
|
|
if len(chunk) > remaining:
|
|
byte_budget['remaining'] = 0
|
|
raise CommandOutputLimitError('target byte budget exhausted')
|
|
byte_budget['remaining'] = remaining - len(chunk)
|
|
output.write(chunk)
|
|
total += len(chunk)
|
|
output.flush()
|
|
os.fsync(output.fileno())
|
|
harden_private_file(temporary)
|
|
durable_replace(temporary, destination)
|
|
if not private_file_ready(destination):
|
|
raise OSError('download destination lost its private file identity')
|
|
return DownloadOutcome(destination, response.status_code, total)
|
|
except (CommandOutputLimitError, OSError, requests.RequestException) as exc:
|
|
return DownloadOutcome('', response.status_code, total, str(exc))
|
|
finally:
|
|
response.close()
|
|
if descriptor is not None:
|
|
os.close(descriptor)
|
|
if os.path.exists(temporary):
|
|
try:
|
|
os.remove(temporary)
|
|
except OSError:
|
|
pass
|
|
return DownloadOutcome('', 0, 0, 'too many redirects')
|
|
|
|
|
|
def github_api_get(url, token=None, timeout=20, stream=False):
|
|
response = api_request(
|
|
'GET', url, headers=github_headers(token), timeout=timeout, stream=stream,
|
|
)
|
|
if token and response.status_code in (401, 403):
|
|
response.close()
|
|
anonymous = api_request(
|
|
'GET', url, headers=github_headers(None), timeout=timeout, stream=stream,
|
|
)
|
|
if anonymous.status_code < 400:
|
|
return anonymous
|
|
response = anonymous
|
|
if response.status_code >= 400:
|
|
raise github_api_error(response)
|
|
return response
|
|
|
|
|
|
def bounded_response_json(response, max_bytes=8 * 1024 * 1024):
|
|
max_bytes = max(1, int(max_bytes))
|
|
declared = response.headers.get('Content-Length')
|
|
if declared:
|
|
try:
|
|
declared = int(declared)
|
|
except ValueError as exc:
|
|
response.close()
|
|
raise ApiRequestError('API JSON response has an invalid Content-Length') from exc
|
|
if declared > max_bytes:
|
|
response.close()
|
|
raise ApiRequestError(f'API JSON response exceeds {max_bytes} bytes')
|
|
payload = bytearray()
|
|
try:
|
|
for chunk in response.iter_content(chunk_size=64 * 1024):
|
|
if not chunk:
|
|
continue
|
|
if len(payload) + len(chunk) > max_bytes:
|
|
raise ApiRequestError(f'API JSON response exceeds {max_bytes} bytes')
|
|
payload.extend(chunk)
|
|
return json.loads(bytes(payload).decode('utf-8', errors='strict'))
|
|
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
|
|
raise ApiRequestError('API response is not bounded valid UTF-8 JSON') from exc
|
|
finally:
|
|
response.close()
|
|
|
|
|
|
def select_ci_runs(runs, runs_per_repo=5, lookback_days=30, failed_first=True):
|
|
cutoff = datetime.now(timezone.utc) - timedelta(days=int(lookback_days or 0)) if int(lookback_days or 0) > 0 else None
|
|
kept = []
|
|
for run in runs or []:
|
|
created = parse_postman_time(run.get('created_at'))
|
|
if cutoff and created and created < cutoff:
|
|
continue
|
|
kept.append(run)
|
|
def created_ts(item):
|
|
parsed = parse_postman_time(item.get('created_at'))
|
|
return parsed.timestamp() if parsed else 0
|
|
if failed_first:
|
|
kept.sort(key=lambda item: (0 if item.get('conclusion') not in ('success', None) else 1, -created_ts(item)))
|
|
else:
|
|
kept.sort(key=lambda item: -created_ts(item))
|
|
return kept[:max(1, int(runs_per_repo or 5))]
|
|
|
|
|
|
def scan_github_actions_repo(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, ci_runs_per_repo=5, ci_lookback_days=30, ci_max_log_archive_mb=50, ci_max_log_file_mb=20, ci_failed_first=True, ci_scan_artifacts=False, ci_max_artifacts_per_run=3, ci_max_artifact_archive_mb=50, ci_max_artifact_file_mb=10, ci_max_artifact_files=1000, ci_target_max_download_mb=500, fetch_timeout=20):
|
|
repo, repo_url = parse_github_repo_target(target)
|
|
if not repo:
|
|
return {"findings": [], "errors": ["Unable to parse GitHub repo target"]}
|
|
logger.info(f"Scanning GitHub Actions logs: {repo}")
|
|
_raise_if_scan_slot_fatal()
|
|
work_dir = create_command_work_dir()
|
|
_raise_if_scan_slot_fatal()
|
|
if not work_dir:
|
|
return {"findings": [], "errors": ["Unable to create GitHub Actions work dir"]}
|
|
try:
|
|
runs_url = f'https://api.github.com/repos/{repo}/actions/runs?per_page={max(1, min(100, int(ci_runs_per_repo or 5) * 3))}'
|
|
response = github_api_get(runs_url, token, fetch_timeout, stream=True)
|
|
payload = bounded_response_json(response)
|
|
if not isinstance(payload, dict) or 'workflow_runs' not in payload or not isinstance(payload.get('workflow_runs'), list):
|
|
raise ApiRequestError('invalid GitHub Actions runs payload')
|
|
runs = select_ci_runs(payload.get('workflow_runs') or [], ci_runs_per_repo, ci_lookback_days, ci_failed_first)
|
|
if not runs:
|
|
return {"findings": [], "errors": [], "skipped": "no recent workflow runs", "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url}}
|
|
max_archive_bytes = int(ci_max_log_archive_mb or 0) * 1024 * 1024
|
|
max_artifact_archive_bytes = int(ci_max_artifact_archive_mb or 0) * 1024 * 1024
|
|
extracted_total = 0
|
|
artifact_extracted_total = 0
|
|
download_failures = []
|
|
downloaded_total = 0
|
|
target_download_limit = int(ci_target_max_download_mb or 0) * 1024 * 1024
|
|
download_budget = {'remaining': target_download_limit} if target_download_limit else None
|
|
run_meta = []
|
|
for run in runs:
|
|
_raise_if_scan_slot_fatal()
|
|
if download_budget is not None and download_budget['remaining'] <= 0:
|
|
download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap')
|
|
break
|
|
run_id = run.get('id')
|
|
if not run_id:
|
|
continue
|
|
extracted = 0
|
|
artifacts_meta = []
|
|
logs_url = f'https://api.github.com/repos/{repo}/actions/runs/{run_id}/logs'
|
|
zip_path = os.path.join(work_dir, f'github_actions_{safe_path_component(repo)}_{run_id}.zip')
|
|
authenticated_status = None
|
|
try:
|
|
download = download_to_file(
|
|
logs_url, zip_path, github_headers(token), fetch_timeout, ci_max_log_archive_mb,
|
|
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
|
|
byte_budget=download_budget,
|
|
)
|
|
authenticated_status = download.status_code
|
|
except requests.RequestException as exc:
|
|
download = DownloadOutcome('', 0, 0, str(exc))
|
|
if not download.ok and token and download.status_code in (401, 403):
|
|
anonymous = download_to_file(
|
|
logs_url, zip_path, github_headers(None), fetch_timeout, ci_max_log_archive_mb,
|
|
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
|
|
byte_budget=download_budget,
|
|
)
|
|
if not anonymous.ok:
|
|
download = DownloadOutcome(
|
|
'', authenticated_status or anonymous.status_code, 0,
|
|
f'authenticated HTTP {authenticated_status}; anonymous HTTP '
|
|
f'{anonymous.status_code}: {anonymous.error}',
|
|
)
|
|
else:
|
|
download = anonymous
|
|
if download.ok:
|
|
downloaded_total += download.bytes_written
|
|
run_dir = os.path.join(work_dir, 'github_actions', safe_path_component(repo), str(run_id))
|
|
ensure_private_directory(run_dir, reject_reparse=True)
|
|
try:
|
|
extracted = safe_extract_zip(
|
|
zip_path, run_dir, ci_max_log_file_mb,
|
|
max_total_size_mb=ci_max_log_archive_mb,
|
|
)
|
|
harden_private_tree(run_dir)
|
|
except (zipfile.BadZipFile, ValueError) as exc:
|
|
logger.warning(f"GitHub Actions logs archive rejected for {repo} run {run_id}: {exc}")
|
|
download_failures.append(f'log archive rejected: {exc}')
|
|
extracted = 0
|
|
remove_file_quiet(zip_path)
|
|
else:
|
|
if download.status_code not in (404, 410):
|
|
download_failures.append(f'log download HTTP {download.status_code}: {download.error}')
|
|
run_dir = os.path.join(work_dir, 'github_actions', safe_path_component(repo), str(run_id))
|
|
ensure_private_directory(run_dir, reject_reparse=True)
|
|
if ci_scan_artifacts:
|
|
artifacts_url = f'https://api.github.com/repos/{repo}/actions/runs/{run_id}/artifacts?per_page={max(1, min(100, int(ci_max_artifacts_per_run or 3)))}'
|
|
try:
|
|
artifacts_response = github_api_get(
|
|
artifacts_url, token, fetch_timeout, stream=True,
|
|
)
|
|
artifact_payload = bounded_response_json(artifacts_response)
|
|
if not isinstance(artifact_payload, dict) or 'artifacts' not in artifact_payload or not isinstance(artifact_payload.get('artifacts'), list):
|
|
raise ApiRequestError('invalid GitHub Actions artifacts payload')
|
|
artifacts = artifact_payload.get('artifacts') or []
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except Exception as exc:
|
|
logger.warning(f"Unable to list GitHub Actions artifacts for {repo} run {run_id}: {exc}")
|
|
download_failures.append(f'artifact listing failed: {exc}')
|
|
artifacts = []
|
|
for artifact in artifacts[:max(0, int(ci_max_artifacts_per_run or 3))]:
|
|
_raise_if_scan_slot_fatal()
|
|
if download_budget is not None and download_budget['remaining'] <= 0:
|
|
download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap')
|
|
break
|
|
if artifact.get('expired'):
|
|
continue
|
|
size = int(artifact.get('size_in_bytes') or 0)
|
|
if max_artifact_archive_bytes and size and size > max_artifact_archive_bytes:
|
|
download_failures.append(f'artifact declared size exceeds {ci_max_artifact_archive_mb} MB')
|
|
continue
|
|
download_url = artifact.get('archive_download_url')
|
|
if not download_url:
|
|
continue
|
|
artifact_id = artifact.get('id') or safe_path_component(artifact.get('name'))
|
|
artifact_zip = os.path.join(work_dir, f'github_actions_artifact_{safe_path_component(repo)}_{run_id}_{artifact_id}.zip')
|
|
download = download_to_file(
|
|
download_url, artifact_zip, github_headers(token), fetch_timeout, ci_max_artifact_archive_mb,
|
|
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
|
|
byte_budget=download_budget,
|
|
)
|
|
authenticated_status = download.status_code
|
|
if not download.ok and token and download.status_code in (401, 403):
|
|
anonymous = download_to_file(
|
|
download_url, artifact_zip, github_headers(None), fetch_timeout, ci_max_artifact_archive_mb,
|
|
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
|
|
byte_budget=download_budget,
|
|
)
|
|
if not anonymous.ok:
|
|
download = DownloadOutcome(
|
|
'', authenticated_status or anonymous.status_code, 0,
|
|
f'authenticated HTTP {authenticated_status}; anonymous HTTP '
|
|
f'{anonymous.status_code}: {anonymous.error}',
|
|
)
|
|
else:
|
|
download = anonymous
|
|
if not download.ok:
|
|
if download.status_code not in (404, 410):
|
|
logger.warning(f"GitHub Actions artifact download HTTP {download.status_code} for {repo} run {run_id}: {download.error}")
|
|
download_failures.append(f'artifact download HTTP {download.status_code}: {download.error}')
|
|
continue
|
|
downloaded_total += download.bytes_written
|
|
artifact_dir = os.path.join(run_dir, 'artifacts', safe_path_component(artifact.get('name') or artifact_id))
|
|
ensure_private_directory(artifact_dir, reject_reparse=True)
|
|
try:
|
|
artifact_files = safe_extract_zip(
|
|
artifact_zip,
|
|
artifact_dir,
|
|
ci_max_artifact_file_mb,
|
|
ci_max_artifact_files,
|
|
CI_ARTIFACT_ALLOWED_SUFFIXES,
|
|
max_total_size_mb=ci_max_artifact_archive_mb,
|
|
)
|
|
harden_private_tree(artifact_dir)
|
|
except (zipfile.BadZipFile, ValueError) as exc:
|
|
logger.warning(f"GitHub Actions artifact rejected for {repo} run {run_id}: {artifact.get('name')}: {exc}")
|
|
download_failures.append(f'artifact archive rejected: {exc}')
|
|
artifact_files = 0
|
|
remove_file_quiet(artifact_zip)
|
|
artifact_extracted_total += artifact_files
|
|
artifacts_meta.append({
|
|
'artifact_id': artifact.get('id'),
|
|
'name': artifact.get('name'),
|
|
'size_in_bytes': size,
|
|
'expired': artifact.get('expired'),
|
|
'extracted_files': artifact_files,
|
|
})
|
|
extracted_total += extracted
|
|
run_meta.append({
|
|
'run_id': run_id,
|
|
'run_number': run.get('run_number'),
|
|
'workflow_name': run.get('name'),
|
|
'status': run.get('status'),
|
|
'conclusion': run.get('conclusion'),
|
|
'created_at': run.get('created_at'),
|
|
'updated_at': run.get('updated_at'),
|
|
'extracted_files': extracted,
|
|
'artifacts': artifacts_meta,
|
|
})
|
|
if extracted_total == 0 and artifact_extracted_total == 0:
|
|
if download_failures:
|
|
text = '; '.join(download_failures[:5])
|
|
all_failures = '; '.join(download_failures)
|
|
auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any(
|
|
token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit')
|
|
)
|
|
return {
|
|
"findings": [], "errors": [text], "error_class": 'source_auth' if auth_failure else 'remote_transient',
|
|
"retryable": True, "source_failure": auth_failure,
|
|
"source_failure_category": 'auth_forbidden' if auth_failure else 'remote_transient',
|
|
"source_failure_auth_related": auth_failure,
|
|
"package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url, "runs": run_meta},
|
|
}
|
|
return {"findings": [], "errors": [], "skipped": "no downloadable workflow logs or artifacts", "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url, "runs": run_meta}}
|
|
_raise_if_scan_slot_fatal()
|
|
results = run_filesystem_scan(work_dir, timeout_sec, detectors, exclude_detectors, no_verification, trufflehog_config)
|
|
results['package'] = {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url, "runs": run_meta, "log_files": extracted_total, "artifact_files": artifact_extracted_total}
|
|
if download_failures:
|
|
text = '; '.join(download_failures[:5])
|
|
all_failures = '; '.join(download_failures)
|
|
auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any(
|
|
token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit')
|
|
)
|
|
results['errors'] = list(results.get('errors') or []) + [text]
|
|
results['error_class'] = 'source_auth' if auth_failure else 'remote_transient'
|
|
results['retryable'] = True
|
|
results['source_failure'] = auth_failure
|
|
results['source_failure_category'] = 'auth_forbidden' if auth_failure else 'remote_transient'
|
|
results['source_failure_auth_related'] = auth_failure
|
|
return results
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except RateLimitError as exc:
|
|
category = getattr(exc, 'category', '')
|
|
if category == 'not_found':
|
|
return {"findings": [], "errors": [], "skipped": str(exc), "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url}}
|
|
return {
|
|
"findings": [], "errors": [str(exc)], "error_class": 'source_auth' if getattr(exc, 'auth_related', True) else 'remote_transient',
|
|
"retryable": True, "source_failure": True,
|
|
"source_failure_category": category or 'rate_limit',
|
|
"source_failure_auth_related": bool(getattr(exc, 'auth_related', True)),
|
|
"package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url},
|
|
**({'_diagnostic_http': exc.diagnostic_http} if exc.diagnostic_http else {}),
|
|
}
|
|
except Exception as exc:
|
|
return {
|
|
"findings": [], "errors": [f"GitHub Actions acquisition failed: {exc}"],
|
|
"error_class": "remote_transient", "retryable": True, "source_failure": True,
|
|
"source_failure_category": "network", "source_failure_auth_related": False,
|
|
"package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url},
|
|
}
|
|
finally:
|
|
cleanup_command_work_dir(work_dir)
|
|
|
|
|
|
def gitlab_api_get(url, token=None, timeout=20, stream=False):
|
|
response = api_request(
|
|
'GET', url, headers=gitlab_headers(token), timeout=timeout, stream=stream,
|
|
)
|
|
if token and response.status_code in (401, 403):
|
|
response.close()
|
|
anonymous = api_request(
|
|
'GET', url, headers=gitlab_headers(None), timeout=timeout, stream=stream,
|
|
)
|
|
if anonymous.status_code < 400:
|
|
return anonymous
|
|
response = anonymous
|
|
if response.status_code >= 400:
|
|
raise gitlab_api_error(response)
|
|
return response
|
|
|
|
|
|
def scan_gitlab_ci_project(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, ci_pipelines_per_project=5, ci_jobs_per_pipeline=20, ci_lookback_days=30, ci_max_trace_mb=20, ci_scan_artifacts=False, ci_max_artifacts_per_pipeline=5, ci_max_artifact_archive_mb=50, ci_max_artifact_file_mb=10, ci_max_artifact_files=1000, ci_target_max_download_mb=500, fetch_timeout=20):
|
|
project, project_url = parse_gitlab_project_target(target)
|
|
if not project:
|
|
return {"findings": [], "errors": ["Unable to parse GitLab project target"]}
|
|
logger.info(f"Scanning GitLab CI traces: {project}")
|
|
_raise_if_scan_slot_fatal()
|
|
work_dir = create_command_work_dir()
|
|
_raise_if_scan_slot_fatal()
|
|
if not work_dir:
|
|
return {"findings": [], "errors": ["Unable to create GitLab CI work dir"]}
|
|
try:
|
|
encoded = quote(project, safe='')
|
|
pipeline_url = f'https://gitlab.com/api/v4/projects/{encoded}/pipelines?per_page={max(1, min(100, int(ci_pipelines_per_project or 5)))}&order_by=updated_at&sort=desc'
|
|
response = gitlab_api_get(pipeline_url, token, fetch_timeout, stream=True)
|
|
pipelines = bounded_response_json(response) or []
|
|
if not isinstance(pipelines, list):
|
|
raise ApiRequestError('invalid GitLab pipelines payload')
|
|
cutoff = datetime.now(timezone.utc) - timedelta(days=int(ci_lookback_days or 0)) if int(ci_lookback_days or 0) > 0 else None
|
|
selected = []
|
|
for pipeline in pipelines:
|
|
updated = parse_postman_time(pipeline.get('updated_at') or pipeline.get('created_at'))
|
|
if cutoff and updated and updated < cutoff:
|
|
continue
|
|
selected.append(pipeline)
|
|
if len(selected) >= int(ci_pipelines_per_project or 5):
|
|
break
|
|
if not selected:
|
|
return {"findings": [], "errors": [], "skipped": "no recent pipelines", "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url}}
|
|
trace_dir = os.path.join(work_dir, 'gitlab_ci', safe_path_component(project))
|
|
ensure_private_directory(trace_dir, reject_reparse=True)
|
|
max_trace_bytes = int(ci_max_trace_mb or 0) * 1024 * 1024
|
|
max_artifact_archive_bytes = int(ci_max_artifact_archive_mb or 0) * 1024 * 1024
|
|
pipeline_meta = []
|
|
written = 0
|
|
artifact_extracted_total = 0
|
|
download_failures = []
|
|
downloaded_total = 0
|
|
target_download_limit = int(ci_target_max_download_mb or 0) * 1024 * 1024
|
|
download_budget = {'remaining': target_download_limit} if target_download_limit else None
|
|
for pipeline in selected:
|
|
_raise_if_scan_slot_fatal()
|
|
if download_budget is not None and download_budget['remaining'] <= 0:
|
|
download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap')
|
|
break
|
|
pipeline_id = pipeline.get('id')
|
|
if not pipeline_id:
|
|
continue
|
|
jobs_url = f'https://gitlab.com/api/v4/projects/{encoded}/pipelines/{pipeline_id}/jobs?per_page={max(1, min(100, int(ci_jobs_per_pipeline or 20)))}'
|
|
try:
|
|
jobs_response = gitlab_api_get(jobs_url, token, fetch_timeout, stream=True)
|
|
jobs = bounded_response_json(jobs_response) or []
|
|
if not isinstance(jobs, list):
|
|
raise ApiRequestError('invalid GitLab jobs payload')
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except Exception as exc:
|
|
logger.warning(f"Unable to list GitLab CI jobs for {project} pipeline {pipeline_id}: {exc}")
|
|
download_failures.append(f'job listing failed: {exc}')
|
|
jobs = []
|
|
job_meta = []
|
|
artifacts_seen = 0
|
|
for job in jobs[:max(1, int(ci_jobs_per_pipeline or 20))]:
|
|
_raise_if_scan_slot_fatal()
|
|
if download_budget is not None and download_budget['remaining'] <= 0:
|
|
download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap')
|
|
break
|
|
job_id = job.get('id')
|
|
if not job_id:
|
|
continue
|
|
artifact_meta = None
|
|
trace_written = False
|
|
trace_url = f'https://gitlab.com/api/v4/projects/{encoded}/jobs/{job_id}/trace'
|
|
filename = f"pipeline_{pipeline_id}_job_{job_id}_{safe_path_component(job.get('name'))}.log"
|
|
trace_path = os.path.join(trace_dir, filename)
|
|
try:
|
|
trace_download = download_to_file(
|
|
trace_url, trace_path, gitlab_headers(token), fetch_timeout, ci_max_trace_mb,
|
|
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
|
|
byte_budget=download_budget,
|
|
)
|
|
except requests.RequestException as exc:
|
|
logger.warning(f"GitLab CI trace download failed for {project} job {job_id}: {exc}")
|
|
download_failures.append(f'trace download failed: {exc}')
|
|
trace_download = DownloadOutcome('', 0, 0, str(exc))
|
|
authenticated_status = trace_download.status_code
|
|
if not trace_download.ok and token and trace_download.status_code in (401, 403):
|
|
anonymous = download_to_file(
|
|
trace_url, trace_path, gitlab_headers(None), fetch_timeout, ci_max_trace_mb,
|
|
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
|
|
byte_budget=download_budget,
|
|
)
|
|
if not anonymous.ok:
|
|
trace_download = DownloadOutcome(
|
|
'', authenticated_status or anonymous.status_code, 0,
|
|
f'authenticated HTTP {authenticated_status}; anonymous HTTP '
|
|
f'{anonymous.status_code}: {anonymous.error}',
|
|
)
|
|
else:
|
|
trace_download = anonymous
|
|
if trace_download.ok:
|
|
downloaded_total += trace_download.bytes_written
|
|
written += 1
|
|
trace_written = True
|
|
elif trace_download.status_code not in (404, 410):
|
|
download_failures.append(
|
|
f'trace download HTTP {trace_download.status_code}: {trace_download.error}'
|
|
)
|
|
if ci_scan_artifacts and artifacts_seen < int(ci_max_artifacts_per_pipeline or 5):
|
|
if download_budget is not None and download_budget['remaining'] <= 0:
|
|
download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap')
|
|
job_meta.append({'job_id': job_id, 'name': job.get('name'), 'status': job.get('status'), 'stage': job.get('stage'), 'trace': trace_written, 'artifact': None})
|
|
break
|
|
artifact_file = job.get('artifacts_file') if isinstance(job.get('artifacts_file'), dict) else {}
|
|
artifact_size = int(artifact_file.get('size') or 0)
|
|
if artifact_file and max_artifact_archive_bytes and artifact_size and artifact_size > max_artifact_archive_bytes:
|
|
download_failures.append(f'artifact declared size exceeds {ci_max_artifact_archive_mb} MB')
|
|
artifact_file = {}
|
|
if artifact_file and (not max_artifact_archive_bytes or not artifact_size or artifact_size <= max_artifact_archive_bytes):
|
|
artifacts_url = f'https://gitlab.com/api/v4/projects/{encoded}/jobs/{job_id}/artifacts'
|
|
artifact_zip = os.path.join(work_dir, f'gitlab_ci_artifact_{safe_path_component(project)}_{job_id}.zip')
|
|
download = download_to_file(
|
|
artifacts_url, artifact_zip, gitlab_headers(token), fetch_timeout, ci_max_artifact_archive_mb,
|
|
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
|
|
byte_budget=download_budget,
|
|
)
|
|
authenticated_status = download.status_code
|
|
if not download.ok and token and download.status_code in (401, 403):
|
|
anonymous = download_to_file(
|
|
artifacts_url, artifact_zip, gitlab_headers(None), fetch_timeout, ci_max_artifact_archive_mb,
|
|
max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None,
|
|
byte_budget=download_budget,
|
|
)
|
|
if not anonymous.ok:
|
|
download = DownloadOutcome(
|
|
'', authenticated_status or anonymous.status_code, 0,
|
|
f'authenticated HTTP {authenticated_status}; anonymous HTTP '
|
|
f'{anonymous.status_code}: {anonymous.error}',
|
|
)
|
|
else:
|
|
download = anonymous
|
|
if download.ok:
|
|
downloaded_total += download.bytes_written
|
|
artifact_dir = os.path.join(trace_dir, 'artifacts', f'pipeline_{pipeline_id}', f'job_{job_id}')
|
|
ensure_private_directory(artifact_dir, reject_reparse=True)
|
|
try:
|
|
artifact_files = safe_extract_zip(
|
|
artifact_zip,
|
|
artifact_dir,
|
|
ci_max_artifact_file_mb,
|
|
ci_max_artifact_files,
|
|
CI_ARTIFACT_ALLOWED_SUFFIXES,
|
|
max_total_size_mb=ci_max_artifact_archive_mb,
|
|
)
|
|
harden_private_tree(artifact_dir)
|
|
except (zipfile.BadZipFile, ValueError) as exc:
|
|
logger.warning(f"GitLab CI artifact rejected for {project} job {job_id}: {exc}")
|
|
download_failures.append(f'artifact archive rejected: {exc}')
|
|
artifact_files = 0
|
|
remove_file_quiet(artifact_zip)
|
|
artifact_extracted_total += artifact_files
|
|
artifacts_seen += 1
|
|
artifact_meta = {
|
|
'filename': artifact_file.get('filename'),
|
|
'size': artifact_size,
|
|
'extracted_files': artifact_files,
|
|
}
|
|
else:
|
|
if download.status_code not in (404, 410):
|
|
logger.warning(f"GitLab CI artifact download HTTP {download.status_code} for {project} job {job_id}: {download.error}")
|
|
download_failures.append(f'artifact download HTTP {download.status_code}: {download.error}')
|
|
job_meta.append({'job_id': job_id, 'name': job.get('name'), 'status': job.get('status'), 'stage': job.get('stage'), 'trace': trace_written, 'artifact': artifact_meta})
|
|
pipeline_meta.append({'pipeline_id': pipeline_id, 'status': pipeline.get('status'), 'ref': pipeline.get('ref'), 'updated_at': pipeline.get('updated_at'), 'jobs': job_meta})
|
|
if written == 0 and artifact_extracted_total == 0:
|
|
if download_failures:
|
|
text = '; '.join(download_failures[:5])
|
|
all_failures = '; '.join(download_failures)
|
|
auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any(
|
|
token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit')
|
|
)
|
|
return {
|
|
"findings": [], "errors": [text], "error_class": 'source_auth' if auth_failure else 'remote_transient',
|
|
"retryable": True, "source_failure": auth_failure,
|
|
"source_failure_category": 'auth_forbidden' if auth_failure else 'remote_transient',
|
|
"source_failure_auth_related": auth_failure,
|
|
"package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url, "pipelines": pipeline_meta},
|
|
}
|
|
return {"findings": [], "errors": [], "skipped": "no downloadable job traces or artifacts", "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url, "pipelines": pipeline_meta}}
|
|
_raise_if_scan_slot_fatal()
|
|
results = run_filesystem_scan(work_dir, timeout_sec, detectors, exclude_detectors, no_verification, trufflehog_config)
|
|
results['package'] = {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url, "pipelines": pipeline_meta, "trace_files": written, "artifact_files": artifact_extracted_total}
|
|
if download_failures:
|
|
text = '; '.join(download_failures[:5])
|
|
all_failures = '; '.join(download_failures)
|
|
auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any(
|
|
token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit')
|
|
)
|
|
results['errors'] = list(results.get('errors') or []) + [text]
|
|
results['error_class'] = 'source_auth' if auth_failure else 'remote_transient'
|
|
results['retryable'] = True
|
|
results['source_failure'] = auth_failure
|
|
results['source_failure_category'] = 'auth_forbidden' if auth_failure else 'remote_transient'
|
|
results['source_failure_auth_related'] = auth_failure
|
|
return results
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except RateLimitError as exc:
|
|
category = getattr(exc, 'category', '')
|
|
if category == 'not_found':
|
|
return {"findings": [], "errors": [], "skipped": str(exc), "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url}}
|
|
return {
|
|
"findings": [], "errors": [str(exc)], "error_class": 'source_auth' if getattr(exc, 'auth_related', True) else 'remote_transient',
|
|
"retryable": True, "source_failure": True,
|
|
"source_failure_category": category or 'rate_limit',
|
|
"source_failure_auth_related": bool(getattr(exc, 'auth_related', True)),
|
|
"package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url},
|
|
**({'_diagnostic_http': exc.diagnostic_http} if exc.diagnostic_http else {}),
|
|
}
|
|
except Exception as exc:
|
|
return {
|
|
"findings": [], "errors": [f"GitLab CI acquisition failed: {exc}"],
|
|
"error_class": "remote_transient", "retryable": True, "source_failure": True,
|
|
"source_failure_category": "network", "source_failure_auth_related": False,
|
|
"package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url},
|
|
}
|
|
finally:
|
|
cleanup_command_work_dir(work_dir)
|
|
|
|
def send_webhook_notification(webhook_url, finding, target):
|
|
"""Send webhook notification for a finding"""
|
|
if not webhook_url:
|
|
return
|
|
|
|
try:
|
|
payload = {
|
|
"text": f"Secret Found in {target}",
|
|
"attachments": [{
|
|
"color": "danger",
|
|
"fields": [
|
|
{"title": "Detector", "value": finding.get('DetectorName', 'Unknown'), "short": True},
|
|
{"title": "Target", "value": target, "short": True},
|
|
{"title": "Verified", "value": str(finding.get('Verified', False)), "short": True}
|
|
]
|
|
}]
|
|
}
|
|
|
|
response = requests.post(webhook_url, json=payload, timeout=10)
|
|
response.raise_for_status()
|
|
logger.info(f"Webhook notification sent for {target}")
|
|
except Exception as e:
|
|
logger.error(f"Failed to send webhook notification: {str(e)}")
|
|
|
|
|
|
def _captured_http_failure(error):
|
|
current = error
|
|
seen = set()
|
|
for _index in range(8):
|
|
if current is None or id(current) in seen:
|
|
break
|
|
seen.add(id(current))
|
|
direct_status = getattr(current, 'status_code', None)
|
|
if type(direct_status) is int:
|
|
direct_body = getattr(current, 'body', b'')
|
|
result = {
|
|
'status_code': direct_status,
|
|
'content_type': getattr(current, 'content_type', None),
|
|
'request_id': getattr(current, 'request_id', None),
|
|
'operation': getattr(current, 'operation', 'provider-api'),
|
|
'headers_b64': base64.b64encode(json.dumps(
|
|
getattr(current, 'headers', None) or {},
|
|
ensure_ascii=True, sort_keys=True, separators=(',', ':'),
|
|
).encode('ascii')).decode('ascii'),
|
|
}
|
|
result['body_capture_truncated'] = bool(getattr(
|
|
current, 'body_capture_truncated', False
|
|
))
|
|
if direct_body is not None:
|
|
if isinstance(direct_body, str):
|
|
direct_body = direct_body.encode('utf-8')
|
|
if not isinstance(direct_body, bytes):
|
|
direct_body = bytes(direct_body or b'')
|
|
result['body_b64'] = base64.b64encode(direct_body).decode('ascii')
|
|
original_size = getattr(current, 'body_original_size', None)
|
|
stored_size = getattr(current, 'body_stored_size', None)
|
|
body_sha256 = getattr(current, 'body_sha256', None)
|
|
if original_size is None or stored_size is None or body_sha256 is None:
|
|
material = make_body_material(direct_body)
|
|
original_size = material.original_size
|
|
stored_size = material.stored_size
|
|
body_sha256 = material.sha256
|
|
result['body_capture_truncated'] = material.truncated
|
|
result.update({
|
|
'body_original_size': original_size,
|
|
'body_stored_size': stored_size,
|
|
'body_sha256': body_sha256,
|
|
})
|
|
return result
|
|
response = getattr(current, 'response', None)
|
|
status = getattr(response, 'status_code', None)
|
|
if type(status) is int:
|
|
body_material = getattr(response, '_truf_diagnostic_body_material', None)
|
|
if body_material is None and (
|
|
not getattr(response, 'raw', None)
|
|
or getattr(response, '_content_consumed', False)
|
|
):
|
|
try:
|
|
body = getattr(response, 'content', b'')
|
|
except RuntimeError:
|
|
body = None
|
|
if isinstance(body, str):
|
|
body = body.encode('utf-8')
|
|
if body is not None and not isinstance(body, bytes):
|
|
body = bytes(body)
|
|
if body is not None:
|
|
body_material = make_body_material(body)
|
|
headers = getattr(response, 'headers', {}) or {}
|
|
result = {
|
|
'status_code': status,
|
|
'content_type': str(headers.get('Content-Type') or '') or None,
|
|
'request_id': str(
|
|
headers.get('X-Request-Id') or headers.get('X-GitHub-Request-Id') or ''
|
|
) or None,
|
|
'operation': 'provider-request',
|
|
'headers_b64': base64.b64encode(json.dumps(
|
|
{str(key): str(value) for key, value in headers.items()},
|
|
ensure_ascii=True, sort_keys=True, separators=(',', ':'),
|
|
).encode('ascii')).decode('ascii'),
|
|
'body_capture_truncated': False,
|
|
}
|
|
if body_material is not None:
|
|
body = diagnostic_material_bytes(body_material)
|
|
result.update({
|
|
'body_b64': base64.b64encode(body).decode('ascii'),
|
|
'body_original_size': body_material.original_size,
|
|
'body_stored_size': body_material.stored_size,
|
|
'body_sha256': body_material.sha256,
|
|
'body_capture_truncated': body_material.truncated,
|
|
})
|
|
return result
|
|
current = getattr(current, '__cause__', None) or getattr(
|
|
current, '__context__', None
|
|
)
|
|
return None
|
|
|
|
|
|
def _captured_http_body_material(payload, capture):
|
|
material = make_body_material(payload)
|
|
metadata_fields = (
|
|
'body_original_size', 'body_stored_size', 'body_sha256',
|
|
)
|
|
if not any(name in capture for name in metadata_fields):
|
|
if capture.get('body_capture_truncated'):
|
|
raise ValueError('truncated diagnostic HTTP material lacks capture metadata')
|
|
return material
|
|
if (
|
|
not all(name in capture for name in metadata_fields)
|
|
or type(capture.get('body_capture_truncated')) is not bool
|
|
or isinstance(capture['body_original_size'], bool)
|
|
or not isinstance(capture['body_original_size'], int)
|
|
or isinstance(capture['body_stored_size'], bool)
|
|
or not isinstance(capture['body_stored_size'], int)
|
|
or capture['body_stored_size'] != len(payload)
|
|
or capture['body_original_size'] < capture['body_stored_size']
|
|
or capture['body_capture_truncated']
|
|
!= (capture['body_original_size'] > capture['body_stored_size'])
|
|
or not isinstance(capture['body_sha256'], str)
|
|
or re.fullmatch(r'[a-f0-9]{64}', capture['body_sha256']) is None
|
|
or (
|
|
not capture['body_capture_truncated']
|
|
and capture['body_sha256'] != hashlib.sha256(payload).hexdigest()
|
|
)
|
|
):
|
|
raise ValueError('captured diagnostic HTTP material metadata is invalid')
|
|
return replace(
|
|
material,
|
|
original_size=capture['body_original_size'],
|
|
stored_size=capture['body_stored_size'],
|
|
sha256=capture['body_sha256'],
|
|
truncated=capture['body_capture_truncated'],
|
|
)
|
|
|
|
|
|
def scan_target_result(target, scan_type, scan_event_id, scan_kwargs=None):
|
|
scan_kwargs = dict(scan_kwargs or {})
|
|
scan_started_at = datetime.now(timezone.utc).isoformat()
|
|
started = time.perf_counter()
|
|
try:
|
|
direct_kind = _client_remote_execution_kind.get()
|
|
expected_direct_kind = {
|
|
'docker': 'docker_direct_v1',
|
|
'huggingface': 'huggingface_space_v1',
|
|
}.get(scan_type)
|
|
if (
|
|
_client_scan_manifest.get() is not None
|
|
and expected_direct_kind is not None
|
|
and direct_kind != expected_direct_kind
|
|
):
|
|
raise RuntimeError(
|
|
'remote direct scan lacks its execution authority'
|
|
)
|
|
if direct_kind is not None and direct_kind != expected_direct_kind:
|
|
raise RuntimeError('remote direct execution platform changed')
|
|
if scan_type in ('git', 'github', 'github_archive', 'gitlab'):
|
|
provider = 'github' if scan_type == 'github_archive' else scan_type if scan_type in ('github', 'gitlab') else None
|
|
result = scan_git_repo(target, provider=provider, **scan_kwargs)
|
|
elif scan_type == 'docker':
|
|
docker_layer_work = scan_kwargs.pop('docker_layer_work', None)
|
|
if docker_layer_work is not None:
|
|
scan_kwargs.pop('trufflehog_concurrency', None)
|
|
scan_kwargs.pop('docker_recovery_limits', None)
|
|
scan_kwargs.pop('docker_recovery_min_free_bytes', None)
|
|
try:
|
|
result = scan_docker_layer_plan(
|
|
target, docker_layer_work, **scan_kwargs,
|
|
)
|
|
except (ScanSlotFatalError, DockerLayerInfrastructureError):
|
|
raise
|
|
except Exception as exc:
|
|
raise DockerLayerInfrastructureError(
|
|
'scanner_infrastructure',
|
|
'Docker layer scanner infrastructure failed',
|
|
category='source_resource',
|
|
) from exc
|
|
else:
|
|
result = scan_docker_image(
|
|
target,
|
|
config_dir=(
|
|
None if direct_kind == 'docker_direct_v1'
|
|
else docker_token_manager.get_next_config()
|
|
),
|
|
**scan_kwargs,
|
|
)
|
|
elif scan_type == 'huggingface':
|
|
result = scan_huggingface_space(target, **scan_kwargs)
|
|
elif scan_type == 'npm':
|
|
result = scan_npm_package(target, **scan_kwargs)
|
|
elif scan_type == 'pypi':
|
|
result = scan_pypi_package(target, **scan_kwargs)
|
|
elif scan_type == 'package_git':
|
|
result = scan_package_git_repo(target, **scan_kwargs)
|
|
elif scan_type in ('postman', 'github_gists', 'github_archive_files'):
|
|
result = scan_postman_target(target, **scan_kwargs)
|
|
elif scan_type == 'github_actions':
|
|
result = scan_github_actions_repo(target, **scan_kwargs)
|
|
elif scan_type == 'gitlab_ci':
|
|
result = scan_gitlab_ci_project(target, **scan_kwargs)
|
|
else:
|
|
result = {'findings': [], 'errors': [f'Unknown scan type: {scan_type}']}
|
|
except (ScanSlotFatalError, DockerLayerInfrastructureError):
|
|
raise
|
|
except Exception as exc:
|
|
result = {'findings': [], 'errors': [str(exc)]}
|
|
http_failure = _captured_http_failure(exc)
|
|
if http_failure is not None:
|
|
result['_diagnostic_http'] = http_failure
|
|
apply_result_error_scope(result)
|
|
result['target'] = target
|
|
result['scan_type'] = scan_type
|
|
result['scan_event_id'] = str(scan_event_id)
|
|
result['scan_started_at'] = scan_started_at
|
|
result['duration_sec'] = time.perf_counter() - started
|
|
result['timestamp'] = datetime.now(timezone.utc).isoformat()
|
|
assign_finding_uids(result)
|
|
return result
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class StagedResult:
|
|
target: str
|
|
scan_event_id: str
|
|
bundle_id: str
|
|
reservation_id: int
|
|
scan_event_hash: str
|
|
actual_bytes: int
|
|
relative_path: str
|
|
frame_count: int
|
|
finding_count: int
|
|
error_count: int
|
|
candidate_count: int
|
|
queue_status: str
|
|
source_failure: bool
|
|
source_failure_category: str
|
|
source_failure_auth_related: bool
|
|
first_error: str
|
|
|
|
def as_dict(self):
|
|
return dict(self.__dict__)
|
|
|
|
|
|
def stage_result_bundle(
|
|
result, reservation, bundle_root, scan_options, queue_disposition,
|
|
candidate_max_items=2000, candidate_max_bytes=2 * 1024 * 1024,
|
|
require_s_drive=False, fault=None, diagnostic_slot_id=0,
|
|
diagnostic_attempt=1,
|
|
):
|
|
reservation = reservation if isinstance(reservation, BundleReservation) else BundleReservation.from_mapping(reservation)
|
|
if not isinstance(result, dict):
|
|
raise ValueError('scan result must be an object')
|
|
canonical_reservation_target = normalize_target(
|
|
reservation.target, reservation.platform,
|
|
)
|
|
if not reservation.normalized_target:
|
|
reservation = replace(
|
|
reservation, normalized_target=canonical_reservation_target,
|
|
)
|
|
|
|
def optional_integer_identity(name, expected):
|
|
if name not in result or result[name] is None:
|
|
return
|
|
value = result[name]
|
|
if isinstance(value, bool):
|
|
raise ValueError('scan result identity does not match its reservation')
|
|
try:
|
|
value = int(value)
|
|
except (TypeError, ValueError, OverflowError):
|
|
raise ValueError('scan result identity does not match its reservation') from None
|
|
if value != int(expected):
|
|
raise ValueError('scan result identity does not match its reservation')
|
|
|
|
optional_integer_identity('reservation_id', reservation.reservation_id)
|
|
optional_integer_identity('result_reservation_id', reservation.reservation_id)
|
|
optional_integer_identity('queue_id', reservation.queue_id)
|
|
for name, expected in (
|
|
('bundle_id', reservation.bundle_id),
|
|
('source', reservation.source),
|
|
('platform', reservation.platform),
|
|
('query', reservation.query),
|
|
('normalized_target', reservation.normalized_target),
|
|
):
|
|
if name in result and result[name] is not None and str(result[name]) != str(expected):
|
|
raise ValueError('scan result identity does not match its reservation')
|
|
if (
|
|
str(result.get('scan_event_id') or '') != reservation.scan_event_id
|
|
or str(result.get('target') or '') != reservation.target
|
|
or str(result.get('scan_type') or '') != reservation.platform
|
|
or normalize_target(result.get('target'), reservation.platform)
|
|
!= canonical_reservation_target
|
|
or reservation.normalized_target != canonical_reservation_target
|
|
):
|
|
raise ValueError('scan result identity does not match its reservation')
|
|
|
|
strip_nearby_context_for_persistence(result)
|
|
findings = result.get('findings') or []
|
|
errors = result.get('errors') or []
|
|
candidates = []
|
|
candidate_bytes = 0
|
|
candidate_identities = set()
|
|
candidate_truncated = False
|
|
|
|
def add_candidate(candidate, attribution):
|
|
nonlocal candidate_bytes, candidate_truncated
|
|
identity = (
|
|
candidate.service, candidate.credential_hash,
|
|
'' if candidate.service == 'provider_resolver' else str(
|
|
(attribution or {}).get('finding_uid') or (attribution or {}).get('origin') or ''
|
|
),
|
|
)
|
|
if identity in candidate_identities:
|
|
return
|
|
frame = candidate.as_frame(attribution)
|
|
encoded_size = len(json.dumps(
|
|
frame, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str,
|
|
).encode('utf-8'))
|
|
if len(candidates) >= max(0, int(candidate_max_items)):
|
|
candidate_truncated = True
|
|
return
|
|
if candidate_bytes + encoded_size > max(0, int(candidate_max_bytes)):
|
|
candidate_truncated = True
|
|
return
|
|
candidate_identities.add(identity)
|
|
candidates.append(frame)
|
|
candidate_bytes += encoded_size
|
|
|
|
for finding in findings:
|
|
attribution = {
|
|
'finding_uid': str(finding.get('finding_uid') or ''),
|
|
'detector_name': str(finding.get('DetectorName') or finding.get('DetectorType') or ''),
|
|
}
|
|
try:
|
|
for candidate in extract_candidates(finding, attribution):
|
|
add_candidate(candidate, attribution)
|
|
except ValueError as exc:
|
|
result.setdefault('warnings', []).append(
|
|
f'Optional keycheck candidate was omitted: {type(exc).__name__}'
|
|
)
|
|
result['degraded'] = True
|
|
if result.get('structured_keycheck_pending') and isinstance(result.get('postman'), dict):
|
|
try:
|
|
postman_data = result['postman']
|
|
cache_path, _ = validate_postman_cache_artifact(
|
|
postman_data,
|
|
int(result.get('postman_max_artifact_size_mb') or 20),
|
|
expected_size=result.get('bytes'),
|
|
)
|
|
contexts = load_postman_context(cache_path)
|
|
origin = artifact_origin_label(postman_data, cache_path)
|
|
attribution = {'origin': origin}
|
|
for candidate in extract_structured_candidates({
|
|
'contexts': contexts, 'origin': origin,
|
|
}, attribution):
|
|
structured_origin = candidate.metadata.get('structured_origin') or origin
|
|
add_candidate(candidate, {'origin': structured_origin})
|
|
except Exception as exc:
|
|
result.setdefault('warnings', []).append(
|
|
f'Optional structured keycheck extraction failed: {type(exc).__name__}'
|
|
)
|
|
result['degraded'] = True
|
|
if candidate_truncated:
|
|
result.setdefault('warnings', []).append(
|
|
'Keycheck candidates reached their pre-reserved per-event bound; remaining candidates were omitted'
|
|
)
|
|
result['degraded'] = True
|
|
status = target_status(result)
|
|
first_error = ''
|
|
for error in errors:
|
|
first_error = next((line.strip() for line in str(error).splitlines() if line.strip()), '')
|
|
if first_error:
|
|
break
|
|
diagnostics = list(result.get('diagnostics') or ())
|
|
if errors and not diagnostics:
|
|
timestamp = str(result.get('timestamp') or result.get('scan_started_at') or '')
|
|
try:
|
|
occurred = datetime.fromisoformat(timestamp.replace('Z', '+00:00'))
|
|
except ValueError:
|
|
occurred = datetime.now(timezone.utc)
|
|
if occurred.tzinfo is None:
|
|
occurred = datetime.now(timezone.utc)
|
|
occurred_at = occurred.astimezone(timezone.utc).isoformat(
|
|
timespec='milliseconds'
|
|
).replace('+00:00', 'Z')
|
|
scan_meta = result.get('scan_meta') or {}
|
|
timed_out = bool(scan_meta.get('command_timed_out')) or result.get('error_class') == 'timeout'
|
|
category_name = str(
|
|
result.get('source_failure_category') or result.get('error_class') or ''
|
|
).lower()
|
|
category = {
|
|
'source_auth': DiagnosticCategory.AUTHORIZATION,
|
|
'auth_forbidden': DiagnosticCategory.AUTHORIZATION,
|
|
'rate_limit': DiagnosticCategory.RATE_LIMIT,
|
|
'not_found': DiagnosticCategory.NOT_FOUND,
|
|
'network': DiagnosticCategory.NETWORK,
|
|
'remote_transient': DiagnosticCategory.NETWORK,
|
|
'timeout': DiagnosticCategory.TIMEOUT,
|
|
'source_resource': DiagnosticCategory.STORAGE,
|
|
'storage': DiagnosticCategory.STORAGE,
|
|
}.get(category_name, DiagnosticCategory.SCANNER)
|
|
phase_name = str(scan_meta.get('timed_out_phase') or WorkerPhase.SCANNING.value)
|
|
try:
|
|
diagnostic_phase = WorkerPhase(phase_name)
|
|
except ValueError:
|
|
diagnostic_phase = WorkerPhase.SCANNING
|
|
scan_outcome = {
|
|
'clean': ScanOutcome.CLEAN,
|
|
'found': ScanOutcome.FOUND,
|
|
'degraded': ScanOutcome.DEGRADED,
|
|
'error': ScanOutcome.ERROR,
|
|
'skipped': ScanOutcome.SKIPPED,
|
|
}.get(status, ScanOutcome.ERROR)
|
|
try:
|
|
raw_stdout = base64.b64decode(
|
|
str(result['_diagnostic_raw_stdout_b64']).encode('ascii'),
|
|
validate=True,
|
|
) if '_diagnostic_raw_stdout_b64' in result else None
|
|
raw_stderr = base64.b64decode(
|
|
str(result['_diagnostic_raw_stderr_b64']).encode('ascii'),
|
|
validate=True,
|
|
) if '_diagnostic_raw_stderr_b64' in result else None
|
|
except (UnicodeEncodeError, ValueError, binascii.Error) as exc:
|
|
raise ValueError('captured diagnostic process material is invalid') from exc
|
|
stdout_limit = stderr_limit = 0
|
|
if raw_stdout is not None or raw_stderr is not None:
|
|
desired = [
|
|
min(len(raw_stdout), MAX_DIAGNOSTIC_LOG_BYTES)
|
|
if raw_stdout is not None else 0,
|
|
min(len(raw_stderr), MAX_DIAGNOSTIC_LOG_BYTES)
|
|
if raw_stderr is not None else 0,
|
|
]
|
|
total = sum(desired)
|
|
if total <= MAX_DIAGNOSTIC_LOG_BYTES:
|
|
stdout_limit, stderr_limit = desired
|
|
elif total:
|
|
stdout_limit = (
|
|
MAX_DIAGNOSTIC_LOG_BYTES * desired[0]
|
|
) // total
|
|
stderr_limit = MAX_DIAGNOSTIC_LOG_BYTES - stdout_limit
|
|
process = None
|
|
if raw_stdout is not None or raw_stderr is not None:
|
|
process = DiagnosticProcessContext(
|
|
name='trufflehog',
|
|
exit_code=(
|
|
int(scan_meta['trufflehog_returncode'])
|
|
if type(scan_meta.get('trufflehog_returncode')) is int else None
|
|
),
|
|
signal=None,
|
|
timed_out=timed_out,
|
|
stdout=(
|
|
make_log_material(raw_stdout, maximum=stdout_limit)
|
|
if raw_stdout is not None else None
|
|
),
|
|
stderr=(
|
|
make_log_material(raw_stderr, maximum=stderr_limit)
|
|
if raw_stderr is not None else None
|
|
),
|
|
)
|
|
http_value = result.get('_diagnostic_http')
|
|
http = None
|
|
if isinstance(http_value, dict) and type(http_value.get('status_code')) is int:
|
|
raw_body = None
|
|
raw_headers = None
|
|
if 'body_b64' in http_value:
|
|
try:
|
|
raw_body = base64.b64decode(
|
|
str(http_value['body_b64']).encode('ascii'),
|
|
validate=True,
|
|
)
|
|
except (UnicodeEncodeError, ValueError, binascii.Error) as exc:
|
|
raise ValueError('captured diagnostic HTTP material is invalid') from exc
|
|
if 'headers_b64' in http_value:
|
|
try:
|
|
raw_headers = base64.b64decode(
|
|
str(http_value['headers_b64']).encode('ascii'),
|
|
validate=True,
|
|
)
|
|
except (UnicodeEncodeError, ValueError, binascii.Error) as exc:
|
|
raise ValueError('captured diagnostic HTTP headers are invalid') from exc
|
|
http = DiagnosticHTTPContext(
|
|
operation=str(http_value.get('operation') or 'provider-request'),
|
|
status_code=http_value['status_code'],
|
|
content_type=http_value.get('content_type'),
|
|
request_id=http_value.get('request_id'),
|
|
body=(
|
|
_captured_http_body_material(raw_body, http_value)
|
|
if raw_body is not None else None
|
|
),
|
|
headers=(
|
|
make_body_material(raw_headers)
|
|
if raw_headers is not None else None
|
|
),
|
|
)
|
|
transformation = str(result.get('_diagnostic_stderr_transformation') or (
|
|
'HTTP status and bounded body evidence preserve captured values; parsed '
|
|
'response headers were serialized as a deterministic mapping because raw '
|
|
'wire order and casing are unavailable'
|
|
+ (
|
|
'; response body capture reached its explicit byte bound'
|
|
if http_value and http_value.get('body_capture_truncated') else ''
|
|
)
|
|
if http is not None else
|
|
'raw provider/process material was unavailable; canonical diagnostic '
|
|
'was projected from the existing legacy error representation'
|
|
))
|
|
fingerprint = hashlib.sha256(
|
|
('\n'.join(str(error) for error in errors)).encode('utf-8')
|
|
).hexdigest()
|
|
kind = (
|
|
DiagnosticKind.PROVIDER_HTTP if http is not None
|
|
else DiagnosticKind.SCANNER_PROCESS if process is not None
|
|
else DiagnosticKind.EXCEPTION
|
|
)
|
|
diagnostics.append(build_diagnostic_envelope(
|
|
occurrence_id=f'{reservation.scan_event_id}:scan-result',
|
|
reservation_id=reservation.reservation_id,
|
|
scan_event_id=reservation.scan_event_id,
|
|
slot_id=int(diagnostic_slot_id),
|
|
source=reservation.source,
|
|
phase=diagnostic_phase,
|
|
kind=kind,
|
|
category=(DiagnosticCategory.TIMEOUT if timed_out else category),
|
|
code=('scan.stage_timeout' if timed_out else 'scan.result_error'),
|
|
summary=(first_error or 'scan returned one or more errors')[:1000],
|
|
retryable=bool(result.get('retryable', False)),
|
|
attempt=max(1, int(diagnostic_attempt)),
|
|
assignment_outcome=AssignmentOutcome.ACCEPTED,
|
|
scan_outcome=scan_outcome,
|
|
occurred_at=occurred_at,
|
|
captured_at=occurred_at,
|
|
http=http,
|
|
process=process,
|
|
exception=DiagnosticExceptionContext(
|
|
type='truf.diagnostic.TechnicalTransformation',
|
|
message=transformation,
|
|
fingerprint=fingerprint,
|
|
),
|
|
))
|
|
metadata = {
|
|
key: value for key, value in result.items()
|
|
if key not in ('findings', 'errors', 'diagnostics')
|
|
and not key.startswith('_diagnostic_')
|
|
}
|
|
metadata.update({
|
|
'reservation_id': reservation.reservation_id,
|
|
'queue_id': reservation.queue_id,
|
|
'bundle_id': reservation.bundle_id,
|
|
'source': reservation.source,
|
|
'platform': reservation.platform,
|
|
'query': reservation.query,
|
|
'normalized_target': reservation.normalized_target,
|
|
'status': status,
|
|
'findings_count': len(findings),
|
|
'verified_findings_count': sum(1 for finding in findings if finding.get('Verified')),
|
|
'error_count': len(errors),
|
|
'first_error_summary': first_error[:500],
|
|
'scan_options': dict(scan_options or {}),
|
|
'derived_postman_targets': list(result.get('postman_targets') or ()),
|
|
**dict(queue_disposition or {}),
|
|
})
|
|
with ResultBundleWriter.open(
|
|
bundle_root, reservation, fault=fault, require_s_drive=require_s_drive,
|
|
) as writer:
|
|
for finding in findings:
|
|
writer.write_finding(finding)
|
|
for error in errors:
|
|
writer.write_error(error)
|
|
for diagnostic in diagnostics:
|
|
writer.write_diagnostic(diagnostic)
|
|
for candidate in candidates:
|
|
writer.write_candidate(candidate)
|
|
commit = writer.finish(metadata)
|
|
findings.clear()
|
|
errors.clear()
|
|
diagnostics.clear()
|
|
candidates.clear()
|
|
return StagedResult(
|
|
target=str(result.get('target') or ''),
|
|
scan_event_id=commit.scan_event_id,
|
|
bundle_id=commit.bundle_id,
|
|
reservation_id=commit.reservation_id,
|
|
scan_event_hash=commit.scan_event_hash,
|
|
actual_bytes=commit.actual_bytes,
|
|
relative_path=commit.relative_path,
|
|
frame_count=commit.frame_count,
|
|
finding_count=commit.finding_count,
|
|
error_count=commit.error_count,
|
|
candidate_count=commit.candidate_count,
|
|
queue_status=str(metadata.get('queue_status') or ''),
|
|
source_failure=bool(result.get('source_failure')),
|
|
source_failure_category=str(result.get('source_failure_category') or ''),
|
|
source_failure_auth_related=bool(result.get('source_failure_auth_related')),
|
|
first_error=first_error[:500],
|
|
)
|
|
|
|
|
|
class ResultSinkError(RuntimeError):
|
|
def __init__(self, failures, results):
|
|
self.failures = list(failures)
|
|
self.results = list(results)
|
|
targets = ', '.join(str(target) for target, _ in self.failures[:5])
|
|
super().__init__(f'result sink failed for {len(self.failures)} completed target(s): {targets}')
|
|
|
|
|
|
def scan_targets_batch(
|
|
targets,
|
|
scan_type,
|
|
progress_callback=None,
|
|
max_workers=4,
|
|
persist_results=True,
|
|
result_sink=None,
|
|
scan_slot_leases=None,
|
|
sink_within_scan_slot=False,
|
|
**kwargs,
|
|
):
|
|
"""Scan multiple targets with parallel processing and progress tracking"""
|
|
from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait
|
|
import threading
|
|
|
|
_raise_if_scan_slot_fatal()
|
|
|
|
results = []
|
|
total = len(targets) if hasattr(targets, '__len__') else None
|
|
completed = 0
|
|
results_lock = threading.Lock()
|
|
sink_failures = []
|
|
provided_leases = list(scan_slot_leases or [])
|
|
if provided_leases and len(provided_leases) != total:
|
|
raise ValueError('one pre-acquired scan slot lease is required for every target')
|
|
|
|
def apply_worker_sink(result):
|
|
if not sink_within_scan_slot or result_sink is None:
|
|
return result
|
|
try:
|
|
result_sink(result)
|
|
result['_result_sink_applied'] = True
|
|
except Exception as exc:
|
|
result['_result_sink_exception'] = exc
|
|
return result
|
|
|
|
def scan_single_target(target, scan_event_id, provided_lease=None):
|
|
"""Scan a single target"""
|
|
_raise_if_scan_slot_fatal()
|
|
scan_started_at = datetime.now(timezone.utc).isoformat()
|
|
started = time.perf_counter()
|
|
try:
|
|
with scan_slot_scope(
|
|
['scan-target', scan_type], kwargs.get('timeout_sec'), lease=provided_lease,
|
|
):
|
|
try:
|
|
_raise_if_scan_slot_fatal()
|
|
result = scan_target_result(target, scan_type, scan_event_id, kwargs)
|
|
return apply_worker_sink(result)
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except Exception as e:
|
|
result = {
|
|
"target": target,
|
|
"scan_type": scan_type,
|
|
"scan_event_id": scan_event_id,
|
|
"scan_started_at": scan_started_at,
|
|
"duration_sec": time.perf_counter() - started,
|
|
"timestamp": datetime.now(timezone.utc).isoformat(),
|
|
"findings": [],
|
|
"errors": [str(e)]
|
|
}
|
|
return apply_worker_sink(apply_result_error_scope(result))
|
|
except ScanSlotFatalError:
|
|
raise
|
|
except Exception as e:
|
|
result = {
|
|
"target": target,
|
|
"scan_type": scan_type,
|
|
"scan_event_id": scan_event_id,
|
|
"scan_started_at": scan_started_at,
|
|
"duration_sec": time.perf_counter() - started,
|
|
"timestamp": datetime.now(timezone.utc).isoformat(),
|
|
"findings": [],
|
|
"errors": [str(e)]
|
|
}
|
|
return apply_worker_sink(apply_result_error_scope(result))
|
|
|
|
max_workers = max(1, int(max_workers or 1))
|
|
executor = ThreadPoolExecutor(max_workers=max_workers)
|
|
future_to_target = {}
|
|
target_iterator = iter(enumerate(targets))
|
|
|
|
def submit_next():
|
|
try:
|
|
index, target = next(target_iterator)
|
|
except StopIteration:
|
|
return False
|
|
provided_lease = provided_leases[index] if provided_leases else None
|
|
future_to_target[
|
|
executor.submit(scan_single_target, target, str(uuid.uuid4()), provided_lease)
|
|
] = target
|
|
return True
|
|
|
|
try:
|
|
for _ in range(max_workers):
|
|
_raise_if_scan_slot_fatal()
|
|
if not submit_next():
|
|
break
|
|
while future_to_target:
|
|
_raise_if_scan_slot_fatal()
|
|
done, _ = wait(tuple(future_to_target), timeout=0.2, return_when=FIRST_COMPLETED)
|
|
for future in done:
|
|
target = future_to_target.pop(future)
|
|
result = future.result()
|
|
|
|
with results_lock:
|
|
results.append(result)
|
|
persistence_error = result.pop('_result_sink_exception', None)
|
|
sink_applied = bool(result.pop('_result_sink_applied', False))
|
|
if result_sink is not None and not sink_applied and persistence_error is None:
|
|
try:
|
|
result_sink(result)
|
|
except Exception as exc:
|
|
persistence_error = exc
|
|
if persist_results:
|
|
try:
|
|
if not save_scan_result(result):
|
|
raise RuntimeError('save_scan_result returned false')
|
|
except Exception as exc:
|
|
persistence_error = persistence_error or exc
|
|
if persistence_error is not None:
|
|
result['persistence_failure'] = str(persistence_error)[:500]
|
|
sink_failures.append((target, persistence_error))
|
|
logger.error(f'Persistence failed for completed target {target}: {persistence_error}')
|
|
submit_next()
|
|
continue
|
|
with results_lock:
|
|
completed += 1
|
|
findings_count = len(result.get('findings', []))
|
|
errors_count = len(result.get('errors', []))
|
|
skipped = bool(result.get('skipped'))
|
|
logger.info(
|
|
f"Completed {completed}/{total}: {target} "
|
|
f"(findings={findings_count}, errors={errors_count}, skipped={skipped})"
|
|
)
|
|
|
|
if progress_callback:
|
|
progress_callback(completed, total, target)
|
|
submit_next()
|
|
except ScanSlotFatalError as exc:
|
|
_set_scan_slot_fatal(str(exc))
|
|
for future in future_to_target:
|
|
future.cancel()
|
|
wait(tuple(future_to_target), timeout=1.0)
|
|
try:
|
|
executor.shutdown(wait=False, cancel_futures=True)
|
|
except TypeError:
|
|
executor.shutdown(wait=False)
|
|
executor = None
|
|
raise
|
|
finally:
|
|
if executor is not None:
|
|
executor.shutdown(wait=True)
|
|
for lease in provided_leases:
|
|
if lease.heartbeat_thread is None:
|
|
lease.release()
|
|
|
|
if progress_callback:
|
|
progress_callback(
|
|
completed,
|
|
total,
|
|
"Scan completed" if not sink_failures else "Scan completed with persistence failures",
|
|
)
|
|
|
|
if sink_failures:
|
|
raise ResultSinkError(sink_failures, results)
|
|
return results
|
|
|
|
def extract_trufflehog_error_lines(output):
|
|
error_lines = []
|
|
for index, line in enumerate(_iter_output_lines(output), 1):
|
|
if index > 2000:
|
|
break
|
|
line = line.strip()
|
|
if not line:
|
|
continue
|
|
try:
|
|
payload = json.loads(line)
|
|
level = str(payload.get('level', '')).lower()
|
|
message = str(payload.get('msg', '')).lower()
|
|
if message == 'error cleaning temporary artifacts':
|
|
continue
|
|
if 'error' in level or 'error' in message or payload.get('error'):
|
|
error_lines.append(line)
|
|
except json.JSONDecodeError:
|
|
if 'error' in line.lower() or 'failed' in line.lower():
|
|
error_lines.append(line)
|
|
return error_lines
|