314 lines
16 KiB
Python
314 lines
16 KiB
Python
import json
|
|
import os
|
|
from pathlib import Path
|
|
import subprocess
|
|
from types import SimpleNamespace
|
|
|
|
import pytest
|
|
|
|
from test_docker_codec_recovery import TARGET, archive_bytes, descriptor, encoded, recovery
|
|
import console_runner
|
|
import docker_shadow
|
|
import scanner
|
|
import scanner_db
|
|
|
|
|
|
ROUTINE = [
|
|
('info-2', 'trufflehog dev'), ('info-2', 'starting scanner workers'),
|
|
('info-2', 'starting detector workers'), ('info-2', 'starting verificationOverlap workers'),
|
|
('info-2', 'starting notifier workers'), ('info-0', 'running source'),
|
|
('info-2', 'enumerating source'), ('info-2', 'scanning image'),
|
|
('info-2', 'scanning image history'), ('info-2', 'scanning image history entry'),
|
|
('info-2', 'scanning image layers'), ('info-2', 'scanning layer'),
|
|
]
|
|
SIZE_SKIP = 'skipping file: size exceeds max allowed'
|
|
|
|
|
|
def line(msg, level='info-2', **extra):
|
|
return json.dumps({'level': level, 'logger': 'trufflehog', 'msg': msg, **extra}, ensure_ascii=False)
|
|
|
|
|
|
FINISHED = line('finished scanning', level='info-0')
|
|
|
|
|
|
def parse(lines, source='docker', rc=0, finished=True):
|
|
return scanner.apply_trufflehog_diagnostics(
|
|
{'findings': [{'DetectorName': 'SyntheticRetained'}], 'errors': []},
|
|
'\n'.join([*lines, *([FINISHED] if finished else [])]), rc, source, require_completion=True,
|
|
)
|
|
|
|
|
|
@pytest.mark.parametrize('source', ['docker', 'filesystem'])
|
|
@pytest.mark.parametrize('rc', [0, 1, -1])
|
|
def test_exact_size_skip_is_incomplete_and_nonretryable(source, rc):
|
|
result = parse([line(SIZE_SKIP, filename='fixture.txt', size=8240, limit=1024)], source, rc)
|
|
assert len(result['findings']) == 1 and result['retryable'] is False
|
|
if rc == 0:
|
|
assert not result['errors']
|
|
assert result['degraded'] and result['warning_classes'] == ['archive_member_size']
|
|
else:
|
|
assert result['errors'] and result['error_class'] == 'archive_member_size'
|
|
assert scanner_db.target_status(result) != 'clean'
|
|
|
|
|
|
def test_size_warning_does_not_replace_missing_completion_failure():
|
|
result = parse([line(SIZE_SKIP)], finished=False)
|
|
assert result['error_class'] == 'command_incomplete' and result['retryable']
|
|
assert result['warning_classes'] == ['archive_member_size']
|
|
|
|
|
|
@pytest.mark.parametrize('message', [SIZE_SKIP, 'scanning layer', 'finished scanning'])
|
|
@pytest.mark.parametrize('detail', [{'error': 'unclassified fatal cause'}, {'errors': ['unclassified fatal cause']},
|
|
{'error': 'unknown flag'}, {'errors': ['unknown flag', 'connection reset']},
|
|
{'message': 'unknown flag'}])
|
|
def test_known_messages_cannot_hide_singular_plural_or_detail_errors(message, detail):
|
|
result = parse([line(message, **detail)])
|
|
assert result['errors'] and not result.get('warnings')
|
|
assert len(result['findings']) == 1
|
|
|
|
|
|
def test_known_message_with_error_level_is_fatal():
|
|
assert parse([line('scanning layer', level='error')])['errors']
|
|
|
|
|
|
def test_routine_progress_does_not_consume_unclassified_budget():
|
|
messages = [line(message, level=level) for level, message in ROUTINE] * 50
|
|
result = parse(messages)
|
|
assert not result['errors']
|
|
assert result['scan_meta']['diagnostic_lines_processed'] == len(messages) + 1
|
|
assert 'stderr_unclassified' not in result['scan_meta']
|
|
|
|
|
|
@pytest.mark.parametrize('message', [line('future unknown progress'), line('scanning layer', logger='different'),
|
|
line('scanning layer', errors=[]), line('scanning layer', error=''),
|
|
'scanning layer', line('scanning layer', level='info-1')])
|
|
def test_unknown_or_error_shaped_info_still_has_retention_bound(message):
|
|
result = parse([message] * 21)
|
|
assert result['error_class'] == 'source_resource'
|
|
assert 'unclassified limit' in result['scan_meta']['diagnostic_output_limit_reason']
|
|
|
|
|
|
def test_size_message_matching_is_exact():
|
|
result = parse([line(SIZE_SKIP + ' (new variant)')])
|
|
assert not result.get('warnings')
|
|
assert result['scan_meta']['stderr_unclassified']
|
|
|
|
|
|
def test_routine_info_remains_unclassified_for_unmodified_git_path():
|
|
result = parse([line('scanning layer')] * 21, source='git')
|
|
assert result['error_class'] == 'source_resource'
|
|
|
|
|
|
def test_routine_info_cannot_hide_nonzero_exit():
|
|
result = parse([line(message, level=level) for level, message in ROUTINE], rc=3)
|
|
assert result['errors'] and result['error_class'] == 'wrapper_exit'
|
|
|
|
|
|
def test_routine_info_still_counts_towards_total_line_cap():
|
|
result = parse([line('scanning layer')] * 2001)
|
|
assert result['error_class'] == 'source_resource'
|
|
assert 'total line limit' in result['scan_meta']['diagnostic_output_limit_reason']
|
|
|
|
|
|
@pytest.mark.parametrize('text,reason', [('x' * 8193, 'line character'), ('\u00e9' * 5000, 'line byte')])
|
|
def test_routine_info_cannot_bypass_per_line_caps(text, reason):
|
|
result = parse([line('scanning layer', layer=text)])
|
|
assert result['error_class'] == 'source_resource'
|
|
assert reason in result['scan_meta']['diagnostic_output_limit_reason']
|
|
|
|
|
|
def test_coverage_warning_volume_still_fails_closed():
|
|
result = parse([line(SIZE_SKIP)] * 201)
|
|
assert result['error_class'] == 'source_resource'
|
|
assert 'warning limit' in result['scan_meta']['diagnostic_output_limit_reason']
|
|
assert len(result['findings']) == 1
|
|
|
|
|
|
def test_file_backed_output_limit_after_routine_info_is_not_ignored():
|
|
def output():
|
|
yield line('scanning layer')
|
|
raise scanner.CommandOutputLimitError('TruffleHog stderr exceeded its byte bound')
|
|
result = scanner.apply_trufflehog_diagnostics({'findings': [{}], 'errors': []}, output(), 0, 'docker')
|
|
assert result['error_class'] == 'source_resource'
|
|
assert result['scan_meta']['diagnostic_output_limited'] and result['findings']
|
|
|
|
|
|
def source_args(**overrides):
|
|
return SimpleNamespace(**{
|
|
**dict(platform='docker', workers=1, timeout=30, save_dir='synthetic-output', detectors=None,
|
|
exclude_detectors=None, no_verification=True, trufflehog_config='', drop_detectors=[],
|
|
trufflehog_concurrency=3, docker_layer_config_max_bytes=2048,
|
|
docker_layer_max_bytes=2 << 20, docker_layer_image_max_bytes=3 << 20,
|
|
docker_layer_max_layers=3, docker_layer_archive_max_size_bytes=4096,
|
|
docker_layer_archive_max_depth=2, docker_layer_archive_timeout_sec=4,
|
|
docker_layer_blob_timeout_sec=17, docker_layer_filesystem_concurrency=1,
|
|
docker_layer_blob_max_attempts=2, docker_layer_min_free_bytes=987654), **overrides})
|
|
|
|
|
|
def test_source_limits_reach_full_scan_but_not_durable_layer_dispatch(monkeypatch):
|
|
args = source_args()
|
|
monkeypatch.setattr(console_runner, 'get_platform_token', lambda args: None)
|
|
monkeypatch.setattr(scanner.scan_config, 'drop_detectors', [])
|
|
monkeypatch.setattr(scanner.docker_token_manager, 'get_next_config', lambda: None)
|
|
options = console_runner.prepare_scan_options(args, 1)
|
|
assert options['docker_recovery_limits'] == console_runner.docker_layer_limits(args)
|
|
assert options['docker_recovery_min_free_bytes'] == 987654
|
|
calls = []
|
|
|
|
def full(target, **kwargs):
|
|
calls.append(kwargs)
|
|
return {'findings': [], 'errors': []}
|
|
|
|
def layer(target, work, *, timeout_sec, detectors, exclude_detectors, no_verification, trufflehog_config):
|
|
calls.append({'work': work})
|
|
return {'findings': [], 'errors': []}
|
|
|
|
monkeypatch.setattr(scanner, 'scan_docker_image', full)
|
|
monkeypatch.setattr(scanner, 'scan_docker_layer_plan', layer)
|
|
scanner.scan_target_result(TARGET, 'docker', 'full-fixture', options)
|
|
assert calls[-1]['docker_recovery_limits']['archive_max_size_bytes'] == 4096
|
|
assert calls[-1]['docker_recovery_min_free_bytes'] == 987654
|
|
work = {'synthetic': 'bound work'}
|
|
scanner.scan_target_result(TARGET, 'docker', 'layer-fixture', dict(options, docker_layer_work=work))
|
|
assert calls[-1] == {'work': work}
|
|
assert 'docker_recovery_limits' in options # Dispatcher copies rather than mutating the caller.
|
|
|
|
|
|
def test_non_docker_options_do_not_receive_recovery_settings(monkeypatch):
|
|
monkeypatch.setattr(console_runner, 'get_platform_token', lambda args: None)
|
|
monkeypatch.setattr(scanner.scan_config, 'drop_detectors', [])
|
|
options = console_runner.prepare_scan_options(source_args(platform='github'), 1)
|
|
assert not any(key.startswith('docker_recovery') for key in options)
|
|
|
|
|
|
def test_full_policy_fingerprint_includes_source_recovery_limits(monkeypatch):
|
|
monkeypatch.setattr(console_runner, 'hash_file', lambda path: 'a' * 64)
|
|
args = source_args()
|
|
options = {'docker_recovery_limits': console_runner.docker_layer_limits(args)}
|
|
before = console_runner.docker_layer_scan_policy_sha256(args, options)
|
|
options['docker_recovery_limits'] = dict(options['docker_recovery_limits'], archive_max_depth=3)
|
|
assert console_runner.docker_layer_scan_policy_sha256(args, options) != before
|
|
|
|
|
|
def test_selection_and_scheduling_caps_do_not_rekey_execution_coverage(monkeypatch):
|
|
monkeypatch.setattr(console_runner, 'hash_file', lambda path: 'a' * 64)
|
|
args = source_args()
|
|
options = {'docker_recovery_limits': console_runner.docker_layer_limits(args)}
|
|
before = console_runner.docker_layer_scan_policy_sha256(args, options)
|
|
options['docker_recovery_limits'] = dict(options['docker_recovery_limits'], max_layers=8,
|
|
image_max_bytes=8 << 20, blob_timeout_sec=30, blob_max_attempts=3)
|
|
assert console_runner.docker_layer_scan_policy_sha256(args, options) == before
|
|
|
|
|
|
def test_full_shadow_forwards_source_recovery_settings(monkeypatch):
|
|
class DB:
|
|
def docker_adaptive_shadow_control_identities(self, scan_id):
|
|
return set(), set()
|
|
calls = []
|
|
monkeypatch.setattr(scanner.docker_token_manager, 'get_next_config', lambda: None)
|
|
monkeypatch.setattr(scanner, 'scan_docker_image', lambda *a, **k: calls.append(k) or {'findings': [], 'errors': []})
|
|
limits = console_runner.docker_layer_limits(source_args())
|
|
metrics = docker_shadow.private_full_side_metrics(DB(), {'target_scan_id': 1, 'normalized_target': TARGET},
|
|
{'timeout_sec': 30, 'docker_recovery_limits': limits, 'docker_recovery_min_free_bytes': 987654})
|
|
assert metrics['failure_count'] == 0
|
|
assert calls[0]['docker_recovery_limits'] == limits and calls[0]['docker_recovery_min_free_bytes'] == 987654
|
|
|
|
|
|
@pytest.fixture
|
|
def native_runner(tmp_path, monkeypatch):
|
|
executable = Path(r'C:\Tools\trufflehog.exe')
|
|
if not executable.is_file():
|
|
pytest.skip('installed TruffleHog unavailable')
|
|
monkeypatch.setattr(scanner, 'get_trufflehog_cmd', lambda: str(executable))
|
|
env = {key: os.environ[key] for key in ('SYSTEMROOT', 'WINDIR', 'PATH') if key in os.environ}
|
|
env.update(TEMP=str(tmp_path), TMP=str(tmp_path), HOME=str(tmp_path), USERPROFILE=str(tmp_path),
|
|
DOCKER_CONFIG=str(tmp_path), HTTP_PROXY='http://127.0.0.1:1', HTTPS_PROXY='http://127.0.0.1:1')
|
|
|
|
def command(args, timeout, unused_env, **kwargs):
|
|
completed = subprocess.run(args, env=env, cwd=tmp_path, capture_output=True, timeout=min(timeout, 30))
|
|
return scanner.streamed_output_from_text(stdout=completed.stdout.decode(), stderr=completed.stderr.decode(), returncode=completed.returncode)
|
|
|
|
monkeypatch.setattr(scanner, 'run_command_streamed', command)
|
|
return command
|
|
|
|
|
|
@pytest.mark.parametrize('sibling', [False, True])
|
|
def test_actual_nested_size_skip_through_shared_helper_is_not_clean(tmp_path, native_runner, sibling):
|
|
import time
|
|
marker = b'OFFLINENESTED_ABCDEFGHIJKLMNOPQRSTUVWXYZ012345'
|
|
nested = archive_bytes([('large.txt', b'x' * 8192 + b'\n' + marker + b'\n')])
|
|
entries = [('inner.tar.gz', encoded(nested, 'gzip'))]
|
|
if sibling:
|
|
entries.insert(0, ('outside.txt', marker + b'\n'))
|
|
payload = archive_bytes(entries)
|
|
path = tmp_path / 'blob'
|
|
path.write_bytes(payload)
|
|
policy = tmp_path / 'policy.yaml'
|
|
policy.write_text('detectors:\n - name: OfflineNested\n keywords: [OFFLINENESTED]\n regex:\n marker: "OFFLINENESTED_[A-Z0-9]{32}"\n', encoding='ascii')
|
|
limits = dict(console_runner.docker_layer_limits(source_args()), archive_max_size_bytes=1024, archive_max_depth=4)
|
|
result = scanner._scan_docker_content_file(str(path), descriptor(payload), limits, time.monotonic() + 30,
|
|
no_verification=True, trufflehog_config=str(policy))
|
|
assert not result['errors'] and result['degraded'] and not result['retryable']
|
|
assert result['warning_classes'] == ['archive_member_size']
|
|
assert bool(result['findings']) == sibling
|
|
assert scanner_db.target_status(result) != 'clean'
|
|
|
|
|
|
@pytest.mark.parametrize('sibling', [False, True])
|
|
def test_actual_nested_skip_prevents_full_recovery_success(recovery, tmp_path, monkeypatch, sibling):
|
|
executable = Path(r'C:\Tools\trufflehog.exe')
|
|
if not executable.is_file():
|
|
pytest.skip('installed TruffleHog unavailable')
|
|
monkeypatch.setattr(scanner, 'get_trufflehog_cmd', lambda: str(executable))
|
|
marker = b'OFFLINENESTED_ABCDEFGHIJKLMNOPQRSTUVWXYZ012345'
|
|
nested = archive_bytes([('large.txt', b'x' * 8192 + b'\n' + marker + b'\n')])
|
|
entries = [('inner.tar.gz', encoded(nested, 'gzip'))]
|
|
if sibling:
|
|
entries.insert(0, ('outside.txt', marker + b'\n'))
|
|
payload = archive_bytes(entries)
|
|
item = descriptor(payload)
|
|
recovery.resolved['layers'] = [item]
|
|
recovery.payloads[item['digest']] = payload
|
|
policy = tmp_path / 'policy.yaml'
|
|
policy.write_text('detectors:\n - name: OfflineNested\n keywords: [OFFLINENESTED]\n regex:\n marker: "OFFLINENESTED_[A-Z0-9]{32}"\n', encoding='ascii')
|
|
limits = dict(console_runner.docker_layer_limits(source_args()), archive_max_size_bytes=1024, archive_max_depth=4)
|
|
recovery.real_cli = True
|
|
result = recovery.run(docker_recovery_limits=limits, no_verification=True, trufflehog_config=str(policy))
|
|
assert result['errors'] and not result['retryable']
|
|
assert result['error_class'] == 'recovery_scan_incomplete'
|
|
assert not result['scan_meta']['docker_full_recovery']['coverage_complete']
|
|
assert len(result['findings']) >= 1 + int(sibling)
|
|
assert all(command[command.index('--log-level') + 1] == '2' for command, _, _ in recovery.commands)
|
|
|
|
|
|
def test_actual_normal_full_docker_with_many_routine_lines_succeeds(tmp_path, native_runner, monkeypatch):
|
|
layers = []
|
|
diffs = []
|
|
for index in range(32):
|
|
raw = archive_bytes([(f'file-{index}.txt', b'ordinary synthetic content\n')])
|
|
layers.append((f'layer-{index}.tar', encoded(raw, 'gzip')))
|
|
diffs.append(descriptor(raw)['digest'])
|
|
config = json.dumps({'os': 'linux', 'architecture': 'amd64', 'config': {},
|
|
'rootfs': {'type': 'layers', 'diff_ids': diffs},
|
|
'history': [{'created_by': 'offline fixture'} for _ in layers]}).encode()
|
|
manifest = json.dumps([{'Config': 'config.json', 'RepoTags': ['offline-fixture:latest'],
|
|
'Layers': [name for name, _ in layers]}]).encode()
|
|
image = tmp_path / 'image.tar'
|
|
image.write_bytes(archive_bytes([('manifest.json', manifest), ('config.json', config), *layers]))
|
|
policy = tmp_path / 'policy.yaml'
|
|
policy.write_text('detectors:\n - name: OfflineNested\n keywords: [OFFLINENESTED]\n regex:\n marker: "OFFLINENESTED_[A-Z0-9]{32}"\n', encoding='ascii')
|
|
|
|
def local_image_only(args, *rest, **kwargs):
|
|
assert args[1] == 'docker' and args[args.index('--image') + 1] == TARGET
|
|
args = list(args)
|
|
args[args.index('--image') + 1] = 'file://' + str(image)
|
|
return native_runner(args, *rest, **kwargs)
|
|
|
|
monkeypatch.setattr(scanner, 'run_command_streamed', local_image_only)
|
|
result = scanner.scan_docker_image(TARGET, config_dir=str(tmp_path), trufflehog_config=str(policy),
|
|
no_verification=True, trufflehog_concurrency=1, log_target=False)
|
|
assert not result['errors'] and not result.get('warnings')
|
|
assert result['scan_meta']['diagnostic_lines_processed'] >= 70
|
|
assert 'stderr_unclassified' not in result['scan_meta']
|
|
assert scanner_db.target_status(result) == 'clean'
|