Initial server source import

This commit is contained in:
sashatrask
2026-09-30 20:30:56 +03:00
commit 170dd941b9
498 changed files with 261563 additions and 0 deletions
+712
View File
@@ -0,0 +1,712 @@
import builtins
import contextlib
import gzip
import hashlib
import io
import json
import os
from pathlib import Path
import shutil
import subprocess
import sys
import tarfile
import time
from types import SimpleNamespace
import pytest
import zstandard
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / 'app'))
import scanner
import scanner_db
TARGET = 'offline/fixture@sha256:' + 'a' * 64
MEDIA = 'application/vnd.oci.image.layer.v1.tar'
FINISHED = json.dumps({'level': 'info-0', 'msg': 'finished scanning'})
GZIP_WARNING = json.dumps({'level': 'error', 'msg': 'error processing layer', 'error': 'gzip: invalid header'})
def archive_bytes(entries=None, format=tarfile.USTAR_FORMAT):
output = io.BytesIO()
with tarfile.open(fileobj=output, mode='w', format=format) as archive:
for name, data in entries or [('first.txt', b'offline fixture\n'), ('second.txt', b'other fixture\n')]:
member = tarfile.TarInfo(name)
member.size = len(data)
archive.addfile(member, io.BytesIO(data))
return output.getvalue()
def encoded(raw, codec):
if codec == 'gzip':
return gzip.compress(raw, mtime=0)
if codec == 'zstd':
return zstandard.ZstdCompressor(write_checksum=True).compress(raw)
return raw
def descriptor(payload, codec='raw', kind='layer'):
return {'digest': 'sha256:' + hashlib.sha256(payload).hexdigest(), 'size': len(payload),
'media_type': 'application/vnd.oci.image.config.v1+json' if kind == 'config' else MEDIA + ('' if codec == 'raw' else '+' + codec),
'kind': kind}
def pax_record(key, value):
data = key + b'=' + value + b'\n'
length = len(data) + 3
while True:
adjusted = len(str(length)) + 1 + len(data)
if adjusted == length:
return str(length).encode() + b' ' + data
length = adjusted
def pax_archive(body, kind=tarfile.XHDTYPE):
header = tarfile.TarInfo('pax')
header.type = kind
header.size = len(body)
return header.tobuf() + body + b'\0' * (-len(body) % 512) + archive_bytes()
@pytest.mark.parametrize('codec', ['raw', 'gzip', 'zstd'])
def test_complete_valid_archives(tmp_path, codec):
raw = archive_bytes()
payload = encoded(raw, codec)
path = tmp_path / 'blob'
path.write_bytes(payload)
scanner.validate_docker_content_artifact(path, descriptor(payload, codec))
@pytest.mark.parametrize('case', ['late_header', 'body_truncated', 'no_terminator', 'trailing_data',
'gzip_crc', 'gzip_truncated', 'zstd_crc', 'zstd_truncated', 'zstd_stub'])
def test_complete_validation_rejects_malformed_content(tmp_path, case):
raw = archive_bytes()
codec = 'raw'
if case == 'late_header':
payload = bytearray(raw)
payload[1024] ^= 1
elif case == 'body_truncated':
payload = raw[:520]
elif case == 'no_terminator':
payload = raw[:2048]
elif case == 'trailing_data':
payload = raw + b'not padding'
else:
codec = 'gzip' if case.startswith('gzip') else 'zstd'
payload = bytearray(encoded(raw, codec))
if case.endswith('crc'):
payload[-8 if codec == 'gzip' else -1] ^= 1
elif case.endswith('truncated'):
payload = payload[:-1]
else:
payload = b'\x28\xb5\x2f\xfd'
path = tmp_path / 'blob'
path.write_bytes(payload)
with pytest.raises(scanner.DockerContentScanError) as error:
scanner.validate_docker_content_artifact(path, descriptor(payload, codec))
assert error.value.error_code == 'invalid_layer_archive'
assert not error.value.retryable
@pytest.mark.parametrize('bound', [{'max_members': 1}, {'max_member_bytes': 1}, {'max_decoded_bytes': 1024}])
def test_validation_limits(tmp_path, bound):
payload = archive_bytes()
path = tmp_path / 'blob'
path.write_bytes(payload)
with pytest.raises(scanner.DockerContentScanError) as error:
scanner.validate_docker_content_artifact(path, descriptor(payload), **bound)
assert error.value.error_code == 'archive_limit'
def test_validation_absolute_deadline(tmp_path):
path = tmp_path / 'blob'
path.write_bytes(archive_bytes())
with pytest.raises(scanner.DockerContentScanError) as error:
scanner.validate_docker_content_artifact(path, descriptor(path.read_bytes()), deadline=time.monotonic() - 1)
assert error.value.error_code == 'archive_timeout'
assert error.value.retryable
def test_missing_zstd_is_infrastructure_capability(tmp_path, monkeypatch):
payload = encoded(archive_bytes(), 'zstd')
path = tmp_path / 'blob'
path.write_bytes(payload)
original = builtins.__import__
def importing(name, *args, **kwargs):
if name == 'zstandard':
raise ImportError('synthetic missing decoder')
return original(name, *args, **kwargs)
monkeypatch.setattr(builtins, '__import__', importing)
with pytest.raises(scanner.DockerLayerInfrastructureError) as error:
scanner.validate_docker_content_artifact(path, descriptor(payload, 'zstd'))
assert error.value.error_code == 'archive_decoder_unavailable'
assert error.value.category == 'source_configuration'
@pytest.mark.parametrize('codec', ['raw', 'gzip', 'zstd'])
def test_pax_and_multiframe_archives(tmp_path, codec):
raw = archive_bytes([('long/' + 'x' * 200, b'bounded')], format=tarfile.PAX_FORMAT)
payload = raw if codec == 'raw' else encoded(raw[:2048], codec) + encoded(raw[2048:], codec)
path = tmp_path / 'blob'
path.write_bytes(payload)
scanner.validate_docker_content_artifact(path, descriptor(payload, codec))
def test_empty_tar_and_zstd_skippable_frames(tmp_path):
empty = b'\0' * 1024
skip = b'\x50\x2a\x4d\x18\x04\0\0\0skip'
payload = skip + encoded(empty, 'zstd') + skip
path = tmp_path / 'blob'
path.write_bytes(payload)
scanner.validate_docker_content_artifact(path, descriptor(payload, 'zstd'))
def test_zstd_unknown_size_window_larger_than_one_megabyte(tmp_path):
raw = archive_bytes([('large.txt', b'x' * (2 << 20))])
payload = zstandard.ZstdCompressor(write_content_size=False).compress(raw)
assert zstandard.get_frame_parameters(payload).window_size > 1 << 20
path = tmp_path / 'blob'
path.write_bytes(payload)
scanner.validate_docker_content_artifact(path, descriptor(payload, 'zstd'))
def test_validation_never_extracts_member_paths(tmp_path):
payload = archive_bytes([('../outside', b'bounded'), ('/absolute/fixture', b'bounded')])
path = tmp_path / 'blob'
path.write_bytes(payload)
scanner.validate_docker_content_artifact(path, descriptor(payload))
assert list(tmp_path.iterdir()) == [path]
def test_oversized_pax_metadata_rejected_before_read(tmp_path):
header = tarfile.TarInfo('pax')
header.type = tarfile.XHDTYPE
header.size = (1 << 20) + 1
payload = header.tobuf()
path = tmp_path / 'blob'
path.write_bytes(payload)
with pytest.raises(scanner.DockerContentScanError) as error:
scanner.validate_docker_content_artifact(path, descriptor(payload))
assert error.value.error_code == 'archive_limit'
@pytest.mark.parametrize('kind', [tarfile.XHDTYPE, tarfile.XGLTYPE, tarfile.SOLARIS_XHDTYPE])
@pytest.mark.parametrize('body', [b'9' * 30000 + b' key=value\n', pax_record(b'uid', b'9' * 30000)])
def test_pax_huge_numbers_cannot_enter_unbounded_stdlib_parser(tmp_path, monkeypatch, kind, body):
def forbidden(*args, **kwargs):
raise AssertionError('unbounded stdlib PAX parser entered')
monkeypatch.setattr(tarfile.TarInfo, '_proc_pax', forbidden)
payload = pax_archive(body, kind)
path = tmp_path / 'blob'
path.write_bytes(payload)
started = time.monotonic()
with pytest.raises(scanner.DockerContentScanError) as error:
scanner.validate_docker_content_artifact(path, descriptor(payload), deadline=started + 0.05)
assert error.value.error_code in ('archive_limit', 'archive_timeout')
assert time.monotonic() - started < 0.5
def test_digit_heavy_pax_path_uses_linear_parser(tmp_path, monkeypatch):
monkeypatch.setattr(tarfile.TarInfo, '_proc_pax', lambda *a, **k: pytest.fail('stdlib PAX regex called'))
payload = pax_archive(pax_record(b'path', b'9' * 30000))
path = tmp_path / 'blob'
path.write_bytes(payload)
scanner.validate_docker_content_artifact(path, descriptor(payload), deadline=time.monotonic() + 0.5)
@pytest.mark.parametrize('size', [(1 << 20) + 1, (64 << 10) + 1])
def test_solaris_pax_metadata_and_parse_caps(tmp_path, size):
header = tarfile.TarInfo('pax')
header.type = tarfile.SOLARIS_XHDTYPE
header.size = size
payload = header.tobuf()
path = tmp_path / 'blob'
path.write_bytes(payload)
with pytest.raises(scanner.DockerContentScanError) as error:
scanner.validate_docker_content_artifact(path, descriptor(payload))
assert error.value.error_code == 'archive_limit'
def test_solaris_pax_valid_header(tmp_path):
payload = pax_archive(pax_record(b'path', b'solaris-name'), tarfile.SOLARIS_XHDTYPE)
path = tmp_path / 'blob'
path.write_bytes(payload)
scanner.validate_docker_content_artifact(path, descriptor(payload))
@pytest.mark.parametrize('key,value', [(b'GNU.sparse.map', b'0,0,' * 1000),
(b'GNU.sparse.major', b'1'), (b'GNU.sparse.size', b'1')])
def test_sparse_pax_rejected_before_map_or_field_materialization(tmp_path, monkeypatch, key, value):
def forbidden(*args, **kwargs):
raise AssertionError('sparse PAX field/map materialization entered')
for method in ('_apply_pax_info', '_proc_gnusparse_00', '_proc_gnusparse_01', '_proc_gnusparse_10'):
monkeypatch.setattr(tarfile.TarInfo, method, forbidden)
payload = pax_archive(pax_record(key, value))
path = tmp_path / 'blob'
path.write_bytes(payload)
with pytest.raises(scanner.DockerContentScanError) as error:
scanner.validate_docker_content_artifact(path, descriptor(payload), max_member_bytes=1024)
assert error.value.error_code in ('archive_limit', 'unsupported_layer_archive')
def test_pax_record_count_bound(tmp_path):
payload = pax_archive(pax_record(b'comment', b'x') * 1025)
path = tmp_path / 'blob'
path.write_bytes(payload)
with pytest.raises(scanner.DockerContentScanError) as error:
scanner.validate_docker_content_artifact(path, descriptor(payload))
assert error.value.error_code == 'archive_limit'
def test_header_parse_checks_deadline_on_return(tmp_path, monkeypatch):
original = tarfile.TarInfo._proc_builtin
def delayed(*args, **kwargs):
member = original(*args, **kwargs)
time.sleep(0.06)
return member
monkeypatch.setattr(tarfile.TarInfo, '_proc_builtin', delayed)
payload = archive_bytes()
path = tmp_path / 'blob'
path.write_bytes(payload)
with pytest.raises(scanner.DockerContentScanError) as error:
scanner.validate_docker_content_artifact(path, descriptor(payload), deadline=time.monotonic() + 0.05)
assert error.value.error_code == 'archive_timeout'
def test_deadline_checked_during_archive_walk(tmp_path, monkeypatch):
payload = archive_bytes()
path = tmp_path / 'blob'
path.write_bytes(payload)
clock = [100.0]
def ticking():
clock[0] += 0.01
return clock[0]
monkeypatch.setattr(scanner.time, 'monotonic', ticking)
with pytest.raises(scanner.DockerContentScanError) as error:
scanner.validate_docker_content_artifact(path, descriptor(payload), deadline=100.08)
assert error.value.error_code == 'archive_timeout'
@pytest.fixture
def recovery(tmp_path, monkeypatch):
monkeypatch.delenv('DOCKER_CONFIG', raising=False)
monkeypatch.delenv('DOCKER_TOKEN', raising=False)
monkeypatch.delenv('REGISTRY_AUTH_FILE', raising=False)
monkeypatch.delenv('HOMEDRIVE', raising=False)
monkeypatch.delenv('HOMEPATH', raising=False)
monkeypatch.setenv('HOME', str(tmp_path))
monkeypatch.setenv('USERPROFILE', str(tmp_path))
monkeypatch.setenv('XDG_RUNTIME_DIR', str(tmp_path))
raw = archive_bytes()
payloads = [b'{"history":[]}', raw, encoded(raw, 'gzip'), encoded(raw, 'zstd')]
items = [descriptor(payloads[0], kind='config')] + [descriptor(data, codec) for data, codec in zip(payloads[1:], ('raw', 'gzip', 'zstd'))]
resolved = {'version': 1, 'image': TARGET, 'repository': 'offline/fixture',
'manifest_digest': TARGET.split('@')[1], 'config': items[0], 'layers': items[1:]}
state = SimpleNamespace(resolved=resolved, payloads={item['digest']: data for item, data in zip(items, payloads)},
commands=[], downloads=[], resolutions=[], full_stderr=GZIP_WARNING + '\n' + FINISHED,
blob_stderr=FINISHED, after_full=lambda: None, real_cli=False)
monkeypatch.setattr(scanner, 'get_work_dir', lambda: str(tmp_path))
monkeypatch.setattr(scanner, 'harden_private_directory', lambda *a, **k: None)
monkeypatch.setattr(scanner, 'write_temp_owner', lambda *a, **k: None)
monkeypatch.setattr(scanner, 'durable_unlink', lambda path: os.unlink(path))
monkeypatch.setattr(scanner, 'cleanup_command_work_dir', lambda path: shutil.rmtree(path))
monkeypatch.setattr(scanner, 'apply_finding_filters', lambda result, *a, **k: result)
monkeypatch.setattr(scanner, 'docker_token_manager', scanner.DockerTokenManager())
def resolve(target, **kwargs):
state.resolutions.append((target, kwargs))
return state.resolved, scanner.DockerRegistryAuth('synthetic-bearer')
def download(repository, item, path, bearer_auth, **kwargs):
state.downloads.append((item, kwargs))
Path(path).write_bytes(state.payloads[item['digest']])
return scanner.DockerBlobDownloadOutcome(path, item['size'], item['size'], 1, bearer_auth)
def command(args, timeout, env, **kwargs):
state.commands.append((args, timeout, kwargs))
if args[1] == 'docker':
state.after_full()
return scanner.streamed_output_from_text(
stdout=json.dumps({'DetectorName': 'Fixture', 'Raw': 'synthetic-full'}), stderr=state.full_stderr)
if state.real_cli:
safe_env = {key: os.environ[key] for key in ('SYSTEMROOT', 'WINDIR', 'PATH') if key in os.environ}
safe_env.update(TEMP=str(tmp_path), TMP=str(tmp_path), HOME=str(tmp_path), USERPROFILE=str(tmp_path),
DOCKER_CONFIG=str(tmp_path), HTTP_PROXY='http://127.0.0.1:1', HTTPS_PROXY='http://127.0.0.1:1')
completed = subprocess.run(args, env=safe_env, cwd=tmp_path, capture_output=True, timeout=30)
return scanner.streamed_output_from_text(stdout=completed.stdout.decode(), stderr=completed.stderr.decode(), returncode=completed.returncode)
return scanner.streamed_output_from_text(
stdout=json.dumps({'DetectorName': 'Fixture', 'Raw': 'synthetic-blob',
'SourceMetadata': {'Data': {'Filesystem': {'file': args[2]}}}}), stderr=state.blob_stderr)
monkeypatch.setattr(scanner, 'resolve_docker_content_manifest', resolve)
monkeypatch.setattr(scanner, 'stream_docker_registry_blob', download)
monkeypatch.setattr(scanner, 'run_command_streamed', command)
state.run = lambda **kwargs: scanner.scan_docker_image(TARGET, log_target=False, docker_recovery_min_free_bytes=0, **kwargs)
return state
def test_recovers_all_codecs_and_config_with_same_deadline(recovery):
result = recovery.run()
assert not result['errors']
assert not result.get('warnings') and not result.get('degraded')
assert len(result['findings']) == 5
scope = result['scan_meta']['docker_full_recovery']
assert scope['coverage_complete'] and scope['scanned_descriptors'] == scope['descriptor_count'] == 4
assert len(recovery.downloads) == 4
deadline = recovery.commands[0][2]['deadline']
assert recovery.resolutions[0][1]['deadline'] == deadline
assert all(kwargs['deadline'] <= deadline for _, kwargs in recovery.downloads)
assert all(kwargs['deadline'] <= deadline for _, _, kwargs in recovery.commands)
for finding in result['findings'][1:]:
data = finding['SourceMetadata']['Data']
assert data['DockerContent']['manifest_digest'] == TARGET.split('@')[1]
assert data['Filesystem']['file'].startswith('docker://offline/fixture@')
def test_duplicate_digest_scanned_once_preserves_positions(recovery):
recovery.resolved['layers'].append(dict(recovery.resolved['layers'][0]))
result = recovery.run()
assert result['scan_meta']['docker_full_recovery']['scanned_descriptors'] == 5
assert len(recovery.downloads) == 4
assert result['findings'][2]['SourceMetadata']['Data']['DockerContent']['positions'] == [1, 4]
@pytest.mark.parametrize('field,value', [('media_type', MEDIA + '+gzip'), ('size', 12), ('kind', 'config')])
def test_conflicting_digest_preflight(recovery, field, value):
if field == 'kind':
recovery.resolved['config']['digest'] = recovery.resolved['layers'][0]['digest']
else:
other = dict(recovery.resolved['layers'][0])
other[field] = value
recovery.resolved['layers'].append(other)
result = recovery.run()
assert result['error_class'] == 'conflicting_descriptors'
assert not recovery.downloads
@pytest.mark.parametrize('limit,value', [('max_layers', 1), ('layer_max_bytes', 1024), ('image_max_bytes', 1024)])
def test_all_fitting_preflight_never_scans_top_n(recovery, limit, value):
limits = {'config_max_bytes': 1024, 'layer_max_bytes': 1 << 20, 'image_max_bytes': 2 << 20,
'max_layers': 8, 'archive_max_size_bytes': 1 << 20, 'archive_max_depth': 4,
'archive_timeout_sec': 5, 'blob_timeout_sec': 10, 'filesystem_concurrency': 1, 'blob_max_attempts': 3}
limits[limit] = value
result = recovery.run(docker_recovery_limits=limits)
assert result['errors'] and result['error_class'] == 'recovery_budget'
assert not recovery.downloads
assert not result['scan_meta']['docker_full_recovery']['coverage_complete']
assert len(result['findings']) == 1
def test_oversized_config_preflight(recovery):
recovery.resolved['config']['size'] = (1 << 20) + 1
result = recovery.run()
assert result['error_class'] == 'recovery_budget'
assert not recovery.downloads
def test_incompatible_decoded_policy_rejected_before_any_recovery_download(recovery):
limits = {'config_max_bytes': 1024, 'layer_max_bytes': 1 << 20, 'image_max_bytes': 2 << 20,
'max_layers': 8, 'archive_max_size_bytes': 4 << 30, 'archive_max_depth': 4,
'archive_timeout_sec': 5, 'blob_timeout_sec': 10, 'filesystem_concurrency': 1, 'blob_max_attempts': 3}
result = recovery.run(docker_recovery_limits=limits)
assert result['error_class'] == 'archive_policy_incompatible'
assert result['source_failure_category'] == 'source_configuration'
assert not recovery.resolutions and not recovery.downloads
def test_incompatible_decoded_policy_rejected_before_layer_download(recovery, monkeypatch):
plan = {'image': TARGET, 'limits': {'archive_max_size_bytes': 4 << 30}}
monkeypatch.setattr(scanner, 'validate_docker_layer_plan', lambda value: value)
monkeypatch.setattr(scanner, 'canonical_docker_layer_plan_bytes', lambda value: b'fixture')
with pytest.raises(scanner.DockerLayerInfrastructureError) as error:
scanner.scan_docker_layer_plan(TARGET, {'plan': plan, 'plan_sha256': hashlib.sha256(b'fixture').hexdigest()})
assert error.value.error_code == 'archive_policy_incompatible'
assert not recovery.downloads
def test_unsupported_layer_media_preflight(recovery):
recovery.resolved['layers'][0]['media_type'] = 'application/octet-stream'
result = recovery.run()
assert result['error_class'] == 'unsupported_media_type'
assert not recovery.downloads
def test_resolved_root_must_match_original_digest(recovery):
recovery.resolved['manifest_digest'] = 'sha256:' + 'b' * 64
result = recovery.run()
assert result['error_class'] == 'recovery_identity_mismatch'
assert not recovery.downloads
def test_no_fresh_budget_after_full_engine(recovery, monkeypatch):
clock = [100.0]
monkeypatch.setattr(scanner.time, 'monotonic', lambda: clock[0])
recovery.after_full = lambda: clock.__setitem__(0, 110.0)
result = recovery.run(timeout_sec=10)
assert result['error_class'] == 'timeout'
assert not recovery.resolutions and not recovery.downloads
assert recovery.commands[0][2]['deadline'] == 110
@pytest.mark.parametrize('stage', ['cleanup', 'filter'])
def test_full_scan_without_fallback_cannot_finish_after_deadline(recovery, monkeypatch, stage):
recovery.full_stderr = FINISHED
clock = [100.0]
monkeypatch.setattr(scanner.time, 'monotonic', lambda: clock[0])
original = scanner.run_command_streamed
if stage == 'cleanup':
@contextlib.contextmanager
def command(*args, **kwargs):
with original(*args, **kwargs) as output:
yield output
clock[0] = 111.0
monkeypatch.setattr(scanner, 'run_command_streamed', command)
else:
def filter_result(result, *args, **kwargs):
clock[0] = 111.0
return result
monkeypatch.setattr(scanner, 'apply_finding_filters', filter_result)
result = recovery.run(timeout_sec=10)
assert result['error_class'] == 'timeout'
assert result['scan_meta']['docker_deadline_exceeded']
assert len(result['findings']) == 1
assert not recovery.resolutions
def test_post_filter_timeout_does_not_erase_unrelated_error(recovery, monkeypatch):
recovery.full_stderr = json.dumps({'level': 'error', 'msg': 'fatal synthetic failure'}) + '\n' + FINISHED
clock = [100.0]
monkeypatch.setattr(scanner.time, 'monotonic', lambda: clock[0])
def filtering(result, *args, **kwargs):
clock[0] = 111.0
return result
monkeypatch.setattr(scanner, 'apply_finding_filters', filtering)
result = recovery.run(timeout_sec=10)
assert result['error_class'] != 'timeout'
assert len(result['errors']) >= 2
assert len(result['findings']) == 1
@pytest.mark.parametrize('extra', [json.dumps({'level': 'error', 'msg': 'fatal scan error'}),
json.dumps({'level': 'error', 'msg': 'non-critical error processing chunk'})])
def test_unrelated_diagnostics_are_never_cleared(recovery, extra):
recovery.full_stderr += '\n' + extra
result = recovery.run()
assert result['errors']
assert not recovery.resolutions
assert len(result['findings']) == 1
def test_only_exact_gzip_warning_triggers_recovery(recovery):
recovery.full_stderr = GZIP_WARNING.replace('gzip: invalid header', 'gzip: invalid header trailing text') + '\n' + FINISHED
assert recovery.run()['errors']
assert not recovery.resolutions
@pytest.mark.parametrize('target', ['offline/fixture:latest', 'example.invalid/fixture@sha256:' + 'a' * 64])
def test_no_mutable_or_external_registry_recovery(recovery, target):
result = scanner.scan_docker_image(target, log_target=False)
assert result['errors']
assert not recovery.resolutions
def test_recovery_warning_retains_partial_findings(recovery):
recovery.blob_stderr = json.dumps({'level': 'error', 'msg': 'non-critical error processing chunk'}) + '\n' + FINISHED
result = recovery.run()
assert result['errors'] and len(result['findings']) == 2
assert not result['scan_meta']['docker_full_recovery']['coverage_complete']
def test_transfer_failure_preserves_prior_findings_without_source_disable(recovery, monkeypatch):
original = scanner.stream_docker_registry_blob
def download(*args, **kwargs):
if len(recovery.downloads) == 1:
raise scanner.DockerContentTransferError('digest_mismatch', 'private synthetic detail', False)
return original(*args, **kwargs)
monkeypatch.setattr(scanner, 'stream_docker_registry_blob', download)
result = recovery.run()
assert result['error_class'] == 'digest_mismatch'
assert not result['retryable'] and not result.get('source_failure')
assert len(result['findings']) == 2
assert 'private synthetic detail' not in json.dumps(result)
def test_malformed_manifest_is_not_infrastructure(recovery, monkeypatch):
def resolve(*args, **kwargs):
raise scanner.DockerRegistryResolutionError('private synthetic detail')
monkeypatch.setattr(scanner, 'resolve_docker_content_manifest', resolve)
result = recovery.run()
assert result['error_class'] == 'recovery_manifest_invalid'
assert not result['retryable'] and not result.get('source_failure')
assert 'private synthetic detail' not in json.dumps(result)
def test_missing_decoder_during_recovery_keeps_already_scanned_findings(recovery, monkeypatch):
original = builtins.__import__
def importing(name, *args, **kwargs):
if name == 'zstandard':
raise ImportError('synthetic missing decoder')
return original(name, *args, **kwargs)
monkeypatch.setattr(builtins, '__import__', importing)
result = recovery.run()
assert result['error_class'] == 'archive_decoder_unavailable'
assert result['source_failure_category'] == 'source_configuration'
assert not result['scan_meta']['docker_full_recovery']['coverage_complete']
assert len(result['findings']) == 4
def test_cleanup_failure_never_reports_full_coverage(recovery, monkeypatch):
def cleanup(path):
shutil.rmtree(path)
raise RuntimeError('private cleanup path')
monkeypatch.setattr(scanner, 'cleanup_command_work_dir', cleanup)
result = recovery.run()
assert result['error_class'] == 'private_cleanup'
assert not result['scan_meta']['docker_full_recovery']['coverage_complete']
assert len(result['findings']) == 5
assert 'private cleanup path' not in json.dumps(result)
def test_cleanup_is_inside_the_original_deadline(recovery, monkeypatch):
clock = [100.0]
monkeypatch.setattr(scanner.time, 'monotonic', lambda: clock[0])
def cleanup(path):
shutil.rmtree(path)
clock[0] = 111.0
monkeypatch.setattr(scanner, 'cleanup_command_work_dir', cleanup)
result = recovery.run(timeout_sec=10)
assert result['error_class'] == 'timeout'
assert not result['scan_meta']['docker_full_recovery']['coverage_complete']
def test_unknown_config_fails_closed_without_loading_or_exposing_it(recovery):
result = recovery.run(config_dir='unmanaged-private-config')
assert result['error_class'] == 'recovery_auth_unavailable'
assert result['source_failure_category'] == 'source_configuration'
assert not recovery.resolutions
assert 'unmanaged-private-config' not in json.dumps(result)
@pytest.mark.parametrize('location', ['home', 'profile', 'podman', 'registry_env'])
def test_implicit_keychain_configuration_blocks_identity_switch(recovery, tmp_path, monkeypatch, location):
if location == 'profile':
root = tmp_path / 'profile'
monkeypatch.setenv('USERPROFILE', str(root))
path = root / '.docker' / 'config.json'
elif location == 'podman':
path = tmp_path / 'containers' / 'auth.json'
elif location == 'registry_env':
path = tmp_path / 'private-auth.json'
monkeypatch.setenv('REGISTRY_AUTH_FILE', str(path))
else:
path = tmp_path / '.docker' / 'config.json'
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text('do-not-load-this-credential-configuration', encoding='ascii')
result = recovery.run()
assert result['error_class'] == 'recovery_auth_unavailable'
assert result['source_failure_category'] == 'source_configuration'
assert not recovery.resolutions
assert 'do-not-load' not in json.dumps(result) and str(path) not in json.dumps(result)
def test_implicit_config_snapshot_survives_removal_during_full_scan(recovery, tmp_path):
path = tmp_path / '.docker' / 'config.json'
path.parent.mkdir()
path.write_text('{}', encoding='ascii')
recovery.after_full = path.unlink
result = recovery.run()
assert result['error_class'] == 'recovery_auth_unavailable'
assert not recovery.resolutions
def test_managed_config_uses_existing_token_exchange(recovery, monkeypatch):
scanner.docker_token_manager.accounts = [SimpleNamespace(name='selected', config_dir='managed'), SimpleNamespace(name='other', config_dir='other')]
calls = []
def token(*args, **kwargs):
calls.append(kwargs)
return scanner.DockerRegistryAuth('private-token', 'selected')
monkeypatch.setattr(scanner, 'docker_registry_bearer_token', token)
result = recovery.run(config_dir='managed')
assert not result['errors']
assert calls[0]['excluded_accounts'] == {'other'}
assert recovery.resolutions[0][1]['bearer_auth'].account_name == 'selected'
assert 'private-token' not in json.dumps(result)
def test_policy_versions_invalidate_old_coverage():
assert scanner_db.DOCKER_ADAPTIVE_EXECUTION_VERSION == 'docker-layer-execution-v4'
limits = {'config_max_bytes': 1024, 'layer_max_bytes': 1024, 'image_max_bytes': 2048,
'max_layers': 2, 'archive_max_size_bytes': 1024, 'archive_max_depth': 2,
'archive_timeout_sec': 5, 'blob_timeout_sec': 10, 'filesystem_concurrency': 1, 'blob_max_attempts': 3}
old_payload = {'limits': limits, 'scan_policy_sha256': 'b' * 64}
old_hash = hashlib.sha256(json.dumps(old_payload, ensure_ascii=True, sort_keys=True, separators=(',', ':')).encode()).hexdigest()
assert scanner_db.docker_layer_coverage_policy_sha256('b' * 64, limits) != old_hash
def test_installed_filesystem_cli_recovers_real_synthetic_codecs(recovery, tmp_path, monkeypatch):
executable = Path(r'C:\Tools\trufflehog.exe')
if not executable.is_file():
pytest.skip('installed TruffleHog is unavailable')
monkeypatch.setattr(scanner, 'get_trufflehog_cmd', lambda: str(executable))
policy = tmp_path / 'fixture-policy.yaml'
policy.write_text('detectors:\n - name: OfflineFixture\n keywords: [offline]\n regex:\n marker: "offline fixture"\n', encoding='ascii')
recovery.real_cli = True
result = recovery.run(trufflehog_config=str(policy), no_verification=True)
assert not result['errors'], result['errors']
assert result['scan_meta']['docker_full_recovery']['coverage_complete']
assert len(recovery.downloads) == 4
assert sum('OfflineFixture' in json.dumps(f) for f in result['findings']) >= 3
def test_outer_validation_is_not_proof_of_nested_coverage(tmp_path):
executable = Path(r'C:\Tools\trufflehog.exe')
if not executable.is_file():
pytest.skip('installed TruffleHog is unavailable')
marker = b'OFFLINENESTED_ABCDEFGHIJKLMNOPQRSTUVWXYZ012345'
inner = archive_bytes([('large.txt', b'x' * 8192 + b'\n' + marker + b'\n')])
payload = archive_bytes([('outside.txt', b'plain outside data\n'), ('inner.tar.gz', encoded(inner, 'gzip'))])
path = tmp_path / 'blob'
path.write_bytes(payload)
scanner.validate_docker_content_artifact(path, descriptor(payload), max_member_bytes=1024)
policy = tmp_path / 'fixture-policy.yaml'
policy.write_text('detectors:\n - name: OfflineNested\n keywords: [OFFLINENESTED]\n regex:\n marker: "OFFLINENESTED_[A-Z0-9]{32}"\n', encoding='ascii')
env = {key: os.environ[key] for key in ('SYSTEMROOT', 'WINDIR', 'PATH') if key in os.environ}
env.update(TEMP=str(tmp_path), TMP=str(tmp_path), HOME=str(tmp_path), USERPROFILE=str(tmp_path),
DOCKER_CONFIG=str(tmp_path), HTTP_PROXY='http://127.0.0.1:1', HTTPS_PROXY='http://127.0.0.1:1')
for limit, level, expected in [('1MB', '0', True), ('1KB', '0', False), ('1KB', '2', False)]:
command = [str(executable), 'filesystem', str(path), '--json', '--no-update', '--no-verification',
'--config', str(policy), '--concurrency', '1', '--archive-max-size', limit,
'--archive-max-depth', '4', '--archive-timeout', '5s', '--log-level', level]
completed = subprocess.run(command, env=env, cwd=tmp_path, capture_output=True, timeout=30)
assert completed.returncode == 0
output = [json.loads(line) for line in completed.stdout.decode().splitlines() if line.strip()]
assert any('OfflineNested' in json.dumps(item) for item in output) == expected
diagnostics = [json.loads(line) for line in completed.stderr.decode().splitlines() if line.startswith('{')]
finished = next(item for item in diagnostics if item.get('msg') == 'finished scanning')
assert finished['chunks'] > 0 # An outer sibling is scanned even when nested data is skipped.
skips = [item for item in diagnostics if item.get('msg') == 'skipping file: size exceeds max allowed']
if limit == '1KB' and level == '2':
assert skips and skips[0]['level'] == 'info-2'
assert skips[0]['filename'] == 'large.txt' and skips[0]['size'] == 8240 and skips[0]['limit'] == 1024
else:
assert not skips