127 lines
4.0 KiB
Python
127 lines
4.0 KiB
Python
import contextlib
|
|
import json
|
|
import os
|
|
from pathlib import Path
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
|
|
|
|
_DYNAMIC_FIELDS = {
|
|
'duration_sec', 'finding_uid', 'scan_duration', 'scan_duration_ms',
|
|
'scan_started_at', 'source_manager_worker_id', 'timestamp', 'trace_id',
|
|
'transfer_duration_ms', 'worker_id',
|
|
}
|
|
_TEMP_COMPONENT = re.compile(
|
|
r'(?i)(?:trufflehog-run|trufflehog-\d+|docker-layer)-[^/\\\s"]+',
|
|
)
|
|
_ISO_TIMESTAMP = re.compile(
|
|
r'\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}:\d{2})',
|
|
)
|
|
_LOG_ID = re.compile(
|
|
r'("(?:trace|worker|source_manager_worker)_id"\s*:\s*")[^"]+("\s*[,}])',
|
|
)
|
|
|
|
|
|
def configured_trufflehog():
|
|
candidates = (
|
|
os.getenv('TRUF_TEST_TRUFFLEHOG'),
|
|
r'C:\Tools\trufflehog.exe',
|
|
shutil.which('trufflehog'),
|
|
)
|
|
return next((Path(value) for value in candidates if value and Path(value).is_file()), None)
|
|
|
|
|
|
@contextlib.contextmanager
|
|
def native_streamed_command(command, timeout, env=None, **_kwargs):
|
|
import scanner
|
|
|
|
completed = subprocess.run(
|
|
command,
|
|
stdin=subprocess.DEVNULL,
|
|
stdout=subprocess.PIPE,
|
|
stderr=subprocess.PIPE,
|
|
env=env,
|
|
check=False,
|
|
timeout=max(1.0, float(timeout)),
|
|
)
|
|
with scanner.streamed_output_from_text(
|
|
completed.stdout.decode('utf-8', errors='replace'),
|
|
completed.stderr.decode('utf-8', errors='replace'),
|
|
completed.returncode,
|
|
) as output:
|
|
yield output
|
|
|
|
|
|
def normalized_bundle_evidence(path):
|
|
from result_bundle import ResultBundleReader
|
|
|
|
def normalize(value):
|
|
if isinstance(value, dict):
|
|
return {
|
|
key: normalize(item)
|
|
for key, item in sorted(value.items())
|
|
if key not in _DYNAMIC_FIELDS
|
|
}
|
|
if isinstance(value, list):
|
|
return [normalize(item) for item in value]
|
|
if isinstance(value, str):
|
|
value = _TEMP_COMPONENT.sub('<temporary-scan>', value)
|
|
value = _ISO_TIMESTAMP.sub('<timestamp>', value)
|
|
value = _LOG_ID.sub(r'\1<log-id>\2', value)
|
|
try:
|
|
parsed = json.loads(value)
|
|
except (TypeError, ValueError):
|
|
return value
|
|
if isinstance(parsed, (dict, list)):
|
|
return json.dumps(
|
|
normalize(parsed), ensure_ascii=True, sort_keys=True,
|
|
separators=(',', ':'),
|
|
)
|
|
return value
|
|
return value
|
|
|
|
def ordered(values):
|
|
normalized = [normalize(value) for value in values]
|
|
return sorted(
|
|
normalized,
|
|
key=lambda value: json.dumps(
|
|
value, ensure_ascii=True, sort_keys=True, separators=(',', ':'),
|
|
),
|
|
)
|
|
|
|
reader = ResultBundleReader(path)
|
|
reader.validate()
|
|
return {
|
|
'metadata': normalize(reader.metadata()),
|
|
'findings': ordered(reader.iter_findings()),
|
|
'errors': ordered(reader.iter_errors()),
|
|
'candidates': ordered(reader.iter_candidates()),
|
|
}
|
|
|
|
|
|
def bundle_evidence_difference_paths(left, right, path='$', limit=50):
|
|
differences = []
|
|
|
|
def compare(first, second, current):
|
|
if len(differences) >= limit:
|
|
return
|
|
if type(first) is not type(second):
|
|
differences.append(current)
|
|
elif isinstance(first, dict):
|
|
for key in sorted(set(first) | set(second)):
|
|
if key not in first or key not in second:
|
|
differences.append(f'{current}.{key}')
|
|
else:
|
|
compare(first[key], second[key], f'{current}.{key}')
|
|
elif isinstance(first, list):
|
|
if len(first) != len(second):
|
|
differences.append(f'{current}.length')
|
|
for index, (first_item, second_item) in enumerate(zip(first, second)):
|
|
compare(first_item, second_item, f'{current}[{index}]')
|
|
elif first != second:
|
|
differences.append(current)
|
|
|
|
compare(left, right, path)
|
|
return differences
|