Initial server source import
This commit is contained in:
@@ -0,0 +1,712 @@
|
||||
import builtins
|
||||
import contextlib
|
||||
import gzip
|
||||
import hashlib
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tarfile
|
||||
import time
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
import zstandard
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / 'app'))
|
||||
import scanner
|
||||
import scanner_db
|
||||
|
||||
|
||||
TARGET = 'offline/fixture@sha256:' + 'a' * 64
|
||||
MEDIA = 'application/vnd.oci.image.layer.v1.tar'
|
||||
FINISHED = json.dumps({'level': 'info-0', 'msg': 'finished scanning'})
|
||||
GZIP_WARNING = json.dumps({'level': 'error', 'msg': 'error processing layer', 'error': 'gzip: invalid header'})
|
||||
|
||||
|
||||
def archive_bytes(entries=None, format=tarfile.USTAR_FORMAT):
|
||||
output = io.BytesIO()
|
||||
with tarfile.open(fileobj=output, mode='w', format=format) as archive:
|
||||
for name, data in entries or [('first.txt', b'offline fixture\n'), ('second.txt', b'other fixture\n')]:
|
||||
member = tarfile.TarInfo(name)
|
||||
member.size = len(data)
|
||||
archive.addfile(member, io.BytesIO(data))
|
||||
return output.getvalue()
|
||||
|
||||
|
||||
def encoded(raw, codec):
|
||||
if codec == 'gzip':
|
||||
return gzip.compress(raw, mtime=0)
|
||||
if codec == 'zstd':
|
||||
return zstandard.ZstdCompressor(write_checksum=True).compress(raw)
|
||||
return raw
|
||||
|
||||
|
||||
def descriptor(payload, codec='raw', kind='layer'):
|
||||
return {'digest': 'sha256:' + hashlib.sha256(payload).hexdigest(), 'size': len(payload),
|
||||
'media_type': 'application/vnd.oci.image.config.v1+json' if kind == 'config' else MEDIA + ('' if codec == 'raw' else '+' + codec),
|
||||
'kind': kind}
|
||||
|
||||
|
||||
def pax_record(key, value):
|
||||
data = key + b'=' + value + b'\n'
|
||||
length = len(data) + 3
|
||||
while True:
|
||||
adjusted = len(str(length)) + 1 + len(data)
|
||||
if adjusted == length:
|
||||
return str(length).encode() + b' ' + data
|
||||
length = adjusted
|
||||
|
||||
|
||||
def pax_archive(body, kind=tarfile.XHDTYPE):
|
||||
header = tarfile.TarInfo('pax')
|
||||
header.type = kind
|
||||
header.size = len(body)
|
||||
return header.tobuf() + body + b'\0' * (-len(body) % 512) + archive_bytes()
|
||||
|
||||
|
||||
@pytest.mark.parametrize('codec', ['raw', 'gzip', 'zstd'])
|
||||
def test_complete_valid_archives(tmp_path, codec):
|
||||
raw = archive_bytes()
|
||||
payload = encoded(raw, codec)
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload, codec))
|
||||
|
||||
|
||||
@pytest.mark.parametrize('case', ['late_header', 'body_truncated', 'no_terminator', 'trailing_data',
|
||||
'gzip_crc', 'gzip_truncated', 'zstd_crc', 'zstd_truncated', 'zstd_stub'])
|
||||
def test_complete_validation_rejects_malformed_content(tmp_path, case):
|
||||
raw = archive_bytes()
|
||||
codec = 'raw'
|
||||
if case == 'late_header':
|
||||
payload = bytearray(raw)
|
||||
payload[1024] ^= 1
|
||||
elif case == 'body_truncated':
|
||||
payload = raw[:520]
|
||||
elif case == 'no_terminator':
|
||||
payload = raw[:2048]
|
||||
elif case == 'trailing_data':
|
||||
payload = raw + b'not padding'
|
||||
else:
|
||||
codec = 'gzip' if case.startswith('gzip') else 'zstd'
|
||||
payload = bytearray(encoded(raw, codec))
|
||||
if case.endswith('crc'):
|
||||
payload[-8 if codec == 'gzip' else -1] ^= 1
|
||||
elif case.endswith('truncated'):
|
||||
payload = payload[:-1]
|
||||
else:
|
||||
payload = b'\x28\xb5\x2f\xfd'
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
with pytest.raises(scanner.DockerContentScanError) as error:
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload, codec))
|
||||
assert error.value.error_code == 'invalid_layer_archive'
|
||||
assert not error.value.retryable
|
||||
|
||||
|
||||
@pytest.mark.parametrize('bound', [{'max_members': 1}, {'max_member_bytes': 1}, {'max_decoded_bytes': 1024}])
|
||||
def test_validation_limits(tmp_path, bound):
|
||||
payload = archive_bytes()
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
with pytest.raises(scanner.DockerContentScanError) as error:
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload), **bound)
|
||||
assert error.value.error_code == 'archive_limit'
|
||||
|
||||
|
||||
def test_validation_absolute_deadline(tmp_path):
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(archive_bytes())
|
||||
with pytest.raises(scanner.DockerContentScanError) as error:
|
||||
scanner.validate_docker_content_artifact(path, descriptor(path.read_bytes()), deadline=time.monotonic() - 1)
|
||||
assert error.value.error_code == 'archive_timeout'
|
||||
assert error.value.retryable
|
||||
|
||||
|
||||
def test_missing_zstd_is_infrastructure_capability(tmp_path, monkeypatch):
|
||||
payload = encoded(archive_bytes(), 'zstd')
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
original = builtins.__import__
|
||||
|
||||
def importing(name, *args, **kwargs):
|
||||
if name == 'zstandard':
|
||||
raise ImportError('synthetic missing decoder')
|
||||
return original(name, *args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(builtins, '__import__', importing)
|
||||
with pytest.raises(scanner.DockerLayerInfrastructureError) as error:
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload, 'zstd'))
|
||||
assert error.value.error_code == 'archive_decoder_unavailable'
|
||||
assert error.value.category == 'source_configuration'
|
||||
|
||||
|
||||
@pytest.mark.parametrize('codec', ['raw', 'gzip', 'zstd'])
|
||||
def test_pax_and_multiframe_archives(tmp_path, codec):
|
||||
raw = archive_bytes([('long/' + 'x' * 200, b'bounded')], format=tarfile.PAX_FORMAT)
|
||||
payload = raw if codec == 'raw' else encoded(raw[:2048], codec) + encoded(raw[2048:], codec)
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload, codec))
|
||||
|
||||
|
||||
def test_empty_tar_and_zstd_skippable_frames(tmp_path):
|
||||
empty = b'\0' * 1024
|
||||
skip = b'\x50\x2a\x4d\x18\x04\0\0\0skip'
|
||||
payload = skip + encoded(empty, 'zstd') + skip
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload, 'zstd'))
|
||||
|
||||
|
||||
def test_zstd_unknown_size_window_larger_than_one_megabyte(tmp_path):
|
||||
raw = archive_bytes([('large.txt', b'x' * (2 << 20))])
|
||||
payload = zstandard.ZstdCompressor(write_content_size=False).compress(raw)
|
||||
assert zstandard.get_frame_parameters(payload).window_size > 1 << 20
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload, 'zstd'))
|
||||
|
||||
|
||||
def test_validation_never_extracts_member_paths(tmp_path):
|
||||
payload = archive_bytes([('../outside', b'bounded'), ('/absolute/fixture', b'bounded')])
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload))
|
||||
assert list(tmp_path.iterdir()) == [path]
|
||||
|
||||
|
||||
def test_oversized_pax_metadata_rejected_before_read(tmp_path):
|
||||
header = tarfile.TarInfo('pax')
|
||||
header.type = tarfile.XHDTYPE
|
||||
header.size = (1 << 20) + 1
|
||||
payload = header.tobuf()
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
with pytest.raises(scanner.DockerContentScanError) as error:
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload))
|
||||
assert error.value.error_code == 'archive_limit'
|
||||
|
||||
|
||||
@pytest.mark.parametrize('kind', [tarfile.XHDTYPE, tarfile.XGLTYPE, tarfile.SOLARIS_XHDTYPE])
|
||||
@pytest.mark.parametrize('body', [b'9' * 30000 + b' key=value\n', pax_record(b'uid', b'9' * 30000)])
|
||||
def test_pax_huge_numbers_cannot_enter_unbounded_stdlib_parser(tmp_path, monkeypatch, kind, body):
|
||||
def forbidden(*args, **kwargs):
|
||||
raise AssertionError('unbounded stdlib PAX parser entered')
|
||||
|
||||
monkeypatch.setattr(tarfile.TarInfo, '_proc_pax', forbidden)
|
||||
payload = pax_archive(body, kind)
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
started = time.monotonic()
|
||||
with pytest.raises(scanner.DockerContentScanError) as error:
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload), deadline=started + 0.05)
|
||||
assert error.value.error_code in ('archive_limit', 'archive_timeout')
|
||||
assert time.monotonic() - started < 0.5
|
||||
|
||||
|
||||
def test_digit_heavy_pax_path_uses_linear_parser(tmp_path, monkeypatch):
|
||||
monkeypatch.setattr(tarfile.TarInfo, '_proc_pax', lambda *a, **k: pytest.fail('stdlib PAX regex called'))
|
||||
payload = pax_archive(pax_record(b'path', b'9' * 30000))
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload), deadline=time.monotonic() + 0.5)
|
||||
|
||||
|
||||
@pytest.mark.parametrize('size', [(1 << 20) + 1, (64 << 10) + 1])
|
||||
def test_solaris_pax_metadata_and_parse_caps(tmp_path, size):
|
||||
header = tarfile.TarInfo('pax')
|
||||
header.type = tarfile.SOLARIS_XHDTYPE
|
||||
header.size = size
|
||||
payload = header.tobuf()
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
with pytest.raises(scanner.DockerContentScanError) as error:
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload))
|
||||
assert error.value.error_code == 'archive_limit'
|
||||
|
||||
|
||||
def test_solaris_pax_valid_header(tmp_path):
|
||||
payload = pax_archive(pax_record(b'path', b'solaris-name'), tarfile.SOLARIS_XHDTYPE)
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload))
|
||||
|
||||
|
||||
@pytest.mark.parametrize('key,value', [(b'GNU.sparse.map', b'0,0,' * 1000),
|
||||
(b'GNU.sparse.major', b'1'), (b'GNU.sparse.size', b'1')])
|
||||
def test_sparse_pax_rejected_before_map_or_field_materialization(tmp_path, monkeypatch, key, value):
|
||||
def forbidden(*args, **kwargs):
|
||||
raise AssertionError('sparse PAX field/map materialization entered')
|
||||
|
||||
for method in ('_apply_pax_info', '_proc_gnusparse_00', '_proc_gnusparse_01', '_proc_gnusparse_10'):
|
||||
monkeypatch.setattr(tarfile.TarInfo, method, forbidden)
|
||||
payload = pax_archive(pax_record(key, value))
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
with pytest.raises(scanner.DockerContentScanError) as error:
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload), max_member_bytes=1024)
|
||||
assert error.value.error_code in ('archive_limit', 'unsupported_layer_archive')
|
||||
|
||||
|
||||
def test_pax_record_count_bound(tmp_path):
|
||||
payload = pax_archive(pax_record(b'comment', b'x') * 1025)
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
with pytest.raises(scanner.DockerContentScanError) as error:
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload))
|
||||
assert error.value.error_code == 'archive_limit'
|
||||
|
||||
|
||||
def test_header_parse_checks_deadline_on_return(tmp_path, monkeypatch):
|
||||
original = tarfile.TarInfo._proc_builtin
|
||||
|
||||
def delayed(*args, **kwargs):
|
||||
member = original(*args, **kwargs)
|
||||
time.sleep(0.06)
|
||||
return member
|
||||
|
||||
monkeypatch.setattr(tarfile.TarInfo, '_proc_builtin', delayed)
|
||||
payload = archive_bytes()
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
with pytest.raises(scanner.DockerContentScanError) as error:
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload), deadline=time.monotonic() + 0.05)
|
||||
assert error.value.error_code == 'archive_timeout'
|
||||
|
||||
|
||||
def test_deadline_checked_during_archive_walk(tmp_path, monkeypatch):
|
||||
payload = archive_bytes()
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
clock = [100.0]
|
||||
|
||||
def ticking():
|
||||
clock[0] += 0.01
|
||||
return clock[0]
|
||||
|
||||
monkeypatch.setattr(scanner.time, 'monotonic', ticking)
|
||||
with pytest.raises(scanner.DockerContentScanError) as error:
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload), deadline=100.08)
|
||||
assert error.value.error_code == 'archive_timeout'
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def recovery(tmp_path, monkeypatch):
|
||||
monkeypatch.delenv('DOCKER_CONFIG', raising=False)
|
||||
monkeypatch.delenv('DOCKER_TOKEN', raising=False)
|
||||
monkeypatch.delenv('REGISTRY_AUTH_FILE', raising=False)
|
||||
monkeypatch.delenv('HOMEDRIVE', raising=False)
|
||||
monkeypatch.delenv('HOMEPATH', raising=False)
|
||||
monkeypatch.setenv('HOME', str(tmp_path))
|
||||
monkeypatch.setenv('USERPROFILE', str(tmp_path))
|
||||
monkeypatch.setenv('XDG_RUNTIME_DIR', str(tmp_path))
|
||||
raw = archive_bytes()
|
||||
payloads = [b'{"history":[]}', raw, encoded(raw, 'gzip'), encoded(raw, 'zstd')]
|
||||
items = [descriptor(payloads[0], kind='config')] + [descriptor(data, codec) for data, codec in zip(payloads[1:], ('raw', 'gzip', 'zstd'))]
|
||||
resolved = {'version': 1, 'image': TARGET, 'repository': 'offline/fixture',
|
||||
'manifest_digest': TARGET.split('@')[1], 'config': items[0], 'layers': items[1:]}
|
||||
state = SimpleNamespace(resolved=resolved, payloads={item['digest']: data for item, data in zip(items, payloads)},
|
||||
commands=[], downloads=[], resolutions=[], full_stderr=GZIP_WARNING + '\n' + FINISHED,
|
||||
blob_stderr=FINISHED, after_full=lambda: None, real_cli=False)
|
||||
monkeypatch.setattr(scanner, 'get_work_dir', lambda: str(tmp_path))
|
||||
monkeypatch.setattr(scanner, 'harden_private_directory', lambda *a, **k: None)
|
||||
monkeypatch.setattr(scanner, 'write_temp_owner', lambda *a, **k: None)
|
||||
monkeypatch.setattr(scanner, 'durable_unlink', lambda path: os.unlink(path))
|
||||
monkeypatch.setattr(scanner, 'cleanup_command_work_dir', lambda path: shutil.rmtree(path))
|
||||
monkeypatch.setattr(scanner, 'apply_finding_filters', lambda result, *a, **k: result)
|
||||
monkeypatch.setattr(scanner, 'docker_token_manager', scanner.DockerTokenManager())
|
||||
|
||||
def resolve(target, **kwargs):
|
||||
state.resolutions.append((target, kwargs))
|
||||
return state.resolved, scanner.DockerRegistryAuth('synthetic-bearer')
|
||||
|
||||
def download(repository, item, path, bearer_auth, **kwargs):
|
||||
state.downloads.append((item, kwargs))
|
||||
Path(path).write_bytes(state.payloads[item['digest']])
|
||||
return scanner.DockerBlobDownloadOutcome(path, item['size'], item['size'], 1, bearer_auth)
|
||||
|
||||
def command(args, timeout, env, **kwargs):
|
||||
state.commands.append((args, timeout, kwargs))
|
||||
if args[1] == 'docker':
|
||||
state.after_full()
|
||||
return scanner.streamed_output_from_text(
|
||||
stdout=json.dumps({'DetectorName': 'Fixture', 'Raw': 'synthetic-full'}), stderr=state.full_stderr)
|
||||
if state.real_cli:
|
||||
safe_env = {key: os.environ[key] for key in ('SYSTEMROOT', 'WINDIR', 'PATH') if key in os.environ}
|
||||
safe_env.update(TEMP=str(tmp_path), TMP=str(tmp_path), HOME=str(tmp_path), USERPROFILE=str(tmp_path),
|
||||
DOCKER_CONFIG=str(tmp_path), HTTP_PROXY='http://127.0.0.1:1', HTTPS_PROXY='http://127.0.0.1:1')
|
||||
completed = subprocess.run(args, env=safe_env, cwd=tmp_path, capture_output=True, timeout=30)
|
||||
return scanner.streamed_output_from_text(stdout=completed.stdout.decode(), stderr=completed.stderr.decode(), returncode=completed.returncode)
|
||||
return scanner.streamed_output_from_text(
|
||||
stdout=json.dumps({'DetectorName': 'Fixture', 'Raw': 'synthetic-blob',
|
||||
'SourceMetadata': {'Data': {'Filesystem': {'file': args[2]}}}}), stderr=state.blob_stderr)
|
||||
|
||||
monkeypatch.setattr(scanner, 'resolve_docker_content_manifest', resolve)
|
||||
monkeypatch.setattr(scanner, 'stream_docker_registry_blob', download)
|
||||
monkeypatch.setattr(scanner, 'run_command_streamed', command)
|
||||
state.run = lambda **kwargs: scanner.scan_docker_image(TARGET, log_target=False, docker_recovery_min_free_bytes=0, **kwargs)
|
||||
return state
|
||||
|
||||
|
||||
def test_recovers_all_codecs_and_config_with_same_deadline(recovery):
|
||||
result = recovery.run()
|
||||
assert not result['errors']
|
||||
assert not result.get('warnings') and not result.get('degraded')
|
||||
assert len(result['findings']) == 5
|
||||
scope = result['scan_meta']['docker_full_recovery']
|
||||
assert scope['coverage_complete'] and scope['scanned_descriptors'] == scope['descriptor_count'] == 4
|
||||
assert len(recovery.downloads) == 4
|
||||
deadline = recovery.commands[0][2]['deadline']
|
||||
assert recovery.resolutions[0][1]['deadline'] == deadline
|
||||
assert all(kwargs['deadline'] <= deadline for _, kwargs in recovery.downloads)
|
||||
assert all(kwargs['deadline'] <= deadline for _, _, kwargs in recovery.commands)
|
||||
for finding in result['findings'][1:]:
|
||||
data = finding['SourceMetadata']['Data']
|
||||
assert data['DockerContent']['manifest_digest'] == TARGET.split('@')[1]
|
||||
assert data['Filesystem']['file'].startswith('docker://offline/fixture@')
|
||||
|
||||
|
||||
def test_duplicate_digest_scanned_once_preserves_positions(recovery):
|
||||
recovery.resolved['layers'].append(dict(recovery.resolved['layers'][0]))
|
||||
result = recovery.run()
|
||||
assert result['scan_meta']['docker_full_recovery']['scanned_descriptors'] == 5
|
||||
assert len(recovery.downloads) == 4
|
||||
assert result['findings'][2]['SourceMetadata']['Data']['DockerContent']['positions'] == [1, 4]
|
||||
|
||||
|
||||
@pytest.mark.parametrize('field,value', [('media_type', MEDIA + '+gzip'), ('size', 12), ('kind', 'config')])
|
||||
def test_conflicting_digest_preflight(recovery, field, value):
|
||||
if field == 'kind':
|
||||
recovery.resolved['config']['digest'] = recovery.resolved['layers'][0]['digest']
|
||||
else:
|
||||
other = dict(recovery.resolved['layers'][0])
|
||||
other[field] = value
|
||||
recovery.resolved['layers'].append(other)
|
||||
result = recovery.run()
|
||||
assert result['error_class'] == 'conflicting_descriptors'
|
||||
assert not recovery.downloads
|
||||
|
||||
|
||||
@pytest.mark.parametrize('limit,value', [('max_layers', 1), ('layer_max_bytes', 1024), ('image_max_bytes', 1024)])
|
||||
def test_all_fitting_preflight_never_scans_top_n(recovery, limit, value):
|
||||
limits = {'config_max_bytes': 1024, 'layer_max_bytes': 1 << 20, 'image_max_bytes': 2 << 20,
|
||||
'max_layers': 8, 'archive_max_size_bytes': 1 << 20, 'archive_max_depth': 4,
|
||||
'archive_timeout_sec': 5, 'blob_timeout_sec': 10, 'filesystem_concurrency': 1, 'blob_max_attempts': 3}
|
||||
limits[limit] = value
|
||||
result = recovery.run(docker_recovery_limits=limits)
|
||||
assert result['errors'] and result['error_class'] == 'recovery_budget'
|
||||
assert not recovery.downloads
|
||||
assert not result['scan_meta']['docker_full_recovery']['coverage_complete']
|
||||
assert len(result['findings']) == 1
|
||||
|
||||
|
||||
def test_oversized_config_preflight(recovery):
|
||||
recovery.resolved['config']['size'] = (1 << 20) + 1
|
||||
result = recovery.run()
|
||||
assert result['error_class'] == 'recovery_budget'
|
||||
assert not recovery.downloads
|
||||
|
||||
|
||||
def test_incompatible_decoded_policy_rejected_before_any_recovery_download(recovery):
|
||||
limits = {'config_max_bytes': 1024, 'layer_max_bytes': 1 << 20, 'image_max_bytes': 2 << 20,
|
||||
'max_layers': 8, 'archive_max_size_bytes': 4 << 30, 'archive_max_depth': 4,
|
||||
'archive_timeout_sec': 5, 'blob_timeout_sec': 10, 'filesystem_concurrency': 1, 'blob_max_attempts': 3}
|
||||
result = recovery.run(docker_recovery_limits=limits)
|
||||
assert result['error_class'] == 'archive_policy_incompatible'
|
||||
assert result['source_failure_category'] == 'source_configuration'
|
||||
assert not recovery.resolutions and not recovery.downloads
|
||||
|
||||
|
||||
def test_incompatible_decoded_policy_rejected_before_layer_download(recovery, monkeypatch):
|
||||
plan = {'image': TARGET, 'limits': {'archive_max_size_bytes': 4 << 30}}
|
||||
monkeypatch.setattr(scanner, 'validate_docker_layer_plan', lambda value: value)
|
||||
monkeypatch.setattr(scanner, 'canonical_docker_layer_plan_bytes', lambda value: b'fixture')
|
||||
with pytest.raises(scanner.DockerLayerInfrastructureError) as error:
|
||||
scanner.scan_docker_layer_plan(TARGET, {'plan': plan, 'plan_sha256': hashlib.sha256(b'fixture').hexdigest()})
|
||||
assert error.value.error_code == 'archive_policy_incompatible'
|
||||
assert not recovery.downloads
|
||||
|
||||
|
||||
def test_unsupported_layer_media_preflight(recovery):
|
||||
recovery.resolved['layers'][0]['media_type'] = 'application/octet-stream'
|
||||
result = recovery.run()
|
||||
assert result['error_class'] == 'unsupported_media_type'
|
||||
assert not recovery.downloads
|
||||
|
||||
|
||||
def test_resolved_root_must_match_original_digest(recovery):
|
||||
recovery.resolved['manifest_digest'] = 'sha256:' + 'b' * 64
|
||||
result = recovery.run()
|
||||
assert result['error_class'] == 'recovery_identity_mismatch'
|
||||
assert not recovery.downloads
|
||||
|
||||
|
||||
def test_no_fresh_budget_after_full_engine(recovery, monkeypatch):
|
||||
clock = [100.0]
|
||||
monkeypatch.setattr(scanner.time, 'monotonic', lambda: clock[0])
|
||||
recovery.after_full = lambda: clock.__setitem__(0, 110.0)
|
||||
result = recovery.run(timeout_sec=10)
|
||||
assert result['error_class'] == 'timeout'
|
||||
assert not recovery.resolutions and not recovery.downloads
|
||||
assert recovery.commands[0][2]['deadline'] == 110
|
||||
|
||||
|
||||
@pytest.mark.parametrize('stage', ['cleanup', 'filter'])
|
||||
def test_full_scan_without_fallback_cannot_finish_after_deadline(recovery, monkeypatch, stage):
|
||||
recovery.full_stderr = FINISHED
|
||||
clock = [100.0]
|
||||
monkeypatch.setattr(scanner.time, 'monotonic', lambda: clock[0])
|
||||
original = scanner.run_command_streamed
|
||||
if stage == 'cleanup':
|
||||
@contextlib.contextmanager
|
||||
def command(*args, **kwargs):
|
||||
with original(*args, **kwargs) as output:
|
||||
yield output
|
||||
clock[0] = 111.0
|
||||
monkeypatch.setattr(scanner, 'run_command_streamed', command)
|
||||
else:
|
||||
def filter_result(result, *args, **kwargs):
|
||||
clock[0] = 111.0
|
||||
return result
|
||||
monkeypatch.setattr(scanner, 'apply_finding_filters', filter_result)
|
||||
result = recovery.run(timeout_sec=10)
|
||||
assert result['error_class'] == 'timeout'
|
||||
assert result['scan_meta']['docker_deadline_exceeded']
|
||||
assert len(result['findings']) == 1
|
||||
assert not recovery.resolutions
|
||||
|
||||
|
||||
def test_post_filter_timeout_does_not_erase_unrelated_error(recovery, monkeypatch):
|
||||
recovery.full_stderr = json.dumps({'level': 'error', 'msg': 'fatal synthetic failure'}) + '\n' + FINISHED
|
||||
clock = [100.0]
|
||||
monkeypatch.setattr(scanner.time, 'monotonic', lambda: clock[0])
|
||||
|
||||
def filtering(result, *args, **kwargs):
|
||||
clock[0] = 111.0
|
||||
return result
|
||||
|
||||
monkeypatch.setattr(scanner, 'apply_finding_filters', filtering)
|
||||
result = recovery.run(timeout_sec=10)
|
||||
assert result['error_class'] != 'timeout'
|
||||
assert len(result['errors']) >= 2
|
||||
assert len(result['findings']) == 1
|
||||
|
||||
|
||||
@pytest.mark.parametrize('extra', [json.dumps({'level': 'error', 'msg': 'fatal scan error'}),
|
||||
json.dumps({'level': 'error', 'msg': 'non-critical error processing chunk'})])
|
||||
def test_unrelated_diagnostics_are_never_cleared(recovery, extra):
|
||||
recovery.full_stderr += '\n' + extra
|
||||
result = recovery.run()
|
||||
assert result['errors']
|
||||
assert not recovery.resolutions
|
||||
assert len(result['findings']) == 1
|
||||
|
||||
|
||||
def test_only_exact_gzip_warning_triggers_recovery(recovery):
|
||||
recovery.full_stderr = GZIP_WARNING.replace('gzip: invalid header', 'gzip: invalid header trailing text') + '\n' + FINISHED
|
||||
assert recovery.run()['errors']
|
||||
assert not recovery.resolutions
|
||||
|
||||
|
||||
@pytest.mark.parametrize('target', ['offline/fixture:latest', 'example.invalid/fixture@sha256:' + 'a' * 64])
|
||||
def test_no_mutable_or_external_registry_recovery(recovery, target):
|
||||
result = scanner.scan_docker_image(target, log_target=False)
|
||||
assert result['errors']
|
||||
assert not recovery.resolutions
|
||||
|
||||
|
||||
def test_recovery_warning_retains_partial_findings(recovery):
|
||||
recovery.blob_stderr = json.dumps({'level': 'error', 'msg': 'non-critical error processing chunk'}) + '\n' + FINISHED
|
||||
result = recovery.run()
|
||||
assert result['errors'] and len(result['findings']) == 2
|
||||
assert not result['scan_meta']['docker_full_recovery']['coverage_complete']
|
||||
|
||||
|
||||
def test_transfer_failure_preserves_prior_findings_without_source_disable(recovery, monkeypatch):
|
||||
original = scanner.stream_docker_registry_blob
|
||||
|
||||
def download(*args, **kwargs):
|
||||
if len(recovery.downloads) == 1:
|
||||
raise scanner.DockerContentTransferError('digest_mismatch', 'private synthetic detail', False)
|
||||
return original(*args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(scanner, 'stream_docker_registry_blob', download)
|
||||
result = recovery.run()
|
||||
assert result['error_class'] == 'digest_mismatch'
|
||||
assert not result['retryable'] and not result.get('source_failure')
|
||||
assert len(result['findings']) == 2
|
||||
assert 'private synthetic detail' not in json.dumps(result)
|
||||
|
||||
|
||||
def test_malformed_manifest_is_not_infrastructure(recovery, monkeypatch):
|
||||
def resolve(*args, **kwargs):
|
||||
raise scanner.DockerRegistryResolutionError('private synthetic detail')
|
||||
|
||||
monkeypatch.setattr(scanner, 'resolve_docker_content_manifest', resolve)
|
||||
result = recovery.run()
|
||||
assert result['error_class'] == 'recovery_manifest_invalid'
|
||||
assert not result['retryable'] and not result.get('source_failure')
|
||||
assert 'private synthetic detail' not in json.dumps(result)
|
||||
|
||||
|
||||
def test_missing_decoder_during_recovery_keeps_already_scanned_findings(recovery, monkeypatch):
|
||||
original = builtins.__import__
|
||||
|
||||
def importing(name, *args, **kwargs):
|
||||
if name == 'zstandard':
|
||||
raise ImportError('synthetic missing decoder')
|
||||
return original(name, *args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(builtins, '__import__', importing)
|
||||
result = recovery.run()
|
||||
assert result['error_class'] == 'archive_decoder_unavailable'
|
||||
assert result['source_failure_category'] == 'source_configuration'
|
||||
assert not result['scan_meta']['docker_full_recovery']['coverage_complete']
|
||||
assert len(result['findings']) == 4
|
||||
|
||||
|
||||
def test_cleanup_failure_never_reports_full_coverage(recovery, monkeypatch):
|
||||
def cleanup(path):
|
||||
shutil.rmtree(path)
|
||||
raise RuntimeError('private cleanup path')
|
||||
|
||||
monkeypatch.setattr(scanner, 'cleanup_command_work_dir', cleanup)
|
||||
result = recovery.run()
|
||||
assert result['error_class'] == 'private_cleanup'
|
||||
assert not result['scan_meta']['docker_full_recovery']['coverage_complete']
|
||||
assert len(result['findings']) == 5
|
||||
assert 'private cleanup path' not in json.dumps(result)
|
||||
|
||||
|
||||
def test_cleanup_is_inside_the_original_deadline(recovery, monkeypatch):
|
||||
clock = [100.0]
|
||||
monkeypatch.setattr(scanner.time, 'monotonic', lambda: clock[0])
|
||||
|
||||
def cleanup(path):
|
||||
shutil.rmtree(path)
|
||||
clock[0] = 111.0
|
||||
|
||||
monkeypatch.setattr(scanner, 'cleanup_command_work_dir', cleanup)
|
||||
result = recovery.run(timeout_sec=10)
|
||||
assert result['error_class'] == 'timeout'
|
||||
assert not result['scan_meta']['docker_full_recovery']['coverage_complete']
|
||||
|
||||
|
||||
def test_unknown_config_fails_closed_without_loading_or_exposing_it(recovery):
|
||||
result = recovery.run(config_dir='unmanaged-private-config')
|
||||
assert result['error_class'] == 'recovery_auth_unavailable'
|
||||
assert result['source_failure_category'] == 'source_configuration'
|
||||
assert not recovery.resolutions
|
||||
assert 'unmanaged-private-config' not in json.dumps(result)
|
||||
|
||||
|
||||
@pytest.mark.parametrize('location', ['home', 'profile', 'podman', 'registry_env'])
|
||||
def test_implicit_keychain_configuration_blocks_identity_switch(recovery, tmp_path, monkeypatch, location):
|
||||
if location == 'profile':
|
||||
root = tmp_path / 'profile'
|
||||
monkeypatch.setenv('USERPROFILE', str(root))
|
||||
path = root / '.docker' / 'config.json'
|
||||
elif location == 'podman':
|
||||
path = tmp_path / 'containers' / 'auth.json'
|
||||
elif location == 'registry_env':
|
||||
path = tmp_path / 'private-auth.json'
|
||||
monkeypatch.setenv('REGISTRY_AUTH_FILE', str(path))
|
||||
else:
|
||||
path = tmp_path / '.docker' / 'config.json'
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text('do-not-load-this-credential-configuration', encoding='ascii')
|
||||
result = recovery.run()
|
||||
assert result['error_class'] == 'recovery_auth_unavailable'
|
||||
assert result['source_failure_category'] == 'source_configuration'
|
||||
assert not recovery.resolutions
|
||||
assert 'do-not-load' not in json.dumps(result) and str(path) not in json.dumps(result)
|
||||
|
||||
|
||||
def test_implicit_config_snapshot_survives_removal_during_full_scan(recovery, tmp_path):
|
||||
path = tmp_path / '.docker' / 'config.json'
|
||||
path.parent.mkdir()
|
||||
path.write_text('{}', encoding='ascii')
|
||||
recovery.after_full = path.unlink
|
||||
result = recovery.run()
|
||||
assert result['error_class'] == 'recovery_auth_unavailable'
|
||||
assert not recovery.resolutions
|
||||
|
||||
|
||||
def test_managed_config_uses_existing_token_exchange(recovery, monkeypatch):
|
||||
scanner.docker_token_manager.accounts = [SimpleNamespace(name='selected', config_dir='managed'), SimpleNamespace(name='other', config_dir='other')]
|
||||
calls = []
|
||||
|
||||
def token(*args, **kwargs):
|
||||
calls.append(kwargs)
|
||||
return scanner.DockerRegistryAuth('private-token', 'selected')
|
||||
|
||||
monkeypatch.setattr(scanner, 'docker_registry_bearer_token', token)
|
||||
result = recovery.run(config_dir='managed')
|
||||
assert not result['errors']
|
||||
assert calls[0]['excluded_accounts'] == {'other'}
|
||||
assert recovery.resolutions[0][1]['bearer_auth'].account_name == 'selected'
|
||||
assert 'private-token' not in json.dumps(result)
|
||||
|
||||
|
||||
def test_policy_versions_invalidate_old_coverage():
|
||||
assert scanner_db.DOCKER_ADAPTIVE_EXECUTION_VERSION == 'docker-layer-execution-v4'
|
||||
limits = {'config_max_bytes': 1024, 'layer_max_bytes': 1024, 'image_max_bytes': 2048,
|
||||
'max_layers': 2, 'archive_max_size_bytes': 1024, 'archive_max_depth': 2,
|
||||
'archive_timeout_sec': 5, 'blob_timeout_sec': 10, 'filesystem_concurrency': 1, 'blob_max_attempts': 3}
|
||||
old_payload = {'limits': limits, 'scan_policy_sha256': 'b' * 64}
|
||||
old_hash = hashlib.sha256(json.dumps(old_payload, ensure_ascii=True, sort_keys=True, separators=(',', ':')).encode()).hexdigest()
|
||||
assert scanner_db.docker_layer_coverage_policy_sha256('b' * 64, limits) != old_hash
|
||||
|
||||
|
||||
def test_installed_filesystem_cli_recovers_real_synthetic_codecs(recovery, tmp_path, monkeypatch):
|
||||
executable = Path(r'C:\Tools\trufflehog.exe')
|
||||
if not executable.is_file():
|
||||
pytest.skip('installed TruffleHog is unavailable')
|
||||
monkeypatch.setattr(scanner, 'get_trufflehog_cmd', lambda: str(executable))
|
||||
policy = tmp_path / 'fixture-policy.yaml'
|
||||
policy.write_text('detectors:\n - name: OfflineFixture\n keywords: [offline]\n regex:\n marker: "offline fixture"\n', encoding='ascii')
|
||||
recovery.real_cli = True
|
||||
result = recovery.run(trufflehog_config=str(policy), no_verification=True)
|
||||
assert not result['errors'], result['errors']
|
||||
assert result['scan_meta']['docker_full_recovery']['coverage_complete']
|
||||
assert len(recovery.downloads) == 4
|
||||
assert sum('OfflineFixture' in json.dumps(f) for f in result['findings']) >= 3
|
||||
|
||||
|
||||
def test_outer_validation_is_not_proof_of_nested_coverage(tmp_path):
|
||||
executable = Path(r'C:\Tools\trufflehog.exe')
|
||||
if not executable.is_file():
|
||||
pytest.skip('installed TruffleHog is unavailable')
|
||||
marker = b'OFFLINENESTED_ABCDEFGHIJKLMNOPQRSTUVWXYZ012345'
|
||||
inner = archive_bytes([('large.txt', b'x' * 8192 + b'\n' + marker + b'\n')])
|
||||
payload = archive_bytes([('outside.txt', b'plain outside data\n'), ('inner.tar.gz', encoded(inner, 'gzip'))])
|
||||
path = tmp_path / 'blob'
|
||||
path.write_bytes(payload)
|
||||
scanner.validate_docker_content_artifact(path, descriptor(payload), max_member_bytes=1024)
|
||||
policy = tmp_path / 'fixture-policy.yaml'
|
||||
policy.write_text('detectors:\n - name: OfflineNested\n keywords: [OFFLINENESTED]\n regex:\n marker: "OFFLINENESTED_[A-Z0-9]{32}"\n', encoding='ascii')
|
||||
env = {key: os.environ[key] for key in ('SYSTEMROOT', 'WINDIR', 'PATH') if key in os.environ}
|
||||
env.update(TEMP=str(tmp_path), TMP=str(tmp_path), HOME=str(tmp_path), USERPROFILE=str(tmp_path),
|
||||
DOCKER_CONFIG=str(tmp_path), HTTP_PROXY='http://127.0.0.1:1', HTTPS_PROXY='http://127.0.0.1:1')
|
||||
for limit, level, expected in [('1MB', '0', True), ('1KB', '0', False), ('1KB', '2', False)]:
|
||||
command = [str(executable), 'filesystem', str(path), '--json', '--no-update', '--no-verification',
|
||||
'--config', str(policy), '--concurrency', '1', '--archive-max-size', limit,
|
||||
'--archive-max-depth', '4', '--archive-timeout', '5s', '--log-level', level]
|
||||
completed = subprocess.run(command, env=env, cwd=tmp_path, capture_output=True, timeout=30)
|
||||
assert completed.returncode == 0
|
||||
output = [json.loads(line) for line in completed.stdout.decode().splitlines() if line.strip()]
|
||||
assert any('OfflineNested' in json.dumps(item) for item in output) == expected
|
||||
diagnostics = [json.loads(line) for line in completed.stderr.decode().splitlines() if line.startswith('{')]
|
||||
finished = next(item for item in diagnostics if item.get('msg') == 'finished scanning')
|
||||
assert finished['chunks'] > 0 # An outer sibling is scanned even when nested data is skipped.
|
||||
skips = [item for item in diagnostics if item.get('msg') == 'skipping file: size exceeds max allowed']
|
||||
if limit == '1KB' and level == '2':
|
||||
assert skips and skips[0]['level'] == 'info-2'
|
||||
assert skips[0]['filename'] == 'large.txt' and skips[0]['size'] == 8240 and skips[0]['limit'] == 1024
|
||||
else:
|
||||
assert not skips
|
||||
Reference in New Issue
Block a user