372 lines
16 KiB
Python
372 lines
16 KiB
Python
import hashlib
|
|
import json
|
|
from pathlib import Path
|
|
import sys
|
|
from unittest import mock
|
|
|
|
import pytest
|
|
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / 'app'))
|
|
|
|
import scanner
|
|
from scanner_db import canonical_git_scan_plan_bytes, matching_git_coverage
|
|
|
|
|
|
FINISHED = json.dumps({'level': 'info-0', 'msg': 'finished scanning'})
|
|
|
|
|
|
def diagnostic(message='error reading chunk', error='brotli: excessive input', **extra):
|
|
return json.dumps({'level': 'error', 'msg': message, 'error': error, **extra})
|
|
|
|
|
|
def parse(stderr, returncode=0, **kwargs):
|
|
return scanner.apply_trufflehog_diagnostics(
|
|
{'findings': [{'DetectorName': 'SyntheticFinding'}], 'errors': []},
|
|
stderr, returncode, 'git', **kwargs,
|
|
)
|
|
|
|
|
|
@pytest.mark.parametrize('returncode,finished,error_class,retryable', [
|
|
(0, True, None, False),
|
|
(0, False, 'command_incomplete', True),
|
|
(1, True, 'trufflehog', True),
|
|
(1, False, 'trufflehog', True),
|
|
(-1, False, 'trufflehog', True),
|
|
])
|
|
def test_known_git_brotli_requires_zero_exit_and_completion(returncode, finished, error_class, retryable):
|
|
stderr = diagnostic() + ('\n' + FINISHED if finished else '')
|
|
result = parse(stderr, returncode)
|
|
assert result.get('error_class') == error_class
|
|
assert result['retryable'] is retryable
|
|
assert len(result['findings']) == 1
|
|
if error_class is None:
|
|
assert not result['errors']
|
|
assert result['warning_classes'] == ['chunk_read']
|
|
assert result['degraded']
|
|
|
|
|
|
@pytest.mark.parametrize('message,error', [
|
|
('error reading chunk', 'brotli: PADDING_2'),
|
|
('error reading chunk', 'brotli: unknown failure'),
|
|
('error reading chunk', 'brotli: excessive input; another failure'),
|
|
('error processing repository', 'brotli: excessive input'),
|
|
('error reading chunk', 'exit status 128'),
|
|
])
|
|
def test_unproven_or_unrelated_errors_remain_fatal(message, error):
|
|
result = parse(diagnostic(message, error) + '\n' + FINISHED)
|
|
assert result['errors']
|
|
assert result['error_class'] == 'trufflehog'
|
|
assert result['retryable']
|
|
|
|
|
|
@pytest.mark.parametrize('fatal,error_class,retryable', [
|
|
('connection reset', 'network', True),
|
|
('context deadline exceeded', 'timeout', True),
|
|
('unknown flag', 'source_configuration', False),
|
|
('no space left on device', 'source_resource', True),
|
|
('authentication failed', 'source_auth', True),
|
|
('TruffleHog stdout exceeded', 'output_limit', False),
|
|
])
|
|
@pytest.mark.parametrize('warning_first', [False, True])
|
|
def test_git_brotli_warning_never_suppresses_fatal_policy(fatal, error_class, retryable, warning_first):
|
|
lines = [diagnostic(), diagnostic('source failed', fatal)]
|
|
if not warning_first:
|
|
lines.reverse()
|
|
result = parse('\n'.join([*lines, FINISHED]))
|
|
assert result['error_class'] == error_class
|
|
assert result['retryable'] is retryable
|
|
assert result['warning_classes'] == ['chunk_read']
|
|
assert result['degraded']
|
|
assert len(result['findings']) == 1
|
|
|
|
|
|
@pytest.mark.parametrize('causes,error_class,retryable', [
|
|
(['connection reset'], 'network', True),
|
|
(['unknown flag'], 'source_configuration', False),
|
|
(['unknown flag', 'connection reset'], 'source_configuration', False),
|
|
(['connection reset', 'unknown flag'], 'source_configuration', False),
|
|
(['TruffleHog stdout exceeded', 'timeout'], 'output_limit', False),
|
|
(['timeout', 'TruffleHog stdout exceeded'], 'output_limit', False),
|
|
(['out of memory', 'unknown flag'], 'memory_limit', False),
|
|
(['connection reset', 'no space left on device'], 'source_resource', True),
|
|
(['no space left on device', 'connection reset'], 'source_resource', True),
|
|
(['authentication failed', 'connection reset'], 'auth_or_permission', True),
|
|
(['connection reset', 'deadline exceeded'], 'mixed', True),
|
|
(['unclassified failure', 'deadline exceeded'], 'mixed', True),
|
|
(['unable to resolve commit: object not found'], 'trufflehog', True),
|
|
])
|
|
def test_plural_causes_are_classified_with_conservative_precedence(causes, error_class, retryable):
|
|
line = diagnostic('encountered errors during scan', '', errors=causes)
|
|
result = parse(line + '\n' + FINISHED)
|
|
assert result['error_class'] == error_class
|
|
assert result['retryable'] is retryable
|
|
assert result['source_failure'] is (error_class in ('source_configuration', 'source_resource', 'source_auth'))
|
|
assert result['errors'] == [line]
|
|
|
|
|
|
@pytest.mark.parametrize('source', ['git', 'filesystem', 'npm'])
|
|
@pytest.mark.parametrize('cause', ['unauthorized', 'authentication failed', 'bad credentials'])
|
|
def test_ambiguous_plural_auth_does_not_implicate_source_credentials(source, cause):
|
|
result = scanner.apply_trufflehog_diagnostics(
|
|
{'errors': []}, diagnostic('encountered errors during scan', '', errors=[cause]) + '\n' + FINISHED,
|
|
0, source,
|
|
)
|
|
assert result['error_class'] == 'auth_or_permission'
|
|
assert result['source_failure'] is False
|
|
assert not result.get('source_failure_auth_related')
|
|
assert not result.get('source_failure_category')
|
|
assert result['retryable'] is True
|
|
|
|
|
|
@pytest.mark.parametrize('plural', [False, True])
|
|
def test_explicit_singular_source_auth_behavior_is_unchanged(plural):
|
|
extra = {'errors': ['connection reset']} if plural else {}
|
|
result = parse(diagnostic('source failed', 'unauthorized', **extra) + '\n' + FINISHED)
|
|
assert result['error_class'] == 'source_auth'
|
|
assert result['source_failure'] is True
|
|
assert result['source_failure_auth_related'] is True
|
|
|
|
|
|
def test_plural_auth_envelope_without_source_attribution_is_target_scoped():
|
|
result = parse(diagnostic('unauthorized', '', errors=['unclassified failure']) + '\n' + FINISHED)
|
|
assert result['error_class'] == 'auth_or_permission'
|
|
assert result['source_failure'] is False
|
|
assert not result.get('source_failure_auth_related')
|
|
|
|
|
|
@pytest.mark.parametrize('message,cause,warning_class', [
|
|
('non-critical error processing chunk', 'invalid archive', 'chunk_processing'),
|
|
('a detector ignored the context timeout', 'context deadline exceeded', 'detector_timeout'),
|
|
('error reading chunk', 'brotli: excessive input', 'chunk_read'),
|
|
])
|
|
@pytest.mark.parametrize('singular_detail', [False, True])
|
|
def test_plural_warning_context_is_preserved(message, cause, warning_class, singular_detail):
|
|
result = parse(diagnostic(message, cause if singular_detail else '', errors=[cause]) + '\n' + FINISHED)
|
|
assert not result['errors']
|
|
assert result['warning_classes'] == [warning_class]
|
|
assert result['degraded']
|
|
assert result['retryable'] is False
|
|
assert len(result['findings']) == 1
|
|
|
|
|
|
@pytest.mark.parametrize('message,warning_cause', [
|
|
('non-critical error processing chunk', 'invalid archive'),
|
|
('a detector ignored the context timeout', 'context deadline exceeded'),
|
|
])
|
|
@pytest.mark.parametrize('fatal,error_class,retryable', [
|
|
('connection reset', 'network', True),
|
|
('unknown flag', 'source_configuration', False),
|
|
('no space left on device', 'source_resource', True),
|
|
('unauthorized', 'auth_or_permission', True),
|
|
])
|
|
@pytest.mark.parametrize('warning_first', [False, True])
|
|
def test_plural_warning_context_cannot_hide_other_fatal_causes(
|
|
message, warning_cause, fatal, error_class, retryable, warning_first,
|
|
):
|
|
causes = [warning_cause, fatal] if warning_first else [fatal, warning_cause]
|
|
result = parse(diagnostic(message, '', errors=causes) + '\n' + FINISHED)
|
|
assert result['error_class'] == error_class
|
|
assert result['retryable'] is retryable
|
|
assert result['errors']
|
|
if error_class == 'auth_or_permission':
|
|
assert result['source_failure'] is False
|
|
|
|
|
|
def test_plural_brotli_warning_still_requires_completion():
|
|
result = parse(diagnostic('error reading chunk', '', errors=['brotli: excessive input']))
|
|
assert result['error_class'] == 'command_incomplete'
|
|
assert result['retryable'] is True
|
|
|
|
|
|
@pytest.mark.parametrize('message,error', [
|
|
('a detector ignored the context timeout', 'context deadline exceeded'),
|
|
('non-critical error processing chunk', 'invalid archive'),
|
|
('error reading chunk', 'brotli: excessive input'),
|
|
])
|
|
def test_plural_fatal_cause_overrides_warning_envelope(message, error):
|
|
result = parse(diagnostic(message, error, errors=['unknown flag']) + '\n' + FINISHED)
|
|
assert result['error_class'] == 'source_configuration'
|
|
assert result['retryable'] is False
|
|
assert not result.get('warnings')
|
|
|
|
|
|
def test_explicit_envelope_error_is_not_hidden_by_plural_causes():
|
|
result = parse(diagnostic('source failed', 'unknown flag', errors=['connection reset']) + '\n' + FINISHED)
|
|
assert result['error_class'] == 'source_configuration'
|
|
assert result['retryable'] is False
|
|
|
|
|
|
@pytest.mark.parametrize('causes', [
|
|
[{'error': 'unknown flag'}, None, 42],
|
|
[['unknown flag']],
|
|
{'error': 'unknown flag'},
|
|
[' ', ''],
|
|
])
|
|
def test_non_string_plural_causes_are_not_stringified_or_downgraded(causes):
|
|
result = parse(diagnostic('non-critical error processing chunk', '', errors=causes) + '\n' + FINISHED)
|
|
assert result['error_class'] == 'trufflehog'
|
|
assert result['retryable']
|
|
assert not result.get('warnings')
|
|
|
|
|
|
def test_non_string_entries_do_not_inject_cause_classification():
|
|
result = parse(diagnostic('encountered errors during scan', '', errors=[
|
|
{'error': 'unknown flag'}, 'connection reset',
|
|
]) + '\n' + FINISHED)
|
|
assert result['error_class'] == 'network'
|
|
|
|
|
|
@pytest.mark.parametrize('setting,limit,causes', [
|
|
('trufflehog_diagnostic_max_errors', 2, ['timeout'] * 3),
|
|
('trufflehog_diagnostic_max_line_chars', 8, ['x' * 9]),
|
|
('trufflehog_diagnostic_max_line_bytes', 8, ['\u00e9' * 5]),
|
|
])
|
|
def test_plural_classification_bounds_fail_closed(setting, limit, causes):
|
|
with mock.patch.object(scanner.scan_config, setting, limit):
|
|
policy = scanner._trufflehog_diagnostic_policy(
|
|
diagnostic('encountered errors during scan', '', errors=causes), 'git', 0,
|
|
)
|
|
assert policy == ('error', 'output_limit', False)
|
|
|
|
|
|
@pytest.mark.parametrize('detail,error_class,retryable', [
|
|
('connection reset; unknown flag', 'source_configuration', False),
|
|
('timeout; TruffleHog stderr exceeded', 'output_limit', False),
|
|
('connection reset; no space left on device', 'source_resource', True),
|
|
])
|
|
def test_singular_terminal_and_resource_causes_precede_transport(detail, error_class, retryable):
|
|
result = parse(diagnostic('source failed', detail) + '\n' + FINISHED)
|
|
assert result['error_class'] == error_class
|
|
assert result['retryable'] is retryable
|
|
|
|
|
|
def run_exact(stderr, mode='baseline', external=False, stdout='', filter_result=None):
|
|
plan = {
|
|
'version': 1, 'provider': 'gitlab', 'repo_url': 'https://gitlab.com/Fixture/Only.git',
|
|
'repo_path': 'Fixture/Only', 'branch': 'main', 'ref': 'refs/heads/main',
|
|
'head_sha': 'a' * 40, 'base_sha': 'b' * 40 if mode == 'delta' else None,
|
|
'mode': mode, 'baseline_depth': 2, 'ref_source': 'explicit',
|
|
}
|
|
encoded = canonical_git_scan_plan_bytes(plan)
|
|
digest = hashlib.sha256(encoded).hexdigest()
|
|
output = scanner.streamed_output_from_text(stdout, stderr, 0)
|
|
with mock.patch.object(scanner, 'get_trufflehog_cmd', return_value='fixture-only'), \
|
|
mock.patch.object(scanner, 'append_trufflehog_scan_args'), \
|
|
mock.patch.object(scanner, 'run_command_streamed', return_value=output) as command, \
|
|
mock.patch.object(scanner, 'apply_finding_filters', side_effect=filter_result or (lambda result, target: result)):
|
|
result = scanner.scan_exact_git_plan(
|
|
plan['repo_url'], plan, digest, 10, None, None, True, None, '', external,
|
|
)
|
|
reservation = {'git_scan_plan_json': encoded.decode('ascii'), 'git_scan_plan_sha256': digest}
|
|
return result, reservation, command
|
|
|
|
|
|
@pytest.mark.parametrize('mode', ['baseline', 'delta'])
|
|
@pytest.mark.parametrize('external', [False, True])
|
|
def test_exact_git_always_requires_completion(mode, external):
|
|
result, reservation, command = run_exact('', mode=mode, external=external)
|
|
assert result['error_class'] == 'command_incomplete'
|
|
assert not result['git_scan_execution']['success']
|
|
assert result['git_scan_execution']['coverage_complete'] is False
|
|
assert command.call_count == 1
|
|
assert matching_git_coverage(reservation, result, 'done', len(result['errors']))[0] is False
|
|
|
|
|
|
@pytest.mark.parametrize('warning', [
|
|
diagnostic(),
|
|
diagnostic('non-critical error processing chunk', 'invalid archive'),
|
|
diagnostic('a detector ignored the context timeout', 'context deadline exceeded'),
|
|
diagnostic('error cleaning temporary artifacts', 'cleanup failure'),
|
|
])
|
|
@pytest.mark.parametrize('with_findings', [False, True])
|
|
def test_degraded_git_execution_cannot_advance_coverage(warning, with_findings):
|
|
stdout = json.dumps({'DetectorName': 'SyntheticFinding'}) if with_findings else ''
|
|
result, reservation, command = run_exact(warning + '\n' + FINISHED, stdout=stdout)
|
|
assert not result['errors']
|
|
assert result['retryable'] is False
|
|
assert result['degraded']
|
|
assert bool(result['findings']) is with_findings
|
|
assert not result['git_scan_execution']['success']
|
|
assert result['git_scan_execution']['coverage_complete'] is False
|
|
assert matching_git_coverage(reservation, result, 'done', 0)[0] is False
|
|
assert command.call_count == 1
|
|
|
|
|
|
def test_complete_clean_exact_git_keeps_pinned_contract_and_coverage():
|
|
result, reservation, command = run_exact(FINISHED)
|
|
assert not result['errors']
|
|
assert result['git_scan_execution']['success']
|
|
assert result['git_scan_execution']['coverage_complete'] is True
|
|
assert matching_git_coverage(reservation, result, 'done', 0)[0] is True
|
|
argv = command.call_args.args[0]
|
|
assert argv[argv.index('--branch') + 1] == 'a' * 40
|
|
|
|
|
|
@pytest.mark.parametrize('metadata', [
|
|
{'degraded': True},
|
|
{'warnings': ['synthetic coverage warning']},
|
|
{'degraded': True, 'findings': [{'DetectorName': 'SyntheticFinding'}]},
|
|
])
|
|
def test_persisted_success_cannot_override_incomplete_git_coverage(metadata):
|
|
result, reservation, _ = run_exact(FINISHED)
|
|
assert result['git_scan_execution']['success']
|
|
del result['git_scan_execution']['coverage_complete']
|
|
result.update(metadata)
|
|
assert matching_git_coverage(reservation, result, 'done', 0)[0] is False
|
|
|
|
|
|
@pytest.mark.parametrize('filter_warning', [False, True])
|
|
def test_explicit_complete_coverage_survives_optional_filter_and_staging_warnings(filter_warning):
|
|
def apply_optional_filter(result, _target):
|
|
assert result['git_scan_execution']['coverage_complete'] is True
|
|
if filter_warning:
|
|
result['degraded'] = True
|
|
result['warnings'] = ['optional filtering unavailable']
|
|
return result
|
|
|
|
result, reservation, _ = run_exact(FINISHED, filter_result=apply_optional_filter)
|
|
result['degraded'] = True
|
|
result.setdefault('warnings', []).append('Optional candidate staging unavailable')
|
|
persisted = json.loads(json.dumps(result))
|
|
assert persisted['git_scan_execution']['success'] is True
|
|
assert persisted['git_scan_execution']['coverage_complete'] is True
|
|
assert matching_git_coverage(reservation, persisted, 'done', 0)[0] is True
|
|
|
|
|
|
@pytest.mark.parametrize('marker', [False, None, 'true', 1])
|
|
def test_explicit_incomplete_or_invalid_coverage_marker_rejects_success(marker):
|
|
result, reservation, _ = run_exact(FINISHED)
|
|
result['git_scan_execution']['coverage_complete'] = marker
|
|
assert matching_git_coverage(reservation, result, 'done', 0)[0] is False
|
|
|
|
|
|
def test_explicit_complete_marker_cannot_override_failed_execution():
|
|
result, reservation, _ = run_exact(FINISHED)
|
|
result['git_scan_execution']['success'] = False
|
|
assert result['git_scan_execution']['coverage_complete'] is True
|
|
assert matching_git_coverage(reservation, result, 'done', 0)[0] is False
|
|
|
|
|
|
def test_legacy_clean_success_without_coverage_marker_is_still_accepted():
|
|
result, reservation, _ = run_exact(FINISHED)
|
|
del result['git_scan_execution']['coverage_complete']
|
|
assert matching_git_coverage(reservation, result, 'done', 0)[0] is True
|
|
|
|
|
|
def test_existing_docker_gzip_warning_is_unchanged():
|
|
result = scanner.apply_trufflehog_diagnostics(
|
|
{'errors': []}, diagnostic('error processing layer', 'gzip: invalid header') + '\n' + FINISHED,
|
|
0, 'docker',
|
|
)
|
|
assert not result['errors']
|
|
assert result['warning_classes'] == ['docker_layer_gzip']
|
|
assert result['degraded']
|
|
assert result['retryable'] is False
|
|
|
|
|
|
def test_existing_diagnostic_smoke_policy_without_database_tests():
|
|
from scanner_error_policy_smoke import assert_diagnostic_policy
|
|
|
|
assert_diagnostic_policy()
|