Initial server source import

This commit is contained in:
sashatrask
2026-09-30 20:30:56 +03:00
commit 170dd941b9
498 changed files with 261563 additions and 0 deletions
+455
View File
@@ -0,0 +1,455 @@
import sys
sys.dont_write_bytecode = True
if not sys.dont_write_bytecode:
raise RuntimeError('janitor could not disable bytecode writes')
import argparse
import hashlib
import json
import os
import stat
import time
from dataclasses import dataclass
from datetime import datetime, timezone
from lifecycle_authority import require_active_supervisor_child
from paths import apply_path_config
from process_identity import exact_process_identity_state
from runtime_security import (
atomic_write_private_json,
canonical_path,
fsync_directory,
is_reparse_point,
private_directory_ready,
private_file_ready,
read_private_json,
reject_reparse_components,
require_private_directory,
)
MARKER_NAME = '.scanner-owner.json'
MARKER_SCHEMA = 2
APPROVED_LAYOUTS = (
('work', '', ('trufflehog-', 'trufflehog-run-', 'trufflehog-probe-', 'hg-run-', 'docker-config-', 'docker-layer-', 'tmp-', 'worker-assignment-')),
('work', 'docker-config', ('docker-config-',)),
('work', 'hg', ('hg-run-',)),
('work', 'tmp', ('trufflehog-', 'trufflehog-run-', 'hg-run-', 'docker-config-', 'docker-layer-', 'tmp-')),
('work', os.path.join('tmp', 'docker-config'), ('docker-config-',)),
('work', 'abandoned', ('worker-assignment-',)),
)
@dataclass
class JanitorBudget:
max_candidates: int = 50
max_entries: int = 10000
max_bytes: int = 1024 * 1024 * 1024
max_seconds: float = 30.0
max_depth: int = 64
max_enumerated: int = 1000
candidates: int = 0
entries: int = 0
bytes: int = 0
started_at: float = 0.0
exhausted: bool = False
enumerated: int = 0
def __post_init__(self):
self.started_at = self.started_at or time.monotonic()
def consume(self, size=0, candidate=False, depth=0, allow_oversized=False):
if candidate:
self.candidates += 1
else:
self.entries += 1
self.bytes += max(0, int(size or 0))
candidate_limit = self.candidates > self.max_candidates
entry_limit = self.entries > self.max_entries
byte_limit = self.bytes > self.max_bytes
depth_limit = depth > self.max_depth
time_limit = time.monotonic() - self.started_at >= self.max_seconds
self.exhausted = bool(
candidate_limit or entry_limit or byte_limit or depth_limit or time_limit
)
# Unlinking one regular file is bounded metadata work regardless of its
# payload size. Directory traversal remains bounded by the other limits.
oversized_progress = bool(
allow_oversized and byte_limit
and not (candidate_limit or entry_limit or depth_limit or time_limit)
)
return not self.exhausted or oversized_progress
def consume_enumerated(self):
self.enumerated += 1
self.exhausted = bool(
self.enumerated > self.max_enumerated
or time.monotonic() - self.started_at >= self.max_seconds
)
return not self.exhausted
def _marker_relative_path(root, path):
relative = os.path.relpath(path, root)
if relative == '.' or relative.startswith('..' + os.sep) or os.path.isabs(relative):
raise ValueError('candidate escapes the approved janitor root')
return relative.replace(os.sep, '/')
def validate_marker(root, path, marker, allowed_executables, minimum_age_sec, now=None):
if marker.get('schema') != MARKER_SCHEMA or marker.get('root_kind') != 'work':
return False, 'unsupported_marker'
try:
if marker.get('relative_path') != _marker_relative_path(root, path):
return False, 'path_mismatch'
created = datetime.fromisoformat(str(marker.get('created_at') or '').replace('Z', '+00:00'))
if created.tzinfo is None:
created = created.replace(tzinfo=timezone.utc)
now_value = now or datetime.now(timezone.utc)
if (now_value - created).total_seconds() < max(0, float(minimum_age_sec)):
return False, 'too_young'
except (TypeError, ValueError):
return False, 'invalid_time_or_path'
allowed = {canonical_path(value) for value in allowed_executables if value}
identities = {}
states = {}
prefixes = ('owner', 'parent')
if any(f'child_{field}' in marker for field in ('pid', 'creation_time', 'executable')):
prefixes += ('child',)
for prefix in prefixes:
if prefix == 'child' and any(not marker.get(f'child_{field}') for field in ('pid', 'creation_time', 'executable')):
return False, 'child_identity_invalid'
identity = {
'pid': marker.get(f'{prefix}_pid'),
'creation_time': marker.get(f'{prefix}_creation_time'),
'executable': marker.get(f'{prefix}_executable'),
}
try:
executable = canonical_path(identity['executable'])
except (OSError, TypeError, ValueError):
return False, f'{prefix}_identity_invalid'
if executable not in allowed:
return False, f'{prefix}_executable_unapproved'
identity['executable'] = executable
identities[prefix] = identity
states[prefix] = exact_process_identity_state(
identity['pid'], identity['creation_time'], identity['executable'],
)
if 'child' in states and states['child'] != 'dead':
return False, 'child_live_or_unknown'
if states['owner'] != 'dead':
return False, 'owner_live_or_unknown'
same_identity = all(
identities['owner'].get(field) == identities['parent'].get(field)
for field in ('pid', 'creation_time', 'executable')
)
if same_identity and states['parent'] != 'dead':
return False, 'parent_owned_live_or_unknown'
return True, 'eligible'
def bounded_remove_tree(path, budget, marker_name=MARKER_NAME):
"""Delete without recursion or reparse traversal; leave the root marker last."""
path = os.path.abspath(path)
reject_reparse_components(path)
if is_reparse_point(path) or not private_directory_ready(path):
raise OSError(f'janitor candidate is not an exact private directory: {path}')
marker_path = os.path.join(path, marker_name)
stack = []
root_iterator = os.scandir(path)
stack.append((path, root_iterator, 0))
try:
while stack:
if time.monotonic() - budget.started_at >= budget.max_seconds:
budget.exhausted = True
return False
directory, iterator, depth = stack[-1]
try:
entry = next(iterator)
except StopIteration:
iterator.close()
stack.pop()
if directory == path:
if os.path.lexists(marker_path):
details = os.stat(marker_path, follow_symlinks=False)
# The verified owner marker is removed only after every payload
# entry is gone, so finishing the empty root must make progress
# even when one oversized payload exhausted this pass's budget.
budget.consume(details.st_size, depth=depth + 1)
if not private_file_ready(marker_path):
raise OSError('janitor owner marker lost its private identity')
os.remove(marker_path)
os.rmdir(directory)
fsync_directory(os.path.dirname(directory))
return True
os.rmdir(directory)
continue
if entry.path == marker_path:
continue
details = entry.stat(follow_symlinks=False)
is_regular = stat.S_ISREG(details.st_mode)
if not budget.consume(
details.st_size, depth=depth + 1, allow_oversized=is_regular,
):
return False
if entry.is_symlink() or is_reparse_point(entry.path):
raise OSError(f'janitor candidate contains a link or reparse point: {entry.path}')
if stat.S_ISDIR(details.st_mode):
child_iterator = os.scandir(entry.path)
stack.append((entry.path, child_iterator, depth + 1))
elif is_regular:
os.chmod(entry.path, stat.S_IWRITE | stat.S_IREAD)
os.remove(entry.path)
else:
raise OSError(f'janitor candidate contains an unsupported entry: {entry.path}')
finally:
for _, iterator, _ in stack:
iterator.close()
return False
def _layout_cursor_name(root, relative_parent, prefixes):
identity = '|'.join((
canonical_path(root), str(relative_parent).replace(os.sep, '/'), ','.join(prefixes),
))
return 'layout:' + hashlib.sha256(identity.encode('utf-8')).hexdigest()
class JanitorCursorStore:
SCHEMA = 1
def __init__(self, path, root):
self.path = os.path.abspath(path)
self.root_hash = hashlib.sha256(canonical_path(root).encode('utf-8')).hexdigest()
self.dirty = False
self.state = {
'schema': self.SCHEMA,
'root_sha256': self.root_hash,
'next_layout': 0,
'layouts': {},
}
self.iterators = {}
self.seeking = {}
if os.path.lexists(self.path):
loaded = read_private_json(self.path, max_bytes=256 * 1024)
if (
not isinstance(loaded, dict)
or loaded.get('schema') != self.SCHEMA
or loaded.get('root_sha256') != self.root_hash
or not isinstance(loaded.get('layouts'), dict)
):
raise RuntimeError('janitor cursor authority is invalid')
self.state = loaded
else:
require_private_directory(os.path.dirname(self.path), create=True)
atomic_write_private_json(self.path, self.state)
def _save(self):
atomic_write_private_json(self.path, self.state)
self.dirty = False
def _mark_dirty(self):
self.dirty = True
def flush(self):
if self.dirty:
self._save()
def next_layout(self, count):
index = int(self.state.get('next_layout') or 0) % max(1, int(count))
self.state['next_layout'] = (index + 1) % max(1, int(count))
self._mark_dirty()
return index
def next_entry(self, layout_name, parent):
iterator = self.iterators.get(layout_name)
if iterator is None:
iterator = os.scandir(parent)
self.iterators[layout_name] = iterator
last_name = str((self.state['layouts'].get(layout_name) or {}).get('last_name') or '')
self.seeking[layout_name] = bool(last_name)
try:
entry = next(iterator)
except StopIteration:
iterator.close()
self.iterators.pop(layout_name, None)
self.seeking.pop(layout_name, None)
current = self.state['layouts'].setdefault(layout_name, {})
current['last_name'] = ''
current['wrap_count'] = int(current.get('wrap_count') or 0) + 1
self._mark_dirty()
return None, False
current = self.state['layouts'].setdefault(layout_name, {'last_name': '', 'wrap_count': 0})
target = str(current.get('last_name') or '')
if self.seeking.get(layout_name):
if entry.name == target:
self.seeking[layout_name] = False
return entry, True
current['last_name'] = entry.name
self._mark_dirty()
return entry, False
def close(self):
for iterator in self.iterators.values():
iterator.close()
self.iterators.clear()
class _MemoryCursorStore(JanitorCursorStore):
def __init__(self):
self.path = ''
self.root_hash = ''
self.dirty = False
self.state = {'schema': 1, 'root_sha256': '', 'next_layout': 0, 'layouts': {}}
self.iterators = {}
self.seeking = {}
def _save(self):
self.dirty = False
return None
def iter_candidates(root, budget, cursor_store):
layouts = list(APPROVED_LAYOUTS)
completed_layouts = set()
while not budget.exhausted and len(completed_layouts) < len(layouts):
if (
budget.enumerated >= budget.max_enumerated
or time.monotonic() - budget.started_at >= budget.max_seconds
):
budget.exhausted = True
return
index = cursor_store.next_layout(len(layouts))
root_kind, relative_parent, prefixes = layouts[index]
layout_name = _layout_cursor_name(root, relative_parent, prefixes)
if layout_name in completed_layouts:
continue
parent = os.path.join(root, relative_parent) if relative_parent else root
try:
if not os.path.isdir(parent) or is_reparse_point(parent):
completed_layouts.add(layout_name)
continue
entry, seeking = cursor_store.next_entry(layout_name, parent)
except OSError:
completed_layouts.add(layout_name)
continue
if entry is None:
completed_layouts.add(layout_name)
continue
if not budget.consume_enumerated():
return
if seeking:
continue
if not entry.name.startswith(prefixes) or not entry.is_dir(follow_symlinks=False):
continue
if not budget.consume(candidate=True):
return
yield root_kind, layout_name, entry.name, entry.path
def run_janitor_pass(
root, allowed_executables, minimum_age_sec=7200, budget=None,
cursor_store=None, excluded_relative_paths=(),
):
budget = budget or JanitorBudget()
root = require_private_directory(root, create=False)
cursor_store = cursor_store or _MemoryCursorStore()
excluded = set()
for value in excluded_relative_paths:
relative = str(value or '').replace('\\', '/')
if (
not relative or relative.startswith('/') or relative.endswith('/')
or any(part in ('', '.', '..') for part in relative.split('/'))
):
raise ValueError('janitor exclusion path is invalid')
excluded.add(relative)
if len(excluded) > 4096:
raise ValueError('janitor exclusion set exceeds its bound')
report = {'considered': 0, 'removed': 0, 'retained': 0, 'errors': 0, 'exhausted': False}
try:
for _, _, _, path in iter_candidates(root, budget, cursor_store):
report['considered'] += 1
if _marker_relative_path(root, path) in excluded:
report['retained'] += 1
continue
marker_path = os.path.join(path, MARKER_NAME)
try:
if not private_file_ready(marker_path):
report['retained'] += 1
continue
marker = read_private_json(marker_path)
eligible, _ = validate_marker(
root, path, marker, allowed_executables, minimum_age_sec,
)
if not eligible:
report['retained'] += 1
continue
if bounded_remove_tree(path, budget):
report['removed'] += 1
else:
report['retained'] += 1
except (OSError, ValueError):
report['errors'] += 1
if budget.exhausted:
break
finally:
cursor_store.flush()
report['exhausted'] = budget.exhausted
report['entries'] = budget.entries
report['bytes'] = budget.bytes
report['enumerated'] = budget.enumerated
return report
def parse_args():
parser = argparse.ArgumentParser(description='Bounded scanner work-directory janitor')
parser.add_argument('--config', required=True)
return parser.parse_args()
def main():
metadata = require_active_supervisor_child(child_kind='janitor', require_dsn=False)
args = parse_args()
import yaml
with open(args.config, 'r', encoding='utf-8') as handle:
config = apply_path_config(yaml.safe_load(handle) or {}, args.config)
global_config = config.get('global') or {}
janitor_config = ((config.get('supervisor') or {}).get('janitor') or {})
manifest = metadata.get('code_manifest') or {}
allowed = [sys.executable]
allowed.extend(
item.get('path') for item in (manifest.get('executables') or {}).values()
if isinstance(item, dict) and item.get('path')
)
interval = max(5.0, float(janitor_config.get('interval_sec', 60) or 60))
cursor_store = JanitorCursorStore(
os.path.join(global_config['state_dir'], 'janitor.cursor.json'),
global_config['work_dir'],
)
try:
while True:
budget = JanitorBudget(
max_candidates=max(1, int(janitor_config.get('max_candidates', 50) or 50)),
max_entries=max(1, int(janitor_config.get('max_entries', 10000) or 10000)),
max_bytes=max(1, int(janitor_config.get('max_bytes', 1024 * 1024 * 1024) or 1)),
max_seconds=max(0.1, float(janitor_config.get('max_seconds', 30) or 30)),
max_depth=max(1, int(janitor_config.get('max_depth', 64) or 64)),
max_enumerated=max(1, int(janitor_config.get('max_enumerated', 1000) or 1000)),
)
report = run_janitor_pass(
global_config['work_dir'], allowed,
minimum_age_sec=max(0, int(janitor_config.get('minimum_age_sec', 7200) or 0)),
budget=budget, cursor_store=cursor_store,
)
print(json.dumps(report, ensure_ascii=True, sort_keys=True), flush=True)
time.sleep(interval)
finally:
cursor_store.close()
if __name__ == '__main__':
main()