Files
truf-server/app/config.linux.yaml
T
2026-09-30 20:30:56 +03:00

1160 lines
63 KiB
YAML

# Linux container profile; host runtime entrypoints remain deliberately disabled.
# See DOCKER_MIGRATION.md. The Windows configuration is preserved in the initial Git commit.
global:
loop: true # true = run forever; false = run one full pass over enabled sources
cooldown: 30 # seconds to sleep after one full pass over all enabled sources
backlog_poll_sec: 0.5 # immediately refill released scan slots while durable queue work remains
root_dir: "/opt/truf" # application image root, never the original Windows checkout
project_dir: "{root_dir}/app"
runtime_dir: "/data/runtime-linux" # new Linux state; do not reuse Windows control metadata
postgres_data_dir: "/data/postgres-linux" # independently initialized, never a Windows cluster copy
postgres_bin_dir: "/usr/lib/postgresql/16/bin"
result_bundle_dir: "/data/scanner-result-bundles"
result_bundle_max_event_bytes: 67108864 # hard limit per bundle; separate from remote reservation
remote_assignment_reserve_bytes: 2097152 # bundle and projection baseline per unresolved remote assignment
remote_assignment_max_active: 50 # global unresolved remote assignments across all users
result_bundle_max_items: 10000
result_bundle_max_total_bytes: 3221225472
result_bundle_min_free_bytes: 21474836480
projection_backlog_max_items: 10000
projection_backlog_max_bytes: 2147483648
projection_backlog_headroom_bytes: 402653184 # one worst-case aggregate scan projection beyond all 3 physical slots
keycheck_queue_max_items: 131072
keycheck_queue_max_bytes: 134217728
pipeline_quarantine_max_items: 10000
pipeline_quarantine_max_bytes: 1073741824
pipeline_metadata_retention_days: 30
pipeline_metadata_retirement_batch: 100
keycheck_candidates_per_event: 2000
keycheck_candidate_bytes_per_event: 2097152
keycheck_result_projection_reserve_bytes: 3145728
keycheck_recheck_batch_items: 10000
legacy_result_spool_dir: "{runtime_dir}/result_spool"
legacy_result_spool_max_event_bytes: 201326592
legacy_result_spool_max_events: 10000
legacy_result_spool_max_total_bytes: 3221225472
results_dir: "{runtime_dir}/results" # findings, scan result JSONL/logs, and scanner.db
queue_dir: "{runtime_dir}/queues" # todo_*.txt and checked_*.txt live here
keycheck_dir: "{runtime_dir}/keychecks" # keychecker outputs grouped by service
postman_cache_dir: "{runtime_dir}/postman_cache" # durable cached Postman collection/environment JSON
postman_cache_max_items: 100000 # aggregate content-addressed artifacts; capacity failure stops discovery
postman_cache_max_bytes: 21474836480 # artifact, metadata, and temporary bytes under the cache root
postman_cache_min_free_bytes: 21474836480 # preserve the shared 20 GiB work-volume reserve
postman_cache_lock_timeout_sec: 300 # match the bounded discovery window under concurrent cache publishers
postman_discovery_max_artifacts_per_cycle: 1000 # shared non-package download/cache attempt cap
postman_discovery_max_artifacts_per_page: 100 # bound one API page or GHArchive hour batch
postman_discovery_max_bytes_per_cycle: 1073741824 # aggregate downloaded artifact bytes
postman_discovery_max_elapsed_sec: 300 # includes network, cache scan, lock, and publication work
postman_package_harvest_max_artifacts: 100 # matching package artifacts examined/published per target
postman_package_harvest_max_bytes: 134217728 # aggregate matching artifact bytes considered per target
postman_package_harvest_max_elapsed_sec: 30 # optional package harvesting wall-clock deadline
postman_context_max_input_bytes: 16777216
postman_context_max_nodes: 100000
postman_context_max_depth: 64
postman_context_max_scalar_bytes: 16777216
postman_context_max_items: 50000
context_enrichment_max_source_bytes: 16777216 # aggregate optional-context source reads per target
context_enrichment_max_findings: 2000 # stop optional enrichment without dropping later findings
context_enrichment_max_postman_comparisons: 200000 # hard cap on fallback substring comparisons
context_enrichment_max_elapsed_sec: 5 # aggregate optional-context wall-clock budget per target
trufflehog_diagnostic_max_lines: 2000 # stop parsing stderr after bounded diagnostic work
trufflehog_diagnostic_max_line_chars: 8192
trufflehog_diagnostic_max_line_bytes: 8192
trufflehog_diagnostic_max_errors: 200
trufflehog_diagnostic_max_warnings: 200
trufflehog_diagnostic_max_unclassified: 20
keycheck_input_max_line_bytes: 16777216 # canonical found_secrets producer/consumer JSONL line limit
keycheck_candidate_artifact_max_items: 2000 # cap candidates derived from one scanned artifact
keycheck_candidate_artifact_max_bytes: 2097152
keycheck_candidate_file_max_items: 100000 # aggregate bounded loader/writer limits
keycheck_candidate_file_max_bytes: 33554432
keycheck_candidate_line_max_bytes: 8192
gharchive_cache_dir: "{state_dir}/gharchive_cache" # shared validated hourly .json.gz cache for both GHArchive sources
gharchive_cache_max_items: 48 # aggregate retained hourly archives
gharchive_cache_max_bytes: 8589934592 # compressed artifacts, lock metadata, and temporary bytes
gharchive_cache_min_free_bytes: 5368709120 # preserve 5 GiB free on the runtime volume
gharchive_download_max_bytes: 536870912 # per-hour compressed response cap
gharchive_decompressed_max_bytes: 8589934592 # per-hour gzip expansion cap
gharchive_max_events: 5000000 # per-hour event/line count cap
gharchive_max_line_bytes: 8388608 # reject oversized individual JSON event lines
gharchive_cache_lock_timeout_sec: 600 # bounded aggregate/per-hour cross-process lock wait
state_dir: "{runtime_dir}/state" # runner state files
log_dir: "{runtime_dir}/logs" # supervisor/source/dashboard logs
control_dir: "/run/truf/control" # ephemeral container identity, never persisted across recreation
proxy_file: "{runtime_dir}/proxy.txt" # shared proxy list for checkers/scanners
api_proxy_enabled: true # discovery/metadata use proxy_file; artifact bodies explicitly bypass proxies
api_proxy_file: "{proxy_file}" # host:port:user:pass or full proxy URL; resolved from proxy_file by default
api_proxy_timeout: 5 # proxy connect cap; caller read timeout remains unchanged (fallback: 5 s)
api_proxy_max_retries: 100 # default total attempts; source-specific attempts/deadlines take precedence
api_proxy_retry_delay: 5 # seconds between API proxy retries
download_proxy_enabled: false # reserved: keep heavy downloads/scans direct for now
download_proxy_file: "" # reserved proxy list for package/artifact downloads when enabled later
max_active_scans: 1 # conservative default within the container's shared memory budget
opportunistic_scan_slots: 0 # host-memory-based opportunistic admission is disabled in containers
opportunistic_scan_sources: [github, gitlab, huggingface]
opportunistic_scan_reserve_overhead_bytes: 1073741824 # reserve Job cap plus 1 GiB process/staging overhead
opportunistic_scan_min_available_after_reserve_bytes: 4294967296 # preserve 4 GiB physical RAM after admission
opportunistic_scan_min_commit_after_reserve_bytes: 6442450944 # preserve 6 GiB commit headroom after admission
scan_limiter_db: "{state_dir}/scan_limiter.db" # separate SQLite DB for cross-process scan slot leasing
scan_slot_wait_sec: 0.5 # sleep between slot-acquire attempts when all scan slots are busy
scan_slot_wait_log_sec: 30 # log long waits at this interval
scan_slot_stale_sec: 7200 # clean slots older than this or owned by dead scanner PIDs
target_retry_max_attempts: 3 # total target attempts before a terminal failed queue state
target_retry_base_delay_sec: 3600 # transient target retry delay; doubles after each failed attempt
target_retry_max_delay_sec: 86400 # cap exponential target retry delay at 24 hours
target_timeout_retry_delay_sec: 21600 # timed-out targets use a non-terminal slow retry after 6 hours
target_claim_batch_size: 1 # fallback only; PostgreSQL slot-first dispatch claims exactly acquired capacity
admission_resolution_attempts: 300 # exact-token probes after an ambiguous PostgreSQL admission response
admission_resolution_seconds: 300 # cover 45s loss grace, 60s stable-ready gate, and reconnect margin
admission_resolution_retry_delay_sec: 1 # bounded pause between fast failed recovery probes
sync_file_queues: false # legacy todo/checked import is complete; PostgreSQL is authoritative
dockerhub_tag_cache_path: "{state_dir}/dockerhub_tag_cache.sqlite" # cache Docker Hub tag resolutions/rate limits
dockerhub_tag_cache_ttl_sec: 21600 # successful tag resolutions are reused for 6 hours
dockerhub_tag_negative_cache_ttl_sec: 3600 # empty/not-found tag lookups are retried after 1 hour
dockerhub_tag_rate_limit_cache_ttl_sec: 1800 # global Docker tag API cooldown fallback
dockerhub_tag_cache_max_rows: 50000
dockerhub_tag_cache_max_age_sec: 604800
dockerhub_tag_cache_max_bytes: 268435456
dockerhub_tag_cache_min_free_bytes: 536870912
database_path: "{results_dir}/scanner_active.db" # SQLite fallback observability DB when SCANNER_DB_URL is empty
database_url: "" # Postgres DSN comes from SCANNER_DB_URL; keep real credentials out of config
dashboard_db_path: "{database_path}" # SQLite dashboard fallback
dashboard_db_url: "" # optional dashboard Postgres DSN override; defaults to SCANNER_DB_URL when empty
dashboard_immutable_db: false # active DB is read-only via SQLite mode=ro, but not immutable because WAL changes
jsonl_rotation_enabled: true # rotate large runtime JSONL files instead of growing multi-GB active files
found_secrets_max_mb: 128 # rotate found_secrets.jsonl after this active-file size
scan_results_max_mb: 256 # rotate scan_results.jsonl after this active-file size
scan_errors_max_mb: 32 # bound each scan_errors.log segment
scan_errors_keep: 5 # retain at most this many rotated scan error segments
jsonl_lock_stale_sec: 300 # stale lock cleanup for cross-process JSONL rotation
jsonl_max_segments: 16 # hard cap; publication pauses until registered consumers catch up
jsonl_ledger_max_rows: 1000000 # durable O(1) publication identity bound
jsonl_ledger_max_bytes: 536870912
jsonl_legacy_index_max_bytes: 16777216 # larger existing files require offline ledger reconciliation
jsonl_tail_scan_max_bytes: 8388608
jsonl_torn_quarantine_max_bytes: 65536
detectors: "" # empty = use all TruffleHog detectors; set IDs to limit intentionally
exclude_detectors: "github.v1,gitlab.v1,GitHubOauth2" # drop noisy legacy GitHub/GitLab detectors; keep modern prefixes
no_verification: true # pass --no-verification to TruffleHog; local checkers classify live/dead later
strict_git_provider_token_filter: true # drop unverified GitHub/GitLab detections that do not match known token prefixes
drop_detectors: "Privacy,URI,JDBC,Postgres,MongoDB,SQLServer,Box,ZohoCRM,Accuweather,Roaring,Flatio,LinkPreview,RailwayApp" # do not persist obvious non-keycheckable/generic noise detectors
versions_per_package: 3 # npm/PyPI: scan up to N recent versions per matching package
work_dir: "/data/scanner-work" # isolated scratch root for clone/download/extract folders
trufflehog_stdout_max_mb: 32 # hard file-backed streamed stdout byte bound per scan
trufflehog_stderr_max_mb: 8 # hard file-backed streamed diagnostic byte bound per scan
trufflehog_config: "{project_dir}/trufflehog-custom-detectors.yaml" # custom detectors loaded by TruffleHog --config
trufflehog_job_memory_limit_bytes: 4294967296 # aggregate Windows Job limit for each TruffleHog tree (4096 MiB)
trufflehog_windows_job_cpu_weight: 2 # normal priority with low relative Job weight; avoids broken BELOW_NORMAL startup
trufflehog_windows_memory_priority: 4 # reclaim scan pages before normal-priority desktop working sets
min_free_gb: 20 # minimum free space required in work_dir before starting new scans
state_file: "{state_dir}/runner_state.json" # stores current query index for each source
secrets_file: "/data/config/secrets.yaml" # runtime-owned private credentials, never baked into the image
stop_on_seen_pages: false # default false; enable per source where API pagination is sequential
seen_page_threshold: 2 # stop after this many consecutive all-known pages
min_pages_before_stop: 1 # always fetch at least this many pages before early stop
supervisor:
enabled_sources: [gitlab, dockerhub, huggingface] # exact distributed discovery-producer profile
interactive: false # canonical foreground container supervisor, no terminal dependency
autostart: true # start only the explicit core allowlist above
poll_sec: 1.0 # how often supervisor checks child process/log state
heartbeat_sec: 60 # rewrite background status snapshot at least this often; 0 disables heartbeat
authority_check_interval_sec: 5 # detect code/config authority drift within this bound
pipeline_status_refresh_sec: 2 # cache authenticated pipeline status queries between loop ticks
postgres_health_interval_sec: 15 # bounded authenticated controller probe interval
postgres_ready_loss_grace_sec: 45 # tolerate transient loss for the same live authenticated postmaster
postgres_stable_ready_sec: 60 # dependency gate opens only after readiness remains stable this long
postgres_connect_timeout_sec: 5 # bounded controller authentication connect timeout
postgres_query_timeout_ms: 5000 # bounded controller identity/readiness query timeout
postgres_stop_timeout_sec: 60 # pg_ctl's bounded identity-verified coordinated stop timeout
postgres_start_settle_timeout_sec: 30 # bound late postmaster publication checks after pg_ctl start -W
postgres_shutdown_timeout_sec: 120 # total supervisor wait for the controller during coordinated shutdown
postgres_log_max_mb: 64 # collector/startup log segment bound
postgres_log_keep: 24 # maximum retained collector and startup log segments
refresh_sec: 5 # non-interactive stdout/status-loop interval
log_dir: "{log_dir}" # one appended child-process log per source
log_max_mb: 64 # live source output rotates at this active-segment bound
log_keep: 5 # keep this many rotated log segments per log file
control_dir: "{control_dir}" # hardened directory; links/junctions are rejected
instance_file: "{control_dir}/supervisor.instance.json" # private authenticated process/control identity
lock_file: "{control_dir}/supervisor.lock" # secondary per-instance lock; cluster authority is data-dir-derived and non-configurable
supervisor_log: "{log_dir}/supervisor.log" # stdout/stderr for background supervisor
status_file: "{log_dir}/supervisor.status.txt" # latest background supervisor status table
control_host: "127.0.0.1" # local only; do not expose externally
control_port: 8765
background_start_timeout_sec: 20 # wait for matching private metadata and authenticated handshake
background_shutdown_timeout_sec: 180 # wait on the retained verified supervisor process handle
attach_poll_sec: 0.2 # --attach watch-mode poll interval
dashboard_log: "{log_dir}/dashboard.log" # stdout/stderr for supervisor-launched dashboard
state_dir: "{state_dir}" # per-source runner_state_*.json files to avoid parallel write races
per_source_state: true # true = supervisor sets RUNNER_STATE_FILE per child process
result_ingester:
enabled: true
poll_sec: 0.2
lease_seconds: 300
jsonl_projector:
enabled: true
poll_sec: 0.2
lease_seconds: 300
worker_api:
enabled: false # fail closed; configure profiles and private ingress before enabling
address: "127.0.0.1" # raw API must remain loopback/private and unexposed
port: 8766
sources: [] # optional narrowing of exact protocol-2 package capability triples
auth_entries: {} # GitLab issuance plus optional legacy GitHub reconciliation auth entries
compatibility_profiles: {} # trusted package manifests; never inferred from clients
assignment_ttl_seconds: 86400 # immutable server-time deadline, default 24 hours
assignment_ttl_seconds_by_source: {}
max_bundle_bytes: 67108864 # hard-capped again by global result_bundle_max_event_bytes
reaper_interval_seconds: 60
reaper_batch_size: 1000
limit_concurrency: 64
body_idle_timeout_seconds: 30
json_body_timeout_seconds: 60
bundle_body_timeout_seconds: 1800
admin:
enabled: false # fail closed; available only behind authenticated Caddy
origin: "" # exact public HTTPS origin when enabled
edge_marker: "" # independent 256-bit secret shared only with Caddy
max_body_bytes: 8192
snapshot_limit: 200
requeue_limit: 100
managed_file_roots:
runtime-keychecks:
path: "/data/runtime-linux/keychecks"
permissions:
list: true
read: true
create_replace: false
delete: false
limits:
max_relative_path_bytes: 1024
max_component_bytes: 255
max_path_depth: 16
max_listing_entries: 500
max_listing_bytes: 262144
max_file_bytes: 67108864
runtime-logs:
path: "/data/runtime-linux/logs"
permissions:
list: true
read: true
create_replace: false
delete: false
limits:
max_relative_path_bytes: 1024
max_component_bytes: 255
max_path_depth: 16
max_listing_entries: 500
max_listing_bytes: 262144
max_file_bytes: 67108864
runtime-results:
path: "/data/runtime-linux/results"
permissions:
list: true
read: true
create_replace: false
delete: false
limits:
max_relative_path_bytes: 1024
max_component_bytes: 255
max_path_depth: 16
max_listing_entries: 500
max_listing_bytes: 262144
max_file_bytes: 268435456
docker_shadow:
enabled: true # manual-only operator command; never autostarted
cohort_size: 50
lease_seconds: 3600
janitor:
enabled: true
interval_sec: 60
minimum_age_sec: 7200
max_candidates: 50
max_entries: 10000
max_bytes: 1073741824
max_seconds: 30
max_depth: 64
interval: 300 # default delay before repeating a --once child source
restart_delay: 30 # initial restart delay after failures or unexpected exits
max_restart_delay: 600 # cap for exponential restart backoff
restart_reset_after: 300 # clear failure streak/old exit after this many stable seconds
dashboard:
enabled: false # true = supervisor also starts dashboard.py
address: "127.0.0.1" # local-only dashboard bind address
port: 5000
startup_grace_sec: 30 # allow Streamlit to initialize before a failed health probe triggers restart
health_interval_sec: 2 # bounded asynchronous /_stcore/health probe interval
health_timeout_sec: 1
restart_base_sec: 2 # exponential dashboard-only restart backoff
restart_max_sec: 60
stable_health_sec: 60 # reset dashboard restart streak only after stable health
defaults:
enabled: true
once: false # false = child console_runner loops internally; true = one source cycle per child run
repeat: true # only relevant when once=true; repeat one-shot cycles after interval
restart: true # restart crashed/exited child processes
extra_args: [] # extra args passed to console_runner.py in config mode
sources:
github:
once: false
github_archive:
enabled: true # controlled broad GitHub discovery via GHArchive; start manually from supervisor
use_system_proxy: true
once: false
repeat: true
restart: true
interval: 3600
github_archive_files:
enabled: true # controlled GHArchive changed-file fetch; start manually from supervisor
use_system_proxy: true
once: false
repeat: true
restart: true
interval: 1800
github_gists:
enabled: true # controlled public gist discovery; start manually from supervisor
once: false
repeat: true
restart: true
interval: 1800
gitlab:
once: false
github_actions:
enabled: false # paused after fresh and retained cohorts produced no strict-usable yield
once: false
repeat: true
restart: true
interval: 3600
gitlab_ci:
enabled: true # show in supervisor; main source remains disabled for normal all-source runner
once: false
repeat: true
restart: true
interval: 3600
dockerhub:
once: false
npm:
once: false
pypi:
once: false
package_git:
enabled: false
once: false
huggingface:
once: false
postman:
enabled: true # allows `supervisor --sources postman`; main sources.postman.enabled controls default all-source inclusion
once: false
env:
PYTHONIOENCODING: utf-8 # avoid Windows console codec crashes on Unicode repository paths
keychecks:
enabled: true # supervisor manages this as pseudo-source "keychecks"
autostart: true # start keychecks when supervisor starts, even if scanner sources wait for manual start
input_mode: postgres # PostgreSQL candidate leases are authoritative; JSONL requires explicit compatibility mode
services: all # all or comma/list: openai,anthropic,qwen,kimi,github,...
interval: 3600 # repeat keycheck_runner every N seconds when in repeat/hourly mode
repeat: true
restart: false # do not auto-restart failed checker batch; wait for next interval/manual restart
max_keys: 0 # 0 = no per-run limit; set N for throttled hourly batches
scheduler_batch_keys: 1000 # per-provider process slice; max_keys=0 keeps rotating until empty/deadline
scheduler_workers: 1 # serialize provider handshakes; avoids transient control-plane startup failures
scheduler_deadline_sec: 1800 # aggregate work-conserving provider deadline
retry_network: true # retry transient network failures each scheduled run
retry_limited: false # set true to retry rate-limited/no-quota statuses hourly
retry_unknown: false
retry_restricted: false
retry_no_balance: false # retry no_balance/no_quota statuses when explicitly enabled
retry_valid: false # valid provider keys are re-probed only when explicitly requested
recheck_all: false # true forces all known keys to be checked again
env:
KEYCHECK_INPUT_TAIL_MB: "0" # first keycheck reads from offset 0; high-watermark state handles later appends
KEYCHECK_INPUT_MAX_LINE_BYTES: "16777216" # must match global.keycheck_input_max_line_bytes
KEYCHECK_CANDIDATE_MAX_UNCONSUMED_ATTEMPTS: "3"
KEYCHECK_DB_INGEST_TAIL_MB: "64" # optional tail size; initial DB ingest uses full bounded offsets unless KEYCHECK_DB_INGEST_TAIL_INITIAL=1
KEYCHECK_RESULTS_MAX_MB: "32" # rotate per-service *Results.jsonl files into manifest segments
KEYCHECK_PROVIDER_RESOLUTION_ORDER: "deepseek,zai,qwen,kimi"
KEYCHECK_EVENT_MAP_BACKFILL_ROWS: "0" # one-time legacy backfill is complete; live ingest maintains this map atomically
KEYCHECK_UID_MAP_BACKFILL_ROWS: "0" # scanner writes finding_uid_map for all new findings
db_ingest:
enabled: false # explicit JSONL compatibility import only
repair_links:
enabled: false # new DB candidates carry exact finding attribution
service_args: # provider-specific probe flags passed by keycheck_runner
gemini:
- --probe-generation # call generateContent; RATE_LIMITED valid keys go to geminiAliveRateLimited.txt
aws:
- --probe-bedrock # after STS, probe Bedrock access using safe validation-style calls
- --bedrock-max-attempts
- "12"
azure:
- --probe-openai-route # after deployments list, probe Azure OpenAI chat route without generation
- --probe-foundry-route # probe Azure AI Foundry/MaaS route for configured models
- --foundry-models
- claude-opus-4-6,claude-fable-5
- --timeout
- "8"
gcp:
- --probe-vertex # after OAuth, probe Vertex AI Gemini countTokens access
- --vertex-timeout
- "6"
- --vertex-max-attempts
- "6"
- --vertex-models
- gemini-3.6-flash,gemini-3.1-pro-preview
- --vertex-locations
- global,us,eu
- --vertex-anthropic-models
- claude-opus-5,claude-opus-4-7,claude-opus-4-6,claude-fable-5
- --vertex-anthropic-locations
- global,us,eu,us-east5,europe-west1
- --vertex-anthropic-max-attempts
- "20"
summary_tsv: "{keycheck_dir}/summary.tsv"
summary_json: "{keycheck_dir}/summary.json"
alive_summary_tsv: "{keycheck_dir}/alive_summary.tsv"
query_policy:
rejected: # reviewed source-specific zero-alive evidence; queue rows remain auditable and reversible
- {source: dockerhub, query: OR, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1631, findings: 29137, unique_credentials: 36, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 36}
- {source: dockerhub, query: agent, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1009, findings: 69486, unique_credentials: 30, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 229}
- {source: github, query: coding, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1696, findings: 251, unique_credentials: 11, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 0}
- {source: github, query: memory, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3190, findings: 254, unique_credentials: 11, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 8}
- {source: npm, query: OR, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1985, findings: 152, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 598}
- {source: npm, query: agent, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 9446, findings: 5833, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3957}
- {source: npm, query: agents, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 8936, findings: 4573, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3346}
- {source: npm, query: ai, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 7676, findings: 49499, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3663}
- {source: npm, query: assistant, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3888, findings: 14831, unique_credentials: 2, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 899}
- {source: npm, query: benchmark, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2371, findings: 298, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1292}
- {source: npm, query: bot, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1784, findings: 2081, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 951}
- {source: npm, query: chat, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2048, findings: 381, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 2087}
- {source: npm, query: chats, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1297, findings: 107, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 719}
- {source: npm, query: completion, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2529, findings: 1351, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1263}
- {source: npm, query: completions, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3058, findings: 1169, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1466}
- {source: npm, query: conversation, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1454, findings: 475, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 954}
- {source: npm, query: gemini, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 4922, findings: 3830, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 891}
- {source: npm, query: groq, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2050, findings: 333, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 484}
- {source: npm, query: langchain, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1264, findings: 169, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 908}
- {source: npm, query: llm, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3682, findings: 17976, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 2333}
- {source: npm, query: mcp, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 8704, findings: 5572, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3524}
- {source: npm, query: open, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 6262, findings: 3288, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1359}
- {source: npm, query: openrouter, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1590, findings: 355, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1014}
- {source: npm, query: prompt, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1899, findings: 91, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1161}
- {source: npm, query: rag, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2242, findings: 1912, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 544}
- {source: npm, query: semantic, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1911, findings: 164, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1168}
- {source: npm, query: studio, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 4158, findings: 484, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1349}
- {source: npm, query: xai, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1390, findings: 2395, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 513}
- {source: package_git, query: agent, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1851, findings: 5395, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 8791}
- {source: package_git, query: bot, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1610, findings: 7260, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 4486}
- {source: package_git, query: completion, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3149, findings: 10028, unique_credentials: 9, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1507}
- {source: package_git, query: llm, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1285, findings: 2116, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 5572}
- {source: package_git, query: open, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2053, findings: 14375, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 9968}
- {source: package_git, query: semantic, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3895, findings: 24064, unique_credentials: 22, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 2414}
- {source: package_git, query: studio, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1028, findings: 2666, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 5709}
- {source: postman, query: XAI_API_KEY, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2935, findings: 83, unique_credentials: 1, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 306}
- {source: pypi, query: langchain, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1095, findings: 47, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 16}
- {source: pypi, query: open, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1253, findings: 731, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 66}
sources:
github:
enabled: false # direct GitHub discovery is paused in favor of the Actions experiment
auth_pool: github_main # auth_pools.<name> in secrets.yaml; leave empty to use env/legacy token
auth_rotation: per_cycle # per_cycle = use next token each source cycle; none = first available
retry_with_next_auth_on_rate_limit: true # switch token and retry when GitHub API rate-limits
rate_limit_cooldown: 3600 # seconds to cool down a rate-limited token if API gives no reset time
mode: recent # recent = updated recently; search = paginated search; custom = target_file URLs
queries: # one query is used per source cycle; state rotates through this list
- llm
- ai
- agent
- agents
- assistant
- bot
- chatbot
- rag
- semantic
- prompt
- completion
- completions
- mcp
- claude
- anthropic
- opus
- sonnet
- haiku
- vertex
- aiplatform
- bedrock
- foundry
- azure-openai
- openrouter
- langchain
- langgraph
- litellm
- crewai
- autogen
- semantic-kernel
- mistral
- groq
- cohere
- xai
- together
- perplexity
- gemini
- qwen
- kimi
- dashscope
- aistudio
- studio
- OR
- open
- chat
- chats
- conversation
- conversational
- dialogue
- OPENAI_API_KEY # bounded high-signal README integration query
- api.openai.com # bounded high-signal README endpoint query
- openai-agents # bounded OpenAI agent SDK/ecosystem query
- copilot
- ai-assistant
- ai-agent
- multi-agent
- workflow
- workflows
- llmops
- benchmark
- tokens
- function-calling
query_overrides:
OPENAI_API_KEY:
pages: 1
per_page: 25
max_targets: 5
api.openai.com:
pages: 1
per_page: 25
max_targets: 5
openai-agents:
pages: 1
per_page: 50
max_targets: 5
pages: 5 # pages to fetch in search/recent mode
per_page: 100 # targets per API page, max is usually 100
workers: 1 # one shared non-Docker scan slot for this source
timeout: 1800 # allow one low-priority Git clone/scan up to 30 minutes
recent_hours: 24 # recent mode: only repos updated within this many hours are discovered
max_repo_age_days: 90 # skip repos older than this by repo_age_field before queueing
repo_age_field: updated_at # metadata field for repo age: created_at, updated_at, pushed_at
max_commit_age_days: 90 # scan only recent git history by passing --since-commit boundary
commit_lookup_pages: 3 # API pages of commits to inspect when finding --since-commit boundary
skip_if_commit_lookup_fails: true # true = do not scan if recent commit boundary cannot be found
max_depth: 100 # TruffleHog --max-depth; extra cap on commit count scanned
exact_git_planning_enabled: true # resolve and bind the exact default/explicit ref head after claim
git_baseline_depth: 100 # bounded first/reset scan depth; deltas use only the covered base
git_ref_resolution_timeout_sec: 10
git_ref_resolution_attempts: 2
git_ref_resolution_max_bytes: 1048576
scan_full_history: false # true = ignore max_depth; still uses max_commit_age_days when configured
sort_by: updated # GitHub search sort: updated, stars, forks, created
sort_order: desc # desc = newest/highest first; asc = oldest/lowest first
created_filter: any # optional GitHub created filter: any, today, week, month, year
stop_on_seen_pages: true # stop fetching pages after consecutive all-known pages
seen_page_threshold: 2
min_pages_before_stop: 1
updated_target_rescan_enabled: true # preserve pushed_at and rescan completed repos after newer pushes
updated_target_rescan_max_per_cycle: 1 # bounded rollout: at most one changed repo per discovery cycle
updated_target_rescan_cooldown_hours: 24
github_archive:
enabled: false # broad discovery source; supervisor exposes it for manual/controlled runs
use_system_proxy: true # direct GHArchive route is unstable; use inherited HTTP(S)_PROXY only for hourly dumps
auth_pool: github_main
auth_rotation: per_cycle
mode: archive # GHArchive hourly events -> GitHub repos -> recent git scan
queries:
- gharchive
archive_hours_back: 6 # read recent completed GHArchive hours
fetch_timeout: 120 # read timeout while downloading large hourly gzip archives
archive_max_repos_per_cycle: 500 # hard cap after event/repo dedupe/scoring
archive_rescan_cooldown_hours: 48 # allow rescanning active repos after cooldown; not forever-checked
archive_event_types:
- PushEvent
- CreateEvent
- PublicEvent
workers: 4
timeout: 1800
max_depth: 75
scan_full_history: false
max_commit_age_days: 0 # avoid GitHub commit-boundary API lookups for broad source
commit_lookup_pages: 0
skip_if_commit_lookup_fails: false
stop_on_seen_pages: false
github_archive_files:
enabled: false # fetch suspicious changed files from GHArchive PushEvents
use_system_proxy: true
auth_pool: github_main
auth_rotation: per_cycle
mode: archive
queries:
- gharchive-files
archive_hours_back: 6
archive_max_files_per_cycle: 100
archive_max_commit_lookups: 200
archive_event_types:
- PushEvent
workers: 1
timeout: 300
fetch_timeout: 120
max_artifact_size_mb: 2
stop_on_seen_pages: false
github_gists:
enabled: false # public gist source; supervisor exposes it for manual/controlled runs
auth_pool: github_main
auth_rotation: per_cycle
mode: search
queries:
- gists
pages: 3
per_page: 100
workers: 3
timeout: 300
fetch_timeout: 20
max_artifact_size_mb: 2
stop_on_seen_pages: true
seen_page_threshold: 2
min_pages_before_stop: 1
gitlab:
enabled: true # include the GitLab discovery producer in distributed runs
target_claim_order: oldest # oldest, newest, or balanced PostgreSQL queue admission
auth_pool: gitlab_main # auth_pools.<name> in secrets.yaml; leave empty to use env/legacy token
auth_rotation: per_cycle # per_cycle = use next token each source cycle; none = first available
retry_with_next_auth_on_rate_limit: true # switch token and retry when GitLab API returns 429
rate_limit_cooldown: 3600 # seconds to cool down a rate-limited token if API gives no reset time
discovery_request_attempts: 3 # bounded retries for idempotent project discovery transport failures
discovery_retry_delay: 5 # seconds between GitLab discovery transport attempts
external_trufflehog_lifecycle: true # bypass embedded overseer and require explicit completion for GitLab scans
mode: recent # recent = last_activity_after; search = paginated project search; custom = target_file URLs
queries: # one query is used per source cycle; state rotates through this list
- llm
- ai
- agent
- agents
- assistant
- bot
- chatbot
- rag
- openai-api # bounded metadata-friendly OpenAI API query
- openai-agents # bounded OpenAI agent SDK/ecosystem query
- librechat # bounded deployable chat application query
- semantic
- prompt
- completion
- completions
- mcp
- openrouter
- groq
- xai
- langchain
- litellm
- gemini
- qwen
- kimi
- dashscope
- aistudio
- studio
- OR
- open
- chat
- chats
- conversation
- conversational
- dialogue
- copilot
- coding
- ai-assistant
- ai-agent
- multi-agent
- workflow
- workflows
- llmops
- benchmark
- tokens
- memory
- function-calling
- hermes
- harnes
- flow
- helpdesk
- paperless
query_overrides:
openai-api:
pages: 1
per_page: 50
max_targets: 5
openai-agents:
pages: 1
per_page: 50
max_targets: 5
librechat:
pages: 1
per_page: 25
max_targets: 5
hermes:
pages: 1
per_page: 50
max_targets: 5
harnes:
pages: 1
per_page: 50
max_targets: 5
flow:
pages: 1
per_page: 50
max_targets: 5
helpdesk:
pages: 1
per_page: 50
max_targets: 5
paperless:
pages: 1
per_page: 50
max_targets: 5
pages: 10 # pages to fetch in search/recent mode
per_page: 100 # targets per API page, max is usually 100
workers: 1 # one shared non-Docker scan slot for this source
timeout: 1800 # allow one low-priority Git clone/scan up to 30 minutes
recent_hours: 96 # recent mode: only projects active within this many hours are discovered
max_repo_age_days: 90 # skip projects older than this by repo_age_field before queueing
repo_age_field: last_activity_at # metadata field: created_at, updated_at, last_activity_at
max_commit_age_days: 90 # scan only recent git history by passing --since-commit boundary
commit_lookup_pages: 3 # API pages of commits to inspect when finding --since-commit boundary
skip_if_commit_lookup_fails: true # true = do not scan if recent commit boundary cannot be found
max_depth: 100 # TruffleHog --max-depth; extra cap on commit count scanned
exact_git_planning_enabled: true # resolve and bind the exact default/explicit ref head after claim
git_baseline_depth: 100 # bounded first/reset scan depth; deltas use only the covered base
git_ref_resolution_timeout_sec: 10
git_ref_resolution_attempts: 2
git_ref_resolution_max_bytes: 1048576
scan_full_history: false # true = ignore max_depth; still uses max_commit_age_days when configured
sort_by: last_activity_at # GitLab order_by field: last_activity_at, created_at, updated_at, name, stars
sort_order: desc # desc = newest/highest first; asc = oldest/lowest first
visibility: public # public, internal, private; private/internal require token permissions
stop_on_seen_pages: true # stop fetching pages after consecutive all-known pages
seen_page_threshold: 2
min_pages_before_stop: 1
updated_target_rescan_enabled: true # preserve last_activity_at and rescan completed projects after new activity
updated_target_rescan_max_per_cycle: 1
updated_target_rescan_cooldown_hours: 24
github_actions:
enabled: false # paused; queue and historical evidence remain intact
auth_pool: github_main
auth_rotation: per_cycle
mode: recent
queries:
- logs
ci_seed_sources: github,github_archive,package_git
ci_max_repos_per_cycle: 25
ci_seed_scan_limit: 15000
ci_seed_query_batch_size: 250 # bound each PostgreSQL history page on HDD
ci_soft_cooldown_days: 7
ci_runs_per_repo: 5
ci_lookback_days: 30
ci_max_log_archive_mb: 150
ci_max_log_file_mb: 50
ci_scan_artifacts: true
ci_max_artifacts_per_run: 3
ci_max_artifact_archive_mb: 150
ci_max_artifact_file_mb: 50
ci_max_artifact_files: 1000
ci_failed_first: true
refresh_registry: true # keep discovering recent repositories while historical targets remain queued
target_claim_order: balanced # split work between fresh discoveries and the retained historical backlog
workers: 1
timeout: 180
gitlab_ci:
enabled: false # controlled experiment: run manually from supervisor
auth_pool: gitlab_main
auth_rotation: per_cycle
mode: recent
queries:
- logs
ci_seed_sources: gitlab,package_git
ci_max_repos_per_cycle: 25
ci_seed_scan_limit: 15000
ci_seed_query_batch_size: 250 # bound each PostgreSQL history page on HDD
ci_soft_cooldown_days: 7
ci_pipelines_per_project: 5
ci_jobs_per_pipeline: 20
ci_lookback_days: 30
ci_max_trace_mb: 100
ci_scan_artifacts: true
ci_max_artifacts_per_pipeline: 5
ci_max_artifact_archive_mb: 150
ci_max_artifact_file_mb: 50
ci_max_artifact_files: 1000
workers: 3
timeout: 180
huggingface:
enabled: true # discover newest HuggingFace Spaces for remote workers
target_claim_order: oldest # oldest, newest, or balanced PostgreSQL queue admission
auth_pool: huggingface_main # optional auth_pools.<name> in secrets.yaml or use HF_TOKEN/HUGGINGFACE_TOKEN
auth_rotation: per_cycle
mode: recent # recent/search both fetch newest spaces; custom = target_file space IDs
queries:
- spaces # placeholder query for state rotation; HuggingFace API fetch ignores it for now
pages: 100 # bounded cursor pages from the newest-lastModified Spaces API
per_page: 100 # newest-modified API page size is fixed at 100 by the runner
workers: 1
timeout: 1800
fetch_timeout: 15
discovery_request_attempts: 3 # one transient page timeout must not restart the whole source
discovery_retry_delay: 5
stop_on_seen_pages: true
seen_page_threshold: 2
min_pages_before_stop: 1
updated_target_rescan_enabled: true # preserve lastModified and revisit changed completed Spaces
updated_target_rescan_max_per_cycle: 1
updated_target_rescan_cooldown_hours: 24
dockerhub:
enabled: true # include the DockerHub discovery producer in distributed runs
target_claim_order: oldest # oldest, newest, or balanced PostgreSQL queue admission
require_digest: true # only immutable repo@sha256:... image targets may reach TruffleHog
auth_pool: dockerhub_main # auth_pools.<name> in secrets.yaml; all available accounts are rotated per image scan
auth_rotation: per_scan # DockerTokenManager rotates Docker accounts for each docker scan
retry_with_next_auth_on_rate_limit: true # rotate Hub/Registry requests to another account after 429/invalid auth
rate_limit_cooldown: 1800 # per-account cooldown; shared cooldown starts only after pool exhaustion
mode: search # search = Docker Hub search; recent = client-side recent tag filtering; custom = target_file images
refresh_registry: true # discover fresh images even while deferred targets remain queued
queries: # one query is used per source cycle; Docker Hub empty query returns no results
- llm
- ai
- agents
- assistant
- bot
- chatbot
- rag
- semantic
- prompt
- completion
- completions
- mcp
- openrouter
- groq
- xai
- langchain
- litellm
- gemini
- qwen
- kimi
- dashscope
- aistudio
- studio
- librechat # bounded deployable chat application query
- lobechat # bounded deployable chat application query
- openai-proxy # bounded OpenAI-compatible proxy query
- open
- chat
- chats
- conversation
- conversational
- dialogue
- copilot
- coding
- ai-assistant
- ai-agent
- multi-agent
- workflow
- workflows
- llmops
- benchmark
- tokens
- memory
- function-calling
- hermes
- harnes
- flow
- helpdesk
- paperless
- open-webui
- ragflow
- dify
- flowise
- crewai
- n8n
- langflow
- autogen
- browser-use
- openhands
- anythingllm
- agent-zero
query_overrides:
librechat:
max_targets: 10
lobechat:
max_targets: 10
openai-proxy:
max_targets: 10
hermes:
max_targets: 10
harnes:
max_targets: 10
flow:
max_targets: 10
helpdesk:
max_targets: 10
paperless:
max_targets: 10
pages: 30 # maximum Docker Hub search pages for every configured query
per_page: 100 # maximum Docker Hub results per search page
workers: 2 # allow two Docker image scans within the global three-slot limit
trufflehog_job_memory_limit_bytes: 6442450944 # Docker images get 6 GiB; other sources retain the 4 GiB global cap
timeout: 600 # bounded full-image scan window after disabling TruffleHog's embedded overseer
trufflehog_concurrency: 4 # bound per-image layer/chunk fan-out; two source workers remain available
recent_days: 7 # recent mode: keep images/tags updated within this many days
fetch_workers: 5 # parallel Docker Hub search page fetches before scanning
fetch_timeout: 15 # seconds per Docker Hub API request
tag_fetch_workers: 4 # bound the in-flight burst before a shared 429 cooldown becomes visible
tag_retry_count: 0 # do not retry individual transport failures during high-volume tag resolution
tag_retry_delay: 10 # base delay if bounded non-429 transport retries are enabled later
tag_resolve_limit: 100 # max old bare todo repos to resolve to tags per cycle; 0 = all
docker_platform_filter_enabled: true # skip tags only when complete metadata proves linux/amd64 is absent
docker_platform_os: linux
docker_platform_arch: amd64
docker_platform_candidate_tags: 20 # inspect extra recent tags so an ARM-only latest tag does not hide an amd64 tag
docker_images_per_repository: 3 # ordinary FIFO resolver stays at the reviewed production depth
docker_content_scan_mode: canary # only prior durable full-image timeouts are eligible for layer fallback
docker_layer_canary_basis_points: 10000 # all prior durable full-image timeouts use bounded layer fallback
docker_adaptive_canary_basis_points: 0 # disabled until an exact aggregate shadow report passes every gate
docker_adaptive_gate_max_age_sec: 604800 # matching shadow evidence expires after seven days
docker_layer_config_max_bytes: 1048576 # image config is selected independently from layer budget
docker_layer_max_bytes: 268435456 # max compressed bytes for one selected application layer
docker_layer_image_max_bytes: 1073741824 # cumulative compressed layer budget per immutable image
docker_layer_max_layers: 8 # select application layers from top to base within the byte budget
docker_layer_archive_max_size_bytes: 268435456 # TruffleHog per-member extraction bound
docker_layer_archive_max_depth: 4 # required for an OCI wrapper plus compressed layer archive
docker_layer_archive_timeout_sec: 30
docker_layer_blob_timeout_sec: 600 # shared transfer+filesystem scan deadline per durable checkpoint
docker_layer_filesystem_concurrency: 2
docker_layer_blob_max_attempts: 3 # blob budget is authoritative; parent checkpoint attempts reset
docker_layer_blob_lease_sec: 1800 # exceeds the bounded 600-second execution plus handoff margin
docker_layer_min_free_bytes: 21474836480 # retain 20 GiB on the private work volume
docker_layer_checkpoint_delay_sec: 60 # resume the next selected blob without a full-image restart
docker_adaptive_checkpoint_max_blobs: 4 # scheduling-only bound for a future gated adaptive checkpoint
docker_adaptive_checkpoint_max_bytes: 536870912 # aggregate compressed bytes per adaptive checkpoint
docker_repository_refresh_interval_sec: 86400 # revisit each resolved repository daily for new immutable digests
docker_repository_refresh_max_per_cycle: 0 # disable periodic refresh of completed repository anchors
sort_by: updated_at # Docker Hub search sort field
sort_order: desc # desc = newest/highest first; asc = oldest/lowest first
npm:
enabled: true # true = include npm registry in config-mode runs
mode: search # npm currently supports search mode
queries: # one query is used per source cycle; state rotates through this list
- chatbot
- litellm
- qwen
- kimi
- dashscope
- aistudio
- conversational
- dialogue
- copilot
- coding
- ai-assistant
- ai-agent
- multi-agent
- workflow
- workflows
- llmops
- tokens
- memory
- function-calling
pages: 30 # npm search pages to fetch for the current query
per_page: 50 # npm packages per search page
workers: 3 # parallel TruffleHog filesystem scans for downloaded packages
timeout: 300 # seconds before killing one package scan process tree
versions_per_package: 3 # scan latest N versions per package within max_version_age_days
max_version_age_days: 90 # skip package versions older than this many days
max_artifact_size_mb: 300 # skip/download-fail package tarballs larger than this
fetch_timeout: 20 # seconds per npm registry request
pypi:
enabled: true # true = include PyPI registry in config-mode runs
mode: search # PyPI currently supports search mode
queries: # one query is used per source cycle; state rotates through this list
- llm
- ai
- agent
- agents
- assistant
- bot
- chatbot
- rag
- semantic
- prompt
- completion
- completions
- mcp
- openrouter
- groq
- xai
- litellm
- gemini
- qwen
- kimi
- dashscope
- aistudio
- studio
- OR
- chat
- chats
- conversation
- conversational
- dialogue
- copilot
- coding
- ai-assistant
- ai-agent
- multi-agent
- workflow
- workflows
- llmops
- benchmark
- tokens
- memory
- function-calling
pages: 40 # PyPI search pages to fetch for the current query
per_page: 50 # max package names to process per PyPI search page
workers: 3 # parallel TruffleHog filesystem scans for downloaded packages
timeout: 300 # seconds before killing one package scan process tree
versions_per_package: 3 # scan latest N release artifacts per package within max_version_age_days
max_version_age_days: 90 # skip package releases older than this many days
max_artifact_size_mb: 300 # skip/download-fail package artifacts larger than this
fetch_timeout: 20 # seconds per PyPI request
package_git:
enabled: false # disabled after package-level discovery became duplicate-heavy
auth_pool: github_main # most package metadata points to GitHub; GitLab 401/403 falls back unauthenticated
auth_rotation: per_cycle
mode: search # package_git currently supports search mode and custom JSON/git URL targets
package_sources: # registries used for package -> repository discovery
- npm
- pypi
queries:
- ai
- agents
- assistant
- chatbot
- rag
- prompt
- completions
- mcp
- openrouter
- groq
- xai
- langchain
- litellm
- gemini
- qwen
- kimi
- dashscope
- aistudio
- OR
- chat
- chats
- conversation
- conversational
- dialogue
- copilot
- coding
- ai-assistant
- ai-agent
- multi-agent
- workflow
- workflows
- llmops
- benchmark
- tokens
- memory
- function-calling
pages: 10 # registry search pages per query for repo discovery
per_page: 50 # packages per search page
refresh_registry: true # merge cached repo candidates with fresh registry discovery each cycle
max_targets: 30 # bound one cycle so registry discovery cannot be starved by historical backlog
target_claim_order: balanced # split claims between oldest backlog and newest discovered repositories
workers: 3 # parallel TruffleHog git scans
timeout: 1800 # allow one low-priority Git clone/scan up to 30 minutes
versions_per_package: 3 # inspect repo metadata for up to N recent package versions
max_version_age_days: 90 # ignore package versions older than this during discovery
max_depth: 500 # git commit depth for package-derived repos
scan_full_history: false
max_commit_age_days: 0 # 0 avoids extra provider API commit-boundary lookup by default
commit_lookup_pages: 3
skip_if_commit_lookup_fails: false
fetch_timeout: 20 # seconds per registry metadata request
postman:
enabled: true # first run: enable manually for controlled backfill/tail scans
auth_pool: github_main # uses GitHub tokens for code search, commit lookup, and content download
auth_rotation: per_cycle # runner state still records a last auth; source-local pool rotates all tokens per request
mode: search # search = GitHub code search for Postman JSON artifacts; custom = target_file JSON targets
queries:
- anthropic
- gemini
- qwen
- kimi
- dashscope
- llm
- azure-openai
- openai.azure.com
- foundry
- services.ai.azure.com
- models.ai.azure.com
- DASHSCOPE_API_KEY
- QWEN_API_KEY
- MOONSHOT_API_KEY
- KIMI_API_KEY
- dashscope.aliyuncs.com
- api.moonshot.ai
- api.moonshot.cn
- GROQ_API_KEY
- api.groq.com
- OPENROUTER_API_KEY
- api.openrouter.ai
- REPLICATE_API_TOKEN
- api.replicate.com
- api.x.ai
- HF_TOKEN
- HUGGINGFACE_TOKEN
- ANTHROPIC_API_KEY
- api.anthropic.com
- rag
- agent
- assistant
search_kinds:
- all # Postman, Insomnia, Bruno, Thunder Client, Hoppscotch, and signature searches
pages: 2 # tail default; use 10 for one-time backfill up to GitHub's 1000-result cap
per_page: 100
workers: 3
timeout: 300
fetch_timeout: 20
max_targets: 0 # set a small value for smoke tests
max_file_age_days: 365 # backfill freshness filter by latest commit touching the file path; 0 disables
max_artifact_size_mb: 200
postman_cache_dir: "{postman_cache_dir}"
github_code_search_rpm: 8 # safe per-token code search request rate; GitHub limit is about 10/min/token
all_tokens_cooldown: 1800 # fallback sleep when all GitHub tokens are rate-limited and no reset is known
stop_on_seen_pages: true # tail mode: stop after consecutive fully known pages
seen_page_threshold: 2
min_pages_before_stop: 1