Initial server source import
This commit is contained in:
@@ -0,0 +1 @@
|
||||
"""Keychecker package for the unified scanner layout."""
|
||||
@@ -0,0 +1,274 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
read_plain_keys,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "anthropic"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "anthropicChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "anthropicResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "anthropicAlive.txt"),
|
||||
"NO_QUOTA": os.path.join(OUTPUT_DIR, "anthropicNoQuota.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "anthropicDead.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "anthropicLimited.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "anthropicRestricted.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "anthropicNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "anthropicUnknown.txt"),
|
||||
}
|
||||
|
||||
ANTHROPIC_REGEX = re.compile(r"sk-ant-(?:api03|admin01)-[A-Za-z0-9\-_]{93}AA|sk-ant-[A-Za-z0-9\-_]{86}")
|
||||
ANTHROPIC_ADMIN_PREFIX = "sk-ant-admin01-"
|
||||
ANTHROPIC_ADMIN_API_KEYS_URL = "https://api.anthropic.com/v1/organizations/api_keys"
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, ["Anthropic"]):
|
||||
key = item["raw"]
|
||||
if key and ANTHROPIC_REGEX.fullmatch(key):
|
||||
yield key, item["source"], item["finding"]
|
||||
for item in read_plain_keys(plain_files, ANTHROPIC_REGEX):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
|
||||
def tier_from_rpm(rpm):
|
||||
mapping = {5: "Free Tier", 50: "Tier 1", 1000: "Tier 2", 2000: "Tier 3", 4000: "Tier 4"}
|
||||
return mapping.get(rpm, "Scale/Unknown")
|
||||
|
||||
|
||||
def anthropic_rate_headers(response):
|
||||
output = {}
|
||||
for name, value in response.headers.items():
|
||||
lowered = name.lower()
|
||||
if lowered.startswith("anthropic-ratelimit-"):
|
||||
output[lowered.replace("anthropic-ratelimit-", "rate_").replace("-", "_")] = value
|
||||
return output
|
||||
|
||||
|
||||
def list_models(key, proxy, timeout):
|
||||
headers = {
|
||||
"anthropic-version": "2023-06-01",
|
||||
"x-api-key": key,
|
||||
}
|
||||
try:
|
||||
response = requests.get("https://api.anthropic.com/v1/models", headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"models_error": str(exc)[:500]}
|
||||
if response.status_code != 200:
|
||||
return {"models_status": response.status_code, "models_error": request_error_message(response)}
|
||||
try:
|
||||
data = response.json()
|
||||
except ValueError:
|
||||
return {"models_status": response.status_code, "models_error": "invalid JSON response"}
|
||||
models = []
|
||||
for item in data.get("data") or []:
|
||||
if isinstance(item, dict) and item.get("id"):
|
||||
models.append(item["id"])
|
||||
return {
|
||||
"models_status": response.status_code,
|
||||
"models_count": len(models),
|
||||
"models": models[:50],
|
||||
}
|
||||
|
||||
|
||||
def check_admin_key(key, proxy, timeout):
|
||||
headers = {
|
||||
"anthropic-version": "2023-06-01",
|
||||
"x-api-key": key,
|
||||
}
|
||||
try:
|
||||
response = requests.get(
|
||||
ANTHROPIC_ADMIN_API_KEYS_URL,
|
||||
headers=headers,
|
||||
params={"limit": 1},
|
||||
proxies=proxy,
|
||||
timeout=timeout,
|
||||
)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "admin": True, "message": str(exc)[:500]}
|
||||
|
||||
if response.status_code == 200:
|
||||
return {
|
||||
"status": "VALID",
|
||||
"admin": True,
|
||||
"http_status": 200,
|
||||
"message": "Anthropic Admin API access confirmed",
|
||||
}
|
||||
|
||||
status = {
|
||||
401: "DEAD",
|
||||
403: "RESTRICTED",
|
||||
429: "LIMITED",
|
||||
}.get(response.status_code, "UNKNOWN")
|
||||
message = request_error_message(response).replace(key, "***REDACTED***")
|
||||
return {
|
||||
"status": status,
|
||||
"admin": True,
|
||||
"http_status": response.status_code,
|
||||
"message": message,
|
||||
}
|
||||
|
||||
|
||||
def check_key(key, proxy, timeout, model="claude-opus-4-6", include_models=False):
|
||||
if key.startswith(ANTHROPIC_ADMIN_PREFIX):
|
||||
return check_admin_key(key, proxy, timeout)
|
||||
|
||||
url = "https://api.anthropic.com/v1/messages"
|
||||
headers = {
|
||||
"content-type": "application/json",
|
||||
"anthropic-version": "2023-06-01",
|
||||
"x-api-key": key,
|
||||
}
|
||||
payload = {
|
||||
"model": model,
|
||||
"messages": [{"role": "user", "content": "ping"}],
|
||||
"max_tokens": 1,
|
||||
}
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)}
|
||||
|
||||
if response.status_code == 200:
|
||||
rpm = 0
|
||||
try:
|
||||
rpm = int(response.headers.get("anthropic-ratelimit-requests-limit", "0"))
|
||||
except ValueError:
|
||||
rpm = 0
|
||||
rate_data = anthropic_rate_headers(response)
|
||||
result = {
|
||||
"status": "VALID",
|
||||
"model": model,
|
||||
"rpm": rpm,
|
||||
"tier": tier_from_rpm(rpm),
|
||||
**rate_data,
|
||||
}
|
||||
if include_models:
|
||||
result.update(list_models(key, proxy, timeout))
|
||||
token_limit = rate_data.get("rate_tokens_limit") or ""
|
||||
token_remaining = rate_data.get("rate_tokens_remaining") or ""
|
||||
token_part = f" tokens={token_remaining}/{token_limit}" if token_limit or token_remaining else ""
|
||||
result["message"] = f"model={model}; rpm={rpm}; tier={result['tier']}{token_part}"
|
||||
return result
|
||||
|
||||
if response.status_code == 429:
|
||||
return {"status": "LIMITED", "http_status": 429, "message": request_error_message(response)}
|
||||
|
||||
message = request_error_message(response)
|
||||
lower = message.lower()
|
||||
if "credit balance is too low" in lower or "usage limits" in lower:
|
||||
return {"status": "NO_QUOTA", "http_status": response.status_code, "message": message}
|
||||
if response.status_code in (401, 403):
|
||||
status = "RESTRICTED" if "disabled" in lower or response.status_code == 403 else "DEAD"
|
||||
return {"status": status, "http_status": response.status_code, "message": message}
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "Anthropic")
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), source,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, "Anthropic")
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Anthropic key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--model", default=os.getenv("ANTHROPIC_CHECK_MODEL", "claude-opus-4-6"))
|
||||
parser.add_argument("--list-models", action="store_true")
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_no_balance:
|
||||
retry_statuses.add("NO_QUOTA")
|
||||
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, source, finding in extract_candidates(args.input, args.plain):
|
||||
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector="Anthropic"):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] Anthropic candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_key(key, proxy, args.timeout, args.model, args.list_models)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,607 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_known_statuses,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
read_plain_keys,
|
||||
recover_status_transaction,
|
||||
record_cached_keycheck_occurrence,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "aws"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "awsChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "awsResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "awsAlive.txt"),
|
||||
"BEDROCK": os.path.join(OUTPUT_DIR, "awsBedrock.txt"),
|
||||
"ADMIN": os.path.join(OUTPUT_DIR, "awsAdmin.txt"),
|
||||
"CANARY": os.path.join(OUTPUT_DIR, "awsCanary.txt"),
|
||||
"QUARANTINED": os.path.join(OUTPUT_DIR, "awsQuarantined.txt"),
|
||||
"ACCESS_DENIED": os.path.join(OUTPUT_DIR, "awsAccessDenied.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "awsDead.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "awsNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "awsUnknown.txt"),
|
||||
}
|
||||
|
||||
BEDROCK_REGIONS = ["us-east-1", "us-west-2", "eu-west-1", "eu-north-1", "ap-northeast-1", "ap-southeast-4"]
|
||||
ANTHROPIC_MESSAGES_PROBE = {
|
||||
"anthropic_version": "bedrock-2023-05-31",
|
||||
"messages": [{"role": "user", "content": "ping"}],
|
||||
"max_tokens": -1,
|
||||
}
|
||||
ANTHROPIC_MESSAGES_LIVE_PING = {
|
||||
"anthropic_version": "bedrock-2023-05-31",
|
||||
"messages": [{"role": "user", "content": "ping"}],
|
||||
"max_tokens": 1,
|
||||
}
|
||||
BEDROCK_MODEL_TESTS = {
|
||||
# Current Anthropic Bedrock runtime IDs. The default probe intentionally uses
|
||||
# invalid max_tokens to validate auth/model access without generating tokens.
|
||||
"anthropic.claude-fable-5": ANTHROPIC_MESSAGES_PROBE,
|
||||
"us.anthropic.claude-fable-5": ANTHROPIC_MESSAGES_PROBE,
|
||||
"global.anthropic.claude-fable-5": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-sonnet-5": ANTHROPIC_MESSAGES_PROBE,
|
||||
"us.anthropic.claude-sonnet-5": ANTHROPIC_MESSAGES_PROBE,
|
||||
"global.anthropic.claude-sonnet-5": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-opus-4-8": ANTHROPIC_MESSAGES_PROBE,
|
||||
"us.anthropic.claude-opus-4-8": ANTHROPIC_MESSAGES_PROBE,
|
||||
"global.anthropic.claude-opus-4-8": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-opus-4-7": ANTHROPIC_MESSAGES_PROBE,
|
||||
"us.anthropic.claude-opus-4-7": ANTHROPIC_MESSAGES_PROBE,
|
||||
"global.anthropic.claude-opus-4-7": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-sonnet-4-6": ANTHROPIC_MESSAGES_PROBE,
|
||||
"us.anthropic.claude-sonnet-4-6": ANTHROPIC_MESSAGES_PROBE,
|
||||
"global.anthropic.claude-sonnet-4-6": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-haiku-4-5-20251001-v1:0": ANTHROPIC_MESSAGES_PROBE,
|
||||
"us.anthropic.claude-haiku-4-5-20251001-v1:0": ANTHROPIC_MESSAGES_PROBE,
|
||||
"global.anthropic.claude-haiku-4-5-20251001-v1:0": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-3-5-sonnet-20241022-v2:0": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-3-5-haiku-20241022-v1:0": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-3-haiku-20240307-v1:0": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-v2": {"prompt": "\n\nHuman:\n\nAssistant:", "max_tokens_to_sample": -1},
|
||||
"anthropic.claude-instant-v1": {"prompt": "\n\nHuman:\n\nAssistant:", "max_tokens_to_sample": -1},
|
||||
}
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, ["AWS"]):
|
||||
key = item["raw_v2"] or item["raw"]
|
||||
if key and ":" in key:
|
||||
yield key, item["source"], item["finding"]
|
||||
import re
|
||||
regex = re.compile(r"AKIA[0-9A-Z]{16}:[A-Za-z0-9+/]{40}")
|
||||
for item in read_plain_keys(plain_files, regex):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
|
||||
def is_dead_aws_error(code):
|
||||
return code in {"InvalidClientTokenId", "SignatureDoesNotMatch", "AuthFailure", "UnrecognizedClientException"}
|
||||
|
||||
|
||||
def is_canary_text(value):
|
||||
value = str(value or "").lower()
|
||||
return "canarytokens" in value or "canary token" in value or "is_canary" in value
|
||||
|
||||
|
||||
def is_canary_finding(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return False
|
||||
extra = finding.get("ExtraData") or {}
|
||||
if isinstance(extra, dict):
|
||||
if str(extra.get("is_canary", "")).lower() == "true":
|
||||
return True
|
||||
if any(is_canary_text(value) for value in extra.values()):
|
||||
return True
|
||||
return is_canary_text(finding.get("Raw")) or is_canary_text(finding.get("RawV2"))
|
||||
|
||||
|
||||
def is_canary_arn(arn):
|
||||
return is_canary_text(arn)
|
||||
|
||||
|
||||
def aws_client(session, service, proxy=None, region_name=None, timeout=20):
|
||||
kwargs = {}
|
||||
if region_name:
|
||||
kwargs["region_name"] = region_name
|
||||
from botocore.config import Config
|
||||
kwargs["config"] = Config(
|
||||
proxies=proxy or None,
|
||||
connect_timeout=timeout,
|
||||
read_timeout=timeout,
|
||||
retries={"max_attempts": 1},
|
||||
)
|
||||
return session.client(service, **kwargs)
|
||||
|
||||
|
||||
def bedrock_validation_allows_invoke(exc):
|
||||
text = str(exc or "").lower()
|
||||
if any(item in text for item in ("operation not allowed", "not authorized", "access denied")):
|
||||
return False
|
||||
# The default probe sends deliberately invalid token limits. If Bedrock only
|
||||
# rejects the payload shape after auth, InvokeModel reached the model path.
|
||||
return any(item in text for item in ("max_tokens", "max_tokens_to_sample", "malformed input", "schema"))
|
||||
|
||||
|
||||
def client_error_code(exc):
|
||||
try:
|
||||
return exc.response.get("Error", {}).get("Code", "ClientError")
|
||||
except Exception:
|
||||
return "ClientError"
|
||||
|
||||
|
||||
def client_error_message(exc):
|
||||
try:
|
||||
return exc.response.get("Error", {}).get("Message", str(exc))
|
||||
except Exception:
|
||||
return str(exc)
|
||||
|
||||
|
||||
def model_arn(region, model_id):
|
||||
# Cross-region inference profile IDs are not foundation-model ARNs.
|
||||
if model_id.startswith(("us.", "eu.", "jp.", "au.", "global.")):
|
||||
return "*"
|
||||
return f"arn:aws:bedrock:{region}::foundation-model/{model_id}"
|
||||
|
||||
|
||||
def iam_policy_source_arn(sts_arn, account):
|
||||
arn = str(sts_arn or "")
|
||||
if ":assumed-role/" in arn:
|
||||
role_part = arn.split(":assumed-role/", 1)[1].split("/", 1)[0]
|
||||
return f"arn:aws:iam::{account}:role/{role_part}"
|
||||
return arn if ":iam::" in arn else ""
|
||||
|
||||
|
||||
def simulate_bedrock_activation(session, arn, account, region, model_id, proxy=None, timeout=20):
|
||||
import botocore.exceptions
|
||||
|
||||
source_arn = iam_policy_source_arn(arn, account)
|
||||
if not source_arn:
|
||||
return {"status": "not_available", "message": "unsupported principal arn for IAM simulation"}
|
||||
actions = [
|
||||
"bedrock:GetFoundationModelAvailability",
|
||||
"bedrock:ListFoundationModelAgreementOffers",
|
||||
"bedrock:GetUseCaseForModelAccess",
|
||||
"bedrock:PutUseCaseForModelAccess",
|
||||
"bedrock:CreateFoundationModelAgreement",
|
||||
"bedrock:GetInferenceProfile",
|
||||
"bedrock:InvokeModel",
|
||||
]
|
||||
try:
|
||||
iam = aws_client(session, "iam", proxy, timeout=timeout)
|
||||
response = iam.simulate_principal_policy(
|
||||
PolicySourceArn=source_arn,
|
||||
ActionNames=actions,
|
||||
ResourceArns=[model_arn(region, model_id)],
|
||||
)
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
return {
|
||||
"status": "access_denied" if client_error_code(exc) == "AccessDenied" else "error",
|
||||
"code": client_error_code(exc),
|
||||
"message": client_error_message(exc)[:500],
|
||||
}
|
||||
decisions = {}
|
||||
for item in response.get("EvaluationResults", []):
|
||||
action = str(item.get("EvalActionName") or "")
|
||||
decisions[action] = str(item.get("EvalDecision") or "")
|
||||
activation_actions = ["bedrock:PutUseCaseForModelAccess", "bedrock:CreateFoundationModelAgreement"]
|
||||
can_activate = all(decisions.get(action) == "allowed" for action in activation_actions)
|
||||
return {"status": "ok", "source_arn": source_arn, "can_activate": can_activate, "decisions": decisions}
|
||||
|
||||
|
||||
def check_bedrock_management(session, arn, account, proxy=None, timeout=20, regions=None, models=None, max_attempts=12, debug=False):
|
||||
import botocore.exceptions
|
||||
|
||||
attempts = []
|
||||
findings = []
|
||||
tried = 0
|
||||
for region in (regions or BEDROCK_REGIONS):
|
||||
bedrock = aws_client(session, "bedrock", proxy, region, timeout)
|
||||
use_case = None
|
||||
try:
|
||||
use_case = bedrock.get_use_case_for_model_access()
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
use_case = {"error_code": client_error_code(exc), "message": client_error_message(exc)[:300]}
|
||||
try:
|
||||
profiles = bedrock.list_inference_profiles(typeEquals="SYSTEM_DEFINED", maxResults=20).get("inferenceProfileSummaries", [])
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
profiles = {"error_code": client_error_code(exc), "message": client_error_message(exc)[:300]}
|
||||
for model_id in (models or list(BEDROCK_MODEL_TESTS.keys())):
|
||||
if max_attempts and tried >= max_attempts:
|
||||
return {"enabled": bool(findings), "findings": findings, "message": "; ".join(attempts[:10])}
|
||||
tried += 1
|
||||
if debug:
|
||||
print(f" BEDROCK MGMT TRY: region={region}, model={model_id}")
|
||||
item = {"region": region, "model": model_id, "use_case": use_case, "profiles": profiles}
|
||||
try:
|
||||
item["foundation_model"] = bedrock.get_foundation_model(modelIdentifier=model_id).get("modelDetails", {})
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
item["foundation_model_error"] = {"code": client_error_code(exc), "message": client_error_message(exc)[:300]}
|
||||
try:
|
||||
item["availability"] = bedrock.get_foundation_model_availability(modelId=model_id)
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
item["availability_error"] = {"code": client_error_code(exc), "message": client_error_message(exc)[:300]}
|
||||
try:
|
||||
item["agreement_offers"] = bedrock.list_foundation_model_agreement_offers(modelId=model_id, offerType="ALL")
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
item["agreement_offers_error"] = {"code": client_error_code(exc), "message": client_error_message(exc)[:300]}
|
||||
item["iam_simulation"] = simulate_bedrock_activation(session, arn, account, region, model_id, proxy, timeout)
|
||||
availability = item.get("availability") or {}
|
||||
simulation = item.get("iam_simulation") or {}
|
||||
can_activate = bool(simulation.get("can_activate"))
|
||||
authorized = str(availability.get("authorizationStatus") or "").lower() in ("authorized", "available")
|
||||
if can_activate or authorized:
|
||||
findings.append(item)
|
||||
else:
|
||||
code = (item.get("availability_error") or item.get("foundation_model_error") or {}).get("code") or "checked"
|
||||
attempts.append(f"{region}:{model_id}:can_activate={can_activate}:authorization={availability.get('authorizationStatus') or code}")
|
||||
return {"enabled": bool(findings), "findings": findings, "message": "; ".join(attempts[:10])}
|
||||
|
||||
|
||||
def check_bedrock(session, proxy=None, timeout=20, debug=False, regions=None, models=None, max_attempts=12, live_invoke=False):
|
||||
import json
|
||||
import botocore.exceptions
|
||||
|
||||
attempts = []
|
||||
accepted = []
|
||||
tried = 0
|
||||
model_ids = models or list(BEDROCK_MODEL_TESTS.keys())
|
||||
for region in (regions or BEDROCK_REGIONS):
|
||||
for model_id in model_ids:
|
||||
if max_attempts and tried >= max_attempts:
|
||||
if accepted:
|
||||
first = accepted[0]
|
||||
return {
|
||||
"enabled": True,
|
||||
"region": first.get("region", ""),
|
||||
"model": first.get("model", ""),
|
||||
"available_models": [f"{item['region']}/{item['model']}" for item in accepted],
|
||||
"message": "Bedrock InvokeModel accepted",
|
||||
}
|
||||
return {"enabled": False, "region": "", "model": "", "available_models": [], "message": "; ".join(attempts[:10]) or "Bedrock probe attempt limit reached"}
|
||||
tried += 1
|
||||
data = BEDROCK_MODEL_TESTS.get(model_id)
|
||||
if data is None:
|
||||
data = ANTHROPIC_MESSAGES_PROBE
|
||||
if live_invoke and data is ANTHROPIC_MESSAGES_PROBE:
|
||||
data = ANTHROPIC_MESSAGES_LIVE_PING
|
||||
client = aws_client(session, "bedrock-runtime", proxy, region, timeout)
|
||||
if debug:
|
||||
print(f" BEDROCK TRY: region={region}, model={model_id}")
|
||||
try:
|
||||
client.invoke_model(body=json.dumps(data), modelId=model_id)
|
||||
if debug:
|
||||
print(" BEDROCK RESULT: invoke_model succeeded")
|
||||
accepted.append({"region": region, "model": model_id, "message": "invoke_model succeeded"})
|
||||
continue
|
||||
except client.exceptions.ValidationException as exc:
|
||||
message = str(exc)
|
||||
if bedrock_validation_allows_invoke(exc):
|
||||
# ValidationException for the intentional bad payload means auth/model access passed.
|
||||
if debug:
|
||||
print(f" BEDROCK RESULT: validation_exception_after_auth: {message[:200]}")
|
||||
accepted.append({"region": region, "model": model_id, "message": message[:300]})
|
||||
else:
|
||||
if debug:
|
||||
print(f" BEDROCK RESULT: validation_rejected: {message[:200]}")
|
||||
attempts.append(f"{region}:{model_id}:validation:{message[:120]}")
|
||||
continue
|
||||
except client.exceptions.AccessDeniedException:
|
||||
if debug:
|
||||
print(" BEDROCK RESULT: access_denied")
|
||||
attempts.append(f"{region}:{model_id}:access_denied")
|
||||
continue
|
||||
except client.exceptions.ResourceNotFoundException:
|
||||
if debug:
|
||||
print(" BEDROCK RESULT: model_not_found")
|
||||
attempts.append(f"{region}:{model_id}:not_found")
|
||||
continue
|
||||
except botocore.exceptions.EndpointConnectionError as exc:
|
||||
if debug:
|
||||
print(f" BEDROCK RESULT: network_error: {str(exc)[:120]}")
|
||||
attempts.append(f"{region}:{model_id}:network:{str(exc)[:80]}")
|
||||
continue
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
code = exc.response.get("Error", {}).get("Code", "ClientError")
|
||||
if debug:
|
||||
print(f" BEDROCK RESULT: {code}: {str(exc)[:160]}")
|
||||
attempts.append(f"{region}:{model_id}:{code}")
|
||||
continue
|
||||
if accepted:
|
||||
first = accepted[0]
|
||||
return {
|
||||
"enabled": True,
|
||||
"region": first.get("region", ""),
|
||||
"model": first.get("model", ""),
|
||||
"available_models": [f"{item['region']}/{item['model']}" for item in accepted],
|
||||
"message": "Bedrock InvokeModel accepted",
|
||||
}
|
||||
return {"enabled": False, "region": "", "model": "", "available_models": [], "message": "; ".join(attempts[:10])}
|
||||
|
||||
|
||||
def inspect_iam(session, arn, proxy=None, timeout=20):
|
||||
import botocore.exceptions
|
||||
|
||||
output = {"admin": False, "quarantined": False, "policy_check": "not_checked", "message": ""}
|
||||
if ":user/" not in arn:
|
||||
output["policy_check"] = "not_user_arn"
|
||||
return output
|
||||
username = arn.rsplit("/", 1)[1]
|
||||
try:
|
||||
iam = aws_client(session, "iam", proxy, timeout=timeout)
|
||||
policies = iam.list_attached_user_policies(UserName=username).get("AttachedPolicies", [])
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
code = exc.response.get("Error", {}).get("Code", "")
|
||||
output["policy_check"] = "access_denied" if code == "AccessDenied" else "error"
|
||||
output["message"] = str(exc)
|
||||
return output
|
||||
output["policy_check"] = "ok"
|
||||
for policy in policies:
|
||||
name = policy.get("PolicyName", "")
|
||||
if "AWSCompromisedKeyQuarantine" in name:
|
||||
output["quarantined"] = True
|
||||
if name == "AdministratorAccess":
|
||||
output["admin"] = True
|
||||
return output
|
||||
|
||||
|
||||
def check_key(
|
||||
key,
|
||||
probe_bedrock=False,
|
||||
bedrock_debug=False,
|
||||
proxy=None,
|
||||
timeout=20,
|
||||
bedrock_regions=None,
|
||||
bedrock_models=None,
|
||||
bedrock_max_attempts=12,
|
||||
bedrock_live_invoke=False,
|
||||
probe_bedrock_management=False,
|
||||
):
|
||||
try:
|
||||
import boto3
|
||||
import botocore.exceptions
|
||||
except ImportError as exc:
|
||||
return {"status": "UNKNOWN", "message": f"boto3/botocore missing: {exc}"}
|
||||
|
||||
access_key, secret = key.split(":", 1)
|
||||
session = boto3.Session(aws_access_key_id=access_key, aws_secret_access_key=secret)
|
||||
try:
|
||||
identity = aws_client(session, "sts", proxy, timeout=timeout).get_caller_identity()
|
||||
except botocore.exceptions.EndpointConnectionError as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)}
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
code = exc.response.get("Error", {}).get("Code", "")
|
||||
status = "DEAD" if is_dead_aws_error(code) else "ACCESS_DENIED"
|
||||
return {"status": status, "code": code, "message": str(exc)}
|
||||
except Exception as exc:
|
||||
return {"status": "UNKNOWN", "message": str(exc)}
|
||||
|
||||
arn = identity.get("Arn", "")
|
||||
if is_canary_arn(arn):
|
||||
return {
|
||||
"status": "CANARY",
|
||||
"account": identity.get("Account", ""),
|
||||
"arn": arn,
|
||||
"admin": False,
|
||||
"quarantined": False,
|
||||
"iam_policy_check": "skipped_canary",
|
||||
"bedrock_enabled": False,
|
||||
"bedrock_region": "",
|
||||
"bedrock_model": "",
|
||||
"bedrock_message": "skipped_canary",
|
||||
"bedrock_management_enabled": False,
|
||||
"bedrock_management_message": "skipped_canary",
|
||||
"message": "canary credential detected from STS arn; skipped IAM/Bedrock probes",
|
||||
}
|
||||
|
||||
iam_info = inspect_iam(session, arn, proxy, timeout)
|
||||
bedrock_info = {"enabled": False, "region": "", "model": "", "message": "not_checked"}
|
||||
bedrock_management_info = {"enabled": False, "findings": [], "message": "not_checked"}
|
||||
if probe_bedrock:
|
||||
bedrock_info = check_bedrock(session, proxy, timeout, bedrock_debug, bedrock_regions, bedrock_models, bedrock_max_attempts, bedrock_live_invoke)
|
||||
if probe_bedrock_management:
|
||||
bedrock_management_info = check_bedrock_management(
|
||||
session,
|
||||
arn,
|
||||
identity.get("Account", ""),
|
||||
proxy,
|
||||
timeout,
|
||||
bedrock_regions,
|
||||
bedrock_models,
|
||||
bedrock_max_attempts,
|
||||
bedrock_debug,
|
||||
)
|
||||
|
||||
if iam_info.get("quarantined"):
|
||||
status = "QUARANTINED"
|
||||
elif bedrock_info.get("enabled"):
|
||||
status = "BEDROCK"
|
||||
else:
|
||||
status = "ADMIN" if iam_info.get("admin") else "VALID"
|
||||
|
||||
message_parts = [
|
||||
"sts_ok",
|
||||
f"iam_policy_check={iam_info.get('policy_check')}",
|
||||
]
|
||||
if probe_bedrock:
|
||||
message_parts.append(f"bedrock_enabled={bedrock_info.get('enabled')}")
|
||||
if bedrock_info.get("region"):
|
||||
message_parts.append(f"bedrock_region={bedrock_info.get('region')}")
|
||||
if bedrock_info.get("model"):
|
||||
message_parts.append(f"bedrock_model={bedrock_info.get('model')}")
|
||||
if probe_bedrock_management:
|
||||
findings = bedrock_management_info.get("findings") or []
|
||||
can_activate = any((item.get("iam_simulation") or {}).get("can_activate") for item in findings)
|
||||
message_parts.append(f"bedrock_mgmt_enabled={bedrock_management_info.get('enabled')}")
|
||||
message_parts.append(f"bedrock_can_activate={can_activate}")
|
||||
if iam_info.get("message") and iam_info.get("policy_check") != "access_denied":
|
||||
message_parts.append(iam_info.get("message")[:300])
|
||||
|
||||
return {
|
||||
"status": status,
|
||||
"account": identity.get("Account", ""),
|
||||
"arn": arn,
|
||||
"admin": iam_info.get("admin", False),
|
||||
"quarantined": iam_info.get("quarantined", False),
|
||||
"iam_policy_check": iam_info.get("policy_check"),
|
||||
"bedrock_enabled": bedrock_info.get("enabled"),
|
||||
"bedrock_region": bedrock_info.get("region"),
|
||||
"bedrock_model": bedrock_info.get("model"),
|
||||
"bedrock_available_models": bedrock_info.get("available_models") or [],
|
||||
"bedrock_message": bedrock_info.get("message"),
|
||||
"bedrock_management_enabled": bedrock_management_info.get("enabled"),
|
||||
"bedrock_management_findings": bedrock_management_info.get("findings") or [],
|
||||
"bedrock_management_message": bedrock_management_info.get("message"),
|
||||
"message": "; ".join(message_parts),
|
||||
}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "AWS")
|
||||
message = result.get("message", "")
|
||||
if result.get("status") == "BEDROCK":
|
||||
models = result.get("bedrock_available_models") or []
|
||||
model_text = ",".join(str(item) for item in models) or f"{result.get('bedrock_region', '')}/{result.get('bedrock_model', '')}".strip("/")
|
||||
message = f"{message}; models={model_text}"
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, result["status"], message, result.get("arn", source),
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, "AWS")
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="AWS key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--probe-bedrock", action="store_true")
|
||||
parser.add_argument("--bedrock-debug", action="store_true")
|
||||
parser.add_argument("--bedrock-regions", default=",".join(BEDROCK_REGIONS))
|
||||
parser.add_argument("--bedrock-models", default=",".join(BEDROCK_MODEL_TESTS))
|
||||
parser.add_argument("--bedrock-max-attempts", type=int, default=12)
|
||||
parser.add_argument("--bedrock-live-invoke", action="store_true")
|
||||
parser.add_argument("--probe-bedrock-management", action="store_true")
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES)
|
||||
known = set(known_statuses)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_valid:
|
||||
retry_statuses.update({"VALID", "BEDROCK", "ADMIN"})
|
||||
processed = 0
|
||||
skipped = 0
|
||||
bedrock_regions = [item.strip() for item in str(args.bedrock_regions or "").split(",") if item.strip()]
|
||||
bedrock_models = [item.strip() for item in str(args.bedrock_models or "").split(",") if item.strip()]
|
||||
for key, source, finding in extract_candidates(args.input, args.plain):
|
||||
if should_skip_key(
|
||||
key, checked, known, args, retry_statuses,
|
||||
service=SERVICE, source=source, finding=finding, detector="AWS", known_statuses=known_statuses,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] AWS candidate {mask_secret(key)} from {source}")
|
||||
if is_canary_finding(finding):
|
||||
result = {
|
||||
"status": "CANARY",
|
||||
"message": "canary credential detected in TruffleHog ExtraData; skipped AWS API probes",
|
||||
}
|
||||
else:
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_key(
|
||||
key,
|
||||
args.probe_bedrock,
|
||||
args.bedrock_debug,
|
||||
proxy,
|
||||
args.timeout,
|
||||
bedrock_regions,
|
||||
bedrock_models,
|
||||
args.bedrock_max_attempts,
|
||||
args.bedrock_live_invoke,
|
||||
args.probe_bedrock_management,
|
||||
)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
if args.probe_bedrock:
|
||||
print(
|
||||
" BEDROCK PING: "
|
||||
f"enabled={result.get('bedrock_enabled')}, "
|
||||
f"region={result.get('bedrock_region') or '-'}, "
|
||||
f"model={result.get('bedrock_model') or '-'}"
|
||||
)
|
||||
if result.get('bedrock_message'):
|
||||
print(f" BEDROCK RESPONSE: {str(result.get('bedrock_message'))[:300]}")
|
||||
if args.probe_bedrock_management:
|
||||
findings = result.get("bedrock_management_findings") or []
|
||||
can_activate = any((item.get("iam_simulation") or {}).get("can_activate") for item in findings)
|
||||
print(
|
||||
" BEDROCK MGMT: "
|
||||
f"enabled={result.get('bedrock_management_enabled')}, "
|
||||
f"can_activate={can_activate}, "
|
||||
f"findings={len(findings)}"
|
||||
)
|
||||
if result.get("bedrock_management_message"):
|
||||
print(f" BEDROCK MGMT RESPONSE: {str(result.get('bedrock_management_message'))[:300]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,946 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))))
|
||||
|
||||
from keycheck_candidates import extract_azure_foundry_parts
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
append_status,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
keycheck_input_mode,
|
||||
iter_findings,
|
||||
iter_bounded_text_lines,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "azure"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "azureChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "azureResults.jsonl")
|
||||
AZURE_OPENAI_LLM_FILE = os.path.join(OUTPUT_DIR, "azureOpenAILLM.txt")
|
||||
AZURE_OPENAI_PLAIN_FILE = os.path.join(OUTPUT_DIR, "azureOpenAI.txt")
|
||||
AZURE_FOUNDRY_PLAIN_FILE = os.path.join(OUTPUT_DIR, "azureFoundry.txt")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "azureAlive.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "azureDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "azureRestricted.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "azureNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "azureUnknown.txt"),
|
||||
"OPENAI_UNRESOLVED": os.path.join(OUTPUT_DIR, "azureOpenAIUnresolved.txt"),
|
||||
"OPENAI_BAD_ENDPOINT": os.path.join(OUTPUT_DIR, "azureOpenAIBadEndpoint.txt"),
|
||||
"FOUNDRY": os.path.join(OUTPUT_DIR, "azureFoundryLLM.txt"),
|
||||
"FOUNDRY_UNRESOLVED": os.path.join(OUTPUT_DIR, "azureFoundryUnresolved.txt"),
|
||||
"FOUNDRY_BAD_ENDPOINT": os.path.join(OUTPUT_DIR, "azureFoundryBadEndpoint.txt"),
|
||||
}
|
||||
|
||||
AZURE_OPENAI_ENDPOINT_RE = re.compile(r"([a-z0-9-]+\.openai\.azure\.com)", re.IGNORECASE)
|
||||
AZURE_FOUNDRY_HOST_RE = r"[a-z0-9-]+(?:\.[a-z0-9-]+)*\.(?:models\.ai\.azure\.com|services\.ai\.azure\.com|inference\.ai\.azure\.com)"
|
||||
AZURE_FOUNDRY_ENDPOINT_RE = re.compile(r"((?:https?://)?" + AZURE_FOUNDRY_HOST_RE + r"(?:/[^\s:\"'<>\\]*)?)", re.IGNORECASE)
|
||||
AZURE_OPENAI_DEPLOYMENTS_API_VERSION = "2023-03-15-preview"
|
||||
AZURE_OPENAI_CHAT_API_VERSION = "2024-02-15-preview"
|
||||
AZURE_FOUNDRY_API_VERSION = "2024-05-01-preview"
|
||||
AZURE_FOUNDRY_KEY_ASSIGNMENT_RE = re.compile(
|
||||
r"(?is)(?:authorization|api[_-]?key|key|token|secret|credential|bearer)[^\n:=]{0,80}[:=]\s*[\"']?(?:bearer\s+)?([A-Za-z0-9_./+=\-]{20,512})"
|
||||
)
|
||||
NON_FOUNDRY_KEY_PREFIXES = (
|
||||
"sk-", "sk_", "sk-or-", "xai-", "ghp_", "gho_", "ghu_", "ghs_", "ghr_", "github_pat_",
|
||||
"glpat-", "glrt-", "hf_", "AIza", "AQ.", "zai-", "gsk_", "r8_", "nvapi-",
|
||||
)
|
||||
|
||||
|
||||
def transaction_status_files():
|
||||
return {**STATUS_FILES, "AUX_OPENAI_LLM": AZURE_OPENAI_LLM_FILE}
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, AZURE_OPENAI_LLM_FILE, AZURE_OPENAI_PLAIN_FILE, AZURE_FOUNDRY_PLAIN_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, transaction_status_files())
|
||||
|
||||
|
||||
def parse_azure_sp(raw_v2):
|
||||
try:
|
||||
data = json.loads(raw_v2)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
client_secret = data.get("clientSecret") or data.get("client_secret")
|
||||
client_id = data.get("clientId") or data.get("client_id")
|
||||
tenant_id = data.get("tenantId") or data.get("tenant_id")
|
||||
if not all([client_secret, client_id, tenant_id]):
|
||||
return None
|
||||
return {"client_secret": client_secret, "client_id": client_id, "tenant_id": tenant_id}
|
||||
|
||||
|
||||
def scanner_context_text(finding):
|
||||
context = finding.get("ScannerContext") if isinstance(finding, dict) else None
|
||||
if isinstance(context, dict):
|
||||
return str(context.get("nearby") or "")
|
||||
return ""
|
||||
|
||||
|
||||
def foundry_keyish(value):
|
||||
text = re.sub(r"(?i)^bearer\s+", "", str(value or "").strip().strip('"\'`,;')).strip()
|
||||
lower = text.lower()
|
||||
if not (20 <= len(text) <= 512):
|
||||
return False
|
||||
if any(marker in lower for marker in ("http://", "https://", "{{", "${", "<", "azure.com")):
|
||||
return False
|
||||
if any(ch.isspace() for ch in text):
|
||||
return False
|
||||
if re.match(r"(?i)^(?:authorization|api[_-]?key|key|token|secret|credential|bearer)\s*[:=]", text):
|
||||
return False
|
||||
if text.startswith(NON_FOUNDRY_KEY_PREFIXES):
|
||||
return False
|
||||
return bool(re.search(r"[A-Za-z]", text) and re.search(r"[0-9]", text))
|
||||
|
||||
|
||||
def normalize_foundry_endpoint(value):
|
||||
text = str(value or "").strip().strip('"\'`,;')
|
||||
if not text:
|
||||
return ""
|
||||
split_text = text if re.match(r"(?i)^https?://", text) else "https://" + text
|
||||
try:
|
||||
parsed = urlsplit(split_text)
|
||||
host = parsed.netloc or parsed.path.split("/", 1)[0]
|
||||
path = parsed.path if parsed.netloc else ("/" + parsed.path.split("/", 1)[1] if "/" in parsed.path else "")
|
||||
except Exception:
|
||||
host, path = re.sub(r"(?i)^https?://", "", text).split("/", 1)[0], ""
|
||||
path = path.rstrip(".,;:)]}/")
|
||||
terminal_routes = (
|
||||
("/models/chat/completions", ""),
|
||||
("/openai/v1/chat/completions", "/openai/v1"),
|
||||
("/v1/chat/completions", "/v1"),
|
||||
("/chat/completions", ""),
|
||||
("/v1/models", "/v1"),
|
||||
("/models", ""),
|
||||
)
|
||||
lower_path = path.lower()
|
||||
for suffix, replacement in terminal_routes:
|
||||
if lower_path.endswith(suffix):
|
||||
path = path[:-len(suffix)] + replacement
|
||||
break
|
||||
return (host + path).strip("/").lower()
|
||||
|
||||
|
||||
def split_foundry_endpoint_key(text):
|
||||
candidate = str(text or "").strip().split("\t", 1)[0].strip()
|
||||
if not candidate:
|
||||
return None
|
||||
endpoint_match = AZURE_FOUNDRY_ENDPOINT_RE.search(candidate)
|
||||
if not endpoint_match:
|
||||
return None
|
||||
endpoint = normalize_foundry_endpoint(endpoint_match.group(1))
|
||||
before = candidate[:endpoint_match.start()].replace("https://", " ").replace("http://", " ").strip(" \t:=,;'\"/")
|
||||
after = candidate[endpoint_match.end():].replace("https://", " ").replace("http://", " ").strip(" \t:=,;'\"/")
|
||||
for key in (after, before):
|
||||
if foundry_keyish(key):
|
||||
return {"key": key, "endpoint": endpoint}
|
||||
return None
|
||||
|
||||
|
||||
def foundry_context_values(finding):
|
||||
values = []
|
||||
for value in (finding.get("Raw"), finding.get("RawV2")) if isinstance(finding, dict) else ():
|
||||
if value:
|
||||
text = str(value)
|
||||
values.append(text)
|
||||
values.extend(str(item) for item in AZURE_FOUNDRY_KEY_ASSIGNMENT_RE.findall(text))
|
||||
context = scanner_context_text(finding)
|
||||
if context:
|
||||
values.append(context)
|
||||
values.extend(str(item) for item in AZURE_FOUNDRY_KEY_ASSIGNMENT_RE.findall(context or ""))
|
||||
extra = finding.get("ExtraData") if isinstance(finding, dict) else None
|
||||
if isinstance(extra, dict):
|
||||
values.extend(str(value) for value in extra.values() if isinstance(value, str))
|
||||
return values
|
||||
|
||||
|
||||
def parse_azure_openai(raw, raw_v2, finding):
|
||||
key = raw or ""
|
||||
endpoint = ""
|
||||
raw_v2 = raw_v2 or ""
|
||||
match = re.match(r"^([a-f0-9]{32}):(.+\.openai\.azure\.com)$", raw_v2, re.IGNORECASE)
|
||||
if match:
|
||||
key = match.group(1)
|
||||
endpoint = match.group(2)
|
||||
if not endpoint:
|
||||
context_match = AZURE_OPENAI_ENDPOINT_RE.search(scanner_context_text(finding))
|
||||
if context_match:
|
||||
endpoint = context_match.group(1)
|
||||
if not key:
|
||||
return None
|
||||
return {"key": key, "endpoint": endpoint}
|
||||
|
||||
|
||||
def parse_azure_openai_line(line):
|
||||
text = str(line or "").strip()
|
||||
if not text:
|
||||
return None
|
||||
text = text.split("\t", 1)[0].strip()
|
||||
if ":" in text:
|
||||
endpoint, key = text.split(":", 1)
|
||||
if AZURE_OPENAI_ENDPOINT_RE.fullmatch(endpoint.strip()) and key.strip():
|
||||
return {"endpoint": endpoint.strip(), "key": key.strip()}
|
||||
endpoint_match = AZURE_OPENAI_ENDPOINT_RE.search(text)
|
||||
key_match = re.search(r"\b[a-f0-9]{32}\b", text, re.IGNORECASE)
|
||||
if endpoint_match and key_match:
|
||||
return {"endpoint": endpoint_match.group(1), "key": key_match.group(0)}
|
||||
if key_match:
|
||||
return {"endpoint": "", "key": key_match.group(0)}
|
||||
return None
|
||||
|
||||
|
||||
def parse_azure_foundry(raw, raw_v2, finding):
|
||||
key = raw or ""
|
||||
endpoint = ""
|
||||
raw_v2 = raw_v2 or ""
|
||||
split = split_foundry_endpoint_key(raw_v2) or split_foundry_endpoint_key(raw)
|
||||
if not split:
|
||||
split = extract_azure_foundry_parts(raw, raw_v2)
|
||||
if split:
|
||||
key = split["key"]
|
||||
endpoint = split["endpoint"]
|
||||
if not endpoint:
|
||||
for endpoint_text in (raw, raw_v2, scanner_context_text(finding)):
|
||||
context_match = AZURE_FOUNDRY_ENDPOINT_RE.search(str(endpoint_text or ""))
|
||||
if context_match:
|
||||
endpoint = normalize_foundry_endpoint(context_match.group(1))
|
||||
break
|
||||
if not foundry_keyish(key):
|
||||
for value in foundry_context_values(finding):
|
||||
if foundry_keyish(value):
|
||||
key = value.strip().strip('"\'`,;')
|
||||
break
|
||||
if not key or not foundry_keyish(key):
|
||||
return None
|
||||
return {"key": key.strip().strip('"\'`,;'), "endpoint": normalize_foundry_endpoint(endpoint)}
|
||||
|
||||
|
||||
def parse_azure_foundry_line(line):
|
||||
parts = str(line or "").strip().split("\t", 1)
|
||||
text = parts[0].strip()
|
||||
if not text:
|
||||
return None
|
||||
split = split_foundry_endpoint_key(text)
|
||||
parsed = split if split else ({"key": text, "endpoint": ""} if foundry_keyish(text) else None)
|
||||
if not parsed:
|
||||
return None
|
||||
if len(parts) > 1:
|
||||
try:
|
||||
metadata = json.loads(parts[1])
|
||||
except ValueError:
|
||||
metadata = {}
|
||||
if isinstance(metadata, dict):
|
||||
parsed["finding_uid"] = metadata.get("finding_uid") or ""
|
||||
parsed["origin"] = metadata.get("origin") or ""
|
||||
return parsed
|
||||
|
||||
|
||||
def parse_azure_acr(raw_v2):
|
||||
try:
|
||||
data = json.loads(raw_v2)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
username = data.get("username")
|
||||
password = data.get("password")
|
||||
if not username or not password:
|
||||
return None
|
||||
return {"username": username, "password": password}
|
||||
|
||||
|
||||
def azure_openai_key(parsed):
|
||||
endpoint = parsed.get("endpoint") or ""
|
||||
return f"{endpoint}:{parsed['key']}" if endpoint else parsed["key"]
|
||||
|
||||
|
||||
def azure_acr_key(parsed):
|
||||
return f"{parsed['username']}:{parsed['password']}"
|
||||
|
||||
|
||||
def azure_sp_key(parsed):
|
||||
return f"{parsed['tenant_id']}:{parsed['client_id']}:{parsed['client_secret']}"
|
||||
|
||||
|
||||
def azure_foundry_key(parsed):
|
||||
endpoint = parsed.get("endpoint") or ""
|
||||
return f"{endpoint}:{parsed['key']}" if endpoint else parsed["key"]
|
||||
|
||||
|
||||
def extract_candidates(input_file):
|
||||
seen_plain = set()
|
||||
foundry_detectors = {"AzureFoundryEndpointBeforeKey", "AzureFoundryKeyBeforeEndpoint"}
|
||||
detector_names = ["AzureOpenAI", "AzureContainerRegistry", "Azure", *sorted(foundry_detectors)]
|
||||
for item in iter_findings(input_file, detector_names):
|
||||
candidate_kind = item.get('candidate_kind') or ''
|
||||
if keycheck_input_mode() == 'postgres' and candidate_kind:
|
||||
secret_text = item.get('credential_secret_text') or ''
|
||||
secret_json = item.get('credential_secret_json') or ''
|
||||
endpoint = item.get('credential_endpoint') or ''
|
||||
parsed = None
|
||||
detector = ''
|
||||
if candidate_kind == 'azure_service_principal':
|
||||
parsed = parse_azure_sp(secret_json)
|
||||
detector = 'Azure'
|
||||
key = azure_sp_key(parsed) if parsed else ''
|
||||
elif candidate_kind == 'azure_container_registry':
|
||||
parsed = parse_azure_acr(secret_json)
|
||||
detector = 'AzureContainerRegistry'
|
||||
key = azure_acr_key(parsed) if parsed else ''
|
||||
elif candidate_kind == 'azure_openai':
|
||||
parsed = {'key': secret_text, 'endpoint': endpoint} if secret_text else None
|
||||
detector = 'AzureOpenAI'
|
||||
key = azure_openai_key(parsed) if parsed else ''
|
||||
elif candidate_kind == 'azure_foundry':
|
||||
parsed = (
|
||||
{'key': secret_text, 'endpoint': normalize_foundry_endpoint(endpoint)}
|
||||
if foundry_keyish(secret_text) and endpoint else
|
||||
parse_azure_foundry(secret_text, '', item.get('finding') or {})
|
||||
)
|
||||
if parsed and endpoint:
|
||||
parsed['endpoint'] = normalize_foundry_endpoint(endpoint)
|
||||
detector = 'AzureFoundry'
|
||||
key = azure_foundry_key(parsed) if parsed else ''
|
||||
else:
|
||||
key = ''
|
||||
if parsed and key:
|
||||
yield key, detector, item['source'], item['finding'], parsed
|
||||
continue
|
||||
unresolved_detector = {
|
||||
'azure_openai': 'AzureOpenAI',
|
||||
'azure_foundry': 'AzureFoundry',
|
||||
'azure_container_registry': 'AzureContainerRegistry',
|
||||
'azure_service_principal': 'Azure',
|
||||
}.get(candidate_kind, 'Azure')
|
||||
unresolved_key = secret_text or secret_json or candidate_kind
|
||||
if unresolved_key:
|
||||
yield unresolved_key, unresolved_detector, item['source'], item['finding'], {
|
||||
'_unresolved_candidate': True,
|
||||
'candidate_kind': candidate_kind,
|
||||
}
|
||||
continue
|
||||
if item["detector"] == "AzureOpenAI":
|
||||
foundry = parse_azure_foundry(item["raw"], item["raw_v2"], item["finding"])
|
||||
if foundry and foundry.get("endpoint"):
|
||||
key = azure_foundry_key(foundry)
|
||||
yield key, "AzureFoundry", item["source"], item["finding"], foundry
|
||||
parsed = parse_azure_openai(item["raw"], item["raw_v2"], item["finding"])
|
||||
if not parsed:
|
||||
continue
|
||||
key = azure_openai_key(parsed)
|
||||
yield key, "AzureOpenAI", item["source"], item["finding"], parsed
|
||||
continue
|
||||
if item["detector"] == "AzureContainerRegistry":
|
||||
parsed = parse_azure_acr(item["raw_v2"])
|
||||
if not parsed:
|
||||
continue
|
||||
key = azure_acr_key(parsed)
|
||||
yield key, "AzureContainerRegistry", item["source"], item["finding"], parsed
|
||||
continue
|
||||
if item["detector"] == "Azure":
|
||||
parsed = parse_azure_sp(item["raw_v2"])
|
||||
if not parsed:
|
||||
continue
|
||||
key = azure_sp_key(parsed)
|
||||
yield key, "Azure", item["source"], item["finding"], parsed
|
||||
continue
|
||||
|
||||
if item["detector"] in foundry_detectors:
|
||||
foundry = parse_azure_foundry(item.get("raw"), item.get("raw_v2"), item.get("finding"))
|
||||
if not foundry or not foundry.get("endpoint"):
|
||||
continue
|
||||
key = azure_foundry_key(foundry)
|
||||
yield key, "AzureFoundry", item["source"], item["finding"], foundry
|
||||
|
||||
if keycheck_input_mode() == 'jsonl' and os.path.exists(AZURE_FOUNDRY_PLAIN_FILE):
|
||||
for line_num, line in enumerate(iter_bounded_text_lines(AZURE_FOUNDRY_PLAIN_FILE), 1):
|
||||
parsed = parse_azure_foundry_line(line)
|
||||
if not parsed:
|
||||
continue
|
||||
key = azure_foundry_key(parsed)
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, "AzureFoundry", f"{AZURE_FOUNDRY_PLAIN_FILE}:{line_num}", {}, parsed
|
||||
|
||||
if keycheck_input_mode() == 'jsonl' and os.path.exists(AZURE_OPENAI_PLAIN_FILE):
|
||||
for line_num, line in enumerate(iter_bounded_text_lines(AZURE_OPENAI_PLAIN_FILE), 1):
|
||||
parsed = parse_azure_openai_line(line)
|
||||
if not parsed:
|
||||
continue
|
||||
key = azure_openai_key(parsed)
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, "AzureOpenAI", f"{AZURE_OPENAI_PLAIN_FILE}:{line_num}", {}, parsed
|
||||
|
||||
|
||||
def permission_matches(action, pattern):
|
||||
action = str(action or "").lower()
|
||||
pattern = str(pattern or "").lower()
|
||||
if pattern == "*":
|
||||
return True
|
||||
if pattern.endswith("/*"):
|
||||
return action.startswith(pattern[:-1])
|
||||
return action == pattern
|
||||
|
||||
|
||||
def has_action(actions, wanted):
|
||||
return any(permission_matches(wanted, action) for action in actions)
|
||||
|
||||
|
||||
def probe_azure_rbac(access_token, proxy, timeout, max_subscriptions=3):
|
||||
if not access_token:
|
||||
return {"azure_rbac_level": "unknown", "message": "rbac_probe=no_access_token"}
|
||||
headers = {"Authorization": f"Bearer {access_token}", "Content-Type": "application/json"}
|
||||
try:
|
||||
response = requests.get(
|
||||
"https://management.azure.com/subscriptions?api-version=2020-01-01",
|
||||
headers=headers,
|
||||
proxies=proxy,
|
||||
timeout=timeout,
|
||||
)
|
||||
except requests.RequestException as exc:
|
||||
return {"azure_rbac_level": "network", "message": f"rbac_probe_network={str(exc)[:200]}"}
|
||||
if response.status_code == 403:
|
||||
return {"azure_rbac_level": "token_only", "azure_subscription_count": 0, "message": "rbac_probe=subscriptions_forbidden"}
|
||||
if response.status_code >= 400:
|
||||
return {"azure_rbac_level": "unknown", "azure_rbac_http_status": response.status_code, "message": f"rbac_probe_http={response.status_code}:{request_error_message(response)[:200]}"}
|
||||
payload = response.json()
|
||||
subscriptions = payload.get("value") if isinstance(payload, dict) else []
|
||||
subscriptions = subscriptions or []
|
||||
sub_ids = [item.get("subscriptionId") for item in subscriptions if isinstance(item, dict) and item.get("subscriptionId")]
|
||||
if not sub_ids:
|
||||
return {"azure_rbac_level": "token_only", "azure_subscription_count": 0, "message": "rbac_probe=no_subscriptions"}
|
||||
|
||||
all_actions = set()
|
||||
all_not_actions = set()
|
||||
permission_errors = []
|
||||
for sub_id in sub_ids[:max_subscriptions]:
|
||||
url = f"https://management.azure.com/subscriptions/{sub_id}/providers/Microsoft.Authorization/permissions?api-version=2022-04-01"
|
||||
try:
|
||||
perms_response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
permission_errors.append(f"{sub_id}:network:{str(exc)[:120]}")
|
||||
continue
|
||||
if perms_response.status_code >= 400:
|
||||
permission_errors.append(f"{sub_id}:http_{perms_response.status_code}:{request_error_message(perms_response)[:120]}")
|
||||
continue
|
||||
data = perms_response.json().get("value") or []
|
||||
for item in data:
|
||||
for action in item.get("actions") or []:
|
||||
all_actions.add(str(action))
|
||||
for action in item.get("notActions") or []:
|
||||
all_not_actions.add(str(action))
|
||||
|
||||
can_all = has_action(all_actions, "*")
|
||||
can_assign_roles = has_action(all_actions, "Microsoft.Authorization/roleAssignments/write") and not has_action(all_not_actions, "Microsoft.Authorization/roleAssignments/write")
|
||||
can_manage_cognitive = any(
|
||||
has_action(all_actions, action) for action in (
|
||||
"Microsoft.CognitiveServices/accounts/write",
|
||||
"Microsoft.CognitiveServices/accounts/deployments/write",
|
||||
"Microsoft.CognitiveServices/*",
|
||||
)
|
||||
) or can_all
|
||||
can_manage_ml = any(
|
||||
has_action(all_actions, action) for action in (
|
||||
"Microsoft.MachineLearningServices/workspaces/write",
|
||||
"Microsoft.MachineLearningServices/*",
|
||||
)
|
||||
) or can_all
|
||||
can_deploy_resources = has_action(all_actions, "Microsoft.Resources/deployments/write") or can_all
|
||||
can_manage_ai = can_manage_cognitive or can_manage_ml
|
||||
|
||||
if can_all and can_assign_roles:
|
||||
level = "owner_like"
|
||||
elif can_all:
|
||||
level = "contributor_like"
|
||||
elif can_manage_ai:
|
||||
level = "ai_manager"
|
||||
elif all_actions:
|
||||
level = "limited"
|
||||
else:
|
||||
level = "subscriptions_visible"
|
||||
|
||||
message = (
|
||||
f"rbac_probe={level}; subscriptions={len(sub_ids)}; "
|
||||
f"can_manage_ai={can_manage_ai}; can_assign_roles={can_assign_roles}; can_deploy_resources={can_deploy_resources}"
|
||||
)
|
||||
if permission_errors and not all_actions:
|
||||
message += "; permission_errors=" + " | ".join(permission_errors[:3])
|
||||
return {
|
||||
"azure_rbac_level": level,
|
||||
"azure_subscription_count": len(sub_ids),
|
||||
"azure_subscription_ids": sub_ids[:10],
|
||||
"azure_can_manage_ai": can_manage_ai,
|
||||
"azure_can_manage_cognitive": can_manage_cognitive,
|
||||
"azure_can_manage_ml": can_manage_ml,
|
||||
"azure_can_assign_roles": can_assign_roles,
|
||||
"azure_can_deploy_resources": can_deploy_resources,
|
||||
"azure_permission_actions_sample": sorted(all_actions)[:40],
|
||||
"message": message,
|
||||
}
|
||||
|
||||
|
||||
def check_service_principal(parsed, proxy, timeout):
|
||||
url = f"https://login.microsoftonline.com/{parsed['tenant_id']}/oauth2/v2.0/token"
|
||||
payload = {
|
||||
"client_id": parsed["client_id"],
|
||||
"client_secret": parsed["client_secret"],
|
||||
"scope": "https://management.azure.com/.default",
|
||||
"grant_type": "client_credentials",
|
||||
}
|
||||
try:
|
||||
response = requests.post(url, data=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)}
|
||||
if response.status_code == 200:
|
||||
data = response.json()
|
||||
rbac = probe_azure_rbac(data.get("access_token"), proxy, timeout)
|
||||
message = "token issued"
|
||||
if rbac.get("message"):
|
||||
message = f"{message}; {rbac.get('message')}"
|
||||
return {
|
||||
"status": "VALID",
|
||||
"tenant_id": parsed["tenant_id"],
|
||||
"client_id": parsed["client_id"],
|
||||
"expires_in": data.get("expires_in"),
|
||||
"message": message,
|
||||
**rbac,
|
||||
}
|
||||
message = request_error_message(response)
|
||||
lower = message.lower()
|
||||
if response.status_code in (400, 401) and ("invalid_client" in lower or "invalid_grant" in lower):
|
||||
return {"status": "DEAD", "http_status": response.status_code, "message": message}
|
||||
if response.status_code in (401, 403):
|
||||
return {"status": "RESTRICTED", "http_status": response.status_code, "message": message}
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message}
|
||||
|
||||
|
||||
def azure_openai_deployment_ids(payload):
|
||||
data = payload.get("data") if isinstance(payload, dict) else None
|
||||
if data is None and isinstance(payload, dict):
|
||||
data = payload.get("value")
|
||||
deployments = []
|
||||
for item in data or []:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
deployment_id = item.get("id") or item.get("name")
|
||||
model = item.get("model") or item.get("modelName") or ""
|
||||
if deployment_id:
|
||||
deployments.append({"id": deployment_id, "model": model})
|
||||
return deployments
|
||||
|
||||
|
||||
def endpoint_url(endpoint, path):
|
||||
endpoint = str(endpoint or "").strip().rstrip("/")
|
||||
if not endpoint.startswith("http://") and not endpoint.startswith("https://"):
|
||||
endpoint = "https://" + endpoint
|
||||
path = "/" + str(path or "").lstrip("/")
|
||||
parsed = urlsplit(endpoint)
|
||||
endpoint_path = parsed.path.rstrip("/")
|
||||
if endpoint_path and path.lower().startswith(endpoint_path.lower() + "/"):
|
||||
path = path[len(endpoint_path):]
|
||||
return endpoint + path
|
||||
|
||||
|
||||
def azure_model_ids(payload):
|
||||
data = payload.get("data") if isinstance(payload, dict) else payload if isinstance(payload, list) else []
|
||||
if data is None and isinstance(payload, dict):
|
||||
data = payload.get("value") or payload.get("models")
|
||||
models = []
|
||||
for item in data or []:
|
||||
if isinstance(item, str):
|
||||
models.append(item)
|
||||
elif isinstance(item, dict):
|
||||
model_id = item.get("id") or item.get("name") or item.get("model") or item.get("modelName")
|
||||
if model_id:
|
||||
models.append(str(model_id))
|
||||
return models
|
||||
|
||||
|
||||
def endpoint_failure_status(error_text, status):
|
||||
lower = str(error_text or "").lower()
|
||||
if any(item in lower for item in (
|
||||
"name resolution", "no such host", "failed to resolve", "getaddrinfo",
|
||||
"unexpected_eof", "eof occurred in violation of protocol", "ssleoferror",
|
||||
)):
|
||||
return status
|
||||
return "NETWORK"
|
||||
|
||||
|
||||
def probe_azure_openai_chat_route(endpoint, key, deployments, proxy, timeout):
|
||||
if not deployments:
|
||||
return {"route_probe": "no_deployments"}
|
||||
preferred = None
|
||||
for item in deployments:
|
||||
text = f"{item.get('id', '')} {item.get('model', '')}".lower()
|
||||
if any(marker in text for marker in ("gpt", "chat", "turbo", "4o")):
|
||||
preferred = item
|
||||
break
|
||||
deployment = preferred or deployments[0]
|
||||
deployment_id = deployment["id"]
|
||||
url = f"https://{endpoint}/openai/deployments/{deployment_id}/chat/completions?api-version={AZURE_OPENAI_CHAT_API_VERSION}"
|
||||
headers = {"api-key": key, "Content-Type": "application/json"}
|
||||
# Empty messages should fail validation after auth/deployment routing, without generating content.
|
||||
payload = {"messages": [], "max_tokens": 1}
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"route_probe": "network", "route_deployment": deployment_id, "route_message": str(exc)[:500]}
|
||||
message = request_error_message(response)
|
||||
if response.status_code in (200, 400):
|
||||
return {
|
||||
"route_probe": "accepted_auth_route",
|
||||
"route_deployment": deployment_id,
|
||||
"route_model": deployment.get("model", ""),
|
||||
"route_http_status": response.status_code,
|
||||
"route_message": message,
|
||||
}
|
||||
if response.status_code in (401, 403):
|
||||
return {"route_probe": "auth_failed", "route_deployment": deployment_id, "route_http_status": response.status_code, "route_message": message}
|
||||
if response.status_code == 404:
|
||||
return {"route_probe": "not_found", "route_deployment": deployment_id, "route_http_status": 404, "route_message": message}
|
||||
return {"route_probe": "unknown", "route_deployment": deployment_id, "route_http_status": response.status_code, "route_message": message}
|
||||
|
||||
|
||||
def check_azure_openai(parsed, proxy, timeout, probe_openai_route=False):
|
||||
endpoint = (parsed.get("endpoint") or "").strip().strip("/")
|
||||
key = parsed.get("key")
|
||||
if not endpoint:
|
||||
return {"status": "OPENAI_UNRESOLVED", "message": "AzureOpenAI key found without endpoint/resource name"}
|
||||
url = f"https://{endpoint}/openai/deployments?api-version={AZURE_OPENAI_DEPLOYMENTS_API_VERSION}"
|
||||
headers = {"api-key": key, "Content-Type": "application/json"}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
first_error = str(exc)
|
||||
def endpoint_failure_status(error_text):
|
||||
lower = str(error_text or "").lower()
|
||||
if any(item in lower for item in (
|
||||
"name resolution", "no such host", "failed to resolve", "getaddrinfo",
|
||||
"unexpected_eof", "eof occurred in violation of protocol", "ssleoferror",
|
||||
)):
|
||||
return {"status": "OPENAI_BAD_ENDPOINT", "endpoint": endpoint, "message": error_text}
|
||||
return None
|
||||
endpoint_status = endpoint_failure_status(first_error)
|
||||
if endpoint_status:
|
||||
return endpoint_status
|
||||
return {"status": "NETWORK", "endpoint": endpoint, "message": first_error}
|
||||
if response.status_code == 200:
|
||||
deployments = azure_openai_deployment_ids(response.json())
|
||||
result = {
|
||||
"status": "VALID",
|
||||
"endpoint": endpoint,
|
||||
"deployment_count": len(deployments),
|
||||
"deployments": [item.get("id") for item in deployments[:20]],
|
||||
"message": f"deployments endpoint accepted key; deployments={len(deployments)}",
|
||||
}
|
||||
if probe_openai_route:
|
||||
result.update(probe_azure_openai_chat_route(endpoint, key, deployments, proxy, timeout))
|
||||
return result
|
||||
message = request_error_message(response)
|
||||
if response.status_code in (401, 403):
|
||||
return {"status": "DEAD", "endpoint": endpoint, "http_status": response.status_code, "message": message}
|
||||
if response.status_code == 404:
|
||||
return {"status": "UNKNOWN", "endpoint": endpoint, "http_status": 404, "message": message}
|
||||
if response.status_code >= 500:
|
||||
return {"status": "NETWORK", "endpoint": endpoint, "http_status": response.status_code, "message": message}
|
||||
return {"status": "UNKNOWN", "endpoint": endpoint, "http_status": response.status_code, "message": message}
|
||||
|
||||
|
||||
def foundry_model_routes(endpoint):
|
||||
return [
|
||||
endpoint_url(endpoint, f"/models?api-version={AZURE_FOUNDRY_API_VERSION}"),
|
||||
endpoint_url(endpoint, "/models"),
|
||||
endpoint_url(endpoint, "/v1/models"),
|
||||
]
|
||||
|
||||
|
||||
def foundry_auth_headers(key):
|
||||
return [
|
||||
{"api-key": key, "Content-Type": "application/json"},
|
||||
{"Authorization": f"Bearer {key}", "Content-Type": "application/json"},
|
||||
]
|
||||
|
||||
|
||||
def check_foundry_models(endpoint, key, proxy, timeout):
|
||||
attempts = []
|
||||
auth_failures = 0
|
||||
attempted = 0
|
||||
for url in foundry_model_routes(endpoint):
|
||||
for headers in foundry_auth_headers(key):
|
||||
attempted += 1
|
||||
auth_kind = "bearer" if "Authorization" in headers else "api-key"
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
status = endpoint_failure_status(str(exc), "FOUNDRY_BAD_ENDPOINT")
|
||||
attempts.append(f"{url}:{auth_kind}:network:{str(exc)[:180]}")
|
||||
if status == "FOUNDRY_BAD_ENDPOINT":
|
||||
return {"status": status, "endpoint": endpoint, "message": str(exc)}
|
||||
continue
|
||||
message = request_error_message(response)
|
||||
if response.status_code == 200:
|
||||
models = azure_model_ids(response.json())
|
||||
return {
|
||||
"status": "FOUNDRY",
|
||||
"endpoint": endpoint,
|
||||
"auth_scheme": auth_kind,
|
||||
"model_count": len(models),
|
||||
"models": models[:50],
|
||||
"message": f"models endpoint accepted key; auth={auth_kind}; models={len(models)}",
|
||||
}
|
||||
if response.status_code == 429:
|
||||
return {
|
||||
"status": "FOUNDRY",
|
||||
"endpoint": endpoint,
|
||||
"auth_scheme": auth_kind,
|
||||
"model_count": 0,
|
||||
"models": [],
|
||||
"message": f"models endpoint rate limited after auth; auth={auth_kind}; {message[:200]}",
|
||||
}
|
||||
if response.status_code in (401, 403):
|
||||
auth_failures += 1
|
||||
attempts.append(f"{url}:{auth_kind}:auth_{response.status_code}:{message[:160]}")
|
||||
continue
|
||||
if response.status_code == 404:
|
||||
attempts.append(f"{url}:{auth_kind}:http_404:{message[:160]}")
|
||||
continue
|
||||
if response.status_code >= 500:
|
||||
attempts.append(f"{url}:{auth_kind}:server_{response.status_code}:{message[:160]}")
|
||||
continue
|
||||
attempts.append(f"{url}:{auth_kind}:http_{response.status_code}:{message[:160]}")
|
||||
if attempted and auth_failures == attempted:
|
||||
return {"status": "DEAD", "endpoint": endpoint, "message": "; ".join(attempts[:4])}
|
||||
return {"status": "UNKNOWN", "endpoint": endpoint, "message": "; ".join(attempts[:4])}
|
||||
|
||||
|
||||
def probe_foundry_route(endpoint, key, models, proxy, timeout):
|
||||
configured = [item.strip() for item in (models or []) if item.strip()]
|
||||
if not configured:
|
||||
return {"foundry_route_probe": "not_configured"}
|
||||
attempts = []
|
||||
accepted = []
|
||||
for model in configured:
|
||||
route_specs = [
|
||||
(endpoint_url(endpoint, f"/models/chat/completions?api-version={AZURE_FOUNDRY_API_VERSION}"), {"model": model, "messages": [], "max_tokens": 1}),
|
||||
(endpoint_url(endpoint, "/chat/completions"), {"model": model, "messages": [], "max_tokens": 1}),
|
||||
(endpoint_url(endpoint, "/v1/chat/completions"), {"model": model, "messages": [], "max_tokens": 1}),
|
||||
(endpoint_url(endpoint, "/openai/v1/chat/completions"), {"model": model, "messages": [], "max_tokens": 1}),
|
||||
]
|
||||
for url, payload in route_specs:
|
||||
for headers in foundry_auth_headers(key):
|
||||
auth_kind = "bearer" if "Authorization" in headers else "api-key"
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
attempts.append(f"{model}:{auth_kind}:network:{str(exc)[:120]}")
|
||||
continue
|
||||
message = request_error_message(response)
|
||||
if response.status_code == 200:
|
||||
accepted.append(model)
|
||||
break
|
||||
if response.status_code == 400 and any(item in message.lower() for item in ("messages", "content", "validation", "empty")):
|
||||
accepted.append(model)
|
||||
break
|
||||
if response.status_code == 429:
|
||||
accepted.append(model)
|
||||
break
|
||||
if response.status_code in (401, 403, 404):
|
||||
attempts.append(f"{model}:{auth_kind}:http_{response.status_code}:{message[:160]}")
|
||||
continue
|
||||
attempts.append(f"{model}:{auth_kind}:http_{response.status_code}:{message[:160]}")
|
||||
if model in accepted:
|
||||
break
|
||||
if accepted:
|
||||
return {"foundry_route_probe": "accepted", "foundry_route_models": accepted, "foundry_route_message": "route accepted"}
|
||||
return {"foundry_route_probe": "not_accepted", "foundry_route_models": [], "foundry_route_message": "; ".join(attempts[:8])}
|
||||
|
||||
|
||||
def check_azure_foundry(parsed, proxy, timeout, probe_foundry_route_enabled=False, foundry_models=None):
|
||||
endpoint = (parsed.get("endpoint") or "").strip().strip("/")
|
||||
key = parsed.get("key")
|
||||
if not endpoint:
|
||||
return {"status": "FOUNDRY_UNRESOLVED", "message": "Azure Foundry key found without endpoint"}
|
||||
result = check_foundry_models(endpoint, key, proxy, timeout)
|
||||
if probe_foundry_route_enabled:
|
||||
route = probe_foundry_route(endpoint, key, foundry_models or [], proxy, timeout)
|
||||
if result.get("status") != "FOUNDRY" and route.get("foundry_route_probe") == "accepted":
|
||||
result = {"status": "FOUNDRY", "endpoint": endpoint, "model_count": 0, "models": [], "message": "route accepted without model-list support"}
|
||||
result.update(route)
|
||||
return result
|
||||
|
||||
|
||||
def is_azure_openai_llm(result):
|
||||
if result.get("status") != "VALID":
|
||||
return False
|
||||
if int(result.get("deployment_count") or 0) <= 0:
|
||||
return False
|
||||
route_probe = result.get("route_probe")
|
||||
if route_probe and route_probe != "accepted_auth_route":
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def append_azure_openai_llm(key, result, source):
|
||||
deployments = result.get("deployments") or []
|
||||
deployment_text = ",".join(str(item) for item in deployments[:20])
|
||||
details = " ".join(part for part in [
|
||||
f"deployments={int(result.get('deployment_count') or 0)}",
|
||||
f"route_probe={result.get('route_probe') or ''}" if result.get("route_probe") else "",
|
||||
f"route_deployment={result.get('route_deployment') or ''}" if result.get("route_deployment") else "",
|
||||
f"route_model={result.get('route_model') or ''}" if result.get("route_model") else "",
|
||||
f"deployment_ids={deployment_text}" if deployment_text else "",
|
||||
] if part)
|
||||
append_status(AZURE_OPENAI_LLM_FILE, key, result.get("status", "VALID"), details, source)
|
||||
|
||||
|
||||
def check_azure_acr(parsed, proxy, timeout):
|
||||
username = parsed["username"]
|
||||
password = parsed["password"]
|
||||
url = f"https://{username}.azurecr.io/v2/"
|
||||
try:
|
||||
response = requests.get(url, auth=(username, password), proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
text = str(exc)
|
||||
if "no such host" in text.lower():
|
||||
return {"status": "DEAD", "registry": username, "message": text}
|
||||
return {"status": "NETWORK", "registry": username, "message": text}
|
||||
if response.status_code == 200:
|
||||
return {"status": "VALID", "registry": username, "message": "ACR /v2 accepted basic auth"}
|
||||
message = request_error_message(response)
|
||||
if response.status_code == 401:
|
||||
return {"status": "DEAD", "registry": username, "http_status": 401, "message": message}
|
||||
if response.status_code == 403:
|
||||
return {"status": "RESTRICTED", "registry": username, "http_status": 403, "message": message}
|
||||
if response.status_code >= 500:
|
||||
return {"status": "NETWORK", "registry": username, "http_status": response.status_code, "message": message}
|
||||
return {"status": "UNKNOWN", "registry": username, "http_status": response.status_code, "message": message}
|
||||
|
||||
|
||||
def write_result(key, detector, result, source, finding):
|
||||
safe_finding = strip_finding_nearby_context(finding)
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, {"detector": detector, **result}, source, safe_finding, detector)
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE,
|
||||
transaction_status_files(),
|
||||
key,
|
||||
result["status"],
|
||||
result.get("message", ""),
|
||||
source,
|
||||
)
|
||||
if detector == "AzureOpenAI" and is_azure_openai_llm(result):
|
||||
append_azure_openai_llm(key, result, source)
|
||||
record_validation_result(SERVICE, key, {"detector": detector, **result}, source, safe_finding, detector)
|
||||
|
||||
|
||||
def strip_finding_nearby_context(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return finding
|
||||
output = dict(finding)
|
||||
context = output.get("ScannerContext")
|
||||
if isinstance(context, dict) and "nearby" in context:
|
||||
output["ScannerContext"] = {key: value for key, value in context.items() if key != "nearby"}
|
||||
return output
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Azure key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--probe-openai-route", action="store_true", help="Probe Azure OpenAI chat route with an invalid no-generation request after deployment listing succeeds")
|
||||
parser.add_argument("--probe-foundry-route", action="store_true", help="Probe Azure Foundry/MaaS chat route for configured models")
|
||||
parser.add_argument("--foundry-models", default="", help="Comma-separated Azure Foundry model IDs to route-probe")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.update({"NETWORK", "FOUNDRY_BAD_ENDPOINT", "OPENAI_BAD_ENDPOINT"})
|
||||
if args.retry_unknown:
|
||||
retry_statuses.update({"UNKNOWN", "FOUNDRY_UNRESOLVED", "OPENAI_UNRESOLVED"})
|
||||
if args.retry_valid:
|
||||
retry_statuses.update({"VALID", "FOUNDRY"})
|
||||
foundry_models = [item.strip() for item in str(args.foundry_models or "").split(",") if item.strip()]
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, detector, source, finding, parsed in extract_candidates(args.input):
|
||||
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=detector):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] {detector} candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
if parsed.get('_unresolved_candidate'):
|
||||
result = {
|
||||
'status': (
|
||||
'FOUNDRY_UNRESOLVED'
|
||||
if detector == 'AzureFoundry' else
|
||||
'OPENAI_UNRESOLVED'
|
||||
if detector == 'AzureOpenAI' else
|
||||
'UNKNOWN'
|
||||
),
|
||||
'message': f"normalized {parsed.get('candidate_kind') or 'azure'} candidate is incomplete",
|
||||
}
|
||||
elif detector == "AzureOpenAI":
|
||||
result = check_azure_openai(parsed, proxy, args.timeout, args.probe_openai_route)
|
||||
elif detector == "AzureFoundry":
|
||||
result = check_azure_foundry(parsed, proxy, args.timeout, args.probe_foundry_route, foundry_models)
|
||||
if parsed.get("finding_uid"):
|
||||
result["finding_uid"] = parsed.get("finding_uid")
|
||||
elif detector == "AzureContainerRegistry":
|
||||
result = check_azure_acr(parsed, proxy, args.timeout)
|
||||
else:
|
||||
result = check_service_principal(parsed, proxy, args.timeout)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, detector, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,292 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
classify_common_http_status,
|
||||
combined_provider_routing_hint,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
provider_routing_database_failed,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from keycheckers.provider_resolution import resolve_provider_key
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "deepseek"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "deepseekChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "deepseekResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "deepseekAlive.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "deepseekNoBalance.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "deepseekDead.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "deepseekLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "deepseekNetwork.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "deepseekNoContext.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "deepseekUnknown.txt"),
|
||||
}
|
||||
|
||||
DEEPSEEK_REGEX = re.compile(r"sk-[a-z0-9]{32}")
|
||||
DEEPSEEK_DETECTOR_NAMES = {"deepseek", "deepseekapikey", "deepseek_api_key"}
|
||||
DEEPSEEK_EXPLICIT_DETECTOR_NAMES = {"deepseekapikey", "deepseek_api_key"}
|
||||
QWEN_EXPLICIT_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope"}
|
||||
KIMI_EXPLICIT_DETECTOR_NAMES = {"kimimoonshot", "moonshotai"}
|
||||
QWEN_CONTEXT_REGEX = re.compile(
|
||||
r"(?:DASHSCOPE_API_KEY|QWEN_API_KEY|dashscope|qwen|model[_-]?studio|bailian)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
DEEPSEEK_CONTEXT_REGEX = re.compile(r"(?:DEEPSEEK_API_KEY|deepseek|api\.deepseek\.com)", re.IGNORECASE)
|
||||
KIMI_CONTEXT_REGEX = re.compile(
|
||||
r"(?:MOONSHOT_API_KEY|KIMI_API_KEY|api\.moonshot\.(?:ai|cn)|platform\.kimi\.(?:ai|com))",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
AMBIGUOUS_PROVIDER_HINT = "ambiguous_qwen_deepseek"
|
||||
AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk"
|
||||
GENERIC_SK_PROVIDERS = {"qwen", "deepseek", "kimi", "zai"}
|
||||
EXPLICIT_ASSIGNMENT_HINT_SOURCE = "explicit_assignment"
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def iter_candidate_decisions(input_file, plain_files):
|
||||
detector_names = ["DeepSeek", "DeepSeekApiKey", "DeepSeek_API_Key", "CustomRegex"]
|
||||
for item in iter_findings(input_file, detector_names):
|
||||
finding = item.get("finding") or {}
|
||||
if not finding_has_deepseek_detector(finding):
|
||||
continue
|
||||
key = item.get("credential_secret_text") or item["raw"]
|
||||
if key and DEEPSEEK_REGEX.fullmatch(key):
|
||||
hint = combined_provider_routing_hint(key, finding_provider_routing_hint(finding))
|
||||
if provider_routing_database_failed():
|
||||
raise RuntimeError("provider routing evidence lookup failed closed")
|
||||
yield key, item["source"], finding, hint
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
for key, source, finding, hint in iter_candidate_decisions(input_file, plain_files):
|
||||
if hint == "deepseek":
|
||||
yield key, source, finding
|
||||
|
||||
|
||||
def route_rejection_result(hint):
|
||||
normalized = str(hint or "missing").strip().lower()
|
||||
return {
|
||||
"status": "NO_CONTEXT",
|
||||
"routing_hint": normalized,
|
||||
"message": f"candidate is not safely attributable to DeepSeek; routing_hint={normalized}",
|
||||
}
|
||||
|
||||
|
||||
def finding_detector_names(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return set()
|
||||
extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {}
|
||||
names = {
|
||||
str(finding.get("DetectorName") or finding.get("detector") or "").strip().lower(),
|
||||
str(extra.get("name") or "").strip().lower(),
|
||||
}
|
||||
return {name for name in names if name}
|
||||
|
||||
|
||||
def finding_has_deepseek_detector(finding):
|
||||
return bool(finding_detector_names(finding) & DEEPSEEK_DETECTOR_NAMES)
|
||||
|
||||
|
||||
def finding_has_explicit_detector(finding, detector_names):
|
||||
return bool(finding_detector_names(finding) & set(detector_names))
|
||||
|
||||
|
||||
def finding_provider_routing_hint(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return ""
|
||||
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
|
||||
persisted_hint = context.get("provider_hint")
|
||||
if (
|
||||
context.get("provider_hint_source") == EXPLICIT_ASSIGNMENT_HINT_SOURCE
|
||||
and persisted_hint in (*GENERIC_SK_PROVIDERS, AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
|
||||
):
|
||||
return persisted_hint
|
||||
parts = []
|
||||
for key in ("nearby", "file"):
|
||||
if context.get(key):
|
||||
parts.append(str(context.get(key)))
|
||||
metadata = finding.get("SourceMetadata") if isinstance(finding.get("SourceMetadata"), dict) else {}
|
||||
data = metadata.get("Data") if isinstance(metadata.get("Data"), dict) else {}
|
||||
for details in data.values():
|
||||
if not isinstance(details, dict):
|
||||
continue
|
||||
for key in ("file", "repository", "repo", "link", "image"):
|
||||
if details.get(key):
|
||||
parts.append(str(details.get(key)))
|
||||
text = "\n".join(parts)
|
||||
evidence = set()
|
||||
if QWEN_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, QWEN_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("qwen")
|
||||
if DEEPSEEK_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, DEEPSEEK_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("deepseek")
|
||||
if KIMI_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, KIMI_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("kimi")
|
||||
if persisted_hint == AMBIGUOUS_PROVIDER_HINT:
|
||||
evidence.update(("qwen", "deepseek"))
|
||||
elif persisted_hint == AMBIGUOUS_GENERIC_SK_HINT:
|
||||
evidence.update(GENERIC_SK_PROVIDERS)
|
||||
elif persisted_hint in GENERIC_SK_PROVIDERS:
|
||||
evidence.add(persisted_hint)
|
||||
if len(evidence) > 1:
|
||||
return AMBIGUOUS_PROVIDER_HINT if evidence == {"qwen", "deepseek"} else AMBIGUOUS_GENERIC_SK_HINT
|
||||
return next(iter(evidence)) if evidence else ""
|
||||
|
||||
|
||||
def finding_has_ambiguous_provider_hint(finding):
|
||||
return finding_provider_routing_hint(finding) in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
|
||||
|
||||
|
||||
def finding_looks_like_qwen_context(finding):
|
||||
return finding_provider_routing_hint(finding) == "qwen"
|
||||
|
||||
|
||||
def check_key(key, proxy, timeout):
|
||||
url = "https://api.deepseek.com/user/balance"
|
||||
headers = {"Authorization": f"Bearer {key}"}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)}
|
||||
|
||||
if response.status_code == 200:
|
||||
data = response.json()
|
||||
balance_infos = data.get("balance_infos", [])
|
||||
total_usd = 0.0
|
||||
for balance in balance_infos:
|
||||
amount = float(balance.get("total_balance", "0") or 0)
|
||||
currency = balance.get("currency", "USD")
|
||||
if currency == "CNY":
|
||||
amount *= 0.14
|
||||
total_usd += amount
|
||||
available = bool(data.get("is_available", False))
|
||||
status = "VALID" if available and total_usd > 0 else "NO_BALANCE"
|
||||
return {
|
||||
"status": status,
|
||||
"authenticated": True,
|
||||
"available": available,
|
||||
"balance_usd": round(total_usd, 4),
|
||||
"message": f"available={available}; balance=${total_usd:.4f}",
|
||||
}
|
||||
|
||||
status = classify_common_http_status(response.status_code)
|
||||
return {"status": status, "http_status": response.status_code, "message": request_error_message(response)}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "DeepSeek")
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), source,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, "DeepSeek")
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="DeepSeek key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
|
||||
if args.retry_no_balance:
|
||||
retry_statuses.add("NO_BALANCE")
|
||||
if args.retry_valid:
|
||||
retry_statuses.add("VALID")
|
||||
|
||||
processed = 0
|
||||
skipped = 0
|
||||
postgres_mode = keycheck_input_mode() == "postgres"
|
||||
for key, source, finding, routing_hint in iter_candidate_decisions(args.input, args.plain):
|
||||
route_rejected = routing_hint != "deepseek"
|
||||
ambiguous_route = routing_hint in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
|
||||
if route_rejected and not postgres_mode:
|
||||
skipped += 1
|
||||
continue
|
||||
if not route_rejected and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector="DeepSeek"):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] DeepSeek candidate {mask_secret(key)} from {source}")
|
||||
if route_rejected and ambiguous_route:
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = resolve_provider_key(
|
||||
key, finding, proxy, args.timeout,
|
||||
hint=routing_hint, origin_service=SERVICE,
|
||||
)
|
||||
elif route_rejected:
|
||||
result = route_rejection_result(routing_hint)
|
||||
else:
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_key(key, proxy, args.timeout)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,240 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "dockerhub"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
PLAIN_FILE = os.path.join(OUTPUT_DIR, "dockerhub.txt")
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "dockerhubChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "dockerhubResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "dockerhubAlive.txt"),
|
||||
"VALID_2FA": os.path.join(OUTPUT_DIR, "dockerhubAlive.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "dockerhubDead.txt"),
|
||||
"NO_USERNAME": os.path.join(OUTPUT_DIR, "dockerhubNoUsername.txt"),
|
||||
"RATE_LIMITED": os.path.join(OUTPUT_DIR, "dockerhubRateLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "dockerhubNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "dockerhubUnknown.txt"),
|
||||
}
|
||||
|
||||
DOCKER_PAT_RE = re.compile(r"\bdckr_pat_[A-Za-z0-9_-]{27}\b")
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *set(STATUS_FILES.values()), PLAIN_FILE])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def candidate_key(username, token):
|
||||
return f"{username}:{token}" if username else token
|
||||
|
||||
|
||||
def parse_username_token(raw, raw_v2, finding):
|
||||
raw = raw or ""
|
||||
raw_v2 = raw_v2 or ""
|
||||
token = ""
|
||||
username = ""
|
||||
|
||||
if ":" in raw_v2:
|
||||
maybe_user, maybe_token = raw_v2.split(":", 1)
|
||||
if DOCKER_PAT_RE.fullmatch(maybe_token):
|
||||
username, token = maybe_user.strip(), maybe_token.strip()
|
||||
if not token:
|
||||
match = DOCKER_PAT_RE.search(raw_v2) or DOCKER_PAT_RE.search(raw)
|
||||
if match:
|
||||
token = match.group(0)
|
||||
|
||||
extra = finding.get("ExtraData") if isinstance(finding, dict) else {}
|
||||
analysis = finding.get("AnalysisInfo") if isinstance(finding, dict) else {}
|
||||
if isinstance(extra, dict):
|
||||
username = username or extra.get("hub_username") or ""
|
||||
if isinstance(analysis, dict):
|
||||
username = username or analysis.get("username") or ""
|
||||
return username, token
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_file):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, ["Dockerhub"]):
|
||||
username, token = parse_username_token(item.get("raw"), item.get("raw_v2"), item.get("finding") or {})
|
||||
if not token:
|
||||
continue
|
||||
key = candidate_key(username, token)
|
||||
yield key, username, token, item["source"], item["finding"]
|
||||
|
||||
if keycheck_input_mode() == 'jsonl' and os.path.exists(plain_file):
|
||||
with open(plain_file, "r", encoding="utf-8") as f:
|
||||
for line_num, line in enumerate(f, 1):
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
username = ""
|
||||
token = ""
|
||||
if ":" in line:
|
||||
maybe_user, rest = line.split(":", 1)
|
||||
match = DOCKER_PAT_RE.search(rest)
|
||||
if match:
|
||||
username, token = maybe_user.strip(), match.group(0)
|
||||
else:
|
||||
match = DOCKER_PAT_RE.search(line)
|
||||
if match:
|
||||
token = match.group(0)
|
||||
if not token:
|
||||
continue
|
||||
key = candidate_key(username, token)
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, username, token, f"{plain_file}:{line_num}", {}
|
||||
|
||||
|
||||
def decode_jwt_payload(jwt_token):
|
||||
try:
|
||||
payload = jwt_token.split(".")[1]
|
||||
payload += "=" * (-len(payload) % 4)
|
||||
return json.loads(base64.urlsafe_b64decode(payload.encode()).decode())
|
||||
except Exception:
|
||||
return {}
|
||||
|
||||
|
||||
def check_token(username, token, proxy, timeout):
|
||||
if not username:
|
||||
return {"status": "NO_USERNAME", "message": "DockerHub PAT requires username/email for login check"}
|
||||
|
||||
url = "https://hub.docker.com/v2/users/login"
|
||||
payload = {"username": username, "password": token}
|
||||
try:
|
||||
response = requests.post(url, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc), "username": username}
|
||||
|
||||
message = request_error_message(response)
|
||||
if response.status_code == 200:
|
||||
data = response.json()
|
||||
hub_token = data.get("token", "")
|
||||
claims = decode_jwt_payload(hub_token) if hub_token else {}
|
||||
hub_claims = claims.get("https://hub.docker.com", {}) if isinstance(claims, dict) else {}
|
||||
return {
|
||||
"status": "VALID",
|
||||
"message": "login accepted",
|
||||
"username": username,
|
||||
"hub_username": hub_claims.get("username", username),
|
||||
"hub_email": hub_claims.get("email", ""),
|
||||
"scope": claims.get("scope", "") if isinstance(claims, dict) else "",
|
||||
}
|
||||
if response.status_code == 401:
|
||||
try:
|
||||
data = response.json()
|
||||
except ValueError:
|
||||
data = {}
|
||||
if data.get("login_2fa_token"):
|
||||
return {"status": "VALID_2FA", "message": "credentials accepted; 2FA required", "username": username}
|
||||
return {"status": "DEAD", "http_status": 401, "message": message, "username": username}
|
||||
if response.status_code == 429:
|
||||
return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "username": username}
|
||||
if response.status_code >= 500:
|
||||
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "username": username}
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "username": username}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "Dockerhub")
|
||||
extra = result.get("hub_username") or result.get("username") or source
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), extra,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, "Dockerhub")
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="DockerHub PAT checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", default=PLAIN_FILE)
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-no-username", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("RATE_LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_no_username:
|
||||
retry_statuses.add("NO_USERNAME")
|
||||
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, username, token, source, finding in extract_candidates(args.input, args.plain):
|
||||
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector="Dockerhub"):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] DockerHub candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_token(username, token, proxy, args.timeout)
|
||||
print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
time.sleep(0.1)
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,789 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import base64
|
||||
import binascii
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
append_status,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_known_statuses,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
record_cached_keycheck_occurrence,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "gcp"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
PLAIN_FILE = os.path.join(OUTPUT_DIR, "gcp.txt")
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "gcpChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "gcpResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "gcpAlive.txt"),
|
||||
"VERTEX": os.path.join(OUTPUT_DIR, "gcpVertex.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "gcpDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "gcpRestricted.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "gcpNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "gcpUnknown.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "gcpNoContext.txt"),
|
||||
}
|
||||
VERTEX_GEMINI_FILE = os.path.join(OUTPUT_DIR, "gcpVertexGemini.txt")
|
||||
VERTEX_ANTHROPIC_FILE = os.path.join(OUTPUT_DIR, "gcpVertexAnthropic.txt")
|
||||
|
||||
GOOGLE_TOKEN_URL = "https://oauth2.googleapis.com/token"
|
||||
TRUSTED_GOOGLE_TOKEN_ENDPOINTS = frozenset({
|
||||
GOOGLE_TOKEN_URL,
|
||||
"https://accounts.google.com/o/oauth2/token",
|
||||
})
|
||||
TOKEN_REDIRECT_STATUSES = {301, 302, 303, 307, 308}
|
||||
SA_SCOPE = "https://www.googleapis.com/auth/cloud-platform"
|
||||
MAX_PEM_BYTES = 24 * 1024
|
||||
MAX_DER_BYTES = 16 * 1024
|
||||
MAX_DER_LENGTH_BYTES = 2
|
||||
MIN_RSA_BITS = 2048
|
||||
MAX_RSA_BITS = 8192
|
||||
MAX_RSA_INTEGER_BYTES = MAX_RSA_BITS // 8
|
||||
RSA_ENCRYPTION_OID = bytes.fromhex("2a864886f70d010101")
|
||||
VERTEX_LOCATIONS = ["global", "us", "eu"]
|
||||
VERTEX_MODELS = ["gemini-3.6-flash", "gemini-3.1-pro-preview"]
|
||||
VERTEX_ANTHROPIC_LOCATIONS = ["global", "us", "eu", "us-east5", "europe-west1"]
|
||||
VERTEX_ANTHROPIC_MODELS = ["claude-opus-5", "claude-opus-4-7", "claude-opus-4-6", "claude-fable-5"]
|
||||
|
||||
|
||||
def transaction_status_files():
|
||||
return {
|
||||
**STATUS_FILES,
|
||||
"RATE_LIMITED": STATUS_FILES["UNKNOWN"],
|
||||
"AUX_VERTEX_GEMINI": VERTEX_GEMINI_FILE,
|
||||
"AUX_VERTEX_ANTHROPIC": VERTEX_ANTHROPIC_FILE,
|
||||
}
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values(), VERTEX_GEMINI_FILE, VERTEX_ANTHROPIC_FILE, PLAIN_FILE])
|
||||
recover_status_transaction(CHECKED_FILE, transaction_status_files())
|
||||
|
||||
|
||||
def b64url(data):
|
||||
return base64.urlsafe_b64encode(data).rstrip(b"=").decode()
|
||||
|
||||
|
||||
def validate_google_token_uri(value):
|
||||
token_uri = str(value or GOOGLE_TOKEN_URL).strip()
|
||||
try:
|
||||
parsed = urlsplit(token_uri)
|
||||
port = parsed.port
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise ValueError("invalid Google OAuth token_uri") from exc
|
||||
if (
|
||||
parsed.scheme != "https"
|
||||
or parsed.username is not None
|
||||
or parsed.password is not None
|
||||
or port not in (None, 443)
|
||||
or parsed.query
|
||||
or parsed.fragment
|
||||
or token_uri not in TRUSTED_GOOGLE_TOKEN_ENDPOINTS
|
||||
):
|
||||
raise ValueError("untrusted Google OAuth token_uri")
|
||||
return token_uri
|
||||
|
||||
|
||||
class InvalidRSAPrivateKey(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
class DERReader:
|
||||
def __init__(self, data):
|
||||
if not isinstance(data, (bytes, bytearray, memoryview)):
|
||||
raise InvalidRSAPrivateKey("DER value is not binary")
|
||||
if len(data) > MAX_DER_BYTES:
|
||||
raise InvalidRSAPrivateKey("DER value exceeds size limit")
|
||||
self.data = data
|
||||
self.pos = 0
|
||||
|
||||
def read_tlv(self):
|
||||
if len(self.data) - self.pos < 2:
|
||||
raise InvalidRSAPrivateKey("truncated DER tag or length")
|
||||
tag = self.data[self.pos]
|
||||
self.pos += 1
|
||||
first_len = self.data[self.pos]
|
||||
self.pos += 1
|
||||
if first_len & 0x80:
|
||||
length_len = first_len & 0x7F
|
||||
if length_len == 0:
|
||||
raise InvalidRSAPrivateKey("indefinite DER length is not allowed")
|
||||
if length_len > MAX_DER_LENGTH_BYTES:
|
||||
raise InvalidRSAPrivateKey("DER length-of-length exceeds limit")
|
||||
if len(self.data) - self.pos < length_len:
|
||||
raise InvalidRSAPrivateKey("truncated DER length")
|
||||
length_bytes = self.data[self.pos:self.pos + length_len]
|
||||
if length_bytes[0] == 0:
|
||||
raise InvalidRSAPrivateKey("non-minimal DER length")
|
||||
length = int.from_bytes(length_bytes, "big")
|
||||
self.pos += length_len
|
||||
if length < 0x80:
|
||||
raise InvalidRSAPrivateKey("non-minimal DER length")
|
||||
else:
|
||||
length = first_len
|
||||
if length > MAX_DER_BYTES:
|
||||
raise InvalidRSAPrivateKey("DER value length exceeds limit")
|
||||
if length > len(self.data) - self.pos:
|
||||
raise InvalidRSAPrivateKey("truncated DER value")
|
||||
value = self.data[self.pos:self.pos + length]
|
||||
self.pos += length
|
||||
return tag, value
|
||||
|
||||
def expect(self, tag):
|
||||
actual, value = self.read_tlv()
|
||||
if actual != tag:
|
||||
raise InvalidRSAPrivateKey(f"expected DER tag {tag:#x}, got {actual:#x}")
|
||||
return value
|
||||
|
||||
def at_end(self):
|
||||
return self.pos == len(self.data)
|
||||
|
||||
def require_eof(self, context="DER structure"):
|
||||
if not self.at_end():
|
||||
raise InvalidRSAPrivateKey(f"trailing data in {context}")
|
||||
|
||||
|
||||
def der_int(value, name="integer", max_bytes=MAX_RSA_INTEGER_BYTES):
|
||||
if not value:
|
||||
raise InvalidRSAPrivateKey(f"empty RSA {name}")
|
||||
if len(value) > max_bytes + 1:
|
||||
raise InvalidRSAPrivateKey(f"RSA {name} exceeds size limit")
|
||||
if value[0] & 0x80:
|
||||
raise InvalidRSAPrivateKey(f"negative RSA {name}")
|
||||
if value[0] == 0:
|
||||
if len(value) > 1 and not value[1] & 0x80:
|
||||
raise InvalidRSAPrivateKey(f"non-minimal RSA {name}")
|
||||
unsigned = value[1:]
|
||||
else:
|
||||
unsigned = value
|
||||
if len(unsigned) > max_bytes:
|
||||
raise InvalidRSAPrivateKey(f"RSA {name} exceeds size limit")
|
||||
return int.from_bytes(unsigned, "big") if unsigned else 0
|
||||
|
||||
|
||||
def parse_pkcs1_rsa_private_key(data):
|
||||
rsa = DERReader(data)
|
||||
version = der_int(rsa.expect(0x02), "version", 1)
|
||||
if version != 0:
|
||||
raise InvalidRSAPrivateKey("unsupported RSA private key version")
|
||||
n = der_int(rsa.expect(0x02), "modulus")
|
||||
public_exponent = der_int(rsa.expect(0x02), "public exponent")
|
||||
d = der_int(rsa.expect(0x02), "private exponent")
|
||||
for name in ("prime1", "prime2", "exponent1", "exponent2", "coefficient"):
|
||||
der_int(rsa.expect(0x02), name)
|
||||
rsa.require_eof("RSA private key")
|
||||
|
||||
modulus_bits = n.bit_length()
|
||||
if not MIN_RSA_BITS <= modulus_bits <= MAX_RSA_BITS:
|
||||
raise InvalidRSAPrivateKey(
|
||||
f"RSA modulus must be between {MIN_RSA_BITS} and {MAX_RSA_BITS} bits"
|
||||
)
|
||||
if public_exponent == 0:
|
||||
raise InvalidRSAPrivateKey("RSA public exponent is zero")
|
||||
if d == 0 or d >= n:
|
||||
raise InvalidRSAPrivateKey("RSA private exponent is out of range")
|
||||
return n, d
|
||||
|
||||
|
||||
def validate_rsa_algorithm_identifier(data):
|
||||
algorithm = DERReader(data)
|
||||
if algorithm.expect(0x06) != RSA_ENCRYPTION_OID:
|
||||
raise InvalidRSAPrivateKey("PKCS#8 key does not use rsaEncryption")
|
||||
if not algorithm.at_end() and algorithm.expect(0x05):
|
||||
raise InvalidRSAPrivateKey("invalid rsaEncryption parameters")
|
||||
algorithm.require_eof("PKCS#8 algorithm identifier")
|
||||
|
||||
|
||||
def parse_rsa_private_key_from_pem(pem):
|
||||
pem = str(pem or "")
|
||||
if len(pem) > MAX_PEM_BYTES:
|
||||
raise InvalidRSAPrivateKey("PEM private key exceeds size limit")
|
||||
try:
|
||||
pem_bytes = pem.encode("utf-8")
|
||||
except UnicodeEncodeError as exc:
|
||||
raise InvalidRSAPrivateKey("PEM private key is not valid UTF-8") from exc
|
||||
if len(pem_bytes) > MAX_PEM_BYTES:
|
||||
raise InvalidRSAPrivateKey("PEM private key exceeds size limit")
|
||||
pem = pem.replace("\\n", "\n")
|
||||
match = re.fullmatch(
|
||||
r"\s*-----BEGIN (RSA PRIVATE KEY|PRIVATE KEY)-----\s*(.*?)\s*-----END \1-----\s*",
|
||||
pem,
|
||||
re.DOTALL,
|
||||
)
|
||||
if not match:
|
||||
raise InvalidRSAPrivateKey("missing complete PEM private key block")
|
||||
body = re.sub(r"\s+", "", match.group(2))
|
||||
if len(body) < 256:
|
||||
raise InvalidRSAPrivateKey("PEM private key body is too short")
|
||||
try:
|
||||
der = base64.b64decode(body + ("=" * (-len(body) % 4)), validate=True)
|
||||
except (binascii.Error, ValueError) as exc:
|
||||
raise InvalidRSAPrivateKey("invalid PEM base64") from exc
|
||||
if len(der) > MAX_DER_BYTES:
|
||||
raise InvalidRSAPrivateKey("DER private key exceeds size limit")
|
||||
|
||||
reader = DERReader(der)
|
||||
top_bytes = reader.expect(0x30)
|
||||
reader.require_eof("DER private key")
|
||||
top = DERReader(top_bytes)
|
||||
|
||||
# PKCS#8 PrivateKeyInfo: SEQUENCE(version, alg, OCTET STRING(RSAPrivateKey))
|
||||
first_tag, first_val = top.read_tlv()
|
||||
if first_tag != 0x02:
|
||||
raise InvalidRSAPrivateKey("unexpected private key structure")
|
||||
second_tag, second_val = top.read_tlv()
|
||||
if second_tag == 0x30:
|
||||
if der_int(first_val, "PKCS#8 version", 1) != 0:
|
||||
raise InvalidRSAPrivateKey("unsupported PKCS#8 version")
|
||||
validate_rsa_algorithm_identifier(second_val)
|
||||
private_octet = top.expect(0x04)
|
||||
top.require_eof("PKCS#8 private key")
|
||||
wrapped = DERReader(private_octet)
|
||||
rsa_bytes = wrapped.expect(0x30)
|
||||
wrapped.require_eof("PKCS#8 private key octets")
|
||||
return parse_pkcs1_rsa_private_key(rsa_bytes)
|
||||
if second_tag != 0x02:
|
||||
raise InvalidRSAPrivateKey("unexpected private key structure")
|
||||
return parse_pkcs1_rsa_private_key(top_bytes)
|
||||
|
||||
|
||||
def rsa_pkcs1v15_sha256_sign(message, pem):
|
||||
n, d = parse_rsa_private_key_from_pem(pem)
|
||||
digest = hashlib.sha256(message).digest()
|
||||
digest_info = bytes.fromhex("3031300d060960864801650304020105000420") + digest
|
||||
key_len = (n.bit_length() + 7) // 8
|
||||
if key_len < len(digest_info) + 11:
|
||||
raise InvalidRSAPrivateKey("RSA key too small")
|
||||
encoded = b"\x00\x01" + b"\xff" * (key_len - len(digest_info) - 3) + b"\x00" + digest_info
|
||||
sig = pow(int.from_bytes(encoded, "big"), d, n).to_bytes(key_len, "big")
|
||||
return sig
|
||||
|
||||
|
||||
def make_service_account_assertion(creds):
|
||||
now = int(time.time())
|
||||
token_uri = validate_google_token_uri(creds.get("token_uri"))
|
||||
header = {"alg": "RS256", "typ": "JWT", "kid": creds.get("private_key_id")}
|
||||
payload = {
|
||||
"iss": creds["client_email"],
|
||||
"scope": SA_SCOPE,
|
||||
"aud": token_uri,
|
||||
"iat": now,
|
||||
"exp": now + 3600,
|
||||
}
|
||||
signing_input = (b64url(json.dumps(header, separators=(",", ":")).encode()) + "." + b64url(json.dumps(payload, separators=(",", ":")).encode())).encode()
|
||||
signature = rsa_pkcs1v15_sha256_sign(signing_input, creds["private_key"])
|
||||
return signing_input.decode() + "." + b64url(signature), token_uri
|
||||
|
||||
|
||||
def compact_json(data):
|
||||
return json.dumps(data, ensure_ascii=False, separators=(",", ":"), sort_keys=True)
|
||||
|
||||
|
||||
def scanner_context_text(finding):
|
||||
context = finding.get("ScannerContext") if isinstance(finding, dict) else None
|
||||
if isinstance(context, dict):
|
||||
return str(context.get("nearby") or "")
|
||||
return ""
|
||||
|
||||
|
||||
def parse_json_object(text):
|
||||
try:
|
||||
return json.loads(text)
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
match = re.search(r"\{.*\}", str(text or ""), re.DOTALL)
|
||||
if not match:
|
||||
return None
|
||||
try:
|
||||
return json.loads(match.group(0))
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
def parse_service_account(raw_v2):
|
||||
data = parse_json_object(raw_v2)
|
||||
if not isinstance(data, dict):
|
||||
return None
|
||||
required = ["client_email", "private_key", "private_key_id"]
|
||||
if not all(data.get(item) for item in required):
|
||||
return None
|
||||
private_key = str(data.get("private_key") or "")
|
||||
if "-----BEGIN" not in private_key or "-----END" not in private_key or len(private_key) < 800:
|
||||
return None
|
||||
return data
|
||||
|
||||
|
||||
def parse_adc(raw_v2, finding):
|
||||
data = parse_json_object(scanner_context_text(finding)) or parse_json_object(raw_v2)
|
||||
if not isinstance(data, dict):
|
||||
return None
|
||||
required = ["client_id", "client_secret", "refresh_token"]
|
||||
if not all(data.get(item) for item in required):
|
||||
return None
|
||||
return data
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_file):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, ["GCP", "GCPApplicationDefaultCredentials"]):
|
||||
if item["detector"] == "GCP":
|
||||
parsed = parse_service_account(item["raw_v2"])
|
||||
detector = "GCP"
|
||||
else:
|
||||
parsed = parse_adc(item["raw_v2"], item["finding"])
|
||||
detector = "GCPApplicationDefaultCredentials"
|
||||
if not parsed:
|
||||
key = f"{detector}:no_context:{item['source']}"
|
||||
yield key, detector, item["source"], item["finding"], None
|
||||
continue
|
||||
key = compact_json(parsed)
|
||||
yield key, detector, item["source"], item["finding"], parsed
|
||||
|
||||
if keycheck_input_mode() == 'jsonl' and os.path.exists(plain_file):
|
||||
with open(plain_file, "r", encoding="utf-8", errors="replace") as f:
|
||||
content = f.read()
|
||||
candidates = []
|
||||
whole_file = parse_json_object(content)
|
||||
if whole_file:
|
||||
candidates.append((plain_file, whole_file))
|
||||
for line_num, line in enumerate(content.splitlines(), 1):
|
||||
key_text = line.split("\t", 1)[0].strip()
|
||||
parsed = parse_json_object(key_text) or parse_json_object(line)
|
||||
if parsed:
|
||||
candidates.append((f"{plain_file}:{line_num}", parsed))
|
||||
for source, parsed in candidates:
|
||||
detector = "GCP" if parsed.get("private_key") else "GCPApplicationDefaultCredentials"
|
||||
key = compact_json(parsed)
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, detector, source, {}, parsed
|
||||
|
||||
|
||||
def vertex_api_host(location):
|
||||
location = str(location or "").strip().lower()
|
||||
if location == "global":
|
||||
return "aiplatform.googleapis.com"
|
||||
if location in ("us", "eu"):
|
||||
return f"aiplatform.{location}.rep.googleapis.com"
|
||||
return f"{location}-aiplatform.googleapis.com"
|
||||
|
||||
|
||||
def probe_vertex_llm(access_token, project_id, proxy, timeout, locations=None, models=None, max_attempts=2):
|
||||
if not project_id:
|
||||
return {"enabled": False, "message": "project_id unavailable"}
|
||||
headers = {"Authorization": f"Bearer {access_token}", "Content-Type": "application/json"}
|
||||
payload = {"contents": [{"role": "user", "parts": [{"text": "ping"}]}]}
|
||||
attempts = []
|
||||
accepted = []
|
||||
tried = 0
|
||||
for location in (locations or VERTEX_LOCATIONS):
|
||||
for model in (models or VERTEX_MODELS):
|
||||
if max_attempts and tried >= max_attempts:
|
||||
if accepted:
|
||||
first = accepted[0]
|
||||
return {
|
||||
"enabled": True,
|
||||
"location": first.get("location", ""),
|
||||
"model": first.get("model", ""),
|
||||
"total_tokens": first.get("total_tokens"),
|
||||
"available_models": [f"google/{item['location']}/{item['model']}" for item in accepted],
|
||||
"message": "Vertex countTokens accepted",
|
||||
}
|
||||
return {"enabled": False, "message": "; ".join(attempts[:8]) or "Vertex probe attempt limit reached"}
|
||||
tried += 1
|
||||
url = f"https://{vertex_api_host(location)}/v1/projects/{project_id}/locations/{location}/publishers/google/models/{model}:countTokens"
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
attempts.append(f"{location}:{model}:network:{str(exc)[:120]}")
|
||||
continue
|
||||
if response.status_code == 200:
|
||||
data = response.json()
|
||||
accepted.append({
|
||||
"location": location,
|
||||
"model": model,
|
||||
"total_tokens": data.get("totalTokens") or data.get("total_tokens"),
|
||||
})
|
||||
continue
|
||||
message = request_error_message(response)
|
||||
if response.status_code in (400, 401, 403, 404, 429):
|
||||
attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}")
|
||||
continue
|
||||
if response.status_code >= 500:
|
||||
attempts.append(f"{location}:{model}:server_{response.status_code}:{message[:160]}")
|
||||
continue
|
||||
attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}")
|
||||
if accepted:
|
||||
first = accepted[0]
|
||||
return {
|
||||
"enabled": True,
|
||||
"location": first.get("location", ""),
|
||||
"model": first.get("model", ""),
|
||||
"total_tokens": first.get("total_tokens"),
|
||||
"available_models": [f"google/{item['location']}/{item['model']}" for item in accepted],
|
||||
"message": "Vertex countTokens accepted",
|
||||
}
|
||||
return {"enabled": False, "message": "; ".join(attempts[:8])}
|
||||
|
||||
|
||||
def probe_vertex_anthropic(access_token, project_id, proxy, timeout, locations=None, models=None, max_attempts=2):
|
||||
if not project_id:
|
||||
return {"enabled": False, "message": "project_id unavailable"}
|
||||
models = models or []
|
||||
if not models:
|
||||
return {"enabled": False, "message": "no Anthropic models configured"}
|
||||
headers = {"Authorization": f"Bearer {access_token}", "Content-Type": "application/json"}
|
||||
payload = {
|
||||
"anthropic_version": "vertex-2023-10-16",
|
||||
"messages": [{"role": "user", "content": "ping"}],
|
||||
"max_tokens": 1,
|
||||
}
|
||||
attempts = []
|
||||
accepted = []
|
||||
tried = 0
|
||||
for location in (locations or VERTEX_ANTHROPIC_LOCATIONS):
|
||||
for model in models:
|
||||
if max_attempts and tried >= max_attempts:
|
||||
if accepted:
|
||||
first = accepted[0]
|
||||
return {
|
||||
"enabled": True,
|
||||
"location": first.get("location", ""),
|
||||
"model": first.get("model", ""),
|
||||
"available_models": [f"anthropic/{item['location']}/{item['model']}" for item in accepted],
|
||||
"message": "Vertex Anthropic rawPredict accepted",
|
||||
}
|
||||
return {"enabled": False, "message": "; ".join(attempts[:8]) or "Vertex Anthropic probe attempt limit reached"}
|
||||
tried += 1
|
||||
url = f"https://{vertex_api_host(location)}/v1/projects/{project_id}/locations/{location}/publishers/anthropic/models/{model}:rawPredict"
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
attempts.append(f"{location}:{model}:network:{str(exc)[:120]}")
|
||||
continue
|
||||
if response.status_code == 200:
|
||||
accepted.append({"location": location, "model": model})
|
||||
continue
|
||||
message = request_error_message(response)
|
||||
if response.status_code in (400, 401, 403, 404, 429):
|
||||
attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}")
|
||||
continue
|
||||
if response.status_code >= 500:
|
||||
attempts.append(f"{location}:{model}:server_{response.status_code}:{message[:160]}")
|
||||
continue
|
||||
attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}")
|
||||
if accepted:
|
||||
first = accepted[0]
|
||||
return {
|
||||
"enabled": True,
|
||||
"location": first.get("location", ""),
|
||||
"model": first.get("model", ""),
|
||||
"available_models": [f"anthropic/{item['location']}/{item['model']}" for item in accepted],
|
||||
"message": "Vertex Anthropic rawPredict accepted",
|
||||
}
|
||||
return {"enabled": False, "message": "; ".join(attempts[:8])}
|
||||
|
||||
|
||||
def merge_vertex_results(google_vertex, anthropic_vertex):
|
||||
google_vertex = google_vertex or {"enabled": False, "message": ""}
|
||||
anthropic_vertex = anthropic_vertex or {"enabled": False, "message": ""}
|
||||
available = []
|
||||
available.extend(google_vertex.get("available_models") or [])
|
||||
available.extend(anthropic_vertex.get("available_models") or [])
|
||||
first = google_vertex if google_vertex.get("enabled") else anthropic_vertex if anthropic_vertex.get("enabled") else {}
|
||||
messages = []
|
||||
if google_vertex.get("message"):
|
||||
messages.append(f"google: {google_vertex.get('message')}")
|
||||
if anthropic_vertex.get("message"):
|
||||
messages.append(f"anthropic: {anthropic_vertex.get('message')}")
|
||||
return {
|
||||
"enabled": bool(available),
|
||||
"location": first.get("location", ""),
|
||||
"model": first.get("model", ""),
|
||||
"total_tokens": first.get("total_tokens"),
|
||||
"available_models": available,
|
||||
"google_enabled": bool(google_vertex.get("enabled")),
|
||||
"anthropic_enabled": bool(anthropic_vertex.get("enabled")),
|
||||
"message": "; ".join(messages),
|
||||
}
|
||||
|
||||
|
||||
def check_service_account(creds, proxy, timeout, probe_vertex=False, vertex_timeout=6, vertex_locations=None, vertex_models=None, vertex_max_attempts=2, vertex_anthropic_locations=None, vertex_anthropic_models=None, vertex_anthropic_max_attempts=2):
|
||||
try:
|
||||
assertion, token_uri = make_service_account_assertion(creds)
|
||||
except InvalidRSAPrivateKey as exc:
|
||||
return {
|
||||
"status": "DEAD",
|
||||
"classification": "invalid_private_key",
|
||||
"message": f"invalid RSA private key: {exc}",
|
||||
"project_id": creds.get("project_id"),
|
||||
"client_email": creds.get("client_email"),
|
||||
}
|
||||
except Exception as exc:
|
||||
return {"status": "UNKNOWN", "message": f"failed to build JWT assertion: {exc}"}
|
||||
data = {"grant_type": "urn:ietf:params:oauth:grant-type:jwt-bearer", "assertion": assertion}
|
||||
try:
|
||||
response = requests.post(
|
||||
token_uri, data=data, proxies=proxy, timeout=timeout, allow_redirects=False,
|
||||
)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc), "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
|
||||
if response.status_code in TOKEN_REDIRECT_STATUSES:
|
||||
return {
|
||||
"status": "UNKNOWN",
|
||||
"http_status": response.status_code,
|
||||
"message": "Google OAuth token endpoint redirect refused",
|
||||
"project_id": creds.get("project_id"),
|
||||
"client_email": creds.get("client_email"),
|
||||
}
|
||||
if response.status_code == 200:
|
||||
payload = response.json()
|
||||
result = {
|
||||
"status": "VALID",
|
||||
"message": "OAuth token issued",
|
||||
"project_id": creds.get("project_id"),
|
||||
"client_email": creds.get("client_email"),
|
||||
"private_key_id": creds.get("private_key_id"),
|
||||
"expires_in": payload.get("expires_in"),
|
||||
}
|
||||
if probe_vertex:
|
||||
google_vertex = probe_vertex_llm(
|
||||
payload.get("access_token"), creds.get("project_id"), proxy,
|
||||
vertex_timeout, vertex_locations, vertex_models, vertex_max_attempts,
|
||||
)
|
||||
anthropic_vertex = probe_vertex_anthropic(
|
||||
payload.get("access_token"), creds.get("project_id"), proxy,
|
||||
vertex_timeout, vertex_anthropic_locations, vertex_anthropic_models, vertex_anthropic_max_attempts,
|
||||
) if vertex_anthropic_models else {"enabled": False, "message": ""}
|
||||
vertex = merge_vertex_results(google_vertex, anthropic_vertex)
|
||||
result.update({
|
||||
"vertex_enabled": vertex.get("enabled"),
|
||||
"vertex_location": vertex.get("location", ""),
|
||||
"vertex_model": vertex.get("model", ""),
|
||||
"vertex_available_models": vertex.get("available_models") or [],
|
||||
"vertex_google_enabled": vertex.get("google_enabled"),
|
||||
"vertex_anthropic_enabled": vertex.get("anthropic_enabled"),
|
||||
"vertex_total_tokens": vertex.get("total_tokens"),
|
||||
"vertex_message": vertex.get("message", ""),
|
||||
})
|
||||
if vertex.get("enabled"):
|
||||
result["status"] = "VERTEX"
|
||||
result["message"] = "OAuth token issued; Vertex countTokens accepted"
|
||||
return result
|
||||
message = request_error_message(response)
|
||||
lower = message.lower()
|
||||
if response.status_code in (400, 401) and any(item in lower for item in ("invalid_grant", "invalid_client", "invalid jwt", "invalid signature")):
|
||||
return {"status": "DEAD", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
|
||||
if response.status_code in (401, 403):
|
||||
return {"status": "RESTRICTED", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
|
||||
if response.status_code == 429:
|
||||
return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
|
||||
if response.status_code >= 500:
|
||||
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
|
||||
|
||||
|
||||
def check_adc(creds, proxy, timeout, probe_vertex=False, vertex_timeout=6, vertex_locations=None, vertex_models=None, vertex_max_attempts=2, vertex_anthropic_locations=None, vertex_anthropic_models=None, vertex_anthropic_max_attempts=2):
|
||||
try:
|
||||
token_uri = validate_google_token_uri(creds.get("token_uri"))
|
||||
except ValueError as exc:
|
||||
return {"status": "UNKNOWN", "message": str(exc), "client_id": creds.get("client_id")}
|
||||
data = {
|
||||
"client_id": creds["client_id"],
|
||||
"client_secret": creds["client_secret"],
|
||||
"refresh_token": creds["refresh_token"],
|
||||
"grant_type": "refresh_token",
|
||||
}
|
||||
try:
|
||||
response = requests.post(
|
||||
token_uri, data=data, proxies=proxy, timeout=timeout, allow_redirects=False,
|
||||
)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc), "client_id": creds.get("client_id")}
|
||||
if response.status_code in TOKEN_REDIRECT_STATUSES:
|
||||
return {
|
||||
"status": "UNKNOWN",
|
||||
"http_status": response.status_code,
|
||||
"message": "Google OAuth token endpoint redirect refused",
|
||||
"client_id": creds.get("client_id"),
|
||||
}
|
||||
if response.status_code == 200:
|
||||
payload = response.json()
|
||||
project_id = creds.get("quota_project_id") or creds.get("project_id")
|
||||
result = {"status": "VALID", "message": "refresh token accepted", "client_id": creds.get("client_id"), "project_id": project_id, "expires_in": payload.get("expires_in")}
|
||||
if probe_vertex:
|
||||
google_vertex = probe_vertex_llm(
|
||||
payload.get("access_token"), project_id, proxy,
|
||||
vertex_timeout, vertex_locations, vertex_models, vertex_max_attempts,
|
||||
)
|
||||
anthropic_vertex = probe_vertex_anthropic(
|
||||
payload.get("access_token"), project_id, proxy,
|
||||
vertex_timeout, vertex_anthropic_locations, vertex_anthropic_models, vertex_anthropic_max_attempts,
|
||||
) if vertex_anthropic_models else {"enabled": False, "message": ""}
|
||||
vertex = merge_vertex_results(google_vertex, anthropic_vertex)
|
||||
result.update({
|
||||
"vertex_enabled": vertex.get("enabled"),
|
||||
"vertex_location": vertex.get("location", ""),
|
||||
"vertex_model": vertex.get("model", ""),
|
||||
"vertex_available_models": vertex.get("available_models") or [],
|
||||
"vertex_google_enabled": vertex.get("google_enabled"),
|
||||
"vertex_anthropic_enabled": vertex.get("anthropic_enabled"),
|
||||
"vertex_total_tokens": vertex.get("total_tokens"),
|
||||
"vertex_message": vertex.get("message", ""),
|
||||
})
|
||||
if vertex.get("enabled"):
|
||||
result["status"] = "VERTEX"
|
||||
result["message"] = "refresh token accepted; Vertex countTokens accepted"
|
||||
return result
|
||||
message = request_error_message(response)
|
||||
lower = message.lower()
|
||||
if response.status_code in (400, 401) and any(item in lower for item in ("invalid_grant", "invalid_client", "unauthorized_client")):
|
||||
return {"status": "DEAD", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")}
|
||||
if response.status_code in (401, 403):
|
||||
return {"status": "RESTRICTED", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")}
|
||||
if response.status_code == 429:
|
||||
return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "client_id": creds.get("client_id")}
|
||||
if response.status_code >= 500:
|
||||
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")}
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")}
|
||||
|
||||
|
||||
def write_result(key, detector, result, source, finding):
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, {"detector": detector, **result}, source, finding, detector)
|
||||
extra = result.get("client_email") or result.get("client_id") or result.get("project_id") or source
|
||||
message = result.get("message", "")
|
||||
if result.get("status") == "VERTEX":
|
||||
models = result.get("vertex_available_models") or []
|
||||
model_text = ",".join(str(item) for item in models) or f"{result.get('vertex_location', '')}/{result.get('vertex_model', '')}".strip("/")
|
||||
message = f"{message}; models={model_text}"
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, transaction_status_files(), key, result["status"], message, extra,
|
||||
)
|
||||
if result.get("status") == "VERTEX" and result.get("vertex_google_enabled"):
|
||||
append_status(VERTEX_GEMINI_FILE, key, result["status"], message, extra)
|
||||
if result.get("status") == "VERTEX" and result.get("vertex_anthropic_enabled"):
|
||||
append_status(VERTEX_ANTHROPIC_FILE, key, result["status"], message, extra)
|
||||
record_validation_result(SERVICE, key, {"detector": detector, **result}, source, finding, detector)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="GCP credential checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", default=PLAIN_FILE)
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=25)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--probe-vertex", action="store_true", help="After OAuth succeeds, probe Vertex AI Gemini with countTokens through the configured proxy")
|
||||
parser.add_argument("--vertex-timeout", type=int, default=6, help="Seconds per Vertex countTokens request")
|
||||
parser.add_argument("--vertex-max-attempts", type=int, default=6, help="Maximum location/model countTokens attempts per credential")
|
||||
parser.add_argument("--vertex-locations", default=",".join(VERTEX_LOCATIONS), help="Comma-separated Vertex locations to probe")
|
||||
parser.add_argument("--vertex-models", default=",".join(VERTEX_MODELS), help="Comma-separated Vertex models to probe")
|
||||
parser.add_argument("--vertex-anthropic-locations", default=",".join(VERTEX_ANTHROPIC_LOCATIONS), help="Comma-separated Vertex Anthropic locations to probe")
|
||||
parser.add_argument("--vertex-anthropic-models", default=",".join(VERTEX_ANTHROPIC_MODELS), help="Comma-separated Vertex Anthropic model IDs to probe with rawPredict")
|
||||
parser.add_argument("--vertex-anthropic-max-attempts", type=int, default=20, help="Maximum Anthropic location/model attempts per credential")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES)
|
||||
known = set(known_statuses)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("RATE_LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_valid:
|
||||
retry_statuses.update({"VALID", "VERTEX"})
|
||||
vertex_locations = [item.strip() for item in str(args.vertex_locations or "").split(",") if item.strip()]
|
||||
vertex_models = [item.strip() for item in str(args.vertex_models or "").split(",") if item.strip()]
|
||||
vertex_anthropic_locations = [item.strip() for item in str(args.vertex_anthropic_locations or "").split(",") if item.strip()]
|
||||
vertex_anthropic_models = [item.strip() for item in str(args.vertex_anthropic_models or "").split(",") if item.strip()]
|
||||
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, detector, source, finding, parsed in extract_candidates(args.input, args.plain):
|
||||
if should_skip_key(
|
||||
key, checked, known, args, retry_statuses,
|
||||
service=SERVICE, source=source, finding=finding, detector=detector, known_statuses=known_statuses,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] {detector} candidate {mask_secret(key)} from {source}", flush=True)
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
if not parsed:
|
||||
result = {"status": "NO_CONTEXT", "message": "credential JSON is incomplete or unavailable"}
|
||||
elif detector == "GCP":
|
||||
result = check_service_account(
|
||||
parsed, proxy, args.timeout, args.probe_vertex,
|
||||
args.vertex_timeout, vertex_locations, vertex_models, args.vertex_max_attempts,
|
||||
vertex_anthropic_locations, vertex_anthropic_models, args.vertex_anthropic_max_attempts,
|
||||
)
|
||||
else:
|
||||
result = check_adc(
|
||||
parsed, proxy, args.timeout, args.probe_vertex,
|
||||
args.vertex_timeout, vertex_locations, vertex_models, args.vertex_max_attempts,
|
||||
vertex_anthropic_locations, vertex_anthropic_models, args.vertex_anthropic_max_attempts,
|
||||
)
|
||||
print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}", flush=True)
|
||||
write_result(key, detector, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
time.sleep(0.1)
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,738 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
from itertools import cycle
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
acquire_file_lock,
|
||||
append_checked,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
env_int,
|
||||
ensure_output_files as ensure_private_output_files,
|
||||
iter_findings,
|
||||
iter_bounded_text_lines,
|
||||
keycheck_input_mode,
|
||||
load_known_statuses,
|
||||
private_atomic_writer,
|
||||
record_cached_keycheck_occurrence,
|
||||
record_validation_result,
|
||||
release_file_lock,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from runtime_security import durable_replace, reject_reparse_components, require_private_directory, require_private_file
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "gemini"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
|
||||
def here(*parts):
|
||||
return os.path.join(SCRIPT_DIR, *parts)
|
||||
|
||||
|
||||
def out(*parts):
|
||||
return os.path.join(OUTPUT_DIR, *parts)
|
||||
|
||||
|
||||
def parent(*parts):
|
||||
return os.path.join(PARENT_DIR, *parts)
|
||||
|
||||
|
||||
# --- Configuration ---
|
||||
DEFAULT_INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
DEFAULT_PLAIN_INPUT_FILES = [out("gem.txt")]
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
CHECKED_FILE = out("geminiChecked.txt")
|
||||
RESULTS_FILE = out("geminiResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": out("geminiAlive.txt"),
|
||||
"VALID_RATE_LIMITED": out("geminiAliveRateLimited.txt"),
|
||||
"INVALID": out("geminiDead.txt"),
|
||||
"EXPIRED": out("geminiExpired.txt"),
|
||||
"LEAKED_REVOKED": out("geminiLeaked.txt"),
|
||||
"API_DISABLED": out("geminiDisabled.txt"),
|
||||
"RESTRICTED": out("geminiRestricted.txt"),
|
||||
"RATE_LIMITED": out("geminiRateLimited.txt"),
|
||||
"NETWORK_ERROR": out("geminiNetwork.txt"),
|
||||
"UNKNOWN": out("geminiUnknown.txt"),
|
||||
}
|
||||
|
||||
GEMINI_KEY_REGEX = re.compile(r"(?:AIza[0-9A-Za-z\-_]{35}|AQ\.[0-9A-Za-z\-_]{50})")
|
||||
GEMINI_DETECTOR_NAMES = {"googleai", "googleaistudio"}
|
||||
MODELS_URL = "https://generativelanguage.googleapis.com/v1beta/models"
|
||||
|
||||
PROBE_MODEL_PRIORITY = [
|
||||
"gemini-3.1-pro-preview",
|
||||
"gemini-3.7-flash",
|
||||
]
|
||||
|
||||
MODEL_PRIORITY = [
|
||||
"gemini-3",
|
||||
"gemini-2.5-pro",
|
||||
"gemini-2.5-flash",
|
||||
"gemini-2.0-flash",
|
||||
"gemini-1.5-pro",
|
||||
"gemini-1.5-flash",
|
||||
"imagen",
|
||||
"embedding",
|
||||
]
|
||||
|
||||
|
||||
def now_iso():
|
||||
return datetime.now(timezone.utc).isoformat(timespec="seconds")
|
||||
|
||||
|
||||
def mask_key(key):
|
||||
if not key or len(key) < 12:
|
||||
return key
|
||||
return f"{key[:8]}...{key[-4:]}"
|
||||
|
||||
|
||||
def redact_key_text(text, key):
|
||||
if not isinstance(text, str):
|
||||
return text
|
||||
redacted = text.replace(key, "***REDACTED***") if key else text
|
||||
return GEMINI_KEY_REGEX.sub("***REDACTED***", redacted)
|
||||
|
||||
|
||||
def redact_result_text(result, key):
|
||||
if isinstance(result, dict):
|
||||
return {k: redact_result_text(v, key) for k, v in result.items()}
|
||||
if isinstance(result, list):
|
||||
return [redact_result_text(v, key) for v in result]
|
||||
return redact_key_text(result, key)
|
||||
|
||||
|
||||
def key_from_line(line):
|
||||
line = line.strip()
|
||||
if not line:
|
||||
return None
|
||||
if "\t" in line:
|
||||
return line.split("\t", 1)[0].strip()
|
||||
return line.split(":", 1)[0].strip()
|
||||
|
||||
|
||||
def load_keys_from_file(filepath):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return set()
|
||||
if not os.path.exists(filepath):
|
||||
return set()
|
||||
keys = set()
|
||||
for line in iter_bounded_text_lines(filepath):
|
||||
key = key_from_line(line)
|
||||
if key:
|
||||
keys.add(key)
|
||||
return keys
|
||||
|
||||
|
||||
def load_checked_statuses(filepath=CHECKED_FILE):
|
||||
statuses = {}
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return statuses
|
||||
if not os.path.exists(filepath):
|
||||
return statuses
|
||||
for line in iter_bounded_text_lines(filepath):
|
||||
parts = line.rstrip("\n").split("\t")
|
||||
if not parts or not parts[0]:
|
||||
continue
|
||||
key = parts[0]
|
||||
status = parts[1] if len(parts) > 1 else "UNKNOWN"
|
||||
statuses[key] = status
|
||||
return statuses
|
||||
|
||||
|
||||
def load_all_known_keys():
|
||||
known = set(load_checked_statuses().keys())
|
||||
for path in STATUS_FILES.values():
|
||||
known.update(load_keys_from_file(path))
|
||||
return known
|
||||
|
||||
|
||||
def ensure_output_files():
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
require_private_directory(OUTPUT_DIR, create=True)
|
||||
|
||||
legacy_rate_limited = out("geminiLimited.txt")
|
||||
rate_limited = STATUS_FILES["RATE_LIMITED"]
|
||||
if os.path.exists(legacy_rate_limited) and not os.path.exists(rate_limited):
|
||||
require_private_file(legacy_rate_limited)
|
||||
durable_replace(legacy_rate_limited, rate_limited)
|
||||
require_private_file(rate_limited)
|
||||
|
||||
paths = {CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()}
|
||||
ensure_private_output_files(paths)
|
||||
migrate_legacy_alive_rate_limited()
|
||||
|
||||
|
||||
def effective_status(result):
|
||||
status = result.get("status")
|
||||
probe_status = (result.get("probe") or {}).get("status")
|
||||
if status == "VALID" and probe_status == "RATE_LIMITED":
|
||||
return "VALID_RATE_LIMITED"
|
||||
return status
|
||||
|
||||
|
||||
def _gemini_status_layout():
|
||||
paths_by_status = {
|
||||
status: os.path.abspath(os.fspath(path))
|
||||
for status, path in STATUS_FILES.items()
|
||||
}
|
||||
paths = list(dict.fromkeys(paths_by_status.values()))
|
||||
directories = {os.path.normcase(os.path.dirname(path)) for path in paths}
|
||||
if len(paths) != len(paths_by_status) or len(directories) != 1:
|
||||
raise RuntimeError("Gemini status files must be unique files in one directory")
|
||||
directory = os.path.dirname(paths[0])
|
||||
require_private_directory(directory, create=True)
|
||||
return paths_by_status, paths, os.path.join(directory, "geminiStatus.lock")
|
||||
|
||||
|
||||
def migrate_legacy_alive_rate_limited():
|
||||
paths_by_status, _, lock_path = _gemini_status_layout()
|
||||
alive_path = paths_by_status["VALID"]
|
||||
limited_path = paths_by_status["VALID_RATE_LIMITED"]
|
||||
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
||||
try:
|
||||
if not os.path.lexists(alive_path):
|
||||
return
|
||||
require_private_file(alive_path)
|
||||
|
||||
keep = []
|
||||
moved = {}
|
||||
for line in iter_bounded_text_lines(alive_path):
|
||||
key = key_from_line(line)
|
||||
if key and line.rstrip("\r\n").endswith(":RATE_LIMITED"):
|
||||
moved.setdefault(key, line if line.endswith("\n") else f"{line}\n")
|
||||
else:
|
||||
keep.append(line)
|
||||
if not moved:
|
||||
return
|
||||
|
||||
if os.path.lexists(limited_path):
|
||||
require_private_file(limited_path)
|
||||
existing = _normalized_status_lines(list(iter_bounded_text_lines(limited_path)))
|
||||
limited = []
|
||||
published = set()
|
||||
for line in existing:
|
||||
key = key_from_line(line)
|
||||
if key in moved:
|
||||
if key in published:
|
||||
continue
|
||||
published.add(key)
|
||||
limited.append(line)
|
||||
for key, line in moved.items():
|
||||
if key not in published:
|
||||
limited.append(line)
|
||||
published.add(key)
|
||||
|
||||
_validate_status_snapshot(limited_path, limited)
|
||||
_validate_status_snapshot(alive_path, keep)
|
||||
|
||||
# Make every moved key durable before publishing the source snapshot
|
||||
# that removes it. An interruption can therefore only leave duplicates.
|
||||
_replace_status_snapshot(limited_path, limited)
|
||||
confirmed = {key: 0 for key in moved}
|
||||
for line in iter_bounded_text_lines(limited_path):
|
||||
key = key_from_line(line)
|
||||
if key in confirmed:
|
||||
confirmed[key] += 1
|
||||
if any(count != 1 for count in confirmed.values()):
|
||||
raise RuntimeError("Gemini legacy rate-limited status publication was incomplete")
|
||||
_replace_status_snapshot(alive_path, keep)
|
||||
finally:
|
||||
release_file_lock(lock, lock_path)
|
||||
|
||||
|
||||
def load_proxies(proxy_file):
|
||||
if not os.path.exists(proxy_file):
|
||||
print(f"Info: {proxy_file} not found. Requests will go directly.")
|
||||
return None
|
||||
|
||||
proxies = []
|
||||
with open(proxy_file, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
ip, port, login, password = line.split(":")
|
||||
proxy_url = f"http://{login}:{password}@{ip}:{port}"
|
||||
proxies.append({"http": proxy_url, "https": proxy_url})
|
||||
except ValueError:
|
||||
print(f"Warning: bad proxy format: {line}. Skipping.")
|
||||
|
||||
if not proxies:
|
||||
print(f"Warning: {proxy_file} is empty. Requests will go directly.")
|
||||
return None
|
||||
|
||||
print(f"Loaded proxies: {len(proxies)}")
|
||||
return cycle(proxies)
|
||||
|
||||
|
||||
def status_file_line(key, result, status):
|
||||
if status in ("VALID", "VALID_RATE_LIMITED"):
|
||||
models_str = ",".join(result.get("notable_models", [])) or "models-only"
|
||||
probe_status = result.get("probe", {}).get("status", "not_probed")
|
||||
return f"{key}:[{models_str}]:{result.get('model_class', 'unknown')}:{probe_status}\n"
|
||||
message = (result.get("error", {}).get("message") or "").replace("\n", " ")[:300]
|
||||
return f"{key}\t{status}\t{message}\n"
|
||||
|
||||
|
||||
def _normalized_status_lines(lines):
|
||||
return [line if line.endswith("\n") else f"{line}\n" for line in lines]
|
||||
|
||||
|
||||
def _validate_status_snapshot(path, lines):
|
||||
max_bytes = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_BYTES", 32 * 1024 * 1024))
|
||||
max_items = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_ITEMS", 100000))
|
||||
max_line_bytes = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_LINE_BYTES", 8192))
|
||||
if len(lines) > max_items:
|
||||
raise RuntimeError(f"Gemini status file exceeds its item bound: {path}")
|
||||
total = 0
|
||||
for index, line in enumerate(lines, 1):
|
||||
encoded = line.encode("utf-8")
|
||||
if len(encoded) > max_line_bytes:
|
||||
raise RuntimeError(f"Gemini status line exceeds its byte bound: {path}:{index}")
|
||||
total += len(encoded)
|
||||
if total > max_bytes:
|
||||
raise RuntimeError(f"Gemini status file exceeds its aggregate byte bound: {path}")
|
||||
|
||||
|
||||
def _replace_status_snapshot(path, lines):
|
||||
with private_atomic_writer(path, binary=True, suffix=".status.tmp") as handle:
|
||||
for line in lines:
|
||||
handle.write(line.encode("utf-8"))
|
||||
|
||||
|
||||
def append_status_file(key, result):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
paths_by_status, paths, lock_path = _gemini_status_layout()
|
||||
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
||||
try:
|
||||
status = effective_status(result)
|
||||
target_path = paths_by_status.get(status, paths_by_status["UNKNOWN"])
|
||||
new_line = status_file_line(key, result, status)
|
||||
snapshots = {}
|
||||
for path in paths:
|
||||
if os.path.lexists(path):
|
||||
reject_reparse_components(path)
|
||||
snapshots[path] = _normalized_status_lines(list(iter_bounded_text_lines(path)))
|
||||
|
||||
rewritten = {
|
||||
path: [line for line in lines if key_from_line(line) != key]
|
||||
for path, lines in snapshots.items()
|
||||
}
|
||||
rewritten[target_path].insert(0, new_line)
|
||||
for path, lines in rewritten.items():
|
||||
_validate_status_snapshot(path, lines)
|
||||
|
||||
# Publish the new classification before removing any old copies. A
|
||||
# failure after this point can leave duplicates, but never no status.
|
||||
_replace_status_snapshot(target_path, rewritten[target_path])
|
||||
for path in paths:
|
||||
if path == target_path or rewritten[path] == snapshots[path]:
|
||||
continue
|
||||
_replace_status_snapshot(path, rewritten[path])
|
||||
finally:
|
||||
release_file_lock(lock, lock_path)
|
||||
|
||||
|
||||
def append_checked_file(key, result):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
append_checked(CHECKED_FILE, key, effective_status(result))
|
||||
|
||||
|
||||
def is_gemini_detector(detector):
|
||||
return str(detector or "").lower() in GEMINI_DETECTOR_NAMES
|
||||
|
||||
|
||||
def custom_detector_name(data):
|
||||
if not isinstance(data, dict):
|
||||
return ""
|
||||
extra = data.get("ExtraData") if isinstance(data.get("ExtraData"), dict) else {}
|
||||
name = extra.get("name") or ""
|
||||
if str(data.get("DetectorName") or "").lower() == "customregex" and is_gemini_detector(name):
|
||||
return name
|
||||
return ""
|
||||
|
||||
|
||||
def detector_name_from_finding(data):
|
||||
if not isinstance(data, dict):
|
||||
return ""
|
||||
if is_gemini_detector(data.get("DetectorName")):
|
||||
return data.get("DetectorName")
|
||||
custom_name = custom_detector_name(data)
|
||||
if custom_name:
|
||||
return custom_name
|
||||
|
||||
# Old wrapped format from earlier scanner versions.
|
||||
if is_gemini_detector(data.get("detector")):
|
||||
return data.get("detector")
|
||||
|
||||
finding = data.get("finding")
|
||||
if isinstance(finding, dict) and is_gemini_detector(finding.get("DetectorName")):
|
||||
return finding.get("DetectorName")
|
||||
custom_name = custom_detector_name(finding)
|
||||
if custom_name:
|
||||
return custom_name
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def extract_key_from_finding(data):
|
||||
if is_gemini_detector(data.get("DetectorName")):
|
||||
return data.get("Raw") or data.get("RawV2")
|
||||
if custom_detector_name(data):
|
||||
return data.get("Raw") or data.get("RawV2")
|
||||
|
||||
# Old wrapped format from earlier scanner versions.
|
||||
if is_gemini_detector(data.get("detector")):
|
||||
return data.get("raw") or data.get("raw_v2")
|
||||
|
||||
finding = data.get("finding")
|
||||
if isinstance(finding, dict) and is_gemini_detector(finding.get("DetectorName")):
|
||||
return finding.get("Raw") or finding.get("RawV2")
|
||||
if custom_detector_name(finding):
|
||||
return finding.get("Raw") or finding.get("RawV2")
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def iter_candidate_keys(input_file, plain_files):
|
||||
for item in iter_findings(input_file, ["GoogleAI", "GoogleAIStudio", "CustomRegex"]):
|
||||
finding = item.get("finding") or {}
|
||||
key = item.get("raw") or extract_key_from_finding(finding)
|
||||
if key and GEMINI_KEY_REGEX.fullmatch(key) and detector_name_from_finding(finding):
|
||||
yield item.get("source") or input_file, key, finding
|
||||
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
for path in plain_files:
|
||||
if not os.path.exists(path):
|
||||
print(f"Info: plain input {path} not found. Skipping.")
|
||||
continue
|
||||
try:
|
||||
keys = set()
|
||||
for line in iter_bounded_text_lines(path):
|
||||
keys.update(GEMINI_KEY_REGEX.findall(line))
|
||||
except (OSError, RuntimeError) as e:
|
||||
print(f"Warning: cannot read {path}: {e}")
|
||||
continue
|
||||
for idx, key in enumerate(sorted(keys), 1):
|
||||
yield f"{path}:plain:{idx}", key, {}
|
||||
|
||||
|
||||
def parse_error_response(response):
|
||||
try:
|
||||
payload = response.json()
|
||||
except json.JSONDecodeError:
|
||||
payload = {}
|
||||
|
||||
error = payload.get("error", {}) if isinstance(payload, dict) else {}
|
||||
return {
|
||||
"http_status": response.status_code,
|
||||
"code": error.get("code", response.status_code),
|
||||
"status": error.get("status", ""),
|
||||
"message": error.get("message", response.text[:500]),
|
||||
}
|
||||
|
||||
|
||||
def classify_error(error):
|
||||
http_status = int(error.get("http_status") or 0)
|
||||
status = str(error.get("status") or "").lower()
|
||||
message = str(error.get("message") or "").lower()
|
||||
|
||||
if "reported as leaked" in message or "leaked" in message:
|
||||
return "LEAKED_REVOKED"
|
||||
if "api key expired" in message or "expired" in message:
|
||||
return "EXPIRED"
|
||||
if "api key not valid" in message or "invalid api key" in message:
|
||||
return "INVALID"
|
||||
if "has not been used" in message or "it is disabled" in message or "api is disabled" in message:
|
||||
return "API_DISABLED"
|
||||
if "requests to this api" in message and "blocked" in message:
|
||||
return "RESTRICTED"
|
||||
if "api key restrictions" in message or "permission_denied" in status:
|
||||
return "RESTRICTED"
|
||||
if http_status == 429 or "resource_exhausted" in status or "quota" in message:
|
||||
return "RATE_LIMITED"
|
||||
if http_status in (400, 401):
|
||||
return "INVALID"
|
||||
if http_status == 403:
|
||||
return "RESTRICTED"
|
||||
return "UNKNOWN"
|
||||
|
||||
|
||||
def fetch_models(key, proxy, timeout, debug=False):
|
||||
try:
|
||||
response = requests.get(MODELS_URL, params={"key": key}, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as e:
|
||||
return {
|
||||
"status": "NETWORK_ERROR",
|
||||
"error": {"message": str(e)},
|
||||
"models": [],
|
||||
"model_infos": [],
|
||||
}
|
||||
|
||||
if debug:
|
||||
print(f" DEBUG /models: HTTP {response.status_code}: {redact_key_text(response.text[:500], key)}")
|
||||
|
||||
if response.status_code != 200:
|
||||
error = parse_error_response(response)
|
||||
return {
|
||||
"status": classify_error(error),
|
||||
"error": error,
|
||||
"models": [],
|
||||
"model_infos": [],
|
||||
}
|
||||
|
||||
payload = response.json()
|
||||
model_infos = payload.get("models", [])
|
||||
models = sorted({m.get("name", "").replace("models/", "") for m in model_infos if m.get("name")})
|
||||
return {
|
||||
"status": "VALID",
|
||||
"error": {},
|
||||
"models": models,
|
||||
"model_infos": model_infos,
|
||||
}
|
||||
|
||||
|
||||
def supported_methods_by_model(model_infos):
|
||||
output = {}
|
||||
for model in model_infos:
|
||||
name = model.get("name", "").replace("models/", "")
|
||||
if not name:
|
||||
continue
|
||||
output[name] = sorted(model.get("supportedGenerationMethods", []))
|
||||
return output
|
||||
|
||||
|
||||
def classify_models(models, methods_by_model):
|
||||
notable = []
|
||||
lower_models = {m.lower(): m for m in models}
|
||||
for marker in MODEL_PRIORITY:
|
||||
for lower, original in lower_models.items():
|
||||
if marker in lower and original not in notable:
|
||||
notable.append(original)
|
||||
|
||||
generation_models = sorted([
|
||||
model for model, methods in methods_by_model.items()
|
||||
if "generateContent" in methods
|
||||
])
|
||||
|
||||
if any("gemini-2.5-pro" in m.lower() for m in generation_models):
|
||||
model_class = "pro_generation"
|
||||
elif generation_models:
|
||||
model_class = "generation"
|
||||
elif models:
|
||||
model_class = "models_only"
|
||||
else:
|
||||
model_class = "no_models"
|
||||
|
||||
return notable[:20], generation_models, model_class
|
||||
|
||||
|
||||
def choose_probe_model(generation_models):
|
||||
available = set(generation_models)
|
||||
for model in PROBE_MODEL_PRIORITY:
|
||||
if model in available:
|
||||
return model
|
||||
return generation_models[0] if generation_models else None
|
||||
|
||||
|
||||
def probe_generation(key, model, proxy, timeout, debug=False):
|
||||
if not model:
|
||||
return {"status": "NO_GENERATION_MODEL", "model": None}
|
||||
|
||||
url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent"
|
||||
headers = {"x-goog-api-key": key, "Content-Type": "application/json"}
|
||||
payload = {
|
||||
"contents": [{"parts": [{"text": "ping"}]}],
|
||||
"generationConfig": {"maxOutputTokens": 1},
|
||||
}
|
||||
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as e:
|
||||
return {"status": "NETWORK_ERROR", "model": model, "error": {"message": str(e)}}
|
||||
|
||||
if debug:
|
||||
print(f" DEBUG probe {model}: HTTP {response.status_code}: {redact_key_text(response.text[:500], key)}")
|
||||
|
||||
if response.status_code == 200:
|
||||
return {"status": "GENERATION_OK", "model": model}
|
||||
|
||||
error = parse_error_response(response)
|
||||
return {"status": classify_error(error), "model": model, "error": error}
|
||||
|
||||
|
||||
def check_key(key, proxy, args):
|
||||
result = fetch_models(key, proxy, args.timeout, args.debug)
|
||||
result = redact_result_text(result, key)
|
||||
result.update({
|
||||
"checked_at": now_iso(),
|
||||
"key_masked": mask_key(key),
|
||||
"model_count": len(result.get("models", [])),
|
||||
})
|
||||
|
||||
if result["status"] != "VALID":
|
||||
result["notable_models"] = []
|
||||
result["generation_models"] = []
|
||||
result["model_class"] = "none"
|
||||
return result
|
||||
|
||||
methods_by_model = supported_methods_by_model(result.get("model_infos", []))
|
||||
notable, generation_models, model_class = classify_models(result["models"], methods_by_model)
|
||||
|
||||
result["methods_by_model"] = methods_by_model
|
||||
result["notable_models"] = notable
|
||||
result["generation_models"] = generation_models[:50]
|
||||
result["model_class"] = model_class
|
||||
result["billing_status"] = "unknown"
|
||||
|
||||
if args.probe_generation:
|
||||
probe_model = choose_probe_model(generation_models)
|
||||
result["probe"] = redact_result_text(probe_generation(key, probe_model, proxy, args.timeout, args.debug), key)
|
||||
else:
|
||||
result["probe"] = {"status": "not_probed", "model": None}
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def print_result(index, source, key, result):
|
||||
print(f"\n[{index}] Candidate {mask_key(key)} from {source}")
|
||||
print(f" STATUS: {result['status']}")
|
||||
|
||||
if result["status"] == "VALID":
|
||||
print(f" MODELS: {result.get('model_count', 0)} total; class={result.get('model_class')}")
|
||||
notable = result.get("notable_models", [])[:8]
|
||||
if notable:
|
||||
print(f" NOTABLE: {', '.join(notable)}")
|
||||
probe = result.get("probe", {})
|
||||
print(f" PROBE: {probe.get('status')} ({probe.get('model')})")
|
||||
if effective_status(result) == "VALID_RATE_LIMITED":
|
||||
print(f" OUT: {STATUS_FILES['VALID_RATE_LIMITED']}")
|
||||
else:
|
||||
error = result.get("error", {})
|
||||
message = (error.get("message") or "").replace("\n", " ")[:300]
|
||||
if message:
|
||||
print(f" MESSAGE: {message}")
|
||||
|
||||
print(f" OUT: {STATUS_FILES.get(result['status'], STATUS_FILES['UNKNOWN'])}")
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Gemini / Google AI API key classifier")
|
||||
parser.add_argument("--input", default=DEFAULT_INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=None, help="Plain text file with Gemini keys. Can be repeated.")
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--probe-generation", action="store_true", help="Optionally call generateContent, preferring gemini-3.1-pro-preview when available.")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
plain_files = args.plain if args.plain is not None else DEFAULT_PLAIN_INPUT_FILES
|
||||
ensure_output_files()
|
||||
|
||||
print("--- Gemini key checker ---")
|
||||
print("Default mode: /models only. Use --probe-generation for runtime/billing probe.")
|
||||
print(f"Workspace: {SCRIPT_DIR}")
|
||||
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked_statuses = load_checked_statuses()
|
||||
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES)
|
||||
known_keys = set(known_statuses)
|
||||
retry_statuses = set()
|
||||
if args.retry_limited:
|
||||
retry_statuses.update(("RATE_LIMITED", "VALID_RATE_LIMITED"))
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK_ERROR")
|
||||
if args.retry_valid:
|
||||
retry_statuses.update(("VALID", "VALID_RATE_LIMITED"))
|
||||
print(f"Loaded known keys: {len(known_keys)}; checked records: {len(checked_statuses)}")
|
||||
|
||||
seen_this_run = set()
|
||||
processed = 0
|
||||
skipped = 0
|
||||
|
||||
for source, key, finding in iter_candidate_keys(args.input, plain_files):
|
||||
if keycheck_input_mode() != 'postgres' and key in seen_this_run:
|
||||
cached_status = checked_statuses.get(key) or known_statuses.get(key) or 'UNKNOWN'
|
||||
detector = detector_name_from_finding(finding) or "GoogleAI"
|
||||
record_cached_keycheck_occurrence(SERVICE, key, cached_status, source, finding, detector)
|
||||
skipped += 1
|
||||
continue
|
||||
seen_this_run.add(key)
|
||||
|
||||
detector = detector_name_from_finding(finding) or "GoogleAI"
|
||||
if should_skip_key(
|
||||
key, checked_statuses, known_keys, args, retry_statuses,
|
||||
service=SERVICE, source=source, finding=finding, detector=detector,
|
||||
known_statuses=known_statuses,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
|
||||
processed += 1
|
||||
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_key(key, proxy, args)
|
||||
result["source"] = source
|
||||
event_result = {**result, "status": effective_status(result)}
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, event_result, source, finding, detector)
|
||||
|
||||
print_result(processed, source, key, result)
|
||||
append_status_file(key, result)
|
||||
append_checked_file(key, result)
|
||||
record_validation_result(SERVICE, key, {**result, "status": effective_status(result)}, source, finding, detector)
|
||||
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = effective_status(result)
|
||||
|
||||
# Small pause helps when many keys hit the same API/proxy.
|
||||
time.sleep(0.1)
|
||||
|
||||
print("\n--- Done ---")
|
||||
print(f"Processed: {processed}")
|
||||
print(f"Skipped: {skipped}")
|
||||
print(f"Results: {RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,212 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
read_plain_keys,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "github"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
PLAIN_FILE = os.path.join(OUTPUT_DIR, "github.txt")
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "githubChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "githubResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "githubAlive.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "githubDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "githubRestricted.txt"),
|
||||
"RATE_LIMITED": os.path.join(OUTPUT_DIR, "githubRateLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "githubNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "githubUnknown.txt"),
|
||||
"REFRESH_TOKEN": os.path.join(OUTPUT_DIR, "githubRefreshToken.txt"),
|
||||
}
|
||||
|
||||
GITHUB_TOKEN_RE = re.compile(r"\b(?:gh[pousr]_[A-Za-z0-9_]{20,}|github_pat_[A-Za-z0-9_]{20,})\b")
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values(), PLAIN_FILE])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, ["Github", "GitHubOauth2"]):
|
||||
text = "\n".join(str(value or "") for value in [item.get("raw"), item.get("raw_v2")])
|
||||
for match in GITHUB_TOKEN_RE.findall(text):
|
||||
yield match, item["source"], item["finding"]
|
||||
|
||||
for item in read_plain_keys(plain_files, GITHUB_TOKEN_RE):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
|
||||
def is_rate_limited(response):
|
||||
remaining = response.headers.get("X-RateLimit-Remaining")
|
||||
return response.status_code in (403, 429) and remaining == "0"
|
||||
|
||||
|
||||
def check_token(token, proxy, timeout):
|
||||
if token.startswith("ghr_"):
|
||||
return {
|
||||
"status": "REFRESH_TOKEN",
|
||||
"message": "GitHub refresh tokens cannot be checked directly as bearer API tokens",
|
||||
}
|
||||
|
||||
if token.startswith("ghs_"):
|
||||
url = "https://api.github.com/installation/repositories"
|
||||
token_kind = "installation"
|
||||
else:
|
||||
url = "https://api.github.com/user"
|
||||
token_kind = "user"
|
||||
|
||||
headers = {
|
||||
"Authorization": f"Bearer {token}",
|
||||
"Accept": "application/vnd.github+json",
|
||||
"X-GitHub-Api-Version": "2022-11-28",
|
||||
"User-Agent": "local-keycheck-github",
|
||||
}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc), "token_kind": token_kind}
|
||||
|
||||
message = request_error_message(response)
|
||||
scopes = response.headers.get("X-OAuth-Scopes", "")
|
||||
accepted_scopes = response.headers.get("X-Accepted-OAuth-Scopes", "")
|
||||
rate_remaining = response.headers.get("X-RateLimit-Remaining", "")
|
||||
rate_reset = response.headers.get("X-RateLimit-Reset", "")
|
||||
|
||||
if response.status_code == 200:
|
||||
payload = response.json()
|
||||
extra = {
|
||||
"token_kind": token_kind,
|
||||
"scopes": scopes,
|
||||
"accepted_scopes": accepted_scopes,
|
||||
"rate_remaining": rate_remaining,
|
||||
"rate_reset": rate_reset,
|
||||
}
|
||||
if token_kind == "installation":
|
||||
extra["repo_count"] = payload.get("total_count")
|
||||
return {"status": "VALID", "message": "installation token accepted", **extra}
|
||||
return {
|
||||
"status": "VALID",
|
||||
"message": "user token accepted",
|
||||
"login": payload.get("login"),
|
||||
"account_type": payload.get("type"),
|
||||
**extra,
|
||||
}
|
||||
|
||||
if response.status_code == 401:
|
||||
return {"status": "DEAD", "http_status": 401, "message": message, "token_kind": token_kind}
|
||||
if is_rate_limited(response):
|
||||
return {"status": "RATE_LIMITED", "http_status": response.status_code, "message": message, "token_kind": token_kind, "rate_reset": rate_reset}
|
||||
if response.status_code == 403:
|
||||
return {"status": "RESTRICTED", "http_status": 403, "message": message, "token_kind": token_kind, "scopes": scopes}
|
||||
if response.status_code in (404, 422):
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "token_kind": token_kind}
|
||||
if response.status_code >= 500:
|
||||
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "token_kind": token_kind}
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "token_kind": token_kind}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding)
|
||||
extra = result.get("login") or result.get("repo_count") or result.get("token_kind") or source
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), extra,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="GitHub token checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("RATE_LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_restricted:
|
||||
retry_statuses.add("RESTRICTED")
|
||||
|
||||
plain_files = args.plain or [PLAIN_FILE]
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, source, finding in extract_candidates(args.input, plain_files):
|
||||
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] GitHub candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_token(key, proxy, args.timeout)
|
||||
print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
time.sleep(0.1)
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,178 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
read_plain_keys,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "gitlab"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
PLAIN_FILE = os.path.join(OUTPUT_DIR, "gitlab.txt")
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "gitlabChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "gitlabResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "gitlabAlive.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "gitlabDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "gitlabRestricted.txt"),
|
||||
"RATE_LIMITED": os.path.join(OUTPUT_DIR, "gitlabRateLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "gitlabNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "gitlabUnknown.txt"),
|
||||
}
|
||||
|
||||
GITLAB_TOKEN_RE = re.compile(r"\b(?:glpat|gloas|glcbt|glimt|glrt|glft|glsoat)-[A-Za-z0-9_\-=]{20,}\b")
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values(), PLAIN_FILE])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, ["Gitlab"]):
|
||||
text = "\n".join(str(value or "") for value in [item.get("raw"), item.get("raw_v2")])
|
||||
for match in GITLAB_TOKEN_RE.findall(text):
|
||||
yield match, item["source"], item["finding"]
|
||||
|
||||
for item in read_plain_keys(plain_files, GITLAB_TOKEN_RE):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
|
||||
def check_token(token, base_url, proxy, timeout):
|
||||
base_url = base_url.rstrip("/")
|
||||
url = f"{base_url}/api/v4/user"
|
||||
headers = {"Authorization": f"Bearer {token}", "User-Agent": "local-keycheck-gitlab"}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc), "base_url": base_url}
|
||||
|
||||
message = request_error_message(response)
|
||||
retry_after = response.headers.get("Retry-After", "")
|
||||
rate_remaining = response.headers.get("RateLimit-Remaining") or response.headers.get("X-RateLimit-Remaining") or ""
|
||||
|
||||
if response.status_code == 200:
|
||||
payload = response.json()
|
||||
return {
|
||||
"status": "VALID",
|
||||
"message": "token accepted",
|
||||
"username": payload.get("username"),
|
||||
"name": payload.get("name"),
|
||||
"user_id": payload.get("id"),
|
||||
"base_url": base_url,
|
||||
"rate_remaining": rate_remaining,
|
||||
}
|
||||
if response.status_code == 401:
|
||||
return {"status": "DEAD", "http_status": 401, "message": message, "base_url": base_url}
|
||||
if response.status_code == 403:
|
||||
# TruffleHog treats 403 as a live token with insufficient scope or blocked account.
|
||||
return {"status": "RESTRICTED", "http_status": 403, "message": message, "base_url": base_url}
|
||||
if response.status_code == 429:
|
||||
return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "base_url": base_url, "retry_after": retry_after}
|
||||
if response.status_code >= 500:
|
||||
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "base_url": base_url}
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "base_url": base_url}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding)
|
||||
extra = result.get("username") or result.get("user_id") or source
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), extra,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="GitLab token checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--base-url", default="https://gitlab.com")
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("RATE_LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_restricted:
|
||||
retry_statuses.add("RESTRICTED")
|
||||
|
||||
plain_files = args.plain or [PLAIN_FILE]
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, source, finding in extract_candidates(args.input, plain_files):
|
||||
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] GitLab candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_token(key, args.base_url, proxy, args.timeout)
|
||||
print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
time.sleep(0.1)
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,275 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
classify_common_http_status,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
read_plain_keys,
|
||||
record_validation_result,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SERVICE = "groq"
|
||||
DETECTOR = "Groq"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "groqChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "groqResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "groqAlive.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "groqNoBalance.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "groqDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "groqRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "groqLimited.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "groqNoContext.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "groqNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "groqUnknown.txt"),
|
||||
}
|
||||
|
||||
GROQ_KEY_REGEX = re.compile(r"\bgsk_[A-Za-z0-9_-]{20,}\b")
|
||||
MODELS_URL = "https://api.groq.com/openai/v1/models"
|
||||
CHAT_URL = "https://api.groq.com/openai/v1/chat/completions"
|
||||
CHAT_MODEL_PRIORITY = (
|
||||
"llama-3.1-8b-instant",
|
||||
"llama-3.3-70b-versatile",
|
||||
"llama3-8b-8192",
|
||||
"llama3-70b-8192",
|
||||
"mixtral-8x7b-32768",
|
||||
"gemma2-9b-it",
|
||||
)
|
||||
NO_BALANCE_MARKERS = (
|
||||
"quota",
|
||||
"billing",
|
||||
"balance",
|
||||
"credit",
|
||||
"payment",
|
||||
"insufficient",
|
||||
"depleted",
|
||||
)
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, [DETECTOR]):
|
||||
key = item["raw"]
|
||||
if key and GROQ_KEY_REGEX.fullmatch(key):
|
||||
yield key, item["source"], item["finding"]
|
||||
|
||||
for item in read_plain_keys(plain_files, GROQ_KEY_REGEX):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
|
||||
def classify_groq_response(response):
|
||||
message = request_error_message(response).lower()
|
||||
if response.status_code == 401:
|
||||
return "DEAD"
|
||||
if response.status_code == 403:
|
||||
return "RESTRICTED"
|
||||
if response.status_code == 429:
|
||||
if any(marker in message for marker in NO_BALANCE_MARKERS):
|
||||
return "NO_BALANCE"
|
||||
return "LIMITED"
|
||||
return classify_common_http_status(response.status_code)
|
||||
|
||||
|
||||
def notable_models(payload):
|
||||
models = payload.get("data", []) if isinstance(payload, dict) else []
|
||||
ids = []
|
||||
for item in models:
|
||||
if isinstance(item, dict) and item.get("id"):
|
||||
ids.append(str(item.get("id")))
|
||||
priority = []
|
||||
for marker in ("llama", "mixtral", "gemma", "whisper"):
|
||||
for model in ids:
|
||||
if marker in model.lower() and model not in priority:
|
||||
priority.append(model)
|
||||
return priority[:20], len(ids), ids
|
||||
|
||||
|
||||
def choose_chat_model(model_ids):
|
||||
model_ids = [str(model or "") for model in model_ids if model]
|
||||
by_lower = {model.lower(): model for model in model_ids}
|
||||
for model in CHAT_MODEL_PRIORITY:
|
||||
if model.lower() in by_lower:
|
||||
return by_lower[model.lower()]
|
||||
for marker in ("llama", "mixtral", "gemma"):
|
||||
for model in model_ids:
|
||||
lowered = model.lower()
|
||||
if marker in lowered and "whisper" not in lowered and "guard" not in lowered:
|
||||
return model
|
||||
return ""
|
||||
|
||||
|
||||
def probe_chat_completion(key, model, proxy, timeout, debug=False):
|
||||
if not model:
|
||||
return {"status": "NO_CONTEXT", "message": "no chat-capable model from /models", "model": ""}
|
||||
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
|
||||
payload = {"model": model, "messages": [{"role": "user", "content": "ping"}], "max_tokens": 1}
|
||||
try:
|
||||
response = requests.post(CHAT_URL, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc), "model": model}
|
||||
|
||||
if debug:
|
||||
print(f" DEBUG chat ping {model}: HTTP {response.status_code}: {response.text[:500].replace(key, '***REDACTED***')}")
|
||||
|
||||
if response.status_code == 200:
|
||||
return {"status": "GENERATION_OK", "message": "chat completion accepted", "model": model}
|
||||
return {
|
||||
"status": classify_groq_response(response),
|
||||
"http_status": response.status_code,
|
||||
"message": request_error_message(response).replace(key, "***REDACTED***"),
|
||||
"model": model,
|
||||
}
|
||||
|
||||
|
||||
def check_key(key, proxy, timeout, debug=False):
|
||||
headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"}
|
||||
try:
|
||||
response = requests.get(MODELS_URL, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)}
|
||||
|
||||
if debug:
|
||||
print(f" DEBUG /models: HTTP {response.status_code}: {response.text[:500].replace(key, '***REDACTED***')}")
|
||||
|
||||
if response.status_code == 200:
|
||||
try:
|
||||
payload = response.json()
|
||||
except ValueError:
|
||||
payload = {}
|
||||
models, model_count, model_ids = notable_models(payload)
|
||||
chat_model = choose_chat_model(model_ids)
|
||||
probe = probe_chat_completion(key, chat_model, proxy, timeout, debug)
|
||||
if probe.get("status") != "GENERATION_OK":
|
||||
return {
|
||||
"status": probe.get("status") or "UNKNOWN",
|
||||
"message": probe.get("message", ""),
|
||||
"model_count": model_count,
|
||||
"models": models,
|
||||
"llm_probe_status": probe.get("status"),
|
||||
"llm_probe_model": probe.get("model", chat_model),
|
||||
"llm_probe_http_status": probe.get("http_status"),
|
||||
}
|
||||
return {
|
||||
"status": "VALID",
|
||||
"message": f"chat ping ok; model={chat_model}; models={model_count}",
|
||||
"model_count": model_count,
|
||||
"models": models,
|
||||
"llm_probe_status": probe.get("status"),
|
||||
"llm_probe_model": chat_model,
|
||||
}
|
||||
|
||||
return {
|
||||
"status": classify_groq_response(response),
|
||||
"http_status": response.status_code,
|
||||
"message": request_error_message(response).replace(key, "***REDACTED***"),
|
||||
}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
extra = ",".join(result.get("models") or [])[:500] if status == "VALID" else source
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), extra,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Groq key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_restricted:
|
||||
retry_statuses.add("RESTRICTED")
|
||||
if args.retry_no_balance:
|
||||
retry_statuses.add("NO_BALANCE")
|
||||
if args.retry_valid:
|
||||
retry_statuses.add("VALID")
|
||||
|
||||
print("--- Groq key checker ---")
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, source, finding in extract_candidates(args.input, args.plain):
|
||||
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] Groq candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_key(key, proxy, args.timeout, args.debug)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,142 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl, classify_common_http_status, commit_status_transaction,
|
||||
default_input_file, default_proxy_file, ensure_output_files, iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses, load_known_keys, load_proxies, mask_secret,
|
||||
read_plain_keys, record_validation_result, recover_status_transaction,
|
||||
request_error_message, require_provider_authority, service_output_dir, should_skip_key, write_keycheck_event,
|
||||
)
|
||||
|
||||
SERVICE = "huggingface"
|
||||
DETECTOR_NAMES = ["HuggingFace", "Huggingface"]
|
||||
DETECTOR = "HuggingFace"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "huggingfaceChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "huggingfaceResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "huggingfaceAlive.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "huggingfaceDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "huggingfaceRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "huggingfaceLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "huggingfaceNetwork.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "huggingfaceNoContext.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "huggingfaceUnknown.txt"),
|
||||
}
|
||||
KEY_REGEX = re.compile(r"\bhf_[A-Za-z0-9]{20,}\b")
|
||||
WHOAMI_URL = "https://huggingface.co/api/whoami-v2"
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def iter_candidate_decisions(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, DETECTOR_NAMES):
|
||||
key = item.get("credential_secret_text") or item["raw"]
|
||||
if key:
|
||||
yield key, item["source"], item["finding"], bool(KEY_REGEX.fullmatch(key))
|
||||
for item in read_plain_keys(plain_files, KEY_REGEX):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}, True
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
for key, source, finding, valid_format in iter_candidate_decisions(input_file, plain_files):
|
||||
if valid_format:
|
||||
yield key, source, finding
|
||||
|
||||
|
||||
def check_key(key, proxy, timeout):
|
||||
try:
|
||||
response = requests.get(WHOAMI_URL, headers={"Authorization": f"Bearer {key}"}, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)}
|
||||
if response.status_code == 200:
|
||||
data = response.json() if response.text else {}
|
||||
return {"status": "VALID", "message": "whoami accepted", "username": data.get("name") or data.get("fullname") or ""}
|
||||
status = "RESTRICTED" if response.status_code == 403 else classify_common_http_status(response.status_code)
|
||||
return {"status": status, "http_status": response.status_code, "message": request_error_message(response).replace(key, "***REDACTED***")}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), source,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="HuggingFace key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network: retry_statuses.add("NETWORK")
|
||||
if args.retry_limited: retry_statuses.add("LIMITED")
|
||||
if args.retry_unknown: retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
|
||||
if args.retry_restricted: retry_statuses.add("RESTRICTED")
|
||||
processed = skipped = 0
|
||||
print("--- HuggingFace key checker ---")
|
||||
postgres_mode = keycheck_input_mode() == "postgres"
|
||||
for key, source, finding, valid_format in iter_candidate_decisions(args.input, args.plain):
|
||||
if not valid_format and not postgres_mode:
|
||||
skipped += 1
|
||||
continue
|
||||
if valid_format and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] HuggingFace candidate {mask_secret(key)} from {source}")
|
||||
result = (
|
||||
check_key(key, next(proxy_cycler) if proxy_cycler else None, args.timeout)
|
||||
if valid_format else
|
||||
{"status": "NO_CONTEXT", "message": "candidate does not match canonical Hugging Face token format"}
|
||||
)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,499 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
combined_provider_routing_hint,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
provider_routing_database_failed,
|
||||
read_plain_keys,
|
||||
record_validation_result,
|
||||
recover_status_transaction,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from keycheckers.provider_resolution import resolve_provider_key
|
||||
|
||||
|
||||
SERVICE = "kimi"
|
||||
DETECTOR = "KimiMoonshot"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "kimiChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "kimiResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "kimiAlive.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "kimiNoBalance.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "kimiDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "kimiRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "kimiLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "kimiNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "kimiUnknown.txt"),
|
||||
}
|
||||
|
||||
KIMI_DETECTOR_NAMES = {"kimimoonshot", "moonshotai", "moonshot", "kimi"}
|
||||
KIMI_EXPLICIT_DETECTOR_NAMES = {"kimimoonshot", "moonshotai"}
|
||||
QWEN_EXPLICIT_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope"}
|
||||
DEEPSEEK_EXPLICIT_DETECTOR_NAMES = {"deepseekapikey", "deepseek_api_key"}
|
||||
GENERIC_SK_PROVIDERS = {"qwen", "deepseek", "kimi", "zai"}
|
||||
AMBIGUOUS_QWEN_DEEPSEEK_HINT = "ambiguous_qwen_deepseek"
|
||||
AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk"
|
||||
EXPLICIT_ASSIGNMENT_HINT_SOURCE = "explicit_assignment"
|
||||
CANDIDATE_PROVIDER_ROUTE_FIELD = "_keycheck_candidate_provider_route"
|
||||
KIMI_KEY_MAX_BYTES = 512
|
||||
KIMI_KEY_REGEX = re.compile(
|
||||
r"(?<![A-Za-z0-9_-])sk-[A-Za-z0-9][A-Za-z0-9_-]{20,505}(?![A-Za-z0-9_-])"
|
||||
)
|
||||
FOREIGN_KEY_PREFIXES = ("sk-ant-", "sk-or-", "sk-proj-", "sk-svcacct-", "sk-admin-")
|
||||
DEFAULT_BASE_URLS = (
|
||||
"https://api.moonshot.ai/v1",
|
||||
"https://api.moonshot.cn/v1",
|
||||
)
|
||||
QWEN_CONTEXT_REGEX = re.compile(
|
||||
r"(?:DASHSCOPE_API_KEY|QWEN_API_KEY|dashscope|qwen|model[_-]?studio|bailian)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
DEEPSEEK_CONTEXT_REGEX = re.compile(
|
||||
r"(?:DEEPSEEK_API_KEY|deepseek|api\.deepseek\.com)", re.IGNORECASE,
|
||||
)
|
||||
KIMI_CONTEXT_REGEX = re.compile(
|
||||
r"(?:MOONSHOT_API_KEY|KIMI_API_KEY|api\.moonshot\.(?:ai|cn)|platform\.kimi\.(?:ai|com))",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def normalize_base_url(value):
|
||||
return str(value or "").strip().rstrip("/")
|
||||
|
||||
|
||||
def split_csv(value):
|
||||
if not value:
|
||||
return []
|
||||
if isinstance(value, str):
|
||||
return [item.strip() for item in value.split(",") if item.strip()]
|
||||
return [str(item).strip() for item in value if str(item).strip()]
|
||||
|
||||
|
||||
def unique_ordered(values):
|
||||
output = []
|
||||
seen = set()
|
||||
for value in values:
|
||||
normalized = normalize_base_url(value)
|
||||
if normalized and normalized not in seen:
|
||||
seen.add(normalized)
|
||||
output.append(normalized)
|
||||
return output
|
||||
|
||||
|
||||
def endpoint_label(base_url):
|
||||
parsed = urlparse(base_url)
|
||||
return parsed.netloc or base_url
|
||||
|
||||
|
||||
def finding_detector_names(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return set()
|
||||
extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {}
|
||||
names = {
|
||||
str(finding.get("DetectorName") or finding.get("detector") or "").strip().lower(),
|
||||
str(extra.get("name") or "").strip().lower(),
|
||||
}
|
||||
return {name for name in names if name}
|
||||
|
||||
|
||||
def finding_has_detector(finding, detector_names):
|
||||
return bool(finding_detector_names(finding) & set(detector_names))
|
||||
|
||||
|
||||
def key_from_text(*values):
|
||||
for value in values:
|
||||
for match in KIMI_KEY_REGEX.finditer(str(value or "")):
|
||||
key = match.group(0)
|
||||
if not key.startswith(FOREIGN_KEY_PREFIXES):
|
||||
return key
|
||||
return ""
|
||||
|
||||
|
||||
def key_rejection_reason(key):
|
||||
value = str(key or "")
|
||||
try:
|
||||
key_bytes = len(value.encode("utf-8", errors="strict"))
|
||||
except UnicodeEncodeError:
|
||||
return "candidate is not valid UTF-8"
|
||||
if key_bytes > KIMI_KEY_MAX_BYTES:
|
||||
return f"candidate exceeds the {KIMI_KEY_MAX_BYTES}-byte key limit"
|
||||
if value.count("sk-") != 1:
|
||||
return "candidate contains multiple key prefixes"
|
||||
if not KIMI_KEY_REGEX.fullmatch(value) or value.startswith(FOREIGN_KEY_PREFIXES):
|
||||
return "candidate does not match the bounded Kimi/Moonshot key format"
|
||||
return ""
|
||||
|
||||
|
||||
def finding_provider_routing_hint(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return ""
|
||||
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
|
||||
persisted_hint = context.get("provider_hint")
|
||||
if context.get("provider_hint_source") == EXPLICIT_ASSIGNMENT_HINT_SOURCE and persisted_hint in (
|
||||
*GENERIC_SK_PROVIDERS, AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT,
|
||||
):
|
||||
return persisted_hint
|
||||
|
||||
parts = [str(context.get(key) or "") for key in ("nearby", "file")]
|
||||
metadata = finding.get("SourceMetadata") if isinstance(finding.get("SourceMetadata"), dict) else {}
|
||||
data = metadata.get("Data") if isinstance(metadata.get("Data"), dict) else {}
|
||||
for details in data.values():
|
||||
if isinstance(details, dict):
|
||||
parts.extend(str(details.get(key) or "") for key in ("file", "repository", "repo", "link", "image"))
|
||||
|
||||
text = "\n".join(parts)
|
||||
evidence = set()
|
||||
if QWEN_CONTEXT_REGEX.search(text) or finding_has_detector(finding, QWEN_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("qwen")
|
||||
if DEEPSEEK_CONTEXT_REGEX.search(text) or finding_has_detector(finding, DEEPSEEK_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("deepseek")
|
||||
if KIMI_CONTEXT_REGEX.search(text) or finding_has_detector(finding, KIMI_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("kimi")
|
||||
if persisted_hint == AMBIGUOUS_QWEN_DEEPSEEK_HINT:
|
||||
evidence.update(("qwen", "deepseek"))
|
||||
elif persisted_hint == AMBIGUOUS_GENERIC_SK_HINT:
|
||||
evidence.update(GENERIC_SK_PROVIDERS)
|
||||
elif persisted_hint in GENERIC_SK_PROVIDERS:
|
||||
evidence.add(persisted_hint)
|
||||
if len(evidence) > 1:
|
||||
return (
|
||||
AMBIGUOUS_QWEN_DEEPSEEK_HINT
|
||||
if evidence == {"qwen", "deepseek"}
|
||||
else AMBIGUOUS_GENERIC_SK_HINT
|
||||
)
|
||||
return next(iter(evidence)) if evidence else ""
|
||||
|
||||
|
||||
def iter_candidate_keys(input_file, plain_files, trusted_retry_files=None):
|
||||
detector_names = [
|
||||
"KimiMoonshot", "MoonshotAI", "Moonshot", "Kimi",
|
||||
"kimimoonshot", "moonshotai", "moonshot", "kimi", "CustomRegex",
|
||||
]
|
||||
routing_decisions = {}
|
||||
seen = set()
|
||||
for item in iter_findings(input_file, detector_names):
|
||||
finding = dict(item.get("finding") or {})
|
||||
finding.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, None)
|
||||
candidate_metadata = item.get("candidate_metadata")
|
||||
persisted_route = ""
|
||||
if keycheck_input_mode() == "postgres" and isinstance(candidate_metadata, dict):
|
||||
persisted_route = str(candidate_metadata.get("provider_hint") or "").lower()
|
||||
if persisted_route == SERVICE:
|
||||
finding[CANDIDATE_PROVIDER_ROUTE_FIELD] = SERVICE
|
||||
if persisted_route != SERVICE and not finding_has_detector(finding, KIMI_DETECTOR_NAMES):
|
||||
continue
|
||||
key = key_from_text(item.get("raw"), item.get("raw_v2"), finding.get("Raw"), finding.get("RawV2"))
|
||||
if not key or key_rejection_reason(key):
|
||||
continue
|
||||
if persisted_route == SERVICE:
|
||||
hint = SERVICE
|
||||
else:
|
||||
local_hint = finding_provider_routing_hint(finding)
|
||||
if key not in routing_decisions:
|
||||
routing_decisions[key] = combined_provider_routing_hint(key, local_hint)
|
||||
hint = routing_decisions[key]
|
||||
if provider_routing_database_failed():
|
||||
raise RuntimeError("provider routing evidence lookup failed closed")
|
||||
if hint == "kimi" or (
|
||||
keycheck_input_mode() == "postgres"
|
||||
and hint in (AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT)
|
||||
):
|
||||
seen.add(key)
|
||||
yield key, item.get("source") or input_file, finding
|
||||
|
||||
owned_retry_paths = {
|
||||
os.path.normcase(os.path.abspath(path)) for path in STATUS_FILES.values()
|
||||
}
|
||||
retry_files = [
|
||||
path for path in (trusted_retry_files or [])
|
||||
if os.path.normcase(os.path.abspath(path)) in owned_retry_paths
|
||||
]
|
||||
for item in read_plain_keys(retry_files, KIMI_KEY_REGEX):
|
||||
key = item["key"]
|
||||
if key_rejection_reason(key) or key in seen:
|
||||
continue
|
||||
hint = combined_provider_routing_hint(key, "kimi")
|
||||
if provider_routing_database_failed():
|
||||
raise RuntimeError("provider routing evidence lookup failed closed")
|
||||
if hint == "kimi":
|
||||
seen.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
|
||||
def redact_text(value, key):
|
||||
text = str(value or "")[:1000]
|
||||
if key:
|
||||
text = text.replace(key, "***REDACTED***")
|
||||
return KIMI_KEY_REGEX.sub("***REDACTED***", text)
|
||||
|
||||
|
||||
def parse_error(response, key):
|
||||
try:
|
||||
payload = response.json()
|
||||
except ValueError:
|
||||
payload = {}
|
||||
error = payload.get("error") if isinstance(payload, dict) else {}
|
||||
if not isinstance(error, dict):
|
||||
error = {}
|
||||
message = error.get("message") or response.text[:500]
|
||||
return {
|
||||
"http_status": response.status_code,
|
||||
"code": error.get("code") or error.get("type") or "",
|
||||
"message": redact_text(message, key),
|
||||
}
|
||||
|
||||
|
||||
def classify_error(error):
|
||||
http_status = int(error.get("http_status") or 0)
|
||||
code = str(error.get("code") or "").lower()
|
||||
message = str(error.get("message") or "").lower()
|
||||
if http_status == 401 or any(marker in code for marker in ("invalid_authentication", "invalid_api_key")):
|
||||
return "DEAD"
|
||||
if http_status == 403:
|
||||
return "RESTRICTED"
|
||||
if http_status == 429:
|
||||
if any(marker in code + " " + message for marker in ("quota", "balance", "payment")):
|
||||
return "NO_BALANCE"
|
||||
return "LIMITED"
|
||||
if 500 <= http_status <= 599:
|
||||
return "NETWORK"
|
||||
return "UNKNOWN"
|
||||
|
||||
|
||||
def check_base_url(key, base_url, proxy, timeout, debug=False):
|
||||
url = f"{normalize_base_url(base_url)}/users/me/balance"
|
||||
headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {
|
||||
"status": "NETWORK", "base_url": base_url,
|
||||
"region": endpoint_label(base_url), "message": redact_text(exc, key),
|
||||
}
|
||||
if debug:
|
||||
print(
|
||||
f" DEBUG {endpoint_label(base_url)} balance: HTTP {response.status_code}: "
|
||||
f"{redact_text(response.text[:500], key)}"
|
||||
)
|
||||
if response.status_code == 200:
|
||||
try:
|
||||
payload = response.json()
|
||||
data = payload.get("data") if isinstance(payload, dict) else None
|
||||
if not isinstance(data, dict) or "available_balance" not in data:
|
||||
raise ValueError("missing data.available_balance")
|
||||
available = float(data.get("available_balance"))
|
||||
voucher = float(data.get("voucher_balance", 0) or 0)
|
||||
cash = float(data.get("cash_balance", 0) or 0)
|
||||
except (TypeError, ValueError, json.JSONDecodeError) as exc:
|
||||
return {
|
||||
"status": "UNKNOWN", "base_url": base_url,
|
||||
"region": endpoint_label(base_url),
|
||||
"message": f"invalid balance response: {exc}",
|
||||
}
|
||||
status = "VALID" if available > 0 else "NO_BALANCE"
|
||||
return {
|
||||
"status": status,
|
||||
"authenticated": True,
|
||||
"base_url": base_url,
|
||||
"region": endpoint_label(base_url),
|
||||
"balance_usd": round(available, 6),
|
||||
"voucher_balance_usd": round(voucher, 6),
|
||||
"cash_balance_usd": round(cash, 6),
|
||||
"message": f"available_balance=${available:.6f}",
|
||||
}
|
||||
error = parse_error(response, key)
|
||||
return {
|
||||
"status": classify_error(error),
|
||||
"base_url": base_url,
|
||||
"region": endpoint_label(base_url),
|
||||
"http_status": response.status_code,
|
||||
"error": error,
|
||||
"message": error.get("message") or "",
|
||||
}
|
||||
|
||||
|
||||
def choose_final_status(attempts):
|
||||
statuses = [attempt.get("status") for attempt in attempts]
|
||||
for status in ("VALID", "NO_BALANCE", "LIMITED", "RESTRICTED", "UNKNOWN", "NETWORK"):
|
||||
if status in statuses:
|
||||
return status
|
||||
return "DEAD"
|
||||
|
||||
|
||||
def check_key(key, base_urls, proxy, timeout, debug=False):
|
||||
rejection = key_rejection_reason(key)
|
||||
if rejection:
|
||||
return {"status": "UNKNOWN", "message": rejection, "candidate_rejected": True}
|
||||
attempts = []
|
||||
for base_url in base_urls:
|
||||
result = check_base_url(key, base_url, proxy, timeout, debug)
|
||||
attempts.append(result)
|
||||
if result.get("status") in ("VALID", "NO_BALANCE"):
|
||||
return {**result, "attempts": attempts}
|
||||
status = choose_final_status(attempts)
|
||||
selected = next((attempt for attempt in attempts if attempt.get("status") == status), {})
|
||||
return {**selected, "status": status, "attempts": attempts}
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status,
|
||||
result.get("message", ""), result.get("region") or source,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def retry_statuses_from_args(args):
|
||||
statuses = set()
|
||||
if args.retry_network:
|
||||
statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
statuses.add("LIMITED")
|
||||
if args.retry_unknown:
|
||||
statuses.add("UNKNOWN")
|
||||
if args.retry_restricted:
|
||||
statuses.add("RESTRICTED")
|
||||
if args.retry_no_balance:
|
||||
statuses.add("NO_BALANCE")
|
||||
if args.retry_valid:
|
||||
statuses.add("VALID")
|
||||
return statuses
|
||||
|
||||
|
||||
def retry_input_files_from_args(args):
|
||||
if args.recheck_all:
|
||||
statuses = list(STATUS_FILES)
|
||||
else:
|
||||
statuses = [
|
||||
status for flag, status in (
|
||||
(args.retry_network, "NETWORK"),
|
||||
(args.retry_limited, "LIMITED"),
|
||||
(args.retry_unknown, "UNKNOWN"),
|
||||
(args.retry_restricted, "RESTRICTED"),
|
||||
(args.retry_no_balance, "NO_BALANCE"),
|
||||
(args.retry_valid, "VALID"),
|
||||
) if flag
|
||||
]
|
||||
return [STATUS_FILES[status] for status in statuses]
|
||||
|
||||
|
||||
def base_urls_from_args(args):
|
||||
custom = []
|
||||
for value in args.base_url:
|
||||
custom.extend(split_csv(value))
|
||||
custom.extend(split_csv(os.getenv("KIMI_BASE_URLS") or os.getenv("MOONSHOT_BASE_URLS")))
|
||||
defaults = [] if args.no_default_base_urls else DEFAULT_BASE_URLS
|
||||
return unique_ordered([*custom, *defaults])
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Kimi / Moonshot AI key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--base-url", action="append", default=[])
|
||||
parser.add_argument("--no-default-base-urls", action="store_true")
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = retry_statuses_from_args(args)
|
||||
retry_files = retry_input_files_from_args(args)
|
||||
base_urls = base_urls_from_args(args)
|
||||
if not base_urls:
|
||||
raise SystemExit("No Kimi/Moonshot base URLs configured")
|
||||
|
||||
print("--- Kimi / Moonshot AI key checker ---")
|
||||
print("base_urls: " + ", ".join(endpoint_label(url) for url in base_urls))
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, source, finding in iter_candidate_keys(args.input, args.plain, retry_files):
|
||||
finding = dict(finding or {})
|
||||
candidate_route = str(finding.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, "") or "").lower()
|
||||
if should_skip_key(
|
||||
key, checked, known, args, retry_statuses, service=SERVICE,
|
||||
source=source, finding=finding, detector=DETECTOR,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] Kimi/Moonshot candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
routing_hint = "kimi"
|
||||
if keycheck_input_mode() == "postgres":
|
||||
if candidate_route == SERVICE:
|
||||
routing_hint = SERVICE
|
||||
else:
|
||||
routing_hint = combined_provider_routing_hint(key, finding_provider_routing_hint(finding))
|
||||
if provider_routing_database_failed():
|
||||
raise RuntimeError("provider routing evidence lookup failed closed")
|
||||
if routing_hint in (AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT):
|
||||
result = resolve_provider_key(
|
||||
key, finding, proxy, args.timeout, args.debug,
|
||||
hint=routing_hint, origin_service=SERVICE,
|
||||
)
|
||||
else:
|
||||
result = check_key(key, base_urls, proxy, args.timeout, args.debug)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,572 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import requests
|
||||
import json
|
||||
import os
|
||||
import argparse
|
||||
import re
|
||||
from itertools import cycle
|
||||
import time
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files as ensure_private_output_files,
|
||||
iter_findings,
|
||||
iter_bounded_text_lines,
|
||||
keycheck_input_mode,
|
||||
load_known_statuses,
|
||||
private_atomic_writer,
|
||||
record_cached_keycheck_occurrence,
|
||||
record_validation_result,
|
||||
recover_status_transaction,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
try:
|
||||
sys.stdout.reconfigure(encoding='utf-8', errors='replace')
|
||||
sys.stderr.reconfigure(encoding='utf-8', errors='replace')
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# --- Конфигурация ---
|
||||
SERVICE = "openai"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
ALIVE_FILE = os.path.join(OUTPUT_DIR, "openaiAlive.txt")
|
||||
DEAD_FILE = os.path.join(OUTPUT_DIR, "openaiDead.txt")
|
||||
NETWORK_FILE = os.path.join(OUTPUT_DIR, "openaiNetwork.txt")
|
||||
LIMITED_FILE = os.path.join(OUTPUT_DIR, "openaiLimited.txt")
|
||||
RESTRICTED_FILE = os.path.join(OUTPUT_DIR, "openaiRestricted.txt")
|
||||
UNKNOWN_FILE = os.path.join(OUTPUT_DIR, "openaiUnknown.txt")
|
||||
NO_TARGET_FILE = os.path.join(OUTPUT_DIR, "openaiNoTarget.txt")
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "openaiChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "openaiResults.jsonl")
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
MODEL_PRIORITY_FOR_TEST = (
|
||||
'gpt-5.6-sol',
|
||||
'gpt-5.6',
|
||||
'gpt-5.6-luna',
|
||||
'gpt-5.6-terra',
|
||||
'gpt-5',
|
||||
'o3-pro',
|
||||
'o3',
|
||||
)
|
||||
TARGET_MODELS = {'gpt-5', 'o3-pro', 'o3'}
|
||||
NON_CHAT_MODEL_MARKERS = (
|
||||
'embedding', 'image', 'audio', 'tts', 'transcribe', 'realtime', 'search', 'moderation',
|
||||
)
|
||||
PROBE_MAX_COMPLETION_TOKENS = 16
|
||||
STATUS_FILES = [ALIVE_FILE, DEAD_FILE, NETWORK_FILE, LIMITED_FILE, RESTRICTED_FILE, UNKNOWN_FILE, NO_TARGET_FILE]
|
||||
STATUS_BY_FILE = {
|
||||
ALIVE_FILE: 'ALIVE',
|
||||
DEAD_FILE: 'DEAD',
|
||||
NETWORK_FILE: 'NETWORK',
|
||||
LIMITED_FILE: 'LIMITED',
|
||||
RESTRICTED_FILE: 'RESTRICTED',
|
||||
UNKNOWN_FILE: 'UNKNOWN',
|
||||
NO_TARGET_FILE: 'NO_TARGET_MODELS',
|
||||
}
|
||||
OPENAI_KEY_REGEX = re.compile(r'sk-[A-Za-z0-9_-]{20,}')
|
||||
|
||||
# --- Вспомогательные функции ---
|
||||
def key_from_line(line):
|
||||
line = line.strip()
|
||||
if not line:
|
||||
return None
|
||||
if '\t' in line:
|
||||
return line.split('\t', 1)[0].strip()
|
||||
if ':[' in line:
|
||||
return line.split(':[', 1)[0].strip()
|
||||
match = OPENAI_KEY_REGEX.search(line)
|
||||
if match:
|
||||
return match.group(0)
|
||||
return line.split()[0].strip()
|
||||
|
||||
def load_set_from_file(filepath):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return set()
|
||||
if not os.path.exists(filepath): return set()
|
||||
return {key for key in (key_from_line(line) for line in iter_bounded_text_lines(filepath)) if key}
|
||||
|
||||
|
||||
def iter_plain_openai_keys(paths):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
seen = set()
|
||||
items = []
|
||||
for path in paths or []:
|
||||
if not path or not os.path.exists(path):
|
||||
continue
|
||||
for line in iter_bounded_text_lines(path):
|
||||
key = key_from_line(line)
|
||||
if not key or key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
items.append({'key': key, 'source': path, 'finding': {}})
|
||||
for item in items:
|
||||
yield item
|
||||
|
||||
|
||||
def retry_plain_files(args):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return []
|
||||
files = list(args.plain or [])
|
||||
if args.recheck_all:
|
||||
files.extend(STATUS_FILES)
|
||||
else:
|
||||
if args.retry_limited:
|
||||
files.append(LIMITED_FILE)
|
||||
if args.retry_network:
|
||||
files.append(NETWORK_FILE)
|
||||
if args.retry_unknown:
|
||||
files.append(UNKNOWN_FILE)
|
||||
if args.retry_restricted:
|
||||
files.append(RESTRICTED_FILE)
|
||||
if args.retry_no_balance:
|
||||
files.append(LIMITED_FILE)
|
||||
out = []
|
||||
seen = set()
|
||||
for path in files:
|
||||
if path and path not in seen:
|
||||
seen.add(path)
|
||||
out.append(path)
|
||||
return out
|
||||
|
||||
def load_checked_statuses():
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return {}
|
||||
statuses = {}
|
||||
if not os.path.exists(CHECKED_FILE):
|
||||
return statuses
|
||||
for line in iter_bounded_text_lines(CHECKED_FILE):
|
||||
parts = line.rstrip('\n').split('\t')
|
||||
if parts and parts[0]:
|
||||
statuses[parts[0]] = parts[1] if len(parts) > 1 else 'UNKNOWN'
|
||||
return statuses
|
||||
|
||||
def load_known_keys():
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return set()
|
||||
known = set(load_checked_statuses().keys())
|
||||
for path in STATUS_FILES:
|
||||
known.update(load_set_from_file(path))
|
||||
return known
|
||||
|
||||
def ensure_output_files():
|
||||
ensure_private_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_BY_FILE)
|
||||
|
||||
def compact_status_file(path):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
if not os.path.exists(path):
|
||||
return
|
||||
last_by_key = {}
|
||||
order = []
|
||||
for line in iter_bounded_text_lines(path):
|
||||
key = key_from_line(line)
|
||||
if not key:
|
||||
continue
|
||||
if key not in last_by_key:
|
||||
order.append(key)
|
||||
last_by_key[key] = line
|
||||
for attempt in range(6):
|
||||
try:
|
||||
with private_atomic_writer(path) as f:
|
||||
for key in order:
|
||||
f.write(last_by_key[key])
|
||||
except PermissionError:
|
||||
if attempt == 5:
|
||||
print(f"Warning: unable to compact {path}; leaving existing file as-is")
|
||||
return
|
||||
time.sleep(0.1 * (attempt + 1))
|
||||
else:
|
||||
return
|
||||
|
||||
def compact_all_status_files():
|
||||
for path in [CHECKED_FILE, *STATUS_FILES]:
|
||||
compact_status_file(path)
|
||||
|
||||
def backfill_checked_file():
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
checked = load_checked_statuses()
|
||||
changed = False
|
||||
for path, status in STATUS_BY_FILE.items():
|
||||
for key in load_set_from_file(path):
|
||||
if key not in checked:
|
||||
checked[key] = status
|
||||
changed = True
|
||||
if not changed:
|
||||
return
|
||||
with private_atomic_writer(CHECKED_FILE) as f:
|
||||
for key, status in sorted(checked.items()):
|
||||
f.write(f"{key}\t{status}\tbackfilled\n")
|
||||
|
||||
def load_proxies(proxy_file=None):
|
||||
proxy_file = proxy_file or PROXY_FILE
|
||||
if not os.path.exists(proxy_file): return None
|
||||
proxies = []
|
||||
with open(proxy_file, 'r') as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line: continue
|
||||
try:
|
||||
ip, port, login, password = line.split(':')
|
||||
proxy_url = f"http://{login}:{password}@{ip}:{port}"
|
||||
proxies.append({"http": proxy_url, "https": proxy_url})
|
||||
except ValueError:
|
||||
print(f"⚠️ Неверный формат прокси: '{line}'. Пропускаем.")
|
||||
if not proxies:
|
||||
print("⚠️ Файл proxy.txt пуст. Запросы будут идти напрямую.")
|
||||
return None
|
||||
print(f"✅ Загружено {len(proxies)} прокси.")
|
||||
return cycle(proxies)
|
||||
|
||||
def move_key_to_alive(
|
||||
key, available_target_models, service_tier, source='', finding=None,
|
||||
model_inventory=None, probe_model='',
|
||||
):
|
||||
"""
|
||||
Перемещает ключ из DEAD_FILE в ALIVE_FILE, записывая модели и service_tier.
|
||||
"""
|
||||
models_str = ",".join(sorted(list(available_target_models)))
|
||||
tier_str = str(service_tier or 'unknown').replace('\r', ' ').replace('\n', ' ')[:1000]
|
||||
print(f" -> ✅ Ключ рабочий! Модели: {models_str}, Тир: {tier_str}. Перемещаем в {ALIVE_FILE}")
|
||||
|
||||
model_inventory = sorted(set(model_inventory or available_target_models))
|
||||
result = {
|
||||
'status': 'ALIVE',
|
||||
'models': sorted(list(available_target_models)),
|
||||
'model_inventory': model_inventory,
|
||||
'model_count': len(model_inventory),
|
||||
'llm_probe_model': probe_model,
|
||||
'llm_probe_status': 'GENERATION_OK',
|
||||
'service_tier': service_tier,
|
||||
'message': f'chat ping ok; model={probe_model}; models={len(model_inventory)}',
|
||||
}
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, 'OpenAI')
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE,
|
||||
STATUS_BY_FILE,
|
||||
key,
|
||||
'ALIVE',
|
||||
status_line=f"{key}:[{models_str}]:{tier_str}\n",
|
||||
checked_line=f"{key}\tALIVE\t{time.strftime('%Y-%m-%dT%H:%M:%S')}\n",
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, 'OpenAI')
|
||||
|
||||
def redact_message(message, key):
|
||||
message = str(message).replace('\r', ' ').replace('\n', ' ')[:1000]
|
||||
if key:
|
||||
message = message.replace(key, '***REDACTED***')
|
||||
return OPENAI_KEY_REGEX.sub('***REDACTED***', message)
|
||||
|
||||
def write_key_status(key, path, status, message='', source='', finding=None, metadata=None):
|
||||
status_upper = status.upper()
|
||||
projection_status = STATUS_BY_FILE.get(path)
|
||||
if not projection_status:
|
||||
raise ValueError(f'unknown OpenAI status projection: {path}')
|
||||
message = redact_message(message, key)
|
||||
result = {**(metadata or {}), 'status': status_upper, 'message': message}
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, 'OpenAI')
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE,
|
||||
STATUS_BY_FILE,
|
||||
key,
|
||||
projection_status,
|
||||
message,
|
||||
source,
|
||||
status_line=f"{key}\t{status}\t{message}\n",
|
||||
checked_line=f"{key}\t{projection_status}\t{time.strftime('%Y-%m-%dT%H:%M:%S')}\n",
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, 'OpenAI')
|
||||
|
||||
def extract_openai_key(data):
|
||||
if data.get("DetectorName") == "OpenAI":
|
||||
return data.get("Raw") or data.get("RawV2")
|
||||
if data.get("detector") == "OpenAI":
|
||||
return data.get("raw") or data.get("raw_v2")
|
||||
finding = data.get("finding")
|
||||
if isinstance(finding, dict) and finding.get("DetectorName") == "OpenAI":
|
||||
return finding.get("Raw") or finding.get("RawV2")
|
||||
return None
|
||||
|
||||
# --- Функции проверки ---
|
||||
def check_authentication(key, proxy):
|
||||
print(f" [1/2] Проверка аутентификации...")
|
||||
url = "https://api.openai.com/v1/models"
|
||||
headers = {"Authorization": f"Bearer {key}"}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=15)
|
||||
if response.status_code == 200:
|
||||
print(" -> Аутентификация пройдена.")
|
||||
return 'valid', response.json().get('data', [])
|
||||
elif response.status_code == 401:
|
||||
print(" -> Ошибка 401: Ключ недействителен или отозван.")
|
||||
return 'dead', response.text
|
||||
elif response.status_code == 403:
|
||||
print(f" -> Ошибка 403: ключ ограничен/заблокирован: {response.text[:300]}")
|
||||
return 'restricted', response.text
|
||||
elif response.status_code == 429:
|
||||
print(f" -> Ошибка 429: rate limit / quota: {response.text[:300]}")
|
||||
return 'limited', response.text
|
||||
else:
|
||||
print(f" -> Ошибка {response.status_code}: {response.text}")
|
||||
return 'unknown', response.text
|
||||
except requests.RequestException as e:
|
||||
print(f" -> Ошибка сети: {e}")
|
||||
return 'network', str(e)
|
||||
|
||||
def reportable_target_models(model_ids):
|
||||
return {
|
||||
model for model in model_ids
|
||||
if model in TARGET_MODELS or model.startswith('gpt-5.6-') or model == 'gpt-5.6'
|
||||
}
|
||||
|
||||
|
||||
def choose_probe_model(model_ids):
|
||||
models = [str(model or '') for model in model_ids if model]
|
||||
by_lower = {model.lower(): model for model in models}
|
||||
for model in MODEL_PRIORITY_FOR_TEST:
|
||||
if model.lower() in by_lower:
|
||||
return by_lower[model.lower()]
|
||||
for model in models:
|
||||
lowered = model.lower()
|
||||
if lowered.startswith(('gpt-', 'o')) and not any(
|
||||
marker in lowered for marker in NON_CHAT_MODEL_MARKERS
|
||||
):
|
||||
return model
|
||||
return ''
|
||||
|
||||
|
||||
def check_balance_and_tier(key, model_to_test, proxy):
|
||||
"""
|
||||
Проверяет баланс и возвращает service_tier в случае успеха.
|
||||
"""
|
||||
print(f" [2/2] Проверка баланса и тира...")
|
||||
if not model_to_test:
|
||||
print(" -> Не найдено подходящих моделей для теста баланса.")
|
||||
return 'unknown', 'no chat-capable model from /models'
|
||||
|
||||
print(f" -> Используем модель для теста: {model_to_test}")
|
||||
url = "https://api.openai.com/v1/chat/completions"
|
||||
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
|
||||
payload = {
|
||||
"model": model_to_test,
|
||||
"messages": [{"role": "user", "content": "Reply with one digit."}],
|
||||
"max_completion_tokens": PROBE_MAX_COMPLETION_TOKENS,
|
||||
}
|
||||
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=20)
|
||||
if response.status_code == 200:
|
||||
# Успех, извлекаем service_tier
|
||||
response_data = response.json()
|
||||
service_tier = response_data.get('service_tier')
|
||||
return 'ok', service_tier
|
||||
elif response.status_code == 429:
|
||||
print(f" -> Ошибка 429: Нет баланса или превышен лимит.")
|
||||
return 'limited', response.text
|
||||
elif response.status_code == 401:
|
||||
print(f" -> Ошибка 401: ключ недействителен или отозван.")
|
||||
return 'dead', response.text
|
||||
elif response.status_code == 403:
|
||||
print(f" -> Ошибка 403: ключ ограничен/заблокирован: {response.text[:300]}")
|
||||
return 'restricted', response.text
|
||||
else:
|
||||
print(f" -> Ошибка {response.status_code}: {response.text}")
|
||||
return 'unknown', response.text
|
||||
except requests.RequestException as e:
|
||||
print(f" -> Ошибка сети: {e}")
|
||||
return 'network', str(e)
|
||||
|
||||
# --- Основной процесс ---
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description='OpenAI key checker')
|
||||
parser.add_argument('--input', default=INPUT_FILE)
|
||||
parser.add_argument('--plain', action='append', default=[])
|
||||
parser.add_argument('--proxy-file', default=PROXY_FILE)
|
||||
parser.add_argument('--max-keys', type=int, default=0)
|
||||
parser.add_argument('--retry-network', action='store_true')
|
||||
parser.add_argument('--retry-limited', action='store_true')
|
||||
parser.add_argument('--retry-unknown', action='store_true')
|
||||
parser.add_argument('--retry-restricted', action='store_true')
|
||||
parser.add_argument('--retry-no-balance', action='store_true')
|
||||
parser.add_argument('--recheck-all', action='store_true')
|
||||
return parser.parse_args()
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_output_files()
|
||||
compact_all_status_files()
|
||||
backfill_checked_file()
|
||||
print("--- 🚀 Запуск чекера ключей OpenAI 🚀 ---")
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked_statuses = load_checked_statuses()
|
||||
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_BY_FILE)
|
||||
known_keys = set(known_statuses)
|
||||
alive_keys = load_set_from_file(ALIVE_FILE)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add('NETWORK')
|
||||
if args.retry_limited:
|
||||
retry_statuses.update(('LIMITED', 'LIMITED_OR_NO_BALANCE', 'LIMITED_OR_QUOTA'))
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add('UNKNOWN')
|
||||
if args.retry_restricted:
|
||||
retry_statuses.add('RESTRICTED')
|
||||
if args.retry_no_balance:
|
||||
retry_statuses.update(('NO_BALANCE', 'NO_QUOTA', 'LIMITED_OR_NO_BALANCE', 'LIMITED_OR_QUOTA'))
|
||||
print(f"📖 Загружено: {len(alive_keys)} живых ключей, {len(known_keys)} уже классифицированных ключей, {len(checked_statuses)} checked.")
|
||||
|
||||
if keycheck_input_mode() == 'jsonl' and not os.path.exists(args.input):
|
||||
print(f"❌ Файл с секретами {args.input} не найден. Завершение.")
|
||||
return
|
||||
|
||||
processed = 0
|
||||
skipped = 0
|
||||
seen_this_run = set()
|
||||
def candidates():
|
||||
for item in iter_findings(args.input, ['OpenAI']):
|
||||
data = item.get('finding') or {}
|
||||
key = item.get('raw') or extract_openai_key(data)
|
||||
if key:
|
||||
yield {'key': key, 'source': item.get('source') or args.input, 'finding': data}
|
||||
seen_plain = set()
|
||||
for item in iter_plain_openai_keys(retry_plain_files(args)):
|
||||
key = item.get('key')
|
||||
if key and key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield item
|
||||
|
||||
for item in candidates():
|
||||
data = item.get('finding') or {}
|
||||
source_line = item.get('source') or args.input
|
||||
key = item.get('key')
|
||||
if not key:
|
||||
continue
|
||||
if keycheck_input_mode() != 'postgres' and key in seen_this_run:
|
||||
cached_status = checked_statuses.get(key) or known_statuses.get(key) or 'UNKNOWN'
|
||||
record_cached_keycheck_occurrence(SERVICE, key, cached_status, source_line, data, 'OpenAI')
|
||||
skipped += 1
|
||||
continue
|
||||
seen_this_run.add(key)
|
||||
if should_skip_key(
|
||||
key, checked_statuses, known_keys, args, retry_statuses,
|
||||
service=SERVICE, source=source_line, finding=data, detector='OpenAI',
|
||||
known_statuses=known_statuses,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
|
||||
print(f"\n[{processed}] 🎯 Новый кандидат: {key[:8]}...{key[-4:]} from {source_line}")
|
||||
|
||||
current_proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
|
||||
auth_status, auth_data = check_authentication(key, current_proxy)
|
||||
|
||||
if auth_status == 'valid' and auth_data:
|
||||
all_available_models_data = auth_data
|
||||
all_model_ids = sorted({
|
||||
str(model.get('id') or '') for model in all_available_models_data
|
||||
if isinstance(model, dict) and model.get('id')
|
||||
})
|
||||
found_target_models = reportable_target_models(all_model_ids)
|
||||
model_to_test = choose_probe_model(all_model_ids)
|
||||
|
||||
if not model_to_test:
|
||||
print(" -> Ключ валиден, но не имеет подходящей chat-модели. Пропускаем.")
|
||||
write_key_status(
|
||||
key, NO_TARGET_FILE, 'no_target_models', ','.join(all_model_ids)[:500],
|
||||
source_line, data, {
|
||||
'models': sorted(found_target_models),
|
||||
'model_inventory': all_model_ids,
|
||||
'model_count': len(all_model_ids),
|
||||
'llm_probe_status': 'NO_CONTEXT',
|
||||
},
|
||||
)
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = 'NO_TARGET_MODELS'
|
||||
continue
|
||||
|
||||
balance_status, balance_data = check_balance_and_tier(key, model_to_test, current_proxy)
|
||||
probe_metadata = {
|
||||
'models': sorted(found_target_models or {model_to_test}),
|
||||
'model_inventory': all_model_ids,
|
||||
'model_count': len(all_model_ids),
|
||||
'llm_probe_model': model_to_test,
|
||||
'llm_probe_status': {
|
||||
'ok': 'GENERATION_OK',
|
||||
'limited': 'LIMITED',
|
||||
'network': 'NETWORK',
|
||||
'restricted': 'RESTRICTED',
|
||||
'dead': 'DEAD',
|
||||
}.get(balance_status, 'UNKNOWN'),
|
||||
}
|
||||
|
||||
# Проверяем, что результат не None (успешная проверка баланса)
|
||||
if balance_status == 'ok':
|
||||
move_key_to_alive(
|
||||
key, found_target_models or {model_to_test}, balance_data, source_line, data,
|
||||
model_inventory=all_model_ids, probe_model=model_to_test,
|
||||
)
|
||||
alive_keys.add(key)
|
||||
final_status = 'ALIVE'
|
||||
elif balance_status == 'limited':
|
||||
write_key_status(key, LIMITED_FILE, 'limited_or_no_balance', balance_data, source_line, data, probe_metadata)
|
||||
final_status = 'LIMITED_OR_NO_BALANCE'
|
||||
elif balance_status == 'network':
|
||||
write_key_status(key, NETWORK_FILE, 'network_error', balance_data, source_line, data, probe_metadata)
|
||||
final_status = 'NETWORK'
|
||||
elif balance_status == 'restricted':
|
||||
write_key_status(key, RESTRICTED_FILE, 'restricted', balance_data, source_line, data, probe_metadata)
|
||||
final_status = 'RESTRICTED'
|
||||
elif balance_status == 'dead':
|
||||
write_key_status(key, DEAD_FILE, 'invalid_or_revoked', balance_data, source_line, data, probe_metadata)
|
||||
final_status = 'DEAD'
|
||||
else:
|
||||
write_key_status(key, UNKNOWN_FILE, 'unknown', balance_data, source_line, data, probe_metadata)
|
||||
final_status = 'UNKNOWN'
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = final_status
|
||||
elif auth_status == 'network':
|
||||
write_key_status(key, NETWORK_FILE, 'network_error', auth_data, source_line, data)
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = 'NETWORK'
|
||||
elif auth_status == 'limited':
|
||||
write_key_status(key, LIMITED_FILE, 'limited_or_quota', auth_data, source_line, data)
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = 'LIMITED'
|
||||
elif auth_status == 'restricted':
|
||||
write_key_status(key, RESTRICTED_FILE, 'restricted', auth_data, source_line, data)
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = 'RESTRICTED'
|
||||
elif auth_status == 'dead':
|
||||
write_key_status(key, DEAD_FILE, 'invalid_or_revoked', auth_data, source_line, data)
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = 'DEAD'
|
||||
else:
|
||||
write_key_status(key, UNKNOWN_FILE, 'unknown', auth_data, source_line, data)
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = 'UNKNOWN'
|
||||
|
||||
print("\n--- ✅ Проверка завершена. ---")
|
||||
print(f"Processed={processed}, skipped={skipped}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,498 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import requests
|
||||
import json
|
||||
import os
|
||||
import argparse
|
||||
from itertools import cycle
|
||||
from datetime import datetime, timezone
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
try:
|
||||
sys.stdout.reconfigure(encoding='utf-8', errors='replace')
|
||||
sys.stderr.reconfigure(encoding='utf-8', errors='replace')
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
from keycheck_common import (
|
||||
acquire_file_lock,
|
||||
append_checked,
|
||||
append_jsonl,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
env_int,
|
||||
ensure_output_files as ensure_private_output_files,
|
||||
finding_detector_secret_hash,
|
||||
iter_findings,
|
||||
iter_bounded_text_lines,
|
||||
keycheck_input_mode,
|
||||
load_known_statuses,
|
||||
load_checked_statuses,
|
||||
mask_secret,
|
||||
now_iso,
|
||||
physical_jsonl_segments,
|
||||
private_append_writer,
|
||||
private_atomic_writer,
|
||||
reconcile_keycheck_jsonl_segments,
|
||||
record_validation_result,
|
||||
recover_status_transaction,
|
||||
release_file_lock,
|
||||
repair_keycheck_jsonl_tail,
|
||||
require_provider_authority,
|
||||
rotate_jsonl_if_needed,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
sha256_text,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from runtime_security import reject_reparse_components, require_private_file
|
||||
|
||||
# --- Конфигурация ---
|
||||
SERVICE = "openrouter"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
ALIVE_FILE = os.path.join(OUTPUT_DIR, "openrouterAlive.txt")
|
||||
DEAD_FILE = os.path.join(OUTPUT_DIR, "openrouterDead.txt")
|
||||
LIMITED_FILE = os.path.join(OUTPUT_DIR, "openrouterLimited.txt")
|
||||
NO_BALANCE_FILE = os.path.join(OUTPUT_DIR, "openrouterNoBalance.txt")
|
||||
NETWORK_FILE = os.path.join(OUTPUT_DIR, "openrouterNetwork.txt")
|
||||
UNKNOWN_FILE = os.path.join(OUTPUT_DIR, "openrouterUnknown.txt")
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "openrouterChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "openrouterResults.jsonl")
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
STATUS_FILES = {
|
||||
"VALID": ALIVE_FILE,
|
||||
"NO_BALANCE": NO_BALANCE_FILE,
|
||||
"DEAD": DEAD_FILE,
|
||||
"LIMITED": LIMITED_FILE,
|
||||
"NETWORK": NETWORK_FILE,
|
||||
"UNKNOWN": UNKNOWN_FILE,
|
||||
}
|
||||
|
||||
CREDITS_URL = "https://openrouter.ai/api/v1/credits"
|
||||
|
||||
# --- Вспомогательные функции ---
|
||||
def load_set_from_file(filepath):
|
||||
"""Загружает ключи из файла в set для быстрой проверки."""
|
||||
if not os.path.exists(filepath):
|
||||
return set()
|
||||
return {line.strip().split(':')[0] for line in iter_bounded_text_lines(filepath) if line.strip()}
|
||||
|
||||
def ensure_output_files():
|
||||
ensure_private_output_files((CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()))
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def legacy_status_key(line):
|
||||
value = line.strip()
|
||||
if not value:
|
||||
return None
|
||||
if '\t' in value:
|
||||
return value.split('\t', 1)[0].strip()
|
||||
return value.split(':', 1)[0].strip()
|
||||
|
||||
|
||||
def load_openrouter_keys(path):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return set()
|
||||
if not os.path.exists(path):
|
||||
return set()
|
||||
return {key for key in (legacy_status_key(line) for line in iter_bounded_text_lines(path)) if key}
|
||||
|
||||
|
||||
def iter_plain_openrouter_keys(paths):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
seen = set()
|
||||
items = []
|
||||
for path in paths or []:
|
||||
if not path or not os.path.exists(path):
|
||||
continue
|
||||
for line in iter_bounded_text_lines(path):
|
||||
key = legacy_status_key(line)
|
||||
if not key or key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
items.append({'raw': key, 'source': path, 'finding': {}})
|
||||
for item in items:
|
||||
yield item
|
||||
|
||||
|
||||
def retry_plain_files(args):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return []
|
||||
files = list(args.plain or [])
|
||||
if args.recheck_all:
|
||||
files.extend(STATUS_FILES.values())
|
||||
else:
|
||||
if args.retry_valid:
|
||||
files.append(ALIVE_FILE)
|
||||
if args.retry_no_balance:
|
||||
files.append(NO_BALANCE_FILE)
|
||||
if args.retry_limited:
|
||||
files.append(LIMITED_FILE)
|
||||
if args.retry_network:
|
||||
files.append(NETWORK_FILE)
|
||||
if args.retry_unknown:
|
||||
files.append(UNKNOWN_FILE)
|
||||
out = []
|
||||
seen = set()
|
||||
for path in files:
|
||||
if path and path not in seen:
|
||||
seen.add(path)
|
||||
out.append(path)
|
||||
return out
|
||||
|
||||
|
||||
def migrate_legacy_checked():
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
for key in sorted(load_openrouter_keys(ALIVE_FILE)):
|
||||
if key not in checked:
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, {'status': 'VALID', 'message': 'legacy alive status migration'}, 'legacy:openrouterAlive.txt', {}, 'OpenRouter', 'legacy_status')
|
||||
append_checked(CHECKED_FILE, key, 'VALID')
|
||||
checked[key] = 'VALID'
|
||||
for key in sorted(load_openrouter_keys(DEAD_FILE)):
|
||||
if key not in checked:
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, {'status': 'DEAD', 'message': 'legacy dead status migration'}, 'legacy:openrouterDead.txt', {}, 'OpenRouter', 'legacy_status')
|
||||
append_checked(CHECKED_FILE, key, 'DEAD')
|
||||
checked[key] = 'DEAD'
|
||||
|
||||
|
||||
def _legacy_alive_checked_at(path):
|
||||
details = os.stat(path, follow_symlinks=False)
|
||||
return datetime.fromtimestamp(details.st_mtime, timezone.utc).isoformat(timespec='seconds')
|
||||
|
||||
|
||||
def _legacy_balance_event_payload(key, balance, checked_at):
|
||||
balance_text = f'{balance:.6f}'
|
||||
key_hash = sha256_text(key)
|
||||
event_id = sha256_text('|'.join([
|
||||
SERVICE,
|
||||
'legacy_status',
|
||||
'openrouterAlive.txt',
|
||||
key_hash,
|
||||
'NO_BALANCE',
|
||||
balance_text,
|
||||
]))
|
||||
return {
|
||||
'key_masked': mask_secret(key),
|
||||
'key_hash': key_hash,
|
||||
'secret_hash': key_hash,
|
||||
'detector_secret_hash': finding_detector_secret_hash({}),
|
||||
'finding_uid': '',
|
||||
'detector': 'OpenRouter',
|
||||
'source': 'legacy:openrouterAlive.txt',
|
||||
'finding': {},
|
||||
'checked_at': checked_at,
|
||||
'result_source': 'legacy_status',
|
||||
'status': 'NO_BALANCE',
|
||||
'message': f'credits={balance_text}',
|
||||
'event_id': event_id,
|
||||
}
|
||||
|
||||
|
||||
def _publish_legacy_no_balance(moved):
|
||||
lock_path = f'{NO_BALANCE_FILE}.lock'
|
||||
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
||||
try:
|
||||
require_private_file(NO_BALANCE_FILE)
|
||||
existing = load_openrouter_keys(NO_BALANCE_FILE)
|
||||
with private_append_writer(NO_BALANCE_FILE) as handle:
|
||||
for key, balance in moved:
|
||||
if key in existing:
|
||||
continue
|
||||
balance_text = f'{balance:.6f}'
|
||||
handle.write(f'{key}\tNO_BALANCE\tcredits={balance_text}\tmigrated_from_alive\n')
|
||||
existing.add(key)
|
||||
finally:
|
||||
release_file_lock(lock, lock_path)
|
||||
|
||||
|
||||
def _existing_legacy_event_ids(expected):
|
||||
found = set()
|
||||
max_line_bytes = max(1024, env_int('KEYCHECK_INPUT_MAX_LINE_BYTES', 16 * 1024 * 1024))
|
||||
max_file_bytes = max(
|
||||
max_line_bytes,
|
||||
max(1, env_int('KEYCHECK_INPUT_LIST_MAX_BYTES', 32 * 1024 * 1024)),
|
||||
max(0, env_int('KEYCHECK_RESULTS_MAX_MB', 32)) * 1024 * 1024 + max_line_bytes,
|
||||
)
|
||||
paths = [path for _, path in physical_jsonl_segments(RESULTS_FILE)]
|
||||
if os.path.isfile(RESULTS_FILE):
|
||||
paths.append(os.path.abspath(RESULTS_FILE))
|
||||
for path in paths:
|
||||
reject_reparse_components(path)
|
||||
if os.path.getsize(path) > max_file_bytes:
|
||||
raise RuntimeError(f'OpenRouter result file exceeds the bounded migration scan size: {path}')
|
||||
with open(path, 'rb') as handle:
|
||||
while True:
|
||||
raw_line = handle.readline(max_line_bytes + 1)
|
||||
if not raw_line:
|
||||
break
|
||||
if len(raw_line) > max_line_bytes:
|
||||
raise RuntimeError(f'OpenRouter result line exceeds the bounded migration scan size: {path}')
|
||||
if not raw_line.endswith(b'\n'):
|
||||
raise RuntimeError(f'torn OpenRouter result line during legacy migration: {path}')
|
||||
try:
|
||||
payload = json.loads(raw_line.decode('utf-8', errors='replace'))
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
event_id = str(payload.get('event_id') or '') if isinstance(payload, dict) else ''
|
||||
if event_id not in expected:
|
||||
continue
|
||||
wanted = expected[event_id]
|
||||
for field in ('key_hash', 'status', 'source', 'result_source'):
|
||||
if str(payload.get(field) or '') != str(wanted.get(field) or ''):
|
||||
raise RuntimeError(f'conflicting OpenRouter legacy migration event: {event_id}')
|
||||
found.add(event_id)
|
||||
return found
|
||||
|
||||
|
||||
def _publish_legacy_balance_events(moved, checked_at):
|
||||
payloads = [_legacy_balance_event_payload(key, balance, checked_at) for key, balance in moved]
|
||||
expected = {payload['event_id']: payload for payload in payloads}
|
||||
lock_path = f'{RESULTS_FILE}.lock'
|
||||
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
||||
try:
|
||||
require_private_file(RESULTS_FILE)
|
||||
repair_keycheck_jsonl_tail(RESULTS_FILE)
|
||||
reconcile_keycheck_jsonl_segments(RESULTS_FILE)
|
||||
existing = _existing_legacy_event_ids(expected)
|
||||
max_bytes = max(0, env_int('KEYCHECK_RESULTS_MAX_MB', 32)) * 1024 * 1024
|
||||
for payload in payloads:
|
||||
event_id = payload['event_id']
|
||||
if event_id in existing:
|
||||
continue
|
||||
rotate_jsonl_if_needed(RESULTS_FILE, max_bytes)
|
||||
with private_append_writer(RESULTS_FILE) as handle:
|
||||
handle.write(json.dumps(payload, ensure_ascii=False, default=str) + '\n')
|
||||
existing.add(event_id)
|
||||
finally:
|
||||
release_file_lock(lock, lock_path)
|
||||
|
||||
|
||||
def _publish_legacy_checked(moved, checked_at):
|
||||
lock_path = f'{CHECKED_FILE}.lock'
|
||||
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
||||
try:
|
||||
require_private_file(CHECKED_FILE)
|
||||
existing = set()
|
||||
for line in iter_bounded_text_lines(CHECKED_FILE):
|
||||
parts = line.rstrip('\r\n').split('\t')
|
||||
if len(parts) >= 2:
|
||||
existing.add((parts[0], parts[1]))
|
||||
with private_append_writer(CHECKED_FILE) as handle:
|
||||
for key, _ in moved:
|
||||
identity = (key, 'NO_BALANCE')
|
||||
if identity in existing:
|
||||
continue
|
||||
handle.write(f'{key}\tNO_BALANCE\t{checked_at}\n')
|
||||
existing.add(identity)
|
||||
finally:
|
||||
release_file_lock(lock, lock_path)
|
||||
|
||||
|
||||
def _rewrite_legacy_alive(keep):
|
||||
with private_atomic_writer(ALIVE_FILE, binary=True, suffix='.legacy.tmp') as handle:
|
||||
for line in keep:
|
||||
handle.write(line.encode('utf-8'))
|
||||
|
||||
|
||||
def migrate_legacy_alive_balances():
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
lock_path = f'{ALIVE_FILE}.lock'
|
||||
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
||||
try:
|
||||
if not os.path.exists(ALIVE_FILE):
|
||||
return
|
||||
require_private_file(ALIVE_FILE)
|
||||
checked_at = _legacy_alive_checked_at(ALIVE_FILE)
|
||||
keep = []
|
||||
moved = []
|
||||
seen = set()
|
||||
for line in iter_bounded_text_lines(ALIVE_FILE):
|
||||
text = line.strip()
|
||||
if not text:
|
||||
keep.append(line)
|
||||
continue
|
||||
key = legacy_status_key(text)
|
||||
balance = None
|
||||
if ':' in text and '\t' not in text:
|
||||
try:
|
||||
balance = float(text.rsplit(':', 1)[1])
|
||||
except ValueError:
|
||||
balance = None
|
||||
if key and balance is not None and balance <= 0:
|
||||
if key not in seen:
|
||||
moved.append((key, balance))
|
||||
seen.add(key)
|
||||
else:
|
||||
keep.append(line)
|
||||
if not moved:
|
||||
return
|
||||
_publish_legacy_no_balance(moved)
|
||||
_publish_legacy_balance_events(moved, checked_at)
|
||||
_publish_legacy_checked(moved, checked_at)
|
||||
_rewrite_legacy_alive(keep)
|
||||
finally:
|
||||
release_file_lock(lock, lock_path)
|
||||
|
||||
|
||||
def load_proxies(proxy_file=None):
|
||||
"""Загружает и подготавливает прокси."""
|
||||
proxy_file = proxy_file or PROXY_FILE
|
||||
if not os.path.exists(proxy_file):
|
||||
print("ℹ️ Файл proxy.txt не найден, запросы будут идти напрямую.")
|
||||
return None
|
||||
proxies = []
|
||||
with open(proxy_file, 'r') as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line: continue
|
||||
try:
|
||||
ip, port, login, password = line.split(':')
|
||||
proxy_url = f"http://{login}:{password}@{ip}:{port}"
|
||||
proxies.append({"http": proxy_url, "https": proxy_url})
|
||||
except ValueError:
|
||||
print(f"⚠️ Неверный формат прокси: '{line}'. Пропускаем.")
|
||||
if not proxies:
|
||||
print("⚠️ Файл proxy.txt пуст. Запросы будут идти напрямую.")
|
||||
return None
|
||||
print(f"✅ Загружено {len(proxies)} прокси.")
|
||||
return cycle(proxies)
|
||||
|
||||
def write_result(key, result, source_line, finding=None, previous_status=None):
|
||||
status = result.get('status') or 'UNKNOWN'
|
||||
credits = result.get('remaining_credits')
|
||||
extra = f"credits={credits:.6f}" if isinstance(credits, (int, float)) else source_line
|
||||
checked_at = now_iso()
|
||||
result = {**result, 'checked_at': result.get('checked_at') or checked_at}
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source_line, finding, 'OpenRouter')
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status, result.get('message', ''), extra,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source_line, finding, 'OpenRouter')
|
||||
|
||||
# --- Функция проверки ---
|
||||
def check_openrouter_key(key, proxy):
|
||||
"""Проверяет один ключ OpenRouter и возвращает normalized result."""
|
||||
headers = {"Authorization": f"Bearer {key}"}
|
||||
try:
|
||||
resp = requests.get(CREDITS_URL, headers=headers, proxies=proxy, timeout=15)
|
||||
except requests.exceptions.RequestException as e:
|
||||
return {'status': 'NETWORK', 'message': str(e)}
|
||||
|
||||
if resp.status_code == 200:
|
||||
try:
|
||||
data = resp.json().get("data", {})
|
||||
total = float(data.get("total_credits", 0.0) or 0.0)
|
||||
used = float(data.get("total_usage", 0.0) or 0.0)
|
||||
except (TypeError, ValueError, json.JSONDecodeError) as exc:
|
||||
return {'status': 'UNKNOWN', 'http_status': resp.status_code, 'message': f'invalid credits response: {exc}'}
|
||||
remaining = total - used
|
||||
if remaining > 0:
|
||||
return {'status': 'VALID', 'remaining_credits': remaining, 'message': f'credits={remaining:.6f}'}
|
||||
return {'status': 'NO_BALANCE', 'remaining_credits': remaining, 'message': f'credits={remaining:.6f}'}
|
||||
if resp.status_code in (401, 403):
|
||||
return {'status': 'DEAD', 'http_status': resp.status_code, 'message': resp.text[:1000]}
|
||||
if resp.status_code == 429:
|
||||
return {'status': 'LIMITED', 'http_status': resp.status_code, 'message': resp.text[:1000]}
|
||||
if 500 <= resp.status_code <= 599:
|
||||
return {'status': 'NETWORK', 'http_status': resp.status_code, 'message': resp.text[:1000]}
|
||||
return {'status': 'UNKNOWN', 'http_status': resp.status_code, 'message': resp.text[:1000]}
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description='OpenRouter key checker')
|
||||
parser.add_argument('--input', default=INPUT_FILE)
|
||||
parser.add_argument('--plain', action='append', default=[])
|
||||
parser.add_argument('--proxy-file', default=PROXY_FILE)
|
||||
parser.add_argument('--max-keys', type=int, default=0)
|
||||
parser.add_argument('--retry-network', action='store_true')
|
||||
parser.add_argument('--retry-limited', action='store_true')
|
||||
parser.add_argument('--retry-unknown', action='store_true')
|
||||
parser.add_argument('--retry-no-balance', action='store_true')
|
||||
parser.add_argument('--retry-valid', action='store_true')
|
||||
parser.add_argument('--recheck-all', action='store_true')
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
# --- Основной процесс ---
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_output_files()
|
||||
migrate_legacy_alive_balances()
|
||||
migrate_legacy_checked()
|
||||
print("--- 🚀 Запуск чекера ключей OpenRouter 🚀 ---")
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
|
||||
checked_statuses = load_checked_statuses(CHECKED_FILE)
|
||||
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES)
|
||||
known_keys = set(known_statuses)
|
||||
for path in STATUS_FILES.values():
|
||||
known_keys.update(load_openrouter_keys(path))
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add('NETWORK')
|
||||
if args.retry_limited:
|
||||
retry_statuses.add('LIMITED')
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add('UNKNOWN')
|
||||
if args.retry_no_balance:
|
||||
retry_statuses.add('NO_BALANCE')
|
||||
if args.retry_valid:
|
||||
retry_statuses.add('VALID')
|
||||
print(f"📖 Загружено: {len(load_openrouter_keys(ALIVE_FILE))} живых ключей, {len(known_keys)} классифицированных ключей.")
|
||||
|
||||
if keycheck_input_mode() == 'jsonl' and not os.path.exists(args.input):
|
||||
print(f"❌ Файл с секретами {args.input} не найден. Завершение.")
|
||||
return
|
||||
|
||||
processed = 0
|
||||
def candidates():
|
||||
for item in iter_findings(args.input, ["OpenRouter"]):
|
||||
key = item.get("raw") or ""
|
||||
if key:
|
||||
yield item
|
||||
seen = set()
|
||||
for item in iter_plain_openrouter_keys(retry_plain_files(args)):
|
||||
key = item.get("raw") or ""
|
||||
if key and key not in seen:
|
||||
seen.add(key)
|
||||
yield item
|
||||
|
||||
for item in candidates():
|
||||
key = item.get("raw") or ""
|
||||
if not key:
|
||||
continue
|
||||
|
||||
source = item.get("source") or args.input
|
||||
finding = item.get("finding") or {}
|
||||
if should_skip_key(
|
||||
key, checked_statuses, known_keys, args, retry_statuses,
|
||||
service=SERVICE, source=source, finding=finding, detector='OpenRouter', known_statuses=known_statuses,
|
||||
):
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
|
||||
print(f"\n[{processed}] 🎯 Новый кандидат: {key[:8]}...{key[-4:]} from {source}")
|
||||
current_proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_openrouter_key(key, current_proxy)
|
||||
print(f" STATUS: {result.get('status')} | {str(result.get('message', ''))[:200]}")
|
||||
previous_status = known_statuses.get(key) or checked_statuses.get(key)
|
||||
write_result(key, result, source, finding, previous_status)
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = result.get('status')
|
||||
|
||||
print("\n--- ✅ Проверка завершена. ---")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,189 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import os
|
||||
|
||||
|
||||
SUPPORTED_PROVIDERS = ("deepseek", "zai", "qwen", "kimi")
|
||||
DEFAULT_PROVIDER_ORDER = SUPPORTED_PROVIDERS
|
||||
AMBIGUOUS_QWEN_DEEPSEEK_HINT = "ambiguous_qwen_deepseek"
|
||||
AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk"
|
||||
|
||||
|
||||
def split_csv(value):
|
||||
if not value:
|
||||
return []
|
||||
if isinstance(value, str):
|
||||
values = value.split(",")
|
||||
else:
|
||||
values = value
|
||||
return [str(item).strip().lower() for item in values if str(item).strip()]
|
||||
|
||||
|
||||
def unique_supported(values):
|
||||
output = []
|
||||
seen = set()
|
||||
for value in values:
|
||||
provider = str(value or "").strip().lower()
|
||||
if provider in SUPPORTED_PROVIDERS and provider not in seen:
|
||||
seen.add(provider)
|
||||
output.append(provider)
|
||||
return output
|
||||
|
||||
|
||||
def providers_for_hint(hint):
|
||||
hint = str(hint or "").strip().lower()
|
||||
if hint == AMBIGUOUS_QWEN_DEEPSEEK_HINT:
|
||||
return ["qwen", "deepseek"]
|
||||
if hint == AMBIGUOUS_GENERIC_SK_HINT:
|
||||
return list(SUPPORTED_PROVIDERS)
|
||||
return [hint] if hint in SUPPORTED_PROVIDERS else []
|
||||
|
||||
|
||||
def detector_provider(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return ""
|
||||
extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {}
|
||||
names = {
|
||||
str(finding.get("DetectorName") or finding.get("DetectorType") or "").strip().lower(),
|
||||
str(extra.get("name") or "").strip().lower(),
|
||||
}
|
||||
mappings = (
|
||||
("deepseek", {"deepseek", "deepseekapikey", "deepseek_api_key"}),
|
||||
("zai", {"zaiglm"}),
|
||||
("qwen", {"qwendashscope", "qwen_dashscope", "qwen", "dashscope"}),
|
||||
("kimi", {"kimimoonshot", "moonshotai", "moonshot", "kimi"}),
|
||||
)
|
||||
for provider, detectors in mappings:
|
||||
if names & detectors:
|
||||
return provider
|
||||
return ""
|
||||
|
||||
|
||||
def ordered_providers(finding=None, hint="", origin_service="", configured_order=None):
|
||||
context = finding.get("ScannerContext") if isinstance(finding, dict) and isinstance(
|
||||
finding.get("ScannerContext"), dict
|
||||
) else {}
|
||||
hint = str(hint or context.get("provider_hint") or "").strip().lower()
|
||||
compatible = unique_supported(context.get("provider_candidates") or providers_for_hint(hint))
|
||||
if not compatible:
|
||||
compatible = providers_for_hint(hint)
|
||||
if not compatible:
|
||||
compatible = list(SUPPORTED_PROVIDERS)
|
||||
|
||||
configured = unique_supported(
|
||||
configured_order
|
||||
if configured_order is not None
|
||||
else split_csv(os.getenv("KEYCHECK_PROVIDER_RESOLUTION_ORDER"))
|
||||
)
|
||||
base_order = configured or list(DEFAULT_PROVIDER_ORDER)
|
||||
origin = str(origin_service or detector_provider(finding)).strip().lower()
|
||||
ordered = []
|
||||
if origin in compatible:
|
||||
ordered.append(origin)
|
||||
ordered.extend(provider for provider in base_order if provider in compatible)
|
||||
ordered.extend(provider for provider in compatible if provider not in ordered)
|
||||
return unique_supported(ordered)
|
||||
|
||||
|
||||
def provider_result_outcome(result):
|
||||
result = result if isinstance(result, dict) else {}
|
||||
status = str(result.get("status") or "UNKNOWN").strip().upper()
|
||||
if result.get("authenticated") is True or status in ("VALID", "ALIVE"):
|
||||
return "match"
|
||||
if result.get("candidate_rejected") or status in (
|
||||
"DEAD", "INVALID", "EXPIRED", "LEAKED_REVOKED", "INVALID_OR_REVOKED",
|
||||
):
|
||||
return "no_match"
|
||||
return "retry"
|
||||
|
||||
|
||||
def bounded_attempt(provider, result, outcome):
|
||||
result = result if isinstance(result, dict) else {}
|
||||
error = result.get("error") if isinstance(result.get("error"), dict) else {}
|
||||
return {
|
||||
"provider": provider,
|
||||
"outcome": outcome,
|
||||
"status": str(result.get("status") or "UNKNOWN").upper(),
|
||||
"authenticated": bool(result.get("authenticated")),
|
||||
"http_status": int(result.get("http_status") or error.get("http_status") or 0),
|
||||
"business_code": str(result.get("business_code") or error.get("code") or "")[:80],
|
||||
"region": str(result.get("region") or "")[:160],
|
||||
"message": str(result.get("message") or "").replace("\r", " ").replace("\n", " ")[:300],
|
||||
}
|
||||
|
||||
|
||||
def default_provider_probe(provider, key, proxy, timeout, debug=False):
|
||||
if provider == "deepseek":
|
||||
from keycheckers.deepseek import deepseekKeycheck
|
||||
|
||||
return deepseekKeycheck.check_key(key, proxy, timeout)
|
||||
if provider == "zai":
|
||||
from keycheckers.zai import zaiKeycheck
|
||||
|
||||
return zaiKeycheck.check_key(
|
||||
key, zaiKeycheck.base_urls_from_environment(), proxy, timeout, debug,
|
||||
)
|
||||
if provider == "qwen":
|
||||
from keycheckers.qwen import qwenKeycheck
|
||||
|
||||
custom = qwenKeycheck.split_csv(
|
||||
os.getenv("QWEN_BASE_URLS") or os.getenv("DASHSCOPE_BASE_URLS")
|
||||
)
|
||||
base_urls = qwenKeycheck.unique_ordered([*custom, *qwenKeycheck.DEFAULT_BASE_URLS])
|
||||
return qwenKeycheck.check_key(key, base_urls, bool(custom), proxy, timeout, debug)
|
||||
if provider == "kimi":
|
||||
from keycheckers.kimi import kimiKeycheck
|
||||
|
||||
custom = kimiKeycheck.split_csv(
|
||||
os.getenv("KIMI_BASE_URLS") or os.getenv("MOONSHOT_BASE_URLS")
|
||||
)
|
||||
base_urls = kimiKeycheck.unique_ordered([*custom, *kimiKeycheck.DEFAULT_BASE_URLS])
|
||||
return kimiKeycheck.check_key(key, base_urls, proxy, timeout, debug)
|
||||
raise ValueError(f"unsupported provider resolution adapter: {provider}")
|
||||
|
||||
|
||||
def resolve_provider_key(
|
||||
key, finding=None, proxy=None, timeout=15, debug=False, hint="",
|
||||
origin_service="", configured_order=None, probe=None,
|
||||
):
|
||||
providers = ordered_providers(finding, hint, origin_service, configured_order)
|
||||
probe = probe or default_provider_probe
|
||||
attempts = []
|
||||
retry_results = []
|
||||
for provider in providers:
|
||||
result = probe(provider, key, proxy, timeout, debug)
|
||||
result = result if isinstance(result, dict) else {"status": "UNKNOWN"}
|
||||
outcome = provider_result_outcome(result)
|
||||
attempts.append(bounded_attempt(provider, result, outcome))
|
||||
if outcome == "match":
|
||||
return {
|
||||
**result,
|
||||
"resolved_provider": provider,
|
||||
"provider_resolution": "matched",
|
||||
"provider_resolution_order": providers,
|
||||
"provider_resolution_attempts": attempts,
|
||||
"result_source": "provider_resolution",
|
||||
}
|
||||
if outcome == "retry":
|
||||
retry_results.append(result)
|
||||
|
||||
if retry_results:
|
||||
selected = retry_results[0]
|
||||
return {
|
||||
**selected,
|
||||
"provider_resolution": "retry",
|
||||
"provider_resolution_order": providers,
|
||||
"provider_resolution_attempts": attempts,
|
||||
"result_source": "provider_resolution",
|
||||
"message": str(selected.get("message") or "provider resolution remains inconclusive")[:1000],
|
||||
}
|
||||
return {
|
||||
"status": "DEAD",
|
||||
"provider_resolution": "exhausted",
|
||||
"provider_resolution_order": providers,
|
||||
"provider_resolution_attempts": attempts,
|
||||
"result_source": "provider_resolution",
|
||||
"message": "all compatible providers rejected the credential",
|
||||
}
|
||||
@@ -0,0 +1,203 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
record_validation_result,
|
||||
recover_status_transaction,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from keycheckers.provider_resolution import (
|
||||
AMBIGUOUS_GENERIC_SK_HINT,
|
||||
AMBIGUOUS_QWEN_DEEPSEEK_HINT,
|
||||
resolve_provider_key,
|
||||
)
|
||||
|
||||
|
||||
SERVICE = "provider_resolver"
|
||||
DETECTOR = "ProviderResolver"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "providerResolverChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "providerResolverResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "providerResolverAlive.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "providerResolverNoBalance.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "providerResolverDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "providerResolverRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "providerResolverLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "providerResolverNetwork.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "providerResolverNoContext.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "providerResolverUnknown.txt"),
|
||||
}
|
||||
|
||||
RESOLVABLE_KEY_REGEX = re.compile(
|
||||
r"(?<![A-Za-z0-9_.-])(?:"
|
||||
r"(?:zai|sk)-[A-Za-z0-9][A-Za-z0-9_-]{20,505}|"
|
||||
r"[A-Fa-f0-9]{32}\.[A-Za-z0-9_-]{16,128}"
|
||||
r")(?![A-Za-z0-9_.-])"
|
||||
)
|
||||
FOREIGN_KEY_PREFIXES = ("sk-ant-", "sk-or-", "sk-proj-", "sk-svcacct-", "sk-admin-")
|
||||
AMBIGUOUS_HINTS = {AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT}
|
||||
|
||||
|
||||
def key_rejection_reason(key):
|
||||
value = str(key or "")
|
||||
try:
|
||||
encoded = value.encode("utf-8", errors="strict")
|
||||
except UnicodeEncodeError:
|
||||
return "candidate is not valid UTF-8"
|
||||
if len(encoded) > 512:
|
||||
return "candidate exceeds the 512-byte key limit"
|
||||
if value.startswith(FOREIGN_KEY_PREFIXES):
|
||||
return "candidate has a foreign provider prefix"
|
||||
if not RESOLVABLE_KEY_REGEX.fullmatch(value):
|
||||
return "candidate does not match a bounded resolvable provider-key format"
|
||||
return ""
|
||||
|
||||
|
||||
def iter_candidate_keys(input_file):
|
||||
detector_names = [
|
||||
"ProviderResolver", "CustomRegex", "QwenDashScope", "Qwen_DashScope",
|
||||
"Qwen", "DashScope", "DeepSeek", "DeepSeekApiKey", "DeepSeek_API_Key",
|
||||
"KimiMoonshot", "MoonshotAI", "Moonshot", "Kimi", "ZaiGLM",
|
||||
"qwendashscope", "qwen_dashscope", "qwen", "dashscope", "deepseek",
|
||||
"deepseekapikey", "deepseek_api_key", "kimimoonshot", "moonshotai",
|
||||
"moonshot", "kimi", "zaiglm",
|
||||
]
|
||||
for item in iter_findings(input_file, detector_names):
|
||||
finding = item.get("finding") or {}
|
||||
key = item.get("credential_secret_text") or ""
|
||||
if not key:
|
||||
for value in (item.get("raw"), item.get("raw_v2"), finding.get("Raw"), finding.get("RawV2")):
|
||||
match = RESOLVABLE_KEY_REGEX.search(str(value or ""))
|
||||
if match:
|
||||
key = match.group(0)
|
||||
break
|
||||
if not key or key_rejection_reason(key):
|
||||
continue
|
||||
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
|
||||
metadata_hint = ""
|
||||
active_metadata = item.get("candidate_metadata")
|
||||
if isinstance(active_metadata, dict):
|
||||
metadata_hint = str(active_metadata.get("provider_hint") or "")
|
||||
metadata_candidates = active_metadata.get("provider_candidates")
|
||||
if metadata_hint or isinstance(metadata_candidates, list):
|
||||
context = dict(context)
|
||||
if metadata_hint:
|
||||
context.setdefault("provider_hint", metadata_hint)
|
||||
if isinstance(metadata_candidates, list):
|
||||
context.setdefault("provider_candidates", metadata_candidates)
|
||||
finding = dict(finding)
|
||||
finding["ScannerContext"] = context
|
||||
hint = str(context.get("provider_hint") or metadata_hint or AMBIGUOUS_GENERIC_SK_HINT)
|
||||
if hint not in AMBIGUOUS_HINTS:
|
||||
hint = AMBIGUOUS_GENERIC_SK_HINT
|
||||
yield key, item.get("source") or input_file, finding, hint
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status,
|
||||
result.get("message", ""), result.get("resolved_provider") or source,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def retry_statuses_from_args(args):
|
||||
statuses = set()
|
||||
for enabled, status in (
|
||||
(args.retry_network, "NETWORK"),
|
||||
(args.retry_limited, "LIMITED"),
|
||||
(args.retry_unknown, "UNKNOWN"),
|
||||
(args.retry_restricted, "RESTRICTED"),
|
||||
(args.retry_no_balance, "NO_BALANCE"),
|
||||
(args.retry_valid, "VALID"),
|
||||
):
|
||||
if enabled:
|
||||
statuses.add(status)
|
||||
return statuses
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Ambiguous generic provider key resolver")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = retry_statuses_from_args(args)
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, source, finding, hint in iter_candidate_keys(args.input):
|
||||
if should_skip_key(
|
||||
key, checked, known, args, retry_statuses, service=SERVICE,
|
||||
source=source, finding=finding, detector=DETECTOR,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] Ambiguous provider candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = resolve_provider_key(
|
||||
key, finding, proxy, args.timeout, args.debug, hint=hint,
|
||||
)
|
||||
print(
|
||||
f" STATUS: {result['status']} provider={result.get('resolved_provider', '')} "
|
||||
f"| {result.get('message', '')[:200]}"
|
||||
)
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,773 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
try:
|
||||
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
||||
sys.stderr.reconfigure(encoding="utf-8", errors="replace")
|
||||
except (AttributeError, OSError, ValueError):
|
||||
pass
|
||||
|
||||
from keycheck_common import (
|
||||
combined_provider_routing_hint,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
provider_routing_database_failed,
|
||||
read_plain_keys,
|
||||
record_validation_result,
|
||||
recover_status_transaction,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from keycheckers.provider_resolution import resolve_provider_key
|
||||
|
||||
|
||||
SERVICE = "qwen"
|
||||
DETECTOR = "QwenDashScope"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "qwenChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "qwenResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "qwenAlive.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "qwenDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "qwenRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "qwenLimited.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "qwenNoBalance.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "qwenNoContext.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "qwenNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "qwenUnknown.txt"),
|
||||
}
|
||||
|
||||
QWEN_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope", "dashscope", "qwen"}
|
||||
QWEN_EXPLICIT_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope"}
|
||||
DEEPSEEK_EXPLICIT_DETECTOR_NAMES = {"deepseekapikey", "deepseek_api_key"}
|
||||
KIMI_EXPLICIT_DETECTOR_NAMES = {"kimimoonshot", "moonshotai"}
|
||||
QWEN_KEY_MAX_BYTES = 512
|
||||
QWEN_KEY_REGEX = re.compile(
|
||||
r"(?<![A-Za-z0-9_-])sk-(?:sp-)?[A-Za-z0-9][A-Za-z0-9_-]{20,505}(?![A-Za-z0-9_-])"
|
||||
)
|
||||
QWEN_OVERSIZED_KEY_PREFIX_REGEX = re.compile(
|
||||
r"(?<![A-Za-z0-9_-])sk-(?:sp-)?[A-Za-z0-9][A-Za-z0-9_-]{506}"
|
||||
)
|
||||
OVERLAPPING_QWEN_DEEPSEEK_REGEX = re.compile(r"sk-[a-z0-9]{32}")
|
||||
FOREIGN_QWEN_KEY_PREFIXES = ("sk-ant-", "sk-or-", "sk-proj-", "sk-svcacct-", "sk-admin-")
|
||||
OPENAI_LEGACY_KEY_MARKER = "T3BlbkFJ"
|
||||
QWEN_CONTEXT_REGEX = re.compile(
|
||||
r"(?:DASHSCOPE_API_KEY|QWEN_API_KEY|dashscope|qwen|model[_-]?studio|bailian)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
DEEPSEEK_CONTEXT_REGEX = re.compile(r"(?:DEEPSEEK_API_KEY|deepseek|api\.deepseek\.com)", re.IGNORECASE)
|
||||
KIMI_CONTEXT_REGEX = re.compile(
|
||||
r"(?:MOONSHOT_API_KEY|KIMI_API_KEY|api\.moonshot\.(?:ai|cn)|platform\.kimi\.(?:ai|com))",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
AMBIGUOUS_PROVIDER_HINT = "ambiguous_qwen_deepseek"
|
||||
AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk"
|
||||
GENERIC_SK_PROVIDERS = {"qwen", "deepseek", "kimi", "zai"}
|
||||
EXPLICIT_ASSIGNMENT_HINT_SOURCE = "explicit_assignment"
|
||||
CANDIDATE_PROVIDER_ROUTE_FIELD = "_keycheck_candidate_provider_route"
|
||||
DEFAULT_BASE_URLS = [
|
||||
"https://coding-intl.dashscope.aliyuncs.com/v1",
|
||||
"https://dashscope-intl.aliyuncs.com/compatible-mode/v1",
|
||||
"https://dashscope-us.aliyuncs.com/compatible-mode/v1",
|
||||
"https://dashscope.aliyuncs.com/compatible-mode/v1",
|
||||
"https://cn-hongkong.dashscope.aliyuncs.com/compatible-mode/v1",
|
||||
]
|
||||
MODEL_MARKERS = ("qwen", "qwq", "qvq", "wan", "text-embedding", "multimodal-embedding")
|
||||
CHAT_MODEL_PRIORITY = (
|
||||
"qwen-plus",
|
||||
"qwen-turbo",
|
||||
"qwen-max",
|
||||
"qwen3-235b-a22b",
|
||||
"qwen3-32b",
|
||||
"qwen2.5-72b-instruct",
|
||||
"qwen2.5-32b-instruct",
|
||||
"qwen2.5-14b-instruct",
|
||||
"qwen2.5-7b-instruct",
|
||||
"qwq-32b",
|
||||
)
|
||||
NON_CHAT_MODEL_MARKERS = ("embedding", "rerank", "wan", "image", "audio", "tts", "asr", "vision", "vl")
|
||||
|
||||
|
||||
def normalize_base_url(value):
|
||||
value = str(value or "").strip()
|
||||
if not value:
|
||||
return ""
|
||||
return value.rstrip("/")
|
||||
|
||||
|
||||
def split_csv(value):
|
||||
if not value:
|
||||
return []
|
||||
if isinstance(value, str):
|
||||
return [item.strip() for item in value.split(",") if item.strip()]
|
||||
return [str(item).strip() for item in value if str(item).strip()]
|
||||
|
||||
|
||||
def unique_ordered(values):
|
||||
seen = set()
|
||||
output = []
|
||||
for value in values:
|
||||
normalized = normalize_base_url(value)
|
||||
if normalized and normalized not in seen:
|
||||
seen.add(normalized)
|
||||
output.append(normalized)
|
||||
return output
|
||||
|
||||
|
||||
def endpoint_label(base_url):
|
||||
parsed = urlparse(base_url)
|
||||
return parsed.netloc or base_url
|
||||
|
||||
|
||||
def is_qwen_detector(value):
|
||||
return str(value or "").lower() in QWEN_DETECTOR_NAMES
|
||||
|
||||
|
||||
def custom_detector_name(data):
|
||||
if not isinstance(data, dict):
|
||||
return ""
|
||||
extra = data.get("ExtraData") if isinstance(data.get("ExtraData"), dict) else {}
|
||||
name = str(extra.get("name") or "")
|
||||
if str(data.get("DetectorName") or "").lower() == "customregex" and is_qwen_detector(name):
|
||||
return name
|
||||
return ""
|
||||
|
||||
|
||||
def finding_detector_names(data):
|
||||
if not isinstance(data, dict):
|
||||
return set()
|
||||
extra = data.get("ExtraData") if isinstance(data.get("ExtraData"), dict) else {}
|
||||
names = {
|
||||
str(data.get("DetectorName") or data.get("detector") or "").strip().lower(),
|
||||
str(extra.get("name") or "").strip().lower(),
|
||||
}
|
||||
return {name for name in names if name}
|
||||
|
||||
|
||||
def finding_has_explicit_detector(data, detector_names):
|
||||
if not isinstance(data, dict):
|
||||
return False
|
||||
if finding_detector_names(data) & set(detector_names):
|
||||
return True
|
||||
nested = data.get("finding")
|
||||
return isinstance(nested, dict) and bool(finding_detector_names(nested) & set(detector_names))
|
||||
|
||||
|
||||
def detector_name_from_finding(data):
|
||||
if not isinstance(data, dict):
|
||||
return ""
|
||||
if is_qwen_detector(data.get("DetectorName")):
|
||||
return data.get("DetectorName")
|
||||
custom_name = custom_detector_name(data)
|
||||
if custom_name:
|
||||
return custom_name
|
||||
if is_qwen_detector(data.get("detector")):
|
||||
return data.get("detector")
|
||||
|
||||
finding = data.get("finding")
|
||||
if isinstance(finding, dict):
|
||||
if is_qwen_detector(finding.get("DetectorName")):
|
||||
return finding.get("DetectorName")
|
||||
custom_name = custom_detector_name(finding)
|
||||
if custom_name:
|
||||
return custom_name
|
||||
return ""
|
||||
|
||||
|
||||
def key_from_text(*values):
|
||||
for value in values:
|
||||
for match in QWEN_KEY_REGEX.finditer(str(value or "")):
|
||||
key = match.group(0)
|
||||
if not key.startswith(FOREIGN_QWEN_KEY_PREFIXES) and OPENAI_LEGACY_KEY_MARKER not in key:
|
||||
return key
|
||||
return ""
|
||||
|
||||
|
||||
def qwen_key_rejection_reason(key):
|
||||
value = str(key or "")
|
||||
try:
|
||||
key_bytes = len(value.encode("utf-8", errors="strict"))
|
||||
except UnicodeEncodeError:
|
||||
return "candidate is not valid UTF-8"
|
||||
if key_bytes > QWEN_KEY_MAX_BYTES:
|
||||
return f"candidate exceeds the {QWEN_KEY_MAX_BYTES}-byte key limit"
|
||||
if OPENAI_LEGACY_KEY_MARKER in value:
|
||||
return "candidate is a recognizable OpenAI legacy key"
|
||||
if value.count("sk-") != 1:
|
||||
return "candidate contains multiple concatenated key prefixes"
|
||||
if not QWEN_KEY_REGEX.fullmatch(value) or value.startswith(FOREIGN_QWEN_KEY_PREFIXES):
|
||||
return "candidate does not match the bounded Qwen key format"
|
||||
return ""
|
||||
|
||||
|
||||
def is_qwen_key(key):
|
||||
return not qwen_key_rejection_reason(key)
|
||||
|
||||
|
||||
def finding_has_oversized_qwen_key(finding, *raw_values):
|
||||
values = list(raw_values)
|
||||
if isinstance(finding, dict):
|
||||
values.extend((finding.get("Raw"), finding.get("RawV2"), finding.get("raw"), finding.get("raw_v2")))
|
||||
nested = finding.get("finding")
|
||||
if isinstance(nested, dict):
|
||||
values.extend((nested.get("Raw"), nested.get("RawV2"), nested.get("raw"), nested.get("raw_v2")))
|
||||
return any(
|
||||
QWEN_OVERSIZED_KEY_PREFIX_REGEX.search(str(value or ""))
|
||||
for value in values
|
||||
)
|
||||
|
||||
|
||||
def warn_rejected_candidate(reason, source, key=""):
|
||||
reason = str(reason or "candidate rejected")
|
||||
source = str(source or "unknown")
|
||||
if key:
|
||||
reason = reason.replace(key, "***REDACTED***")
|
||||
source = source.replace(key, "***REDACTED***")
|
||||
reason = reason.replace("\r", " ").replace("\n", " ")[:300]
|
||||
source = source.replace("\r", " ").replace("\n", " ")[:300]
|
||||
print(f"Warning: skipped Qwen candidate from {source}: {reason}", flush=True)
|
||||
|
||||
|
||||
def warn_candidate_failure(reason, source, key=""):
|
||||
reason = str(reason or "candidate failure")
|
||||
source = str(source or "unknown")
|
||||
if key:
|
||||
reason = reason.replace(key, "***REDACTED***")
|
||||
source = source.replace(key, "***REDACTED***")
|
||||
reason = QWEN_KEY_REGEX.sub("***REDACTED***", reason).replace("\r", " ").replace("\n", " ")[:300]
|
||||
source = QWEN_KEY_REGEX.sub("***REDACTED***", source).replace("\r", " ").replace("\n", " ")[:300]
|
||||
print(f"Warning: Qwen candidate failure from {source}: {reason}", flush=True)
|
||||
|
||||
|
||||
def extract_key_from_finding(data):
|
||||
if not detector_name_from_finding(data):
|
||||
return ""
|
||||
if data.get("Raw") or data.get("RawV2"):
|
||||
return key_from_text(data.get("Raw"), data.get("RawV2"))
|
||||
if data.get("raw") or data.get("raw_v2"):
|
||||
return key_from_text(data.get("raw"), data.get("raw_v2"))
|
||||
finding = data.get("finding")
|
||||
if isinstance(finding, dict):
|
||||
return key_from_text(finding.get("Raw"), finding.get("RawV2"))
|
||||
return ""
|
||||
|
||||
|
||||
def finding_provider_routing_hint(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return ""
|
||||
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
|
||||
persisted_hint = context.get("provider_hint")
|
||||
if (
|
||||
context.get("provider_hint_source") == EXPLICIT_ASSIGNMENT_HINT_SOURCE
|
||||
and persisted_hint in (*GENERIC_SK_PROVIDERS, AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
|
||||
):
|
||||
return persisted_hint
|
||||
parts = [str(context.get(key) or "") for key in ("nearby", "file")]
|
||||
metadata = finding.get("SourceMetadata") if isinstance(finding.get("SourceMetadata"), dict) else {}
|
||||
data = metadata.get("Data") if isinstance(metadata.get("Data"), dict) else {}
|
||||
for details in data.values():
|
||||
if not isinstance(details, dict):
|
||||
continue
|
||||
parts.extend(str(details.get(key) or "") for key in ("file", "repository", "repo", "link", "image"))
|
||||
|
||||
text = "\n".join(parts)
|
||||
evidence = set()
|
||||
if QWEN_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, QWEN_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("qwen")
|
||||
if DEEPSEEK_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, DEEPSEEK_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("deepseek")
|
||||
if KIMI_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, KIMI_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("kimi")
|
||||
if persisted_hint == AMBIGUOUS_PROVIDER_HINT:
|
||||
evidence.update(("qwen", "deepseek"))
|
||||
elif persisted_hint == AMBIGUOUS_GENERIC_SK_HINT:
|
||||
evidence.update(GENERIC_SK_PROVIDERS)
|
||||
elif persisted_hint in GENERIC_SK_PROVIDERS:
|
||||
evidence.add(persisted_hint)
|
||||
if len(evidence) > 1:
|
||||
return AMBIGUOUS_PROVIDER_HINT if evidence == {"qwen", "deepseek"} else AMBIGUOUS_GENERIC_SK_HINT
|
||||
return next(iter(evidence)) if evidence else ""
|
||||
|
||||
|
||||
def finding_has_ambiguous_provider_hint(finding):
|
||||
return finding_provider_routing_hint(finding) in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def iter_candidate_keys(input_file, plain_files, trusted_retry_files=None):
|
||||
seen_plain = set()
|
||||
seen_candidates = set()
|
||||
routing_decisions = {}
|
||||
detector_names = [
|
||||
"QwenDashScope", "Qwen_DashScope", "qwendashscope", "qwen_dashscope",
|
||||
"Qwen", "DashScope", "qwen", "dashscope", "CustomRegex",
|
||||
]
|
||||
for item in iter_findings(input_file, detector_names):
|
||||
data = dict(item.get("finding") or {})
|
||||
data.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, None)
|
||||
candidate_metadata = item.get("candidate_metadata")
|
||||
persisted_route = ""
|
||||
if keycheck_input_mode() == "postgres" and isinstance(candidate_metadata, dict):
|
||||
persisted_route = str(candidate_metadata.get("provider_hint") or "").lower()
|
||||
if persisted_route == SERVICE:
|
||||
data[CANDIDATE_PROVIDER_ROUTE_FIELD] = SERVICE
|
||||
if finding_has_oversized_qwen_key(data, item.get("raw"), item.get("raw_v2")):
|
||||
warn_rejected_candidate(
|
||||
f"candidate exceeds the {QWEN_KEY_MAX_BYTES}-byte key limit",
|
||||
item.get("source") or input_file,
|
||||
)
|
||||
continue
|
||||
if persisted_route == SERVICE:
|
||||
key = key_from_text(
|
||||
item.get("raw"), item.get("raw_v2"),
|
||||
data.get("Raw"), data.get("RawV2"),
|
||||
)
|
||||
else:
|
||||
key = extract_key_from_finding(data)
|
||||
if key and is_qwen_key(key):
|
||||
if not key.startswith("sk-sp-"):
|
||||
if persisted_route == SERVICE:
|
||||
hint, lookup_failed = SERVICE, False
|
||||
else:
|
||||
local_hint = finding_provider_routing_hint(data)
|
||||
if key in routing_decisions:
|
||||
hint, lookup_failed = routing_decisions[key]
|
||||
if local_hint == AMBIGUOUS_PROVIDER_HINT or (
|
||||
local_hint and hint and local_hint != hint
|
||||
):
|
||||
hint = AMBIGUOUS_PROVIDER_HINT
|
||||
elif not hint:
|
||||
hint = local_hint
|
||||
routing_decisions[key] = (hint, lookup_failed)
|
||||
else:
|
||||
hint = combined_provider_routing_hint(key, local_hint)
|
||||
lookup_failed = provider_routing_database_failed()
|
||||
routing_decisions[key] = (hint, lookup_failed)
|
||||
if lookup_failed:
|
||||
message = "provider routing evidence lookup failed closed"
|
||||
warn_candidate_failure(message, item.get("source") or input_file, key)
|
||||
raise RuntimeError(message)
|
||||
if hint != "qwen" and not (
|
||||
keycheck_input_mode() == "postgres"
|
||||
and hint in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
|
||||
):
|
||||
continue
|
||||
seen_candidates.add(key)
|
||||
yield key, item.get("source") or input_file, data
|
||||
|
||||
for item in read_plain_keys(plain_files, QWEN_KEY_REGEX):
|
||||
key = item["key"]
|
||||
if not is_qwen_key(key) or not key.startswith("sk-sp-"):
|
||||
continue
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
seen_candidates.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
owned_retry_paths = {
|
||||
os.path.normcase(os.path.abspath(path)) for path in STATUS_FILES.values()
|
||||
}
|
||||
retry_files = [
|
||||
path for path in (trusted_retry_files or [])
|
||||
if os.path.normcase(os.path.abspath(path)) in owned_retry_paths
|
||||
]
|
||||
for item in read_plain_keys(retry_files, QWEN_KEY_REGEX):
|
||||
key = item["key"]
|
||||
if not is_qwen_key(key) or key in seen_candidates:
|
||||
continue
|
||||
if not key.startswith("sk-sp-"):
|
||||
hint = combined_provider_routing_hint(key, "qwen")
|
||||
if provider_routing_database_failed():
|
||||
message = "provider routing evidence lookup failed closed"
|
||||
warn_candidate_failure(message, item["source"], key)
|
||||
raise RuntimeError(message)
|
||||
if hint != "qwen":
|
||||
continue
|
||||
seen_candidates.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
|
||||
def redact_text(text, key):
|
||||
redacted = str(text or "")[:1000]
|
||||
if key:
|
||||
redacted = redacted.replace(key, "***REDACTED***")
|
||||
return QWEN_KEY_REGEX.sub("***REDACTED***", redacted)
|
||||
|
||||
|
||||
def parse_error_response(response, key):
|
||||
try:
|
||||
payload = response.json()
|
||||
except ValueError:
|
||||
payload = {}
|
||||
error = payload.get("error") if isinstance(payload, dict) else {}
|
||||
if not isinstance(error, dict):
|
||||
error = {}
|
||||
message = error.get("message") or response.text[:500]
|
||||
return {
|
||||
"http_status": response.status_code,
|
||||
"code": error.get("code") or error.get("type") or "",
|
||||
"type": error.get("type") or "",
|
||||
"message": redact_text(message, key),
|
||||
}
|
||||
|
||||
|
||||
def classify_error(error):
|
||||
http_status = int(error.get("http_status") or 0)
|
||||
code = str(error.get("code") or "").lower()
|
||||
message = str(error.get("message") or "").lower()
|
||||
|
||||
if http_status == 401 or "invalid_api_key" in code or "incorrect api key" in message:
|
||||
return "DEAD"
|
||||
if http_status == 402 or "arrearage" in code or any(item in message for item in (
|
||||
"arrearage", "arrears", "billing", "balance", "overdue", "payment",
|
||||
"insufficient credit", "credit balance",
|
||||
)):
|
||||
return "NO_BALANCE"
|
||||
if http_status == 403:
|
||||
return "RESTRICTED"
|
||||
if http_status == 429:
|
||||
return "LIMITED"
|
||||
if 500 <= http_status <= 599:
|
||||
return "NETWORK"
|
||||
return "UNKNOWN"
|
||||
|
||||
|
||||
def choose_chat_model(models):
|
||||
models = [str(model or "").replace("models/", "") for model in models if model]
|
||||
by_lower = {model.lower(): model for model in models}
|
||||
for model in CHAT_MODEL_PRIORITY:
|
||||
if model.lower() in by_lower:
|
||||
return by_lower[model.lower()]
|
||||
for model in models:
|
||||
lowered = model.lower()
|
||||
if any(marker in lowered for marker in NON_CHAT_MODEL_MARKERS):
|
||||
continue
|
||||
if any(marker in lowered for marker in ("qwen", "qwq", "qvq")):
|
||||
return model
|
||||
return ""
|
||||
|
||||
|
||||
def probe_chat_completion(key, base_url, model, proxy, timeout, debug=False):
|
||||
if not model:
|
||||
return {"status": "NO_CONTEXT", "message": "no chat-capable model from /models", "model": ""}
|
||||
url = f"{normalize_base_url(base_url)}/chat/completions"
|
||||
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
|
||||
payload = {"model": model, "messages": [{"role": "user", "content": "ping"}], "max_tokens": 1}
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)[:1000], "model": model}
|
||||
if debug:
|
||||
print(f" DEBUG {endpoint_label(base_url)} chat ping {model}: HTTP {response.status_code}: {redact_text(response.text[:500], key)}")
|
||||
if response.status_code == 200:
|
||||
return {"status": "GENERATION_OK", "message": "chat completion accepted", "model": model}
|
||||
error = parse_error_response(response, key)
|
||||
return {"status": classify_error(error), "error": error, "message": error.get("message") or "", "model": model}
|
||||
|
||||
|
||||
def parse_models(payload):
|
||||
if not isinstance(payload, dict):
|
||||
return [], []
|
||||
model_infos = payload.get("data")
|
||||
if not isinstance(model_infos, list):
|
||||
model_infos = payload.get("models") if isinstance(payload.get("models"), list) else []
|
||||
models = []
|
||||
for item in model_infos:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
model_id = item.get("id") or item.get("model") or item.get("name")
|
||||
if model_id:
|
||||
models.append(str(model_id).replace("models/", ""))
|
||||
return sorted(set(models)), model_infos
|
||||
|
||||
|
||||
def notable_models(models):
|
||||
notable = []
|
||||
for model in models:
|
||||
lowered = model.lower()
|
||||
if any(marker in lowered for marker in MODEL_MARKERS):
|
||||
notable.append(model)
|
||||
return notable[:30]
|
||||
|
||||
|
||||
def check_base_url(key, base_url, proxy, timeout, debug=False):
|
||||
url = f"{normalize_base_url(base_url)}/models"
|
||||
headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {
|
||||
"base_url": base_url,
|
||||
"region": endpoint_label(base_url),
|
||||
"status": "NETWORK",
|
||||
"message": str(exc)[:1000],
|
||||
}
|
||||
|
||||
if debug:
|
||||
print(f" DEBUG {endpoint_label(base_url)} /models: HTTP {response.status_code}: {redact_text(response.text[:500], key)}")
|
||||
|
||||
if response.status_code == 200:
|
||||
try:
|
||||
payload = response.json()
|
||||
except ValueError:
|
||||
payload = {}
|
||||
models, model_infos = parse_models(payload)
|
||||
chat_model = choose_chat_model(models)
|
||||
probe = probe_chat_completion(key, base_url, chat_model, proxy, timeout, debug)
|
||||
probe_status = probe.get("status") or "UNKNOWN"
|
||||
status = "VALID" if probe_status in ("GENERATION_OK", "NO_CONTEXT") else probe_status
|
||||
probe_message = probe.get("message") or json.dumps(probe.get("error") or {}, ensure_ascii=False)[:1000]
|
||||
return {
|
||||
"base_url": base_url,
|
||||
"region": endpoint_label(base_url),
|
||||
"status": status,
|
||||
"authenticated": True,
|
||||
"model_count": len(models),
|
||||
"models": notable_models(models),
|
||||
"all_model_count": len(models),
|
||||
"model_infos_count": len(model_infos),
|
||||
"llm_probe_status": probe_status,
|
||||
"llm_probe_model": probe.get("model", chat_model),
|
||||
"message": (
|
||||
f"chat ping ok; model={chat_model}; models={len(models)}"
|
||||
if probe_status == "GENERATION_OK"
|
||||
else f"models authenticated; generation_probe={probe_status}; models={len(models)}; {probe_message}"
|
||||
)[:1000],
|
||||
"error": probe.get("error") or {},
|
||||
}
|
||||
|
||||
error = parse_error_response(response, key)
|
||||
return {
|
||||
"base_url": base_url,
|
||||
"region": endpoint_label(base_url),
|
||||
"status": classify_error(error),
|
||||
"error": error,
|
||||
"message": error.get("message") or "",
|
||||
}
|
||||
|
||||
|
||||
def choose_final_status(key, attempts, has_custom_base_urls):
|
||||
statuses = [attempt.get("status") for attempt in attempts]
|
||||
for status in ("VALID", "NO_BALANCE", "LIMITED", "RESTRICTED", "UNKNOWN", "NO_CONTEXT", "NETWORK"):
|
||||
if status in statuses:
|
||||
return status
|
||||
return "DEAD"
|
||||
|
||||
|
||||
def check_key(key, base_urls, has_custom_base_urls, proxy, timeout, debug=False):
|
||||
rejection = qwen_key_rejection_reason(key)
|
||||
if rejection:
|
||||
return {"status": "UNKNOWN", "message": rejection, "candidate_rejected": True}
|
||||
attempts = []
|
||||
for base_url in base_urls:
|
||||
result = check_base_url(key, base_url, proxy, timeout, debug)
|
||||
attempts.append(result)
|
||||
if result.get("status") == "VALID":
|
||||
return {
|
||||
"status": "VALID",
|
||||
"region": result.get("region"),
|
||||
"base_url": result.get("base_url"),
|
||||
"model_count": result.get("model_count", 0),
|
||||
"models": result.get("models", []),
|
||||
"llm_probe_status": result.get("llm_probe_status", ""),
|
||||
"llm_probe_model": result.get("llm_probe_model", ""),
|
||||
"authenticated": bool(result.get("authenticated")),
|
||||
"attempts": attempts,
|
||||
"message": f"models={result.get('model_count', 0)} region={result.get('region')}",
|
||||
}
|
||||
|
||||
status = choose_final_status(key, attempts, has_custom_base_urls)
|
||||
message = ""
|
||||
for attempt in attempts:
|
||||
if attempt.get("status") == status:
|
||||
message = attempt.get("message") or json.dumps(attempt.get("error") or {}, ensure_ascii=False)[:1000]
|
||||
break
|
||||
return {"status": status, "attempts": attempts, "message": message}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
rejection = qwen_key_rejection_reason(key)
|
||||
if rejection:
|
||||
raise ValueError(rejection)
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
if status == "VALID":
|
||||
extra = f"{result.get('region', '')};probe_model={result.get('llm_probe_model', '')};models={','.join(result.get('models', []))[:500]}"
|
||||
else:
|
||||
extra = source
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE,
|
||||
STATUS_FILES,
|
||||
key,
|
||||
status,
|
||||
result.get("message", ""),
|
||||
extra,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def retry_statuses_from_args(args):
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
|
||||
if args.retry_restricted:
|
||||
retry_statuses.add("RESTRICTED")
|
||||
if args.retry_no_balance:
|
||||
retry_statuses.add("NO_BALANCE")
|
||||
if args.retry_valid:
|
||||
retry_statuses.add("VALID")
|
||||
return retry_statuses
|
||||
|
||||
|
||||
def retry_input_files_from_args(args):
|
||||
if args.recheck_all:
|
||||
statuses = list(STATUS_FILES)
|
||||
else:
|
||||
statuses = []
|
||||
if args.retry_network:
|
||||
statuses.append("NETWORK")
|
||||
if args.retry_limited:
|
||||
statuses.append("LIMITED")
|
||||
if args.retry_unknown:
|
||||
statuses.extend(("UNKNOWN", "NO_CONTEXT"))
|
||||
if args.retry_restricted:
|
||||
statuses.append("RESTRICTED")
|
||||
if args.retry_no_balance:
|
||||
statuses.append("NO_BALANCE")
|
||||
if args.retry_valid:
|
||||
statuses.append("VALID")
|
||||
return list(dict.fromkeys(STATUS_FILES[status] for status in statuses))
|
||||
|
||||
|
||||
def base_urls_from_args(args):
|
||||
env_urls = split_csv(os.getenv("QWEN_BASE_URLS") or os.getenv("DASHSCOPE_BASE_URLS"))
|
||||
custom_urls = []
|
||||
for value in args.base_url:
|
||||
custom_urls.extend(split_csv(value))
|
||||
custom_urls.extend(env_urls)
|
||||
default_urls = [] if args.no_default_base_urls else DEFAULT_BASE_URLS
|
||||
return unique_ordered(custom_urls + default_urls), bool(custom_urls)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Qwen/DashScope key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--base-url", action="append", default=[], help="Extra DashScope/OpenAI-compatible base URL; can be repeated")
|
||||
parser.add_argument("--no-default-base-urls", action="store_true")
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = retry_statuses_from_args(args)
|
||||
retry_input_files = retry_input_files_from_args(args)
|
||||
base_urls, has_custom_base_urls = base_urls_from_args(args)
|
||||
if not base_urls:
|
||||
raise SystemExit("No Qwen/DashScope base URLs configured")
|
||||
|
||||
print("--- Qwen/DashScope key checker ---")
|
||||
print("base_urls: " + ", ".join(endpoint_label(url) for url in base_urls))
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, source, finding in iter_candidate_keys(args.input, args.plain, retry_input_files):
|
||||
finding = dict(finding or {})
|
||||
candidate_route = str(finding.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, "") or "").lower()
|
||||
rejection = qwen_key_rejection_reason(key)
|
||||
if rejection:
|
||||
skipped += 1
|
||||
warn_rejected_candidate(rejection, source, key)
|
||||
continue
|
||||
try:
|
||||
should_skip = should_skip_key(
|
||||
key, checked, known, args, retry_statuses, service=SERVICE,
|
||||
source=source, finding=finding, detector=DETECTOR,
|
||||
)
|
||||
except Exception as exc:
|
||||
warn_candidate_failure(f"candidate preparation failed: {exc}", source, key)
|
||||
raise
|
||||
if should_skip:
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
try:
|
||||
print(f"\n[{processed}] Qwen/DashScope candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
routing_hint = "qwen"
|
||||
if keycheck_input_mode() == "postgres":
|
||||
if candidate_route == SERVICE:
|
||||
routing_hint = SERVICE
|
||||
else:
|
||||
routing_hint = combined_provider_routing_hint(key, finding_provider_routing_hint(finding))
|
||||
if provider_routing_database_failed():
|
||||
raise RuntimeError("provider routing evidence lookup failed closed")
|
||||
if routing_hint in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT):
|
||||
result = resolve_provider_key(
|
||||
key, finding, proxy, args.timeout, args.debug,
|
||||
hint=routing_hint, origin_service=SERVICE,
|
||||
)
|
||||
else:
|
||||
result = check_key(key, base_urls, has_custom_base_urls, proxy, args.timeout, args.debug)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
except Exception as exc:
|
||||
warn_candidate_failure(f"candidate processing failed: {exc}", source, key)
|
||||
raise
|
||||
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,403 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
from collections import Counter
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl, classify_common_http_status, commit_status_transaction,
|
||||
default_input_file, default_proxy_file, ensure_output_files, iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses, load_known_keys, load_proxies, mask_secret,
|
||||
read_plain_keys, record_validation_result, recover_status_transaction,
|
||||
request_error_message, require_provider_authority, service_output_dir, should_skip_key, write_keycheck_event,
|
||||
)
|
||||
|
||||
SERVICE = "replicate"
|
||||
DETECTOR = "Replicate"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "replicateChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "replicateResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "replicateAlive.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "replicateDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "replicateRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "replicateLimited.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "replicateNoBalance.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "replicateNetwork.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "replicateNoContext.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "replicateUnknown.txt"),
|
||||
}
|
||||
KEY_REGEX = re.compile(r"\br8_[A-Za-z0-9]{30,}\b")
|
||||
API_BASE = "https://api.replicate.com/v1"
|
||||
ACCOUNT_URL = f"{API_BASE}/account"
|
||||
RESOURCE_ENDPOINTS = {
|
||||
"predictions": f"{API_BASE}/predictions",
|
||||
"deployments": f"{API_BASE}/deployments",
|
||||
"trainings": f"{API_BASE}/trainings",
|
||||
}
|
||||
NO_BALANCE_MARKERS = (
|
||||
"balance",
|
||||
"billing",
|
||||
"credit",
|
||||
"credits",
|
||||
"payment",
|
||||
"insufficient",
|
||||
"depleted",
|
||||
"no credits",
|
||||
"out of credit",
|
||||
"run out of credit",
|
||||
)
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def auth_headers(key):
|
||||
return {"Authorization": f"Bearer {key}", "Accept": "application/json"}
|
||||
|
||||
|
||||
def redacted_error_message(response, key):
|
||||
return request_error_message(response).replace(key, "***REDACTED***")
|
||||
|
||||
|
||||
def classify_replicate_response(response, key):
|
||||
message = redacted_error_message(response, key).lower()
|
||||
if response.status_code == 402 or any(marker in message for marker in NO_BALANCE_MARKERS):
|
||||
return "NO_BALANCE"
|
||||
if response.status_code == 403:
|
||||
return "RESTRICTED"
|
||||
return classify_common_http_status(response.status_code)
|
||||
|
||||
|
||||
def api_get(key, url, proxy, timeout, debug=False):
|
||||
try:
|
||||
response = requests.get(url, headers=auth_headers(key), proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)[:1000], "payload": None}
|
||||
if debug:
|
||||
detail = "ok" if response.status_code == 200 else redacted_error_message(response, key)[:500]
|
||||
print(f" DEBUG GET {url}: HTTP {response.status_code}: {detail}")
|
||||
if response.status_code != 200:
|
||||
return {
|
||||
"status": classify_replicate_response(response, key),
|
||||
"http_status": response.status_code,
|
||||
"message": redacted_error_message(response, key),
|
||||
"payload": None,
|
||||
}
|
||||
try:
|
||||
payload = response.json() if response.text else {}
|
||||
except ValueError:
|
||||
payload = {}
|
||||
return {"status": "OK", "http_status": 200, "message": "ok", "payload": payload}
|
||||
|
||||
|
||||
def paginated_items(payload):
|
||||
if isinstance(payload, list):
|
||||
return payload
|
||||
if not isinstance(payload, dict):
|
||||
return []
|
||||
for key in ("results", "data", "items"):
|
||||
value = payload.get(key)
|
||||
if isinstance(value, list):
|
||||
return value
|
||||
return []
|
||||
|
||||
|
||||
def text_value(value):
|
||||
return str(value or "").strip()
|
||||
|
||||
|
||||
def compact_model_ref(value):
|
||||
if isinstance(value, str):
|
||||
return value.strip()
|
||||
if not isinstance(value, dict):
|
||||
return ""
|
||||
owner = text_value(value.get("owner") or value.get("model_owner"))
|
||||
name = text_value(value.get("name") or value.get("model_name"))
|
||||
if owner and name:
|
||||
return f"{owner}/{name}"
|
||||
for key in ("model", "id", "slug"):
|
||||
item = text_value(value.get(key))
|
||||
if item:
|
||||
return item
|
||||
url = text_value(value.get("url") or value.get("web_url"))
|
||||
if "replicate.com/" in url:
|
||||
return url.rstrip("/").split("replicate.com/", 1)[-1]
|
||||
return ""
|
||||
|
||||
|
||||
def model_refs_from_item(item):
|
||||
if not isinstance(item, dict):
|
||||
return []
|
||||
refs = []
|
||||
for key in ("model", "destination", "source_model", "base_model"):
|
||||
ref = compact_model_ref(item.get(key))
|
||||
if ref:
|
||||
refs.append(ref)
|
||||
for key in ("version", "latest_version", "current_release"):
|
||||
value = item.get(key)
|
||||
if isinstance(value, dict):
|
||||
ref = compact_model_ref(value.get("model") or value.get("destination"))
|
||||
if ref:
|
||||
refs.append(ref)
|
||||
return sorted(set(refs))
|
||||
|
||||
|
||||
def summarize_predictions(payload, limit=10):
|
||||
items = paginated_items(payload)
|
||||
models = sorted({text_value(item.get("model")) for item in items if isinstance(item, dict) and item.get("model")})
|
||||
statuses = Counter(text_value(item.get("status")) for item in items if isinstance(item, dict) and item.get("status"))
|
||||
samples = []
|
||||
for item in items[:limit]:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
samples.append({
|
||||
"id": text_value(item.get("id"))[:80],
|
||||
"status": text_value(item.get("status")),
|
||||
"model": text_value(item.get("model")),
|
||||
"source": text_value(item.get("source")),
|
||||
"data_removed": bool(item.get("data_removed")),
|
||||
"created_at": text_value(item.get("created_at")),
|
||||
"completed_at": text_value(item.get("completed_at")),
|
||||
})
|
||||
return {
|
||||
"prediction_count_sample": len(items),
|
||||
"prediction_has_next_page": bool(isinstance(payload, dict) and payload.get("next")),
|
||||
"prediction_status_counts": dict(statuses),
|
||||
"prediction_models": models[:50],
|
||||
"prediction_samples": samples,
|
||||
}
|
||||
|
||||
|
||||
def deployment_name(item):
|
||||
owner = text_value(item.get("owner") or item.get("deployment_owner"))
|
||||
name = text_value(item.get("name") or item.get("deployment_name"))
|
||||
if owner and name and "/" not in name:
|
||||
return f"{owner}/{name}"
|
||||
return name or owner
|
||||
|
||||
|
||||
def summarize_deployments(payload, limit=20):
|
||||
items = paginated_items(payload)
|
||||
models = sorted({ref for item in items for ref in model_refs_from_item(item)})
|
||||
deployments = []
|
||||
for item in items[:limit]:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
current_release = item.get("current_release") if isinstance(item.get("current_release"), dict) else {}
|
||||
deployments.append({
|
||||
"name": deployment_name(item),
|
||||
"model": next(iter(model_refs_from_item(item)), ""),
|
||||
"version": text_value(item.get("version") or current_release.get("version"))[:80],
|
||||
"hardware": text_value(item.get("hardware") or current_release.get("hardware")),
|
||||
"min_instances": item.get("min_instances"),
|
||||
"max_instances": item.get("max_instances"),
|
||||
})
|
||||
return {
|
||||
"deployment_count": len(items),
|
||||
"deployment_has_next_page": bool(isinstance(payload, dict) and payload.get("next")),
|
||||
"deployment_models": models[:50],
|
||||
"deployments": deployments,
|
||||
}
|
||||
|
||||
|
||||
def summarize_trainings(payload, limit=10):
|
||||
items = paginated_items(payload)
|
||||
models = sorted({ref for item in items for ref in model_refs_from_item(item)})
|
||||
statuses = Counter(text_value(item.get("status")) for item in items if isinstance(item, dict) and item.get("status"))
|
||||
samples = []
|
||||
for item in items[:limit]:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
samples.append({
|
||||
"id": text_value(item.get("id"))[:80],
|
||||
"status": text_value(item.get("status")),
|
||||
"model": next(iter(model_refs_from_item(item)), ""),
|
||||
"created_at": text_value(item.get("created_at")),
|
||||
"completed_at": text_value(item.get("completed_at")),
|
||||
})
|
||||
return {
|
||||
"training_count_sample": len(items),
|
||||
"training_has_next_page": bool(isinstance(payload, dict) and payload.get("next")),
|
||||
"training_status_counts": dict(statuses),
|
||||
"training_models": models[:50],
|
||||
"training_samples": samples,
|
||||
}
|
||||
|
||||
|
||||
def probe_account_resources(key, proxy, timeout, debug=False):
|
||||
summaries = {}
|
||||
endpoint_statuses = {}
|
||||
model_refs = set()
|
||||
ok_count = 0
|
||||
total_items = 0
|
||||
summarizers = {
|
||||
"predictions": summarize_predictions,
|
||||
"deployments": summarize_deployments,
|
||||
"trainings": summarize_trainings,
|
||||
}
|
||||
for name, url in RESOURCE_ENDPOINTS.items():
|
||||
result = api_get(key, url, proxy, timeout, debug)
|
||||
endpoint_statuses[name] = {k: v for k, v in result.items() if k in ("status", "http_status", "message")}
|
||||
if result.get("status") != "OK":
|
||||
continue
|
||||
ok_count += 1
|
||||
summary = summarizers[name](result.get("payload"))
|
||||
summaries.update(summary)
|
||||
for key_name, value in summary.items():
|
||||
if key_name.endswith("_models") and isinstance(value, list):
|
||||
model_refs.update(value)
|
||||
total_items += sum(
|
||||
int(summary.get(field, 0) or 0)
|
||||
for field in ("prediction_count_sample", "deployment_count", "training_count_sample")
|
||||
)
|
||||
if ok_count == len(RESOURCE_ENDPOINTS):
|
||||
probe_status = "RESOURCE_OK" if total_items else "NO_RESOURCES"
|
||||
elif ok_count:
|
||||
probe_status = "PARTIAL"
|
||||
else:
|
||||
probe_status = next((item.get("status") for item in endpoint_statuses.values() if item.get("status")), "UNKNOWN")
|
||||
return {
|
||||
"probe": {"status": probe_status, "endpoints": endpoint_statuses},
|
||||
"models": sorted(model_refs)[:50],
|
||||
"model_count": len(model_refs),
|
||||
**summaries,
|
||||
}
|
||||
|
||||
|
||||
def iter_candidate_decisions(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, [DETECTOR]):
|
||||
key = item.get("credential_secret_text") or item["raw"]
|
||||
if key:
|
||||
yield key, item["source"], item["finding"], bool(KEY_REGEX.fullmatch(key))
|
||||
for item in read_plain_keys(plain_files, KEY_REGEX):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}, True
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
for key, source, finding, valid_format in iter_candidate_decisions(input_file, plain_files):
|
||||
if valid_format:
|
||||
yield key, source, finding
|
||||
|
||||
|
||||
def check_key(key, proxy, args):
|
||||
account = api_get(key, ACCOUNT_URL, proxy, args.timeout, args.debug)
|
||||
if account.get("status") != "OK":
|
||||
return {k: v for k, v in account.items() if k != "payload"}
|
||||
data = account.get("payload") if isinstance(account.get("payload"), dict) else {}
|
||||
result = {
|
||||
"status": "VALID",
|
||||
"message": "account endpoint accepted",
|
||||
"account": data.get("username") or data.get("name") or "",
|
||||
"account_type": data.get("type") or "",
|
||||
}
|
||||
if not args.no_resource_probe:
|
||||
result.update(probe_account_resources(key, proxy, args.timeout, args.debug))
|
||||
result["message"] = (
|
||||
f"account endpoint accepted; probe={result.get('probe', {}).get('status')}; "
|
||||
f"models={result.get('model_count', 0)}; "
|
||||
f"deployments={result.get('deployment_count', 0)}; "
|
||||
f"predictions={result.get('prediction_count_sample', 0)}; "
|
||||
f"trainings={result.get('training_count_sample', 0)}"
|
||||
)
|
||||
else:
|
||||
result.update({"probe": {"status": "not_probed"}, "models": [], "model_count": 0})
|
||||
return result
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
extra = ",".join(result.get("models") or [])[:1000] if status == "VALID" else source
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), extra or source,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Replicate key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--no-resource-probe", action="store_true", help="Only call /account; skip read-only predictions/deployments/trainings probes.")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network: retry_statuses.add("NETWORK")
|
||||
if args.retry_limited: retry_statuses.add("LIMITED")
|
||||
if args.retry_unknown: retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
|
||||
if args.retry_restricted: retry_statuses.add("RESTRICTED")
|
||||
if args.retry_no_balance: retry_statuses.add("NO_BALANCE")
|
||||
if args.retry_valid: retry_statuses.add("VALID")
|
||||
processed = skipped = 0
|
||||
print("--- Replicate key checker ---")
|
||||
print("Default mode: /account plus read-only /predictions, /deployments and /trainings probes. Use --no-resource-probe for /account only.")
|
||||
print(f"proxy: {args.proxy_file}")
|
||||
postgres_mode = keycheck_input_mode() == "postgres"
|
||||
for key, source, finding, valid_format in iter_candidate_decisions(args.input, args.plain):
|
||||
if not valid_format and not postgres_mode:
|
||||
skipped += 1
|
||||
continue
|
||||
if valid_format and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] Replicate candidate {mask_secret(key)} from {source}")
|
||||
result = (
|
||||
check_key(key, next(proxy_cycler) if proxy_cycler else None, args)
|
||||
if valid_format else
|
||||
{"status": "NO_CONTEXT", "message": "candidate does not match canonical Replicate token format"}
|
||||
)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
if result.get("status") == "VALID":
|
||||
print(f" ACCOUNT: {result.get('account') or 'unknown'}")
|
||||
print(f" MODELS: {result.get('model_count', 0)} from account resources")
|
||||
notable = result.get("models") or []
|
||||
if notable:
|
||||
print(f" MODEL REFS: {', '.join(notable[:8])}")
|
||||
print(f" PROBE: {(result.get('probe') or {}).get('status')}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,231 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl, classify_common_http_status, commit_status_transaction,
|
||||
default_input_file, default_proxy_file, ensure_output_files, iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses, load_known_keys, load_proxies, mask_secret,
|
||||
read_plain_keys, record_validation_result, recover_status_transaction,
|
||||
request_error_message, require_provider_authority, service_output_dir, should_skip_key, write_keycheck_event,
|
||||
)
|
||||
|
||||
SERVICE = "xai"
|
||||
DETECTOR_NAMES = ["XAI", "XAi", "Xai"]
|
||||
DETECTOR = "XAI"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "xaiChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "xaiResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "xaiAlive.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "xaiNoBalance.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "xaiDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "xaiRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "xaiLimited.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "xaiNoContext.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "xaiNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "xaiUnknown.txt"),
|
||||
}
|
||||
KEY_REGEX = re.compile(r"\bxai-[A-Za-z0-9_-]{20,}\b")
|
||||
MODELS_URL = "https://api.x.ai/v1/models"
|
||||
CHAT_URL = "https://api.x.ai/v1/chat/completions"
|
||||
CHAT_MODEL_PRIORITY = (
|
||||
"grok-4.6",
|
||||
"grok-4.5",
|
||||
"grok-4.3",
|
||||
"grok-4.20-0309-reasoning",
|
||||
"grok-4.20-0309-non-reasoning",
|
||||
)
|
||||
NO_BALANCE_MARKERS = (
|
||||
"quota",
|
||||
"billing",
|
||||
"balance",
|
||||
"credit",
|
||||
"credits",
|
||||
"payment",
|
||||
"insufficient",
|
||||
"depleted",
|
||||
"spending limit",
|
||||
"no credits",
|
||||
"used all available credits",
|
||||
"doesn't have any credits",
|
||||
)
|
||||
DEAD_MARKERS = ("incorrect api key", "invalid api key", "api key provided", "invalid-argument")
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def iter_candidate_decisions(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, DETECTOR_NAMES):
|
||||
key = item.get("credential_secret_text") or item["raw"]
|
||||
if key:
|
||||
yield key, item["source"], item["finding"], bool(KEY_REGEX.fullmatch(key))
|
||||
for item in read_plain_keys(plain_files, KEY_REGEX):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}, True
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
for key, source, finding, valid_format in iter_candidate_decisions(input_file, plain_files):
|
||||
if valid_format:
|
||||
yield key, source, finding
|
||||
|
||||
|
||||
def check_key(key, proxy, timeout):
|
||||
try:
|
||||
response = requests.get(MODELS_URL, headers={"Authorization": f"Bearer {key}", "Accept": "application/json"}, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)}
|
||||
if response.status_code == 200:
|
||||
data = response.json() if response.text else {}
|
||||
models = [item.get("id") for item in data.get("data", []) if isinstance(item, dict) and item.get("id")]
|
||||
model_inventory = sorted(set(models))
|
||||
model = choose_chat_model(models)
|
||||
probe = probe_chat_completion(key, model, proxy, timeout)
|
||||
if probe.get("status") != "GENERATION_OK":
|
||||
return {
|
||||
"status": probe.get("status") or "UNKNOWN",
|
||||
"message": probe.get("message", ""),
|
||||
"model_count": len(models),
|
||||
"models": models[:20],
|
||||
"model_inventory": model_inventory,
|
||||
"llm_probe_status": probe.get("status"),
|
||||
"llm_probe_model": probe.get("model", model),
|
||||
"llm_probe_http_status": probe.get("http_status"),
|
||||
}
|
||||
return {
|
||||
"status": "VALID", "message": f"chat ping ok; model={model}; models={len(models)}",
|
||||
"model_count": len(models), "models": models[:20], "model_inventory": model_inventory,
|
||||
"llm_probe_status": probe.get("status"), "llm_probe_model": model,
|
||||
}
|
||||
status = classify_xai_response(response)
|
||||
return {"status": status, "http_status": response.status_code, "message": request_error_message(response).replace(key, "***REDACTED***")}
|
||||
|
||||
|
||||
def classify_xai_response(response):
|
||||
message = request_error_message(response).lower()
|
||||
if any(marker in message for marker in DEAD_MARKERS):
|
||||
return "DEAD"
|
||||
if any(marker in message for marker in NO_BALANCE_MARKERS):
|
||||
return "NO_BALANCE"
|
||||
if response.status_code == 401:
|
||||
return "DEAD"
|
||||
if response.status_code == 403:
|
||||
return "RESTRICTED"
|
||||
if response.status_code == 429:
|
||||
return "LIMITED"
|
||||
return classify_common_http_status(response.status_code)
|
||||
|
||||
|
||||
def choose_chat_model(models):
|
||||
models = [str(model or "") for model in models if model]
|
||||
by_lower = {model.lower(): model for model in models}
|
||||
for model in CHAT_MODEL_PRIORITY:
|
||||
if model.lower() in by_lower:
|
||||
return by_lower[model.lower()]
|
||||
for model in models:
|
||||
if "grok" in model.lower():
|
||||
return model
|
||||
return models[0] if models else ""
|
||||
|
||||
|
||||
def probe_chat_completion(key, model, proxy, timeout):
|
||||
if not model:
|
||||
return {"status": "NO_CONTEXT", "message": "no chat-capable model from /models", "model": ""}
|
||||
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
|
||||
payload = {"model": model, "messages": [{"role": "user", "content": "ping"}], "max_tokens": 1}
|
||||
try:
|
||||
response = requests.post(CHAT_URL, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc), "model": model}
|
||||
if response.status_code == 200:
|
||||
return {"status": "GENERATION_OK", "message": "chat completion accepted", "model": model}
|
||||
return {"status": classify_xai_response(response), "http_status": response.status_code, "message": request_error_message(response).replace(key, "***REDACTED***"), "model": model}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
extra = ",".join(result.get("models") or [])[:500] if status == "VALID" else source
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), extra,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="xAI key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network: retry_statuses.add("NETWORK")
|
||||
if args.retry_limited: retry_statuses.add("LIMITED")
|
||||
if args.retry_unknown: retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
|
||||
if args.retry_restricted: retry_statuses.add("RESTRICTED")
|
||||
if args.retry_no_balance: retry_statuses.add("NO_BALANCE")
|
||||
if args.retry_valid: retry_statuses.add("VALID")
|
||||
processed = skipped = 0
|
||||
print("--- xAI key checker ---")
|
||||
postgres_mode = keycheck_input_mode() == "postgres"
|
||||
for key, source, finding, valid_format in iter_candidate_decisions(args.input, args.plain):
|
||||
if not valid_format and not postgres_mode:
|
||||
skipped += 1
|
||||
continue
|
||||
if valid_format and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] xAI candidate {mask_secret(key)} from {source}")
|
||||
result = (
|
||||
check_key(key, next(proxy_cycler) if proxy_cycler else None, args.timeout)
|
||||
if valid_format else
|
||||
{"status": "NO_CONTEXT", "message": "candidate does not match canonical xAI token format"}
|
||||
)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,508 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
combined_provider_routing_hint,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
provider_routing_database_failed,
|
||||
read_plain_keys,
|
||||
record_validation_result,
|
||||
recover_status_transaction,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from keycheckers.provider_resolution import (
|
||||
AMBIGUOUS_GENERIC_SK_HINT,
|
||||
AMBIGUOUS_QWEN_DEEPSEEK_HINT,
|
||||
resolve_provider_key,
|
||||
)
|
||||
|
||||
|
||||
SERVICE = "zai"
|
||||
DETECTOR = "ZaiGLM"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "zaiChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "zaiResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "zaiAlive.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "zaiNoBalance.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "zaiDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "zaiRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "zaiLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "zaiNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "zaiUnknown.txt"),
|
||||
}
|
||||
|
||||
DEFAULT_BASE_URLS = (
|
||||
"https://api.z.ai/api/paas/v4",
|
||||
"https://open.bigmodel.cn/api/paas/v4",
|
||||
)
|
||||
ZAI_KEY_REGEX = re.compile(
|
||||
r"(?<![A-Za-z0-9_.-])(?:"
|
||||
r"(?:zai|sk)-[A-Za-z0-9][A-Za-z0-9_-]{20,505}|"
|
||||
r"[A-Fa-f0-9]{32}\.[A-Za-z0-9_-]{16,128}"
|
||||
r")(?![A-Za-z0-9_.-])"
|
||||
)
|
||||
ZAI_DOTTED_KEY_REGEX = re.compile(r"^[A-Fa-f0-9]{32}\.[A-Za-z0-9_-]{16,128}$")
|
||||
ZAI_CONTEXT_REGEX = re.compile(
|
||||
r"(?:ZAI_API_KEY|GLM_API_KEY|ZHIPUAI_API_KEY|BIGMODEL_API_KEY|api\.z\.ai|"
|
||||
r"open\.bigmodel\.cn|zhipuai|chatglm)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
FOREIGN_KEY_PREFIXES = ("sk-ant-", "sk-or-", "sk-proj-", "sk-svcacct-", "sk-admin-")
|
||||
AMBIGUOUS_HINTS = {AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT}
|
||||
AUTH_FAILURE_CODES = {"1000", "1001", "1003"}
|
||||
AUTHENTICATED_NO_BALANCE_CODES = {"1113"}
|
||||
AUTHENTICATED_LIMIT_CODES = {"1302", "1308", "1309", "1310", "1311"}
|
||||
AUTHENTICATED_RESTRICTED_CODES = {"1005", "1220"}
|
||||
PROBE_MODEL = "glm-5.2"
|
||||
|
||||
|
||||
def normalize_base_url(value):
|
||||
return str(value or "").strip().rstrip("/")
|
||||
|
||||
|
||||
def split_csv(value):
|
||||
if not value:
|
||||
return []
|
||||
values = value.split(",") if isinstance(value, str) else value
|
||||
return [str(item).strip() for item in values if str(item).strip()]
|
||||
|
||||
|
||||
def unique_ordered(values):
|
||||
output = []
|
||||
seen = set()
|
||||
for value in values:
|
||||
normalized = normalize_base_url(value)
|
||||
if normalized and normalized not in seen:
|
||||
seen.add(normalized)
|
||||
output.append(normalized)
|
||||
return output
|
||||
|
||||
|
||||
def base_urls_from_environment(extra=None, include_defaults=True):
|
||||
configured = []
|
||||
for value in extra or ():
|
||||
configured.extend(split_csv(value))
|
||||
configured.extend(split_csv(os.getenv("ZAI_BASE_URLS") or os.getenv("ZHIPU_BASE_URLS")))
|
||||
defaults = DEFAULT_BASE_URLS if include_defaults else ()
|
||||
return unique_ordered([*configured, *defaults])
|
||||
|
||||
|
||||
def endpoint_label(base_url):
|
||||
parsed = urlparse(base_url)
|
||||
return parsed.netloc or base_url
|
||||
|
||||
|
||||
def key_from_text(*values):
|
||||
for value in values:
|
||||
match = ZAI_KEY_REGEX.search(str(value or ""))
|
||||
if match:
|
||||
return match.group(0)
|
||||
return ""
|
||||
|
||||
|
||||
def key_rejection_reason(key):
|
||||
value = str(key or "")
|
||||
try:
|
||||
encoded = value.encode("utf-8", errors="strict")
|
||||
except UnicodeEncodeError:
|
||||
return "candidate is not valid UTF-8"
|
||||
if len(encoded) > 512:
|
||||
return "candidate exceeds the 512-byte key limit"
|
||||
if value.startswith(FOREIGN_KEY_PREFIXES):
|
||||
return "candidate has a foreign provider prefix"
|
||||
if not ZAI_KEY_REGEX.fullmatch(value):
|
||||
return "candidate does not match a bounded ZAI key format"
|
||||
return ""
|
||||
|
||||
|
||||
def finding_detector_names(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return set()
|
||||
extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {}
|
||||
return {
|
||||
name for name in (
|
||||
str(finding.get("DetectorName") or finding.get("DetectorType") or "").strip().lower(),
|
||||
str(extra.get("name") or "").strip().lower(),
|
||||
) if name
|
||||
}
|
||||
|
||||
|
||||
def finding_provider_routing_hint(finding, key=""):
|
||||
if not isinstance(finding, dict):
|
||||
return "zai" if ZAI_DOTTED_KEY_REGEX.fullmatch(str(key or "")) else ""
|
||||
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
|
||||
persisted = str(context.get("provider_hint") or "").strip().lower()
|
||||
if persisted:
|
||||
return persisted
|
||||
if "zaiglm" in finding_detector_names(finding) or ZAI_DOTTED_KEY_REGEX.fullmatch(str(key or "")):
|
||||
return "zai"
|
||||
text = "\n".join(str(context.get(name) or "") for name in ("nearby", "file"))
|
||||
return "zai" if ZAI_CONTEXT_REGEX.search(text) else ""
|
||||
|
||||
|
||||
def iter_candidate_decisions(input_file, plain_files, trusted_retry_files=None):
|
||||
detectors = [
|
||||
"ZaiGLM", "zaiglm", "CustomRegex", "QwenDashScope", "Qwen_DashScope",
|
||||
"Qwen", "DashScope", "DeepSeek", "DeepSeekApiKey", "DeepSeek_API_Key",
|
||||
"KimiMoonshot", "MoonshotAI", "Moonshot", "Kimi",
|
||||
]
|
||||
seen = set()
|
||||
for item in iter_findings(input_file, detectors):
|
||||
finding = item.get("finding") or {}
|
||||
key = item.get("credential_secret_text") or key_from_text(
|
||||
item.get("raw"), item.get("raw_v2"), finding.get("Raw"), finding.get("RawV2"),
|
||||
)
|
||||
if not key or key_rejection_reason(key):
|
||||
continue
|
||||
local_hint = finding_provider_routing_hint(finding, key)
|
||||
hint = combined_provider_routing_hint(key, local_hint)
|
||||
if provider_routing_database_failed():
|
||||
raise RuntimeError("provider routing evidence lookup failed closed")
|
||||
seen.add(key)
|
||||
yield key, item.get("source") or input_file, finding, hint
|
||||
|
||||
owned_retry_paths = {
|
||||
os.path.normcase(os.path.abspath(path)) for path in STATUS_FILES.values()
|
||||
}
|
||||
retry_paths = [
|
||||
path for path in trusted_retry_files or ()
|
||||
if os.path.normcase(os.path.abspath(path)) in owned_retry_paths
|
||||
]
|
||||
for item in read_plain_keys([*plain_files, *retry_paths], ZAI_KEY_REGEX):
|
||||
key = item["key"]
|
||||
if key in seen or key_rejection_reason(key):
|
||||
continue
|
||||
yield key, item["source"], {}, "zai"
|
||||
|
||||
|
||||
def redact_text(value, key):
|
||||
text = str(value or "")[:1000]
|
||||
if key:
|
||||
text = text.replace(key, "***REDACTED***")
|
||||
return ZAI_KEY_REGEX.sub("***REDACTED***", text)
|
||||
|
||||
|
||||
def parse_error(response, key):
|
||||
try:
|
||||
payload = response.json()
|
||||
except ValueError:
|
||||
payload = {}
|
||||
error = payload.get("error") if isinstance(payload, dict) else {}
|
||||
if not isinstance(error, dict):
|
||||
error = {}
|
||||
return {
|
||||
"http_status": int(response.status_code),
|
||||
"code": str(error.get("code") or (payload.get("code") if isinstance(payload, dict) else "") or ""),
|
||||
"message": redact_text(
|
||||
error.get("message") or error.get("msg") or (
|
||||
payload.get("message") or payload.get("msg") if isinstance(payload, dict) else ""
|
||||
) or response.text[:500],
|
||||
key,
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def classify_error(error):
|
||||
http_status = int(error.get("http_status") or 0)
|
||||
code = str(error.get("code") or "")
|
||||
message = str(error.get("message") or "").lower()
|
||||
if http_status == 402 or any(marker in message for marker in (
|
||||
"insufficient balance", "balance is insufficient", "no balance", "account balance",
|
||||
"recharge", "payment required", "billing arrears", "credit balance",
|
||||
)):
|
||||
return "NO_BALANCE", True
|
||||
if code in AUTHENTICATED_NO_BALANCE_CODES:
|
||||
return "NO_BALANCE", True
|
||||
if any(marker in message for marker in (
|
||||
"quota", "rate limit", "rate-limit", "too many requests", "resource exhausted",
|
||||
"concurrency limit", "usage limit",
|
||||
)):
|
||||
return "LIMITED", True
|
||||
if code in AUTHENTICATED_LIMIT_CODES:
|
||||
return "LIMITED", True
|
||||
if any(marker in message for marker in (
|
||||
"permission denied", "access denied", "not authorized for", "model access", "forbidden",
|
||||
)):
|
||||
return "RESTRICTED", True
|
||||
if code in AUTHENTICATED_RESTRICTED_CODES:
|
||||
return "RESTRICTED", True
|
||||
if http_status == 401 or code in AUTH_FAILURE_CODES:
|
||||
return "DEAD", False
|
||||
if 500 <= http_status <= 599 or code in {"1200", "1230", "1234", "1305"}:
|
||||
return "NETWORK", False
|
||||
if http_status == 429:
|
||||
return "LIMITED", False
|
||||
if http_status == 403:
|
||||
return "RESTRICTED", False
|
||||
return "UNKNOWN", False
|
||||
|
||||
|
||||
def probe_chat_completion(key, base_url, model, proxy, timeout, debug=False):
|
||||
if not model:
|
||||
return {
|
||||
"status": "UNKNOWN", "model": "",
|
||||
"message": "no chat-capable model returned by /models",
|
||||
}
|
||||
url = f"{normalize_base_url(base_url)}/chat/completions"
|
||||
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
|
||||
payload = {
|
||||
"model": model,
|
||||
"messages": [{"role": "user", "content": "ping"}],
|
||||
"max_tokens": 1,
|
||||
"stream": False,
|
||||
}
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {
|
||||
"status": "NETWORK", "model": model,
|
||||
"message": redact_text(exc, key),
|
||||
}
|
||||
if debug:
|
||||
print(
|
||||
f" DEBUG {endpoint_label(base_url)} chat probe {model}: HTTP {response.status_code}: "
|
||||
f"{redact_text(response.text[:500], key)}"
|
||||
)
|
||||
if response.status_code == 200:
|
||||
try:
|
||||
response_payload = response.json()
|
||||
except ValueError as exc:
|
||||
return {
|
||||
"status": "UNKNOWN", "model": model, "http_status": response.status_code,
|
||||
"message": f"invalid chat completion response: {exc}",
|
||||
}
|
||||
if isinstance(response_payload, dict) and response_payload.get("choices"):
|
||||
return {
|
||||
"status": "GENERATION_OK", "model": model,
|
||||
"http_status": response.status_code, "message": "chat completion accepted",
|
||||
}
|
||||
error = parse_error(response, key)
|
||||
status, authenticated = classify_error(error)
|
||||
return {
|
||||
"status": status, "authenticated": authenticated, "model": model,
|
||||
"http_status": response.status_code, "business_code": error.get("code") or "",
|
||||
"error": error, "message": error.get("message") or "",
|
||||
}
|
||||
|
||||
|
||||
def check_base_url(key, base_url, proxy, timeout, debug=False):
|
||||
url = f"{normalize_base_url(base_url)}/models"
|
||||
headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {
|
||||
"status": "NETWORK", "base_url": base_url,
|
||||
"region": endpoint_label(base_url), "message": redact_text(exc, key),
|
||||
}
|
||||
if debug:
|
||||
print(
|
||||
f" DEBUG {endpoint_label(base_url)} /models: HTTP {response.status_code}: "
|
||||
f"{redact_text(response.text[:500], key)}"
|
||||
)
|
||||
if response.status_code == 200:
|
||||
try:
|
||||
payload = response.json()
|
||||
data = payload.get("data") if isinstance(payload, dict) else None
|
||||
if not isinstance(data, list):
|
||||
raise ValueError("missing data model list")
|
||||
models = sorted({
|
||||
str(item.get("id") or item.get("name") or "")
|
||||
for item in data if isinstance(item, dict) and (item.get("id") or item.get("name"))
|
||||
})
|
||||
except (TypeError, ValueError, json.JSONDecodeError) as exc:
|
||||
return {
|
||||
"status": "UNKNOWN", "base_url": base_url,
|
||||
"region": endpoint_label(base_url), "message": f"invalid models response: {exc}",
|
||||
}
|
||||
probe_model = PROBE_MODEL
|
||||
probe = probe_chat_completion(key, base_url, probe_model, proxy, timeout, debug)
|
||||
probe_status = probe.get("status") or "UNKNOWN"
|
||||
if probe_status == "GENERATION_OK":
|
||||
status = "VALID"
|
||||
elif probe_status in {"NO_BALANCE", "LIMITED", "RESTRICTED", "NETWORK"}:
|
||||
status = probe_status
|
||||
elif probe_status == "DEAD":
|
||||
status = "RESTRICTED"
|
||||
else:
|
||||
status = "UNKNOWN"
|
||||
probe_message = probe.get("message") or json.dumps(probe.get("error") or {}, ensure_ascii=False)
|
||||
message = (
|
||||
f"chat probe ok; model={probe_model}; models={len(data)}"
|
||||
if status == "VALID"
|
||||
else f"models authenticated; generation_probe={probe_status}; model={probe_model}; "
|
||||
f"models={len(data)}; {probe_message}"
|
||||
)
|
||||
return {
|
||||
"status": status, "authenticated": True, "base_url": base_url,
|
||||
"region": endpoint_label(base_url), "model_count": len(data),
|
||||
"models": models[:30], "llm_probe_status": probe_status,
|
||||
"llm_probe_model": probe.get("model") or probe_model,
|
||||
"llm_probe_http_status": probe.get("http_status"),
|
||||
"business_code": probe.get("business_code") or "",
|
||||
"probe": probe, "error": probe.get("error") or {},
|
||||
"message": message[:1000],
|
||||
}
|
||||
error = parse_error(response, key)
|
||||
status, authenticated = classify_error(error)
|
||||
return {
|
||||
"status": status, "authenticated": authenticated, "base_url": base_url,
|
||||
"region": endpoint_label(base_url), "http_status": response.status_code,
|
||||
"business_code": error.get("code") or "", "error": error,
|
||||
"message": error.get("message") or "",
|
||||
}
|
||||
|
||||
|
||||
def check_key(key, base_urls, proxy, timeout, debug=False):
|
||||
rejection = key_rejection_reason(key)
|
||||
if rejection:
|
||||
return {"status": "UNKNOWN", "message": rejection, "candidate_rejected": True}
|
||||
attempts = []
|
||||
for base_url in base_urls:
|
||||
result = check_base_url(key, base_url, proxy, timeout, debug)
|
||||
attempts.append(result)
|
||||
if result.get("authenticated"):
|
||||
return {**result, "attempts": attempts}
|
||||
statuses = [attempt.get("status") for attempt in attempts]
|
||||
status = next(
|
||||
(candidate for candidate in ("NETWORK", "LIMITED", "RESTRICTED", "UNKNOWN", "DEAD") if candidate in statuses),
|
||||
"UNKNOWN",
|
||||
)
|
||||
selected = next((attempt for attempt in attempts if attempt.get("status") == status), {})
|
||||
return {**selected, "status": status, "attempts": attempts}
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status,
|
||||
result.get("message", ""), result.get("region") or source,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def retry_statuses_from_args(args):
|
||||
statuses = set()
|
||||
for enabled, status in (
|
||||
(args.retry_network, "NETWORK"),
|
||||
(args.retry_limited, "LIMITED"),
|
||||
(args.retry_unknown, "UNKNOWN"),
|
||||
(args.retry_restricted, "RESTRICTED"),
|
||||
(args.retry_no_balance, "NO_BALANCE"),
|
||||
(args.retry_valid, "VALID"),
|
||||
):
|
||||
if enabled:
|
||||
statuses.add(status)
|
||||
return statuses
|
||||
|
||||
|
||||
def retry_input_files_from_args(args):
|
||||
if args.recheck_all:
|
||||
statuses = list(STATUS_FILES)
|
||||
else:
|
||||
statuses = list(retry_statuses_from_args(args))
|
||||
return [STATUS_FILES[status] for status in statuses]
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="ZAI / Zhipu GLM key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--base-url", action="append", default=[])
|
||||
parser.add_argument("--no-default-base-urls", action="store_true")
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = retry_statuses_from_args(args)
|
||||
retry_files = retry_input_files_from_args(args)
|
||||
base_urls = base_urls_from_environment(args.base_url, not args.no_default_base_urls)
|
||||
if not base_urls:
|
||||
raise SystemExit("No ZAI base URLs configured")
|
||||
|
||||
processed = 0
|
||||
skipped = 0
|
||||
postgres_mode = keycheck_input_mode() == "postgres"
|
||||
for key, source, finding, routing_hint in iter_candidate_decisions(args.input, args.plain, retry_files):
|
||||
is_ambiguous = routing_hint in AMBIGUOUS_HINTS
|
||||
if routing_hint != "zai" and not (postgres_mode and is_ambiguous):
|
||||
skipped += 1
|
||||
continue
|
||||
if should_skip_key(
|
||||
key, checked, known, args, retry_statuses, service=SERVICE,
|
||||
source=source, finding=finding, detector=DETECTOR,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] ZAI candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
if is_ambiguous:
|
||||
result = resolve_provider_key(
|
||||
key, finding, proxy, args.timeout, args.debug,
|
||||
hint=routing_hint, origin_service=SERVICE,
|
||||
)
|
||||
else:
|
||||
result = check_key(key, base_urls, proxy, args.timeout, args.debug)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user