Initial server source import
This commit is contained in:
@@ -0,0 +1,738 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
from itertools import cycle
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
acquire_file_lock,
|
||||
append_checked,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
env_int,
|
||||
ensure_output_files as ensure_private_output_files,
|
||||
iter_findings,
|
||||
iter_bounded_text_lines,
|
||||
keycheck_input_mode,
|
||||
load_known_statuses,
|
||||
private_atomic_writer,
|
||||
record_cached_keycheck_occurrence,
|
||||
record_validation_result,
|
||||
release_file_lock,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from runtime_security import durable_replace, reject_reparse_components, require_private_directory, require_private_file
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "gemini"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
|
||||
def here(*parts):
|
||||
return os.path.join(SCRIPT_DIR, *parts)
|
||||
|
||||
|
||||
def out(*parts):
|
||||
return os.path.join(OUTPUT_DIR, *parts)
|
||||
|
||||
|
||||
def parent(*parts):
|
||||
return os.path.join(PARENT_DIR, *parts)
|
||||
|
||||
|
||||
# --- Configuration ---
|
||||
DEFAULT_INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
DEFAULT_PLAIN_INPUT_FILES = [out("gem.txt")]
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
CHECKED_FILE = out("geminiChecked.txt")
|
||||
RESULTS_FILE = out("geminiResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": out("geminiAlive.txt"),
|
||||
"VALID_RATE_LIMITED": out("geminiAliveRateLimited.txt"),
|
||||
"INVALID": out("geminiDead.txt"),
|
||||
"EXPIRED": out("geminiExpired.txt"),
|
||||
"LEAKED_REVOKED": out("geminiLeaked.txt"),
|
||||
"API_DISABLED": out("geminiDisabled.txt"),
|
||||
"RESTRICTED": out("geminiRestricted.txt"),
|
||||
"RATE_LIMITED": out("geminiRateLimited.txt"),
|
||||
"NETWORK_ERROR": out("geminiNetwork.txt"),
|
||||
"UNKNOWN": out("geminiUnknown.txt"),
|
||||
}
|
||||
|
||||
GEMINI_KEY_REGEX = re.compile(r"(?:AIza[0-9A-Za-z\-_]{35}|AQ\.[0-9A-Za-z\-_]{50})")
|
||||
GEMINI_DETECTOR_NAMES = {"googleai", "googleaistudio"}
|
||||
MODELS_URL = "https://generativelanguage.googleapis.com/v1beta/models"
|
||||
|
||||
PROBE_MODEL_PRIORITY = [
|
||||
"gemini-3.1-pro-preview",
|
||||
"gemini-3.7-flash",
|
||||
]
|
||||
|
||||
MODEL_PRIORITY = [
|
||||
"gemini-3",
|
||||
"gemini-2.5-pro",
|
||||
"gemini-2.5-flash",
|
||||
"gemini-2.0-flash",
|
||||
"gemini-1.5-pro",
|
||||
"gemini-1.5-flash",
|
||||
"imagen",
|
||||
"embedding",
|
||||
]
|
||||
|
||||
|
||||
def now_iso():
|
||||
return datetime.now(timezone.utc).isoformat(timespec="seconds")
|
||||
|
||||
|
||||
def mask_key(key):
|
||||
if not key or len(key) < 12:
|
||||
return key
|
||||
return f"{key[:8]}...{key[-4:]}"
|
||||
|
||||
|
||||
def redact_key_text(text, key):
|
||||
if not isinstance(text, str):
|
||||
return text
|
||||
redacted = text.replace(key, "***REDACTED***") if key else text
|
||||
return GEMINI_KEY_REGEX.sub("***REDACTED***", redacted)
|
||||
|
||||
|
||||
def redact_result_text(result, key):
|
||||
if isinstance(result, dict):
|
||||
return {k: redact_result_text(v, key) for k, v in result.items()}
|
||||
if isinstance(result, list):
|
||||
return [redact_result_text(v, key) for v in result]
|
||||
return redact_key_text(result, key)
|
||||
|
||||
|
||||
def key_from_line(line):
|
||||
line = line.strip()
|
||||
if not line:
|
||||
return None
|
||||
if "\t" in line:
|
||||
return line.split("\t", 1)[0].strip()
|
||||
return line.split(":", 1)[0].strip()
|
||||
|
||||
|
||||
def load_keys_from_file(filepath):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return set()
|
||||
if not os.path.exists(filepath):
|
||||
return set()
|
||||
keys = set()
|
||||
for line in iter_bounded_text_lines(filepath):
|
||||
key = key_from_line(line)
|
||||
if key:
|
||||
keys.add(key)
|
||||
return keys
|
||||
|
||||
|
||||
def load_checked_statuses(filepath=CHECKED_FILE):
|
||||
statuses = {}
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return statuses
|
||||
if not os.path.exists(filepath):
|
||||
return statuses
|
||||
for line in iter_bounded_text_lines(filepath):
|
||||
parts = line.rstrip("\n").split("\t")
|
||||
if not parts or not parts[0]:
|
||||
continue
|
||||
key = parts[0]
|
||||
status = parts[1] if len(parts) > 1 else "UNKNOWN"
|
||||
statuses[key] = status
|
||||
return statuses
|
||||
|
||||
|
||||
def load_all_known_keys():
|
||||
known = set(load_checked_statuses().keys())
|
||||
for path in STATUS_FILES.values():
|
||||
known.update(load_keys_from_file(path))
|
||||
return known
|
||||
|
||||
|
||||
def ensure_output_files():
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
require_private_directory(OUTPUT_DIR, create=True)
|
||||
|
||||
legacy_rate_limited = out("geminiLimited.txt")
|
||||
rate_limited = STATUS_FILES["RATE_LIMITED"]
|
||||
if os.path.exists(legacy_rate_limited) and not os.path.exists(rate_limited):
|
||||
require_private_file(legacy_rate_limited)
|
||||
durable_replace(legacy_rate_limited, rate_limited)
|
||||
require_private_file(rate_limited)
|
||||
|
||||
paths = {CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()}
|
||||
ensure_private_output_files(paths)
|
||||
migrate_legacy_alive_rate_limited()
|
||||
|
||||
|
||||
def effective_status(result):
|
||||
status = result.get("status")
|
||||
probe_status = (result.get("probe") or {}).get("status")
|
||||
if status == "VALID" and probe_status == "RATE_LIMITED":
|
||||
return "VALID_RATE_LIMITED"
|
||||
return status
|
||||
|
||||
|
||||
def _gemini_status_layout():
|
||||
paths_by_status = {
|
||||
status: os.path.abspath(os.fspath(path))
|
||||
for status, path in STATUS_FILES.items()
|
||||
}
|
||||
paths = list(dict.fromkeys(paths_by_status.values()))
|
||||
directories = {os.path.normcase(os.path.dirname(path)) for path in paths}
|
||||
if len(paths) != len(paths_by_status) or len(directories) != 1:
|
||||
raise RuntimeError("Gemini status files must be unique files in one directory")
|
||||
directory = os.path.dirname(paths[0])
|
||||
require_private_directory(directory, create=True)
|
||||
return paths_by_status, paths, os.path.join(directory, "geminiStatus.lock")
|
||||
|
||||
|
||||
def migrate_legacy_alive_rate_limited():
|
||||
paths_by_status, _, lock_path = _gemini_status_layout()
|
||||
alive_path = paths_by_status["VALID"]
|
||||
limited_path = paths_by_status["VALID_RATE_LIMITED"]
|
||||
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
||||
try:
|
||||
if not os.path.lexists(alive_path):
|
||||
return
|
||||
require_private_file(alive_path)
|
||||
|
||||
keep = []
|
||||
moved = {}
|
||||
for line in iter_bounded_text_lines(alive_path):
|
||||
key = key_from_line(line)
|
||||
if key and line.rstrip("\r\n").endswith(":RATE_LIMITED"):
|
||||
moved.setdefault(key, line if line.endswith("\n") else f"{line}\n")
|
||||
else:
|
||||
keep.append(line)
|
||||
if not moved:
|
||||
return
|
||||
|
||||
if os.path.lexists(limited_path):
|
||||
require_private_file(limited_path)
|
||||
existing = _normalized_status_lines(list(iter_bounded_text_lines(limited_path)))
|
||||
limited = []
|
||||
published = set()
|
||||
for line in existing:
|
||||
key = key_from_line(line)
|
||||
if key in moved:
|
||||
if key in published:
|
||||
continue
|
||||
published.add(key)
|
||||
limited.append(line)
|
||||
for key, line in moved.items():
|
||||
if key not in published:
|
||||
limited.append(line)
|
||||
published.add(key)
|
||||
|
||||
_validate_status_snapshot(limited_path, limited)
|
||||
_validate_status_snapshot(alive_path, keep)
|
||||
|
||||
# Make every moved key durable before publishing the source snapshot
|
||||
# that removes it. An interruption can therefore only leave duplicates.
|
||||
_replace_status_snapshot(limited_path, limited)
|
||||
confirmed = {key: 0 for key in moved}
|
||||
for line in iter_bounded_text_lines(limited_path):
|
||||
key = key_from_line(line)
|
||||
if key in confirmed:
|
||||
confirmed[key] += 1
|
||||
if any(count != 1 for count in confirmed.values()):
|
||||
raise RuntimeError("Gemini legacy rate-limited status publication was incomplete")
|
||||
_replace_status_snapshot(alive_path, keep)
|
||||
finally:
|
||||
release_file_lock(lock, lock_path)
|
||||
|
||||
|
||||
def load_proxies(proxy_file):
|
||||
if not os.path.exists(proxy_file):
|
||||
print(f"Info: {proxy_file} not found. Requests will go directly.")
|
||||
return None
|
||||
|
||||
proxies = []
|
||||
with open(proxy_file, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
ip, port, login, password = line.split(":")
|
||||
proxy_url = f"http://{login}:{password}@{ip}:{port}"
|
||||
proxies.append({"http": proxy_url, "https": proxy_url})
|
||||
except ValueError:
|
||||
print(f"Warning: bad proxy format: {line}. Skipping.")
|
||||
|
||||
if not proxies:
|
||||
print(f"Warning: {proxy_file} is empty. Requests will go directly.")
|
||||
return None
|
||||
|
||||
print(f"Loaded proxies: {len(proxies)}")
|
||||
return cycle(proxies)
|
||||
|
||||
|
||||
def status_file_line(key, result, status):
|
||||
if status in ("VALID", "VALID_RATE_LIMITED"):
|
||||
models_str = ",".join(result.get("notable_models", [])) or "models-only"
|
||||
probe_status = result.get("probe", {}).get("status", "not_probed")
|
||||
return f"{key}:[{models_str}]:{result.get('model_class', 'unknown')}:{probe_status}\n"
|
||||
message = (result.get("error", {}).get("message") or "").replace("\n", " ")[:300]
|
||||
return f"{key}\t{status}\t{message}\n"
|
||||
|
||||
|
||||
def _normalized_status_lines(lines):
|
||||
return [line if line.endswith("\n") else f"{line}\n" for line in lines]
|
||||
|
||||
|
||||
def _validate_status_snapshot(path, lines):
|
||||
max_bytes = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_BYTES", 32 * 1024 * 1024))
|
||||
max_items = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_ITEMS", 100000))
|
||||
max_line_bytes = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_LINE_BYTES", 8192))
|
||||
if len(lines) > max_items:
|
||||
raise RuntimeError(f"Gemini status file exceeds its item bound: {path}")
|
||||
total = 0
|
||||
for index, line in enumerate(lines, 1):
|
||||
encoded = line.encode("utf-8")
|
||||
if len(encoded) > max_line_bytes:
|
||||
raise RuntimeError(f"Gemini status line exceeds its byte bound: {path}:{index}")
|
||||
total += len(encoded)
|
||||
if total > max_bytes:
|
||||
raise RuntimeError(f"Gemini status file exceeds its aggregate byte bound: {path}")
|
||||
|
||||
|
||||
def _replace_status_snapshot(path, lines):
|
||||
with private_atomic_writer(path, binary=True, suffix=".status.tmp") as handle:
|
||||
for line in lines:
|
||||
handle.write(line.encode("utf-8"))
|
||||
|
||||
|
||||
def append_status_file(key, result):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
paths_by_status, paths, lock_path = _gemini_status_layout()
|
||||
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
||||
try:
|
||||
status = effective_status(result)
|
||||
target_path = paths_by_status.get(status, paths_by_status["UNKNOWN"])
|
||||
new_line = status_file_line(key, result, status)
|
||||
snapshots = {}
|
||||
for path in paths:
|
||||
if os.path.lexists(path):
|
||||
reject_reparse_components(path)
|
||||
snapshots[path] = _normalized_status_lines(list(iter_bounded_text_lines(path)))
|
||||
|
||||
rewritten = {
|
||||
path: [line for line in lines if key_from_line(line) != key]
|
||||
for path, lines in snapshots.items()
|
||||
}
|
||||
rewritten[target_path].insert(0, new_line)
|
||||
for path, lines in rewritten.items():
|
||||
_validate_status_snapshot(path, lines)
|
||||
|
||||
# Publish the new classification before removing any old copies. A
|
||||
# failure after this point can leave duplicates, but never no status.
|
||||
_replace_status_snapshot(target_path, rewritten[target_path])
|
||||
for path in paths:
|
||||
if path == target_path or rewritten[path] == snapshots[path]:
|
||||
continue
|
||||
_replace_status_snapshot(path, rewritten[path])
|
||||
finally:
|
||||
release_file_lock(lock, lock_path)
|
||||
|
||||
|
||||
def append_checked_file(key, result):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
append_checked(CHECKED_FILE, key, effective_status(result))
|
||||
|
||||
|
||||
def is_gemini_detector(detector):
|
||||
return str(detector or "").lower() in GEMINI_DETECTOR_NAMES
|
||||
|
||||
|
||||
def custom_detector_name(data):
|
||||
if not isinstance(data, dict):
|
||||
return ""
|
||||
extra = data.get("ExtraData") if isinstance(data.get("ExtraData"), dict) else {}
|
||||
name = extra.get("name") or ""
|
||||
if str(data.get("DetectorName") or "").lower() == "customregex" and is_gemini_detector(name):
|
||||
return name
|
||||
return ""
|
||||
|
||||
|
||||
def detector_name_from_finding(data):
|
||||
if not isinstance(data, dict):
|
||||
return ""
|
||||
if is_gemini_detector(data.get("DetectorName")):
|
||||
return data.get("DetectorName")
|
||||
custom_name = custom_detector_name(data)
|
||||
if custom_name:
|
||||
return custom_name
|
||||
|
||||
# Old wrapped format from earlier scanner versions.
|
||||
if is_gemini_detector(data.get("detector")):
|
||||
return data.get("detector")
|
||||
|
||||
finding = data.get("finding")
|
||||
if isinstance(finding, dict) and is_gemini_detector(finding.get("DetectorName")):
|
||||
return finding.get("DetectorName")
|
||||
custom_name = custom_detector_name(finding)
|
||||
if custom_name:
|
||||
return custom_name
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def extract_key_from_finding(data):
|
||||
if is_gemini_detector(data.get("DetectorName")):
|
||||
return data.get("Raw") or data.get("RawV2")
|
||||
if custom_detector_name(data):
|
||||
return data.get("Raw") or data.get("RawV2")
|
||||
|
||||
# Old wrapped format from earlier scanner versions.
|
||||
if is_gemini_detector(data.get("detector")):
|
||||
return data.get("raw") or data.get("raw_v2")
|
||||
|
||||
finding = data.get("finding")
|
||||
if isinstance(finding, dict) and is_gemini_detector(finding.get("DetectorName")):
|
||||
return finding.get("Raw") or finding.get("RawV2")
|
||||
if custom_detector_name(finding):
|
||||
return finding.get("Raw") or finding.get("RawV2")
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def iter_candidate_keys(input_file, plain_files):
|
||||
for item in iter_findings(input_file, ["GoogleAI", "GoogleAIStudio", "CustomRegex"]):
|
||||
finding = item.get("finding") or {}
|
||||
key = item.get("raw") or extract_key_from_finding(finding)
|
||||
if key and GEMINI_KEY_REGEX.fullmatch(key) and detector_name_from_finding(finding):
|
||||
yield item.get("source") or input_file, key, finding
|
||||
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
for path in plain_files:
|
||||
if not os.path.exists(path):
|
||||
print(f"Info: plain input {path} not found. Skipping.")
|
||||
continue
|
||||
try:
|
||||
keys = set()
|
||||
for line in iter_bounded_text_lines(path):
|
||||
keys.update(GEMINI_KEY_REGEX.findall(line))
|
||||
except (OSError, RuntimeError) as e:
|
||||
print(f"Warning: cannot read {path}: {e}")
|
||||
continue
|
||||
for idx, key in enumerate(sorted(keys), 1):
|
||||
yield f"{path}:plain:{idx}", key, {}
|
||||
|
||||
|
||||
def parse_error_response(response):
|
||||
try:
|
||||
payload = response.json()
|
||||
except json.JSONDecodeError:
|
||||
payload = {}
|
||||
|
||||
error = payload.get("error", {}) if isinstance(payload, dict) else {}
|
||||
return {
|
||||
"http_status": response.status_code,
|
||||
"code": error.get("code", response.status_code),
|
||||
"status": error.get("status", ""),
|
||||
"message": error.get("message", response.text[:500]),
|
||||
}
|
||||
|
||||
|
||||
def classify_error(error):
|
||||
http_status = int(error.get("http_status") or 0)
|
||||
status = str(error.get("status") or "").lower()
|
||||
message = str(error.get("message") or "").lower()
|
||||
|
||||
if "reported as leaked" in message or "leaked" in message:
|
||||
return "LEAKED_REVOKED"
|
||||
if "api key expired" in message or "expired" in message:
|
||||
return "EXPIRED"
|
||||
if "api key not valid" in message or "invalid api key" in message:
|
||||
return "INVALID"
|
||||
if "has not been used" in message or "it is disabled" in message or "api is disabled" in message:
|
||||
return "API_DISABLED"
|
||||
if "requests to this api" in message and "blocked" in message:
|
||||
return "RESTRICTED"
|
||||
if "api key restrictions" in message or "permission_denied" in status:
|
||||
return "RESTRICTED"
|
||||
if http_status == 429 or "resource_exhausted" in status or "quota" in message:
|
||||
return "RATE_LIMITED"
|
||||
if http_status in (400, 401):
|
||||
return "INVALID"
|
||||
if http_status == 403:
|
||||
return "RESTRICTED"
|
||||
return "UNKNOWN"
|
||||
|
||||
|
||||
def fetch_models(key, proxy, timeout, debug=False):
|
||||
try:
|
||||
response = requests.get(MODELS_URL, params={"key": key}, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as e:
|
||||
return {
|
||||
"status": "NETWORK_ERROR",
|
||||
"error": {"message": str(e)},
|
||||
"models": [],
|
||||
"model_infos": [],
|
||||
}
|
||||
|
||||
if debug:
|
||||
print(f" DEBUG /models: HTTP {response.status_code}: {redact_key_text(response.text[:500], key)}")
|
||||
|
||||
if response.status_code != 200:
|
||||
error = parse_error_response(response)
|
||||
return {
|
||||
"status": classify_error(error),
|
||||
"error": error,
|
||||
"models": [],
|
||||
"model_infos": [],
|
||||
}
|
||||
|
||||
payload = response.json()
|
||||
model_infos = payload.get("models", [])
|
||||
models = sorted({m.get("name", "").replace("models/", "") for m in model_infos if m.get("name")})
|
||||
return {
|
||||
"status": "VALID",
|
||||
"error": {},
|
||||
"models": models,
|
||||
"model_infos": model_infos,
|
||||
}
|
||||
|
||||
|
||||
def supported_methods_by_model(model_infos):
|
||||
output = {}
|
||||
for model in model_infos:
|
||||
name = model.get("name", "").replace("models/", "")
|
||||
if not name:
|
||||
continue
|
||||
output[name] = sorted(model.get("supportedGenerationMethods", []))
|
||||
return output
|
||||
|
||||
|
||||
def classify_models(models, methods_by_model):
|
||||
notable = []
|
||||
lower_models = {m.lower(): m for m in models}
|
||||
for marker in MODEL_PRIORITY:
|
||||
for lower, original in lower_models.items():
|
||||
if marker in lower and original not in notable:
|
||||
notable.append(original)
|
||||
|
||||
generation_models = sorted([
|
||||
model for model, methods in methods_by_model.items()
|
||||
if "generateContent" in methods
|
||||
])
|
||||
|
||||
if any("gemini-2.5-pro" in m.lower() for m in generation_models):
|
||||
model_class = "pro_generation"
|
||||
elif generation_models:
|
||||
model_class = "generation"
|
||||
elif models:
|
||||
model_class = "models_only"
|
||||
else:
|
||||
model_class = "no_models"
|
||||
|
||||
return notable[:20], generation_models, model_class
|
||||
|
||||
|
||||
def choose_probe_model(generation_models):
|
||||
available = set(generation_models)
|
||||
for model in PROBE_MODEL_PRIORITY:
|
||||
if model in available:
|
||||
return model
|
||||
return generation_models[0] if generation_models else None
|
||||
|
||||
|
||||
def probe_generation(key, model, proxy, timeout, debug=False):
|
||||
if not model:
|
||||
return {"status": "NO_GENERATION_MODEL", "model": None}
|
||||
|
||||
url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent"
|
||||
headers = {"x-goog-api-key": key, "Content-Type": "application/json"}
|
||||
payload = {
|
||||
"contents": [{"parts": [{"text": "ping"}]}],
|
||||
"generationConfig": {"maxOutputTokens": 1},
|
||||
}
|
||||
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as e:
|
||||
return {"status": "NETWORK_ERROR", "model": model, "error": {"message": str(e)}}
|
||||
|
||||
if debug:
|
||||
print(f" DEBUG probe {model}: HTTP {response.status_code}: {redact_key_text(response.text[:500], key)}")
|
||||
|
||||
if response.status_code == 200:
|
||||
return {"status": "GENERATION_OK", "model": model}
|
||||
|
||||
error = parse_error_response(response)
|
||||
return {"status": classify_error(error), "model": model, "error": error}
|
||||
|
||||
|
||||
def check_key(key, proxy, args):
|
||||
result = fetch_models(key, proxy, args.timeout, args.debug)
|
||||
result = redact_result_text(result, key)
|
||||
result.update({
|
||||
"checked_at": now_iso(),
|
||||
"key_masked": mask_key(key),
|
||||
"model_count": len(result.get("models", [])),
|
||||
})
|
||||
|
||||
if result["status"] != "VALID":
|
||||
result["notable_models"] = []
|
||||
result["generation_models"] = []
|
||||
result["model_class"] = "none"
|
||||
return result
|
||||
|
||||
methods_by_model = supported_methods_by_model(result.get("model_infos", []))
|
||||
notable, generation_models, model_class = classify_models(result["models"], methods_by_model)
|
||||
|
||||
result["methods_by_model"] = methods_by_model
|
||||
result["notable_models"] = notable
|
||||
result["generation_models"] = generation_models[:50]
|
||||
result["model_class"] = model_class
|
||||
result["billing_status"] = "unknown"
|
||||
|
||||
if args.probe_generation:
|
||||
probe_model = choose_probe_model(generation_models)
|
||||
result["probe"] = redact_result_text(probe_generation(key, probe_model, proxy, args.timeout, args.debug), key)
|
||||
else:
|
||||
result["probe"] = {"status": "not_probed", "model": None}
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def print_result(index, source, key, result):
|
||||
print(f"\n[{index}] Candidate {mask_key(key)} from {source}")
|
||||
print(f" STATUS: {result['status']}")
|
||||
|
||||
if result["status"] == "VALID":
|
||||
print(f" MODELS: {result.get('model_count', 0)} total; class={result.get('model_class')}")
|
||||
notable = result.get("notable_models", [])[:8]
|
||||
if notable:
|
||||
print(f" NOTABLE: {', '.join(notable)}")
|
||||
probe = result.get("probe", {})
|
||||
print(f" PROBE: {probe.get('status')} ({probe.get('model')})")
|
||||
if effective_status(result) == "VALID_RATE_LIMITED":
|
||||
print(f" OUT: {STATUS_FILES['VALID_RATE_LIMITED']}")
|
||||
else:
|
||||
error = result.get("error", {})
|
||||
message = (error.get("message") or "").replace("\n", " ")[:300]
|
||||
if message:
|
||||
print(f" MESSAGE: {message}")
|
||||
|
||||
print(f" OUT: {STATUS_FILES.get(result['status'], STATUS_FILES['UNKNOWN'])}")
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Gemini / Google AI API key classifier")
|
||||
parser.add_argument("--input", default=DEFAULT_INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=None, help="Plain text file with Gemini keys. Can be repeated.")
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--probe-generation", action="store_true", help="Optionally call generateContent, preferring gemini-3.1-pro-preview when available.")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
plain_files = args.plain if args.plain is not None else DEFAULT_PLAIN_INPUT_FILES
|
||||
ensure_output_files()
|
||||
|
||||
print("--- Gemini key checker ---")
|
||||
print("Default mode: /models only. Use --probe-generation for runtime/billing probe.")
|
||||
print(f"Workspace: {SCRIPT_DIR}")
|
||||
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked_statuses = load_checked_statuses()
|
||||
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES)
|
||||
known_keys = set(known_statuses)
|
||||
retry_statuses = set()
|
||||
if args.retry_limited:
|
||||
retry_statuses.update(("RATE_LIMITED", "VALID_RATE_LIMITED"))
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK_ERROR")
|
||||
if args.retry_valid:
|
||||
retry_statuses.update(("VALID", "VALID_RATE_LIMITED"))
|
||||
print(f"Loaded known keys: {len(known_keys)}; checked records: {len(checked_statuses)}")
|
||||
|
||||
seen_this_run = set()
|
||||
processed = 0
|
||||
skipped = 0
|
||||
|
||||
for source, key, finding in iter_candidate_keys(args.input, plain_files):
|
||||
if keycheck_input_mode() != 'postgres' and key in seen_this_run:
|
||||
cached_status = checked_statuses.get(key) or known_statuses.get(key) or 'UNKNOWN'
|
||||
detector = detector_name_from_finding(finding) or "GoogleAI"
|
||||
record_cached_keycheck_occurrence(SERVICE, key, cached_status, source, finding, detector)
|
||||
skipped += 1
|
||||
continue
|
||||
seen_this_run.add(key)
|
||||
|
||||
detector = detector_name_from_finding(finding) or "GoogleAI"
|
||||
if should_skip_key(
|
||||
key, checked_statuses, known_keys, args, retry_statuses,
|
||||
service=SERVICE, source=source, finding=finding, detector=detector,
|
||||
known_statuses=known_statuses,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
|
||||
processed += 1
|
||||
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_key(key, proxy, args)
|
||||
result["source"] = source
|
||||
event_result = {**result, "status": effective_status(result)}
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, event_result, source, finding, detector)
|
||||
|
||||
print_result(processed, source, key, result)
|
||||
append_status_file(key, result)
|
||||
append_checked_file(key, result)
|
||||
record_validation_result(SERVICE, key, {**result, "status": effective_status(result)}, source, finding, detector)
|
||||
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = effective_status(result)
|
||||
|
||||
# Small pause helps when many keys hit the same API/proxy.
|
||||
time.sleep(0.1)
|
||||
|
||||
print("\n--- Done ---")
|
||||
print(f"Processed: {processed}")
|
||||
print(f"Skipped: {skipped}")
|
||||
print(f"Results: {RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user