Files
2026-09-30 20:30:56 +03:00

739 lines
25 KiB
Python

import sys
sys.dont_write_bytecode = True
import argparse
import json
import os
import re
import time
from datetime import datetime, timezone
from itertools import cycle
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
acquire_file_lock,
append_checked,
default_input_file,
default_proxy_file,
env_int,
ensure_output_files as ensure_private_output_files,
iter_findings,
iter_bounded_text_lines,
keycheck_input_mode,
load_known_statuses,
private_atomic_writer,
record_cached_keycheck_occurrence,
record_validation_result,
release_file_lock,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
from runtime_security import durable_replace, reject_reparse_components, require_private_directory, require_private_file
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
SERVICE = "gemini"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
def here(*parts):
return os.path.join(SCRIPT_DIR, *parts)
def out(*parts):
return os.path.join(OUTPUT_DIR, *parts)
def parent(*parts):
return os.path.join(PARENT_DIR, *parts)
# --- Configuration ---
DEFAULT_INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
DEFAULT_PLAIN_INPUT_FILES = [out("gem.txt")]
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
CHECKED_FILE = out("geminiChecked.txt")
RESULTS_FILE = out("geminiResults.jsonl")
STATUS_FILES = {
"VALID": out("geminiAlive.txt"),
"VALID_RATE_LIMITED": out("geminiAliveRateLimited.txt"),
"INVALID": out("geminiDead.txt"),
"EXPIRED": out("geminiExpired.txt"),
"LEAKED_REVOKED": out("geminiLeaked.txt"),
"API_DISABLED": out("geminiDisabled.txt"),
"RESTRICTED": out("geminiRestricted.txt"),
"RATE_LIMITED": out("geminiRateLimited.txt"),
"NETWORK_ERROR": out("geminiNetwork.txt"),
"UNKNOWN": out("geminiUnknown.txt"),
}
GEMINI_KEY_REGEX = re.compile(r"(?:AIza[0-9A-Za-z\-_]{35}|AQ\.[0-9A-Za-z\-_]{50})")
GEMINI_DETECTOR_NAMES = {"googleai", "googleaistudio"}
MODELS_URL = "https://generativelanguage.googleapis.com/v1beta/models"
PROBE_MODEL_PRIORITY = [
"gemini-3.1-pro-preview",
"gemini-3.7-flash",
]
MODEL_PRIORITY = [
"gemini-3",
"gemini-2.5-pro",
"gemini-2.5-flash",
"gemini-2.0-flash",
"gemini-1.5-pro",
"gemini-1.5-flash",
"imagen",
"embedding",
]
def now_iso():
return datetime.now(timezone.utc).isoformat(timespec="seconds")
def mask_key(key):
if not key or len(key) < 12:
return key
return f"{key[:8]}...{key[-4:]}"
def redact_key_text(text, key):
if not isinstance(text, str):
return text
redacted = text.replace(key, "***REDACTED***") if key else text
return GEMINI_KEY_REGEX.sub("***REDACTED***", redacted)
def redact_result_text(result, key):
if isinstance(result, dict):
return {k: redact_result_text(v, key) for k, v in result.items()}
if isinstance(result, list):
return [redact_result_text(v, key) for v in result]
return redact_key_text(result, key)
def key_from_line(line):
line = line.strip()
if not line:
return None
if "\t" in line:
return line.split("\t", 1)[0].strip()
return line.split(":", 1)[0].strip()
def load_keys_from_file(filepath):
if keycheck_input_mode() == 'postgres':
return set()
if not os.path.exists(filepath):
return set()
keys = set()
for line in iter_bounded_text_lines(filepath):
key = key_from_line(line)
if key:
keys.add(key)
return keys
def load_checked_statuses(filepath=CHECKED_FILE):
statuses = {}
if keycheck_input_mode() == 'postgres':
return statuses
if not os.path.exists(filepath):
return statuses
for line in iter_bounded_text_lines(filepath):
parts = line.rstrip("\n").split("\t")
if not parts or not parts[0]:
continue
key = parts[0]
status = parts[1] if len(parts) > 1 else "UNKNOWN"
statuses[key] = status
return statuses
def load_all_known_keys():
known = set(load_checked_statuses().keys())
for path in STATUS_FILES.values():
known.update(load_keys_from_file(path))
return known
def ensure_output_files():
if keycheck_input_mode() == 'postgres':
return
require_private_directory(OUTPUT_DIR, create=True)
legacy_rate_limited = out("geminiLimited.txt")
rate_limited = STATUS_FILES["RATE_LIMITED"]
if os.path.exists(legacy_rate_limited) and not os.path.exists(rate_limited):
require_private_file(legacy_rate_limited)
durable_replace(legacy_rate_limited, rate_limited)
require_private_file(rate_limited)
paths = {CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()}
ensure_private_output_files(paths)
migrate_legacy_alive_rate_limited()
def effective_status(result):
status = result.get("status")
probe_status = (result.get("probe") or {}).get("status")
if status == "VALID" and probe_status == "RATE_LIMITED":
return "VALID_RATE_LIMITED"
return status
def _gemini_status_layout():
paths_by_status = {
status: os.path.abspath(os.fspath(path))
for status, path in STATUS_FILES.items()
}
paths = list(dict.fromkeys(paths_by_status.values()))
directories = {os.path.normcase(os.path.dirname(path)) for path in paths}
if len(paths) != len(paths_by_status) or len(directories) != 1:
raise RuntimeError("Gemini status files must be unique files in one directory")
directory = os.path.dirname(paths[0])
require_private_directory(directory, create=True)
return paths_by_status, paths, os.path.join(directory, "geminiStatus.lock")
def migrate_legacy_alive_rate_limited():
paths_by_status, _, lock_path = _gemini_status_layout()
alive_path = paths_by_status["VALID"]
limited_path = paths_by_status["VALID_RATE_LIMITED"]
lock = acquire_file_lock(lock_path, timeout_sec=30)
try:
if not os.path.lexists(alive_path):
return
require_private_file(alive_path)
keep = []
moved = {}
for line in iter_bounded_text_lines(alive_path):
key = key_from_line(line)
if key and line.rstrip("\r\n").endswith(":RATE_LIMITED"):
moved.setdefault(key, line if line.endswith("\n") else f"{line}\n")
else:
keep.append(line)
if not moved:
return
if os.path.lexists(limited_path):
require_private_file(limited_path)
existing = _normalized_status_lines(list(iter_bounded_text_lines(limited_path)))
limited = []
published = set()
for line in existing:
key = key_from_line(line)
if key in moved:
if key in published:
continue
published.add(key)
limited.append(line)
for key, line in moved.items():
if key not in published:
limited.append(line)
published.add(key)
_validate_status_snapshot(limited_path, limited)
_validate_status_snapshot(alive_path, keep)
# Make every moved key durable before publishing the source snapshot
# that removes it. An interruption can therefore only leave duplicates.
_replace_status_snapshot(limited_path, limited)
confirmed = {key: 0 for key in moved}
for line in iter_bounded_text_lines(limited_path):
key = key_from_line(line)
if key in confirmed:
confirmed[key] += 1
if any(count != 1 for count in confirmed.values()):
raise RuntimeError("Gemini legacy rate-limited status publication was incomplete")
_replace_status_snapshot(alive_path, keep)
finally:
release_file_lock(lock, lock_path)
def load_proxies(proxy_file):
if not os.path.exists(proxy_file):
print(f"Info: {proxy_file} not found. Requests will go directly.")
return None
proxies = []
with open(proxy_file, "r", encoding="utf-8") as f:
for line in f:
line = line.strip()
if not line:
continue
try:
ip, port, login, password = line.split(":")
proxy_url = f"http://{login}:{password}@{ip}:{port}"
proxies.append({"http": proxy_url, "https": proxy_url})
except ValueError:
print(f"Warning: bad proxy format: {line}. Skipping.")
if not proxies:
print(f"Warning: {proxy_file} is empty. Requests will go directly.")
return None
print(f"Loaded proxies: {len(proxies)}")
return cycle(proxies)
def status_file_line(key, result, status):
if status in ("VALID", "VALID_RATE_LIMITED"):
models_str = ",".join(result.get("notable_models", [])) or "models-only"
probe_status = result.get("probe", {}).get("status", "not_probed")
return f"{key}:[{models_str}]:{result.get('model_class', 'unknown')}:{probe_status}\n"
message = (result.get("error", {}).get("message") or "").replace("\n", " ")[:300]
return f"{key}\t{status}\t{message}\n"
def _normalized_status_lines(lines):
return [line if line.endswith("\n") else f"{line}\n" for line in lines]
def _validate_status_snapshot(path, lines):
max_bytes = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_BYTES", 32 * 1024 * 1024))
max_items = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_ITEMS", 100000))
max_line_bytes = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_LINE_BYTES", 8192))
if len(lines) > max_items:
raise RuntimeError(f"Gemini status file exceeds its item bound: {path}")
total = 0
for index, line in enumerate(lines, 1):
encoded = line.encode("utf-8")
if len(encoded) > max_line_bytes:
raise RuntimeError(f"Gemini status line exceeds its byte bound: {path}:{index}")
total += len(encoded)
if total > max_bytes:
raise RuntimeError(f"Gemini status file exceeds its aggregate byte bound: {path}")
def _replace_status_snapshot(path, lines):
with private_atomic_writer(path, binary=True, suffix=".status.tmp") as handle:
for line in lines:
handle.write(line.encode("utf-8"))
def append_status_file(key, result):
if keycheck_input_mode() == 'postgres':
return
paths_by_status, paths, lock_path = _gemini_status_layout()
lock = acquire_file_lock(lock_path, timeout_sec=30)
try:
status = effective_status(result)
target_path = paths_by_status.get(status, paths_by_status["UNKNOWN"])
new_line = status_file_line(key, result, status)
snapshots = {}
for path in paths:
if os.path.lexists(path):
reject_reparse_components(path)
snapshots[path] = _normalized_status_lines(list(iter_bounded_text_lines(path)))
rewritten = {
path: [line for line in lines if key_from_line(line) != key]
for path, lines in snapshots.items()
}
rewritten[target_path].insert(0, new_line)
for path, lines in rewritten.items():
_validate_status_snapshot(path, lines)
# Publish the new classification before removing any old copies. A
# failure after this point can leave duplicates, but never no status.
_replace_status_snapshot(target_path, rewritten[target_path])
for path in paths:
if path == target_path or rewritten[path] == snapshots[path]:
continue
_replace_status_snapshot(path, rewritten[path])
finally:
release_file_lock(lock, lock_path)
def append_checked_file(key, result):
if keycheck_input_mode() == 'postgres':
return
append_checked(CHECKED_FILE, key, effective_status(result))
def is_gemini_detector(detector):
return str(detector or "").lower() in GEMINI_DETECTOR_NAMES
def custom_detector_name(data):
if not isinstance(data, dict):
return ""
extra = data.get("ExtraData") if isinstance(data.get("ExtraData"), dict) else {}
name = extra.get("name") or ""
if str(data.get("DetectorName") or "").lower() == "customregex" and is_gemini_detector(name):
return name
return ""
def detector_name_from_finding(data):
if not isinstance(data, dict):
return ""
if is_gemini_detector(data.get("DetectorName")):
return data.get("DetectorName")
custom_name = custom_detector_name(data)
if custom_name:
return custom_name
# Old wrapped format from earlier scanner versions.
if is_gemini_detector(data.get("detector")):
return data.get("detector")
finding = data.get("finding")
if isinstance(finding, dict) and is_gemini_detector(finding.get("DetectorName")):
return finding.get("DetectorName")
custom_name = custom_detector_name(finding)
if custom_name:
return custom_name
return ""
def extract_key_from_finding(data):
if is_gemini_detector(data.get("DetectorName")):
return data.get("Raw") or data.get("RawV2")
if custom_detector_name(data):
return data.get("Raw") or data.get("RawV2")
# Old wrapped format from earlier scanner versions.
if is_gemini_detector(data.get("detector")):
return data.get("raw") or data.get("raw_v2")
finding = data.get("finding")
if isinstance(finding, dict) and is_gemini_detector(finding.get("DetectorName")):
return finding.get("Raw") or finding.get("RawV2")
if custom_detector_name(finding):
return finding.get("Raw") or finding.get("RawV2")
return None
def iter_candidate_keys(input_file, plain_files):
for item in iter_findings(input_file, ["GoogleAI", "GoogleAIStudio", "CustomRegex"]):
finding = item.get("finding") or {}
key = item.get("raw") or extract_key_from_finding(finding)
if key and GEMINI_KEY_REGEX.fullmatch(key) and detector_name_from_finding(finding):
yield item.get("source") or input_file, key, finding
if keycheck_input_mode() == 'postgres':
return
for path in plain_files:
if not os.path.exists(path):
print(f"Info: plain input {path} not found. Skipping.")
continue
try:
keys = set()
for line in iter_bounded_text_lines(path):
keys.update(GEMINI_KEY_REGEX.findall(line))
except (OSError, RuntimeError) as e:
print(f"Warning: cannot read {path}: {e}")
continue
for idx, key in enumerate(sorted(keys), 1):
yield f"{path}:plain:{idx}", key, {}
def parse_error_response(response):
try:
payload = response.json()
except json.JSONDecodeError:
payload = {}
error = payload.get("error", {}) if isinstance(payload, dict) else {}
return {
"http_status": response.status_code,
"code": error.get("code", response.status_code),
"status": error.get("status", ""),
"message": error.get("message", response.text[:500]),
}
def classify_error(error):
http_status = int(error.get("http_status") or 0)
status = str(error.get("status") or "").lower()
message = str(error.get("message") or "").lower()
if "reported as leaked" in message or "leaked" in message:
return "LEAKED_REVOKED"
if "api key expired" in message or "expired" in message:
return "EXPIRED"
if "api key not valid" in message or "invalid api key" in message:
return "INVALID"
if "has not been used" in message or "it is disabled" in message or "api is disabled" in message:
return "API_DISABLED"
if "requests to this api" in message and "blocked" in message:
return "RESTRICTED"
if "api key restrictions" in message or "permission_denied" in status:
return "RESTRICTED"
if http_status == 429 or "resource_exhausted" in status or "quota" in message:
return "RATE_LIMITED"
if http_status in (400, 401):
return "INVALID"
if http_status == 403:
return "RESTRICTED"
return "UNKNOWN"
def fetch_models(key, proxy, timeout, debug=False):
try:
response = requests.get(MODELS_URL, params={"key": key}, proxies=proxy, timeout=timeout)
except requests.RequestException as e:
return {
"status": "NETWORK_ERROR",
"error": {"message": str(e)},
"models": [],
"model_infos": [],
}
if debug:
print(f" DEBUG /models: HTTP {response.status_code}: {redact_key_text(response.text[:500], key)}")
if response.status_code != 200:
error = parse_error_response(response)
return {
"status": classify_error(error),
"error": error,
"models": [],
"model_infos": [],
}
payload = response.json()
model_infos = payload.get("models", [])
models = sorted({m.get("name", "").replace("models/", "") for m in model_infos if m.get("name")})
return {
"status": "VALID",
"error": {},
"models": models,
"model_infos": model_infos,
}
def supported_methods_by_model(model_infos):
output = {}
for model in model_infos:
name = model.get("name", "").replace("models/", "")
if not name:
continue
output[name] = sorted(model.get("supportedGenerationMethods", []))
return output
def classify_models(models, methods_by_model):
notable = []
lower_models = {m.lower(): m for m in models}
for marker in MODEL_PRIORITY:
for lower, original in lower_models.items():
if marker in lower and original not in notable:
notable.append(original)
generation_models = sorted([
model for model, methods in methods_by_model.items()
if "generateContent" in methods
])
if any("gemini-2.5-pro" in m.lower() for m in generation_models):
model_class = "pro_generation"
elif generation_models:
model_class = "generation"
elif models:
model_class = "models_only"
else:
model_class = "no_models"
return notable[:20], generation_models, model_class
def choose_probe_model(generation_models):
available = set(generation_models)
for model in PROBE_MODEL_PRIORITY:
if model in available:
return model
return generation_models[0] if generation_models else None
def probe_generation(key, model, proxy, timeout, debug=False):
if not model:
return {"status": "NO_GENERATION_MODEL", "model": None}
url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent"
headers = {"x-goog-api-key": key, "Content-Type": "application/json"}
payload = {
"contents": [{"parts": [{"text": "ping"}]}],
"generationConfig": {"maxOutputTokens": 1},
}
try:
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
except requests.RequestException as e:
return {"status": "NETWORK_ERROR", "model": model, "error": {"message": str(e)}}
if debug:
print(f" DEBUG probe {model}: HTTP {response.status_code}: {redact_key_text(response.text[:500], key)}")
if response.status_code == 200:
return {"status": "GENERATION_OK", "model": model}
error = parse_error_response(response)
return {"status": classify_error(error), "model": model, "error": error}
def check_key(key, proxy, args):
result = fetch_models(key, proxy, args.timeout, args.debug)
result = redact_result_text(result, key)
result.update({
"checked_at": now_iso(),
"key_masked": mask_key(key),
"model_count": len(result.get("models", [])),
})
if result["status"] != "VALID":
result["notable_models"] = []
result["generation_models"] = []
result["model_class"] = "none"
return result
methods_by_model = supported_methods_by_model(result.get("model_infos", []))
notable, generation_models, model_class = classify_models(result["models"], methods_by_model)
result["methods_by_model"] = methods_by_model
result["notable_models"] = notable
result["generation_models"] = generation_models[:50]
result["model_class"] = model_class
result["billing_status"] = "unknown"
if args.probe_generation:
probe_model = choose_probe_model(generation_models)
result["probe"] = redact_result_text(probe_generation(key, probe_model, proxy, args.timeout, args.debug), key)
else:
result["probe"] = {"status": "not_probed", "model": None}
return result
def print_result(index, source, key, result):
print(f"\n[{index}] Candidate {mask_key(key)} from {source}")
print(f" STATUS: {result['status']}")
if result["status"] == "VALID":
print(f" MODELS: {result.get('model_count', 0)} total; class={result.get('model_class')}")
notable = result.get("notable_models", [])[:8]
if notable:
print(f" NOTABLE: {', '.join(notable)}")
probe = result.get("probe", {})
print(f" PROBE: {probe.get('status')} ({probe.get('model')})")
if effective_status(result) == "VALID_RATE_LIMITED":
print(f" OUT: {STATUS_FILES['VALID_RATE_LIMITED']}")
else:
error = result.get("error", {})
message = (error.get("message") or "").replace("\n", " ")[:300]
if message:
print(f" MESSAGE: {message}")
print(f" OUT: {STATUS_FILES.get(result['status'], STATUS_FILES['UNKNOWN'])}")
def parse_args():
parser = argparse.ArgumentParser(description="Gemini / Google AI API key classifier")
parser.add_argument("--input", default=DEFAULT_INPUT_FILE)
parser.add_argument("--plain", action="append", default=None, help="Plain text file with Gemini keys. Can be repeated.")
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=20)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--probe-generation", action="store_true", help="Optionally call generateContent, preferring gemini-3.1-pro-preview when available.")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-valid", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
parser.add_argument("--debug", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
plain_files = args.plain if args.plain is not None else DEFAULT_PLAIN_INPUT_FILES
ensure_output_files()
print("--- Gemini key checker ---")
print("Default mode: /models only. Use --probe-generation for runtime/billing probe.")
print(f"Workspace: {SCRIPT_DIR}")
proxy_cycler = load_proxies(args.proxy_file)
checked_statuses = load_checked_statuses()
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES)
known_keys = set(known_statuses)
retry_statuses = set()
if args.retry_limited:
retry_statuses.update(("RATE_LIMITED", "VALID_RATE_LIMITED"))
if args.retry_unknown:
retry_statuses.add("UNKNOWN")
if args.retry_network:
retry_statuses.add("NETWORK_ERROR")
if args.retry_valid:
retry_statuses.update(("VALID", "VALID_RATE_LIMITED"))
print(f"Loaded known keys: {len(known_keys)}; checked records: {len(checked_statuses)}")
seen_this_run = set()
processed = 0
skipped = 0
for source, key, finding in iter_candidate_keys(args.input, plain_files):
if keycheck_input_mode() != 'postgres' and key in seen_this_run:
cached_status = checked_statuses.get(key) or known_statuses.get(key) or 'UNKNOWN'
detector = detector_name_from_finding(finding) or "GoogleAI"
record_cached_keycheck_occurrence(SERVICE, key, cached_status, source, finding, detector)
skipped += 1
continue
seen_this_run.add(key)
detector = detector_name_from_finding(finding) or "GoogleAI"
if should_skip_key(
key, checked_statuses, known_keys, args, retry_statuses,
service=SERVICE, source=source, finding=finding, detector=detector,
known_statuses=known_statuses,
):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
proxy = next(proxy_cycler) if proxy_cycler else None
result = check_key(key, proxy, args)
result["source"] = source
event_result = {**result, "status": effective_status(result)}
write_keycheck_event(SERVICE, RESULTS_FILE, key, event_result, source, finding, detector)
print_result(processed, source, key, result)
append_status_file(key, result)
append_checked_file(key, result)
record_validation_result(SERVICE, key, {**result, "status": effective_status(result)}, source, finding, detector)
known_keys.add(key)
checked_statuses[key] = effective_status(result)
# Small pause helps when many keys hit the same API/proxy.
time.sleep(0.1)
print("\n--- Done ---")
print(f"Processed: {processed}")
print(f"Skipped: {skipped}")
print(f"Results: {RESULTS_FILE}")
if __name__ == "__main__":
main()