import sys sys.dont_write_bytecode = True import argparse import json import os import re import time from datetime import datetime, timezone from itertools import cycle import requests sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) from keycheck_common import ( acquire_file_lock, append_checked, default_input_file, default_proxy_file, env_int, ensure_output_files as ensure_private_output_files, iter_findings, iter_bounded_text_lines, keycheck_input_mode, load_known_statuses, private_atomic_writer, record_cached_keycheck_occurrence, record_validation_result, release_file_lock, require_provider_authority, service_output_dir, should_skip_key, write_keycheck_event, ) from runtime_security import durable_replace, reject_reparse_components, require_private_directory, require_private_file SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) PARENT_DIR = os.path.dirname(SCRIPT_DIR) SERVICE = "gemini" OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) def here(*parts): return os.path.join(SCRIPT_DIR, *parts) def out(*parts): return os.path.join(OUTPUT_DIR, *parts) def parent(*parts): return os.path.join(PARENT_DIR, *parts) # --- Configuration --- DEFAULT_INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() DEFAULT_PLAIN_INPUT_FILES = [out("gem.txt")] PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() CHECKED_FILE = out("geminiChecked.txt") RESULTS_FILE = out("geminiResults.jsonl") STATUS_FILES = { "VALID": out("geminiAlive.txt"), "VALID_RATE_LIMITED": out("geminiAliveRateLimited.txt"), "INVALID": out("geminiDead.txt"), "EXPIRED": out("geminiExpired.txt"), "LEAKED_REVOKED": out("geminiLeaked.txt"), "API_DISABLED": out("geminiDisabled.txt"), "RESTRICTED": out("geminiRestricted.txt"), "RATE_LIMITED": out("geminiRateLimited.txt"), "NETWORK_ERROR": out("geminiNetwork.txt"), "UNKNOWN": out("geminiUnknown.txt"), } GEMINI_KEY_REGEX = re.compile(r"(?:AIza[0-9A-Za-z\-_]{35}|AQ\.[0-9A-Za-z\-_]{50})") GEMINI_DETECTOR_NAMES = {"googleai", "googleaistudio"} MODELS_URL = "https://generativelanguage.googleapis.com/v1beta/models" PROBE_MODEL_PRIORITY = [ "gemini-3.1-pro-preview", "gemini-3.7-flash", ] MODEL_PRIORITY = [ "gemini-3", "gemini-2.5-pro", "gemini-2.5-flash", "gemini-2.0-flash", "gemini-1.5-pro", "gemini-1.5-flash", "imagen", "embedding", ] def now_iso(): return datetime.now(timezone.utc).isoformat(timespec="seconds") def mask_key(key): if not key or len(key) < 12: return key return f"{key[:8]}...{key[-4:]}" def redact_key_text(text, key): if not isinstance(text, str): return text redacted = text.replace(key, "***REDACTED***") if key else text return GEMINI_KEY_REGEX.sub("***REDACTED***", redacted) def redact_result_text(result, key): if isinstance(result, dict): return {k: redact_result_text(v, key) for k, v in result.items()} if isinstance(result, list): return [redact_result_text(v, key) for v in result] return redact_key_text(result, key) def key_from_line(line): line = line.strip() if not line: return None if "\t" in line: return line.split("\t", 1)[0].strip() return line.split(":", 1)[0].strip() def load_keys_from_file(filepath): if keycheck_input_mode() == 'postgres': return set() if not os.path.exists(filepath): return set() keys = set() for line in iter_bounded_text_lines(filepath): key = key_from_line(line) if key: keys.add(key) return keys def load_checked_statuses(filepath=CHECKED_FILE): statuses = {} if keycheck_input_mode() == 'postgres': return statuses if not os.path.exists(filepath): return statuses for line in iter_bounded_text_lines(filepath): parts = line.rstrip("\n").split("\t") if not parts or not parts[0]: continue key = parts[0] status = parts[1] if len(parts) > 1 else "UNKNOWN" statuses[key] = status return statuses def load_all_known_keys(): known = set(load_checked_statuses().keys()) for path in STATUS_FILES.values(): known.update(load_keys_from_file(path)) return known def ensure_output_files(): if keycheck_input_mode() == 'postgres': return require_private_directory(OUTPUT_DIR, create=True) legacy_rate_limited = out("geminiLimited.txt") rate_limited = STATUS_FILES["RATE_LIMITED"] if os.path.exists(legacy_rate_limited) and not os.path.exists(rate_limited): require_private_file(legacy_rate_limited) durable_replace(legacy_rate_limited, rate_limited) require_private_file(rate_limited) paths = {CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()} ensure_private_output_files(paths) migrate_legacy_alive_rate_limited() def effective_status(result): status = result.get("status") probe_status = (result.get("probe") or {}).get("status") if status == "VALID" and probe_status == "RATE_LIMITED": return "VALID_RATE_LIMITED" return status def _gemini_status_layout(): paths_by_status = { status: os.path.abspath(os.fspath(path)) for status, path in STATUS_FILES.items() } paths = list(dict.fromkeys(paths_by_status.values())) directories = {os.path.normcase(os.path.dirname(path)) for path in paths} if len(paths) != len(paths_by_status) or len(directories) != 1: raise RuntimeError("Gemini status files must be unique files in one directory") directory = os.path.dirname(paths[0]) require_private_directory(directory, create=True) return paths_by_status, paths, os.path.join(directory, "geminiStatus.lock") def migrate_legacy_alive_rate_limited(): paths_by_status, _, lock_path = _gemini_status_layout() alive_path = paths_by_status["VALID"] limited_path = paths_by_status["VALID_RATE_LIMITED"] lock = acquire_file_lock(lock_path, timeout_sec=30) try: if not os.path.lexists(alive_path): return require_private_file(alive_path) keep = [] moved = {} for line in iter_bounded_text_lines(alive_path): key = key_from_line(line) if key and line.rstrip("\r\n").endswith(":RATE_LIMITED"): moved.setdefault(key, line if line.endswith("\n") else f"{line}\n") else: keep.append(line) if not moved: return if os.path.lexists(limited_path): require_private_file(limited_path) existing = _normalized_status_lines(list(iter_bounded_text_lines(limited_path))) limited = [] published = set() for line in existing: key = key_from_line(line) if key in moved: if key in published: continue published.add(key) limited.append(line) for key, line in moved.items(): if key not in published: limited.append(line) published.add(key) _validate_status_snapshot(limited_path, limited) _validate_status_snapshot(alive_path, keep) # Make every moved key durable before publishing the source snapshot # that removes it. An interruption can therefore only leave duplicates. _replace_status_snapshot(limited_path, limited) confirmed = {key: 0 for key in moved} for line in iter_bounded_text_lines(limited_path): key = key_from_line(line) if key in confirmed: confirmed[key] += 1 if any(count != 1 for count in confirmed.values()): raise RuntimeError("Gemini legacy rate-limited status publication was incomplete") _replace_status_snapshot(alive_path, keep) finally: release_file_lock(lock, lock_path) def load_proxies(proxy_file): if not os.path.exists(proxy_file): print(f"Info: {proxy_file} not found. Requests will go directly.") return None proxies = [] with open(proxy_file, "r", encoding="utf-8") as f: for line in f: line = line.strip() if not line: continue try: ip, port, login, password = line.split(":") proxy_url = f"http://{login}:{password}@{ip}:{port}" proxies.append({"http": proxy_url, "https": proxy_url}) except ValueError: print(f"Warning: bad proxy format: {line}. Skipping.") if not proxies: print(f"Warning: {proxy_file} is empty. Requests will go directly.") return None print(f"Loaded proxies: {len(proxies)}") return cycle(proxies) def status_file_line(key, result, status): if status in ("VALID", "VALID_RATE_LIMITED"): models_str = ",".join(result.get("notable_models", [])) or "models-only" probe_status = result.get("probe", {}).get("status", "not_probed") return f"{key}:[{models_str}]:{result.get('model_class', 'unknown')}:{probe_status}\n" message = (result.get("error", {}).get("message") or "").replace("\n", " ")[:300] return f"{key}\t{status}\t{message}\n" def _normalized_status_lines(lines): return [line if line.endswith("\n") else f"{line}\n" for line in lines] def _validate_status_snapshot(path, lines): max_bytes = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_BYTES", 32 * 1024 * 1024)) max_items = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_ITEMS", 100000)) max_line_bytes = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_LINE_BYTES", 8192)) if len(lines) > max_items: raise RuntimeError(f"Gemini status file exceeds its item bound: {path}") total = 0 for index, line in enumerate(lines, 1): encoded = line.encode("utf-8") if len(encoded) > max_line_bytes: raise RuntimeError(f"Gemini status line exceeds its byte bound: {path}:{index}") total += len(encoded) if total > max_bytes: raise RuntimeError(f"Gemini status file exceeds its aggregate byte bound: {path}") def _replace_status_snapshot(path, lines): with private_atomic_writer(path, binary=True, suffix=".status.tmp") as handle: for line in lines: handle.write(line.encode("utf-8")) def append_status_file(key, result): if keycheck_input_mode() == 'postgres': return paths_by_status, paths, lock_path = _gemini_status_layout() lock = acquire_file_lock(lock_path, timeout_sec=30) try: status = effective_status(result) target_path = paths_by_status.get(status, paths_by_status["UNKNOWN"]) new_line = status_file_line(key, result, status) snapshots = {} for path in paths: if os.path.lexists(path): reject_reparse_components(path) snapshots[path] = _normalized_status_lines(list(iter_bounded_text_lines(path))) rewritten = { path: [line for line in lines if key_from_line(line) != key] for path, lines in snapshots.items() } rewritten[target_path].insert(0, new_line) for path, lines in rewritten.items(): _validate_status_snapshot(path, lines) # Publish the new classification before removing any old copies. A # failure after this point can leave duplicates, but never no status. _replace_status_snapshot(target_path, rewritten[target_path]) for path in paths: if path == target_path or rewritten[path] == snapshots[path]: continue _replace_status_snapshot(path, rewritten[path]) finally: release_file_lock(lock, lock_path) def append_checked_file(key, result): if keycheck_input_mode() == 'postgres': return append_checked(CHECKED_FILE, key, effective_status(result)) def is_gemini_detector(detector): return str(detector or "").lower() in GEMINI_DETECTOR_NAMES def custom_detector_name(data): if not isinstance(data, dict): return "" extra = data.get("ExtraData") if isinstance(data.get("ExtraData"), dict) else {} name = extra.get("name") or "" if str(data.get("DetectorName") or "").lower() == "customregex" and is_gemini_detector(name): return name return "" def detector_name_from_finding(data): if not isinstance(data, dict): return "" if is_gemini_detector(data.get("DetectorName")): return data.get("DetectorName") custom_name = custom_detector_name(data) if custom_name: return custom_name # Old wrapped format from earlier scanner versions. if is_gemini_detector(data.get("detector")): return data.get("detector") finding = data.get("finding") if isinstance(finding, dict) and is_gemini_detector(finding.get("DetectorName")): return finding.get("DetectorName") custom_name = custom_detector_name(finding) if custom_name: return custom_name return "" def extract_key_from_finding(data): if is_gemini_detector(data.get("DetectorName")): return data.get("Raw") or data.get("RawV2") if custom_detector_name(data): return data.get("Raw") or data.get("RawV2") # Old wrapped format from earlier scanner versions. if is_gemini_detector(data.get("detector")): return data.get("raw") or data.get("raw_v2") finding = data.get("finding") if isinstance(finding, dict) and is_gemini_detector(finding.get("DetectorName")): return finding.get("Raw") or finding.get("RawV2") if custom_detector_name(finding): return finding.get("Raw") or finding.get("RawV2") return None def iter_candidate_keys(input_file, plain_files): for item in iter_findings(input_file, ["GoogleAI", "GoogleAIStudio", "CustomRegex"]): finding = item.get("finding") or {} key = item.get("raw") or extract_key_from_finding(finding) if key and GEMINI_KEY_REGEX.fullmatch(key) and detector_name_from_finding(finding): yield item.get("source") or input_file, key, finding if keycheck_input_mode() == 'postgres': return for path in plain_files: if not os.path.exists(path): print(f"Info: plain input {path} not found. Skipping.") continue try: keys = set() for line in iter_bounded_text_lines(path): keys.update(GEMINI_KEY_REGEX.findall(line)) except (OSError, RuntimeError) as e: print(f"Warning: cannot read {path}: {e}") continue for idx, key in enumerate(sorted(keys), 1): yield f"{path}:plain:{idx}", key, {} def parse_error_response(response): try: payload = response.json() except json.JSONDecodeError: payload = {} error = payload.get("error", {}) if isinstance(payload, dict) else {} return { "http_status": response.status_code, "code": error.get("code", response.status_code), "status": error.get("status", ""), "message": error.get("message", response.text[:500]), } def classify_error(error): http_status = int(error.get("http_status") or 0) status = str(error.get("status") or "").lower() message = str(error.get("message") or "").lower() if "reported as leaked" in message or "leaked" in message: return "LEAKED_REVOKED" if "api key expired" in message or "expired" in message: return "EXPIRED" if "api key not valid" in message or "invalid api key" in message: return "INVALID" if "has not been used" in message or "it is disabled" in message or "api is disabled" in message: return "API_DISABLED" if "requests to this api" in message and "blocked" in message: return "RESTRICTED" if "api key restrictions" in message or "permission_denied" in status: return "RESTRICTED" if http_status == 429 or "resource_exhausted" in status or "quota" in message: return "RATE_LIMITED" if http_status in (400, 401): return "INVALID" if http_status == 403: return "RESTRICTED" return "UNKNOWN" def fetch_models(key, proxy, timeout, debug=False): try: response = requests.get(MODELS_URL, params={"key": key}, proxies=proxy, timeout=timeout) except requests.RequestException as e: return { "status": "NETWORK_ERROR", "error": {"message": str(e)}, "models": [], "model_infos": [], } if debug: print(f" DEBUG /models: HTTP {response.status_code}: {redact_key_text(response.text[:500], key)}") if response.status_code != 200: error = parse_error_response(response) return { "status": classify_error(error), "error": error, "models": [], "model_infos": [], } payload = response.json() model_infos = payload.get("models", []) models = sorted({m.get("name", "").replace("models/", "") for m in model_infos if m.get("name")}) return { "status": "VALID", "error": {}, "models": models, "model_infos": model_infos, } def supported_methods_by_model(model_infos): output = {} for model in model_infos: name = model.get("name", "").replace("models/", "") if not name: continue output[name] = sorted(model.get("supportedGenerationMethods", [])) return output def classify_models(models, methods_by_model): notable = [] lower_models = {m.lower(): m for m in models} for marker in MODEL_PRIORITY: for lower, original in lower_models.items(): if marker in lower and original not in notable: notable.append(original) generation_models = sorted([ model for model, methods in methods_by_model.items() if "generateContent" in methods ]) if any("gemini-2.5-pro" in m.lower() for m in generation_models): model_class = "pro_generation" elif generation_models: model_class = "generation" elif models: model_class = "models_only" else: model_class = "no_models" return notable[:20], generation_models, model_class def choose_probe_model(generation_models): available = set(generation_models) for model in PROBE_MODEL_PRIORITY: if model in available: return model return generation_models[0] if generation_models else None def probe_generation(key, model, proxy, timeout, debug=False): if not model: return {"status": "NO_GENERATION_MODEL", "model": None} url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent" headers = {"x-goog-api-key": key, "Content-Type": "application/json"} payload = { "contents": [{"parts": [{"text": "ping"}]}], "generationConfig": {"maxOutputTokens": 1}, } try: response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout) except requests.RequestException as e: return {"status": "NETWORK_ERROR", "model": model, "error": {"message": str(e)}} if debug: print(f" DEBUG probe {model}: HTTP {response.status_code}: {redact_key_text(response.text[:500], key)}") if response.status_code == 200: return {"status": "GENERATION_OK", "model": model} error = parse_error_response(response) return {"status": classify_error(error), "model": model, "error": error} def check_key(key, proxy, args): result = fetch_models(key, proxy, args.timeout, args.debug) result = redact_result_text(result, key) result.update({ "checked_at": now_iso(), "key_masked": mask_key(key), "model_count": len(result.get("models", [])), }) if result["status"] != "VALID": result["notable_models"] = [] result["generation_models"] = [] result["model_class"] = "none" return result methods_by_model = supported_methods_by_model(result.get("model_infos", [])) notable, generation_models, model_class = classify_models(result["models"], methods_by_model) result["methods_by_model"] = methods_by_model result["notable_models"] = notable result["generation_models"] = generation_models[:50] result["model_class"] = model_class result["billing_status"] = "unknown" if args.probe_generation: probe_model = choose_probe_model(generation_models) result["probe"] = redact_result_text(probe_generation(key, probe_model, proxy, args.timeout, args.debug), key) else: result["probe"] = {"status": "not_probed", "model": None} return result def print_result(index, source, key, result): print(f"\n[{index}] Candidate {mask_key(key)} from {source}") print(f" STATUS: {result['status']}") if result["status"] == "VALID": print(f" MODELS: {result.get('model_count', 0)} total; class={result.get('model_class')}") notable = result.get("notable_models", [])[:8] if notable: print(f" NOTABLE: {', '.join(notable)}") probe = result.get("probe", {}) print(f" PROBE: {probe.get('status')} ({probe.get('model')})") if effective_status(result) == "VALID_RATE_LIMITED": print(f" OUT: {STATUS_FILES['VALID_RATE_LIMITED']}") else: error = result.get("error", {}) message = (error.get("message") or "").replace("\n", " ")[:300] if message: print(f" MESSAGE: {message}") print(f" OUT: {STATUS_FILES.get(result['status'], STATUS_FILES['UNKNOWN'])}") def parse_args(): parser = argparse.ArgumentParser(description="Gemini / Google AI API key classifier") parser.add_argument("--input", default=DEFAULT_INPUT_FILE) parser.add_argument("--plain", action="append", default=None, help="Plain text file with Gemini keys. Can be repeated.") parser.add_argument("--proxy-file", default=PROXY_FILE) parser.add_argument("--timeout", type=int, default=20) parser.add_argument("--max-keys", type=int, default=0) parser.add_argument("--probe-generation", action="store_true", help="Optionally call generateContent, preferring gemini-3.1-pro-preview when available.") parser.add_argument("--retry-limited", action="store_true") parser.add_argument("--retry-unknown", action="store_true") parser.add_argument("--retry-network", action="store_true") parser.add_argument("--retry-valid", action="store_true") parser.add_argument("--recheck-all", action="store_true") parser.add_argument("--debug", action="store_true") return parser.parse_args() def main(): require_provider_authority(SERVICE) args = parse_args() plain_files = args.plain if args.plain is not None else DEFAULT_PLAIN_INPUT_FILES ensure_output_files() print("--- Gemini key checker ---") print("Default mode: /models only. Use --probe-generation for runtime/billing probe.") print(f"Workspace: {SCRIPT_DIR}") proxy_cycler = load_proxies(args.proxy_file) checked_statuses = load_checked_statuses() known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES) known_keys = set(known_statuses) retry_statuses = set() if args.retry_limited: retry_statuses.update(("RATE_LIMITED", "VALID_RATE_LIMITED")) if args.retry_unknown: retry_statuses.add("UNKNOWN") if args.retry_network: retry_statuses.add("NETWORK_ERROR") if args.retry_valid: retry_statuses.update(("VALID", "VALID_RATE_LIMITED")) print(f"Loaded known keys: {len(known_keys)}; checked records: {len(checked_statuses)}") seen_this_run = set() processed = 0 skipped = 0 for source, key, finding in iter_candidate_keys(args.input, plain_files): if keycheck_input_mode() != 'postgres' and key in seen_this_run: cached_status = checked_statuses.get(key) or known_statuses.get(key) or 'UNKNOWN' detector = detector_name_from_finding(finding) or "GoogleAI" record_cached_keycheck_occurrence(SERVICE, key, cached_status, source, finding, detector) skipped += 1 continue seen_this_run.add(key) detector = detector_name_from_finding(finding) or "GoogleAI" if should_skip_key( key, checked_statuses, known_keys, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=detector, known_statuses=known_statuses, ): skipped += 1 continue if args.max_keys and processed >= args.max_keys: break processed += 1 proxy = next(proxy_cycler) if proxy_cycler else None result = check_key(key, proxy, args) result["source"] = source event_result = {**result, "status": effective_status(result)} write_keycheck_event(SERVICE, RESULTS_FILE, key, event_result, source, finding, detector) print_result(processed, source, key, result) append_status_file(key, result) append_checked_file(key, result) record_validation_result(SERVICE, key, {**result, "status": effective_status(result)}, source, finding, detector) known_keys.add(key) checked_statuses[key] = effective_status(result) # Small pause helps when many keys hit the same API/proxy. time.sleep(0.1) print("\n--- Done ---") print(f"Processed: {processed}") print(f"Skipped: {skipped}") print(f"Results: {RESULTS_FILE}") if __name__ == "__main__": main()