Files
hack/tools/scripts/llm-key-hunter/content_cache.py
T
chaos 5d215e1649 Add LLM key-hunter toolkit, vault, and skill
- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*,
  pivot miner, two-layer verify/content caches, per-provider verification)
- usable_keys: verified key vault across 12 providers (deepseek, minimax,
  volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.)
- .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow
- NewAPI channel import scripts and CDP capture helpers
- Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
2026-08-02 06:02:58 +08:00

207 lines
7.5 KiB
Python

#!/usr/bin/env python3
"""Persistent content cache for GitHub raw-file fetching.
GitHub code search returns the same file over and over across dozens of
queries. A blob URL embeds the file's git SHA:
https://github.com/<owner>/<repo>/blob/<sha>/<path>
https://raw.githubusercontent.com/<owner>/<repo>/<sha>/<path>
That SHA is immutable — once we've fetched a given (repo, path, sha), the
bytes can never change, so we never need to crawl it again. For default-
branch raw URLs that omit the SHA, we cache by (repo, path) plus a hash of
the prior content; the caller can force a revalidation when GitHub reports
the file changed, but in practice the hunters only fetch blob-SHA URLs.
Cache files live under results/_cache/content/<sharded>.json so a single
large file doesn't get rewritten on every save. Each entry:
key: {"t": epoch, "n": length, "h": sha1[:16], "c": text}
The big win: `get_cached(repo, path, sha)` returns the stored text
immediately when the SHA matches — zero network, zero extraction, and the
hunter's dedup logic then skips keys it already pulled out of that file.
Usage:
from content_cache import ContentCache
cache = ContentCache() # auto-loads results/_cache/content
text = cache.get(repo, path, sha)
if text is None:
text = fetch_raw(url)
cache.put(repo, path, sha, text)
...
cache.save() # or use as context manager
All public methods are thread-safe.
"""
import hashlib
import json
import os
import threading
import time
from pathlib import Path
HERE = Path(__file__).parent
DEFAULT_DIR = HERE / "results" / "_cache" / "content"
def _norm_repo(repo):
# strip leading "gitee:" or other mirrors consistently; keep as-is otherwise
return (repo or "").strip().lower()
def _content_hash(text):
return hashlib.sha1((text or "").encode("utf-8", "replace")).hexdigest()[:16]
class ContentCache:
"""In-memory + on-disk cache of fetched file contents keyed by blob."""
def __init__(self, cache_dir=None, force=False):
self.dir = Path(cache_dir) if cache_dir else DEFAULT_DIR
self.dir.mkdir(parents=True, exist_ok=True)
self.force = force or os.environ.get("NO_CONTENT_CACHE") == "1"
self._lock = threading.RLock()
# repo_lower -> { path: { sha_or_branch: entry } }
self._data = {}
self._dirty = False
self.hits = 0
self.misses = 0
self.bytes_served = 0
self._load()
# ── persistence ────────────────────────────────────────────
def _shard_path(self, repo):
# one json file per repo keeps saves small and collision-free
safe = "".join(c if c.isalnum() or c in "._-" else "_" for c in repo)
if not safe:
safe = "_misc"
return self.dir / f"{safe}.json"
def _load(self):
if self.force:
return
for p in self.dir.glob("*.json"):
try:
d = json.loads(p.read_text())
except Exception:
continue
# file shape: {"repo": ..., "files": {path: {sha: entry}}}
repo = _norm_repo(d.get("repo", p.stem))
files = d.get("files")
if isinstance(files, dict):
self._data.setdefault(repo, {}).update(files)
def save(self, repo=None):
with self._lock:
if not self._dirty:
return
targets = [repo] if repo else list(self._data.keys())
for r in targets:
r = _norm_repo(r)
files = self._data.get(r)
if not files:
continue
tmp = self._shard_path(r).with_suffix(".tmp")
tmp.write_text(json.dumps(
{"repo": r, "files": files}, ensure_ascii=False))
os.replace(tmp, self._shard_path(r))
self._dirty = False
def __enter__(self):
return self
def __exit__(self, *exc):
self.save()
# ── core API ───────────────────────────────────────────────
def get(self, repo, path, sha=None):
"""Return cached text for (repo, path, sha) or None.
If `sha` is given, a match requires the exact blob SHA. If `sha` is
None (a branch-tip URL), returns the most recent cached content for
that path regardless of SHA — useful only as a cheap pre-screen.
"""
if self.force:
return None
r = _norm_repo(repo)
path = (path or "").strip()
with self._lock:
files = self._data.get(r)
if not files:
return None
versions = files.get(path)
if not versions:
return None
if sha:
ent = versions.get(sha)
else:
# branch URL: return newest version we have
ent = max(versions.values(),
key=lambda e: e.get("t", 0), default=None)
if not ent:
return None
self.hits += 1
self.bytes_served += ent.get("n", 0)
return ent.get("c", "")
def put(self, repo, path, sha, text, ttl=None):
"""Store fetched text. Empty/None text is not cached (avoids
permanently poisoning a transient fetch failure)."""
if self.force:
return
if text is None or not isinstance(text, str) or text == "":
return
r = _norm_repo(repo)
path = (path or "").strip()
with self._lock:
self._data.setdefault(r, {}).setdefault(path, {})[sha or "_branch"] = {
"t": time.time(),
"n": len(text),
"h": _content_hash(text),
"c": text,
}
self._dirty = True
self.misses += 1
def has(self, repo, path, sha):
"""Fast existence check without pulling the (potentially large)
content string into the caller."""
if self.force:
return False
r = _norm_repo(repo)
with self._lock:
return bool(self._data.get(r, {}).get(path, {}).get(sha))
def stats(self):
with self._lock:
repos = len(self._data)
files = sum(len(v) for v in self._data.values())
blobs = sum(len(vers) for v in self._data.values()
for vers in v.values())
return {"repos": repos, "files": files, "blobs": blobs,
"hits": self.hits, "misses": self.misses,
"bytes_served": self.bytes_served}
def parse_raw_url(url):
"""Split a raw.githubusercontent or github blob URL into
(repo, sha, path). Returns (None, None, None) if not parseable."""
u = (url or "").strip()
# raw: https://raw.githubusercontent.com/<owner>/<repo>/<sha>/<path>
if "raw.githubusercontent.com/" in u:
part = u.split("raw.githubusercontent.com/", 1)[1]
seg = part.split("/", 3)
if len(seg) >= 4:
return f"{seg[0]}/{seg[1]}", seg[2], seg[3]
return None, None, None
# blob: https://github.com/<owner>/<repo>/blob/<sha>/<path>
if "github.com/" in u and "/blob/" in u:
part = u.split("github.com/", 1)[1]
seg = part.split("/", 4)
# owner, repo, 'blob', sha, path
if len(seg) >= 5 and seg[2] == "blob":
return f"{seg[0]}/{seg[1]}", seg[3], seg[4]
return None, None, None