- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
207 lines
7.5 KiB
Python
207 lines
7.5 KiB
Python
#!/usr/bin/env python3
|
|
"""Persistent content cache for GitHub raw-file fetching.
|
|
|
|
GitHub code search returns the same file over and over across dozens of
|
|
queries. A blob URL embeds the file's git SHA:
|
|
|
|
https://github.com/<owner>/<repo>/blob/<sha>/<path>
|
|
https://raw.githubusercontent.com/<owner>/<repo>/<sha>/<path>
|
|
|
|
That SHA is immutable — once we've fetched a given (repo, path, sha), the
|
|
bytes can never change, so we never need to crawl it again. For default-
|
|
branch raw URLs that omit the SHA, we cache by (repo, path) plus a hash of
|
|
the prior content; the caller can force a revalidation when GitHub reports
|
|
the file changed, but in practice the hunters only fetch blob-SHA URLs.
|
|
|
|
Cache files live under results/_cache/content/<sharded>.json so a single
|
|
large file doesn't get rewritten on every save. Each entry:
|
|
|
|
key: {"t": epoch, "n": length, "h": sha1[:16], "c": text}
|
|
|
|
The big win: `get_cached(repo, path, sha)` returns the stored text
|
|
immediately when the SHA matches — zero network, zero extraction, and the
|
|
hunter's dedup logic then skips keys it already pulled out of that file.
|
|
|
|
Usage:
|
|
|
|
from content_cache import ContentCache
|
|
cache = ContentCache() # auto-loads results/_cache/content
|
|
text = cache.get(repo, path, sha)
|
|
if text is None:
|
|
text = fetch_raw(url)
|
|
cache.put(repo, path, sha, text)
|
|
...
|
|
cache.save() # or use as context manager
|
|
|
|
All public methods are thread-safe.
|
|
"""
|
|
import hashlib
|
|
import json
|
|
import os
|
|
import threading
|
|
import time
|
|
from pathlib import Path
|
|
|
|
HERE = Path(__file__).parent
|
|
DEFAULT_DIR = HERE / "results" / "_cache" / "content"
|
|
|
|
|
|
def _norm_repo(repo):
|
|
# strip leading "gitee:" or other mirrors consistently; keep as-is otherwise
|
|
return (repo or "").strip().lower()
|
|
|
|
|
|
def _content_hash(text):
|
|
return hashlib.sha1((text or "").encode("utf-8", "replace")).hexdigest()[:16]
|
|
|
|
|
|
class ContentCache:
|
|
"""In-memory + on-disk cache of fetched file contents keyed by blob."""
|
|
|
|
def __init__(self, cache_dir=None, force=False):
|
|
self.dir = Path(cache_dir) if cache_dir else DEFAULT_DIR
|
|
self.dir.mkdir(parents=True, exist_ok=True)
|
|
self.force = force or os.environ.get("NO_CONTENT_CACHE") == "1"
|
|
self._lock = threading.RLock()
|
|
# repo_lower -> { path: { sha_or_branch: entry } }
|
|
self._data = {}
|
|
self._dirty = False
|
|
self.hits = 0
|
|
self.misses = 0
|
|
self.bytes_served = 0
|
|
self._load()
|
|
|
|
# ── persistence ────────────────────────────────────────────
|
|
def _shard_path(self, repo):
|
|
# one json file per repo keeps saves small and collision-free
|
|
safe = "".join(c if c.isalnum() or c in "._-" else "_" for c in repo)
|
|
if not safe:
|
|
safe = "_misc"
|
|
return self.dir / f"{safe}.json"
|
|
|
|
def _load(self):
|
|
if self.force:
|
|
return
|
|
for p in self.dir.glob("*.json"):
|
|
try:
|
|
d = json.loads(p.read_text())
|
|
except Exception:
|
|
continue
|
|
# file shape: {"repo": ..., "files": {path: {sha: entry}}}
|
|
repo = _norm_repo(d.get("repo", p.stem))
|
|
files = d.get("files")
|
|
if isinstance(files, dict):
|
|
self._data.setdefault(repo, {}).update(files)
|
|
|
|
def save(self, repo=None):
|
|
with self._lock:
|
|
if not self._dirty:
|
|
return
|
|
targets = [repo] if repo else list(self._data.keys())
|
|
for r in targets:
|
|
r = _norm_repo(r)
|
|
files = self._data.get(r)
|
|
if not files:
|
|
continue
|
|
tmp = self._shard_path(r).with_suffix(".tmp")
|
|
tmp.write_text(json.dumps(
|
|
{"repo": r, "files": files}, ensure_ascii=False))
|
|
os.replace(tmp, self._shard_path(r))
|
|
self._dirty = False
|
|
|
|
def __enter__(self):
|
|
return self
|
|
|
|
def __exit__(self, *exc):
|
|
self.save()
|
|
|
|
# ── core API ───────────────────────────────────────────────
|
|
def get(self, repo, path, sha=None):
|
|
"""Return cached text for (repo, path, sha) or None.
|
|
|
|
If `sha` is given, a match requires the exact blob SHA. If `sha` is
|
|
None (a branch-tip URL), returns the most recent cached content for
|
|
that path regardless of SHA — useful only as a cheap pre-screen.
|
|
"""
|
|
if self.force:
|
|
return None
|
|
r = _norm_repo(repo)
|
|
path = (path or "").strip()
|
|
with self._lock:
|
|
files = self._data.get(r)
|
|
if not files:
|
|
return None
|
|
versions = files.get(path)
|
|
if not versions:
|
|
return None
|
|
if sha:
|
|
ent = versions.get(sha)
|
|
else:
|
|
# branch URL: return newest version we have
|
|
ent = max(versions.values(),
|
|
key=lambda e: e.get("t", 0), default=None)
|
|
if not ent:
|
|
return None
|
|
self.hits += 1
|
|
self.bytes_served += ent.get("n", 0)
|
|
return ent.get("c", "")
|
|
|
|
def put(self, repo, path, sha, text, ttl=None):
|
|
"""Store fetched text. Empty/None text is not cached (avoids
|
|
permanently poisoning a transient fetch failure)."""
|
|
if self.force:
|
|
return
|
|
if text is None or not isinstance(text, str) or text == "":
|
|
return
|
|
r = _norm_repo(repo)
|
|
path = (path or "").strip()
|
|
with self._lock:
|
|
self._data.setdefault(r, {}).setdefault(path, {})[sha or "_branch"] = {
|
|
"t": time.time(),
|
|
"n": len(text),
|
|
"h": _content_hash(text),
|
|
"c": text,
|
|
}
|
|
self._dirty = True
|
|
self.misses += 1
|
|
|
|
def has(self, repo, path, sha):
|
|
"""Fast existence check without pulling the (potentially large)
|
|
content string into the caller."""
|
|
if self.force:
|
|
return False
|
|
r = _norm_repo(repo)
|
|
with self._lock:
|
|
return bool(self._data.get(r, {}).get(path, {}).get(sha))
|
|
|
|
def stats(self):
|
|
with self._lock:
|
|
repos = len(self._data)
|
|
files = sum(len(v) for v in self._data.values())
|
|
blobs = sum(len(vers) for v in self._data.values()
|
|
for vers in v.values())
|
|
return {"repos": repos, "files": files, "blobs": blobs,
|
|
"hits": self.hits, "misses": self.misses,
|
|
"bytes_served": self.bytes_served}
|
|
|
|
|
|
def parse_raw_url(url):
|
|
"""Split a raw.githubusercontent or github blob URL into
|
|
(repo, sha, path). Returns (None, None, None) if not parseable."""
|
|
u = (url or "").strip()
|
|
# raw: https://raw.githubusercontent.com/<owner>/<repo>/<sha>/<path>
|
|
if "raw.githubusercontent.com/" in u:
|
|
part = u.split("raw.githubusercontent.com/", 1)[1]
|
|
seg = part.split("/", 3)
|
|
if len(seg) >= 4:
|
|
return f"{seg[0]}/{seg[1]}", seg[2], seg[3]
|
|
return None, None, None
|
|
# blob: https://github.com/<owner>/<repo>/blob/<sha>/<path>
|
|
if "github.com/" in u and "/blob/" in u:
|
|
part = u.split("github.com/", 1)[1]
|
|
seg = part.split("/", 4)
|
|
# owner, repo, 'blob', sha, path
|
|
if len(seg) >= 5 and seg[2] == "blob":
|
|
return f"{seg[0]}/{seg[1]}", seg[3], seg[4]
|
|
return None, None, None
|