#!/usr/bin/env python3 """Persistent content cache for GitHub raw-file fetching. GitHub code search returns the same file over and over across dozens of queries. A blob URL embeds the file's git SHA: https://github.com///blob// https://raw.githubusercontent.com//// That SHA is immutable — once we've fetched a given (repo, path, sha), the bytes can never change, so we never need to crawl it again. For default- branch raw URLs that omit the SHA, we cache by (repo, path) plus a hash of the prior content; the caller can force a revalidation when GitHub reports the file changed, but in practice the hunters only fetch blob-SHA URLs. Cache files live under results/_cache/content/.json so a single large file doesn't get rewritten on every save. Each entry: key: {"t": epoch, "n": length, "h": sha1[:16], "c": text} The big win: `get_cached(repo, path, sha)` returns the stored text immediately when the SHA matches — zero network, zero extraction, and the hunter's dedup logic then skips keys it already pulled out of that file. Usage: from content_cache import ContentCache cache = ContentCache() # auto-loads results/_cache/content text = cache.get(repo, path, sha) if text is None: text = fetch_raw(url) cache.put(repo, path, sha, text) ... cache.save() # or use as context manager All public methods are thread-safe. """ import hashlib import json import os import threading import time from pathlib import Path HERE = Path(__file__).parent DEFAULT_DIR = HERE / "results" / "_cache" / "content" def _norm_repo(repo): # strip leading "gitee:" or other mirrors consistently; keep as-is otherwise return (repo or "").strip().lower() def _content_hash(text): return hashlib.sha1((text or "").encode("utf-8", "replace")).hexdigest()[:16] class ContentCache: """In-memory + on-disk cache of fetched file contents keyed by blob.""" def __init__(self, cache_dir=None, force=False): self.dir = Path(cache_dir) if cache_dir else DEFAULT_DIR self.dir.mkdir(parents=True, exist_ok=True) self.force = force or os.environ.get("NO_CONTENT_CACHE") == "1" self._lock = threading.RLock() # repo_lower -> { path: { sha_or_branch: entry } } self._data = {} self._dirty = False self.hits = 0 self.misses = 0 self.bytes_served = 0 self._load() # ── persistence ──────────────────────────────────────────── def _shard_path(self, repo): # one json file per repo keeps saves small and collision-free safe = "".join(c if c.isalnum() or c in "._-" else "_" for c in repo) if not safe: safe = "_misc" return self.dir / f"{safe}.json" def _load(self): if self.force: return for p in self.dir.glob("*.json"): try: d = json.loads(p.read_text()) except Exception: continue # file shape: {"repo": ..., "files": {path: {sha: entry}}} repo = _norm_repo(d.get("repo", p.stem)) files = d.get("files") if isinstance(files, dict): self._data.setdefault(repo, {}).update(files) def save(self, repo=None): with self._lock: if not self._dirty: return targets = [repo] if repo else list(self._data.keys()) for r in targets: r = _norm_repo(r) files = self._data.get(r) if not files: continue tmp = self._shard_path(r).with_suffix(".tmp") tmp.write_text(json.dumps( {"repo": r, "files": files}, ensure_ascii=False)) os.replace(tmp, self._shard_path(r)) self._dirty = False def __enter__(self): return self def __exit__(self, *exc): self.save() # ── core API ─────────────────────────────────────────────── def get(self, repo, path, sha=None): """Return cached text for (repo, path, sha) or None. If `sha` is given, a match requires the exact blob SHA. If `sha` is None (a branch-tip URL), returns the most recent cached content for that path regardless of SHA — useful only as a cheap pre-screen. """ if self.force: return None r = _norm_repo(repo) path = (path or "").strip() with self._lock: files = self._data.get(r) if not files: return None versions = files.get(path) if not versions: return None if sha: ent = versions.get(sha) else: # branch URL: return newest version we have ent = max(versions.values(), key=lambda e: e.get("t", 0), default=None) if not ent: return None self.hits += 1 self.bytes_served += ent.get("n", 0) return ent.get("c", "") def put(self, repo, path, sha, text, ttl=None): """Store fetched text. Empty/None text is not cached (avoids permanently poisoning a transient fetch failure).""" if self.force: return if text is None or not isinstance(text, str) or text == "": return r = _norm_repo(repo) path = (path or "").strip() with self._lock: self._data.setdefault(r, {}).setdefault(path, {})[sha or "_branch"] = { "t": time.time(), "n": len(text), "h": _content_hash(text), "c": text, } self._dirty = True self.misses += 1 def has(self, repo, path, sha): """Fast existence check without pulling the (potentially large) content string into the caller.""" if self.force: return False r = _norm_repo(repo) with self._lock: return bool(self._data.get(r, {}).get(path, {}).get(sha)) def stats(self): with self._lock: repos = len(self._data) files = sum(len(v) for v in self._data.values()) blobs = sum(len(vers) for v in self._data.values() for vers in v.values()) return {"repos": repos, "files": files, "blobs": blobs, "hits": self.hits, "misses": self.misses, "bytes_served": self.bytes_served} def parse_raw_url(url): """Split a raw.githubusercontent or github blob URL into (repo, sha, path). Returns (None, None, None) if not parseable.""" u = (url or "").strip() # raw: https://raw.githubusercontent.com//// if "raw.githubusercontent.com/" in u: part = u.split("raw.githubusercontent.com/", 1)[1] seg = part.split("/", 3) if len(seg) >= 4: return f"{seg[0]}/{seg[1]}", seg[2], seg[3] return None, None, None # blob: https://github.com///blob// if "github.com/" in u and "/blob/" in u: part = u.split("github.com/", 1)[1] seg = part.split("/", 4) # owner, repo, 'blob', sha, path if len(seg) >= 5 and seg[2] == "blob": return f"{seg[0]}/{seg[1]}", seg[3], seg[4] return None, None, None