Local branch had diverged from origin/master (sibling commits on the same base). Rewrote local history linearly on top of origin/master, folding in all local content: exa/hot-platform discovery hunters, leaks ledger, nightly sweep orchestrator, updated .gitignore and skill docs, plus the latest hunt output and vault state. Remote-only files (vault keys, channel scripts) were restored rather than dropped, so the resulting tree is a full union of both sides.
141 lines
4.6 KiB
Python
141 lines
4.6 KiB
Python
#!/usr/bin/env python3
|
|
"""Leaks ledger — record every leaked file and key the hunters found.
|
|
|
|
Scans all results/<hunt>/ dirs for:
|
|
- extracted_keys.txt / bucket .txt files (provider|key|url|detail)
|
|
- candidates.json (github_search.py output)
|
|
and maintains results/leaks_ledger.json:
|
|
|
|
files: {html_url: {repo, path, sha, first_seen, last_seen, keys, providers}}
|
|
keys: {key: {providers, sources[], first_seen, last_seen}}
|
|
|
|
"Changed" detection: a file's git blob SHA is immutable, so the same URL
|
|
with the same SHA means nothing changed. New SHAs/URLs count as new leaks.
|
|
|
|
Usage:
|
|
python3 leaks_ledger.py # scan + update + print stats
|
|
python3 leaks_ledger.py --stats # just print stats
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import re
|
|
import sys
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
|
|
HERE = Path(__file__).resolve().parent
|
|
RESULTS = HERE / "results"
|
|
LEDGER = RESULTS / "leaks_ledger.json"
|
|
|
|
URL_RE = re.compile(r"https?://[^\s\"'<>|]+")
|
|
|
|
|
|
def load_ledger():
|
|
if LEDGER.exists():
|
|
try:
|
|
return json.loads(LEDGER.read_text())
|
|
except Exception:
|
|
pass
|
|
return {"updated": None, "files": {}, "keys": {}}
|
|
|
|
|
|
def parse_url_sha(url):
|
|
"""Extract (repo, path, sha) from a github blob/raw URL, else (None,...)."""
|
|
u = url.split("|")[0].strip().rstrip("),;")
|
|
if "/blob/" in u and "github.com/" in u:
|
|
seg = u.split("github.com/", 1)[1].split("/", 4)
|
|
if len(seg) >= 5 and seg[2] == "blob":
|
|
return f"{seg[0]}/{seg[1]}", seg[4], seg[3]
|
|
return None, None, None
|
|
|
|
|
|
def scan_files():
|
|
"""Yield (url, key) pairs from every hunt output file."""
|
|
seen_urls = set()
|
|
# extracted_keys.txt style: provider|key|url|... or key|url|detail
|
|
for txt in RESULTS.glob("*/extracted_keys.txt"):
|
|
for ln in txt.read_text(errors="replace").splitlines():
|
|
parts = ln.split("|")
|
|
if len(parts) >= 3:
|
|
url = parts[2].strip()
|
|
if url.startswith("http"):
|
|
seen_urls.add((url, parts[1].strip()))
|
|
elif len(parts) == 2:
|
|
seen_urls.add((parts[1].strip(), parts[0].strip()))
|
|
# bucket files: key|url|detail (usable/no_balance/...)
|
|
for txt in RESULTS.glob("*/usable.txt"):
|
|
for ln in txt.read_text(errors="replace").splitlines():
|
|
parts = ln.split("|")
|
|
if len(parts) >= 2 and parts[1].strip().startswith("http"):
|
|
seen_urls.add((parts[1].strip(), parts[0].strip()))
|
|
# candidates.json
|
|
for cj in RESULTS.glob("*/candidates.json"):
|
|
try:
|
|
data = json.loads(cj.read_text())
|
|
except Exception:
|
|
continue
|
|
for it in data:
|
|
u = it.get("url", "")
|
|
if u.startswith("http"):
|
|
seen_urls.add((u, ""))
|
|
return seen_urls
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--stats", action="store_true")
|
|
args = ap.parse_args()
|
|
|
|
ledger = load_ledger()
|
|
files, keys = ledger["files"], ledger["keys"]
|
|
now = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%MZ")
|
|
|
|
pairs = scan_files()
|
|
new_files = new_keys = 0
|
|
|
|
for url, key in pairs:
|
|
repo, path, sha = parse_url_sha(url)
|
|
ent = files.get(url)
|
|
if ent is None:
|
|
files[url] = {
|
|
"repo": repo or "", "path": path or "", "sha": sha or "",
|
|
"first_seen": now, "last_seen": now, "keys": 0, "providers": []}
|
|
new_files += 1
|
|
else:
|
|
ent["last_seen"] = now
|
|
if sha and sha != ent.get("sha"):
|
|
ent["sha"] = sha # file changed (new version)
|
|
if key:
|
|
files[url]["keys"] += 1
|
|
ke = keys.get(key)
|
|
if ke is None:
|
|
keys[key] = {"providers": [], "sources": [url],
|
|
"first_seen": now, "last_seen": now}
|
|
new_keys += 1
|
|
else:
|
|
if url not in ke["sources"]:
|
|
ke["sources"].append(url)
|
|
ke["last_seen"] = now
|
|
|
|
ledger["updated"] = now
|
|
LEDGER.write_text(json.dumps(ledger, indent=1, ensure_ascii=False) + "\n")
|
|
|
|
print(f"leaks ledger: {len(files)} files ({new_files} new), "
|
|
f"{len(keys)} keys ({new_keys} new)")
|
|
print(f"updated: {now}")
|
|
|
|
if args.stats:
|
|
return
|
|
|
|
# today's new files (for the daily report)
|
|
today = now[:10]
|
|
fresh = [u for u, e in files.items() if e["first_seen"].startswith(today)]
|
|
if fresh:
|
|
print(f"\ntoday's new leaked files ({len(fresh)}):")
|
|
for u in sorted(fresh)[:20]:
|
|
print(f" {u}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main() |