Files
ginnoir dd26aafc42 feat(vault): synthesize Discord scrape + web research into the ROM-hack catalog
Enrich the Obsidian ROM-hack vault from the valhalla Discord scrape
(/storage1/labdata/romhacks/metadata) plus web research, growing the catalog
from 285 to 368 notes and preparing a data export for a wiki site.

Tooling (all dry-run by default, --apply to write; .scrape/ is gitignored):
- scrape_match.py    reconcile scrape entries vs notes (43 overlap / 211 match / 83 new)
- enrich_vault.py    backfill frontmatter, merge scrape into curated notes,
                     create 83 new notes, rebuild import bodies to one layout
- enrich_web.py      apply hand/web-verified facts from .scrape/web_facts*.json
- build-wiki-data.py export wiki/catalog.json (the site data source)

Results: 368 notes, 327 with served banner art, 332 rich summaries, 200 with
feature lists; new frontmatter developer/release_date/banner/homepage/scrape_dir;
12 flagship hacks web-verified. wiki/SPEC.md describes the site build.
build-vault-mocs.py Index callout updated for the new counts.
2026-06-08 20:10:59 -05:00

160 lines
5.9 KiB
Python

#!/usr/bin/env python3
"""Match Discord-scrape metadata.json entries to the live vault notes.
Reads:
- .scrape/<Dir>/metadata.json (pulled from valhalla; {name, description, links, art})
- <VAULT>/Pokémon ROM Hacks/Hacks/*.md (live notes; frontmatter source of truth)
Produces a JSON report on stdout (and --report path) classifying every scrape
entry as one of:
curated_overlap -> a curated (non-Discord) note exists; merge description in
discord_match -> a Discord-import note exists; backfill frontmatter
unmatched -> no note exists; candidate for a brand-new note
Matching is by a normalized key (lowercase, strip 'pokemon/pokémon', drop
non-alphanumerics). This is analysis only; it writes no vault notes.
"""
from __future__ import annotations
import json, os, re, sys, glob, unicodedata
VAULT = r"C:\Users\MattC\Documents\Obsidian Vault"
HACKS = os.path.join(VAULT, "Pokémon ROM Hacks", "Hacks")
SCRAPE = os.path.join(os.path.dirname(__file__), "..", ".scrape")
def norm(name: str) -> str:
"""Normalized match key: NFKD, drop accents, lowercase, strip 'pokemon',
keep only [a-z0-9]."""
name = unicodedata.normalize("NFKD", name)
name = "".join(c for c in name if not unicodedata.combining(c))
name = name.lower()
name = re.sub(r"\bpok[eé]?mon\b", " ", name)
name = re.sub(r"[^a-z0-9]+", "", name)
return name
def parse_frontmatter(text: str) -> dict:
m = re.match(r"^---\n(.*?)\n---", text, re.S)
if not m:
return {}
fm: dict = {}
lines = m.group(1).split("\n")
i = 0
while i < len(lines):
km = re.match(r"^([\w]+):\s*(.*)$", lines[i])
if km:
k, v = km.group(1), km.group(2).strip()
if v == "" and i + 1 < len(lines) and re.match(r"^\s*-\s", lines[i + 1]):
items = []
while i + 1 < len(lines) and re.match(r"^\s*-\s", lines[i + 1]):
items.append(lines[i + 1].strip()[2:].strip().strip('"'))
i += 1
fm[k] = items
else:
fm[k] = v.strip('"')
i += 1
return fm
# semi-structured fields commonly present in Discord descriptions
FIELD_PATS = {
"version": re.compile(r"(?im)^\s*version\s*[:\-]\s*(.+?)\s*$"),
"status": re.compile(r"(?im)^\s*status\s*[:\-]\s*(.+?)\s*$"),
"creator": re.compile(r"(?im)^\s*(?:creator|developer|author|made by|dev)\s*[:\-]\s*(.+?)\s*$"),
"language": re.compile(r"(?im)^\s*language\s*[:\-]\s*(.+?)\s*$"),
"base": re.compile(r"(?im)^\s*(?:base|base game|hack of|rom base)\s*[:\-]\s*(.+?)\s*$"),
}
def extract_fields(desc: str) -> dict:
out = {}
for k, pat in FIELD_PATS.items():
m = pat.search(desc or "")
if m:
out[k] = m.group(1).strip()
return out
def main() -> None:
# index vault notes by normalized key
notes = {}
for f in glob.glob(os.path.join(HACKS, "*.md")):
text = open(f, encoding="utf-8").read()
fm = parse_frontmatter(text)
stem = os.path.splitext(os.path.basename(f))[0]
title = fm.get("title", stem)
key = norm(title) or norm(stem)
notes[key] = {
"stem": stem,
"title": title,
"is_discord": "source/discord" in (fm.get("tags") or []),
"base": fm.get("base", "—"),
"version": fm.get("version", "—"),
"status": fm.get("status", "—"),
"type": fm.get("type") or [],
"body_len": len(text),
}
results = {"curated_overlap": [], "discord_match": [], "unmatched": []}
scrape_dirs = sorted(glob.glob(os.path.join(SCRAPE, "*", "metadata.json")))
for jf in scrape_dirs:
meta = json.load(open(jf, encoding="utf-8"))
name = meta.get("name", "")
key = norm(name)
desc = meta.get("description", "")
fields = extract_fields(desc)
entry = {
"scrape_dir": os.path.basename(os.path.dirname(jf)),
"name": name,
"desc_len": len(desc),
"fields": fields,
"n_links": len(meta.get("links") or []),
"n_art": len(meta.get("art") or []),
}
note = notes.get(key)
if note is None:
results["unmatched"].append(entry)
elif note["is_discord"]:
entry["note_stem"] = note["stem"]
results["discord_match"].append(entry)
else:
entry["note_stem"] = note["stem"]
entry["note_body_len"] = note["body_len"]
results["curated_overlap"].append(entry)
# curated notes with NO scrape match (web-research only)
matched_keys = {norm(json.load(open(jf, encoding="utf-8")).get("name", ""))
for jf in scrape_dirs}
curated_no_scrape = [
n["stem"] for k, n in notes.items()
if not n["is_discord"] and k not in matched_keys
]
summary = {
"total_scrape": len(scrape_dirs),
"total_notes": len(notes),
"curated_overlap": len(results["curated_overlap"]),
"discord_match": len(results["discord_match"]),
"unmatched_scrape": len(results["unmatched"]),
"curated_without_scrape": len(curated_no_scrape),
}
out = {"summary": summary, "results": results,
"curated_without_scrape": sorted(curated_no_scrape)}
if "--report" in sys.argv:
p = sys.argv[sys.argv.index("--report") + 1]
json.dump(out, open(p, "w", encoding="utf-8"), indent=2, ensure_ascii=False)
print(f"wrote {p}")
print(json.dumps(summary, indent=2))
if "--list" in sys.argv:
print("\n== curated_overlap (merge Discord desc into curated note) ==")
for e in results["curated_overlap"]:
print(f" {e['note_stem']:40s} <- {e['scrape_dir']} ({e['desc_len']}c, {e['n_art']} art) {e['fields']}")
print("\n== unmatched scrape (no note yet) ==")
for e in results["unmatched"]:
print(f" {e['name']:45s} [{e['scrape_dir']}] ({e['desc_len']}c)")
if __name__ == "__main__":
main()