The changelog engine: turn git diffs into a civic changelog
The diff is the product — now literally. generate_changelog.py reads a
git diff of the entity tree between any two refs and says, in plain
language, what changed in the government: entities added / modified /
removed, grouped by type and by jurisdiction, dated from the commit.
CHANGELOG.md is seeded from the real history so far:
- initial load: +11,285 people across 51 states
- institutional gap: +233 bodies, ~532 people enriched
- elections + demographics: +2,494 candidates, +3,494 jurisdictions
(TX +541, FL +396, CA +341 ...)
On an initial load every entity is an addition; on a re-sync of the same
source, only real government changes will surface — which is the whole
point. When nothing changed, it says so.
make changelog [BASE=.. HEAD=..]. No third-party deps.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,174 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Turn a git diff of the entity tree into a human-readable civic changelog.
|
||||
|
||||
The diff is the product. Every sync that changes the mirror leaves a git
|
||||
diff; this reads that diff and says, in plain language, what changed in the
|
||||
government — added, modified, and removed entities, grouped by type and by
|
||||
jurisdiction.
|
||||
|
||||
Usage:
|
||||
python scripts/generate_changelog.py [BASE] [HEAD] # default HEAD~1..HEAD
|
||||
python scripts/generate_changelog.py <BASE> <HEAD> --write # prepend CHANGELOG.md
|
||||
|
||||
With no data changes between the refs, it says so. On an initial load every
|
||||
entity is an addition; on a re-sync of the same source, only real government
|
||||
changes surface — which is the point.
|
||||
"""
|
||||
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from collections import defaultdict
|
||||
|
||||
PREFIX = "data/jurisdictions/"
|
||||
|
||||
|
||||
def git(*args):
|
||||
return subprocess.run(["git", *args], capture_output=True, text=True,
|
||||
check=True).stdout
|
||||
|
||||
|
||||
def classify(path):
|
||||
"""(entity type, jurisdiction label) from a repo path — fast, no file reads."""
|
||||
if not path.startswith(PREFIX):
|
||||
return None, None
|
||||
rel = path[len(PREFIX):]
|
||||
# jurisdiction label
|
||||
m = re.match(r"us/states/([a-z]{2})/", rel)
|
||||
juris = m.group(1).upper() if m else ("US (federal)" if rel.startswith("us/") else "?")
|
||||
# entity type
|
||||
if "/candidates/" in rel:
|
||||
etype = "Candidate"
|
||||
elif "/bodies/" in rel:
|
||||
etype = "Body"
|
||||
elif rel.endswith("/index.md") or "/districts/" in rel:
|
||||
etype = "Jurisdiction"
|
||||
elif "/people/" in rel:
|
||||
etype = "Person"
|
||||
else:
|
||||
etype = "Other"
|
||||
return etype, juris
|
||||
|
||||
|
||||
PLURAL = {"Person": "people", "Body": "bodies", "Candidate": "candidates",
|
||||
"Jurisdiction": "jurisdictions", "Other": "files"}
|
||||
|
||||
|
||||
def pluralize(etype, n):
|
||||
return PLURAL.get(etype, etype.lower() + "s") if n != 1 else etype.lower()
|
||||
|
||||
|
||||
def nice_body(path):
|
||||
"""A readable name for a body file, derived from its path."""
|
||||
slug = path.rsplit("/", 1)[-1][:-3]
|
||||
kind = "subcommittee" if "/subcommittees/" in path else \
|
||||
"committee" if "/committees/" in path else "chamber"
|
||||
m = re.search(r"us/bodies/([a-z]+)/", path)
|
||||
chamber = m.group(1).title() + " " if m and kind != "chamber" else ""
|
||||
return f"{chamber}{slug.replace('-', ' ').title()} ({kind})"
|
||||
|
||||
|
||||
def main():
|
||||
argv = [a for a in sys.argv[1:] if not a.startswith("--")]
|
||||
write = "--write" in sys.argv
|
||||
base = argv[0] if len(argv) > 0 else "HEAD~1"
|
||||
head = argv[1] if len(argv) > 1 else "HEAD"
|
||||
|
||||
status = git("diff", "--name-status", f"{base}..{head}", "--", PREFIX)
|
||||
date = git("show", "-s", "--format=%cs", head).strip()
|
||||
short = git("rev-parse", "--short", base).strip() + ".." + \
|
||||
git("rev-parse", "--short", head).strip()
|
||||
|
||||
# tallies[etype][op] = count ; per_state[state][op] = count
|
||||
tallies = defaultdict(lambda: defaultdict(int))
|
||||
per_state = defaultdict(lambda: defaultdict(int))
|
||||
added_bodies = []
|
||||
added_by_state = defaultdict(lambda: defaultdict(int)) # state -> etype -> n added
|
||||
|
||||
for line in status.splitlines():
|
||||
parts = line.split("\t")
|
||||
code = parts[0][0] # A / M / D / R
|
||||
path = parts[-1] # new path for renames
|
||||
op = {"A": "added", "M": "modified", "D": "removed", "R": "modified"}.get(code)
|
||||
if not op:
|
||||
continue
|
||||
etype, juris = classify(path)
|
||||
if etype is None:
|
||||
continue
|
||||
tallies[etype][op] += 1
|
||||
per_state[juris][op] += 1
|
||||
if op == "added":
|
||||
added_by_state[juris][etype] += 1
|
||||
if etype == "Body":
|
||||
added_bodies.append(nice_body(path))
|
||||
|
||||
total = sum(sum(v.values()) for v in tallies.values())
|
||||
|
||||
out = []
|
||||
out.append(f"# Government Changelog — {date}")
|
||||
out.append("")
|
||||
out.append(f"*Diff `{short}` over the mirror.*")
|
||||
out.append("")
|
||||
if total == 0:
|
||||
out.append("No entity changes in this range — the government, as mirrored, held still.")
|
||||
emit(out, write)
|
||||
return 0
|
||||
|
||||
# summary line
|
||||
def phrase(t):
|
||||
d = tallies[t]
|
||||
bits = [f"+{d['added']}" if d["added"] else "",
|
||||
f"~{d['modified']}" if d["modified"] else "",
|
||||
f"−{d['removed']}" if d["removed"] else ""]
|
||||
return " ".join(b for b in bits if b)
|
||||
|
||||
out.append("## Summary")
|
||||
out.append("")
|
||||
for t in ("Person", "Body", "Candidate", "Jurisdiction", "Other"):
|
||||
if t in tallies:
|
||||
out.append(f"- **{t}**: {phrase(t)}")
|
||||
out.append("")
|
||||
|
||||
# bodies get named — they are few and load-bearing
|
||||
if added_bodies:
|
||||
out.append("## New bodies")
|
||||
out.append("")
|
||||
for b in sorted(added_bodies)[:40]:
|
||||
out.append(f"- {b}")
|
||||
if len(added_bodies) > 40:
|
||||
out.append(f"- … and {len(added_bodies) - 40} more")
|
||||
out.append("")
|
||||
|
||||
# jurisdiction rollup
|
||||
out.append("## By jurisdiction")
|
||||
out.append("")
|
||||
for juris in sorted(per_state, key=lambda s: (-sum(per_state[s].values()), s)):
|
||||
d = per_state[juris]
|
||||
bits = [f"+{d['added']}" if d["added"] else "",
|
||||
f"~{d['modified']}" if d["modified"] else "",
|
||||
f"−{d['removed']}" if d["removed"] else ""]
|
||||
detail = added_by_state.get(juris)
|
||||
types = ""
|
||||
if detail:
|
||||
types = " (" + ", ".join(f"{n} {pluralize(t, n)}"
|
||||
for t, n in sorted(detail.items())) + ")"
|
||||
out.append(f"- **{juris}**: {' '.join(b for b in bits if b)}{types}")
|
||||
out.append("")
|
||||
out.append(f"*{total} entity changes total.*")
|
||||
|
||||
emit(out, write)
|
||||
return 0
|
||||
|
||||
|
||||
def emit(lines, write):
|
||||
text = "\n".join(lines) + "\n"
|
||||
print(text)
|
||||
if write:
|
||||
from pathlib import Path
|
||||
cl = Path(__file__).resolve().parent.parent / "CHANGELOG.md"
|
||||
prev = cl.read_text() if cl.exists() else ""
|
||||
cl.write_text(text + "\n---\n\n" + prev)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user