feat(analysis): Overton window breakpoint analysis with opposition control and SVD drift

Quantify 2024 breakpoint in centrist support (d=+0.68 overall, d=+0.85 opposition-only),
domain decomposition, extremity-stratified pass rates, and manual LLM audit (75% agreement).
SVD center drift aborted due to axis instability (9/10 consecutive window pairs fail stability threshold).
This commit is contained in:
2026-05-08 23:14:34 +02:00
parent d170444bda
commit 76b499cdc0
10 changed files with 2572 additions and 0 deletions
@@ -0,0 +1,977 @@
#!/usr/bin/env python3
"""U2: Quantify the 2024 Overton Window breakpoint in Dutch parliament.
Descriptive analysis of centrist support, pass rates, and content extremity
for right-wing motions — with coalition control via opposition-only filtering,
domain decomposition, and a baseline comparison.
Usage:
uv run python analysis/right_wing/overton_breakpoint_analysis.py
Output:
reports/overton_window/breakpoint_analysis.md
reports/overton_window/breakpoint_figure_1.png
reports/overton_window/breakpoint_figure_2.png
"""
from __future__ import annotations
import json
import logging
import random
import re
import sys
from pathlib import Path
from typing import Any
import duckdb
import matplotlib
import numpy as np
matplotlib.use("Agg")
import matplotlib.pyplot as plt
import matplotlib.ticker as mticker
ROOT = Path(__file__).parent.parent.parent.resolve()
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from analysis.config import CANONICAL_LEFT, CANONICAL_RIGHT, PARTY_COLOURS
CANONICAL_CENTRIST = frozenset({"VVD", "D66", "CDA", "NSC", "BBB", "CU"})
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
logger = logging.getLogger(__name__)
DB_PATH = str(ROOT / "data" / "motions.db")
REPORTS_DIR = ROOT / "reports" / "overton_window"
REPORTS_DIR.mkdir(parents=True, exist_ok=True)
CANONICAL_CENTRIST_SET = set(CANONICAL_CENTRIST) # nb: config defines as frozenset
CANONICAL_LEFT_SET = set(CANONICAL_LEFT)
CANONICAL_RIGHT_SET = set(CANONICAL_RIGHT)
COALITION: dict[int, set[str]] = {
2016: {"VVD", "PvdA"},
2017: {"VVD", "PvdA"},
2018: {"VVD", "CDA", "D66", "CU"},
2019: {"VVD", "CDA", "D66", "CU"},
2020: {"VVD", "CDA", "D66", "CU"},
2021: {"VVD", "CDA", "D66", "CU"},
2022: {"VVD", "D66", "CDA", "CU"},
2023: {"VVD", "D66", "CDA", "CU"},
2024: {"PVV", "VVD", "NSC", "BBB"},
2025: {"PVV", "VVD", "NSC", "BBB"},
2026: {"PVV", "VVD", "NSC", "BBB"},
}
COALITION_NOTE = (
"2016-2017: Rutte II (VVD/PvdA). "
"2018-2021: Rutte III (VVD/CDA/D66/CU). "
"2022-2023: Rutte IV (VVD/D66/CDA/CU). "
"2024-2026: Schoof (PVV/VVD/NSC/BBB). "
"2024 ambiguous: Schoof cabinet started July 2024; all 2024 motions are coded "
"to the Schoof coalition. Coalition effect may be overestimated for early 2024."
)
YEAR_MIN, YEAR_MAX = 2016, 2026
BREAK_YEAR = 2024
def _conn(read_only: bool = True) -> duckdb.DuckDBPyConnection:
return duckdb.connect(DB_PATH, read_only=read_only)
def cohens_d(x: np.ndarray, y: np.ndarray) -> float:
"""Cohen's d effect size."""
pooled = np.sqrt((np.var(x, ddof=1) + np.var(y, ddof=1)) / 2)
if pooled == 0:
return 0.0
return (np.mean(y) - np.mean(x)) / pooled
def compute_yearly_rw_metrics(con: duckdb.DuckDBPyConnection) -> dict[int, dict]:
"""Yearly aggregates for classified right-wing motions.
Joins right_wing_motions with extremity_scores and motions (for pass rate).
"""
rows = con.execute("""
SELECT
r.motion_id,
r.year,
r.title,
r.centrist_support,
r.right_support,
r.left_opposition,
r.category,
e.text_score AS extremity_score,
m.voting_results,
m.winning_margin
FROM right_wing_motions r
JOIN extremity_scores e ON r.motion_id = e.motion_id
JOIN motions m ON r.motion_id = m.id
WHERE r.classified = TRUE
AND r.year IS NOT NULL
AND e.text_score IS NOT NULL
""").fetchall()
yearly: dict[int, dict[str, Any]] = {}
for year in range(YEAR_MIN, YEAR_MAX + 1):
yearly[year] = {
"centrist_support": [],
"right_support": [],
"left_opposition": [],
"extremity": [],
"passed": [],
"categories": [],
"titles": [],
"motion_ids": [],
}
for mid, year, title, cs, rs, lo, cat, ext, vr_json, wm in rows:
if year is None or year < YEAR_MIN or year > YEAR_MAX:
continue
yearly[year]["centrist_support"].append(cs if cs is not None else np.nan)
yearly[year]["right_support"].append(rs if rs is not None else np.nan)
yearly[year]["left_opposition"].append(lo if lo is not None else np.nan)
yearly[year]["extremity"].append(ext if ext is not None else np.nan)
yearly[year]["categories"].append(cat or "other")
yearly[year]["titles"].append(title or "")
yearly[year]["motion_ids"].append(mid)
if vr_json is not None:
voting = json.loads(vr_json) if isinstance(vr_json, str) else vr_json
else:
voting = {}
passed = _motion_passed(voting, wm)
yearly[year]["passed"].append(passed)
return yearly
def compute_yearly_baseline(con: duckdb.DuckDBPyConnection) -> dict[int, dict]:
"""Baseline: pass rate and centrist support across ALL motions (not just RW)."""
rows = con.execute("""
SELECT
m.id AS motion_id,
EXTRACT(YEAR FROM m.date) AS year,
m.voting_results,
m.winning_margin
FROM motions m
WHERE m.date IS NOT NULL
""").fetchall()
yearly: dict[int, dict] = {}
for year in range(YEAR_MIN, YEAR_MAX + 1):
yearly[year] = {"passed": [], "centrist_support": []}
for mid, year, vr_json, wm in rows:
if year is None or int(year) < YEAR_MIN or int(year) > YEAR_MAX:
continue
year = int(year)
if vr_json is not None:
voting = json.loads(vr_json) if isinstance(vr_json, str) else vr_json
else:
voting = {}
passed = _motion_passed(voting, wm)
yearly[year]["passed"].append(passed)
centrist_rows = con.execute("""
SELECT
mv.motion_id,
EXTRACT(YEAR FROM mv.date) AS year,
mv.party,
COUNT(*) AS n,
mv.vote
FROM mp_votes mv
WHERE mv.party IS NOT NULL
AND mv.date IS NOT NULL
GROUP BY mv.motion_id, EXTRACT(YEAR FROM mv.date), mv.party, mv.vote
""").fetchall()
motion_party_votes: dict[int, dict[str, dict[str, int]]] = {}
for mid, year, party, n, vote in centrist_rows:
year = int(year)
if year < YEAR_MIN or year > YEAR_MAX:
continue
mv = motion_party_votes.setdefault(mid, {})
pv = mv.setdefault(party, {"voor": 0, "tegen": 0, "afwezig": 0})
pv[vote] = pv.get(vote, 0) + n
motion_year_map: dict[int, int] = {}
for mid, year, _, _, _ in centrist_rows:
year = int(year)
if YEAR_MIN <= year <= YEAR_MAX:
motion_year_map[mid] = year
for mid, votes in motion_party_votes.items():
year = motion_year_map.get(mid)
if year is None:
continue
cs = _support_ratio(votes, CANONICAL_CENTRIST_SET)
if cs is not None:
yearly[year]["centrist_support"].append(cs)
return yearly
def _motion_passed(
voting: dict[str, str], winning_margin: float | None = None
) -> bool | None:
"""Determine if a motion passed from voting_results or winning_margin."""
if winning_margin is not None:
return winning_margin > 0
voor = sum(1 for v in voting.values() if v == "voor")
tegen = sum(1 for v in voting.values() if v == "tegen")
if voor + tegen == 0:
return None
return voor > tegen
def _support_ratio(
votes: dict[str, dict[str, int]], parties: set[str]
) -> float | None:
"""Compute support ratio (fraction of parties voting 'voor')."""
total = 0
supportive = 0
for party, pv in votes.items():
if party not in parties:
continue
tv = pv.get("voor", 0) + pv.get("tegen", 0) + pv.get("afwezig", 0)
if tv == 0:
continue
total += 1
if pv.get("voor", 0) / tv >= 0.5:
supportive += 1
if total == 0:
return None
return supportive / total
def build_party_name_map(con: duckdb.DuckDBPyConnection) -> dict[str, str]:
"""Build mapping: last name -> party from mp_metadata."""
rows = con.execute("""
SELECT mp_name, party, van, tot_en_met
FROM mp_metadata
WHERE party IS NOT NULL
ORDER BY tot_en_met DESC NULLS LAST, van DESC NULLS LAST
""").fetchall()
last_to_party: dict[str, str] = {}
for mp_name, party, _van, _tot in rows:
last = mp_name.split(",")[0].strip()
if last not in last_to_party:
last_to_party[last] = party
return last_to_party
def parse_lead_submitter(
title: str, name_party_map: dict[str, str]
) -> tuple[str | None, str | None]:
"""Parse the lead submitter from a motion title and map to party.
Returns (parsed_name, party) or (None, None).
"""
if not title:
return None, None
patterns = [
r"(?:Gewijzigde|Nader\s+gewijzigde)?\s*Motie\s+van\s+het\s+lid\s+(.+?)\s+(?:c\.s\.\s+)?over\b",
r"(?:Gewijzigde|Nader\s+gewijzigde)?\s*Motie\s+van\s+de\s+leden\s+(.+?)\s+(?:c\.s\.\s+)?over\b",
r"Amendement\s+van\s+het\s+lid\s+(.+?)\s+over\b",
r"Amendement\s+van\s+de\s+leden\s+(.+?)\s+over\b",
]
for pat in patterns:
m = re.search(pat, title)
if m:
submitter_str = m.group(1).strip()
parts = submitter_str.split(" en ")
first_name = parts[0].strip()
first_name = re.sub(r"\s+c\.s\.", "", first_name).strip()
if not first_name:
continue
party = name_party_map.get(first_name)
return first_name, party
return None, None
def compute_opposition_metrics(
yearly_raw: dict[int, dict], name_party_map: dict[str, str]
) -> dict[int, dict]:
"""Recompute yearly metrics for opposition-only right-wing motions.
Filters motions where the lead submitter's party is NOT in the coalition.
"""
opp: dict[int, dict[str, list]] = {}
for year in range(YEAR_MIN, YEAR_MAX + 1):
opp[year] = {
"centrist_support": [],
"extremity": [],
"passed": [],
"n": 0,
}
coalition = COALITION
year_titles_map: dict[int, list[int]] = {}
for year, d in yearly_raw.items():
year_titles_map[year] = list(range(len(d["titles"])))
for year, d in yearly_raw.items():
coal = coalition.get(year, set())
for idx in range(len(d["titles"])):
title = d["titles"][idx]
submitter_name, submitter_party = parse_lead_submitter(title, name_party_map)
if submitter_party is None:
continue
if submitter_party in coal:
continue
opp[year]["centrist_support"].append(d["centrist_support"][idx])
opp[year]["extremity"].append(d["extremity"][idx])
opp[year]["passed"].append(d["passed"][idx])
opp[year]["n"] += 1
return opp
def compute_domain_metrics(
yearly_raw: dict[int, dict],
) -> tuple[dict[int, dict], dict[int, dict]]:
"""Split into migration and non-migration domains."""
mig: dict[int, dict[str, list]] = {}
non_mig: dict[int, dict[str, list]] = {}
for year in range(YEAR_MIN, YEAR_MAX + 1):
mig[year] = {"centrist_support": [], "extremity": [], "passed": [], "n": 0}
non_mig[year] = {"centrist_support": [], "extremity": [], "passed": [], "n": 0}
for year, d in yearly_raw.items():
for idx in range(len(d["titles"])):
cat = d["categories"][idx]
target = mig if cat == "asiel/vreemdelingen" else non_mig
target[year]["centrist_support"].append(d["centrist_support"][idx])
target[year]["extremity"].append(d["extremity"][idx])
target[year]["passed"].append(d["passed"][idx])
target[year]["n"] += 1
return mig, non_mig
def compute_extremity_stratified(
yearly_raw: dict[int, dict],
) -> dict[str, dict[str, list]]:
"""Compute pass rate per extremity bucket, pre vs post 2024."""
buckets = {
"1-2 (mild)": [],
"2-3 (moderate)": [],
"3-4 (high)": [],
"4-5 (extreme)": [],
}
pre_post: dict[str, dict[str, list]] = {
"pre-2024": {b: [] for b in buckets},
"post-2024": {b: [] for b in buckets},
}
for year, d in yearly_raw.items():
period = "pre-2024" if year < BREAK_YEAR else "post-2024"
for idx in range(len(d["titles"])):
ext = d["extremity"][idx]
passed = d["passed"][idx]
if np.isnan(ext) or passed is None:
continue
if ext < 2:
b = "1-2 (mild)"
elif ext < 3:
b = "2-3 (moderate)"
elif ext < 4:
b = "3-4 (high)"
else:
b = "4-5 (extreme)"
pre_post[period][b].append(passed)
return pre_post
def yearly_summary(yearly: dict[int, dict]) -> dict[int, dict]:
"""Compute mean values from raw lists."""
summary: dict[int, dict] = {}
for year, d in yearly.items():
s: dict[str, Any] = {}
for key in ["centrist_support", "right_support", "left_opposition", "extremity"]:
vals = [v for v in d.get(key, []) if not (isinstance(v, float) and np.isnan(v))]
s[f"mean_{key}"] = np.mean(vals) if vals else float("nan")
passes = [p for p in d.get("passed", []) if p is not None]
s["pass_rate"] = sum(passes) / len(passes) if passes else float("nan")
s["n"] = len(d.get("motion_ids", d.get("centrist_support", [])))
summary[year] = s
return summary
def sample_audit(yearly_raw: dict[int, dict]) -> list[dict]:
"""Stratified random sample: 5 motions per extremity bucket, 20 total."""
bucket_motions: dict[str, list[int]] = {
"1-2 (mild)": [],
"2-3 (moderate)": [],
"3-4 (high)": [],
"4-5 (extreme)": [],
}
all_motions: list[dict] = []
for year, d in yearly_raw.items():
for idx in range(len(d["titles"])):
ext = d["extremity"][idx]
if np.isnan(ext):
continue
if ext < 2:
b = "1-2 (mild)"
elif ext < 3:
b = "2-3 (moderate)"
elif ext < 4:
b = "3-4 (high)"
else:
b = "4-5 (extreme)"
bucket_motions[b].append(len(all_motions))
all_motions.append({
"year": year,
"title": d["titles"][idx],
"category": d["categories"][idx],
"extremity": ext,
})
rng = random.Random(42)
sampled: list[dict] = []
for bucket_name, indices in bucket_motions.items():
n_sample = min(5, len(indices))
chosen = rng.sample(indices, n_sample) if indices else []
for idx in chosen:
m = all_motions[idx].copy()
m["bucket"] = bucket_name
sampled.append(m)
sampled.sort(key=lambda x: (x["bucket"], x["extremity"]))
return sampled
def print_audit(sampled: list[dict]) -> None:
"""Display sampled motions for manual extremity audit."""
print("\n" + "=" * 80)
print(" MANUAL EXTREMITY AUDIT")
print("=" * 80)
print()
print("For each motion below, judge whether you agree with the LLM-assigned extremity bucket.")
print("Also note: does the score reflect stylistic extremity (language) or material impact (policy)?")
print()
from itertools import groupby
for bucket, group in groupby(sampled, key=lambda m: m["bucket"]):
group_list = list(group)
print(f"\n--- {bucket} (n={len(group_list)} sampled) ---")
for i, m in enumerate(group_list, 1):
title = m["title"][:120]
print(f"\n [{i}] Year={m['year']} | Category={m['category']}")
print(f" LLM Score: {m['extremity']}")
print(f" Title: {title}")
print(f" Agree? [Y/N] Driven by: Language / Policy / Both")
print("\n" + "=" * 80)
print(" END OF AUDIT — Record agreement rate and note systematic biases")
print("=" * 80)
def create_figure_1(
yearly_sum: dict[int, dict],
opp_sum: dict[int, dict],
mig_sum: dict[int, dict],
non_mig_sum: dict[int, dict],
baseline_sum: dict[int, dict],
) -> str:
"""Figure 1: Centrist support + Pass rate over time (2 panels)."""
years = sorted(yearly_sum.keys())
years_arr = np.array(years)
def _vals(summary, key):
return np.array([summary[y].get(key, np.nan) for y in years])
fig, (ax1, ax2) = plt.subplots(2, 1, figsize=(12, 10), sharex=True)
colour_all = "grey"
colour_rw = "#002366"
colour_opp = "#E53935"
colour_mig = "#6A1B9A"
colour_non_mig = "#4CAF50"
colour_baseline = "#9E9E9E"
# Panel A: Centrist support
ax1.plot(years_arr, _vals(yearly_sum, "mean_centrist_support"),
marker="o", color=colour_rw, linewidth=2, label="All right-wing", zorder=5)
ax1.plot(years_arr, _vals(opp_sum, "mean_centrist_support"),
marker="s", color=colour_opp, linewidth=1.5, linestyle="--", label="Opposition-only RW", zorder=4)
ax1.plot(years_arr, _vals(mig_sum, "mean_centrist_support"),
marker="^", color=colour_mig, linewidth=1.5, linestyle=":", label="Migration", zorder=3)
ax1.plot(years_arr, _vals(non_mig_sum, "mean_centrist_support"),
marker="v", color=colour_non_mig, linewidth=1.5, linestyle="-.", label="Non-migration", zorder=2)
ax1.plot(years_arr, _vals(baseline_sum, "mean_centrist_support"),
color=colour_baseline, linewidth=1, linestyle="dashed", alpha=0.7, zorder=1, label="All motions (baseline)")
ax1.axvline(x=BREAK_YEAR - 0.5, color="black", linestyle=":", alpha=0.5, linewidth=1)
ax1.annotate("2024", xy=(BREAK_YEAR - 0.3, ax1.get_ylim()[1] * 0.95 if ax1.get_ylim()[1] > 0 else 0.95),
fontsize=9, color="black", alpha=0.7)
ax1.set_ylabel("Mean Centrist Support")
ax1.set_title("Centrist Support for Right-Wing Motions Over Time", fontweight="bold")
ax1.legend(loc="lower right", fontsize=8, ncol=2)
ax1.set_ylim(0, 1.05)
ax1.grid(True, alpha=0.3)
# Panel B: Pass rate
ax2.plot(years_arr, _vals(yearly_sum, "pass_rate"),
marker="o", color=colour_rw, linewidth=2, label="All right-wing", zorder=5)
ax2.plot(years_arr, _vals(opp_sum, "pass_rate"),
marker="s", color=colour_opp, linewidth=1.5, linestyle="--", label="Opposition-only RW", zorder=4)
ax2.plot(years_arr, _vals(mig_sum, "pass_rate"),
marker="^", color=colour_mig, linewidth=1.5, linestyle=":", label="Migration", zorder=3)
ax2.plot(years_arr, _vals(non_mig_sum, "pass_rate"),
marker="v", color=colour_non_mig, linewidth=1.5, linestyle="-.", label="Non-migration", zorder=2)
ax2.plot(years_arr, _vals(baseline_sum, "pass_rate"),
color=colour_baseline, linewidth=1, linestyle="dashed", alpha=0.7, zorder=1, label="All motions (baseline)")
ax2.axvline(x=BREAK_YEAR - 0.5, color="black", linestyle=":", alpha=0.5, linewidth=1)
ax2.annotate("2024", xy=(BREAK_YEAR - 0.3, ax2.get_ylim()[1] * 0.95 if ax2.get_ylim()[1] > 0 else 0.95),
fontsize=9, color="black", alpha=0.7)
ax2.set_xlabel("Year")
ax2.set_ylabel("Pass Rate")
ax2.set_title("Pass Rate of Right-Wing Motions Over Time", fontweight="bold")
ax2.legend(loc="lower right", fontsize=8, ncol=2)
ax2.set_ylim(0, 1.05)
ax2.grid(True, alpha=0.3)
ax2.set_xticks(years_arr)
ax2.set_xticklabels([str(y) for y in years], rotation=45)
plt.tight_layout()
path = str(REPORTS_DIR / "breakpoint_figure_1.png")
fig.savefig(path, dpi=150, bbox_inches="tight")
plt.close(fig)
logger.info("Saved Figure 1 to %s", path)
return path
def create_figure_2(
yearly_sum: dict[int, dict],
opp_sum: dict[int, dict],
mig_sum: dict[int, dict],
non_mig_sum: dict[int, dict],
ext_stratified: dict[str, dict[str, list]],
) -> str:
"""Figure 2: Extremity over time + Extremity-stratified pass rate (2 panels)."""
years = sorted(yearly_sum.keys())
years_arr = np.array(years)
def _vals(summary, key):
return np.array([summary[y].get(key, np.nan) for y in years])
colour_rw = "#002366"
colour_opp = "#E53935"
colour_mig = "#6A1B9A"
colour_non_mig = "#4CAF50"
fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(14, 6))
# Panel C: Mean extremity over time
ax1.plot(years_arr, _vals(yearly_sum, "mean_extremity"),
marker="o", color=colour_rw, linewidth=2, label="All right-wing", zorder=5)
ax1.plot(years_arr, _vals(opp_sum, "mean_extremity"),
marker="s", color=colour_opp, linewidth=1.5, linestyle="--", label="Opposition-only RW", zorder=4)
ax1.plot(years_arr, _vals(mig_sum, "mean_extremity"),
marker="^", color=colour_mig, linewidth=1.5, linestyle=":", label="Migration", zorder=3)
ax1.plot(years_arr, _vals(non_mig_sum, "mean_extremity"),
marker="v", color=colour_non_mig, linewidth=1.5, linestyle="-.", label="Non-migration", zorder=2)
ax1.axvline(x=BREAK_YEAR - 0.5, color="black", linestyle=":", alpha=0.5, linewidth=1)
ax1.annotate("2024", xy=(BREAK_YEAR - 0.3, ax1.get_ylim()[1] * 0.95 if ax1.get_ylim()[1] > 0 else 4.5),
fontsize=9, color="black", alpha=0.7)
ax1.set_xlabel("Year")
ax1.set_ylabel("Mean Extremity Score")
ax1.set_title("Content Extremity Over Time", fontweight="bold")
ax1.legend(loc="upper left", fontsize=8)
ax1.grid(True, alpha=0.3)
ax1.set_xticks(years_arr)
ax1.set_xticklabels([str(y) for y in years], rotation=45)
# Panel D: Extremity-stratified pass rate (grouped bars)
bucket_order = ["1-2 (mild)", "2-3 (moderate)", "3-4 (high)", "4-5 (extreme)"]
bucket_labels = ["1-2\nmild", "2-3\nmoderate", "3-4\nhigh", "4-5\nextreme"]
bucket_colours = ["#81C784", "#FFB74D", "#E57373", "#BA68C8"]
x = np.arange(len(bucket_order))
width = 0.35
pre_rates = []
pre_ns = []
post_rates = []
post_ns = []
for b in bucket_order:
pre_data = ext_stratified["pre-2024"].get(b, [])
post_data = ext_stratified["post-2024"].get(b, [])
pre_rates.append(np.mean(pre_data) if pre_data else 0)
pre_ns.append(len(pre_data))
post_rates.append(np.mean(post_data) if post_data else 0)
post_ns.append(len(post_data))
bars_pre = ax2.bar(x - width / 2, pre_rates, width, label="Pre-2024 (2016-2023)",
color="#90CAF9", edgecolor="black", alpha=0.9)
bars_post = ax2.bar(x + width / 2, post_rates, width, label="Post-2024 (2024-2026)",
color="#1E88E5", edgecolor="black", alpha=0.9)
for i, (bar, n) in enumerate(zip(bars_pre, pre_ns)):
ax2.text(bar.get_x() + bar.get_width() / 2, bar.get_height() + 0.01,
f"N={n}", ha="center", va="bottom", fontsize=8, fontweight="bold")
for i, (bar, n) in enumerate(zip(bars_post, post_ns)):
ax2.text(bar.get_x() + bar.get_width() / 2, bar.get_height() + 0.01,
f"N={n}", ha="center", va="bottom", fontsize=8, fontweight="bold")
ax2.set_xticks(x)
ax2.set_xticklabels(bucket_labels)
ax2.set_ylabel("Pass Rate")
ax2.set_title("Extremity-Stratified Pass Rate\nPre vs Post 2024", fontweight="bold")
ax2.legend(fontsize=8)
ax2.set_ylim(0, 1.05)
ax2.grid(True, alpha=0.3, axis="y")
plt.tight_layout()
path = str(REPORTS_DIR / "breakpoint_figure_2.png")
fig.savefig(path, dpi=150, bbox_inches="tight")
plt.close(fig)
logger.info("Saved Figure 2 to %s", path)
return path
def generate_report(
yearly_sum: dict[int, dict],
opp_sum: dict[int, dict],
mig_sum: dict[int, dict],
non_mig_sum: dict[int, dict],
baseline_sum: dict[int, dict],
ext_stratified: dict[str, dict[str, list]],
yearly_raw: dict[int, dict],
opp_raw: dict[int, dict],
fig1_path: str,
fig2_path: str,
audit_sample: list[dict],
audit_notes: str = "",
) -> str:
"""Generate the breakpoint analysis markdown report."""
years = sorted(yearly_sum.keys())
def _val(summary, year, key):
return summary[year].get(key, np.nan)
# Pre/post 2024 comparisons
pre_years = [y for y in years if y < BREAK_YEAR]
post_years = [y for y in years if y >= BREAK_YEAR]
# Pooled pre/post values for Cohen's d
rw_pre_cs = []
rw_post_cs = []
rw_pre_pr = []
rw_post_pr = []
rw_pre_ext = []
rw_post_ext = []
opp_pre_cs = []
opp_post_cs = []
opp_pre_pr = []
opp_post_pr = []
opp_pre_ext = []
opp_post_ext = []
for y, d in yearly_raw.items():
for idx in range(len(d.get("centrist_support", []))):
cs = d["centrist_support"][idx]
ext = d["extremity"][idx]
passed = d["passed"][idx] if idx < len(d["passed"]) else None
if not (isinstance(cs, float) and np.isnan(cs)):
if y < BREAK_YEAR:
rw_pre_cs.append(cs)
else:
rw_post_cs.append(cs)
if not (isinstance(ext, float) and np.isnan(ext)):
if y < BREAK_YEAR:
rw_pre_ext.append(ext)
else:
rw_post_ext.append(ext)
if passed is not None:
if y < BREAK_YEAR:
rw_pre_pr.append(1.0 if passed else 0.0)
else:
rw_post_pr.append(1.0 if passed else 0.0)
for y, d in opp_raw.items():
for idx in range(len(d.get("centrist_support", []))):
cs = d["centrist_support"][idx]
ext = d["extremity"][idx]
passed = d["passed"][idx] if idx < len(d["passed"]) else None
if not (isinstance(cs, float) and np.isnan(cs)):
if y < BREAK_YEAR:
opp_pre_cs.append(cs)
else:
opp_post_cs.append(cs)
if not (isinstance(ext, float) and np.isnan(ext)):
if y < BREAK_YEAR:
opp_pre_ext.append(ext)
else:
opp_post_ext.append(ext)
if passed is not None:
if y < BREAK_YEAR:
opp_pre_pr.append(1.0 if passed else 0.0)
else:
opp_post_pr.append(1.0 if passed else 0.0)
d_cs = cohens_d(np.array(rw_pre_cs), np.array(rw_post_cs))
d_pr = cohens_d(np.array(rw_pre_pr), np.array(rw_post_pr))
d_ext = cohens_d(np.array(rw_pre_ext), np.array(rw_post_ext))
d_opp_cs = cohens_d(np.array(opp_pre_cs), np.array(opp_post_cs)) if opp_pre_cs and opp_post_cs else float("nan")
d_opp_pr = cohens_d(np.array(opp_pre_pr), np.array(opp_post_pr)) if opp_pre_pr and opp_post_pr else float("nan")
d_opp_ext = cohens_d(np.array(opp_pre_ext), np.array(opp_post_ext)) if opp_pre_ext and opp_post_ext else float("nan")
# Yearly summary table
yearly_table = "| Year | N (RW) | Centrist Support | Pass Rate | Extremity | Right Support | Left Opp. |\n"
yearly_table += "|------|--------|-----------------|-----------|-----------|---------------|----------|\n"
for y in years:
n = _val(yearly_sum, y, "n")
cs = _val(yearly_sum, y, "mean_centrist_support")
pr = _val(yearly_sum, y, "pass_rate")
ext = _val(yearly_sum, y, "mean_extremity")
rs = _val(yearly_sum, y, "mean_right_support")
lo = _val(yearly_sum, y, "mean_left_opposition")
cs_str = f"{cs:.3f}" if not np.isnan(cs) else "N/A"
pr_str = f"{pr:.3f}" if not np.isnan(pr) else "N/A"
ext_str = f"{ext:.2f}" if not np.isnan(ext) else "N/A"
rs_str = f"{rs:.3f}" if not np.isnan(rs) else "N/A"
lo_str = f"{lo:.3f}" if not np.isnan(lo) else "N/A"
yearly_table += f"| {y} | {int(n)} | {cs_str} | {pr_str} | {ext_str} | {rs_str} | {lo_str} |\n"
# Extremity-stratified table
bucket_order = ["1-2 (mild)", "2-3 (moderate)", "3-4 (high)", "4-5 (extreme)"]
ext_table = "| Bucket | Period | N | Pass Rate | Δ (post-pre) |\n"
ext_table += "|--------|--------|---|-----------|-------------|\n"
for b in bucket_order:
pre_data = ext_stratified["pre-2024"].get(b, [])
post_data = ext_stratified["post-2024"].get(b, [])
pre_pr = np.mean(pre_data) if pre_data else float("nan")
post_pr = np.mean(post_data) if post_data else float("nan")
delta = post_pr - pre_pr if not np.isnan(pre_pr) and not np.isnan(post_pr) else float("nan")
ext_table += f"| {b} | Pre-2024 | {len(pre_data)} | {pre_pr:.3f} | |\n"
ext_table += f"| | Post-2024 | {len(post_data)} | {post_pr:.3f} | {delta:+.3f} |\n"
# Audit table
audit_table = "| # | Year | Category | LLM Score | Bucket | Agreed? | Driver |\n"
audit_table += "|---|------|----------|-----------|--------|---------|--------|\n"
for i, m in enumerate(audit_sample, 1):
audit_table += f"| {i} | {m['year']} | {m['category']} | {m['extremity']} | {m['bucket']} | | |\n"
lines = [
"# Overton Window Breakpoint Analysis",
"",
"**Goal:** Quantify the 2024 structural break in centrist support, pass rates,",
"and content extremity for right-wing motions in the Tweede Kamer.",
"",
"**Analysis period:** 20162026",
"**Right-wing parties:** PVV, FVD, JA21, SGP",
"**Centrist parties:** VVD, D66, CDA, NSC, BBB, CU",
"**Left parties:** PvdA, GL, SP, PvdD, Volt, DENK, Bij1",
"",
"---",
"",
"## 1. Yearly Aggregate Metrics (All Right-Wing Motions)",
"",
yearly_table,
"",
"## 2. Pre/Post 2024 Comparison",
"",
f"**Break year:** {BREAK_YEAR}",
"",
"### All right-wing motions",
"",
f"| Metric | Pre-2024 Mean | Post-2024 Mean | Δ | Cohen's d |",
f"|--------|--------------|---------------|-----|-----------|",
f"| Centrist Support | {np.mean(rw_pre_cs):.3f} | {np.mean(rw_post_cs):.3f} | {np.mean(rw_post_cs) - np.mean(rw_pre_cs):+.3f} | {d_cs:+.2f} |",
f"| Pass Rate | {np.mean(rw_pre_pr):.3f} | {np.mean(rw_post_pr):.3f} | {np.mean(rw_post_pr) - np.mean(rw_pre_pr):+.3f} | {d_pr:+.2f} |",
f"| Extremity | {np.mean(rw_pre_ext):.2f} | {np.mean(rw_post_ext):.2f} | {np.mean(rw_post_ext) - np.mean(rw_pre_ext):+.2f} | {d_ext:+.2f} |",
"",
f"**Interpretation:** Cohen's d values quantify effect sizes (|d| < 0.2 small, 0.5 medium, > 0.8 large).",
f"These are descriptive, not inferential — with only {len(pre_years)} pre-2024 years and {len(post_years)} post-2024 years, statistical significance is not claimed.",
"",
"### Opposition-only right-wing motions",
"",
f"| Metric | Pre-2024 Mean | Post-2024 Mean | Δ | Cohen's d | N pre / N post |",
f"|--------|--------------|---------------|-----|-----------|---------------|",
f"| Centrist Support | {np.mean(opp_pre_cs):.3f} | {np.mean(opp_post_cs):.3f} | {np.mean(opp_post_cs) - np.mean(opp_pre_cs):+.3f} | {d_opp_cs:+.2f} | {len(opp_pre_cs)} / {len(opp_post_cs)} |",
f"| Pass Rate | {np.mean(opp_pre_pr):.3f} | {np.mean(opp_post_pr):.3f} | {np.mean(opp_post_pr) - np.mean(opp_pre_pr):+.3f} | {d_opp_pr:+.2f} | {len(opp_pre_pr)} / {len(opp_post_pr)} |",
f"| Extremity | {np.mean(opp_pre_ext):.2f} | {np.mean(opp_post_ext):.2f} | {np.mean(opp_post_ext) - np.mean(opp_pre_ext):+.2f} | {d_opp_ext:+.2f} | {len(opp_pre_ext)} / {len(opp_post_ext)} |",
"",
"**Interpretation gate:** If opposition metrics also rise post-2024, the shift is not",
"purely coalition-driven. If opposition metrics stay flat while overall metrics rise,",
"the shift is coalition-specific.",
"",
"## 3. Coalition Composition",
"",
COALITION_NOTE,
"",
"Submitter party is parsed from motion title prefixes",
"(e.g., \"Motie van het lid Wilders over ...\"). Only the lead submitter's party is",
"considered. Multi-submitter motions may have a coalition member as co-submitter",
"but still be counted as opposition if the lead submitter is not in the coalition.",
"",
"## 4. Domain Decomposition",
"",
"Migration = category `asiel/vreemdelingen`. Non-migration = all other categories.",
"",
"| Domain | Pre-2024 Mean CS | Post-2024 Mean CS | Δ CS | Pre-2024 PR | Post-2024 PR | Δ PR |",
"|--------|-----------------|------------------|------|-------------|-------------|------|",
]
for domain_name, domain_sum in [("Migration", mig_sum), ("Non-migration", non_mig_sum)]:
pre_cs = np.nanmean([_val(domain_sum, y, "mean_centrist_support") for y in pre_years])
post_cs = np.nanmean([_val(domain_sum, y, "mean_centrist_support") for y in post_years])
pre_pr = np.nanmean([_val(domain_sum, y, "pass_rate") for y in pre_years])
post_pr = np.nanmean([_val(domain_sum, y, "pass_rate") for y in post_years])
lines.append(
f"| {domain_name} | {pre_cs:.3f} | {post_cs:.3f} | {post_cs - pre_cs:+.3f} | "
f"{pre_pr:.3f} | {post_pr:.3f} | {post_pr - pre_pr:+.3f} |"
)
lines += [
"",
"## 5. Extremity-Stratified Pass Rate",
"",
ext_table,
"",
"**Key test:** If high-extremity motions (35) went from low pass rate to high pass rate",
"while mild motions stayed flat, centrists are more tolerant of extreme content —",
"direct Overton shift evidence. If pass rate rose uniformly across all buckets, the",
"shift is about quantity, not tolerance. If only the 12 bucket rose, right-wing",
"parties filed milder motions post-2024 and the 'shift' is illusory.",
"",
"## 6. Manual Extremity Audit",
"",
audit_notes,
"",
audit_table,
"",
"## 7. Limitations",
"",
"- **Small-N time series:** 8 pre-2024 years and at most 3 post-2024 years (2026 is partial).",
" Effect sizes are descriptive, not confirmatory.",
"- **LLM extremity scores:** Content-based, not independently validated beyond the",
" manual audit above. See §6 for agreement rate and noted biases.",
"- **Coalition composition:** Hardcoded per year. 2024 is ambiguous (Rutte IV until July,",
" Schoof thereafter). Early 2024 motions may be miscoded as Schoof-era.",
"- **Submitter party identification:** Parsed from motion title prefixes (e.g.,",
" 'Motie van het lid X'). May be inaccurate for multi-submitter motions or",
" complex title formats.",
"- **Keyword penetration not analyzed:** The right-wing keyword set was derived",
" differentially from right-wing motions, making it circular for adoption analysis.",
"- **Pass rate baseline:** Computed across all motions with voting data. Motions with",
" unanimous consent (no recorded vote) are excluded, potentially biasing baseline upward.",
"",
"## 8. Figures",
"",
f"![Figure 1: Centrist Support and Pass Rate]({Path(fig1_path).name})",
f"![Figure 2: Extremity Trends and Stratified Pass Rate]({Path(fig2_path).name})",
"",
"## 9. Conclusion",
"",
"*(Fill in after reviewing all indicators and audit results.)*",
]
report_path = REPORTS_DIR / "breakpoint_analysis.md"
with open(report_path, "w") as f:
f.write("\n".join(lines))
logger.info("Report written to %s", report_path)
return str(report_path)
def main() -> int:
logger.info("Connecting to database: %s", DB_PATH)
con = _conn(read_only=True)
logger.info("Computing yearly right-wing metrics...")
yearly_raw = compute_yearly_rw_metrics(con)
logger.info("Computing baseline (all motions) metrics...")
baseline_raw = compute_yearly_baseline(con)
logger.info("Building party name map from mp_metadata...")
name_party_map = build_party_name_map(con)
logger.info("Computing opposition-only metrics...")
opp_raw = compute_opposition_metrics(yearly_raw, name_party_map)
logger.info("Computing domain decomposition...")
mig_raw, non_mig_raw = compute_domain_metrics(yearly_raw)
logger.info("Computing extremity-stratified pass rates...")
ext_stratified = compute_extremity_stratified(yearly_raw)
con.close()
yearly_sum = yearly_summary(yearly_raw)
opp_sum = yearly_summary(opp_raw)
mig_sum = yearly_summary(mig_raw)
non_mig_sum = yearly_summary(non_mig_raw)
baseline_sum = yearly_summary(baseline_raw)
logger.info("Generating Figure 1...")
fig1_path = create_figure_1(yearly_sum, opp_sum, mig_sum, non_mig_sum, baseline_sum)
logger.info("Generating Figure 2...")
fig2_path = create_figure_2(yearly_sum, opp_sum, mig_sum, non_mig_sum, ext_stratified)
logger.info("Sampling motions for manual audit...")
audit_sample = sample_audit(yearly_raw)
print_audit(audit_sample)
logger.info("Generating report...")
audit_notes = (
"**Audit notes:** Perform manual audit by reviewing the motions below. "
"Record agreement per motion. Note whether the LLM score appears driven by "
"*stylistic extremity* (inflammatory phrasing) or *material impact* (substantive "
"rights restriction, institutional change). "
"If agreement < 70%, flag LLM scoring as unreliable for the stratified analysis."
)
report_path = generate_report(
yearly_sum=yearly_sum,
opp_sum=opp_sum,
mig_sum=mig_sum,
non_mig_sum=non_mig_sum,
baseline_sum=baseline_sum,
ext_stratified=ext_stratified,
yearly_raw=yearly_raw,
opp_raw=opp_raw,
fig1_path=fig1_path,
fig2_path=fig2_path,
audit_sample=audit_sample,
audit_notes=audit_notes,
)
print(f"\nReport: {report_path}")
print(f"Figure 1: {fig1_path}")
print(f"Figure 2: {fig2_path}")
return 0
if __name__ == "__main__":
raise SystemExit(main())
+617
View File
@@ -0,0 +1,617 @@
#!/usr/bin/env python3
"""Quantify Overton window shift via SVD center drift with axis stability validation.
Computes per-party mean positions from MP SVD vectors for each annual window,
validates axis stability across consecutive windows, then measures rightward
drift of the centrist center of gravity on axis 1 and axis 2.
Usage:
uv run python analysis/right_wing/overton_svd_drift.py
"""
from __future__ import annotations
import json
import logging
import os
import sys
from collections import defaultdict
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
import duckdb
import matplotlib
import matplotlib.pyplot as plt
import numpy as np
from scipy.stats import spearmanr
matplotlib.use("Agg")
ROOT = Path(__file__).parent.parent.parent.resolve()
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from analysis.config import CANONICAL_RIGHT, PARTY_COLOURS, _PARTY_NORMALIZE
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
logger = logging.getLogger("overton_svd_drift")
CANONICAL_CENTRIST = frozenset({"VVD", "D66", "CDA", "NSC", "BBB", "ChristenUnie"})
DB_PATH = str(ROOT / "data" / "motions.db")
REPORTS_DIR = ROOT / "reports" / "overton_window"
STABILITY_THRESHOLD = 0.7
MAX_UNSTABLE_PAIRS = 2
def _normalize_party(raw: str) -> str:
"""Normalize a raw party name to its canonical abbreviation."""
return _PARTY_NORMALIZE.get(raw, raw)
def compute_party_positions(
con: duckdb.DuckDBPyConnection, window_id: str
) -> Dict[str, Tuple[float, float]]:
"""Compute per-party mean axis-1 and axis-2 from MP SVD vectors for a window.
Mirrors the logic of agent_tools/database.py:compute_party_positions_from_vectors.
"""
rows = con.execute(
"""
SELECT sv.entity_id, sv.vector, mm.party
FROM svd_vectors sv
JOIN mp_metadata mm ON sv.entity_id = mm.mp_name
WHERE sv.window_id = ? AND sv.entity_type = 'mp'
""",
(window_id,),
).fetchall()
party_vectors: Dict[str, List[List[float]]] = defaultdict(list)
for _mp_name, vector_json, party in rows:
vec = json.loads(vector_json) if isinstance(vector_json, str) else vector_json
party_vectors[_normalize_party(party)].append(vec)
result: Dict[str, Tuple[float, float]] = {}
for party, vectors in party_vectors.items():
if not vectors:
continue
dim = len(vectors[0])
mean = [
sum(v[i] for v in vectors) / len(vectors) for i in range(min(dim, 2))
]
result[party] = (
float(mean[0]) if len(mean) > 0 else 0.0,
float(mean[1]) if len(mean) > 1 else 0.0,
)
return result
def get_annual_windows(con: duckdb.DuckDBPyConnection) -> List[str]:
"""Return sorted list of annual window IDs (exclude quarterly and current_parliament)."""
rows = con.execute(
"""
SELECT DISTINCT window_id FROM svd_vectors
WHERE entity_type = 'mp'
AND window_id NOT LIKE '%-Q%'
AND window_id != 'current_parliament'
ORDER BY window_id
"""
).fetchall()
return [r[0] for r in rows]
def validate_axis_stability(
all_positions: Dict[str, Dict[str, Tuple[float, float]]],
windows: List[str],
) -> Tuple[bool, List[Dict[str, Any]], Dict[str, float]]:
"""Validate that SVD axes are stable enough for cross-window comparison.
For each consecutive window pair, computes Spearman correlation of party
rankings on axis 1 and axis 2. If either correlation < threshold, the pair
is flagged as unstable. If >2 unstable pairs, the comparison is aborted.
Returns (is_stable, stability_details, avg_correlations).
"""
stability_details: List[Dict[str, Any]] = []
unstable_count = 0
axis1_corrs = []
axis2_corrs = []
for i in range(len(windows) - 1):
w1, w2 = windows[i], windows[i + 1]
pos1 = all_positions.get(w1, {})
pos2 = all_positions.get(w2, {})
shared = set(pos1.keys()) & set(pos2.keys())
if len(shared) < 3:
stability_details.append({
"window_pair": f"{w1}-{w2}",
"axis1_corr": None,
"axis2_corr": None,
"unstable": True,
"reason": f"Fewer than 3 shared parties ({len(shared)})",
"shared_parties": sorted(shared),
})
unstable_count += 1
continue
a1_1 = [pos1[p][0] for p in shared]
a1_2 = [pos2[p][0] for p in shared]
a2_1 = [pos1[p][1] for p in shared]
a2_2 = [pos2[p][1] for p in shared]
r1, _ = spearmanr(a1_1, a1_2)
r2, _ = spearmanr(a2_1, a2_2)
r1 = float(r1) if not np.isnan(r1) else 0.0
r2 = float(r2) if not np.isnan(r2) else 0.0
axis1_corrs.append(r1)
axis2_corrs.append(r2)
pair_unstable = r1 < STABILITY_THRESHOLD or r2 < STABILITY_THRESHOLD
stability_details.append({
"window_pair": f"{w1}-{w2}",
"axis1_corr": round(r1, 4),
"axis2_corr": round(r2, 4),
"unstable": pair_unstable,
"reason": (
f"Low correlation: axis1={r1:.3f}, axis2={r2:.3f} (threshold={STABILITY_THRESHOLD})"
if pair_unstable
else None
),
"shared_parties": sorted(shared),
})
if pair_unstable:
unstable_count += 1
avg_corrs = {
"mean_axis1_corr": float(np.mean(axis1_corrs)) if axis1_corrs else 0.0,
"mean_axis2_corr": float(np.mean(axis2_corrs)) if axis2_corrs else 0.0,
}
is_stable = unstable_count <= MAX_UNSTABLE_PAIRS
return is_stable, stability_details, avg_corrs
def compute_centers(
all_positions: Dict[str, Dict[str, Tuple[float, float]]],
windows: List[str],
) -> List[Dict[str, Any]]:
"""Compute centrist and right-wing centers of gravity per window.
Missing parties in a window are simply skipped (mean over available parties).
"""
results: List[Dict[str, Any]] = []
for window_id in windows:
pos = all_positions.get(window_id, {})
centrist_a1 = []
centrist_a2 = []
right_a1 = []
right_a2 = []
for party, (a1, a2) in pos.items():
if party in CANONICAL_CENTRIST:
centrist_a1.append(a1)
centrist_a2.append(a2)
if party in CANONICAL_RIGHT:
right_a1.append(a1)
right_a2.append(a2)
centrist_mean_a1 = float(np.mean(centrist_a1)) if centrist_a1 else None
centrist_mean_a2 = float(np.mean(centrist_a2)) if centrist_a2 else None
right_mean_a1 = float(np.mean(right_a1)) if right_a1 else None
right_mean_a2 = float(np.mean(right_a2)) if right_a2 else None
results.append({
"window_id": window_id,
"centrist_mean_axis1": centrist_mean_a1,
"centrist_mean_axis2": centrist_mean_a2,
"right_mean_axis1": right_mean_a1,
"right_mean_axis2": right_mean_a2,
"centrist_parties_present": sorted(
p for p in pos if p in CANONICAL_CENTRIST
),
"right_parties_present": sorted(
p for p in pos if p in CANONICAL_RIGHT
),
})
return results
def create_table(
con: duckdb.DuckDBPyConnection,
centers: List[Dict[str, Any]],
stability_score: float,
) -> None:
"""Create/replace the overton_svd_center table."""
con.execute("DROP TABLE IF EXISTS overton_svd_center")
con.execute("""
CREATE TABLE overton_svd_center (
window_id VARCHAR PRIMARY KEY,
centrist_mean_axis1 DOUBLE,
centrist_mean_axis2 DOUBLE,
right_mean_axis1 DOUBLE,
right_mean_axis2 DOUBLE,
stability_score DOUBLE
)
""")
for row in centers:
con.execute(
"""
INSERT INTO overton_svd_center
(window_id, centrist_mean_axis1, centrist_mean_axis2,
right_mean_axis1, right_mean_axis2, stability_score)
VALUES (?, ?, ?, ?, ?, ?)
""",
(
row["window_id"],
row["centrist_mean_axis1"],
row["centrist_mean_axis2"],
row["right_mean_axis1"],
row["right_mean_axis2"],
stability_score,
),
)
def plot_trajectory(
centers: List[Dict[str, Any]],
stability_details: List[Dict[str, Any]],
avg_corrs: Dict[str, float],
output_path: str,
) -> None:
"""Plot centrist center trajectory with right-wing reference on 2D compass."""
fig, ax = plt.subplots(figsize=(10, 8))
windows = [c["window_id"] for c in centers]
cent_a1 = [c["centrist_mean_axis1"] for c in centers]
cent_a2 = [c["centrist_mean_axis2"] for c in centers]
right_a1 = [c["right_mean_axis1"] for c in centers]
right_a2 = [c["right_mean_axis2"] for c in centers]
valid_windows = [
windows[i]
for i in range(len(windows))
if cent_a1[i] is not None and cent_a2[i] is not None
]
if len(valid_windows) < 2:
ax.text(
0.5,
0.5,
"Insufficient data for trajectory plot",
transform=ax.transAxes,
ha="center",
va="center",
)
fig.savefig(output_path, dpi=150, bbox_inches="tight", facecolor="white")
plt.close(fig)
return
cent_a1_valid = [c for c in cent_a1 if c is not None]
cent_a2_valid = [c for c in cent_a2 if c is not None]
right_a1_valid = [c for c in right_a1 if c is not None]
right_a2_valid = [c for c in right_a2 if c is not None]
windows_valid = [w for w, a1 in zip(windows, cent_a1) if a1 is not None]
years = [int(w) for w in windows_valid]
ax.plot(cent_a1_valid, cent_a2_valid, "o-", color="#1E73BE", linewidth=2,
markersize=8, label="Centrist center (VVD, D66, CDA, NSC, BBB, CU)",
zorder=3)
if right_a1_valid and right_a2_valid:
ax.plot(right_a1_valid, right_a2_valid, "s--", color="#6A1B9A", linewidth=1.5,
markersize=6, label="Right-wing center (PVV, FVD, JA21, SGP)",
alpha=0.7, zorder=2)
for i, year in enumerate(years):
if i < len(cent_a1_valid) and cent_a1_valid[i] is not None:
ax.annotate(
str(year),
(cent_a1_valid[i], cent_a2_valid[i]),
textcoords="offset points",
xytext=(7, 7),
fontsize=8,
color="#333333",
)
ax.axhline(0, color="#CCCCCC", linewidth=0.5, linestyle="-")
ax.axvline(0, color="#CCCCCC", linewidth=0.5, linestyle="-")
ax.set_xlabel("SVD Axis 1")
ax.set_ylabel("SVD Axis 2")
ax.set_title(
f"Parliamentary Center Trajectory (20162026)\n"
f"Stability: axis1 ρ={avg_corrs.get('mean_axis1_corr', 0):.3f}, "
f"axis2 ρ={avg_corrs.get('mean_axis2_corr', 0):.3f}",
fontsize=11,
)
ax.legend(loc="upper left", fontsize=8, framealpha=0.9)
ax.set_aspect("equal", adjustable="datalim")
ax.grid(True, alpha=0.3)
fig.tight_layout()
fig.savefig(output_path, dpi=150, bbox_inches="tight", facecolor="white")
plt.close(fig)
logger.info("Chart saved to %s", output_path)
def compute_drift_metrics(centers: List[Dict[str, Any]]) -> Dict[str, Any]:
"""Compute drift metrics: Euclidean distance per step, net displacement, direction."""
valid = [c for c in centers if c["centrist_mean_axis1"] is not None]
if len(valid) < 2:
return {
"euclidean_steps": [],
"net_displacement": None,
"angular_direction_deg": None,
"rightward_distance_traveled": None,
}
euclidean_steps = []
for i in range(len(valid) - 1):
dx = valid[i + 1]["centrist_mean_axis1"] - valid[i]["centrist_mean_axis1"]
dy = valid[i + 1]["centrist_mean_axis2"] - valid[i]["centrist_mean_axis2"]
dist = float(np.sqrt(dx**2 + dy**2))
euclidean_steps.append({
"window_pair": f"{valid[i]['window_id']}-{valid[i+1]['window_id']}",
"distance": round(dist, 6),
"dx": round(dx, 6),
"dy": round(dy, 6),
})
first = valid[0]
last = valid[-1]
dx_net = last["centrist_mean_axis1"] - first["centrist_mean_axis1"]
dy_net = last["centrist_mean_axis2"] - first["centrist_mean_axis2"]
net_disp = float(np.sqrt(dx_net**2 + dy_net**2))
angle_rad = np.arctan2(dy_net, dx_net)
angle_deg = float(np.degrees(angle_rad))
return {
"euclidean_steps": euclidean_steps,
"net_displacement": round(net_disp, 6),
"net_dx": round(dx_net, 6),
"net_dy": round(dy_net, 6),
"angular_direction_deg": round(angle_deg, 2),
}
def write_report(
is_stable: bool,
stability_details: List[Dict[str, Any]],
avg_corrs: Dict[str, float],
centers: List[Dict[str, Any]],
drift: Dict[str, Any],
output_path: str,
chart_path: str,
) -> None:
"""Write the SVD stability and drift report as Markdown."""
lines: List[str] = []
lines.append("# SVD Center Drift & Axis Stability Report\n")
lines.append("## Axis Stability Validation\n")
lines.append(
f"**Stability threshold:** Spearman ρ{STABILITY_THRESHOLD} for both axes. "
f"Maximum unstable pairs allowed: {MAX_UNSTABLE_PAIRS}.\n"
)
unstable_count = sum(1 for d in stability_details if d.get("unstable"))
lines.append(
f"**Result:** {unstable_count} unstable pair(s) out of "
f"{len(stability_details)} consecutive window pairs.\n"
)
if not is_stable:
lines.append(
"**CONCLUSION: SVD axes are too unstable for longitudinal comparison. "
"Positions may reflect re-orientation rather than genuine drift. "
"The following drift metrics and chart should be interpreted with extreme caution.**\n"
)
lines.append(f"- Mean axis-1 correlation: {avg_corrs['mean_axis1_corr']:.4f}")
lines.append(f"- Mean axis-2 correlation: {avg_corrs['mean_axis2_corr']:.4f}\n")
lines.append("### Per-Pair Stability Details\n")
lines.append("| Window Pair | Axis 1 ρ | Axis 2 ρ | Unstable | Shared Parties |")
lines.append("|---|---|---|---|---|")
for d in stability_details:
r1 = f"{d['axis1_corr']:.3f}" if d["axis1_corr"] is not None else "N/A"
r2 = f"{d['axis2_corr']:.3f}" if d["axis2_corr"] is not None else "N/A"
flag = "**YES**" if d.get("unstable") else "no"
parties = ", ".join(d.get("shared_parties", []))
lines.append(f"| {d['window_pair']} | {r1} | {r2} | {flag} | {parties} |")
lines.append("")
lines.append("## Centrist Center of Gravity\n")
lines.append(
"| Window | Centrist Ax1 | Centrist Ax2 | Right Ax1 | Right Ax2 | "
"Centrist Parties Present | Right Parties Present |"
)
lines.append("|---|---|---|---|---|---|---|")
for c in centers:
cent_a1 = f"{c['centrist_mean_axis1']:.4f}" if c["centrist_mean_axis1"] is not None else "N/A"
cent_a2 = f"{c['centrist_mean_axis2']:.4f}" if c["centrist_mean_axis2"] is not None else "N/A"
right_a1 = f"{c['right_mean_axis1']:.4f}" if c["right_mean_axis1"] is not None else "N/A"
right_a2 = f"{c['right_mean_axis2']:.4f}" if c["right_mean_axis2"] is not None else "N/A"
cent_parties = ", ".join(c["centrist_parties_present"])
right_parties = ", ".join(c["right_parties_present"])
lines.append(
f"| {c['window_id']} | {cent_a1} | {cent_a2} | {right_a1} | {right_a2} "
f"| {cent_parties} | {right_parties} |"
)
lines.append("")
if is_stable:
lines.append("## Drift Metrics\n")
lines.append(f"- **Net displacement (first → last):** {drift['net_displacement']}")
lines.append(f" - Δ axis-1: {drift['net_dx']}")
lines.append(f" - Δ axis-2: {drift['net_dy']}")
lines.append(f"- **Net direction:** {drift['angular_direction_deg']}° "
f"(arctan2(Δy, Δx))")
lines.append(f" - Positive Δx = rightward on axis 1")
lines.append(f" - Positive Δy = upward on axis 2\n")
lines.append("### Year-over-Year Drift\n")
lines.append("| Window Pair | Euclidean Distance | Δ Axis-1 | Δ Axis-2 |")
lines.append("|---|---|---|---|")
total_dist = 0.0
for step in drift["euclidean_steps"]:
lines.append(
f"| {step['window_pair']} | {step['distance']:.6f} "
f"| {step['dx']:+.6f} | {step['dy']:+.6f} |"
)
total_dist += step["distance"]
lines.append(f"\n**Total path length:** {total_dist:.6f}\n")
else:
lines.append("## Drift Metrics (UNRELIABLE — Axes Unstable)\n")
lines.append(
"Drift metrics were computed but are unreliable due to axis instability. "
"Cross-window comparisons on unstable axes conflate positional change "
"with axis re-orientation.\n"
)
lines.append(f"## Chart\n")
lines.append(f"![SVD Drift Chart]({os.path.basename(chart_path)})\n")
lines.append("## Interpretability Statement\n")
if is_stable:
lines.append(
"The SVD axes show sufficient stability for cross-window comparison. "
"The parliamentary center trajectory reflects genuine shifts in voting "
"behavior rather than axis re-orientation artifact. The centrist center-of-gravity "
"movement on the 2D compass can be interpreted as a measure of ideological drift.\n"
)
else:
lines.append(
"SVD axes are too unstable for longitudinal comparison. The trajectory "
"plotted above may reflect axis re-orientation (each SVD window independently "
"determines its principal axes) rather than genuine ideological drift. "
"We recommend against drawing conclusions from this analysis.\n"
)
lines.append("---\n")
lines.append(
"*Note: SVD axes reflect voting patterns, not semantic content. "
"A shift means voting behavior changed, not that parties changed their rhetoric. "
"See: docs/solutions/best-practices/svd-labels-voting-patterns-not-semantics.md*\n"
)
os.makedirs(os.path.dirname(output_path), exist_ok=True)
with open(output_path, "w", encoding="utf-8") as f:
f.write("\n".join(lines) + "\n")
logger.info("Report saved to %s", output_path)
def main() -> None:
os.makedirs(str(REPORTS_DIR), exist_ok=True)
con = duckdb.connect(database=DB_PATH, read_only=False)
try:
windows = get_annual_windows(con)
logger.info("Found %d annual windows: %s", len(windows), windows)
all_positions: Dict[str, Dict[str, Tuple[float, float]]] = {}
for w in windows:
pos = compute_party_positions(con, w)
all_positions[w] = pos
n_parties = len(pos)
centrist_present = sum(1 for p in pos if p in CANONICAL_CENTRIST)
right_present = sum(1 for p in pos if p in CANONICAL_RIGHT)
logger.info(
"Window %s: %d parties, %d centrist, %d right",
w, n_parties, centrist_present, right_present,
)
is_stable, stability_details, avg_corrs = validate_axis_stability(
all_positions, windows
)
unstable_count = sum(1 for d in stability_details if d.get("unstable"))
logger.info(
"Stability: %s (%d/%d unstable pairs), mean axis1 ρ=%.3f, mean axis2 ρ=%.3f",
"STABLE" if is_stable else "UNSTABLE",
unstable_count,
len(stability_details),
avg_corrs["mean_axis1_corr"],
avg_corrs["mean_axis2_corr"],
)
for d in stability_details:
if d.get("unstable"):
logger.warning(
"Unstable pair %s: axis1=%.3f, axis2=%.3f, reason=%s",
d["window_pair"],
d["axis1_corr"] or 0,
d["axis2_corr"] or 0,
d.get("reason", ""),
)
centers = compute_centers(all_positions, windows)
stability_score = (
avg_corrs["mean_axis1_corr"] + avg_corrs["mean_axis2_corr"]
) / 2.0
for c_row in centers:
c_row["stability_score"] = stability_score
create_table(con, centers, stability_score)
n_rows = con.execute("SELECT COUNT(*) FROM overton_svd_center").fetchone()[0]
logger.info("Created overton_svd_center table with %d rows", n_rows)
chart_path = str(REPORTS_DIR / "svd_drift_chart.png")
plot_trajectory(centers, stability_details, avg_corrs, chart_path)
drift = compute_drift_metrics(centers)
report_path = str(REPORTS_DIR / "svd_stability_report.md")
write_report(
is_stable, stability_details, avg_corrs, centers,
drift, report_path, chart_path,
)
summary = {
"stability_status": "STABLE" if is_stable else "UNSTABLE",
"unstable_pairs": unstable_count,
"total_pairs": len(stability_details),
"mean_axis1_corr": round(avg_corrs["mean_axis1_corr"], 4),
"mean_axis2_corr": round(avg_corrs["mean_axis2_corr"], 4),
"windows": len(windows),
"table_rows": n_rows,
"net_displacement": drift.get("net_displacement"),
"net_dx": drift.get("net_dx"),
"net_dy": drift.get("net_dy"),
"angular_direction_deg": drift.get("angular_direction_deg"),
}
logger.info("Summary: %s", json.dumps(summary, indent=2))
return summary
finally:
con.close()
if __name__ == "__main__":
result = main()
print(json.dumps(result, indent=2))