refactor: delete stale files and consolidate .mindmodel structure
Deleted stale root-level Python files: - main.py (unused 'Hello world' script) - verify.py (unused table info script) - scraper.py (unused MotionScraper class) - scheduler.py (unused DataUpdateScheduler class) Deleted duplicate .mindmodel root YAML files (subdirectory versions are more comprehensive): - anti-patterns.yaml, architecture.yaml, conventions.yaml - dependencies.yaml, domain.yaml, domain-glossary.yaml - stack.yaml, tech-stack.yaml, workflows.yaml Added comprehensive .mindmodel subdirectories: - constraints/ (naming, db-schema, error-handling, types, etc.) - patterns/ (api, architecture, database, python, streamlit, etc.) - examples/ (code examples for each pattern) - anti-patterns/, architecture/, conventions/, dependencies/, domain/, stack/ Updated ARCHITECTURE.md to reflect current codebase: - Removed references to non-existent files - Added missing files (explorer.py, explorer_helpers.py, pipeline/) - Added directory structure documentation - Updated tech stack to include scipy, sklearn, umap Updated .gitignore: - Added patterns for generated analysis files - Added .worktrees/ pattern (was already in gitignore but dir was deleted) Removed empty .worktrees/ directory
This commit is contained in:
@@ -0,0 +1,265 @@
|
||||
# API Client Patterns
|
||||
|
||||
## Base API Client Pattern
|
||||
|
||||
Using requests.Session for connection pooling:
|
||||
|
||||
```python
|
||||
# api_client.py
|
||||
import requests
|
||||
from typing import Dict, List, Optional
|
||||
from config import config
|
||||
|
||||
class TweedeKamerAPI:
|
||||
def __init__(self):
|
||||
self.odata_base_url = "https://gegevensmagazijn.tweedekamer.nl/OData/v4/2.0"
|
||||
self.session = requests.Session()
|
||||
self.session.headers.update({
|
||||
"Accept": "application/json",
|
||||
"User-Agent": "Dutch-Political-Compass-Tool/1.0",
|
||||
})
|
||||
|
||||
def get_motions(
|
||||
self,
|
||||
start_date: datetime = None,
|
||||
end_date: datetime = None,
|
||||
limit: int = 500,
|
||||
) -> List[Dict]:
|
||||
"""Get motions with voting results using OData API."""
|
||||
if not start_date:
|
||||
start_date = datetime.now() - timedelta(days=730)
|
||||
|
||||
try:
|
||||
voting_records, besluit_meta = self._get_voting_records(
|
||||
start_date, end_date, limit
|
||||
)
|
||||
return self._process_voting_records(voting_records, besluit_meta)
|
||||
except Exception as e:
|
||||
print(f"Error fetching motions from API: {e}")
|
||||
return []
|
||||
```
|
||||
|
||||
## OData Pagination Pattern
|
||||
|
||||
Handle server-side pagination with $skip:
|
||||
|
||||
```python
|
||||
def _get_voting_records(
|
||||
self,
|
||||
start_date: datetime,
|
||||
end_date: datetime = None,
|
||||
limit: int = 50000
|
||||
) -> tuple:
|
||||
"""Fetch with automatic pagination."""
|
||||
|
||||
filter_query = (
|
||||
f"GewijzigdOp ge {start_date.strftime('%Y-%m-%d')}T00:00:00Z"
|
||||
" and StemmingsSoort ne null"
|
||||
" and Verwijderd eq false"
|
||||
)
|
||||
|
||||
page_size = 250 # API caps $top at 250
|
||||
base_url = f"{self.odata_base_url}/Besluit"
|
||||
base_params = {
|
||||
"$filter": filter_query,
|
||||
"$top": page_size,
|
||||
"$expand": "Stemming",
|
||||
"$orderby": "GewijzigdOp desc",
|
||||
}
|
||||
|
||||
all_records = []
|
||||
skip = 0
|
||||
|
||||
while len(all_records) < limit:
|
||||
params = {**base_params, "$skip": skip}
|
||||
response = self.session.get(
|
||||
base_url,
|
||||
params=params,
|
||||
timeout=config.API_TIMEOUT
|
||||
)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
|
||||
besluit_page = data.get("value", [])
|
||||
if not besluit_page:
|
||||
break
|
||||
|
||||
# Process page
|
||||
for besluit in besluit_page:
|
||||
all_records.extend(self._extract_votes(besluit))
|
||||
|
||||
skip += page_size
|
||||
|
||||
return all_records
|
||||
```
|
||||
|
||||
## Retry with Backoff Pattern
|
||||
|
||||
For transient failures:
|
||||
|
||||
```python
|
||||
# ai_provider.py
|
||||
import time
|
||||
import random
|
||||
from requests.exceptions import ConnectionError
|
||||
|
||||
def _post_with_retries(
|
||||
path: str,
|
||||
json: dict,
|
||||
retries: int = 3
|
||||
) -> requests.Response:
|
||||
"""POST with exponential backoff retry."""
|
||||
|
||||
backoff = 0.5
|
||||
for attempt in range(1, retries + 1):
|
||||
try:
|
||||
resp = requests.post(url, json=json, headers=headers, timeout=10)
|
||||
|
||||
# Handle rate limiting
|
||||
if resp.status_code == 429:
|
||||
if attempt == retries:
|
||||
raise ProviderError("Rate limited")
|
||||
|
||||
retry_after = resp.headers.get("Retry-After")
|
||||
if retry_after:
|
||||
time.sleep(int(retry_after))
|
||||
else:
|
||||
sleep = backoff * (2 ** (attempt - 1))
|
||||
sleep += random.uniform(0, sleep * 0.1)
|
||||
time.sleep(sleep)
|
||||
continue
|
||||
|
||||
# Handle server errors
|
||||
if 500 <= resp.status_code < 600:
|
||||
if attempt == retries:
|
||||
raise ProviderError(f"Server error: {resp.status_code}")
|
||||
time.sleep(backoff * (2 ** (attempt - 1)))
|
||||
continue
|
||||
|
||||
return resp
|
||||
|
||||
except ConnectionError as exc:
|
||||
if attempt == retries:
|
||||
raise ProviderError(f"Connection error: {exc}")
|
||||
time.sleep(backoff * (2 ** (attempt - 1)))
|
||||
|
||||
raise ProviderError("Failed after retries")
|
||||
```
|
||||
|
||||
## Batch Processing Pattern
|
||||
|
||||
Process items in batches to manage API limits:
|
||||
|
||||
```python
|
||||
def get_embeddings_with_retry(
|
||||
texts: List[str],
|
||||
batch_size: int = 50,
|
||||
retries: int = 3,
|
||||
) -> List[Optional[List[float]]]:
|
||||
"""Process embeddings in batches with fallback to single items."""
|
||||
|
||||
results = [None] * len(texts)
|
||||
|
||||
i = 0
|
||||
while i < len(texts):
|
||||
end = min(len(texts), i + batch_size)
|
||||
chunk = texts[i:end]
|
||||
|
||||
# Try batch first
|
||||
try:
|
||||
emb_chunk = get_embeddings_batch(chunk)
|
||||
for j, emb in enumerate(emb_chunk):
|
||||
results[i + j] = emb
|
||||
i = end
|
||||
continue
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Fallback: single items
|
||||
for j, text in enumerate(chunk):
|
||||
try:
|
||||
results[i + j] = get_embedding(text)
|
||||
except Exception:
|
||||
results[i + j] = None
|
||||
|
||||
i = end
|
||||
|
||||
return results
|
||||
```
|
||||
|
||||
## Response Validation Pattern
|
||||
|
||||
Validate API responses before processing:
|
||||
|
||||
```python
|
||||
def _process_response(self, response: requests.Response) -> Dict:
|
||||
"""Validate and parse API response."""
|
||||
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
|
||||
if "value" not in data:
|
||||
raise ValueError("Unexpected response format: missing 'value' key")
|
||||
|
||||
return data
|
||||
|
||||
def _validate_besluit(self, besluit: Dict) -> bool:
|
||||
"""Check required fields exist."""
|
||||
required = ["Id", "GewijzigdOp"]
|
||||
return all(field in besluit for field in required)
|
||||
```
|
||||
|
||||
## Error Handling Patterns
|
||||
|
||||
Always provide safe fallbacks:
|
||||
|
||||
```python
|
||||
def safe_api_call(self, endpoint: str, params: Dict = None) -> List[Dict]:
|
||||
"""Call API with error handling and fallback."""
|
||||
try:
|
||||
response = self.session.get(
|
||||
endpoint,
|
||||
params=params,
|
||||
timeout=config.API_TIMEOUT
|
||||
)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
return data.get("value", [])
|
||||
except requests.Timeout:
|
||||
_logger.warning(f"API timeout for {endpoint}")
|
||||
return []
|
||||
except requests.HTTPError as e:
|
||||
_logger.error(f"HTTP error: {e}")
|
||||
return []
|
||||
except Exception as e:
|
||||
_logger.error(f"API call failed: {e}")
|
||||
return []
|
||||
```
|
||||
|
||||
## Session Management
|
||||
|
||||
Reuse session for connection pooling:
|
||||
|
||||
```python
|
||||
class TweedeKamerAPI:
|
||||
def __init__(self):
|
||||
self.session = requests.Session()
|
||||
self.session.headers.update({
|
||||
"Accept": "application/json",
|
||||
"User-Agent": "Dutch-Political-Compass-Tool/1.0",
|
||||
})
|
||||
|
||||
def close(self):
|
||||
"""Clean up session when done."""
|
||||
self.session.close()
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *args):
|
||||
self.close()
|
||||
|
||||
# Usage
|
||||
with TweedeKamerAPI() as api:
|
||||
motions = api.get_motions(start_date)
|
||||
```
|
||||
@@ -0,0 +1,230 @@
|
||||
# Architectural Patterns
|
||||
|
||||
## Repository Pattern
|
||||
|
||||
The `MotionDatabase` class acts as a repository, encapsulating all database operations behind a clean interface.
|
||||
|
||||
```python
|
||||
# database.py
|
||||
class MotionDatabase:
|
||||
def __init__(self, db_path: str = config.DATABASE_PATH):
|
||||
self.db_path = db_path
|
||||
self._init_database()
|
||||
|
||||
def get_motion(self, motion_id: int) -> Optional[Dict]:
|
||||
"""Get a single motion by ID."""
|
||||
conn = duckdb.connect(self.db_path)
|
||||
try:
|
||||
result = conn.execute(
|
||||
"SELECT * FROM motions WHERE id = ?", (motion_id,)
|
||||
).fetchone()
|
||||
return result
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
def get_filtered_motions(
|
||||
self,
|
||||
policy_area: str = "Alle",
|
||||
min_margin: float = 0.0,
|
||||
max_margin: float = 1.0,
|
||||
limit: int = 10
|
||||
) -> List[Dict]:
|
||||
"""Get filtered list of motions."""
|
||||
...
|
||||
```
|
||||
|
||||
**Usage**: Import the singleton instance for all DB operations.
|
||||
```python
|
||||
from database import db
|
||||
|
||||
motions = db.get_filtered_motions(policy_area="Klimaat", limit=20)
|
||||
```
|
||||
|
||||
## Facade Pattern
|
||||
|
||||
Simplified interfaces over complex subsystems.
|
||||
|
||||
### MotionDatabase Facade
|
||||
```python
|
||||
# Single entry point for all database operations
|
||||
db = MotionDatabase() # Singleton instance
|
||||
|
||||
# Operations are abstracted:
|
||||
db.create_session(total_motions)
|
||||
db.record_vote(session_id, motion_id, vote)
|
||||
db.get_party_results(session_id)
|
||||
```
|
||||
|
||||
### API Client Facade
|
||||
```python
|
||||
# api_client.py
|
||||
class TweedeKamerAPI:
|
||||
def __init__(self):
|
||||
self.session = requests.Session() # Connection pooling
|
||||
|
||||
def get_motions(self, start_date, end_date) -> List[Dict]:
|
||||
"""Simple interface hiding OData pagination details."""
|
||||
voting_records, besluit_meta = self._get_voting_records(start_date, end_date)
|
||||
return self._process_voting_records(voting_records, besluit_meta)
|
||||
```
|
||||
|
||||
### MotionScraper Facade
|
||||
```python
|
||||
# scraper.py (if used)
|
||||
class MotionScraper:
|
||||
def get_motion_content(self, url: str) -> Optional[str]:
|
||||
"""Extract body text from official website."""
|
||||
...
|
||||
```
|
||||
|
||||
## Pipeline Pattern
|
||||
|
||||
Sequential phases with explicit dependencies:
|
||||
|
||||
```
|
||||
pipeline/run_pipeline.py
|
||||
├── Phase 1: fetch_mp_metadata
|
||||
│ └── pipeline/fetch_mp_metadata.py
|
||||
├── Phase 2: extract_mp_votes
|
||||
│ └── pipeline/extract_mp_votes.py
|
||||
├── Phase 3: svd_pipeline
|
||||
│ └── pipeline/svd_pipeline.py
|
||||
├── Phase 4: text_pipeline (gap-fill)
|
||||
│ └── pipeline/text_pipeline.py
|
||||
└── Phase 5: fusion (combine SVD + text)
|
||||
└── pipeline/fusion.py
|
||||
```
|
||||
|
||||
### Phase Orchestration
|
||||
```python
|
||||
# pipeline/run_pipeline.py
|
||||
def run(args: argparse.Namespace) -> int:
|
||||
db = MotionDatabase(args.db_path)
|
||||
|
||||
# Phase 1: MP metadata
|
||||
if not args.skip_metadata:
|
||||
from pipeline.fetch_mp_metadata import fetch_mp_metadata
|
||||
fetch_mp_metadata(db_path=db.db_path)
|
||||
|
||||
# Phase 2: Extract votes
|
||||
if not args.skip_extract:
|
||||
from pipeline.extract_mp_votes import extract_mp_votes
|
||||
extract_mp_votes(db_path=db.db_path)
|
||||
|
||||
# Phase 3: SVD per window
|
||||
if not args.skip_svd:
|
||||
from pipeline.svd_pipeline import run_svd_pipeline
|
||||
run_svd_pipeline(db, windows, args.svd_k)
|
||||
|
||||
# ... additional phases
|
||||
```
|
||||
|
||||
## Strategy Pattern
|
||||
|
||||
Interchangeable algorithms for axis computation:
|
||||
|
||||
```python
|
||||
# analysis/political_axis.py
|
||||
def compute_political_axis(
|
||||
vectors: Dict[str, np.ndarray],
|
||||
method: str = "pca" # or "anchor"
|
||||
) -> Tuple[np.ndarray, np.ndarray]:
|
||||
"""Compute political axis using specified method.
|
||||
|
||||
Methods:
|
||||
- 'pca': Use first principal component
|
||||
- 'anchor': Use predefined anchor motions
|
||||
"""
|
||||
if method == "pca":
|
||||
return _compute_pca_axis(vectors)
|
||||
elif method == "anchor":
|
||||
return _compute_anchor_axis(vectors)
|
||||
```
|
||||
|
||||
## Visitor Pattern
|
||||
|
||||
External operations on data structures:
|
||||
|
||||
```python
|
||||
# analysis/trajectory.py
|
||||
def _procrustes_align_windows(
|
||||
window_vecs: Dict[str, Dict[str, np.ndarray]],
|
||||
min_overlap: int = 5,
|
||||
) -> Dict[str, Dict[str, np.ndarray]]:
|
||||
"""Align SVD vectors across windows using Procrustes rotations.
|
||||
|
||||
Takes the first window as reference and aligns each subsequent window
|
||||
to it via orthogonal Procrustes on the set of common entities.
|
||||
"""
|
||||
```
|
||||
|
||||
## Builder Pattern
|
||||
|
||||
Configuration via method chaining:
|
||||
|
||||
```python
|
||||
# CLI argument parsing
|
||||
parser = argparse.ArgumentParser(description="Pipeline runner")
|
||||
parser.add_argument("--db-path", default="data/motions.db")
|
||||
parser.add_argument("--start-date", default=None)
|
||||
parser.add_argument("--end-date", default=None)
|
||||
parser.add_argument("--window-size", choices=["quarterly", "annual"], default="quarterly")
|
||||
parser.add_argument("--svd-k", type=int, default=50)
|
||||
```
|
||||
|
||||
## Decorator Pattern
|
||||
|
||||
Retry logic for transient failures:
|
||||
|
||||
```python
|
||||
# pipeline/ai_provider_wrapper.py
|
||||
def get_embeddings_with_retry(
|
||||
texts: List[str],
|
||||
retries: int = 3,
|
||||
batch_size: int = 50,
|
||||
) -> List[Optional[List[float]]]:
|
||||
"""Return embeddings with automatic retry on failure."""
|
||||
for attempt in range(1, retries + 1):
|
||||
try:
|
||||
return _embedder(texts, batch_size=len(texts))
|
||||
except Exception as exc:
|
||||
if attempt == retries:
|
||||
break
|
||||
time.sleep(backoff * (2 ** (attempt - 1)))
|
||||
return [None] * len(texts) # Safe fallback
|
||||
```
|
||||
|
||||
## Data Patterns
|
||||
|
||||
### Batch Processing
|
||||
Process items in chunks to manage memory and API limits:
|
||||
```python
|
||||
for i in range(0, len(items), batch_size):
|
||||
chunk = items[i:i + batch_size]
|
||||
process_batch(chunk)
|
||||
```
|
||||
|
||||
### Caching
|
||||
Pre-compute and store expensive results:
|
||||
```python
|
||||
# SimilarityCache table stores computed similarities
|
||||
db.get_similarity(motion_a, motion_b)
|
||||
```
|
||||
|
||||
### Lazy Loading
|
||||
Load data only when needed:
|
||||
```python
|
||||
class MotionDatabase:
|
||||
@property
|
||||
def _connection(self):
|
||||
if self._conn is None:
|
||||
self._conn = duckdb.connect(self.db_path)
|
||||
return self._conn
|
||||
```
|
||||
|
||||
### Vectorization
|
||||
Use numpy for batch operations:
|
||||
```python
|
||||
vectors = np.array([v for v in entity_vectors.values()])
|
||||
normalized = vectors / np.linalg.norm(vectors, axis=1, keepdims=True)
|
||||
```
|
||||
@@ -0,0 +1,239 @@
|
||||
# DuckDB Database Patterns
|
||||
|
||||
## Connection Management
|
||||
|
||||
### Pattern 1: Short-lived per Method (Most Common)
|
||||
|
||||
Always create a new connection, use try/finally for cleanup:
|
||||
|
||||
```python
|
||||
# database.py
|
||||
class MotionDatabase:
|
||||
def get_motion(self, motion_id: int) -> Optional[Dict]:
|
||||
conn = duckdb.connect(self.db_path)
|
||||
try:
|
||||
result = conn.execute(
|
||||
"SELECT * FROM motions WHERE id = ?",
|
||||
(motion_id,)
|
||||
).fetchone()
|
||||
conn.close()
|
||||
return result
|
||||
except Exception:
|
||||
conn.close()
|
||||
return None
|
||||
|
||||
def get_filtered_motions(
|
||||
self,
|
||||
policy_area: str = "Alle",
|
||||
min_margin: float = 0.0,
|
||||
max_margin: float = 1.0,
|
||||
limit: int = 10
|
||||
) -> List[Dict]:
|
||||
conn = duckdb.connect(self.db_path)
|
||||
try:
|
||||
query = """
|
||||
SELECT * FROM motions
|
||||
WHERE (? = 'Alle' OR policy_area = ?)
|
||||
AND winning_margin BETWEEN ? AND ?
|
||||
ORDER BY RANDOM()
|
||||
LIMIT ?
|
||||
"""
|
||||
rows = conn.execute(query, (policy_area, policy_area, min_margin, max_margin, limit)).fetchall()
|
||||
conn.close()
|
||||
return rows
|
||||
except Exception:
|
||||
conn.close()
|
||||
return []
|
||||
```
|
||||
|
||||
### Pattern 2: With Statement (Cleaner)
|
||||
|
||||
```python
|
||||
def execute_query(self, query: str, params: tuple = ()):
|
||||
with duckdb.connect(self.db_path) as conn:
|
||||
return conn.execute(query, params).fetchall()
|
||||
```
|
||||
|
||||
### Pattern 3: Lazy Connection Caching
|
||||
|
||||
For frequently accessed connections:
|
||||
|
||||
```python
|
||||
class MotionDatabase:
|
||||
def __init__(self, db_path: str = config.DATABASE_PATH):
|
||||
self.db_path = db_path
|
||||
self._conn = None
|
||||
|
||||
@property
|
||||
def connection(self):
|
||||
if self._conn is None:
|
||||
self._conn = duckdb.connect(self.db_path)
|
||||
return self._conn
|
||||
|
||||
def close(self):
|
||||
if self._conn:
|
||||
self._conn.close()
|
||||
self._conn = None
|
||||
```
|
||||
|
||||
## Table Initialization
|
||||
|
||||
Create tables with proper constraints and sequences:
|
||||
|
||||
```python
|
||||
def _init_database(self):
|
||||
conn = duckdb.connect(self.db_path)
|
||||
|
||||
# Create sequence for auto-incrementing IDs
|
||||
try:
|
||||
conn.execute("CREATE SEQUENCE IF NOT EXISTS motions_id_seq START 1")
|
||||
except:
|
||||
pass
|
||||
|
||||
# Create tables
|
||||
conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS motions (
|
||||
id INTEGER DEFAULT nextval('motions_id_seq'),
|
||||
title TEXT NOT NULL,
|
||||
description TEXT,
|
||||
date DATE,
|
||||
policy_area TEXT,
|
||||
voting_results JSON,
|
||||
winning_margin FLOAT,
|
||||
controversy_score FLOAT,
|
||||
layman_explanation TEXT,
|
||||
externe_identifier TEXT,
|
||||
body_text TEXT,
|
||||
url TEXT UNIQUE,
|
||||
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||
PRIMARY KEY (id)
|
||||
)
|
||||
""")
|
||||
|
||||
# Add columns to existing tables safely
|
||||
try:
|
||||
conn.execute("ALTER TABLE motions ADD COLUMN IF NOT EXISTS body_text TEXT")
|
||||
except Exception:
|
||||
pass # Column may already exist
|
||||
|
||||
conn.close()
|
||||
```
|
||||
|
||||
## JSON Column Handling
|
||||
|
||||
Store and retrieve JSON data:
|
||||
|
||||
```python
|
||||
# Insert JSON
|
||||
def store_motion(self, motion: Dict):
|
||||
conn = duckdb.connect(self.db_path)
|
||||
try:
|
||||
conn.execute(
|
||||
"INSERT INTO motions (title, voting_results) VALUES (?, ?)",
|
||||
(motion["title"], json.dumps(motion["voting_results"]))
|
||||
)
|
||||
conn.close()
|
||||
except Exception:
|
||||
conn.close()
|
||||
|
||||
# Query JSON
|
||||
def get_motions_with_votes(self, party: str) -> List[Dict]:
|
||||
conn = duckdb.connect(self.db_path)
|
||||
try:
|
||||
rows = conn.execute("""
|
||||
SELECT title, voting_results
|
||||
FROM motions
|
||||
WHERE JSON_EXTRACT(voting_results, '$.party') = ?
|
||||
""", (party,)).fetchall()
|
||||
conn.close()
|
||||
return rows
|
||||
except Exception:
|
||||
conn.close()
|
||||
return []
|
||||
```
|
||||
|
||||
## Query Patterns
|
||||
|
||||
### Parameterized Queries (Always!)
|
||||
```python
|
||||
# SAFE - uses parameterized query
|
||||
conn.execute("SELECT * FROM motions WHERE id = ?", (motion_id,))
|
||||
|
||||
# AVOID - SQL injection risk
|
||||
# conn.execute(f"SELECT * FROM motions WHERE id = {motion_id}") # BAD!
|
||||
```
|
||||
|
||||
### Batch Inserts
|
||||
```python
|
||||
def bulk_insert_motions(self, motions: List[Dict]):
|
||||
conn = duckdb.connect(self.db_path)
|
||||
try:
|
||||
for motion in motions:
|
||||
conn.execute(
|
||||
"""INSERT OR IGNORE INTO motions
|
||||
(title, date, policy_area) VALUES (?, ?, ?)""",
|
||||
(motion["title"], motion["date"], motion["policy_area"])
|
||||
)
|
||||
conn.close()
|
||||
except Exception:
|
||||
conn.close()
|
||||
```
|
||||
|
||||
### Aggregation Queries
|
||||
```python
|
||||
def get_party_vote_stats(self, party: str) -> Dict:
|
||||
conn = duckdb.connect(self.db_path)
|
||||
try:
|
||||
result = conn.execute("""
|
||||
SELECT
|
||||
COUNT(*) as total_votes,
|
||||
SUM(CASE WHEN vote = 'Voor' THEN 1 ELSE 0 END) as voor,
|
||||
SUM(CASE WHEN vote = 'Tegen' THEN 1 ELSE 0 END) as tegen
|
||||
FROM mp_votes
|
||||
WHERE party = ?
|
||||
""", (party,)).fetchone()
|
||||
conn.close()
|
||||
return {"total": result[0], "voor": result[1], "tegen": result[2]}
|
||||
except Exception:
|
||||
conn.close()
|
||||
return {"total": 0, "voor": 0, "tegen": 0}
|
||||
```
|
||||
|
||||
## Error Handling
|
||||
|
||||
Always close connections in finally block or with context manager:
|
||||
|
||||
```python
|
||||
def safe_query(self, query: str, params: tuple = ()):
|
||||
conn = None
|
||||
try:
|
||||
conn = duckdb.connect(self.db_path)
|
||||
result = conn.execute(query, params).fetchall()
|
||||
return result
|
||||
except Exception as e:
|
||||
_logger.error(f"Query failed: {e}")
|
||||
return []
|
||||
finally:
|
||||
if conn:
|
||||
conn.close()
|
||||
```
|
||||
|
||||
## Testing with Mock
|
||||
|
||||
For unit tests without DuckDB:
|
||||
|
||||
```python
|
||||
# In MotionDatabase.__init__
|
||||
def __init__(self, db_path: str = config.DATABASE_PATH):
|
||||
self.db_path = db_path
|
||||
self._file_mode = duckdb is None
|
||||
|
||||
if duckdb is None:
|
||||
# Create JSON fallback files
|
||||
for p in (f"{db_path}.embeddings.json", f"{db_path}.similarity_cache.json"):
|
||||
if not os.path.exists(p):
|
||||
with open(p, "w") as fh:
|
||||
fh.write("[]")
|
||||
else:
|
||||
self._init_database()
|
||||
```
|
||||
@@ -0,0 +1,228 @@
|
||||
# Code Patterns
|
||||
|
||||
## 1. Page Wrapper Pattern
|
||||
Thin Streamlit page files delegate to core modules. Pages contain only route logic, not business logic.
|
||||
|
||||
**Example** (pages/1_🗳️_Stemwijzer.py):
|
||||
```python
|
||||
import streamlit as st
|
||||
from quiz_module import render_quiz_page
|
||||
|
||||
st.set_page_config(...)
|
||||
render_quiz_page()
|
||||
```
|
||||
|
||||
**Example** (pages/2_🔍_Explorer.py):
|
||||
```python
|
||||
import streamlit as st
|
||||
from explorer import render_explorer
|
||||
|
||||
st.set_page_config(...)
|
||||
render_explorer()
|
||||
```
|
||||
|
||||
**Rule**: Pages should have <20 lines of logic. All complexity lives in modules.
|
||||
|
||||
---
|
||||
|
||||
## 2. Pipeline Pattern
|
||||
Data flows: fetch → transform → store
|
||||
|
||||
**Location**: `pipeline/` directory
|
||||
|
||||
**Pattern**:
|
||||
```python
|
||||
def run_pipeline():
|
||||
raw_data = fetch_from_source()
|
||||
transformed = transform(raw_data)
|
||||
store(transformed)
|
||||
|
||||
def fetch_from_source():
|
||||
# API call or DB query
|
||||
...
|
||||
|
||||
def transform(raw):
|
||||
# Clean, normalize, compute derived fields
|
||||
...
|
||||
```
|
||||
|
||||
**Usage**: SVD computation pipeline, data ingestion, motion processing
|
||||
|
||||
---
|
||||
|
||||
## 3. API Client Pattern
|
||||
HTTP client with retry/backoff for external data sources.
|
||||
|
||||
**Pattern**:
|
||||
```python
|
||||
import time
|
||||
import requests
|
||||
|
||||
def fetch_with_retry(url, max_retries=3):
|
||||
for attempt in range(max_retries):
|
||||
try:
|
||||
response = requests.get(url)
|
||||
response.raise_for_status()
|
||||
return response.json()
|
||||
except requests.RequestException:
|
||||
if attempt < max_retries - 1:
|
||||
time.sleep(2 ** attempt) # exponential backoff
|
||||
else:
|
||||
raise
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 4. Pure Helper Functions
|
||||
Functions in `explorer_helpers.py` have no side effects, no IO.
|
||||
|
||||
**Pattern**:
|
||||
```python
|
||||
def compute_party_coords(svd_df, party_map, window):
|
||||
"""Pure function: same inputs → same outputs, no side effects."""
|
||||
# Filter, compute, return
|
||||
return result_df
|
||||
|
||||
def build_scatter_trace(df, color_col, marker_size=8):
|
||||
"""Pure: returns Plotly trace dict, no rendering."""
|
||||
trace = go.Scatter(x=df.x, y=df.y, mode='markers', ...)
|
||||
return trace
|
||||
```
|
||||
|
||||
**Rule**: No `import streamlit` in helper modules. No file I/O. No global state.
|
||||
|
||||
---
|
||||
|
||||
## 5. Dummy Fallbacks for Optional Dependencies
|
||||
Gracefully degrade when optional packages are unavailable.
|
||||
|
||||
**Pattern**:
|
||||
```python
|
||||
try:
|
||||
import umap
|
||||
HAS_UMAP = True
|
||||
except ImportError:
|
||||
HAS_UMAP = False
|
||||
# or provide dummy stub
|
||||
|
||||
def project_to_2d(vectors):
|
||||
if HAS_UMAP:
|
||||
return umap.UMAP().fit_transform(vectors)
|
||||
else:
|
||||
return vectors[:, :2] # fallback: just take first 2 dims
|
||||
```
|
||||
|
||||
**Used for**: UMAP, Plotly (with fallback to altair or text-only)
|
||||
|
||||
---
|
||||
|
||||
## 6. Cached Data Loaders
|
||||
Expensive DB queries wrapped with `@st.cache_data`.
|
||||
|
||||
**Pattern**:
|
||||
```python
|
||||
@st.cache_data
|
||||
def load_svd_vectors(window: str) -> pd.DataFrame:
|
||||
return db.query("SELECT * FROM svd_vectors WHERE window = ?", window)
|
||||
|
||||
@st.cache_data
|
||||
def load_party_centroids(window: str) -> pd.DataFrame:
|
||||
return db.query("SELECT * FROM party_centroids WHERE window = ?", window)
|
||||
|
||||
# Clear cache when data updates
|
||||
@st.cache_data
|
||||
def load_motions(category: str | None = None) -> pd.DataFrame:
|
||||
...
|
||||
```
|
||||
|
||||
**Rule**: Use `ttl=3600` for large datasets. Use `show_spinner=False` where appropriate.
|
||||
|
||||
---
|
||||
|
||||
## 7. Plotly Dual-Layer Charts
|
||||
Charts built with two traces: scatter points + text annotations.
|
||||
|
||||
**Pattern**:
|
||||
```python
|
||||
def build_dual_layer_chart(df, x_col, y_col, label_col):
|
||||
# Layer 1: markers
|
||||
scatter = go.Scatter(
|
||||
x=df[x_col], y=df[y_col],
|
||||
mode='markers',
|
||||
marker=dict(size=10, color=df['color']),
|
||||
name='Parties'
|
||||
)
|
||||
# Layer 2: labels (smaller, non-hoverable)
|
||||
labels = go.Scatter(
|
||||
x=df[x_col], y=df[y_col],
|
||||
mode='text',
|
||||
text=df[label_col],
|
||||
textposition='top center',
|
||||
showlegend=False
|
||||
)
|
||||
return [scatter, labels]
|
||||
```
|
||||
|
||||
**Used in**: Explorer tab charts, party position plots
|
||||
|
||||
---
|
||||
|
||||
## 8. Singleton Module Instances
|
||||
One shared instance per module, created at import time.
|
||||
|
||||
**Pattern**:
|
||||
```python
|
||||
# database.py
|
||||
class MotionDatabase:
|
||||
def __init__(self, db_path=None):
|
||||
self.conn = ibis.duckdb.connect(db_path)
|
||||
self._load_schema()
|
||||
|
||||
_db = None
|
||||
def get_db():
|
||||
global _db
|
||||
if _db is None:
|
||||
_db = MotionDatabase()
|
||||
return _db
|
||||
|
||||
# At module bottom:
|
||||
db = MotionDatabase() # singleton instance
|
||||
```
|
||||
|
||||
**Also used in**: `config.py` exports `config` and `PARTY_COLOURS`
|
||||
|
||||
---
|
||||
|
||||
## 9. Dataclass Config Pattern
|
||||
Configuration centralized in a `@dataclass`.
|
||||
|
||||
**Pattern**:
|
||||
```python
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
@dataclass
|
||||
class Config:
|
||||
db_path: str = "data/stemwijzer.duckdb"
|
||||
default_window: str = "2023"
|
||||
cache_ttl: int = 3600
|
||||
party_colours: dict = field(default_factory=lambda: PARTY_COLOURS)
|
||||
|
||||
def __post_init__(self):
|
||||
if not Path(self.db_path).exists():
|
||||
raise FileNotFoundError(f"Database not found: {self.db_path}")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 10. Graceful Degradation with try/except
|
||||
Core pattern throughout: attempt operation, fall back gracefully.
|
||||
|
||||
**Pattern**:
|
||||
```python
|
||||
def get_political_position(mp_name, window):
|
||||
try:
|
||||
vectors = load_svd_vectors(window)
|
||||
return vectors[vectors['mp_name'] == mp_name]['vector_2d'].iloc[0]
|
||||
except (KeyError, IndexError):
|
||||
return [0.0, 0.0] # neutral fallback
|
||||
```
|
||||
@@ -0,0 +1,196 @@
|
||||
# Python-Specific Patterns
|
||||
|
||||
## Singleton Pattern
|
||||
|
||||
Use module-level instances for shared resources:
|
||||
|
||||
```python
|
||||
# database.py
|
||||
class MotionDatabase:
|
||||
def __init__(self, db_path: str = config.DATABASE_PATH):
|
||||
self.db_path = db_path
|
||||
self._init_database()
|
||||
|
||||
def _init_database(self):
|
||||
# Initialize tables on first instantiation
|
||||
...
|
||||
|
||||
# Bottom of file - the singleton
|
||||
db = MotionDatabase()
|
||||
```
|
||||
|
||||
**Usage across the codebase:**
|
||||
```python
|
||||
# In other modules
|
||||
from database import db
|
||||
|
||||
def some_function():
|
||||
motions = db.get_filtered_motions(limit=10)
|
||||
return motions
|
||||
```
|
||||
|
||||
Similarly for other singletons:
|
||||
```python
|
||||
# summarizer.py
|
||||
class MotionSummarizer:
|
||||
def __init__(self):
|
||||
pass # Stateless
|
||||
|
||||
def generate_layman_explanation(self, title: str, body: str) -> str:
|
||||
...
|
||||
|
||||
summarizer = MotionSummarizer()
|
||||
```
|
||||
|
||||
## Dataclass Config Pattern
|
||||
|
||||
Use dataclass for configuration with environment variable support:
|
||||
|
||||
```python
|
||||
# config.py
|
||||
from dataclasses import dataclass
|
||||
from typing import List
|
||||
import os
|
||||
|
||||
@dataclass
|
||||
class Config:
|
||||
# Database settings
|
||||
DATABASE_PATH = "data/motions.db"
|
||||
|
||||
# API settings
|
||||
TWEEDE_KAMER_ODATA_API = "https://gegevensmagazijn.tweedekamer.nl/OData/v4/2.0"
|
||||
API_TIMEOUT = 30
|
||||
API_BATCH_SIZE = 250
|
||||
|
||||
# AI settings
|
||||
OPENROUTER_API_KEY = os.getenv("OPENROUTER_API_KEY")
|
||||
OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1"
|
||||
QWEN_MODEL = "qwen/qwen-2.5-72b-instruct"
|
||||
|
||||
# App settings
|
||||
DEFAULT_MOTION_COUNT = 10
|
||||
SESSION_TIMEOUT_DAYS = 30
|
||||
|
||||
# Policy areas
|
||||
POLICY_AREAS: List[str] = None
|
||||
def __post_init__(self):
|
||||
self.POLICY_AREAS = [
|
||||
"Alle", "Economie", "Klimaat", "Immigratie",
|
||||
"Zorg", "Onderwijs", "Defensie", "Sociale Zaken", "Algemeen"
|
||||
]
|
||||
|
||||
config = Config()
|
||||
```
|
||||
|
||||
**Usage:**
|
||||
```python
|
||||
from config import config
|
||||
|
||||
# Access as attributes
|
||||
timeout = config.API_TIMEOUT
|
||||
areas = config.POLICY_AREAS
|
||||
```
|
||||
|
||||
## DuckDB Connection Pattern
|
||||
|
||||
Short-lived connections with explicit cleanup:
|
||||
|
||||
```python
|
||||
class MotionDatabase:
|
||||
def get_motion(self, motion_id: int) -> Optional[Dict]:
|
||||
conn = duckdb.connect(self.db_path)
|
||||
try:
|
||||
result = conn.execute(
|
||||
"SELECT * FROM motions WHERE id = ?",
|
||||
(motion_id,)
|
||||
).fetchone()
|
||||
return result
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
def get_filtered_motions(self, **kwargs) -> List[Dict]:
|
||||
conn = duckdb.connect(self.db_path)
|
||||
try:
|
||||
rows = conn.execute(query, params).fetchall()
|
||||
return rows
|
||||
except Exception:
|
||||
return [] # Safe fallback
|
||||
finally:
|
||||
conn.close()
|
||||
```
|
||||
|
||||
**Context manager alternative (preferred when applicable):**
|
||||
```python
|
||||
def some_operation(self):
|
||||
with duckdb.connect(self.db_path) as conn:
|
||||
result = conn.execute("SELECT ...").fetchall()
|
||||
return result
|
||||
```
|
||||
|
||||
## Try/Except with Fallback Pattern
|
||||
|
||||
Always provide safe fallbacks:
|
||||
|
||||
```python
|
||||
def get_motion_or_default(self, motion_id: int) -> Dict:
|
||||
try:
|
||||
conn = duckdb.connect(self.db_path)
|
||||
result = conn.execute("SELECT * FROM motions WHERE id = ?", (motion_id,)).fetchone()
|
||||
conn.close()
|
||||
return result if result else {}
|
||||
except Exception:
|
||||
return {}
|
||||
```
|
||||
|
||||
## Optional Import Pattern
|
||||
|
||||
Handle optional dependencies gracefully:
|
||||
|
||||
```python
|
||||
try:
|
||||
import duckdb
|
||||
except Exception: # pragma: no cover
|
||||
duckdb = None
|
||||
|
||||
class MotionDatabase:
|
||||
def __init__(self, db_path: str = config.DATABASE_PATH):
|
||||
self._file_mode = duckdb is None
|
||||
...
|
||||
```
|
||||
|
||||
## Property Pattern
|
||||
|
||||
Lazy initialization of expensive resources:
|
||||
|
||||
```python
|
||||
class MotionDatabase:
|
||||
def __init__(self, db_path: str = config.DATABASE_PATH):
|
||||
self.db_path = db_path
|
||||
self._session_cache = None
|
||||
|
||||
@property
|
||||
def session(self):
|
||||
"""Lazy-load expensive resources."""
|
||||
if self._session_cache is None:
|
||||
self._session_cache = self._create_session()
|
||||
return self._session_cache
|
||||
```
|
||||
|
||||
## Type Annotation Patterns
|
||||
|
||||
```python
|
||||
from typing import Dict, List, Optional, Tuple, Any
|
||||
|
||||
# Optional with None default
|
||||
def get_motion(self, motion_id: Optional[int] = None) -> Optional[Dict]:
|
||||
...
|
||||
|
||||
# Multiple return types
|
||||
def parse_vote(self, vote_str: str) -> Tuple[bool, str]:
|
||||
"""Returns (success, error_message)"""
|
||||
...
|
||||
|
||||
# Generic types
|
||||
def get_batch(self, ids: List[int]) -> Dict[str, Any]:
|
||||
...
|
||||
```
|
||||
@@ -0,0 +1,225 @@
|
||||
# Streamlit Patterns
|
||||
|
||||
## Session State Initialization
|
||||
|
||||
Always initialize session state at the start of the main function:
|
||||
|
||||
```python
|
||||
# app.py
|
||||
import streamlit as st
|
||||
|
||||
def main():
|
||||
# Initialize all session state variables
|
||||
if "session_id" not in st.session_state:
|
||||
st.session_state.session_id = None
|
||||
if "current_motion_index" not in st.session_state:
|
||||
st.session_state.current_motion_index = 0
|
||||
if "motions" not in st.session_state:
|
||||
st.session_state.motions = []
|
||||
if "show_results" not in st.session_state:
|
||||
st.session_state.show_results = False
|
||||
|
||||
# Rest of app...
|
||||
```
|
||||
|
||||
## Page Configuration
|
||||
|
||||
Set page config at the top of each page file:
|
||||
|
||||
```python
|
||||
# pages/1_Stemwijzer.py
|
||||
import streamlit as st
|
||||
|
||||
st.set_page_config(
|
||||
page_title="Stemwijzer",
|
||||
page_icon="🗳️",
|
||||
layout="centered",
|
||||
)
|
||||
|
||||
from explorer import build_mp_quiz_tab
|
||||
build_mp_quiz_tab("data/motions.db")
|
||||
```
|
||||
|
||||
## Thin Page Wrapper Pattern
|
||||
|
||||
Pages delegate to shared functions in main modules:
|
||||
|
||||
```python
|
||||
# pages/2_Explorer.py
|
||||
import streamlit as st
|
||||
|
||||
st.set_page_config(
|
||||
page_title="Explorer",
|
||||
page_icon="🔭",
|
||||
layout="wide",
|
||||
)
|
||||
|
||||
from explorer import build_explorer_tab
|
||||
build_explorer_tab()
|
||||
```
|
||||
|
||||
```python
|
||||
# explorer.py
|
||||
def build_explorer_tab():
|
||||
st.header("🔭 Politiek Explorer")
|
||||
|
||||
tab1, tab2, tab3 = st.tabs([
|
||||
"Compass",
|
||||
"Trajectories",
|
||||
"Zoeken"
|
||||
])
|
||||
|
||||
with tab1:
|
||||
render_compass()
|
||||
with tab2:
|
||||
render_trajectories()
|
||||
with tab3:
|
||||
render_search()
|
||||
```
|
||||
|
||||
## Sidebar Pattern
|
||||
|
||||
Use sidebar for configuration and navigation:
|
||||
|
||||
```python
|
||||
# app.py
|
||||
def main():
|
||||
with st.sidebar:
|
||||
st.header("Instellingen")
|
||||
|
||||
motion_count = st.slider(
|
||||
"Aantal moties",
|
||||
min_value=5,
|
||||
max_value=25,
|
||||
value=10,
|
||||
)
|
||||
|
||||
policy_area = st.selectbox("Beleidsgebied", config.POLICY_AREAS)
|
||||
|
||||
if st.button("Start Nieuwe Sessie"):
|
||||
start_new_session(motion_count, policy_area)
|
||||
```
|
||||
|
||||
## Callback Pattern for State Updates
|
||||
|
||||
Use callbacks to handle user interactions:
|
||||
|
||||
```python
|
||||
def on_motion_vote(motion_id: int, vote: str):
|
||||
"""Callback when user votes on a motion."""
|
||||
st.session_state.user_votes[motion_id] = vote
|
||||
|
||||
# Move to next motion
|
||||
if st.session_state.current_motion_index < len(st.session_state.motions) - 1:
|
||||
st.session_state.current_motion_index += 1
|
||||
else:
|
||||
st.session_state.show_results = True
|
||||
|
||||
st.rerun()
|
||||
|
||||
# In UI
|
||||
col1, col2, col3 = st.columns(3)
|
||||
with col1:
|
||||
st.button("👍 Voor", on_click=on_motion_vote, args=(motion_id, "Voor"))
|
||||
with col2:
|
||||
st.button("👎 Tegen", on_click=on_motion_vote, args=(motion_id, "Tegen"))
|
||||
with col3:
|
||||
st.button("❓ Onthouden", on_click=on_motion_vote, args=(motion_id, "Onthouden"))
|
||||
```
|
||||
|
||||
## Container Pattern for Dynamic Content
|
||||
|
||||
Use containers for dynamic rendering:
|
||||
|
||||
```python
|
||||
def show_motion_interface():
|
||||
if not st.session_state.motions:
|
||||
st.warning("Geen moties geladen")
|
||||
return
|
||||
|
||||
current_idx = st.session_state.current_motion_index
|
||||
motion = st.session_state.motions[current_idx]
|
||||
|
||||
with st.container():
|
||||
st.subheader(f"Motie {current_idx + 1} van {len(st.session_state.motions)}")
|
||||
st.markdown(f"**{motion['title']}**")
|
||||
st.caption(f"📅 {motion['date']} | 🏷️ {motion['policy_area']}")
|
||||
|
||||
if motion.get("layman_explanation"):
|
||||
st.info(motion["layman_explanation"])
|
||||
|
||||
# Voting buttons...
|
||||
```
|
||||
|
||||
## Expander Pattern for Details
|
||||
|
||||
Use expanders for collapsible content:
|
||||
|
||||
```python
|
||||
with st.expander("Meer details"):
|
||||
st.markdown(f"**Beschrijving:** {motion.get('description', 'N/A')}")
|
||||
|
||||
if motion.get("voting_results"):
|
||||
results = json.loads(motion["voting_results"])
|
||||
st.json(results)
|
||||
```
|
||||
|
||||
## Form Pattern for Batch Updates
|
||||
|
||||
Use forms for multiple related inputs:
|
||||
|
||||
```python
|
||||
with st.form("session_settings"):
|
||||
st.subheader("Sessie Instellingen")
|
||||
|
||||
col1, col2 = st.columns(2)
|
||||
with col1:
|
||||
count = st.number_input("Aantal moties", min_value=5, max_value=25)
|
||||
with col2:
|
||||
area = st.selectbox("Beleidsgebied", config.POLICY_AREAS)
|
||||
|
||||
submitted = st.form_submit_button("Start Sessie")
|
||||
if submitted:
|
||||
start_session(count, area)
|
||||
```
|
||||
|
||||
## Caching Pattern
|
||||
|
||||
Cache expensive computations:
|
||||
|
||||
```python
|
||||
@st.cache_data(ttl=3600) # Cache for 1 hour
|
||||
def load_party_positions(window_id: str) -> Dict:
|
||||
"""Load party positions from database."""
|
||||
return db.get_party_positions(window_id)
|
||||
|
||||
@st.cache_resource
|
||||
def init_database():
|
||||
"""Initialize database connection."""
|
||||
return MotionDatabase(config.DATABASE_PATH)
|
||||
```
|
||||
|
||||
## Home Page Pattern
|
||||
|
||||
Landing page with navigation:
|
||||
|
||||
```python
|
||||
# Home.py
|
||||
import streamlit as st
|
||||
|
||||
st.set_page_config(
|
||||
page_title="Motief: de stematlas",
|
||||
page_icon="🗺️",
|
||||
layout="centered",
|
||||
)
|
||||
|
||||
def main():
|
||||
st.title("🗺️ Motief: de stematlas")
|
||||
st.markdown("**Motief** brengt de Nederlandse Tweede Kamer in kaart...")
|
||||
|
||||
col1, col2 = st.columns(2)
|
||||
with col1:
|
||||
st.page_link("pages/1_Stemwijzer.py", label="Open Stemwijzer", icon="🗳️")
|
||||
with col2:
|
||||
st.page_link("pages/2_Explorer.py", label="Open Explorer", icon="🔭")
|
||||
```
|
||||
Reference in New Issue
Block a user