Search, chat and research over parliamentary speeches and documents
You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
 
 
 
 
 

281 lines
11 KiB

"""Adapter for the Swedish Riksdag's open data (data.riksdagen.se).
This module is the *only* place that knows what the Riksdag calls its fields.
Everything downstream works in plenum's own column names. Copy this file as the
starting point for another parliament — see docs/PORTING.md.
Two things about this source that will probably bite you elsewhere too:
* Its JSON is converted from XML, so a repeated element arrives as an object when
there is one of them and as a list when there are several. `_as_list` normalises
that; forgetting it produces code that works until the day a document has two
authors.
* Null is serialised as the *string* `"None"`, which is truthy and will happily be
written to the database as text unless it is stripped.
"""
from __future__ import annotations
import re
from datetime import date, datetime
from typing import Any, Iterable, Iterator, Optional
# ── source quirks ─────────────────────────────────────────────────────────────
def _as_list(value: Any) -> list[dict]:
"""One child arrives as an object, several as a list, none as null.
Non-object entries are dropped rather than raising: a handful of records in
the archive carry a bare string where a child element is expected, and one
malformed row should not abort a multi-hour load.
"""
if value is None:
return []
items = value if isinstance(value, list) else [value]
return [i for i in items if isinstance(i, dict)]
def _clean(value: Any) -> Any:
"""The API serialises null as the string "None"."""
if value is None or value == "None":
return None
return value
def _parse_date(raw: Optional[str]) -> Optional[date]:
"""Dates arrive as "YYYY-MM-DD HH:MM:SS", sometimes without the time."""
if not raw:
return None
for fmt in ("%Y-%m-%d %H:%M:%S", "%Y-%m-%dT%H:%M:%S", "%Y-%m-%d"):
try:
return datetime.strptime(raw.strip()[: len(fmt) + 2], fmt).date()
except ValueError:
continue
return None
def _session_start_year(session_label: Optional[str]) -> Optional[int]:
""""2022/23" -> 2022."""
if not session_label:
return None
try:
return int(str(session_label)[:4])
except ValueError:
return None
_TITLE_PREFIXES = re.compile(
r"^(?:statsrådet|ministern|talman(?:nen)?|herr|fru)\s+", re.IGNORECASE
)
# The transcript appends the party in parentheses: "Mikael Dahlqvist (S)". The
# party is stored in its own column, and leaving it here would break joins to the
# member register and show up duplicated in the UI.
_TRAILING_PARTY = re.compile(r"\s*\([A-ZÅÄÖ\-]{1,4}\)\s*$")
def _clean_speaker(name: Optional[str]) -> Optional[str]:
"""Strip honorifics and the trailing party, so names join to the register."""
if not name:
return None
name = _TITLE_PREFIXES.sub("", name.strip())
return _TRAILING_PARTY.sub("", name).strip() or None
# ── speeches ──────────────────────────────────────────────────────────────────
# Left: plenum column. Right: the Riksdag's field name. Written this way round
# because the column names are the contract and the source names are the detail.
SPEECH_FIELDS: dict[str, str] = {
"source_speech_id": "anforande_id",
"text": "anforandetext",
"section_title": "avsnittsrubrik",
"sequence": "anforande_nummer",
"activity_type": "kammaraktivitet",
"speaker_name": "talare",
"party": "parti",
"person_id": "intressent_id",
"source_datetime": "dok_datum",
"source_doc_id": "dok_id",
"related_doc_id": "rel_dok_id",
"source_doc_number": "dok_nummer",
"source_record_id": "dok_hangar_id",
"title": "dok_titel",
}
def adapt_speech(payload: dict) -> Optional[dict]:
"""Turn one source record into a `speeches` row, or None if unusable."""
doc = payload.get("anforande") if "anforande" in payload else payload
if not doc:
return None
row = {col: _clean(doc.get(src)) for col, src in SPEECH_FIELDS.items()}
# The primary key is the protocol document plus the position within it. The
# source's own `anforande_id` is a UUID that changed in an earlier migration,
# so it is kept for reference but is not the key.
if not row["source_doc_id"] or row["sequence"] in (None, ""):
return None
row["id"] = f"{row['source_doc_id']}-{row['sequence']}"
row["speaker_name"] = _clean_speaker(row["speaker_name"])
# Speech text arrives as HTML fragments; the corpus stores plain text so that
# full-text search and snippet extraction do not have to strip tags per query.
row["text"] = _html_to_text(row["text"]) if row["text"] else None
row["date"] = _parse_date(row["source_datetime"])
row["year"] = _session_start_year(doc.get("dok_rm"))
row["session_year"] = row["year"]
# "Y"/"N", not a boolean.
row["is_reply"] = _clean(doc.get("replik")) == "Y"
return row
# ── documents ─────────────────────────────────────────────────────────────────
DOCUMENT_FIELDS: dict[str, str] = {
"doc_id": "dok_id",
"source_record_id": "hangar_id",
"session_label": "rm",
"designation": "beteckning",
"subtype": "subtyp",
"committee": "organ",
"status": "status",
"source_updated_at": "systemdatum",
"published_at": "publicerad",
"title": "titel",
"subtitle": "undertitel",
"url_text": "dokument_url_text",
"url_html": "dokument_url_html",
}
def adapt_document(payload: dict) -> Optional[dict]:
"""Turn one `dokumentstatus` record into a `documents` row plus its children.
Returns a dict with keys `document`, `authors` and `proposals`; the caller
decides how to persist them.
"""
status = payload.get("dokumentstatus") or payload
doc = status.get("dokument")
if not doc or not doc.get("dok_id"):
return None
row = {col: _clean(doc.get(src)) for col, src in DOCUMENT_FIELDS.items()}
row["doc_type"] = "motion"
row["date"] = _parse_date(doc.get("datum"))
row["session_year"] = _session_start_year(row["session_label"])
authors = []
for i, person in enumerate(_as_list((status.get("dokintressent") or {}).get("intressent"))):
authors.append({
"doc_id": row["doc_id"],
"ordinal": i,
"person_id": _clean(person.get("intressent_id")),
"name": _clean(person.get("namn")),
# The archive carries both "S" and "s" for the same party. Left as-is,
# a filter on 'S' silently misses half the documents.
"party": (_clean(person.get("partibet")) or "").upper() or None,
"role": _clean(person.get("roll")),
})
proposals = []
for i, item in enumerate(_as_list((status.get("dokforslag") or {}).get("forslag"))):
proposals.append({
"id": f"{row['doc_id']}:{i}",
"doc_id": row["doc_id"],
"ordinal": i,
"number": _clean(item.get("nummer")),
"text": _clean(item.get("lydelse")),
"committee_recommendation": _clean(item.get("utskottet")),
"chamber_decision": _clean(item.get("kammaren")),
"handled_in": _clean(item.get("behandlas_i")),
})
# Denormalised so a party filter does not need a join.
row["parties"] = sorted({a["party"] for a in authors if a["party"]})
row["author_names"] = [a["name"] for a in authors if a["name"]]
row["num_proposals"] = len(proposals)
row["proposals_text"] = "\n".join(p["text"] for p in proposals if p["text"]) or None
row["proposals_raw"] = (status.get("dokforslag") or {}).get("forslag")
row["attachments"] = (status.get("dokbilaga") or {}).get("bilaga")
# `html` is the full text; documents that exist only as scanned PDFs have none,
# which is why has_text is stored rather than inferred at query time.
html = _clean(doc.get("html"))
row["text"] = _html_to_text(html) if html else None
row["has_text"] = bool(row["text"])
row["url_pdf"] = _pdf_url(status)
return {"document": row, "authors": authors, "proposals": proposals}
def _html_to_text(html: str) -> Optional[str]:
from bs4 import BeautifulSoup
text = BeautifulSoup(html, "html.parser").get_text("\n")
return re.sub(r"\n{3,}", "\n\n", text).strip() or None
def _pdf_url(status: dict) -> Optional[str]:
for attachment in _as_list((status.get("dokbilaga") or {}).get("bilaga")):
url = _clean(attachment.get("fil_url"))
if url and url.lower().endswith(".pdf"):
return url
return None
# ── people ────────────────────────────────────────────────────────────────────
PERSON_FIELDS: dict[str, str] = {
"person_id": "intressent_id",
"source_record_id": "hangar_id",
"source_record_guid": "hangar_guid",
"source_id": "sourceid",
"birth_year": "fodd_ar",
"gender": "kon",
"last_name": "efternamn",
"first_name": "tilltalsnamn",
"sort_name": "sorteringsnamn",
"home_town": "iort",
"party": "parti",
"constituency": "valkrets",
"status": "status",
"source_url": "person_url_xml",
"image_url_small": "bild_url_80",
"image_url_medium": "bild_url_192",
"image_url_large": "bild_url_max",
}
def adapt_person(payload: dict) -> Optional[dict]:
if not payload.get("intressent_id"):
return None
row = {col: _clean(payload.get(src)) for col, src in PERSON_FIELDS.items()}
row["name"] = " ".join(x for x in (row["first_name"], row["last_name"]) if x) or None
row["assignments"] = payload.get("personuppdrag")
row["contact_details"] = payload.get("personuppgift")
row["active"] = _clean(payload.get("status")) not in (None, "Avgången")
return row
ADAPTERS = {
"speeches": adapt_speech,
"documents": adapt_document,
"people": adapt_person,
}
def adapt(kind: str, payload: dict) -> Optional[dict]:
"""Adapt one record of the named kind."""
try:
return ADAPTERS[kind](payload)
except KeyError:
raise ValueError(f"Unknown record kind {kind!r}; expected one of {sorted(ADAPTERS)}")
def adapt_many(kind: str, payloads: Iterable[dict]) -> Iterator[dict]:
"""Adapt a stream, skipping records the source could not supply usably."""
for payload in payloads:
row = adapt(kind, payload)
if row is not None:
yield row