1647 lines
68 KiB
Python
1647 lines
68 KiB
Python
from __future__ import annotations
|
|
|
|
import copy
|
|
import hashlib
|
|
import json
|
|
import re
|
|
import threading
|
|
from datetime import date, datetime, timezone
|
|
from typing import Any
|
|
|
|
from sqlalchemy import and_, func, or_, select
|
|
from sqlalchemy.orm import Session, aliased
|
|
|
|
from app.data_management import dataset_row_counts
|
|
from app.models import (
|
|
CanonicalStopLink,
|
|
Dataset,
|
|
GtfsCalendar,
|
|
GtfsCalendarDate,
|
|
GtfsHarmonizedSnapshot,
|
|
GtfsHarmonizedSnapshotDataset,
|
|
GtfsHarmonizedSnapshotRoute,
|
|
GtfsRoute,
|
|
GtfsStop,
|
|
GtfsStopTime,
|
|
GtfsTrip,
|
|
RouteMatch,
|
|
Source,
|
|
)
|
|
|
|
|
|
GTFS_QA_NOTE_PREFIX = "[GTFS QA]"
|
|
GTFS_LICENSE_FLAG_KEYS = (
|
|
"can_import",
|
|
"can_derive",
|
|
"can_redistribute",
|
|
"requires_attribution",
|
|
"commercial_restrictions",
|
|
)
|
|
GTFS_AUTHORITY_LEVELS = {
|
|
"unknown",
|
|
"national_official",
|
|
"regional_authority",
|
|
"operator",
|
|
"aggregator",
|
|
"mirror",
|
|
"secondary_discovery",
|
|
}
|
|
GTFS_AUTHORITY_SCORES = {
|
|
"national_official": 600,
|
|
"regional_authority": 500,
|
|
"operator": 420,
|
|
"unknown": 320,
|
|
"aggregator": 260,
|
|
"mirror": 120,
|
|
"secondary_discovery": 80,
|
|
}
|
|
GTFS_PRIORITY_SCORES = {
|
|
"P0": 60,
|
|
"P1": 40,
|
|
"P2": 20,
|
|
"P3": 10,
|
|
}
|
|
GTFS_SHADOW_ROUTE_RATIO = 0.65
|
|
GTFS_SHADOW_ROUTE_MIN_OVERLAP = 10
|
|
GTFS_SHADOW_SMALL_ROUTE_RATIO = 0.80
|
|
GTFS_SHADOW_SMALL_ROUTE_MIN_OVERLAP = 2
|
|
GTFS_ROUTE_SHADOW_BBOX_MARGIN_DEG = 0.02
|
|
_SNAPSHOT_CACHE_LOCK = threading.RLock()
|
|
_SNAPSHOT_CACHE_REVISION: tuple[object, ...] | None = None
|
|
_SNAPSHOT_CACHE: dict[str, Any] | None = None
|
|
_ROUTE_SHADOW_CACHE_LOCK = threading.RLock()
|
|
_ROUTE_SHADOW_CACHE_REVISION: tuple[object, ...] | None = None
|
|
_ROUTE_SHADOW_CACHE: set[int] | None = None
|
|
|
|
|
|
def active_harmonized_gtfs_dataset_ids(session: Session, source_ids: list[int] | None = None) -> list[int]:
|
|
if source_ids:
|
|
return _raw_active_gtfs_dataset_ids(session, source_ids=source_ids)
|
|
persisted_ids = _active_persisted_gtfs_dataset_ids(session)
|
|
if persisted_ids:
|
|
return persisted_ids
|
|
snapshot = computed_gtfs_harmonized_snapshot(session)
|
|
return [int(item["dataset_id"]) for item in snapshot["datasets"] if item["role"] == "included"]
|
|
|
|
|
|
def active_harmonized_gtfs_shadowed_route_ids(session: Session, source_ids: list[int] | None = None) -> set[int]:
|
|
if source_ids:
|
|
return set()
|
|
persisted = _active_persisted_gtfs_route_shadow_map(session)
|
|
if persisted is not None:
|
|
return set(persisted)
|
|
global _ROUTE_SHADOW_CACHE, _ROUTE_SHADOW_CACHE_REVISION
|
|
revision = _snapshot_revision(session)
|
|
with _ROUTE_SHADOW_CACHE_LOCK:
|
|
if _ROUTE_SHADOW_CACHE is not None and _ROUTE_SHADOW_CACHE_REVISION == revision:
|
|
return set(_ROUTE_SHADOW_CACHE)
|
|
items = _active_gtfs_snapshot_items(session)
|
|
_apply_dataset_shadowing(items)
|
|
result = _route_shadow_result(session, items)
|
|
shadowed_route_ids = set(result["shadowed_route_ids"])
|
|
with _ROUTE_SHADOW_CACHE_LOCK:
|
|
_ROUTE_SHADOW_CACHE_REVISION = revision
|
|
_ROUTE_SHADOW_CACHE = set(shadowed_route_ids)
|
|
return shadowed_route_ids
|
|
|
|
|
|
def active_harmonized_gtfs_route_shadow_map(session: Session, source_ids: list[int] | None = None) -> dict[int, int]:
|
|
if source_ids:
|
|
return {}
|
|
persisted = _active_persisted_gtfs_route_shadow_map(session)
|
|
if persisted is not None:
|
|
return persisted
|
|
return _computed_gtfs_route_shadow_map(session)
|
|
|
|
|
|
def _computed_gtfs_route_shadow_map(session: Session) -> dict[int, int]:
|
|
items = _active_gtfs_snapshot_items(session)
|
|
_apply_dataset_shadowing(items)
|
|
result = _route_shadow_result(session, items)
|
|
return dict(result["shadowed_route_map"])
|
|
|
|
|
|
def gtfs_harmonized_snapshot(session: Session) -> dict[str, Any]:
|
|
persisted = active_persisted_gtfs_snapshot(session)
|
|
if persisted is not None:
|
|
return persisted
|
|
return computed_gtfs_harmonized_snapshot(session)
|
|
|
|
|
|
def gtfs_harmonized_snapshot_diagnostics(session: Session) -> dict[str, Any]:
|
|
active = active_persisted_gtfs_snapshot(session)
|
|
computed = computed_gtfs_harmonized_snapshot(session)
|
|
computed_summary = computed.get("summary") or {}
|
|
active_summary = active.get("summary") if active is not None else {}
|
|
if not isinstance(active_summary, dict):
|
|
active_summary = {}
|
|
computed_key = _materialized_snapshot_key(computed)
|
|
active_key = str(active.get("snapshot_key") or "") if active is not None else ""
|
|
active_route_rows = _summary_int(active_summary, "snapshot_route_rows")
|
|
active_route_shadows = _summary_int(active_summary, "route_shadowed_routes")
|
|
computed_included_datasets = _summary_int(computed_summary, "included_datasets")
|
|
computed_route_shadows = _summary_int(computed_summary, "route_shadowed_routes")
|
|
reasons: list[str] = []
|
|
if active is None:
|
|
reasons.append("no_active_snapshot")
|
|
elif active_key != computed_key:
|
|
reasons.append("snapshot_inputs_changed")
|
|
if active is not None and computed_included_datasets > 0 and active_route_rows == 0:
|
|
reasons.append("route_ownership_not_materialized")
|
|
if active is not None and active_route_rows > 0 and active_route_shadows != computed_route_shadows:
|
|
reasons.append("route_shadow_count_changed")
|
|
return {
|
|
"needs_rebuild": bool(reasons),
|
|
"reasons": reasons,
|
|
"active_snapshot_id": active.get("id") if active is not None else None,
|
|
"active_snapshot_key": active_key or None,
|
|
"computed_snapshot_key": computed_key,
|
|
"route_rows_materialized": active_route_rows > 0,
|
|
"active": {
|
|
"persisted": active is not None,
|
|
"included_datasets": _summary_int(active_summary, "included_datasets"),
|
|
"shadowed_datasets": _summary_int(active_summary, "shadowed_datasets"),
|
|
"route_rows": active_route_rows,
|
|
"route_shadowed_routes": active_route_shadows,
|
|
},
|
|
"computed": {
|
|
"raw_active_datasets": _summary_int(computed_summary, "raw_active_datasets"),
|
|
"included_datasets": computed_included_datasets,
|
|
"shadowed_datasets": _summary_int(computed_summary, "shadowed_datasets"),
|
|
"route_shadowed_routes": computed_route_shadows,
|
|
},
|
|
}
|
|
|
|
|
|
def list_active_harmonized_snapshot_routes(
|
|
session: Session,
|
|
*,
|
|
role: str | None = None,
|
|
dataset_id: int | None = None,
|
|
source_id: int | None = None,
|
|
shadowed_by_dataset_id: int | None = None,
|
|
mode: str | None = None,
|
|
q: str | None = None,
|
|
limit: int = 200,
|
|
offset: int = 0,
|
|
max_limit: int = 1000,
|
|
) -> dict[str, Any]:
|
|
snapshot = session.scalar(
|
|
select(GtfsHarmonizedSnapshot)
|
|
.where(GtfsHarmonizedSnapshot.status == "active")
|
|
.order_by(GtfsHarmonizedSnapshot.activated_at.desc(), GtfsHarmonizedSnapshot.id.desc())
|
|
.limit(1)
|
|
)
|
|
if snapshot is None:
|
|
return {"snapshot": None, "summary": {}, "routes": []}
|
|
conditions = [GtfsHarmonizedSnapshotRoute.snapshot_id == snapshot.id]
|
|
if role:
|
|
conditions.append(GtfsHarmonizedSnapshotRoute.role == role)
|
|
if dataset_id is not None:
|
|
conditions.append(GtfsHarmonizedSnapshotRoute.dataset_id == int(dataset_id))
|
|
if source_id is not None:
|
|
conditions.append(GtfsHarmonizedSnapshotRoute.source_id == int(source_id))
|
|
if shadowed_by_dataset_id is not None:
|
|
conditions.append(GtfsHarmonizedSnapshotRoute.shadowed_by_dataset_id == int(shadowed_by_dataset_id))
|
|
mode_text = str(mode or "").strip()
|
|
if mode_text:
|
|
conditions.append(GtfsRoute.mode == mode_text)
|
|
query_text = " ".join(str(q or "").strip().split())
|
|
if query_text:
|
|
pattern = f"%{query_text}%"
|
|
conditions.append(
|
|
or_(
|
|
GtfsHarmonizedSnapshotRoute.route_id.ilike(pattern),
|
|
GtfsHarmonizedSnapshotRoute.overlap_key.ilike(pattern),
|
|
GtfsRoute.short_name.ilike(pattern),
|
|
GtfsRoute.long_name.ilike(pattern),
|
|
GtfsRoute.operator_name.ilike(pattern),
|
|
)
|
|
)
|
|
stmt = (
|
|
select(GtfsHarmonizedSnapshotRoute)
|
|
.join(GtfsRoute, GtfsRoute.id == GtfsHarmonizedSnapshotRoute.gtfs_route_id)
|
|
.where(*conditions)
|
|
)
|
|
count_stmt = (
|
|
select(func.count())
|
|
.select_from(GtfsHarmonizedSnapshotRoute)
|
|
.join(GtfsRoute, GtfsRoute.id == GtfsHarmonizedSnapshotRoute.gtfs_route_id)
|
|
.where(*conditions)
|
|
)
|
|
selected_limit = max(1, min(int(limit), max(1, int(max_limit))))
|
|
selected_offset = max(0, int(offset))
|
|
rows = session.scalars(
|
|
stmt.order_by(
|
|
GtfsHarmonizedSnapshotRoute.role,
|
|
GtfsHarmonizedSnapshotRoute.dataset_id,
|
|
GtfsHarmonizedSnapshotRoute.overlap_key,
|
|
GtfsHarmonizedSnapshotRoute.id,
|
|
)
|
|
.offset(selected_offset)
|
|
.limit(selected_limit)
|
|
).all()
|
|
return {
|
|
"snapshot": {
|
|
"id": snapshot.id,
|
|
"snapshot_key": snapshot.snapshot_key,
|
|
"status": snapshot.status,
|
|
"activated_at": _iso(snapshot.activated_at),
|
|
},
|
|
"summary": {
|
|
**_snapshot_route_row_counts(session, snapshot.id),
|
|
"filtered_total": int(session.scalar(count_stmt) or 0),
|
|
"limit": selected_limit,
|
|
"offset": selected_offset,
|
|
},
|
|
"routes": [_snapshot_route_payload(row) for row in rows],
|
|
}
|
|
|
|
|
|
def computed_gtfs_harmonized_snapshot(session: Session) -> dict[str, Any]:
|
|
global _SNAPSHOT_CACHE, _SNAPSHOT_CACHE_REVISION
|
|
revision = _snapshot_revision(session)
|
|
with _SNAPSHOT_CACHE_LOCK:
|
|
if _SNAPSHOT_CACHE is not None and _SNAPSHOT_CACHE_REVISION == revision:
|
|
return copy.deepcopy(_SNAPSHOT_CACHE)
|
|
snapshot = _compute_gtfs_harmonized_snapshot(session)
|
|
with _SNAPSHOT_CACHE_LOCK:
|
|
_SNAPSHOT_CACHE_REVISION = revision
|
|
_SNAPSHOT_CACHE = copy.deepcopy(snapshot)
|
|
return snapshot
|
|
|
|
|
|
def active_persisted_gtfs_snapshot(session: Session) -> dict[str, Any] | None:
|
|
snapshot = session.scalar(
|
|
select(GtfsHarmonizedSnapshot)
|
|
.where(GtfsHarmonizedSnapshot.status == "active")
|
|
.order_by(GtfsHarmonizedSnapshot.activated_at.desc(), GtfsHarmonizedSnapshot.id.desc())
|
|
.limit(1)
|
|
)
|
|
if snapshot is None:
|
|
return None
|
|
return _persisted_snapshot_payload(session, snapshot)
|
|
|
|
|
|
def create_gtfs_harmonized_snapshot(session: Session, *, activate: bool = True, note: str | None = None) -> dict[str, Any]:
|
|
computed = computed_gtfs_harmonized_snapshot(session)
|
|
snapshot_key = _materialized_snapshot_key(computed)
|
|
snapshot = session.scalar(select(GtfsHarmonizedSnapshot).where(GtfsHarmonizedSnapshot.snapshot_key == snapshot_key))
|
|
now = datetime.now(timezone.utc)
|
|
dataset_items_by_id = {int(item["dataset_id"]): item for item in computed["datasets"]}
|
|
if snapshot is None:
|
|
snapshot = GtfsHarmonizedSnapshot(
|
|
snapshot_key=snapshot_key,
|
|
status="draft",
|
|
source="computed",
|
|
note=note,
|
|
summary_json=json.dumps(computed["summary"], sort_keys=True, separators=(",", ":"), default=str),
|
|
metadata_json=json.dumps(
|
|
{
|
|
"computed_at": now.isoformat(),
|
|
"revision": _snapshot_revision(session),
|
|
"dataset_ids": computed["dataset_ids"],
|
|
},
|
|
sort_keys=True,
|
|
separators=(",", ":"),
|
|
default=str,
|
|
),
|
|
created_at=now,
|
|
)
|
|
session.add(snapshot)
|
|
session.flush()
|
|
for item in computed["datasets"]:
|
|
session.add(
|
|
GtfsHarmonizedSnapshotDataset(
|
|
snapshot_id=snapshot.id,
|
|
dataset_id=int(item["dataset_id"]),
|
|
source_id=int(item["source_id"]),
|
|
source_name=item.get("source_name"),
|
|
role=str(item["role"]),
|
|
reason=item.get("reason"),
|
|
shadowed_by_dataset_id=_optional_int(item.get("shadowed_by_dataset_id")),
|
|
shadowed_by_source_id=_optional_int(item.get("shadowed_by_source_id")),
|
|
authority_level=item.get("authority_level"),
|
|
review_authority_level=item.get("review_authority_level"),
|
|
route_count=int(item.get("route_count") or 0),
|
|
route_key_count=int(item.get("route_key_count") or 0),
|
|
route_key_overlap=int(item.get("route_key_overlap") or 0),
|
|
route_key_overlap_ratio=float(item.get("route_key_overlap_ratio") or 0),
|
|
metadata_json=json.dumps(item, sort_keys=True, separators=(",", ":"), default=str),
|
|
created_at=now,
|
|
)
|
|
)
|
|
_persist_harmonized_snapshot_routes(session, snapshot, dataset_items_by_id=dataset_items_by_id, now=now)
|
|
elif note is not None and note != snapshot.note:
|
|
snapshot.note = note
|
|
if snapshot.id is not None and not _snapshot_has_route_rows(session, int(snapshot.id)):
|
|
_persist_harmonized_snapshot_routes(session, snapshot, dataset_items_by_id=dataset_items_by_id, now=now)
|
|
if activate:
|
|
for older in session.scalars(
|
|
select(GtfsHarmonizedSnapshot).where(GtfsHarmonizedSnapshot.status == "active", GtfsHarmonizedSnapshot.id != snapshot.id)
|
|
).all():
|
|
older.status = "archived"
|
|
snapshot.status = "active"
|
|
if snapshot.activated_at is None:
|
|
snapshot.activated_at = now
|
|
session.flush()
|
|
return _persisted_snapshot_payload(session, snapshot)
|
|
|
|
|
|
def _compute_gtfs_harmonized_snapshot(session: Session) -> dict[str, Any]:
|
|
items = _active_gtfs_snapshot_items(session)
|
|
_apply_dataset_shadowing(items)
|
|
route_shadow = _route_shadow_result(session, items)
|
|
for item in items:
|
|
dataset_shadow = route_shadow["by_dataset"].get(int(item["dataset_id"]), {})
|
|
shadow_count = int(dataset_shadow.get("route_shadow_count") or 0)
|
|
item["route_shadow_count"] = shadow_count
|
|
item["route_shadow_ratio"] = round(shadow_count / max(1, int(item.get("route_count") or 0)), 3)
|
|
item["route_shadowed_by_dataset_ids"] = dataset_shadow.get("route_shadowed_by_dataset_ids", [])
|
|
|
|
public_items = [_snapshot_public_item(item) for item in items]
|
|
included = [item for item in public_items if item["role"] == "included"]
|
|
shadowed = [item for item in public_items if item["role"] == "shadowed"]
|
|
excluded = [item for item in public_items if item["role"] == "excluded"]
|
|
return {
|
|
"summary": {
|
|
"raw_active_datasets": len(items),
|
|
"included_datasets": len(included),
|
|
"shadowed_datasets": len(shadowed),
|
|
"excluded_datasets": len(excluded),
|
|
"included_sources": len({item["source_id"] for item in included}),
|
|
"shadowed_sources": len({item["source_id"] for item in shadowed}),
|
|
"route_shadowed_routes": route_shadow["route_shadowed_routes"],
|
|
"route_shadow_digest": _route_shadow_digest(route_shadow["shadowed_route_map"]),
|
|
},
|
|
"dataset_ids": [item["dataset_id"] for item in included],
|
|
"datasets": public_items,
|
|
"included": included,
|
|
"shadowed": shadowed,
|
|
"excluded": excluded,
|
|
"persisted": False,
|
|
"status": "computed",
|
|
}
|
|
|
|
|
|
def gtfs_harmonization_inventory(session: Session) -> dict[str, Any]:
|
|
feeds = [_feed_inventory_item(session, source) for source in _gtfs_sources(session)]
|
|
snapshot = gtfs_harmonized_snapshot(session)
|
|
snapshot_diagnostics = gtfs_harmonized_snapshot_diagnostics(session)
|
|
snapshot_by_dataset_id = {int(item["dataset_id"]): item for item in snapshot["datasets"]}
|
|
for feed in feeds:
|
|
active_dataset = feed.get("active_dataset")
|
|
feed["snapshot"] = (
|
|
snapshot_by_dataset_id.get(int(active_dataset["id"]))
|
|
if active_dataset is not None
|
|
else None
|
|
)
|
|
summary = {
|
|
"sources": len(feeds),
|
|
"active_sources": sum(1 for feed in feeds if feed["active_dataset"] is not None),
|
|
"datasets": sum(len(feed["datasets"]) for feed in feeds),
|
|
"snapshot_included_datasets": snapshot["summary"]["included_datasets"],
|
|
"snapshot_shadowed_datasets": snapshot["summary"]["shadowed_datasets"],
|
|
"snapshot_excluded_datasets": snapshot["summary"]["excluded_datasets"],
|
|
"ready": sum(1 for feed in feeds if feed["qa_status"] == "ready"),
|
|
"needs_review": sum(1 for feed in feeds if feed["qa_status"] == "needs_review"),
|
|
"blocked": sum(1 for feed in feeds if feed["qa_status"] == "blocked"),
|
|
}
|
|
return {
|
|
"summary": summary,
|
|
"snapshot": snapshot,
|
|
"snapshot_diagnostics": snapshot_diagnostics,
|
|
"feeds": feeds,
|
|
}
|
|
|
|
|
|
def gtfs_harmonization_feed_detail(session: Session, source_id: int) -> dict[str, Any] | None:
|
|
source = session.get(Source, source_id)
|
|
if source is None or source.kind != "gtfs":
|
|
return None
|
|
feed = _feed_inventory_item(session, source)
|
|
snapshot = gtfs_harmonized_snapshot(session)
|
|
if feed["active_dataset"] is not None:
|
|
snapshot_by_dataset_id = {int(item["dataset_id"]): item for item in snapshot["datasets"]}
|
|
feed["snapshot"] = snapshot_by_dataset_id.get(int(feed["active_dataset"]["id"]))
|
|
else:
|
|
feed["snapshot"] = None
|
|
return {
|
|
**feed,
|
|
"snapshot_summary": snapshot["summary"],
|
|
"sections": _feed_sections(feed),
|
|
}
|
|
|
|
|
|
def gtfs_qa_review_payload(notes: str | None) -> dict[str, Any]:
|
|
return _qa_review_payload(notes)
|
|
|
|
|
|
def gtfs_route_overlap_key(route: GtfsRoute) -> str | None:
|
|
return _route_overlap_key(
|
|
route_id=route.route_id,
|
|
route_key=route.route_key,
|
|
mode=route.mode,
|
|
short_name=route.short_name,
|
|
long_name=route.long_name,
|
|
operator_key=route.operator_key,
|
|
operator_name=route.operator_name,
|
|
)
|
|
|
|
|
|
def _gtfs_sources(session: Session) -> list[Source]:
|
|
return session.scalars(select(Source).where(Source.kind == "gtfs").order_by(Source.country, Source.priority, Source.name, Source.id)).all()
|
|
|
|
|
|
def _active_persisted_gtfs_dataset_ids(session: Session) -> list[int]:
|
|
snapshot = session.scalar(
|
|
select(GtfsHarmonizedSnapshot.id)
|
|
.where(GtfsHarmonizedSnapshot.status == "active")
|
|
.order_by(GtfsHarmonizedSnapshot.activated_at.desc(), GtfsHarmonizedSnapshot.id.desc())
|
|
.limit(1)
|
|
)
|
|
if snapshot is None:
|
|
return []
|
|
rows = session.execute(
|
|
select(GtfsHarmonizedSnapshotDataset.dataset_id)
|
|
.join(Dataset, Dataset.id == GtfsHarmonizedSnapshotDataset.dataset_id)
|
|
.where(
|
|
GtfsHarmonizedSnapshotDataset.snapshot_id == int(snapshot),
|
|
GtfsHarmonizedSnapshotDataset.role == "included",
|
|
Dataset.kind == "gtfs",
|
|
Dataset.is_active.is_(True),
|
|
Dataset.status == "imported",
|
|
)
|
|
.order_by(GtfsHarmonizedSnapshotDataset.id)
|
|
).all()
|
|
return [int(row[0]) for row in rows]
|
|
|
|
|
|
def _active_persisted_gtfs_route_shadow_map(session: Session) -> dict[int, int] | None:
|
|
snapshot_id = session.scalar(
|
|
select(GtfsHarmonizedSnapshot.id)
|
|
.where(GtfsHarmonizedSnapshot.status == "active")
|
|
.order_by(GtfsHarmonizedSnapshot.activated_at.desc(), GtfsHarmonizedSnapshot.id.desc())
|
|
.limit(1)
|
|
)
|
|
if snapshot_id is None:
|
|
return None
|
|
rows = session.execute(
|
|
select(GtfsHarmonizedSnapshotRoute.gtfs_route_id, GtfsHarmonizedSnapshotRoute.shadowed_by_gtfs_route_id)
|
|
.where(
|
|
GtfsHarmonizedSnapshotRoute.snapshot_id == int(snapshot_id),
|
|
GtfsHarmonizedSnapshotRoute.role == "shadowed",
|
|
GtfsHarmonizedSnapshotRoute.shadowed_by_gtfs_route_id.is_not(None),
|
|
)
|
|
).all()
|
|
if not rows:
|
|
has_route_rows = bool(
|
|
session.scalar(
|
|
select(GtfsHarmonizedSnapshotRoute.id)
|
|
.where(GtfsHarmonizedSnapshotRoute.snapshot_id == int(snapshot_id))
|
|
.limit(1)
|
|
)
|
|
)
|
|
return {} if has_route_rows else None
|
|
return {int(route_id): int(shadowed_by_route_id) for route_id, shadowed_by_route_id in rows if shadowed_by_route_id is not None}
|
|
|
|
|
|
def _persisted_snapshot_payload(session: Session, snapshot: GtfsHarmonizedSnapshot) -> dict[str, Any]:
|
|
rows = session.scalars(
|
|
select(GtfsHarmonizedSnapshotDataset)
|
|
.where(GtfsHarmonizedSnapshotDataset.snapshot_id == snapshot.id)
|
|
.order_by(GtfsHarmonizedSnapshotDataset.id)
|
|
).all()
|
|
active_dataset_ids = set(_active_dataset_ids_for_snapshot_rows(session, rows))
|
|
datasets = [_persisted_snapshot_dataset_payload(row, dataset_active=int(row.dataset_id) in active_dataset_ids) for row in rows]
|
|
included = [item for item in datasets if item["role"] == "included" and item.get("dataset_active")]
|
|
shadowed = [item for item in datasets if item["role"] == "shadowed"]
|
|
excluded = [item for item in datasets if item["role"] == "excluded"]
|
|
try:
|
|
summary = json.loads(snapshot.summary_json)
|
|
except json.JSONDecodeError:
|
|
summary = {}
|
|
if not isinstance(summary, dict):
|
|
summary = {}
|
|
route_rows = _snapshot_route_row_counts(session, snapshot.id)
|
|
summary = {
|
|
**summary,
|
|
"included_datasets": len(included),
|
|
"shadowed_datasets": len(shadowed),
|
|
"excluded_datasets": len(excluded),
|
|
"inactive_included_datasets": sum(1 for item in datasets if item["role"] == "included" and not item.get("dataset_active")),
|
|
"snapshot_route_rows": route_rows["total"],
|
|
"snapshot_route_included_rows": route_rows["included"],
|
|
"snapshot_route_shadowed_rows": route_rows["shadowed"],
|
|
"snapshot_route_excluded_rows": route_rows["excluded"],
|
|
"route_shadowed_routes": route_rows["shadowed"] if route_rows["total"] else summary.get("route_shadowed_routes", 0),
|
|
}
|
|
return {
|
|
"id": snapshot.id,
|
|
"snapshot_key": snapshot.snapshot_key,
|
|
"status": snapshot.status,
|
|
"persisted": True,
|
|
"source": snapshot.source,
|
|
"note": snapshot.note,
|
|
"created_at": _iso(snapshot.created_at),
|
|
"activated_at": _iso(snapshot.activated_at),
|
|
"summary": summary,
|
|
"dataset_ids": [item["dataset_id"] for item in included],
|
|
"datasets": datasets,
|
|
"included": included,
|
|
"shadowed": shadowed,
|
|
"excluded": excluded,
|
|
}
|
|
|
|
|
|
def _snapshot_route_row_counts(session: Session, snapshot_id: int) -> dict[str, int]:
|
|
rows = session.execute(
|
|
select(GtfsHarmonizedSnapshotRoute.role, func.count())
|
|
.where(GtfsHarmonizedSnapshotRoute.snapshot_id == int(snapshot_id))
|
|
.group_by(GtfsHarmonizedSnapshotRoute.role)
|
|
).all()
|
|
by_role = {str(role): int(count or 0) for role, count in rows}
|
|
return {
|
|
"total": sum(by_role.values()),
|
|
"included": by_role.get("included", 0),
|
|
"shadowed": by_role.get("shadowed", 0),
|
|
"excluded": by_role.get("excluded", 0),
|
|
}
|
|
|
|
|
|
def _snapshot_has_route_rows(session: Session, snapshot_id: int) -> bool:
|
|
return bool(
|
|
session.scalar(
|
|
select(GtfsHarmonizedSnapshotRoute.id)
|
|
.where(GtfsHarmonizedSnapshotRoute.snapshot_id == int(snapshot_id))
|
|
.limit(1)
|
|
)
|
|
)
|
|
|
|
|
|
def _summary_int(summary: dict[str, Any], key: str) -> int:
|
|
try:
|
|
return int(summary.get(key) or 0)
|
|
except (TypeError, ValueError):
|
|
return 0
|
|
|
|
|
|
def _snapshot_route_payload(row: GtfsHarmonizedSnapshotRoute) -> dict[str, Any]:
|
|
metadata = _json_object(row.metadata_json)
|
|
return {
|
|
"id": row.id,
|
|
"snapshot_id": row.snapshot_id,
|
|
"gtfs_route_id": row.gtfs_route_id,
|
|
"dataset_id": row.dataset_id,
|
|
"source_id": row.source_id,
|
|
"source_name": metadata.get("source_name"),
|
|
"route_id": row.route_id,
|
|
"route_ref": metadata.get("route_ref"),
|
|
"route_name": metadata.get("route_name"),
|
|
"mode": metadata.get("mode"),
|
|
"operator": metadata.get("operator"),
|
|
"overlap_key": row.overlap_key,
|
|
"role": row.role,
|
|
"reason": row.reason,
|
|
"shadowed_by_gtfs_route_id": row.shadowed_by_gtfs_route_id,
|
|
"shadowed_by_dataset_id": row.shadowed_by_dataset_id,
|
|
"shadowed_by_source_id": row.shadowed_by_source_id,
|
|
"metadata": metadata,
|
|
"created_at": _iso(row.created_at),
|
|
}
|
|
|
|
|
|
def _persist_harmonized_snapshot_routes(
|
|
session: Session,
|
|
snapshot: GtfsHarmonizedSnapshot,
|
|
*,
|
|
dataset_items_by_id: dict[int, dict[str, Any]],
|
|
now: datetime,
|
|
) -> None:
|
|
if not dataset_items_by_id:
|
|
return
|
|
route_shadow_map = _computed_gtfs_route_shadow_map(session)
|
|
shadow_winners = {
|
|
int(route.id): route
|
|
for route in session.scalars(
|
|
select(GtfsRoute).where(GtfsRoute.id.in_(sorted(set(route_shadow_map.values()))))
|
|
).all()
|
|
} if route_shadow_map else {}
|
|
routes = session.scalars(
|
|
select(GtfsRoute)
|
|
.where(GtfsRoute.dataset_id.in_(sorted(dataset_items_by_id)))
|
|
.order_by(GtfsRoute.dataset_id, GtfsRoute.id)
|
|
).all()
|
|
rows: list[GtfsHarmonizedSnapshotRoute] = []
|
|
for route in routes:
|
|
dataset_item = dataset_items_by_id.get(int(route.dataset_id))
|
|
if dataset_item is None:
|
|
continue
|
|
winner = shadow_winners.get(int(route_shadow_map.get(int(route.id), 0)))
|
|
if winner is not None:
|
|
role = "shadowed"
|
|
reason = "route_covered_by_higher_precedence_dataset"
|
|
shadowed_by_route_id = int(winner.id)
|
|
shadowed_by_dataset_id = int(winner.dataset_id)
|
|
shadowed_by_source_id = _optional_int(dataset_items_by_id.get(int(winner.dataset_id), {}).get("source_id"))
|
|
else:
|
|
role = str(dataset_item.get("role") or "included")
|
|
reason = dataset_item.get("reason")
|
|
shadowed_by_route_id = None
|
|
shadowed_by_dataset_id = _optional_int(dataset_item.get("shadowed_by_dataset_id"))
|
|
shadowed_by_source_id = _optional_int(dataset_item.get("shadowed_by_source_id"))
|
|
metadata = {
|
|
"route_ref": route.short_name,
|
|
"route_name": route.long_name,
|
|
"mode": route.mode,
|
|
"operator": route.operator_name,
|
|
"route_key": route.route_key,
|
|
"operator_key": route.operator_key,
|
|
"source_name": dataset_item.get("source_name"),
|
|
"authority_level": dataset_item.get("authority_level"),
|
|
"review_authority_level": dataset_item.get("review_authority_level"),
|
|
}
|
|
rows.append(
|
|
GtfsHarmonizedSnapshotRoute(
|
|
snapshot_id=snapshot.id,
|
|
gtfs_route_id=int(route.id),
|
|
dataset_id=int(route.dataset_id),
|
|
source_id=int(dataset_item["source_id"]),
|
|
route_id=str(route.route_id),
|
|
overlap_key=gtfs_route_overlap_key(route),
|
|
role=role,
|
|
reason=reason,
|
|
shadowed_by_gtfs_route_id=shadowed_by_route_id,
|
|
shadowed_by_dataset_id=shadowed_by_dataset_id,
|
|
shadowed_by_source_id=shadowed_by_source_id,
|
|
metadata_json=json.dumps(metadata, sort_keys=True, separators=(",", ":"), default=str),
|
|
created_at=now,
|
|
)
|
|
)
|
|
if len(rows) >= 5000:
|
|
session.add_all(rows)
|
|
session.flush()
|
|
rows.clear()
|
|
if rows:
|
|
session.add_all(rows)
|
|
session.flush()
|
|
|
|
|
|
def _active_dataset_ids_for_snapshot_rows(session: Session, rows: list[GtfsHarmonizedSnapshotDataset]) -> list[int]:
|
|
dataset_ids = [int(row.dataset_id) for row in rows]
|
|
if not dataset_ids:
|
|
return []
|
|
return [
|
|
int(row[0])
|
|
for row in session.execute(
|
|
select(Dataset.id).where(
|
|
Dataset.id.in_(dataset_ids),
|
|
Dataset.kind == "gtfs",
|
|
Dataset.status == "imported",
|
|
Dataset.is_active.is_(True),
|
|
)
|
|
).all()
|
|
]
|
|
|
|
|
|
def _persisted_snapshot_dataset_payload(row: GtfsHarmonizedSnapshotDataset, *, dataset_active: bool) -> dict[str, Any]:
|
|
return {
|
|
"dataset_id": row.dataset_id,
|
|
"source_id": row.source_id,
|
|
"source_name": row.source_name,
|
|
"country": _snapshot_dataset_metadata(row).get("country"),
|
|
"priority": _snapshot_dataset_metadata(row).get("priority"),
|
|
"authority_level": row.authority_level,
|
|
"review_authority_level": row.review_authority_level,
|
|
"review_status": _snapshot_dataset_metadata(row).get("review_status"),
|
|
"route_count": row.route_count,
|
|
"route_key_count": row.route_key_count,
|
|
"role": row.role,
|
|
"reason": row.reason,
|
|
"shadowed_by_dataset_id": row.shadowed_by_dataset_id,
|
|
"shadowed_by_source_id": row.shadowed_by_source_id,
|
|
"route_key_overlap": row.route_key_overlap,
|
|
"route_key_overlap_ratio": row.route_key_overlap_ratio,
|
|
"route_shadow_count": int(_snapshot_dataset_metadata(row).get("route_shadow_count") or 0),
|
|
"route_shadow_ratio": float(_snapshot_dataset_metadata(row).get("route_shadow_ratio") or 0),
|
|
"route_shadowed_by_dataset_ids": _snapshot_dataset_metadata(row).get("route_shadowed_by_dataset_ids") or [],
|
|
"dataset_active": dataset_active,
|
|
}
|
|
|
|
|
|
def _snapshot_dataset_metadata(row: GtfsHarmonizedSnapshotDataset) -> dict[str, Any]:
|
|
if not row.metadata_json:
|
|
return {}
|
|
try:
|
|
payload = json.loads(row.metadata_json)
|
|
except json.JSONDecodeError:
|
|
return {}
|
|
return payload if isinstance(payload, dict) else {}
|
|
|
|
|
|
def _materialized_snapshot_key(snapshot: dict[str, Any]) -> str:
|
|
payload = {
|
|
"datasets": [
|
|
{
|
|
"dataset_id": item.get("dataset_id"),
|
|
"source_id": item.get("source_id"),
|
|
"role": item.get("role"),
|
|
"reason": item.get("reason"),
|
|
"shadowed_by_dataset_id": item.get("shadowed_by_dataset_id"),
|
|
"route_count": item.get("route_count"),
|
|
"route_key_count": item.get("route_key_count"),
|
|
"route_key_overlap": item.get("route_key_overlap"),
|
|
"route_shadow_count": item.get("route_shadow_count"),
|
|
}
|
|
for item in snapshot.get("datasets", [])
|
|
],
|
|
"dataset_ids": snapshot.get("dataset_ids", []),
|
|
"route_shadow_digest": (snapshot.get("summary") or {}).get("route_shadow_digest"),
|
|
}
|
|
digest = hashlib.sha256(json.dumps(payload, sort_keys=True, separators=(",", ":"), default=str).encode("utf-8")).hexdigest()
|
|
return f"gtfs-snapshot-{digest[:32]}"
|
|
|
|
|
|
def _route_shadow_digest(route_shadow_map: dict[int, int]) -> str:
|
|
if not route_shadow_map:
|
|
return ""
|
|
pairs = sorted((int(route_id), int(shadowed_by_route_id)) for route_id, shadowed_by_route_id in route_shadow_map.items())
|
|
encoded = json.dumps(pairs, separators=(",", ":")).encode("utf-8")
|
|
return hashlib.sha256(encoded).hexdigest()[:32]
|
|
|
|
|
|
def _snapshot_revision(session: Session) -> tuple[object, ...]:
|
|
rows = session.execute(
|
|
select(
|
|
Dataset.id,
|
|
Dataset.sha256,
|
|
Dataset.status,
|
|
Source.id,
|
|
Source.name,
|
|
Source.country,
|
|
Source.priority,
|
|
Source.enabled,
|
|
Source.status,
|
|
Source.last_run_at,
|
|
Source.source_basis,
|
|
Source.notes,
|
|
func.count(GtfsRoute.id),
|
|
func.max(GtfsRoute.id),
|
|
)
|
|
.join(Source, Source.id == Dataset.source_id)
|
|
.outerjoin(GtfsRoute, GtfsRoute.dataset_id == Dataset.id)
|
|
.where(Dataset.kind == "gtfs", Dataset.is_active.is_(True))
|
|
.group_by(
|
|
Dataset.id,
|
|
Dataset.sha256,
|
|
Dataset.status,
|
|
Source.id,
|
|
Source.name,
|
|
Source.country,
|
|
Source.priority,
|
|
Source.enabled,
|
|
Source.status,
|
|
Source.last_run_at,
|
|
Source.source_basis,
|
|
Source.notes,
|
|
)
|
|
.order_by(Dataset.id)
|
|
).all()
|
|
return tuple(
|
|
(
|
|
int(row[0]),
|
|
row[1],
|
|
row[2],
|
|
int(row[3]),
|
|
row[4],
|
|
row[5],
|
|
row[6],
|
|
bool(row[7]),
|
|
row[8],
|
|
_iso(row[9]),
|
|
row[10],
|
|
row[11],
|
|
int(row[12] or 0),
|
|
int(row[13] or 0),
|
|
)
|
|
for row in rows
|
|
)
|
|
|
|
|
|
def _raw_active_gtfs_dataset_ids(session: Session, source_ids: list[int] | None = None) -> list[int]:
|
|
stmt = select(Dataset.id).where(Dataset.kind == "gtfs", Dataset.is_active.is_(True)).order_by(Dataset.id)
|
|
if source_ids:
|
|
stmt = stmt.where(Dataset.source_id.in_(source_ids))
|
|
return [int(row[0]) for row in session.execute(stmt).all()]
|
|
|
|
|
|
def _active_gtfs_snapshot_items(session: Session) -> list[dict[str, Any]]:
|
|
active_rows = session.execute(
|
|
select(Dataset, Source)
|
|
.join(Source, Source.id == Dataset.source_id)
|
|
.where(Dataset.kind == "gtfs", Dataset.is_active.is_(True))
|
|
.order_by(Source.country, Source.priority, Source.name, Dataset.id)
|
|
).all()
|
|
dataset_ids = [int(dataset.id) for dataset, _source in active_rows]
|
|
route_profiles = _route_profiles_by_dataset(session, dataset_ids)
|
|
bboxes = _route_bboxes_by_dataset(session, dataset_ids)
|
|
items: list[dict[str, Any]] = []
|
|
for dataset, source in active_rows:
|
|
profile = route_profiles.get(int(dataset.id), {"route_count": 0, "route_keys": set()})
|
|
route_keys = profile["route_keys"]
|
|
review = _qa_review_payload(source.notes)
|
|
authority_level = _effective_authority_level(source, review)
|
|
exclusion = _source_snapshot_exclusion(source, review)
|
|
items.append(
|
|
{
|
|
"dataset_id": int(dataset.id),
|
|
"source_id": int(source.id),
|
|
"source_name": source.name,
|
|
"country": source.country,
|
|
"priority": source.priority,
|
|
"authority_level": authority_level,
|
|
"review_authority_level": review["authority_level"],
|
|
"review_status": review["status"],
|
|
"route_count": int(profile["route_count"]),
|
|
"route_key_count": len(route_keys),
|
|
"route_keys": route_keys,
|
|
"bbox": bboxes.get(int(dataset.id)),
|
|
"score": _source_precedence_score(source, authority_level, len(route_keys)),
|
|
"role": "excluded" if exclusion else "included",
|
|
"reason": exclusion,
|
|
"shadowed_by_dataset_id": None,
|
|
"shadowed_by_source_id": None,
|
|
"route_key_overlap": 0,
|
|
"route_key_overlap_ratio": 0.0,
|
|
"route_shadow_count": 0,
|
|
"route_shadow_ratio": 0.0,
|
|
"route_shadowed_by_dataset_ids": [],
|
|
}
|
|
)
|
|
return items
|
|
|
|
|
|
def _apply_dataset_shadowing(items: list[dict[str, Any]]) -> None:
|
|
eligible = [item for item in items if item["role"] == "included"]
|
|
for item in eligible:
|
|
shadow = _best_shadowing_dataset(item, eligible)
|
|
if shadow is None:
|
|
continue
|
|
other, overlap, ratio = shadow
|
|
item["role"] = "shadowed"
|
|
item["reason"] = "covered_by_higher_precedence_dataset"
|
|
item["shadowed_by_dataset_id"] = other["dataset_id"]
|
|
item["shadowed_by_source_id"] = other["source_id"]
|
|
item["route_key_overlap"] = overlap
|
|
item["route_key_overlap_ratio"] = round(ratio, 3)
|
|
|
|
|
|
def _route_profiles_by_dataset(session: Session, dataset_ids: list[int]) -> dict[int, dict[str, Any]]:
|
|
profiles: dict[int, dict[str, Any]] = {
|
|
int(dataset_id): {"route_count": 0, "route_keys": set()}
|
|
for dataset_id in dataset_ids
|
|
}
|
|
if not dataset_ids:
|
|
return profiles
|
|
rows = session.execute(
|
|
select(
|
|
GtfsRoute.dataset_id,
|
|
GtfsRoute.route_id,
|
|
GtfsRoute.route_key,
|
|
GtfsRoute.mode,
|
|
GtfsRoute.short_name,
|
|
GtfsRoute.long_name,
|
|
GtfsRoute.operator_key,
|
|
GtfsRoute.operator_name,
|
|
).where(GtfsRoute.dataset_id.in_(dataset_ids))
|
|
).all()
|
|
for row in rows:
|
|
dataset_id = int(row.dataset_id)
|
|
profile = profiles.setdefault(dataset_id, {"route_count": 0, "route_keys": set()})
|
|
profile["route_count"] += 1
|
|
key = _route_overlap_key(
|
|
route_id=row.route_id,
|
|
route_key=row.route_key,
|
|
mode=row.mode,
|
|
short_name=row.short_name,
|
|
long_name=row.long_name,
|
|
operator_key=row.operator_key,
|
|
operator_name=row.operator_name,
|
|
)
|
|
if key:
|
|
profile["route_keys"].add(key)
|
|
return profiles
|
|
|
|
|
|
def _route_bboxes_by_dataset(session: Session, dataset_ids: list[int]) -> dict[int, tuple[float, float, float, float]]:
|
|
if not dataset_ids:
|
|
return {}
|
|
rows = session.execute(
|
|
select(
|
|
GtfsRoute.dataset_id,
|
|
func.min(GtfsRoute.min_lon),
|
|
func.min(GtfsRoute.min_lat),
|
|
func.max(GtfsRoute.max_lon),
|
|
func.max(GtfsRoute.max_lat),
|
|
)
|
|
.where(GtfsRoute.dataset_id.in_(dataset_ids))
|
|
.group_by(GtfsRoute.dataset_id)
|
|
).all()
|
|
bboxes: dict[int, tuple[float, float, float, float]] = {}
|
|
for dataset_id, min_lon, min_lat, max_lon, max_lat in rows:
|
|
if None in {min_lon, min_lat, max_lon, max_lat}:
|
|
continue
|
|
bboxes[int(dataset_id)] = (float(min_lon), float(min_lat), float(max_lon), float(max_lat))
|
|
return bboxes
|
|
|
|
|
|
def _route_shadow_result(session: Session, items: list[dict[str, Any]]) -> dict[str, Any]:
|
|
candidate_items = {int(item["dataset_id"]): item for item in items if item["role"] == "included"}
|
|
if len(candidate_items) < 2:
|
|
return {"shadowed_route_ids": set(), "shadowed_route_map": {}, "route_shadowed_routes": 0, "by_dataset": {}}
|
|
routes_by_key: dict[str, list[GtfsRoute]] = {}
|
|
for route in session.scalars(
|
|
select(GtfsRoute)
|
|
.where(GtfsRoute.dataset_id.in_(list(candidate_items)))
|
|
.order_by(GtfsRoute.dataset_id, GtfsRoute.route_id, GtfsRoute.id)
|
|
).all():
|
|
key = gtfs_route_overlap_key(route)
|
|
if not key:
|
|
continue
|
|
routes_by_key.setdefault(key, []).append(route)
|
|
|
|
shadowed_route_ids: set[int] = set()
|
|
shadowed_route_map: dict[int, int] = {}
|
|
by_dataset: dict[int, dict[str, Any]] = {}
|
|
for routes in routes_by_key.values():
|
|
if len({int(route.dataset_id) for route in routes}) < 2:
|
|
continue
|
|
for route in routes:
|
|
loser_item = candidate_items.get(int(route.dataset_id))
|
|
if loser_item is None:
|
|
continue
|
|
winner = _best_route_shadow(route, loser_item, routes, candidate_items)
|
|
if winner is None:
|
|
continue
|
|
shadowed_route_ids.add(int(route.id))
|
|
shadowed_route_map[int(route.id)] = int(winner.id)
|
|
dataset_shadow = by_dataset.setdefault(
|
|
int(route.dataset_id),
|
|
{"route_shadow_count": 0, "route_shadowed_by_dataset_ids": set()},
|
|
)
|
|
dataset_shadow["route_shadow_count"] += 1
|
|
dataset_shadow["route_shadowed_by_dataset_ids"].add(int(winner.dataset_id))
|
|
public_by_dataset = {
|
|
dataset_id: {
|
|
"route_shadow_count": int(payload["route_shadow_count"]),
|
|
"route_shadowed_by_dataset_ids": sorted(payload["route_shadowed_by_dataset_ids"]),
|
|
}
|
|
for dataset_id, payload in by_dataset.items()
|
|
}
|
|
return {
|
|
"shadowed_route_ids": shadowed_route_ids,
|
|
"shadowed_route_map": shadowed_route_map,
|
|
"route_shadowed_routes": len(shadowed_route_ids),
|
|
"by_dataset": public_by_dataset,
|
|
}
|
|
|
|
|
|
def _best_route_shadow(
|
|
route: GtfsRoute,
|
|
loser_item: dict[str, Any],
|
|
candidates: list[GtfsRoute],
|
|
candidate_items: dict[int, dict[str, Any]],
|
|
) -> GtfsRoute | None:
|
|
route_bbox = _route_bbox(route)
|
|
if route_bbox is None:
|
|
return None
|
|
matches: list[tuple[tuple[int, int, int, int], GtfsRoute]] = []
|
|
for candidate in candidates:
|
|
if int(candidate.id) == int(route.id) or int(candidate.dataset_id) == int(route.dataset_id):
|
|
continue
|
|
winner_item = candidate_items.get(int(candidate.dataset_id))
|
|
if winner_item is None or not _can_shadow_route(winner_item, loser_item):
|
|
continue
|
|
candidate_bbox = _route_bbox(candidate)
|
|
if candidate_bbox is None:
|
|
continue
|
|
if not _snapshot_bboxes_compatible(candidate_bbox, route_bbox, margin=GTFS_ROUTE_SHADOW_BBOX_MARGIN_DEG):
|
|
continue
|
|
matches.append((winner_item["score"], candidate))
|
|
if not matches:
|
|
return None
|
|
return max(matches, key=lambda item: item[0])[1]
|
|
|
|
|
|
def _can_shadow_route(other: dict[str, Any], item: dict[str, Any]) -> bool:
|
|
if other["country"] and item["country"] and str(other["country"]).upper() != str(item["country"]).upper():
|
|
return False
|
|
return tuple(other["score"]) > tuple(item["score"])
|
|
|
|
|
|
def _route_bbox(route: GtfsRoute) -> tuple[float, float, float, float] | None:
|
|
values = (route.min_lon, route.min_lat, route.max_lon, route.max_lat)
|
|
if any(value is None for value in values):
|
|
return None
|
|
return (float(route.min_lon), float(route.min_lat), float(route.max_lon), float(route.max_lat))
|
|
|
|
|
|
def _route_overlap_key(
|
|
*,
|
|
route_id: str | None,
|
|
route_key: str | None,
|
|
mode: str | None,
|
|
short_name: str | None,
|
|
long_name: str | None,
|
|
operator_key: str | None,
|
|
operator_name: str | None,
|
|
) -> str | None:
|
|
_ = operator_key, operator_name
|
|
route_part = route_key or short_name or long_name or route_id
|
|
normalized_route = _snapshot_key_text(route_part)
|
|
if not normalized_route:
|
|
return None
|
|
mode_part = _snapshot_key_text(mode)
|
|
return "|".join(part for part in [mode_part, normalized_route] if part)
|
|
|
|
|
|
def _snapshot_key_text(value: str | None) -> str:
|
|
text = str(value or "").casefold().strip()
|
|
text = text.replace("ß", "ss")
|
|
text = re.sub(r"[^a-z0-9]+", " ", text)
|
|
return re.sub(r"\s+", " ", text).strip()
|
|
|
|
|
|
def _source_snapshot_exclusion(source: Source, review: dict[str, Any]) -> str | None:
|
|
if source.enabled is False:
|
|
return "source_disabled"
|
|
if review["status"] in {"blocked", "rejected"}:
|
|
return f"review_{review['status']}"
|
|
if any(review.get(key) == "no" for key in ("can_import", "can_derive", "can_redistribute")):
|
|
return "license_blocks_publication"
|
|
if review.get("commercial_restrictions") == "yes":
|
|
return "commercial_restrictions"
|
|
return None
|
|
|
|
|
|
def _effective_authority_level(source: Source, review: dict[str, Any]) -> str:
|
|
reviewed = str(review.get("authority_level") or "unknown")
|
|
if reviewed != "unknown":
|
|
return reviewed
|
|
text = " ".join(str(value or "") for value in [source.name, source.source_basis, source.notes]).casefold()
|
|
if "delfi" in text or "gtfs.de" in text or "national gtfs" in text or "national official" in text:
|
|
return "national_official"
|
|
if "regional authority" in text:
|
|
return "regional_authority"
|
|
if "operator feed" in text:
|
|
return "operator"
|
|
if "mobility database mirror" in text or "transitland" in text:
|
|
return "secondary_discovery"
|
|
return "unknown"
|
|
|
|
|
|
def _source_precedence_score(source: Source, authority_level: str, route_key_count: int) -> tuple[int, int, int, int]:
|
|
authority = GTFS_AUTHORITY_SCORES.get(str(authority_level or "unknown"), GTFS_AUTHORITY_SCORES["unknown"])
|
|
priority = GTFS_PRIORITY_SCORES.get(str(source.priority or "").upper(), 0)
|
|
coverage = min(max(int(route_key_count), 0), 100_000)
|
|
source_id_tiebreak = -int(source.id or 0)
|
|
return (authority, priority, coverage, source_id_tiebreak)
|
|
|
|
|
|
def _best_shadowing_dataset(item: dict[str, Any], candidates: list[dict[str, Any]]) -> tuple[dict[str, Any], int, float] | None:
|
|
item_keys = item["route_keys"]
|
|
if not item_keys:
|
|
return None
|
|
matches: list[tuple[tuple[int, int, int, int], float, int, dict[str, Any]]] = []
|
|
for other in candidates:
|
|
if other["dataset_id"] == item["dataset_id"]:
|
|
continue
|
|
if not _can_shadow_dataset(other, item):
|
|
continue
|
|
if not _snapshot_bboxes_compatible(other.get("bbox"), item.get("bbox")):
|
|
continue
|
|
overlap = len(item_keys & other["route_keys"])
|
|
if overlap <= 0:
|
|
continue
|
|
ratio = overlap / max(1, len(item_keys))
|
|
if not _route_overlap_is_shadowing(item, overlap, ratio):
|
|
continue
|
|
matches.append((other["score"], ratio, overlap, other))
|
|
if not matches:
|
|
return None
|
|
_score, ratio, overlap, other = max(matches, key=lambda row: (row[0], row[1], row[2]))
|
|
return other, overlap, ratio
|
|
|
|
|
|
def _can_shadow_dataset(other: dict[str, Any], item: dict[str, Any]) -> bool:
|
|
if other["country"] and item["country"] and str(other["country"]).upper() != str(item["country"]).upper():
|
|
return False
|
|
if tuple(other["score"]) > tuple(item["score"]):
|
|
return True
|
|
if tuple(other["score"]) == tuple(item["score"]):
|
|
return int(other["route_key_count"]) > int(item["route_key_count"])
|
|
return False
|
|
|
|
|
|
def _snapshot_bboxes_compatible(
|
|
left: tuple[float, float, float, float] | None,
|
|
right: tuple[float, float, float, float] | None,
|
|
*,
|
|
margin: float = 0.0,
|
|
) -> bool:
|
|
if left is None or right is None:
|
|
return True
|
|
left_min_lon, left_min_lat, left_max_lon, left_max_lat = left
|
|
right_min_lon, right_min_lat, right_max_lon, right_max_lat = right
|
|
return not (
|
|
left_max_lon + margin < right_min_lon
|
|
or left_min_lon - margin > right_max_lon
|
|
or left_max_lat + margin < right_min_lat
|
|
or left_min_lat - margin > right_max_lat
|
|
)
|
|
|
|
|
|
def _route_overlap_is_shadowing(item: dict[str, Any], overlap: int, ratio: float) -> bool:
|
|
route_key_count = int(item["route_key_count"])
|
|
if route_key_count < GTFS_SHADOW_ROUTE_MIN_OVERLAP:
|
|
return overlap >= GTFS_SHADOW_SMALL_ROUTE_MIN_OVERLAP and ratio >= GTFS_SHADOW_SMALL_ROUTE_RATIO
|
|
return overlap >= GTFS_SHADOW_ROUTE_MIN_OVERLAP and ratio >= GTFS_SHADOW_ROUTE_RATIO
|
|
|
|
|
|
def _snapshot_public_item(item: dict[str, Any]) -> dict[str, Any]:
|
|
return {
|
|
"dataset_id": item["dataset_id"],
|
|
"source_id": item["source_id"],
|
|
"source_name": item["source_name"],
|
|
"country": item["country"],
|
|
"priority": item["priority"],
|
|
"authority_level": item["authority_level"],
|
|
"review_authority_level": item["review_authority_level"],
|
|
"review_status": item["review_status"],
|
|
"route_count": item["route_count"],
|
|
"route_key_count": item["route_key_count"],
|
|
"role": item["role"],
|
|
"reason": item["reason"],
|
|
"shadowed_by_dataset_id": item["shadowed_by_dataset_id"],
|
|
"shadowed_by_source_id": item["shadowed_by_source_id"],
|
|
"route_key_overlap": item["route_key_overlap"],
|
|
"route_key_overlap_ratio": item["route_key_overlap_ratio"],
|
|
"route_shadow_count": item["route_shadow_count"],
|
|
"route_shadow_ratio": item["route_shadow_ratio"],
|
|
"route_shadowed_by_dataset_ids": item["route_shadowed_by_dataset_ids"],
|
|
}
|
|
|
|
|
|
def _feed_inventory_item(session: Session, source: Source) -> dict[str, Any]:
|
|
datasets = sorted([dataset for dataset in source.datasets if dataset.kind == "gtfs"], key=lambda item: (not item.is_active, item.created_at, item.id))
|
|
active_dataset = next((dataset for dataset in datasets if dataset.is_active), None)
|
|
counts = dataset_row_counts(session, active_dataset.id, active_dataset.kind) if active_dataset is not None else {}
|
|
validation = _validate_gtfs_dataset(session, source, active_dataset, counts)
|
|
overlap = _overlap_summary(session, active_dataset)
|
|
service = _service_horizon(session, active_dataset)
|
|
issues = [*validation["issues"], *service["issues"], *overlap["issues"], *_license_issues(source)]
|
|
qa_status = _qa_status(issues, active_dataset)
|
|
return {
|
|
"source": _source_payload(source),
|
|
"active_dataset": None if active_dataset is None else _dataset_payload(active_dataset, counts),
|
|
"datasets": [_dataset_payload(dataset, dataset_row_counts(session, dataset.id, dataset.kind)) for dataset in datasets],
|
|
"counts": counts,
|
|
"validation": validation,
|
|
"service": service,
|
|
"overlap": overlap,
|
|
"license": _license_payload(source),
|
|
"issues": issues,
|
|
"qa_status": qa_status,
|
|
}
|
|
|
|
|
|
def _source_payload(source: Source) -> dict[str, Any]:
|
|
return {
|
|
"id": source.id,
|
|
"name": source.name,
|
|
"country": source.country,
|
|
"license": source.license,
|
|
"priority": source.priority,
|
|
"mode_scope": source.mode_scope,
|
|
"source_basis": source.source_basis,
|
|
"status": source.status,
|
|
"enabled": source.enabled,
|
|
"last_error": source.last_error,
|
|
"last_run_at": _iso(source.last_run_at),
|
|
"url": source.url,
|
|
"catalog_entry_id": source.catalog_entry_id,
|
|
"notes": source.notes,
|
|
"qa_review": _qa_review_payload(source.notes),
|
|
}
|
|
|
|
|
|
def _dataset_payload(dataset: Dataset, counts: dict[str, Any]) -> dict[str, Any]:
|
|
return {
|
|
"id": dataset.id,
|
|
"kind": dataset.kind,
|
|
"is_active": dataset.is_active,
|
|
"status": dataset.status,
|
|
"sha256": dataset.sha256,
|
|
"local_path": dataset.local_path,
|
|
"created_at": _iso(dataset.created_at),
|
|
"counts": counts,
|
|
}
|
|
|
|
|
|
def _validate_gtfs_dataset(session: Session, source: Source, dataset: Dataset | None, counts: dict[str, Any]) -> dict[str, Any]:
|
|
if dataset is None:
|
|
return {
|
|
"status": "blocked",
|
|
"items": [],
|
|
"issues": [_issue("missing_active_dataset", "bad", "No active GTFS dataset", "Import this source before harmonization.")],
|
|
}
|
|
items = [
|
|
_metric("Agencies", counts.get("agencies", 0), "bad" if not counts.get("agencies", 0) else "good"),
|
|
_metric("Stops", counts.get("stops", 0), "bad" if not counts.get("stops", 0) else "good"),
|
|
_metric("Routes", counts.get("routes", 0), "bad" if not counts.get("routes", 0) else "good"),
|
|
_metric("Trips", counts.get("trips", 0), "bad" if not counts.get("trips", 0) else "good"),
|
|
_metric("Stop times", counts.get("stop_times", 0), "bad" if not counts.get("stop_times", 0) else "good"),
|
|
_metric("Shapes", counts.get("shapes", 0), "warn" if not counts.get("shapes", 0) else "good"),
|
|
]
|
|
missing_coords = _count(session, GtfsStop, dataset.id, (GtfsStop.lat.is_(None) | GtfsStop.lon.is_(None)))
|
|
invalid_coords = _count(
|
|
session,
|
|
GtfsStop,
|
|
dataset.id,
|
|
(GtfsStop.lat < -90) | (GtfsStop.lat > 90) | (GtfsStop.lon < -180) | (GtfsStop.lon > 180),
|
|
)
|
|
routes_without_trips = _routes_without_trips(session, dataset.id)
|
|
trips_without_stop_times = _trips_without_stop_times(session, dataset.id)
|
|
stop_times_without_seconds = _stop_times_without_seconds(session, dataset.id)
|
|
route_geometry_missing = _count(session, GtfsRoute, dataset.id, GtfsRoute.geometry_geojson.is_(None))
|
|
canonical_links = _count(session, CanonicalStopLink, dataset.id, CanonicalStopLink.object_type == "gtfs_stop")
|
|
match_counts = counts.get("match_counts", {}) if isinstance(counts.get("match_counts"), dict) else {}
|
|
|
|
items.extend(
|
|
[
|
|
_metric("Stops missing coordinates", missing_coords, "bad" if missing_coords else "good"),
|
|
_metric("Stops with invalid coordinates", invalid_coords, "bad" if invalid_coords else "good"),
|
|
_metric("Routes without trips", routes_without_trips, "bad" if routes_without_trips else "good"),
|
|
_metric("Trips without stop_times", trips_without_stop_times, "bad" if trips_without_stop_times else "good"),
|
|
_metric("Stop times without parsed seconds", stop_times_without_seconds, "warn" if stop_times_without_seconds else "good"),
|
|
_metric("Routes without geometry", route_geometry_missing, "warn" if route_geometry_missing else "good"),
|
|
_metric("Canonical stop links", canonical_links, "warn" if counts.get("stops", 0) and canonical_links == 0 else "good"),
|
|
_metric("Route matches", counts.get("matches", 0), "warn" if counts.get("routes", 0) and not counts.get("matches", 0) else "good"),
|
|
]
|
|
)
|
|
issues: list[dict[str, str]] = []
|
|
if counts.get("missing_sidecar"):
|
|
issues.append(_issue("missing_sidecar", "bad", "GTFS sidecar is missing", "Queue a recovery import for this dataset."))
|
|
for key, label in [
|
|
("agencies", "No agencies imported"),
|
|
("stops", "No stops imported"),
|
|
("routes", "No routes imported"),
|
|
("trips", "No trips imported"),
|
|
("stop_times", "No stop_times imported"),
|
|
]:
|
|
if not counts.get(key, 0):
|
|
issues.append(_issue(f"missing_{key}", "bad", label, "Required GTFS content is absent or failed to import."))
|
|
if missing_coords:
|
|
issues.append(_issue("missing_stop_coordinates", "bad", f"{missing_coords:,} stops have no coordinates", "Stop coordinates are required for deduplication and routing access."))
|
|
if invalid_coords:
|
|
issues.append(_issue("invalid_stop_coordinates", "bad", f"{invalid_coords:,} stops have invalid coordinates", "Fix or exclude invalid stop coordinates before publication."))
|
|
if routes_without_trips:
|
|
issues.append(_issue("routes_without_trips", "warn", f"{routes_without_trips:,} routes have no trips", "These routes cannot contribute timetable service."))
|
|
if trips_without_stop_times:
|
|
issues.append(_issue("trips_without_stop_times", "bad", f"{trips_without_stop_times:,} trips have no stop_times", "These trips cannot be routed."))
|
|
if route_geometry_missing:
|
|
issues.append(_issue("route_geometry_missing", "warn", f"{route_geometry_missing:,} routes have no geometry", "Use GTFS shapes, route-layer matching, or stop-by-stop fallback."))
|
|
if counts.get("routes", 0) and not counts.get("shapes", 0):
|
|
issues.append(_issue("missing_shapes", "warn", "No GTFS shapes imported", "OSM route matching or generated geometry will be needed."))
|
|
if counts.get("routes", 0) and not match_counts:
|
|
issues.append(_issue("no_route_matching", "warn", "No route-match rows", "Run route matching before route-layer publication QA."))
|
|
return {
|
|
"status": _qa_status(issues, dataset),
|
|
"items": items,
|
|
"issues": issues,
|
|
}
|
|
|
|
|
|
def _service_horizon(session: Session, dataset: Dataset | None) -> dict[str, Any]:
|
|
if dataset is None:
|
|
return {"start_date": None, "end_date": None, "days_until_end": None, "items": [], "issues": []}
|
|
cal_min, cal_max = session.execute(
|
|
select(func.min(GtfsCalendar.start_date), func.max(GtfsCalendar.end_date)).where(GtfsCalendar.dataset_id == dataset.id)
|
|
).one()
|
|
date_min, date_max = session.execute(
|
|
select(func.min(GtfsCalendarDate.date), func.max(GtfsCalendarDate.date)).where(GtfsCalendarDate.dataset_id == dataset.id)
|
|
).one()
|
|
start_int = _min_int(cal_min, date_min)
|
|
end_int = _max_int(cal_max, date_max)
|
|
start_date = _gtfs_date(start_int)
|
|
end_date = _gtfs_date(end_int)
|
|
today = datetime.now(timezone.utc).date()
|
|
days_until_end = None if end_date is None else (end_date - today).days
|
|
issues: list[dict[str, str]] = []
|
|
if end_date is None:
|
|
issues.append(_issue("service_horizon_missing", "bad", "No service calendar horizon", "calendar.txt or calendar_dates.txt is required for reliable routing."))
|
|
elif days_until_end is not None and days_until_end < 0:
|
|
issues.append(_issue("service_horizon_expired", "bad", f"Service expired {abs(days_until_end):,} days ago", "Update or exclude this feed."))
|
|
elif days_until_end is not None and days_until_end < 30:
|
|
issues.append(_issue("service_horizon_short", "warn", f"Service ends in {days_until_end:,} days", "Update cadence is too close for publication confidence."))
|
|
return {
|
|
"start_date": None if start_date is None else start_date.isoformat(),
|
|
"end_date": None if end_date is None else end_date.isoformat(),
|
|
"days_until_end": days_until_end,
|
|
"items": [
|
|
_metric("Service starts", start_date.isoformat() if start_date else "n/a", "info"),
|
|
_metric("Service ends", end_date.isoformat() if end_date else "n/a", "bad" if end_date is None or (days_until_end is not None and days_until_end < 0) else "warn" if days_until_end is not None and days_until_end < 30 else "good"),
|
|
],
|
|
"issues": issues,
|
|
}
|
|
|
|
|
|
def _overlap_summary(session: Session, dataset: Dataset | None) -> dict[str, Any]:
|
|
if dataset is None:
|
|
return {"items": [], "issues": []}
|
|
route_key_overlaps = _shared_route_keys(session, dataset.id)
|
|
canonical_stop_overlaps = _shared_canonical_stops(session, dataset.id)
|
|
issues: list[dict[str, str]] = []
|
|
if route_key_overlaps:
|
|
issues.append(_issue("shared_route_keys", "warn", f"{route_key_overlaps:,} route keys also exist in another active feed", "Deduplicate or rank source authority for overlapping routes."))
|
|
if canonical_stop_overlaps:
|
|
issues.append(_issue("shared_canonical_stops", "warn", f"{canonical_stop_overlaps:,} canonical stops are shared with another active feed", "This is useful linking evidence, but conflicts need review."))
|
|
return {
|
|
"items": [
|
|
_metric("Shared route keys", route_key_overlaps, "warn" if route_key_overlaps else "good"),
|
|
_metric("Shared canonical stops", canonical_stop_overlaps, "warn" if canonical_stop_overlaps else "good"),
|
|
],
|
|
"issues": issues,
|
|
}
|
|
|
|
|
|
def _license_payload(source: Source) -> dict[str, Any]:
|
|
text = (source.license or "").strip()
|
|
review = _qa_review_payload(source.notes)
|
|
flags = {key: review.get(key, "unknown") for key in GTFS_LICENSE_FLAG_KEYS}
|
|
missing_required = any(flags[key] != "yes" for key in ("can_import", "can_derive", "can_redistribute"))
|
|
blocked = (
|
|
any(flags[key] == "no" for key in ("can_import", "can_derive", "can_redistribute"))
|
|
or flags["commercial_restrictions"] == "yes"
|
|
)
|
|
unknown = not text or "unknown" in text.lower()
|
|
if blocked:
|
|
redistribution_status = "blocked"
|
|
tone = "bad"
|
|
elif not missing_required and not unknown:
|
|
redistribution_status = "allowed"
|
|
tone = "good"
|
|
elif unknown:
|
|
redistribution_status = "unknown"
|
|
tone = "warn"
|
|
else:
|
|
redistribution_status = "review_required"
|
|
tone = "warn"
|
|
return {
|
|
"label": text or "unknown",
|
|
"redistribution_status": redistribution_status,
|
|
"tone": tone,
|
|
"flags": flags,
|
|
"authority_level": review.get("authority_level", "unknown"),
|
|
}
|
|
|
|
|
|
def _license_issues(source: Source) -> list[dict[str, str]]:
|
|
payload = _license_payload(source)
|
|
review = _qa_review_payload(source.notes)
|
|
flags = payload["flags"]
|
|
issues: list[dict[str, str]] = []
|
|
if review["status"] in {"blocked", "rejected"}:
|
|
issues.append(_issue("manual_review_blocked", "bad", "Feed review blocks publication", f"Reviewer decision is {review['status']}."))
|
|
elif review["status"] == "needs_review":
|
|
issues.append(_issue("manual_review_needed", "warn", "Feed needs manual review", review["note"] or "Review decision is marked as needs review."))
|
|
if payload["redistribution_status"] == "blocked":
|
|
issues.append(_issue("license_blocked", "bad", "License blocks publication", "Import, derivation, redistribution, or commercial-use review explicitly blocks publication."))
|
|
elif payload["redistribution_status"] == "unknown":
|
|
issues.append(_issue("license_unknown", "warn", "License/redistribution status is unknown", "Publication needs explicit import, derivation, redistribution, and attribution flags."))
|
|
elif payload["redistribution_status"] == "review_required":
|
|
missing = [
|
|
label
|
|
for key, label in [
|
|
("can_import", "import"),
|
|
("can_derive", "derivation"),
|
|
("can_redistribute", "redistribution"),
|
|
]
|
|
if flags.get(key) != "yes"
|
|
]
|
|
issues.append(_issue("license_flags_incomplete", "warn", "License decision is incomplete", f"Missing explicit approval for: {', '.join(missing)}."))
|
|
if payload.get("authority_level") == "unknown":
|
|
issues.append(_issue("authority_unknown", "warn", "Source authority ranking is unknown", "Rank the source as national official, regional authority, operator, aggregator, mirror, or secondary discovery."))
|
|
return issues
|
|
|
|
|
|
def _qa_review_payload(notes: str | None) -> dict[str, Any]:
|
|
default = _default_qa_review_payload()
|
|
if not notes:
|
|
return default
|
|
for line in str(notes).splitlines():
|
|
if not line.startswith(GTFS_QA_NOTE_PREFIX):
|
|
continue
|
|
raw = line[len(GTFS_QA_NOTE_PREFIX) :].strip()
|
|
if raw.startswith("{"):
|
|
try:
|
|
parsed = json.loads(raw)
|
|
except json.JSONDecodeError:
|
|
parsed = {}
|
|
if isinstance(parsed, dict):
|
|
return _normalise_qa_review_payload({**default, **parsed})
|
|
payload: dict[str, str] = {}
|
|
for part in raw.split(";"):
|
|
if "=" not in part:
|
|
continue
|
|
key, value = part.split("=", 1)
|
|
payload[key.strip()] = value.strip()
|
|
return _normalise_qa_review_payload({**default, **payload})
|
|
return default
|
|
|
|
|
|
def _default_qa_review_payload() -> dict[str, Any]:
|
|
return {
|
|
"status": "unreviewed",
|
|
"note": "",
|
|
"updated_at": None,
|
|
"can_import": "unknown",
|
|
"can_derive": "unknown",
|
|
"can_redistribute": "unknown",
|
|
"requires_attribution": "unknown",
|
|
"commercial_restrictions": "unknown",
|
|
"authority_level": "unknown",
|
|
}
|
|
|
|
|
|
def _normalise_qa_review_payload(payload: dict[str, Any]) -> dict[str, Any]:
|
|
result = _default_qa_review_payload()
|
|
result.update(
|
|
{
|
|
"status": _choice(payload.get("status"), {"unreviewed", "approved", "needs_review", "blocked", "rejected"}, "unreviewed"),
|
|
"note": str(payload.get("note") or ""),
|
|
"updated_at": payload.get("updated_at"),
|
|
"authority_level": _choice(payload.get("authority_level"), GTFS_AUTHORITY_LEVELS, "unknown"),
|
|
}
|
|
)
|
|
for key in GTFS_LICENSE_FLAG_KEYS:
|
|
result[key] = _choice(payload.get(key), {"unknown", "yes", "no"}, "unknown")
|
|
return result
|
|
|
|
|
|
def _choice(value: Any, allowed: set[str], default: str) -> str:
|
|
text = str(value or "").strip()
|
|
return text if text in allowed else default
|
|
|
|
|
|
def _routes_without_trips(session: Session, dataset_id: int) -> int:
|
|
trip_exists = select(GtfsTrip.id).where(GtfsTrip.dataset_id == dataset_id, GtfsTrip.route_id == GtfsRoute.route_id).exists()
|
|
return int(session.scalar(select(func.count()).select_from(GtfsRoute).where(GtfsRoute.dataset_id == dataset_id, ~trip_exists)) or 0)
|
|
|
|
|
|
def _trips_without_stop_times(session: Session, dataset_id: int) -> int:
|
|
stop_time_exists = select(GtfsStopTime.id).where(GtfsStopTime.dataset_id == dataset_id, GtfsStopTime.trip_id == GtfsTrip.trip_id).exists()
|
|
return int(session.scalar(select(func.count()).select_from(GtfsTrip).where(GtfsTrip.dataset_id == dataset_id, ~stop_time_exists)) or 0)
|
|
|
|
|
|
def _stop_times_without_seconds(session: Session, dataset_id: int) -> int:
|
|
return int(
|
|
session.scalar(
|
|
select(func.count())
|
|
.select_from(GtfsStopTime)
|
|
.where(GtfsStopTime.dataset_id == dataset_id, GtfsStopTime.arrival_seconds.is_(None), GtfsStopTime.departure_seconds.is_(None))
|
|
)
|
|
or 0
|
|
)
|
|
|
|
|
|
def _shared_route_keys(session: Session, dataset_id: int) -> int:
|
|
current = aliased(GtfsRoute)
|
|
other = aliased(GtfsRoute)
|
|
other_dataset = aliased(Dataset)
|
|
return int(
|
|
session.scalar(
|
|
select(func.count(func.distinct(current.route_key)))
|
|
.select_from(current)
|
|
.join(other, and_(other.route_key == current.route_key, other.dataset_id != current.dataset_id))
|
|
.join(other_dataset, other_dataset.id == other.dataset_id)
|
|
.where(
|
|
current.dataset_id == dataset_id,
|
|
current.route_key.is_not(None),
|
|
current.route_key != "",
|
|
other_dataset.kind == "gtfs",
|
|
other_dataset.is_active.is_(True),
|
|
)
|
|
)
|
|
or 0
|
|
)
|
|
|
|
|
|
def _shared_canonical_stops(session: Session, dataset_id: int) -> int:
|
|
current = aliased(CanonicalStopLink)
|
|
other = aliased(CanonicalStopLink)
|
|
other_dataset = aliased(Dataset)
|
|
return int(
|
|
session.scalar(
|
|
select(func.count(func.distinct(current.canonical_stop_id)))
|
|
.select_from(current)
|
|
.join(other, and_(other.canonical_stop_id == current.canonical_stop_id, other.dataset_id != current.dataset_id))
|
|
.join(other_dataset, other_dataset.id == other.dataset_id)
|
|
.where(
|
|
current.dataset_id == dataset_id,
|
|
current.object_type == "gtfs_stop",
|
|
other.object_type == "gtfs_stop",
|
|
other_dataset.kind == "gtfs",
|
|
other_dataset.is_active.is_(True),
|
|
)
|
|
)
|
|
or 0
|
|
)
|
|
|
|
|
|
def _count(session: Session, model: Any, dataset_id: int, *criteria: Any) -> int:
|
|
stmt = select(func.count()).select_from(model).where(model.dataset_id == dataset_id)
|
|
if criteria:
|
|
stmt = stmt.where(*criteria)
|
|
return int(session.scalar(stmt) or 0)
|
|
|
|
|
|
def _metric(label: str, value: Any, tone: str = "info", description: str = "") -> dict[str, Any]:
|
|
return {"label": label, "value": value, "tone": tone, "description": description}
|
|
|
|
|
|
def _issue(issue_id: str, severity: str, title: str, detail: str) -> dict[str, str]:
|
|
return {"id": issue_id, "severity": severity, "title": title, "detail": detail}
|
|
|
|
|
|
def _qa_status(issues: list[dict[str, str]], dataset: Dataset | None) -> str:
|
|
if dataset is None or any(issue.get("severity") == "bad" for issue in issues):
|
|
return "blocked"
|
|
if any(issue.get("severity") == "warn" for issue in issues):
|
|
return "needs_review"
|
|
return "ready"
|
|
|
|
|
|
def _feed_sections(feed: dict[str, Any]) -> list[dict[str, Any]]:
|
|
license_payload = feed["license"]
|
|
flags = license_payload.get("flags", {})
|
|
return [
|
|
{"id": "validation", "title": "GTFS Validation", "items": feed["validation"]["items"]},
|
|
{"id": "service", "title": "Service Horizon", "items": feed["service"]["items"]},
|
|
{"id": "overlap", "title": "Overlap and Deduplication", "items": feed["overlap"]["items"]},
|
|
{
|
|
"id": "license",
|
|
"title": "License and Authority",
|
|
"items": [
|
|
_metric("Redistribution", license_payload["redistribution_status"], license_payload["tone"]),
|
|
_metric("License", license_payload["label"], license_payload["tone"]),
|
|
_metric("Can import", flags.get("can_import", "unknown"), "good" if flags.get("can_import") == "yes" else "warn"),
|
|
_metric("Can derive", flags.get("can_derive", "unknown"), "good" if flags.get("can_derive") == "yes" else "warn"),
|
|
_metric("Can redistribute", flags.get("can_redistribute", "unknown"), "good" if flags.get("can_redistribute") == "yes" else "warn"),
|
|
_metric("Attribution", flags.get("requires_attribution", "unknown"), "info"),
|
|
_metric("Commercial restrictions", flags.get("commercial_restrictions", "unknown"), "bad" if flags.get("commercial_restrictions") == "yes" else "info"),
|
|
_metric("Authority", license_payload.get("authority_level", "unknown"), "warn" if license_payload.get("authority_level") == "unknown" else "good"),
|
|
],
|
|
},
|
|
]
|
|
|
|
|
|
def _gtfs_date(value: int | None) -> date | None:
|
|
if value is None:
|
|
return None
|
|
try:
|
|
return datetime.strptime(str(int(value)), "%Y%m%d").date()
|
|
except ValueError:
|
|
return None
|
|
|
|
|
|
def _min_int(*values: int | None) -> int | None:
|
|
clean = [int(value) for value in values if value is not None]
|
|
return min(clean) if clean else None
|
|
|
|
|
|
def _max_int(*values: int | None) -> int | None:
|
|
clean = [int(value) for value in values if value is not None]
|
|
return max(clean) if clean else None
|
|
|
|
|
|
def _optional_int(value: object) -> int | None:
|
|
try:
|
|
return None if value is None else int(value)
|
|
except (TypeError, ValueError):
|
|
return None
|
|
|
|
|
|
def _json_object(value: str | None) -> dict[str, Any]:
|
|
if not value:
|
|
return {}
|
|
try:
|
|
payload = json.loads(value)
|
|
except json.JSONDecodeError:
|
|
return {}
|
|
return payload if isinstance(payload, dict) else {}
|
|
|
|
|
|
def _iso(value: datetime | None) -> str | None:
|
|
return None if value is None else value.isoformat()
|