14 Commits
Author SHA1 Message Date
zemion 32cac8835b Release v0.1.16
Module Package Release / publish-packages (push) Successful in 11s
2026-08-05 19:52:00 +02:00
zemion c69d68f1da Release v0.1.15
Module Package Release / publish-packages (push) Successful in 11s
2026-08-04 15:10:18 +02:00
zemion 1caae6e49e Make package publication retries hash-safe 2026-08-04 14:32:19 +02:00
zemion 84c9bb7711 Harden module package publication 2026-08-04 14:02:39 +02:00
zemion 20146ef8fe Bound connector source previews 2026-08-04 12:05:27 +02:00
zemion c33380b957 Add protected package release workflow 2026-08-04 04:14:03 +02:00
zemion dfa717b9ba Fence governed connector acquisitions 2026-08-03 05:43:54 +02:00
zemion 26f8898d11 Announce shared connector runtime contract 2026-08-02 14:54:56 +02:00
zemion 7be93785a2 feat: declare governed external provider state 2026-08-01 17:48:25 +02:00
zemion 52fe33568c Add governed RSS and Atom connectors 2026-07-31 22:48:07 +02:00
zemion c5a43b3dae feat: acquire immutable sanctions snapshots 2026-07-29 18:46:53 +02:00
zemion 27302f0c39 feat: expose connector datasource origins 2026-07-28 12:43:26 +02:00
zemion ba5ccea5b0 Implement governed tabular source snapshots 2026-07-28 11:13:22 +02:00
zemion dd45d9bd36 Release v0.1.8 2026-07-11 16:49:04 +02:00
32 changed files with 6658 additions and 8 deletions
+270
View File
@@ -0,0 +1,270 @@
name: Module Package Release
on:
push:
tags:
- "v*"
workflow_dispatch:
inputs:
release_tag:
description: Existing protected version tag to publish
required: true
type: string
jobs:
publish-packages:
runs-on: ubuntu-latest
env:
GITEA_REPOSITORY: ${{ gitea.repository }}
steps:
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5
with:
fetch-depth: 0
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065
with:
python-version: "3.12"
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020
with:
node-version: "22"
- name: Select and validate protected release tag
shell: bash
env:
REQUESTED_TAG: ${{ inputs.release_tag }}
TRIGGER_TAG: ${{ gitea.ref_name }}
run: |
set -euo pipefail
tag="${REQUESTED_TAG:-$TRIGGER_TAG}"
case "$tag" in
v[0-9]*.[0-9]*.[0-9]*) ;;
*) echo "Release tag must start with a SemVer-shaped vX.Y.Z value" >&2; exit 1 ;;
esac
git fetch --force origin "refs/tags/$tag:refs/tags/$tag" refs/heads/main:refs/remotes/origin/main
tag_commit="$(git rev-list -n 1 "$tag")"
git merge-base --is-ancestor "$tag_commit" refs/remotes/origin/main || {
echo "Release tag is not contained in main" >&2
exit 1
}
git checkout --detach "$tag"
printf 'RELEASE_TAG=%s\n' "$tag" >> "$GITEA_ENV"
printf 'SOURCE_DATE_EPOCH=%s\n' "$(git show -s --format=%ct HEAD)" >> "$GITEA_ENV"
- name: Validate package versions
run: |
python - <<'PY'
import json
from pathlib import Path
import os
import re
import tomllib
tag = os.environ["RELEASE_TAG"]
expected = tag.removeprefix("v")
project = tomllib.loads(Path("pyproject.toml").read_text(encoding="utf-8"))["project"]
if project.get("version") != expected:
raise SystemExit(f"pyproject version {project.get('version')!r} does not match {tag}")
if re.fullmatch(r"govoplan-[a-z0-9-]+", str(project.get("name", ""))) is None:
raise SystemExit("Python distribution name must use the govoplan-* namespace")
webui = Path("webui/package.json")
if webui.is_file():
package = json.loads(webui.read_text(encoding="utf-8"))
if package.get("version") != expected:
raise SystemExit(f"WebUI version {package.get('version')!r} does not match {tag}")
if re.fullmatch(r"@govoplan/[a-z0-9-]+-webui", str(package.get("name", ""))) is None:
raise SystemExit("WebUI package name must use the @govoplan/*-webui namespace")
release = Path("webui/package.release.json")
if release.is_file():
release_package = json.loads(release.read_text(encoding="utf-8"))
if (
release_package.get("name") != package.get("name")
or release_package.get("version") != expected
):
raise SystemExit("WebUI release package identity does not match package.json and the release tag")
PY
- name: Build immutable package artifacts
shell: bash
run: |
set -euo pipefail
python -m pip install --disable-pip-version-check build==1.5.0 twine==7.0.0
rm -rf dist .package-webui
python -m build --wheel --outdir dist
python -m twine check dist/*.whl
if [[ -f webui/package.json ]]; then
mkdir .package-webui
cp -a webui/. .package-webui/
rm -rf .package-webui/node_modules .package-webui/dist
if [[ -f .package-webui/package.release.json ]]; then
cp .package-webui/package.release.json .package-webui/package.json
fi
node <<'NODE'
const fs = require("node:fs");
const path = ".package-webui/package.json";
const packageJson = JSON.parse(fs.readFileSync(path, "utf8"));
const groups = ["dependencies", "optionalDependencies", "peerDependencies"];
for (const group of groups) {
for (const [name, specifier] of Object.entries(packageJson[group] || {})) {
if (!name.startsWith("@govoplan/")) continue;
if (typeof specifier !== "string") {
throw new Error(`${group}.${name} must use a string version`);
}
const packageSlug = name.slice("@govoplan/".length);
if (!packageSlug.endsWith("-webui")) {
throw new Error(`${group}.${name} is outside the WebUI package namespace`);
}
const repository = `govoplan-${packageSlug.slice(0, -"-webui".length)}`;
const escapedRepository = repository.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
const gitTag = specifier.match(
new RegExp(
`^git\\+(?:ssh://git@|https://)git\\.add-ideas\\.de/(?:GovOPlaN|add-ideas)/${escapedRepository}\\.git#v([0-9]+\\.[0-9]+\\.[0-9]+)$`,
),
);
if (gitTag) {
packageJson[group][name] = gitTag[1];
continue;
}
if (specifier.startsWith("file:") || specifier.startsWith("git+")) {
throw new Error(
`${group}.${name} must resolve to an exact registry version for publication`,
);
}
}
}
delete packageJson.private;
fs.writeFileSync(path, `${JSON.stringify(packageJson, null, 2)}\n`);
NODE
npm pkg delete private --prefix .package-webui
(cd .package-webui && npm pack --ignore-scripts --pack-destination ../dist)
fi
python - <<'PY'
import hashlib
import json
from pathlib import Path
import os
import subprocess
artifacts = []
for path in sorted(Path("dist").iterdir()):
if path.suffix not in {".whl", ".tgz"}:
continue
digest = hashlib.sha256(path.read_bytes()).hexdigest()
artifacts.append({"filename": path.name, "sha256": digest, "size": path.stat().st_size})
payload = {
"schema_version": "1",
"repository": os.environ["GITEA_REPOSITORY"],
"tag": os.environ["RELEASE_TAG"],
"commit": subprocess.check_output(["git", "rev-parse", "HEAD"], text=True).strip(),
"artifacts": artifacts,
}
Path("dist/package-artifacts.json").write_text(
json.dumps(payload, indent=2, sort_keys=True) + "\n",
encoding="utf-8",
)
PY
- name: Retain package hash evidence
uses: actions/upload-artifact@a8a3f3ad30e3422c9c7b888a15615d19a852ae32
with:
name: module-packages-${{ gitea.ref_name }}
path: dist/package-artifacts.json
- name: Check immutable registry state
shell: bash
env:
PACKAGE_TOKEN: ${{ secrets.GOVOPLAN_PACKAGE_TOKEN }}
run: |
set -euo pipefail
test -n "$PACKAGE_TOKEN"
python - <<'PY'
import hashlib
import json
import os
from pathlib import Path
import tomllib
from urllib.error import HTTPError
from urllib.parse import quote
from urllib.request import Request, urlopen
api_root = "https://git.add-ideas.de/api/v1/packages/GovOPlaN"
token = os.environ["PACKAGE_TOKEN"]
def should_publish(kind, name, version, path):
package_url = "/".join(
(api_root, kind, quote(name, safe=""), quote(version, safe=""), "files")
)
request = Request(
package_url,
headers={"Accept": "application/json", "Authorization": f"token {token}"},
)
try:
with urlopen(request, timeout=30) as response:
files = json.load(response)
except HTTPError as exc:
if exc.code == 404:
print(f"{kind} package {name}=={version} is not published yet")
return True
raise
if not isinstance(files, list) or len(files) != 1:
raise SystemExit(
f"immutable {kind} package {name}=={version} has an unexpected file set"
)
expected_sha256 = hashlib.sha256(path.read_bytes()).hexdigest()
if files[0].get("sha256") != expected_sha256:
raise SystemExit(
f"immutable {kind} package {name}=={version} already exists with a different SHA-256"
)
print(f"verified existing {kind} package {name}=={version} ({expected_sha256})")
return False
project = tomllib.loads(Path("pyproject.toml").read_text(encoding="utf-8"))["project"]
wheels = tuple(Path("dist").glob("*.whl"))
if len(wheels) != 1:
raise SystemExit("release build must contain exactly one wheel")
publish_pypi = should_publish(
"pypi", str(project["name"]), str(project["version"]), wheels[0]
)
tarballs = tuple(Path("dist").glob("*.tgz"))
if len(tarballs) > 1:
raise SystemExit("release build must contain at most one npm package")
publish_npm = False
if tarballs:
webui = json.loads(
Path(".package-webui/package.json").read_text(encoding="utf-8")
)
publish_npm = should_publish(
"npm", str(webui["name"]), str(webui["version"]), tarballs[0]
)
with Path(os.environ["GITEA_ENV"]).open("a", encoding="utf-8") as env_file:
env_file.write(f"PUBLISH_PYPI={int(publish_pypi)}\n")
env_file.write(f"PUBLISH_NPM={int(publish_npm)}\n")
PY
- name: Publish wheel and WebUI package
shell: bash
env:
PACKAGE_USERNAME: ${{ secrets.GOVOPLAN_PACKAGE_USERNAME }}
PACKAGE_TOKEN: ${{ secrets.GOVOPLAN_PACKAGE_TOKEN }}
run: |
set -euo pipefail
test -n "$PACKAGE_USERNAME"
test -n "$PACKAGE_TOKEN"
if [[ "$PUBLISH_PYPI" == 1 ]]; then
TWINE_USERNAME="$PACKAGE_USERNAME" TWINE_PASSWORD="$PACKAGE_TOKEN" \
python -m twine upload --non-interactive \
--repository-url https://git.add-ideas.de/api/packages/GovOPlaN/pypi \
dist/*.whl
else
echo "Exact wheel is already present; skipping immutable retry."
fi
shopt -s nullglob
webui_packages=(dist/*.tgz)
if (( ${#webui_packages[@]} )) && [[ "$PUBLISH_NPM" == 1 ]]; then
npmrc="$(mktemp)"
trap 'rm -f "$npmrc"' EXIT
chmod 600 "$npmrc"
printf '%s\n' \
'@govoplan:registry=https://git.add-ideas.de/api/packages/GovOPlaN/npm/' \
"//git.add-ideas.de/api/packages/GovOPlaN/npm/:_authToken=$PACKAGE_TOKEN" \
> "$npmrc"
NPM_CONFIG_USERCONFIG="$npmrc" npm publish "./${webui_packages[0]}" \
--ignore-scripts --access public \
--registry https://git.add-ideas.de/api/packages/GovOPlaN/npm/
elif (( ${#webui_packages[@]} )); then
echo "Exact WebUI package is already present; skipping immutable retry."
fi
+16
View File
@@ -0,0 +1,16 @@
# GovOPlaN Connectors Codex Guide
## Scope
This repository owns reusable external connection profiles, protocol adapters, governed snapshots, and connector capability contracts.
## Documentation Contract
- Treat documentation as part of every behavior change. Update this module's manifest-driven `DocumentationTopic` contributions for affected user and administrator behavior.
- Keep feature content here; `govoplan-docs` projects it without importing Connectors internals.
- Maintain a static user/admin baseline and run `/mnt/DATA/git/govoplan/tools/checks/check-manifest-shapes.py` after behavior or manifest changes.
## Boundaries
- Domain modules own business semantics; connectors own transport, credentials, synchronization, and diagnostics.
- Keep optional adapters behind capabilities and enforce egress and peer-validation policy.
+47 -2
View File
@@ -1,13 +1,58 @@
# govoplan-connectors # govoplan-connectors
`govoplan-connectors` will own integration catalogues and generic external <!-- govoplan-repository-type:start -->
system connection patterns for GovOPlaN. **Repository type:** connector (connector-hub).
<!-- govoplan-repository-type:end -->
`govoplan-connectors` owns integration catalogues and generic external system
connection patterns for GovOPlaN.
The module should make external systems discoverable, testable, and usable The module should make external systems discoverable, testable, and usable
without taking ownership of their business semantics. Domain-specific modules without taking ownership of their business semantics. Domain-specific modules
remain responsible for case, file, workflow, payment, mail, identity, document, remain responsible for case, file, workflow, payment, mail, identity, document,
or reporting behavior. or reporting behavior.
## Executable First Slice
The first executable connector capability provides tenant-isolated tabular
origins. Operators can import bounded JSON or CSV snapshots, inspect inferred
schemas, and expose immutable source references and content fingerprints
through `connectors.datasource_origins@0.1.0`. Preview reads enforce provider
ceilings for rows, serialized bytes, and elapsed time and report the effective
limits and any truncation as structured diagnostics.
Connectors owns acquisition, connection profiles, credentials, discovery, and
provider health. `govoplan-datasources` registers an origin as a governed live
or cached datasource and owns staging, materializations, frozen states, and
consumer access. Dataflow consumes that Datasources contract and never imports
connector implementations or stores connector credentials.
Each origin declares whether it is live, cached, file-backed, or static, its
structured health state, and which projection, filter, aggregation, sorting,
and pagination operations it can push down. The immutable snapshot provider
currently supports projection and pagination only; consumers must keep other
operations in Dataflow rather than assuming transport-side execution.
Database, REST/HTTP, directory, managed-file, and warehouse providers can
implement the same origin contract without changing Datasources or Dataflow.
Governed sanctions and feed snapshot acquisitions use Core recovery operations.
The source revision/cursor, redacted dry-run decision, canonical request digest,
and distributed lease are durable before network I/O. Immutable snapshot rows
and the terminal recovery checkpoint commit atomically, and an
`Idempotency-Key` replays the committed result without contacting the provider.
Current acquisition transports are read-only. The exported external-mutation
recovery contract requires stable idempotency, provider verification, and
operator reconciliation, but no connector currently claims a production write
or delete path.
Development:
```bash
/mnt/DATA/git/govoplan/.venv/bin/python -m pip install -e .
/mnt/DATA/git/govoplan/.venv/bin/python -m unittest discover -s tests
```
See: See:
- [Connector concept](docs/CONCEPT.md) - [Connector concept](docs/CONCEPT.md)
+34
View File
@@ -11,6 +11,13 @@ results, credential references, and generic integration events. Protocol-heavy
or domain-heavy integrations may live in dedicated modules once their scope is or domain-heavy integrations may live in dedicated modules once their scope is
clear. clear.
Connector capability is not source ownership. Each configured binding also
declares whether GovOPlaN is authoritative, the external system is
authoritative, GovOPlaN keeps a mirror, both sides use governed synchronization,
GovOPlaN supplies only a governance overlay, or the object is link-only. The
same connector may be configured differently by tenant, service, object type,
or field group.
Detailed follow-up documents: Detailed follow-up documents:
- [Public-sector integration catalogue](PUBLIC_SECTOR_INTEGRATION_CATALOGUE.md) - [Public-sector integration catalogue](PUBLIC_SECTOR_INTEGRATION_CATALOGUE.md)
@@ -18,6 +25,20 @@ Detailed follow-up documents:
- [OpenProject connector concept](OPENPROJECT_CONNECTOR.md) - [OpenProject connector concept](OPENPROJECT_CONNECTOR.md)
- [OpenDesk integration map](OPENDESK_INTEGRATION_MAP.md) - [OpenDesk integration map](OPENDESK_INTEGRATION_MAP.md)
## Shared runtime boundary
Connector transports use the versioned Core runtime contract for bounded dry
runs and diagnostics. Connectors owns endpoint discovery, authentication
hand-off, protocol reads, retry/backoff, and source health. A consuming module
owns domain mappings and mutations: Addresses, for example, owns contact and
vCard semantics even when Connectors supplies reusable LDAP, CardDAV, Exchange,
or Google transport patterns.
Dry runs carry an immutable input hash, source revision and fingerprint,
redacted effects, diagnostics, truncation state, and an apply token. Apply must
reject a changed input or source revision. URLs and diagnostics never contain
credential material; they retain only credential-envelope references.
## Ownership ## Ownership
The module owns: The module owns:
@@ -29,9 +50,13 @@ The module owns:
- connector health status and last-test evidence - connector health status and last-test evidence
- operator-visible integration inventory - operator-visible integration inventory
- cross-module discovery of available external capabilities - cross-module discovery of available external capabilities
- supported integration maturity, source-authority modes, operation limits,
and effect/reconciliation behavior for each connector type
The module does not own: The module does not own:
- governed datasource identity, staging, materializations, frozen states, or
consumer read semantics, owned by `govoplan-datasources`
- file storage semantics, owned by files/DMS - file storage semantics, owned by files/DMS
- identity provisioning semantics, owned by IDM/access - identity provisioning semantics, owned by IDM/access
- mail/calendar semantics, owned by mail/calendar - mail/calendar semantics, owned by mail/calendar
@@ -59,6 +84,9 @@ The module should integrate through:
- module manifest metadata, route factories, permissions, and migrations - module manifest metadata, route factories, permissions, and migrations
- a connector catalogue API for listing available connector types - a connector catalogue API for listing available connector types
- a connection profile API with secret references, not plaintext secrets - a connection profile API with secret references, not plaintext secrets
- a provider declaration that composes authority mode, maturity, supported
operations, revisions/freshness, health, limits, idempotency, conflicts,
evidence, and reconciliation behavior
- capability declarations such as `connectors.catalog`, - capability declarations such as `connectors.catalog`,
`connectors.profileTester`, and `connectors.health` `connectors.profileTester`, and `connectors.health`
- events such as `connector.profile_created`, `connector.test_succeeded`, - events such as `connector.profile_created`, `connector.test_succeeded`,
@@ -69,6 +97,11 @@ Domain modules should ask whether a connector capability exists and request a
profile/test result through core-mediated capabilities. They must not import profile/test result through core-mediated capabilities. They must not import
connector implementation modules directly. connector implementation modules directly.
Data-oriented consumers use a two-layer path: Connectors publishes a
provider-specific datasource origin, then Datasources registers and governs it.
Dataflow, Workflow, Reporting, and other consumers use Datasources rather than
calling the connector origin directly.
## Reference Journeys ## Reference Journeys
### OpenProject Connector First ### OpenProject Connector First
@@ -106,6 +139,7 @@ The first implementation should provide:
- WebUI catalogue and profile pages - WebUI catalogue and profile pages
- configuration-package fragment support - configuration-package fragment support
- generic external-reference DTOs - generic external-reference DTOs
- source-authority binding and provider-operation metadata
- health summary provider - health summary provider
## Permissions ## Permissions
+68 -6
View File
@@ -14,6 +14,22 @@ predictable and avoids hidden module imports.
- `bidirectional`: GovOPlaN supports both directions with conflict detection and - `bidirectional`: GovOPlaN supports both directions with conflict detection and
reconciliation rules. reconciliation rules.
Direction describes transport. Every binding also needs a source-authority
mode:
- `native_authoritative`
- `external_authoritative`
- `external_mirror`
- `governed_sync`
- `governance_overlay`
- `linked_reference`
The authority mode and the connector's integration maturity are orthogonal. A
bidirectional connector may be configured as an external mirror, and a native
GovOPlaN object may publish to an external target without transferring
authority. The effective binding must identify its scope and provenance rather
than relying on a profile-wide `sync` boolean.
## Source Data Lifecycle ## Source Data Lifecycle
Connector profiles have operational states, while individual external records Connector profiles have operational states, while individual external records
@@ -88,10 +104,12 @@ this lifecycle when a connector publishes status.
2. Fetch only the minimal remote data required for the declared use case. 2. Fetch only the minimal remote data required for the declared use case.
3. Normalize into a connector-owned staging payload. 3. Normalize into a connector-owned staging payload.
4. Validate shape, required fields, and source trust level. 4. Validate shape, required fields, and source trust level.
5. Emit a core-mediated event such as `connector.record_discovered`. 5. Publish data-shaped inputs as versioned datasource origins.
6. Let domain modules claim or transform staged data through capabilities, not 6. Let Datasources register live/cached origins or stage immutable snapshots.
imports. 7. Let domain modules consume governed datasource references through
7. Store external references with source system, object type, object id, version capabilities, not imports.
8. Emit a core-mediated event such as `connector.record_discovered`.
9. Store external references with source system, object type, object id, version
or ETag, and last-seen timestamp. or ETag, and last-seen timestamp.
## Publish Flow ## Publish Flow
@@ -101,8 +119,11 @@ this lifecycle when a connector publishes status.
shape. shape.
3. Connector sends the remote request. 3. Connector sends the remote request.
4. Connector stores the remote id, version/ETag, and response diagnostics. 4. Connector stores the remote id, version/ETag, and response diagnostics.
5. Connector emits `connector.record_published` or `connector.publish_failed`. 5. A timeout or lost acknowledgement after dispatch becomes outcome-unknown,
6. Domain module stores only the external-reference DTO and any domain result. not an ordinary failure or permission to duplicate the command.
6. Connector emits a confirmed, retryable, outcome-unknown, reconciled, or
corrected result event.
7. Domain module stores only the external-reference DTO and any domain result.
## Reconciliation ## Reconciliation
@@ -114,6 +135,45 @@ Every connector that writes to an external system needs a reconciliation story:
- retry policy for temporary failures - retry policy for temporary failures
- explicit operator action for destructive overwrite or deletion - explicit operator action for destructive overwrite or deletion
- audit trace from GovOPlaN record to external request and response summary - audit trace from GovOPlaN record to external request and response summary
- explicit requested, approved, dispatched, possibly-executed, confirmed, and
reconciled/corrected effect states
## Durable recovery operations
Connectors declares two recovery classes. A read-only acquisition into an
immutable snapshot is `atomic`: the source revision or conditional cursor,
redacted dry-run decision, canonical request digest, and distributed
tenant/provider lease are durable before the fetch. The acquired domain rows
and terminal Core recovery checkpoint commit in one PostgreSQL transaction. A
caller-supplied `Idempotency-Key` replays that committed result without a second
provider request. A failed or stale transaction has no remote mutation and may
be repeated only as a new deliberate acquisition.
An external create, update, publish, or delete is `forward_recovery`. It must
start through the connector mutation recovery contract with a stable
idempotency key, SHA-256 request digest, source revision/cursor, and dry-run
evidence. Definitive rejection is terminal. A timeout or lost acknowledgement
after dispatch is `outcome_unknown` and blocks replay until the owning connector
verifies provider state. The contract and conformance tests exist; no current
connector advertises a production external mutation, so write/delete adoption
remains explicitly planned rather than implied.
## Provider Declaration
An executable connector type should publish machine-readable metadata for:
- owned object and field groups, plus supported authority modes;
- supported discovery, link, search, read, publish, synchronize, migrate, and
replacement maturity;
- read/write/delete/preview/dry-run operations and bounded response limits;
- revision/concurrency token, freshness, health, timeout, retry, and conflict
semantics;
- idempotency and outcome-unknown handling;
- evidence, rollback/compensation, correction, and reconciliation paths;
- classification, purpose, retention, secret, degraded, and outage behavior.
This declaration composes Core contracts. It does not move protocol behavior
or domain semantics into Core or Connectors.
## Capability Boundary ## Capability Boundary
@@ -124,6 +184,7 @@ should ask core for capabilities such as:
- `connectors.profileTester` - `connectors.profileTester`
- `connectors.health` - `connectors.health`
- `connectors.externalReferences` - `connectors.externalReferences`
- `connectors.datasourceOrigins`
- `connectors.sourceConsumer` - `connectors.sourceConsumer`
- `connectors.sourcePublisher` - `connectors.sourcePublisher`
@@ -155,5 +216,6 @@ Before shipping an executable connector type:
- Add unavailable-optional-module tests for every consuming domain module. - Add unavailable-optional-module tests for every consuming domain module.
- Add profile test and health status fixtures. - Add profile test and health status fixtures.
- Add external-reference DTO tests. - Add external-reference DTO tests.
- Add source-authority and provider-declaration validation tests.
- Add lifecycle transition tests for pause, retry, retirement, and uninstall - Add lifecycle transition tests for pause, retry, retirement, and uninstall
guard behavior. guard behavior.
+25
View File
@@ -0,0 +1,25 @@
[build-system]
requires = ["setuptools>=69", "wheel"]
build-backend = "setuptools.build_meta"
[project]
name = "govoplan-connectors"
version = "0.1.16"
description = "Governed connector catalogue and tabular source capabilities for GovOPlaN."
readme = "README.md"
requires-python = ">=3.12"
license = "AGPL-3.0-or-later"
authors = [{ name = "GovOPlaN" }]
dependencies = [
"defusedxml>=0.7,<1",
"govoplan-core>=0.1.16",
]
[tool.setuptools.packages.find]
where = ["src"]
[tool.setuptools.package-data]
govoplan_connectors = ["py.typed"]
[project.entry-points."govoplan.modules"]
connectors = "govoplan_connectors.backend.manifest:get_manifest"
+3
View File
@@ -0,0 +1,3 @@
"""GovOPlaN Connectors module."""
__all__: list[str] = []
@@ -0,0 +1 @@
"""Connector backend package."""
@@ -0,0 +1,147 @@
from __future__ import annotations
from govoplan_core.core.datasources import (
DatasourceAccessError,
DatasourceField,
DatasourceNotFoundError,
DatasourceOrigin,
DatasourceOriginReadRequest,
DatasourceOriginReadResult,
DatasourceUnavailableError,
DatasourceValidationError,
)
from govoplan_core.core.tabular_sources import (
TabularReadRequest,
TabularSource,
TabularSourceAccessError,
TabularSourceError,
TabularSourceNotFoundError,
TabularSourceUnavailableError,
TabularSourceValidationError,
)
from govoplan_connectors.backend.tabular_sources import SqlTabularSourceProvider
class ConnectorDatasourceOriginProvider:
"""Expose connector-owned sources through the Datasources origin contract."""
def __init__(self, provider: SqlTabularSourceProvider | None = None) -> None:
self._provider = provider or SqlTabularSourceProvider()
def list_origins(
self,
session: object,
principal: object,
*,
query: str = "",
limit: int = 100,
):
try:
rows = self._provider.list_sources(
session,
principal,
query=query,
limit=limit,
)
except TabularSourceError as exc:
raise _datasource_error(exc) from exc
return tuple(_origin(source) for source in rows)
def get_origin(
self,
session: object,
principal: object,
*,
origin_ref: str,
) -> DatasourceOrigin | None:
try:
source = self._provider.get_source(
session,
principal,
source_ref=origin_ref,
)
except TabularSourceError as exc:
raise _datasource_error(exc) from exc
return _origin(source) if source is not None else None
def read_origin(
self,
session: object,
principal: object,
*,
request: DatasourceOriginReadRequest,
) -> DatasourceOriginReadResult:
try:
result = self._provider.read_source(
session,
principal,
request=TabularReadRequest(
source_ref=request.origin_ref,
limit=request.limit,
offset=request.offset,
columns=request.columns,
expected_fingerprint=request.expected_fingerprint,
max_bytes=request.max_bytes,
timeout_ms=request.timeout_ms,
),
)
except TabularSourceError as exc:
raise _datasource_error(exc) from exc
return DatasourceOriginReadResult(
origin=_origin(result.source),
rows=result.rows,
total_rows=result.total_rows,
truncated=result.truncated,
returned_bytes=result.returned_bytes,
elapsed_ms=result.elapsed_ms,
effective_row_limit=result.effective_row_limit,
effective_byte_limit=result.effective_byte_limit,
effective_timeout_ms=result.effective_timeout_ms,
diagnostics=result.diagnostics,
)
def _origin(source: TabularSource) -> DatasourceOrigin:
return DatasourceOrigin(
ref=source.ref,
source_name=source.source_name,
name=source.name,
description=source.description,
kind="upload",
shape="tabular",
supported_modes=("live", "cached"),
provider=f"connectors.{source.provider}",
schema=tuple(
DatasourceField(
name=column.name,
data_type=column.data_type,
nullable=column.nullable,
)
for column in source.schema
),
schema_version=source.schema_version,
fingerprint=source.fingerprint,
row_count=source.row_count,
byte_count=source.byte_count,
updated_at=source.updated_at,
capabilities=source.capabilities,
metadata=dict(source.metadata),
source_mode=source.source_mode,
pushdown=source.pushdown,
health=source.health,
)
def _datasource_error(exc: TabularSourceError):
if isinstance(exc, TabularSourceAccessError):
return DatasourceAccessError(str(exc))
if isinstance(exc, TabularSourceNotFoundError):
return DatasourceNotFoundError(str(exc))
if isinstance(exc, TabularSourceUnavailableError):
return DatasourceUnavailableError(str(exc))
if isinstance(exc, TabularSourceValidationError):
return DatasourceValidationError(str(exc))
return DatasourceValidationError(str(exc))
__all__ = ["ConnectorDatasourceOriginProvider"]
@@ -0,0 +1,3 @@
from govoplan_connectors.backend.db.models import ConnectorTabularSource
__all__ = ["ConnectorTabularSource"]
@@ -0,0 +1,263 @@
from __future__ import annotations
import uuid
from datetime import datetime
from typing import Any
from sqlalchemy import (
DateTime,
ForeignKey,
Index,
Integer,
JSON,
LargeBinary,
String,
Text,
UniqueConstraint,
)
from sqlalchemy.orm import Mapped, mapped_column
from govoplan_core.db.base import Base, TimestampMixin
def new_uuid() -> str:
return str(uuid.uuid4())
class ConnectorTabularSource(Base, TimestampMixin):
__tablename__ = "connector_tabular_sources"
__table_args__ = (
UniqueConstraint("tenant_id", "source_name", name="uq_connector_tabular_source_name"),
Index("ix_connector_tabular_sources_tenant_status", "tenant_id", "status"),
Index("ix_connector_tabular_sources_tenant_updated", "tenant_id", "updated_at"),
)
id: Mapped[str] = mapped_column(String(36), primary_key=True, default=new_uuid)
tenant_id: Mapped[str] = mapped_column(String(36), nullable=False, index=True)
provider: Mapped[str] = mapped_column(String(50), default="snapshot", nullable=False, index=True)
source_name: Mapped[str] = mapped_column(String(120), nullable=False)
name: Mapped[str] = mapped_column(String(300), nullable=False)
description: Mapped[str | None] = mapped_column(Text)
status: Mapped[str] = mapped_column(String(30), default="active", nullable=False, index=True)
schema_version: Mapped[int] = mapped_column(Integer, default=1, nullable=False)
schema_: Mapped[list[dict[str, Any]]] = mapped_column("schema", JSON, default=list, nullable=False)
rows: Mapped[list[dict[str, Any]]] = mapped_column(JSON, default=list, nullable=False)
fingerprint: Mapped[str] = mapped_column(String(64), nullable=False, index=True)
row_count: Mapped[int] = mapped_column(Integer, nullable=False)
byte_count: Mapped[int] = mapped_column(Integer, nullable=False)
metadata_: Mapped[dict[str, Any]] = mapped_column("metadata", JSON, default=dict, nullable=False)
created_by: Mapped[str | None] = mapped_column(String(255), nullable=True, index=True)
updated_by: Mapped[str | None] = mapped_column(String(255), nullable=True, index=True)
deleted_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True, index=True)
class ConnectorSanctionsAcquisitionRun(Base, TimestampMixin):
__tablename__ = "connector_sanctions_acquisition_runs"
__table_args__ = (
Index(
"ix_connector_sanctions_run_health",
"tenant_id",
"provider_id",
"status",
"started_at",
),
)
id: Mapped[str] = mapped_column(
String(36),
primary_key=True,
default=new_uuid,
)
tenant_id: Mapped[str] = mapped_column(
String(36),
nullable=False,
index=True,
)
provider_id: Mapped[str] = mapped_column(
String(100),
nullable=False,
index=True,
)
source_id: Mapped[str] = mapped_column(
String(200),
nullable=False,
index=True,
)
status: Mapped[str] = mapped_column(
String(40),
default="running",
nullable=False,
index=True,
)
attempt_count: Mapped[int] = mapped_column(
Integer,
default=0,
nullable=False,
)
request_evidence: Mapped[dict[str, Any]] = mapped_column(
JSON,
default=dict,
nullable=False,
)
response_evidence: Mapped[dict[str, Any]] = mapped_column(
JSON,
default=dict,
nullable=False,
)
started_at: Mapped[datetime] = mapped_column(
DateTime(timezone=True),
nullable=False,
index=True,
)
finished_at: Mapped[datetime | None] = mapped_column(
DateTime(timezone=True),
nullable=True,
)
snapshot_id: Mapped[str | None] = mapped_column(
String(36),
nullable=True,
index=True,
)
error: Mapped[str | None] = mapped_column(
Text,
nullable=True,
)
created_by: Mapped[str | None] = mapped_column(
String(255),
nullable=True,
index=True,
)
class ConnectorSanctionsSnapshot(Base, TimestampMixin):
__tablename__ = "connector_sanctions_snapshots"
__table_args__ = (
UniqueConstraint(
"connector_run_id",
name="uq_connector_sanctions_snapshot_run",
),
Index(
"ix_connector_sanctions_snapshot_source",
"tenant_id",
"provider_id",
"acquired_at",
),
Index(
"ix_connector_sanctions_snapshot_version",
"provider_id",
"source_id",
"source_version",
),
)
id: Mapped[str] = mapped_column(
String(36),
primary_key=True,
default=new_uuid,
)
tenant_id: Mapped[str] = mapped_column(
String(36),
nullable=False,
index=True,
)
provider_id: Mapped[str] = mapped_column(
String(100),
nullable=False,
index=True,
)
publisher: Mapped[str] = mapped_column(
String(300),
nullable=False,
)
jurisdiction: Mapped[str] = mapped_column(
String(100),
nullable=False,
index=True,
)
list_type: Mapped[str] = mapped_column(
String(100),
nullable=False,
index=True,
)
source_id: Mapped[str] = mapped_column(
String(200),
nullable=False,
index=True,
)
source_version: Mapped[str] = mapped_column(
String(255),
nullable=False,
index=True,
)
publication_at: Mapped[datetime | None] = mapped_column(
DateTime(timezone=True),
nullable=True,
)
effective_at: Mapped[datetime | None] = mapped_column(
DateTime(timezone=True),
nullable=True,
)
acquired_at: Mapped[datetime] = mapped_column(
DateTime(timezone=True),
nullable=False,
index=True,
)
source_url: Mapped[str | None] = mapped_column(
String(1500),
nullable=True,
)
content_type: Mapped[str] = mapped_column(
String(200),
nullable=False,
)
byte_count: Mapped[int] = mapped_column(
Integer,
nullable=False,
)
sha256: Mapped[str] = mapped_column(
String(64),
nullable=False,
index=True,
)
signature_evidence: Mapped[dict[str, Any]] = mapped_column(
JSON,
default=dict,
nullable=False,
)
parser_version: Mapped[str] = mapped_column(
String(100),
nullable=False,
)
licence_notes: Mapped[str | None] = mapped_column(
Text,
nullable=True,
)
trust_notes: Mapped[str | None] = mapped_column(
Text,
nullable=True,
)
connector_run_id: Mapped[str] = mapped_column(
ForeignKey(
"connector_sanctions_acquisition_runs.id",
ondelete="RESTRICT",
),
nullable=False,
index=True,
)
transport_evidence: Mapped[dict[str, Any]] = mapped_column(
JSON,
default=dict,
nullable=False,
)
raw_content: Mapped[bytes] = mapped_column(
LargeBinary,
nullable=False,
)
__all__ = [
"ConnectorSanctionsAcquisitionRun",
"ConnectorSanctionsSnapshot",
"ConnectorTabularSource",
"new_uuid",
]
+413
View File
@@ -0,0 +1,413 @@
from __future__ import annotations
import hashlib
import re
import xml.etree.ElementTree as ET
from collections.abc import Mapping
from dataclasses import replace
from datetime import datetime, timedelta, timezone
from email.utils import format_datetime, parsedate_to_datetime
from defusedxml import ElementTree as SafeET
from defusedxml.common import DefusedXmlException
from govoplan_core.core.feeds import (
FeedCapabilityError,
FeedDocument,
FeedEntry,
FeedProvider,
FeedRenderRequest,
FeedRenderResult,
)
from govoplan_core.security.http_fetch import fetch_http
MAX_FEED_BYTES = 5_000_000
ATOM_NS = "http://www.w3.org/2005/Atom"
class ConnectorFeedProvider(FeedProvider):
def fetch(
self,
url: str,
*,
timeout: float = 15,
max_entries: int = 2_000,
) -> FeedDocument:
try:
response = fetch_http(
url,
timeout=timeout,
label="RSS/Atom feed URL",
headers={
"Accept": (
"application/atom+xml, application/rss+xml, "
"application/xml;q=0.9, text/xml;q=0.8"
)
},
max_bytes=MAX_FEED_BYTES,
)
except Exception as exc:
raise FeedCapabilityError(f"Feed acquisition failed: {exc}") from exc
if response.status < 200 or response.status >= 300:
raise FeedCapabilityError(
f"Feed acquisition returned HTTP {response.status}."
)
content_type = _header(response.headers, "content-type")
document = self.parse(
response.body,
source_url=url,
content_type=content_type,
max_entries=max_entries,
)
acquired_at = datetime.now(timezone.utc)
return replace(
document,
acquired_at=acquired_at,
fresh_until=_fresh_until(response.headers, acquired_at),
etag=_header(response.headers, "etag"),
last_modified=_header(response.headers, "last-modified"),
metadata={
**dict(document.metadata),
"http_status": response.status,
"byte_count": len(response.body),
},
)
def parse(
self,
content: bytes,
*,
source_url: str,
content_type: str | None = None,
max_entries: int = 2_000,
) -> FeedDocument:
if not content:
raise FeedCapabilityError("Feed content is empty.")
if len(content) > MAX_FEED_BYTES:
raise FeedCapabilityError(
f"Feeds are limited to {MAX_FEED_BYTES // 1_000_000} MB."
)
try:
root = SafeET.fromstring(content)
except (ET.ParseError, DefusedXmlException) as exc:
raise FeedCapabilityError(f"Feed XML is not safe or valid: {exc}") from exc
local_name = _local_name(root.tag)
if local_name == "rss":
document = _parse_rss(root, source_url=source_url, max_entries=max_entries)
elif local_name == "feed":
document = _parse_atom(root, source_url=source_url, max_entries=max_entries)
else:
raise FeedCapabilityError("The document is neither an RSS nor an Atom feed.")
return replace(
document,
content_type=content_type,
sha256=hashlib.sha256(content).hexdigest(),
)
def render(self, request: FeedRenderRequest) -> FeedRenderResult:
if not request.title.strip() or not request.feed_url.strip():
raise FeedCapabilityError("Feed title and feed URL are required.")
entries = tuple(
entry
for entry in request.entries
if entry.visibility in request.allowed_visibilities
)
root = (
_render_rss(request, entries)
if request.format == "rss"
else _render_atom(request, entries)
)
body = ET.tostring(root, encoding="utf-8", xml_declaration=True)
return FeedRenderResult(
format=request.format,
content_type=(
"application/rss+xml; charset=utf-8"
if request.format == "rss"
else "application/atom+xml; charset=utf-8"
),
body=body,
included_entries=len(entries),
excluded_entries=len(request.entries) - len(entries),
)
def feed_rows(document: FeedDocument) -> tuple[Mapping[str, object], ...]:
"""Map feed entries to the connector tabular shape used by Datasources."""
return tuple(
{
"id": entry.id,
"title": entry.title,
"url": entry.url,
"summary": entry.summary,
"content": entry.content,
"author": entry.author,
"published_at": (
entry.published_at.isoformat() if entry.published_at else None
),
"updated_at": entry.updated_at.isoformat() if entry.updated_at else None,
"categories": list(entry.categories),
"enclosures": [dict(item) for item in entry.enclosures],
}
for entry in document.entries
)
def _parse_rss(root: ET.Element, *, source_url: str, max_entries: int) -> FeedDocument:
channel = _first_child(root, "channel")
if channel is None:
raise FeedCapabilityError("RSS feed is missing its channel element.")
entries: list[FeedEntry] = []
for item in _children(channel, "item"):
if len(entries) >= max_entries:
raise FeedCapabilityError(f"Feeds are limited to {max_entries:,} entries.")
url = _text(item, "link")
identifier = _text(item, "guid") or url or _entry_fallback_id(item)
entries.append(
FeedEntry(
id=identifier,
title=_text(item, "title") or "(Untitled)",
url=url,
summary=_text(item, "description"),
content=_text(item, "encoded"),
author=_text(item, "author") or _text(item, "creator"),
published_at=_parse_date(_text(item, "pubDate")),
categories=tuple(
value for child in _children(item, "category")
if (value := (child.text or "").strip())
),
enclosures=tuple(
{
"url": child.attrib.get("url"),
"media_type": child.attrib.get("type"),
"size_bytes": _integer(child.attrib.get("length")),
}
for child in _children(item, "enclosure")
),
)
)
return FeedDocument(
format="rss",
title=_text(channel, "title") or "Untitled feed",
source_url=source_url,
entries=tuple(entries),
description=_text(channel, "description"),
home_url=_text(channel, "link"),
language=_text(channel, "language"),
updated_at=_parse_date(
_text(channel, "lastBuildDate") or _text(channel, "pubDate")
),
)
def _parse_atom(root: ET.Element, *, source_url: str, max_entries: int) -> FeedDocument:
entries: list[FeedEntry] = []
for item in _children(root, "entry"):
if len(entries) >= max_entries:
raise FeedCapabilityError(f"Feeds are limited to {max_entries:,} entries.")
alternate = _atom_link(item, "alternate")
identifier = _text(item, "id") or alternate or _entry_fallback_id(item)
author = _first_child(item, "author")
entries.append(
FeedEntry(
id=identifier,
title=_text(item, "title") or "(Untitled)",
url=alternate,
summary=_text(item, "summary"),
content=_text(item, "content"),
author=_text(author, "name") if author is not None else None,
published_at=_parse_date(_text(item, "published")),
updated_at=_parse_date(_text(item, "updated")),
categories=tuple(
value for child in _children(item, "category")
if (value := (child.attrib.get("term") or "").strip())
),
enclosures=tuple(
{
"url": child.attrib.get("href"),
"media_type": child.attrib.get("type"),
"size_bytes": _integer(child.attrib.get("length")),
}
for child in _children(item, "link")
if child.attrib.get("rel") == "enclosure"
),
)
)
return FeedDocument(
format="atom",
title=_text(root, "title") or "Untitled feed",
source_url=source_url,
entries=tuple(entries),
description=_text(root, "subtitle"),
home_url=_atom_link(root, "alternate"),
updated_at=_parse_date(_text(root, "updated")),
)
def _render_rss(request: FeedRenderRequest, entries: tuple[FeedEntry, ...]) -> ET.Element:
ET.register_namespace("atom", ATOM_NS)
root = ET.Element("rss", {"version": "2.0"})
channel = ET.SubElement(root, "channel")
_element(channel, "title", request.title)
_element(channel, "link", request.home_url)
_element(channel, "description", request.description or request.title)
_element(channel, f"{{{ATOM_NS}}}link", None, {
"href": request.feed_url,
"rel": "self",
"type": "application/rss+xml",
})
if request.language:
_element(channel, "language", request.language)
for entry in entries:
item = ET.SubElement(channel, "item")
_element(item, "guid", entry.id, {"isPermaLink": "false"})
_element(item, "title", entry.title)
if entry.url:
_element(item, "link", entry.url)
if entry.summary or entry.content:
_element(item, "description", entry.summary or entry.content)
if entry.author:
_element(item, "author", entry.author)
date = entry.published_at or entry.updated_at
if date:
_element(item, "pubDate", format_datetime(_utc(date)))
for category in entry.categories:
_element(item, "category", category)
for enclosure in entry.enclosures:
attributes = {
"url": str(enclosure.get("url") or ""),
"type": str(enclosure.get("media_type") or "application/octet-stream"),
"length": str(enclosure.get("size_bytes") or 0),
}
if attributes["url"]:
_element(item, "enclosure", None, attributes)
return root
def _render_atom(request: FeedRenderRequest, entries: tuple[FeedEntry, ...]) -> ET.Element:
ET.register_namespace("", ATOM_NS)
root = ET.Element(f"{{{ATOM_NS}}}feed")
_element(root, f"{{{ATOM_NS}}}id", request.feed_url)
_element(root, f"{{{ATOM_NS}}}title", request.title)
_element(root, f"{{{ATOM_NS}}}link", None, {"href": request.home_url})
_element(
root,
f"{{{ATOM_NS}}}link",
None,
{"href": request.feed_url, "rel": "self", "type": "application/atom+xml"},
)
latest = max(
(date for entry in entries for date in (entry.updated_at, entry.published_at) if date),
default=datetime.now(timezone.utc),
)
_element(root, f"{{{ATOM_NS}}}updated", _utc(latest).isoformat().replace("+00:00", "Z"))
if request.description:
_element(root, f"{{{ATOM_NS}}}subtitle", request.description)
for value in entries:
entry = ET.SubElement(root, f"{{{ATOM_NS}}}entry")
_element(entry, f"{{{ATOM_NS}}}id", value.id)
_element(entry, f"{{{ATOM_NS}}}title", value.title)
if value.url:
_element(entry, f"{{{ATOM_NS}}}link", None, {"href": value.url})
if value.summary:
_element(entry, f"{{{ATOM_NS}}}summary", value.summary)
if value.content:
_element(entry, f"{{{ATOM_NS}}}content", value.content, {"type": "html"})
updated = value.updated_at or value.published_at or latest
_element(entry, f"{{{ATOM_NS}}}updated", _utc(updated).isoformat().replace("+00:00", "Z"))
if value.published_at:
_element(entry, f"{{{ATOM_NS}}}published", _utc(value.published_at).isoformat().replace("+00:00", "Z"))
if value.author:
author = ET.SubElement(entry, f"{{{ATOM_NS}}}author")
_element(author, f"{{{ATOM_NS}}}name", value.author)
for category in value.categories:
_element(entry, f"{{{ATOM_NS}}}category", None, {"term": category})
return root
def _children(element: ET.Element, name: str) -> tuple[ET.Element, ...]:
return tuple(child for child in element if _local_name(child.tag) == name)
def _first_child(element: ET.Element, name: str) -> ET.Element | None:
return next((child for child in element if _local_name(child.tag) == name), None)
def _text(element: ET.Element | None, name: str) -> str | None:
if element is None:
return None
child = _first_child(element, name)
if child is None:
return None
value = "".join(child.itertext()).strip()
return value or None
def _local_name(tag: str) -> str:
return tag.rsplit("}", 1)[-1].split(":", 1)[-1]
def _atom_link(element: ET.Element, relation: str) -> str | None:
for child in _children(element, "link"):
if (child.attrib.get("rel") or "alternate") == relation:
return child.attrib.get("href")
return None
def _parse_date(value: str | None) -> datetime | None:
if not value:
return None
try:
parsed = parsedate_to_datetime(value)
except (TypeError, ValueError, OverflowError):
try:
parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
except ValueError:
return None
return _utc(parsed)
def _utc(value: datetime) -> datetime:
if value.tzinfo is None:
return value.replace(tzinfo=timezone.utc)
return value.astimezone(timezone.utc)
def _integer(value: str | None) -> int | None:
try:
return int(value) if value is not None else None
except ValueError:
return None
def _entry_fallback_id(element: ET.Element) -> str:
body = ET.tostring(element, encoding="utf-8")
return f"urn:sha256:{hashlib.sha256(body).hexdigest()}"
def _header(headers: Mapping[str, str], name: str) -> str | None:
lowered = name.casefold()
return next((value for key, value in headers.items() if key.casefold() == lowered), None)
def _fresh_until(headers: Mapping[str, str], acquired_at: datetime) -> datetime | None:
cache_control = _header(headers, "cache-control") or ""
match = re.search(r"(?:^|,)\s*max-age\s*=\s*(\d+)", cache_control, re.IGNORECASE)
if match:
return acquired_at + timedelta(seconds=int(match.group(1)))
return _parse_date(_header(headers, "expires"))
def _element(
parent: ET.Element,
tag: str,
text: str | None,
attributes: Mapping[str, str] | None = None,
) -> ET.Element:
child = ET.SubElement(parent, tag, dict(attributes or {}))
child.text = text
return child
__all__ = ["ConnectorFeedProvider", "MAX_FEED_BYTES", "feed_rows"]
+547
View File
@@ -0,0 +1,547 @@
from __future__ import annotations
from pathlib import Path
from govoplan_core.core.access import (
CAPABILITY_AUTH_PERMISSION_EVALUATOR,
CAPABILITY_AUTH_PRINCIPAL_RESOLVER,
)
from govoplan_core.core.module_guards import (
drop_table_retirement_provider,
persistent_table_uninstall_guard,
)
from govoplan_core.core.datasources import CAPABILITY_DATASOURCE_ORIGINS
from govoplan_core.core.feeds import CAPABILITY_CONNECTORS_FEEDS
from govoplan_core.core.modules import (
DocumentationTopic,
MigrationSpec,
ModuleInterfaceProvider,
ModuleManifest,
PermissionDefinition,
RoleTemplate,
)
from govoplan_core.core.provider_governance import (
ExternalProviderDeclaration,
ExternalProviderStateProviderRegistration,
ModuleArchitectureDeclaration,
ModuleArchitectureDocumentation,
ModuleMaturityEvidence,
ProviderBehaviorDeclaration,
ProviderObjectDeclaration,
)
from govoplan_core.core.tabular_sources import (
CAPABILITY_CONNECTORS_TABULAR_SNAPSHOT_WRITER,
CAPABILITY_CONNECTORS_TABULAR_SOURCES,
)
from govoplan_core.core.sanctions import (
CAPABILITY_CONNECTORS_SANCTIONS_SNAPSHOTS,
)
from govoplan_core.db.base import Base
from govoplan_connectors.backend.db.models import (
ConnectorSanctionsAcquisitionRun,
ConnectorSanctionsSnapshot,
ConnectorTabularSource,
)
from govoplan_connectors.backend.sanctions_sources import (
SANCTIONS_READ_SCOPE,
SANCTIONS_REFRESH_SCOPE,
SqlSanctionsSnapshotProvider,
)
from govoplan_connectors.backend.tabular_sources import (
ADMIN_SCOPE,
READ_SCOPE,
WRITE_SCOPE,
SqlTabularSourceProvider,
)
from govoplan_connectors.backend.datasource_origins import (
ConnectorDatasourceOriginProvider,
)
from govoplan_connectors.backend.feeds import ConnectorFeedProvider
from govoplan_connectors.backend.provider_state import (
SANCTIONS_PROVIDER_ID,
TABULAR_PROVIDER_ID,
sanctions_provider_states,
tabular_provider_states,
)
MODULE_ID = "connectors"
MODULE_VERSION = "0.1.16"
TABULAR_SOURCE_INTERFACE_VERSION = "0.1.0"
DATASOURCE_ORIGIN_INTERFACE_VERSION = "0.1.0"
SANCTIONS_SNAPSHOT_INTERFACE_VERSION = "1.0.0"
FEED_INTERFACE_VERSION = "0.1.0"
CONNECTOR_RUNTIME_INTERFACE_VERSION = "1.0.0"
ARCHITECTURE = ModuleArchitectureDeclaration(
layer="data_reporting_integration",
kind="integration",
maturity="vertical_slice",
evidence=(
ModuleMaturityEvidence(
kind="test",
reference="tests/test_tabular_sources.py",
summary="Exercises tenant-safe immutable tabular snapshots and bounded reads.",
),
ModuleMaturityEvidence(
kind="test",
reference="tests/test_sanctions_sources.py",
summary="Exercises source acquisition health, checksums, retries, and immutable evidence.",
),
ModuleMaturityEvidence(
kind="recovery",
reference="tests/test_recovery.py",
summary="Proves atomic snapshot commits, idempotent replay, distributed fences, tamper rejection, and unknown external-effect handling.",
),
ModuleMaturityEvidence(
kind="documentation",
reference="docs/CONNECTOR_SOURCE_LIFECYCLE.md",
summary="Defines source lifecycle, authority, evidence, and outage boundaries.",
),
),
known_limits=(
"The executable generic datasource origin is an immutable tabular snapshot; database and arbitrary REST profiles remain future providers.",
"Feed publication renders a governed document but does not yet push it to an external publishing endpoint.",
),
supported_authority_modes=(
"external_authoritative",
"external_mirror",
"linked_reference",
),
owned_concepts=(
"external transport profiles",
"protocol interaction",
"immutable connector snapshots",
"connector acquisition health",
),
non_owned_concepts=(
"datasource catalogue identity and lifecycle",
"domain records and business semantics",
"data transformations",
"screening dispositions",
),
target_tested_providers=(
TABULAR_PROVIDER_ID,
SANCTIONS_PROVIDER_ID,
),
documentation=ModuleArchitectureDocumentation(
migration=("src/govoplan_connectors/backend/migrations/versions",),
upgrade=("docs/CONNECTOR_SOURCE_LIFECYCLE.md",),
recovery=("docs/CONNECTOR_SOURCE_LIFECYCLE.md",),
security=("docs/CONNECTOR_SOURCE_LIFECYCLE.md",),
operations=("docs/CONNECTOR_SOURCE_LIFECYCLE.md",),
),
)
EXTERNAL_PROVIDERS = (
ExternalProviderDeclaration(
id=TABULAR_PROVIDER_ID,
module_id=MODULE_ID,
label="Immutable tabular snapshot provider",
maturity="read",
operations=("discover", "search", "read", "preview", "dry_run"),
objects=(
ProviderObjectDeclaration(
object_type="tabular_source_snapshot",
field_groups=("identity", "schema", "rows", "source_provenance"),
authority_modes=("external_authoritative", "external_mirror"),
default_authority_mode="external_mirror",
),
),
behavior=ProviderBehaviorDeclaration(
revision_tokens="Source fingerprints and immutable snapshot ids are retained.",
concurrency="Reads may require the expected fingerprint; snapshots never mutate in place.",
freshness="Snapshot acquisition time and source timestamp are exposed.",
health="Import validation and source-read failures are explicit.",
max_read_items=1000,
idempotency="Feed imports accept a caller request key and replay the same committed immutable source without refetching.",
retry="Read-only acquisition may be retried only as a new deliberate request after a failed atomic operation.",
outcome_unknown="Provider reads do not mutate remote state; an uncertain database commit is resolved by the atomic recovery transaction.",
outcome_unknown_supported=False,
evidence="Rows, schema, fingerprint, source metadata, and acquisition provenance remain linked.",
correction="Import a replacement snapshot; retain the prior snapshot as evidence.",
rollback="Snapshot rows and the terminal recovery checkpoint commit or roll back together.",
reconciliation="Compare source and snapshot fingerprints before selecting a new current state.",
outage="Existing snapshots remain available and visibly stale; no live-source claim is made.",
classifications=("internal", "confidential", "restricted"),
purposes=("governed import", "dataflow input", "evidence reconstruction"),
retention="Datasources or the consuming domain supplies retention and hold policy.",
secret_handling="Generic snapshots contain no connector credential; transport credentials stay in credential envelopes.",
),
capability_names=(
CAPABILITY_CONNECTORS_TABULAR_SOURCES,
CAPABILITY_DATASOURCE_ORIGINS,
),
interface_names=(
"connectors.tabular_sources",
"connectors.datasource_origins",
),
documentation_topic_ids=(
"connectors.authority-and-effects",
"connectors.tabular-sources",
),
),
ExternalProviderDeclaration(
id=SANCTIONS_PROVIDER_ID,
module_id=MODULE_ID,
label="Sanctions source snapshot provider",
maturity="read",
operations=("discover", "search", "read", "preview"),
objects=(
ProviderObjectDeclaration(
object_type="sanctions_source_snapshot",
field_groups=("source_identity", "raw_evidence", "entries", "acquisition_health"),
authority_modes=("external_authoritative", "external_mirror"),
default_authority_mode="external_mirror",
),
),
behavior=ProviderBehaviorDeclaration(
revision_tokens="Provider source version, ETag, Last-Modified, and SHA-256 digest are retained when available.",
concurrency="Refreshes use conditional source requests, a distributed per-tenant/provider fence, and immutable snapshots.",
freshness="Latest successful acquisition, source timestamp, and stale health are reported.",
health="Transport, parsing, source-change, and malformed-source states are explicit.",
max_read_items=5000,
idempotency="A caller request key identifies one acquisition run and replays its committed result without contacting the source again.",
retry="Bounded HTTP retries are safe because acquisition is read-only; failed runs require a new deliberate request key.",
timeout_seconds=30,
outcome_unknown="The external operation is read-only; snapshot rows and recovery evidence commit atomically.",
outcome_unknown_supported=False,
evidence="Raw source bytes, checksum, acquisition run, parser result, and normalized entry count are linked.",
correction="A corrected source creates a new immutable snapshot and acquisition run.",
rollback="A failed database transaction leaves no snapshot and the stale atomic fence resolves as failed.",
reconciliation="Compare source version and digest, then preserve both prior and corrected evidence.",
outage="The latest accepted snapshot stays usable with stale/unavailable source health.",
classifications=("public", "internal"),
purposes=("sanctions source acquisition", "compliance screening evidence"),
retention="Risk and Records policies determine accepted snapshot retention and legal holds.",
secret_handling="Public sources require no subject data or source credential; configured proxy secrets remain external to snapshots.",
),
capability_names=(CAPABILITY_CONNECTORS_SANCTIONS_SNAPSHOTS,),
interface_names=("connectors.sanctions_snapshots",),
documentation_topic_ids=(
"connectors.authority-and-effects",
"connectors.sanctions-snapshots",
),
),
)
def _permission(scope: str, label: str, description: str) -> PermissionDefinition:
module_id, resource, action = scope.split(":", 2)
return PermissionDefinition(
scope=scope,
label=label,
description=description,
category="Connectors",
level="tenant",
module_id=module_id,
resource=resource,
action=action,
)
PERMISSIONS = (
_permission(
READ_SCOPE,
"View tabular sources",
"Discover and preview policy-visible tabular connector sources.",
),
_permission(
WRITE_SCOPE,
"Manage tabular sources",
"Import and retire bounded tabular snapshots.",
),
_permission(
ADMIN_SCOPE,
"Administer connector sources",
"Manage every tenant connector source and future source policies.",
),
_permission(
SANCTIONS_READ_SCOPE,
"View sanctions source evidence",
"Inspect immutable sanctions snapshots and acquisition health.",
),
_permission(
SANCTIONS_REFRESH_SCOPE,
"Refresh sanctions sources",
"Acquire a new immutable sanctions source snapshot.",
),
)
ROLE_TEMPLATES = (
RoleTemplate(
slug="connector_source_manager",
name="Connector source manager",
description="Discover, import, preview, and retire tabular sources.",
permissions=(
READ_SCOPE,
WRITE_SCOPE,
SANCTIONS_READ_SCOPE,
SANCTIONS_REFRESH_SCOPE,
),
),
RoleTemplate(
slug="connector_source_reader",
name="Connector source reader",
description="Discover and preview tabular connector sources.",
permissions=(READ_SCOPE, SANCTIONS_READ_SCOPE),
),
)
def _router(_context):
from govoplan_connectors.backend.router import router
return router
def _provider(_context) -> SqlTabularSourceProvider:
return SqlTabularSourceProvider()
def _datasource_origin_provider(_context) -> ConnectorDatasourceOriginProvider:
return ConnectorDatasourceOriginProvider()
def _sanctions_snapshot_provider(
_context,
) -> SqlSanctionsSnapshotProvider:
return SqlSanctionsSnapshotProvider()
def _feed_provider(_context) -> ConnectorFeedProvider:
return ConnectorFeedProvider()
def _tenant_summary(session, tenant_id: str) -> dict[str, int]:
return {
"connector_tabular_sources": (
session.query(ConnectorTabularSource)
.filter(
ConnectorTabularSource.tenant_id == tenant_id,
ConnectorTabularSource.deleted_at.is_(None),
)
.count()
),
"connector_sanctions_snapshots": (
session.query(ConnectorSanctionsSnapshot)
.filter(ConnectorSanctionsSnapshot.tenant_id == tenant_id)
.count()
),
"connector_sanctions_runs": (
session.query(ConnectorSanctionsAcquisitionRun)
.filter(ConnectorSanctionsAcquisitionRun.tenant_id == tenant_id)
.count()
),
}
manifest = ModuleManifest(
id=MODULE_ID,
name="Connectors",
version=MODULE_VERSION,
optional_dependencies=(
"access",
"audit",
"files",
"policy",
"datasources",
"portal",
"reporting",
"risk_compliance",
),
required_capabilities=(
CAPABILITY_AUTH_PRINCIPAL_RESOLVER,
CAPABILITY_AUTH_PERMISSION_EVALUATOR,
),
provides_interfaces=(
ModuleInterfaceProvider(
name="connectors.tabular_sources",
version=TABULAR_SOURCE_INTERFACE_VERSION,
),
ModuleInterfaceProvider(
name="connectors.tabular_snapshot_writer",
version=TABULAR_SOURCE_INTERFACE_VERSION,
),
ModuleInterfaceProvider(
name="connectors.datasource_origins",
version=DATASOURCE_ORIGIN_INTERFACE_VERSION,
),
ModuleInterfaceProvider(
name="connectors.sanctions_snapshots",
version=SANCTIONS_SNAPSHOT_INTERFACE_VERSION,
),
ModuleInterfaceProvider(
name="connectors.feeds",
version=FEED_INTERFACE_VERSION,
),
ModuleInterfaceProvider(
name="connectors.runtime_contract",
version=CONNECTOR_RUNTIME_INTERFACE_VERSION,
),
),
permissions=PERMISSIONS,
role_templates=ROLE_TEMPLATES,
route_factory=_router,
capability_factories={
CAPABILITY_CONNECTORS_TABULAR_SOURCES: _provider,
CAPABILITY_CONNECTORS_TABULAR_SNAPSHOT_WRITER: _provider,
CAPABILITY_DATASOURCE_ORIGINS: _datasource_origin_provider,
CAPABILITY_CONNECTORS_SANCTIONS_SNAPSHOTS: (_sanctions_snapshot_provider),
CAPABILITY_CONNECTORS_FEEDS: _feed_provider,
},
tenant_summary_providers=(_tenant_summary,),
architecture=ARCHITECTURE,
external_providers=EXTERNAL_PROVIDERS,
external_provider_state_providers=(
ExternalProviderStateProviderRegistration(
module_id=MODULE_ID,
provider_id=TABULAR_PROVIDER_ID,
provider=tabular_provider_states,
),
ExternalProviderStateProviderRegistration(
module_id=MODULE_ID,
provider_id=SANCTIONS_PROVIDER_ID,
provider=sanctions_provider_states,
),
),
migration_spec=MigrationSpec(
module_id=MODULE_ID,
metadata=Base.metadata,
script_location=str(Path(__file__).with_name("migrations") / "versions"),
retirement_supported=True,
retirement_provider=drop_table_retirement_provider(
ConnectorSanctionsSnapshot,
ConnectorSanctionsAcquisitionRun,
ConnectorTabularSource,
label="Connectors",
),
retirement_notes=(
"Destructive retirement drops connector-owned source snapshots after "
"the installer captures a database snapshot."
),
),
uninstall_guard_providers=(
persistent_table_uninstall_guard(
ConnectorSanctionsSnapshot,
ConnectorSanctionsAcquisitionRun,
ConnectorTabularSource,
label="Connectors",
),
),
documentation=(
DocumentationTopic(
id="connectors.authority-and-effects",
title="Connector authority and effect behavior",
summary="Connector direction, technical maturity, and configured source authority are separate and must remain visible.",
body=(
"A connector can consume, publish, or work bidirectionally and can mature from discovery through replacement. "
"Each binding separately states whether GovOPlaN is authoritative, follows an external authority, keeps a mirror, synchronizes under conflict rules, adds a governance overlay, or retains only a link. "
"Writable providers must explain revisions, limits, idempotency, outcome-unknown handling, evidence, reconciliation, correction, outage behavior, and secret requirements."
),
layer="available",
documentation_types=("admin", "user"),
audience=("operator", "module_admin", "power_user", "product_owner"),
related_modules=("datasources", "dataflow", "ops", "policy", "audit"),
order=39,
),
DocumentationTopic(
id="connectors.runtime-preview-contract",
title="Connector previews and diagnostics",
summary="Use one bounded, redacted dry-run shape across external transports.",
body=(
"Connectors owns endpoint discovery, authentication hand-off, transport limits, retries, and protocol health. "
"Domain modules own field mapping, validation, reconciliation, and record mutation. The shared Core runtime "
"contract reports redacted effects and diagnostics with source revisions, fingerprints, and immutable input hashes. "
"Tabular previews enforce effective row, serialized-byte, and elapsed-time ceilings and report limit truncation "
"as structured diagnostics. A commit must reject stale, truncated, conflicting, or error-bearing previews, and "
"credentials never appear in URLs or samples."
),
layer="available",
documentation_types=("admin", "user"),
audience=("operator", "module_admin", "power_user"),
related_modules=("addresses", "datasources", "dataflow", "policy", "audit"),
order=40,
),
DocumentationTopic(
id="connectors.tabular-sources",
title="Governed tabular sources",
summary="Provider-neutral source discovery and bounded reads for Dataflow.",
body=(
"Connectors owns source configuration, access checks, schema discovery, "
"fingerprints, and bounded reads. Dataflow stores only opaque source "
"references and expected fingerprints. Each source declares its live, "
"cached, file-backed, or static mode, structured health, and supported "
"projection, filter, aggregation, sorting, and pagination pushdown. The "
"first executable provider imports immutable JSON or CSV snapshots, "
"supports projection and pagination, and exposes them as Datasource "
"origins. Database and API providers can implement the same origin "
"contract without changing Datasources or Dataflow."
),
layer="available",
documentation_types=("admin", "user"),
audience=("operator", "module_admin", "power_user"),
related_modules=("dataflow", "files", "reporting", "risk_compliance"),
order=40,
),
DocumentationTopic(
id="connectors.rss-atom",
title="RSS and Atom feeds",
summary="Import governed feed snapshots and emit visibility-filtered feeds.",
body=(
"Connectors owns bounded, SSRF-protected RSS/Atom transport and XML "
"parsing. Imported entries become immutable tabular snapshots exposed "
"through Datasources, including acquisition, freshness, ETag, content "
"digest, and source provenance. Portal or Reporting owns publication "
"routes and must pass the allowed visibility set when rendering output. "
"A separate RSS module is only warranted if GovOPlaN later needs a "
"dedicated feed-reader product surface."
),
layer="available",
documentation_types=("admin", "user"),
audience=("operator", "module_admin", "power_user"),
related_modules=("datasources", "dataflow", "portal", "reporting"),
order=42,
),
DocumentationTopic(
id="connectors.sanctions-snapshots",
title="Sanctions source snapshots",
summary=(
"Acquire immutable, checksum-verifiable sanctions list "
"evidence without transmitting screening subjects."
),
body=(
"Connectors provides a deterministic synthetic fixture and "
"the official United Nations Security Council consolidated "
"XML source. Each fetch records conditional transport "
"evidence, bounded retries, health state, source metadata, "
"raw evidence, and a SHA-256 checksum. Refreshes acquire a "
"distributed recovery fence before provider I/O; the immutable "
"snapshot and terminal recovery checkpoint then commit in one "
"transaction. A repeated request key returns the same result. "
"Risk Compliance owns "
"normalization, matching, legal review, and dispositions."
),
layer="available",
documentation_types=("admin", "user"),
audience=("operator", "module_admin", "compliance_reviewer"),
related_modules=("risk_compliance", "dataflow"),
order=41,
),
),
)
def get_manifest() -> ModuleManifest:
return manifest
__all__ = [
"MODULE_ID",
"MODULE_VERSION",
"DATASOURCE_ORIGIN_INTERFACE_VERSION",
"SANCTIONS_SNAPSHOT_INTERFACE_VERSION",
"TABULAR_SOURCE_INTERFACE_VERSION",
"get_manifest",
"manifest",
]
@@ -0,0 +1 @@
"""Connectors migrations."""
@@ -0,0 +1 @@
"""Connectors migration revisions."""
@@ -0,0 +1,141 @@
"""v0.1.14 Connectors baseline
Revision ID: e6b7c8d9f0a1
Revises: None
Create Date: 2026-07-28 00:00:00.000000
"""
from __future__ import annotations
from alembic import op
import sqlalchemy as sa
revision = "e6b7c8d9f0a1"
down_revision = None
branch_labels = None
depends_on = None
def upgrade() -> None:
op.create_table(
"connector_tabular_sources",
sa.Column("id", sa.String(length=36), nullable=False),
sa.Column("tenant_id", sa.String(length=36), nullable=False),
sa.Column("provider", sa.String(length=50), nullable=False),
sa.Column("source_name", sa.String(length=120), nullable=False),
sa.Column("name", sa.String(length=300), nullable=False),
sa.Column("description", sa.Text(), nullable=True),
sa.Column("status", sa.String(length=30), nullable=False),
sa.Column("schema_version", sa.Integer(), nullable=False),
sa.Column("schema", sa.JSON(), nullable=False),
sa.Column("rows", sa.JSON(), nullable=False),
sa.Column("fingerprint", sa.String(length=64), nullable=False),
sa.Column("row_count", sa.Integer(), nullable=False),
sa.Column("byte_count", sa.Integer(), nullable=False),
sa.Column("metadata", sa.JSON(), nullable=False),
sa.Column("created_by", sa.String(length=255), nullable=True),
sa.Column("updated_by", sa.String(length=255), nullable=True),
sa.Column("deleted_at", sa.DateTime(timezone=True), nullable=True),
sa.Column("created_at", sa.DateTime(timezone=True), nullable=False),
sa.Column("updated_at", sa.DateTime(timezone=True), nullable=False),
sa.PrimaryKeyConstraint("id", name=op.f("pk_connector_tabular_sources")),
sa.UniqueConstraint(
"tenant_id",
"source_name",
name="uq_connector_tabular_source_name",
),
)
op.create_index(
op.f("ix_connector_tabular_sources_created_by"),
"connector_tabular_sources",
["created_by"],
unique=False,
)
op.create_index(
op.f("ix_connector_tabular_sources_deleted_at"),
"connector_tabular_sources",
["deleted_at"],
unique=False,
)
op.create_index(
op.f("ix_connector_tabular_sources_fingerprint"),
"connector_tabular_sources",
["fingerprint"],
unique=False,
)
op.create_index(
op.f("ix_connector_tabular_sources_provider"),
"connector_tabular_sources",
["provider"],
unique=False,
)
op.create_index(
op.f("ix_connector_tabular_sources_status"),
"connector_tabular_sources",
["status"],
unique=False,
)
op.create_index(
op.f("ix_connector_tabular_sources_tenant_id"),
"connector_tabular_sources",
["tenant_id"],
unique=False,
)
op.create_index(
op.f("ix_connector_tabular_sources_updated_by"),
"connector_tabular_sources",
["updated_by"],
unique=False,
)
op.create_index(
"ix_connector_tabular_sources_tenant_status",
"connector_tabular_sources",
["tenant_id", "status"],
unique=False,
)
op.create_index(
"ix_connector_tabular_sources_tenant_updated",
"connector_tabular_sources",
["tenant_id", "updated_at"],
unique=False,
)
def downgrade() -> None:
op.drop_index(
"ix_connector_tabular_sources_tenant_updated",
table_name="connector_tabular_sources",
)
op.drop_index(
"ix_connector_tabular_sources_tenant_status",
table_name="connector_tabular_sources",
)
op.drop_index(
op.f("ix_connector_tabular_sources_updated_by"),
table_name="connector_tabular_sources",
)
op.drop_index(
op.f("ix_connector_tabular_sources_tenant_id"),
table_name="connector_tabular_sources",
)
op.drop_index(
op.f("ix_connector_tabular_sources_status"),
table_name="connector_tabular_sources",
)
op.drop_index(
op.f("ix_connector_tabular_sources_provider"),
table_name="connector_tabular_sources",
)
op.drop_index(
op.f("ix_connector_tabular_sources_fingerprint"),
table_name="connector_tabular_sources",
)
op.drop_index(
op.f("ix_connector_tabular_sources_deleted_at"),
table_name="connector_tabular_sources",
)
op.drop_index(
op.f("ix_connector_tabular_sources_created_by"),
table_name="connector_tabular_sources",
)
op.drop_table("connector_tabular_sources")
@@ -0,0 +1,192 @@
"""Add immutable sanctions source snapshots.
Revision ID: f7c8d9e0a1b2
Revises: e6b7c8d9f0a1
Create Date: 2026-07-29
"""
from __future__ import annotations
from alembic import op
import sqlalchemy as sa
revision = "f7c8d9e0a1b2"
down_revision = "e6b7c8d9f0a1"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.create_table(
"connector_sanctions_acquisition_runs",
sa.Column("id", sa.String(length=36), nullable=False),
sa.Column("tenant_id", sa.String(length=36), nullable=False),
sa.Column("provider_id", sa.String(length=100), nullable=False),
sa.Column("source_id", sa.String(length=200), nullable=False),
sa.Column("status", sa.String(length=40), nullable=False),
sa.Column("attempt_count", sa.Integer(), nullable=False),
sa.Column("request_evidence", sa.JSON(), nullable=False),
sa.Column("response_evidence", sa.JSON(), nullable=False),
sa.Column(
"started_at",
sa.DateTime(timezone=True),
nullable=False,
),
sa.Column(
"finished_at",
sa.DateTime(timezone=True),
nullable=True,
),
sa.Column("snapshot_id", sa.String(length=36), nullable=True),
sa.Column("error", sa.Text(), nullable=True),
sa.Column("created_by", sa.String(length=255), nullable=True),
sa.Column(
"created_at",
sa.DateTime(timezone=True),
nullable=False,
),
sa.Column(
"updated_at",
sa.DateTime(timezone=True),
nullable=False,
),
sa.PrimaryKeyConstraint(
"id",
name=op.f("pk_connector_sanctions_acquisition_runs"),
),
)
for column in (
"tenant_id",
"provider_id",
"source_id",
"status",
"started_at",
"snapshot_id",
"created_by",
):
op.create_index(
op.f(
"ix_connector_sanctions_acquisition_runs_"
f"{column}"
),
"connector_sanctions_acquisition_runs",
[column],
)
op.create_index(
"ix_connector_sanctions_run_health",
"connector_sanctions_acquisition_runs",
["tenant_id", "provider_id", "status", "started_at"],
)
op.create_table(
"connector_sanctions_snapshots",
sa.Column("id", sa.String(length=36), nullable=False),
sa.Column("tenant_id", sa.String(length=36), nullable=False),
sa.Column("provider_id", sa.String(length=100), nullable=False),
sa.Column("publisher", sa.String(length=300), nullable=False),
sa.Column("jurisdiction", sa.String(length=100), nullable=False),
sa.Column("list_type", sa.String(length=100), nullable=False),
sa.Column("source_id", sa.String(length=200), nullable=False),
sa.Column(
"source_version",
sa.String(length=255),
nullable=False,
),
sa.Column(
"publication_at",
sa.DateTime(timezone=True),
nullable=True,
),
sa.Column(
"effective_at",
sa.DateTime(timezone=True),
nullable=True,
),
sa.Column(
"acquired_at",
sa.DateTime(timezone=True),
nullable=False,
),
sa.Column("source_url", sa.String(length=1500), nullable=True),
sa.Column(
"content_type",
sa.String(length=200),
nullable=False,
),
sa.Column("byte_count", sa.Integer(), nullable=False),
sa.Column("sha256", sa.String(length=64), nullable=False),
sa.Column("signature_evidence", sa.JSON(), nullable=False),
sa.Column(
"parser_version",
sa.String(length=100),
nullable=False,
),
sa.Column("licence_notes", sa.Text(), nullable=True),
sa.Column("trust_notes", sa.Text(), nullable=True),
sa.Column(
"connector_run_id",
sa.String(length=36),
nullable=False,
),
sa.Column("transport_evidence", sa.JSON(), nullable=False),
sa.Column("raw_content", sa.LargeBinary(), nullable=False),
sa.Column(
"created_at",
sa.DateTime(timezone=True),
nullable=False,
),
sa.Column(
"updated_at",
sa.DateTime(timezone=True),
nullable=False,
),
sa.ForeignKeyConstraint(
["connector_run_id"],
["connector_sanctions_acquisition_runs.id"],
name=op.f(
"fk_connector_sanctions_snapshots_connector_run_id_"
"connector_sanctions_acquisition_runs"
),
ondelete="RESTRICT",
),
sa.PrimaryKeyConstraint(
"id",
name=op.f("pk_connector_sanctions_snapshots"),
),
sa.UniqueConstraint(
"connector_run_id",
name="uq_connector_sanctions_snapshot_run",
),
)
for column in (
"tenant_id",
"provider_id",
"jurisdiction",
"list_type",
"source_id",
"source_version",
"acquired_at",
"sha256",
"connector_run_id",
):
op.create_index(
op.f(f"ix_connector_sanctions_snapshots_{column}"),
"connector_sanctions_snapshots",
[column],
)
op.create_index(
"ix_connector_sanctions_snapshot_source",
"connector_sanctions_snapshots",
["tenant_id", "provider_id", "acquired_at"],
)
op.create_index(
"ix_connector_sanctions_snapshot_version",
"connector_sanctions_snapshots",
["provider_id", "source_id", "source_version"],
)
def downgrade() -> None:
op.drop_table("connector_sanctions_snapshots")
op.drop_table("connector_sanctions_acquisition_runs")
@@ -0,0 +1,207 @@
from __future__ import annotations
from collections import defaultdict
from datetime import UTC, datetime
from hashlib import sha256
from sqlalchemy import func, select
from sqlalchemy.orm import Session
from govoplan_connectors.backend.db.models import (
ConnectorSanctionsAcquisitionRun,
ConnectorSanctionsSnapshot,
ConnectorTabularSource,
)
from govoplan_core.core.provider_governance import (
ExternalProviderRuntimeState,
ExternalProviderStateContext,
)
TABULAR_PROVIDER_ID = "connectors.tabular_snapshot"
SANCTIONS_PROVIDER_ID = "connectors.sanctions_snapshot"
def tabular_provider_states(
context: ExternalProviderStateContext,
) -> tuple[ExternalProviderRuntimeState, ...]:
session = _session(context)
statement = select(ConnectorTabularSource).where(
ConnectorTabularSource.deleted_at.is_(None)
)
if context.tenant_id is not None:
statement = statement.where(
ConnectorTabularSource.tenant_id == context.tenant_id
)
sources = tuple(
session.scalars(
statement.order_by(
ConnectorTabularSource.tenant_id,
ConnectorTabularSource.id,
).limit(context.max_items + 1)
)
)
observed_at = datetime.now(UTC)
return tuple(_tabular_state(item, observed_at=observed_at) for item in sources)
def sanctions_provider_states(
context: ExternalProviderStateContext,
) -> tuple[ExternalProviderRuntimeState, ...]:
session = _session(context)
statement = select(ConnectorSanctionsAcquisitionRun)
if context.tenant_id is not None:
statement = statement.where(
ConnectorSanctionsAcquisitionRun.tenant_id == context.tenant_id
)
runs = tuple(
session.scalars(
statement.order_by(
ConnectorSanctionsAcquisitionRun.tenant_id,
ConnectorSanctionsAcquisitionRun.provider_id,
ConnectorSanctionsAcquisitionRun.source_id,
ConnectorSanctionsAcquisitionRun.started_at.desc(),
).limit(max(context.max_items * 10, context.max_items + 1))
)
)
latest_by_binding: dict[tuple[str, str, str], ConnectorSanctionsAcquisitionRun] = {}
for run in runs:
key = (run.tenant_id, run.provider_id, run.source_id)
latest_by_binding.setdefault(key, run)
if len(latest_by_binding) >= context.max_items + 1:
break
snapshot_counts = _snapshot_counts(
session,
binding_keys=tuple(latest_by_binding),
)
observed_at = datetime.now(UTC)
return tuple(
_sanctions_state(
run,
observed_at=observed_at,
snapshot_count=snapshot_counts.get(key, 0),
)
for key, run in latest_by_binding.items()
)
def _session(context: ExternalProviderStateContext) -> Session:
if not isinstance(context.session, Session):
raise RuntimeError("Connectors provider state requires a database session.")
return context.session
def _tabular_state(
source: ConnectorTabularSource,
*,
observed_at: datetime,
) -> ExternalProviderRuntimeState:
active = source.status == "active"
return ExternalProviderRuntimeState(
provider_id=TABULAR_PROVIDER_ID,
binding_ref=f"connectors:tabular-source:{source.id}",
authority_mode="external_mirror",
observed_at=observed_at,
configured=True,
active=active,
health="healthy" if active else "inactive",
freshness="not_applicable",
conflict="not_applicable",
recovery="ready" if active else "not_applicable",
last_success_at=_aware(source.updated_at or source.created_at),
detail=(
"Immutable tabular snapshot is available."
if active
else "Immutable tabular snapshot is inactive."
),
metrics={
"row_count": int(source.row_count),
"byte_count": int(source.byte_count),
"schema_version": int(source.schema_version),
},
)
def _snapshot_counts(
session: Session,
*,
binding_keys: tuple[tuple[str, str, str], ...],
) -> dict[tuple[str, str, str], int]:
if not binding_keys:
return {}
tenant_ids = {item[0] for item in binding_keys}
rows = session.execute(
select(
ConnectorSanctionsSnapshot.tenant_id,
ConnectorSanctionsSnapshot.provider_id,
ConnectorSanctionsSnapshot.source_id,
func.count(ConnectorSanctionsSnapshot.id),
)
.where(ConnectorSanctionsSnapshot.tenant_id.in_(tenant_ids))
.group_by(
ConnectorSanctionsSnapshot.tenant_id,
ConnectorSanctionsSnapshot.provider_id,
ConnectorSanctionsSnapshot.source_id,
)
)
return {
(str(tenant_id), str(provider_id), str(source_id)): int(count)
for tenant_id, provider_id, source_id, count in rows
if (str(tenant_id), str(provider_id), str(source_id)) in binding_keys
}
def _sanctions_state(
run: ConnectorSanctionsAcquisitionRun,
*,
observed_at: datetime,
snapshot_count: int,
) -> ExternalProviderRuntimeState:
status = str(run.status)
success = status in {"succeeded", "success", "not_modified"}
running = status in {"running", "pending", "retry"}
has_snapshot = bool(run.snapshot_id) or snapshot_count > 0
health = "healthy" if success else "warning" if running else "error"
binding_digest = sha256(
f"{run.tenant_id}\0{run.provider_id}\0{run.source_id}".encode("utf-8")
).hexdigest()[:24]
return ExternalProviderRuntimeState(
provider_id=SANCTIONS_PROVIDER_ID,
binding_ref=f"connectors:sanctions-source:{binding_digest}",
authority_mode="external_mirror",
observed_at=observed_at,
configured=True,
active=True,
health=health,
freshness="unknown",
conflict="not_applicable",
recovery="ready" if success and has_snapshot else "attention",
last_success_at=_aware(run.finished_at) if success else None,
detail=(
"Latest sanctions acquisition completed."
if success
else "Sanctions acquisition is in progress."
if running
else "Latest sanctions acquisition failed; prior accepted snapshots remain separate evidence."
),
metrics={
"latest_status": status,
"attempt_count": int(run.attempt_count),
"accepted_snapshots": int(snapshot_count),
},
)
def _aware(value: datetime | None) -> datetime | None:
if value is None:
return None
return value.replace(tzinfo=UTC) if value.tzinfo is None else value.astimezone(UTC)
__all__ = [
"SANCTIONS_PROVIDER_ID",
"TABULAR_PROVIDER_ID",
"sanctions_provider_states",
"tabular_provider_states",
]
+381
View File
@@ -0,0 +1,381 @@
from __future__ import annotations
from dataclasses import dataclass
import hashlib
from typing import Any
from uuid import NAMESPACE_URL, uuid4, uuid5
from sqlalchemy.orm import Session, sessionmaker
from govoplan_core.core.recovery import (
RecoveryGuaranteeError,
RecoveryMode,
RecoveryPlan,
RecoveryStatus,
)
from govoplan_core.core.recovery_runtime import (
DurableRecoveryOperation,
RecoveryOperationBusy,
RecoveryOperationStateConflict,
begin_durable_recovery_operation,
)
from govoplan_core.core.runtime_coordination import process_runtime_identity
class ConnectorRecoveryError(RuntimeError):
pass
@dataclass(frozen=True, slots=True)
class ConnectorRecoveryDeclaration:
operation_type: str
mode: RecoveryMode
provider_mutation: bool
idempotency: str
verification: tuple[str, ...]
recovery: tuple[str, ...]
implemented: bool
CONNECTOR_RECOVERY_OPERATIONS = (
ConnectorRecoveryDeclaration(
operation_type="read-snapshot",
mode=RecoveryMode.ATOMIC,
provider_mutation=False,
idempotency=(
"Caller-supplied request keys replay a committed immutable snapshot; "
"otherwise each deliberate acquisition receives a generated key."
),
verification=(
"provider revision or conditional cursor is recorded before fetch",
"domain snapshot and terminal recovery checkpoint commit together",
"stored bytes and provider evidence are checksum verified",
),
recovery=(
"a stale running transaction is failed after its database transaction rolls back",
"a new deliberate acquisition may then use a new request key",
),
implemented=True,
),
ConnectorRecoveryDeclaration(
operation_type="external-mutation",
mode=RecoveryMode.FORWARD_RECOVERY,
provider_mutation=True,
idempotency="A stable caller key and canonical request digest are mandatory.",
verification=(
"record the remote revision and bounded provider result",
"verify the provider state before reporting success",
),
recovery=(
"unknown outcomes remain unresolved until provider-backed reconciliation",
"never retry the same remote effect solely to reconstruct local state",
),
implemented=False,
),
)
def connector_session_factory(session: Session) -> sessionmaker[Session]:
bind = session.get_bind()
if bind is None:
raise ConnectorRecoveryError("Connector recovery requires a bound database session")
return sessionmaker(bind=bind, expire_on_commit=False)
def _digest(value: str) -> str:
return hashlib.sha256(value.encode("utf-8")).hexdigest()
def _clean_key(value: str | None) -> str:
clean = str(value or "").strip()
if clean and len(clean) > 500:
raise ConnectorRecoveryError("Connector idempotency keys are limited to 500 characters")
return clean or str(uuid4())
def _stable_resource_id(
*,
tenant_id: str,
provider_id: str,
operation_type: str,
request_key: str,
) -> str:
return str(
uuid5(
NAMESPACE_URL,
f"govoplan:{tenant_id}:{provider_id}:{operation_type}:{request_key}",
)
)
@dataclass(slots=True)
class ConnectorReadSnapshotRecovery:
operation: DurableRecoveryOperation | None
operation_id: str
request_key: str
resource_id: str
replayed: bool
def commit_success(self, session: Session, *, evidence: dict[str, Any]) -> None:
if self.operation is None:
raise ConnectorRecoveryError("A replayed connector read cannot be committed again")
try:
self.operation.commit_atomic_success(session, evidence=evidence)
except Exception as exc:
raise ConnectorRecoveryError(
"The connector snapshot and recovery evidence did not commit atomically"
) from exc
def commit_failure(
self,
session: Session,
*,
summary: str,
evidence: dict[str, Any],
) -> None:
if self.operation is None:
raise ConnectorRecoveryError("A replayed connector read cannot be failed again")
try:
self.operation.commit_atomic_failure(
session,
summary=summary,
evidence=evidence,
)
except Exception as exc:
raise ConnectorRecoveryError(
"The connector failure evidence did not commit atomically"
) from exc
def fail_without_projection(
self,
*,
summary: str,
code: str,
) -> None:
if self.operation is None:
return
self.operation.fail(
summary=summary,
evidence={
"verified": True,
"checks": {
"provider_mutation": False,
"projection_committed": False,
"failure_code": code,
},
},
)
def begin_connector_read_snapshot(
session: Session,
*,
tenant_id: str,
provider_id: str,
idempotency_key: str | None,
source_revision: str | None,
cursor: str | None,
dry_run_evidence: dict[str, Any],
request_metadata: dict[str, Any] | None = None,
resource_type: str = "connector_sync_run",
) -> ConnectorReadSnapshotRecovery:
request_key = _clean_key(idempotency_key)
resource_id = _stable_resource_id(
tenant_id=tenant_id,
provider_id=provider_id,
operation_type="read-snapshot",
request_key=request_key,
)
request = {
"tenant_id": tenant_id,
"provider_id": provider_id,
"dry_run": dry_run_evidence,
"request_key_sha256": _digest(request_key),
**dict(request_metadata or {}),
}
try:
started = begin_durable_recovery_operation(
connector_session_factory(session),
identity=process_runtime_identity(),
module_id="connectors",
operation_type="read-snapshot",
idempotency_key=f"connector-read:{_digest(f'{tenant_id}:{provider_id}:{request_key}')}",
request=request,
recovery_plan=RecoveryPlan(
mode=RecoveryMode.ATOMIC,
preconditions=(
"the actor is authorized for the connector source",
"the provider request is read-only",
"the source revision, cursor, and dry-run decision are durable",
),
verification_steps=(
"validate the bounded provider response and source revision",
"commit the immutable snapshot and terminal checkpoint atomically",
"compare the stored content digest with the acquired bytes",
),
),
precondition_evidence={
"provider_id": provider_id,
"source_revision": source_revision,
"cursor_sha256": _digest(cursor) if cursor else None,
"dry_run": dry_run_evidence,
"provider_mutation": False,
},
lease_resource_key=f"connectors:read:{tenant_id}:{_digest(provider_id)[:40]}",
lease_ttl_seconds=15 * 60,
resource_type=resource_type,
resource_id=resource_id,
metadata={
"resources": ["postgresql", "external-provider"],
"provider_mutation": False,
"recovery_declaration": "read-snapshot",
},
)
except RecoveryOperationBusy as exc:
raise ConnectorRecoveryError(
"Another runtime is already acquiring this connector source"
) from exc
except RecoveryOperationStateConflict as exc:
raise ConnectorRecoveryError(
"This connector request is active or unresolved; reconcile it before retrying"
) from exc
except (RecoveryGuaranteeError, RuntimeError) as exc:
raise ConnectorRecoveryError(
"The connector recovery ledger is unavailable; the provider was not contacted"
) from exc
return ConnectorReadSnapshotRecovery(
operation=started.operation,
operation_id=started.operation_id,
request_key=request_key,
resource_id=resource_id,
replayed=started.replayed,
)
@dataclass(slots=True)
class ConnectorExternalMutationRecovery:
operation: DurableRecoveryOperation | None
operation_id: str
replayed: bool
def succeed(self, *, provider_evidence: dict[str, Any]) -> None:
if self.operation is not None:
self.operation.succeed(evidence=provider_evidence)
def reject(self, *, summary: str, provider_code: str) -> None:
if self.operation is not None:
self.operation.reject(
summary=summary,
evidence={
"verified": True,
"checks": {"provider_rejection": provider_code},
},
)
def outcome_unknown(self, *, summary: str, provider_code: str) -> None:
if self.operation is not None:
self.operation.unresolved(
status=RecoveryStatus.OUTCOME_UNKNOWN,
summary=summary,
evidence={"effect_started": True, "provider_code": provider_code},
failure_summary="Inspect provider state before any retry",
)
def begin_connector_external_mutation(
session: Session,
*,
tenant_id: str,
provider_id: str,
idempotency_key: str,
request_sha256: str,
source_revision: str | None,
cursor: str | None,
dry_run_evidence: dict[str, Any],
resource_type: str,
resource_id: str,
) -> ConnectorExternalMutationRecovery:
if not str(idempotency_key or "").strip():
raise ConnectorRecoveryError("External connector mutations require an idempotency key")
request_key = _clean_key(idempotency_key)
if len(request_sha256) != 64 or any(
character not in "0123456789abcdefABCDEF" for character in request_sha256
):
raise ConnectorRecoveryError("External connector mutations require a SHA-256 request digest")
try:
started = begin_durable_recovery_operation(
connector_session_factory(session),
identity=process_runtime_identity(),
module_id="connectors",
operation_type="external-mutation",
idempotency_key=f"connector-write:{_digest(f'{tenant_id}:{provider_id}:{request_key}')}",
request={
"tenant_id": tenant_id,
"provider_id": provider_id,
"request_sha256": request_sha256,
"source_revision": source_revision,
"cursor_sha256": _digest(cursor) if cursor else None,
"dry_run": dry_run_evidence,
},
recovery_plan=RecoveryPlan(
mode=RecoveryMode.FORWARD_RECOVERY,
preconditions=(
"the actor and effective connector policy authorize the mutation",
"a stable idempotency key and canonical request digest are present",
"the dry-run and source revision evidence are durable",
),
forward_recovery_steps=(
"inspect provider state without repeating the mutation",
"record whether the provider accepted the requested revision",
"retry only under a new deliberate key when absence is proven",
),
verification_steps=(
"compare provider identity and revision with the canonical request",
"verify the consuming domain state independently",
),
),
precondition_evidence={
"request_sha256": request_sha256,
"source_revision": source_revision,
"cursor_sha256": _digest(cursor) if cursor else None,
"dry_run": dry_run_evidence,
"provider_mutation": True,
},
lease_resource_key=(
f"connectors:write:{tenant_id}:{_digest(provider_id)[:24]}:"
f"{_digest(resource_id)[:24]}"
),
lease_ttl_seconds=15 * 60,
resource_type=resource_type,
resource_id=resource_id,
metadata={
"resources": ["postgresql", "queue", "external-provider"],
"provider_mutation": True,
"recovery_declaration": "external-mutation",
},
)
except (RecoveryOperationBusy, RecoveryOperationStateConflict) as exc:
raise ConnectorRecoveryError(
"This external connector effect is active or unresolved"
) from exc
except (RecoveryGuaranteeError, RuntimeError) as exc:
raise ConnectorRecoveryError(
"The connector recovery ledger is unavailable; no external mutation started"
) from exc
return ConnectorExternalMutationRecovery(
operation=started.operation,
operation_id=started.operation_id,
replayed=started.replayed,
)
__all__ = [
"CONNECTOR_RECOVERY_OPERATIONS",
"ConnectorExternalMutationRecovery",
"ConnectorReadSnapshotRecovery",
"ConnectorRecoveryDeclaration",
"ConnectorRecoveryError",
"begin_connector_external_mutation",
"begin_connector_read_snapshot",
"connector_session_factory",
]
+689
View File
@@ -0,0 +1,689 @@
from __future__ import annotations
from dataclasses import asdict
import hashlib
from typing import Annotated
from fastapi import APIRouter, Depends, Header, HTTPException, Query, Response, status
from sqlalchemy.orm import Session
from govoplan_core.audit.logging import audit_event
from govoplan_core.auth import ApiPrincipal, get_api_principal, has_scope
from govoplan_core.core.tabular_sources import (
TabularReadRequest,
TabularSnapshotInput,
TabularSource,
TabularSourceAccessError,
TabularSourceError,
TabularSourceNotFoundError,
TabularSourceUnavailableError,
)
from govoplan_core.core.feeds import (
FeedCapabilityError,
FeedEntry,
FeedRenderRequest,
)
from govoplan_core.core.sanctions import SanctionsSnapshotReference
from govoplan_core.db.session import get_session
from govoplan_connectors.backend.schemas import (
FeedAcquireRequest,
FeedDocumentResponse,
FeedImportRequest,
FeedRenderPayload,
SanctionsAcquisitionRunListResponse,
SanctionsAcquisitionRunResponse,
SanctionsRefreshResponse,
SanctionsSnapshotListResponse,
SanctionsSnapshotResponse,
SanctionsSourceListResponse,
SanctionsSourceResponse,
SnapshotCreateRequest,
TabularColumnResponse,
TabularHealthResponse,
TabularPreviewDiagnosticResponse,
TabularPushdownResponse,
TabularSourceDeleteResponse,
TabularSourceListResponse,
TabularSourcePreviewResponse,
TabularSourceResponse,
)
from govoplan_connectors.backend.feeds import ConnectorFeedProvider, feed_rows
from govoplan_connectors.backend.recovery import (
ConnectorRecoveryError,
begin_connector_read_snapshot,
)
from govoplan_connectors.backend.sanctions_sources import (
SANCTIONS_READ_SCOPE,
SANCTIONS_REFRESH_SCOPE,
SanctionsSourceAccessError,
SanctionsSourceError,
SanctionsSourceNotFoundError,
SqlSanctionsSnapshotProvider,
)
from govoplan_connectors.backend.tabular_sources import (
ADMIN_SCOPE,
READ_SCOPE,
WRITE_SCOPE,
SqlTabularSourceProvider,
parse_csv_snapshot,
)
router = APIRouter(prefix="/connectors", tags=["connectors"])
provider = SqlTabularSourceProvider()
sanctions_provider = SqlSanctionsSnapshotProvider()
feed_transport = ConnectorFeedProvider()
def _require_any_scope(principal: ApiPrincipal, *scopes: str) -> None:
if any(has_scope(principal, scope) for scope in scopes):
return
raise HTTPException(
status_code=status.HTTP_403_FORBIDDEN,
detail=f"Missing one of the required scopes: {', '.join(scopes)}",
)
def _http_error(exc: TabularSourceError) -> HTTPException:
if isinstance(exc, TabularSourceNotFoundError):
return HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail=str(exc))
if isinstance(exc, TabularSourceAccessError):
return HTTPException(status_code=status.HTTP_403_FORBIDDEN, detail=str(exc))
if isinstance(exc, TabularSourceUnavailableError):
return HTTPException(
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
detail=str(exc),
)
return HTTPException(status_code=status.HTTP_422_UNPROCESSABLE_CONTENT, detail=str(exc))
def _sanctions_http_error(
exc: SanctionsSourceError,
) -> HTTPException:
if isinstance(exc, SanctionsSourceNotFoundError):
return HTTPException(
status_code=status.HTTP_404_NOT_FOUND,
detail=str(exc),
)
if isinstance(exc, SanctionsSourceAccessError):
return HTTPException(
status_code=status.HTTP_403_FORBIDDEN,
detail=str(exc),
)
return HTTPException(
status_code=status.HTTP_422_UNPROCESSABLE_CONTENT,
detail=str(exc),
)
def _feed_http_error(exc: FeedCapabilityError) -> HTTPException:
return HTTPException(
status_code=status.HTTP_422_UNPROCESSABLE_CONTENT,
detail=str(exc),
)
def _recovery_http_error(exc: ConnectorRecoveryError) -> HTTPException:
detail = str(exc)
return HTTPException(
status_code=(
status.HTTP_409_CONFLICT
if "already" in detail.casefold() or "active" in detail.casefold()
else status.HTTP_503_SERVICE_UNAVAILABLE
),
detail=detail,
)
@router.post("/feeds/preview", response_model=FeedDocumentResponse)
def api_preview_feed(
payload: FeedAcquireRequest,
principal: ApiPrincipal = Depends(get_api_principal),
) -> FeedDocumentResponse:
_require_any_scope(principal, READ_SCOPE, ADMIN_SCOPE)
try:
document = feed_transport.fetch(
payload.url,
max_entries=payload.max_entries,
)
except FeedCapabilityError as exc:
raise _feed_http_error(exc) from exc
return FeedDocumentResponse.model_validate(asdict(document))
@router.post(
"/feeds/import",
response_model=TabularSourceResponse,
status_code=status.HTTP_201_CREATED,
)
def api_import_feed_snapshot(
payload: FeedImportRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
idempotency_key: Annotated[
str | None,
Header(alias="Idempotency-Key", max_length=500),
] = None,
) -> TabularSourceResponse:
_require_any_scope(principal, WRITE_SCOPE, ADMIN_SCOPE)
try:
recovery = begin_connector_read_snapshot(
session,
tenant_id=principal.tenant_id,
provider_id="connectors.feed_snapshot",
idempotency_key=idempotency_key,
source_revision=None,
cursor=None,
dry_run_evidence={
"performed": False,
"reason": "read-only acquisition into an immutable snapshot",
},
request_metadata={
"source_url_sha256": hashlib.sha256(
payload.url.encode("utf-8")
).hexdigest(),
"source_name": payload.source_name,
"max_entries": payload.max_entries,
},
resource_type="connector_tabular_source",
)
except ConnectorRecoveryError as exc:
raise _recovery_http_error(exc) from exc
if recovery.replayed:
try:
source = provider.get_source(
session,
principal,
source_ref=f"snapshot:{recovery.resource_id}",
)
except TabularSourceError as exc:
raise _http_error(exc) from exc
return _source_response(source)
try:
document = feed_transport.fetch(
payload.url,
max_entries=payload.max_entries,
)
source = provider.create_snapshot(
session,
principal,
snapshot=TabularSnapshotInput(
name=payload.name,
source_name=payload.source_name,
description=payload.description or document.description,
rows=feed_rows(document),
metadata={
"import_format": document.format,
"feed": {
"source_url": document.source_url,
"home_url": document.home_url,
"acquired_at": (
document.acquired_at.isoformat()
if document.acquired_at
else None
),
"fresh_until": (
document.fresh_until.isoformat()
if document.fresh_until
else None
),
"etag": document.etag,
"last_modified": document.last_modified,
"content_type": document.content_type,
"sha256": document.sha256,
},
},
),
source_id=recovery.resource_id,
)
except (FeedCapabilityError, TabularSourceError) as exc:
session.rollback()
recovery.fail_without_projection(
summary="The read-only feed import failed before a snapshot committed",
code=exc.__class__.__name__,
)
if isinstance(exc, FeedCapabilityError):
raise _feed_http_error(exc) from exc
raise _http_error(exc) from exc
audit_event(
session,
tenant_id=principal.tenant_id,
user_id=getattr(principal.user, "id", None),
api_key_id=principal.api_key_id,
action="connectors.feed_snapshot.created",
object_type="connector_tabular_source",
object_id=source.ref,
details={
"source_url": document.source_url,
"format": document.format,
"sha256": document.sha256,
"row_count": source.row_count,
},
)
try:
recovery.commit_success(
session,
evidence={
"verified": True,
"checks": {
"snapshot_ref": source.ref,
"snapshot_fingerprint": source.fingerprint,
"feed_sha256": document.sha256,
"row_count": source.row_count,
"provider_mutation": False,
},
},
)
except ConnectorRecoveryError as exc:
raise _recovery_http_error(exc) from exc
return _source_response(source)
@router.post("/feeds/render")
def api_render_feed(
payload: FeedRenderPayload,
principal: ApiPrincipal = Depends(get_api_principal),
) -> Response:
_require_any_scope(principal, READ_SCOPE, ADMIN_SCOPE)
try:
rendered = feed_transport.render(
FeedRenderRequest(
format=payload.format,
title=payload.title,
feed_url=payload.feed_url,
home_url=payload.home_url,
description=payload.description,
language=payload.language,
entries=tuple(
FeedEntry(**item.model_dump()) for item in payload.entries
),
allowed_visibilities=frozenset(payload.allowed_visibilities),
)
)
except FeedCapabilityError as exc:
raise _feed_http_error(exc) from exc
return Response(
content=rendered.body,
media_type=rendered.content_type,
headers={
"X-GovOPlaN-Feed-Included": str(rendered.included_entries),
"X-GovOPlaN-Feed-Excluded": str(rendered.excluded_entries),
},
)
@router.get("/tabular-sources", response_model=TabularSourceListResponse)
def api_list_tabular_sources(
query: str = Query(default="", max_length=200),
limit: int = Query(default=100, ge=1, le=100),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
) -> TabularSourceListResponse:
_require_any_scope(principal, READ_SCOPE, ADMIN_SCOPE)
sources = provider.list_sources(
session,
principal,
query=query,
limit=limit,
)
return TabularSourceListResponse(sources=[_source_response(source) for source in sources])
@router.post(
"/tabular-sources/snapshots",
response_model=TabularSourceResponse,
status_code=status.HTTP_201_CREATED,
)
def api_create_tabular_snapshot(
payload: SnapshotCreateRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
) -> TabularSourceResponse:
_require_any_scope(principal, WRITE_SCOPE, ADMIN_SCOPE)
try:
rows = (
tuple(payload.rows or ())
if payload.format == "json"
else parse_csv_snapshot(payload.csv_text or "", delimiter=payload.delimiter)
)
source = provider.create_snapshot(
session,
principal,
snapshot=TabularSnapshotInput(
name=payload.name,
source_name=payload.source_name,
description=payload.description,
rows=rows,
metadata={"import_format": payload.format},
),
)
except TabularSourceError as exc:
raise _http_error(exc) from exc
audit_event(
session,
tenant_id=principal.tenant_id,
user_id=getattr(principal.user, "id", None),
api_key_id=principal.api_key_id,
action="connectors.tabular_snapshot.created",
object_type="connector_tabular_source",
object_id=source.ref,
details={
"provider": source.provider,
"source_name": source.source_name,
"fingerprint": source.fingerprint,
"row_count": source.row_count,
},
)
session.commit()
return _source_response(source)
@router.get(
"/tabular-sources/{source_id}/preview",
response_model=TabularSourcePreviewResponse,
)
def api_preview_tabular_source(
source_id: str,
limit: int = Query(default=100, ge=1, le=500),
offset: int = Query(default=0, ge=0),
max_bytes: int = Query(default=1_000_000, ge=2, le=5_000_000),
timeout_ms: int = Query(default=2_000, ge=1, le=10_000),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
) -> TabularSourcePreviewResponse:
_require_any_scope(principal, READ_SCOPE, ADMIN_SCOPE)
try:
result = provider.read_source(
session,
principal,
request=TabularReadRequest(
source_ref=f"snapshot:{source_id}",
limit=limit,
offset=offset,
max_bytes=max_bytes,
timeout_ms=timeout_ms,
),
)
except TabularSourceError as exc:
raise _http_error(exc) from exc
return TabularSourcePreviewResponse(
source=_source_response(result.source),
rows=[dict(row) for row in result.rows],
total_rows=result.total_rows,
truncated=result.truncated,
returned_bytes=result.returned_bytes,
elapsed_ms=result.elapsed_ms,
effective_row_limit=result.effective_row_limit,
effective_byte_limit=result.effective_byte_limit,
effective_timeout_ms=result.effective_timeout_ms,
diagnostics=[
TabularPreviewDiagnosticResponse(
severity=item.severity,
code=item.code,
message=item.message,
details=dict(item.details),
)
for item in result.diagnostics
],
)
@router.delete(
"/tabular-sources/{source_id}",
response_model=TabularSourceDeleteResponse,
)
def api_delete_tabular_source(
source_id: str,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
) -> TabularSourceDeleteResponse:
_require_any_scope(principal, WRITE_SCOPE, ADMIN_SCOPE)
source_ref = f"snapshot:{source_id}"
try:
source = provider.delete_snapshot(
session,
principal,
source_ref=source_ref,
)
except TabularSourceError as exc:
raise _http_error(exc) from exc
audit_event(
session,
tenant_id=principal.tenant_id,
user_id=getattr(principal.user, "id", None),
api_key_id=principal.api_key_id,
action="connectors.tabular_snapshot.deleted",
object_type="connector_tabular_source",
object_id=source_ref,
details={"source_name": source.source_name, "fingerprint": source.fingerprint},
)
session.commit()
return TabularSourceDeleteResponse(deleted=True, source_ref=source_ref)
@router.get(
"/sanctions/sources",
response_model=SanctionsSourceListResponse,
)
def api_list_sanctions_sources(
principal: ApiPrincipal = Depends(get_api_principal),
) -> SanctionsSourceListResponse:
_require_any_scope(
principal,
SANCTIONS_READ_SCOPE,
SANCTIONS_REFRESH_SCOPE,
ADMIN_SCOPE,
)
return SanctionsSourceListResponse(
sources=[
SanctionsSourceResponse.model_validate(
source,
from_attributes=True,
)
for source in sanctions_provider.available_sources()
]
)
@router.post(
"/sanctions/sources/{provider_id}/refresh",
response_model=SanctionsRefreshResponse,
)
def api_refresh_sanctions_source(
provider_id: str,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
idempotency_key: Annotated[
str | None,
Header(alias="Idempotency-Key", max_length=500),
] = None,
) -> SanctionsRefreshResponse:
_require_any_scope(
principal,
SANCTIONS_REFRESH_SCOPE,
ADMIN_SCOPE,
)
try:
result = sanctions_provider.refresh_source(
session,
principal,
provider_id=provider_id,
idempotency_key=idempotency_key,
)
except ConnectorRecoveryError as exc:
raise _recovery_http_error(exc) from exc
except SanctionsSourceError as exc:
raise _sanctions_http_error(exc) from exc
audit_event(
session,
tenant_id=principal.tenant_id,
user_id=getattr(principal.user, "id", None),
api_key_id=principal.api_key_id,
action="connectors.sanctions_source.refreshed",
object_type="connector_sanctions_acquisition_run",
object_id=result.run_id,
details={
"provider_id": provider_id,
"status": result.status,
"snapshot_ref": (
result.snapshot.ref
if result.snapshot is not None
else None
),
},
)
session.commit()
return SanctionsRefreshResponse(
run_id=result.run_id,
provider_id=result.provider_id,
status=result.status,
snapshot=(
_sanctions_snapshot_response(result.snapshot)
if result.snapshot is not None
else None
),
error=result.error,
)
@router.get(
"/sanctions/snapshots",
response_model=SanctionsSnapshotListResponse,
)
def api_list_sanctions_snapshots(
limit: int = Query(default=100, ge=1, le=500),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
) -> SanctionsSnapshotListResponse:
_require_any_scope(
principal,
SANCTIONS_READ_SCOPE,
ADMIN_SCOPE,
)
try:
snapshots = sanctions_provider.list_snapshots(
session,
principal,
limit=limit,
)
except SanctionsSourceError as exc:
raise _sanctions_http_error(exc) from exc
return SanctionsSnapshotListResponse(
snapshots=[
_sanctions_snapshot_response(item)
for item in snapshots
]
)
@router.get(
"/sanctions/runs",
response_model=SanctionsAcquisitionRunListResponse,
)
def api_list_sanctions_runs(
limit: int = Query(default=100, ge=1, le=500),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
) -> SanctionsAcquisitionRunListResponse:
_require_any_scope(
principal,
SANCTIONS_READ_SCOPE,
ADMIN_SCOPE,
)
try:
runs = sanctions_provider.list_runs(
session,
principal,
limit=limit,
)
except SanctionsSourceError as exc:
raise _sanctions_http_error(exc) from exc
return SanctionsAcquisitionRunListResponse(
runs=[
SanctionsAcquisitionRunResponse.model_validate(
item,
from_attributes=True,
)
for item in runs
]
)
def _source_response(source: TabularSource) -> TabularSourceResponse:
return TabularSourceResponse(
ref=source.ref,
provider=source.provider,
source_name=source.source_name,
name=source.name,
description=source.description,
columns=[
TabularColumnResponse(
name=column.name,
data_type=column.data_type,
nullable=column.nullable,
)
for column in source.schema
],
schema_version=source.schema_version,
fingerprint=source.fingerprint,
row_count=source.row_count,
byte_count=source.byte_count,
updated_at=source.updated_at.isoformat() if source.updated_at else None,
capabilities=list(source.capabilities),
metadata=dict(source.metadata),
source_mode=source.source_mode,
pushdown=TabularPushdownResponse(
projections=source.pushdown.projections,
pagination=source.pushdown.pagination,
filters=list(source.pushdown.filters),
aggregations=list(source.pushdown.aggregations),
sorting=list(source.pushdown.sorting),
),
health=TabularHealthResponse(
status=source.health.status,
code=source.health.code,
summary=source.health.summary,
checked_at=(
source.health.checked_at.isoformat()
if source.health.checked_at
else None
),
details=dict(source.health.details),
),
)
def _sanctions_snapshot_response(
snapshot: SanctionsSnapshotReference,
) -> SanctionsSnapshotResponse:
return SanctionsSnapshotResponse.model_validate(
{
"ref": snapshot.ref,
"provider_id": snapshot.provider_id,
"publisher": snapshot.publisher,
"jurisdiction": snapshot.jurisdiction,
"list_type": snapshot.list_type,
"source_id": snapshot.source_id,
"source_version": snapshot.source_version,
"publication_at": snapshot.publication_at,
"effective_at": snapshot.effective_at,
"acquired_at": snapshot.acquired_at,
"content_type": snapshot.content_type,
"byte_count": snapshot.byte_count,
"sha256": snapshot.sha256,
"parser_version": snapshot.parser_version,
"raw_evidence_ref": snapshot.raw_evidence_ref,
"connector_run_id": snapshot.connector_run_id,
"signature_evidence": dict(
snapshot.signature_evidence
),
"licence_notes": snapshot.licence_notes,
"trust_notes": snapshot.trust_notes,
"transport_evidence": dict(
snapshot.transport_evidence
),
}
)
__all__ = ["router"]
File diff suppressed because it is too large Load Diff
+258
View File
@@ -0,0 +1,258 @@
from __future__ import annotations
from datetime import datetime
from typing import Any, Literal
from pydantic import BaseModel, Field, model_validator
class FeedAcquireRequest(BaseModel):
url: str = Field(min_length=1, max_length=2000)
max_entries: int = Field(default=2_000, ge=1, le=10_000)
class FeedImportRequest(FeedAcquireRequest):
name: str = Field(min_length=1, max_length=300)
source_name: str = Field(
min_length=1,
max_length=120,
pattern=r"^[A-Za-z_][A-Za-z0-9_]*$",
)
description: str | None = Field(default=None, max_length=4000)
class FeedEntryPayload(BaseModel):
id: str = Field(min_length=1, max_length=2000)
title: str = Field(min_length=1, max_length=1000)
url: str | None = Field(default=None, max_length=2000)
summary: str | None = None
content: str | None = None
author: str | None = Field(default=None, max_length=500)
published_at: datetime | None = None
updated_at: datetime | None = None
categories: list[str] = Field(default_factory=list, max_length=100)
enclosures: list[dict[str, Any]] = Field(default_factory=list, max_length=100)
visibility: Literal["public", "tenant", "private"] = "public"
metadata: dict[str, Any] = Field(default_factory=dict)
class FeedDocumentResponse(BaseModel):
format: Literal["rss", "atom"]
title: str
source_url: str
description: str | None = None
home_url: str | None = None
language: str | None = None
updated_at: datetime | None = None
acquired_at: datetime | None = None
fresh_until: datetime | None = None
etag: str | None = None
last_modified: str | None = None
content_type: str | None = None
sha256: str
entries: list[FeedEntryPayload]
metadata: dict[str, Any] = Field(default_factory=dict)
class FeedRenderPayload(BaseModel):
format: Literal["rss", "atom"]
title: str = Field(min_length=1, max_length=1000)
feed_url: str = Field(min_length=1, max_length=2000)
home_url: str = Field(min_length=1, max_length=2000)
description: str | None = None
language: str | None = Field(default=None, max_length=100)
entries: list[FeedEntryPayload] = Field(default_factory=list, max_length=10_000)
allowed_visibilities: list[Literal["public", "tenant", "private"]] = Field(
default_factory=lambda: ["public"],
max_length=3,
)
class SnapshotCreateRequest(BaseModel):
name: str = Field(min_length=1, max_length=300)
source_name: str = Field(
min_length=1,
max_length=120,
pattern=r"^[A-Za-z_][A-Za-z0-9_]*$",
)
description: str | None = Field(default=None, max_length=4000)
format: Literal["json", "csv"] = "json"
rows: list[dict[str, Any]] | None = Field(default=None, max_length=10_000)
csv_text: str | None = Field(default=None, max_length=5_000_000)
delimiter: Literal[",", ";", "\t", "|"] = ","
@model_validator(mode="after")
def validate_payload(self) -> "SnapshotCreateRequest":
if self.format == "json" and self.rows is None:
raise ValueError("JSON snapshots require rows.")
if self.format == "json" and self.csv_text is not None:
raise ValueError("JSON snapshots cannot include CSV text.")
if self.format == "csv" and not self.csv_text:
raise ValueError("CSV snapshots require CSV text.")
if self.format == "csv" and self.rows is not None:
raise ValueError("CSV snapshots cannot include JSON rows.")
return self
class TabularColumnResponse(BaseModel):
name: str
data_type: str
nullable: bool
class TabularPushdownResponse(BaseModel):
projections: bool
pagination: bool
filters: list[str]
aggregations: list[str]
sorting: list[str]
class TabularHealthResponse(BaseModel):
status: Literal["healthy", "warning", "error", "unknown"]
code: str
summary: str
checked_at: str | None
details: dict[str, Any]
class TabularPreviewDiagnosticResponse(BaseModel):
severity: Literal["info", "warning", "error"]
code: str
message: str
details: dict[str, Any]
class TabularSourceResponse(BaseModel):
ref: str
provider: str
source_name: str
name: str
description: str | None
columns: list[TabularColumnResponse]
schema_version: str
fingerprint: str
row_count: int | None
byte_count: int | None
updated_at: str | None
capabilities: list[str]
metadata: dict[str, Any]
source_mode: Literal["live", "cached", "file_backed", "static"]
pushdown: TabularPushdownResponse
health: TabularHealthResponse
class TabularSourceListResponse(BaseModel):
sources: list[TabularSourceResponse]
class TabularSourcePreviewResponse(BaseModel):
source: TabularSourceResponse
rows: list[dict[str, Any]]
total_rows: int
truncated: bool
returned_bytes: int
elapsed_ms: int
effective_row_limit: int
effective_byte_limit: int
effective_timeout_ms: int
diagnostics: list[TabularPreviewDiagnosticResponse]
class TabularSourceDeleteResponse(BaseModel):
deleted: bool
source_ref: str
class SanctionsSourceResponse(BaseModel):
provider_id: str
publisher: str
jurisdiction: str
list_type: str
source_id: str
source_url: str | None
parser_version: str
licence_notes: str
trust_notes: str
class SanctionsSourceListResponse(BaseModel):
sources: list[SanctionsSourceResponse]
class SanctionsSnapshotResponse(BaseModel):
ref: str
provider_id: str
publisher: str
jurisdiction: str
list_type: str
source_id: str
source_version: str
publication_at: datetime | None
effective_at: datetime | None
acquired_at: datetime
content_type: str
byte_count: int
sha256: str
parser_version: str
raw_evidence_ref: str
connector_run_id: str
signature_evidence: dict[str, Any]
licence_notes: str | None
trust_notes: str | None
transport_evidence: dict[str, Any]
class SanctionsSnapshotListResponse(BaseModel):
snapshots: list[SanctionsSnapshotResponse]
class SanctionsAcquisitionRunResponse(BaseModel):
id: str
provider_id: str
source_id: str
status: str
attempt_count: int
request_evidence: dict[str, Any]
response_evidence: dict[str, Any]
started_at: datetime
finished_at: datetime | None
snapshot_id: str | None
error: str | None
class SanctionsAcquisitionRunListResponse(BaseModel):
runs: list[SanctionsAcquisitionRunResponse]
class SanctionsRefreshResponse(BaseModel):
run_id: str
provider_id: str
status: str
snapshot: SanctionsSnapshotResponse | None
error: str | None
__all__ = [
"FeedAcquireRequest",
"FeedDocumentResponse",
"FeedEntryPayload",
"FeedImportRequest",
"FeedRenderPayload",
"SnapshotCreateRequest",
"SanctionsAcquisitionRunListResponse",
"SanctionsAcquisitionRunResponse",
"SanctionsRefreshResponse",
"SanctionsSnapshotListResponse",
"SanctionsSnapshotResponse",
"SanctionsSourceListResponse",
"SanctionsSourceResponse",
"TabularColumnResponse",
"TabularHealthResponse",
"TabularPreviewDiagnosticResponse",
"TabularPushdownResponse",
"TabularSourceDeleteResponse",
"TabularSourceListResponse",
"TabularSourcePreviewResponse",
"TabularSourceResponse",
]
@@ -0,0 +1,498 @@
from __future__ import annotations
import hashlib
import json
import time
from collections.abc import Mapping, Sequence
from datetime import datetime
from decimal import Decimal
from typing import Any, Callable
from sqlalchemy import or_, select
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal, has_scope
from govoplan_core.core.tabular_sources import (
TabularColumn,
TabularPreviewDiagnostic,
TabularPushdown,
TabularReadRequest,
TabularReadResult,
TabularSnapshotInput,
TabularSource,
TabularSourceAccessError,
TabularSourceNotFoundError,
TabularSourceHealth,
TabularSourceUnavailableError,
TabularSourceValidationError,
parse_tabular_csv,
)
from govoplan_core.db.base import utcnow
from govoplan_connectors.backend.db.models import ConnectorTabularSource
READ_SCOPE = "connectors:source:read"
WRITE_SCOPE = "connectors:source:write"
ADMIN_SCOPE = "connectors:source:admin"
MAX_SNAPSHOT_ROWS = 10_000
MAX_SNAPSHOT_BYTES = 5_000_000
MAX_READ_ROWS = 500
MAX_READ_BYTES = 1_000_000
MAX_READ_TIMEOUT_MS = 2_000
class SqlTabularSourceProvider:
def __init__(self, *, clock: Callable[[], float] = time.monotonic) -> None:
self._clock = clock
def list_sources(
self,
session: object,
principal: object,
*,
query: str = "",
limit: int = 100,
) -> Sequence[TabularSource]:
db, api_principal = _context(session, principal, READ_SCOPE)
normalized_query = str(query or "").strip()
statement = select(ConnectorTabularSource).where(
ConnectorTabularSource.tenant_id == api_principal.tenant_id,
ConnectorTabularSource.deleted_at.is_(None),
ConnectorTabularSource.status == "active",
)
if normalized_query:
pattern = f"%{_escape_like(normalized_query)}%"
statement = statement.where(
or_(
ConnectorTabularSource.name.ilike(pattern, escape="\\"),
ConnectorTabularSource.source_name.ilike(pattern, escape="\\"),
)
)
statement = statement.order_by(
ConnectorTabularSource.updated_at.desc(),
ConnectorTabularSource.name,
).limit(max(1, min(int(limit), 100)))
return tuple(_source_dto(item) for item in db.scalars(statement))
def get_source(
self,
session: object,
principal: object,
*,
source_ref: str,
) -> TabularSource | None:
db, api_principal = _context(session, principal, READ_SCOPE)
item = _source_record(
db,
tenant_id=api_principal.tenant_id,
source_ref=source_ref,
)
return _source_dto(item) if item is not None else None
def read_source(
self,
session: object,
principal: object,
*,
request: TabularReadRequest,
) -> TabularReadResult:
db, api_principal = _context(session, principal, READ_SCOPE)
item = _source_record(
db,
tenant_id=api_principal.tenant_id,
source_ref=request.source_ref,
)
if item is None:
raise TabularSourceNotFoundError("Tabular source not found.")
if request.expected_fingerprint and request.expected_fingerprint != item.fingerprint:
raise TabularSourceValidationError(
"The source fingerprint changed; refresh the source node before running it."
)
started = self._clock()
limit = max(1, min(int(request.limit), MAX_READ_ROWS))
byte_limit = max(2, min(int(request.max_bytes), MAX_READ_BYTES))
timeout_ms = max(1, min(int(request.timeout_ms), MAX_READ_TIMEOUT_MS))
offset = max(0, int(request.offset))
diagnostics: list[TabularPreviewDiagnostic] = []
if limit != request.limit:
diagnostics.append(
_preview_diagnostic(
"preview.row_limit_tightened",
"The provider tightened the requested row limit.",
requested=request.limit,
effective=limit,
)
)
if byte_limit != request.max_bytes:
diagnostics.append(
_preview_diagnostic(
"preview.byte_limit_tightened",
"The provider tightened the requested byte limit.",
requested=request.max_bytes,
effective=byte_limit,
)
)
if timeout_ms != request.timeout_ms:
diagnostics.append(
_preview_diagnostic(
"preview.timeout_tightened",
"The provider tightened the requested time limit.",
requested=request.timeout_ms,
effective=timeout_ms,
)
)
selected_columns = tuple(dict.fromkeys(request.columns))
known_columns = {column["name"] for column in item.schema_}
unknown_columns = [column for column in selected_columns if column not in known_columns]
if unknown_columns:
raise TabularSourceValidationError(
f"Unknown source columns: {', '.join(unknown_columns)}"
)
rows: list[dict[str, object]] = []
returned_bytes = 2
stopped_for = ""
for row in item.rows[offset:]:
if len(rows) >= limit:
stopped_for = "rows"
break
elapsed_ms = int(max(0.0, self._clock() - started) * 1_000)
if elapsed_ms >= timeout_ms:
if not rows:
raise TabularSourceUnavailableError(
"Tabular source preview exceeded its time budget."
)
stopped_for = "time"
break
selected = {
key: value
for key, value in row.items()
if not selected_columns or key in selected_columns
}
row_bytes = len(
json.dumps(
selected,
sort_keys=True,
separators=(",", ":"),
default=str,
).encode("utf-8")
)
additional_bytes = row_bytes + (1 if rows else 0)
if returned_bytes + additional_bytes > byte_limit:
if not rows:
raise TabularSourceValidationError(
"A single source row exceeds the preview byte limit."
)
stopped_for = "bytes"
break
rows.append(selected)
returned_bytes += additional_bytes
elapsed_ms = int(max(0.0, self._clock() - started) * 1_000)
if stopped_for:
labels = {
"rows": ("preview.row_limit_reached", "row"),
"bytes": ("preview.byte_limit_reached", "byte"),
"time": ("preview.timeout_reached", "time"),
}
code, label = labels[stopped_for]
diagnostics.append(
TabularPreviewDiagnostic(
severity="warning",
code=code,
message=f"The preview stopped at its effective {label} limit.",
)
)
return TabularReadResult(
source=_source_dto(item),
rows=tuple(rows),
total_rows=item.row_count,
truncated=offset + len(rows) < item.row_count,
returned_bytes=returned_bytes,
elapsed_ms=elapsed_ms,
effective_row_limit=limit,
effective_byte_limit=byte_limit,
effective_timeout_ms=timeout_ms,
diagnostics=tuple(diagnostics),
)
def create_snapshot(
self,
session: object,
principal: object,
*,
snapshot: TabularSnapshotInput,
source_id: str | None = None,
) -> TabularSource:
db, api_principal = _context(session, principal, WRITE_SCOPE)
name = snapshot.name.strip()
source_name = snapshot.source_name.strip()
if not name:
raise TabularSourceValidationError("Snapshot name is required.")
if not source_name:
raise TabularSourceValidationError("Snapshot source name is required.")
if len(snapshot.rows) > MAX_SNAPSHOT_ROWS:
raise TabularSourceValidationError(
f"Snapshots are limited to {MAX_SNAPSHOT_ROWS:,} rows."
)
rows = [_json_row(row) for row in snapshot.rows]
encoded = json.dumps(rows, sort_keys=True, separators=(",", ":"), default=str).encode("utf-8")
if len(encoded) > MAX_SNAPSHOT_BYTES:
raise TabularSourceValidationError(
f"Snapshots are limited to {MAX_SNAPSHOT_BYTES // 1_000_000} MB."
)
existing = db.scalar(
select(ConnectorTabularSource.id).where(
ConnectorTabularSource.tenant_id == api_principal.tenant_id,
ConnectorTabularSource.source_name == source_name,
)
)
if existing is not None:
raise TabularSourceValidationError(
f"A tabular source named {source_name!r} already exists."
)
schema = infer_schema(rows)
fingerprint = snapshot_fingerprint(rows, schema)
actor_id = _actor_id(api_principal)
item = ConnectorTabularSource(
tenant_id=api_principal.tenant_id,
provider="snapshot",
source_name=source_name,
name=name,
description=_clean_optional(snapshot.description),
status="active",
schema_version=1,
schema_=[_column_payload(column) for column in schema],
rows=rows,
fingerprint=fingerprint,
row_count=len(rows),
byte_count=len(encoded),
metadata_=dict(snapshot.metadata),
created_by=actor_id,
updated_by=actor_id,
)
if source_id:
item.id = source_id
db.add(item)
db.flush()
return _source_dto(item)
def delete_snapshot(
self,
session: object,
principal: object,
*,
source_ref: str,
) -> TabularSource:
db, api_principal = _context(session, principal, WRITE_SCOPE)
item = _source_record(
db,
tenant_id=api_principal.tenant_id,
source_ref=source_ref,
)
if item is None:
raise TabularSourceNotFoundError("Tabular source not found.")
item.deleted_at = utcnow()
item.status = "retired"
item.updated_by = _actor_id(api_principal)
db.flush()
return _source_dto(item)
def parse_csv_snapshot(csv_text: str, *, delimiter: str) -> tuple[Mapping[str, object], ...]:
return parse_tabular_csv(
csv_text,
delimiter=delimiter,
max_rows=MAX_SNAPSHOT_ROWS,
)
def infer_schema(rows: Sequence[Mapping[str, object]]) -> tuple[TabularColumn, ...]:
names: list[str] = []
for row in rows:
for name in row:
if name not in names:
names.append(name)
result: list[TabularColumn] = []
for name in names:
values = [row.get(name) for row in rows]
concrete = [value for value in values if value is not None]
data_type = _type_name(concrete[0]) if concrete else "unknown"
if any(_type_name(value) != data_type for value in concrete[1:]):
data_type = "mixed"
result.append(
TabularColumn(
name=name,
data_type=data_type,
nullable=len(concrete) != len(values),
)
)
return tuple(result)
def snapshot_fingerprint(
rows: Sequence[Mapping[str, object]],
schema: Sequence[TabularColumn],
) -> str:
payload = {
"schema": [_column_payload(column) for column in schema],
"rows": [dict(row) for row in rows],
}
encoded = json.dumps(payload, sort_keys=True, separators=(",", ":"), default=str)
return hashlib.sha256(encoded.encode("utf-8")).hexdigest()
def _source_record(
session: Session,
*,
tenant_id: str,
source_ref: str,
) -> ConnectorTabularSource | None:
source_id = source_ref.removeprefix("snapshot:")
if not source_id or source_id == source_ref:
return None
return session.scalar(
select(ConnectorTabularSource).where(
ConnectorTabularSource.id == source_id,
ConnectorTabularSource.tenant_id == tenant_id,
ConnectorTabularSource.deleted_at.is_(None),
)
)
def _source_dto(item: ConnectorTabularSource) -> TabularSource:
return TabularSource(
ref=f"snapshot:{item.id}",
provider=item.provider,
source_name=item.source_name,
name=item.name,
description=item.description,
schema=tuple(TabularColumn(**column) for column in item.schema_),
schema_version=str(item.schema_version),
fingerprint=item.fingerprint,
row_count=item.row_count,
byte_count=item.byte_count,
updated_at=item.updated_at,
capabilities=("read", "preview"),
metadata=dict(item.metadata_),
source_mode="cached",
pushdown=TabularPushdown(
projections=True,
pagination=True,
),
health=TabularSourceHealth(
status="healthy",
code="snapshot.ready",
summary="The immutable connector snapshot is ready.",
checked_at=item.updated_at,
details={"immutable": True},
),
)
def _preview_diagnostic(
code: str,
message: str,
*,
requested: int,
effective: int,
) -> TabularPreviewDiagnostic:
return TabularPreviewDiagnostic(
severity="info",
code=code,
message=message,
details={"requested": requested, "effective": effective},
)
def _context(
session: object,
principal: object,
required_scope: str,
) -> tuple[Session, ApiPrincipal]:
if not isinstance(session, Session):
raise TypeError("Tabular source providers require a SQLAlchemy session.")
if not isinstance(principal, ApiPrincipal):
raise TabularSourceAccessError("A tenant API principal is required.")
accepted_scopes = {required_scope, ADMIN_SCOPE}
if required_scope == READ_SCOPE:
accepted_scopes.update(
{
WRITE_SCOPE,
"datasources:catalogue:read",
"datasources:source:write",
"datasources:source:admin",
}
)
if not any(has_scope(principal, scope) for scope in accepted_scopes):
raise TabularSourceAccessError(f"Missing scope: {required_scope}")
return session, principal
def _json_row(row: Mapping[str, object]) -> dict[str, Any]:
normalized = {str(key).strip(): value for key, value in row.items()}
if not normalized or any(not key for key in normalized):
raise TabularSourceValidationError("Every snapshot row needs named columns.")
try:
json.dumps(normalized, default=_unsupported_json)
except (TypeError, ValueError) as exc:
raise TabularSourceValidationError(f"Snapshot values must be JSON compatible: {exc}") from exc
return json.loads(json.dumps(normalized, default=_unsupported_json))
def _unsupported_json(value: object) -> object:
if isinstance(value, (datetime, Decimal)):
return str(value)
raise TypeError(f"{type(value).__name__} is not JSON serializable")
def _type_name(value: object) -> str:
if isinstance(value, bool):
return "boolean"
if isinstance(value, int):
return "integer"
if isinstance(value, (float, Decimal)):
return "number"
if isinstance(value, str):
return "string"
if isinstance(value, list):
return "array"
if isinstance(value, dict):
return "object"
return type(value).__name__.lower()
def _column_payload(column: TabularColumn) -> dict[str, object]:
return {
"name": column.name,
"data_type": column.data_type,
"nullable": column.nullable,
}
def _actor_id(principal: ApiPrincipal) -> str | None:
return principal.account_id or principal.membership_id or principal.identity_id
def _clean_optional(value: str | None) -> str | None:
cleaned = str(value or "").strip()
return cleaned or None
def _escape_like(value: str) -> str:
return value.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_")
__all__ = [
"ADMIN_SCOPE",
"MAX_READ_ROWS",
"MAX_READ_BYTES",
"MAX_READ_TIMEOUT_MS",
"MAX_SNAPSHOT_BYTES",
"MAX_SNAPSHOT_ROWS",
"READ_SCOPE",
"SqlTabularSourceProvider",
"WRITE_SCOPE",
"infer_schema",
"parse_csv_snapshot",
"snapshot_fingerprint",
]
+1
View File
@@ -0,0 +1 @@
+92
View File
@@ -0,0 +1,92 @@
from __future__ import annotations
import unittest
from sqlalchemy import create_engine
from sqlalchemy.orm import sessionmaker
from govoplan_core.auth import ApiPrincipal
from govoplan_core.core.access import PrincipalRef
from govoplan_core.core.datasources import DatasourceOriginReadRequest
from govoplan_core.core.tabular_sources import TabularSnapshotInput
from govoplan_core.db.base import Base
from govoplan_connectors.backend.datasource_origins import (
ConnectorDatasourceOriginProvider,
)
from govoplan_connectors.backend.db.models import ConnectorTabularSource
from govoplan_connectors.backend.tabular_sources import (
WRITE_SCOPE,
SqlTabularSourceProvider,
)
def principal(*, scopes: tuple[str, ...]) -> ApiPrincipal:
return ApiPrincipal(
principal=PrincipalRef(
account_id="account-1",
membership_id="membership-1",
tenant_id="tenant-1",
scopes=frozenset(scopes),
),
account=object(),
user=object(),
)
class ConnectorDatasourceOriginTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite:///:memory:")
Base.metadata.create_all(
self.engine,
tables=[ConnectorTabularSource.__table__],
)
self.Session = sessionmaker(bind=self.engine)
self.session = self.Session()
provider = SqlTabularSourceProvider()
self.source = provider.create_snapshot(
self.session,
principal(scopes=(WRITE_SCOPE,)),
snapshot=TabularSnapshotInput(
name="Imported cases",
source_name="imported_cases",
rows=({"id": 1, "name": "Ada"},),
),
)
self.session.commit()
self.origins = ConnectorDatasourceOriginProvider(provider)
def tearDown(self) -> None:
self.session.close()
Base.metadata.drop_all(
self.engine,
tables=[ConnectorTabularSource.__table__],
)
self.engine.dispose()
def test_datasource_reader_can_discover_and_read_connector_origin(self) -> None:
datasource_principal = principal(
scopes=("datasources:catalogue:read",),
)
origins = self.origins.list_origins(
self.session,
datasource_principal,
)
result = self.origins.read_origin(
self.session,
datasource_principal,
request=DatasourceOriginReadRequest(origin_ref=self.source.ref),
)
self.assertEqual((self.source.ref,), tuple(item.ref for item in origins))
self.assertEqual(("live", "cached"), origins[0].supported_modes)
self.assertEqual(({"id": 1, "name": "Ada"},), result.rows)
self.assertEqual("cached", origins[0].source_mode)
self.assertTrue(origins[0].pushdown.projections)
self.assertEqual("healthy", origins[0].health.status)
self.assertGreater(result.returned_bytes, 2)
self.assertEqual(1_000_000, result.effective_byte_limit)
if __name__ == "__main__":
unittest.main()
+103
View File
@@ -0,0 +1,103 @@
from __future__ import annotations
import unittest
from unittest.mock import patch
from defusedxml import ElementTree as SafeET
from govoplan_connectors.backend.feeds import ConnectorFeedProvider, feed_rows
from govoplan_core.core.feeds import (
FeedCapabilityError,
FeedEntry,
FeedRenderRequest,
)
from govoplan_core.security.http_fetch import HttpFetchResponse
RSS = b"""<?xml version="1.0"?>
<rss version="2.0"><channel>
<title>Decisions</title><link>https://example.test/</link>
<description>Published decisions</description>
<item><guid>decision-1</guid><title>Decision one</title>
<link>https://example.test/1</link>
<pubDate>Fri, 31 Jul 2026 10:00:00 GMT</pubDate>
<category>planning</category>
</item>
</channel></rss>"""
ATOM = b"""<?xml version="1.0"?>
<feed xmlns="http://www.w3.org/2005/Atom">
<id>https://example.test/feed</id><title>Updates</title>
<updated>2026-07-31T10:00:00Z</updated>
<link href="https://example.test/" />
<entry><id>update-1</id><title>Update one</title>
<updated>2026-07-31T10:00:00Z</updated>
<link href="https://example.test/update-1" />
</entry>
</feed>"""
class ConnectorFeedProviderTests(unittest.TestCase):
def setUp(self) -> None:
self.provider = ConnectorFeedProvider()
def test_rss_and_atom_are_normalized_to_tabular_entries(self) -> None:
rss = self.provider.parse(RSS, source_url="https://example.test/rss")
atom = self.provider.parse(ATOM, source_url="https://example.test/atom")
self.assertEqual("rss", rss.format)
self.assertEqual("decision-1", rss.entries[0].id)
self.assertEqual("atom", atom.format)
self.assertEqual("https://example.test/update-1", atom.entries[0].url)
self.assertEqual("decision-1", feed_rows(rss)[0]["id"])
self.assertEqual(64, len(rss.sha256))
def test_fetch_records_transport_freshness_and_provenance(self) -> None:
with patch(
"govoplan_connectors.backend.feeds.fetch_http",
return_value=HttpFetchResponse(
status=200,
headers={
"Content-Type": "application/rss+xml",
"Cache-Control": "public, max-age=600",
"ETag": '"feed-1"',
},
body=RSS,
),
):
document = self.provider.fetch("https://example.test/rss")
self.assertEqual('"feed-1"', document.etag)
self.assertIsNotNone(document.acquired_at)
self.assertEqual(600, int((document.fresh_until - document.acquired_at).total_seconds()))
self.assertEqual(len(RSS), document.metadata["byte_count"])
def test_render_filters_entries_by_explicit_visibility(self) -> None:
request = FeedRenderRequest(
format="atom",
title="Public updates",
feed_url="https://example.test/feed.atom",
home_url="https://example.test/",
entries=(
FeedEntry(id="public-1", title="Public", visibility="public"),
FeedEntry(id="tenant-1", title="Tenant", visibility="tenant"),
),
allowed_visibilities=frozenset({"public"}),
)
result = self.provider.render(request)
root = SafeET.fromstring(result.body)
self.assertEqual(1, result.included_entries)
self.assertEqual(1, result.excluded_entries)
self.assertIn(b"Public", result.body)
self.assertNotIn(b"Tenant", result.body)
self.assertTrue(root.tag.endswith("feed"))
def test_unsafe_xml_is_rejected(self) -> None:
payload = b'<!DOCTYPE x [<!ENTITY y SYSTEM "file:///etc/passwd">]><rss>&y;</rss>'
with self.assertRaisesRegex(FeedCapabilityError, "not safe or valid"):
self.provider.parse(payload, source_url="https://example.test/rss")
if __name__ == "__main__":
unittest.main()
+39
View File
@@ -0,0 +1,39 @@
from __future__ import annotations
import unittest
from govoplan_core.core.tabular_sources import (
CAPABILITY_CONNECTORS_TABULAR_SNAPSHOT_WRITER,
CAPABILITY_CONNECTORS_TABULAR_SOURCES,
)
from govoplan_core.core.sanctions import (
CAPABILITY_CONNECTORS_SANCTIONS_SNAPSHOTS,
)
from govoplan_connectors.backend.manifest import manifest
class ConnectorsManifestTests(unittest.TestCase):
def test_manifest_exposes_versioned_tabular_capabilities(self) -> None:
self.assertEqual("connectors", manifest.id)
self.assertIn("access", manifest.optional_dependencies)
self.assertIn(
"connectors.tabular_sources",
{interface.name for interface in manifest.provides_interfaces},
)
self.assertIn(
CAPABILITY_CONNECTORS_TABULAR_SOURCES,
manifest.capability_factories,
)
self.assertIn(
CAPABILITY_CONNECTORS_TABULAR_SNAPSHOT_WRITER,
manifest.capability_factories,
)
self.assertIn(
CAPABILITY_CONNECTORS_SANCTIONS_SNAPSHOTS,
manifest.capability_factories,
)
self.assertIsNotNone(manifest.migration_spec)
if __name__ == "__main__":
unittest.main()
+47
View File
@@ -0,0 +1,47 @@
from __future__ import annotations
import tempfile
import unittest
from pathlib import Path
from alembic.runtime.migration import MigrationContext
from sqlalchemy import create_engine, inspect
from govoplan_connectors.backend.manifest import get_manifest
from govoplan_core.db.migrations import migrate_database
class ConnectorsMigrationTests(unittest.TestCase):
def test_baseline_creates_connector_tables_and_head(self) -> None:
with tempfile.TemporaryDirectory(prefix="govoplan-connectors-migration-") as directory:
url = f"sqlite:///{Path(directory) / 'connectors.db'}"
migrate_database(
database_url=url,
enabled_modules=("connectors",),
manifest_factories=(get_manifest,),
)
engine = create_engine(url)
try:
with engine.connect() as connection:
self.assertIn(
"f7c8d9e0a1b2",
set(MigrationContext.configure(connection).get_current_heads()),
)
self.assertIn(
"connector_tabular_sources",
inspect(connection).get_table_names(),
)
self.assertIn(
"connector_sanctions_snapshots",
inspect(connection).get_table_names(),
)
self.assertIn(
"connector_sanctions_acquisition_runs",
inspect(connection).get_table_names(),
)
finally:
engine.dispose()
if __name__ == "__main__":
unittest.main()
+118
View File
@@ -0,0 +1,118 @@
from __future__ import annotations
from datetime import UTC, datetime
import unittest
from sqlalchemy import create_engine
from sqlalchemy.orm import sessionmaker
from govoplan_connectors.backend.db.models import (
ConnectorSanctionsAcquisitionRun,
ConnectorSanctionsSnapshot,
ConnectorTabularSource,
)
from govoplan_connectors.backend.manifest import manifest
from govoplan_connectors.backend.provider_state import (
SANCTIONS_PROVIDER_ID,
TABULAR_PROVIDER_ID,
sanctions_provider_states,
tabular_provider_states,
)
from govoplan_core.core.provider_governance import ExternalProviderStateContext
from govoplan_core.db.base import Base
class ConnectorsProviderStateTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite+pysqlite:///:memory:", future=True)
Base.metadata.create_all(
self.engine,
tables=(
ConnectorTabularSource.__table__,
ConnectorSanctionsAcquisitionRun.__table__,
ConnectorSanctionsSnapshot.__table__,
),
)
self.session = sessionmaker(bind=self.engine, expire_on_commit=False)()
def tearDown(self) -> None:
self.session.close()
self.engine.dispose()
def test_immutable_tabular_snapshot_reports_ready_state(self) -> None:
self.session.add(
ConnectorTabularSource(
id="tabular-1",
tenant_id="tenant-1",
source_name="monthly",
name="Monthly input",
status="active",
schema_version=1,
schema_=[{"name": "id", "type": "string"}],
rows=[{"id": "1"}],
fingerprint="a" * 64,
row_count=1,
byte_count=10,
)
)
self.session.commit()
state = tabular_provider_states(
ExternalProviderStateContext(session=self.session, tenant_id="tenant-1")
)[0]
self.assertEqual("healthy", state.health)
self.assertEqual("ready", state.recovery)
self.assertEqual("not_applicable", state.freshness)
def test_sanctions_state_hashes_binding_and_manifest_registers_state(self) -> None:
now = datetime.now(UTC)
run = ConnectorSanctionsAcquisitionRun(
id="run-1",
tenant_id="tenant-1",
provider_id="eu",
source_id="secret-source-name",
status="succeeded",
attempt_count=1,
started_at=now,
finished_at=now,
)
snapshot = ConnectorSanctionsSnapshot(
id="snapshot-1",
tenant_id="tenant-1",
provider_id="eu",
publisher="European Union",
jurisdiction="EU",
list_type="sanctions",
source_id="secret-source-name",
source_version="2026-08-01",
acquired_at=now,
source_url="https://source.example.test/list.xml",
content_type="application/xml",
byte_count=8,
sha256="b" * 64,
parser_version="1",
connector_run_id=run.id,
raw_content=b"<list/>",
)
run.snapshot_id = snapshot.id
self.session.add_all((run, snapshot))
self.session.commit()
state = sanctions_provider_states(
ExternalProviderStateContext(session=self.session, tenant_id="tenant-1")
)[0]
self.assertEqual("healthy", state.health)
self.assertEqual("ready", state.recovery)
rendered = str(state.to_dict())
self.assertNotIn("secret-source-name", rendered)
self.assertNotIn("source.example.test", rendered)
self.assertEqual(
{TABULAR_PROVIDER_ID, SANCTIONS_PROVIDER_ID},
{item.provider_id for item in manifest.external_provider_state_providers},
)
if __name__ == "__main__":
unittest.main()
+263
View File
@@ -0,0 +1,263 @@
from __future__ import annotations
from datetime import datetime, timedelta, timezone
import unittest
from unittest.mock import patch
from sqlalchemy import create_engine, select
from sqlalchemy.orm import Session
from govoplan_connectors.backend.db.models import ConnectorTabularSource
from govoplan_connectors.backend.feeds import ConnectorFeedProvider
from govoplan_connectors.backend.recovery import (
CONNECTOR_RECOVERY_OPERATIONS,
ConnectorRecoveryError,
begin_connector_external_mutation,
begin_connector_read_snapshot,
)
from govoplan_connectors.backend.router import api_import_feed_snapshot
from govoplan_connectors.backend.schemas import FeedImportRequest
from govoplan_connectors.backend.tabular_sources import WRITE_SCOPE
from govoplan_core.auth import ApiPrincipal
from govoplan_core.core.access import PrincipalRef
from govoplan_core.core.recovery import (
RecoveryCheckpoint,
RecoveryOperation,
RecoveryStatus,
)
from govoplan_core.core.recovery_runtime import (
RecoveryOperationStateConflict,
claim_durable_recovery_operation,
)
from govoplan_core.core.tabular_sources import TabularSnapshotInput
from govoplan_core.core.runtime_coordination import (
DistributedLease,
RuntimeIdentity,
bind_process_runtime_identity,
)
from govoplan_core.db.base import Base
from govoplan_connectors.backend.tabular_sources import SqlTabularSourceProvider
RSS = b"""<?xml version="1.0"?>
<rss version="2.0"><channel><title>Updates</title>
<link>https://example.test/</link><description>Updates</description>
<item><guid>1</guid><title>One</title></item></channel></rss>"""
def _identity(node: str, incarnation: str) -> RuntimeIdentity:
return RuntimeIdentity(
installation_id="connector-recovery-tests",
node_id=node,
incarnation=incarnation,
role="worker",
software_version="test",
composition_hash="a" * 64,
)
def _principal() -> ApiPrincipal:
return ApiPrincipal(
principal=PrincipalRef(
account_id="account-1",
membership_id="membership-1",
tenant_id="tenant-1",
scopes=frozenset({WRITE_SCOPE}),
),
account=object(),
user=object(),
)
class ConnectorRecoveryTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite+pysqlite:///:memory:")
Base.metadata.create_all(
self.engine,
tables=(
DistributedLease.__table__,
RecoveryOperation.__table__,
RecoveryCheckpoint.__table__,
ConnectorTabularSource.__table__,
),
)
self.session = Session(self.engine, expire_on_commit=False)
bind_process_runtime_identity(_identity("node-1", "incarnation-1"))
def tearDown(self) -> None:
bind_process_runtime_identity(None)
self.session.close()
self.engine.dispose()
def test_feed_snapshot_and_recovery_checkpoint_commit_atomically_and_replay(self) -> None:
document = ConnectorFeedProvider().parse(
RSS,
source_url="https://example.test/feed.xml",
)
payload = FeedImportRequest(
url="https://example.test/feed.xml",
name="Updates",
source_name="updates",
)
with (
patch(
"govoplan_connectors.backend.router.feed_transport.fetch",
return_value=document,
) as fetch,
patch("govoplan_connectors.backend.router.audit_event"),
):
first = api_import_feed_snapshot(
payload,
session=self.session,
principal=_principal(),
idempotency_key="feed-import-1",
)
replay = api_import_feed_snapshot(
payload,
session=self.session,
principal=_principal(),
idempotency_key="feed-import-1",
)
self.assertEqual(first.ref, replay.ref)
fetch.assert_called_once()
operation = self.session.scalar(select(RecoveryOperation))
assert operation is not None
self.assertEqual(RecoveryStatus.SUCCEEDED.value, operation.status)
self.assertEqual(
first.ref.removeprefix("snapshot:"),
operation.resource_id,
)
def test_recovery_metadata_distinguishes_reads_from_external_mutations(self) -> None:
declarations = {
item.operation_type: item for item in CONNECTOR_RECOVERY_OPERATIONS
}
self.assertFalse(declarations["read-snapshot"].provider_mutation)
self.assertTrue(declarations["read-snapshot"].implemented)
self.assertTrue(declarations["external-mutation"].provider_mutation)
self.assertFalse(declarations["external-mutation"].implemented)
def test_stale_atomic_connector_fence_fails_without_claiming_an_effect(self) -> None:
recovery = begin_connector_read_snapshot(
self.session,
tenant_id="tenant-1",
provider_id="provider-1",
idempotency_key="read-1",
source_revision="revision-1",
cursor="cursor-1",
dry_run_evidence={"performed": True, "approved": True},
)
lease = self.session.scalar(select(DistributedLease))
assert lease is not None
lease.expires_at = datetime.now(timezone.utc) - timedelta(seconds=1)
self.session.commit()
bind_process_runtime_identity(_identity("node-2", "incarnation-2"))
with self.assertRaises(RecoveryOperationStateConflict):
claim_durable_recovery_operation(
recovery.operation.session_factory,
identity=_identity("node-2", "incarnation-2"),
operation_id=recovery.operation_id,
)
operation = self.session.get(RecoveryOperation, recovery.operation_id)
self.session.refresh(operation)
self.assertEqual(RecoveryStatus.FAILED.value, operation.status)
def test_external_mutation_unknown_outcome_blocks_blind_retry(self) -> None:
kwargs = {
"tenant_id": "tenant-1",
"provider_id": "provider-1",
"idempotency_key": "publish-1",
"request_sha256": "b" * 64,
"source_revision": "revision-1",
"cursor": None,
"dry_run_evidence": {"performed": True, "approved": True},
"resource_type": "external_record",
"resource_id": "record-1",
}
recovery = begin_connector_external_mutation(self.session, **kwargs)
recovery.outcome_unknown(
summary="The provider connection closed after dispatch",
provider_code="connection_closed",
)
with self.assertRaises(ConnectorRecoveryError):
begin_connector_external_mutation(self.session, **kwargs)
operation = self.session.get(RecoveryOperation, recovery.operation_id)
self.session.refresh(operation)
self.assertEqual(RecoveryStatus.OUTCOME_UNKNOWN.value, operation.status)
def test_tampered_chain_rolls_back_the_atomic_connector_projection(self) -> None:
recovery = begin_connector_read_snapshot(
self.session,
tenant_id="tenant-1",
provider_id="provider-1",
idempotency_key="tampered-read",
source_revision=None,
cursor=None,
dry_run_evidence={"performed": False, "reason": "read-only"},
)
checkpoint = self.session.scalar(
select(RecoveryCheckpoint)
.where(RecoveryCheckpoint.operation_id == recovery.operation_id)
.order_by(RecoveryCheckpoint.sequence)
.limit(1)
)
assert checkpoint is not None
checkpoint.summary = "tampered"
self.session.commit()
source = SqlTabularSourceProvider().create_snapshot(
self.session,
_principal(),
snapshot=TabularSnapshotInput(
name="Tampered",
source_name="tampered",
rows=({"id": 1},),
),
source_id=recovery.resource_id,
)
with self.assertRaises(ConnectorRecoveryError):
recovery.commit_success(
self.session,
evidence={
"verified": True,
"checks": {"snapshot_ref": source.ref},
},
)
self.assertIsNone(
self.session.get(ConnectorTabularSource, recovery.resource_id)
)
operation = self.session.get(RecoveryOperation, recovery.operation_id)
self.session.refresh(operation)
self.assertEqual(RecoveryStatus.RUNNING.value, operation.status)
def test_definitive_external_rejection_is_terminal(self) -> None:
recovery = begin_connector_external_mutation(
self.session,
tenant_id="tenant-1",
provider_id="provider-1",
idempotency_key="publish-rejected",
request_sha256="c" * 64,
source_revision="revision-1",
cursor=None,
dry_run_evidence={"performed": True, "approved": True},
resource_type="external_record",
resource_id="record-2",
)
recovery.reject(
summary="The provider rejected the requested revision",
provider_code="revision_conflict",
)
operation = self.session.get(RecoveryOperation, recovery.operation_id)
self.session.refresh(operation)
self.assertEqual(RecoveryStatus.REJECTED.value, operation.status)
if __name__ == "__main__":
unittest.main()
+391
View File
@@ -0,0 +1,391 @@
from __future__ import annotations
from datetime import timedelta
import hashlib
import unittest
from unittest.mock import patch
from urllib.error import URLError
from sqlalchemy import create_engine
from sqlalchemy.orm import Session
from govoplan_connectors.backend.db.models import (
ConnectorSanctionsAcquisitionRun,
ConnectorSanctionsSnapshot,
)
from govoplan_connectors.backend.sanctions_sources import (
SANCTIONS_READ_SCOPE,
SANCTIONS_REFRESH_SCOPE,
SOURCE_DEFINITIONS,
SYNTHETIC_PROVIDER_ID,
SYNTHETIC_UN_XML,
SanctionsSourceError,
SqlSanctionsSnapshotProvider,
TransportResponse,
UNSC_PROVIDER_ID,
UrllibSanctionsTransport,
)
from govoplan_core.auth import ApiPrincipal
from govoplan_core.core.access import PrincipalRef
from govoplan_core.core.recovery import (
RecoveryCheckpoint,
RecoveryOperation,
RecoveryStatus,
)
from govoplan_core.core.runtime_coordination import (
DistributedLease,
RuntimeIdentity,
bind_process_runtime_identity,
)
from govoplan_core.db.base import Base, utcnow
def principal(
tenant_id: str = "tenant-1",
*,
scopes: tuple[str, ...] = (
SANCTIONS_READ_SCOPE,
SANCTIONS_REFRESH_SCOPE,
),
) -> ApiPrincipal:
return ApiPrincipal(
principal=PrincipalRef(
account_id="account-1",
membership_id="membership-1",
tenant_id=tenant_id,
scopes=frozenset(scopes),
),
account=object(),
user=object(),
)
class _Transport:
def __init__(self, responses):
self.responses = list(responses)
self.headers = []
def fetch(self, definition, *, headers):
del definition
self.headers.append(dict(headers))
response = self.responses.pop(0)
if isinstance(response, Exception):
raise response
return response
def response(
content: bytes = SYNTHETIC_UN_XML,
*,
status: int = 200,
content_type: str = "application/xml",
etag: str = '"fixture-v1"',
) -> TransportResponse:
return TransportResponse(
status=status,
final_url="https://scsanctions.un.org/consolidated.xml",
headers={
"content-type": content_type,
"etag": etag,
},
content=content,
attempts=1,
)
class SanctionsSourcesTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite:///:memory:")
Base.metadata.create_all(
self.engine,
tables=(
DistributedLease.__table__,
RecoveryOperation.__table__,
RecoveryCheckpoint.__table__,
ConnectorSanctionsAcquisitionRun.__table__,
ConnectorSanctionsSnapshot.__table__,
),
)
self.session = Session(self.engine)
bind_process_runtime_identity(
RuntimeIdentity(
installation_id="connectors-tests",
node_id="connectors-test-node",
incarnation="connectors-test-incarnation",
role="worker",
software_version="test",
composition_hash="a" * 64,
)
)
def tearDown(self) -> None:
bind_process_runtime_identity(None)
self.session.close()
self.engine.dispose()
def test_fixture_refreshes_are_immutable_and_evidence_is_readable(
self,
) -> None:
provider = SqlSanctionsSnapshotProvider()
first = provider.refresh_source(
self.session,
principal(),
provider_id=SYNTHETIC_PROVIDER_ID,
)
second = provider.refresh_source(
self.session,
principal(),
provider_id=SYNTHETIC_PROVIDER_ID,
)
self.session.commit()
self.assertEqual("succeeded", first.status)
self.assertEqual("succeeded", second.status)
self.assertNotEqual(first.snapshot.ref, second.snapshot.ref)
self.assertEqual(
hashlib.sha256(SYNTHETIC_UN_XML).hexdigest(),
first.snapshot.sha256,
)
self.assertEqual(
SYNTHETIC_UN_XML,
provider.read_snapshot(
self.session,
principal(),
snapshot_ref=first.snapshot.ref,
).content,
)
runs = provider.list_runs(
self.session,
principal(),
)
self.assertTrue(
all(
run.request_evidence["subject_data_transmitted"]
is False
for run in runs
)
)
def test_conditional_fetch_reuses_prior_immutable_snapshot(self) -> None:
transport = _Transport(
(
response(),
response(b"", status=304),
)
)
provider = SqlSanctionsSnapshotProvider(transport)
first = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
)
second = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
)
self.assertEqual("not_modified", second.status)
self.assertEqual(first.snapshot.ref, second.snapshot.ref)
self.assertEqual(
{'If-None-Match': '"fixture-v1"'},
transport.headers[1],
)
self.assertEqual(
1,
self.session.query(ConnectorSanctionsSnapshot).count(),
)
def test_malformed_and_changed_sources_have_explicit_health(
self,
) -> None:
cases = (
(
b"<CONSOLIDATED_LIST>",
"application/xml",
"malformed",
),
(
b"<DIFFERENT><INDIVIDUALS/><ENTITIES/></DIFFERENT>",
"application/xml",
"unexpected_change",
),
(
SYNTHETIC_UN_XML,
"text/html",
"unexpected_change",
),
)
for payload, content_type, expected in cases:
with self.subTest(expected=expected, content_type=content_type):
provider = SqlSanctionsSnapshotProvider(
_Transport(
(
response(
payload,
content_type=content_type,
),
)
)
)
result = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
)
self.assertEqual(expected, result.status)
self.assertIsNotNone(result.error)
def test_unavailable_source_becomes_stale_when_evidence_is_old(
self,
) -> None:
provider = SqlSanctionsSnapshotProvider(
_Transport(
(
response(),
SanctionsSourceError("offline"),
)
)
)
first = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
)
record = self.session.get(
ConnectorSanctionsSnapshot,
first.snapshot.ref.removeprefix("sanctions-snapshot:"),
)
record.acquired_at = utcnow() - timedelta(days=3)
self.session.commit()
result = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
)
self.assertEqual("stale", result.status)
self.assertEqual(first.snapshot.ref, result.snapshot.ref)
def test_idempotent_refresh_replays_the_committed_acquisition(self) -> None:
transport = _Transport((response(),))
provider = SqlSanctionsSnapshotProvider(transport)
first = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
idempotency_key="scheduled-refresh-1",
)
replay = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
idempotency_key="scheduled-refresh-1",
)
self.assertEqual(first.run_id, replay.run_id)
self.assertEqual(first.snapshot.ref, replay.snapshot.ref)
self.assertEqual([], transport.responses)
operation = self.session.query(RecoveryOperation).one()
self.assertEqual(RecoveryStatus.SUCCEEDED.value, operation.status)
def test_provider_failure_commits_failed_run_and_terminal_recovery(self) -> None:
provider = SqlSanctionsSnapshotProvider(
_Transport((SanctionsSourceError("offline"),))
)
result = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
)
self.assertEqual("unavailable", result.status)
operation = self.session.query(RecoveryOperation).one()
self.assertEqual(RecoveryStatus.FAILED.value, operation.status)
self.assertEqual(
result.run_id,
self.session.query(ConnectorSanctionsAcquisitionRun).one().id,
)
def test_snapshot_access_is_tenant_and_scope_isolated(self) -> None:
provider = SqlSanctionsSnapshotProvider()
created = provider.refresh_source(
self.session,
principal(),
provider_id=SYNTHETIC_PROVIDER_ID,
)
self.assertIsNone(
provider.get_snapshot(
self.session,
principal("tenant-2"),
snapshot_ref=created.snapshot.ref,
)
)
with self.assertRaisesRegex(Exception, "Missing scope"):
provider.list_snapshots(
self.session,
principal(scopes=()),
)
def test_transport_retries_transient_network_failures(self) -> None:
class _Headers(dict):
pass
class _Response:
status = 200
headers = _Headers(
{
"Content-Type": "application/xml",
"Content-Length": str(len(SYNTHETIC_UN_XML)),
}
)
def __enter__(self):
return self
def __exit__(self, *args):
return False
def geturl(self):
return (
"https://scsanctions.un.org/"
"resources/xml/en/consolidated.xml"
)
def read(self, size):
del size
if hasattr(self, "_read"):
return b""
self._read = True
return SYNTHETIC_UN_XML
opener = unittest.mock.Mock()
opener.open.side_effect = (
URLError("temporary"),
URLError("temporary"),
_Response(),
)
sleeps = []
transport = UrllibSanctionsTransport(
sleeper=sleeps.append
)
with patch(
"govoplan_connectors.backend.sanctions_sources.build_opener",
return_value=opener,
):
fetched = transport.fetch(
SOURCE_DEFINITIONS[UNSC_PROVIDER_ID],
headers={},
)
self.assertEqual(3, fetched.attempts)
self.assertEqual([1.0, 2.0], sleeps)
if __name__ == "__main__":
unittest.main()
+268
View File
@@ -0,0 +1,268 @@
from __future__ import annotations
import unittest
from fastapi import HTTPException
from sqlalchemy import create_engine
from sqlalchemy.orm import sessionmaker
from govoplan_core.auth import ApiPrincipal
from govoplan_core.core.access import PrincipalRef
from govoplan_core.core.tabular_sources import (
TabularReadRequest,
TabularSnapshotInput,
TabularSourceAccessError,
TabularSourceUnavailableError,
TabularSourceValidationError,
)
from govoplan_core.db.base import Base
from govoplan_connectors.backend.db.models import ConnectorTabularSource
from govoplan_connectors.backend.router import api_create_tabular_snapshot
from govoplan_connectors.backend.schemas import SnapshotCreateRequest
from govoplan_connectors.backend.tabular_sources import (
READ_SCOPE,
WRITE_SCOPE,
SqlTabularSourceProvider,
parse_csv_snapshot,
)
def principal(
tenant_id: str = "tenant-1",
*,
scopes: tuple[str, ...] = (READ_SCOPE, WRITE_SCOPE),
) -> ApiPrincipal:
return ApiPrincipal(
principal=PrincipalRef(
account_id="account-1",
membership_id="membership-1",
tenant_id=tenant_id,
scopes=frozenset(scopes),
),
account=object(),
user=object(),
)
class ConnectorsTabularSourceTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite:///:memory:")
Base.metadata.create_all(self.engine, tables=[ConnectorTabularSource.__table__])
self.Session = sessionmaker(bind=self.engine)
self.session = self.Session()
self.provider = SqlTabularSourceProvider()
def tearDown(self) -> None:
self.session.close()
Base.metadata.drop_all(self.engine, tables=[ConnectorTabularSource.__table__])
self.engine.dispose()
def test_snapshot_round_trip_preserves_schema_fingerprint_and_bounds(self) -> None:
created = self.provider.create_snapshot(
self.session,
principal(),
snapshot=TabularSnapshotInput(
name="Monthly cases",
source_name="monthly_cases_2026_07",
rows=(
{"case_id": "A-1", "amount": 12, "active": True},
{"case_id": "A-2", "amount": None, "active": False},
),
),
)
self.session.commit()
listed = self.provider.list_sources(self.session, principal())
preview = self.provider.read_source(
self.session,
principal(),
request=TabularReadRequest(
source_ref=created.ref,
limit=1,
expected_fingerprint=created.fingerprint,
),
)
self.assertEqual((created.ref,), tuple(source.ref for source in listed))
self.assertEqual(
["case_id", "amount", "active"],
[column.name for column in created.schema],
)
self.assertEqual(2, preview.total_rows)
self.assertEqual(1, len(preview.rows))
self.assertTrue(preview.truncated)
self.assertEqual(created.fingerprint, preview.source.fingerprint)
self.assertEqual("cached", preview.source.source_mode)
self.assertTrue(preview.source.pushdown.projections)
self.assertTrue(preview.source.pushdown.pagination)
self.assertEqual("healthy", preview.source.health.status)
self.assertGreater(preview.returned_bytes, 2)
self.assertEqual(1, preview.effective_row_limit)
self.assertEqual("preview.row_limit_reached", preview.diagnostics[0].code)
def test_preview_enforces_byte_time_and_provider_ceiling_budgets(self) -> None:
created = self.provider.create_snapshot(
self.session,
principal(),
snapshot=TabularSnapshotInput(
name="Bounded",
source_name="bounded",
rows=(
{"id": 1, "value": "first"},
{"id": 2, "value": "second"},
),
),
)
self.session.commit()
bounded = self.provider.read_source(
self.session,
principal(),
request=TabularReadRequest(
source_ref=created.ref,
limit=500,
max_bytes=35,
timeout_ms=2_000,
),
)
self.assertEqual(1, len(bounded.rows))
self.assertTrue(bounded.truncated)
self.assertEqual(
"preview.byte_limit_reached",
bounded.diagnostics[-1].code,
)
with self.assertRaisesRegex(
TabularSourceValidationError,
"single source row exceeds",
):
self.provider.read_source(
self.session,
principal(),
request=TabularReadRequest(
source_ref=created.ref,
max_bytes=2,
),
)
tightened = self.provider.read_source(
self.session,
principal(),
request=TabularReadRequest(
source_ref=created.ref,
limit=5_000,
max_bytes=5_000_000,
timeout_ms=10_000,
),
)
self.assertEqual(500, tightened.effective_row_limit)
self.assertEqual(1_000_000, tightened.effective_byte_limit)
self.assertEqual(2_000, tightened.effective_timeout_ms)
self.assertEqual(
{
"preview.row_limit_tightened",
"preview.byte_limit_tightened",
"preview.timeout_tightened",
},
{item.code for item in tightened.diagnostics},
)
times = iter((0.0, 0.01))
timeout_provider = SqlTabularSourceProvider(clock=lambda: next(times))
with self.assertRaisesRegex(
TabularSourceUnavailableError,
"time budget",
):
timeout_provider.read_source(
self.session,
principal(),
request=TabularReadRequest(
source_ref=created.ref,
timeout_ms=1,
),
)
def test_tenant_and_scope_isolation_are_enforced(self) -> None:
created = self.provider.create_snapshot(
self.session,
principal(),
snapshot=TabularSnapshotInput(
name="Private",
source_name="private_source",
rows=({"id": 1},),
),
)
self.session.commit()
self.assertEqual((), self.provider.list_sources(self.session, principal("tenant-2")))
self.assertIsNone(
self.provider.get_source(
self.session,
principal("tenant-2"),
source_ref=created.ref,
)
)
with self.assertRaises(TabularSourceAccessError):
self.provider.list_sources(
self.session,
principal(scopes=()),
)
def test_duplicate_source_name_and_stale_fingerprint_are_rejected(self) -> None:
snapshot = TabularSnapshotInput(
name="Cases",
source_name="cases",
rows=({"id": 1},),
)
created = self.provider.create_snapshot(self.session, principal(), snapshot=snapshot)
self.session.commit()
with self.assertRaises(TabularSourceValidationError):
self.provider.create_snapshot(self.session, principal(), snapshot=snapshot)
with self.assertRaises(TabularSourceValidationError):
self.provider.read_source(
self.session,
principal(),
request=TabularReadRequest(
source_ref=created.ref,
expected_fingerprint="stale",
),
)
def test_csv_parser_infers_scalar_values_and_rejects_duplicate_headers(self) -> None:
rows = parse_csv_snapshot(
"\ufeffid;amount;active;note\n0012;12.5;true;\n2;7;false;ok\n",
delimiter=";",
)
self.assertEqual(
(
{"id": "0012", "amount": 12.5, "active": True, "note": None},
{"id": 2, "amount": 7, "active": False, "note": "ok"},
),
rows,
)
with self.assertRaises(TabularSourceValidationError):
parse_csv_snapshot("id,id\n1,2\n", delimiter=",")
with self.assertRaises(TabularSourceValidationError):
parse_csv_snapshot("id,name\n1,Ada,extra\n", delimiter=",")
def test_malformed_csv_api_request_is_reported_as_validation_error(self) -> None:
payload = SnapshotCreateRequest(
name="Malformed",
source_name="malformed",
format="csv",
csv_text="id,name\n1,Ada,extra\n",
)
with self.assertRaises(HTTPException) as raised:
api_create_tabular_snapshot(
payload,
session=self.session,
principal=principal(),
)
self.assertEqual(422, raised.exception.status_code)
if __name__ == "__main__":
unittest.main()