16 Commits
Author SHA1 Message Date
zemion 32cac8835b Release v0.1.16
Module Package Release / publish-packages (push) Successful in 11s
2026-08-05 19:52:00 +02:00
zemion c69d68f1da Release v0.1.15
Module Package Release / publish-packages (push) Successful in 11s
2026-08-04 15:10:18 +02:00
zemion 1caae6e49e Make package publication retries hash-safe 2026-08-04 14:32:19 +02:00
zemion 84c9bb7711 Harden module package publication 2026-08-04 14:02:39 +02:00
zemion 20146ef8fe Bound connector source previews 2026-08-04 12:05:27 +02:00
zemion c33380b957 Add protected package release workflow 2026-08-04 04:14:03 +02:00
zemion dfa717b9ba Fence governed connector acquisitions 2026-08-03 05:43:54 +02:00
zemion 26f8898d11 Announce shared connector runtime contract 2026-08-02 14:54:56 +02:00
zemion 7be93785a2 feat: declare governed external provider state 2026-08-01 17:48:25 +02:00
zemion 52fe33568c Add governed RSS and Atom connectors 2026-07-31 22:48:07 +02:00
zemion c5a43b3dae feat: acquire immutable sanctions snapshots 2026-07-29 18:46:53 +02:00
zemion 27302f0c39 feat: expose connector datasource origins 2026-07-28 12:43:26 +02:00
zemion ba5ccea5b0 Implement governed tabular source snapshots 2026-07-28 11:13:22 +02:00
zemion dd45d9bd36 Release v0.1.8 2026-07-11 16:49:04 +02:00
zemion 853d12151f Add shared GovOPlaN gitignore 2026-07-10 21:57:21 +02:00
zemion d2ba4ce4d8 chore: sync GovOPlaN module split state 2026-07-10 12:51:19 +02:00
37 changed files with 7710 additions and 0 deletions
+270
View File
@@ -0,0 +1,270 @@
name: Module Package Release
on:
push:
tags:
- "v*"
workflow_dispatch:
inputs:
release_tag:
description: Existing protected version tag to publish
required: true
type: string
jobs:
publish-packages:
runs-on: ubuntu-latest
env:
GITEA_REPOSITORY: ${{ gitea.repository }}
steps:
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5
with:
fetch-depth: 0
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065
with:
python-version: "3.12"
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020
with:
node-version: "22"
- name: Select and validate protected release tag
shell: bash
env:
REQUESTED_TAG: ${{ inputs.release_tag }}
TRIGGER_TAG: ${{ gitea.ref_name }}
run: |
set -euo pipefail
tag="${REQUESTED_TAG:-$TRIGGER_TAG}"
case "$tag" in
v[0-9]*.[0-9]*.[0-9]*) ;;
*) echo "Release tag must start with a SemVer-shaped vX.Y.Z value" >&2; exit 1 ;;
esac
git fetch --force origin "refs/tags/$tag:refs/tags/$tag" refs/heads/main:refs/remotes/origin/main
tag_commit="$(git rev-list -n 1 "$tag")"
git merge-base --is-ancestor "$tag_commit" refs/remotes/origin/main || {
echo "Release tag is not contained in main" >&2
exit 1
}
git checkout --detach "$tag"
printf 'RELEASE_TAG=%s\n' "$tag" >> "$GITEA_ENV"
printf 'SOURCE_DATE_EPOCH=%s\n' "$(git show -s --format=%ct HEAD)" >> "$GITEA_ENV"
- name: Validate package versions
run: |
python - <<'PY'
import json
from pathlib import Path
import os
import re
import tomllib
tag = os.environ["RELEASE_TAG"]
expected = tag.removeprefix("v")
project = tomllib.loads(Path("pyproject.toml").read_text(encoding="utf-8"))["project"]
if project.get("version") != expected:
raise SystemExit(f"pyproject version {project.get('version')!r} does not match {tag}")
if re.fullmatch(r"govoplan-[a-z0-9-]+", str(project.get("name", ""))) is None:
raise SystemExit("Python distribution name must use the govoplan-* namespace")
webui = Path("webui/package.json")
if webui.is_file():
package = json.loads(webui.read_text(encoding="utf-8"))
if package.get("version") != expected:
raise SystemExit(f"WebUI version {package.get('version')!r} does not match {tag}")
if re.fullmatch(r"@govoplan/[a-z0-9-]+-webui", str(package.get("name", ""))) is None:
raise SystemExit("WebUI package name must use the @govoplan/*-webui namespace")
release = Path("webui/package.release.json")
if release.is_file():
release_package = json.loads(release.read_text(encoding="utf-8"))
if (
release_package.get("name") != package.get("name")
or release_package.get("version") != expected
):
raise SystemExit("WebUI release package identity does not match package.json and the release tag")
PY
- name: Build immutable package artifacts
shell: bash
run: |
set -euo pipefail
python -m pip install --disable-pip-version-check build==1.5.0 twine==7.0.0
rm -rf dist .package-webui
python -m build --wheel --outdir dist
python -m twine check dist/*.whl
if [[ -f webui/package.json ]]; then
mkdir .package-webui
cp -a webui/. .package-webui/
rm -rf .package-webui/node_modules .package-webui/dist
if [[ -f .package-webui/package.release.json ]]; then
cp .package-webui/package.release.json .package-webui/package.json
fi
node <<'NODE'
const fs = require("node:fs");
const path = ".package-webui/package.json";
const packageJson = JSON.parse(fs.readFileSync(path, "utf8"));
const groups = ["dependencies", "optionalDependencies", "peerDependencies"];
for (const group of groups) {
for (const [name, specifier] of Object.entries(packageJson[group] || {})) {
if (!name.startsWith("@govoplan/")) continue;
if (typeof specifier !== "string") {
throw new Error(`${group}.${name} must use a string version`);
}
const packageSlug = name.slice("@govoplan/".length);
if (!packageSlug.endsWith("-webui")) {
throw new Error(`${group}.${name} is outside the WebUI package namespace`);
}
const repository = `govoplan-${packageSlug.slice(0, -"-webui".length)}`;
const escapedRepository = repository.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
const gitTag = specifier.match(
new RegExp(
`^git\\+(?:ssh://git@|https://)git\\.add-ideas\\.de/(?:GovOPlaN|add-ideas)/${escapedRepository}\\.git#v([0-9]+\\.[0-9]+\\.[0-9]+)$`,
),
);
if (gitTag) {
packageJson[group][name] = gitTag[1];
continue;
}
if (specifier.startsWith("file:") || specifier.startsWith("git+")) {
throw new Error(
`${group}.${name} must resolve to an exact registry version for publication`,
);
}
}
}
delete packageJson.private;
fs.writeFileSync(path, `${JSON.stringify(packageJson, null, 2)}\n`);
NODE
npm pkg delete private --prefix .package-webui
(cd .package-webui && npm pack --ignore-scripts --pack-destination ../dist)
fi
python - <<'PY'
import hashlib
import json
from pathlib import Path
import os
import subprocess
artifacts = []
for path in sorted(Path("dist").iterdir()):
if path.suffix not in {".whl", ".tgz"}:
continue
digest = hashlib.sha256(path.read_bytes()).hexdigest()
artifacts.append({"filename": path.name, "sha256": digest, "size": path.stat().st_size})
payload = {
"schema_version": "1",
"repository": os.environ["GITEA_REPOSITORY"],
"tag": os.environ["RELEASE_TAG"],
"commit": subprocess.check_output(["git", "rev-parse", "HEAD"], text=True).strip(),
"artifacts": artifacts,
}
Path("dist/package-artifacts.json").write_text(
json.dumps(payload, indent=2, sort_keys=True) + "\n",
encoding="utf-8",
)
PY
- name: Retain package hash evidence
uses: actions/upload-artifact@a8a3f3ad30e3422c9c7b888a15615d19a852ae32
with:
name: module-packages-${{ gitea.ref_name }}
path: dist/package-artifacts.json
- name: Check immutable registry state
shell: bash
env:
PACKAGE_TOKEN: ${{ secrets.GOVOPLAN_PACKAGE_TOKEN }}
run: |
set -euo pipefail
test -n "$PACKAGE_TOKEN"
python - <<'PY'
import hashlib
import json
import os
from pathlib import Path
import tomllib
from urllib.error import HTTPError
from urllib.parse import quote
from urllib.request import Request, urlopen
api_root = "https://git.add-ideas.de/api/v1/packages/GovOPlaN"
token = os.environ["PACKAGE_TOKEN"]
def should_publish(kind, name, version, path):
package_url = "/".join(
(api_root, kind, quote(name, safe=""), quote(version, safe=""), "files")
)
request = Request(
package_url,
headers={"Accept": "application/json", "Authorization": f"token {token}"},
)
try:
with urlopen(request, timeout=30) as response:
files = json.load(response)
except HTTPError as exc:
if exc.code == 404:
print(f"{kind} package {name}=={version} is not published yet")
return True
raise
if not isinstance(files, list) or len(files) != 1:
raise SystemExit(
f"immutable {kind} package {name}=={version} has an unexpected file set"
)
expected_sha256 = hashlib.sha256(path.read_bytes()).hexdigest()
if files[0].get("sha256") != expected_sha256:
raise SystemExit(
f"immutable {kind} package {name}=={version} already exists with a different SHA-256"
)
print(f"verified existing {kind} package {name}=={version} ({expected_sha256})")
return False
project = tomllib.loads(Path("pyproject.toml").read_text(encoding="utf-8"))["project"]
wheels = tuple(Path("dist").glob("*.whl"))
if len(wheels) != 1:
raise SystemExit("release build must contain exactly one wheel")
publish_pypi = should_publish(
"pypi", str(project["name"]), str(project["version"]), wheels[0]
)
tarballs = tuple(Path("dist").glob("*.tgz"))
if len(tarballs) > 1:
raise SystemExit("release build must contain at most one npm package")
publish_npm = False
if tarballs:
webui = json.loads(
Path(".package-webui/package.json").read_text(encoding="utf-8")
)
publish_npm = should_publish(
"npm", str(webui["name"]), str(webui["version"]), tarballs[0]
)
with Path(os.environ["GITEA_ENV"]).open("a", encoding="utf-8") as env_file:
env_file.write(f"PUBLISH_PYPI={int(publish_pypi)}\n")
env_file.write(f"PUBLISH_NPM={int(publish_npm)}\n")
PY
- name: Publish wheel and WebUI package
shell: bash
env:
PACKAGE_USERNAME: ${{ secrets.GOVOPLAN_PACKAGE_USERNAME }}
PACKAGE_TOKEN: ${{ secrets.GOVOPLAN_PACKAGE_TOKEN }}
run: |
set -euo pipefail
test -n "$PACKAGE_USERNAME"
test -n "$PACKAGE_TOKEN"
if [[ "$PUBLISH_PYPI" == 1 ]]; then
TWINE_USERNAME="$PACKAGE_USERNAME" TWINE_PASSWORD="$PACKAGE_TOKEN" \
python -m twine upload --non-interactive \
--repository-url https://git.add-ideas.de/api/packages/GovOPlaN/pypi \
dist/*.whl
else
echo "Exact wheel is already present; skipping immutable retry."
fi
shopt -s nullglob
webui_packages=(dist/*.tgz)
if (( ${#webui_packages[@]} )) && [[ "$PUBLISH_NPM" == 1 ]]; then
npmrc="$(mktemp)"
trap 'rm -f "$npmrc"' EXIT
chmod 600 "$npmrc"
printf '%s\n' \
'@govoplan:registry=https://git.add-ideas.de/api/packages/GovOPlaN/npm/' \
"//git.add-ideas.de/api/packages/GovOPlaN/npm/:_authToken=$PACKAGE_TOKEN" \
> "$npmrc"
NPM_CONFIG_USERCONFIG="$npmrc" npm publish "./${webui_packages[0]}" \
--ignore-scripts --access public \
--registry https://git.add-ideas.de/api/packages/GovOPlaN/npm/
elif (( ${#webui_packages[@]} )); then
echo "Exact WebUI package is already present; skipping immutable retry."
fi
+276
View File
@@ -0,0 +1,276 @@
__pycache__/
*.py[cod]
*.egg-info/
.pytest_cache/
.mypy_cache/
.ruff_cache/
.venv/
build/
dist/
node_modules/
webui/node_modules/
webui/dist/
*.tsbuildinfo
.component-test-build/
.module-test-build/
.policy-test-build/
.template-preview-test-build/
.import-test-build/
webui/.component-test-build/
webui/.module-test-build/
webui/.policy-test-build/
webui/.template-preview-test-build/
webui/.import-test-build/
# GovOPlaN shared ignore rules from govoplan-core
# ---> Node
# Logs
logs
*.log
npm-debug.log*
yarn-debug.log*
yarn-error.log*
lerna-debug.log*
.pnpm-debug.log*
# Diagnostic reports (https://nodejs.org/api/report.html)
report.[0-9]*.[0-9]*.[0-9]*.[0-9]*.json
# Runtime data
pids
*.pid
*.seed
*.pid.lock
# Directory for instrumented libs generated by jscoverage/JSCover
lib-cov
# Coverage directory used by tools like istanbul
coverage
*.lcov
# nyc test coverage
.nyc_output
# Grunt intermediate storage (https://gruntjs.com/creating-plugins#storing-task-files)
.grunt
# Bower dependency directory (https://bower.io/)
bower_components
# node-waf configuration
.lock-wscript
# Compiled binary addons (https://nodejs.org/api/addons.html)
build/Release
# Dependency directories
jspm_packages/
# Snowpack dependency directory (https://snowpack.dev/)
web_modules/
# TypeScript cache
# Optional npm cache directory
.npm
# Optional eslint cache
.eslintcache
# Optional stylelint cache
.stylelintcache
# Microbundle cache
.rpt2_cache/
.rts2_cache_cjs/
.rts2_cache_es/
.rts2_cache_umd/
# Optional REPL history
.node_repl_history
# Output of 'npm pack'
*.tgz
# Yarn Integrity file
.yarn-integrity
# dotenv environment variable files
.env
.env.development.local
.env.test.local
.env.production.local
.env.local
# parcel-bundler cache (https://parceljs.org/)
.cache
.parcel-cache
# Next.js build output
.next
out
# Nuxt.js build / generate output
.nuxt
dist
# Gatsby files
.cache/
# Comment in the public line in if your project uses Gatsby and not Next.js
# https://nextjs.org/blog/next-9-1#public-directory-support
# public
# vuepress build output
.vuepress/dist
# vuepress v2.x temp and cache directory
.temp
.cache
# vitepress build output
**/.vitepress/dist
# vitepress cache directory
**/.vitepress/cache
# Docusaurus cache and generated files
.docusaurus
# Serverless directories
.serverless/
# FuseBox cache
.fusebox/
# DynamoDB Local files
.dynamodb/
# TernJS port file
.tern-port
# Stores VSCode versions used for testing VSCode extensions
.vscode-test
# yarn v2
.yarn/cache
.yarn/unplugged
.yarn/build-state.yml
.yarn/install-state.gz
.pnp.*
# Local WebUI test/build scratch directories
# ---> Python
# Byte-compiled / optimized / DLL files
*$py.class
# C extensions
*.so
# Distribution / packaging
.Python
develop-eggs/
downloads/
eggs/
.eggs/
lib/
lib64/
parts/
sdist/
var/
wheels/
share/python-wheels/
.installed.cfg
*.egg
MANIFEST
# PyInstaller
# Usually these files are written by a python script from a template
# before PyInstaller builds the exe, so as to inject date/other infos into it.
*.manifest
*.spec
# Installer logs
pip-log.txt
pip-delete-this-directory.txt
# Unit test / coverage reports
htmlcov/
.tox/
.nox/
.coverage
.coverage.*
.cache
nosetests.xml
coverage.xml
*.cover
*.py,cover
.hypothesis/
cover/
# Translations
*.mo
*.pot
# Django stuff:
*.log
local_settings.py
db.sqlite3
db.sqlite3-journal
# Flask stuff:
instance/
.webassets-cache
# Scrapy stuff:
.scrapy
# Sphinx documentation
docs/_build/
# PyBuilder
.pybuilder/
target/
# Jupyter Notebook
.ipynb_checkpoints
# IPython
profile_default/
ipython_config.py
# pyenv
# For a library or package, you might want to ignore these files since the code is
# intended to run in multiple environments; otherwise, check them in:
# .python-version
# pipenv
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
# However, in case of collaboration, if having platform-specific dependencies or dependencies
# having no cross-platform support, pipenv may install dependencies that don't work, or not
# install all needed dependencies.
#Pipfile.lock
# UV
# Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control.
# This is especially recommended for binary packages to ensure reproducibility, and is more
# commonly ignored for libraries.
#uv.lock
# poetry
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
# This is especially recommended for binary packages to ensure reproducibility, and is more
# commonly ignored for libraries.
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
#poetry.lock
# pdm
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
#pdm.lock
# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
# in version control.
# https://pdm.fming.dev/latest/usage/project/#working-with-version-control
.pdm.toml
.pdm-python
.pdm-build/
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
__pypackages__/
# Celery stuff
celerybeat-schedule
celerybeat.pid
# SageMath parsed files
*.sage.py
# Environments
.env
.venv
env/
venv/
ENV/
env.bak/
venv.bak/
# Spyder project settings
.spyderproject
.spyproject
# Rope project settings
.ropeproject
# mkdocs documentation
/site
# mypy
.dmypy.json
dmypy.json
# Pyre type checker
.pyre/
# pytype static type analyzer
.pytype/
# Cython debug symbols
cython_debug/
# PyCharm
# JetBrains specific template is maintained in a separate JetBrains.gitignore that can
# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
# and can be added to the global gitignore or merged into this file. For a more nuclear
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
#.idea/
# Ruff stuff:
# PyPI configuration file
.pypirc
# ---> VisualStudioCode
.vscode/*
!.vscode/settings.json
!.vscode/tasks.json
!.vscode/launch.json
!.vscode/extensions.json
!.vscode/*.code-snippets
# Local History for Visual Studio Code
.history/
# Built Visual Studio Code Extensions
*.vsix
*.db
# GovOPlaN local runtime state
runtime/
# GovOPlaN WebUI test output
+16
View File
@@ -0,0 +1,16 @@
# GovOPlaN Connectors Codex Guide
## Scope
This repository owns reusable external connection profiles, protocol adapters, governed snapshots, and connector capability contracts.
## Documentation Contract
- Treat documentation as part of every behavior change. Update this module's manifest-driven `DocumentationTopic` contributions for affected user and administrator behavior.
- Keep feature content here; `govoplan-docs` projects it without importing Connectors internals.
- Maintain a static user/admin baseline and run `/mnt/DATA/git/govoplan/tools/checks/check-manifest-shapes.py` after behavior or manifest changes.
## Boundaries
- Domain modules own business semantics; connectors own transport, credentials, synchronization, and diagnostics.
- Keep optional adapters behind capabilities and enforce egress and peer-validation policy.
+62
View File
@@ -0,0 +1,62 @@
# govoplan-connectors
<!-- govoplan-repository-type:start -->
**Repository type:** connector (connector-hub).
<!-- govoplan-repository-type:end -->
`govoplan-connectors` owns integration catalogues and generic external system
connection patterns for GovOPlaN.
The module should make external systems discoverable, testable, and usable
without taking ownership of their business semantics. Domain-specific modules
remain responsible for case, file, workflow, payment, mail, identity, document,
or reporting behavior.
## Executable First Slice
The first executable connector capability provides tenant-isolated tabular
origins. Operators can import bounded JSON or CSV snapshots, inspect inferred
schemas, and expose immutable source references and content fingerprints
through `connectors.datasource_origins@0.1.0`. Preview reads enforce provider
ceilings for rows, serialized bytes, and elapsed time and report the effective
limits and any truncation as structured diagnostics.
Connectors owns acquisition, connection profiles, credentials, discovery, and
provider health. `govoplan-datasources` registers an origin as a governed live
or cached datasource and owns staging, materializations, frozen states, and
consumer access. Dataflow consumes that Datasources contract and never imports
connector implementations or stores connector credentials.
Each origin declares whether it is live, cached, file-backed, or static, its
structured health state, and which projection, filter, aggregation, sorting,
and pagination operations it can push down. The immutable snapshot provider
currently supports projection and pagination only; consumers must keep other
operations in Dataflow rather than assuming transport-side execution.
Database, REST/HTTP, directory, managed-file, and warehouse providers can
implement the same origin contract without changing Datasources or Dataflow.
Governed sanctions and feed snapshot acquisitions use Core recovery operations.
The source revision/cursor, redacted dry-run decision, canonical request digest,
and distributed lease are durable before network I/O. Immutable snapshot rows
and the terminal recovery checkpoint commit atomically, and an
`Idempotency-Key` replays the committed result without contacting the provider.
Current acquisition transports are read-only. The exported external-mutation
recovery contract requires stable idempotency, provider verification, and
operator reconciliation, but no connector currently claims a production write
or delete path.
Development:
```bash
/mnt/DATA/git/govoplan/.venv/bin/python -m pip install -e .
/mnt/DATA/git/govoplan/.venv/bin/python -m unittest discover -s tests
```
See:
- [Connector concept](docs/CONCEPT.md)
- [Public-sector integration catalogue](docs/PUBLIC_SECTOR_INTEGRATION_CATALOGUE.md)
- [Connector source lifecycle](docs/CONNECTOR_SOURCE_LIFECYCLE.md)
- [OpenProject connector concept](docs/OPENPROJECT_CONNECTOR.md)
- [OpenDesk integration map](docs/OPENDESK_INTEGRATION_MAP.md)
+197
View File
@@ -0,0 +1,197 @@
# govoplan-connectors Concept
## Purpose
`govoplan-connectors` is the integration catalogue and connector coordination
module. It helps GovOPlaN connect to existing public-sector and organizational
systems without pretending to replace every specialist platform.
The module owns connector metadata, connection profiles, health checks, test
results, credential references, and generic integration events. Protocol-heavy
or domain-heavy integrations may live in dedicated modules once their scope is
clear.
Connector capability is not source ownership. Each configured binding also
declares whether GovOPlaN is authoritative, the external system is
authoritative, GovOPlaN keeps a mirror, both sides use governed synchronization,
GovOPlaN supplies only a governance overlay, or the object is link-only. The
same connector may be configured differently by tenant, service, object type,
or field group.
Detailed follow-up documents:
- [Public-sector integration catalogue](PUBLIC_SECTOR_INTEGRATION_CATALOGUE.md)
- [Connector source lifecycle](CONNECTOR_SOURCE_LIFECYCLE.md)
- [OpenProject connector concept](OPENPROJECT_CONNECTOR.md)
- [OpenDesk integration map](OPENDESK_INTEGRATION_MAP.md)
## Shared runtime boundary
Connector transports use the versioned Core runtime contract for bounded dry
runs and diagnostics. Connectors owns endpoint discovery, authentication
hand-off, protocol reads, retry/backoff, and source health. A consuming module
owns domain mappings and mutations: Addresses, for example, owns contact and
vCard semantics even when Connectors supplies reusable LDAP, CardDAV, Exchange,
or Google transport patterns.
Dry runs carry an immutable input hash, source revision and fingerprint,
redacted effects, diagnostics, truncation state, and an apply token. Apply must
reject a changed input or source revision. URLs and diagnostics never contain
credential material; they retain only credential-envelope references.
## Ownership
The module owns:
- connector catalogue entries and capability metadata
- connection profiles and endpoint configuration
- credential references and test diagnostics
- generic webhook/polling/job coordination metadata
- connector health status and last-test evidence
- operator-visible integration inventory
- cross-module discovery of available external capabilities
- supported integration maturity, source-authority modes, operation limits,
and effect/reconciliation behavior for each connector type
The module does not own:
- governed datasource identity, staging, materializations, frozen states, or
consumer read semantics, owned by `govoplan-datasources`
- file storage semantics, owned by files/DMS
- identity provisioning semantics, owned by IDM/access
- mail/calendar semantics, owned by mail/calendar
- case/workflow/task/domain records
- payment, ledger, XRechnung, XTA/OSCI, FIT-Connect, or XOE/V protocol
semantics once those are dedicated modules
## Connector Categories
Initial catalogue categories:
- project management and task systems such as OpenProject
- DMS/e-file/archive systems
- file providers such as Nextcloud, Seafile, WebDAV, SMB/NFS, object storage
- identity providers such as LDAP, Active Directory, OIDC, SAML, OpenDesk IDM
- groupware such as Open-Xchange mail/calendar
- ERP, finance, accounting, payment, and cash-register systems
- public-sector protocols such as FIT-Connect, XTA/OSCI, XRechnung, XOE/V
- reporting, BI, RSS/API publication, and open-data endpoints
## Core Contracts
The module should integrate through:
- module manifest metadata, route factories, permissions, and migrations
- a connector catalogue API for listing available connector types
- a connection profile API with secret references, not plaintext secrets
- a provider declaration that composes authority mode, maturity, supported
operations, revisions/freshness, health, limits, idempotency, conflicts,
evidence, and reconciliation behavior
- capability declarations such as `connectors.catalog`,
`connectors.profileTester`, and `connectors.health`
- events such as `connector.profile_created`, `connector.test_succeeded`,
`connector.test_failed`, and `connector.health_changed`
- configuration-package fragments for required external systems
Domain modules should ask whether a connector capability exists and request a
profile/test result through core-mediated capabilities. They must not import
connector implementation modules directly.
Data-oriented consumers use a two-layer path: Connectors publishes a
provider-specific datasource origin, then Datasources registers and governs it.
Dataflow, Workflow, Reporting, and other consumers use Datasources rather than
calling the connector origin directly.
## Reference Journeys
### OpenProject Connector First
1. Operator registers an OpenProject connection profile.
2. Connector tests API reachability and authentication.
3. A future project-management decision can use the connector before a native
`govoplan-projects` module exists.
4. Cases/tasks/workflow may link to external project/task references through
stable external-reference DTOs.
### Public-Sector Integration Catalogue
1. Operator records which external systems exist in an organization.
2. GovOPlaN identifies common protocols and missing connectors.
3. Configuration packages can declare required connector profiles.
4. Health/status pages show whether required integrations are ready.
### OpenDesk Profile
1. Operator records OpenDesk component profiles for identity, mail, calendar,
files/documents, and OpenProject where present.
2. Connectors shows which components are configured, tested, degraded, or
missing.
3. Domain modules enable optional behavior by checking capabilities through
core, not by importing connector or OpenDesk-specific implementation code.
## MVP Slice
The first implementation should provide:
- connector type registry
- connection profile CRUD with secret references
- connection test result records
- WebUI catalogue and profile pages
- configuration-package fragment support
- generic external-reference DTOs
- source-authority binding and provider-operation metadata
- health summary provider
## Permissions
Candidate scopes:
- `connectors:catalog:read`
- `connectors:profile:read`
- `connectors:profile:write`
- `connectors:profile:test`
- `connectors:secret:manage`
- `connectors:admin`
## Data Model Sketch
Candidate tables:
- `connector_types`
- `connector_profiles`
- `connector_profile_tests`
- `connector_health_status`
- `external_references`
Plaintext credentials must never be stored in connector tables. Use secret
references and the platform secret contract.
## WebUI
Initial route contributions:
- `/connectors`
- `/connectors/profiles/:profileId`
The UI should show profile status, last test result, capability labels, required
configuration-package dependencies, and external-reference search where a
connector supports it.
## Tests
Minimum tests:
- core starts with connectors installed and no domain modules present
- profile validation rejects plaintext secret echoing
- connection tests record success/failure diagnostics without leaking secrets
- configuration package can require a connector profile
- domain-module optional behavior can detect connector capabilities without
imports
## Open Decisions
- Whether protocol-specific connector modules depend on `govoplan-connectors`
or only share kernel contracts.
- Which connector type should be first after OpenProject.
- How much polling/webhook scheduling belongs here versus workflow/ops.
- Whether external-reference indexing should move to search/dataflow later.
+221
View File
@@ -0,0 +1,221 @@
# Connector Source Lifecycle
GovOPlaN modules should treat external systems as sources with explicit
lifecycle state. A connector profile can consume records from a source, publish
records into a source, or do both. The lifecycle below keeps connectors
predictable and avoids hidden module imports.
## Source Directions
- `consume`: GovOPlaN reads external records, normalizes them, and exposes them
to modules as external references, events, or staged imports.
- `publish`: GovOPlaN creates or updates external records and stores the external
identifiers as immutable references.
- `bidirectional`: GovOPlaN supports both directions with conflict detection and
reconciliation rules.
Direction describes transport. Every binding also needs a source-authority
mode:
- `native_authoritative`
- `external_authoritative`
- `external_mirror`
- `governed_sync`
- `governance_overlay`
- `linked_reference`
The authority mode and the connector's integration maturity are orthogonal. A
bidirectional connector may be configured as an external mirror, and a native
GovOPlaN object may publish to an external target without transferring
authority. The effective binding must identify its scope and provenance rather
than relying on a profile-wide `sync` boolean.
## Source Data Lifecycle
Connector profiles have operational states, while individual external records
or source datasets move through a data lifecycle:
1. `discovered`
A source, record, file, feed item, webhook event, or remote object is known
but not yet trusted for domain use.
2. `connected`
GovOPlaN can authenticate and fetch or publish against the source profile.
3. `imported`
Minimal source data has been staged with external id, version/ETag, source
timestamp, and provenance.
4. `validated`
Shape, permissions, freshness, and required fields passed connector and
domain validation.
5. `transformed`
A dataflow, workflow, or domain module normalized the staged payload into a
domain-specific form.
6. `published`
GovOPlaN exposed or wrote an output through API, RSS, report, export, or a
downstream connector.
7. `archived`
The source/output is no longer active but remains available under retention,
audit, and external-reference rules.
8. `deprecated`
The source/output remains readable for history but must not be used for new
workflows.
Every transition must preserve provenance, permissions context, freshness, and
audit trace. Domain modules may add stricter states, but they should map back to
this lifecycle when a connector publishes status.
## Lifecycle States
1. `draft`
Profile exists but is not used by runtime jobs.
2. `configured`
Required endpoint and credential references are present.
3. `tested`
A health/test run succeeded and recorded non-secret diagnostics.
4. `active`
Runtime jobs may consume or publish data.
5. `degraded`
The connector is active but health checks or recent jobs show failures.
6. `paused`
Operators intentionally stop scheduled connector activity.
7. `retiring`
The connector is being removed from active workflows while references remain
readable.
8. `retired`
No new runtime activity is allowed. Historical references remain available.
## State Transition Gates
| Transition | Required Evidence | Blockers |
| --- | --- | --- |
| `draft` -> `configured` | endpoint fields are valid, credential references exist, owner/tenant scope is set | plaintext secret in profile payload, unsupported connector type |
| `configured` -> `tested` | latest profile test succeeded and diagnostics were redacted | failed auth, unreachable endpoint, TLS/policy error |
| `tested` -> `active` | operator enabled runtime use, required modules/capabilities are present, schedule/webhook is valid | missing module, missing permission, no idempotency strategy for publish jobs |
| `active` -> `degraded` | health check or job telemetry reports failures | none; this is automatic diagnostic state |
| `degraded` -> `active` | health/test succeeds or failed jobs are reconciled | unresolved conflict or repeated failure threshold |
| any running state -> `paused` | operator pause request or maintenance preflight | active critical transaction that cannot be interrupted |
| `paused` -> `active` | successful re-test when credentials/endpoints changed | failed profile test |
| any state -> `retiring` | uninstall/disable plan accepted, schedulers/workers stopped | active domain references that require operator decision |
| `retiring` -> `retired` | non-destructive retirement complete, references remain readable | destructive retirement requested without provider and backup |
## Consume Flow
1. Discover changes through polling, webhook, batch upload, or manual operator
action.
2. Fetch only the minimal remote data required for the declared use case.
3. Normalize into a connector-owned staging payload.
4. Validate shape, required fields, and source trust level.
5. Publish data-shaped inputs as versioned datasource origins.
6. Let Datasources register live/cached origins or stage immutable snapshots.
7. Let domain modules consume governed datasource references through
capabilities, not imports.
8. Emit a core-mediated event such as `connector.record_discovered`.
9. Store external references with source system, object type, object id, version
or ETag, and last-seen timestamp.
## Publish Flow
1. Domain module requests publish through a core-mediated connector capability.
2. Connector validates profile state, permission, idempotency key, and payload
shape.
3. Connector sends the remote request.
4. Connector stores the remote id, version/ETag, and response diagnostics.
5. A timeout or lost acknowledgement after dispatch becomes outcome-unknown,
not an ordinary failure or permission to duplicate the command.
6. Connector emits a confirmed, retryable, outcome-unknown, reconciled, or
corrected result event.
7. Domain module stores only the external-reference DTO and any domain result.
## Reconciliation
Every connector that writes to an external system needs a reconciliation story:
- idempotency key for create/update jobs
- remote object version, ETag, or last-modified value where available
- conflict state when local and remote records diverge
- retry policy for temporary failures
- explicit operator action for destructive overwrite or deletion
- audit trace from GovOPlaN record to external request and response summary
- explicit requested, approved, dispatched, possibly-executed, confirmed, and
reconciled/corrected effect states
## Durable recovery operations
Connectors declares two recovery classes. A read-only acquisition into an
immutable snapshot is `atomic`: the source revision or conditional cursor,
redacted dry-run decision, canonical request digest, and distributed
tenant/provider lease are durable before the fetch. The acquired domain rows
and terminal Core recovery checkpoint commit in one PostgreSQL transaction. A
caller-supplied `Idempotency-Key` replays that committed result without a second
provider request. A failed or stale transaction has no remote mutation and may
be repeated only as a new deliberate acquisition.
An external create, update, publish, or delete is `forward_recovery`. It must
start through the connector mutation recovery contract with a stable
idempotency key, SHA-256 request digest, source revision/cursor, and dry-run
evidence. Definitive rejection is terminal. A timeout or lost acknowledgement
after dispatch is `outcome_unknown` and blocks replay until the owning connector
verifies provider state. The contract and conformance tests exist; no current
connector advertises a production external mutation, so write/delete adoption
remains explicitly planned rather than implied.
## Provider Declaration
An executable connector type should publish machine-readable metadata for:
- owned object and field groups, plus supported authority modes;
- supported discovery, link, search, read, publish, synchronize, migrate, and
replacement maturity;
- read/write/delete/preview/dry-run operations and bounded response limits;
- revision/concurrency token, freshness, health, timeout, retry, and conflict
semantics;
- idempotency and outcome-unknown handling;
- evidence, rollback/compensation, correction, and reconciliation paths;
- classification, purpose, retention, secret, degraded, and outage behavior.
This declaration composes Core contracts. It does not move protocol behavior
or domain semantics into Core or Connectors.
## Capability Boundary
Domain modules must not import connector implementation packages directly. They
should ask core for capabilities such as:
- `connectors.catalog`
- `connectors.profileTester`
- `connectors.health`
- `connectors.externalReferences`
- `connectors.datasourceOrigins`
- `connectors.sourceConsumer`
- `connectors.sourcePublisher`
Connector payloads should be DTOs or protocol objects from kernel/core
contracts. Protocol-specific clients stay inside the connector module that owns
them.
## Safety Rules
- Secret values never leave the secret contract and are never stored in test
result payloads.
- Runtime jobs must include profile id, connector type, direction, idempotency
key, and triggering principal/system actor.
- Profile tests must redact tokens, passwords, cookies, authorization headers,
and remote personal data not needed for diagnostics.
- Deactivation must stop schedulers/workers before profile removal.
- Uninstall defaults to non-destructive retirement; domain data and external
references remain readable.
- Destructive retirement requires a module-owned retirement provider and an
explicit operator choice.
## Release Checklist
Before shipping an executable connector type:
- Add catalogue metadata and capability names.
- Add profile schema validation that rejects plaintext secrets.
- Add redaction tests for success and failure diagnostics.
- Add unavailable-optional-module tests for every consuming domain module.
- Add profile test and health status fixtures.
- Add external-reference DTO tests.
- Add source-authority and provider-declaration validation tests.
- Add lifecycle transition tests for pause, retry, retirement, and uninstall
guard behavior.
+48
View File
@@ -0,0 +1,48 @@
# Governed Connector Configuration
GovOPlaN connectors should make integration behavior inspectable and testable.
The target is not hardcoded glue hidden in module code, but governed connector
definitions with schemas, mappings, test runs, simulation, versioning, and
audit-visible execution.
## Connector Definition
A connector definition should describe:
- provider type and protocol
- endpoint and credential requirements
- supported capabilities
- input and output schemas
- mapping and transformation versions
- validation rules
- dry-run and test operations
- privacy and retention classification
- expected events and audit records
- operational limits and retry behavior
Provider-specific code may still be required, but the configured integration
logic should remain visible and reviewable.
## Runtime Expectations
Connectors should support:
- discovery where possible
- typed configuration through UI-managed controls
- secret references instead of plaintext secrets
- dry-run plans before writes
- simulation with sample payloads
- provenance for consumed and produced data
- idempotent external writes where supported
- quarantine/manual-review state for unsafe or ambiguous results
Configuration packages may install connector definitions, but local overrides
must be protected from accidental package updates.
## Relationship To Datasources And Dataflow
Recurring extraction and transformation should start as configuration across
connectors, files, workflow, reporting, and templates. Create dedicated
datasource or dataflow modules only when repeated source-catalog, lineage,
mapping, scheduling, or publication contracts clearly outgrow connector
ownership.
+71
View File
@@ -0,0 +1,71 @@
# OpenDesk Integration Map
OpenDesk is an integration profile across GovOPlaN modules, not a monolithic
GovOPlaN module. The profile should let an operator see which OpenDesk
components are connected, which GovOPlaN module owns each behavior, and which
optional capabilities are available.
The core boundary decision register is in
`/mnt/DATA/git/govoplan-core/docs/MODULE_ARCHITECTURE.md`.
## Component Routing
| OpenDesk area | Example component | GovOPlaN owner | Integration behavior |
| --- | --- | --- | --- |
| Identity and directory | OpenDesk IDM, LDAP, AD, OIDC, SAML, SCIM | `govoplan-idm`, `govoplan-access` | integrate, synchronize selected accounts/groups, map principals and memberships |
| Mail/groupware | Open-Xchange mail | `govoplan-mail` | integrate, link profiles, test mailbox/send/append, keep mail semantics in mail |
| Calendar/groupware | Open-Xchange calendar, CalDAV/CardDAV | `govoplan-calendar` | integrate, free/busy lookup, selected event sync, resource calendars |
| Files/documents | Nextcloud/WebDAV/files, office integrations | `govoplan-files`, later `govoplan-dms` | integrate, import/link files, keep document lifecycle in DMS |
| Project management | OpenProject | `govoplan-connectors`, consumers in tasks/workflow/cases | connector-first, link/synchronize selected work packages |
| Portal/collaboration | Portal/chat/video/office services where present | `govoplan-portal`, `govoplan-connectors`, `govoplan-dms`, `govoplan-workflow` | link/integrate only when a process needs it |
| Inventory and diagnostics | endpoint catalogue, profile health, version checks | `govoplan-connectors` | catalogue, profile test, health summary, optional capability discovery |
## Integration Behavior
- `integrate`: call a stable API or protocol for the component.
- `link`: store external references and open the external tool for
source-of-truth work.
- `import`: bring selected files/records into GovOPlaN-owned storage or
evidence.
- `synchronize`: keep selected records aligned through explicit source-of-truth
rules.
- `replace selected workflow`: only when GovOPlaN owns tighter governance,
audit, retention, or configuration-package state than the OpenDesk component.
## Shared Assumptions
- Identity is the first dependency. Mail, calendar, files, and project
connectors should record which identity profile or tenant mapping they expect.
- Connector profiles store references to secrets, never secret values.
- Module consumers discover optional behavior through core capabilities and
module metadata.
- The profile must work partially: an installation can have OpenProject without
Open-Xchange, or calendar without files.
## Candidate Profile Shape
```json
{
"id": "opendesk-main",
"display_name": "OpenDesk",
"components": {
"identity": {"profile_id": "opendesk-idm", "owner": "govoplan-idm"},
"mail": {"profile_id": "ox-mail", "owner": "govoplan-mail"},
"calendar": {"profile_id": "ox-calendar", "owner": "govoplan-calendar"},
"files": {"profile_id": "nextcloud-main", "owner": "govoplan-files"},
"projects": {"profile_id": "openproject-main", "owner": "govoplan-connectors"}
}
}
```
## Follow-Up Implementation Issues
Existing high-priority module issues:
- `govoplan-idm#1`: LDAP, Active Directory, OpenDesk identity services.
- `govoplan-mail#5`: Open-Xchange mail/groupware adapter boundary.
- `govoplan-calendar#2`: Open-Xchange calendar adapter boundary.
- `govoplan-connectors#1`: OpenProject connector.
Future executable connector issues should be created in the owning module
repository when the first concrete API/profile slice is selected.
+170
View File
@@ -0,0 +1,170 @@
# OpenProject Connector Concept
OpenProject is the first proposed concrete connector for
`govoplan-connectors`. It gives GovOPlaN a public-sector-friendly project and
work-package integration target without making project management a core
platform dependency.
## Goals
- Register OpenProject connection profiles.
- Test API reachability and authentication without exposing secrets.
- Read projects, users, statuses, and work packages for linking.
- Create or update work packages from GovOPlaN tasks/cases/workflows once those
modules request the capability.
- Receive or poll changes for external-reference synchronization.
- Keep all OpenProject-specific client code inside the connector module.
## Non-Goals
- Replacing a future native GovOPlaN project-management module.
- Importing workflow, tasks, cases, or access implementation modules directly.
- Mirroring complete OpenProject project state into GovOPlaN by default.
- Storing OpenProject tokens outside the platform secret contract.
## Profile Fields
Candidate profile payload:
- `base_url`
- `api_version`, default `v3`
- `credential_ref`
- `verify_tls`
- `timeout_seconds`
- `allowed_project_ids`
- `default_project_id`
- `webhook_secret_ref`, optional
- `poll_interval_seconds`, optional
## Health Check
The connection test should:
1. Normalize and validate `base_url`.
2. Resolve the credential reference.
3. Call the OpenProject API root or a small read-only endpoint.
4. Record API version, authenticated principal where available, latency,
response status, and safe capability hints.
5. Redact token, Authorization headers, cookies, and any server-provided secret
fields from diagnostics.
## Candidate Capabilities
- `connectors.openproject.profileTester`
- `connectors.openproject.projects`
- `connectors.openproject.workPackages.read`
- `connectors.openproject.workPackages.write`
- `connectors.openproject.webhooks`
- `connectors.openproject.externalReferences`
Domain modules request these through core-mediated capabilities. For example,
`govoplan-tasks` can publish a task as an OpenProject work package without
importing OpenProject client code.
## Data Boundary Decision
OpenProject should be referenced live by default, not mirrored wholesale into
GovOPlaN. The connector stores stable external references and safe metadata:
- profile id and connector type
- project id and work-package id
- external URL
- remote version, ETag, or lock version where available
- last-seen timestamp and safe status/type labels
- GovOPlaN trace id for publish or synchronization jobs
GovOPlaN should import only the subset needed by a requesting domain module,
for example a work-package title/status for display, a link-back reference for a
task, or evidence that a publish operation succeeded. Full project state,
comments, attachments, membership lists, and custom fields remain remote unless
a future domain module explicitly owns that synchronization. This keeps cases,
tasks, workflow, and reporting decoupled from OpenProject while still allowing
link-out, link-back, selected publish, and selected read views.
## Runtime Events
- `openproject.profile_tested`
- `openproject.project_seen`
- `openproject.work_package_seen`
- `openproject.work_package_published`
- `openproject.webhook_received`
- `openproject.sync_failed`
Events should carry GovOPlaN ids, external ids, safe diagnostics, and trace
context. They must not contain credentials or raw personal data beyond what the
requesting domain module is authorized to process.
## Webhook And Polling Strategy
OpenProject supports API and webhook administration. The connector should allow
both:
- Webhook-first when an operator registers a webhook for selected project/work
package events.
- Polling fallback for installations where webhooks cannot be exposed.
The first implementation can start with manual test plus read-only project/work
package lookup, then add publishing, then webhook/polling synchronization.
## First Implementation Slice
1. Add connector type metadata for `openproject`.
2. Add connection profile CRUD using secret references.
3. Add a read-only test endpoint.
4. Add project/work-package lookup DTOs.
5. Add external-reference storage for linked OpenProject work packages.
6. Add a WebUI profile page with last-test diagnostics.
7. Add tests for redaction, unavailable connector behavior, and optional module
capability discovery.
## Minimum DTOs
Profile summary:
```json
{
"id": "openproject-main",
"connector_type": "openproject",
"name": "OpenProject",
"base_url": "https://openproject.example",
"state": "tested",
"last_test_at": "2026-07-09T10:00:00Z",
"last_test_status": "success"
}
```
External reference:
```json
{
"connector_type": "openproject",
"profile_id": "openproject-main",
"object_type": "work_package",
"external_id": "1234",
"external_url": "https://openproject.example/work_packages/1234",
"version": "etag-or-lock-version",
"metadata": {
"project_id": "42"
}
}
```
Diagnostics must be redacted and should include only endpoint, version,
authenticated principal label where safe, latency, status code, and capability
hints.
## First Tests To Add
- profile create/update rejects plaintext token fields
- profile test redacts Authorization, cookies, and token-like response fields
- lookup capabilities are absent when the connector module is disabled
- a task/workflow/case module can detect OpenProject capabilities without
importing connector internals
- external-reference round-trip stores profile id, object type, external id,
version/ETag, URL, and safe metadata
## Reference Sources
- OpenProject API v3 documentation: https://www.openproject.org/docs/api/
- OpenProject API introduction: https://www.openproject.org/docs/api/introduction/
- OpenProject API and webhooks administration: https://www.openproject.org/docs/system-admin-guide/api-and-webhooks/
+156
View File
@@ -0,0 +1,156 @@
# Public-Sector Integration Catalogue
`govoplan-connectors` should maintain an operator-visible catalogue of common
external systems, protocols, and integration patterns. The catalogue is not a
promise that GovOPlaN replaces those systems. It is the map that lets modules
discover what exists, test connections, and decide which optional behavior can
be enabled.
## Catalogue Entry Shape
Each connector type should define:
- stable connector type key, for example `openproject`, `fit-connect`,
`xrepository`, or `sap`
- category and owning GovOPlaN module, if any
- supported direction: consume, publish, or bidirectional
- supported trigger modes: manual test, polling, webhook, batch import, export
- credential method and whether secrets are stored through the platform secret
contract
- health check and diagnostic payload shape
- external reference shape for records created or linked through the connector
- required capabilities and optional module combinations
- lifecycle support: activate, pause, re-test, rotate credential, retire
## Initial Target Categories
The catalogue should be maintained as a ranked inventory. A target can start as
an inventory entry before there is executable connector code.
| Target | Scope/Jurisdiction | Category | Mode | Likely Owner | First Useful Capability | Priority |
| --- | --- | --- | --- | --- | --- | --- |
| OpenProject | international/open source | Project/task management | link, synchronize selected records, publish tasks | `govoplan-connectors`, later tasks/workflow/projects | profile test, project/work-package lookup, external references | Wave 0 |
| Nextcloud/WebDAV/SMB/Seafile | broad public-sector/self-hosted | File providers | integrate, import, link | `govoplan-files` with connector inventory | profile health and managed-file provenance | Wave 0/in progress |
| OpenDesk IDM, LDAP, Active Directory, OIDC, SAML | Germany/EU and general enterprise | Identity | integrate, synchronize | `govoplan-idm`, `govoplan-access` | endpoint inventory, login/provisioning preflight | Wave 1 |
| Open-Xchange mail/calendar | Germany/OpenDesk and groupware deployments | Groupware | integrate, link | `govoplan-mail`, `govoplan-calendar` | profile test, mailbox/calendar diagnostics | Wave 1 |
| FIT-Connect | German public-sector transport | Public-sector transport | integrate, publish, receive | dedicated protocol module with connectors inventory | destination profile, test, receipt reference | Wave 1 |
| XRepository/XÖV lookup | German public-sector standards | Standards registry | link, import schema metadata | `govoplan-connectors`, later XÖV modules | read-only catalogue lookup/cache | Wave 1 |
| RSS/API publication | public data/external services | Publication/data exchange | consume, publish | `govoplan-connectors`, `govoplan-dataflow` | consume/publish feed profiles | Wave 2 |
| DMS/e-file/archive systems | German municipal/state/federal administration | DMS/records | link, import, synchronize selected metadata | `govoplan-dms`, `govoplan-files` | external document reference and health | Wave 2 |
| ERP/finance/procurement/payment systems | German municipal finance/procurement plus EU standards | ERP/payment | export, import, synchronize, replace only by domain decision | dedicated modules | profile inventory and export/import staging | Wave 2 |
### Project And Task Management
- OpenProject
- Jira or Jira-compatible APIs
- Redmine
- Microsoft Planner/Project where available through Microsoft Graph
GovOPlaN should start with OpenProject because it is open source, common in
public-sector environments, and has API/webhook documentation suitable for a
first connector.
### DMS, E-File, Records, And Archive
- d.velop/d.3
- enaio
- Fabasoft eGov-Suite
- ELO
- VIS/eAkte environments
- CMIS-capable repositories
- S3/object storage used as archive staging
These targets should usually be owned by DMS/files/records modules once a
domain module exists. `govoplan-connectors` should still provide inventory,
profiles, and generic health checks.
### File Providers
- SMB/CIFS
- WebDAV
- Nextcloud
- Seafile
- S3-compatible object storage
- SFTP
The files module owns file semantics. The connectors catalogue should record
profile metadata and health, but must not import files-module internals.
### Identity And Access
- LDAP
- Active Directory
- OIDC
- SAML
- OpenDesk IDM and comparable identity platforms
The access/IDM modules own principal synchronization and authorization effects.
Connectors own endpoint inventory and diagnostics.
### Mail, Calendar, And Collaboration
- Microsoft Exchange/M365
- Open-Xchange
- IMAP/SMTP where represented as external infrastructure
- CalDAV/CardDAV
- chat, video, and collaboration systems such as Matrix, Jitsi, BigBlueButton,
Nextcloud Talk, or Collabora/OnlyOffice environments
Mail/calendar/collaboration modules own business semantics. Connector profiles
can expose reachability and version diagnostics.
### ERP, Finance, Procurement, And Payment
- SAP
- MACH
- Infoma/new system
- DATEV interfaces
- XRechnung/Peppol access points
- XBestellung and procurement feeds
- payment providers and cash-register systems
Protocol-heavy parts should move into dedicated modules such as
`govoplan-xrechnung`, `govoplan-erp`, `govoplan-procurement`, or
`govoplan-payments`.
### Public-Sector Protocols And Registries
- FIT-Connect
- XTA/OSCI
- XÖV standards and XRepository lookup
- XRechnung/XBestellung
- register and Fachverfahren interfaces discovered by implementation projects
These are integration priorities because they model common administrative
processes. They should be represented as connector categories even when a
dedicated module later owns the actual protocol implementation.
## Wave 0 Catalogue Priorities
1. OpenProject connector concept and profile shape.
2. Generic connector profile, health, and secret-reference model.
3. Public-sector target inventory table with category, owner module, and
priority.
4. Consume/publish source lifecycle contract.
5. External-reference DTO shared through kernel/core contracts.
6. Configuration-package declaration for required connector profiles.
## Catalogue Maintenance Rules
- Prefer one stable connector type key per external product or protocol family.
- Record when GovOPlaN should integrate with an existing product instead of
replacing it.
- Keep protocol/client implementation in the owning connector or protocol
module; domain modules consume capabilities and DTOs only.
- Treat "inventory only" entries as useful: operators can document a landscape
before GovOPlaN can automate it.
- Every executable connector type needs a redaction-safe test plan, lifecycle
states, external-reference shape, and uninstall/retirement behavior.
## Reference Sources
- OpenProject API v3 documentation: https://www.openproject.org/docs/api/
- OpenProject API and webhooks administration: https://www.openproject.org/docs/system-admin-guide/api-and-webhooks/
- FIT-Connect Destination API documentation: https://docs.fitko.de/en/resources/fit-connect-destination-api/
- XÖV overview by KoSIT: https://www.xoev.de/xoev-4987
- XRepository overview: https://www.xrepository.de/
+25
View File
@@ -0,0 +1,25 @@
[build-system]
requires = ["setuptools>=69", "wheel"]
build-backend = "setuptools.build_meta"
[project]
name = "govoplan-connectors"
version = "0.1.16"
description = "Governed connector catalogue and tabular source capabilities for GovOPlaN."
readme = "README.md"
requires-python = ">=3.12"
license = "AGPL-3.0-or-later"
authors = [{ name = "GovOPlaN" }]
dependencies = [
"defusedxml>=0.7,<1",
"govoplan-core>=0.1.16",
]
[tool.setuptools.packages.find]
where = ["src"]
[tool.setuptools.package-data]
govoplan_connectors = ["py.typed"]
[project.entry-points."govoplan.modules"]
connectors = "govoplan_connectors.backend.manifest:get_manifest"
+3
View File
@@ -0,0 +1,3 @@
"""GovOPlaN Connectors module."""
__all__: list[str] = []
@@ -0,0 +1 @@
"""Connector backend package."""
@@ -0,0 +1,147 @@
from __future__ import annotations
from govoplan_core.core.datasources import (
DatasourceAccessError,
DatasourceField,
DatasourceNotFoundError,
DatasourceOrigin,
DatasourceOriginReadRequest,
DatasourceOriginReadResult,
DatasourceUnavailableError,
DatasourceValidationError,
)
from govoplan_core.core.tabular_sources import (
TabularReadRequest,
TabularSource,
TabularSourceAccessError,
TabularSourceError,
TabularSourceNotFoundError,
TabularSourceUnavailableError,
TabularSourceValidationError,
)
from govoplan_connectors.backend.tabular_sources import SqlTabularSourceProvider
class ConnectorDatasourceOriginProvider:
"""Expose connector-owned sources through the Datasources origin contract."""
def __init__(self, provider: SqlTabularSourceProvider | None = None) -> None:
self._provider = provider or SqlTabularSourceProvider()
def list_origins(
self,
session: object,
principal: object,
*,
query: str = "",
limit: int = 100,
):
try:
rows = self._provider.list_sources(
session,
principal,
query=query,
limit=limit,
)
except TabularSourceError as exc:
raise _datasource_error(exc) from exc
return tuple(_origin(source) for source in rows)
def get_origin(
self,
session: object,
principal: object,
*,
origin_ref: str,
) -> DatasourceOrigin | None:
try:
source = self._provider.get_source(
session,
principal,
source_ref=origin_ref,
)
except TabularSourceError as exc:
raise _datasource_error(exc) from exc
return _origin(source) if source is not None else None
def read_origin(
self,
session: object,
principal: object,
*,
request: DatasourceOriginReadRequest,
) -> DatasourceOriginReadResult:
try:
result = self._provider.read_source(
session,
principal,
request=TabularReadRequest(
source_ref=request.origin_ref,
limit=request.limit,
offset=request.offset,
columns=request.columns,
expected_fingerprint=request.expected_fingerprint,
max_bytes=request.max_bytes,
timeout_ms=request.timeout_ms,
),
)
except TabularSourceError as exc:
raise _datasource_error(exc) from exc
return DatasourceOriginReadResult(
origin=_origin(result.source),
rows=result.rows,
total_rows=result.total_rows,
truncated=result.truncated,
returned_bytes=result.returned_bytes,
elapsed_ms=result.elapsed_ms,
effective_row_limit=result.effective_row_limit,
effective_byte_limit=result.effective_byte_limit,
effective_timeout_ms=result.effective_timeout_ms,
diagnostics=result.diagnostics,
)
def _origin(source: TabularSource) -> DatasourceOrigin:
return DatasourceOrigin(
ref=source.ref,
source_name=source.source_name,
name=source.name,
description=source.description,
kind="upload",
shape="tabular",
supported_modes=("live", "cached"),
provider=f"connectors.{source.provider}",
schema=tuple(
DatasourceField(
name=column.name,
data_type=column.data_type,
nullable=column.nullable,
)
for column in source.schema
),
schema_version=source.schema_version,
fingerprint=source.fingerprint,
row_count=source.row_count,
byte_count=source.byte_count,
updated_at=source.updated_at,
capabilities=source.capabilities,
metadata=dict(source.metadata),
source_mode=source.source_mode,
pushdown=source.pushdown,
health=source.health,
)
def _datasource_error(exc: TabularSourceError):
if isinstance(exc, TabularSourceAccessError):
return DatasourceAccessError(str(exc))
if isinstance(exc, TabularSourceNotFoundError):
return DatasourceNotFoundError(str(exc))
if isinstance(exc, TabularSourceUnavailableError):
return DatasourceUnavailableError(str(exc))
if isinstance(exc, TabularSourceValidationError):
return DatasourceValidationError(str(exc))
return DatasourceValidationError(str(exc))
__all__ = ["ConnectorDatasourceOriginProvider"]
@@ -0,0 +1,3 @@
from govoplan_connectors.backend.db.models import ConnectorTabularSource
__all__ = ["ConnectorTabularSource"]
@@ -0,0 +1,263 @@
from __future__ import annotations
import uuid
from datetime import datetime
from typing import Any
from sqlalchemy import (
DateTime,
ForeignKey,
Index,
Integer,
JSON,
LargeBinary,
String,
Text,
UniqueConstraint,
)
from sqlalchemy.orm import Mapped, mapped_column
from govoplan_core.db.base import Base, TimestampMixin
def new_uuid() -> str:
return str(uuid.uuid4())
class ConnectorTabularSource(Base, TimestampMixin):
__tablename__ = "connector_tabular_sources"
__table_args__ = (
UniqueConstraint("tenant_id", "source_name", name="uq_connector_tabular_source_name"),
Index("ix_connector_tabular_sources_tenant_status", "tenant_id", "status"),
Index("ix_connector_tabular_sources_tenant_updated", "tenant_id", "updated_at"),
)
id: Mapped[str] = mapped_column(String(36), primary_key=True, default=new_uuid)
tenant_id: Mapped[str] = mapped_column(String(36), nullable=False, index=True)
provider: Mapped[str] = mapped_column(String(50), default="snapshot", nullable=False, index=True)
source_name: Mapped[str] = mapped_column(String(120), nullable=False)
name: Mapped[str] = mapped_column(String(300), nullable=False)
description: Mapped[str | None] = mapped_column(Text)
status: Mapped[str] = mapped_column(String(30), default="active", nullable=False, index=True)
schema_version: Mapped[int] = mapped_column(Integer, default=1, nullable=False)
schema_: Mapped[list[dict[str, Any]]] = mapped_column("schema", JSON, default=list, nullable=False)
rows: Mapped[list[dict[str, Any]]] = mapped_column(JSON, default=list, nullable=False)
fingerprint: Mapped[str] = mapped_column(String(64), nullable=False, index=True)
row_count: Mapped[int] = mapped_column(Integer, nullable=False)
byte_count: Mapped[int] = mapped_column(Integer, nullable=False)
metadata_: Mapped[dict[str, Any]] = mapped_column("metadata", JSON, default=dict, nullable=False)
created_by: Mapped[str | None] = mapped_column(String(255), nullable=True, index=True)
updated_by: Mapped[str | None] = mapped_column(String(255), nullable=True, index=True)
deleted_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True, index=True)
class ConnectorSanctionsAcquisitionRun(Base, TimestampMixin):
__tablename__ = "connector_sanctions_acquisition_runs"
__table_args__ = (
Index(
"ix_connector_sanctions_run_health",
"tenant_id",
"provider_id",
"status",
"started_at",
),
)
id: Mapped[str] = mapped_column(
String(36),
primary_key=True,
default=new_uuid,
)
tenant_id: Mapped[str] = mapped_column(
String(36),
nullable=False,
index=True,
)
provider_id: Mapped[str] = mapped_column(
String(100),
nullable=False,
index=True,
)
source_id: Mapped[str] = mapped_column(
String(200),
nullable=False,
index=True,
)
status: Mapped[str] = mapped_column(
String(40),
default="running",
nullable=False,
index=True,
)
attempt_count: Mapped[int] = mapped_column(
Integer,
default=0,
nullable=False,
)
request_evidence: Mapped[dict[str, Any]] = mapped_column(
JSON,
default=dict,
nullable=False,
)
response_evidence: Mapped[dict[str, Any]] = mapped_column(
JSON,
default=dict,
nullable=False,
)
started_at: Mapped[datetime] = mapped_column(
DateTime(timezone=True),
nullable=False,
index=True,
)
finished_at: Mapped[datetime | None] = mapped_column(
DateTime(timezone=True),
nullable=True,
)
snapshot_id: Mapped[str | None] = mapped_column(
String(36),
nullable=True,
index=True,
)
error: Mapped[str | None] = mapped_column(
Text,
nullable=True,
)
created_by: Mapped[str | None] = mapped_column(
String(255),
nullable=True,
index=True,
)
class ConnectorSanctionsSnapshot(Base, TimestampMixin):
__tablename__ = "connector_sanctions_snapshots"
__table_args__ = (
UniqueConstraint(
"connector_run_id",
name="uq_connector_sanctions_snapshot_run",
),
Index(
"ix_connector_sanctions_snapshot_source",
"tenant_id",
"provider_id",
"acquired_at",
),
Index(
"ix_connector_sanctions_snapshot_version",
"provider_id",
"source_id",
"source_version",
),
)
id: Mapped[str] = mapped_column(
String(36),
primary_key=True,
default=new_uuid,
)
tenant_id: Mapped[str] = mapped_column(
String(36),
nullable=False,
index=True,
)
provider_id: Mapped[str] = mapped_column(
String(100),
nullable=False,
index=True,
)
publisher: Mapped[str] = mapped_column(
String(300),
nullable=False,
)
jurisdiction: Mapped[str] = mapped_column(
String(100),
nullable=False,
index=True,
)
list_type: Mapped[str] = mapped_column(
String(100),
nullable=False,
index=True,
)
source_id: Mapped[str] = mapped_column(
String(200),
nullable=False,
index=True,
)
source_version: Mapped[str] = mapped_column(
String(255),
nullable=False,
index=True,
)
publication_at: Mapped[datetime | None] = mapped_column(
DateTime(timezone=True),
nullable=True,
)
effective_at: Mapped[datetime | None] = mapped_column(
DateTime(timezone=True),
nullable=True,
)
acquired_at: Mapped[datetime] = mapped_column(
DateTime(timezone=True),
nullable=False,
index=True,
)
source_url: Mapped[str | None] = mapped_column(
String(1500),
nullable=True,
)
content_type: Mapped[str] = mapped_column(
String(200),
nullable=False,
)
byte_count: Mapped[int] = mapped_column(
Integer,
nullable=False,
)
sha256: Mapped[str] = mapped_column(
String(64),
nullable=False,
index=True,
)
signature_evidence: Mapped[dict[str, Any]] = mapped_column(
JSON,
default=dict,
nullable=False,
)
parser_version: Mapped[str] = mapped_column(
String(100),
nullable=False,
)
licence_notes: Mapped[str | None] = mapped_column(
Text,
nullable=True,
)
trust_notes: Mapped[str | None] = mapped_column(
Text,
nullable=True,
)
connector_run_id: Mapped[str] = mapped_column(
ForeignKey(
"connector_sanctions_acquisition_runs.id",
ondelete="RESTRICT",
),
nullable=False,
index=True,
)
transport_evidence: Mapped[dict[str, Any]] = mapped_column(
JSON,
default=dict,
nullable=False,
)
raw_content: Mapped[bytes] = mapped_column(
LargeBinary,
nullable=False,
)
__all__ = [
"ConnectorSanctionsAcquisitionRun",
"ConnectorSanctionsSnapshot",
"ConnectorTabularSource",
"new_uuid",
]
+413
View File
@@ -0,0 +1,413 @@
from __future__ import annotations
import hashlib
import re
import xml.etree.ElementTree as ET
from collections.abc import Mapping
from dataclasses import replace
from datetime import datetime, timedelta, timezone
from email.utils import format_datetime, parsedate_to_datetime
from defusedxml import ElementTree as SafeET
from defusedxml.common import DefusedXmlException
from govoplan_core.core.feeds import (
FeedCapabilityError,
FeedDocument,
FeedEntry,
FeedProvider,
FeedRenderRequest,
FeedRenderResult,
)
from govoplan_core.security.http_fetch import fetch_http
MAX_FEED_BYTES = 5_000_000
ATOM_NS = "http://www.w3.org/2005/Atom"
class ConnectorFeedProvider(FeedProvider):
def fetch(
self,
url: str,
*,
timeout: float = 15,
max_entries: int = 2_000,
) -> FeedDocument:
try:
response = fetch_http(
url,
timeout=timeout,
label="RSS/Atom feed URL",
headers={
"Accept": (
"application/atom+xml, application/rss+xml, "
"application/xml;q=0.9, text/xml;q=0.8"
)
},
max_bytes=MAX_FEED_BYTES,
)
except Exception as exc:
raise FeedCapabilityError(f"Feed acquisition failed: {exc}") from exc
if response.status < 200 or response.status >= 300:
raise FeedCapabilityError(
f"Feed acquisition returned HTTP {response.status}."
)
content_type = _header(response.headers, "content-type")
document = self.parse(
response.body,
source_url=url,
content_type=content_type,
max_entries=max_entries,
)
acquired_at = datetime.now(timezone.utc)
return replace(
document,
acquired_at=acquired_at,
fresh_until=_fresh_until(response.headers, acquired_at),
etag=_header(response.headers, "etag"),
last_modified=_header(response.headers, "last-modified"),
metadata={
**dict(document.metadata),
"http_status": response.status,
"byte_count": len(response.body),
},
)
def parse(
self,
content: bytes,
*,
source_url: str,
content_type: str | None = None,
max_entries: int = 2_000,
) -> FeedDocument:
if not content:
raise FeedCapabilityError("Feed content is empty.")
if len(content) > MAX_FEED_BYTES:
raise FeedCapabilityError(
f"Feeds are limited to {MAX_FEED_BYTES // 1_000_000} MB."
)
try:
root = SafeET.fromstring(content)
except (ET.ParseError, DefusedXmlException) as exc:
raise FeedCapabilityError(f"Feed XML is not safe or valid: {exc}") from exc
local_name = _local_name(root.tag)
if local_name == "rss":
document = _parse_rss(root, source_url=source_url, max_entries=max_entries)
elif local_name == "feed":
document = _parse_atom(root, source_url=source_url, max_entries=max_entries)
else:
raise FeedCapabilityError("The document is neither an RSS nor an Atom feed.")
return replace(
document,
content_type=content_type,
sha256=hashlib.sha256(content).hexdigest(),
)
def render(self, request: FeedRenderRequest) -> FeedRenderResult:
if not request.title.strip() or not request.feed_url.strip():
raise FeedCapabilityError("Feed title and feed URL are required.")
entries = tuple(
entry
for entry in request.entries
if entry.visibility in request.allowed_visibilities
)
root = (
_render_rss(request, entries)
if request.format == "rss"
else _render_atom(request, entries)
)
body = ET.tostring(root, encoding="utf-8", xml_declaration=True)
return FeedRenderResult(
format=request.format,
content_type=(
"application/rss+xml; charset=utf-8"
if request.format == "rss"
else "application/atom+xml; charset=utf-8"
),
body=body,
included_entries=len(entries),
excluded_entries=len(request.entries) - len(entries),
)
def feed_rows(document: FeedDocument) -> tuple[Mapping[str, object], ...]:
"""Map feed entries to the connector tabular shape used by Datasources."""
return tuple(
{
"id": entry.id,
"title": entry.title,
"url": entry.url,
"summary": entry.summary,
"content": entry.content,
"author": entry.author,
"published_at": (
entry.published_at.isoformat() if entry.published_at else None
),
"updated_at": entry.updated_at.isoformat() if entry.updated_at else None,
"categories": list(entry.categories),
"enclosures": [dict(item) for item in entry.enclosures],
}
for entry in document.entries
)
def _parse_rss(root: ET.Element, *, source_url: str, max_entries: int) -> FeedDocument:
channel = _first_child(root, "channel")
if channel is None:
raise FeedCapabilityError("RSS feed is missing its channel element.")
entries: list[FeedEntry] = []
for item in _children(channel, "item"):
if len(entries) >= max_entries:
raise FeedCapabilityError(f"Feeds are limited to {max_entries:,} entries.")
url = _text(item, "link")
identifier = _text(item, "guid") or url or _entry_fallback_id(item)
entries.append(
FeedEntry(
id=identifier,
title=_text(item, "title") or "(Untitled)",
url=url,
summary=_text(item, "description"),
content=_text(item, "encoded"),
author=_text(item, "author") or _text(item, "creator"),
published_at=_parse_date(_text(item, "pubDate")),
categories=tuple(
value for child in _children(item, "category")
if (value := (child.text or "").strip())
),
enclosures=tuple(
{
"url": child.attrib.get("url"),
"media_type": child.attrib.get("type"),
"size_bytes": _integer(child.attrib.get("length")),
}
for child in _children(item, "enclosure")
),
)
)
return FeedDocument(
format="rss",
title=_text(channel, "title") or "Untitled feed",
source_url=source_url,
entries=tuple(entries),
description=_text(channel, "description"),
home_url=_text(channel, "link"),
language=_text(channel, "language"),
updated_at=_parse_date(
_text(channel, "lastBuildDate") or _text(channel, "pubDate")
),
)
def _parse_atom(root: ET.Element, *, source_url: str, max_entries: int) -> FeedDocument:
entries: list[FeedEntry] = []
for item in _children(root, "entry"):
if len(entries) >= max_entries:
raise FeedCapabilityError(f"Feeds are limited to {max_entries:,} entries.")
alternate = _atom_link(item, "alternate")
identifier = _text(item, "id") or alternate or _entry_fallback_id(item)
author = _first_child(item, "author")
entries.append(
FeedEntry(
id=identifier,
title=_text(item, "title") or "(Untitled)",
url=alternate,
summary=_text(item, "summary"),
content=_text(item, "content"),
author=_text(author, "name") if author is not None else None,
published_at=_parse_date(_text(item, "published")),
updated_at=_parse_date(_text(item, "updated")),
categories=tuple(
value for child in _children(item, "category")
if (value := (child.attrib.get("term") or "").strip())
),
enclosures=tuple(
{
"url": child.attrib.get("href"),
"media_type": child.attrib.get("type"),
"size_bytes": _integer(child.attrib.get("length")),
}
for child in _children(item, "link")
if child.attrib.get("rel") == "enclosure"
),
)
)
return FeedDocument(
format="atom",
title=_text(root, "title") or "Untitled feed",
source_url=source_url,
entries=tuple(entries),
description=_text(root, "subtitle"),
home_url=_atom_link(root, "alternate"),
updated_at=_parse_date(_text(root, "updated")),
)
def _render_rss(request: FeedRenderRequest, entries: tuple[FeedEntry, ...]) -> ET.Element:
ET.register_namespace("atom", ATOM_NS)
root = ET.Element("rss", {"version": "2.0"})
channel = ET.SubElement(root, "channel")
_element(channel, "title", request.title)
_element(channel, "link", request.home_url)
_element(channel, "description", request.description or request.title)
_element(channel, f"{{{ATOM_NS}}}link", None, {
"href": request.feed_url,
"rel": "self",
"type": "application/rss+xml",
})
if request.language:
_element(channel, "language", request.language)
for entry in entries:
item = ET.SubElement(channel, "item")
_element(item, "guid", entry.id, {"isPermaLink": "false"})
_element(item, "title", entry.title)
if entry.url:
_element(item, "link", entry.url)
if entry.summary or entry.content:
_element(item, "description", entry.summary or entry.content)
if entry.author:
_element(item, "author", entry.author)
date = entry.published_at or entry.updated_at
if date:
_element(item, "pubDate", format_datetime(_utc(date)))
for category in entry.categories:
_element(item, "category", category)
for enclosure in entry.enclosures:
attributes = {
"url": str(enclosure.get("url") or ""),
"type": str(enclosure.get("media_type") or "application/octet-stream"),
"length": str(enclosure.get("size_bytes") or 0),
}
if attributes["url"]:
_element(item, "enclosure", None, attributes)
return root
def _render_atom(request: FeedRenderRequest, entries: tuple[FeedEntry, ...]) -> ET.Element:
ET.register_namespace("", ATOM_NS)
root = ET.Element(f"{{{ATOM_NS}}}feed")
_element(root, f"{{{ATOM_NS}}}id", request.feed_url)
_element(root, f"{{{ATOM_NS}}}title", request.title)
_element(root, f"{{{ATOM_NS}}}link", None, {"href": request.home_url})
_element(
root,
f"{{{ATOM_NS}}}link",
None,
{"href": request.feed_url, "rel": "self", "type": "application/atom+xml"},
)
latest = max(
(date for entry in entries for date in (entry.updated_at, entry.published_at) if date),
default=datetime.now(timezone.utc),
)
_element(root, f"{{{ATOM_NS}}}updated", _utc(latest).isoformat().replace("+00:00", "Z"))
if request.description:
_element(root, f"{{{ATOM_NS}}}subtitle", request.description)
for value in entries:
entry = ET.SubElement(root, f"{{{ATOM_NS}}}entry")
_element(entry, f"{{{ATOM_NS}}}id", value.id)
_element(entry, f"{{{ATOM_NS}}}title", value.title)
if value.url:
_element(entry, f"{{{ATOM_NS}}}link", None, {"href": value.url})
if value.summary:
_element(entry, f"{{{ATOM_NS}}}summary", value.summary)
if value.content:
_element(entry, f"{{{ATOM_NS}}}content", value.content, {"type": "html"})
updated = value.updated_at or value.published_at or latest
_element(entry, f"{{{ATOM_NS}}}updated", _utc(updated).isoformat().replace("+00:00", "Z"))
if value.published_at:
_element(entry, f"{{{ATOM_NS}}}published", _utc(value.published_at).isoformat().replace("+00:00", "Z"))
if value.author:
author = ET.SubElement(entry, f"{{{ATOM_NS}}}author")
_element(author, f"{{{ATOM_NS}}}name", value.author)
for category in value.categories:
_element(entry, f"{{{ATOM_NS}}}category", None, {"term": category})
return root
def _children(element: ET.Element, name: str) -> tuple[ET.Element, ...]:
return tuple(child for child in element if _local_name(child.tag) == name)
def _first_child(element: ET.Element, name: str) -> ET.Element | None:
return next((child for child in element if _local_name(child.tag) == name), None)
def _text(element: ET.Element | None, name: str) -> str | None:
if element is None:
return None
child = _first_child(element, name)
if child is None:
return None
value = "".join(child.itertext()).strip()
return value or None
def _local_name(tag: str) -> str:
return tag.rsplit("}", 1)[-1].split(":", 1)[-1]
def _atom_link(element: ET.Element, relation: str) -> str | None:
for child in _children(element, "link"):
if (child.attrib.get("rel") or "alternate") == relation:
return child.attrib.get("href")
return None
def _parse_date(value: str | None) -> datetime | None:
if not value:
return None
try:
parsed = parsedate_to_datetime(value)
except (TypeError, ValueError, OverflowError):
try:
parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
except ValueError:
return None
return _utc(parsed)
def _utc(value: datetime) -> datetime:
if value.tzinfo is None:
return value.replace(tzinfo=timezone.utc)
return value.astimezone(timezone.utc)
def _integer(value: str | None) -> int | None:
try:
return int(value) if value is not None else None
except ValueError:
return None
def _entry_fallback_id(element: ET.Element) -> str:
body = ET.tostring(element, encoding="utf-8")
return f"urn:sha256:{hashlib.sha256(body).hexdigest()}"
def _header(headers: Mapping[str, str], name: str) -> str | None:
lowered = name.casefold()
return next((value for key, value in headers.items() if key.casefold() == lowered), None)
def _fresh_until(headers: Mapping[str, str], acquired_at: datetime) -> datetime | None:
cache_control = _header(headers, "cache-control") or ""
match = re.search(r"(?:^|,)\s*max-age\s*=\s*(\d+)", cache_control, re.IGNORECASE)
if match:
return acquired_at + timedelta(seconds=int(match.group(1)))
return _parse_date(_header(headers, "expires"))
def _element(
parent: ET.Element,
tag: str,
text: str | None,
attributes: Mapping[str, str] | None = None,
) -> ET.Element:
child = ET.SubElement(parent, tag, dict(attributes or {}))
child.text = text
return child
__all__ = ["ConnectorFeedProvider", "MAX_FEED_BYTES", "feed_rows"]
+547
View File
@@ -0,0 +1,547 @@
from __future__ import annotations
from pathlib import Path
from govoplan_core.core.access import (
CAPABILITY_AUTH_PERMISSION_EVALUATOR,
CAPABILITY_AUTH_PRINCIPAL_RESOLVER,
)
from govoplan_core.core.module_guards import (
drop_table_retirement_provider,
persistent_table_uninstall_guard,
)
from govoplan_core.core.datasources import CAPABILITY_DATASOURCE_ORIGINS
from govoplan_core.core.feeds import CAPABILITY_CONNECTORS_FEEDS
from govoplan_core.core.modules import (
DocumentationTopic,
MigrationSpec,
ModuleInterfaceProvider,
ModuleManifest,
PermissionDefinition,
RoleTemplate,
)
from govoplan_core.core.provider_governance import (
ExternalProviderDeclaration,
ExternalProviderStateProviderRegistration,
ModuleArchitectureDeclaration,
ModuleArchitectureDocumentation,
ModuleMaturityEvidence,
ProviderBehaviorDeclaration,
ProviderObjectDeclaration,
)
from govoplan_core.core.tabular_sources import (
CAPABILITY_CONNECTORS_TABULAR_SNAPSHOT_WRITER,
CAPABILITY_CONNECTORS_TABULAR_SOURCES,
)
from govoplan_core.core.sanctions import (
CAPABILITY_CONNECTORS_SANCTIONS_SNAPSHOTS,
)
from govoplan_core.db.base import Base
from govoplan_connectors.backend.db.models import (
ConnectorSanctionsAcquisitionRun,
ConnectorSanctionsSnapshot,
ConnectorTabularSource,
)
from govoplan_connectors.backend.sanctions_sources import (
SANCTIONS_READ_SCOPE,
SANCTIONS_REFRESH_SCOPE,
SqlSanctionsSnapshotProvider,
)
from govoplan_connectors.backend.tabular_sources import (
ADMIN_SCOPE,
READ_SCOPE,
WRITE_SCOPE,
SqlTabularSourceProvider,
)
from govoplan_connectors.backend.datasource_origins import (
ConnectorDatasourceOriginProvider,
)
from govoplan_connectors.backend.feeds import ConnectorFeedProvider
from govoplan_connectors.backend.provider_state import (
SANCTIONS_PROVIDER_ID,
TABULAR_PROVIDER_ID,
sanctions_provider_states,
tabular_provider_states,
)
MODULE_ID = "connectors"
MODULE_VERSION = "0.1.16"
TABULAR_SOURCE_INTERFACE_VERSION = "0.1.0"
DATASOURCE_ORIGIN_INTERFACE_VERSION = "0.1.0"
SANCTIONS_SNAPSHOT_INTERFACE_VERSION = "1.0.0"
FEED_INTERFACE_VERSION = "0.1.0"
CONNECTOR_RUNTIME_INTERFACE_VERSION = "1.0.0"
ARCHITECTURE = ModuleArchitectureDeclaration(
layer="data_reporting_integration",
kind="integration",
maturity="vertical_slice",
evidence=(
ModuleMaturityEvidence(
kind="test",
reference="tests/test_tabular_sources.py",
summary="Exercises tenant-safe immutable tabular snapshots and bounded reads.",
),
ModuleMaturityEvidence(
kind="test",
reference="tests/test_sanctions_sources.py",
summary="Exercises source acquisition health, checksums, retries, and immutable evidence.",
),
ModuleMaturityEvidence(
kind="recovery",
reference="tests/test_recovery.py",
summary="Proves atomic snapshot commits, idempotent replay, distributed fences, tamper rejection, and unknown external-effect handling.",
),
ModuleMaturityEvidence(
kind="documentation",
reference="docs/CONNECTOR_SOURCE_LIFECYCLE.md",
summary="Defines source lifecycle, authority, evidence, and outage boundaries.",
),
),
known_limits=(
"The executable generic datasource origin is an immutable tabular snapshot; database and arbitrary REST profiles remain future providers.",
"Feed publication renders a governed document but does not yet push it to an external publishing endpoint.",
),
supported_authority_modes=(
"external_authoritative",
"external_mirror",
"linked_reference",
),
owned_concepts=(
"external transport profiles",
"protocol interaction",
"immutable connector snapshots",
"connector acquisition health",
),
non_owned_concepts=(
"datasource catalogue identity and lifecycle",
"domain records and business semantics",
"data transformations",
"screening dispositions",
),
target_tested_providers=(
TABULAR_PROVIDER_ID,
SANCTIONS_PROVIDER_ID,
),
documentation=ModuleArchitectureDocumentation(
migration=("src/govoplan_connectors/backend/migrations/versions",),
upgrade=("docs/CONNECTOR_SOURCE_LIFECYCLE.md",),
recovery=("docs/CONNECTOR_SOURCE_LIFECYCLE.md",),
security=("docs/CONNECTOR_SOURCE_LIFECYCLE.md",),
operations=("docs/CONNECTOR_SOURCE_LIFECYCLE.md",),
),
)
EXTERNAL_PROVIDERS = (
ExternalProviderDeclaration(
id=TABULAR_PROVIDER_ID,
module_id=MODULE_ID,
label="Immutable tabular snapshot provider",
maturity="read",
operations=("discover", "search", "read", "preview", "dry_run"),
objects=(
ProviderObjectDeclaration(
object_type="tabular_source_snapshot",
field_groups=("identity", "schema", "rows", "source_provenance"),
authority_modes=("external_authoritative", "external_mirror"),
default_authority_mode="external_mirror",
),
),
behavior=ProviderBehaviorDeclaration(
revision_tokens="Source fingerprints and immutable snapshot ids are retained.",
concurrency="Reads may require the expected fingerprint; snapshots never mutate in place.",
freshness="Snapshot acquisition time and source timestamp are exposed.",
health="Import validation and source-read failures are explicit.",
max_read_items=1000,
idempotency="Feed imports accept a caller request key and replay the same committed immutable source without refetching.",
retry="Read-only acquisition may be retried only as a new deliberate request after a failed atomic operation.",
outcome_unknown="Provider reads do not mutate remote state; an uncertain database commit is resolved by the atomic recovery transaction.",
outcome_unknown_supported=False,
evidence="Rows, schema, fingerprint, source metadata, and acquisition provenance remain linked.",
correction="Import a replacement snapshot; retain the prior snapshot as evidence.",
rollback="Snapshot rows and the terminal recovery checkpoint commit or roll back together.",
reconciliation="Compare source and snapshot fingerprints before selecting a new current state.",
outage="Existing snapshots remain available and visibly stale; no live-source claim is made.",
classifications=("internal", "confidential", "restricted"),
purposes=("governed import", "dataflow input", "evidence reconstruction"),
retention="Datasources or the consuming domain supplies retention and hold policy.",
secret_handling="Generic snapshots contain no connector credential; transport credentials stay in credential envelopes.",
),
capability_names=(
CAPABILITY_CONNECTORS_TABULAR_SOURCES,
CAPABILITY_DATASOURCE_ORIGINS,
),
interface_names=(
"connectors.tabular_sources",
"connectors.datasource_origins",
),
documentation_topic_ids=(
"connectors.authority-and-effects",
"connectors.tabular-sources",
),
),
ExternalProviderDeclaration(
id=SANCTIONS_PROVIDER_ID,
module_id=MODULE_ID,
label="Sanctions source snapshot provider",
maturity="read",
operations=("discover", "search", "read", "preview"),
objects=(
ProviderObjectDeclaration(
object_type="sanctions_source_snapshot",
field_groups=("source_identity", "raw_evidence", "entries", "acquisition_health"),
authority_modes=("external_authoritative", "external_mirror"),
default_authority_mode="external_mirror",
),
),
behavior=ProviderBehaviorDeclaration(
revision_tokens="Provider source version, ETag, Last-Modified, and SHA-256 digest are retained when available.",
concurrency="Refreshes use conditional source requests, a distributed per-tenant/provider fence, and immutable snapshots.",
freshness="Latest successful acquisition, source timestamp, and stale health are reported.",
health="Transport, parsing, source-change, and malformed-source states are explicit.",
max_read_items=5000,
idempotency="A caller request key identifies one acquisition run and replays its committed result without contacting the source again.",
retry="Bounded HTTP retries are safe because acquisition is read-only; failed runs require a new deliberate request key.",
timeout_seconds=30,
outcome_unknown="The external operation is read-only; snapshot rows and recovery evidence commit atomically.",
outcome_unknown_supported=False,
evidence="Raw source bytes, checksum, acquisition run, parser result, and normalized entry count are linked.",
correction="A corrected source creates a new immutable snapshot and acquisition run.",
rollback="A failed database transaction leaves no snapshot and the stale atomic fence resolves as failed.",
reconciliation="Compare source version and digest, then preserve both prior and corrected evidence.",
outage="The latest accepted snapshot stays usable with stale/unavailable source health.",
classifications=("public", "internal"),
purposes=("sanctions source acquisition", "compliance screening evidence"),
retention="Risk and Records policies determine accepted snapshot retention and legal holds.",
secret_handling="Public sources require no subject data or source credential; configured proxy secrets remain external to snapshots.",
),
capability_names=(CAPABILITY_CONNECTORS_SANCTIONS_SNAPSHOTS,),
interface_names=("connectors.sanctions_snapshots",),
documentation_topic_ids=(
"connectors.authority-and-effects",
"connectors.sanctions-snapshots",
),
),
)
def _permission(scope: str, label: str, description: str) -> PermissionDefinition:
module_id, resource, action = scope.split(":", 2)
return PermissionDefinition(
scope=scope,
label=label,
description=description,
category="Connectors",
level="tenant",
module_id=module_id,
resource=resource,
action=action,
)
PERMISSIONS = (
_permission(
READ_SCOPE,
"View tabular sources",
"Discover and preview policy-visible tabular connector sources.",
),
_permission(
WRITE_SCOPE,
"Manage tabular sources",
"Import and retire bounded tabular snapshots.",
),
_permission(
ADMIN_SCOPE,
"Administer connector sources",
"Manage every tenant connector source and future source policies.",
),
_permission(
SANCTIONS_READ_SCOPE,
"View sanctions source evidence",
"Inspect immutable sanctions snapshots and acquisition health.",
),
_permission(
SANCTIONS_REFRESH_SCOPE,
"Refresh sanctions sources",
"Acquire a new immutable sanctions source snapshot.",
),
)
ROLE_TEMPLATES = (
RoleTemplate(
slug="connector_source_manager",
name="Connector source manager",
description="Discover, import, preview, and retire tabular sources.",
permissions=(
READ_SCOPE,
WRITE_SCOPE,
SANCTIONS_READ_SCOPE,
SANCTIONS_REFRESH_SCOPE,
),
),
RoleTemplate(
slug="connector_source_reader",
name="Connector source reader",
description="Discover and preview tabular connector sources.",
permissions=(READ_SCOPE, SANCTIONS_READ_SCOPE),
),
)
def _router(_context):
from govoplan_connectors.backend.router import router
return router
def _provider(_context) -> SqlTabularSourceProvider:
return SqlTabularSourceProvider()
def _datasource_origin_provider(_context) -> ConnectorDatasourceOriginProvider:
return ConnectorDatasourceOriginProvider()
def _sanctions_snapshot_provider(
_context,
) -> SqlSanctionsSnapshotProvider:
return SqlSanctionsSnapshotProvider()
def _feed_provider(_context) -> ConnectorFeedProvider:
return ConnectorFeedProvider()
def _tenant_summary(session, tenant_id: str) -> dict[str, int]:
return {
"connector_tabular_sources": (
session.query(ConnectorTabularSource)
.filter(
ConnectorTabularSource.tenant_id == tenant_id,
ConnectorTabularSource.deleted_at.is_(None),
)
.count()
),
"connector_sanctions_snapshots": (
session.query(ConnectorSanctionsSnapshot)
.filter(ConnectorSanctionsSnapshot.tenant_id == tenant_id)
.count()
),
"connector_sanctions_runs": (
session.query(ConnectorSanctionsAcquisitionRun)
.filter(ConnectorSanctionsAcquisitionRun.tenant_id == tenant_id)
.count()
),
}
manifest = ModuleManifest(
id=MODULE_ID,
name="Connectors",
version=MODULE_VERSION,
optional_dependencies=(
"access",
"audit",
"files",
"policy",
"datasources",
"portal",
"reporting",
"risk_compliance",
),
required_capabilities=(
CAPABILITY_AUTH_PRINCIPAL_RESOLVER,
CAPABILITY_AUTH_PERMISSION_EVALUATOR,
),
provides_interfaces=(
ModuleInterfaceProvider(
name="connectors.tabular_sources",
version=TABULAR_SOURCE_INTERFACE_VERSION,
),
ModuleInterfaceProvider(
name="connectors.tabular_snapshot_writer",
version=TABULAR_SOURCE_INTERFACE_VERSION,
),
ModuleInterfaceProvider(
name="connectors.datasource_origins",
version=DATASOURCE_ORIGIN_INTERFACE_VERSION,
),
ModuleInterfaceProvider(
name="connectors.sanctions_snapshots",
version=SANCTIONS_SNAPSHOT_INTERFACE_VERSION,
),
ModuleInterfaceProvider(
name="connectors.feeds",
version=FEED_INTERFACE_VERSION,
),
ModuleInterfaceProvider(
name="connectors.runtime_contract",
version=CONNECTOR_RUNTIME_INTERFACE_VERSION,
),
),
permissions=PERMISSIONS,
role_templates=ROLE_TEMPLATES,
route_factory=_router,
capability_factories={
CAPABILITY_CONNECTORS_TABULAR_SOURCES: _provider,
CAPABILITY_CONNECTORS_TABULAR_SNAPSHOT_WRITER: _provider,
CAPABILITY_DATASOURCE_ORIGINS: _datasource_origin_provider,
CAPABILITY_CONNECTORS_SANCTIONS_SNAPSHOTS: (_sanctions_snapshot_provider),
CAPABILITY_CONNECTORS_FEEDS: _feed_provider,
},
tenant_summary_providers=(_tenant_summary,),
architecture=ARCHITECTURE,
external_providers=EXTERNAL_PROVIDERS,
external_provider_state_providers=(
ExternalProviderStateProviderRegistration(
module_id=MODULE_ID,
provider_id=TABULAR_PROVIDER_ID,
provider=tabular_provider_states,
),
ExternalProviderStateProviderRegistration(
module_id=MODULE_ID,
provider_id=SANCTIONS_PROVIDER_ID,
provider=sanctions_provider_states,
),
),
migration_spec=MigrationSpec(
module_id=MODULE_ID,
metadata=Base.metadata,
script_location=str(Path(__file__).with_name("migrations") / "versions"),
retirement_supported=True,
retirement_provider=drop_table_retirement_provider(
ConnectorSanctionsSnapshot,
ConnectorSanctionsAcquisitionRun,
ConnectorTabularSource,
label="Connectors",
),
retirement_notes=(
"Destructive retirement drops connector-owned source snapshots after "
"the installer captures a database snapshot."
),
),
uninstall_guard_providers=(
persistent_table_uninstall_guard(
ConnectorSanctionsSnapshot,
ConnectorSanctionsAcquisitionRun,
ConnectorTabularSource,
label="Connectors",
),
),
documentation=(
DocumentationTopic(
id="connectors.authority-and-effects",
title="Connector authority and effect behavior",
summary="Connector direction, technical maturity, and configured source authority are separate and must remain visible.",
body=(
"A connector can consume, publish, or work bidirectionally and can mature from discovery through replacement. "
"Each binding separately states whether GovOPlaN is authoritative, follows an external authority, keeps a mirror, synchronizes under conflict rules, adds a governance overlay, or retains only a link. "
"Writable providers must explain revisions, limits, idempotency, outcome-unknown handling, evidence, reconciliation, correction, outage behavior, and secret requirements."
),
layer="available",
documentation_types=("admin", "user"),
audience=("operator", "module_admin", "power_user", "product_owner"),
related_modules=("datasources", "dataflow", "ops", "policy", "audit"),
order=39,
),
DocumentationTopic(
id="connectors.runtime-preview-contract",
title="Connector previews and diagnostics",
summary="Use one bounded, redacted dry-run shape across external transports.",
body=(
"Connectors owns endpoint discovery, authentication hand-off, transport limits, retries, and protocol health. "
"Domain modules own field mapping, validation, reconciliation, and record mutation. The shared Core runtime "
"contract reports redacted effects and diagnostics with source revisions, fingerprints, and immutable input hashes. "
"Tabular previews enforce effective row, serialized-byte, and elapsed-time ceilings and report limit truncation "
"as structured diagnostics. A commit must reject stale, truncated, conflicting, or error-bearing previews, and "
"credentials never appear in URLs or samples."
),
layer="available",
documentation_types=("admin", "user"),
audience=("operator", "module_admin", "power_user"),
related_modules=("addresses", "datasources", "dataflow", "policy", "audit"),
order=40,
),
DocumentationTopic(
id="connectors.tabular-sources",
title="Governed tabular sources",
summary="Provider-neutral source discovery and bounded reads for Dataflow.",
body=(
"Connectors owns source configuration, access checks, schema discovery, "
"fingerprints, and bounded reads. Dataflow stores only opaque source "
"references and expected fingerprints. Each source declares its live, "
"cached, file-backed, or static mode, structured health, and supported "
"projection, filter, aggregation, sorting, and pagination pushdown. The "
"first executable provider imports immutable JSON or CSV snapshots, "
"supports projection and pagination, and exposes them as Datasource "
"origins. Database and API providers can implement the same origin "
"contract without changing Datasources or Dataflow."
),
layer="available",
documentation_types=("admin", "user"),
audience=("operator", "module_admin", "power_user"),
related_modules=("dataflow", "files", "reporting", "risk_compliance"),
order=40,
),
DocumentationTopic(
id="connectors.rss-atom",
title="RSS and Atom feeds",
summary="Import governed feed snapshots and emit visibility-filtered feeds.",
body=(
"Connectors owns bounded, SSRF-protected RSS/Atom transport and XML "
"parsing. Imported entries become immutable tabular snapshots exposed "
"through Datasources, including acquisition, freshness, ETag, content "
"digest, and source provenance. Portal or Reporting owns publication "
"routes and must pass the allowed visibility set when rendering output. "
"A separate RSS module is only warranted if GovOPlaN later needs a "
"dedicated feed-reader product surface."
),
layer="available",
documentation_types=("admin", "user"),
audience=("operator", "module_admin", "power_user"),
related_modules=("datasources", "dataflow", "portal", "reporting"),
order=42,
),
DocumentationTopic(
id="connectors.sanctions-snapshots",
title="Sanctions source snapshots",
summary=(
"Acquire immutable, checksum-verifiable sanctions list "
"evidence without transmitting screening subjects."
),
body=(
"Connectors provides a deterministic synthetic fixture and "
"the official United Nations Security Council consolidated "
"XML source. Each fetch records conditional transport "
"evidence, bounded retries, health state, source metadata, "
"raw evidence, and a SHA-256 checksum. Refreshes acquire a "
"distributed recovery fence before provider I/O; the immutable "
"snapshot and terminal recovery checkpoint then commit in one "
"transaction. A repeated request key returns the same result. "
"Risk Compliance owns "
"normalization, matching, legal review, and dispositions."
),
layer="available",
documentation_types=("admin", "user"),
audience=("operator", "module_admin", "compliance_reviewer"),
related_modules=("risk_compliance", "dataflow"),
order=41,
),
),
)
def get_manifest() -> ModuleManifest:
return manifest
__all__ = [
"MODULE_ID",
"MODULE_VERSION",
"DATASOURCE_ORIGIN_INTERFACE_VERSION",
"SANCTIONS_SNAPSHOT_INTERFACE_VERSION",
"TABULAR_SOURCE_INTERFACE_VERSION",
"get_manifest",
"manifest",
]
@@ -0,0 +1 @@
"""Connectors migrations."""
@@ -0,0 +1 @@
"""Connectors migration revisions."""
@@ -0,0 +1,141 @@
"""v0.1.14 Connectors baseline
Revision ID: e6b7c8d9f0a1
Revises: None
Create Date: 2026-07-28 00:00:00.000000
"""
from __future__ import annotations
from alembic import op
import sqlalchemy as sa
revision = "e6b7c8d9f0a1"
down_revision = None
branch_labels = None
depends_on = None
def upgrade() -> None:
op.create_table(
"connector_tabular_sources",
sa.Column("id", sa.String(length=36), nullable=False),
sa.Column("tenant_id", sa.String(length=36), nullable=False),
sa.Column("provider", sa.String(length=50), nullable=False),
sa.Column("source_name", sa.String(length=120), nullable=False),
sa.Column("name", sa.String(length=300), nullable=False),
sa.Column("description", sa.Text(), nullable=True),
sa.Column("status", sa.String(length=30), nullable=False),
sa.Column("schema_version", sa.Integer(), nullable=False),
sa.Column("schema", sa.JSON(), nullable=False),
sa.Column("rows", sa.JSON(), nullable=False),
sa.Column("fingerprint", sa.String(length=64), nullable=False),
sa.Column("row_count", sa.Integer(), nullable=False),
sa.Column("byte_count", sa.Integer(), nullable=False),
sa.Column("metadata", sa.JSON(), nullable=False),
sa.Column("created_by", sa.String(length=255), nullable=True),
sa.Column("updated_by", sa.String(length=255), nullable=True),
sa.Column("deleted_at", sa.DateTime(timezone=True), nullable=True),
sa.Column("created_at", sa.DateTime(timezone=True), nullable=False),
sa.Column("updated_at", sa.DateTime(timezone=True), nullable=False),
sa.PrimaryKeyConstraint("id", name=op.f("pk_connector_tabular_sources")),
sa.UniqueConstraint(
"tenant_id",
"source_name",
name="uq_connector_tabular_source_name",
),
)
op.create_index(
op.f("ix_connector_tabular_sources_created_by"),
"connector_tabular_sources",
["created_by"],
unique=False,
)
op.create_index(
op.f("ix_connector_tabular_sources_deleted_at"),
"connector_tabular_sources",
["deleted_at"],
unique=False,
)
op.create_index(
op.f("ix_connector_tabular_sources_fingerprint"),
"connector_tabular_sources",
["fingerprint"],
unique=False,
)
op.create_index(
op.f("ix_connector_tabular_sources_provider"),
"connector_tabular_sources",
["provider"],
unique=False,
)
op.create_index(
op.f("ix_connector_tabular_sources_status"),
"connector_tabular_sources",
["status"],
unique=False,
)
op.create_index(
op.f("ix_connector_tabular_sources_tenant_id"),
"connector_tabular_sources",
["tenant_id"],
unique=False,
)
op.create_index(
op.f("ix_connector_tabular_sources_updated_by"),
"connector_tabular_sources",
["updated_by"],
unique=False,
)
op.create_index(
"ix_connector_tabular_sources_tenant_status",
"connector_tabular_sources",
["tenant_id", "status"],
unique=False,
)
op.create_index(
"ix_connector_tabular_sources_tenant_updated",
"connector_tabular_sources",
["tenant_id", "updated_at"],
unique=False,
)
def downgrade() -> None:
op.drop_index(
"ix_connector_tabular_sources_tenant_updated",
table_name="connector_tabular_sources",
)
op.drop_index(
"ix_connector_tabular_sources_tenant_status",
table_name="connector_tabular_sources",
)
op.drop_index(
op.f("ix_connector_tabular_sources_updated_by"),
table_name="connector_tabular_sources",
)
op.drop_index(
op.f("ix_connector_tabular_sources_tenant_id"),
table_name="connector_tabular_sources",
)
op.drop_index(
op.f("ix_connector_tabular_sources_status"),
table_name="connector_tabular_sources",
)
op.drop_index(
op.f("ix_connector_tabular_sources_provider"),
table_name="connector_tabular_sources",
)
op.drop_index(
op.f("ix_connector_tabular_sources_fingerprint"),
table_name="connector_tabular_sources",
)
op.drop_index(
op.f("ix_connector_tabular_sources_deleted_at"),
table_name="connector_tabular_sources",
)
op.drop_index(
op.f("ix_connector_tabular_sources_created_by"),
table_name="connector_tabular_sources",
)
op.drop_table("connector_tabular_sources")
@@ -0,0 +1,192 @@
"""Add immutable sanctions source snapshots.
Revision ID: f7c8d9e0a1b2
Revises: e6b7c8d9f0a1
Create Date: 2026-07-29
"""
from __future__ import annotations
from alembic import op
import sqlalchemy as sa
revision = "f7c8d9e0a1b2"
down_revision = "e6b7c8d9f0a1"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.create_table(
"connector_sanctions_acquisition_runs",
sa.Column("id", sa.String(length=36), nullable=False),
sa.Column("tenant_id", sa.String(length=36), nullable=False),
sa.Column("provider_id", sa.String(length=100), nullable=False),
sa.Column("source_id", sa.String(length=200), nullable=False),
sa.Column("status", sa.String(length=40), nullable=False),
sa.Column("attempt_count", sa.Integer(), nullable=False),
sa.Column("request_evidence", sa.JSON(), nullable=False),
sa.Column("response_evidence", sa.JSON(), nullable=False),
sa.Column(
"started_at",
sa.DateTime(timezone=True),
nullable=False,
),
sa.Column(
"finished_at",
sa.DateTime(timezone=True),
nullable=True,
),
sa.Column("snapshot_id", sa.String(length=36), nullable=True),
sa.Column("error", sa.Text(), nullable=True),
sa.Column("created_by", sa.String(length=255), nullable=True),
sa.Column(
"created_at",
sa.DateTime(timezone=True),
nullable=False,
),
sa.Column(
"updated_at",
sa.DateTime(timezone=True),
nullable=False,
),
sa.PrimaryKeyConstraint(
"id",
name=op.f("pk_connector_sanctions_acquisition_runs"),
),
)
for column in (
"tenant_id",
"provider_id",
"source_id",
"status",
"started_at",
"snapshot_id",
"created_by",
):
op.create_index(
op.f(
"ix_connector_sanctions_acquisition_runs_"
f"{column}"
),
"connector_sanctions_acquisition_runs",
[column],
)
op.create_index(
"ix_connector_sanctions_run_health",
"connector_sanctions_acquisition_runs",
["tenant_id", "provider_id", "status", "started_at"],
)
op.create_table(
"connector_sanctions_snapshots",
sa.Column("id", sa.String(length=36), nullable=False),
sa.Column("tenant_id", sa.String(length=36), nullable=False),
sa.Column("provider_id", sa.String(length=100), nullable=False),
sa.Column("publisher", sa.String(length=300), nullable=False),
sa.Column("jurisdiction", sa.String(length=100), nullable=False),
sa.Column("list_type", sa.String(length=100), nullable=False),
sa.Column("source_id", sa.String(length=200), nullable=False),
sa.Column(
"source_version",
sa.String(length=255),
nullable=False,
),
sa.Column(
"publication_at",
sa.DateTime(timezone=True),
nullable=True,
),
sa.Column(
"effective_at",
sa.DateTime(timezone=True),
nullable=True,
),
sa.Column(
"acquired_at",
sa.DateTime(timezone=True),
nullable=False,
),
sa.Column("source_url", sa.String(length=1500), nullable=True),
sa.Column(
"content_type",
sa.String(length=200),
nullable=False,
),
sa.Column("byte_count", sa.Integer(), nullable=False),
sa.Column("sha256", sa.String(length=64), nullable=False),
sa.Column("signature_evidence", sa.JSON(), nullable=False),
sa.Column(
"parser_version",
sa.String(length=100),
nullable=False,
),
sa.Column("licence_notes", sa.Text(), nullable=True),
sa.Column("trust_notes", sa.Text(), nullable=True),
sa.Column(
"connector_run_id",
sa.String(length=36),
nullable=False,
),
sa.Column("transport_evidence", sa.JSON(), nullable=False),
sa.Column("raw_content", sa.LargeBinary(), nullable=False),
sa.Column(
"created_at",
sa.DateTime(timezone=True),
nullable=False,
),
sa.Column(
"updated_at",
sa.DateTime(timezone=True),
nullable=False,
),
sa.ForeignKeyConstraint(
["connector_run_id"],
["connector_sanctions_acquisition_runs.id"],
name=op.f(
"fk_connector_sanctions_snapshots_connector_run_id_"
"connector_sanctions_acquisition_runs"
),
ondelete="RESTRICT",
),
sa.PrimaryKeyConstraint(
"id",
name=op.f("pk_connector_sanctions_snapshots"),
),
sa.UniqueConstraint(
"connector_run_id",
name="uq_connector_sanctions_snapshot_run",
),
)
for column in (
"tenant_id",
"provider_id",
"jurisdiction",
"list_type",
"source_id",
"source_version",
"acquired_at",
"sha256",
"connector_run_id",
):
op.create_index(
op.f(f"ix_connector_sanctions_snapshots_{column}"),
"connector_sanctions_snapshots",
[column],
)
op.create_index(
"ix_connector_sanctions_snapshot_source",
"connector_sanctions_snapshots",
["tenant_id", "provider_id", "acquired_at"],
)
op.create_index(
"ix_connector_sanctions_snapshot_version",
"connector_sanctions_snapshots",
["provider_id", "source_id", "source_version"],
)
def downgrade() -> None:
op.drop_table("connector_sanctions_snapshots")
op.drop_table("connector_sanctions_acquisition_runs")
@@ -0,0 +1,207 @@
from __future__ import annotations
from collections import defaultdict
from datetime import UTC, datetime
from hashlib import sha256
from sqlalchemy import func, select
from sqlalchemy.orm import Session
from govoplan_connectors.backend.db.models import (
ConnectorSanctionsAcquisitionRun,
ConnectorSanctionsSnapshot,
ConnectorTabularSource,
)
from govoplan_core.core.provider_governance import (
ExternalProviderRuntimeState,
ExternalProviderStateContext,
)
TABULAR_PROVIDER_ID = "connectors.tabular_snapshot"
SANCTIONS_PROVIDER_ID = "connectors.sanctions_snapshot"
def tabular_provider_states(
context: ExternalProviderStateContext,
) -> tuple[ExternalProviderRuntimeState, ...]:
session = _session(context)
statement = select(ConnectorTabularSource).where(
ConnectorTabularSource.deleted_at.is_(None)
)
if context.tenant_id is not None:
statement = statement.where(
ConnectorTabularSource.tenant_id == context.tenant_id
)
sources = tuple(
session.scalars(
statement.order_by(
ConnectorTabularSource.tenant_id,
ConnectorTabularSource.id,
).limit(context.max_items + 1)
)
)
observed_at = datetime.now(UTC)
return tuple(_tabular_state(item, observed_at=observed_at) for item in sources)
def sanctions_provider_states(
context: ExternalProviderStateContext,
) -> tuple[ExternalProviderRuntimeState, ...]:
session = _session(context)
statement = select(ConnectorSanctionsAcquisitionRun)
if context.tenant_id is not None:
statement = statement.where(
ConnectorSanctionsAcquisitionRun.tenant_id == context.tenant_id
)
runs = tuple(
session.scalars(
statement.order_by(
ConnectorSanctionsAcquisitionRun.tenant_id,
ConnectorSanctionsAcquisitionRun.provider_id,
ConnectorSanctionsAcquisitionRun.source_id,
ConnectorSanctionsAcquisitionRun.started_at.desc(),
).limit(max(context.max_items * 10, context.max_items + 1))
)
)
latest_by_binding: dict[tuple[str, str, str], ConnectorSanctionsAcquisitionRun] = {}
for run in runs:
key = (run.tenant_id, run.provider_id, run.source_id)
latest_by_binding.setdefault(key, run)
if len(latest_by_binding) >= context.max_items + 1:
break
snapshot_counts = _snapshot_counts(
session,
binding_keys=tuple(latest_by_binding),
)
observed_at = datetime.now(UTC)
return tuple(
_sanctions_state(
run,
observed_at=observed_at,
snapshot_count=snapshot_counts.get(key, 0),
)
for key, run in latest_by_binding.items()
)
def _session(context: ExternalProviderStateContext) -> Session:
if not isinstance(context.session, Session):
raise RuntimeError("Connectors provider state requires a database session.")
return context.session
def _tabular_state(
source: ConnectorTabularSource,
*,
observed_at: datetime,
) -> ExternalProviderRuntimeState:
active = source.status == "active"
return ExternalProviderRuntimeState(
provider_id=TABULAR_PROVIDER_ID,
binding_ref=f"connectors:tabular-source:{source.id}",
authority_mode="external_mirror",
observed_at=observed_at,
configured=True,
active=active,
health="healthy" if active else "inactive",
freshness="not_applicable",
conflict="not_applicable",
recovery="ready" if active else "not_applicable",
last_success_at=_aware(source.updated_at or source.created_at),
detail=(
"Immutable tabular snapshot is available."
if active
else "Immutable tabular snapshot is inactive."
),
metrics={
"row_count": int(source.row_count),
"byte_count": int(source.byte_count),
"schema_version": int(source.schema_version),
},
)
def _snapshot_counts(
session: Session,
*,
binding_keys: tuple[tuple[str, str, str], ...],
) -> dict[tuple[str, str, str], int]:
if not binding_keys:
return {}
tenant_ids = {item[0] for item in binding_keys}
rows = session.execute(
select(
ConnectorSanctionsSnapshot.tenant_id,
ConnectorSanctionsSnapshot.provider_id,
ConnectorSanctionsSnapshot.source_id,
func.count(ConnectorSanctionsSnapshot.id),
)
.where(ConnectorSanctionsSnapshot.tenant_id.in_(tenant_ids))
.group_by(
ConnectorSanctionsSnapshot.tenant_id,
ConnectorSanctionsSnapshot.provider_id,
ConnectorSanctionsSnapshot.source_id,
)
)
return {
(str(tenant_id), str(provider_id), str(source_id)): int(count)
for tenant_id, provider_id, source_id, count in rows
if (str(tenant_id), str(provider_id), str(source_id)) in binding_keys
}
def _sanctions_state(
run: ConnectorSanctionsAcquisitionRun,
*,
observed_at: datetime,
snapshot_count: int,
) -> ExternalProviderRuntimeState:
status = str(run.status)
success = status in {"succeeded", "success", "not_modified"}
running = status in {"running", "pending", "retry"}
has_snapshot = bool(run.snapshot_id) or snapshot_count > 0
health = "healthy" if success else "warning" if running else "error"
binding_digest = sha256(
f"{run.tenant_id}\0{run.provider_id}\0{run.source_id}".encode("utf-8")
).hexdigest()[:24]
return ExternalProviderRuntimeState(
provider_id=SANCTIONS_PROVIDER_ID,
binding_ref=f"connectors:sanctions-source:{binding_digest}",
authority_mode="external_mirror",
observed_at=observed_at,
configured=True,
active=True,
health=health,
freshness="unknown",
conflict="not_applicable",
recovery="ready" if success and has_snapshot else "attention",
last_success_at=_aware(run.finished_at) if success else None,
detail=(
"Latest sanctions acquisition completed."
if success
else "Sanctions acquisition is in progress."
if running
else "Latest sanctions acquisition failed; prior accepted snapshots remain separate evidence."
),
metrics={
"latest_status": status,
"attempt_count": int(run.attempt_count),
"accepted_snapshots": int(snapshot_count),
},
)
def _aware(value: datetime | None) -> datetime | None:
if value is None:
return None
return value.replace(tzinfo=UTC) if value.tzinfo is None else value.astimezone(UTC)
__all__ = [
"SANCTIONS_PROVIDER_ID",
"TABULAR_PROVIDER_ID",
"sanctions_provider_states",
"tabular_provider_states",
]
+381
View File
@@ -0,0 +1,381 @@
from __future__ import annotations
from dataclasses import dataclass
import hashlib
from typing import Any
from uuid import NAMESPACE_URL, uuid4, uuid5
from sqlalchemy.orm import Session, sessionmaker
from govoplan_core.core.recovery import (
RecoveryGuaranteeError,
RecoveryMode,
RecoveryPlan,
RecoveryStatus,
)
from govoplan_core.core.recovery_runtime import (
DurableRecoveryOperation,
RecoveryOperationBusy,
RecoveryOperationStateConflict,
begin_durable_recovery_operation,
)
from govoplan_core.core.runtime_coordination import process_runtime_identity
class ConnectorRecoveryError(RuntimeError):
pass
@dataclass(frozen=True, slots=True)
class ConnectorRecoveryDeclaration:
operation_type: str
mode: RecoveryMode
provider_mutation: bool
idempotency: str
verification: tuple[str, ...]
recovery: tuple[str, ...]
implemented: bool
CONNECTOR_RECOVERY_OPERATIONS = (
ConnectorRecoveryDeclaration(
operation_type="read-snapshot",
mode=RecoveryMode.ATOMIC,
provider_mutation=False,
idempotency=(
"Caller-supplied request keys replay a committed immutable snapshot; "
"otherwise each deliberate acquisition receives a generated key."
),
verification=(
"provider revision or conditional cursor is recorded before fetch",
"domain snapshot and terminal recovery checkpoint commit together",
"stored bytes and provider evidence are checksum verified",
),
recovery=(
"a stale running transaction is failed after its database transaction rolls back",
"a new deliberate acquisition may then use a new request key",
),
implemented=True,
),
ConnectorRecoveryDeclaration(
operation_type="external-mutation",
mode=RecoveryMode.FORWARD_RECOVERY,
provider_mutation=True,
idempotency="A stable caller key and canonical request digest are mandatory.",
verification=(
"record the remote revision and bounded provider result",
"verify the provider state before reporting success",
),
recovery=(
"unknown outcomes remain unresolved until provider-backed reconciliation",
"never retry the same remote effect solely to reconstruct local state",
),
implemented=False,
),
)
def connector_session_factory(session: Session) -> sessionmaker[Session]:
bind = session.get_bind()
if bind is None:
raise ConnectorRecoveryError("Connector recovery requires a bound database session")
return sessionmaker(bind=bind, expire_on_commit=False)
def _digest(value: str) -> str:
return hashlib.sha256(value.encode("utf-8")).hexdigest()
def _clean_key(value: str | None) -> str:
clean = str(value or "").strip()
if clean and len(clean) > 500:
raise ConnectorRecoveryError("Connector idempotency keys are limited to 500 characters")
return clean or str(uuid4())
def _stable_resource_id(
*,
tenant_id: str,
provider_id: str,
operation_type: str,
request_key: str,
) -> str:
return str(
uuid5(
NAMESPACE_URL,
f"govoplan:{tenant_id}:{provider_id}:{operation_type}:{request_key}",
)
)
@dataclass(slots=True)
class ConnectorReadSnapshotRecovery:
operation: DurableRecoveryOperation | None
operation_id: str
request_key: str
resource_id: str
replayed: bool
def commit_success(self, session: Session, *, evidence: dict[str, Any]) -> None:
if self.operation is None:
raise ConnectorRecoveryError("A replayed connector read cannot be committed again")
try:
self.operation.commit_atomic_success(session, evidence=evidence)
except Exception as exc:
raise ConnectorRecoveryError(
"The connector snapshot and recovery evidence did not commit atomically"
) from exc
def commit_failure(
self,
session: Session,
*,
summary: str,
evidence: dict[str, Any],
) -> None:
if self.operation is None:
raise ConnectorRecoveryError("A replayed connector read cannot be failed again")
try:
self.operation.commit_atomic_failure(
session,
summary=summary,
evidence=evidence,
)
except Exception as exc:
raise ConnectorRecoveryError(
"The connector failure evidence did not commit atomically"
) from exc
def fail_without_projection(
self,
*,
summary: str,
code: str,
) -> None:
if self.operation is None:
return
self.operation.fail(
summary=summary,
evidence={
"verified": True,
"checks": {
"provider_mutation": False,
"projection_committed": False,
"failure_code": code,
},
},
)
def begin_connector_read_snapshot(
session: Session,
*,
tenant_id: str,
provider_id: str,
idempotency_key: str | None,
source_revision: str | None,
cursor: str | None,
dry_run_evidence: dict[str, Any],
request_metadata: dict[str, Any] | None = None,
resource_type: str = "connector_sync_run",
) -> ConnectorReadSnapshotRecovery:
request_key = _clean_key(idempotency_key)
resource_id = _stable_resource_id(
tenant_id=tenant_id,
provider_id=provider_id,
operation_type="read-snapshot",
request_key=request_key,
)
request = {
"tenant_id": tenant_id,
"provider_id": provider_id,
"dry_run": dry_run_evidence,
"request_key_sha256": _digest(request_key),
**dict(request_metadata or {}),
}
try:
started = begin_durable_recovery_operation(
connector_session_factory(session),
identity=process_runtime_identity(),
module_id="connectors",
operation_type="read-snapshot",
idempotency_key=f"connector-read:{_digest(f'{tenant_id}:{provider_id}:{request_key}')}",
request=request,
recovery_plan=RecoveryPlan(
mode=RecoveryMode.ATOMIC,
preconditions=(
"the actor is authorized for the connector source",
"the provider request is read-only",
"the source revision, cursor, and dry-run decision are durable",
),
verification_steps=(
"validate the bounded provider response and source revision",
"commit the immutable snapshot and terminal checkpoint atomically",
"compare the stored content digest with the acquired bytes",
),
),
precondition_evidence={
"provider_id": provider_id,
"source_revision": source_revision,
"cursor_sha256": _digest(cursor) if cursor else None,
"dry_run": dry_run_evidence,
"provider_mutation": False,
},
lease_resource_key=f"connectors:read:{tenant_id}:{_digest(provider_id)[:40]}",
lease_ttl_seconds=15 * 60,
resource_type=resource_type,
resource_id=resource_id,
metadata={
"resources": ["postgresql", "external-provider"],
"provider_mutation": False,
"recovery_declaration": "read-snapshot",
},
)
except RecoveryOperationBusy as exc:
raise ConnectorRecoveryError(
"Another runtime is already acquiring this connector source"
) from exc
except RecoveryOperationStateConflict as exc:
raise ConnectorRecoveryError(
"This connector request is active or unresolved; reconcile it before retrying"
) from exc
except (RecoveryGuaranteeError, RuntimeError) as exc:
raise ConnectorRecoveryError(
"The connector recovery ledger is unavailable; the provider was not contacted"
) from exc
return ConnectorReadSnapshotRecovery(
operation=started.operation,
operation_id=started.operation_id,
request_key=request_key,
resource_id=resource_id,
replayed=started.replayed,
)
@dataclass(slots=True)
class ConnectorExternalMutationRecovery:
operation: DurableRecoveryOperation | None
operation_id: str
replayed: bool
def succeed(self, *, provider_evidence: dict[str, Any]) -> None:
if self.operation is not None:
self.operation.succeed(evidence=provider_evidence)
def reject(self, *, summary: str, provider_code: str) -> None:
if self.operation is not None:
self.operation.reject(
summary=summary,
evidence={
"verified": True,
"checks": {"provider_rejection": provider_code},
},
)
def outcome_unknown(self, *, summary: str, provider_code: str) -> None:
if self.operation is not None:
self.operation.unresolved(
status=RecoveryStatus.OUTCOME_UNKNOWN,
summary=summary,
evidence={"effect_started": True, "provider_code": provider_code},
failure_summary="Inspect provider state before any retry",
)
def begin_connector_external_mutation(
session: Session,
*,
tenant_id: str,
provider_id: str,
idempotency_key: str,
request_sha256: str,
source_revision: str | None,
cursor: str | None,
dry_run_evidence: dict[str, Any],
resource_type: str,
resource_id: str,
) -> ConnectorExternalMutationRecovery:
if not str(idempotency_key or "").strip():
raise ConnectorRecoveryError("External connector mutations require an idempotency key")
request_key = _clean_key(idempotency_key)
if len(request_sha256) != 64 or any(
character not in "0123456789abcdefABCDEF" for character in request_sha256
):
raise ConnectorRecoveryError("External connector mutations require a SHA-256 request digest")
try:
started = begin_durable_recovery_operation(
connector_session_factory(session),
identity=process_runtime_identity(),
module_id="connectors",
operation_type="external-mutation",
idempotency_key=f"connector-write:{_digest(f'{tenant_id}:{provider_id}:{request_key}')}",
request={
"tenant_id": tenant_id,
"provider_id": provider_id,
"request_sha256": request_sha256,
"source_revision": source_revision,
"cursor_sha256": _digest(cursor) if cursor else None,
"dry_run": dry_run_evidence,
},
recovery_plan=RecoveryPlan(
mode=RecoveryMode.FORWARD_RECOVERY,
preconditions=(
"the actor and effective connector policy authorize the mutation",
"a stable idempotency key and canonical request digest are present",
"the dry-run and source revision evidence are durable",
),
forward_recovery_steps=(
"inspect provider state without repeating the mutation",
"record whether the provider accepted the requested revision",
"retry only under a new deliberate key when absence is proven",
),
verification_steps=(
"compare provider identity and revision with the canonical request",
"verify the consuming domain state independently",
),
),
precondition_evidence={
"request_sha256": request_sha256,
"source_revision": source_revision,
"cursor_sha256": _digest(cursor) if cursor else None,
"dry_run": dry_run_evidence,
"provider_mutation": True,
},
lease_resource_key=(
f"connectors:write:{tenant_id}:{_digest(provider_id)[:24]}:"
f"{_digest(resource_id)[:24]}"
),
lease_ttl_seconds=15 * 60,
resource_type=resource_type,
resource_id=resource_id,
metadata={
"resources": ["postgresql", "queue", "external-provider"],
"provider_mutation": True,
"recovery_declaration": "external-mutation",
},
)
except (RecoveryOperationBusy, RecoveryOperationStateConflict) as exc:
raise ConnectorRecoveryError(
"This external connector effect is active or unresolved"
) from exc
except (RecoveryGuaranteeError, RuntimeError) as exc:
raise ConnectorRecoveryError(
"The connector recovery ledger is unavailable; no external mutation started"
) from exc
return ConnectorExternalMutationRecovery(
operation=started.operation,
operation_id=started.operation_id,
replayed=started.replayed,
)
__all__ = [
"CONNECTOR_RECOVERY_OPERATIONS",
"ConnectorExternalMutationRecovery",
"ConnectorReadSnapshotRecovery",
"ConnectorRecoveryDeclaration",
"ConnectorRecoveryError",
"begin_connector_external_mutation",
"begin_connector_read_snapshot",
"connector_session_factory",
]
+689
View File
@@ -0,0 +1,689 @@
from __future__ import annotations
from dataclasses import asdict
import hashlib
from typing import Annotated
from fastapi import APIRouter, Depends, Header, HTTPException, Query, Response, status
from sqlalchemy.orm import Session
from govoplan_core.audit.logging import audit_event
from govoplan_core.auth import ApiPrincipal, get_api_principal, has_scope
from govoplan_core.core.tabular_sources import (
TabularReadRequest,
TabularSnapshotInput,
TabularSource,
TabularSourceAccessError,
TabularSourceError,
TabularSourceNotFoundError,
TabularSourceUnavailableError,
)
from govoplan_core.core.feeds import (
FeedCapabilityError,
FeedEntry,
FeedRenderRequest,
)
from govoplan_core.core.sanctions import SanctionsSnapshotReference
from govoplan_core.db.session import get_session
from govoplan_connectors.backend.schemas import (
FeedAcquireRequest,
FeedDocumentResponse,
FeedImportRequest,
FeedRenderPayload,
SanctionsAcquisitionRunListResponse,
SanctionsAcquisitionRunResponse,
SanctionsRefreshResponse,
SanctionsSnapshotListResponse,
SanctionsSnapshotResponse,
SanctionsSourceListResponse,
SanctionsSourceResponse,
SnapshotCreateRequest,
TabularColumnResponse,
TabularHealthResponse,
TabularPreviewDiagnosticResponse,
TabularPushdownResponse,
TabularSourceDeleteResponse,
TabularSourceListResponse,
TabularSourcePreviewResponse,
TabularSourceResponse,
)
from govoplan_connectors.backend.feeds import ConnectorFeedProvider, feed_rows
from govoplan_connectors.backend.recovery import (
ConnectorRecoveryError,
begin_connector_read_snapshot,
)
from govoplan_connectors.backend.sanctions_sources import (
SANCTIONS_READ_SCOPE,
SANCTIONS_REFRESH_SCOPE,
SanctionsSourceAccessError,
SanctionsSourceError,
SanctionsSourceNotFoundError,
SqlSanctionsSnapshotProvider,
)
from govoplan_connectors.backend.tabular_sources import (
ADMIN_SCOPE,
READ_SCOPE,
WRITE_SCOPE,
SqlTabularSourceProvider,
parse_csv_snapshot,
)
router = APIRouter(prefix="/connectors", tags=["connectors"])
provider = SqlTabularSourceProvider()
sanctions_provider = SqlSanctionsSnapshotProvider()
feed_transport = ConnectorFeedProvider()
def _require_any_scope(principal: ApiPrincipal, *scopes: str) -> None:
if any(has_scope(principal, scope) for scope in scopes):
return
raise HTTPException(
status_code=status.HTTP_403_FORBIDDEN,
detail=f"Missing one of the required scopes: {', '.join(scopes)}",
)
def _http_error(exc: TabularSourceError) -> HTTPException:
if isinstance(exc, TabularSourceNotFoundError):
return HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail=str(exc))
if isinstance(exc, TabularSourceAccessError):
return HTTPException(status_code=status.HTTP_403_FORBIDDEN, detail=str(exc))
if isinstance(exc, TabularSourceUnavailableError):
return HTTPException(
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
detail=str(exc),
)
return HTTPException(status_code=status.HTTP_422_UNPROCESSABLE_CONTENT, detail=str(exc))
def _sanctions_http_error(
exc: SanctionsSourceError,
) -> HTTPException:
if isinstance(exc, SanctionsSourceNotFoundError):
return HTTPException(
status_code=status.HTTP_404_NOT_FOUND,
detail=str(exc),
)
if isinstance(exc, SanctionsSourceAccessError):
return HTTPException(
status_code=status.HTTP_403_FORBIDDEN,
detail=str(exc),
)
return HTTPException(
status_code=status.HTTP_422_UNPROCESSABLE_CONTENT,
detail=str(exc),
)
def _feed_http_error(exc: FeedCapabilityError) -> HTTPException:
return HTTPException(
status_code=status.HTTP_422_UNPROCESSABLE_CONTENT,
detail=str(exc),
)
def _recovery_http_error(exc: ConnectorRecoveryError) -> HTTPException:
detail = str(exc)
return HTTPException(
status_code=(
status.HTTP_409_CONFLICT
if "already" in detail.casefold() or "active" in detail.casefold()
else status.HTTP_503_SERVICE_UNAVAILABLE
),
detail=detail,
)
@router.post("/feeds/preview", response_model=FeedDocumentResponse)
def api_preview_feed(
payload: FeedAcquireRequest,
principal: ApiPrincipal = Depends(get_api_principal),
) -> FeedDocumentResponse:
_require_any_scope(principal, READ_SCOPE, ADMIN_SCOPE)
try:
document = feed_transport.fetch(
payload.url,
max_entries=payload.max_entries,
)
except FeedCapabilityError as exc:
raise _feed_http_error(exc) from exc
return FeedDocumentResponse.model_validate(asdict(document))
@router.post(
"/feeds/import",
response_model=TabularSourceResponse,
status_code=status.HTTP_201_CREATED,
)
def api_import_feed_snapshot(
payload: FeedImportRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
idempotency_key: Annotated[
str | None,
Header(alias="Idempotency-Key", max_length=500),
] = None,
) -> TabularSourceResponse:
_require_any_scope(principal, WRITE_SCOPE, ADMIN_SCOPE)
try:
recovery = begin_connector_read_snapshot(
session,
tenant_id=principal.tenant_id,
provider_id="connectors.feed_snapshot",
idempotency_key=idempotency_key,
source_revision=None,
cursor=None,
dry_run_evidence={
"performed": False,
"reason": "read-only acquisition into an immutable snapshot",
},
request_metadata={
"source_url_sha256": hashlib.sha256(
payload.url.encode("utf-8")
).hexdigest(),
"source_name": payload.source_name,
"max_entries": payload.max_entries,
},
resource_type="connector_tabular_source",
)
except ConnectorRecoveryError as exc:
raise _recovery_http_error(exc) from exc
if recovery.replayed:
try:
source = provider.get_source(
session,
principal,
source_ref=f"snapshot:{recovery.resource_id}",
)
except TabularSourceError as exc:
raise _http_error(exc) from exc
return _source_response(source)
try:
document = feed_transport.fetch(
payload.url,
max_entries=payload.max_entries,
)
source = provider.create_snapshot(
session,
principal,
snapshot=TabularSnapshotInput(
name=payload.name,
source_name=payload.source_name,
description=payload.description or document.description,
rows=feed_rows(document),
metadata={
"import_format": document.format,
"feed": {
"source_url": document.source_url,
"home_url": document.home_url,
"acquired_at": (
document.acquired_at.isoformat()
if document.acquired_at
else None
),
"fresh_until": (
document.fresh_until.isoformat()
if document.fresh_until
else None
),
"etag": document.etag,
"last_modified": document.last_modified,
"content_type": document.content_type,
"sha256": document.sha256,
},
},
),
source_id=recovery.resource_id,
)
except (FeedCapabilityError, TabularSourceError) as exc:
session.rollback()
recovery.fail_without_projection(
summary="The read-only feed import failed before a snapshot committed",
code=exc.__class__.__name__,
)
if isinstance(exc, FeedCapabilityError):
raise _feed_http_error(exc) from exc
raise _http_error(exc) from exc
audit_event(
session,
tenant_id=principal.tenant_id,
user_id=getattr(principal.user, "id", None),
api_key_id=principal.api_key_id,
action="connectors.feed_snapshot.created",
object_type="connector_tabular_source",
object_id=source.ref,
details={
"source_url": document.source_url,
"format": document.format,
"sha256": document.sha256,
"row_count": source.row_count,
},
)
try:
recovery.commit_success(
session,
evidence={
"verified": True,
"checks": {
"snapshot_ref": source.ref,
"snapshot_fingerprint": source.fingerprint,
"feed_sha256": document.sha256,
"row_count": source.row_count,
"provider_mutation": False,
},
},
)
except ConnectorRecoveryError as exc:
raise _recovery_http_error(exc) from exc
return _source_response(source)
@router.post("/feeds/render")
def api_render_feed(
payload: FeedRenderPayload,
principal: ApiPrincipal = Depends(get_api_principal),
) -> Response:
_require_any_scope(principal, READ_SCOPE, ADMIN_SCOPE)
try:
rendered = feed_transport.render(
FeedRenderRequest(
format=payload.format,
title=payload.title,
feed_url=payload.feed_url,
home_url=payload.home_url,
description=payload.description,
language=payload.language,
entries=tuple(
FeedEntry(**item.model_dump()) for item in payload.entries
),
allowed_visibilities=frozenset(payload.allowed_visibilities),
)
)
except FeedCapabilityError as exc:
raise _feed_http_error(exc) from exc
return Response(
content=rendered.body,
media_type=rendered.content_type,
headers={
"X-GovOPlaN-Feed-Included": str(rendered.included_entries),
"X-GovOPlaN-Feed-Excluded": str(rendered.excluded_entries),
},
)
@router.get("/tabular-sources", response_model=TabularSourceListResponse)
def api_list_tabular_sources(
query: str = Query(default="", max_length=200),
limit: int = Query(default=100, ge=1, le=100),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
) -> TabularSourceListResponse:
_require_any_scope(principal, READ_SCOPE, ADMIN_SCOPE)
sources = provider.list_sources(
session,
principal,
query=query,
limit=limit,
)
return TabularSourceListResponse(sources=[_source_response(source) for source in sources])
@router.post(
"/tabular-sources/snapshots",
response_model=TabularSourceResponse,
status_code=status.HTTP_201_CREATED,
)
def api_create_tabular_snapshot(
payload: SnapshotCreateRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
) -> TabularSourceResponse:
_require_any_scope(principal, WRITE_SCOPE, ADMIN_SCOPE)
try:
rows = (
tuple(payload.rows or ())
if payload.format == "json"
else parse_csv_snapshot(payload.csv_text or "", delimiter=payload.delimiter)
)
source = provider.create_snapshot(
session,
principal,
snapshot=TabularSnapshotInput(
name=payload.name,
source_name=payload.source_name,
description=payload.description,
rows=rows,
metadata={"import_format": payload.format},
),
)
except TabularSourceError as exc:
raise _http_error(exc) from exc
audit_event(
session,
tenant_id=principal.tenant_id,
user_id=getattr(principal.user, "id", None),
api_key_id=principal.api_key_id,
action="connectors.tabular_snapshot.created",
object_type="connector_tabular_source",
object_id=source.ref,
details={
"provider": source.provider,
"source_name": source.source_name,
"fingerprint": source.fingerprint,
"row_count": source.row_count,
},
)
session.commit()
return _source_response(source)
@router.get(
"/tabular-sources/{source_id}/preview",
response_model=TabularSourcePreviewResponse,
)
def api_preview_tabular_source(
source_id: str,
limit: int = Query(default=100, ge=1, le=500),
offset: int = Query(default=0, ge=0),
max_bytes: int = Query(default=1_000_000, ge=2, le=5_000_000),
timeout_ms: int = Query(default=2_000, ge=1, le=10_000),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
) -> TabularSourcePreviewResponse:
_require_any_scope(principal, READ_SCOPE, ADMIN_SCOPE)
try:
result = provider.read_source(
session,
principal,
request=TabularReadRequest(
source_ref=f"snapshot:{source_id}",
limit=limit,
offset=offset,
max_bytes=max_bytes,
timeout_ms=timeout_ms,
),
)
except TabularSourceError as exc:
raise _http_error(exc) from exc
return TabularSourcePreviewResponse(
source=_source_response(result.source),
rows=[dict(row) for row in result.rows],
total_rows=result.total_rows,
truncated=result.truncated,
returned_bytes=result.returned_bytes,
elapsed_ms=result.elapsed_ms,
effective_row_limit=result.effective_row_limit,
effective_byte_limit=result.effective_byte_limit,
effective_timeout_ms=result.effective_timeout_ms,
diagnostics=[
TabularPreviewDiagnosticResponse(
severity=item.severity,
code=item.code,
message=item.message,
details=dict(item.details),
)
for item in result.diagnostics
],
)
@router.delete(
"/tabular-sources/{source_id}",
response_model=TabularSourceDeleteResponse,
)
def api_delete_tabular_source(
source_id: str,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
) -> TabularSourceDeleteResponse:
_require_any_scope(principal, WRITE_SCOPE, ADMIN_SCOPE)
source_ref = f"snapshot:{source_id}"
try:
source = provider.delete_snapshot(
session,
principal,
source_ref=source_ref,
)
except TabularSourceError as exc:
raise _http_error(exc) from exc
audit_event(
session,
tenant_id=principal.tenant_id,
user_id=getattr(principal.user, "id", None),
api_key_id=principal.api_key_id,
action="connectors.tabular_snapshot.deleted",
object_type="connector_tabular_source",
object_id=source_ref,
details={"source_name": source.source_name, "fingerprint": source.fingerprint},
)
session.commit()
return TabularSourceDeleteResponse(deleted=True, source_ref=source_ref)
@router.get(
"/sanctions/sources",
response_model=SanctionsSourceListResponse,
)
def api_list_sanctions_sources(
principal: ApiPrincipal = Depends(get_api_principal),
) -> SanctionsSourceListResponse:
_require_any_scope(
principal,
SANCTIONS_READ_SCOPE,
SANCTIONS_REFRESH_SCOPE,
ADMIN_SCOPE,
)
return SanctionsSourceListResponse(
sources=[
SanctionsSourceResponse.model_validate(
source,
from_attributes=True,
)
for source in sanctions_provider.available_sources()
]
)
@router.post(
"/sanctions/sources/{provider_id}/refresh",
response_model=SanctionsRefreshResponse,
)
def api_refresh_sanctions_source(
provider_id: str,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
idempotency_key: Annotated[
str | None,
Header(alias="Idempotency-Key", max_length=500),
] = None,
) -> SanctionsRefreshResponse:
_require_any_scope(
principal,
SANCTIONS_REFRESH_SCOPE,
ADMIN_SCOPE,
)
try:
result = sanctions_provider.refresh_source(
session,
principal,
provider_id=provider_id,
idempotency_key=idempotency_key,
)
except ConnectorRecoveryError as exc:
raise _recovery_http_error(exc) from exc
except SanctionsSourceError as exc:
raise _sanctions_http_error(exc) from exc
audit_event(
session,
tenant_id=principal.tenant_id,
user_id=getattr(principal.user, "id", None),
api_key_id=principal.api_key_id,
action="connectors.sanctions_source.refreshed",
object_type="connector_sanctions_acquisition_run",
object_id=result.run_id,
details={
"provider_id": provider_id,
"status": result.status,
"snapshot_ref": (
result.snapshot.ref
if result.snapshot is not None
else None
),
},
)
session.commit()
return SanctionsRefreshResponse(
run_id=result.run_id,
provider_id=result.provider_id,
status=result.status,
snapshot=(
_sanctions_snapshot_response(result.snapshot)
if result.snapshot is not None
else None
),
error=result.error,
)
@router.get(
"/sanctions/snapshots",
response_model=SanctionsSnapshotListResponse,
)
def api_list_sanctions_snapshots(
limit: int = Query(default=100, ge=1, le=500),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
) -> SanctionsSnapshotListResponse:
_require_any_scope(
principal,
SANCTIONS_READ_SCOPE,
ADMIN_SCOPE,
)
try:
snapshots = sanctions_provider.list_snapshots(
session,
principal,
limit=limit,
)
except SanctionsSourceError as exc:
raise _sanctions_http_error(exc) from exc
return SanctionsSnapshotListResponse(
snapshots=[
_sanctions_snapshot_response(item)
for item in snapshots
]
)
@router.get(
"/sanctions/runs",
response_model=SanctionsAcquisitionRunListResponse,
)
def api_list_sanctions_runs(
limit: int = Query(default=100, ge=1, le=500),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(get_api_principal),
) -> SanctionsAcquisitionRunListResponse:
_require_any_scope(
principal,
SANCTIONS_READ_SCOPE,
ADMIN_SCOPE,
)
try:
runs = sanctions_provider.list_runs(
session,
principal,
limit=limit,
)
except SanctionsSourceError as exc:
raise _sanctions_http_error(exc) from exc
return SanctionsAcquisitionRunListResponse(
runs=[
SanctionsAcquisitionRunResponse.model_validate(
item,
from_attributes=True,
)
for item in runs
]
)
def _source_response(source: TabularSource) -> TabularSourceResponse:
return TabularSourceResponse(
ref=source.ref,
provider=source.provider,
source_name=source.source_name,
name=source.name,
description=source.description,
columns=[
TabularColumnResponse(
name=column.name,
data_type=column.data_type,
nullable=column.nullable,
)
for column in source.schema
],
schema_version=source.schema_version,
fingerprint=source.fingerprint,
row_count=source.row_count,
byte_count=source.byte_count,
updated_at=source.updated_at.isoformat() if source.updated_at else None,
capabilities=list(source.capabilities),
metadata=dict(source.metadata),
source_mode=source.source_mode,
pushdown=TabularPushdownResponse(
projections=source.pushdown.projections,
pagination=source.pushdown.pagination,
filters=list(source.pushdown.filters),
aggregations=list(source.pushdown.aggregations),
sorting=list(source.pushdown.sorting),
),
health=TabularHealthResponse(
status=source.health.status,
code=source.health.code,
summary=source.health.summary,
checked_at=(
source.health.checked_at.isoformat()
if source.health.checked_at
else None
),
details=dict(source.health.details),
),
)
def _sanctions_snapshot_response(
snapshot: SanctionsSnapshotReference,
) -> SanctionsSnapshotResponse:
return SanctionsSnapshotResponse.model_validate(
{
"ref": snapshot.ref,
"provider_id": snapshot.provider_id,
"publisher": snapshot.publisher,
"jurisdiction": snapshot.jurisdiction,
"list_type": snapshot.list_type,
"source_id": snapshot.source_id,
"source_version": snapshot.source_version,
"publication_at": snapshot.publication_at,
"effective_at": snapshot.effective_at,
"acquired_at": snapshot.acquired_at,
"content_type": snapshot.content_type,
"byte_count": snapshot.byte_count,
"sha256": snapshot.sha256,
"parser_version": snapshot.parser_version,
"raw_evidence_ref": snapshot.raw_evidence_ref,
"connector_run_id": snapshot.connector_run_id,
"signature_evidence": dict(
snapshot.signature_evidence
),
"licence_notes": snapshot.licence_notes,
"trust_notes": snapshot.trust_notes,
"transport_evidence": dict(
snapshot.transport_evidence
),
}
)
__all__ = ["router"]
File diff suppressed because it is too large Load Diff
+258
View File
@@ -0,0 +1,258 @@
from __future__ import annotations
from datetime import datetime
from typing import Any, Literal
from pydantic import BaseModel, Field, model_validator
class FeedAcquireRequest(BaseModel):
url: str = Field(min_length=1, max_length=2000)
max_entries: int = Field(default=2_000, ge=1, le=10_000)
class FeedImportRequest(FeedAcquireRequest):
name: str = Field(min_length=1, max_length=300)
source_name: str = Field(
min_length=1,
max_length=120,
pattern=r"^[A-Za-z_][A-Za-z0-9_]*$",
)
description: str | None = Field(default=None, max_length=4000)
class FeedEntryPayload(BaseModel):
id: str = Field(min_length=1, max_length=2000)
title: str = Field(min_length=1, max_length=1000)
url: str | None = Field(default=None, max_length=2000)
summary: str | None = None
content: str | None = None
author: str | None = Field(default=None, max_length=500)
published_at: datetime | None = None
updated_at: datetime | None = None
categories: list[str] = Field(default_factory=list, max_length=100)
enclosures: list[dict[str, Any]] = Field(default_factory=list, max_length=100)
visibility: Literal["public", "tenant", "private"] = "public"
metadata: dict[str, Any] = Field(default_factory=dict)
class FeedDocumentResponse(BaseModel):
format: Literal["rss", "atom"]
title: str
source_url: str
description: str | None = None
home_url: str | None = None
language: str | None = None
updated_at: datetime | None = None
acquired_at: datetime | None = None
fresh_until: datetime | None = None
etag: str | None = None
last_modified: str | None = None
content_type: str | None = None
sha256: str
entries: list[FeedEntryPayload]
metadata: dict[str, Any] = Field(default_factory=dict)
class FeedRenderPayload(BaseModel):
format: Literal["rss", "atom"]
title: str = Field(min_length=1, max_length=1000)
feed_url: str = Field(min_length=1, max_length=2000)
home_url: str = Field(min_length=1, max_length=2000)
description: str | None = None
language: str | None = Field(default=None, max_length=100)
entries: list[FeedEntryPayload] = Field(default_factory=list, max_length=10_000)
allowed_visibilities: list[Literal["public", "tenant", "private"]] = Field(
default_factory=lambda: ["public"],
max_length=3,
)
class SnapshotCreateRequest(BaseModel):
name: str = Field(min_length=1, max_length=300)
source_name: str = Field(
min_length=1,
max_length=120,
pattern=r"^[A-Za-z_][A-Za-z0-9_]*$",
)
description: str | None = Field(default=None, max_length=4000)
format: Literal["json", "csv"] = "json"
rows: list[dict[str, Any]] | None = Field(default=None, max_length=10_000)
csv_text: str | None = Field(default=None, max_length=5_000_000)
delimiter: Literal[",", ";", "\t", "|"] = ","
@model_validator(mode="after")
def validate_payload(self) -> "SnapshotCreateRequest":
if self.format == "json" and self.rows is None:
raise ValueError("JSON snapshots require rows.")
if self.format == "json" and self.csv_text is not None:
raise ValueError("JSON snapshots cannot include CSV text.")
if self.format == "csv" and not self.csv_text:
raise ValueError("CSV snapshots require CSV text.")
if self.format == "csv" and self.rows is not None:
raise ValueError("CSV snapshots cannot include JSON rows.")
return self
class TabularColumnResponse(BaseModel):
name: str
data_type: str
nullable: bool
class TabularPushdownResponse(BaseModel):
projections: bool
pagination: bool
filters: list[str]
aggregations: list[str]
sorting: list[str]
class TabularHealthResponse(BaseModel):
status: Literal["healthy", "warning", "error", "unknown"]
code: str
summary: str
checked_at: str | None
details: dict[str, Any]
class TabularPreviewDiagnosticResponse(BaseModel):
severity: Literal["info", "warning", "error"]
code: str
message: str
details: dict[str, Any]
class TabularSourceResponse(BaseModel):
ref: str
provider: str
source_name: str
name: str
description: str | None
columns: list[TabularColumnResponse]
schema_version: str
fingerprint: str
row_count: int | None
byte_count: int | None
updated_at: str | None
capabilities: list[str]
metadata: dict[str, Any]
source_mode: Literal["live", "cached", "file_backed", "static"]
pushdown: TabularPushdownResponse
health: TabularHealthResponse
class TabularSourceListResponse(BaseModel):
sources: list[TabularSourceResponse]
class TabularSourcePreviewResponse(BaseModel):
source: TabularSourceResponse
rows: list[dict[str, Any]]
total_rows: int
truncated: bool
returned_bytes: int
elapsed_ms: int
effective_row_limit: int
effective_byte_limit: int
effective_timeout_ms: int
diagnostics: list[TabularPreviewDiagnosticResponse]
class TabularSourceDeleteResponse(BaseModel):
deleted: bool
source_ref: str
class SanctionsSourceResponse(BaseModel):
provider_id: str
publisher: str
jurisdiction: str
list_type: str
source_id: str
source_url: str | None
parser_version: str
licence_notes: str
trust_notes: str
class SanctionsSourceListResponse(BaseModel):
sources: list[SanctionsSourceResponse]
class SanctionsSnapshotResponse(BaseModel):
ref: str
provider_id: str
publisher: str
jurisdiction: str
list_type: str
source_id: str
source_version: str
publication_at: datetime | None
effective_at: datetime | None
acquired_at: datetime
content_type: str
byte_count: int
sha256: str
parser_version: str
raw_evidence_ref: str
connector_run_id: str
signature_evidence: dict[str, Any]
licence_notes: str | None
trust_notes: str | None
transport_evidence: dict[str, Any]
class SanctionsSnapshotListResponse(BaseModel):
snapshots: list[SanctionsSnapshotResponse]
class SanctionsAcquisitionRunResponse(BaseModel):
id: str
provider_id: str
source_id: str
status: str
attempt_count: int
request_evidence: dict[str, Any]
response_evidence: dict[str, Any]
started_at: datetime
finished_at: datetime | None
snapshot_id: str | None
error: str | None
class SanctionsAcquisitionRunListResponse(BaseModel):
runs: list[SanctionsAcquisitionRunResponse]
class SanctionsRefreshResponse(BaseModel):
run_id: str
provider_id: str
status: str
snapshot: SanctionsSnapshotResponse | None
error: str | None
__all__ = [
"FeedAcquireRequest",
"FeedDocumentResponse",
"FeedEntryPayload",
"FeedImportRequest",
"FeedRenderPayload",
"SnapshotCreateRequest",
"SanctionsAcquisitionRunListResponse",
"SanctionsAcquisitionRunResponse",
"SanctionsRefreshResponse",
"SanctionsSnapshotListResponse",
"SanctionsSnapshotResponse",
"SanctionsSourceListResponse",
"SanctionsSourceResponse",
"TabularColumnResponse",
"TabularHealthResponse",
"TabularPreviewDiagnosticResponse",
"TabularPushdownResponse",
"TabularSourceDeleteResponse",
"TabularSourceListResponse",
"TabularSourcePreviewResponse",
"TabularSourceResponse",
]
@@ -0,0 +1,498 @@
from __future__ import annotations
import hashlib
import json
import time
from collections.abc import Mapping, Sequence
from datetime import datetime
from decimal import Decimal
from typing import Any, Callable
from sqlalchemy import or_, select
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal, has_scope
from govoplan_core.core.tabular_sources import (
TabularColumn,
TabularPreviewDiagnostic,
TabularPushdown,
TabularReadRequest,
TabularReadResult,
TabularSnapshotInput,
TabularSource,
TabularSourceAccessError,
TabularSourceNotFoundError,
TabularSourceHealth,
TabularSourceUnavailableError,
TabularSourceValidationError,
parse_tabular_csv,
)
from govoplan_core.db.base import utcnow
from govoplan_connectors.backend.db.models import ConnectorTabularSource
READ_SCOPE = "connectors:source:read"
WRITE_SCOPE = "connectors:source:write"
ADMIN_SCOPE = "connectors:source:admin"
MAX_SNAPSHOT_ROWS = 10_000
MAX_SNAPSHOT_BYTES = 5_000_000
MAX_READ_ROWS = 500
MAX_READ_BYTES = 1_000_000
MAX_READ_TIMEOUT_MS = 2_000
class SqlTabularSourceProvider:
def __init__(self, *, clock: Callable[[], float] = time.monotonic) -> None:
self._clock = clock
def list_sources(
self,
session: object,
principal: object,
*,
query: str = "",
limit: int = 100,
) -> Sequence[TabularSource]:
db, api_principal = _context(session, principal, READ_SCOPE)
normalized_query = str(query or "").strip()
statement = select(ConnectorTabularSource).where(
ConnectorTabularSource.tenant_id == api_principal.tenant_id,
ConnectorTabularSource.deleted_at.is_(None),
ConnectorTabularSource.status == "active",
)
if normalized_query:
pattern = f"%{_escape_like(normalized_query)}%"
statement = statement.where(
or_(
ConnectorTabularSource.name.ilike(pattern, escape="\\"),
ConnectorTabularSource.source_name.ilike(pattern, escape="\\"),
)
)
statement = statement.order_by(
ConnectorTabularSource.updated_at.desc(),
ConnectorTabularSource.name,
).limit(max(1, min(int(limit), 100)))
return tuple(_source_dto(item) for item in db.scalars(statement))
def get_source(
self,
session: object,
principal: object,
*,
source_ref: str,
) -> TabularSource | None:
db, api_principal = _context(session, principal, READ_SCOPE)
item = _source_record(
db,
tenant_id=api_principal.tenant_id,
source_ref=source_ref,
)
return _source_dto(item) if item is not None else None
def read_source(
self,
session: object,
principal: object,
*,
request: TabularReadRequest,
) -> TabularReadResult:
db, api_principal = _context(session, principal, READ_SCOPE)
item = _source_record(
db,
tenant_id=api_principal.tenant_id,
source_ref=request.source_ref,
)
if item is None:
raise TabularSourceNotFoundError("Tabular source not found.")
if request.expected_fingerprint and request.expected_fingerprint != item.fingerprint:
raise TabularSourceValidationError(
"The source fingerprint changed; refresh the source node before running it."
)
started = self._clock()
limit = max(1, min(int(request.limit), MAX_READ_ROWS))
byte_limit = max(2, min(int(request.max_bytes), MAX_READ_BYTES))
timeout_ms = max(1, min(int(request.timeout_ms), MAX_READ_TIMEOUT_MS))
offset = max(0, int(request.offset))
diagnostics: list[TabularPreviewDiagnostic] = []
if limit != request.limit:
diagnostics.append(
_preview_diagnostic(
"preview.row_limit_tightened",
"The provider tightened the requested row limit.",
requested=request.limit,
effective=limit,
)
)
if byte_limit != request.max_bytes:
diagnostics.append(
_preview_diagnostic(
"preview.byte_limit_tightened",
"The provider tightened the requested byte limit.",
requested=request.max_bytes,
effective=byte_limit,
)
)
if timeout_ms != request.timeout_ms:
diagnostics.append(
_preview_diagnostic(
"preview.timeout_tightened",
"The provider tightened the requested time limit.",
requested=request.timeout_ms,
effective=timeout_ms,
)
)
selected_columns = tuple(dict.fromkeys(request.columns))
known_columns = {column["name"] for column in item.schema_}
unknown_columns = [column for column in selected_columns if column not in known_columns]
if unknown_columns:
raise TabularSourceValidationError(
f"Unknown source columns: {', '.join(unknown_columns)}"
)
rows: list[dict[str, object]] = []
returned_bytes = 2
stopped_for = ""
for row in item.rows[offset:]:
if len(rows) >= limit:
stopped_for = "rows"
break
elapsed_ms = int(max(0.0, self._clock() - started) * 1_000)
if elapsed_ms >= timeout_ms:
if not rows:
raise TabularSourceUnavailableError(
"Tabular source preview exceeded its time budget."
)
stopped_for = "time"
break
selected = {
key: value
for key, value in row.items()
if not selected_columns or key in selected_columns
}
row_bytes = len(
json.dumps(
selected,
sort_keys=True,
separators=(",", ":"),
default=str,
).encode("utf-8")
)
additional_bytes = row_bytes + (1 if rows else 0)
if returned_bytes + additional_bytes > byte_limit:
if not rows:
raise TabularSourceValidationError(
"A single source row exceeds the preview byte limit."
)
stopped_for = "bytes"
break
rows.append(selected)
returned_bytes += additional_bytes
elapsed_ms = int(max(0.0, self._clock() - started) * 1_000)
if stopped_for:
labels = {
"rows": ("preview.row_limit_reached", "row"),
"bytes": ("preview.byte_limit_reached", "byte"),
"time": ("preview.timeout_reached", "time"),
}
code, label = labels[stopped_for]
diagnostics.append(
TabularPreviewDiagnostic(
severity="warning",
code=code,
message=f"The preview stopped at its effective {label} limit.",
)
)
return TabularReadResult(
source=_source_dto(item),
rows=tuple(rows),
total_rows=item.row_count,
truncated=offset + len(rows) < item.row_count,
returned_bytes=returned_bytes,
elapsed_ms=elapsed_ms,
effective_row_limit=limit,
effective_byte_limit=byte_limit,
effective_timeout_ms=timeout_ms,
diagnostics=tuple(diagnostics),
)
def create_snapshot(
self,
session: object,
principal: object,
*,
snapshot: TabularSnapshotInput,
source_id: str | None = None,
) -> TabularSource:
db, api_principal = _context(session, principal, WRITE_SCOPE)
name = snapshot.name.strip()
source_name = snapshot.source_name.strip()
if not name:
raise TabularSourceValidationError("Snapshot name is required.")
if not source_name:
raise TabularSourceValidationError("Snapshot source name is required.")
if len(snapshot.rows) > MAX_SNAPSHOT_ROWS:
raise TabularSourceValidationError(
f"Snapshots are limited to {MAX_SNAPSHOT_ROWS:,} rows."
)
rows = [_json_row(row) for row in snapshot.rows]
encoded = json.dumps(rows, sort_keys=True, separators=(",", ":"), default=str).encode("utf-8")
if len(encoded) > MAX_SNAPSHOT_BYTES:
raise TabularSourceValidationError(
f"Snapshots are limited to {MAX_SNAPSHOT_BYTES // 1_000_000} MB."
)
existing = db.scalar(
select(ConnectorTabularSource.id).where(
ConnectorTabularSource.tenant_id == api_principal.tenant_id,
ConnectorTabularSource.source_name == source_name,
)
)
if existing is not None:
raise TabularSourceValidationError(
f"A tabular source named {source_name!r} already exists."
)
schema = infer_schema(rows)
fingerprint = snapshot_fingerprint(rows, schema)
actor_id = _actor_id(api_principal)
item = ConnectorTabularSource(
tenant_id=api_principal.tenant_id,
provider="snapshot",
source_name=source_name,
name=name,
description=_clean_optional(snapshot.description),
status="active",
schema_version=1,
schema_=[_column_payload(column) for column in schema],
rows=rows,
fingerprint=fingerprint,
row_count=len(rows),
byte_count=len(encoded),
metadata_=dict(snapshot.metadata),
created_by=actor_id,
updated_by=actor_id,
)
if source_id:
item.id = source_id
db.add(item)
db.flush()
return _source_dto(item)
def delete_snapshot(
self,
session: object,
principal: object,
*,
source_ref: str,
) -> TabularSource:
db, api_principal = _context(session, principal, WRITE_SCOPE)
item = _source_record(
db,
tenant_id=api_principal.tenant_id,
source_ref=source_ref,
)
if item is None:
raise TabularSourceNotFoundError("Tabular source not found.")
item.deleted_at = utcnow()
item.status = "retired"
item.updated_by = _actor_id(api_principal)
db.flush()
return _source_dto(item)
def parse_csv_snapshot(csv_text: str, *, delimiter: str) -> tuple[Mapping[str, object], ...]:
return parse_tabular_csv(
csv_text,
delimiter=delimiter,
max_rows=MAX_SNAPSHOT_ROWS,
)
def infer_schema(rows: Sequence[Mapping[str, object]]) -> tuple[TabularColumn, ...]:
names: list[str] = []
for row in rows:
for name in row:
if name not in names:
names.append(name)
result: list[TabularColumn] = []
for name in names:
values = [row.get(name) for row in rows]
concrete = [value for value in values if value is not None]
data_type = _type_name(concrete[0]) if concrete else "unknown"
if any(_type_name(value) != data_type for value in concrete[1:]):
data_type = "mixed"
result.append(
TabularColumn(
name=name,
data_type=data_type,
nullable=len(concrete) != len(values),
)
)
return tuple(result)
def snapshot_fingerprint(
rows: Sequence[Mapping[str, object]],
schema: Sequence[TabularColumn],
) -> str:
payload = {
"schema": [_column_payload(column) for column in schema],
"rows": [dict(row) for row in rows],
}
encoded = json.dumps(payload, sort_keys=True, separators=(",", ":"), default=str)
return hashlib.sha256(encoded.encode("utf-8")).hexdigest()
def _source_record(
session: Session,
*,
tenant_id: str,
source_ref: str,
) -> ConnectorTabularSource | None:
source_id = source_ref.removeprefix("snapshot:")
if not source_id or source_id == source_ref:
return None
return session.scalar(
select(ConnectorTabularSource).where(
ConnectorTabularSource.id == source_id,
ConnectorTabularSource.tenant_id == tenant_id,
ConnectorTabularSource.deleted_at.is_(None),
)
)
def _source_dto(item: ConnectorTabularSource) -> TabularSource:
return TabularSource(
ref=f"snapshot:{item.id}",
provider=item.provider,
source_name=item.source_name,
name=item.name,
description=item.description,
schema=tuple(TabularColumn(**column) for column in item.schema_),
schema_version=str(item.schema_version),
fingerprint=item.fingerprint,
row_count=item.row_count,
byte_count=item.byte_count,
updated_at=item.updated_at,
capabilities=("read", "preview"),
metadata=dict(item.metadata_),
source_mode="cached",
pushdown=TabularPushdown(
projections=True,
pagination=True,
),
health=TabularSourceHealth(
status="healthy",
code="snapshot.ready",
summary="The immutable connector snapshot is ready.",
checked_at=item.updated_at,
details={"immutable": True},
),
)
def _preview_diagnostic(
code: str,
message: str,
*,
requested: int,
effective: int,
) -> TabularPreviewDiagnostic:
return TabularPreviewDiagnostic(
severity="info",
code=code,
message=message,
details={"requested": requested, "effective": effective},
)
def _context(
session: object,
principal: object,
required_scope: str,
) -> tuple[Session, ApiPrincipal]:
if not isinstance(session, Session):
raise TypeError("Tabular source providers require a SQLAlchemy session.")
if not isinstance(principal, ApiPrincipal):
raise TabularSourceAccessError("A tenant API principal is required.")
accepted_scopes = {required_scope, ADMIN_SCOPE}
if required_scope == READ_SCOPE:
accepted_scopes.update(
{
WRITE_SCOPE,
"datasources:catalogue:read",
"datasources:source:write",
"datasources:source:admin",
}
)
if not any(has_scope(principal, scope) for scope in accepted_scopes):
raise TabularSourceAccessError(f"Missing scope: {required_scope}")
return session, principal
def _json_row(row: Mapping[str, object]) -> dict[str, Any]:
normalized = {str(key).strip(): value for key, value in row.items()}
if not normalized or any(not key for key in normalized):
raise TabularSourceValidationError("Every snapshot row needs named columns.")
try:
json.dumps(normalized, default=_unsupported_json)
except (TypeError, ValueError) as exc:
raise TabularSourceValidationError(f"Snapshot values must be JSON compatible: {exc}") from exc
return json.loads(json.dumps(normalized, default=_unsupported_json))
def _unsupported_json(value: object) -> object:
if isinstance(value, (datetime, Decimal)):
return str(value)
raise TypeError(f"{type(value).__name__} is not JSON serializable")
def _type_name(value: object) -> str:
if isinstance(value, bool):
return "boolean"
if isinstance(value, int):
return "integer"
if isinstance(value, (float, Decimal)):
return "number"
if isinstance(value, str):
return "string"
if isinstance(value, list):
return "array"
if isinstance(value, dict):
return "object"
return type(value).__name__.lower()
def _column_payload(column: TabularColumn) -> dict[str, object]:
return {
"name": column.name,
"data_type": column.data_type,
"nullable": column.nullable,
}
def _actor_id(principal: ApiPrincipal) -> str | None:
return principal.account_id or principal.membership_id or principal.identity_id
def _clean_optional(value: str | None) -> str | None:
cleaned = str(value or "").strip()
return cleaned or None
def _escape_like(value: str) -> str:
return value.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_")
__all__ = [
"ADMIN_SCOPE",
"MAX_READ_ROWS",
"MAX_READ_BYTES",
"MAX_READ_TIMEOUT_MS",
"MAX_SNAPSHOT_BYTES",
"MAX_SNAPSHOT_ROWS",
"READ_SCOPE",
"SqlTabularSourceProvider",
"WRITE_SCOPE",
"infer_schema",
"parse_csv_snapshot",
"snapshot_fingerprint",
]
+1
View File
@@ -0,0 +1 @@
+92
View File
@@ -0,0 +1,92 @@
from __future__ import annotations
import unittest
from sqlalchemy import create_engine
from sqlalchemy.orm import sessionmaker
from govoplan_core.auth import ApiPrincipal
from govoplan_core.core.access import PrincipalRef
from govoplan_core.core.datasources import DatasourceOriginReadRequest
from govoplan_core.core.tabular_sources import TabularSnapshotInput
from govoplan_core.db.base import Base
from govoplan_connectors.backend.datasource_origins import (
ConnectorDatasourceOriginProvider,
)
from govoplan_connectors.backend.db.models import ConnectorTabularSource
from govoplan_connectors.backend.tabular_sources import (
WRITE_SCOPE,
SqlTabularSourceProvider,
)
def principal(*, scopes: tuple[str, ...]) -> ApiPrincipal:
return ApiPrincipal(
principal=PrincipalRef(
account_id="account-1",
membership_id="membership-1",
tenant_id="tenant-1",
scopes=frozenset(scopes),
),
account=object(),
user=object(),
)
class ConnectorDatasourceOriginTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite:///:memory:")
Base.metadata.create_all(
self.engine,
tables=[ConnectorTabularSource.__table__],
)
self.Session = sessionmaker(bind=self.engine)
self.session = self.Session()
provider = SqlTabularSourceProvider()
self.source = provider.create_snapshot(
self.session,
principal(scopes=(WRITE_SCOPE,)),
snapshot=TabularSnapshotInput(
name="Imported cases",
source_name="imported_cases",
rows=({"id": 1, "name": "Ada"},),
),
)
self.session.commit()
self.origins = ConnectorDatasourceOriginProvider(provider)
def tearDown(self) -> None:
self.session.close()
Base.metadata.drop_all(
self.engine,
tables=[ConnectorTabularSource.__table__],
)
self.engine.dispose()
def test_datasource_reader_can_discover_and_read_connector_origin(self) -> None:
datasource_principal = principal(
scopes=("datasources:catalogue:read",),
)
origins = self.origins.list_origins(
self.session,
datasource_principal,
)
result = self.origins.read_origin(
self.session,
datasource_principal,
request=DatasourceOriginReadRequest(origin_ref=self.source.ref),
)
self.assertEqual((self.source.ref,), tuple(item.ref for item in origins))
self.assertEqual(("live", "cached"), origins[0].supported_modes)
self.assertEqual(({"id": 1, "name": "Ada"},), result.rows)
self.assertEqual("cached", origins[0].source_mode)
self.assertTrue(origins[0].pushdown.projections)
self.assertEqual("healthy", origins[0].health.status)
self.assertGreater(result.returned_bytes, 2)
self.assertEqual(1_000_000, result.effective_byte_limit)
if __name__ == "__main__":
unittest.main()
+103
View File
@@ -0,0 +1,103 @@
from __future__ import annotations
import unittest
from unittest.mock import patch
from defusedxml import ElementTree as SafeET
from govoplan_connectors.backend.feeds import ConnectorFeedProvider, feed_rows
from govoplan_core.core.feeds import (
FeedCapabilityError,
FeedEntry,
FeedRenderRequest,
)
from govoplan_core.security.http_fetch import HttpFetchResponse
RSS = b"""<?xml version="1.0"?>
<rss version="2.0"><channel>
<title>Decisions</title><link>https://example.test/</link>
<description>Published decisions</description>
<item><guid>decision-1</guid><title>Decision one</title>
<link>https://example.test/1</link>
<pubDate>Fri, 31 Jul 2026 10:00:00 GMT</pubDate>
<category>planning</category>
</item>
</channel></rss>"""
ATOM = b"""<?xml version="1.0"?>
<feed xmlns="http://www.w3.org/2005/Atom">
<id>https://example.test/feed</id><title>Updates</title>
<updated>2026-07-31T10:00:00Z</updated>
<link href="https://example.test/" />
<entry><id>update-1</id><title>Update one</title>
<updated>2026-07-31T10:00:00Z</updated>
<link href="https://example.test/update-1" />
</entry>
</feed>"""
class ConnectorFeedProviderTests(unittest.TestCase):
def setUp(self) -> None:
self.provider = ConnectorFeedProvider()
def test_rss_and_atom_are_normalized_to_tabular_entries(self) -> None:
rss = self.provider.parse(RSS, source_url="https://example.test/rss")
atom = self.provider.parse(ATOM, source_url="https://example.test/atom")
self.assertEqual("rss", rss.format)
self.assertEqual("decision-1", rss.entries[0].id)
self.assertEqual("atom", atom.format)
self.assertEqual("https://example.test/update-1", atom.entries[0].url)
self.assertEqual("decision-1", feed_rows(rss)[0]["id"])
self.assertEqual(64, len(rss.sha256))
def test_fetch_records_transport_freshness_and_provenance(self) -> None:
with patch(
"govoplan_connectors.backend.feeds.fetch_http",
return_value=HttpFetchResponse(
status=200,
headers={
"Content-Type": "application/rss+xml",
"Cache-Control": "public, max-age=600",
"ETag": '"feed-1"',
},
body=RSS,
),
):
document = self.provider.fetch("https://example.test/rss")
self.assertEqual('"feed-1"', document.etag)
self.assertIsNotNone(document.acquired_at)
self.assertEqual(600, int((document.fresh_until - document.acquired_at).total_seconds()))
self.assertEqual(len(RSS), document.metadata["byte_count"])
def test_render_filters_entries_by_explicit_visibility(self) -> None:
request = FeedRenderRequest(
format="atom",
title="Public updates",
feed_url="https://example.test/feed.atom",
home_url="https://example.test/",
entries=(
FeedEntry(id="public-1", title="Public", visibility="public"),
FeedEntry(id="tenant-1", title="Tenant", visibility="tenant"),
),
allowed_visibilities=frozenset({"public"}),
)
result = self.provider.render(request)
root = SafeET.fromstring(result.body)
self.assertEqual(1, result.included_entries)
self.assertEqual(1, result.excluded_entries)
self.assertIn(b"Public", result.body)
self.assertNotIn(b"Tenant", result.body)
self.assertTrue(root.tag.endswith("feed"))
def test_unsafe_xml_is_rejected(self) -> None:
payload = b'<!DOCTYPE x [<!ENTITY y SYSTEM "file:///etc/passwd">]><rss>&y;</rss>'
with self.assertRaisesRegex(FeedCapabilityError, "not safe or valid"):
self.provider.parse(payload, source_url="https://example.test/rss")
if __name__ == "__main__":
unittest.main()
+39
View File
@@ -0,0 +1,39 @@
from __future__ import annotations
import unittest
from govoplan_core.core.tabular_sources import (
CAPABILITY_CONNECTORS_TABULAR_SNAPSHOT_WRITER,
CAPABILITY_CONNECTORS_TABULAR_SOURCES,
)
from govoplan_core.core.sanctions import (
CAPABILITY_CONNECTORS_SANCTIONS_SNAPSHOTS,
)
from govoplan_connectors.backend.manifest import manifest
class ConnectorsManifestTests(unittest.TestCase):
def test_manifest_exposes_versioned_tabular_capabilities(self) -> None:
self.assertEqual("connectors", manifest.id)
self.assertIn("access", manifest.optional_dependencies)
self.assertIn(
"connectors.tabular_sources",
{interface.name for interface in manifest.provides_interfaces},
)
self.assertIn(
CAPABILITY_CONNECTORS_TABULAR_SOURCES,
manifest.capability_factories,
)
self.assertIn(
CAPABILITY_CONNECTORS_TABULAR_SNAPSHOT_WRITER,
manifest.capability_factories,
)
self.assertIn(
CAPABILITY_CONNECTORS_SANCTIONS_SNAPSHOTS,
manifest.capability_factories,
)
self.assertIsNotNone(manifest.migration_spec)
if __name__ == "__main__":
unittest.main()
+47
View File
@@ -0,0 +1,47 @@
from __future__ import annotations
import tempfile
import unittest
from pathlib import Path
from alembic.runtime.migration import MigrationContext
from sqlalchemy import create_engine, inspect
from govoplan_connectors.backend.manifest import get_manifest
from govoplan_core.db.migrations import migrate_database
class ConnectorsMigrationTests(unittest.TestCase):
def test_baseline_creates_connector_tables_and_head(self) -> None:
with tempfile.TemporaryDirectory(prefix="govoplan-connectors-migration-") as directory:
url = f"sqlite:///{Path(directory) / 'connectors.db'}"
migrate_database(
database_url=url,
enabled_modules=("connectors",),
manifest_factories=(get_manifest,),
)
engine = create_engine(url)
try:
with engine.connect() as connection:
self.assertIn(
"f7c8d9e0a1b2",
set(MigrationContext.configure(connection).get_current_heads()),
)
self.assertIn(
"connector_tabular_sources",
inspect(connection).get_table_names(),
)
self.assertIn(
"connector_sanctions_snapshots",
inspect(connection).get_table_names(),
)
self.assertIn(
"connector_sanctions_acquisition_runs",
inspect(connection).get_table_names(),
)
finally:
engine.dispose()
if __name__ == "__main__":
unittest.main()
+118
View File
@@ -0,0 +1,118 @@
from __future__ import annotations
from datetime import UTC, datetime
import unittest
from sqlalchemy import create_engine
from sqlalchemy.orm import sessionmaker
from govoplan_connectors.backend.db.models import (
ConnectorSanctionsAcquisitionRun,
ConnectorSanctionsSnapshot,
ConnectorTabularSource,
)
from govoplan_connectors.backend.manifest import manifest
from govoplan_connectors.backend.provider_state import (
SANCTIONS_PROVIDER_ID,
TABULAR_PROVIDER_ID,
sanctions_provider_states,
tabular_provider_states,
)
from govoplan_core.core.provider_governance import ExternalProviderStateContext
from govoplan_core.db.base import Base
class ConnectorsProviderStateTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite+pysqlite:///:memory:", future=True)
Base.metadata.create_all(
self.engine,
tables=(
ConnectorTabularSource.__table__,
ConnectorSanctionsAcquisitionRun.__table__,
ConnectorSanctionsSnapshot.__table__,
),
)
self.session = sessionmaker(bind=self.engine, expire_on_commit=False)()
def tearDown(self) -> None:
self.session.close()
self.engine.dispose()
def test_immutable_tabular_snapshot_reports_ready_state(self) -> None:
self.session.add(
ConnectorTabularSource(
id="tabular-1",
tenant_id="tenant-1",
source_name="monthly",
name="Monthly input",
status="active",
schema_version=1,
schema_=[{"name": "id", "type": "string"}],
rows=[{"id": "1"}],
fingerprint="a" * 64,
row_count=1,
byte_count=10,
)
)
self.session.commit()
state = tabular_provider_states(
ExternalProviderStateContext(session=self.session, tenant_id="tenant-1")
)[0]
self.assertEqual("healthy", state.health)
self.assertEqual("ready", state.recovery)
self.assertEqual("not_applicable", state.freshness)
def test_sanctions_state_hashes_binding_and_manifest_registers_state(self) -> None:
now = datetime.now(UTC)
run = ConnectorSanctionsAcquisitionRun(
id="run-1",
tenant_id="tenant-1",
provider_id="eu",
source_id="secret-source-name",
status="succeeded",
attempt_count=1,
started_at=now,
finished_at=now,
)
snapshot = ConnectorSanctionsSnapshot(
id="snapshot-1",
tenant_id="tenant-1",
provider_id="eu",
publisher="European Union",
jurisdiction="EU",
list_type="sanctions",
source_id="secret-source-name",
source_version="2026-08-01",
acquired_at=now,
source_url="https://source.example.test/list.xml",
content_type="application/xml",
byte_count=8,
sha256="b" * 64,
parser_version="1",
connector_run_id=run.id,
raw_content=b"<list/>",
)
run.snapshot_id = snapshot.id
self.session.add_all((run, snapshot))
self.session.commit()
state = sanctions_provider_states(
ExternalProviderStateContext(session=self.session, tenant_id="tenant-1")
)[0]
self.assertEqual("healthy", state.health)
self.assertEqual("ready", state.recovery)
rendered = str(state.to_dict())
self.assertNotIn("secret-source-name", rendered)
self.assertNotIn("source.example.test", rendered)
self.assertEqual(
{TABULAR_PROVIDER_ID, SANCTIONS_PROVIDER_ID},
{item.provider_id for item in manifest.external_provider_state_providers},
)
if __name__ == "__main__":
unittest.main()
+263
View File
@@ -0,0 +1,263 @@
from __future__ import annotations
from datetime import datetime, timedelta, timezone
import unittest
from unittest.mock import patch
from sqlalchemy import create_engine, select
from sqlalchemy.orm import Session
from govoplan_connectors.backend.db.models import ConnectorTabularSource
from govoplan_connectors.backend.feeds import ConnectorFeedProvider
from govoplan_connectors.backend.recovery import (
CONNECTOR_RECOVERY_OPERATIONS,
ConnectorRecoveryError,
begin_connector_external_mutation,
begin_connector_read_snapshot,
)
from govoplan_connectors.backend.router import api_import_feed_snapshot
from govoplan_connectors.backend.schemas import FeedImportRequest
from govoplan_connectors.backend.tabular_sources import WRITE_SCOPE
from govoplan_core.auth import ApiPrincipal
from govoplan_core.core.access import PrincipalRef
from govoplan_core.core.recovery import (
RecoveryCheckpoint,
RecoveryOperation,
RecoveryStatus,
)
from govoplan_core.core.recovery_runtime import (
RecoveryOperationStateConflict,
claim_durable_recovery_operation,
)
from govoplan_core.core.tabular_sources import TabularSnapshotInput
from govoplan_core.core.runtime_coordination import (
DistributedLease,
RuntimeIdentity,
bind_process_runtime_identity,
)
from govoplan_core.db.base import Base
from govoplan_connectors.backend.tabular_sources import SqlTabularSourceProvider
RSS = b"""<?xml version="1.0"?>
<rss version="2.0"><channel><title>Updates</title>
<link>https://example.test/</link><description>Updates</description>
<item><guid>1</guid><title>One</title></item></channel></rss>"""
def _identity(node: str, incarnation: str) -> RuntimeIdentity:
return RuntimeIdentity(
installation_id="connector-recovery-tests",
node_id=node,
incarnation=incarnation,
role="worker",
software_version="test",
composition_hash="a" * 64,
)
def _principal() -> ApiPrincipal:
return ApiPrincipal(
principal=PrincipalRef(
account_id="account-1",
membership_id="membership-1",
tenant_id="tenant-1",
scopes=frozenset({WRITE_SCOPE}),
),
account=object(),
user=object(),
)
class ConnectorRecoveryTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite+pysqlite:///:memory:")
Base.metadata.create_all(
self.engine,
tables=(
DistributedLease.__table__,
RecoveryOperation.__table__,
RecoveryCheckpoint.__table__,
ConnectorTabularSource.__table__,
),
)
self.session = Session(self.engine, expire_on_commit=False)
bind_process_runtime_identity(_identity("node-1", "incarnation-1"))
def tearDown(self) -> None:
bind_process_runtime_identity(None)
self.session.close()
self.engine.dispose()
def test_feed_snapshot_and_recovery_checkpoint_commit_atomically_and_replay(self) -> None:
document = ConnectorFeedProvider().parse(
RSS,
source_url="https://example.test/feed.xml",
)
payload = FeedImportRequest(
url="https://example.test/feed.xml",
name="Updates",
source_name="updates",
)
with (
patch(
"govoplan_connectors.backend.router.feed_transport.fetch",
return_value=document,
) as fetch,
patch("govoplan_connectors.backend.router.audit_event"),
):
first = api_import_feed_snapshot(
payload,
session=self.session,
principal=_principal(),
idempotency_key="feed-import-1",
)
replay = api_import_feed_snapshot(
payload,
session=self.session,
principal=_principal(),
idempotency_key="feed-import-1",
)
self.assertEqual(first.ref, replay.ref)
fetch.assert_called_once()
operation = self.session.scalar(select(RecoveryOperation))
assert operation is not None
self.assertEqual(RecoveryStatus.SUCCEEDED.value, operation.status)
self.assertEqual(
first.ref.removeprefix("snapshot:"),
operation.resource_id,
)
def test_recovery_metadata_distinguishes_reads_from_external_mutations(self) -> None:
declarations = {
item.operation_type: item for item in CONNECTOR_RECOVERY_OPERATIONS
}
self.assertFalse(declarations["read-snapshot"].provider_mutation)
self.assertTrue(declarations["read-snapshot"].implemented)
self.assertTrue(declarations["external-mutation"].provider_mutation)
self.assertFalse(declarations["external-mutation"].implemented)
def test_stale_atomic_connector_fence_fails_without_claiming_an_effect(self) -> None:
recovery = begin_connector_read_snapshot(
self.session,
tenant_id="tenant-1",
provider_id="provider-1",
idempotency_key="read-1",
source_revision="revision-1",
cursor="cursor-1",
dry_run_evidence={"performed": True, "approved": True},
)
lease = self.session.scalar(select(DistributedLease))
assert lease is not None
lease.expires_at = datetime.now(timezone.utc) - timedelta(seconds=1)
self.session.commit()
bind_process_runtime_identity(_identity("node-2", "incarnation-2"))
with self.assertRaises(RecoveryOperationStateConflict):
claim_durable_recovery_operation(
recovery.operation.session_factory,
identity=_identity("node-2", "incarnation-2"),
operation_id=recovery.operation_id,
)
operation = self.session.get(RecoveryOperation, recovery.operation_id)
self.session.refresh(operation)
self.assertEqual(RecoveryStatus.FAILED.value, operation.status)
def test_external_mutation_unknown_outcome_blocks_blind_retry(self) -> None:
kwargs = {
"tenant_id": "tenant-1",
"provider_id": "provider-1",
"idempotency_key": "publish-1",
"request_sha256": "b" * 64,
"source_revision": "revision-1",
"cursor": None,
"dry_run_evidence": {"performed": True, "approved": True},
"resource_type": "external_record",
"resource_id": "record-1",
}
recovery = begin_connector_external_mutation(self.session, **kwargs)
recovery.outcome_unknown(
summary="The provider connection closed after dispatch",
provider_code="connection_closed",
)
with self.assertRaises(ConnectorRecoveryError):
begin_connector_external_mutation(self.session, **kwargs)
operation = self.session.get(RecoveryOperation, recovery.operation_id)
self.session.refresh(operation)
self.assertEqual(RecoveryStatus.OUTCOME_UNKNOWN.value, operation.status)
def test_tampered_chain_rolls_back_the_atomic_connector_projection(self) -> None:
recovery = begin_connector_read_snapshot(
self.session,
tenant_id="tenant-1",
provider_id="provider-1",
idempotency_key="tampered-read",
source_revision=None,
cursor=None,
dry_run_evidence={"performed": False, "reason": "read-only"},
)
checkpoint = self.session.scalar(
select(RecoveryCheckpoint)
.where(RecoveryCheckpoint.operation_id == recovery.operation_id)
.order_by(RecoveryCheckpoint.sequence)
.limit(1)
)
assert checkpoint is not None
checkpoint.summary = "tampered"
self.session.commit()
source = SqlTabularSourceProvider().create_snapshot(
self.session,
_principal(),
snapshot=TabularSnapshotInput(
name="Tampered",
source_name="tampered",
rows=({"id": 1},),
),
source_id=recovery.resource_id,
)
with self.assertRaises(ConnectorRecoveryError):
recovery.commit_success(
self.session,
evidence={
"verified": True,
"checks": {"snapshot_ref": source.ref},
},
)
self.assertIsNone(
self.session.get(ConnectorTabularSource, recovery.resource_id)
)
operation = self.session.get(RecoveryOperation, recovery.operation_id)
self.session.refresh(operation)
self.assertEqual(RecoveryStatus.RUNNING.value, operation.status)
def test_definitive_external_rejection_is_terminal(self) -> None:
recovery = begin_connector_external_mutation(
self.session,
tenant_id="tenant-1",
provider_id="provider-1",
idempotency_key="publish-rejected",
request_sha256="c" * 64,
source_revision="revision-1",
cursor=None,
dry_run_evidence={"performed": True, "approved": True},
resource_type="external_record",
resource_id="record-2",
)
recovery.reject(
summary="The provider rejected the requested revision",
provider_code="revision_conflict",
)
operation = self.session.get(RecoveryOperation, recovery.operation_id)
self.session.refresh(operation)
self.assertEqual(RecoveryStatus.REJECTED.value, operation.status)
if __name__ == "__main__":
unittest.main()
+391
View File
@@ -0,0 +1,391 @@
from __future__ import annotations
from datetime import timedelta
import hashlib
import unittest
from unittest.mock import patch
from urllib.error import URLError
from sqlalchemy import create_engine
from sqlalchemy.orm import Session
from govoplan_connectors.backend.db.models import (
ConnectorSanctionsAcquisitionRun,
ConnectorSanctionsSnapshot,
)
from govoplan_connectors.backend.sanctions_sources import (
SANCTIONS_READ_SCOPE,
SANCTIONS_REFRESH_SCOPE,
SOURCE_DEFINITIONS,
SYNTHETIC_PROVIDER_ID,
SYNTHETIC_UN_XML,
SanctionsSourceError,
SqlSanctionsSnapshotProvider,
TransportResponse,
UNSC_PROVIDER_ID,
UrllibSanctionsTransport,
)
from govoplan_core.auth import ApiPrincipal
from govoplan_core.core.access import PrincipalRef
from govoplan_core.core.recovery import (
RecoveryCheckpoint,
RecoveryOperation,
RecoveryStatus,
)
from govoplan_core.core.runtime_coordination import (
DistributedLease,
RuntimeIdentity,
bind_process_runtime_identity,
)
from govoplan_core.db.base import Base, utcnow
def principal(
tenant_id: str = "tenant-1",
*,
scopes: tuple[str, ...] = (
SANCTIONS_READ_SCOPE,
SANCTIONS_REFRESH_SCOPE,
),
) -> ApiPrincipal:
return ApiPrincipal(
principal=PrincipalRef(
account_id="account-1",
membership_id="membership-1",
tenant_id=tenant_id,
scopes=frozenset(scopes),
),
account=object(),
user=object(),
)
class _Transport:
def __init__(self, responses):
self.responses = list(responses)
self.headers = []
def fetch(self, definition, *, headers):
del definition
self.headers.append(dict(headers))
response = self.responses.pop(0)
if isinstance(response, Exception):
raise response
return response
def response(
content: bytes = SYNTHETIC_UN_XML,
*,
status: int = 200,
content_type: str = "application/xml",
etag: str = '"fixture-v1"',
) -> TransportResponse:
return TransportResponse(
status=status,
final_url="https://scsanctions.un.org/consolidated.xml",
headers={
"content-type": content_type,
"etag": etag,
},
content=content,
attempts=1,
)
class SanctionsSourcesTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite:///:memory:")
Base.metadata.create_all(
self.engine,
tables=(
DistributedLease.__table__,
RecoveryOperation.__table__,
RecoveryCheckpoint.__table__,
ConnectorSanctionsAcquisitionRun.__table__,
ConnectorSanctionsSnapshot.__table__,
),
)
self.session = Session(self.engine)
bind_process_runtime_identity(
RuntimeIdentity(
installation_id="connectors-tests",
node_id="connectors-test-node",
incarnation="connectors-test-incarnation",
role="worker",
software_version="test",
composition_hash="a" * 64,
)
)
def tearDown(self) -> None:
bind_process_runtime_identity(None)
self.session.close()
self.engine.dispose()
def test_fixture_refreshes_are_immutable_and_evidence_is_readable(
self,
) -> None:
provider = SqlSanctionsSnapshotProvider()
first = provider.refresh_source(
self.session,
principal(),
provider_id=SYNTHETIC_PROVIDER_ID,
)
second = provider.refresh_source(
self.session,
principal(),
provider_id=SYNTHETIC_PROVIDER_ID,
)
self.session.commit()
self.assertEqual("succeeded", first.status)
self.assertEqual("succeeded", second.status)
self.assertNotEqual(first.snapshot.ref, second.snapshot.ref)
self.assertEqual(
hashlib.sha256(SYNTHETIC_UN_XML).hexdigest(),
first.snapshot.sha256,
)
self.assertEqual(
SYNTHETIC_UN_XML,
provider.read_snapshot(
self.session,
principal(),
snapshot_ref=first.snapshot.ref,
).content,
)
runs = provider.list_runs(
self.session,
principal(),
)
self.assertTrue(
all(
run.request_evidence["subject_data_transmitted"]
is False
for run in runs
)
)
def test_conditional_fetch_reuses_prior_immutable_snapshot(self) -> None:
transport = _Transport(
(
response(),
response(b"", status=304),
)
)
provider = SqlSanctionsSnapshotProvider(transport)
first = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
)
second = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
)
self.assertEqual("not_modified", second.status)
self.assertEqual(first.snapshot.ref, second.snapshot.ref)
self.assertEqual(
{'If-None-Match': '"fixture-v1"'},
transport.headers[1],
)
self.assertEqual(
1,
self.session.query(ConnectorSanctionsSnapshot).count(),
)
def test_malformed_and_changed_sources_have_explicit_health(
self,
) -> None:
cases = (
(
b"<CONSOLIDATED_LIST>",
"application/xml",
"malformed",
),
(
b"<DIFFERENT><INDIVIDUALS/><ENTITIES/></DIFFERENT>",
"application/xml",
"unexpected_change",
),
(
SYNTHETIC_UN_XML,
"text/html",
"unexpected_change",
),
)
for payload, content_type, expected in cases:
with self.subTest(expected=expected, content_type=content_type):
provider = SqlSanctionsSnapshotProvider(
_Transport(
(
response(
payload,
content_type=content_type,
),
)
)
)
result = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
)
self.assertEqual(expected, result.status)
self.assertIsNotNone(result.error)
def test_unavailable_source_becomes_stale_when_evidence_is_old(
self,
) -> None:
provider = SqlSanctionsSnapshotProvider(
_Transport(
(
response(),
SanctionsSourceError("offline"),
)
)
)
first = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
)
record = self.session.get(
ConnectorSanctionsSnapshot,
first.snapshot.ref.removeprefix("sanctions-snapshot:"),
)
record.acquired_at = utcnow() - timedelta(days=3)
self.session.commit()
result = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
)
self.assertEqual("stale", result.status)
self.assertEqual(first.snapshot.ref, result.snapshot.ref)
def test_idempotent_refresh_replays_the_committed_acquisition(self) -> None:
transport = _Transport((response(),))
provider = SqlSanctionsSnapshotProvider(transport)
first = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
idempotency_key="scheduled-refresh-1",
)
replay = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
idempotency_key="scheduled-refresh-1",
)
self.assertEqual(first.run_id, replay.run_id)
self.assertEqual(first.snapshot.ref, replay.snapshot.ref)
self.assertEqual([], transport.responses)
operation = self.session.query(RecoveryOperation).one()
self.assertEqual(RecoveryStatus.SUCCEEDED.value, operation.status)
def test_provider_failure_commits_failed_run_and_terminal_recovery(self) -> None:
provider = SqlSanctionsSnapshotProvider(
_Transport((SanctionsSourceError("offline"),))
)
result = provider.refresh_source(
self.session,
principal(),
provider_id=UNSC_PROVIDER_ID,
)
self.assertEqual("unavailable", result.status)
operation = self.session.query(RecoveryOperation).one()
self.assertEqual(RecoveryStatus.FAILED.value, operation.status)
self.assertEqual(
result.run_id,
self.session.query(ConnectorSanctionsAcquisitionRun).one().id,
)
def test_snapshot_access_is_tenant_and_scope_isolated(self) -> None:
provider = SqlSanctionsSnapshotProvider()
created = provider.refresh_source(
self.session,
principal(),
provider_id=SYNTHETIC_PROVIDER_ID,
)
self.assertIsNone(
provider.get_snapshot(
self.session,
principal("tenant-2"),
snapshot_ref=created.snapshot.ref,
)
)
with self.assertRaisesRegex(Exception, "Missing scope"):
provider.list_snapshots(
self.session,
principal(scopes=()),
)
def test_transport_retries_transient_network_failures(self) -> None:
class _Headers(dict):
pass
class _Response:
status = 200
headers = _Headers(
{
"Content-Type": "application/xml",
"Content-Length": str(len(SYNTHETIC_UN_XML)),
}
)
def __enter__(self):
return self
def __exit__(self, *args):
return False
def geturl(self):
return (
"https://scsanctions.un.org/"
"resources/xml/en/consolidated.xml"
)
def read(self, size):
del size
if hasattr(self, "_read"):
return b""
self._read = True
return SYNTHETIC_UN_XML
opener = unittest.mock.Mock()
opener.open.side_effect = (
URLError("temporary"),
URLError("temporary"),
_Response(),
)
sleeps = []
transport = UrllibSanctionsTransport(
sleeper=sleeps.append
)
with patch(
"govoplan_connectors.backend.sanctions_sources.build_opener",
return_value=opener,
):
fetched = transport.fetch(
SOURCE_DEFINITIONS[UNSC_PROVIDER_ID],
headers={},
)
self.assertEqual(3, fetched.attempts)
self.assertEqual([1.0, 2.0], sleeps)
if __name__ == "__main__":
unittest.main()
+268
View File
@@ -0,0 +1,268 @@
from __future__ import annotations
import unittest
from fastapi import HTTPException
from sqlalchemy import create_engine
from sqlalchemy.orm import sessionmaker
from govoplan_core.auth import ApiPrincipal
from govoplan_core.core.access import PrincipalRef
from govoplan_core.core.tabular_sources import (
TabularReadRequest,
TabularSnapshotInput,
TabularSourceAccessError,
TabularSourceUnavailableError,
TabularSourceValidationError,
)
from govoplan_core.db.base import Base
from govoplan_connectors.backend.db.models import ConnectorTabularSource
from govoplan_connectors.backend.router import api_create_tabular_snapshot
from govoplan_connectors.backend.schemas import SnapshotCreateRequest
from govoplan_connectors.backend.tabular_sources import (
READ_SCOPE,
WRITE_SCOPE,
SqlTabularSourceProvider,
parse_csv_snapshot,
)
def principal(
tenant_id: str = "tenant-1",
*,
scopes: tuple[str, ...] = (READ_SCOPE, WRITE_SCOPE),
) -> ApiPrincipal:
return ApiPrincipal(
principal=PrincipalRef(
account_id="account-1",
membership_id="membership-1",
tenant_id=tenant_id,
scopes=frozenset(scopes),
),
account=object(),
user=object(),
)
class ConnectorsTabularSourceTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite:///:memory:")
Base.metadata.create_all(self.engine, tables=[ConnectorTabularSource.__table__])
self.Session = sessionmaker(bind=self.engine)
self.session = self.Session()
self.provider = SqlTabularSourceProvider()
def tearDown(self) -> None:
self.session.close()
Base.metadata.drop_all(self.engine, tables=[ConnectorTabularSource.__table__])
self.engine.dispose()
def test_snapshot_round_trip_preserves_schema_fingerprint_and_bounds(self) -> None:
created = self.provider.create_snapshot(
self.session,
principal(),
snapshot=TabularSnapshotInput(
name="Monthly cases",
source_name="monthly_cases_2026_07",
rows=(
{"case_id": "A-1", "amount": 12, "active": True},
{"case_id": "A-2", "amount": None, "active": False},
),
),
)
self.session.commit()
listed = self.provider.list_sources(self.session, principal())
preview = self.provider.read_source(
self.session,
principal(),
request=TabularReadRequest(
source_ref=created.ref,
limit=1,
expected_fingerprint=created.fingerprint,
),
)
self.assertEqual((created.ref,), tuple(source.ref for source in listed))
self.assertEqual(
["case_id", "amount", "active"],
[column.name for column in created.schema],
)
self.assertEqual(2, preview.total_rows)
self.assertEqual(1, len(preview.rows))
self.assertTrue(preview.truncated)
self.assertEqual(created.fingerprint, preview.source.fingerprint)
self.assertEqual("cached", preview.source.source_mode)
self.assertTrue(preview.source.pushdown.projections)
self.assertTrue(preview.source.pushdown.pagination)
self.assertEqual("healthy", preview.source.health.status)
self.assertGreater(preview.returned_bytes, 2)
self.assertEqual(1, preview.effective_row_limit)
self.assertEqual("preview.row_limit_reached", preview.diagnostics[0].code)
def test_preview_enforces_byte_time_and_provider_ceiling_budgets(self) -> None:
created = self.provider.create_snapshot(
self.session,
principal(),
snapshot=TabularSnapshotInput(
name="Bounded",
source_name="bounded",
rows=(
{"id": 1, "value": "first"},
{"id": 2, "value": "second"},
),
),
)
self.session.commit()
bounded = self.provider.read_source(
self.session,
principal(),
request=TabularReadRequest(
source_ref=created.ref,
limit=500,
max_bytes=35,
timeout_ms=2_000,
),
)
self.assertEqual(1, len(bounded.rows))
self.assertTrue(bounded.truncated)
self.assertEqual(
"preview.byte_limit_reached",
bounded.diagnostics[-1].code,
)
with self.assertRaisesRegex(
TabularSourceValidationError,
"single source row exceeds",
):
self.provider.read_source(
self.session,
principal(),
request=TabularReadRequest(
source_ref=created.ref,
max_bytes=2,
),
)
tightened = self.provider.read_source(
self.session,
principal(),
request=TabularReadRequest(
source_ref=created.ref,
limit=5_000,
max_bytes=5_000_000,
timeout_ms=10_000,
),
)
self.assertEqual(500, tightened.effective_row_limit)
self.assertEqual(1_000_000, tightened.effective_byte_limit)
self.assertEqual(2_000, tightened.effective_timeout_ms)
self.assertEqual(
{
"preview.row_limit_tightened",
"preview.byte_limit_tightened",
"preview.timeout_tightened",
},
{item.code for item in tightened.diagnostics},
)
times = iter((0.0, 0.01))
timeout_provider = SqlTabularSourceProvider(clock=lambda: next(times))
with self.assertRaisesRegex(
TabularSourceUnavailableError,
"time budget",
):
timeout_provider.read_source(
self.session,
principal(),
request=TabularReadRequest(
source_ref=created.ref,
timeout_ms=1,
),
)
def test_tenant_and_scope_isolation_are_enforced(self) -> None:
created = self.provider.create_snapshot(
self.session,
principal(),
snapshot=TabularSnapshotInput(
name="Private",
source_name="private_source",
rows=({"id": 1},),
),
)
self.session.commit()
self.assertEqual((), self.provider.list_sources(self.session, principal("tenant-2")))
self.assertIsNone(
self.provider.get_source(
self.session,
principal("tenant-2"),
source_ref=created.ref,
)
)
with self.assertRaises(TabularSourceAccessError):
self.provider.list_sources(
self.session,
principal(scopes=()),
)
def test_duplicate_source_name_and_stale_fingerprint_are_rejected(self) -> None:
snapshot = TabularSnapshotInput(
name="Cases",
source_name="cases",
rows=({"id": 1},),
)
created = self.provider.create_snapshot(self.session, principal(), snapshot=snapshot)
self.session.commit()
with self.assertRaises(TabularSourceValidationError):
self.provider.create_snapshot(self.session, principal(), snapshot=snapshot)
with self.assertRaises(TabularSourceValidationError):
self.provider.read_source(
self.session,
principal(),
request=TabularReadRequest(
source_ref=created.ref,
expected_fingerprint="stale",
),
)
def test_csv_parser_infers_scalar_values_and_rejects_duplicate_headers(self) -> None:
rows = parse_csv_snapshot(
"\ufeffid;amount;active;note\n0012;12.5;true;\n2;7;false;ok\n",
delimiter=";",
)
self.assertEqual(
(
{"id": "0012", "amount": 12.5, "active": True, "note": None},
{"id": 2, "amount": 7, "active": False, "note": "ok"},
),
rows,
)
with self.assertRaises(TabularSourceValidationError):
parse_csv_snapshot("id,id\n1,2\n", delimiter=",")
with self.assertRaises(TabularSourceValidationError):
parse_csv_snapshot("id,name\n1,Ada,extra\n", delimiter=",")
def test_malformed_csv_api_request_is_reported_as_validation_error(self) -> None:
payload = SnapshotCreateRequest(
name="Malformed",
source_name="malformed",
format="csv",
csv_text="id,name\n1,Ada,extra\n",
)
with self.assertRaises(HTTPException) as raised:
api_create_tabular_snapshot(
payload,
session=self.session,
principal=principal(),
)
self.assertEqual(422, raised.exception.status_code)
if __name__ == "__main__":
unittest.main()