Compare commits

..
55 Commits
Author SHA1 Message Date
zemion edee29af51 Release v0.1.16
Module Package Release / publish-packages (push) Successful in 12s
2026-08-05 19:52:09 +02:00
zemion 0293e49def Release v0.1.15
Module Package Release / publish-packages (push) Successful in 12s
2026-08-04 15:18:12 +02:00
zemion aaa364b8d4 Make package publication retries hash-safe 2026-08-04 14:32:19 +02:00
zemion 4bb17c0f4c Harden module package publication 2026-08-04 14:02:40 +02:00
zemion 92e649477f Pin S3 and SMB connector peers 2026-08-04 10:40:46 +02:00
zemion 0c68e904cf Add protected package release workflow 2026-08-04 04:14:04 +02:00
zemion 8ec31b16ee Add native permission-aware file search source 2026-08-04 03:03:26 +02:00
zemion 7e6be4b017 Add governed Files integrity operations UI 2026-08-04 01:04:39 +02:00
zemion 04882f1628 Fix PostgreSQL file visibility queries 2026-08-04 01:04:39 +02:00
zemion d8ae506ff8 Migrate Files surfaces to interface patterns 2026-08-03 09:35:52 +02:00
zemion 6baf2a421b Fence managed file object effects 2026-08-03 04:21:52 +02:00
zemion b993d8e31a Provide managed generated artifact storage 2026-08-02 12:38:15 +02:00
zemion 359c4e9570 feat: support optional encrypted file content 2026-08-02 03:40:54 +02:00
zemion 233ce40983 feat: declare governed external provider state 2026-08-01 17:48:32 +02:00
zemion f084a0f3a9 Add managed storage operational probes 2026-07-31 22:48:07 +02:00
zemion b752dea610 Add safe archive preview and extraction workflows 2026-07-31 02:48:56 +02:00
zemion 159a012833 feat: classify file administration surfaces 2026-07-30 17:42:06 +02:00
zemion 835eacfc5d feat: harden file sharing and integrity 2026-07-30 14:26:47 +02:00
zemion 85606d5580 Add server-backed file property filters 2026-07-30 05:22:09 +02:00
zemion 632cc6cf7d perf(files): batch tenant summaries 2026-07-30 01:15:53 +02:00
zemion 86a905a3a7 refactor(api): split files workflow routers 2026-07-29 20:11:17 +02:00
zemion 5b868272b9 security: harden connector policy and add dashboard widget 2026-07-29 14:16:29 +02:00
zemion 93eea839c8 Declare file workspace View surfaces 2026-07-28 21:04:55 +02:00
zemion b667e7ff0e Integrate file connectors with credential envelopes 2026-07-28 19:33:01 +02:00
zemion 0fd8f6e02a docs: update GovOPlaN repository links 2026-07-27 15:46:51 +02:00
zemion 2b34f6e305 fix(files): align adaptive tasks with live access 2026-07-21 19:15:47 +02:00
zemion 58af1a20a7 security(files): filter connectors before secret resolution 2026-07-21 19:15:41 +02:00
zemion 4722161592 feat(files): surface configured handbook tasks 2026-07-21 18:54:26 +02:00
zemion 1444ba80a0 refactor(files): centralize connector visibility 2026-07-21 18:52:33 +02:00
zemion 02ef83ecee test: commit Files retirement transaction 2026-07-21 18:18:33 +02:00
zemion 1069f85796 Refactor connector profile updates 2026-07-21 18:16:55 +02:00
zemion 1401c78c8a docs: add files workflow assurance topics 2026-07-21 17:13:53 +02:00
zemion b6109245a7 docs(files): describe current sharing permission 2026-07-21 17:07:40 +02:00
zemion f5d40b23c2 docs(files): link lifecycle reliability backlog 2026-07-21 16:48:13 +02:00
zemion 94ea629635 docs(files): require import workflow permissions 2026-07-21 16:33:06 +02:00
zemion 6c8a8c655d docs(files): expose adaptive handbook topics 2026-07-21 16:31:49 +02:00
zemion 062ad5ddfb docs(files): add adaptive operations handbook 2026-07-21 16:27:23 +02:00
zemion 9adfa91e74 docs(files): define connector credential retirement 2026-07-21 16:16:53 +02:00
zemion 65d8ed80b5 security(files): scrub connector credentials on deletion 2026-07-21 16:16:47 +02:00
zemion cffe161f29 Document governed file connector boundaries 2026-07-21 15:44:00 +02:00
zemion 5248e7de4a Fail closed for SMB referral transports 2026-07-21 15:43:56 +02:00
zemion d5d0df792b Disable connector transports without peer pinning 2026-07-21 15:38:09 +02:00
zemion 15ade8df75 Use shared settings target layout 2026-07-21 13:48:04 +02:00
zemion f3c485ef61 Use Core explorer work-surface styling 2026-07-21 13:19:29 +02:00
zemion 0e36b20a14 refactor(webui): use core access explanation 2026-07-21 13:19:11 +02:00
zemion 06e6e7191b refactor(files): decompose S3 connector imports 2026-07-21 12:57:44 +02:00
zemion 3449cbc8a5 refactor(files): decompose campaign snapshot preparation 2026-07-21 12:57:40 +02:00
zemion 92950af6f4 refactor(files): simplify asset and delta responses 2026-07-21 12:57:37 +02:00
zemion 8826cf2890 Use managed secret controls for Files connectors 2026-07-21 12:13:00 +02:00
zemion f2dfb6c90e Harden external file connector boundaries 2026-07-21 12:10:23 +02:00
zemion 3bc1d3489e Narrow Files backend import surfaces 2026-07-21 03:16:23 +02:00
zemion d1051293b2 feat(files): expose campaign attachment linking 2026-07-20 20:06:04 +02:00
zemion 8c5f149f07 fix(files): route drops through the selected target 2026-07-20 16:57:34 +02:00
zemion f3210234d3 intermittent commit 2026-07-14 13:22:11 +02:00
zemion b8b395e8b5 Run SMB dev connector as non-root 2026-07-11 18:37:51 +02:00
111 changed files with 22887 additions and 4724 deletions
+270
View File
@@ -0,0 +1,270 @@
name: Module Package Release
on:
push:
tags:
- "v*"
workflow_dispatch:
inputs:
release_tag:
description: Existing protected version tag to publish
required: true
type: string
jobs:
publish-packages:
runs-on: ubuntu-latest
env:
GITEA_REPOSITORY: ${{ gitea.repository }}
steps:
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5
with:
fetch-depth: 0
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065
with:
python-version: "3.12"
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020
with:
node-version: "22"
- name: Select and validate protected release tag
shell: bash
env:
REQUESTED_TAG: ${{ inputs.release_tag }}
TRIGGER_TAG: ${{ gitea.ref_name }}
run: |
set -euo pipefail
tag="${REQUESTED_TAG:-$TRIGGER_TAG}"
case "$tag" in
v[0-9]*.[0-9]*.[0-9]*) ;;
*) echo "Release tag must start with a SemVer-shaped vX.Y.Z value" >&2; exit 1 ;;
esac
git fetch --force origin "refs/tags/$tag:refs/tags/$tag" refs/heads/main:refs/remotes/origin/main
tag_commit="$(git rev-list -n 1 "$tag")"
git merge-base --is-ancestor "$tag_commit" refs/remotes/origin/main || {
echo "Release tag is not contained in main" >&2
exit 1
}
git checkout --detach "$tag"
printf 'RELEASE_TAG=%s\n' "$tag" >> "$GITEA_ENV"
printf 'SOURCE_DATE_EPOCH=%s\n' "$(git show -s --format=%ct HEAD)" >> "$GITEA_ENV"
- name: Validate package versions
run: |
python - <<'PY'
import json
from pathlib import Path
import os
import re
import tomllib
tag = os.environ["RELEASE_TAG"]
expected = tag.removeprefix("v")
project = tomllib.loads(Path("pyproject.toml").read_text(encoding="utf-8"))["project"]
if project.get("version") != expected:
raise SystemExit(f"pyproject version {project.get('version')!r} does not match {tag}")
if re.fullmatch(r"govoplan-[a-z0-9-]+", str(project.get("name", ""))) is None:
raise SystemExit("Python distribution name must use the govoplan-* namespace")
webui = Path("webui/package.json")
if webui.is_file():
package = json.loads(webui.read_text(encoding="utf-8"))
if package.get("version") != expected:
raise SystemExit(f"WebUI version {package.get('version')!r} does not match {tag}")
if re.fullmatch(r"@govoplan/[a-z0-9-]+-webui", str(package.get("name", ""))) is None:
raise SystemExit("WebUI package name must use the @govoplan/*-webui namespace")
release = Path("webui/package.release.json")
if release.is_file():
release_package = json.loads(release.read_text(encoding="utf-8"))
if (
release_package.get("name") != package.get("name")
or release_package.get("version") != expected
):
raise SystemExit("WebUI release package identity does not match package.json and the release tag")
PY
- name: Build immutable package artifacts
shell: bash
run: |
set -euo pipefail
python -m pip install --disable-pip-version-check build==1.5.0 twine==7.0.0
rm -rf dist .package-webui
python -m build --wheel --outdir dist
python -m twine check dist/*.whl
if [[ -f webui/package.json ]]; then
mkdir .package-webui
cp -a webui/. .package-webui/
rm -rf .package-webui/node_modules .package-webui/dist
if [[ -f .package-webui/package.release.json ]]; then
cp .package-webui/package.release.json .package-webui/package.json
fi
node <<'NODE'
const fs = require("node:fs");
const path = ".package-webui/package.json";
const packageJson = JSON.parse(fs.readFileSync(path, "utf8"));
const groups = ["dependencies", "optionalDependencies", "peerDependencies"];
for (const group of groups) {
for (const [name, specifier] of Object.entries(packageJson[group] || {})) {
if (!name.startsWith("@govoplan/")) continue;
if (typeof specifier !== "string") {
throw new Error(`${group}.${name} must use a string version`);
}
const packageSlug = name.slice("@govoplan/".length);
if (!packageSlug.endsWith("-webui")) {
throw new Error(`${group}.${name} is outside the WebUI package namespace`);
}
const repository = `govoplan-${packageSlug.slice(0, -"-webui".length)}`;
const escapedRepository = repository.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
const gitTag = specifier.match(
new RegExp(
`^git\\+(?:ssh://git@|https://)git\\.add-ideas\\.de/(?:GovOPlaN|add-ideas)/${escapedRepository}\\.git#v([0-9]+\\.[0-9]+\\.[0-9]+)$`,
),
);
if (gitTag) {
packageJson[group][name] = gitTag[1];
continue;
}
if (specifier.startsWith("file:") || specifier.startsWith("git+")) {
throw new Error(
`${group}.${name} must resolve to an exact registry version for publication`,
);
}
}
}
delete packageJson.private;
fs.writeFileSync(path, `${JSON.stringify(packageJson, null, 2)}\n`);
NODE
npm pkg delete private --prefix .package-webui
(cd .package-webui && npm pack --ignore-scripts --pack-destination ../dist)
fi
python - <<'PY'
import hashlib
import json
from pathlib import Path
import os
import subprocess
artifacts = []
for path in sorted(Path("dist").iterdir()):
if path.suffix not in {".whl", ".tgz"}:
continue
digest = hashlib.sha256(path.read_bytes()).hexdigest()
artifacts.append({"filename": path.name, "sha256": digest, "size": path.stat().st_size})
payload = {
"schema_version": "1",
"repository": os.environ["GITEA_REPOSITORY"],
"tag": os.environ["RELEASE_TAG"],
"commit": subprocess.check_output(["git", "rev-parse", "HEAD"], text=True).strip(),
"artifacts": artifacts,
}
Path("dist/package-artifacts.json").write_text(
json.dumps(payload, indent=2, sort_keys=True) + "\n",
encoding="utf-8",
)
PY
- name: Retain package hash evidence
uses: actions/upload-artifact@a8a3f3ad30e3422c9c7b888a15615d19a852ae32
with:
name: module-packages-${{ gitea.ref_name }}
path: dist/package-artifacts.json
- name: Check immutable registry state
shell: bash
env:
PACKAGE_TOKEN: ${{ secrets.GOVOPLAN_PACKAGE_TOKEN }}
run: |
set -euo pipefail
test -n "$PACKAGE_TOKEN"
python - <<'PY'
import hashlib
import json
import os
from pathlib import Path
import tomllib
from urllib.error import HTTPError
from urllib.parse import quote
from urllib.request import Request, urlopen
api_root = "https://git.add-ideas.de/api/v1/packages/GovOPlaN"
token = os.environ["PACKAGE_TOKEN"]
def should_publish(kind, name, version, path):
package_url = "/".join(
(api_root, kind, quote(name, safe=""), quote(version, safe=""), "files")
)
request = Request(
package_url,
headers={"Accept": "application/json", "Authorization": f"token {token}"},
)
try:
with urlopen(request, timeout=30) as response:
files = json.load(response)
except HTTPError as exc:
if exc.code == 404:
print(f"{kind} package {name}=={version} is not published yet")
return True
raise
if not isinstance(files, list) or len(files) != 1:
raise SystemExit(
f"immutable {kind} package {name}=={version} has an unexpected file set"
)
expected_sha256 = hashlib.sha256(path.read_bytes()).hexdigest()
if files[0].get("sha256") != expected_sha256:
raise SystemExit(
f"immutable {kind} package {name}=={version} already exists with a different SHA-256"
)
print(f"verified existing {kind} package {name}=={version} ({expected_sha256})")
return False
project = tomllib.loads(Path("pyproject.toml").read_text(encoding="utf-8"))["project"]
wheels = tuple(Path("dist").glob("*.whl"))
if len(wheels) != 1:
raise SystemExit("release build must contain exactly one wheel")
publish_pypi = should_publish(
"pypi", str(project["name"]), str(project["version"]), wheels[0]
)
tarballs = tuple(Path("dist").glob("*.tgz"))
if len(tarballs) > 1:
raise SystemExit("release build must contain at most one npm package")
publish_npm = False
if tarballs:
webui = json.loads(
Path(".package-webui/package.json").read_text(encoding="utf-8")
)
publish_npm = should_publish(
"npm", str(webui["name"]), str(webui["version"]), tarballs[0]
)
with Path(os.environ["GITEA_ENV"]).open("a", encoding="utf-8") as env_file:
env_file.write(f"PUBLISH_PYPI={int(publish_pypi)}\n")
env_file.write(f"PUBLISH_NPM={int(publish_npm)}\n")
PY
- name: Publish wheel and WebUI package
shell: bash
env:
PACKAGE_USERNAME: ${{ secrets.GOVOPLAN_PACKAGE_USERNAME }}
PACKAGE_TOKEN: ${{ secrets.GOVOPLAN_PACKAGE_TOKEN }}
run: |
set -euo pipefail
test -n "$PACKAGE_USERNAME"
test -n "$PACKAGE_TOKEN"
if [[ "$PUBLISH_PYPI" == 1 ]]; then
TWINE_USERNAME="$PACKAGE_USERNAME" TWINE_PASSWORD="$PACKAGE_TOKEN" \
python -m twine upload --non-interactive \
--repository-url https://git.add-ideas.de/api/packages/GovOPlaN/pypi \
dist/*.whl
else
echo "Exact wheel is already present; skipping immutable retry."
fi
shopt -s nullglob
webui_packages=(dist/*.tgz)
if (( ${#webui_packages[@]} )) && [[ "$PUBLISH_NPM" == 1 ]]; then
npmrc="$(mktemp)"
trap 'rm -f "$npmrc"' EXIT
chmod 600 "$npmrc"
printf '%s\n' \
'@govoplan:registry=https://git.add-ideas.de/api/packages/GovOPlaN/npm/' \
"//git.add-ideas.de/api/packages/GovOPlaN/npm/:_authToken=$PACKAGE_TOKEN" \
> "$npmrc"
NPM_CONFIG_USERCONFIG="$npmrc" npm publish "./${webui_packages[0]}" \
--ignore-scripts --access public \
--registry https://git.add-ideas.de/api/packages/GovOPlaN/npm/
elif (( ${#webui_packages[@]} )); then
echo "Exact WebUI package is already present; skipping immutable retry."
fi
+6
View File
@@ -1,5 +1,11 @@
# GovOPlaN Files Codex Guide
## Documentation Contract
- Treat documentation as part of every behavior change. Update this module's manifest-driven `DocumentationTopic` contributions for affected user and administrator behavior.
- Keep feature content here; `govoplan-docs` projects it without importing Files internals.
- Maintain a static user/admin baseline and run `/mnt/DATA/git/govoplan/tools/checks/check-manifest-shapes.py` after behavior or manifest changes.
## Scope
This repository owns the `files` module: managed file storage APIs, file metadata, shares, uploads/downloads, folder and pattern helpers, backend module manifest, and `@govoplan/files-webui`.
+74 -8
View File
@@ -78,7 +78,17 @@ Profiles can be supplied as JSON through
`GOVOPLAN_FILES_CONNECTOR_PROFILES_JSON`, or from a JSON file path through
`GOVOPLAN_FILES_CONNECTOR_PROFILES_FILE`. Each profile has an `id`, `provider`,
`endpoint_url`, governance `scope_type`/`scope_id`, optional `capabilities`, and
credential references such as `password_env`, `token_env`, or `secret_ref`.
deployment-owned credential references such as `password_env`, `token_env`, or
`secret_ref`. Environment references require an exact name in the deployment-wide
`GOVOPLAN_CONNECTOR_SECRET_ENV_ALLOWLIST`; API-managed profiles cannot select
process environment variables and may use only Files-owned encrypted password or
token values. API-created `secret_ref` values fail closed until Files has an
ownership contract that can confirm provider-side deletion. Legacy external
references are treated as non-owned: deleting a profile or credential detaches
and audits the reference but never passes it to an arbitrary secret provider.
Profile and credential deletion immediately clears encrypted values, credential
identities, deployment references, and private metadata in the same transaction
as the non-secret audit record; an audit failure rolls the deletion back.
`GET /api/v1/files/connectors/profiles` returns only profiles visible to the
current principal (system, tenant, user, group, or accessible campaign scope) and
redacts secret values and environment variable names. Use the returned
@@ -95,15 +105,45 @@ run profile policy with the `browse` operation, and support Seafile,
WebDAV/Nextcloud, and SMB when the optional `smb` extra is installed. Seafile
profiles browse libraries and directories via Seafile's read-only API; profiles
can still opt into WebDAV browsing by setting `metadata.webdav_endpoint_url` or
`metadata.browse_protocol` to `webdav`.
`metadata.browse_protocol` to `webdav`. S3-compatible profiles support bucket
and prefix browsing when the optional `s3` extra is installed.
`POST /api/v1/files/connectors/profiles/{profile_id}/import` imports a Seafile
or WebDAV/Nextcloud/SMB file into managed storage through the same governance,
conflict handling, source provenance, and connector audit path as direct
uploads. The Seafile provider uses account-token auth and the native file
download-link API; Nextcloud and generic WebDAV profiles use authenticated `GET`
requests against the configured WebDAV endpoint. SMB profiles use
`smb://server[:port]/share[/path]` endpoints and environment-backed credentials
through `smbprotocol`.
`smb://server[:port]/share[/path]` endpoints and deployment-owned or encrypted
stored credentials through `smbprotocol`. Files installs a pinned transport for
every initial session, reconnect, alias, and DFS referral target. The deployment
private-network policy is re-evaluated immediately before each socket opens.
S3 connector browse/import binds botocore HTTP and HTTPS pools to the same pinned
socket policy. Retries, redirects, endpoint discovery, bucket aliases, and new
connections are therefore revalidated while TLS keeps the configured hostname
for SNI and certificate checks. Connector clients do not use outbound proxies or
ambient AWS credential discovery; configure credentials on the governed profile,
or explicitly use an anonymous profile for public objects. An incompatible SDK
upgrade fails closed before a usable client/session is returned.
Durable platform storage has a separate deployment boundary: installer-owned
Garage is accepted only at its exact generated endpoint, while an
operator-selected external backend requires a clean HTTPS origin and
`FILE_STORAGE_S3_ENDPOINT_TRUSTED=true`. That flag is deployment configuration,
cannot be supplied through a Files connector profile, and does not replace
operator responsibility for DNS, certificates, egress, bucket policy,
versioning, and recovery.
The actual local/S3 backend implementation is owned by Core so Files, Campaign,
and workers resolve the same object namespace. Files owns file metadata and key
layout. Node-local storage is supported only for `local` or one-host
`host-shared` profiles; multi-host `shared` deployments require S3.
Destructive Files-module retirement applies the same credential lifecycle before
dropping tables. Every remaining Files-owned encrypted connector secret is
scrubbed and audited first, while legacy non-owned external references are
detached and identified as such in the audit record. Retirement does not claim
or attempt provider-side deletion for references Files cannot prove it owns.
Local connector development assets live in `dev/connectors/`. The compose stack
boots Nextcloud, Seafile, WebDAV, and SMB endpoints for provider development and
@@ -111,17 +151,43 @@ manual interoperability testing.
Connector and collaboration ownership boundaries are documented in
`docs/CONNECTOR_BOUNDARY.md` and `docs/DOCUMENT_COLLABORATION_BOUNDARY.md`.
The role-adaptive user, administration, integration, and operator guide is the
[Files handbook](docs/FILES_HANDBOOK.md).
The Files route, connector surfaces, consequence classes, and pattern-language
verification are recorded in
[Files interface pattern migration](docs/INTERFACE_PATTERN_MIGRATION.md).
ZIP uploads are processed without buffering the whole archive in memory. The API
spools incoming ZIP request bodies to a bounded temporary file, then extracts
members with per-file and total extracted-size limits before storing managed
files.
Archive imports use a two-phase preview and confirmation flow for ZIP, TAR,
TAR.GZ, TAR.BZ2, and TAR.XZ. Requests are spooled to bounded temporary files;
the server validates paths, entry count, expanded size, and expansion ratio,
then returns a 30-minute tenant/user-bound preview token. Confirmation reuploads
the original archive and stores only the selected members. Password-protected
ZIP passwords remain request-only and are never included in the preview token.
Managed blob writes and applied orphan cleanup use Core's durable recovery
ledger. Intent, request digests, recovery mode, and a distributed lease are
committed before physical storage effects; the Files session commit verifies
database and streamed object evidence, while rollback compensates only a newly
reserved unreferenced key. New object keys are opaque and do not retain the
uploaded filename. Uncertain or mismatched effects remain visible through Ops.
Operators with `files:file:admin` can run bounded, resumable integrity scans in
Administration. Scan batches and finding actions carry monotonic revisions;
stale resume, recheck, or cleanup requests fail before touching object storage.
Orphan cleanup always requires a dry-run preview followed by separate
confirmation and records recovery-ledger evidence.
Bulk rename and transfer APIs are owner-scoped: callers must provide the active
user or group file space with `owner_type` and `owner_id`. The storage layer
keeps a named legacy file-only helper for historical callers that lack owner
context, and regression tests cover its write-access checks.
Optional producer modules can persist generated output through the provider-
neutral `files.artifact_store` capability. Files remains responsible for upload
authorization, ownership, path normalization, versioning, and blob storage;
producers receive stable file/version references without importing Files
internals. See [Generated Artifact Store](docs/GENERATED_ARTIFACT_STORE.md).
## Release packaging
The repository root includes a `package.json` for git-based WebUI installs. It exports the package `@govoplan/files-webui` from `webui/src` so release builds can depend on tagged git refs instead of local `file:` paths.
+6
View File
@@ -17,3 +17,9 @@ SMB_HOST_PORT=1445
SMB_SHARE_NAME=files
SMB_USER=govoplan
SMB_PASSWORD=govoplan-smb
MINIO_API_HOST_PORT=9000
MINIO_CONSOLE_HOST_PORT=9001
MINIO_ROOT_USER=govoplan
MINIO_ROOT_PASSWORD=govoplan-minio
MINIO_BUCKET=govoplan
+49 -7
View File
@@ -9,7 +9,8 @@ binds services to localhost high ports.
```bash
cd /mnt/DATA/git/govoplan-files/dev/connectors
cp .env.example .env
docker compose up -d nextcloud nextcloud-db webdav smb
mkdir -p data/smb data/webdav
docker compose up -d nextcloud nextcloud-db webdav smb minio
docker compose up -d seafile-db seafile-memcached seafile
```
@@ -29,14 +30,17 @@ Endpoints:
`http://127.0.0.1:9082/seafdav/`
- WebDAV: `http://127.0.0.1:9083`
- SMB: `smb://127.0.0.1:1445/files`
- MinIO/S3 API: `http://127.0.0.1:9000`, console
`http://127.0.0.1:9001`
The local fixture data under `data/` is ignored by git.
## GovOPlaN Profile Config
Connector profiles are read from `GOVOPLAN_FILES_CONNECTOR_PROFILES_JSON` or
`GOVOPLAN_FILES_CONNECTOR_PROFILES_FILE`. Credentials should be referenced by
secret refs or environment variables; profile API responses only expose the
`GOVOPLAN_FILES_CONNECTOR_PROFILES_FILE`. This deployment-owned configuration
may reference environment variables only when their exact names are listed in
`GOVOPLAN_CONNECTOR_SECRET_ENV_ALLOWLIST`; profile API responses expose only the
credential source and configured state.
Example local profile file:
@@ -92,6 +96,25 @@ Example local profile file:
"password_env": "SMB_PASSWORD",
"capabilities": ["browse", "import"],
"policy": { "allow": { "providers": ["smb"] } }
},
{
"id": "dev-s3",
"label": "Dev MinIO",
"provider": "s3",
"endpoint_url": "http://127.0.0.1:9000",
"base_path": "",
"scope_type": "system",
"credential_mode": "basic",
"username": "govoplan",
"password_env": "MINIO_ROOT_PASSWORD",
"capabilities": ["browse", "import"],
"metadata": {
"bucket": "govoplan",
"region": "us-east-1",
"path_style": true,
"verify_tls": false
},
"policy": { "allow": { "providers": ["s3"] } }
}
]
}
@@ -101,6 +124,8 @@ Start GovOPlaN with:
```bash
export GOVOPLAN_FILES_CONNECTOR_PROFILES_FILE=/mnt/DATA/git/govoplan-files/dev/connectors/profiles.local.json
export GOVOPLAN_CONNECTOR_SECRET_ENV_ALLOWLIST=SEAFILE_ADMIN_PASSWORD,NEXTCLOUD_ADMIN_PASSWORD,WEBDAV_PASSWORD,SMB_PASSWORD,MINIO_ROOT_PASSWORD
export GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS=true
```
## Smoke Test
@@ -113,10 +138,13 @@ cd /mnt/DATA/git/govoplan-files/dev/connectors
/mnt/DATA/git/govoplan-core/.venv/bin/python smoke.py
```
The script reads `.env` from this directory when present, seeds tiny WebDAV,
Nextcloud, and SMB fixtures, then browses and imports them through the connector
helper layer. Pass `--require-smb` after recreating the SMB container to fail on
SMB access errors instead of reporting them as an optional skip.
The script reads `.env` from this directory when present and seeds tiny WebDAV,
Nextcloud, SMB, and MinIO fixtures. WebDAV and Nextcloud are browsed and imported
through the pinned connector transport. SMB and S3 product access deliberately
fails closed until their SDK transports support peer pinning, so their fixtures
are retained for transport development and reported as expected optional skips.
The `--require-smb` and `--require-s3` switches are useful only while developing
that transport support and currently make the smoke check fail by design.
SMB smoke checks need the optional Python dependency in the environment running
the script:
@@ -125,6 +153,20 @@ the script:
/mnt/DATA/git/govoplan-core/.venv/bin/python -m pip install -e /mnt/DATA/git/govoplan-files[smb]
```
S3 smoke checks need the optional boto3 dependency in the environment running
the script:
```bash
/mnt/DATA/git/govoplan-core/.venv/bin/python -m pip install -e /mnt/DATA/git/govoplan-files[s3]
```
The SMB service is built from `dev/connectors/smb/` so the development share is
deterministic: one `files` share backed by `data/smb`, with the credentials from
`.env`.
The SMB image runs as the non-root `govoplan` user with UID/GID `1000` and
listens on unprivileged container port `1445`. The default host endpoint remains
`smb://127.0.0.1:1445/files`. `SMB_USER` must stay `govoplan` unless the image is
rebuilt with a matching user; `SMB_PORT` must be `1024` or higher. `SMB_PASSWORD`
is applied when the image is built, so rebuild the `smb` service after changing
it in `.env`.
+18 -4
View File
@@ -86,21 +86,35 @@ services:
smb:
build:
context: ./smb
args:
SMB_PASSWORD: ${SMB_PASSWORD:-govoplan-smb}
image: govoplan-files-samba-dev:latest
restart: unless-stopped
environment:
SMB_SHARE_NAME: ${SMB_SHARE_NAME:-files}
SMB_USER: ${SMB_USER:-govoplan}
SMB_PASSWORD: ${SMB_PASSWORD:-govoplan-smb}
SMB_UID: ${SMB_UID:-1000}
SMB_GID: ${SMB_GID:-1000}
SMB_PORT: ${SMB_PORT:-1445}
ports:
- "127.0.0.1:${SMB_HOST_PORT:-1445}:445"
- "127.0.0.1:${SMB_HOST_PORT:-1445}:${SMB_PORT:-1445}"
volumes:
- ./data/smb:/storage
minio:
image: minio/minio:latest
restart: unless-stopped
command: server /data --console-address ":9001"
environment:
MINIO_ROOT_USER: ${MINIO_ROOT_USER:-govoplan}
MINIO_ROOT_PASSWORD: ${MINIO_ROOT_PASSWORD:-govoplan-minio}
ports:
- "127.0.0.1:${MINIO_API_HOST_PORT:-9000}:9000"
- "127.0.0.1:${MINIO_CONSOLE_HOST_PORT:-9001}:9001"
volumes:
- minio:/data
volumes:
nextcloud-db:
nextcloud:
seafile-db:
seafile-data:
minio:
+21 -2
View File
@@ -1,10 +1,29 @@
FROM alpine:3.20
RUN apk add --no-cache samba-server samba-common-tools
ARG SMB_PASSWORD=govoplan-smb
RUN apk add --no-cache samba-server samba-common-tools \
&& addgroup -S -g 1000 govoplan \
&& adduser -S -D -H -h /nonexistent -s /sbin/nologin -u 1000 -G govoplan govoplan \
&& mkdir -p /storage /var/lib/samba/private /var/cache/samba /run/samba /etc/samba /var/log/samba/cores \
&& printf '%s\n' \
'[global]' \
' passdb backend = tdbsam' \
' private dir = /var/lib/samba/private' \
' lock directory = /run/samba' \
' state directory = /var/lib/samba' \
' cache directory = /var/cache/samba' \
' pid directory = /run/samba' \
> /etc/samba/smb.conf \
&& printf '%s\n%s\n' "$SMB_PASSWORD" "$SMB_PASSWORD" | smbpasswd -s -a govoplan \
&& smbpasswd -e govoplan \
&& chown -R govoplan:govoplan /storage /var/lib/samba /var/cache/samba /run/samba /etc/samba /var/log/samba
COPY entrypoint.sh /usr/local/bin/govoplan-samba-entrypoint
RUN chmod +x /usr/local/bin/govoplan-samba-entrypoint
EXPOSE 445
EXPOSE 1445
USER govoplan:govoplan
ENTRYPOINT ["/usr/local/bin/govoplan-samba-entrypoint"]
+46 -14
View File
@@ -3,21 +3,42 @@ set -eu
share_name="${SMB_SHARE_NAME:-files}"
user_name="${SMB_USER:-govoplan}"
password="${SMB_PASSWORD:-govoplan-smb}"
uid="${SMB_UID:-1000}"
gid="${SMB_GID:-1000}"
smb_port="${SMB_PORT:-1445}"
runtime_user="$(id -un)"
runtime_group="$(id -gn)"
mkdir -p /storage /var/lib/samba/private /run/samba
case "$share_name" in
"" | *[!A-Za-z0-9_.-]*)
printf 'SMB_SHARE_NAME must contain only letters, numbers, dot, dash, or underscore.\n' >&2
exit 1
;;
esac
if ! getent group "$user_name" >/dev/null 2>&1; then
addgroup -g "$gid" "$user_name"
case "$user_name" in
"" | *[!A-Za-z0-9_.-]*)
printf 'SMB_USER must contain only letters, numbers, dot, dash, or underscore.\n' >&2
exit 1
;;
esac
case "$smb_port" in
"" | *[!0-9]*)
printf 'SMB_PORT must be numeric.\n' >&2
exit 1
;;
esac
if [ "$smb_port" -lt 1024 ]; then
printf 'SMB_PORT must be >= 1024 because the dev Samba container runs as a non-root user.\n' >&2
exit 1
fi
if ! id "$user_name" >/dev/null 2>&1; then
adduser -D -H -s /sbin/nologin -u "$uid" -G "$user_name" "$user_name"
if [ "$user_name" != "$runtime_user" ]; then
printf 'SMB_USER=%s is not supported by the non-root dev image; use %s or rebuild the image with a matching user.\n' "$user_name" "$runtime_user" >&2
exit 1
fi
chown -R "$user_name:$user_name" /storage
mkdir -p /storage /var/lib/samba/private /var/cache/samba /run/samba /etc/samba
cat > /etc/samba/smb.conf <<EOF
[global]
@@ -25,12 +46,20 @@ cat > /etc/samba/smb.conf <<EOF
workgroup = WORKGROUP
security = user
map to guest = Never
guest account = $runtime_user
server min protocol = SMB2
server signing = mandatory
smb ports = $smb_port
load printers = no
printing = bsd
disable spoolss = yes
log level = 1
passdb backend = tdbsam
private dir = /var/lib/samba/private
lock directory = /run/samba
state directory = /var/lib/samba
cache directory = /var/cache/samba
pid directory = /run/samba
[$share_name]
path = /storage
@@ -38,13 +67,16 @@ cat > /etc/samba/smb.conf <<EOF
read only = no
guest ok = no
valid users = $user_name
force user = $user_name
force group = $user_name
force user = $runtime_user
force group = $runtime_group
create mask = 0664
directory mask = 0775
EOF
printf '%s\n%s\n' "$password" "$password" | smbpasswd -s -a "$user_name" >/dev/null
smbpasswd -e "$user_name" >/dev/null
if ! pdbedit -L -u "$user_name" >/dev/null 2>&1; then
printf 'Samba user %s is missing from passdb. Rebuild the smb image so the non-root passdb is seeded.\n' "$user_name" >&2
printf 'Run: docker compose build --no-cache smb && docker compose up -d smb\n' >&2
exit 1
fi
exec smbd --foreground --no-process-group --debug-stdout
exec smbd --foreground --no-process-group --debug-stdout -p "$smb_port"
+70 -3
View File
@@ -18,16 +18,17 @@ ROOT = Path(__file__).resolve().parent
def main() -> int:
parser = argparse.ArgumentParser(description="Smoke-test GovOPlaN connector helpers against the local dev compose stack.")
parser.add_argument("--require-smb", action="store_true", help="Fail if the SMB connector cannot browse/import.")
parser.add_argument("--require-s3", action="store_true", help="Fail if the S3 connector cannot browse/import.")
parser.add_argument("--debug-smb", action="store_true", help="Print direct smbclient probes before the connector smoke check.")
args = parser.parse_args()
_load_dotenv()
_default_env()
preflight_failures = _preflight_services(require_smb=args.require_smb)
preflight_failures = _preflight_services(require_smb=args.require_smb, require_s3=args.require_s3)
if preflight_failures:
for failure in preflight_failures:
print(f"FAIL {failure}")
print("Start or recreate the connector stack from this directory with: docker compose up -d nextcloud nextcloud-db webdav smb")
print("Start or recreate the connector stack from this directory with: docker compose up -d nextcloud nextcloud-db webdav smb minio")
return 1
_seed_webdav_fixture()
_seed_smb_fixture()
@@ -36,6 +37,13 @@ def main() -> int:
except httpx.HTTPError as exc:
print(f"FAIL dev-nextcloud seed failed: {exc}")
return 1
try:
_seed_s3_fixture()
except Exception as exc:
if args.require_s3:
print(f"FAIL dev-s3 seed failed: {exc}")
return 1
print(f"SKIP dev-s3 seed failed: {exc}")
profiles = {profile.id: profile for profile in connector_profiles_from_payload({"profiles": _profile_payloads()})}
if args.debug_smb:
@@ -59,6 +67,15 @@ def main() -> int:
else:
print(f"SKIP {message}")
try:
_exercise_profile(profiles["dev-s3"], folder="GovOPlaN", file_path="GovOPlaN/s3-live.txt", expected="s3 live fixture")
except Exception as exc:
message = f"dev-s3: {exc}"
if args.require_s3:
failures.append(message)
else:
print(f"SKIP {message}")
if failures:
for failure in failures:
print(f"FAIL {failure}")
@@ -76,6 +93,9 @@ def _default_env() -> None:
"SMB_SHARE_NAME": "files",
"SMB_USER": "govoplan",
"SMB_PASSWORD": "govoplan-smb",
"MINIO_ROOT_USER": "govoplan",
"MINIO_ROOT_PASSWORD": "govoplan-minio",
"MINIO_BUCKET": "govoplan",
}
for key, value in defaults.items():
os.environ.setdefault(key, value)
@@ -96,7 +116,7 @@ def _load_dotenv() -> None:
os.environ[key] = value.strip().strip("\"'")
def _preflight_services(*, require_smb: bool) -> list[str]:
def _preflight_services(*, require_smb: bool, require_s3: bool) -> list[str]:
failures: list[str] = []
nextcloud_url = f"http://127.0.0.1:{os.getenv('NEXTCLOUD_HOST_PORT', '9081')}/status.php"
try:
@@ -121,6 +141,14 @@ def _preflight_services(*, require_smb: bool) -> list[str]:
pass
except OSError as exc:
failures.append(f"dev-smb is not reachable at 127.0.0.1:{smb_port}: {exc}")
if require_s3:
minio_url = f"http://127.0.0.1:{os.getenv('MINIO_API_HOST_PORT', '9000')}/minio/health/live"
try:
response = httpx.get(minio_url, timeout=3.0)
if response.status_code != 200:
failures.append(f"dev-s3 expected HTTP 200 at {minio_url}, got HTTP {response.status_code}")
except httpx.HTTPError as exc:
failures.append(f"dev-s3 is not reachable at {minio_url}: {exc}")
return failures
@@ -150,6 +178,20 @@ def _profile_payloads() -> list[dict[str, object]]:
"username": os.getenv("SMB_USER", "govoplan"),
"password_env": "SMB_PASSWORD",
},
{
"id": "dev-s3",
"provider": "s3",
"endpoint_url": f"http://127.0.0.1:{os.getenv('MINIO_API_HOST_PORT', '9000')}",
"credential_mode": "basic",
"username": os.getenv("MINIO_ROOT_USER", "govoplan"),
"password_env": "MINIO_ROOT_PASSWORD",
"metadata": {
"bucket": os.getenv("MINIO_BUCKET", "govoplan"),
"region": "us-east-1",
"path_style": True,
"verify_tls": False,
},
},
]
@@ -211,5 +253,30 @@ def _seed_nextcloud_fixture() -> None:
raise RuntimeError(f"Nextcloud PUT failed with HTTP {response.status_code}: {response.text[:200]}")
def _seed_s3_fixture() -> None:
try:
import boto3
from botocore.config import Config
from botocore.exceptions import ClientError
except ImportError as exc:
raise RuntimeError("boto3 is not installed; install govoplan-files[s3]") from exc
bucket = os.getenv("MINIO_BUCKET", "govoplan")
client = boto3.client(
"s3",
endpoint_url=f"http://127.0.0.1:{os.getenv('MINIO_API_HOST_PORT', '9000')}",
aws_access_key_id=os.getenv("MINIO_ROOT_USER", "govoplan"),
aws_secret_access_key=os.getenv("MINIO_ROOT_PASSWORD", "govoplan-minio"),
region_name="us-east-1",
config=Config(s3={"addressing_style": "path"}),
)
try:
client.create_bucket(Bucket=bucket)
except ClientError as exc:
code = exc.response.get("Error", {}).get("Code")
if code not in {"BucketAlreadyOwnedByYou", "BucketAlreadyExists"}:
raise
client.put_object(Bucket=bucket, Key="GovOPlaN/s3-live.txt", Body=b"s3 live fixture\n", ContentType="text/plain")
if __name__ == "__main__":
raise SystemExit(main())
+15 -2
View File
@@ -19,6 +19,7 @@ normal product workflows and can share the same governance model:
- Nextcloud through WebDAV
- generic WebDAV
- SMB through the optional `smb` extra
- S3-compatible stores through the optional `s3` extra
These providers are surfaced through connector descriptors at
`GET /api/v1/files/connectors/providers`. Provider descriptors declare whether
@@ -49,13 +50,25 @@ any remote write behavior.
Every provider must:
- enforce GovOPlaN profile visibility and connector policy before browse/import
- keep credentials as environment variables or secret references, never API
response values
- keep credentials as encrypted values or scoped secret references, never API
response values; process-environment references are allowed only in
deployment-owned profiles with an exact deployment allowlist
- import external files into managed storage before they are used in campaigns
or workflows
- preserve source provenance and revision metadata
- emit connector audit events for imported or accessed files
- treat remote ACLs as upstream checks, not as a replacement for GovOPlaN policy
- use a transport that pins every connection to a policy-validated DNS/IP answer
and revalidates redirects; SDK transports without that guarantee fail closed
SMB initial connections, reconnects, aliases, and DFS referral targets are
created through a Files-owned pinned `smbprotocol` transport. S3 HTTP/HTTPS pools
use the equivalent botocore adapter for every connection selected by retries,
redirects, endpoint discovery, and virtual-host addressing. Both adapters apply
the deployment-wide private-network policy at socket creation. S3 keeps the
configured hostname for TLS SNI and certificate verification, but does not use
outbound proxies or ambient AWS credential discovery. An incompatible optional
SDK release fails closed before a usable session or client is returned.
## Non-Goals For Files
+3 -2
View File
@@ -34,8 +34,9 @@ Credential profile:
- provider: a specific provider or any provider
- scope: `system` or `tenant` for administered credentials
- credential mode: anonymous, environment reference, secret reference, or
encrypted stored password/token
- credential mode: anonymous, scoped secret reference, or encrypted stored
password/token; environment references are reserved for deployment-owned JSON
profiles and exact deployment allowlisting
- username and redacted secret configuration
- credential-local policy for allow/deny rules
+925
View File
@@ -0,0 +1,925 @@
# GovOPlaN Files Handbook
This handbook describes the Files module as implemented in version `0.1.9`.
It is the operational source of truth for users, process owners, administrators,
operators, auditors, and module integrators. Statements about future behavior
are marked **planned**; an unmarked statement describes the current code.
Files is a governed snapshot store. It owns managed file content, versions,
logical folders, shares, source provenance, and the evidence that another
GovOPlaN module used a particular file version. It can browse selected external
stores read-only and import a frozen copy. It is not a general remote filesystem,
a document collaboration engine, or a records-management system.
## Choose a reading path
| If you need to... | Start with... |
| --- | --- |
| Upload, find, organize, download, or import a file | [User tasks](#user-tasks) |
| Design a governed process that uses files | [Process perspective](#process-perspective) |
| Decide who can use a file or connector | [Ownership and access](#ownership-and-access) and [Administration and policy](#administration-and-policy) |
| Operate storage, connectors, backups, or recovery | [Operator runbook](#operator-runbook) |
| Integrate another GovOPlaN module | [Capabilities and integration](#capabilities-and-integration) |
| Review evidence, deletion, or security behavior | [Security, provenance, audit, deletion, and retention](#security-provenance-audit-deletion-and-retention) |
| Verify a release or scenario | [Acceptance scenarios](#acceptance-scenarios) |
| Check whether an idea exists today | [Implemented and planned boundary](#implemented-and-planned-boundary) |
The Files-owned interface archetypes, consequence classes, disabled-state
wording, and verification evidence are recorded in
[Files Interface Pattern Migration](INTERFACE_PATTERN_MIGRATION.md).
## The service contract
A managed file is a tenant-scoped logical asset with exactly one user or group
owner, a normalized path, and a current version. The current version points to a
blob record containing the storage location, SHA-256 checksum, byte size, and
content type. A protected blob also records its Encryption envelope, protection
discriminator, stored-ciphertext checksum, and stored-ciphertext size. Current
service paths append a version when connector sync finds
changed content; they do not mutate the previous version record.
Unprotected content with the same tenant, plaintext SHA-256 checksum, size, and
protection discriminator can reuse one blob. Protected content is deduplicated
only inside the same vault/profile discriminator; ciphertext is never silently
reused across protection boundaries. Plaintext checksums remain semantic
version evidence, while download and integrity scans verify stored ciphertext
before asking Encryption to open it. Neither digest proves authorship or source
authenticity.
The main domain objects are:
| Object | Meaning | Lifecycle today |
| --- | --- | --- |
| File asset | The user-facing file identity, owner, logical path, description, metadata, and current version | Created, organized, shared, and soft-deleted |
| File version | A numbered snapshot of one asset and its blob | Appended by connector sync when bytes change; retained |
| File blob | Stored plaintext semantics plus stored-byte integrity, backend key, optional Encryption envelope, reference count, and retention timestamp | Reused only within a tenant and matching protection boundary; no automated garbage collection |
| Folder | An explicit logical path in a user or group space | Created, moved/renamed through organize operations, and soft-deleted |
| Share | A grant from an asset to a user, group, tenant, or campaign with `read`, `write`, or `manage` permission | Created or updated; no public revocation endpoint yet |
| Connector profile | A governed external endpoint, scope, optional credential link, policy, and descriptive capabilities | Created, updated, disabled, or credential-scrubbed on deletion |
| Connector credential | Reusable authentication material, optionally limited to a provider and scope | Encrypted when database-managed; immediately scrubbed on deletion |
| Connector policy | Allow and deny rules inherited from system through tenant to one leaf scope | Evaluated before configuration and connector use |
| Connector space | A read-only, manually synchronized remote folder/library linked to a user or group space | Created, updated, disabled, and soft-deleted |
| Campaign attachment use | Evidence connecting a campaign job or entry to an exact asset, version, blob, checksum, and stage | Retained for campaign execution evidence |
## User tasks
The Files page is available at `/files`. Actions appear only when the current
principal has the required permission and resource access.
The configured Help Center projects these sections as independently authorized
tasks, so upload, ZIP import, organization, download, sharing, and deletion do
not disappear merely because an unrelated permission is absent. It states the
deployment's actual upload limits. External import appears as a task only when
the actor has the required permissions and at least one actor-visible profile
is eligible for the current fail-closed browse/import path; endpoint, path, and
item policy are still enforced when the operation runs.
### Choose a space
Every file user sees **My files**. Group spaces are added for active groups of
which the user is a member. The space list also exposes every tenant group to a
Files administrator; governed API operations can administer other tenant-owned
Files resources when their owner is specified. Linked connector spaces appear
beside managed spaces when they are active and visible to the user.
A connector space is intentionally read-only. It is a view of an approved
remote location and a starting point for importing or synchronizing selected
files into managed storage; it is not a mounted write-through filesystem.
### Upload files
The managed-space UI supports upload and drag-and-drop. A caller chooses the
user or group owner and destination folder. The normal per-file deployment
limit defaults to 50 MiB through `FILE_UPLOAD_MAX_BYTES`.
When a target path already exists, the operation must use one of these conflict
strategies:
- `reject` stops instead of silently replacing content;
- `rename` selects the next available `copy` name;
- `overwrite` soft-deletes the conflicting asset and creates the new asset;
- an item-specific conflict resolution can also `skip` that item.
An ordinary upload does not append a new version to an existing asset. The
`overwrite` strategy retires the old asset at that path. Version-preserving
updates currently belong to connector sync.
### Preview and unpack archives
The UI previews ZIP, TAR, TAR.GZ, TAR.BZ2, and TAR.XZ before writing managed
files. Users may select individual files or complete folders. The request is
spooled to a bounded temporary file rather than buffered wholly in memory. The
defaults are:
- 250 MiB compressed request;
- 50 MiB per extracted member;
- 2 GiB total expanded data;
- 10,000 declared entries;
- 100:1 maximum expansion ratio;
- a 30-minute preview token bound to the tenant, user, archive digest, and
destination;
- password-protected ZIP support with request-only password handling;
- traversal, duplicate paths, links, devices, and other special entries are
rejected;
- actual bytes read are counted, not only archive header declarations.
The browser retains the selected archive and password until confirmation.
Confirmation reuploads the archive, verifies the token and digest, repeats all
safety checks, and commits only the selected files. No preview archive or
password is retained server-side.
### Organize files and folders
With `files:file:organize`, a user can:
- create logical folders;
- rename one selection directly;
- preview and apply bulk prefix, suffix, or replacement renames;
- move or copy files and folder trees between accessible user/group spaces;
- resolve target conflicts by rejecting, renaming, overwriting, or skipping;
- use drag-and-drop for move operations in the file explorer.
Copies create new assets and versions while reusing the immutable blob bytes.
Moves keep the asset identity and change its owner/path. All source and target
owners are validated; knowing an identifier does not bypass space membership.
Recursive folder deletion is the default. It soft-deletes the selected folder,
its child folders, and files below it. A non-recursive delete fails when the
folder is not empty.
### Find and download files
Files can list by owner and path, use cursor pagination, and consume incremental
changes through a watermark. The UI supports path/name pattern search and
sorting. The pattern API can resolve campaign-style wildcard selections and can
return unmatched files.
A user with download permission and resource access can download one current
version or create a ZIP archive from a selection. Downloads use an attachment
content disposition. The archive is generated in a temporary file and removed
after the response completes.
There is no dedicated content-preview service in the Files API today. File
responses expose content type, size, checksum, and current version metadata.
### Share files
The API can grant or update a share for a user, group, the tenant, or a campaign.
The supported permissions are `read`, `write`, and `manage`. Ownership remains
unchanged. A write operation accepts a `write` or `manage` share; read/download
accepts any of the three.
The current Files page shows campaign linkage but does not offer a general
user/group share editor. The API also has no share-revocation route yet. Treat
share revocation and a complete share-management UI as planned work.
### Browse and import an external file
With a visible connector profile, a user can browse the allowed remote path,
select one file, and import it into an accessible managed space. The imported
asset records the connector, provider, remote identity/path/URL, selected remote
metadata, and source revision when supplied by the provider.
Manual sync looks for an existing managed asset with the same source identity
inside the chosen owner space:
- no match creates a managed asset;
- identical checksum and size updates provenance and returns `unchanged`;
- changed bytes append a version and return `updated`.
Browse, import, and sync never write, rename, or delete the remote source.
Connector administration separates endpoint profiles, reusable credentials,
and inherited policy. Ordinary setup uses typed fields and provider discovery;
provider metadata JSON is available only under advanced compatibility options.
Read-only deployment entries explain where they must be changed, and disabled
actions identify the missing permission, target, input, or running operation.
The contextual help icon opens the configured Help Center topic when Docs is
enabled and the hosted GovOPlaN documentation otherwise.
## Process perspective
### Managed ingestion
The managed-file flow is:
1. Core authenticates the principal and evaluates the operation permission.
2. Files validates the tenant, owner, group membership, or applicable share.
3. Files normalizes the logical path and resolves conflicts explicitly.
4. Upload or connector response limits are enforced before content is retained.
5. Files calculates SHA-256 and stores or reuses a tenant blob.
6. Files creates the asset/version records and optional campaign share.
7. The database transaction commits and emits change-sequence entries.
8. Connector-originated operations also emit their connector audit evidence.
Blob storage is not part of the database transaction. An object may therefore
be left without committed metadata after a process or database failure. The
operator integrity API scans database blobs and the tenant storage prefix in
bounded, resumable phases. It reports orphan objects before any cleanup and
never deletes them as part of a scan.
### Governed connector import
The connector flow separates four concerns:
1. An administrator defines reusable credential material.
2. An administrator defines a scoped endpoint profile that may reference that
credential.
3. System, tenant, and leaf policy sources narrow the allowed profile,
credential, provider, URL, and remote path.
4. A user optionally links an allowed remote root as a user/group connector
space, then browses and imports selected content.
Before each network operation, Files checks profile visibility, connector
policy, endpoint safety, transport support, and response size. A successful
import becomes an independent managed snapshot. Later source changes have no
effect until an explicit sync.
### Campaign attachment evidence
When Campaign uses Files, the integration follows a freeze-before-send model:
1. The campaign refers to managed user/group sources and attachment patterns.
2. Files verifies access and resolves matching managed assets.
3. A prepared campaign snapshot records the exact asset, version, blob,
checksum, size, relative path, and source provenance.
4. Files materializes those bytes for the campaign build without exposing its
database models to Campaign.
5. Campaign job/entry use is recorded and later marked as sent.
Changing the current file after preparation does not change the version already
recorded as campaign evidence. A file response is marked `audit_relevant` once
the asset has a sent campaign attachment-use record.
### Process ownership
| Concern | Owner |
| --- | --- |
| Authentication, tenants, RBAC evaluation, audit service, change sequence, settings, and module lifecycle | GovOPlaN Core |
| Managed assets, blobs, versions, folders, shares, connector baseline, provenance, and campaign attachment evidence | Files |
| Campaign definition, recipient data, message build/send state, and delivery policy | Campaign |
| Remote ACLs, remote source content, and upstream revision semantics | The external provider |
| Storage durability, egress policy, master key, secret environment, backup, and recovery | Deployment operator |
| Collaborative editing, comments, review, locks, and semantic document workflows | A future Documents/workflow/provider module |
## Ownership and access
Files applies both permission checks and resource checks. A broad operation
permission alone does not make another user's file visible.
### Resource access
- A user owns their personal space.
- A group member can use the group's file space.
- A Files administrator can access all Files resources in the active tenant.
- A file share can grant read or write access to a user, group, or the tenant.
- Campaign shares are resolved only in a verified campaign context and do not
become ordinary user shares.
- Folders are owned by a user or group; they are not independently shared.
- Soft-deleted resources are excluded from normal access and listing.
- Tenant identifiers are checked on every managed object lookup.
The `files.access` capability can explain why a principal has access: resource,
owner, administrator scope, or active share. It also explains virtual folders
that exist through child assets even when there is no explicit folder row.
Deleting an organization/access group is vetoed while it owns Files assets,
folders, connector spaces, or is the target of file shares. Reassign or remove
those relationships first.
### Operation permissions
| Permission | Allows |
| --- | --- |
| `files:file:read` | List and inspect accessible files, folders, spaces, and visible connectors |
| `files:file:download` | Download an accessible current version or ZIP archive |
| `files:file:upload` | Upload managed assets and import/sync selected connector files |
| `files:file:organize` | Create folders, rename, move/copy, and manage linked connector spaces |
| `files:file:share` | Create or update file shares |
| `files:file:delete` | Soft-delete accessible writable files and folders |
| `files:file:admin` | Administer all Files spaces and connector settings in the active tenant |
The `file_manager` role template grants all normal file operations except
`files:file:admin`. The `file_viewer` template grants read and download.
System and tenant settings permissions can also authorize the corresponding
connector administration endpoints.
## Administration and policy
### Profiles, credentials, policies, and spaces
Keep these definitions separate:
- a **credential** holds reusable authentication material and may be restricted
to a provider;
- a **profile** holds the endpoint, scope, base path, credential reference,
local policy, and descriptive operation capabilities;
- a **policy** restricts what a scope may configure or use;
- a **connector space** links one approved profile/library/path to one user or
group and always uses manual, read-only synchronization today.
Profile capability values such as `browse`, `import`, and `sync` are stored and
returned, but they are descriptive today. Provider implementation and policy
checks enforce actual availability; do not use the capability list as the sole
security control.
Profiles, credentials, and policies support `system`, `tenant`, `user`,
`group`, and `campaign` scopes. A normal user sees system and active-tenant
profiles plus leaf profiles that match the user, one of their groups, or an
accessible campaign. Disabled definitions are visible only through authorized
administrative reads.
### Policy evaluation
For a leaf scope, the effective source chain is:
```text
system -> tenant -> user | group | campaign
```
Policy fields are:
- connector/profile IDs;
- credential IDs;
- providers;
- external IDs;
- external path prefixes or glob patterns;
- external URL glob patterns.
Rules use `allow` and `deny` objects. Legacy synonyms `allowlist`, `whitelist`,
`denylist`, and `blacklist` are normalized. A matching deny at any source wins.
Every allow field defined by a source must match, so lower sources can narrow an
inherited set. The effective-policy response includes the contributing source
path and applied fields for explanation.
A parent may set `allow_lower_level_limits` for individual fields such as
`allow.providers` or `deny.external_paths`. An explicit `false` prevents a lower
scope from configuring that field. Absence does not lock the field.
Example tenant policy:
```json
{
"policy": {
"allow": {
"providers": ["webdav", "nextcloud"],
"external_urls": ["https://files.example.edu/*"],
"external_paths": ["departments/finance"]
},
"deny": {
"external_paths": ["departments/finance/private"]
},
"allow_lower_level_limits": {
"allow.providers": true,
"deny.external_paths": true
}
}
}
```
Use `POST /api/v1/files/connector-policy/evaluate` for an explainable preflight.
The normal profile browse/import/sync routes perform their own policy checks;
preflight does not replace enforcement.
### Credential rules
Database-created passwords and tokens use Core's Fernet encryption and require
the deployment `MASTER_KEY_B64` outside development/test/local environments.
Secret values, environment variable names, and local CA paths are never returned
in connector profile responses.
API-managed profiles and credentials cannot select process environment
variables, create an external `secret_ref`, or conceal secret-like values in
nested metadata. Deployment-owned JSON/file profiles may use `password_env` or
`token_env` only when the exact environment variable name appears in
`GOVOPLAN_CONNECTOR_SECRET_ENV_ALLOWLIST`.
Custom CA bundles must be absolute existing files in
`GOVOPLAN_CONNECTOR_CA_BUNDLE_ALLOWLIST`. TLS verification can be disabled only
in a development/test runtime.
### Provider status
| Provider | Current status |
| --- | --- |
| Seafile | Read-only native API browse/download-link import and manual sync implemented using the pinned HTTP transport; WebDAV opt-in supported |
| Nextcloud | Read-only WebDAV browse/import/manual sync implemented using the pinned HTTP transport |
| Generic WebDAV | Read-only browse/import/manual sync implemented using the pinned HTTP transport |
| SMB | Read-only browse/import/manual sync implemented through a pinned smbprotocol transport for initial peers, reconnects, aliases, and DFS referral targets |
| S3 connector | Read-only bucket/prefix browse, import, and manual sync implemented through pinned botocore pools covering retries, redirects, endpoint discovery, and provider aliases |
| SharePoint and OneDrive | Provider keys/descriptors reserved; live Microsoft Graph browse/import is planned |
| NFS and local connector | Described as optional future providers; the local managed-storage backend is a different feature |
Provider descriptors are available from
`GET /api/v1/files/connectors/providers`. Use their `implemented`, `installed`,
and support fields for display. An incompatible optional SDK release fails closed
before it can return a usable client or session.
## Operator runbook
### Storage configuration
| Setting | Default | Purpose |
| --- | --- | --- |
| `FILE_STORAGE_BACKEND` | `local` | Selects `local` or `s3` managed blob storage |
| `FILE_STORAGE_LOCAL_ROOT` | `runtime/files` | Primary local write/read root |
| `FILE_STORAGE_LOCAL_FALLBACK_ROOTS` | empty | Comma-separated older read-only roots checked after the primary root |
| `FILE_STORAGE_S3_ENDPOINT_URL` and related `FILE_STORAGE_S3_*` values | deployment-specific | S3-compatible endpoint, region, credentials, and bucket |
| `FILE_STORAGE_S3_DEPLOYMENT_MANAGED` | `false` | Installer-only trust marker for the exact `http://garage:3900` service; never use it for another endpoint |
| `FILE_STORAGE_S3_ENDPOINT_TRUSTED` | `false` | Deployment-owner acknowledgement for one clean HTTPS external S3 origin; never expose this through connector configuration |
| `GOVOPLAN_STATE_PROFILE` | `local` | Selects `local`, one-host `host-shared`, or multi-host `shared` state validation |
| `FILE_UPLOAD_MAX_BYTES` | 50 MiB | Direct-upload and extracted archive-member maximum |
| `FILE_UPLOAD_ZIP_MAX_BYTES` | 250 MiB | Compressed archive request maximum (legacy name retained for compatibility) |
| `FILE_ARCHIVE_MAX_ENTRIES` | 10,000 | Maximum declared archive entries |
| `FILE_ARCHIVE_MAX_EXPANDED_BYTES` | 2 GiB | Maximum expanded archive bytes |
| `FILE_ARCHIVE_MAX_EXPANSION_RATIO` | 100 | Maximum expanded-to-compressed ratio |
| `FILE_ARCHIVE_PREVIEW_TTL_SECONDS` | 1,800 | Lifetime of the sealed archive preview token |
| `MASTER_KEY_B64` | development fallback only | Encrypts database-managed connector secrets |
The local backend is the operational baseline. It resolves every storage key
under the configured root and rejects escape attempts. Fallback roots support a
controlled storage-root migration: new writes go to the primary root while
reads can still find older objects.
The supported installer may provision a deployment-owned Garage service at the
exact `http://garage:3900` endpoint and set
`FILE_STORAGE_S3_DEPLOYMENT_MANAGED=true`. An operator-selected external S3
backend instead requires a clean HTTPS origin and
`FILE_STORAGE_S3_ENDPOINT_TRUSTED=true`. Both are deployment authority, not a
general connector or private-network bypass. Core owns the shared backend
implementation; Files owns metadata and the Files key namespace.
Multiple API replicas require the same durable blob namespace. Separate local
container filesystems will produce incomplete reads. Use `host-shared` with one
durable shared mount only for same-host replicas. Independent hosts require the
`shared` profile with external S3, PostgreSQL, Redis, a stable installation id,
and one immutable module composition.
### Connector egress
Connector access to private networks is a deployment-wide decision:
```text
GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS=true|false
```
Production-like configuration validation requires an explicit value. Public-only
mode rejects any hostname whose DNS answers include a non-public address.
Private-enabled mode still rejects link-local, multicast, unspecified, and
limited-broadcast addresses.
The built-in HTTP transport:
- resolves and validates every connection attempt;
- connects the socket to the exact approved address while retaining the
original hostname for HTTP Host, TLS SNI, and certificate verification;
- does not inherit proxy settings;
- refuses redirects instead of following a new peer implicitly;
- bounds structured responses to 16 MiB and file transfers to 512 MiB by
default.
The S3 SDK adapter applies the same socket rule to every botocore pool selected
for a retry, redirect, discovered endpoint, or virtual-host bucket alias. The
original authority remains in the request and TLS SNI/certificate check. S3
connector clients use no outbound proxy and never discover ambient AWS
credentials: configure both access and secret keys on the governed profile, or
use an anonymous profile for a public source.
The SMB adapter owns a separate connection cache and replaces smbprotocol's TCP
factory process-wide with the stricter pinned socket. Initial peers, reconnects,
server aliases, domain-controller connections, and DFS referral targets therefore
pass the same policy at connection time. Signing is required by default; enable
SMB encryption on the profile where the server supports it.
Override the connector response limits with
`GOVOPLAN_CONNECTOR_MAX_STRUCTURED_RESPONSE_BYTES` and
`GOVOPLAN_CONNECTOR_MAX_FILE_TRANSFER_BYTES`. The smaller applicable limit wins
when an import is also subject to `FILE_UPLOAD_MAX_BYTES`.
Never work around a connector pinning failure by adding a raw IP, disabling TLS,
or enabling private networks. A failure means the peer policy rejected an actual
connection destination or the installed SDK no longer exposes the verified
transport seam. The separately configured platform S3 backend is trusted only by
the deployment owner and is not selectable by a user or connector profile.
### Backup and restore
The database and blob namespace are one logical backup set. A usable backup must
include:
- Files database rows, including asset/version/blob relationships, shares,
connector settings, and campaign attachment-use evidence;
- every object below `FILE_STORAGE_LOCAL_ROOT` and any still-used fallback root;
or the complete S3 bucket/prefix and version/lifecycle evidence for an S3
backend;
- the exact `MASTER_KEY_B64` needed to decrypt retained connector credentials;
- deployment-owned connector profile files, referenced CA bundles, and secret
environment configuration where those definitions are in use.
There is no Files backup/restore API. Use a write quiesce or coordinated
snapshots so database references and objects represent the same recovery point.
The integrity API verifies a restored set, but it does not replace a coordinated
backup.
Operators normally use **Administration > File integrity**. The equivalent API
creates a scan with `POST /api/v1/files/integrity/scans`, then calls
`POST /api/v1/files/integrity/scans/{scan_id}/run` with the scan's current
`expected_revision` until it reports `completed`. Each call advances at most
the persisted batch size, so a stopped operator or worker can resume from the
committed blob/object cursors. Concurrent or stale actions receive `409` before
the storage backend is invoked; reload the scan and inspect the newer state.
Findings distinguish:
- `missing`: metadata references an absent object;
- `size_mismatch` or `checksum_mismatch`: bytes do not match immutable blob
metadata and the blob is quarantined;
- `orphan_object`: an object exists in the tenant Files prefix without a
corresponding blob row.
Missing or corrupt blobs fail closed for ordinary downloads and Campaign
attachment materialization. After restoring the expected bytes, use the finding
`recheck` action with its current `expected_revision`. Orphan cleanup starts
with a dry-run preview and requires separate destructive confirmation. The
confirmation reuses the finding revision from that preview, rechecks that no
database reference exists, remains scoped to the scanned tenant prefix, and is
idempotent. Both applied and dry-run actions emit audit evidence. A shared
reference blocks deletion. Files currently has no legal-hold or hard-purge
model, so retention-controlled objects must not be treated as cleanup
candidates until those controls are implemented.
### Recovery ledger for object effects
Every managed blob creation or integrity repair starts a Core recovery
operation in an independent committed transaction before Files protects or
writes bytes. The operation records tenant/blob identifiers, an opaque object
locator or locator digest, semantic SHA-256/size evidence, the recovery mode,
and a distributed lease fence. It never records file contents, ZIP passwords,
connector credentials, or a newly uploaded filename. New object keys are opaque;
legacy filename-bearing keys remain readable but repair operations record only
their digest and recover through the blob ID.
The Files business transaction then creates or updates the blob, version, and
asset rows. Its actual SQLAlchemy commit or rollback settles every pending
operation:
- commit reloads the blob through an independent session and streams the object
to verify its stored-byte SHA-256 and size before recording success;
- rollback deletes only a newly reserved object after independently proving
that no `FileBlob` references it, then records verified compensation;
- a repaired existing object is forward-completed only when its identity,
envelope, semantic evidence, and stored bytes all match;
- missing or mismatched bytes quarantine a committed blob and leave the
operation `recovery_required`; an unavailable probe remains
`outcome_unknown` rather than becoming an ordinary upload failure.
Applied orphan cleanup has its own forward-recovery operation. The database
reference check and tenant-prefix check happen before deletion; object absence
and the durable finding state are verified afterward. If the caller transaction
rolls back after deletion, Files may forward-complete only that existing
finding after rechecking that the key is still unreferenced.
Archive preview and confirmation use bounded process-local temporary staging.
Staging is not authoritative and is removed on every handled exit; extracted
members enter the same per-blob recovery boundary as direct uploads. A hard
process loss may leave a temporary OS file for normal host temporary-file
cleanup, but cannot make that staging path a managed Files object.
Use the Ops recovery-operation view to inspect `files` operations. Do not retry
a busy or unresolved blob blindly: first verify the FileBlob row, object hash,
integrity state, and any Encryption envelope named by the blob. Hard purge,
legal hold, and two-way remote connector mutation are not implemented yet, so
they cannot claim recovery-ledger adoption; their owning work remains tracked
separately.
After restore:
1. Verify the active tenant and module migration state.
2. Verify the storage backend and roots before allowing writes.
3. Verify that the original master key is available before testing connectors.
4. Download representative files and compare bytes with their recorded SHA-256.
5. Test one authorized and one unauthorized owner/share path.
6. Test a permitted pinned HTTP connector, if connectors are configured.
7. Review audit and change-sequence continuity around the recovery point.
An inconsistent restore should fail closed for missing objects or undecryptable
credentials. Do not repair it by deleting evidence rows without an approved,
audited data-recovery decision.
### Disable, uninstall, and retire
Disabling a module preserves its persistent data. Ordinary uninstall is guarded
while Files tables contain persistent rows.
Destructive retirement is separate and irreversible at the application level.
The installer records a database snapshot, then Files scrubs and audits
remaining encrypted connector material before its database tables are dropped.
Legacy external secret references are detached and identified as non-owned;
Files never calls a provider delete operation for them.
The retirement executor drops database tables but does not delete corresponding
objects from the configured blob backend. Operators must include those objects
in the approved retention/destruction plan, report them with an integrity scan,
and explicitly approve cleanup. Validate the installer snapshot and independent
blob backup before retirement.
### Operational signals
Use these symptoms as routing hints:
| Symptom | Likely boundary |
| --- | --- |
| `Stored object does not exist` | Database/blob restore mismatch, wrong root, or missing shared storage |
| `Stored secret cannot be decrypted` | Wrong or rotated `MASTER_KEY_B64` |
| Private/non-public endpoint blocked | Deployment-wide egress policy is public-only or DNS returned a forbidden answer |
| SDK peer-pinning seam is unavailable | Optional S3/SMB SDK is incompatible; keep access fail-closed and validate the supported dependency range before upgrade |
| Connector response exceeds limit | Remote payload exceeds connector or upload limit |
| Profile is not visible | Scope, disabled state, campaign access, or policy mismatch |
| Group removal is vetoed | The group still owns a file/folder/connector space or is a share target |
## Capabilities and integration
Other modules must integrate through Core contracts, Files capabilities, or the
public HTTP API. They must not import Files ORM models or storage helpers.
### Provided capabilities
| Capability | Purpose |
| --- | --- |
| `files.access` (`0.1.6`) | Explain resource access provenance for managed files, explicit folders, and virtual folders |
| `files.campaign_attachments` (`0.1.6`) | Resolve managed attachment matches, prepare frozen campaign snapshots, annotate built messages, share assets with a campaign, and record/mark exact attachment use |
Files requires Core principal resolution and permission evaluation. Campaign is
an optional dependency; when installed, Files consumes the optional
`campaigns.access` interface to verify campaign existence and access. Missing
optional Campaign support fails explicitly rather than bypassing the check.
### API families
All routes below are under `/api/v1/files`.
| Area | Routes |
| --- | --- |
| Spaces and content | `GET /spaces`, `GET /`, `GET /folders`, `GET /delta` |
| Upload and folders | `POST /upload`, `POST /upload-zip` (compatibility), `POST /archive-preview`, `POST /archive-confirm`, `POST /folders`, `POST /folders/delete` |
| File access | `GET /{file_id}`, `GET /{file_id}/download`, `DELETE /{file_id}`, `POST /bulk-delete` |
| Organization | `POST /bulk-rename`, `POST /transfer`, `POST /archive.zip`, `POST /resolve-patterns` |
| Sharing | `POST /{file_id}/shares`, `POST /bulk-shares` |
| Connector spaces | `GET/POST /connector-spaces`, `PATCH/DELETE /connector-spaces/{space_id}` |
| Connector catalog/discovery | `GET /connectors/providers`, `POST /connectors/discover` |
| Connector profiles | `GET/POST /connectors/profiles`, `GET/PATCH/DELETE /connectors/profiles/{profile_id}` |
| Browse/import/sync | `GET /connectors/profiles/{profile_id}/browse`, `POST /connectors/profiles/{profile_id}/import`, `POST /connectors/profiles/{profile_id}/sync` |
| Credentials | `GET/POST /connectors/credentials`, `GET/PATCH/DELETE /connectors/credentials/{credential_id}` |
| Policy | `GET/PUT /connectors/policies/{scope_type}`, `POST /connector-policy/evaluate` |
| Incremental connector settings | `GET /connectors/settings/delta` |
Consumers should use cursor/watermark contracts instead of assuming an
unbounded complete list. The default full-list page size is 500 and public page
sizes are capped at 1,000.
### Integration invariants
An integrating module should:
- ask Files/Core for access rather than trusting a submitted file ID;
- retain an exact version/blob/checksum at every governed evidence point;
- import external content before using it in a campaign, report, workflow, or
generated document;
- preserve Files provenance when producing a derived managed snapshot;
- treat source provenance as captured context, not cryptographic attestation;
- avoid writing remote providers through the baseline connector layer;
- define a separate capability when it needs behavior beyond managed snapshots;
- tolerate Files being absent when the integration is declared optional.
Future Postbox, Templates/Reports, BI, Documents, DMS, and workflow modules
should keep their own domain state and use Files for governed input/output
snapshots. Long-running provider sync, OAuth, remote mutation, and
provider-specific health belong in connector modules rather than expanding the
Files baseline indiscriminately.
## Security, provenance, audit, deletion, and retention
### Security controls implemented
- Core authentication, tenant scoping, CSRF/API handling, and RBAC protect the
public routes.
- Owner and share checks protect individual resources after operation-scope
checks.
- Logical paths reject traversal and storage keys cannot escape the local root.
- Upload, archive extraction, connector response, and S3 stream code use bounded
reads. Archive previews are sealed and short-lived; ZIP passwords are
request-only.
- Connector HTTP sockets use connection-time DNS/IP validation and pinning,
redirects are refused, and unsafe SDK transports fail before client creation.
- Database-managed connector passwords/tokens are encrypted; responses redact
secrets and deployment references.
- API metadata is recursively checked for secret-like values.
- Downloads use sanitized attachment filenames.
- Plaintext semantic and stored-byte SHA-256/size evidence are recorded for
every protected blob; they are identical for unprotected blobs.
- Upload and archive-confirm APIs can select an Encryption vault. Protected
writes and reads fail closed if the optional Encryption capability is absent.
- Managed-object writes and applied orphan cleanup start lease-fenced Core
recovery operations before their physical effects; terminal success and
compensation require independent database and object checks.
The module does **not** currently provide malware scanning, content disarm and
reconstruction, a file-type allowlist, per-user quota, automatic encryption
policy assignment, client E2EE, or a dedicated preview sandbox. The optional
server-envelope profile protects selected managed blob bytes at rest but remains
server-decryptable. Deployments that require the other controls must supply
them outside Files until explicit module contracts exist.
### Provenance
Connector-originated assets can retain:
- source type;
- connector/profile ID and provider;
- external ID, path, and URL;
- revision and revision label;
- observation/import timestamps when supplied;
- selected provider metadata.
The normalized provenance and source revision are returned in file responses,
carried into campaign attachment matches, and included in connector audit
events. External metadata is provider/user input and is not a digital signature.
### Audit and change evidence
Files records canonical audit events for:
- connector discovery attempts, before the attempted external I/O;
- connector imports and manual syncs;
- download/archive access to connector-originated managed files;
- immediate connector profile and credential deletion/scrubbing;
- credential scrubbing during destructive module retirement.
Connector audit details include the managed asset/version/blob, checksum, size,
operation, source revision, and provenance where applicable. Deletion audit
details name secret/reference kinds but never the secret values.
Assets, folders, shares, profiles, credentials, policies, and connector spaces
also feed Core's incremental change sequence for UI synchronization. A change
entry is not equivalent to a canonical audit event. Ordinary local upload,
rename, share, and soft-delete operations do not yet all emit dedicated Files
audit events.
Campaign attachment-use records provide separate domain evidence for an exact
file version used in campaign preparation and delivery.
### Deletion semantics
The word "delete" has different meanings by object type:
| Object | Current delete behavior |
| --- | --- |
| File asset | Sets `deleted_at`; content, versions, blob references, and campaign evidence remain |
| Folder | Sets `deleted_at`; recursive deletion also soft-deletes descendants |
| Connector space | Sets `deleted_at` and disables the space |
| Connector credential | Immediately disables the tombstone and clears username, encrypted password/token, environment references, legacy external reference, mode, and private metadata; dependent profiles are disabled and detached |
| Connector profile | Immediately disables the tombstone and clears credential links/material/references and private metadata |
| Legacy `secret_ref` | Detached and audited as an unowned external reference; no provider deletion is attempted or claimed |
| Module retirement | Scrubs/audits credential material, then drops Files database tables; blob-backend cleanup is an operator responsibility |
Credential/profile scrubbing and its audit event use the same database
transaction. If audit creation fails, the deletion rolls back. Repeating a
delete against an already scrubbed tombstone does not recreate secret evidence.
### Retention boundary
File versions and blobs are effectively retained indefinitely today. Although a
blob has `ref_count` and `retained_until` fields, no complete retention-policy,
legal-hold, hard-purge, or garbage-collection service enforces them. There is
also no supported user restore endpoint for soft-deleted assets/folders.
Integrity reconciliation is operator-triggered and is not a retention or
automatic garbage-collection policy.
Do not promise erasure, timed retention, legal hold, or self-service recovery
from the current soft-delete behavior. Those require an explicit, auditable
retention/purge design that preserves campaign and other evidence references.
## Acceptance scenarios
These scenarios describe expected behavior at the current boundary.
### Personal and group ownership
Given Alice has `files:file:upload` and belongs to Finance, when she uploads to
her personal or Finance space, then the asset has the selected owner and tenant.
Given Bob is outside Finance and has no share or Files admin permission, the
same asset ID must not make the asset readable to Bob.
### Safe archive import
Given an archive member contains `../../secret.txt`, is a link or special
filesystem object, exceeds 50 MiB, pushes actual expanded bytes above 2 GiB, or
exceeds the 100:1 expansion ratio, preview or confirmation must fail without a
committed managed asset. A password-protected ZIP requires the correct
request-only password. A valid confirmation must match its unexpired preview
token and preserve normalized relative paths below the chosen logical folder.
### Explicit conflicts
Given a target path already exists, `reject` must leave it unchanged, `rename`
must choose a non-conflicting path, and `overwrite` must soft-delete the old
asset before creating the replacement. A per-item `skip` must not create that
item.
### Policy-denied connector
Given the tenant permits WebDAV only below `departments/finance` and denies its
`private` child, a browse/import/sync request below the denied child must return
an explainable policy denial. Directly invoking the import endpoint must not
bypass the same rule.
### Pinned connector transport
Given public-only mode and DNS returns any private address, the connection must
be rejected before a socket opens. Given private mode, every HTTP, S3, and SMB
connection must still use an address from the answer validated for that exact
attempt. Botocore retries, redirects, endpoint discovery, and aliases, plus SMB
reconnects and DFS referrals, must pass through the pinned factories. A changed
or unsupported SDK seam must fail before a usable client/session is returned.
### Imported evidence and sync
Given a permitted WebDAV file is imported, its managed response and audit event
must carry source identity, revision when available, current version, checksum,
and size. Re-syncing identical bytes must return `unchanged`; changed bytes must
create a higher version while preserving the previous version.
### Campaign freeze
Given a campaign snapshot selected version V1, when the managed asset later
advances to V2, the prepared/sent attachment-use evidence must still identify
V1 and its original blob/checksum.
### Immediate credential deletion
Given a database credential contains an encrypted password and is referenced by
two profiles, deleting it must scrub the credential, disable/detach both
profiles, record non-secret audit evidence, and publish connector-setting
changes in one transaction. If audit creation fails, no part of the deletion
may commit.
### Recovery
Given a coordinated database/blob backup and the original master key, restoring
it must allow representative downloads whose bytes match recorded SHA-256
values. A missing blob or wrong key must surface an error rather than silently
returning different content or credentials.
## Implemented and planned boundary
| Area | Implemented now | Planned or explicitly outside the current boundary |
| --- | --- | --- |
| Managed storage | Core local/S3 backend, exact managed-Garage or explicitly trusted HTTPS external S3, state-profile validation, fallback local read roots, tenant blob deduplication, checksums, bounded resumable integrity scans, quarantine, dry-run-first orphan cleanup, and Core-ledger verification/forward recovery | Scheduled scan execution and deployment-specific S3 HA/backup automation |
| Upload | Bounded direct upload, drag-and-drop UI, archive preview/selective extraction, password-protected ZIP support, explicit conflicts, opaque new object keys, and rollback compensation | Malware scanning, quotas, type policy, resumable/chunked upload |
| Organization | Folders, bulk rename preview/apply, move/copy, drag-and-drop, ZIP download, pattern resolution | General file-history UI and user-driven append-version/restore |
| Sharing | User/group/tenant/campaign grants, expiry, idempotent revocation, searchable share-management UI, and campaign linkage display | Richer policy-driven share lifecycles |
| Deletion/retention | Soft-delete assets/folders/spaces; immediate audited connector-secret scrubbing | File restore API, hard purge, retention policy, legal hold, and blob GC ([#38](https://git.add-ideas.de/GovOPlaN/govoplan-files/issues/38)) |
| Connector governance | Scoped profiles/credentials/policies, effective source explanation, separate credentials, linked user/group spaces | Provider-owned external secret lifecycle; API `secret_ref` remains rejected |
| HTTP connectors | Pinned, bounded, no-redirect Seafile and WebDAV/Nextcloud browse/import/manual sync | Background/folder sync, remote mutation, long-running transfer workers |
| SMB and S3 connectors | Provider descriptors, browse/import/manual sync, pinned SDK transports, and redirect/retry/referral transport-contract tests | Live topology smoke evidence, provider-specific OAuth, remote writes, and background indexing remain separate deployment or connector-module concerns |
| Other providers | Reserved SharePoint/OneDrive keys and NFS/local descriptors | Graph/OAuth/provider paging, NFS deployment integration, DMS connectors |
| Connector spaces | User/group link, browse, manual selected-file sync, edit/disable/delete | Background sync, remote writes/deletes, full conflict-reporting jobs |
| Profile capabilities | Stored and displayed | Enforce capability flags as an independent operation gate |
| Audit | Connector discovery/import/sync/access and connector deletion; campaign exact-use evidence | Dedicated canonical audit events for every ordinary Files mutation |
| Preview | File metadata and attachment download | Dedicated safe content-preview service |
| Campaign | Stable capability-based frozen attachments and sent-use evidence | Campaign-specific process state remains in Campaign |
| Collaboration | Governed input/output snapshots | Co-editing, comments, review, presence, locks, and semantic document versions belong to Documents/workflow/provider modules |
## Release and change checklist
Before releasing Files:
1. Keep `pyproject.toml`, root `package.json`, WebUI `package.json`, and the
module manifest version aligned. Version alignment is a release gate.
2. Run the Files Python tests and security/static-analysis gates from the
GovOPlaN meta repository.
3. Build/type-check the WebUI through the Core host application.
4. Verify database migrations on an upgrade copy and a clean database.
5. Exercise an allowed and denied owner/share path.
6. Exercise upload, ZIP bounds, conflict handling, download, and soft deletion.
7. Exercise connector policy explanation and one pinned HTTP provider where
configured; verify the deployment-managed or trusted external S3 backend if
selected, and verify one configured S3 and SMB connector while recording the
actual target topology and private-network policy.
8. Verify credential deletion scrubs dependents and produces audit evidence.
9. Verify a campaign attachment snapshot still identifies its exact version and
checksum after the current file changes.
10. Exercise a committed upload, a rolled-back upload, object tamper detection,
and applied orphan cleanup; inspect their `files` operations in Ops.
11. Update the implemented/planned table whenever a boundary changes.
## Related documents
- [Repository overview](../README.md)
- [Connector ownership boundary](CONNECTOR_BOUNDARY.md)
- [Connector spaces design and implementation history](CONNECTOR_SPACES.md)
- [Document collaboration boundary](DOCUMENT_COLLABORATION_BOUNDARY.md)
Where an older planning section conflicts with current code or this handbook's
implemented/planned table, verify the code and update both documents in the same
reviewable documentation slice.
+23
View File
@@ -0,0 +1,23 @@
# Generated Artifact Store
Files implements Core's optional `files.artifact_store` capability for modules
that generate deterministic output without owning file storage.
The producer supplies bytes, filename, content type, destination folder,
optional idempotency key, and bounded non-secret provenance. Files applies the
actor's `files:file:upload` permission, tenant/user ownership, path rules,
versioning, configured blob backend, and conflict behavior. An idempotency key
is represented as source provenance so an unchanged retry does not create an
unrelated file version.
The shared Files session owns finalization. Before a new managed object is
written, Files commits a lease-fenced Core recovery operation containing only
identifiers and digests. The caller's eventual session commit independently
verifies both `FileBlob` metadata and stored bytes; rollback compensates only an
unreferenced key. Producers must therefore complete the supplied transaction
normally and must not bypass or replace Files session lifecycle handling.
The response contains only file/version identifiers, display path, media type,
size, digest, and storage provenance. Producers must not put credentials,
tokens, or rendered plaintext into metadata. Storing an artifact proves Files
accepted it; it does not prove printing, mailing, or any other external effect.
+55
View File
@@ -0,0 +1,55 @@
# Files Interface Pattern Migration
This inventory records the Files-owned part of the GovOPlaN interface pattern
language. Core owns the shell and shared components; Files owns the composition
and consequences described here.
## Surface inventory
| Surface | Primary task | Archetype | Consequence | Pattern evidence |
| --- | --- | --- | --- | --- |
| `/files` space and folder panes | Browse managed and connected content without losing location | Directory/explorer | Low for navigation; medium for exposing filenames and provenance | Full-height two-pane workspace, bounded panes, stable selection and contextual Help Center link |
| `/files` toolbar and property filters | Find and act on the current selection | Explorer actions and local filtering | Medium for upload, move, copy, share and synchronization; high for delete | Actions remain beside the affected list, disabled controls explain permission/state/selection blockers, destructive work uses `ConfirmDialog` |
| Upload/archive, transfer, rename and connector-import dialogs | Supply and review one bounded change | Adaptive create/edit or guided import | Medium to high because files, paths and external bytes change | Shared `Dialog`, `FileDropZone`, validation, conflict review, unsaved inputs and explicit confirmation |
| File share dialog | Inspect and change access | Review/decision | High because another actor gains access | Shared dialog, access explanation, stable row actions and destructive confirmation |
| System/tenant/group/user connector surfaces | Compare connections, credentials and effective policy | Administration/configuration | High because endpoints, secrets and inherited policy control external access | Shared `ConnectionTree`, adaptive forms, `ActionBlockerHint`, policy provenance and contextual admin help |
| Connection and credential editors | Create or edit one governed endpoint or secret | Adaptive create/edit | High because a saved change may enable remote access | Relevant fields only, typed credential controls, discovery/test, advanced compatibility section, unsaved-change guard and disabled-save reasons |
| Connector policy card | Narrow inherited connector access | Effective-policy editor | High because deny/allow changes affect lower scopes | Typed reference selectors, deny precedence warning, effective source evidence and permission blocker |
| `files.widget.spaces` | See available managed/connected spaces and open Files | Dashboard widget | Low; names and provider state may still be sensitive | Shared loading/alert/status components, bounded item count, permission-filtered contribution |
| Files chooser capability used by another module | Select a managed snapshot without importing Files internals | Directory chooser | Medium because the exact selected version becomes another module's input | Shared dialog/confirmation, capability boundary and exact file/version evidence |
## State and consequence contract
- Loading, errors, success, empty results, access explanations and confirmation
use Core components. Files does not reproduce the application shell.
- A connector that comes from deployment settings remains visible but read-only;
its action explains that bootstrap configuration and a restart are required.
- Missing permission, target, selection, endpoint, or compatible provider is an
explained disabled state. It is not represented only by color or absence.
- Provider metadata JSON is an expert compatibility escape hatch inside the
collapsed shared advanced-options component. Ordinary connector setup uses
typed provider, endpoint, credential, capability and policy controls.
- Endpoint discovery and credential tests are explicit and report their result;
they do not save the draft. Save remains the only committing action.
- Connector/profile disable and managed-file delete remain confirmed actions and
state their immediate effect. Soft deletion must not be described as purge.
- External connector data, paths and credential references are rendered only in
already-authorized administration or explorer contexts. Secret values are
never returned for rendering.
## Accessibility and responsive evidence
Shared `Dialog` owns focus entry, Escape handling and focus return. Form and
toolbar DOM order is the keyboard order; disabled-action tooltips are themselves
focusable and expose the reason. The connector form uses semantic sections and
labels, status is textual as well as colored, and result alerts are announced by
the shared alert component. The explorer collapses to one column below 1050 px;
connector forms and action rows collapse below 760 px while preserving source
order. Long provider choices scroll inside the segmented control rather than
expanding the page.
The focused structural test guards these contracts, optional-module boundaries,
confirmation, contextual help, advanced-only JSON and responsive rules. Core's
TypeScript build, structural localization audit, module-permutation suite and
full-product bundle check provide the integration gates.
+8 -8
View File
@@ -1,6 +1,6 @@
{
"name": "@govoplan/files-webui",
"version": "0.1.8",
"version": "0.1.16",
"private": true,
"type": "module",
"main": "webui/src/index.ts",
@@ -19,14 +19,14 @@
"LICENSE"
],
"peerDependencies": {
"@govoplan/core-webui": "^0.1.8",
"lucide-react": "^1.23.0",
"react": "^19.0.0",
"react-dom": "^19.0.0",
"react-router-dom": "^7.1.1",
"@vitejs/plugin-react": "^4.3.4",
"@vitejs/plugin-react": "^5.2.0",
"vite": "^7.3.6",
"typescript": "^5.7.2",
"vite": "^6.0.6"
"react": ">=19.2.7 <20",
"react-dom": ">=19.2.7 <20",
"react-router": ">=8.3.0 <9",
"lucide-react": "^1.23.0",
"@govoplan/core-webui": "^0.1.16"
},
"peerDependenciesMeta": {
"@govoplan/core-webui": {
+6 -2
View File
@@ -4,19 +4,23 @@ build-backend = "setuptools.build_meta"
[project]
name = "govoplan-files"
version = "0.1.8"
version = "0.1.16"
description = "GovOPlaN files module with backend and WebUI integration."
readme = "README.md"
requires-python = ">=3.12"
license = { file = "LICENSE" }
authors = [{ name = "GovOPlaN" }]
dependencies = [
"govoplan-core>=0.1.8",
"govoplan-core>=0.1.16",
"defusedxml>=0.7,<1",
"pyzipper>=0.3.6,<1",
"python-multipart>=0.0.31,<1",
]
[project.optional-dependencies]
s3 = [
"boto3>=1.34,<2",
]
smb = [
"smbprotocol>=1.13",
]
+91 -2
View File
@@ -8,6 +8,11 @@ from sqlalchemy import or_
from govoplan_core.core.access import AccessDecisionProvenance, PrincipalRef
from govoplan_core.core.files import FileAccessProvider
from govoplan_core.core.files import (
ManagedArtifactRef,
ManagedArtifactStore,
ManagedArtifactWriteRequest,
)
from govoplan_core.core.modules import ModuleContext
from govoplan_core.security.module_permissions import scopes_grant_compatible
from govoplan_files.backend.db.models import FileAsset, FileFolder, FileShare
@@ -17,10 +22,16 @@ from govoplan_files.backend.storage.campaign_attachments import (
managed_match_payloads,
prepared_campaign_snapshot,
public_attachment_summary_payload,
share_assets_with_campaign,
)
from govoplan_files.backend.storage.campaign_usage import record_campaign_attachment_uses_for_jobs
from govoplan_files.backend.storage.files import current_version_and_blob
from govoplan_files.backend.storage.files import (
create_file_asset,
current_version_and_blob,
sync_file_asset_from_source,
)
from govoplan_files.backend.storage.paths import normalize_folder
from govoplan_files.backend.storage.share_state import effective_file_share_clause
VIRTUAL_FOLDER_RESOURCE_PREFIX = "virtual-folder:v1"
@@ -44,6 +55,7 @@ class FilesCampaignCapability:
annotate_built_messages_with_managed_files = staticmethod(annotate_built_messages_with_managed_files)
record_campaign_attachment_uses_for_jobs = staticmethod(record_campaign_attachment_uses_for_jobs)
current_version_and_blob = staticmethod(current_version_and_blob)
share_assets_with_campaign = staticmethod(share_assets_with_campaign)
@staticmethod
def mark_job_attachment_uses_sent(session: Any, job: Any) -> None:
@@ -57,6 +69,83 @@ def campaign_capability(context: ModuleContext) -> FilesCampaignCapability:
return FilesCampaignCapability()
class FilesArtifactStore(ManagedArtifactStore):
def store_artifact(
self,
session: object,
principal: object,
*,
request: ManagedArtifactWriteRequest,
) -> ManagedArtifactRef:
if not hasattr(session, "query") or not hasattr(session, "flush"):
raise TypeError("Files artifact storage requires a SQLAlchemy session.")
if not hasattr(principal, "has") or not principal.has("files:file:upload"):
raise PermissionError("Managed artifact storage requires files:file:upload.")
user = getattr(principal, "user", None)
user_id = str(getattr(user, "id", "") or "")
tenant_id = str(getattr(principal, "tenant_id", "") or "")
if not user_id or not tenant_id:
raise PermissionError("Managed artifact storage requires a tenant user principal.")
metadata = dict(request.metadata)
if request.idempotency_key:
metadata["source_provenance"] = {
"source_type": "generated_artifact",
"connector_id": "files.artifact_store",
"provider": str(metadata.get("producer_module") or "platform"),
"external_id": request.idempotency_key,
"revision": str(metadata.get("output_sha256") or "") or None,
}
stored, _action, _previous_version_id = sync_file_asset_from_source(
session,
tenant_id=tenant_id,
owner_type="user",
owner_id=user_id,
user_id=user_id,
filename=request.filename,
data=request.payload,
metadata=metadata,
folder=request.folder,
content_type=request.content_type,
conflict_strategy="rename",
is_admin=principal.has("files:file:admin"),
)
else:
stored = create_file_asset(
session,
tenant_id=tenant_id,
owner_type="user",
owner_id=user_id,
user_id=user_id,
filename=request.filename,
data=request.payload,
folder=request.folder,
content_type=request.content_type,
description=request.description,
metadata=metadata,
conflict_strategy="rename",
is_admin=principal.has("files:file:admin"),
)
return ManagedArtifactRef(
file_asset_id=stored.asset.id,
file_version_id=stored.version.id,
filename=stored.version.filename_at_upload,
display_path=stored.asset.display_path,
content_type=stored.version.content_type or request.content_type,
size_bytes=stored.version.size_bytes,
sha256=stored.version.checksum_sha256,
provenance={
"module": "files",
"owner_type": stored.asset.owner_type,
"managed": True,
},
)
def artifact_store_capability(context: ModuleContext) -> FilesArtifactStore:
configure_runtime(registry=context.registry, settings=context.settings)
return FilesArtifactStore()
class FilesAccessService(FileAccessProvider):
def explain_resource_provenance(
self,
@@ -94,7 +183,7 @@ class FilesAccessService(FileAccessProvider):
.filter(
FileShare.tenant_id == asset.tenant_id,
FileShare.file_asset_id == asset.id,
FileShare.revoked_at.is_(None),
effective_file_share_clause(),
FileShare.permission.in_(sorted(permission_values)),
or_(
(FileShare.target_type == "user") & (FileShare.target_id == principal.membership_id),
+26 -46
View File
@@ -1,12 +1,18 @@
from __future__ import annotations
from datetime import datetime
from typing import Any
from sqlalchemy import event, inspect
from sqlalchemy import event
from sqlalchemy.orm import Session as OrmSession
from govoplan_core.core.change_sequence import record_change
from govoplan_core.core.sqlalchemy_change_tracking import (
ensure_object_id,
has_attr_changes,
object_state,
operation_for_soft_deletable,
previous_value,
)
from govoplan_files.backend.db.models import FileAsset, FileFolder, FileShare, new_uuid
FILES_MODULE_ID = "files"
@@ -39,7 +45,7 @@ def _record_files_changes(session: OrmSession, _flush_context: object, _instance
def _record_asset_change(session: OrmSession, asset: FileAsset) -> None:
operation = _operation_for_soft_deletable(
operation = operation_for_soft_deletable(
asset,
changed_attrs=(
"owner_type",
@@ -70,7 +76,7 @@ def _record_asset_change(session: OrmSession, asset: FileAsset) -> None:
"owner_type": asset.owner_type,
"owner_id": _owner_id(asset.owner_type, asset.owner_user_id, asset.owner_group_id),
"path": asset.display_path,
"previous_path": _previous_value(asset, "display_path"),
"previous_path": previous_value(asset, "display_path"),
"filename": asset.filename,
"deleted_at": _isoformat(asset.deleted_at),
},
@@ -78,7 +84,7 @@ def _record_asset_change(session: OrmSession, asset: FileAsset) -> None:
def _record_folder_change(session: OrmSession, folder: FileFolder) -> None:
operation = _operation_for_soft_deletable(
operation = operation_for_soft_deletable(
folder,
changed_attrs=("owner_type", "owner_user_id", "owner_group_id", "path", "deleted_at", "metadata_"),
)
@@ -99,19 +105,27 @@ def _record_folder_change(session: OrmSession, folder: FileFolder) -> None:
"owner_type": folder.owner_type,
"owner_id": _owner_id(folder.owner_type, folder.owner_user_id, folder.owner_group_id),
"path": folder.path,
"previous_path": _previous_value(folder, "path"),
"previous_path": previous_value(folder, "path"),
"deleted_at": _isoformat(folder.deleted_at),
},
)
def _record_share_visibility_change(session: OrmSession, share: FileShare) -> None:
state = inspect(share)
state = object_state(share)
if not share.file_asset_id:
return
if not state.pending and not _has_attr_changes(
if not state.pending and not has_attr_changes(
state,
("file_asset_id", "target_type", "target_id", "permission", "revoked_at"),
(
"file_asset_id",
"target_type",
"target_id",
"permission",
"expires_at",
"revoked_at",
"revoked_by_user_id",
),
):
return
_ensure_id(share)
@@ -130,49 +144,15 @@ def _record_share_visibility_change(session: OrmSession, share: FileShare) -> No
"share_target_type": share.target_type,
"share_target_id": share.target_id,
"share_permission": share.permission,
"share_expires_at": _isoformat(share.expires_at),
"share_revoked_at": _isoformat(share.revoked_at),
"share_revoked_by_user_id": share.revoked_by_user_id,
},
)
def _operation_for_soft_deletable(obj: object, *, changed_attrs: tuple[str, ...]) -> str | None:
state = inspect(obj)
if state.pending:
return "created"
if not _has_attr_changes(state, changed_attrs):
return None
if "deleted_at" in state.attrs:
history = state.attrs.deleted_at.history
if history.has_changes():
if any(value is not None for value in history.added):
return "deleted"
if any(value is not None for value in history.deleted):
return "created"
return "updated"
def _has_attr_changes(state: Any, attrs: tuple[str, ...]) -> bool:
return any(name in state.attrs and state.attrs[name].history.has_changes() for name in attrs)
def _previous_value(obj: object, attr_name: str) -> str | None:
state = inspect(obj)
if attr_name not in state.attrs:
return None
history = state.attrs[attr_name].history
if not history.has_changes() or not history.deleted:
return None
value = history.deleted[0]
return str(value) if value is not None else None
def _ensure_id(obj: object) -> str:
resource_id = getattr(obj, "id", None)
if resource_id:
return str(resource_id)
resource_id = new_uuid()
setattr(obj, "id", resource_id)
return resource_id
return ensure_object_id(obj, new_uuid)
def _owner_id(owner_type: str, owner_user_id: str | None, owner_group_id: str | None) -> str | None:
+25 -1
View File
@@ -1 +1,25 @@
from govoplan_files.backend.db.models import *
from govoplan_files.backend.db.models import (
CampaignAttachmentUse,
FileAsset,
FileBlob,
FileConnectorCredential,
FileConnectorPolicy,
FileConnectorProfile,
FileConnectorSpace,
FileFolder,
FileShare,
FileVersion,
)
__all__ = [
"CampaignAttachmentUse",
"FileAsset",
"FileBlob",
"FileConnectorCredential",
"FileConnectorPolicy",
"FileConnectorProfile",
"FileConnectorSpace",
"FileFolder",
"FileShare",
"FileVersion",
]
+66 -1
View File
@@ -16,7 +16,15 @@ def new_uuid() -> str:
class FileBlob(Base, TimestampMixin):
__tablename__ = "file_blobs"
__table_args__ = (UniqueConstraint("tenant_id", "checksum_sha256", "size_bytes", name="uq_file_blobs_tenant_checksum_size"),)
__table_args__ = (
UniqueConstraint(
"tenant_id",
"checksum_sha256",
"size_bytes",
"protection_discriminator",
name="uq_file_blobs_tenant_checksum_size_protection",
),
)
id: Mapped[str] = mapped_column(String(36), primary_key=True, default=new_uuid)
tenant_id: Mapped[str] = mapped_column(String(36), nullable=False, index=True)
@@ -25,9 +33,64 @@ class FileBlob(Base, TimestampMixin):
storage_key: Mapped[str] = mapped_column(String(1000), nullable=False)
checksum_sha256: Mapped[str] = mapped_column(String(64), nullable=False, index=True)
size_bytes: Mapped[int] = mapped_column(Integer, nullable=False)
protection_discriminator: Mapped[str] = mapped_column(String(320), default="plaintext", nullable=False, index=True)
encryption_envelope_id: Mapped[str | None] = mapped_column(String(255), nullable=True, index=True)
storage_checksum_sha256: Mapped[str | None] = mapped_column(String(64), nullable=True)
storage_size_bytes: Mapped[int | None] = mapped_column(Integer, nullable=True)
content_type: Mapped[str | None] = mapped_column(String(255))
ref_count: Mapped[int] = mapped_column(Integer, default=1, nullable=False)
retained_until: Mapped[datetime | None] = mapped_column(DateTime(timezone=True))
integrity_status: Mapped[str] = mapped_column(String(30), default="unchecked", nullable=False, index=True)
integrity_checked_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True, index=True)
integrity_failure: Mapped[str | None] = mapped_column(String(100), nullable=True)
quarantined_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True, index=True)
class FileIntegrityScan(Base, TimestampMixin):
__tablename__ = "file_integrity_scans"
id: Mapped[str] = mapped_column(String(36), primary_key=True, default=new_uuid)
tenant_id: Mapped[str] = mapped_column(String(36), nullable=False, index=True)
storage_backend: Mapped[str] = mapped_column(String(50), nullable=False)
storage_prefix: Mapped[str] = mapped_column(String(1000), nullable=False)
status: Mapped[str] = mapped_column(String(30), default="pending", nullable=False, index=True)
revision: Mapped[int] = mapped_column(Integer, default=1, nullable=False)
phase: Mapped[str] = mapped_column(String(30), default="blobs", nullable=False)
verify_checksums: Mapped[bool] = mapped_column(Boolean, default=True, nullable=False)
batch_size: Mapped[int] = mapped_column(Integer, default=100, nullable=False)
blob_cursor: Mapped[str | None] = mapped_column(String(36), nullable=True)
object_cursor: Mapped[str | None] = mapped_column(String(1000), nullable=True)
scanned_blob_count: Mapped[int] = mapped_column(Integer, default=0, nullable=False)
verified_blob_count: Mapped[int] = mapped_column(Integer, default=0, nullable=False)
quarantined_blob_count: Mapped[int] = mapped_column(Integer, default=0, nullable=False)
scanned_object_count: Mapped[int] = mapped_column(Integer, default=0, nullable=False)
orphan_object_count: Mapped[int] = mapped_column(Integer, default=0, nullable=False)
created_by_user_id: Mapped[str | None] = mapped_column(ForeignKey("access_users.id", ondelete="SET NULL"), nullable=True, index=True)
started_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
completed_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
last_error: Mapped[str | None] = mapped_column(String(255), nullable=True)
class FileIntegrityFinding(Base, TimestampMixin):
__tablename__ = "file_integrity_findings"
__table_args__ = (
Index("ix_file_integrity_findings_scan_state", "scan_id", "state"),
)
id: Mapped[str] = mapped_column(String(36), primary_key=True, default=new_uuid)
scan_id: Mapped[str] = mapped_column(ForeignKey("file_integrity_scans.id", ondelete="CASCADE"), nullable=False, index=True)
tenant_id: Mapped[str] = mapped_column(String(36), nullable=False, index=True)
kind: Mapped[str] = mapped_column(String(40), nullable=False, index=True)
state: Mapped[str] = mapped_column(String(30), default="open", nullable=False, index=True)
revision: Mapped[int] = mapped_column(Integer, default=1, nullable=False)
blob_id: Mapped[str | None] = mapped_column(ForeignKey("file_blobs.id", ondelete="SET NULL"), nullable=True, index=True)
storage_key: Mapped[str] = mapped_column(String(1000), nullable=False)
expected_size_bytes: Mapped[int | None] = mapped_column(Integer, nullable=True)
observed_size_bytes: Mapped[int | None] = mapped_column(Integer, nullable=True)
expected_checksum_sha256: Mapped[str | None] = mapped_column(String(64), nullable=True)
observed_checksum_sha256: Mapped[str | None] = mapped_column(String(64), nullable=True)
resolved_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
resolved_by_user_id: Mapped[str | None] = mapped_column(ForeignKey("access_users.id", ondelete="SET NULL"), nullable=True, index=True)
class FileFolder(Base, TimestampMixin):
@@ -105,7 +168,9 @@ class FileShare(Base, TimestampMixin):
target_id: Mapped[str] = mapped_column(String(36), nullable=False, index=True)
permission: Mapped[str] = mapped_column(String(20), default="read", nullable=False)
created_by_user_id: Mapped[str | None] = mapped_column(ForeignKey("access_users.id", ondelete="SET NULL"), nullable=True, index=True)
expires_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True, index=True)
revoked_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True, index=True)
revoked_by_user_id: Mapped[str | None] = mapped_column(ForeignKey("access_users.id", ondelete="SET NULL"), nullable=True, index=True)
class FileConnectorProfile(Base, TimestampMixin):
+517
View File
@@ -0,0 +1,517 @@
from __future__ import annotations
from sqlalchemy.orm import Session
from govoplan_core.core.campaigns import (
CAPABILITY_CAMPAIGNS_ACCESS,
CampaignAccessProvider,
)
from govoplan_core.core.modules import (
DocumentationCondition,
DocumentationContext,
DocumentationLink,
DocumentationTopic,
)
from govoplan_files.backend.storage.archives import ARCHIVE_UPLOAD_MAX_ENTRIES
from govoplan_files.backend.storage.access import user_group_ids
from govoplan_files.backend.storage.connector_visibility import (
connector_profile_usable_for_import,
visible_connector_profiles_for_actor,
)
_DEFAULT_UPLOAD_MAX_BYTES = 50 * 1024 * 1024
_DEFAULT_ZIP_MAX_BYTES = 250 * 1024 * 1024
_DEFAULT_ARCHIVE_MAX_EXPANDED_BYTES = 2 * 1024 * 1024 * 1024
_DEFAULT_ARCHIVE_MAX_EXPANSION_RATIO = 100
_DEFAULT_ARCHIVE_PREVIEW_TTL_SECONDS = 30 * 60
_FILES_READ_SCOPE = "files:file:read"
_FILES_UPLOAD_SCOPE = "files:file:upload"
def documentation_topics(
context: DocumentationContext,
) -> tuple[DocumentationTopic, ...]:
if context.documentation_type != "user":
return ()
upload_limit = _configured_positive_int(
context.settings,
"file_upload_max_bytes",
default=_DEFAULT_UPLOAD_MAX_BYTES,
)
zip_limit = _configured_positive_int(
context.settings,
"file_upload_zip_max_bytes",
default=_DEFAULT_ZIP_MAX_BYTES,
)
archive_expanded_limit = _configured_positive_int(
context.settings,
"file_archive_max_expanded_bytes",
default=_DEFAULT_ARCHIVE_MAX_EXPANDED_BYTES,
)
archive_entry_limit = _configured_positive_int(
context.settings,
"file_archive_max_entries",
default=ARCHIVE_UPLOAD_MAX_ENTRIES,
)
archive_ratio_limit = _configured_positive_int(
context.settings,
"file_archive_max_expansion_ratio",
default=_DEFAULT_ARCHIVE_MAX_EXPANSION_RATIO,
)
archive_preview_ttl = _configured_positive_int(
context.settings,
"file_archive_preview_ttl_seconds",
default=_DEFAULT_ARCHIVE_PREVIEW_TTL_SECONDS,
)
topics: list[DocumentationTopic] = []
if upload_limit is not None:
topics.append(_upload_topic(upload_limit))
if (
upload_limit is not None
and zip_limit is not None
and archive_expanded_limit is not None
and archive_entry_limit is not None
and archive_ratio_limit is not None
and archive_preview_ttl is not None
):
topics.append(
_archive_topic(
upload_limit,
zip_limit,
archive_expanded_limit,
archive_entry_limit,
archive_ratio_limit,
archive_preview_ttl,
)
)
topics.append(_connector_import_topic(context))
return tuple(topics)
def _upload_topic(max_bytes: int) -> DocumentationTopic:
limit = _format_byte_limit(max_bytes)
return DocumentationTopic(
id="files.workflow.upload-managed-files",
title="Upload managed files",
summary=f"Upload files to a personal or accessible group space; each uploaded file may contain at most {limit}.",
body=(
f"The current deployment accepts at most {limit} for each ordinary upload. "
"A name conflict is never resolved silently: reject stops the upload, rename chooses a copy name, and overwrite retires the old asset before creating a new one."
),
layer="configured",
documentation_types=("user",),
audience=("file_user", "file_manager", "process_participant"),
order=39,
conditions=(
DocumentationCondition(
required_modules=("files",),
required_scopes=(_FILES_READ_SCOPE, _FILES_UPLOAD_SCOPE),
),
),
links=(
DocumentationLink(label="Files", href="/files", kind="runtime"),
DocumentationLink(
label="Files handbook",
href="govoplan-files/docs/FILES_HANDBOOK.md",
kind="repository",
),
),
related_modules=("campaigns",),
unlocks=(
"Users and connected processes can place governed content in managed storage.",
),
source_module_id="files",
metadata={
"kind": "workflow",
"route": "/files",
"screen": "Files",
"help_contexts": ["files.list"],
"prerequisites": [
"You may view and upload managed files.",
"The destination personal or group space grants this account write access; upload permission alone does not grant access to every space.",
],
"steps": [
"Open Files and choose My files or an accessible group space.",
"Open the intended destination folder and choose Upload, or drag files into the file list.",
"For every name conflict, explicitly reject, rename, overwrite, or skip the affected item.",
"Wait for the upload to finish, then open the resulting file details.",
],
"outcome": "Each accepted file is stored as a governed managed asset in the selected space.",
"verification": "Confirm the owner, logical path, size, checksum, and current version in Files.",
"constraints": [
{
"id": "ordinary-upload-size",
"label": "Maximum size per file",
"description": f"The current safe upload limit is {limit} per file.",
"values": [limit],
},
],
"related_topic_ids": [
"files.workflow.upload-and-unpack-zip",
"files.workflow.organize-managed-files",
"files.workflow.find-and-download-files",
],
},
)
def _archive_topic(
max_file_bytes: int,
max_archive_request_bytes: int,
max_expanded_bytes: int,
max_entries: int,
max_expansion_ratio: int,
preview_ttl_seconds: int,
) -> DocumentationTopic:
member_limit = _format_byte_limit(max_file_bytes)
request_limit = _format_byte_limit(max_archive_request_bytes)
expanded_limit = _format_byte_limit(max_expanded_bytes)
preview_minutes = max(1, preview_ttl_seconds // 60)
return DocumentationTopic(
id="files.workflow.upload-and-unpack-zip",
title="Preview and unpack an archive",
summary=(
f"Review and selectively unpack up to {max_entries:,} entries from ZIP or TAR archives before any managed file is created."
),
body=(
f"The deployment accepts ZIP, TAR, TAR.GZ, TAR.BZ2, and TAR.XZ requests up to {request_limit}, limits actual expanded data to {expanded_limit}, "
f"each member to {member_limit}, the archive to {max_entries:,} entries, and expansion to {max_expansion_ratio}:1. "
f"The server-issued preview expires after {preview_minutes} minutes. Password-protected ZIP archives are supported; passwords remain request-only. "
"Unsafe paths and special filesystem entries are rejected. Actual extracted bytes are counted instead of trusting archive headers."
),
layer="configured",
documentation_types=("user",),
audience=("file_user", "file_manager", "process_participant"),
order=40,
conditions=(
DocumentationCondition(
required_modules=("files",),
required_scopes=(_FILES_READ_SCOPE, _FILES_UPLOAD_SCOPE),
),
),
links=(
DocumentationLink(label="Files", href="/files", kind="runtime"),
DocumentationLink(
label="Files handbook",
href="govoplan-files/docs/FILES_HANDBOOK.md",
kind="repository",
),
),
unlocks=(
"A bounded archive can become governed managed files without trusting archive paths or size declarations.",
),
source_module_id="files",
metadata={
"kind": "workflow",
"route": "/files",
"screen": "Files",
"help_contexts": ["files.list"],
"prerequisites": [
"You may view and upload managed files.",
"The archive fits the configured request, expansion, size, and entry limits.",
"The destination personal or group space grants this account write access.",
],
"steps": [
"Open Files and choose the managed destination space and folder.",
"Enable Preview and unpack archive, then choose or drag one supported archive.",
"Review the discovered entries, supply a ZIP password when required, and select the files or folders to import.",
"Resolve every destination conflict explicitly.",
"Confirm the selection, then wait for extraction and finalization to finish before leaving the page.",
],
"outcome": "Accepted archive members are stored as separate governed managed assets below the selected folder.",
"verification": "Confirm the expected member paths and inspect representative file sizes, checksums, and versions.",
"constraints": [
{
"id": "archive-request-and-total",
"label": "Maximum archive request and expanded total",
"description": f"The compressed request is limited to {request_limit}; actual expanded data is limited to {expanded_limit}.",
"values": [request_limit, expanded_limit],
},
{
"id": "archive-member-size",
"label": "Maximum extracted member size",
"description": f"Each extracted file is limited to {member_limit}.",
"values": [member_limit],
},
{
"id": "archive-entry-count",
"label": "Maximum entry count",
"description": f"An archive may contain at most {max_entries:,} declared entries.",
"values": [f"{max_entries:,} entries"],
},
{
"id": "archive-expansion-ratio",
"label": "Maximum expansion ratio",
"description": f"Declared and actual output may not exceed {max_expansion_ratio} times the compressed request size.",
"values": [f"{max_expansion_ratio}:1"],
},
],
"related_topic_ids": [
"files.workflow.upload-managed-files",
"files.workflow.organize-managed-files",
"files.workflow.find-and-download-files",
],
},
)
def _connector_import_topic(context: DocumentationContext) -> DocumentationTopic:
principal = context.principal
if not _has_all_scopes(principal, (_FILES_READ_SCOPE, _FILES_UPLOAD_SCOPE)):
return _connector_import_limitation(
"Connector import is not available to this account because both permission to view Files and permission to upload managed files are required."
)
tenant_id = _safe_text_attribute(principal, "tenant_id")
user_id = _safe_text_attribute(getattr(principal, "user", None), "id")
session = context.session
if not tenant_id or not user_id or not isinstance(session, Session):
return _connector_import_limitation(
"Connector import availability could not be safely evaluated for this request. Try again, or ask a Files administrator to verify an actor-visible connection."
)
try:
member_group_ids = _actor_group_ids(
session,
principal=principal,
tenant_id=tenant_id,
user_id=user_id,
include_admin_groups=False,
)
connector_group_ids = (
_actor_group_ids(
session,
principal=principal,
tenant_id=tenant_id,
user_id=user_id,
include_admin_groups=True,
)
if _has_all_scopes(principal, ("files:file:admin",))
else member_group_ids
)
profiles = visible_connector_profiles_for_actor(
session,
tenant_id=tenant_id,
user_id=user_id,
group_ids=connector_group_ids,
settings=context.settings,
campaign_visible=_campaign_visibility(
context,
session=session,
tenant_id=tenant_id,
user_id=user_id,
group_ids=member_group_ids,
),
include_effective_policy=True,
)
usable_profiles = tuple(
profile
for profile in profiles
if connector_profile_usable_for_import(profile)
)
except Exception:
return _connector_import_limitation(
"Connector import availability could not be safely evaluated for this request. Try again, or ask a Files administrator to verify an actor-visible connection."
)
if not usable_profiles:
return _connector_import_limitation(
"No connection visible to this account is currently eligible to offer an enabled, credential-ready, policy-allowed browse/import path through a pinning-safe provider. Ask a Files administrator to configure or authorize one."
)
return DocumentationTopic(
id="files.workflow.import-managed-snapshot",
title="Import an external file as a governed snapshot",
summary="Browse an authorized connection read-only and import one selected file into managed storage as a frozen, traceable snapshot.",
body=(
"At least one connection visible to this account is currently eligible to offer the governed browse/import path. "
"The selected remote path and item are re-authorized when used, and browse, import, and sync never mutate the remote source. "
"The managed snapshot stays unchanged until an explicit manual sync."
),
layer="configured",
documentation_types=("user",),
audience=("file_user", "campaign_manager", "report_author"),
order=41,
conditions=(
DocumentationCondition(
required_modules=("files",),
required_scopes=(_FILES_READ_SCOPE, _FILES_UPLOAD_SCOPE),
),
),
links=(
DocumentationLink(label="Files", href="/files", kind="runtime"),
DocumentationLink(
label="Files handbook",
href="govoplan-files/docs/FILES_HANDBOOK.md",
kind="repository",
),
),
related_modules=("campaigns",),
unlocks=(
"Campaigns, reports, and workflows can consume a managed snapshot with stable source evidence.",
),
source_module_id="files",
metadata={
"kind": "workflow",
"route": "/files",
"screen": "Files",
"help_contexts": ["files.list", "files.connector-import"],
"prerequisites": [
"You may view and upload managed files.",
"At least one enabled, credential-ready, policy-allowed connection using a pinning-safe provider is visible to this account.",
"The selected remote path and item must pass their operation-time policy checks.",
"The managed destination space grants this account write access.",
],
"steps": [
"Open Files and choose a managed destination space.",
"Choose Sync from connection, select an available connection, and browse to the permitted remote file.",
"Import the selected file and resolve any destination conflict explicitly.",
"Review the managed file's source and current-version details before using it in another task.",
],
"current_configuration": [
"At least one actor-visible connection is eligible to offer a safe browse/import operation; the selected endpoint, path, and item are still checked when used.",
"Every selected remote path and item is re-authorized at operation time.",
],
"outcome": "The external content is a tenant-managed snapshot with a checksum, exact version, and recorded source context.",
"verification": "Reopen the managed file and confirm its recorded source context, source revision when available, checksum, and current version.",
"related_topic_ids": [
"files.governed-connectors-and-provenance",
"files.reference.integrity-recovery-and-fail-closed-transports",
"files.reference.snapshot-provenance-and-capabilities",
],
},
)
def _connector_import_limitation(message: str) -> DocumentationTopic:
return DocumentationTopic(
id="files.connector-import-unavailable",
title="External file import is not currently available",
summary=message,
body=(
f"{message} Files never exposes connection endpoints, storage paths, credential references, or raw connector policies in user documentation."
),
layer="available",
documentation_types=("user",),
order=41,
conditions=(
DocumentationCondition(
required_modules=("files",),
any_scopes=(_FILES_READ_SCOPE, _FILES_UPLOAD_SCOPE),
),
),
links=(DocumentationLink(label="Files", href="/files", kind="runtime"),),
source_module_id="files",
metadata={
"kind": "reference",
"screen": "Files",
"help_contexts": ["files.list", "files.connector-import"],
"limitations": [message],
},
)
def _configured_positive_int(
settings: object | None, name: str, *, default: int
) -> int | None:
raw_value = getattr(settings, name, default) if settings is not None else default
try:
value = int(raw_value)
except (TypeError, ValueError):
return None
return value if value > 0 else None
def _format_byte_limit(value: int) -> str:
units = ((1024**3, "GiB"), (1024**2, "MiB"), (1024, "KiB"))
for divisor, label in units:
if value % divisor == 0:
return f"{value // divisor:,} {label} ({value:,} bytes)"
return f"{value:,} bytes"
def _has_all_scopes(principal: object | None, scopes: tuple[str, ...]) -> bool:
checker = getattr(principal, "has", None)
if callable(checker):
try:
return all(bool(checker(scope)) for scope in scopes)
except Exception:
return False
granted = {str(scope) for scope in getattr(principal, "scopes", ())}
return all(scope in granted for scope in scopes)
def _safe_text_attribute(value: object | None, name: str) -> str:
try:
result = getattr(value, name, "")
except Exception:
return ""
return str(result or "")
def _principal_group_ids(principal: object) -> tuple[str, ...]:
try:
values = getattr(principal, "group_ids", ())
except Exception:
return ()
return tuple(str(value) for value in values if str(value))
def _actor_group_ids(
session: Session,
*,
principal: object,
tenant_id: str,
user_id: str,
include_admin_groups: bool,
) -> tuple[str, ...]:
try:
values = user_group_ids(
session,
tenant_id=tenant_id,
user_id=user_id,
include_admin_groups=include_admin_groups,
)
except Exception:
return _principal_group_ids(principal)
return tuple(str(value) for value in values if str(value))
def _campaign_visibility(
context: DocumentationContext,
*,
session: Session,
tenant_id: str,
user_id: str,
group_ids: tuple[str, ...],
):
principal = context.principal
registry = context.registry
if not _has_all_scopes(principal, ("campaigns:campaign:read",)):
return None
try:
if not registry.has_capability(CAPABILITY_CAMPAIGNS_ACCESS):
return None
capability = registry.require_capability(CAPABILITY_CAMPAIGNS_ACCESS)
except Exception:
return None
if not isinstance(capability, CampaignAccessProvider):
return None
def campaign_visible(campaign_id: str) -> bool:
try:
return capability.campaign_exists(
session, tenant_id=tenant_id, campaign_id=campaign_id
) and capability.can_read_campaign(
session,
tenant_id=tenant_id,
campaign_id=campaign_id,
user_id=user_id,
group_ids=group_ids,
tenant_admin=_has_all_scopes(principal, ("tenant:*",)),
)
except Exception:
return False
return campaign_visible
+782 -17
View File
@@ -1,12 +1,23 @@
from __future__ import annotations
from dataclasses import replace
from pathlib import Path
from sqlalchemy import inspect
from govoplan_core.core.access import CAPABILITY_AUTH_PERMISSION_EVALUATOR, CAPABILITY_AUTH_PRINCIPAL_RESOLVER
from govoplan_core.core.files import CAPABILITY_FILES_ACCESS
from govoplan_core.core.encryption import CAPABILITY_ENCRYPTION_CONTENT_CIPHER
from govoplan_core.core.files import (
CAPABILITY_FILES_ACCESS,
CAPABILITY_FILES_ARTIFACT_STORE,
)
from govoplan_core.core.module_guards import drop_table_retirement_provider, persistent_table_uninstall_guard
from govoplan_core.core.modules import (
DocumentationCondition,
DocumentationLink,
DocumentationTopic,
FrontendModule,
FrontendRoute,
MigrationSpec,
ModuleContext,
ModuleInterfaceProvider,
@@ -16,13 +27,80 @@ from govoplan_core.core.modules import (
PermissionDefinition,
RoleTemplate,
)
from govoplan_core.core.operations import OperationalCheckProviderRegistration
from govoplan_core.core.provider_governance import (
ExternalProviderDeclaration,
ExternalProviderStateProviderRegistration,
ProviderBehaviorDeclaration,
ProviderObjectDeclaration,
declared_module_architecture,
)
from govoplan_core.core.search import SearchSourceProviderRegistration
from govoplan_core.core.views import ViewSurface
from govoplan_core.db.base import Base
from govoplan_files.backend.change_tracking import register_files_change_tracking
from govoplan_files.backend.db import models as file_models # noqa: F401 - populate Files ORM metadata
from govoplan_files.backend.documentation import documentation_topics
from govoplan_files.backend.provider_state import (
REMOTE_STORAGE_PROVIDER_ID,
remote_storage_provider_states,
)
from govoplan_files.backend.search_source import create_files_search_source
register_files_change_tracking()
_files_table_retirement_provider = drop_table_retirement_provider(
file_models.FileBlob,
file_models.FileIntegrityScan,
file_models.FileIntegrityFinding,
file_models.FileFolder,
file_models.FileAsset,
file_models.FileVersion,
file_models.FileShare,
file_models.FileConnectorCredential,
file_models.FileConnectorPolicy,
file_models.FileConnectorProfile,
file_models.FileConnectorSpace,
file_models.CampaignAttachmentUse,
label="Files",
)
def _files_retirement_provider(session: object | None, module_id: str):
plan = _files_table_retirement_provider(session, module_id)
base_executor = plan.destroy_data_executor
if base_executor is None:
return plan
def executor(execute_session: object, execute_module_id: str) -> None:
if not hasattr(execute_session, "get_bind") or not hasattr(execute_session, "query"):
raise RuntimeError("No database session is available for Files credential retirement.")
live_inspector = inspect(execute_session.get_bind())
if any(
live_inspector.has_table(table_name)
for table_name in (
file_models.FileConnectorCredential.__tablename__,
file_models.FileConnectorProfile.__tablename__,
)
):
from govoplan_files.backend.storage.connector_credential_deletion import (
delete_connector_credentials_for_retirement,
)
delete_connector_credentials_for_retirement(execute_session)
base_executor(execute_session, execute_module_id)
return replace(
plan,
destroy_data_warnings=(
*plan.destroy_data_warnings,
"Files-owned encrypted connector credentials are scrubbed and audited immediately before tables are dropped; legacy non-owned external references are detached without claiming provider-side deletion.",
),
destroy_data_executor=executor,
)
def _permission(scope: str, label: str, description: str) -> PermissionDefinition:
module_id, resource, action = scope.split(":", 2)
return PermissionDefinition(
@@ -42,7 +120,7 @@ PERMISSIONS = (
_permission("files:file:download", "Download files", "Download managed files and generated archives."),
_permission("files:file:upload", "Upload files", "Upload new managed file versions."),
_permission("files:file:organize", "Organize files", "Create folders, rename, move or copy managed files."),
_permission("files:file:share", "Share files", "Grant or revoke managed file shares."),
_permission("files:file:share", "Share files", "List, grant, update, expire, and revoke managed file shares."),
_permission("files:file:delete", "Delete files", "Delete or hide managed files and folders where policy allows it."),
_permission("files:file:admin", "Administer file spaces", "Administer all file spaces in the tenant."),
)
@@ -82,6 +160,43 @@ def _tenant_summary(session, tenant_id: str) -> dict[str, int]:
}
def _tenant_summary_batch(session, tenant_ids) -> dict[str, dict[str, int]]:
from sqlalchemy import func
from govoplan_files.backend.db.models import FileAsset, FileConnectorCredential, FileConnectorPolicy, FileConnectorProfile, FileConnectorSpace
ids = tuple(dict.fromkeys(str(tenant_id) for tenant_id in tenant_ids if tenant_id))
if not ids:
return {}
counts: dict[str, dict[str, int]] = {
tenant_id: {
"files": 0,
"connector_credentials": 0,
"connector_policies": 0,
"connector_profiles": 0,
"connector_spaces": 0,
}
for tenant_id in ids
}
models = (
("files", FileAsset),
("connector_credentials", FileConnectorCredential),
("connector_policies", FileConnectorPolicy),
("connector_profiles", FileConnectorProfile),
("connector_spaces", FileConnectorSpace),
)
for count_key, model in models:
rows = (
session.query(model.tenant_id, func.count(model.id))
.filter(model.tenant_id.in_(ids))
.group_by(model.tenant_id)
.all()
)
for tenant_id, count in rows:
counts[tenant_id][count_key] = int(count)
return counts
def _veto_group_delete(session, tenant_id: str, group_id: str) -> None:
from govoplan_files.backend.db.models import FileAsset, FileConnectorSpace, FileFolder, FileShare
@@ -109,15 +224,68 @@ def _files_router(context: ModuleContext):
return router
REMOTE_STORAGE_PROVIDER = ExternalProviderDeclaration(
id=REMOTE_STORAGE_PROVIDER_ID,
module_id="files",
label="Remote file storage mirror",
maturity="synchronize",
operations=("discover", "search", "read", "synchronize", "preview"),
objects=(
ProviderObjectDeclaration(
object_type="remote_folder",
field_groups=("identity", "hierarchy", "display", "source_metadata"),
authority_modes=("external_authoritative", "external_mirror"),
default_authority_mode="external_mirror",
),
ProviderObjectDeclaration(
object_type="remote_file",
field_groups=("identity", "content", "version", "source_metadata"),
authority_modes=("external_authoritative", "external_mirror"),
default_authority_mode="external_mirror",
),
),
behavior=ProviderBehaviorDeclaration(
revision_tokens="Remote path, provider revision, size, and content digest are retained on managed imports.",
concurrency="A sync compares the frozen source reference and revision before creating a new managed version.",
freshness="Provider state distinguishes software availability from an unobserved live remote binding.",
health="Unsupported providers or missing optional transports fail closed; live health remains unknown without a probe.",
max_read_items=5000,
idempotency="Source profile, remote object reference, and revision/digest suppress duplicate managed versions.",
retry="Operators repeat bounded browse/import after a classified transport failure; effects are not blindly retried.",
timeout_seconds=30,
conflicts="Managed-file conflict policy is explicit; the current provider never mutates the remote source.",
outcome_unknown="An interrupted download is discarded unless its complete digest and managed version commit are confirmed.",
outcome_unknown_supported=True,
evidence="Managed versions retain connector profile, remote object identity, source revision, digest, and acquisition time.",
audit_event_types=(
"files.connector.accessed",
"files.connector.imported",
"files.connector.synced",
),
correction="A later acquisition creates a new managed version and preserves prior provenance.",
rollback="Remote reads require no remote rollback; incomplete local objects are reconciled as orphans.",
compensation="A wrongly imported managed version can be retired under Files policy without deleting the source.",
reconciliation="Re-read source metadata and digest, then compare the committed managed version and object-store inventory.",
outage="Previously imported managed versions remain available while remote spaces report unknown or stale state.",
classifications=("internal", "confidential", "personal"),
purposes=("governed file acquisition", "managed evidence snapshot"),
retention="Files retention applies to managed versions; external retention remains provider-owned.",
secret_handling="Credentials remain encrypted or deployment-owned and never appear in provider state or provenance.",
),
documentation_topic_ids=("files.governed-connectors-and-provenance",),
)
manifest = ModuleManifest(
id="files",
name="Files",
version="0.1.8",
version="0.1.16",
required_capabilities=(CAPABILITY_AUTH_PRINCIPAL_RESOLVER, CAPABILITY_AUTH_PERMISSION_EVALUATOR),
optional_dependencies=("campaigns",),
optional_dependencies=("campaigns", "encryption", "search"),
provides_interfaces=(
ModuleInterfaceProvider(name="files.access", version="0.1.6"),
ModuleInterfaceProvider(name="files.campaign_attachments", version="0.1.6"),
ModuleInterfaceProvider(name=CAPABILITY_FILES_ARTIFACT_STORE, version="0.1.14"),
),
requires_interfaces=(
ModuleInterfaceRequirement(
@@ -126,36 +294,594 @@ manifest = ModuleManifest(
version_max_exclusive="0.2.0",
optional=True,
),
ModuleInterfaceRequirement(
name=CAPABILITY_ENCRYPTION_CONTENT_CIPHER,
version_min="1.0.0",
version_max_exclusive="2.0.0",
optional=True,
),
ModuleInterfaceRequirement(
name="search.source",
version_min="1.0.0",
version_max_exclusive="2.0.0",
optional=True,
),
),
permissions=PERMISSIONS,
route_factory=_files_router,
role_templates=ROLE_TEMPLATES,
tenant_summary_providers=(_tenant_summary,),
tenant_summary_batch_providers=(_tenant_summary_batch,),
search_sources=(
SearchSourceProviderRegistration(
id="files.objects",
factory=create_files_search_source,
),
),
delete_veto_providers={"group": (_veto_group_delete,)},
nav_items=(NavItem(path="/files", label="Files", icon="folder", required_any=("files:file:read",), order=40),),
frontend=FrontendModule(
module_id="files",
package_name="@govoplan/files-webui",
routes=(
FrontendRoute(
path="/files",
component="FilesPage",
required_any=("files:file:read",),
order=40,
),
),
nav_items=(NavItem(path="/files", label="Files", icon="folder", required_any=("files:file:read",), order=40),),
view_surfaces=(
ViewSurface(id="files.admin.system-connectors", module_id="files", kind="section", label="System file connections", order=75),
ViewSurface(id="files.admin.tenant-connectors", module_id="files", kind="section", label="Tenant file connections", order=65),
ViewSurface(id="files.admin.tenant-integrity", module_id="files", kind="section", label="File integrity", order=66),
ViewSurface(id="files.admin.group-connectors", module_id="files", kind="section", label="Group file connections", order=65),
ViewSurface(id="files.admin.user-connectors", module_id="files", kind="section", label="User file connections", order=65),
ViewSurface(id="files.settings.connectors", module_id="files", kind="section", label="Personal file connections", order=20),
ViewSurface(id="files.widget.spaces", module_id="files", kind="section", label="File spaces widget", order=35),
),
),
documentation=(
DocumentationTopic(
id="files.search.managed-content",
title="Search managed files and folders",
summary="Expose file names, logical paths, and descriptions to permission-aware platform Search.",
body=(
"When Search is installed, Files contributes managed files and folders to its derived index. "
"Every result is tenant-bounded and rechecks current ownership, group membership, direct shares, "
"expiry, revocation, deletion, and Files permissions before it is returned. Committed file and "
"share changes are delivered through the platform event outbox; an administrator can rebuild the "
"derived index without changing authoritative Files data."
),
layer="configured",
documentation_types=("admin", "user"),
audience=("file_user", "file_manager", "administrator"),
related_modules=("search",),
order=41,
conditions=(
DocumentationCondition(
required_modules=("files",),
any_scopes=("files:file:read", "files:file:admin"),
),
),
links=(
DocumentationLink(label="Files", href="/files", kind="runtime"),
DocumentationLink(label="Search", href="/search", kind="runtime"),
DocumentationLink(label="Files handbook", href="govoplan-files/docs/FILES_HANDBOOK.md", kind="repository"),
),
metadata={
"kind": "reference",
"route": "/files",
"screen": "Files search contribution",
"help_contexts": ["files.list"],
},
),
DocumentationTopic(
id="files.workflow.organize-managed-files",
title="Organize managed files and folders",
summary="Create folders and rename, move, or copy accessible managed content with explicit conflict handling.",
body=(
"Organization stays inside governed personal or group spaces. Moves preserve the asset identity, while copies create new assets and versions that reuse immutable blob bytes. "
"Every target conflict must be rejected, renamed, overwritten, or skipped explicitly."
),
layer="configured",
documentation_types=("user",),
audience=("file_user", "file_manager", "process_participant"),
order=42,
conditions=(
DocumentationCondition(
required_modules=("files",),
required_scopes=("files:file:read", "files:file:organize"),
),
),
links=(
DocumentationLink(label="Files", href="/files", kind="runtime"),
DocumentationLink(label="Move or copy API", href="/api/v1/files/transfer", kind="api"),
DocumentationLink(label="Files handbook", href="govoplan-files/docs/FILES_HANDBOOK.md", kind="repository"),
),
unlocks=("Managed content can be placed at stable logical paths without bypassing space access.",),
metadata={
"kind": "workflow",
"route": "/files",
"screen": "Files",
"help_contexts": ["files.list"],
"prerequisites": [
"You may view and organize managed files.",
"You have write or owner access to every source item and destination space used by the operation; the global organize permission alone does not grant resource access.",
],
"steps": [
"Open Files and select the personal or group space to organize.",
"Create the required destination folders or select the files and folders to rename, move, or copy.",
"Choose the destination and resolve each target conflict explicitly.",
"Apply the operation and reopen the destination folder.",
],
"outcome": "The selected content has the intended governed owner and logical path.",
"verification": "Confirm each resulting path and owner; for a move, also confirm the old path is gone, and for a copy, confirm the source remains.",
"related_topic_ids": [
"files.workflow.upload-managed-files",
"files.workflow.find-and-download-files",
"files.workflow.delete-managed-files",
],
},
),
DocumentationTopic(
id="files.workflow.find-and-download-files",
title="Find and download managed files",
summary="Search accessible managed content and download a current file version or a ZIP archive of a selection.",
body=(
"Files can be sorted and searched by logical path or name pattern. A download always uses the accessible current version; a multi-file selection can be generated as a temporary ZIP archive. "
"There is no dedicated content-preview service yet."
),
layer="configured",
documentation_types=("user",),
audience=("file_user", "file_manager", "process_participant"),
order=43,
conditions=(
DocumentationCondition(
required_modules=("files",),
required_scopes=("files:file:read", "files:file:download"),
),
),
links=(
DocumentationLink(label="Files", href="/files", kind="runtime"),
DocumentationLink(label="Files handbook", href="govoplan-files/docs/FILES_HANDBOOK.md", kind="repository"),
),
unlocks=("Authorized users can retrieve governed current versions without gaining organization or sharing authority.",),
metadata={
"kind": "workflow",
"route": "/files",
"screen": "Files",
"help_contexts": ["files.list"],
"prerequisites": ["You may view and download managed files in the relevant space."],
"steps": [
"Open Files and select the relevant personal or group space.",
"Navigate folders, sort the list, or use a path/name pattern to find the intended content.",
"Review the displayed owner, path, size, checksum, and version details.",
"Download one current file version, or select several files and choose Download ZIP.",
],
"outcome": "The authorized current file bytes or generated archive are downloaded to the local device.",
"verification": "Confirm the downloaded names and, where integrity matters, compare the file bytes with the displayed checksum.",
"related_topic_ids": [
"files.workflow.organize-managed-files",
"files.reference.snapshot-provenance-and-capabilities",
],
},
),
DocumentationTopic(
id="files.workflow.share-managed-files",
title="Manage access to managed files",
summary="List, grant, update, expire, or revoke direct read, write, and manage access without changing ownership.",
body=(
"File owners and file administrators can manage direct shares for users, groups, the tenant, and Campaign. Expired and revoked grants stop authorizing access immediately while independent active grants remain effective. "
"The Files share dialog lists active and historical grants, and revocation is idempotent."
),
layer="available",
documentation_types=("user",),
audience=("file_manager", "process_participant"),
order=44,
conditions=(
DocumentationCondition(
required_modules=("files",),
required_scopes=("files:file:read", "files:file:share"),
),
),
links=(
DocumentationLink(label="Files", href="/files", kind="runtime"),
DocumentationLink(label="List and manage shares", href="/api/v1/files/{file_id}/shares", kind="api"),
DocumentationLink(label="Files handbook", href="govoplan-files/docs/FILES_HANDBOOK.md", kind="repository"),
),
related_modules=("campaigns",),
unlocks=("A supporting process can grant governed file access without changing file ownership.",),
metadata={
"kind": "workflow",
"route": "/files",
"screen": "Files",
"help_contexts": ["files.list"],
"prerequisites": [
"You have the Files share permission and own the file, or administer file spaces for the active tenant.",
"The intended user, group, tenant, or Campaign target exists and is active.",
],
"steps": [
"Select one managed file, choose Manage shares, and review the effective and historical direct grants.",
"Select the intended user, group, or tenant and choose read, write, or manage access plus an optional expiry.",
"Grant or update the share without changing file ownership.",
"Revoke a grant when it is no longer needed; repeated revocation is a no-op.",
],
"limitations": [
"Campaign-target grants are normally created by the Campaign integration rather than selected manually in the Files dialog.",
"A user may retain access through another active direct grant or ownership path after one share is revoked.",
],
"outcome": "Direct access has the requested permission and lifetime while the managed asset keeps its owner.",
"verification": "Test one intended and one denied path, then expire or revoke the grant and verify that only independent access paths remain.",
"related_topic_ids": [
"files.workflow.find-and-download-files",
"files.assurance.process-and-release-readiness",
],
},
),
DocumentationTopic(
id="files.workflow.delete-managed-files",
title="Delete managed files and folders",
summary="Soft-delete accessible managed files or a folder tree where current policy allows it.",
body=(
"Deletion hides the selected managed assets rather than hard-purging their stored evidence. Folder deletion is recursive by default and includes child folders and files; a non-recursive request fails for a non-empty folder. "
"There is no self-service restore or hard-purge workflow today."
),
layer="configured",
documentation_types=("user",),
audience=("file_user", "file_manager", "process_participant"),
order=45,
conditions=(
DocumentationCondition(
required_modules=("files",),
required_scopes=("files:file:read", "files:file:delete"),
),
),
links=(
DocumentationLink(label="Files", href="/files", kind="runtime"),
DocumentationLink(label="Files handbook", href="govoplan-files/docs/FILES_HANDBOOK.md", kind="repository"),
),
unlocks=("Authorized users can remove obsolete content from active file views while preserving the current soft-delete boundary.",),
metadata={
"kind": "workflow",
"route": "/files",
"screen": "Files",
"help_contexts": ["files.list"],
"prerequisites": [
"You may view and delete the selected managed content and have write or owner access to every affected asset or folder.",
"You have reviewed the complete folder tree when deleting recursively.",
],
"steps": [
"Open Files and select the files or folder to delete.",
"Review the selection and, for a folder, all content below it.",
"Confirm the delete action.",
"Refresh or reopen the space and verify that the selected paths are no longer active.",
],
"limitations": [
"Deletion is soft deletion, not a hard purge.",
"There is no self-service restore or hard-purge workflow.",
],
"outcome": "The selected content is hidden from active Files views under the current soft-delete model.",
"verification": "Confirm the deleted paths no longer appear in the active space; do not treat the action as physical erasure.",
"related_topic_ids": [
"files.workflow.organize-managed-files",
"files.assurance.process-and-release-readiness",
],
},
),
DocumentationTopic(
id="files.governed-connectors-and-provenance",
title="Govern file connections and credential deletion",
summary="Keep endpoint profiles, reusable credentials, and inherited connector policy separate, and understand what DELETE removes immediately.",
body=(
"System, tenant, and one user/group/campaign leaf form the effective policy chain: deny rules win and every configured allow rule must match. "
"Responses redact secret values and deployment references. Deleting a database-managed credential or profile immediately scrubs Files-owned encrypted material and private metadata in the same transaction as a non-secret audit event; dependent profiles are disabled, while legacy non-owned references are only detached and audited."
),
layer="configured",
documentation_types=("admin",),
audience=("file_admin", "tenant_admin", "system_admin", "security_auditor"),
order=50,
conditions=(
DocumentationCondition(
required_modules=("files",),
any_scopes=(
"files:file:admin",
"admin:settings:read",
"admin:settings:write",
"system:settings:read",
"system:settings:write",
),
),
),
links=(
DocumentationLink(label="Tenant connector administration", href="/admin?section=tenant-file-connectors", kind="runtime"),
DocumentationLink(label="System connector administration", href="/admin?section=system-file-connectors", kind="runtime"),
DocumentationLink(label="Tenant connector policy API", href="/api/v1/files/connectors/policies/tenant", kind="api"),
DocumentationLink(label="Connector credentials API", href="/api/v1/files/connectors/credentials", kind="api"),
DocumentationLink(label="Files handbook", href="govoplan-files/docs/FILES_HANDBOOK.md", kind="repository"),
),
related_modules=("access", "audit", "mail"),
unlocks=("Scoped, explainable external-file access without exposing credentials to consuming modules.",),
configuration_keys=(
"GOVOPLAN_FILES_CONNECTOR_PROFILES_JSON",
"GOVOPLAN_FILES_CONNECTOR_PROFILES_FILE",
"GOVOPLAN_CONNECTOR_SECRET_ENV_ALLOWLIST",
"GOVOPLAN_CONNECTOR_CA_BUNDLE_ALLOWLIST",
"MASTER_KEY_B64",
),
metadata={
"kind": "reference",
"route": "/admin?section=tenant-file-connectors",
"screen": "File connections",
"section": "Profiles, credentials, and effective connector policy",
"help_contexts": [
"files.connectors",
"files.connector.credentials",
"files.connector.policy",
],
"security_invariants": [
"New API-managed external secret references fail closed until Files can prove ownership and provider-side deletion.",
"Deletion and destructive retirement scrub Files-owned encrypted connector material before completion and emit non-secret audit evidence.",
"Legacy non-owned external references are detached and audited, never sent to an arbitrary provider delete operation.",
],
"related_topic_ids": [
"files.workflow.import-managed-snapshot",
"files.reference.integrity-recovery-and-fail-closed-transports",
"mail.profiles-and-policy",
],
},
),
DocumentationTopic(
id="files.reference.integrity-recovery-and-fail-closed-transports",
title="Operate Files integrity, recovery, and connector transport safety",
summary="Back up database evidence, blob ciphertext, and Encryption custody as one recovery unit, and pin every SDK-managed connector peer.",
body=(
"Local durable storage is the operational baseline. Recover Files from a coordinated database/blob snapshot with the matching Encryption tables and original deployment master key, then run the bounded resumable integrity scan from Administration and verify representative protected and unprotected access paths. Each scan batch and finding action requires the revision shown to the operator, so a stale screen cannot recheck or delete after concurrent reconciliation. Protected scans verify stored ciphertext before decryption and then verify plaintext semantic evidence. Managed blob creation/repair and applied orphan cleanup commit lease-fenced Core recovery intent before object effects; success, compensation, and forward completion require independent database and object checks, while mismatch is quarantined and unresolved work remains visible in Ops. Missing or mismatched blobs are quarantined; orphan objects are reported before dry-run-first, explicitly authorized cleanup. "
"S3 connector pools pin every retry, redirect, discovered endpoint, and provider alias while retaining the configured TLS authority; outbound proxies and ambient credential discovery are disabled. SMB initial connections, reconnects, aliases, and DFS referrals use a Files-owned pinned transport and cache. Both apply the deployment private-network policy immediately before each socket opens and fail closed if an SDK no longer exposes the verified transport seam. Installer-owned Garage storage is supported only at the exact deployment service endpoint with its explicit trust marker. Destructive module retirement drops database tables but does not remove backend blob objects."
),
layer="configured",
documentation_types=("admin",),
audience=("operator", "system_admin", "security_auditor"),
order=51,
conditions=(
DocumentationCondition(
required_modules=("files",),
any_scopes=(
"files:file:admin",
"admin:settings:read",
"system:settings:read",
"system:audit:read",
),
),
),
links=(
DocumentationLink(label="System file connections", href="/admin?section=system-file-connectors", kind="runtime"),
DocumentationLink(label="File integrity operations", href="/admin?section=tenant-file-integrity", kind="runtime"),
DocumentationLink(label="Connector provider status", href="/api/v1/files/connectors/providers", kind="api"),
DocumentationLink(label="Create an integrity scan", href="/api/v1/files/integrity/scans", kind="api"),
DocumentationLink(label="Files handbook", href="govoplan-files/docs/FILES_HANDBOOK.md", kind="repository"),
),
related_modules=("audit", "encryption", "ops"),
unlocks=("Recoverable managed-file evidence without weakening outbound peer validation.",),
configuration_keys=(
"FILE_STORAGE_BACKEND",
"FILE_STORAGE_LOCAL_ROOT",
"FILE_STORAGE_LOCAL_FALLBACK_ROOTS",
"FILE_STORAGE_S3_DEPLOYMENT_MANAGED",
"FILE_UPLOAD_MAX_BYTES",
"FILE_UPLOAD_ZIP_MAX_BYTES",
"FILE_ARCHIVE_MAX_ENTRIES",
"FILE_ARCHIVE_MAX_EXPANDED_BYTES",
"FILE_ARCHIVE_MAX_EXPANSION_RATIO",
"FILE_ARCHIVE_PREVIEW_TTL_SECONDS",
"MASTER_KEY_B64",
"GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS",
"GOVOPLAN_CONNECTOR_MAX_STRUCTURED_RESPONSE_BYTES",
"GOVOPLAN_CONNECTOR_MAX_FILE_TRANSFER_BYTES",
),
metadata={
"kind": "reference",
"route": "/admin?section=system-file-connectors",
"screen": "System file connections and deployment operations",
"section": "Storage integrity, backup/recovery, and fail-closed transports",
"recovery_unit": ["Files database rows", "Encryption envelope and wrapped-key rows", "managed blob namespace", "MASTER_KEY_B64", "deployment-owned connector configuration"],
"verification": "After restore, complete a checksum-enabled integrity scan, resolve every missing/corrupt finding, approve or retain every reported orphan, inspect Files recovery operations in Ops, verify authorized and denied access, and test configured pinned HTTP, S3, and SMB connectors against their recorded target topology.",
"related_topic_ids": [
"files.governed-connectors-and-provenance",
"files.reference.snapshot-provenance-and-capabilities",
],
},
),
DocumentationTopic(
id="files.reference.snapshot-provenance-and-capabilities",
title="Integrate through managed snapshots and Files capabilities",
summary="Other modules consume stable Files capabilities or HTTP contracts and retain exact version evidence instead of importing Files internals.",
body=(
"Use files.access to explain resource access and files.campaign_attachments to freeze campaign inputs at an exact asset, version, blob, checksum, and source revision. "
"Import external content before a governed use, preserve provenance on derived snapshots, and keep collaboration, provider sync, OAuth, remote mutation, and domain workflow state in their owning modules."
),
layer="available",
documentation_types=("admin", "user"),
audience=("module_integrator", "file_admin", "campaign_admin", "process_designer"),
order=52,
conditions=(
DocumentationCondition(
required_modules=("files",),
any_scopes=("files:file:read", "files:file:upload", "files:file:admin"),
),
),
links=(
DocumentationLink(label="Files", href="/files", kind="runtime"),
DocumentationLink(label="Files API", href="/api/v1/files", kind="api"),
DocumentationLink(label="Files handbook", href="govoplan-files/docs/FILES_HANDBOOK.md", kind="repository"),
),
related_modules=("campaigns", "docs"),
unlocks=("Campaign, report, template, workflow, and document modules can exchange governed input/output snapshots.",),
metadata={
"kind": "reference",
"route": "/files",
"screen": "Files and module integration",
"section": "Managed snapshot provenance and capability boundaries",
"provided_interfaces": ["files.access@0.1.6", "files.campaign_attachments@0.1.6"],
"provenance_fields": ["connector_id", "provider", "external_id", "external_path", "revision", "metadata"],
"related_topic_ids": [
"files.workflow.import-managed-snapshot",
"files.governed-connectors-and-provenance",
"files.reference.integrity-recovery-and-fail-closed-transports",
],
},
),
DocumentationTopic(
id="files.assurance.process-and-release-readiness",
title="Assure a Files-backed process and release",
summary="Check a process against the implemented Files boundary, exercise permitted and denied paths, and retain evidence before approving a release or operational use.",
body=(
"A process owner must distinguish implemented controls from planned capabilities before relying on Files. A release is not ready until all package and manifest versions align and representative authorization, upload limits, conflict handling, download, deletion, connector, and recovery paths have been exercised. "
"Current limitations include no self-service restore or hard purge, no enforced retention or legal hold, and no dedicated canonical audit event for every ordinary Files mutation. Share grant, change, expiry, and revocation operations do emit dedicated audit records. Record remaining limitations in the process assessment instead of treating soft deletion or change-sequence entries as stronger evidence."
),
layer="configured",
documentation_types=("admin", "user"),
audience=("process_owner", "release_manager", "file_admin", "operator", "security_auditor"),
order=53,
conditions=(
DocumentationCondition(
required_modules=("files",),
any_scopes=(
"files:file:read",
"files:file:admin",
"admin:settings:read",
"system:settings:read",
"system:audit:read",
),
),
),
links=(
DocumentationLink(label="Files", href="/files", kind="runtime"),
DocumentationLink(label="Files API", href="/api/v1/files", kind="api"),
DocumentationLink(label="Connector provider status", href="/api/v1/files/connectors/providers", kind="api"),
DocumentationLink(label="Files handbook", href="govoplan-files/docs/FILES_HANDBOOK.md", kind="repository"),
),
related_modules=("campaigns", "audit", "ops"),
unlocks=("Process owners and release managers can approve Files use against explicit controls, evidence, and known gaps.",),
configuration_keys=(
"FILE_STORAGE_BACKEND",
"FILE_STORAGE_LOCAL_ROOT",
"FILE_UPLOAD_MAX_BYTES",
"FILE_UPLOAD_ZIP_MAX_BYTES",
"FILE_ARCHIVE_MAX_ENTRIES",
"FILE_ARCHIVE_MAX_EXPANDED_BYTES",
"FILE_ARCHIVE_MAX_EXPANSION_RATIO",
"FILE_ARCHIVE_PREVIEW_TTL_SECONDS",
"MASTER_KEY_B64",
"GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS",
),
metadata={
"kind": "workflow",
"route": "/files",
"screen": "Files process and release assurance",
"help_contexts": ["files.list"],
"prerequisites": [
"A named process owner has defined the intended users, data classification, retention expectations, and integrations.",
"A candidate release is installed on a clean database and an upgrade copy with aligned Python, root package, WebUI package, and module-manifest versions.",
"Representative permitted and denied accounts, bounded test files, and a coordinated database/blob/key recovery set are available.",
],
"steps": [
"Compare the process requirements with the handbook's implemented/planned boundary and record every unsupported requirement or compensating control.",
"Confirm version alignment, apply migrations on clean and upgrade databases, and run the repository plus meta-repository security and static-analysis gates.",
"Exercise allowed and denied personal, group, and share access with representative accounts.",
"Exercise bounded archive preview and confirmation, every required conflict strategy, organization, download, and soft deletion.",
"Where connectors are configured, verify policy explanation and one pinned HTTP provider; verify installer-owned Garage when selected, then exercise configured S3 retries/aliases and SMB reconnect/referral targets under the deployment private-network policy, confirming an incompatible SDK transport seam fails closed.",
"Restore a coordinated database/blob/key backup and compare representative downloaded bytes with their recorded SHA-256 checksums.",
"Record the tested versions, results, known limitations, evidence locations, residual risks, owner, and approval decision.",
],
"outcome": "The process or release has an explicit approval record tied to aligned versions, representative evidence, known limitations, and owned residual risks.",
"verification": "A reviewer can reproduce the recorded allow/deny, integrity, connector, and recovery checks and can trace each unmet requirement to a documented limitation or accepted compensating control.",
"related_topic_ids": [
"files.workflow.upload-managed-files",
"files.workflow.upload-and-unpack-zip",
"files.workflow.organize-managed-files",
"files.workflow.find-and-download-files",
"files.workflow.share-managed-files",
"files.workflow.delete-managed-files",
"files.workflow.import-managed-snapshot",
"files.governed-connectors-and-provenance",
"files.reference.integrity-recovery-and-fail-closed-transports",
"files.reference.snapshot-provenance-and-capabilities",
],
},
),
DocumentationTopic(
id="files.reference.shared-storage-profile",
title="Operate Files with shared object storage",
summary="Choose local, host-shared, or S3-backed storage consistently with the runtime topology.",
body="Core supplies the common local/S3 object-storage backend while Files owns file metadata and object-key semantics. Local storage is valid for one runtime process; a shared host volume supports same-host replicas; independent hosts require an explicitly trusted HTTPS S3-compatible endpoint. Restore PostgreSQL, objects, and the master key to one coordinated recovery point.",
layer="configured",
documentation_types=("admin",),
audience=("file_admin", "operator", "system_admin"),
order=54,
conditions=(
DocumentationCondition(
required_modules=("files",),
any_scopes=("files:file:admin", "system:settings:read"),
),
),
links=(
DocumentationLink(label="Files", href="/files", kind="runtime"),
DocumentationLink(label="Files handbook", href="govoplan-files/docs/FILES_HANDBOOK.md", kind="repository"),
),
related_modules=("ops", "campaigns"),
configuration_keys=(
"GOVOPLAN_STATE_PROFILE",
"GOVOPLAN_INSTALLATION_ID",
"FILE_STORAGE_BACKEND",
"FILE_STORAGE_LOCAL_ROOT",
"FILE_STORAGE_S3_ENDPOINT_URL",
"FILE_STORAGE_S3_ENDPOINT_TRUSTED",
"FILE_STORAGE_S3_DEPLOYMENT_MANAGED",
),
metadata={
"kind": "reference",
"route": "/ops",
"screen": "Storage and runtime posture",
"limitations": [
"Installer-managed Garage is single-node unless an external multi-node cluster is operated separately.",
"The application does not create or verify production PostgreSQL/object/key backups.",
],
"verification": "Run the Files storage round-trip check and a coordinated restore drill against the exact deployment topology.",
},
),
DocumentationTopic(
id="files.reference.generated-artifact-store",
title="Store generated module artifacts",
summary="Let optional producer modules persist generated output through the Files authority boundary.",
body="The files.artifact_store capability accepts generated bytes plus bounded non-secret provenance, applies Files upload authorization, ownership, path, version, and blob-storage rules, and returns provider-neutral file/version references. Idempotency uses source provenance. Artifact acceptance does not prove printing, mailing, or another external effect.",
layer="available",
documentation_types=("admin", "user"),
audience=("file_admin", "operator", "module_admin", "integrator"),
order=55,
conditions=(
DocumentationCondition(
required_modules=("files",),
any_scopes=("files:file:upload", "files:file:admin"),
),
),
links=(
DocumentationLink(label="Files", href="/files", kind="runtime"),
DocumentationLink(label="Files handbook", href="govoplan-files/docs/FILES_HANDBOOK.md", kind="repository"),
DocumentationLink(label="Generated artifact contract", href="govoplan-files/docs/GENERATED_ARTIFACT_STORE.md", kind="repository"),
),
related_modules=("templates", "campaigns", "reporting"),
metadata={"kind": "reference", "route": "/files"},
),
),
documentation_providers=(documentation_topics,),
migration_spec=MigrationSpec(
module_id="files",
metadata=Base.metadata,
script_location=str(Path(__file__).with_name("migrations") / "versions"),
retirement_supported=True,
retirement_provider=drop_table_retirement_provider(
file_models.FileBlob,
file_models.FileFolder,
file_models.FileAsset,
file_models.FileVersion,
file_models.FileShare,
file_models.FileConnectorCredential,
file_models.FileConnectorPolicy,
file_models.FileConnectorProfile,
file_models.FileConnectorSpace,
file_models.CampaignAttachmentUse,
label="Files",
),
retirement_provider=_files_retirement_provider,
retirement_notes="Destructive retirement drops files-owned database tables after the installer captures a database snapshot.",
),
uninstall_guard_providers=(
@@ -175,8 +901,47 @@ manifest = ModuleManifest(
),
capability_factories={
CAPABILITY_FILES_ACCESS: lambda context: __import__("govoplan_files.backend.capabilities", fromlist=["access_capability"]).access_capability(context),
CAPABILITY_FILES_ARTIFACT_STORE: lambda context: __import__("govoplan_files.backend.capabilities", fromlist=["artifact_store_capability"]).artifact_store_capability(context),
"files.campaign_attachments": lambda context: __import__("govoplan_files.backend.capabilities", fromlist=["campaign_capability"]).campaign_capability(context),
},
operational_check_providers=(
OperationalCheckProviderRegistration(
module_id="files",
check_id="files.managed_storage_roundtrip",
provider=lambda: __import__(
"govoplan_files.backend.operational_checks",
fromlist=["managed_storage_roundtrip_check"],
).managed_storage_roundtrip_check(),
cache_seconds=60,
),
),
external_providers=(REMOTE_STORAGE_PROVIDER,),
external_provider_state_providers=(
ExternalProviderStateProviderRegistration(
module_id="files",
provider_id=REMOTE_STORAGE_PROVIDER_ID,
provider=remote_storage_provider_states,
),
),
architecture=declared_module_architecture(
layer="content_records_evidence",
kind="domain",
maturity="vertical_slice",
documentation_ref="docs/FILES_HANDBOOK.md",
test_ref="tests/test_storage_backends.py",
known_limits=("Target-environment multi-node recovery drills and writable remote connector effects are not reference-ready; hard purge and legal hold remain unimplemented.",),
supported_authority_modes=(
"native_authoritative",
"external_authoritative",
"external_mirror",
),
owned_concepts=("file asset", "file version", "folder", "share", "connector space"),
non_owned_concepts=("record disposition", "campaign attachment rule", "external storage object"),
target_tested_providers=(REMOTE_STORAGE_PROVIDER_ID,),
recovery_docs=("docs/FILES_HANDBOOK.md",),
security_docs=("docs/CONNECTOR_BOUNDARY.md",),
operations_docs=("docs/FILES_HANDBOOK.md",),
),
)
@@ -0,0 +1,53 @@
"""file share lifecycle
Revision ID: b8c9d0e1f2a4
Revises: a7b8c9d0e1f3
Create Date: 2026-07-30 00:00:00.000000
"""
from __future__ import annotations
from alembic import op
import sqlalchemy as sa
revision = "b8c9d0e1f2a4"
down_revision = "a7b8c9d0e1f3"
branch_labels = None
depends_on = None
def upgrade() -> None:
with op.batch_alter_table("file_shares") as batch_op:
batch_op.add_column(
sa.Column("expires_at", sa.DateTime(timezone=True), nullable=True)
)
batch_op.add_column(
sa.Column("revoked_by_user_id", sa.String(length=36), nullable=True)
)
batch_op.create_foreign_key(
op.f("fk_file_shares_revoked_by_user_id_access_users"),
"access_users",
["revoked_by_user_id"],
["id"],
ondelete="SET NULL",
)
batch_op.create_index(
op.f("ix_file_shares_expires_at"), ["expires_at"], unique=False
)
batch_op.create_index(
op.f("ix_file_shares_revoked_by_user_id"),
["revoked_by_user_id"],
unique=False,
)
def downgrade() -> None:
with op.batch_alter_table("file_shares") as batch_op:
batch_op.drop_index(op.f("ix_file_shares_revoked_by_user_id"))
batch_op.drop_index(op.f("ix_file_shares_expires_at"))
batch_op.drop_constraint(
op.f("fk_file_shares_revoked_by_user_id_access_users"),
type_="foreignkey",
)
batch_op.drop_column("revoked_by_user_id")
batch_op.drop_column("expires_at")
@@ -0,0 +1,168 @@
"""file integrity reconciliation
Revision ID: c9d0e1f2a3b5
Revises: b8c9d0e1f2a4
Create Date: 2026-07-30 00:00:00.000000
"""
from __future__ import annotations
from alembic import op
import sqlalchemy as sa
revision = "c9d0e1f2a3b5"
down_revision = "b8c9d0e1f2a4"
branch_labels = None
depends_on = None
def upgrade() -> None:
with op.batch_alter_table("file_blobs") as batch_op:
batch_op.add_column(
sa.Column(
"integrity_status",
sa.String(length=30),
nullable=False,
server_default="unchecked",
)
)
batch_op.add_column(
sa.Column("integrity_checked_at", sa.DateTime(timezone=True), nullable=True)
)
batch_op.add_column(
sa.Column("integrity_failure", sa.String(length=100), nullable=True)
)
batch_op.add_column(
sa.Column("quarantined_at", sa.DateTime(timezone=True), nullable=True)
)
batch_op.create_index(
op.f("ix_file_blobs_integrity_status"),
["integrity_status"],
unique=False,
)
batch_op.create_index(
op.f("ix_file_blobs_integrity_checked_at"),
["integrity_checked_at"],
unique=False,
)
batch_op.create_index(
op.f("ix_file_blobs_quarantined_at"),
["quarantined_at"],
unique=False,
)
op.create_table(
"file_integrity_scans",
sa.Column("id", sa.String(length=36), nullable=False),
sa.Column("tenant_id", sa.String(length=36), nullable=False),
sa.Column("storage_backend", sa.String(length=50), nullable=False),
sa.Column("storage_prefix", sa.String(length=1000), nullable=False),
sa.Column("status", sa.String(length=30), nullable=False),
sa.Column("phase", sa.String(length=30), nullable=False),
sa.Column("verify_checksums", sa.Boolean(), nullable=False),
sa.Column("batch_size", sa.Integer(), nullable=False),
sa.Column("blob_cursor", sa.String(length=36), nullable=True),
sa.Column("object_cursor", sa.String(length=1000), nullable=True),
sa.Column("scanned_blob_count", sa.Integer(), nullable=False),
sa.Column("verified_blob_count", sa.Integer(), nullable=False),
sa.Column("quarantined_blob_count", sa.Integer(), nullable=False),
sa.Column("scanned_object_count", sa.Integer(), nullable=False),
sa.Column("orphan_object_count", sa.Integer(), nullable=False),
sa.Column("created_by_user_id", sa.String(length=36), nullable=True),
sa.Column("started_at", sa.DateTime(timezone=True), nullable=True),
sa.Column("completed_at", sa.DateTime(timezone=True), nullable=True),
sa.Column("last_error", sa.String(length=255), nullable=True),
sa.Column("created_at", sa.DateTime(timezone=True), nullable=False),
sa.Column("updated_at", sa.DateTime(timezone=True), nullable=False),
sa.ForeignKeyConstraint(
["created_by_user_id"],
["access_users.id"],
name=op.f(
"fk_file_integrity_scans_created_by_user_id_access_users"
),
ondelete="SET NULL",
),
sa.PrimaryKeyConstraint("id", name=op.f("pk_file_integrity_scans")),
)
for column in ("tenant_id", "status", "created_by_user_id"):
op.create_index(
op.f(f"ix_file_integrity_scans_{column}"),
"file_integrity_scans",
[column],
unique=False,
)
op.create_table(
"file_integrity_findings",
sa.Column("id", sa.String(length=36), nullable=False),
sa.Column("scan_id", sa.String(length=36), nullable=False),
sa.Column("tenant_id", sa.String(length=36), nullable=False),
sa.Column("kind", sa.String(length=40), nullable=False),
sa.Column("state", sa.String(length=30), nullable=False),
sa.Column("blob_id", sa.String(length=36), nullable=True),
sa.Column("storage_key", sa.String(length=1000), nullable=False),
sa.Column("expected_size_bytes", sa.Integer(), nullable=True),
sa.Column("observed_size_bytes", sa.Integer(), nullable=True),
sa.Column("expected_checksum_sha256", sa.String(length=64), nullable=True),
sa.Column("observed_checksum_sha256", sa.String(length=64), nullable=True),
sa.Column("resolved_at", sa.DateTime(timezone=True), nullable=True),
sa.Column("resolved_by_user_id", sa.String(length=36), nullable=True),
sa.Column("created_at", sa.DateTime(timezone=True), nullable=False),
sa.Column("updated_at", sa.DateTime(timezone=True), nullable=False),
sa.ForeignKeyConstraint(
["blob_id"],
["file_blobs.id"],
name=op.f("fk_file_integrity_findings_blob_id_file_blobs"),
ondelete="SET NULL",
),
sa.ForeignKeyConstraint(
["resolved_by_user_id"],
["access_users.id"],
name=op.f(
"fk_file_integrity_findings_resolved_by_user_id_access_users"
),
ondelete="SET NULL",
),
sa.ForeignKeyConstraint(
["scan_id"],
["file_integrity_scans.id"],
name=op.f(
"fk_file_integrity_findings_scan_id_file_integrity_scans"
),
ondelete="CASCADE",
),
sa.PrimaryKeyConstraint("id", name=op.f("pk_file_integrity_findings")),
)
for column in (
"scan_id",
"tenant_id",
"kind",
"state",
"blob_id",
"resolved_by_user_id",
):
op.create_index(
op.f(f"ix_file_integrity_findings_{column}"),
"file_integrity_findings",
[column],
unique=False,
)
op.create_index(
"ix_file_integrity_findings_scan_state",
"file_integrity_findings",
["scan_id", "state"],
unique=False,
)
def downgrade() -> None:
op.drop_table("file_integrity_findings")
op.drop_table("file_integrity_scans")
with op.batch_alter_table("file_blobs") as batch_op:
batch_op.drop_index(op.f("ix_file_blobs_quarantined_at"))
batch_op.drop_index(op.f("ix_file_blobs_integrity_checked_at"))
batch_op.drop_index(op.f("ix_file_blobs_integrity_status"))
batch_op.drop_column("quarantined_at")
batch_op.drop_column("integrity_failure")
batch_op.drop_column("integrity_checked_at")
batch_op.drop_column("integrity_status")
@@ -0,0 +1,79 @@
"""file content-protection metadata
Revision ID: d0e1f2a3b4c6
Revises: c9d0e1f2a3b5
Create Date: 2026-08-02 00:00:00.000000
"""
from __future__ import annotations
from alembic import op
import sqlalchemy as sa
revision = "d0e1f2a3b4c6"
down_revision = "c9d0e1f2a3b5"
branch_labels = None
depends_on = None
def upgrade() -> None:
with op.batch_alter_table("file_blobs") as batch_op:
batch_op.add_column(
sa.Column(
"protection_discriminator",
sa.String(length=320),
nullable=False,
server_default="plaintext",
)
)
batch_op.add_column(
sa.Column("encryption_envelope_id", sa.String(length=255), nullable=True)
)
batch_op.add_column(
sa.Column("storage_checksum_sha256", sa.String(length=64), nullable=True)
)
batch_op.add_column(
sa.Column("storage_size_bytes", sa.Integer(), nullable=True)
)
batch_op.drop_constraint(
"uq_file_blobs_tenant_checksum_size",
type_="unique",
)
batch_op.create_unique_constraint(
"uq_file_blobs_tenant_checksum_size_protection",
[
"tenant_id",
"checksum_sha256",
"size_bytes",
"protection_discriminator",
],
)
batch_op.create_index(
op.f("ix_file_blobs_protection_discriminator"),
["protection_discriminator"],
unique=False,
)
batch_op.create_index(
op.f("ix_file_blobs_encryption_envelope_id"),
["encryption_envelope_id"],
unique=False,
)
def downgrade() -> None:
with op.batch_alter_table("file_blobs") as batch_op:
batch_op.drop_index(op.f("ix_file_blobs_encryption_envelope_id"))
batch_op.drop_index(op.f("ix_file_blobs_protection_discriminator"))
batch_op.drop_constraint(
"uq_file_blobs_tenant_checksum_size_protection",
type_="unique",
)
batch_op.create_unique_constraint(
"uq_file_blobs_tenant_checksum_size",
["tenant_id", "checksum_sha256", "size_bytes"],
)
batch_op.drop_column("storage_size_bytes")
batch_op.drop_column("storage_checksum_sha256")
batch_op.drop_column("encryption_envelope_id")
batch_op.drop_column("protection_discriminator")
@@ -0,0 +1,34 @@
"""add stale-action revisions to Files integrity operations
Revision ID: f1a2b3c4d5e7
Revises: d0e1f2a3b4c6
"""
from __future__ import annotations
import sqlalchemy as sa
from alembic import op
revision = "f1a2b3c4d5e7"
down_revision = "d0e1f2a3b4c6"
branch_labels = None
depends_on = None
def upgrade() -> None:
with op.batch_alter_table("file_integrity_scans") as batch_op:
batch_op.add_column(
sa.Column("revision", sa.Integer(), nullable=False, server_default="1")
)
with op.batch_alter_table("file_integrity_findings") as batch_op:
batch_op.add_column(
sa.Column("revision", sa.Integer(), nullable=False, server_default="1")
)
def downgrade() -> None:
with op.batch_alter_table("file_integrity_findings") as batch_op:
batch_op.drop_column("revision")
with op.batch_alter_table("file_integrity_scans") as batch_op:
batch_op.drop_column("revision")
@@ -0,0 +1,67 @@
from __future__ import annotations
import hashlib
import secrets
from uuid import uuid4
from govoplan_core.core.operations import OperationalCheck
from govoplan_files.backend.storage.backends import (
StorageBackendError,
get_storage_backend,
)
def managed_storage_roundtrip_check() -> OperationalCheck:
"""Exercise the configured managed store without retaining probe data."""
backend = get_storage_backend()
key = f".govoplan-health/probes/{uuid4().hex}.bin"
payload = secrets.token_bytes(64)
expected_digest = hashlib.sha256(payload).hexdigest()
delete_error: Exception | None = None
try:
backend.put_bytes(key, payload, content_type="application/octet-stream")
stored = backend.get_bytes(key)
info = backend.stat(key)
if info.size_bytes != len(payload):
raise StorageBackendError("Managed storage returned an unexpected object size")
if hashlib.sha256(stored).hexdigest() != expected_digest:
raise StorageBackendError("Managed storage returned different bytes than were written")
except Exception as exc: # noqa: BLE001 - operational boundary reports provider failures.
return OperationalCheck(
id="files.managed_storage_roundtrip",
label="Managed file storage",
state="error",
detail=(
"The configured managed file store failed a bounded write/read/stat/delete "
f"probe ({type(exc).__name__})."
),
readiness_critical=True,
metrics={"backend": backend.name, "probe_bytes": len(payload)},
)
finally:
try:
backend.delete(key)
except Exception as exc: # noqa: BLE001 - reported below when the data probe passed.
delete_error = exc
if delete_error is not None:
return OperationalCheck(
id="files.managed_storage_roundtrip",
label="Managed file storage",
state="error",
detail=(
"Managed file bytes round-tripped, but probe cleanup failed "
f"({type(delete_error).__name__})."
),
readiness_critical=True,
metrics={"backend": backend.name, "probe_bytes": len(payload)},
)
return OperationalCheck(
id="files.managed_storage_roundtrip",
label="Managed file storage",
state="ok",
detail="The configured managed file store passed write, read, stat, integrity, and delete checks.",
metrics={"backend": backend.name, "probe_bytes": len(payload)},
)
@@ -0,0 +1,133 @@
from __future__ import annotations
from collections import defaultdict
from datetime import UTC, datetime
from sqlalchemy import or_, select
from sqlalchemy.orm import Session
from govoplan_core.core.provider_governance import (
ExternalProviderRuntimeState,
ExternalProviderStateContext,
)
from govoplan_files.backend.db.models import FileConnectorProfile, FileConnectorSpace
from govoplan_files.backend.storage.connector_providers import (
ConnectorProviderDescriptor,
connector_provider_descriptors,
)
REMOTE_STORAGE_PROVIDER_ID = "files.remote_storage"
def remote_storage_provider_states(
context: ExternalProviderStateContext,
) -> tuple[ExternalProviderRuntimeState, ...]:
if not isinstance(context.session, Session):
raise RuntimeError("Files provider state requires a database session.")
statement = select(FileConnectorProfile)
if context.tenant_id is not None:
statement = statement.where(
or_(
FileConnectorProfile.tenant_id.is_(None),
FileConnectorProfile.tenant_id == context.tenant_id,
)
)
profiles = tuple(
context.session.scalars(
statement.order_by(
FileConnectorProfile.tenant_id,
FileConnectorProfile.id,
).limit(context.max_items + 1)
)
)
if not profiles:
return ()
profile_ids = tuple(item.id for item in profiles)
space_statement = select(FileConnectorSpace).where(
FileConnectorSpace.connector_profile_id.in_(profile_ids),
FileConnectorSpace.deleted_at.is_(None),
)
if context.tenant_id is not None:
space_statement = space_statement.where(
FileConnectorSpace.tenant_id == context.tenant_id
)
spaces_by_profile: dict[str, list[FileConnectorSpace]] = defaultdict(list)
for space in context.session.scalars(space_statement):
spaces_by_profile[space.connector_profile_id].append(space)
descriptors = {
item.provider: item for item in connector_provider_descriptors()
}
observed_at = datetime.now(UTC)
return tuple(
_profile_state(
profile,
spaces=spaces_by_profile.get(profile.id, []),
descriptor=descriptors.get(profile.provider),
observed_at=observed_at,
)
for profile in profiles
)
def _profile_state(
profile: FileConnectorProfile,
*,
spaces: list[FileConnectorSpace],
descriptor: ConnectorProviderDescriptor | None,
observed_at: datetime,
) -> ExternalProviderRuntimeState:
active_spaces = tuple(item for item in spaces if item.is_active)
active = bool(profile.enabled)
implementation_ready = bool(
descriptor is not None and descriptor.implemented and descriptor.installed
)
health = (
"inactive"
if not active
else "error"
if not implementation_ready
else "unknown"
)
recovery = (
"not_applicable"
if not active
else "unsupported"
if not implementation_ready
else "attention"
)
detail = (
"Remote-storage connector profile is disabled."
if not active
else "The configured provider is not available in this runtime."
if not implementation_ready
else "Software support is available; no live remote health observation is retained."
)
return ExternalProviderRuntimeState(
provider_id=REMOTE_STORAGE_PROVIDER_ID,
binding_ref=f"files:connector-profile:{profile.id}",
authority_mode="external_mirror",
observed_at=observed_at,
configured=True,
active=active,
health=health,
freshness="unknown" if active else "not_applicable",
conflict="not_applicable",
recovery=recovery,
detail=detail,
metrics={
"provider": profile.provider,
"configured_spaces": len(spaces),
"active_spaces": len(active_spaces),
"write_requested_spaces": sum(
1 for item in active_spaces if not item.read_only
),
"software_implemented": bool(descriptor and descriptor.implemented),
"optional_dependency_available": bool(descriptor and descriptor.installed),
},
)
__all__ = ["REMOTE_STORAGE_PROVIDER_ID", "remote_storage_provider_states"]
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
@@ -0,0 +1 @@
"""Focused HTTP route modules for the Files API."""
+134
View File
@@ -0,0 +1,134 @@
from __future__ import annotations
from io import BytesIO
from fastapi import APIRouter, Depends
from fastapi.responses import StreamingResponse
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal, require_scope
from govoplan_files.backend.schemas import (
BulkDeleteRequest,
BulkDeleteResponse,
FileAssetResponse,
)
from govoplan_core.db.session import get_session
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.files import (
get_asset_for_user,
read_asset_bytes,
soft_delete_assets,
)
from govoplan_files.backend.route_support import (
_asset_response,
_attachment_disposition,
_audit_connector_event,
_http_error,
_is_admin,
)
router = APIRouter(prefix="/files", tags=["files"])
@router.get("/{file_id}", response_model=FileAssetResponse)
def get_file(
file_id: str,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:read")),
):
try:
asset = get_asset_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
asset_id=file_id,
is_admin=_is_admin(principal),
)
return _asset_response(session, asset, include_shares=True)
except FileStorageError as exc:
raise _http_error(exc, not_found=True) from exc
@router.get("/{file_id}/download")
def download_file(
file_id: str,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:download")),
):
try:
asset = get_asset_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
asset_id=file_id,
is_admin=_is_admin(principal),
)
data, version, blob = read_asset_bytes(session, asset)
_audit_connector_event(
session,
principal,
action="files.connector.accessed",
asset=asset,
version=version,
blob=blob,
operation="download",
commit=True,
)
except FileStorageError as exc:
raise _http_error(exc, not_found=True) from exc
headers = {"Content-Disposition": _attachment_disposition(asset.filename)}
return StreamingResponse(
BytesIO(data),
media_type=blob.content_type or "application/octet-stream",
headers=headers,
)
@router.delete("/{file_id}", response_model=BulkDeleteResponse)
def delete_file(
file_id: str,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:delete")),
):
try:
asset = get_asset_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
asset_id=file_id,
require_write=True,
is_admin=_is_admin(principal),
)
count = soft_delete_assets(session, [asset])
session.commit()
return BulkDeleteResponse(deleted_count=count)
except FileStorageError as exc:
session.rollback()
raise _http_error(exc, not_found=True) from exc
@router.post("/bulk-delete", response_model=BulkDeleteResponse)
def bulk_delete_files(
payload: BulkDeleteRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:delete")),
):
try:
assets = [
get_asset_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
asset_id=file_id,
require_write=True,
is_admin=_is_admin(principal),
)
for file_id in payload.file_ids
]
count = soft_delete_assets(session, assets)
session.commit()
return BulkDeleteResponse(deleted_count=count)
except FileStorageError as exc:
session.rollback()
raise _http_error(exc) from exc
@@ -0,0 +1,252 @@
from __future__ import annotations
import json
from fastapi import APIRouter, Depends, HTTPException, status
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal, require_any_scope, require_scope
from govoplan_files.backend.schemas import (
FileConnectorBrowseItem,
FileConnectorBrowseResponse,
FileConnectorImportRequest,
FileConnectorSyncResponse,
FileUploadResponse,
)
from govoplan_core.db.session import get_session
from govoplan_files.backend.storage.paths import UnsafeFilePathError
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.connector_browse import (
ConnectorBrowseError,
ConnectorBrowseUnsupported,
browse_connector_profile,
normalize_connector_browse_path,
)
from govoplan_files.backend.storage.connector_imports import (
ConnectorImportError,
ConnectorImportUnsupported,
)
from govoplan_files.backend.storage.connector_deployment import (
connector_effective_endpoint_url,
)
from govoplan_files.backend.storage.connector_policy import (
ConnectorAccessRequest,
ConnectorPolicyDenied,
connector_policy_decision,
)
from govoplan_files.backend.storage.files import (
create_file_asset,
sync_file_asset_from_source,
)
from govoplan_files.backend.route_support import (
_asset_response,
_audit_connector_imports,
_audit_connector_sync,
_connector_browse_next_token,
_connector_policy_error,
_download_connector_payload,
_ensure_campaign_file_access,
_http_error,
_is_admin,
_visible_connector_profile,
)
router = APIRouter(prefix="/files", tags=["files"])
@router.post(
"/connectors/profiles/{profile_id}/import", response_model=FileUploadResponse
)
def import_connector_file(
profile_id: str,
payload: FileConnectorImportRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:upload")),
):
try:
if payload.campaign_id:
_ensure_campaign_file_access(session, principal, payload.campaign_id)
profile = _visible_connector_profile(
session, principal, profile_id, campaign_id=payload.campaign_id
)
_source_path, downloaded, metadata = _download_connector_payload(
profile, payload, operation="import"
)
target_owner = payload.owner_id or principal.user.id
stored = create_file_asset(
session,
tenant_id=principal.tenant_id,
owner_type=payload.owner_type,
owner_id=target_owner,
user_id=principal.user.id,
filename=downloaded.filename,
data=downloaded.data,
folder=payload.target_folder,
display_path=payload.target_path,
content_type=downloaded.content_type,
metadata=metadata,
campaign_id=payload.campaign_id,
conflict_strategy=payload.conflict_strategy,
is_admin=_is_admin(principal),
)
_audit_connector_imports(session, principal, [stored.asset])
session.commit()
except ConnectorPolicyDenied as exc:
session.rollback()
raise _connector_policy_error(exc) from exc
except ConnectorImportUnsupported as exc:
session.rollback()
raise HTTPException(
status_code=status.HTTP_501_NOT_IMPLEMENTED, detail=str(exc)
) from exc
except (
ConnectorImportError,
FileStorageError,
UnsafeFilePathError,
ValueError,
json.JSONDecodeError,
) as exc:
session.rollback()
raise _http_error(exc) from exc
return FileUploadResponse(
files=[_asset_response(session, stored.asset, include_shares=True)]
)
@router.post(
"/connectors/profiles/{profile_id}/sync", response_model=FileConnectorSyncResponse
)
def sync_connector_file(
profile_id: str,
payload: FileConnectorImportRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:upload")),
):
try:
if payload.campaign_id:
_ensure_campaign_file_access(session, principal, payload.campaign_id)
profile = _visible_connector_profile(
session, principal, profile_id, campaign_id=payload.campaign_id
)
_source_path, downloaded, metadata = _download_connector_payload(
profile, payload, operation="sync"
)
target_owner = payload.owner_id or principal.user.id
stored, sync_action, previous_version_id = sync_file_asset_from_source(
session,
tenant_id=principal.tenant_id,
owner_type=payload.owner_type,
owner_id=target_owner,
user_id=principal.user.id,
filename=downloaded.filename,
data=downloaded.data,
folder=payload.target_folder,
display_path=payload.target_path,
content_type=downloaded.content_type,
metadata=metadata,
campaign_id=payload.campaign_id,
conflict_strategy=payload.conflict_strategy,
is_admin=_is_admin(principal),
)
_audit_connector_sync(
session,
principal,
stored.asset,
sync_action=sync_action,
previous_version_id=previous_version_id,
)
session.commit()
except ConnectorPolicyDenied as exc:
session.rollback()
raise _connector_policy_error(exc) from exc
except ConnectorImportUnsupported as exc:
session.rollback()
raise HTTPException(
status_code=status.HTTP_501_NOT_IMPLEMENTED, detail=str(exc)
) from exc
except (
ConnectorImportError,
FileStorageError,
UnsafeFilePathError,
ValueError,
json.JSONDecodeError,
) as exc:
session.rollback()
raise _http_error(exc) from exc
return FileConnectorSyncResponse(
file=_asset_response(session, stored.asset, include_shares=True),
action=sync_action,
previous_version_id=previous_version_id,
current_version_id=stored.version.id,
)
@router.get(
"/connectors/profiles/{profile_id}/browse",
response_model=FileConnectorBrowseResponse,
)
def browse_connector_profile_items(
profile_id: str,
path: str | None = None,
library_id: str | None = None,
continuation_token: str | None = None,
campaign_id: str | None = None,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:read",
"files:file:upload",
"files:file:download",
"files:file:admin",
"system:settings:read",
"admin:settings:read",
)
),
):
try:
profile = _visible_connector_profile(
session, principal, profile_id, campaign_id=campaign_id
)
browse_path = normalize_connector_browse_path(path)
decision = connector_policy_decision(
ConnectorAccessRequest(
connector_id=profile.id,
credential_id=profile.credential_profile_id,
provider=profile.provider,
external_path=browse_path,
external_url=connector_effective_endpoint_url(
provider=profile.provider,
endpoint_url=profile.endpoint_url,
metadata=profile.metadata,
),
operation="browse",
),
profile.policy_sources,
)
if not decision.allowed:
raise ConnectorPolicyDenied(decision)
items = browse_connector_profile(
profile,
path=browse_path,
library_id=library_id,
continuation_token=continuation_token,
)
except ConnectorPolicyDenied as exc:
raise _connector_policy_error(exc) from exc
except ConnectorBrowseUnsupported as exc:
raise HTTPException(
status_code=status.HTTP_501_NOT_IMPLEMENTED, detail=str(exc)
) from exc
except (ConnectorBrowseError, OSError, ValueError, json.JSONDecodeError) as exc:
raise _http_error(exc) from exc
return FileConnectorBrowseResponse(
profile_id=profile.id,
provider=profile.provider,
path=browse_path,
library_id=library_id,
next_continuation_token=_connector_browse_next_token(items),
has_more=any(bool(item.metadata.get("listing_truncated")) for item in items),
decision=decision.to_dict(),
items=[FileConnectorBrowseItem(**item.to_response()) for item in items],
)
@@ -0,0 +1,261 @@
from __future__ import annotations
import json
from fastapi import APIRouter, Depends
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal, require_any_scope
from govoplan_files.backend.change_tracking import (
FILES_CONNECTOR_PROFILES_COLLECTION,
)
from govoplan_files.backend.schemas import (
FileConnectorProfileResponse,
FileConnectorProfileUpdateRequest,
)
from govoplan_core.db.session import get_session
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.connector_credential_deletion import (
delete_connector_profile_row,
)
from govoplan_files.backend.storage.connector_deployment import (
connector_effective_endpoint_url,
)
from govoplan_files.backend.storage.connector_profile_store import (
connector_profile_from_row,
get_connector_profile_row,
update_connector_profile_row,
)
from govoplan_files.backend.storage.connector_policy import (
ConnectorPolicyDenied,
)
from govoplan_files.backend.route_support import (
FILES_CONNECTOR_PROFILE_RESOURCE,
_can_read_disabled_connector_profiles,
_connector_policy_error,
_credential_row_for_profile,
_ensure_connector_configuration_allowed,
_ensure_connector_local_policy_allowed,
_http_error,
_record_connector_settings_change,
_require_connector_profile_write,
_visible_connector_profile,
)
router = APIRouter(prefix="/files", tags=["files"])
@router.get(
"/connectors/profiles/{profile_id}", response_model=FileConnectorProfileResponse
)
def get_connector_profile(
profile_id: str,
campaign_id: str | None = None,
include_disabled: bool = False,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:read",
"files:file:upload",
"files:file:download",
"files:file:admin",
"system:settings:read",
"admin:settings:read",
)
),
):
try:
profile = _visible_connector_profile(
session,
principal,
profile_id,
campaign_id=campaign_id,
include_disabled=include_disabled
and _can_read_disabled_connector_profiles(principal),
include_effective_policy=False,
)
except (OSError, ValueError, json.JSONDecodeError) as exc:
raise _http_error(exc) from exc
return FileConnectorProfileResponse(**profile.to_response())
@router.patch(
"/connectors/profiles/{profile_id}", response_model=FileConnectorProfileResponse
)
def update_connector_profile(
profile_id: str,
payload: FileConnectorProfileUpdateRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:admin", "system:settings:write", "admin:settings:write"
)
),
):
try:
row = get_connector_profile_row(
session,
tenant_id=principal.tenant_id,
profile_id=profile_id,
include_disabled=True,
)
except FileStorageError as exc:
raise _http_error(exc, not_found=True) from exc
_require_connector_profile_write(principal, row.scope_type)
credentials = payload.credentials
try:
credential_profile_id = (
payload.credential_profile_id
if payload.credential_profile_id is not None
else row.credential_profile_id
)
provider = payload.provider if payload.provider is not None else row.provider
credential_row = _credential_row_for_profile(
session,
principal,
credential_profile_id=credential_profile_id,
provider=provider,
profile_id=row.id,
scope_type=row.scope_type,
scope_id=row.scope_id,
include_disabled=True,
)
_ensure_connector_configuration_allowed(
session,
principal,
connector_id=row.id,
credential_id=credential_profile_id,
provider=provider,
endpoint_url=connector_effective_endpoint_url(
provider=provider,
endpoint_url=payload.endpoint_url
if payload.endpoint_url is not None
else row.endpoint_url,
metadata=payload.metadata
if payload.metadata is not None
else row.metadata_,
),
base_path=payload.base_path
if payload.base_path is not None
else row.base_path,
scope_type=row.scope_type,
scope_id=row.scope_id,
operation="configure",
)
if payload.policy is not None:
_ensure_connector_local_policy_allowed(
session,
principal,
scope_type=row.scope_type,
scope_id=row.scope_id,
policy=payload.policy,
)
update_connector_profile_row(
session,
row,
user_id=principal.user.id,
label=payload.label,
provider=payload.provider,
endpoint_url=payload.endpoint_url,
base_path=payload.base_path,
enabled=payload.enabled,
credential_profile_id=payload.credential_profile_id,
credential_mode=payload.credential_mode,
username=credentials.username if credentials else None,
password=credentials.password if credentials else None,
token=credentials.token if credentials else None,
password_env=credentials.password_env if credentials else None,
token_env=credentials.token_env if credentials else None,
secret_ref=credentials.secret_ref if credentials else None,
clear_password=payload.clear_password,
clear_token=payload.clear_token,
capabilities=payload.capabilities,
policy=payload.policy,
metadata=payload.metadata,
)
_record_connector_settings_change(
session,
collection=FILES_CONNECTOR_PROFILES_COLLECTION,
resource_type=FILES_CONNECTOR_PROFILE_RESOURCE,
resource_id=row.id,
operation="updated",
principal=principal,
tenant_id=row.tenant_id,
payload={
"scope_type": row.scope_type,
"scope_id": row.scope_id,
"provider": row.provider,
},
)
session.commit()
session.refresh(row)
return FileConnectorProfileResponse(
**connector_profile_from_row(
row, credential_row=credential_row
).to_response()
)
except ConnectorPolicyDenied as exc:
session.rollback()
raise _connector_policy_error(exc) from exc
except (FileStorageError, ValueError, json.JSONDecodeError) as exc:
session.rollback()
raise _http_error(exc) from exc
@router.delete(
"/connectors/profiles/{profile_id}", response_model=FileConnectorProfileResponse
)
def deactivate_connector_profile(
profile_id: str,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:admin", "system:settings:write", "admin:settings:write"
)
),
):
try:
row = get_connector_profile_row(
session,
tenant_id=principal.tenant_id,
profile_id=profile_id,
include_disabled=True,
)
except FileStorageError as exc:
raise _http_error(exc, not_found=True) from exc
_require_connector_profile_write(principal, row.scope_type)
try:
changed = delete_connector_profile_row(
session,
row,
deletion_reason="api_delete",
user_id=principal.user.id,
api_key_id=principal.api_key.id if principal.api_key else None,
)
if changed:
_record_connector_settings_change(
session,
collection=FILES_CONNECTOR_PROFILES_COLLECTION,
resource_type=FILES_CONNECTOR_PROFILE_RESOURCE,
resource_id=row.id,
operation="deleted",
principal=principal,
tenant_id=row.tenant_id,
payload={
"scope_type": row.scope_type,
"scope_id": row.scope_id,
"provider": row.provider,
},
)
session.commit()
session.refresh(row)
return FileConnectorProfileResponse(
**connector_profile_from_row(row).to_response()
)
except FileStorageError as exc:
session.rollback()
raise _http_error(exc) from exc
except Exception:
session.rollback()
raise
@@ -0,0 +1,852 @@
from __future__ import annotations
import json
from typing import Literal
from fastapi import APIRouter, Depends, Query, status
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal, require_any_scope
from govoplan_files.backend.change_tracking import (
FILES_CONNECTOR_CREDENTIALS_COLLECTION,
FILES_CONNECTOR_POLICIES_COLLECTION,
FILES_CONNECTOR_PROFILES_COLLECTION,
)
from govoplan_files.backend.schemas import (
FileConnectorCredentialCreateRequest,
FileConnectorCredentialResponse,
FileConnectorCredentialsResponse,
FileConnectorCredentialUpdateRequest,
FileConnectorDiscoveryRequest,
FileConnectorDiscoveryResponse,
FileConnectorSettingsDeltaResponse,
FileConnectorPolicyEvaluateRequest,
FileConnectorPolicyEvaluateResponse,
FileConnectorPolicyResponse,
FileConnectorPolicyUpdateRequest,
FileConnectorProfileCreateRequest,
FileConnectorProfileResponse,
FileConnectorProfilesResponse,
FileConnectorProviderResponse,
FileConnectorProvidersResponse,
)
from govoplan_core.db.session import get_session
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.connector_credential_store import (
create_connector_credential_row,
get_connector_credential_row,
list_database_connector_credentials,
update_connector_credential_row,
)
from govoplan_files.backend.storage.connector_credential_deletion import (
delete_connector_credential_row,
)
from govoplan_files.backend.storage.connector_browse import (
ConnectorBrowseError,
ConnectorBrowseUnsupported,
browse_connector_profile,
)
from govoplan_files.backend.storage.connector_deployment import (
connector_effective_endpoint_url,
reject_api_controlled_deployment_references,
)
from govoplan_files.backend.storage.connector_profile_store import (
connector_profile_from_row,
create_connector_profile_row,
)
from govoplan_files.backend.storage.connector_providers import (
connector_provider_descriptors,
)
from govoplan_files.backend.storage.connector_policy import (
ConnectorAccessRequest,
ConnectorPolicyDenied,
connector_policy_decision,
connector_policy_sources_from_payload,
)
from govoplan_files.backend.storage.connector_policy_store import (
connector_policy_response,
set_connector_policy,
)
from govoplan_files.backend.route_support import (
FILES_CONNECTOR_CREDENTIAL_RESOURCE,
FILES_CONNECTOR_POLICY_RESOURCE,
FILES_CONNECTOR_PROFILE_RESOURCE,
_audit_connector_discovery_attempt,
_can_read_disabled_connector_profiles,
_connector_credential_response,
_connector_policy_error,
_credential_row_for_profile,
_discovery_profile_from_payload,
_ensure_campaign_file_access,
_ensure_connector_configuration_allowed,
_ensure_connector_credential_configuration_allowed,
_ensure_connector_local_policy_allowed,
_file_connector_policy_resource_id,
_file_connector_settings_entries,
_http_error,
_record_connector_settings_change,
_require_connector_credential_write,
_require_connector_policy_read,
_require_connector_profile_write,
_same_endpoint,
_visible_connector_profiles,
_webdav_discovery_candidates,
)
from govoplan_files.backend.services.connector_settings_delta import (
_full_file_connector_settings_delta_response,
_incremental_file_connector_settings_delta_response,
)
router = APIRouter(prefix="/files", tags=["files"])
@router.post(
"/connector-policy/evaluate", response_model=FileConnectorPolicyEvaluateResponse
)
def evaluate_connector_policy(
payload: FileConnectorPolicyEvaluateRequest,
principal: ApiPrincipal = Depends(
require_any_scope("files:file:read", "files:file:upload", "files:file:download")
),
):
del principal
sources = connector_policy_sources_from_payload(
[item.model_dump(mode="json") for item in payload.policy_sources]
)
request = ConnectorAccessRequest.from_provenance(
payload.source_provenance.model_dump(mode="json", exclude_none=True),
operation=payload.operation,
)
return FileConnectorPolicyEvaluateResponse(
decision=connector_policy_decision(request, sources).to_dict()
)
@router.get(
"/connectors/settings/delta", response_model=FileConnectorSettingsDeltaResponse
)
def connector_settings_delta(
scope_type: str = Query(default="tenant"),
scope_id: str | None = Query(default=None),
provider: str | None = None,
campaign_id: str | None = None,
include_disabled: bool = False,
include_inactive: bool = False,
owner_type: Literal["user", "group"] | None = None,
owner_id: str | None = None,
since: str | None = None,
limit: int = Query(default=100, ge=1, le=500),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:admin", "system:settings:read", "admin:settings:read"
)
),
):
scope_type = scope_type.strip().casefold()
_require_connector_policy_read(principal, scope_type)
try:
if since is None:
return _full_file_connector_settings_delta_response(
session,
principal,
scope_type=scope_type,
scope_id=scope_id,
provider=provider,
campaign_id=campaign_id,
include_disabled=include_disabled,
include_inactive=include_inactive,
owner_type=owner_type,
owner_id=owner_id,
)
entries, has_more = _file_connector_settings_entries(
session, tenant_id=principal.tenant_id, since=since, limit=limit
)
if entries is None:
return _full_file_connector_settings_delta_response(
session,
principal,
scope_type=scope_type,
scope_id=scope_id,
provider=provider,
campaign_id=campaign_id,
include_disabled=include_disabled,
include_inactive=include_inactive,
owner_type=owner_type,
owner_id=owner_id,
)
return _incremental_file_connector_settings_delta_response(
session,
principal,
entries=entries,
has_more=has_more,
scope_type=scope_type,
scope_id=scope_id,
provider=provider,
campaign_id=campaign_id,
include_disabled=include_disabled,
include_inactive=include_inactive,
owner_type=owner_type,
owner_id=owner_id,
)
except (FileStorageError, ValueError, json.JSONDecodeError) as exc:
raise _http_error(exc) from exc
@router.get("/connectors/providers", response_model=FileConnectorProvidersResponse)
def list_connector_providers(
principal: ApiPrincipal = Depends(
require_any_scope("files:file:read", "files:file:upload", "files:file:admin")
),
):
del principal
return FileConnectorProvidersResponse(
providers=[
FileConnectorProviderResponse(**item.to_response())
for item in connector_provider_descriptors()
]
)
@router.post("/connectors/discover", response_model=FileConnectorDiscoveryResponse)
def discover_connector_endpoint(
payload: FileConnectorDiscoveryRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:admin", "system:settings:write", "admin:settings:write"
)
),
):
try:
reject_api_controlled_deployment_references(
password_env=payload.credentials.password_env,
token_env=payload.credentials.token_env,
secret_ref=payload.credentials.secret_ref,
metadata=payload.metadata,
)
except ValueError as exc:
raise _http_error(exc) from exc
if payload.provider not in {"webdav", "nextcloud"}:
return FileConnectorDiscoveryResponse(
provider=payload.provider,
endpoint_url=None,
base_path=payload.base_path,
status="unsupported",
message=f"Discovery is not implemented for {payload.provider} connectors yet",
)
candidates: list[dict[str, str]] = []
for endpoint_url in _webdav_discovery_candidates(payload):
profile = _discovery_profile_from_payload(
payload, endpoint_url, principal=principal
)
try:
_ensure_connector_configuration_allowed(
session,
principal,
connector_id=None,
credential_id=None,
provider=profile.provider,
endpoint_url=endpoint_url,
base_path=profile.base_path,
scope_type="tenant",
scope_id=principal.tenant_id,
operation="discover",
)
_audit_connector_discovery_attempt(
session,
principal,
provider=profile.provider,
endpoint_url=endpoint_url,
base_path=profile.base_path,
)
browse_connector_profile(profile, path=payload.base_path or "")
except ConnectorPolicyDenied as exc:
raise _connector_policy_error(exc) from exc
except (
ConnectorBrowseError,
ConnectorBrowseUnsupported,
OSError,
ValueError,
json.JSONDecodeError,
) as exc:
message = str(exc)
if "credentials were rejected" in message.casefold():
if payload.require_valid_credentials:
candidates.append(
{
"endpoint_url": endpoint_url,
"status": "credentials_rejected",
"message": "The endpoint exists, but the credentials were rejected.",
}
)
return FileConnectorDiscoveryResponse(
provider=payload.provider,
endpoint_url=endpoint_url,
base_path=payload.base_path,
status="credentials_rejected",
message="The endpoint was found, but login failed with these credentials.",
candidates=candidates,
metadata={"discovered_by": "webdav-auth-challenge"},
)
candidates.append(
{
"endpoint_url": endpoint_url,
"status": "found",
"message": "The endpoint exists, but credentials are required or were rejected.",
}
)
return FileConnectorDiscoveryResponse(
provider=payload.provider,
endpoint_url=endpoint_url,
base_path=payload.base_path,
status="found",
message="The endpoint was found. Add working credentials before saving or testing the connection.",
candidates=candidates,
metadata={"discovered_by": "webdav-auth-challenge"},
)
candidates.append(
{"endpoint_url": endpoint_url, "status": "failed", "message": message}
)
continue
status_value = (
"usable" if _same_endpoint(endpoint_url, payload.endpoint_url) else "found"
)
message = (
"The supplied URL is directly usable."
if status_value == "usable"
else "A usable connector endpoint was discovered."
)
candidates.append(
{"endpoint_url": endpoint_url, "status": status_value, "message": message}
)
return FileConnectorDiscoveryResponse(
provider=payload.provider,
endpoint_url=endpoint_url,
base_path=payload.base_path,
status=status_value,
message=message,
candidates=candidates,
metadata={"discovered_by": "webdav-propfind"},
)
return FileConnectorDiscoveryResponse(
provider=payload.provider,
endpoint_url=None,
base_path=payload.base_path,
status="not_found",
message="No usable WebDAV endpoint was found for this server URL.",
candidates=candidates,
)
@router.get("/connectors/credentials", response_model=FileConnectorCredentialsResponse)
def list_connector_credentials(
provider: str | None = None,
include_disabled: bool = False,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:admin", "system:settings:read", "admin:settings:read"
)
),
):
provider_norm = provider.strip().casefold() if provider else None
credentials = list_database_connector_credentials(
session,
tenant_id=principal.tenant_id,
include_disabled=include_disabled
and _can_read_disabled_connector_profiles(principal),
)
return FileConnectorCredentialsResponse(
credentials=[
FileConnectorCredentialResponse(**credential.to_response())
for credential in credentials
if provider_norm is None or credential.provider in {None, provider_norm}
]
)
@router.get(
"/connectors/policies/{scope_type}", response_model=FileConnectorPolicyResponse
)
def read_connector_policy(
scope_type: str,
scope_id: str | None = Query(default=None),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:admin", "system:settings:read", "admin:settings:read"
)
),
):
_require_connector_policy_read(principal, scope_type)
try:
return FileConnectorPolicyResponse(
**connector_policy_response(
session,
tenant_id=principal.tenant_id,
scope_type=scope_type,
scope_id=scope_id,
)
)
except (FileStorageError, ValueError, json.JSONDecodeError) as exc:
raise _http_error(exc, not_found=True) from exc
@router.put(
"/connectors/policies/{scope_type}", response_model=FileConnectorPolicyResponse
)
def write_connector_policy(
scope_type: str,
payload: FileConnectorPolicyUpdateRequest,
scope_id: str | None = Query(default=None),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:admin", "system:settings:write", "admin:settings:write"
)
),
):
_require_connector_profile_write(principal, scope_type)
try:
set_connector_policy(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
scope_type=scope_type,
scope_id=scope_id,
policy=payload.policy,
)
_record_connector_settings_change(
session,
collection=FILES_CONNECTOR_POLICIES_COLLECTION,
resource_type=FILES_CONNECTOR_POLICY_RESOURCE,
resource_id=_file_connector_policy_resource_id(scope_type, scope_id),
operation="updated",
principal=principal,
tenant_id=None
if scope_type.strip().casefold() == "system"
else principal.tenant_id,
payload={"scope_type": scope_type.strip().casefold(), "scope_id": scope_id},
)
session.commit()
return FileConnectorPolicyResponse(
**connector_policy_response(
session,
tenant_id=principal.tenant_id,
scope_type=scope_type,
scope_id=scope_id,
)
)
except (FileStorageError, ValueError, json.JSONDecodeError) as exc:
session.rollback()
raise _http_error(exc) from exc
@router.post(
"/connectors/credentials",
response_model=FileConnectorCredentialResponse,
status_code=status.HTTP_201_CREATED,
)
def create_connector_credential(
payload: FileConnectorCredentialCreateRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:admin", "system:settings:write", "admin:settings:write"
)
),
):
_require_connector_credential_write(principal, payload.scope_type)
credentials = payload.credentials
try:
_ensure_connector_credential_configuration_allowed(
session,
principal,
credential_id=payload.id,
provider=payload.provider,
scope_type=payload.scope_type,
scope_id=payload.scope_id,
operation="configure_credentials",
)
_ensure_connector_local_policy_allowed(
session,
principal,
scope_type=payload.scope_type,
scope_id=payload.scope_id,
policy=payload.policy,
)
row = create_connector_credential_row(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
credential_id=payload.id,
label=payload.label,
provider=payload.provider,
scope_type=payload.scope_type,
scope_id=payload.scope_id,
enabled=payload.enabled,
credential_mode=payload.credential_mode,
username=credentials.username,
password=credentials.password,
token=credentials.token,
password_env=credentials.password_env,
token_env=credentials.token_env,
secret_ref=credentials.secret_ref,
policy=payload.policy,
metadata=payload.metadata,
)
_record_connector_settings_change(
session,
collection=FILES_CONNECTOR_CREDENTIALS_COLLECTION,
resource_type=FILES_CONNECTOR_CREDENTIAL_RESOURCE,
resource_id=row.id,
operation="created",
principal=principal,
tenant_id=row.tenant_id,
payload={
"scope_type": row.scope_type,
"scope_id": row.scope_id,
"provider": row.provider,
},
)
session.commit()
session.refresh(row)
return _connector_credential_response(row)
except ConnectorPolicyDenied as exc:
session.rollback()
raise _connector_policy_error(exc) from exc
except (FileStorageError, ValueError, json.JSONDecodeError) as exc:
session.rollback()
raise _http_error(exc) from exc
@router.get(
"/connectors/credentials/{credential_id}",
response_model=FileConnectorCredentialResponse,
)
def get_connector_credential(
credential_id: str,
include_disabled: bool = False,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:admin", "system:settings:read", "admin:settings:read"
)
),
):
try:
row = get_connector_credential_row(
session,
tenant_id=principal.tenant_id,
credential_id=credential_id,
include_disabled=include_disabled
and _can_read_disabled_connector_profiles(principal),
)
return _connector_credential_response(row)
except FileStorageError as exc:
raise _http_error(exc, not_found=True) from exc
@router.patch(
"/connectors/credentials/{credential_id}",
response_model=FileConnectorCredentialResponse,
)
def update_connector_credential(
credential_id: str,
payload: FileConnectorCredentialUpdateRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:admin", "system:settings:write", "admin:settings:write"
)
),
):
try:
row = get_connector_credential_row(
session,
tenant_id=principal.tenant_id,
credential_id=credential_id,
include_disabled=True,
)
except FileStorageError as exc:
raise _http_error(exc, not_found=True) from exc
_require_connector_credential_write(principal, row.scope_type)
credentials = payload.credentials
try:
provider = payload.provider if payload.provider is not None else row.provider
_ensure_connector_credential_configuration_allowed(
session,
principal,
credential_id=row.id,
provider=provider,
scope_type=row.scope_type,
scope_id=row.scope_id,
operation="configure_credentials",
)
if payload.policy is not None:
_ensure_connector_local_policy_allowed(
session,
principal,
scope_type=row.scope_type,
scope_id=row.scope_id,
policy=payload.policy,
)
update_connector_credential_row(
session,
row,
user_id=principal.user.id,
label=payload.label,
provider=payload.provider,
enabled=payload.enabled,
credential_mode=payload.credential_mode,
username=credentials.username if credentials else None,
password=credentials.password if credentials else None,
token=credentials.token if credentials else None,
password_env=credentials.password_env if credentials else None,
token_env=credentials.token_env if credentials else None,
secret_ref=credentials.secret_ref if credentials else None,
clear_password=payload.clear_password,
clear_token=payload.clear_token,
policy=payload.policy,
metadata=payload.metadata,
)
_record_connector_settings_change(
session,
collection=FILES_CONNECTOR_CREDENTIALS_COLLECTION,
resource_type=FILES_CONNECTOR_CREDENTIAL_RESOURCE,
resource_id=row.id,
operation="updated",
principal=principal,
tenant_id=row.tenant_id,
payload={
"scope_type": row.scope_type,
"scope_id": row.scope_id,
"provider": row.provider,
},
)
session.commit()
session.refresh(row)
return _connector_credential_response(row)
except ConnectorPolicyDenied as exc:
session.rollback()
raise _connector_policy_error(exc) from exc
except (FileStorageError, ValueError, json.JSONDecodeError) as exc:
session.rollback()
raise _http_error(exc) from exc
@router.delete(
"/connectors/credentials/{credential_id}",
response_model=FileConnectorCredentialResponse,
)
def deactivate_connector_credential(
credential_id: str,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:admin", "system:settings:write", "admin:settings:write"
)
),
):
try:
row = get_connector_credential_row(
session,
tenant_id=principal.tenant_id,
credential_id=credential_id,
include_disabled=True,
)
except FileStorageError as exc:
raise _http_error(exc, not_found=True) from exc
_require_connector_credential_write(principal, row.scope_type)
try:
deletion = delete_connector_credential_row(
session,
row,
deletion_reason="api_delete",
user_id=principal.user.id,
api_key_id=principal.api_key.id if principal.api_key else None,
)
if deletion.changed:
_record_connector_settings_change(
session,
collection=FILES_CONNECTOR_CREDENTIALS_COLLECTION,
resource_type=FILES_CONNECTOR_CREDENTIAL_RESOURCE,
resource_id=row.id,
operation="deleted",
principal=principal,
tenant_id=row.tenant_id,
payload={
"scope_type": row.scope_type,
"scope_id": row.scope_id,
"provider": row.provider,
},
)
for profile in deletion.affected_profiles:
_record_connector_settings_change(
session,
collection=FILES_CONNECTOR_PROFILES_COLLECTION,
resource_type=FILES_CONNECTOR_PROFILE_RESOURCE,
resource_id=profile.id,
operation="updated",
principal=principal,
tenant_id=profile.tenant_id,
payload={
"scope_type": profile.scope_type,
"scope_id": profile.scope_id,
"provider": profile.provider,
"reason": "credential_deleted",
},
)
session.commit()
session.refresh(row)
return _connector_credential_response(row)
except FileStorageError as exc:
session.rollback()
raise _http_error(exc) from exc
except Exception:
session.rollback()
raise
@router.get("/connectors/profiles", response_model=FileConnectorProfilesResponse)
def list_connector_profiles(
provider: str | None = None,
campaign_id: str | None = None,
include_disabled: bool = False,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:read",
"files:file:upload",
"files:file:download",
"files:file:admin",
"system:settings:read",
"admin:settings:read",
)
),
):
try:
if campaign_id:
_ensure_campaign_file_access(session, principal, campaign_id)
profiles = _visible_connector_profiles(
session,
principal,
provider=provider,
campaign_id=campaign_id,
include_disabled=include_disabled
and _can_read_disabled_connector_profiles(principal),
include_admin_scopes=_can_read_disabled_connector_profiles(principal),
include_effective_policy=False,
)
except (OSError, ValueError, json.JSONDecodeError) as exc:
raise _http_error(exc) from exc
return FileConnectorProfilesResponse(
profiles=[
FileConnectorProfileResponse(**profile.to_response())
for profile in profiles
]
)
@router.post(
"/connectors/profiles",
response_model=FileConnectorProfileResponse,
status_code=status.HTTP_201_CREATED,
)
def create_connector_profile(
payload: FileConnectorProfileCreateRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(
require_any_scope(
"files:file:admin", "system:settings:write", "admin:settings:write"
)
),
):
_require_connector_profile_write(principal, payload.scope_type)
credentials = payload.credentials
try:
credential_row = _credential_row_for_profile(
session,
principal,
credential_profile_id=payload.credential_profile_id,
provider=payload.provider,
profile_id=payload.id,
scope_type=payload.scope_type,
scope_id=payload.scope_id,
)
_ensure_connector_configuration_allowed(
session,
principal,
connector_id=payload.id,
credential_id=payload.credential_profile_id,
provider=payload.provider,
endpoint_url=connector_effective_endpoint_url(
provider=payload.provider,
endpoint_url=payload.endpoint_url,
metadata=payload.metadata,
),
base_path=payload.base_path,
scope_type=payload.scope_type,
scope_id=payload.scope_id,
operation="configure",
)
_ensure_connector_local_policy_allowed(
session,
principal,
scope_type=payload.scope_type,
scope_id=payload.scope_id,
policy=payload.policy,
)
row = create_connector_profile_row(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
profile_id=payload.id,
label=payload.label,
provider=payload.provider,
scope_type=payload.scope_type,
scope_id=payload.scope_id,
endpoint_url=payload.endpoint_url,
base_path=payload.base_path,
enabled=payload.enabled,
credential_profile_id=payload.credential_profile_id,
credential_mode=payload.credential_mode,
username=credentials.username,
password=credentials.password,
token=credentials.token,
password_env=credentials.password_env,
token_env=credentials.token_env,
secret_ref=credentials.secret_ref,
capabilities=payload.capabilities,
policy=payload.policy,
metadata=payload.metadata,
)
_record_connector_settings_change(
session,
collection=FILES_CONNECTOR_PROFILES_COLLECTION,
resource_type=FILES_CONNECTOR_PROFILE_RESOURCE,
resource_id=row.id,
operation="created",
principal=principal,
tenant_id=row.tenant_id,
payload={
"scope_type": row.scope_type,
"scope_id": row.scope_id,
"provider": row.provider,
},
)
session.commit()
session.refresh(row)
return FileConnectorProfileResponse(
**connector_profile_from_row(
row, credential_row=credential_row
).to_response()
)
except ConnectorPolicyDenied as exc:
session.rollback()
raise _connector_policy_error(exc) from exc
except (FileStorageError, ValueError, json.JSONDecodeError) as exc:
session.rollback()
raise _http_error(exc) from exc
@@ -0,0 +1,151 @@
from __future__ import annotations
from typing import Literal
from fastapi import APIRouter, Depends, Query
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal, require_scope
from govoplan_files.backend.schemas import (
FileFolderCreateRequest,
FileFolderDeleteRequest,
FileFolderDeleteResponse,
FileFolderResponse,
FileFoldersResponse,
)
from govoplan_core.db.session import get_session
from govoplan_files.backend.storage.paths import UnsafeFilePathError
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.folders import (
create_folder,
list_folders_for_user,
list_folders_for_user_window,
soft_delete_folder,
)
from govoplan_files.backend.route_support import (
_folder_response,
_http_error,
_is_admin,
)
from govoplan_files.backend.services.list_queries import (
FOLDERS_LIST_CURSOR_SCOPE,
_cursor_page_size,
_files_delta_watermark,
_folder_cursor_values,
_folders_list_fingerprint,
_next_folder_list_cursor,
)
router = APIRouter(prefix="/files", tags=["files"])
@router.get("/folders", response_model=FileFoldersResponse)
def list_file_folders(
owner_type: Literal["user", "group"],
owner_id: str,
page_size: int | None = Query(default=None, ge=1, le=1000),
cursor: str | None = None,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:read")),
):
try:
watermark = _files_delta_watermark(session, principal.tenant_id)
effective_page_size = _cursor_page_size(
FOLDERS_LIST_CURSOR_SCOPE, cursor, page_size
)
if effective_page_size is None:
folders = list_folders_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
owner_type=owner_type,
owner_id=owner_id,
is_admin=_is_admin(principal),
)
return FileFoldersResponse(
folders=[_folder_response(folder) for folder in folders],
watermark=watermark,
)
fingerprint = _folders_list_fingerprint(
principal,
owner_type=owner_type,
owner_id=owner_id,
page_size=effective_page_size,
)
after_path, after_id = _folder_cursor_values(cursor, fingerprint=fingerprint)
folders, has_more = list_folders_for_user_window(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
owner_type=owner_type,
owner_id=owner_id,
is_admin=_is_admin(principal),
page_size=effective_page_size,
after_path=after_path,
after_id=after_id,
)
return FileFoldersResponse(
folders=[_folder_response(folder) for folder in folders],
cursor=cursor,
next_cursor=_next_folder_list_cursor(
principal,
folders,
owner_type=owner_type,
owner_id=owner_id,
page_size=effective_page_size,
has_more=has_more,
),
watermark=watermark,
)
except FileStorageError as exc:
raise _http_error(exc) from exc
@router.post("/folders", response_model=FileFolderResponse)
def create_file_folder(
payload: FileFolderCreateRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:organize")),
):
try:
folder = create_folder(
session,
tenant_id=principal.tenant_id,
owner_type=payload.owner_type,
owner_id=payload.owner_id,
user_id=principal.user.id,
path=payload.path,
is_admin=_is_admin(principal),
)
session.commit()
return _folder_response(folder)
except (FileStorageError, UnsafeFilePathError, ValueError) as exc:
session.rollback()
raise _http_error(exc) from exc
@router.post("/folders/delete", response_model=FileFolderDeleteResponse)
def delete_file_folder(
payload: FileFolderDeleteRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:delete")),
):
try:
deleted_folders, deleted_files = soft_delete_folder(
session,
tenant_id=principal.tenant_id,
owner_type=payload.owner_type,
owner_id=payload.owner_id,
user_id=principal.user.id,
path=payload.path,
recursive=payload.recursive,
is_admin=_is_admin(principal),
)
session.commit()
return FileFolderDeleteResponse(
deleted_folders=deleted_folders, deleted_files=deleted_files
)
except (FileStorageError, UnsafeFilePathError, ValueError) as exc:
session.rollback()
raise _http_error(exc) from exc
@@ -0,0 +1,387 @@
from __future__ import annotations
import hashlib
from fastapi import APIRouter, Depends, HTTPException, Query, status
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal, require_scope
from govoplan_core.audit.logging import audit_from_principal
from govoplan_core.db.session import get_session
from govoplan_files.backend.db.models import (
FileIntegrityFinding,
FileIntegrityScan,
)
from govoplan_files.backend.schemas import (
FileIntegrityActionRequest,
FileIntegrityActionResponse,
FileIntegrityFindingResponse,
FileIntegrityFindingsResponse,
FileIntegrityScanCreateRequest,
FileIntegrityScanRunRequest,
FileIntegrityScanResponse,
FileIntegrityScansResponse,
)
from govoplan_files.backend.storage.backends import StorageBackendError
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.integrity import (
cleanup_orphan_finding,
create_integrity_scan,
mark_integrity_scan_failed,
recheck_integrity_finding,
run_integrity_scan_batch,
)
router = APIRouter(prefix="/files/integrity", tags=["files-integrity"])
@router.get("/scans", response_model=FileIntegrityScansResponse)
def list_integrity_scans(
limit: int = Query(default=50, ge=1, le=200),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:admin")),
) -> FileIntegrityScansResponse:
rows = (
session.query(FileIntegrityScan)
.filter(FileIntegrityScan.tenant_id == principal.tenant_id)
.order_by(FileIntegrityScan.created_at.desc(), FileIntegrityScan.id.desc())
.limit(limit)
.all()
)
return FileIntegrityScansResponse(
scans=[_scan_response(row) for row in rows]
)
@router.post(
"/scans",
response_model=FileIntegrityScanResponse,
status_code=status.HTTP_201_CREATED,
)
def create_scan(
payload: FileIntegrityScanCreateRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:admin")),
) -> FileIntegrityScanResponse:
try:
scan = create_integrity_scan(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
verify_checksums=payload.verify_checksums,
batch_size=payload.batch_size,
)
audit_from_principal(
session,
principal,
action="files.integrity.scan_created",
object_type="file_integrity_scan",
object_id=scan.id,
details={
"storage_backend": scan.storage_backend,
"verify_checksums": scan.verify_checksums,
"batch_size": scan.batch_size,
},
)
session.commit()
return _scan_response(scan)
except (FileStorageError, StorageBackendError) as exc:
session.rollback()
raise _integrity_http_error(exc) from exc
@router.post("/scans/{scan_id}/run", response_model=FileIntegrityScanResponse)
def run_scan_batch(
scan_id: str,
payload: FileIntegrityScanRunRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:admin")),
) -> FileIntegrityScanResponse:
scan = _scan_for_tenant(
session,
scan_id,
principal.tenant_id,
for_update=True,
)
_assert_expected_revision(scan.revision, payload.expected_revision)
previous_status = scan.status
try:
run_integrity_scan_batch(session, scan)
if scan.status == "completed" and previous_status != "completed":
audit_from_principal(
session,
principal,
action="files.integrity.scan_completed",
object_type="file_integrity_scan",
object_id=scan.id,
details={
"verified_blob_count": scan.verified_blob_count,
"quarantined_blob_count": scan.quarantined_blob_count,
"orphan_object_count": scan.orphan_object_count,
},
)
session.commit()
return _scan_response(scan)
except (FileStorageError, StorageBackendError) as exc:
session.rollback()
scan = _scan_for_tenant(
session,
scan_id,
principal.tenant_id,
for_update=True,
)
mark_integrity_scan_failed(scan, error=exc)
audit_from_principal(
session,
principal,
action="files.integrity.scan_failed",
object_type="file_integrity_scan",
object_id=scan.id,
details={"error_type": type(exc).__name__},
)
session.commit()
raise _integrity_http_error(exc) from exc
@router.get(
"/scans/{scan_id}/findings",
response_model=FileIntegrityFindingsResponse,
)
def list_integrity_findings(
scan_id: str,
state_filter: str | None = Query(default=None, alias="state"),
limit: int = Query(default=500, ge=1, le=1000),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:admin")),
) -> FileIntegrityFindingsResponse:
scan = _scan_for_tenant(session, scan_id, principal.tenant_id)
query = session.query(FileIntegrityFinding).filter(
FileIntegrityFinding.scan_id == scan.id,
FileIntegrityFinding.tenant_id == principal.tenant_id,
)
if state_filter:
query = query.filter(FileIntegrityFinding.state == state_filter)
rows = (
query.order_by(
FileIntegrityFinding.created_at.asc(),
FileIntegrityFinding.id.asc(),
)
.limit(limit)
.all()
)
return FileIntegrityFindingsResponse(
findings=[_finding_response(row) for row in rows]
)
@router.post(
"/findings/{finding_id}/recheck",
response_model=FileIntegrityActionResponse,
)
def recheck_finding(
finding_id: str,
payload: FileIntegrityActionRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:admin")),
) -> FileIntegrityActionResponse:
finding = _finding_for_tenant(
session,
finding_id,
principal.tenant_id,
for_update=True,
)
_assert_expected_revision(finding.revision, payload.expected_revision)
try:
result = recheck_integrity_finding(
session,
finding,
user_id=principal.user.id,
dry_run=payload.dry_run,
)
_audit_integrity_action(session, principal, result)
session.commit()
return _action_response(result)
except (FileStorageError, StorageBackendError) as exc:
session.rollback()
raise _integrity_http_error(exc) from exc
@router.post(
"/findings/{finding_id}/cleanup",
response_model=FileIntegrityActionResponse,
)
def cleanup_finding(
finding_id: str,
payload: FileIntegrityActionRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:admin")),
) -> FileIntegrityActionResponse:
finding = _finding_for_tenant(
session,
finding_id,
principal.tenant_id,
for_update=True,
)
_assert_expected_revision(finding.revision, payload.expected_revision)
try:
result = cleanup_orphan_finding(
session,
finding,
user_id=principal.user.id,
dry_run=payload.dry_run,
)
_audit_integrity_action(session, principal, result)
session.commit()
return _action_response(result)
except (FileStorageError, StorageBackendError) as exc:
session.rollback()
raise _integrity_http_error(exc) from exc
def _scan_for_tenant(
session: Session,
scan_id: str,
tenant_id: str,
*,
for_update: bool = False,
) -> FileIntegrityScan:
query = session.query(FileIntegrityScan).filter(FileIntegrityScan.id == scan_id)
if for_update:
query = query.populate_existing().with_for_update()
scan = query.one_or_none()
if scan is None or scan.tenant_id != tenant_id:
raise HTTPException(
status_code=status.HTTP_404_NOT_FOUND,
detail="Integrity scan not found",
)
return scan
def _finding_for_tenant(
session: Session,
finding_id: str,
tenant_id: str,
*,
for_update: bool = False,
) -> FileIntegrityFinding:
query = session.query(FileIntegrityFinding).filter(
FileIntegrityFinding.id == finding_id
)
if for_update:
query = query.populate_existing().with_for_update()
finding = query.one_or_none()
if finding is None or finding.tenant_id != tenant_id:
raise HTTPException(
status_code=status.HTTP_404_NOT_FOUND,
detail="Integrity finding not found",
)
return finding
def _assert_expected_revision(current: int, expected: int) -> None:
if current != expected:
raise HTTPException(
status_code=status.HTTP_409_CONFLICT,
detail=(
"The integrity record changed after it was loaded; reload before "
"performing this action."
),
)
def _audit_integrity_action(session, principal, result) -> None:
audit_from_principal(
session,
principal,
action=f"files.integrity.{result.action}",
object_type="file_integrity_finding",
object_id=result.finding.id,
details={
"scan_id": result.finding.scan_id,
"kind": result.finding.kind,
"dry_run": result.dry_run,
"changed": result.changed,
"storage_key_sha256": hashlib.sha256(
result.finding.storage_key.encode("utf-8")
).hexdigest(),
"inspection": result.inspection.kind
if result.inspection
else None,
},
)
def _scan_response(scan: FileIntegrityScan) -> FileIntegrityScanResponse:
return FileIntegrityScanResponse(
id=scan.id,
tenant_id=scan.tenant_id,
storage_backend=scan.storage_backend,
storage_prefix=scan.storage_prefix,
status=scan.status,
revision=scan.revision,
phase=scan.phase,
verify_checksums=scan.verify_checksums,
batch_size=scan.batch_size,
scanned_blob_count=scan.scanned_blob_count,
verified_blob_count=scan.verified_blob_count,
quarantined_blob_count=scan.quarantined_blob_count,
scanned_object_count=scan.scanned_object_count,
orphan_object_count=scan.orphan_object_count,
created_by_user_id=scan.created_by_user_id,
started_at=scan.started_at.isoformat() if scan.started_at else None,
completed_at=scan.completed_at.isoformat()
if scan.completed_at
else None,
last_error=scan.last_error,
created_at=scan.created_at.isoformat(),
updated_at=scan.updated_at.isoformat(),
)
def _finding_response(
finding: FileIntegrityFinding,
) -> FileIntegrityFindingResponse:
return FileIntegrityFindingResponse(
id=finding.id,
scan_id=finding.scan_id,
tenant_id=finding.tenant_id,
kind=finding.kind,
state=finding.state,
revision=finding.revision,
blob_id=finding.blob_id,
storage_key=finding.storage_key,
expected_size_bytes=finding.expected_size_bytes,
observed_size_bytes=finding.observed_size_bytes,
expected_checksum_sha256=finding.expected_checksum_sha256,
observed_checksum_sha256=finding.observed_checksum_sha256,
resolved_at=finding.resolved_at.isoformat()
if finding.resolved_at
else None,
resolved_by_user_id=finding.resolved_by_user_id,
created_at=finding.created_at.isoformat(),
updated_at=finding.updated_at.isoformat(),
)
def _action_response(result) -> FileIntegrityActionResponse:
return FileIntegrityActionResponse(
action=result.action,
changed=result.changed,
dry_run=result.dry_run,
finding=_finding_response(result.finding),
inspection_kind=result.inspection.kind if result.inspection else None,
inspection_valid=result.inspection.valid if result.inspection else None,
)
def _integrity_http_error(exc: Exception) -> HTTPException:
if isinstance(exc, FileStorageError):
return HTTPException(
status_code=status.HTTP_409_CONFLICT,
detail=str(exc),
)
return HTTPException(
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
detail="The configured file storage backend could not complete the integrity operation",
)
@@ -0,0 +1,167 @@
from __future__ import annotations
from typing import Literal
from fastapi import APIRouter, Depends, Query
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal, require_scope
from govoplan_files.backend.schemas import (
FileDeltaResponse,
FileListResponse,
)
from govoplan_core.db.session import get_session
from govoplan_files.backend.storage.files import (
count_assets_for_user,
list_assets_for_user,
list_assets_for_user_window,
)
from govoplan_files.backend.route_support import (
_asset_list_response,
_ensure_campaign_file_access,
_ensure_list_owner_access,
_is_admin,
)
from govoplan_files.backend.services.list_queries import (
FILES_LIST_CURSOR_SCOPE,
_cursor_page_size,
_file_cursor_values,
_files_delta_response,
_files_delta_watermark,
_files_list_fingerprint,
_full_file_delta_response,
_next_file_list_cursor,
)
router = APIRouter(prefix="/files", tags=["files"])
@router.get("/delta", response_model=FileDeltaResponse)
def files_delta(
owner_type: Literal["user", "group"] | None = None,
owner_id: str | None = None,
campaign_id: str | None = None,
path_prefix: str | None = None,
since: str | None = None,
limit: int = Query(default=500, ge=1, le=1000),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:read")),
):
_ensure_list_owner_access(session, principal, owner_type, owner_id)
_ensure_campaign_file_access(session, principal, campaign_id)
if since is None:
return _full_file_delta_response(
session,
principal=principal,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
)
return _files_delta_response(
session,
principal=principal,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
since=since,
limit=limit,
)
@router.get("", response_model=FileListResponse)
def list_files(
owner_type: Literal["user", "group"] | None = None,
owner_id: str | None = None,
campaign_id: str | None = None,
path_prefix: str | None = None,
campaign_usage: Literal["linked", "unlinked"] | None = None,
audit_relevant: bool | None = None,
page_size: int | None = Query(default=None, ge=1, le=1000),
cursor: str | None = None,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:read")),
):
_ensure_list_owner_access(session, principal, owner_type, owner_id)
_ensure_campaign_file_access(session, principal, campaign_id)
watermark = _files_delta_watermark(session, principal.tenant_id)
total = count_assets_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
campaign_usage=campaign_usage,
audit_relevant=audit_relevant,
is_admin=_is_admin(principal),
)
effective_page_size = _cursor_page_size(FILES_LIST_CURSOR_SCOPE, cursor, page_size)
if effective_page_size is not None:
fingerprint = _files_list_fingerprint(
principal,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
campaign_usage=campaign_usage,
audit_relevant=audit_relevant,
page_size=effective_page_size,
)
after_display_path, after_updated_at, after_id = _file_cursor_values(
cursor, fingerprint=fingerprint
)
assets, has_more = list_assets_for_user_window(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
campaign_usage=campaign_usage,
audit_relevant=audit_relevant,
is_admin=_is_admin(principal),
page_size=effective_page_size,
after_display_path=after_display_path,
after_updated_at=after_updated_at,
after_id=after_id,
)
return FileListResponse(
files=_asset_list_response(session, assets, include_shares=True),
total=total,
cursor=cursor,
next_cursor=_next_file_list_cursor(
principal,
assets,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
campaign_usage=campaign_usage,
audit_relevant=audit_relevant,
page_size=effective_page_size,
has_more=has_more,
),
watermark=watermark,
)
assets = list_assets_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
campaign_usage=campaign_usage,
audit_relevant=audit_relevant,
is_admin=_is_admin(principal),
)
return FileListResponse(
files=_asset_list_response(session, assets, include_shares=True),
total=total,
watermark=watermark,
)
+330
View File
@@ -0,0 +1,330 @@
from __future__ import annotations
from datetime import datetime, timezone
from fastapi import APIRouter, Depends, HTTPException, Query, status
from sqlalchemy.orm import Session
from govoplan_core.api.v1.schemas import (
ReferenceOptionListResponse,
ReferenceOptionResponse,
)
from govoplan_core.auth import ApiPrincipal, require_scope
from govoplan_core.audit.logging import audit_from_principal
from govoplan_core.core.references import (
access_scope_reference_page,
access_scope_reference_provider_available,
)
from govoplan_core.db.session import get_session
from govoplan_files.backend.runtime import get_registry
from govoplan_files.backend.schemas import (
BulkFileShareRequest,
BulkFileShareResponse,
FileShareRequest,
FileShareResponse,
FileSharesResponse,
)
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.files import (
current_file_share_for_target,
get_asset_for_share_management,
list_file_shares,
revoke_file_share,
share_file,
share_files,
)
from govoplan_files.backend.route_support import (
_file_share_response,
_http_error,
_is_admin,
)
router = APIRouter(prefix="/files", tags=["files"])
@router.get(
"/{file_id}/share-target-options",
response_model=ReferenceOptionListResponse,
)
def search_share_targets(
file_id: str,
target_type: str,
q: str = "",
selected: list[str] = Query(default=[]),
limit: int = Query(default=50, ge=1, le=200),
cursor: str | None = None,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:share")),
) -> ReferenceOptionListResponse:
if target_type not in {"user", "group"}:
raise HTTPException(
status_code=status.HTTP_422_UNPROCESSABLE_CONTENT,
detail="Share target type must be user or group",
)
try:
get_asset_for_share_management(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
asset_id=file_id,
is_admin=_is_admin(principal),
)
registry = get_registry()
page = access_scope_reference_page(
registry,
principal,
scope_type=target_type,
reference_kind="membership" if target_type == "user" else "group",
query=q,
selected_values=selected,
limit=limit,
cursor=cursor,
administrative=True,
session=session,
)
except ValueError as exc:
raise HTTPException(
status_code=status.HTTP_422_UNPROCESSABLE_CONTENT,
detail=str(exc),
) from exc
except FileStorageError as exc:
raise _http_error(exc) from exc
return ReferenceOptionListResponse(
options=[
ReferenceOptionResponse(**option.to_dict()) for option in page.options
],
provider_available=access_scope_reference_provider_available(registry),
next_cursor=page.next_cursor,
has_more=page.has_more,
)
@router.get("/{file_id}/shares", response_model=FileSharesResponse)
def list_shares(
file_id: str,
include_inactive: bool = False,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:share")),
):
try:
asset = get_asset_for_share_management(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
asset_id=file_id,
is_admin=_is_admin(principal),
)
return FileSharesResponse(
shares=[
_file_share_response(share)
for share in list_file_shares(
session,
tenant_id=principal.tenant_id,
asset_id=asset.id,
include_inactive=include_inactive,
)
]
)
except FileStorageError as exc:
raise _http_error(exc) from exc
@router.post("/{file_id}/shares", response_model=FileShareResponse)
def create_share(
file_id: str,
payload: FileShareRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:share")),
):
try:
asset = get_asset_for_share_management(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
asset_id=file_id,
is_admin=_is_admin(principal),
)
previous = current_file_share_for_target(
session,
tenant_id=principal.tenant_id,
asset_id=asset.id,
target_type=payload.target_type,
target_id=payload.target_id,
)
previous_permission = previous.permission if previous else None
previous_expiry = previous.expires_at if previous else None
share = share_file(
session,
tenant_id=principal.tenant_id,
asset=asset,
target_type=payload.target_type,
target_id=payload.target_id,
permission=payload.permission,
user_id=principal.user.id,
expires_at=payload.expires_at,
)
session.flush()
action = _share_audit_action(
existed=previous is not None,
previous_permission=previous_permission,
permission=share.permission,
previous_expiry=previous_expiry,
expiry=share.expires_at,
)
if action:
_audit_share_change(session, principal, share, action=action)
session.commit()
return _file_share_response(share)
except FileStorageError as exc:
session.rollback()
raise _http_error(exc) from exc
@router.post("/bulk-shares", response_model=BulkFileShareResponse)
def create_bulk_shares(
payload: BulkFileShareRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:share")),
):
try:
file_ids = list(dict.fromkeys(payload.file_ids))
assets = [
get_asset_for_share_management(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
asset_id=file_id,
is_admin=_is_admin(principal),
)
for file_id in file_ids
]
previous_by_asset = {
asset.id: current_file_share_for_target(
session,
tenant_id=principal.tenant_id,
asset_id=asset.id,
target_type=payload.target_type,
target_id=payload.target_id,
)
for asset in assets
}
previous_values = {
asset_id: (
share.permission if share else None,
share.expires_at if share else None,
)
for asset_id, share in previous_by_asset.items()
}
shares = share_files(
session,
tenant_id=principal.tenant_id,
assets=assets,
target_type=payload.target_type,
target_id=payload.target_id,
permission=payload.permission,
user_id=principal.user.id,
expires_at=payload.expires_at,
)
session.flush()
for share in shares:
previous_permission, previous_expiry = previous_values[share.file_asset_id]
action = _share_audit_action(
existed=previous_by_asset[share.file_asset_id] is not None,
previous_permission=previous_permission,
permission=share.permission,
previous_expiry=previous_expiry,
expiry=share.expires_at,
)
if action:
_audit_share_change(session, principal, share, action=action)
session.commit()
return BulkFileShareResponse(
shared_count=len(shares),
shares=[_file_share_response(share) for share in shares],
)
except FileStorageError as exc:
session.rollback()
raise _http_error(exc) from exc
@router.delete("/{file_id}/shares/{share_id}", response_model=FileShareResponse)
def revoke_share(
file_id: str,
share_id: str,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:share")),
):
try:
asset = get_asset_for_share_management(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
asset_id=file_id,
is_admin=_is_admin(principal),
)
share, changed = revoke_file_share(
session,
tenant_id=principal.tenant_id,
asset_id=asset.id,
share_id=share_id,
user_id=principal.user.id,
)
if changed:
_audit_share_change(
session, principal, share, action="files.share.revoked"
)
session.commit()
return _file_share_response(share)
except FileStorageError as exc:
session.rollback()
raise _http_error(exc) from exc
def _share_audit_action(
*,
existed: bool,
previous_permission: str | None,
permission: str,
previous_expiry: datetime | None,
expiry: datetime | None,
) -> str | None:
if not existed:
return "files.share.granted"
permission_changed = previous_permission != permission
expiry_changed = _normalized_expiry(previous_expiry) != _normalized_expiry(expiry)
if permission_changed:
return "files.share.changed"
if expiry_changed:
return "files.share.expiry_changed"
return None
def _audit_share_change(
session: Session,
principal: ApiPrincipal,
share,
*,
action: str,
) -> None:
audit_from_principal(
session,
principal,
action=action,
object_type="file_share",
object_id=share.id,
details={
"file_asset_id": share.file_asset_id,
"target_type": share.target_type,
"target_id": share.target_id,
"permission": share.permission,
"expires_at": _normalized_expiry(share.expires_at),
},
)
def _normalized_expiry(value: datetime | None) -> str | None:
if value is None:
return None
if value.tzinfo is None:
value = value.replace(tzinfo=timezone.utc)
return value.astimezone(timezone.utc).isoformat()
+292
View File
@@ -0,0 +1,292 @@
from __future__ import annotations
import json
from typing import Literal
from fastapi import APIRouter, Depends, HTTPException, status
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal, require_scope
from govoplan_files.backend.change_tracking import (
FILES_CONNECTOR_SPACES_COLLECTION,
)
from govoplan_files.backend.schemas import (
FileConnectorSpaceCreateRequest,
FileConnectorSpaceResponse,
FileConnectorSpacesResponse,
FileConnectorSpaceUpdateRequest,
FileSpaceResponse,
FileSpacesResponse,
)
from govoplan_core.db.session import get_session
from govoplan_files.backend.storage.access import group_refs_for_ids, user_group_ids
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.connector_spaces import (
connector_space_owner_id,
create_connector_space,
get_connector_space_for_user,
list_connector_spaces_for_user,
soft_delete_connector_space,
update_connector_space,
)
from govoplan_files.backend.storage.connector_policy import (
ConnectorPolicyDenied,
)
from govoplan_files.backend.route_support import (
FILES_CONNECTOR_SPACE_RESOURCE,
_connector_policy_error,
_connector_space_file_space_response,
_connector_space_policy_decision,
_connector_space_response,
_http_error,
_is_admin,
_record_connector_settings_change,
_visible_connector_profile,
)
router = APIRouter(prefix="/files", tags=["files"])
@router.get("/spaces", response_model=FileSpacesResponse)
def list_file_spaces(
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:read")),
):
spaces = [
FileSpaceResponse(
id=f"user:{principal.user.id}",
label="My files",
owner_type="user",
owner_id=principal.user.id,
description="Files owned by your user account.",
)
]
group_ids = user_group_ids(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
include_admin_groups=_is_admin(principal),
)
if group_ids:
groups = group_refs_for_ids(tenant_id=principal.tenant_id, group_ids=group_ids)
spaces.extend(
FileSpaceResponse(
id=f"group:{group.id}",
label=f"{group.name} files",
owner_type="group",
owner_id=group.id,
description="Files owned by this group.",
)
for group in groups
)
connector_spaces = list_connector_spaces_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
is_admin=_is_admin(principal),
)
spaces.extend(
_connector_space_file_space_response(space) for space in connector_spaces
)
return FileSpacesResponse(spaces=spaces)
@router.get("/connector-spaces", response_model=FileConnectorSpacesResponse)
def list_file_connector_spaces(
owner_type: Literal["user", "group"] | None = None,
owner_id: str | None = None,
include_inactive: bool = False,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:read")),
):
try:
spaces = list_connector_spaces_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
owner_type=owner_type,
owner_id=owner_id,
include_inactive=include_inactive and _is_admin(principal),
is_admin=_is_admin(principal),
)
return FileConnectorSpacesResponse(
spaces=[_connector_space_response(space) for space in spaces]
)
except FileStorageError as exc:
raise _http_error(exc) from exc
@router.post(
"/connector-spaces",
response_model=FileConnectorSpaceResponse,
status_code=status.HTTP_201_CREATED,
)
def create_file_connector_space(
payload: FileConnectorSpaceCreateRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:organize")),
):
if payload.owner_type == "group" and not payload.owner_id:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail="owner_id is required for group connector spaces",
)
target_owner = payload.owner_id or principal.user.id
try:
profile = _visible_connector_profile(
session, principal, payload.connector_profile_id
)
decision = _connector_space_policy_decision(
profile,
library_id=payload.library_id,
remote_path=payload.remote_path,
operation="link",
)
if not decision.allowed:
raise ConnectorPolicyDenied(decision)
space = create_connector_space(
session,
tenant_id=principal.tenant_id,
owner_type=payload.owner_type,
owner_id=target_owner,
user_id=principal.user.id,
label=payload.label,
profile=profile,
library_id=payload.library_id,
remote_path=payload.remote_path,
sync_mode=payload.sync_mode,
metadata=payload.metadata,
is_admin=_is_admin(principal),
)
_record_connector_settings_change(
session,
collection=FILES_CONNECTOR_SPACES_COLLECTION,
resource_type=FILES_CONNECTOR_SPACE_RESOURCE,
resource_id=space.id,
operation="created",
principal=principal,
tenant_id=space.tenant_id,
payload={
"owner_type": space.owner_type,
"owner_id": connector_space_owner_id(space),
},
)
session.commit()
return _connector_space_response(space)
except ConnectorPolicyDenied as exc:
session.rollback()
raise _connector_policy_error(exc) from exc
except (FileStorageError, ValueError, json.JSONDecodeError) as exc:
session.rollback()
raise _http_error(exc) from exc
@router.patch("/connector-spaces/{space_id}", response_model=FileConnectorSpaceResponse)
def update_file_connector_space(
space_id: str,
payload: FileConnectorSpaceUpdateRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:organize")),
):
try:
space = get_connector_space_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
space_id=space_id,
include_inactive=_is_admin(principal),
is_admin=_is_admin(principal),
)
except FileStorageError as exc:
raise _http_error(exc, not_found=True) from exc
try:
if payload.library_id is not None or payload.remote_path is not None:
profile = _visible_connector_profile(
session, principal, space.connector_profile_id
)
decision = _connector_space_policy_decision(
profile,
library_id=payload.library_id
if payload.library_id is not None
else space.library_id,
remote_path=payload.remote_path
if payload.remote_path is not None
else space.remote_path,
operation="link",
)
if not decision.allowed:
raise ConnectorPolicyDenied(decision)
update_connector_space(
session,
space,
user_id=principal.user.id,
label=payload.label,
library_id=payload.library_id,
remote_path=payload.remote_path,
sync_mode=payload.sync_mode,
is_active=payload.is_active,
metadata=payload.metadata,
is_admin=_is_admin(principal),
)
_record_connector_settings_change(
session,
collection=FILES_CONNECTOR_SPACES_COLLECTION,
resource_type=FILES_CONNECTOR_SPACE_RESOURCE,
resource_id=space.id,
operation="updated",
principal=principal,
tenant_id=space.tenant_id,
payload={
"owner_type": space.owner_type,
"owner_id": connector_space_owner_id(space),
},
)
session.commit()
return _connector_space_response(space)
except ConnectorPolicyDenied as exc:
session.rollback()
raise _connector_policy_error(exc) from exc
except (FileStorageError, ValueError, json.JSONDecodeError) as exc:
session.rollback()
raise _http_error(exc) from exc
@router.delete(
"/connector-spaces/{space_id}", response_model=FileConnectorSpaceResponse
)
def delete_file_connector_space(
space_id: str,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:organize")),
):
try:
space = get_connector_space_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
space_id=space_id,
include_inactive=True,
is_admin=_is_admin(principal),
)
soft_delete_connector_space(
session, space, user_id=principal.user.id, is_admin=_is_admin(principal)
)
_record_connector_settings_change(
session,
collection=FILES_CONNECTOR_SPACES_COLLECTION,
resource_type=FILES_CONNECTOR_SPACE_RESOURCE,
resource_id=space.id,
operation="deleted",
principal=principal,
tenant_id=space.tenant_id,
payload={
"owner_type": space.owner_type,
"owner_id": connector_space_owner_id(space),
},
)
session.commit()
return _connector_space_response(space)
except FileStorageError as exc:
session.rollback()
raise _http_error(exc, not_found=True) from exc
@@ -0,0 +1,213 @@
from __future__ import annotations
import tempfile
from fastapi import APIRouter, Depends
from fastapi.responses import FileResponse
from starlette.background import BackgroundTask
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal, require_scope
from govoplan_files.backend.schemas import (
ArchiveRequest,
PatternMatchResponse,
PatternResolveRequest,
PatternResolveResponse,
RenamePreviewItem,
RenameRequest,
RenameResponse,
TransferRequest,
TransferResponse,
_conflict_resolutions,
)
from govoplan_core.db.session import get_session
from govoplan_files.backend.storage.paths import (
UnsafeFilePathError,
filename_from_path,
normalize_logical_path,
)
from govoplan_files.backend.storage.archives import create_zip_file
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.files import (
get_asset_for_user,
list_assets_for_user,
)
from govoplan_files.backend.storage.search import resolve_patterns
from govoplan_files.backend.storage.transfers import (
rename_selection,
transfer_selection,
)
from govoplan_files.backend.route_support import (
_asset_response,
_attachment_disposition,
_audit_connector_access,
_cleanup_temp_file,
_ensure_campaign_file_access,
_ensure_list_owner_access,
_http_error,
_is_admin,
)
router = APIRouter(prefix="/files", tags=["files"])
@router.post("/bulk-rename", response_model=RenameResponse)
def bulk_rename(
payload: RenameRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:organize")),
):
try:
plan = rename_selection(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
file_ids=payload.file_ids,
folder_paths=payload.folder_paths,
owner_type=payload.owner_type,
owner_id=payload.owner_id,
mode=payload.mode,
new_name=payload.new_name,
find=payload.find,
replacement=payload.replacement,
prefix=payload.prefix,
suffix=payload.suffix,
recursive=payload.recursive,
dry_run=payload.dry_run,
is_admin=_is_admin(principal),
)
if not payload.dry_run:
session.commit()
return RenameResponse(
dry_run=payload.dry_run,
items=[
RenamePreviewItem(
kind=item.kind,
id=item.id,
file_id=item.id if item.kind == "file" else None,
folder_path=item.old_path if item.kind == "folder" else None,
old_path=item.old_path,
new_path=item.new_path,
)
for item in plan
],
)
except (FileStorageError, UnsafeFilePathError, ValueError) as exc:
session.rollback()
raise _http_error(exc) from exc
@router.post("/transfer", response_model=TransferResponse)
def transfer_files(
payload: TransferRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:organize")),
):
try:
files, folders = transfer_selection(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
operation=payload.operation,
file_ids=payload.file_ids,
folder_paths=payload.folder_paths,
source_owner_type=payload.source_owner_type,
source_owner_id=payload.source_owner_id,
target_owner_type=payload.target_owner_type,
target_owner_id=payload.target_owner_id,
target_folder=payload.target_folder,
conflict_strategy=payload.conflict_strategy,
conflict_resolutions=_conflict_resolutions(payload.conflict_resolutions),
is_admin=_is_admin(principal),
)
session.commit()
return TransferResponse(
operation=payload.operation, files=files, folders=folders
)
except (FileStorageError, UnsafeFilePathError, ValueError) as exc:
session.rollback()
raise _http_error(exc) from exc
@router.post("/archive.zip")
def download_archive(
payload: ArchiveRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:download")),
):
try:
assets = [
get_asset_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
asset_id=file_id,
is_admin=_is_admin(principal),
)
for file_id in payload.file_ids
]
tmp = tempfile.NamedTemporaryFile(
prefix="govoplan-files-", suffix=".zip", delete=False
)
tmp_path = tmp.name
tmp.close()
try:
create_zip_file(session, assets, tmp_path)
except Exception:
_cleanup_temp_file(tmp_path)
raise
_audit_connector_access(session, principal, assets, operation="archive")
except FileStorageError as exc:
raise _http_error(exc) from exc
filename = filename_from_path(
normalize_logical_path(payload.filename, fallback_filename="files.zip")
)
headers = {"Content-Disposition": _attachment_disposition(filename)}
return FileResponse(
tmp_path,
media_type="application/zip",
headers=headers,
background=BackgroundTask(_cleanup_temp_file, tmp_path),
)
@router.post("/resolve-patterns", response_model=PatternResolveResponse)
def resolve_file_patterns(
payload: PatternResolveRequest,
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:read")),
):
_ensure_list_owner_access(session, principal, payload.owner_type, payload.owner_id)
_ensure_campaign_file_access(session, principal, payload.campaign_id)
try:
assets = list_assets_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
owner_type=payload.owner_type,
owner_id=payload.owner_id,
campaign_id=payload.campaign_id,
path_prefix=payload.path_prefix,
is_admin=_is_admin(principal),
)
resolved, unmatched = resolve_patterns(
assets,
payload.patterns,
base_path=payload.path_prefix,
case_sensitive=payload.case_sensitive,
)
return PatternResolveResponse(
patterns=[
PatternMatchResponse(
pattern=item.pattern,
matches=[_asset_response(session, asset) for asset in item.matches],
)
for item in resolved
],
unmatched=[_asset_response(session, asset) for asset in unmatched]
if payload.include_unmatched
else [],
)
except (FileStorageError, UnsafeFilePathError, ValueError) as exc:
raise _http_error(exc) from exc
@@ -0,0 +1,487 @@
from __future__ import annotations
import hashlib
import hmac
import json
from datetime import UTC, datetime, timedelta
from typing import Literal
from fastapi import APIRouter, Depends, File as FastAPIFile, Form, UploadFile
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal, require_scope
from govoplan_core.security.secrets import (
TransientPayloadError,
open_transient_payload,
seal_transient_payload,
)
from govoplan_files.backend.schemas import (
ArchiveEntryResponse,
ArchivePreviewResponse,
ConflictResolutionRequest,
FileUploadResponse,
_conflict_resolutions,
)
from govoplan_files.backend.db.models import FileAsset
from govoplan_core.db.session import get_session
from govoplan_files.backend.runtime import settings
from govoplan_files.backend.storage.paths import (
UnsafeFilePathError,
normalize_folder,
)
from govoplan_files.backend.storage.archives import (
archive_format_for_filename,
extract_archive_upload,
extract_zip_upload,
inspect_archive,
)
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.connector_policy import (
ConnectorPolicyDenied,
)
from govoplan_files.backend.storage.files import (
create_file_asset,
)
from govoplan_files.backend.route_support import (
_asset_response,
_audit_connector_imports,
_cleanup_temp_file,
_connector_policy_error,
_enforce_connector_policy,
_http_error,
_is_admin,
_read_limited_upload,
_source_metadata_from_form,
_spool_limited_upload_to_temp,
)
router = APIRouter(prefix="/files", tags=["files"])
_ARCHIVE_PREVIEW_PURPOSE = "files.archive-preview.v1"
def _archive_suffix(filename: str) -> str:
lowered = filename.casefold()
for suffix in (".tar.bz2", ".tar.gz", ".tar.xz", ".tbz2", ".tgz", ".txz", ".tar", ".zip"):
if lowered.endswith(suffix):
return suffix
return ".archive"
def _archive_sha256(path: str) -> str:
digest = hashlib.sha256()
with open(path, "rb") as source:
while chunk := source.read(1024 * 1024):
digest.update(chunk)
return digest.hexdigest()
def _archive_selected_paths(value: str) -> list[str]:
parsed = json.loads(value)
if not isinstance(parsed, list) or not parsed:
raise FileStorageError("Select at least one archive file or folder")
if len(parsed) > settings.file_archive_max_entries:
raise FileStorageError(
"Archive selection exceeds the configured entry limit"
)
selected: list[str] = []
for item in parsed:
if not isinstance(item, str) or not item.strip():
raise FileStorageError("Archive selection contains an invalid path")
if len(item) > 4096:
raise FileStorageError("Archive selection path is too long")
selected.append(item)
return selected
def _validate_archive_preview_token(
token: str,
*,
tenant_id: str,
user_id: str,
owner_type: str,
owner_id: str,
path: str,
campaign_id: str | None,
archive_format: str,
archive_sha256: str,
) -> None:
payload = open_transient_payload(
token,
ttl_seconds=settings.file_archive_preview_ttl_seconds,
)
expected = {
"purpose": _ARCHIVE_PREVIEW_PURPOSE,
"tenant_id": tenant_id,
"user_id": user_id,
"owner_type": owner_type,
"owner_id": owner_id,
"path": path,
"campaign_id": campaign_id or "",
"archive_format": archive_format,
}
if any(payload.get(key) != value for key, value in expected.items()):
raise FileStorageError(
"Archive preview does not match this upload destination"
)
token_digest = payload.get("archive_sha256")
if not isinstance(token_digest, str) or not hmac.compare_digest(
token_digest, archive_sha256
):
raise FileStorageError(
"Archive contents changed after preview; preview it again"
)
@router.post("/archive-preview", response_model=ArchivePreviewResponse)
def preview_archive_upload(
file: UploadFile = FastAPIFile(...),
owner_type: Literal["user", "group"] = Form(default="user"),
owner_id: str | None = Form(default=None),
path: str = Form(default=""),
campaign_id: str | None = Form(default=None),
password: str | None = Form(default=None),
principal: ApiPrincipal = Depends(require_scope("files:file:upload")),
):
target_owner = owner_id or principal.user.id
archive_path: str | None = None
filename = file.filename or "archive"
try:
archive_format = archive_format_for_filename(filename)
archive_path = _spool_limited_upload_to_temp(
file,
max_bytes=settings.file_upload_zip_max_bytes,
suffix=_archive_suffix(filename),
)
inspection = inspect_archive(
archive_path,
filename=filename,
password=password,
max_entries=settings.file_archive_max_entries,
max_expanded_bytes=settings.file_archive_max_expanded_bytes,
max_expansion_ratio=settings.file_archive_max_expansion_ratio,
)
digest = _archive_sha256(archive_path)
normalized_path = normalize_folder(path)
preview_token = seal_transient_payload(
{
"purpose": _ARCHIVE_PREVIEW_PURPOSE,
"tenant_id": principal.tenant_id,
"user_id": principal.user.id,
"owner_type": owner_type,
"owner_id": target_owner,
"path": normalized_path,
"campaign_id": campaign_id or "",
"archive_format": archive_format,
"archive_sha256": digest,
}
)
expires_at = datetime.now(UTC) + timedelta(
seconds=settings.file_archive_preview_ttl_seconds
)
return ArchivePreviewResponse(
preview_token=preview_token,
archive_format=inspection.archive_format,
entries=[
ArchiveEntryResponse(
path=entry.path,
kind=entry.kind,
size_bytes=entry.size_bytes,
compressed_size_bytes=entry.compressed_size_bytes,
encrypted=entry.encrypted,
)
for entry in inspection.entries
],
file_count=inspection.file_count,
directory_count=inspection.directory_count,
expanded_size_bytes=inspection.expanded_size_bytes,
compressed_size_bytes=inspection.compressed_size_bytes,
requires_password=inspection.requires_password,
password_verified=inspection.password_verified,
expires_at=expires_at.isoformat(),
)
except (FileStorageError, UnsafeFilePathError, ValueError) as exc:
raise _http_error(exc) from exc
finally:
if archive_path:
_cleanup_temp_file(archive_path)
@router.post("/archive-confirm", response_model=FileUploadResponse)
def confirm_archive_upload(
file: UploadFile = FastAPIFile(...),
preview_token: str = Form(...),
selected_paths_json: str = Form(...),
owner_type: Literal["user", "group"] = Form(default="user"),
owner_id: str | None = Form(default=None),
path: str = Form(default=""),
campaign_id: str | None = Form(default=None),
password: str | None = Form(default=None),
conflict_strategy: Literal["reject", "overwrite", "rename"] = Form(
default="reject"
),
conflict_resolutions_json: str | None = Form(default=None),
source_provenance_json: str | None = Form(default=None),
source_revision: str | None = Form(default=None),
connector_policy_json: str | None = Form(default=None),
encryption_vault_id: str | None = Form(default=None),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:upload")),
):
target_owner = owner_id or principal.user.id
archive_path: str | None = None
filename = file.filename or "archive"
try:
raw_resolutions = (
json.loads(conflict_resolutions_json)
if conflict_resolutions_json
else []
)
upload_resolutions = _conflict_resolutions(
[ConflictResolutionRequest(**item) for item in raw_resolutions]
)
selected_paths = _archive_selected_paths(selected_paths_json)
_enforce_connector_policy(
source_provenance_json,
connector_policy_json,
operation="import",
)
metadata = _source_metadata_from_form(
source_provenance_json, source_revision
)
archive_format = archive_format_for_filename(filename)
normalized_path = normalize_folder(path)
archive_path = _spool_limited_upload_to_temp(
file,
max_bytes=settings.file_upload_zip_max_bytes,
suffix=_archive_suffix(filename),
)
digest = _archive_sha256(archive_path)
_validate_archive_preview_token(
preview_token,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
owner_type=owner_type,
owner_id=target_owner,
path=normalized_path,
campaign_id=campaign_id,
archive_format=archive_format,
archive_sha256=digest,
)
extracted = extract_archive_upload(
session,
tenant_id=principal.tenant_id,
owner_type=owner_type,
owner_id=target_owner,
user_id=principal.user.id,
archive_data=archive_path,
filename=filename,
folder=normalized_path,
campaign_id=campaign_id,
selected_paths=selected_paths,
password=password,
conflict_strategy=conflict_strategy,
conflict_resolutions=upload_resolutions,
metadata=metadata,
is_admin=_is_admin(principal),
encryption_vault_id=encryption_vault_id,
max_entries=settings.file_archive_max_entries,
max_file_bytes=settings.file_upload_max_bytes,
max_expanded_bytes=settings.file_archive_max_expanded_bytes,
max_expansion_ratio=settings.file_archive_max_expansion_ratio,
)
uploaded_assets = [item.asset for item in extracted]
_audit_connector_imports(session, principal, uploaded_assets)
session.commit()
return FileUploadResponse(
files=[
_asset_response(session, asset, include_shares=True)
for asset in uploaded_assets
]
)
except ConnectorPolicyDenied as exc:
session.rollback()
raise _connector_policy_error(exc) from exc
except (
FileStorageError,
TransientPayloadError,
UnsafeFilePathError,
ValueError,
json.JSONDecodeError,
) as exc:
session.rollback()
raise _http_error(exc) from exc
finally:
if archive_path:
_cleanup_temp_file(archive_path)
@router.post("/upload", response_model=FileUploadResponse)
def upload_files(
files: list[UploadFile] = FastAPIFile(...),
owner_type: Literal["user", "group"] = Form(default="user"),
owner_id: str | None = Form(default=None),
path: str = Form(default=""),
campaign_id: str | None = Form(default=None),
unpack_zip: bool = Form(default=False),
conflict_strategy: Literal["reject", "overwrite", "rename"] = Form(
default="reject"
),
conflict_resolutions_json: str | None = Form(default=None),
source_provenance_json: str | None = Form(default=None),
source_revision: str | None = Form(default=None),
connector_policy_json: str | None = Form(default=None),
encryption_vault_id: str | None = Form(default=None),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:upload")),
):
target_owner = owner_id or principal.user.id
uploaded_assets: list[FileAsset] = []
try:
raw_resolutions = (
json.loads(conflict_resolutions_json) if conflict_resolutions_json else []
)
upload_resolutions = _conflict_resolutions(
[ConflictResolutionRequest(**item) for item in raw_resolutions]
)
_enforce_connector_policy(
source_provenance_json, connector_policy_json, operation="import"
)
metadata = _source_metadata_from_form(source_provenance_json, source_revision)
for upload in files:
filename = upload.filename or "file"
content_type = upload.content_type or None
upload_limit = (
settings.file_upload_zip_max_bytes
if unpack_zip and filename.lower().endswith(".zip")
else settings.file_upload_max_bytes
)
if unpack_zip and filename.lower().endswith(".zip"):
zip_path = _spool_limited_upload_to_temp(
upload, max_bytes=upload_limit, suffix=".zip"
)
try:
extracted = extract_zip_upload(
session,
tenant_id=principal.tenant_id,
owner_type=owner_type,
owner_id=target_owner,
user_id=principal.user.id,
zip_data=zip_path,
folder=path,
campaign_id=campaign_id,
conflict_strategy=conflict_strategy,
conflict_resolutions=upload_resolutions,
metadata=metadata,
is_admin=_is_admin(principal),
encryption_vault_id=encryption_vault_id,
max_file_bytes=settings.file_upload_max_bytes,
max_total_bytes=settings.file_upload_zip_max_bytes,
)
finally:
_cleanup_temp_file(zip_path)
uploaded_assets.extend(item.asset for item in extracted)
continue
data = _read_limited_upload(upload, max_bytes=upload_limit)
stored = create_file_asset(
session,
tenant_id=principal.tenant_id,
owner_type=owner_type,
owner_id=target_owner,
user_id=principal.user.id,
filename=filename,
data=data,
folder=path,
content_type=content_type,
campaign_id=campaign_id,
conflict_strategy=conflict_strategy,
conflict_resolutions=upload_resolutions,
metadata=metadata,
is_admin=_is_admin(principal),
encryption_vault_id=encryption_vault_id,
)
uploaded_assets.append(stored.asset)
_audit_connector_imports(session, principal, uploaded_assets)
session.commit()
except ConnectorPolicyDenied as exc:
session.rollback()
raise _connector_policy_error(exc) from exc
except (FileStorageError, UnsafeFilePathError, ValueError) as exc:
session.rollback()
raise _http_error(exc) from exc
return FileUploadResponse(
files=[
_asset_response(session, asset, include_shares=True)
for asset in uploaded_assets
]
)
@router.post("/upload-zip", response_model=FileUploadResponse)
def upload_zip(
file: UploadFile = FastAPIFile(...),
owner_type: Literal["user", "group"] = Form(default="user"),
owner_id: str | None = Form(default=None),
path: str = Form(default=""),
campaign_id: str | None = Form(default=None),
conflict_strategy: Literal["reject", "overwrite", "rename"] = Form(
default="reject"
),
conflict_resolutions_json: str | None = Form(default=None),
source_provenance_json: str | None = Form(default=None),
source_revision: str | None = Form(default=None),
connector_policy_json: str | None = Form(default=None),
encryption_vault_id: str | None = Form(default=None),
session: Session = Depends(get_session),
principal: ApiPrincipal = Depends(require_scope("files:file:upload")),
):
target_owner = owner_id or principal.user.id
zip_path: str | None = None
try:
raw_resolutions = (
json.loads(conflict_resolutions_json) if conflict_resolutions_json else []
)
upload_resolutions = _conflict_resolutions(
[ConflictResolutionRequest(**item) for item in raw_resolutions]
)
_enforce_connector_policy(
source_provenance_json, connector_policy_json, operation="import"
)
metadata = _source_metadata_from_form(source_provenance_json, source_revision)
zip_path = _spool_limited_upload_to_temp(
file, max_bytes=settings.file_upload_zip_max_bytes, suffix=".zip"
)
extracted = extract_zip_upload(
session,
tenant_id=principal.tenant_id,
owner_type=owner_type,
owner_id=target_owner,
user_id=principal.user.id,
zip_data=zip_path,
folder=path,
campaign_id=campaign_id,
conflict_strategy=conflict_strategy,
conflict_resolutions=upload_resolutions,
metadata=metadata,
is_admin=_is_admin(principal),
encryption_vault_id=encryption_vault_id,
max_file_bytes=settings.file_upload_max_bytes,
max_total_bytes=settings.file_upload_zip_max_bytes,
)
_audit_connector_imports(session, principal, [item.asset for item in extracted])
session.commit()
except ConnectorPolicyDenied as exc:
session.rollback()
raise _connector_policy_error(exc) from exc
except (FileStorageError, UnsafeFilePathError, ValueError) as exc:
session.rollback()
raise _http_error(exc) from exc
finally:
if zip_path:
_cleanup_temp_file(zip_path)
return FileUploadResponse(
files=[
_asset_response(session, item.asset, include_shares=True)
for item in extracted
]
)
+6 -32
View File
@@ -1,36 +1,10 @@
from __future__ import annotations
from typing import Any
from govoplan_core.core.runtime import ModuleRuntimeState
_runtime_registry: object | None = None
_runtime_settings: object | None = None
_runtime = ModuleRuntimeState("Files")
def configure_runtime(*, registry: object | None = None, settings: object | None = None) -> None:
global _runtime_registry, _runtime_settings
if registry is not None:
_runtime_registry = registry
if settings is not None:
_runtime_settings = settings
def get_registry() -> object | None:
return _runtime_registry
def get_settings() -> object:
if _runtime_settings is not None:
return _runtime_settings
try:
from govoplan_core.settings import settings as legacy_settings
except ModuleNotFoundError as exc:
raise RuntimeError("GovOPlaN Files runtime settings are not configured") from exc
return legacy_settings
class SettingsProxy:
def __getattr__(self, name: str) -> Any:
return getattr(get_settings(), name)
settings = SettingsProxy()
configure_runtime = _runtime.configure_runtime
get_registry = _runtime.get_registry
get_settings = _runtime.get_settings
settings = _runtime.settings
+114 -5
View File
@@ -1,5 +1,6 @@
from __future__ import annotations
from datetime import datetime
from typing import Any, Literal
from pydantic import BaseModel, Field
@@ -73,11 +74,93 @@ class FileConnectorSpacesResponse(BaseModel):
class FileShareResponse(BaseModel):
id: str
file_asset_id: str
target_type: str
target_id: str
permission: str
created_by_user_id: str | None = None
created_at: str
expires_at: str | None = None
revoked_at: str | None = None
revoked_by_user_id: str | None = None
active: bool
class FileSharesResponse(BaseModel):
shares: list[FileShareResponse] = Field(default_factory=list)
class FileIntegrityScanCreateRequest(BaseModel):
verify_checksums: bool = True
batch_size: int = Field(default=100, ge=1, le=1000)
class FileIntegrityScanResponse(BaseModel):
id: str
tenant_id: str
storage_backend: str
storage_prefix: str
status: str
revision: int
phase: str
verify_checksums: bool
batch_size: int
scanned_blob_count: int
verified_blob_count: int
quarantined_blob_count: int
scanned_object_count: int
orphan_object_count: int
created_by_user_id: str | None = None
started_at: str | None = None
completed_at: str | None = None
last_error: str | None = None
created_at: str
updated_at: str
class FileIntegrityScansResponse(BaseModel):
scans: list[FileIntegrityScanResponse] = Field(default_factory=list)
class FileIntegrityFindingResponse(BaseModel):
id: str
scan_id: str
tenant_id: str
kind: str
state: str
revision: int
blob_id: str | None = None
storage_key: str
expected_size_bytes: int | None = None
observed_size_bytes: int | None = None
expected_checksum_sha256: str | None = None
observed_checksum_sha256: str | None = None
resolved_at: str | None = None
resolved_by_user_id: str | None = None
created_at: str
updated_at: str
class FileIntegrityFindingsResponse(BaseModel):
findings: list[FileIntegrityFindingResponse] = Field(default_factory=list)
class FileIntegrityActionRequest(BaseModel):
dry_run: bool = True
expected_revision: int = Field(ge=1)
class FileIntegrityScanRunRequest(BaseModel):
expected_revision: int = Field(ge=1)
class FileIntegrityActionResponse(BaseModel):
action: str
changed: bool
dry_run: bool
finding: FileIntegrityFindingResponse
inspection_kind: str | None = None
inspection_valid: bool | None = None
class FileSourceProvenance(BaseModel):
@@ -154,6 +237,7 @@ class FileFolderDeleteResponse(BaseModel):
class FileListResponse(BaseModel):
files: list[FileAssetResponse]
total: int
cursor: str | None = None
next_cursor: str | None = None
watermark: str | None = None
@@ -172,6 +256,27 @@ class FileUploadResponse(BaseModel):
files: list[FileAssetResponse]
class ArchiveEntryResponse(BaseModel):
path: str
kind: Literal["file", "directory"]
size_bytes: int
compressed_size_bytes: int | None = None
encrypted: bool = False
class ArchivePreviewResponse(BaseModel):
preview_token: str
archive_format: str
entries: list[ArchiveEntryResponse] = Field(default_factory=list)
file_count: int
directory_count: int
expanded_size_bytes: int
compressed_size_bytes: int
requires_password: bool
password_verified: bool
expires_at: str
class FileConnectorPolicySource(BaseModel):
scope_type: Literal["system", "tenant", "user", "group", "campaign"] = "system"
scope_id: str | None = None
@@ -287,7 +392,7 @@ class FileConnectorDiscoveryCandidate(BaseModel):
class FileConnectorDiscoveryRequest(BaseModel):
provider: Literal["seafile", "nextcloud", "webdav", "smb", "nfs", "dms", "generic"]
provider: Literal["seafile", "nextcloud", "webdav", "smb", "s3", "sharepoint", "onedrive", "nfs", "dms", "generic"]
endpoint_url: str
base_path: str | None = None
credential_mode: Literal["none", "anonymous", "basic", "token", "secret_ref"] = "none"
@@ -309,7 +414,7 @@ class FileConnectorDiscoveryResponse(BaseModel):
class FileConnectorProfileCreateRequest(BaseModel):
id: str
label: str
provider: Literal["seafile", "nextcloud", "webdav", "smb", "nfs", "dms", "generic"]
provider: Literal["seafile", "nextcloud", "webdav", "smb", "s3", "sharepoint", "onedrive", "nfs", "dms", "generic"]
endpoint_url: str | None = None
base_path: str | None = None
enabled: bool = True
@@ -325,7 +430,7 @@ class FileConnectorProfileCreateRequest(BaseModel):
class FileConnectorProfileUpdateRequest(BaseModel):
label: str | None = None
provider: Literal["seafile", "nextcloud", "webdav", "smb", "nfs", "dms", "generic"] | None = None
provider: Literal["seafile", "nextcloud", "webdav", "smb", "s3", "sharepoint", "onedrive", "nfs", "dms", "generic"] | None = None
endpoint_url: str | None = None
base_path: str | None = None
enabled: bool | None = None
@@ -342,7 +447,7 @@ class FileConnectorProfileUpdateRequest(BaseModel):
class FileConnectorCredentialCreateRequest(BaseModel):
id: str
label: str
provider: Literal["seafile", "nextcloud", "webdav", "smb", "nfs", "dms", "generic"] | None = None
provider: Literal["seafile", "nextcloud", "webdav", "smb", "s3", "sharepoint", "onedrive", "nfs", "dms", "generic"] | None = None
enabled: bool = True
scope_type: Literal["system", "tenant", "user", "group", "campaign"] = "tenant"
scope_id: str | None = None
@@ -354,7 +459,7 @@ class FileConnectorCredentialCreateRequest(BaseModel):
class FileConnectorCredentialUpdateRequest(BaseModel):
label: str | None = None
provider: Literal["seafile", "nextcloud", "webdav", "smb", "nfs", "dms", "generic"] | None = None
provider: Literal["seafile", "nextcloud", "webdav", "smb", "s3", "sharepoint", "onedrive", "nfs", "dms", "generic"] | None = None
enabled: bool | None = None
credential_mode: Literal["none", "anonymous", "basic", "token", "secret_ref"] | None = None
credentials: FileConnectorProfileCredentialsRequest | None = None
@@ -404,6 +509,8 @@ class FileConnectorBrowseResponse(BaseModel):
path: str = ""
library_id: str | None = None
read_only: bool = True
next_continuation_token: str | None = None
has_more: bool = False
decision: dict[str, Any]
items: list[FileConnectorBrowseItem]
@@ -450,6 +557,7 @@ class FileShareRequest(BaseModel):
target_type: Literal["user", "group", "campaign", "tenant"]
target_id: str
permission: Literal["read", "write", "manage"] = "read"
expires_at: datetime | None = None
class BulkFileShareRequest(BaseModel):
@@ -457,6 +565,7 @@ class BulkFileShareRequest(BaseModel):
target_type: Literal["user", "group", "campaign", "tenant"]
target_id: str
permission: Literal["read", "write", "manage"] = "read"
expires_at: datetime | None = None
class BulkFileShareResponse(BaseModel):
+329
View File
@@ -0,0 +1,329 @@
from __future__ import annotations
from collections.abc import Mapping, Sequence
from urllib.parse import quote
from sqlalchemy import func, or_, select
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal
from govoplan_core.core.events import PlatformEvent
from govoplan_core.core.modules import ModuleContext
from govoplan_core.core.search import (
SearchAuthorizationRequest,
SearchBackfillPage,
SearchBackfillRequest,
SearchDocument,
SearchIndexChange,
SearchResourceReference,
SearchResourceType,
)
from govoplan_files.backend.db.models import FileAsset, FileFolder, FileShare
from govoplan_files.backend.storage.share_state import effective_file_share_clause
PROVIDER_ID = "files.objects"
RESOURCE_MODELS = {
"file": FileAsset,
"folder": FileFolder,
}
READ_SCOPE = "files:file:read"
ADMIN_SCOPE = "files:file:admin"
class FilesSearchSource:
def resource_types(self) -> Sequence[SearchResourceType]:
return (
SearchResourceType(
provider_id=PROVIDER_ID,
module_id="files",
resource_type="file",
label="Files",
requires_authorization_recheck=True,
),
SearchResourceType(
provider_id=PROVIDER_ID,
module_id="files",
resource_type="folder",
label="File folders",
requires_authorization_recheck=True,
),
)
def backfill(
self,
session: object,
*,
request: SearchBackfillRequest,
) -> SearchBackfillPage:
db = _session(session)
model = _model(request.provider_id, request.resource_type)
statement = select(model).where(
model.tenant_id == request.tenant_id,
model.deleted_at.is_(None),
)
if request.cursor:
statement = statement.where(model.id > request.cursor)
rows = list(
db.scalars(
statement.order_by(model.id).limit(request.limit + 1)
)
)
has_more = len(rows) > request.limit
selected = rows[: request.limit]
shares = (
_shares_by_asset(db, selected)
if request.resource_type == "file"
else {}
)
high_watermark = db.scalar(
select(func.max(model.updated_at)).where(
model.tenant_id == request.tenant_id,
model.deleted_at.is_(None),
)
)
return SearchBackfillPage(
documents=tuple(
_document(
row,
resource_type=request.resource_type,
shares=shares.get(row.id, ()),
)
for row in selected
),
next_cursor=selected[-1].id if has_more and selected else None,
complete=not has_more,
high_watermark=(
high_watermark.isoformat()
if high_watermark is not None
else None
),
)
def authorize(
self,
session: object,
principal: object,
*,
requests: Sequence[SearchAuthorizationRequest],
) -> Mapping[str, bool]:
decisions = {item.reference.key: False for item in requests}
if not isinstance(principal, ApiPrincipal) or not (
principal.has(READ_SCOPE) or principal.has(ADMIN_SCOPE)
):
return decisions
db = _session(session)
for request in requests:
reference = request.reference
if (
reference.tenant_id != principal.tenant_id
or reference.module_id != "files"
or reference.resource_type not in RESOURCE_MODELS
):
continue
decisions[reference.key] = _can_read(
db,
principal,
reference=reference,
)
return decisions
def index_changes_for_event(
self,
session: object,
*,
event: PlatformEvent,
delivery_key: str,
) -> Sequence[SearchIndexChange]:
if (
event.module_id != "files"
or event.tenant is None
or event.resource is None
or event.resource.id is None
or event.resource.type not in RESOURCE_MODELS
):
return ()
db = _session(session)
reference = SearchResourceReference(
tenant_id=event.tenant.id,
module_id="files",
resource_type=event.resource.type,
resource_id=event.resource.id,
)
model = RESOURCE_MODELS[event.resource.type]
row = db.get(model, event.resource.id)
deleted = row is None or row.tenant_id != event.tenant.id or row.deleted_at is not None
cursor = event.event_id
document = None
if not deleted:
shares = (
tuple(_active_shares(db, row.id))
if event.resource.type == "file"
else ()
)
document = _document(
row,
resource_type=event.resource.type,
shares=shares,
change_cursor=cursor,
)
return (
SearchIndexChange(
change_id=f"{delivery_key}:{PROVIDER_ID}:{event.resource.type}",
provider_id=PROVIDER_ID,
kind="delete" if deleted else "upsert",
reference=reference,
source_revision=(
document.source_revision if document is not None else cursor
),
cursor=cursor,
document=document,
occurred_at=event.occurred_at,
),
)
def create_files_search_source(_context: ModuleContext) -> FilesSearchSource:
return FilesSearchSource()
def _model(provider_id: str, resource_type: str):
if provider_id != PROVIDER_ID or resource_type not in RESOURCE_MODELS:
raise ValueError("Unsupported Files search source.")
return RESOURCE_MODELS[resource_type]
def _document(
row: FileAsset | FileFolder,
*,
resource_type: str,
shares: Sequence[FileShare] = (),
change_cursor: str | None = None,
) -> SearchDocument:
is_file = isinstance(row, FileAsset)
title = row.filename if is_file else (row.path.rsplit("/", 1)[-1] or row.path)
path = row.display_path if is_file else row.path
owner_id = row.owner_user_id if row.owner_type == "user" else row.owner_group_id
tokens = [f"scope:{READ_SCOPE}", f"scope:{ADMIN_SCOPE}"]
if owner_id:
tokens.append(
f"membership:{owner_id}"
if row.owner_type == "user"
else f"group:{owner_id}"
)
for share in shares:
prefix = "membership" if share.target_type == "user" else share.target_type
if prefix in {"membership", "group", "tenant"}:
tokens.append(f"{prefix}:{share.target_id}")
updated_at = row.updated_at or row.created_at
revision = (
f"{row.current_version_id or 'none'}:{updated_at.isoformat()}"
if is_file
else updated_at.isoformat()
)
return SearchDocument(
tenant_id=row.tenant_id,
module_id="files",
provider_id=PROVIDER_ID,
resource_type=resource_type,
resource_id=row.id,
title=title,
url=f"/files?{resource_type}Id={quote(row.id, safe='')}",
summary=((row.description or "") if is_file else path)[:4000] or None,
body=" ".join(
value for value in (path, row.description if is_file else None) if value
)[:200_000],
keywords=(path[:200], row.owner_type[:200]),
visibility="restricted",
acl_tokens=tuple(dict.fromkeys(tokens)),
metadata={
"path": path,
"owner_type": row.owner_type,
"current_version_id": row.current_version_id if is_file else None,
},
source_revision=revision,
change_cursor=change_cursor,
source_updated_at=updated_at,
requires_authorization_recheck=True,
)
def _can_read(
session: Session,
principal: ApiPrincipal,
*,
reference: SearchResourceReference,
) -> bool:
model = RESOURCE_MODELS[reference.resource_type]
row = session.get(model, reference.resource_id)
if row is None or row.tenant_id != principal.tenant_id or row.deleted_at is not None:
return False
if principal.has(ADMIN_SCOPE):
return True
user_id = str(getattr(principal.user, "id", "") or principal.membership_id or "")
if row.owner_type == "user" and row.owner_user_id == user_id:
return True
if row.owner_type == "group" and row.owner_group_id in principal.group_ids:
return True
if reference.resource_type != "file":
return False
target_clauses = [
(FileShare.target_type == "user") & (FileShare.target_id == user_id),
(FileShare.target_type == "tenant")
& (FileShare.target_id == principal.tenant_id),
]
if principal.group_ids:
target_clauses.append(
(FileShare.target_type == "group")
& (FileShare.target_id.in_(tuple(principal.group_ids)))
)
return session.scalar(
select(FileShare.id).where(
FileShare.tenant_id == principal.tenant_id,
FileShare.file_asset_id == row.id,
effective_file_share_clause(),
or_(*target_clauses),
).limit(1)
) is not None
def _active_shares(session: Session, asset_id: str) -> Sequence[FileShare]:
return tuple(
session.scalars(
select(FileShare).where(
FileShare.file_asset_id == asset_id,
effective_file_share_clause(),
)
)
)
def _shares_by_asset(
session: Session,
rows: Sequence[FileAsset | FileFolder],
) -> dict[str, tuple[FileShare, ...]]:
asset_ids = [row.id for row in rows if isinstance(row, FileAsset)]
grouped: dict[str, list[FileShare]] = {asset_id: [] for asset_id in asset_ids}
if not asset_ids:
return {}
for share in session.scalars(
select(FileShare).where(
FileShare.file_asset_id.in_(asset_ids),
effective_file_share_clause(),
)
):
grouped.setdefault(share.file_asset_id, []).append(share)
return {key: tuple(value) for key, value in grouped.items()}
def _session(value: object) -> Session:
if not isinstance(value, Session):
raise TypeError("Files search requires a SQLAlchemy session.")
return value
__all__ = [
"FilesSearchSource",
"PROVIDER_ID",
"create_files_search_source",
]
@@ -0,0 +1 @@
"""Query and response assembly services used by Files routes."""
@@ -0,0 +1,292 @@
from __future__ import annotations
from typing import Literal
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal
from govoplan_core.api.v1.schemas import DeltaDeletedItem
from govoplan_core.core.change_sequence import (
ChangeSequenceEntry,
)
from govoplan_files.backend.change_tracking import (
FILES_CONNECTOR_CREDENTIALS_COLLECTION,
FILES_CONNECTOR_POLICIES_COLLECTION,
FILES_CONNECTOR_PROFILES_COLLECTION,
FILES_CONNECTOR_SPACES_COLLECTION,
)
from govoplan_files.backend.schemas import (
FileConnectorCredentialResponse,
FileConnectorSettingsDeltaResponse,
FileConnectorPolicyResponse,
FileConnectorProfileResponse,
)
from govoplan_files.backend.db.models import FileConnectorSpace
from govoplan_files.backend.storage.connector_credential_store import (
ConnectorCredential,
)
from govoplan_files.backend.storage.connector_profiles import ConnectorProfile
from govoplan_files.backend.storage.connector_policy_store import (
connector_policy_response,
)
from govoplan_files.backend.route_support import (
FILES_CONNECTOR_CREDENTIAL_RESOURCE,
FILES_CONNECTOR_PROFILE_RESOURCE,
FILES_CONNECTOR_SPACE_RESOURCE,
_can_read_disabled_connector_profiles,
_connector_deleted_item,
_connector_space_response,
_ensure_campaign_file_access,
_file_connector_settings_response_watermark,
_file_connector_settings_watermark,
_visible_connector_credentials,
_visible_connector_profiles,
_visible_connector_spaces,
)
def _full_file_connector_settings_delta_response(
session: Session,
principal: ApiPrincipal,
*,
scope_type: str,
scope_id: str | None,
provider: str | None,
campaign_id: str | None,
include_disabled: bool,
include_inactive: bool,
owner_type: Literal["user", "group"] | None,
owner_id: str | None,
) -> FileConnectorSettingsDeltaResponse:
if campaign_id:
_ensure_campaign_file_access(session, principal, campaign_id)
profiles = _visible_connector_profiles(
session,
principal,
provider=provider,
campaign_id=campaign_id,
include_disabled=include_disabled
and _can_read_disabled_connector_profiles(principal),
include_admin_scopes=_can_read_disabled_connector_profiles(principal),
include_effective_policy=False,
)
credentials = _visible_connector_credentials(
session,
principal,
provider=provider,
include_disabled=include_disabled,
)
spaces = _visible_connector_spaces(
session,
principal,
owner_type=owner_type,
owner_id=owner_id,
include_inactive=include_inactive,
)
return FileConnectorSettingsDeltaResponse(
profiles=[
FileConnectorProfileResponse(**profile.to_response())
for profile in profiles
],
credentials=[
FileConnectorCredentialResponse(**credential.to_response())
for credential in credentials
],
spaces=[_connector_space_response(space) for space in spaces],
policy=FileConnectorPolicyResponse(
**connector_policy_response(
session,
tenant_id=principal.tenant_id,
scope_type=scope_type,
scope_id=scope_id,
)
),
changed_sections=["profiles", "credentials", "spaces", "policy"],
deleted=[],
watermark=_file_connector_settings_watermark(
session, tenant_id=principal.tenant_id
),
has_more=False,
full=True,
)
def _changed_file_connector_setting_ids(
entries: list[ChangeSequenceEntry],
) -> tuple[set[str], set[str], set[str], bool]:
changed_profile_ids = {
entry.resource_id
for entry in entries
if entry.collection == FILES_CONNECTOR_PROFILES_COLLECTION
and entry.resource_type == FILES_CONNECTOR_PROFILE_RESOURCE
}
changed_credential_ids = {
entry.resource_id
for entry in entries
if entry.collection == FILES_CONNECTOR_CREDENTIALS_COLLECTION
and entry.resource_type == FILES_CONNECTOR_CREDENTIAL_RESOURCE
}
changed_space_ids = {
entry.resource_id
for entry in entries
if entry.collection == FILES_CONNECTOR_SPACES_COLLECTION
and entry.resource_type == FILES_CONNECTOR_SPACE_RESOURCE
}
policy_changed = any(
entry.collection == FILES_CONNECTOR_POLICIES_COLLECTION for entry in entries
)
return (
changed_profile_ids,
changed_credential_ids,
changed_space_ids,
policy_changed,
)
def _file_connector_settings_changed_sections(
*,
changed_profile_ids: set[str],
changed_credential_ids: set[str],
changed_space_ids: set[str],
policy_changed: bool,
profiles: list[ConnectorProfile],
) -> list[str]:
changed_sections = []
if changed_profile_ids or any(
profile.credential_profile_id
and profile.credential_profile_id in changed_credential_ids
for profile in profiles
):
changed_sections.append("profiles")
if changed_credential_ids:
changed_sections.append("credentials")
if changed_space_ids:
changed_sections.append("spaces")
if policy_changed:
changed_sections.append("policy")
return changed_sections
def _file_connector_settings_deleted_items(
entries: list[ChangeSequenceEntry],
*,
visible_profiles: dict[str, ConnectorProfile],
visible_credentials: dict[str, ConnectorCredential],
visible_spaces: dict[str, FileConnectorSpace],
) -> list[DeltaDeletedItem]:
return [
_connector_deleted_item(entry)
for entry in entries
if (
entry.resource_type == FILES_CONNECTOR_PROFILE_RESOURCE
and entry.resource_id not in visible_profiles
)
or (
entry.resource_type == FILES_CONNECTOR_CREDENTIAL_RESOURCE
and entry.resource_id not in visible_credentials
)
or (
entry.resource_type == FILES_CONNECTOR_SPACE_RESOURCE
and entry.resource_id not in visible_spaces
)
]
def _incremental_file_connector_settings_delta_response(
session: Session,
principal: ApiPrincipal,
*,
entries: list[ChangeSequenceEntry],
has_more: bool,
scope_type: str,
scope_id: str | None,
provider: str | None,
campaign_id: str | None,
include_disabled: bool,
include_inactive: bool,
owner_type: Literal["user", "group"] | None,
owner_id: str | None,
) -> FileConnectorSettingsDeltaResponse:
changed_profile_ids, changed_credential_ids, changed_space_ids, policy_changed = (
_changed_file_connector_setting_ids(entries)
)
profiles = _visible_connector_profiles(
session,
principal,
provider=provider,
campaign_id=campaign_id,
include_disabled=include_disabled
and _can_read_disabled_connector_profiles(principal),
include_admin_scopes=_can_read_disabled_connector_profiles(principal),
include_effective_policy=False,
)
visible_profiles = {
profile.id: profile
for profile in profiles
if profile.id in changed_profile_ids
or (
profile.credential_profile_id
and profile.credential_profile_id in changed_credential_ids
)
}
credentials = _visible_connector_credentials(
session,
principal,
provider=provider,
include_disabled=include_disabled,
)
visible_credentials = {
credential.id: credential
for credential in credentials
if credential.id in changed_credential_ids
}
spaces = _visible_connector_spaces(
session,
principal,
owner_type=owner_type,
owner_id=owner_id,
include_inactive=include_inactive,
)
visible_spaces = {
space.id: space for space in spaces if space.id in changed_space_ids
}
return FileConnectorSettingsDeltaResponse(
profiles=[
FileConnectorProfileResponse(**profile.to_response())
for profile in visible_profiles.values()
],
credentials=[
FileConnectorCredentialResponse(**credential.to_response())
for credential in visible_credentials.values()
],
spaces=[_connector_space_response(space) for space in visible_spaces.values()],
policy=FileConnectorPolicyResponse(
**connector_policy_response(
session,
tenant_id=principal.tenant_id,
scope_type=scope_type,
scope_id=scope_id,
)
)
if policy_changed
else None,
changed_sections=_file_connector_settings_changed_sections(
changed_profile_ids=changed_profile_ids,
changed_credential_ids=changed_credential_ids,
changed_space_ids=changed_space_ids,
policy_changed=policy_changed,
profiles=profiles,
),
deleted=_file_connector_settings_deleted_items(
entries,
visible_profiles=visible_profiles,
visible_credentials=visible_credentials,
visible_spaces=visible_spaces,
),
watermark=_file_connector_settings_response_watermark(
session, tenant_id=principal.tenant_id, entries=entries, has_more=has_more
),
has_more=has_more,
full=False,
)
@@ -0,0 +1,620 @@
from __future__ import annotations
from datetime import datetime
from typing import Literal
from fastapi import HTTPException, status
from sqlalchemy.orm import Session
from govoplan_core.auth import ApiPrincipal
from govoplan_core.api.v1.schemas import DeltaDeletedItem
from govoplan_core.core.change_sequence import (
ChangeSequenceEntry,
decode_sequence_watermark,
encode_sequence_watermark,
max_sequence_id,
sequence_entries_since,
sequence_watermark_is_expired,
)
from govoplan_core.core.pagination import (
KeysetCursorError,
decode_keyset_cursor,
encode_keyset_cursor,
keyset_query_fingerprint,
)
from govoplan_files.backend.change_tracking import (
FILES_ASSETS_COLLECTION,
FILES_FOLDERS_COLLECTION,
FILES_MODULE_ID,
)
from govoplan_files.backend.schemas import (
FileDeltaResponse,
)
from govoplan_files.backend.db.models import FileAsset, FileFolder, FileShare
from govoplan_files.backend.storage.access import user_group_ids
from govoplan_files.backend.storage.files import (
list_assets_for_user,
)
from govoplan_files.backend.storage.folders import list_folders_for_user
from govoplan_files.backend.storage.share_state import effective_file_share_clause
from govoplan_files.backend.route_support import (
_asset_list_response,
_folder_response,
_is_admin,
)
_FILES_DELTA_COLLECTIONS = (FILES_ASSETS_COLLECTION, FILES_FOLDERS_COLLECTION)
FILES_LIST_CURSOR_SCOPE = "files.list.v1"
FOLDERS_LIST_CURSOR_SCOPE = "files.folders.list.v1"
DEFAULT_FILE_LIST_PAGE_SIZE = 500
def _cursor_http_error(exc: Exception) -> HTTPException:
return HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(exc))
def _cursor_page_size(
scope: str, cursor: str | None, explicit_page_size: int | None
) -> int | None:
if explicit_page_size is not None:
return explicit_page_size
if not cursor:
return DEFAULT_FILE_LIST_PAGE_SIZE
try:
values = decode_keyset_cursor(scope, cursor)
except KeysetCursorError as exc:
raise _cursor_http_error(exc) from exc
raw_page_size = values.get("page_size")
if not isinstance(raw_page_size, int) or raw_page_size < 1 or raw_page_size > 1000:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST, detail="Invalid pagination cursor"
)
return raw_page_size
def _files_list_fingerprint(
principal: ApiPrincipal,
*,
owner_type: Literal["user", "group"] | None,
owner_id: str | None,
campaign_id: str | None,
path_prefix: str | None,
campaign_usage: Literal["linked", "unlinked"] | None,
audit_relevant: bool | None,
page_size: int,
) -> str:
return keyset_query_fingerprint(
FILES_LIST_CURSOR_SCOPE,
{
"tenant_id": principal.tenant_id,
"actor": "admin" if _is_admin(principal) else principal.user.id,
"owner_type": owner_type,
"owner_id": owner_id,
"campaign_id": campaign_id,
"path_prefix": path_prefix or "",
"campaign_usage": campaign_usage or "",
"audit_relevant": audit_relevant,
"sort": "display_path.asc,updated_at.desc,id.asc",
"page_size": page_size,
},
)
def _folders_list_fingerprint(
principal: ApiPrincipal,
*,
owner_type: Literal["user", "group"],
owner_id: str,
page_size: int,
) -> str:
return keyset_query_fingerprint(
FOLDERS_LIST_CURSOR_SCOPE,
{
"tenant_id": principal.tenant_id,
"actor": "admin" if _is_admin(principal) else principal.user.id,
"owner_type": owner_type,
"owner_id": owner_id,
"sort": "path.asc,id.asc",
"page_size": page_size,
},
)
def _file_cursor_values(
cursor: str | None, *, fingerprint: str
) -> tuple[str | None, datetime | None, str | None]:
try:
values = decode_keyset_cursor(
FILES_LIST_CURSOR_SCOPE, cursor, fingerprint=fingerprint
)
except KeysetCursorError as exc:
raise _cursor_http_error(exc) from exc
if values is None:
return None, None, None
display_path = values.get("display_path")
updated_at = values.get("updated_at")
asset_id = values.get("id")
if (
not isinstance(display_path, str)
or not isinstance(updated_at, str)
or not isinstance(asset_id, str)
):
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST, detail="Invalid pagination cursor"
)
try:
parsed_updated_at = datetime.fromisoformat(updated_at)
except ValueError as exc:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST, detail="Invalid pagination cursor"
) from exc
return display_path, parsed_updated_at, asset_id
def _folder_cursor_values(
cursor: str | None, *, fingerprint: str
) -> tuple[str | None, str | None]:
try:
values = decode_keyset_cursor(
FOLDERS_LIST_CURSOR_SCOPE, cursor, fingerprint=fingerprint
)
except KeysetCursorError as exc:
raise _cursor_http_error(exc) from exc
if values is None:
return None, None
folder_path = values.get("path")
folder_id = values.get("id")
if not isinstance(folder_path, str) or not isinstance(folder_id, str):
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST, detail="Invalid pagination cursor"
)
return folder_path, folder_id
def _next_file_list_cursor(
principal: ApiPrincipal,
assets: list[FileAsset],
*,
owner_type: Literal["user", "group"] | None,
owner_id: str | None,
campaign_id: str | None,
path_prefix: str | None,
campaign_usage: Literal["linked", "unlinked"] | None,
audit_relevant: bool | None,
page_size: int,
has_more: bool,
) -> str | None:
if not has_more or not assets:
return None
last = assets[-1]
return encode_keyset_cursor(
FILES_LIST_CURSOR_SCOPE,
fingerprint=_files_list_fingerprint(
principal,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
campaign_usage=campaign_usage,
audit_relevant=audit_relevant,
page_size=page_size,
),
values={
"display_path": last.display_path,
"updated_at": last.updated_at,
"id": last.id,
"page_size": page_size,
},
)
def _next_folder_list_cursor(
principal: ApiPrincipal,
folders: list[FileFolder],
*,
owner_type: Literal["user", "group"],
owner_id: str,
page_size: int,
has_more: bool,
) -> str | None:
if not has_more or not folders:
return None
last = folders[-1]
return encode_keyset_cursor(
FOLDERS_LIST_CURSOR_SCOPE,
fingerprint=_folders_list_fingerprint(
principal, owner_type=owner_type, owner_id=owner_id, page_size=page_size
),
values={"path": last.path, "id": last.id, "page_size": page_size},
)
def _files_delta_watermark(session: Session, tenant_id: str) -> str:
return encode_sequence_watermark(
max_sequence_id(
session,
tenant_id=tenant_id,
module_id=FILES_MODULE_ID,
collections=_FILES_DELTA_COLLECTIONS,
)
)
def _full_file_delta_response(
session: Session,
*,
principal: ApiPrincipal,
owner_type: Literal["user", "group"] | None,
owner_id: str | None,
campaign_id: str | None,
path_prefix: str | None,
) -> FileDeltaResponse:
assets = list_assets_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
is_admin=_is_admin(principal),
)
folders = _visible_folders_for_delta(
session,
principal=principal,
owner_type=owner_type,
owner_id=owner_id,
path_prefix=path_prefix,
)
return FileDeltaResponse(
files=_asset_list_response(session, assets, include_shares=True),
folders=[_folder_response(folder) for folder in folders],
deleted=[],
watermark=_files_delta_watermark(session, principal.tenant_id),
has_more=False,
full=True,
)
def _visible_folders_for_delta(
session: Session,
*,
principal: ApiPrincipal,
owner_type: Literal["user", "group"] | None,
owner_id: str | None,
path_prefix: str | None,
) -> list[FileFolder]:
if not owner_type or not owner_id:
return []
folders = list_folders_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
owner_type=owner_type,
owner_id=owner_id,
is_admin=_is_admin(principal),
)
if not path_prefix:
return folders
normalized = path_prefix.strip().strip("/")
if not normalized:
return folders
return [
folder
for folder in folders
if folder.path == normalized or folder.path.startswith(f"{normalized}/")
]
def _entry_path_matches(
entry_payload: dict[str, object], path_prefix: str | None
) -> bool:
if not path_prefix:
return True
normalized = path_prefix.strip().strip("/")
if not normalized:
return True
for key in ("path", "previous_path"):
value = entry_payload.get(key)
if isinstance(value, str) and (
value == normalized or value.startswith(f"{normalized}/")
):
return True
return False
def _entry_owner_matches(
entry_payload: dict[str, object],
owner_type: Literal["user", "group"] | None,
owner_id: str | None,
) -> bool:
if not owner_type or not owner_id:
return True
return (
entry_payload.get("owner_type") == owner_type
and entry_payload.get("owner_id") == owner_id
)
def _entry_campaign_matches(session: Session, entry, campaign_id: str | None) -> bool:
if not campaign_id:
return True
payload = entry.payload or {}
if (
payload.get("share_target_type") == "campaign"
and payload.get("share_target_id") == campaign_id
):
return True
if entry.resource_type != "file":
return False
return (
session.query(FileShare)
.filter(
FileShare.tenant_id == entry.tenant_id,
FileShare.file_asset_id == entry.resource_id,
FileShare.target_type == "campaign",
FileShare.target_id == campaign_id,
effective_file_share_clause(),
)
.first()
is not None
)
def _principal_group_ids_for_delta(
session: Session, principal: ApiPrincipal
) -> set[str]:
cache = session.info.setdefault("files_delta_group_ids", {})
key = (principal.tenant_id, principal.user.id)
if key not in cache:
cache[key] = set(
user_group_ids(
session, tenant_id=principal.tenant_id, user_id=principal.user.id
)
)
return cache[key]
def _entry_subject_matches_principal(
session: Session, principal: ApiPrincipal, entry_payload: dict[str, object]
) -> bool:
if _is_admin(principal):
return True
owner_type = entry_payload.get("owner_type")
owner_id = entry_payload.get("owner_id")
if owner_type == "user" and owner_id == principal.user.id:
return True
if owner_type == "group" and isinstance(owner_id, str):
if owner_id in _principal_group_ids_for_delta(session, principal):
return True
share_target_type = entry_payload.get("share_target_type")
share_target_id = entry_payload.get("share_target_id")
if share_target_type == "user" and share_target_id == principal.user.id:
return True
if share_target_type == "tenant" and share_target_id == principal.tenant_id:
return True
if share_target_type == "group" and isinstance(share_target_id, str):
if share_target_id in _principal_group_ids_for_delta(session, principal):
return True
return False
def _entry_matches_delta_scope(
session: Session,
principal: ApiPrincipal,
entry,
*,
owner_type: Literal["user", "group"] | None,
owner_id: str | None,
campaign_id: str | None,
path_prefix: str | None,
) -> bool:
payload = entry.payload or {}
if (
not owner_type
and not campaign_id
and not _entry_subject_matches_principal(session, principal, payload)
):
return False
return (
_entry_owner_matches(payload, owner_type, owner_id)
and _entry_path_matches(payload, path_prefix)
and _entry_campaign_matches(session, entry, campaign_id)
)
def _decode_files_delta_watermark(since: str) -> int:
try:
return decode_sequence_watermark(since)
except ValueError as exc:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST, detail=str(exc)
) from exc
def _changed_delta_resource_ids(
entries: list[ChangeSequenceEntry],
) -> tuple[list[str], list[str]]:
file_ids = list(
dict.fromkeys(
entry.resource_id for entry in entries if entry.resource_type == "file"
)
)
folder_ids = list(
dict.fromkeys(
entry.resource_id for entry in entries if entry.resource_type == "folder"
)
)
return file_ids, folder_ids
def _visible_assets_for_delta(
session: Session,
*,
principal: ApiPrincipal,
owner_type: Literal["user", "group"] | None,
owner_id: str | None,
campaign_id: str | None,
path_prefix: str | None,
changed_file_ids: list[str],
) -> dict[str, FileAsset]:
return {
asset.id: asset
for asset in list_assets_for_user(
session,
tenant_id=principal.tenant_id,
user_id=principal.user.id,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
is_admin=_is_admin(principal),
)
if asset.id in changed_file_ids
}
def _changed_visible_folders_for_delta(
session: Session,
*,
principal: ApiPrincipal,
owner_type: Literal["user", "group"] | None,
owner_id: str | None,
path_prefix: str | None,
changed_folder_ids: list[str],
) -> dict[str, FileFolder]:
return {
folder.id: folder
for folder in _visible_folders_for_delta(
session,
principal=principal,
owner_type=owner_type,
owner_id=owner_id,
path_prefix=path_prefix,
)
if folder.id in changed_folder_ids
}
def _deleted_delta_items(
session: Session,
*,
principal: ApiPrincipal,
entries: list[ChangeSequenceEntry],
visible_assets: dict[str, FileAsset],
visible_folders: dict[str, FileFolder],
owner_type: Literal["user", "group"] | None,
owner_id: str | None,
campaign_id: str | None,
path_prefix: str | None,
) -> list[DeltaDeletedItem]:
deleted: dict[tuple[str, str], DeltaDeletedItem] = {}
for entry in entries:
if entry.resource_type == "file" and entry.resource_id in visible_assets:
continue
if entry.resource_type == "folder" and entry.resource_id in visible_folders:
continue
if not _entry_matches_delta_scope(
session,
principal,
entry,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
):
continue
deleted[(entry.resource_type, entry.resource_id)] = DeltaDeletedItem(
id=entry.resource_id,
resource_type=entry.resource_type,
revision=encode_sequence_watermark(entry.id),
deleted_at=entry.created_at if entry.operation == "deleted" else None,
)
return list(deleted.values())
def _files_delta_response(
session: Session,
*,
principal: ApiPrincipal,
owner_type: Literal["user", "group"] | None,
owner_id: str | None,
campaign_id: str | None,
path_prefix: str | None,
since: str,
limit: int,
) -> FileDeltaResponse:
since_sequence = _decode_files_delta_watermark(since)
if sequence_watermark_is_expired(
session,
since=since_sequence,
tenant_id=principal.tenant_id,
module_id=FILES_MODULE_ID,
collections=_FILES_DELTA_COLLECTIONS,
):
return _full_file_delta_response(
session,
principal=principal,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
)
entries_plus_one = sequence_entries_since(
session,
since=since_sequence,
tenant_id=principal.tenant_id,
module_id=FILES_MODULE_ID,
collections=_FILES_DELTA_COLLECTIONS,
limit=limit + 1,
)
has_more = len(entries_plus_one) > limit
entries = entries_plus_one[:limit]
changed_file_ids, changed_folder_ids = _changed_delta_resource_ids(entries)
visible_assets = _visible_assets_for_delta(
session,
principal=principal,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
changed_file_ids=changed_file_ids,
)
visible_folders = _changed_visible_folders_for_delta(
session,
principal=principal,
owner_type=owner_type,
owner_id=owner_id,
path_prefix=path_prefix,
changed_folder_ids=changed_folder_ids,
)
deleted = _deleted_delta_items(
session,
principal=principal,
entries=entries,
visible_assets=visible_assets,
visible_folders=visible_folders,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
)
watermark = (
encode_sequence_watermark(entries[-1].id)
if has_more and entries
else _files_delta_watermark(session, principal.tenant_id)
)
return FileDeltaResponse(
files=_asset_list_response(
session, list(visible_assets.values()), include_shares=True
),
folders=[_folder_response(folder) for folder in visible_folders.values()],
deleted=deleted,
watermark=watermark,
has_more=has_more,
full=False,
)
+609 -77
View File
@@ -1,54 +1,91 @@
from __future__ import annotations
import mimetypes
import re
import stat
import tarfile
import zipfile
from dataclasses import dataclass
from io import BytesIO
from os import PathLike
from pathlib import Path
from typing import Any, Iterable
from pathlib import Path, PurePosixPath
from typing import Any, BinaryIO, Iterable, Literal
import pyzipper
from sqlalchemy.orm import Session
from govoplan_files.backend.db.models import FileAsset
from govoplan_files.backend.storage.common import FileConflictResolution, FileStorageError, UploadedStoredFile
from govoplan_files.backend.storage.backends import StorageBackendError, get_storage_backend
from govoplan_files.backend.storage.files import create_file_asset, current_versions_and_blobs
from govoplan_files.backend.storage.paths import filename_from_path, normalize_folder, normalize_logical_path
from govoplan_files.backend.storage.backends import (
StorageBackendError,
get_storage_backend,
)
from govoplan_files.backend.storage.common import (
FileConflictResolution,
FileStorageError,
UploadedStoredFile,
)
from govoplan_files.backend.storage.files import (
create_file_asset,
current_versions_and_blobs,
)
from govoplan_files.backend.storage.paths import (
filename_from_path,
normalize_folder,
normalize_logical_path,
)
_ZIP_READ_CHUNK_SIZE = 1024 * 1024
_ARCHIVE_READ_CHUNK_SIZE = 1024 * 1024
_WINDOWS_DRIVE_RE = re.compile(r"^[A-Za-z]:")
ARCHIVE_UPLOAD_MAX_ENTRIES = 10_000
# Kept for callers and documentation using the former ZIP-specific name.
ZIP_UPLOAD_MAX_FILES = ARCHIVE_UPLOAD_MAX_ENTRIES
SUPPORTED_ARCHIVE_SUFFIXES = (
".tar.bz2",
".tar.gz",
".tar.xz",
".tbz2",
".tgz",
".txz",
".tar",
".zip",
)
def _read_zip_member(
archive: zipfile.ZipFile,
info: zipfile.ZipInfo,
*,
max_file_bytes: int,
max_total_bytes: int,
current_total: int,
) -> tuple[bytes, int]:
parts: list[bytes] = []
actual_size = 0
with archive.open(info) as source:
while True:
read_size = min(_ZIP_READ_CHUNK_SIZE, max_file_bytes + 1 - actual_size)
chunk = source.read(read_size)
if not chunk:
break
actual_size += len(chunk)
if actual_size > max_file_bytes:
raise FileStorageError(f"ZIP member {info.filename!r} exceeds per-file limit")
if current_total + actual_size > max_total_bytes:
raise FileStorageError("ZIP is too large after extraction")
parts.append(chunk)
return b"".join(parts), current_total + actual_size
class ArchivePasswordError(FileStorageError):
pass
def create_zip_file(session: Session, assets: Iterable[FileAsset], output_path: str | Path) -> None:
@dataclass(frozen=True, slots=True)
class ArchiveEntry:
path: str
kind: Literal["file", "directory"]
size_bytes: int
compressed_size_bytes: int | None = None
encrypted: bool = False
@dataclass(frozen=True, slots=True)
class ArchiveInspection:
archive_format: str
entries: tuple[ArchiveEntry, ...]
file_count: int
directory_count: int
expanded_size_bytes: int
compressed_size_bytes: int
requires_password: bool
password_verified: bool
def create_zip_file(
session: Session, assets: Iterable[FileAsset], output_path: str | Path
) -> None:
backend = get_storage_backend()
asset_list = list(assets)
version_blobs = current_versions_and_blobs(session, asset_list)
with zipfile.ZipFile(output_path, mode="w", compression=zipfile.ZIP_DEFLATED) as archive:
with zipfile.ZipFile(
output_path, mode="w", compression=zipfile.ZIP_DEFLATED
) as archive:
for asset in asset_list:
_version, blob = version_blobs[asset.id]
info = zipfile.ZipInfo(asset.display_path)
@@ -62,6 +99,147 @@ def create_zip_file(session: Session, assets: Iterable[FileAsset], output_path:
raise FileStorageError(str(exc)) from exc
def archive_format_for_filename(filename: str) -> str:
lowered = filename.strip().casefold()
if lowered.endswith(".zip"):
return "zip"
if lowered.endswith((".tar.gz", ".tgz")):
return "tar.gz"
if lowered.endswith((".tar.bz2", ".tbz2")):
return "tar.bz2"
if lowered.endswith((".tar.xz", ".txz")):
return "tar.xz"
if lowered.endswith(".tar"):
return "tar"
raise FileStorageError(
"Unsupported archive format. Use ZIP, TAR, TAR.GZ, TAR.BZ2, or TAR.XZ."
)
def is_supported_archive_filename(filename: str) -> bool:
try:
archive_format_for_filename(filename)
except FileStorageError:
return False
return True
def inspect_archive(
archive_data: bytes | str | PathLike[str],
*,
filename: str,
password: str | None = None,
max_entries: int = ARCHIVE_UPLOAD_MAX_ENTRIES,
max_expanded_bytes: int = 2 * 1024 * 1024 * 1024,
max_expansion_ratio: int = 100,
) -> ArchiveInspection:
archive_format = archive_format_for_filename(filename)
compressed_size = _archive_size(archive_data)
if archive_format == "zip":
entries, requires_password, password_verified = _inspect_zip(
archive_data,
password=password,
)
else:
entries = _inspect_tar(archive_data)
requires_password = False
password_verified = True
_validate_archive_limits(
entries,
compressed_size=compressed_size,
max_entries=max_entries,
max_expanded_bytes=max_expanded_bytes,
max_expansion_ratio=max_expansion_ratio,
)
complete_entries = _with_derived_directories(entries)
return ArchiveInspection(
archive_format=archive_format,
entries=tuple(complete_entries),
file_count=sum(entry.kind == "file" for entry in complete_entries),
directory_count=sum(
entry.kind == "directory" for entry in complete_entries
),
expanded_size_bytes=sum(
entry.size_bytes for entry in entries if entry.kind == "file"
),
compressed_size_bytes=compressed_size,
requires_password=requires_password,
password_verified=password_verified,
)
def extract_archive_upload(
session: Session,
*,
tenant_id: str,
owner_type: str,
owner_id: str,
user_id: str,
archive_data: bytes | str | PathLike[str],
filename: str,
folder: str | None,
campaign_id: str | None,
selected_paths: Iterable[str] | None = None,
password: str | None = None,
conflict_strategy: str = "reject",
conflict_resolutions: Iterable[FileConflictResolution] | None = None,
metadata: dict[str, Any] | None = None,
is_admin: bool = False,
encryption_vault_id: str | None = None,
max_entries: int = ARCHIVE_UPLOAD_MAX_ENTRIES,
max_file_bytes: int = 50 * 1024 * 1024,
max_expanded_bytes: int = 2 * 1024 * 1024 * 1024,
max_expansion_ratio: int = 100,
) -> list[UploadedStoredFile]:
inspection = inspect_archive(
archive_data,
filename=filename,
password=password,
max_entries=max_entries,
max_expanded_bytes=max_expanded_bytes,
max_expansion_ratio=max_expansion_ratio,
)
if inspection.requires_password and not inspection.password_verified:
raise ArchivePasswordError("Archive password is required")
selected_files = _selected_file_paths(inspection.entries, selected_paths)
if not selected_files:
raise FileStorageError("Select at least one archive file to import")
actual_total_limit = min(
max_expanded_bytes,
inspection.compressed_size_bytes * max_expansion_ratio,
)
if inspection.archive_format == "zip":
members = _read_selected_zip_members(
archive_data,
selected_files=selected_files,
password=password,
max_file_bytes=max_file_bytes,
max_total_bytes=actual_total_limit,
)
else:
members = _read_selected_tar_members(
archive_data,
selected_files=selected_files,
max_file_bytes=max_file_bytes,
max_total_bytes=actual_total_limit,
)
return _store_archive_members(
session,
members=members,
tenant_id=tenant_id,
owner_type=owner_type,
owner_id=owner_id,
user_id=user_id,
folder=folder,
campaign_id=campaign_id,
conflict_strategy=conflict_strategy,
conflict_resolutions=conflict_resolutions,
metadata=metadata,
is_admin=is_admin,
encryption_vault_id=encryption_vault_id,
)
def extract_zip_upload(
session: Session,
*,
@@ -76,55 +254,409 @@ def extract_zip_upload(
conflict_resolutions: Iterable[FileConflictResolution] | None = None,
metadata: dict[str, Any] | None = None,
is_admin: bool = False,
max_files: int = 1000,
encryption_vault_id: str | None = None,
max_files: int = ZIP_UPLOAD_MAX_FILES,
max_file_bytes: int = 50 * 1024 * 1024,
max_total_bytes: int = 250 * 1024 * 1024,
) -> list[UploadedStoredFile]:
uploaded: list[UploadedStoredFile] = []
total = 0
base_folder = normalize_folder(folder)
"""Backward-compatible wrapper for the original immediate ZIP endpoint."""
return extract_archive_upload(
session,
tenant_id=tenant_id,
owner_type=owner_type,
owner_id=owner_id,
user_id=user_id,
archive_data=zip_data,
filename="archive.zip",
folder=folder,
campaign_id=campaign_id,
conflict_strategy=conflict_strategy,
conflict_resolutions=conflict_resolutions,
metadata=metadata,
is_admin=is_admin,
encryption_vault_id=encryption_vault_id,
max_entries=max_files,
max_file_bytes=max_file_bytes,
max_expanded_bytes=max_total_bytes,
)
def _inspect_zip(
archive_data: bytes | str | PathLike[str],
*,
password: str | None,
) -> tuple[list[ArchiveEntry], bool, bool]:
try:
source = BytesIO(zip_data) if isinstance(zip_data, bytes) else zip_data
with zipfile.ZipFile(source) as archive:
infos = [info for info in archive.infolist() if not info.is_dir()]
if len(infos) > max_files:
raise FileStorageError(f"ZIP contains too many files (limit {max_files})")
for info in infos:
if info.flag_bits & 0x1:
raise FileStorageError("Encrypted ZIP uploads are not supported")
if info.file_size < 0:
raise FileStorageError("Invalid ZIP member")
if info.file_size > max_file_bytes:
raise FileStorageError(f"ZIP member {info.filename!r} exceeds per-file limit")
if total + info.file_size > max_total_bytes:
raise FileStorageError("ZIP is too large after extraction")
inner_path = normalize_logical_path(info.filename)
target_path = f"{base_folder}/{inner_path}" if base_folder else inner_path
data, total = _read_zip_member(
archive,
info,
max_file_bytes=max_file_bytes,
max_total_bytes=max_total_bytes,
current_total=total,
)
uploaded.append(
create_file_asset(
session,
tenant_id=tenant_id,
owner_type=owner_type,
owner_id=owner_id,
user_id=user_id,
filename=filename_from_path(inner_path),
data=data,
display_path=target_path,
content_type=mimetypes.guess_type(inner_path)[0] or "application/octet-stream",
metadata=metadata,
campaign_id=campaign_id,
conflict_strategy=conflict_strategy,
conflict_resolutions=conflict_resolutions,
is_admin=is_admin,
with pyzipper.AESZipFile(_archive_source(archive_data)) as archive:
infos = archive.infolist()
entries = [_zip_entry(info) for info in infos]
encrypted_info = next(
(
info
for info in infos
if not info.is_dir() and bool(info.flag_bits & 0x1)
),
None,
)
if encrypted_info is None:
return entries, False, True
if not password:
return entries, True, False
try:
with archive.open(
encrypted_info, pwd=password.encode("utf-8")
) as member:
member.read(1)
except (RuntimeError, ValueError, zipfile.BadZipFile) as exc:
raise ArchivePasswordError("Archive password is incorrect") from exc
return entries, True, True
except ArchivePasswordError:
raise
except (OSError, ValueError, zipfile.BadZipFile) as exc:
raise FileStorageError("Invalid ZIP upload") from exc
def _inspect_tar(
archive_data: bytes | str | PathLike[str],
) -> list[ArchiveEntry]:
try:
with _open_tar(archive_data) as archive:
entries: list[ArchiveEntry] = []
for member in archive.getmembers():
path = _safe_member_path(member.name)
if member.isdir():
entries.append(
ArchiveEntry(path=path, kind="directory", size_bytes=0)
)
continue
if not member.isfile():
raise FileStorageError(
f"Archive member {member.name!r} is not a regular file or directory"
)
entries.append(
ArchiveEntry(
path=path,
kind="file",
size_bytes=max(0, int(member.size)),
)
)
except zipfile.BadZipFile as exc:
raise FileStorageError("Invalid ZIP upload") from exc
return entries
except FileStorageError:
raise
except (OSError, tarfile.TarError) as exc:
raise FileStorageError("Invalid TAR upload") from exc
def _zip_entry(info: zipfile.ZipInfo) -> ArchiveEntry:
path = _safe_member_path(info.filename)
unix_mode = (info.external_attr >> 16) & 0xFFFF
file_type = stat.S_IFMT(unix_mode)
if file_type and not (
stat.S_ISREG(unix_mode) or stat.S_ISDIR(unix_mode)
):
raise FileStorageError(
f"Archive member {info.filename!r} is not a regular file or directory"
)
if info.file_size < 0 or info.compress_size < 0:
raise FileStorageError(f"Archive member {info.filename!r} has invalid size metadata")
return ArchiveEntry(
path=path,
kind="directory" if info.is_dir() else "file",
size_bytes=0 if info.is_dir() else int(info.file_size),
compressed_size_bytes=(
None if info.is_dir() else int(info.compress_size)
),
encrypted=bool(info.flag_bits & 0x1),
)
def _validate_archive_limits(
entries: list[ArchiveEntry],
*,
compressed_size: int,
max_entries: int,
max_expanded_bytes: int,
max_expansion_ratio: int,
) -> None:
if len(entries) > max_entries:
raise FileStorageError(
f"Archive contains too many entries (limit {max_entries})"
)
seen: dict[str, str] = {}
expanded_size = 0
for entry in entries:
previous_kind = seen.get(entry.path)
if previous_kind is not None:
raise FileStorageError(
f"Archive contains duplicate path {entry.path!r}"
)
seen[entry.path] = entry.kind
if entry.kind == "file":
expanded_size += entry.size_bytes
if expanded_size > max_expanded_bytes:
raise FileStorageError(
"Archive is too large after extraction "
f"(limit {max_expanded_bytes} bytes)"
)
if expanded_size and (
compressed_size <= 0
or expanded_size > compressed_size * max_expansion_ratio
):
raise FileStorageError(
"Archive expansion ratio exceeds "
f"{max_expansion_ratio}:1"
)
def _with_derived_directories(
entries: list[ArchiveEntry],
) -> list[ArchiveEntry]:
by_path = {entry.path: entry for entry in entries}
for entry in entries:
parent = PurePosixPath(entry.path).parent
while str(parent) not in {"", "."}:
path = str(parent)
existing = by_path.get(path)
if existing and existing.kind != "directory":
raise FileStorageError(
f"Archive path {path!r} is both a file and a directory"
)
by_path.setdefault(
path,
ArchiveEntry(path=path, kind="directory", size_bytes=0),
)
parent = parent.parent
return sorted(
by_path.values(),
key=lambda entry: (
tuple(entry.path.casefold().split("/")),
entry.kind != "directory",
),
)
def _selected_file_paths(
entries: tuple[ArchiveEntry, ...],
selected_paths: Iterable[str] | None,
) -> set[str]:
file_paths = {entry.path for entry in entries if entry.kind == "file"}
if selected_paths is None:
return file_paths
known = {entry.path: entry for entry in entries}
normalized = {_safe_member_path(path) for path in selected_paths}
unknown = normalized - known.keys()
if unknown:
raise FileStorageError(
f"Archive selection contains unknown path {sorted(unknown)[0]!r}"
)
selected_files: set[str] = set()
for path in normalized:
entry = known[path]
if entry.kind == "file":
selected_files.add(path)
continue
prefix = f"{path}/"
selected_files.update(
file_path
for file_path in file_paths
if file_path.startswith(prefix)
)
return selected_files
def _read_selected_zip_members(
archive_data: bytes | str | PathLike[str],
*,
selected_files: set[str],
password: str | None,
max_file_bytes: int,
max_total_bytes: int,
) -> list[tuple[str, bytes]]:
result: list[tuple[str, bytes]] = []
total = 0
try:
with pyzipper.AESZipFile(_archive_source(archive_data)) as archive:
for info in archive.infolist():
if info.is_dir():
continue
path = _safe_member_path(info.filename)
if path not in selected_files:
continue
pwd = password.encode("utf-8") if password else None
try:
with archive.open(info, pwd=pwd) as source:
data, total = _read_member(
source,
path=path,
max_file_bytes=max_file_bytes,
max_total_bytes=max_total_bytes,
current_total=total,
)
except (RuntimeError, ValueError, zipfile.BadZipFile) as exc:
if info.flag_bits & 0x1:
raise ArchivePasswordError(
"Archive password is incorrect"
) from exc
raise
result.append((path, data))
except (FileStorageError, ArchivePasswordError):
raise
except (OSError, ValueError, zipfile.BadZipFile) as exc:
raise FileStorageError("ZIP extraction failed") from exc
return result
def _read_selected_tar_members(
archive_data: bytes | str | PathLike[str],
*,
selected_files: set[str],
max_file_bytes: int,
max_total_bytes: int,
) -> list[tuple[str, bytes]]:
result: list[tuple[str, bytes]] = []
total = 0
try:
with _open_tar(archive_data) as archive:
for member in archive.getmembers():
if not member.isfile():
continue
path = _safe_member_path(member.name)
if path not in selected_files:
continue
source = archive.extractfile(member)
if source is None:
raise FileStorageError(
f"Archive member {path!r} could not be read"
)
with source:
data, total = _read_member(
source,
path=path,
max_file_bytes=max_file_bytes,
max_total_bytes=max_total_bytes,
current_total=total,
)
result.append((path, data))
except FileStorageError:
raise
except (OSError, tarfile.TarError) as exc:
raise FileStorageError("TAR extraction failed") from exc
return result
def _read_member(
source: BinaryIO,
*,
path: str,
max_file_bytes: int,
max_total_bytes: int,
current_total: int,
) -> tuple[bytes, int]:
parts: list[bytes] = []
actual_size = 0
while True:
read_size = min(
_ARCHIVE_READ_CHUNK_SIZE,
max_file_bytes + 1 - actual_size,
)
chunk = source.read(read_size)
if not chunk:
break
actual_size += len(chunk)
if actual_size > max_file_bytes:
raise FileStorageError(
f"Archive member {path!r} exceeds per-file limit"
)
if current_total + actual_size > max_total_bytes:
raise FileStorageError("Archive is too large after extraction")
parts.append(chunk)
return b"".join(parts), current_total + actual_size
def _store_archive_members(
session: Session,
*,
members: Iterable[tuple[str, bytes]],
tenant_id: str,
owner_type: str,
owner_id: str,
user_id: str,
folder: str | None,
campaign_id: str | None,
conflict_strategy: str,
conflict_resolutions: Iterable[FileConflictResolution] | None,
metadata: dict[str, Any] | None,
is_admin: bool,
encryption_vault_id: str | None,
) -> list[UploadedStoredFile]:
uploaded: list[UploadedStoredFile] = []
base_folder = normalize_folder(folder)
for inner_path, data in members:
target_path = (
f"{base_folder}/{inner_path}" if base_folder else inner_path
)
uploaded.append(
create_file_asset(
session,
tenant_id=tenant_id,
owner_type=owner_type,
owner_id=owner_id,
user_id=user_id,
filename=filename_from_path(inner_path),
data=data,
display_path=target_path,
content_type=mimetypes.guess_type(inner_path)[0]
or "application/octet-stream",
metadata=metadata,
campaign_id=campaign_id,
conflict_strategy=conflict_strategy,
conflict_resolutions=conflict_resolutions,
is_admin=is_admin,
encryption_vault_id=encryption_vault_id,
)
)
return uploaded
def _safe_member_path(value: str) -> str:
raw = str(value or "").replace("\\", "/").strip()
if (
not raw
or "\x00" in raw
or raw.startswith("/")
or _WINDOWS_DRIVE_RE.match(raw)
):
raise FileStorageError(f"Unsafe archive member path {value!r}")
if any(part == ".." for part in raw.split("/")):
raise FileStorageError(f"Unsafe archive member path {value!r}")
try:
return normalize_logical_path(raw.rstrip("/"))
except ValueError as exc:
raise FileStorageError(f"Unsafe archive member path {value!r}") from exc
def _archive_size(archive_data: bytes | str | PathLike[str]) -> int:
if isinstance(archive_data, bytes):
return len(archive_data)
try:
return Path(archive_data).stat().st_size
except OSError as exc:
raise FileStorageError("Archive upload could not be read") from exc
def _archive_source(
archive_data: bytes | str | PathLike[str],
) -> BytesIO | str | PathLike[str]:
return BytesIO(archive_data) if isinstance(archive_data, bytes) else archive_data
def _open_tar(
archive_data: bytes | str | PathLike[str],
) -> tarfile.TarFile:
if isinstance(archive_data, bytes):
return tarfile.open(fileobj=BytesIO(archive_data), mode="r:*")
try:
return tarfile.open(name=archive_data, mode="r:*")
except OSError as exc:
raise FileStorageError("Archive upload could not be read") from exc
+25 -153
View File
@@ -1,160 +1,32 @@
from __future__ import annotations
from dataclasses import dataclass, field
from pathlib import Path
from typing import Iterable, Protocol
from govoplan_core.core.object_storage import (
LocalFilesystemStorageBackend,
S3StorageBackend,
StorageBackend,
StorageBackendError,
StorageObjectInfo,
StorageObjectMissing,
StorageObjectPage,
configured_storage_backend,
)
from govoplan_files.backend.runtime import settings
class StorageBackendError(RuntimeError):
pass
class StorageBackend(Protocol):
name: str
def put_bytes(self, key: str, data: bytes, *, content_type: str | None = None) -> None: ...
def get_bytes(self, key: str) -> bytes: ...
def iter_bytes(self, key: str, *, chunk_size: int = 1024 * 1024) -> Iterable[bytes]: ...
def delete(self, key: str) -> None: ...
def exists(self, key: str) -> bool: ...
@dataclass(slots=True)
class LocalFilesystemStorageBackend:
root: Path
fallback_roots: tuple[Path, ...] = field(default_factory=tuple)
name: str = "local"
def __post_init__(self) -> None:
self.root = self.root.expanduser().resolve()
self.fallback_roots = tuple(root.expanduser().resolve() for root in self.fallback_roots if root)
self.root.mkdir(parents=True, exist_ok=True)
def _path_for_root(self, root: Path, key: str) -> Path:
path = (root / key).resolve()
if not path.is_relative_to(root):
raise StorageBackendError("Storage key escapes local storage root")
return path
def _path(self, key: str) -> Path:
return self._path_for_root(self.root, key)
def _readable_path(self, key: str) -> Path:
primary = self._path(key)
if primary.exists() and primary.is_file():
return primary
for root in self.fallback_roots:
candidate = self._path_for_root(root, key)
if candidate.exists() and candidate.is_file():
return candidate
raise StorageBackendError("Stored object does not exist")
def put_bytes(self, key: str, data: bytes, *, content_type: str | None = None) -> None:
path = self._path(key)
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(data)
def get_bytes(self, key: str) -> bytes:
return self._readable_path(key).read_bytes()
def iter_bytes(self, key: str, *, chunk_size: int = 1024 * 1024) -> Iterable[bytes]:
path = self._readable_path(key)
with path.open("rb") as handle:
while True:
chunk = handle.read(chunk_size)
if not chunk:
break
yield chunk
def delete(self, key: str) -> None:
path = self._path(key)
if path.exists() and path.is_file():
path.unlink()
def exists(self, key: str) -> bool:
try:
self._readable_path(key)
except StorageBackendError:
return False
return True
@dataclass(slots=True)
class S3StorageBackend:
bucket: str
endpoint_url: str
region_name: str
access_key_id: str
secret_access_key: str
name: str = "s3"
@property
def client(self):
try:
import boto3
except ModuleNotFoundError as exc:
raise StorageBackendError("boto3 is required for the S3 storage backend") from exc
return boto3.client(
"s3",
endpoint_url=self.endpoint_url,
region_name=self.region_name,
aws_access_key_id=self.access_key_id,
aws_secret_access_key=self.secret_access_key,
)
def put_bytes(self, key: str, data: bytes, *, content_type: str | None = None) -> None:
kwargs = {"Bucket": self.bucket, "Key": key, "Body": data}
if content_type:
kwargs["ContentType"] = content_type
self.client.put_object(**kwargs)
def get_bytes(self, key: str) -> bytes:
try:
obj = self.client.get_object(Bucket=self.bucket, Key=key)
return obj["Body"].read()
except Exception as exc: # pragma: no cover - depends on S3 backend
raise StorageBackendError(str(exc)) from exc
def iter_bytes(self, key: str, *, chunk_size: int = 1024 * 1024) -> Iterable[bytes]:
try:
obj = self.client.get_object(Bucket=self.bucket, Key=key)
body = obj["Body"]
while True:
chunk = body.read(chunk_size)
if not chunk:
break
yield chunk
except Exception as exc: # pragma: no cover - depends on S3 backend
raise StorageBackendError(str(exc)) from exc
def delete(self, key: str) -> None:
self.client.delete_object(Bucket=self.bucket, Key=key)
def exists(self, key: str) -> bool:
try:
self.client.head_object(Bucket=self.bucket, Key=key)
return True
except Exception:
return False
def _fallback_roots() -> tuple[Path, ...]:
raw = getattr(settings, "file_storage_local_fallback_roots", "") or ""
return tuple(Path(item.strip()) for item in str(raw).split(",") if item.strip())
def get_storage_backend() -> StorageBackend:
backend = settings.file_storage_backend.lower().strip()
if backend in {"local", "filesystem", "fs"}:
return LocalFilesystemStorageBackend(Path(settings.file_storage_local_root), fallback_roots=_fallback_roots())
if backend in {"s3", "garage"}:
return S3StorageBackend(
bucket=settings.file_storage_s3_bucket or settings.s3_bucket,
endpoint_url=settings.file_storage_s3_endpoint_url or settings.s3_endpoint_url,
region_name=settings.file_storage_s3_region or settings.s3_region,
access_key_id=settings.file_storage_s3_access_key_id or settings.s3_access_key_id,
secret_access_key=settings.file_storage_s3_secret_access_key or settings.s3_secret_access_key,
)
raise StorageBackendError(f"Unsupported file storage backend: {settings.file_storage_backend}")
"""Return the deployment-owned backend for Files-managed objects."""
return configured_storage_backend(settings)
__all__ = [
"LocalFilesystemStorageBackend",
"S3StorageBackend",
"StorageBackend",
"StorageBackendError",
"StorageObjectInfo",
"StorageObjectMissing",
"StorageObjectPage",
"get_storage_backend",
]
@@ -12,12 +12,17 @@ from typing import Any, Iterator
from sqlalchemy.orm import Session
from govoplan_files.backend.db.models import FileAsset
from govoplan_files.backend.storage.backends import StorageBackendError, get_storage_backend
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.files import current_versions_and_blobs, list_assets_for_user
from govoplan_files.backend.db.models import FileAsset, FileBlob, FileShare, FileVersion
from govoplan_files.backend.storage.backends import StorageBackend, get_storage_backend
from govoplan_files.backend.storage.files import current_versions_and_blobs, get_asset_for_user, list_assets_for_user, share_files
from govoplan_files.backend.storage.access import ensure_owner_access
from govoplan_files.backend.storage.paths import normalize_folder, normalize_logical_path, safe_storage_component
from govoplan_files.backend.storage.provenance import source_provenance_from_metadata, source_revision_from_metadata
from govoplan_files.backend.storage.share_state import effective_file_share_clause
from govoplan_files.backend.storage.integrity import (
ensure_blob_is_readable,
read_verified_blob_bytes,
)
MANAGED_SOURCE_PREFIX = "managed:"
@@ -37,6 +42,7 @@ class ManagedAttachmentFile:
checksum_sha256: str
size_bytes: int
content_type: str | None
linked_to_campaign: bool = True
source_provenance: dict[str, Any] | None = None
source_revision: str | None = None
@@ -56,6 +62,7 @@ class PreparedCampaignSnapshot:
raw_json: dict[str, Any]
managed_files_by_local_path: dict[str, ManagedAttachmentFile]
shared_assets: list[FileAsset]
candidate_assets: list[FileAsset]
def parse_managed_source(value: object) -> tuple[str, str] | None:
@@ -117,6 +124,99 @@ def _iter_rule_dicts(attachments: dict[str, Any], raw_json: dict[str, Any]):
yield rule
def _managed_base_sources(raw_json: dict[str, Any]) -> list[tuple[str, str, str]]:
attachments = raw_json.get("attachments")
if not isinstance(attachments, dict):
return []
base_paths = attachments.get("base_paths")
if not isinstance(base_paths, list):
return []
sources: list[tuple[str, str, str]] = []
seen: set[tuple[str, str, str]] = set()
for item in base_paths:
if not isinstance(item, dict):
continue
parsed_source = parse_managed_source(item.get("source"))
if parsed_source is None:
continue
owner_type, owner_id = parsed_source
old_path = str(item.get("path") or ".").strip() or "."
logical_root = "" if old_path in {"", ".", "/"} else normalize_folder(old_path)
key = (owner_type, owner_id, logical_root)
if key in seen:
continue
seen.add(key)
sources.append(key)
return sources
def _campaign_linked_asset_ids(
session: Session,
*,
tenant_id: str,
campaign_id: str,
asset_ids: list[str],
) -> set[str]:
if not asset_ids:
return set()
linked: set[str] = set()
for offset in range(0, len(asset_ids), 900):
chunk = asset_ids[offset : offset + 900]
rows = (
session.query(FileShare.file_asset_id)
.filter(
FileShare.tenant_id == tenant_id,
FileShare.file_asset_id.in_(chunk),
FileShare.target_type == "campaign",
FileShare.target_id == campaign_id,
effective_file_share_clause(),
)
.all()
)
linked.update(row[0] for row in rows)
return linked
def _candidate_assets_for_managed_sources(
session: Session,
*,
tenant_id: str,
campaign_id: str,
raw_json: dict[str, Any],
user_id: str,
is_admin: bool,
shared_assets: list[FileAsset],
) -> tuple[list[FileAsset], set[str]]:
assets_by_id: dict[str, FileAsset] = {asset.id: asset for asset in shared_assets}
for owner_type, owner_id, logical_root in _managed_base_sources(raw_json):
ensure_owner_access(
session,
tenant_id=tenant_id,
owner_type=owner_type,
owner_id=owner_id,
user_id=user_id,
is_admin=is_admin,
)
for asset in list_assets_for_user(
session,
tenant_id=tenant_id,
user_id=user_id,
owner_type=owner_type,
owner_id=owner_id,
path_prefix=logical_root or None,
is_admin=is_admin,
):
assets_by_id.setdefault(asset.id, asset)
candidate_assets = sorted(assets_by_id.values(), key=lambda asset: (asset.display_path, asset.updated_at, asset.id))
linked_ids = _campaign_linked_asset_ids(
session,
tenant_id=tenant_id,
campaign_id=campaign_id,
asset_ids=[asset.id for asset in candidate_assets],
)
return candidate_assets, linked_ids
def _selected_base_path(
rule: dict[str, Any],
prepared_by_id: dict[str, tuple[str, str]],
@@ -136,6 +236,175 @@ def _selected_base_path(
return None
def _prepared_attachment_config(raw_json: dict[str, Any]) -> tuple[dict[str, Any], dict[str, Any], list[Any]]:
prepared_json = copy.deepcopy(raw_json if isinstance(raw_json, dict) else {})
attachments = prepared_json.get("attachments")
if not isinstance(attachments, dict):
attachments = {}
prepared_json["attachments"] = attachments
base_paths = attachments.get("base_paths")
if not isinstance(base_paths, list):
base_paths = []
return prepared_json, attachments, base_paths
def _campaign_snapshot_assets(
session: Session,
*,
tenant_id: str,
campaign_id: str,
prepared_json: dict[str, Any],
include_unlinked_candidates: bool,
user_id: str,
is_admin: bool,
) -> tuple[list[FileAsset], list[FileAsset], set[str]]:
shared_assets = list_assets_for_user(
session,
tenant_id=tenant_id,
user_id="",
campaign_id=campaign_id,
is_admin=True,
)
if not include_unlinked_candidates:
return shared_assets, shared_assets, {asset.id for asset in shared_assets}
candidate_assets, linked_asset_ids = _candidate_assets_for_managed_sources(
session,
tenant_id=tenant_id,
campaign_id=campaign_id,
raw_json=prepared_json,
user_id=user_id,
is_admin=is_admin,
shared_assets=shared_assets,
)
return shared_assets, candidate_assets, linked_asset_ids
def _assets_grouped_by_owner(assets: list[FileAsset]) -> dict[tuple[str, str], list[FileAsset]]:
assets_by_owner: dict[tuple[str, str], list[FileAsset]] = defaultdict(list)
for asset in assets:
owner_id = _asset_owner_id(asset)
if owner_id:
assets_by_owner[(asset.owner_type, owner_id)].append(asset)
return assets_by_owner
def _materialize_managed_asset(
asset: FileAsset,
*,
owner_id: str,
logical_root: str,
local_root: Path,
version_blobs: dict[str, tuple[FileVersion, FileBlob]],
backend: StorageBackend | None,
include_bytes: bool,
linked_asset_ids: set[str],
) -> tuple[str, ManagedAttachmentFile] | None:
relative_path = _relative_asset_path(asset, logical_root)
if not relative_path:
return None
target = _safe_local_target(local_root, relative_path)
target.parent.mkdir(parents=True, exist_ok=True)
version, blob = version_blobs[asset.id]
ensure_blob_is_readable(blob)
if include_bytes:
data = read_verified_blob_bytes(blob, backend=backend) if backend else b""
target.write_bytes(data)
else:
target.touch()
local_key = str(target.resolve())
return local_key, ManagedAttachmentFile(
local_path=local_key,
asset_id=asset.id,
version_id=version.id,
blob_id=blob.id,
display_path=asset.display_path,
relative_path=normalize_logical_path(relative_path),
filename=asset.filename,
owner_type=asset.owner_type,
owner_id=owner_id,
checksum_sha256=blob.checksum_sha256,
size_bytes=blob.size_bytes,
content_type=blob.content_type,
linked_to_campaign=asset.id in linked_asset_ids,
source_provenance=source_provenance_from_metadata(asset.metadata_),
source_revision=source_revision_from_metadata(asset.metadata_),
)
def _prepare_managed_base_paths(
base_paths: list[Any],
*,
materialized_root: Path,
assets_by_owner: dict[tuple[str, str], list[FileAsset]],
version_blobs: dict[str, tuple[FileVersion, FileBlob]],
backend: StorageBackend | None,
include_bytes: bool,
linked_asset_ids: set[str],
) -> tuple[
dict[str, ManagedAttachmentFile],
dict[str, tuple[str, str]],
dict[str, list[tuple[str, str]]],
tuple[str, str] | None,
]:
manifest: dict[str, ManagedAttachmentFile] = {}
prepared_by_id: dict[str, tuple[str, str]] = {}
prepared_by_old_path: dict[str, list[tuple[str, str]]] = {}
first_prepared: tuple[str, str] | None = None
for index, item in enumerate(base_paths):
if not isinstance(item, dict):
continue
parsed_source = parse_managed_source(item.get("source"))
if parsed_source is None:
continue
owner_type, owner_id = parsed_source
old_path = str(item.get("path") or ".").strip() or "."
logical_root = "" if old_path in {"", ".", "/"} else normalize_folder(old_path)
base_path_id = str(item.get("id") or f"base-path-{index + 1}")
local_root = materialized_root / f"{index + 1:03d}-{safe_storage_component(base_path_id)}"
local_root.mkdir(parents=True, exist_ok=True)
local_root_string = str(local_root.resolve())
prepared = (base_path_id, local_root_string)
prepared_by_id[base_path_id] = prepared
prepared_by_old_path.setdefault(old_path, []).append(prepared)
if first_prepared is None:
first_prepared = prepared
item["path"] = local_root_string
for asset in assets_by_owner.get((owner_type, owner_id), []):
materialized = _materialize_managed_asset(
asset,
owner_id=owner_id,
logical_root=logical_root,
local_root=local_root,
version_blobs=version_blobs,
backend=backend,
include_bytes=include_bytes,
linked_asset_ids=linked_asset_ids,
)
if materialized is not None:
local_key, managed_file = materialized
manifest[local_key] = managed_file
return manifest, prepared_by_id, prepared_by_old_path, first_prepared
def _rewrite_managed_attachment_rules(
attachments: dict[str, Any],
prepared_json: dict[str, Any],
*,
prepared_by_id: dict[str, tuple[str, str]],
prepared_by_old_path: dict[str, list[tuple[str, str]]],
first_prepared: tuple[str, str] | None,
) -> None:
for rule in _iter_rule_dicts(attachments, prepared_json):
selected = _selected_base_path(rule, prepared_by_id, prepared_by_old_path, first_prepared)
if selected is None:
continue
base_path_id, local_root_string = selected
rule["base_path_id"] = base_path_id
rule["base_dir"] = local_root_string
if first_prepared is not None:
attachments["base_path"] = first_prepared[1]
def prepare_campaign_snapshot(
session: Session,
*,
@@ -144,6 +413,9 @@ def prepare_campaign_snapshot(
raw_json: dict[str, Any],
destination: Path,
include_bytes: bool,
include_unlinked_candidates: bool = False,
user_id: str = "",
is_admin: bool = False,
) -> PreparedCampaignSnapshot:
"""Create a temporary file-oriented campaign snapshot for managed attachments.
@@ -159,101 +431,35 @@ def prepare_campaign_snapshot(
materialized_root = destination / "managed-attachments"
materialized_root.mkdir(parents=True, exist_ok=True)
prepared_json = copy.deepcopy(raw_json if isinstance(raw_json, dict) else {})
attachments = prepared_json.get("attachments")
if not isinstance(attachments, dict):
attachments = {}
prepared_json["attachments"] = attachments
base_paths = attachments.get("base_paths")
if not isinstance(base_paths, list):
base_paths = []
shared_assets = list_assets_for_user(
prepared_json, attachments, base_paths = _prepared_attachment_config(raw_json)
shared_assets, candidate_assets, linked_asset_ids = _campaign_snapshot_assets(
session,
tenant_id=tenant_id,
user_id="",
campaign_id=campaign_id,
is_admin=True,
prepared_json=prepared_json,
include_unlinked_candidates=include_unlinked_candidates,
user_id=user_id,
is_admin=is_admin,
)
assets_by_owner: dict[tuple[str, str], list[FileAsset]] = defaultdict(list)
for asset in shared_assets:
owner_id = _asset_owner_id(asset)
if owner_id:
assets_by_owner[(asset.owner_type, owner_id)].append(asset)
version_blobs = current_versions_and_blobs(session, shared_assets)
assets_by_owner = _assets_grouped_by_owner(candidate_assets)
version_blobs = current_versions_and_blobs(session, candidate_assets)
backend = get_storage_backend() if include_bytes else None
manifest: dict[str, ManagedAttachmentFile] = {}
prepared_by_id: dict[str, tuple[str, str]] = {}
prepared_by_old_path: dict[str, list[tuple[str, str]]] = {}
first_prepared: tuple[str, str] | None = None
for index, item in enumerate(base_paths):
if not isinstance(item, dict):
continue
parsed_source = parse_managed_source(item.get("source"))
if parsed_source is None:
continue
owner_type, owner_id = parsed_source
old_path = str(item.get("path") or ".").strip() or "."
logical_root = "" if old_path in {"", ".", "/"} else normalize_folder(old_path)
base_path_id = str(item.get("id") or f"base-path-{index + 1}")
local_root = materialized_root / f"{index + 1:03d}-{safe_storage_component(base_path_id)}"
local_root.mkdir(parents=True, exist_ok=True)
local_root_string = str(local_root.resolve())
prepared = (base_path_id, local_root_string)
prepared_by_id[base_path_id] = prepared
prepared_by_old_path.setdefault(old_path, []).append(prepared)
if first_prepared is None:
first_prepared = prepared
item["path"] = local_root_string
for asset in assets_by_owner.get((owner_type, owner_id), []):
relative_path = _relative_asset_path(asset, logical_root)
if not relative_path:
continue
target = _safe_local_target(local_root, relative_path)
target.parent.mkdir(parents=True, exist_ok=True)
version, blob = version_blobs[asset.id]
if include_bytes:
try:
data = backend.get_bytes(blob.storage_key) if backend else b""
except StorageBackendError as exc:
raise FileStorageError(str(exc)) from exc
target.write_bytes(data)
else:
target.touch()
local_key = str(target.resolve())
manifest[local_key] = ManagedAttachmentFile(
local_path=local_key,
asset_id=asset.id,
version_id=version.id,
blob_id=blob.id,
display_path=asset.display_path,
relative_path=normalize_logical_path(relative_path),
filename=asset.filename,
owner_type=asset.owner_type,
owner_id=owner_id,
checksum_sha256=blob.checksum_sha256,
size_bytes=blob.size_bytes,
content_type=blob.content_type,
source_provenance=source_provenance_from_metadata(asset.metadata_),
source_revision=source_revision_from_metadata(asset.metadata_),
)
for rule in _iter_rule_dicts(attachments, prepared_json):
selected = _selected_base_path(rule, prepared_by_id, prepared_by_old_path, first_prepared)
if selected is None:
continue
base_path_id, local_root_string = selected
rule["base_path_id"] = base_path_id
rule["base_dir"] = local_root_string
if first_prepared is not None:
attachments["base_path"] = first_prepared[1]
manifest, prepared_by_id, prepared_by_old_path, first_prepared = _prepare_managed_base_paths(
base_paths,
materialized_root=materialized_root,
assets_by_owner=assets_by_owner,
version_blobs=version_blobs,
backend=backend,
include_bytes=include_bytes,
linked_asset_ids=linked_asset_ids,
)
_rewrite_managed_attachment_rules(
attachments,
prepared_json,
prepared_by_id=prepared_by_id,
prepared_by_old_path=prepared_by_old_path,
first_prepared=first_prepared,
)
snapshot_path = destination / "campaign.json"
snapshot_path.write_text(json.dumps(prepared_json, ensure_ascii=False, indent=2), encoding="utf-8")
@@ -262,6 +468,7 @@ def prepare_campaign_snapshot(
raw_json=prepared_json,
managed_files_by_local_path=manifest,
shared_assets=shared_assets,
candidate_assets=candidate_assets,
)
@@ -273,7 +480,10 @@ def prepared_campaign_snapshot(
campaign_id: str,
raw_json: dict[str, Any],
include_bytes: bool,
prefix: str = "multimailer-managed-campaign-",
prefix: str = "govoplan-managed-campaign-",
include_unlinked_candidates: bool = False,
user_id: str = "",
is_admin: bool = False,
) -> Iterator[PreparedCampaignSnapshot]:
temp_dir = Path(tempfile.mkdtemp(prefix=prefix))
try:
@@ -284,6 +494,9 @@ def prepared_campaign_snapshot(
raw_json=raw_json,
destination=temp_dir,
include_bytes=include_bytes,
include_unlinked_candidates=include_unlinked_candidates,
user_id=user_id,
is_admin=is_admin,
)
finally:
shutil.rmtree(temp_dir, ignore_errors=True)
@@ -301,6 +514,53 @@ def managed_match_payloads(
return payloads
def share_assets_with_campaign(
session: Session,
*,
tenant_id: str,
campaign_id: str,
file_ids: list[str],
user_id: str,
is_admin: bool = False,
) -> list[dict[str, Any]]:
unique_ids = list(dict.fromkeys(file_id for file_id in file_ids if file_id))
if not unique_ids:
return []
assets = [
get_asset_for_user(
session,
tenant_id=tenant_id,
user_id=user_id,
asset_id=file_id,
require_write=True,
is_admin=is_admin,
)
for file_id in unique_ids
]
shares = share_files(
session,
tenant_id=tenant_id,
assets=assets,
target_type="campaign",
target_id=campaign_id,
permission="read",
user_id=user_id,
)
session.flush()
return [
{
"id": share.id,
"file_asset_id": share.file_asset_id,
"target_type": share.target_type,
"target_id": share.target_id,
"permission": share.permission,
"created_at": share.created_at.isoformat() if share.created_at else None,
"revoked_at": share.revoked_at.isoformat() if share.revoked_at else None,
}
for share in shares
]
def public_attachment_summary_payload(value: Any) -> dict[str, Any]:
"""Return an attachment summary without temporary materialization paths.
@@ -1,6 +1,7 @@
from __future__ import annotations
from collections import defaultdict
from dataclasses import dataclass, field
from pathlib import PurePosixPath
from typing import Any, Iterable
@@ -24,6 +25,15 @@ def _candidate_match_keys(raw_match: str) -> set[str]:
AttachmentUseKey = tuple[str, str, str, str]
@dataclass(slots=True)
class _AttachmentBatchRefs:
managed_by_job: dict[str, list[dict[str, object]]] = field(default_factory=lambda: defaultdict(list))
fallback_attachments_by_job: dict[str, list[dict[str, object]]] = field(default_factory=lambda: defaultdict(list))
asset_ids: set[str] = field(default_factory=set)
version_ids: set[str] = field(default_factory=set)
blob_ids: set[str] = field(default_factory=set)
def _attachment_use_key(*, job_id: str, file_version_id: str, filename_used: str, stage: str) -> AttachmentUseKey:
return job_id, file_version_id, filename_used, stage
@@ -134,62 +144,91 @@ def record_campaign_attachment_uses_for_jobs(
if not job_list:
return
managed_by_job: dict[str, list[dict[str, object]]] = defaultdict(list)
fallback_attachments_by_job: dict[str, list[dict[str, object]]] = defaultdict(list)
refs = _collect_attachment_batch_refs(job_list)
job_by_id = {job.id: job for job in job_list}
asset_ids: set[str] = set()
version_ids: set[str] = set()
blob_ids: set[str] = set()
known_keys = _known_use_keys_for_jobs(session, job_list, stage=stage)
assets_by_id, versions_by_id, blobs_by_id = _load_managed_attachment_entities(session, refs)
for job in job_list:
_record_managed_attachment_uses(
session,
job_by_id=job_by_id,
managed_by_job=refs.managed_by_job,
assets_by_id=assets_by_id,
versions_by_id=versions_by_id,
blobs_by_id=blobs_by_id,
stage=stage,
known_keys=known_keys,
)
_record_fallback_attachment_uses(
session,
job_by_id=job_by_id,
fallback_attachments_by_job=refs.fallback_attachments_by_job,
stage=stage,
known_keys=known_keys,
)
def _collect_attachment_batch_refs(jobs: list[CampaignJobLike]) -> _AttachmentBatchRefs:
refs = _AttachmentBatchRefs()
for job in jobs:
attachments = job.resolved_attachments or []
if not isinstance(attachments, list):
continue
for attachment in attachments:
if not isinstance(attachment, dict):
continue
managed_matches = attachment.get("managed_matches")
if isinstance(managed_matches, list) and managed_matches:
for item in managed_matches:
if not isinstance(item, dict):
continue
asset_id = str(item.get("asset_id") or "")
version_id = str(item.get("version_id") or "")
blob_id = str(item.get("blob_id") or "")
if not asset_id or not version_id or not blob_id:
continue
managed_by_job[job.id].append(item)
asset_ids.add(asset_id)
version_ids.add(version_id)
blob_ids.add(blob_id)
else:
fallback_attachments_by_job[job.id].append(attachment)
if not _collect_managed_attachment_refs(refs, job.id, attachment):
refs.fallback_attachments_by_job[job.id].append(attachment)
return refs
assets_by_id = {
item.id: item
for item in session.query(FileAsset).filter(FileAsset.id.in_(asset_ids)).all()
} if asset_ids else {}
versions_by_id = {
item.id: item
for item in session.query(FileVersion).filter(FileVersion.id.in_(version_ids)).all()
} if version_ids else {}
blobs_by_id = {
item.id: item
for item in session.query(FileBlob).filter(FileBlob.id.in_(blob_ids)).all()
} if blob_ids else {}
known_keys = _known_use_keys_for_jobs(session, job_list, stage=stage)
def _collect_managed_attachment_refs(refs: _AttachmentBatchRefs, job_id: str, attachment: dict[str, object]) -> bool:
managed_matches = attachment.get("managed_matches")
if not isinstance(managed_matches, list) or not managed_matches:
return False
for item in managed_matches:
if not isinstance(item, dict):
continue
asset_id = str(item.get("asset_id") or "")
version_id = str(item.get("version_id") or "")
blob_id = str(item.get("blob_id") or "")
if not asset_id or not version_id or not blob_id:
continue
refs.managed_by_job[job_id].append(item)
refs.asset_ids.add(asset_id)
refs.version_ids.add(version_id)
refs.blob_ids.add(blob_id)
return True
def _load_managed_attachment_entities(
session: Session,
refs: _AttachmentBatchRefs,
) -> tuple[dict[str, FileAsset], dict[str, FileVersion], dict[str, FileBlob]]:
assets_by_id = {item.id: item for item in session.query(FileAsset).filter(FileAsset.id.in_(refs.asset_ids)).all()} if refs.asset_ids else {}
versions_by_id = {item.id: item for item in session.query(FileVersion).filter(FileVersion.id.in_(refs.version_ids)).all()} if refs.version_ids else {}
blobs_by_id = {item.id: item for item in session.query(FileBlob).filter(FileBlob.id.in_(refs.blob_ids)).all()} if refs.blob_ids else {}
return assets_by_id, versions_by_id, blobs_by_id
def _record_managed_attachment_uses(
session: Session,
*,
job_by_id: dict[str, CampaignJobLike],
managed_by_job: dict[str, list[dict[str, object]]],
assets_by_id: dict[str, FileAsset],
versions_by_id: dict[str, FileVersion],
blobs_by_id: dict[str, FileBlob],
stage: str,
known_keys: set[AttachmentUseKey],
) -> None:
for job_id, items in managed_by_job.items():
job = job_by_id[job_id]
for item in items:
asset = assets_by_id.get(str(item.get("asset_id") or ""))
version = versions_by_id.get(str(item.get("version_id") or ""))
blob = blobs_by_id.get(str(item.get("blob_id") or ""))
if not asset or not version or not blob:
continue
if asset.tenant_id != job.tenant_id or version.tenant_id != job.tenant_id or blob.tenant_id != job.tenant_id:
continue
if version.file_asset_id != asset.id or version.blob_id != blob.id:
if not _managed_attachment_entities_match_job(job, asset, version, blob):
continue
_add_use(
session,
@@ -202,50 +241,100 @@ def record_campaign_attachment_uses_for_jobs(
known_keys=known_keys,
)
def _managed_attachment_entities_match_job(
job: CampaignJobLike,
asset: FileAsset | None,
version: FileVersion | None,
blob: FileBlob | None,
) -> bool:
if not asset or not version or not blob:
return False
if asset.tenant_id != job.tenant_id or version.tenant_id != job.tenant_id or blob.tenant_id != job.tenant_id:
return False
return version.file_asset_id == asset.id and version.blob_id == blob.id
def _record_fallback_attachment_uses(
session: Session,
*,
job_by_id: dict[str, CampaignJobLike],
fallback_attachments_by_job: dict[str, list[dict[str, object]]],
stage: str,
known_keys: set[AttachmentUseKey],
) -> None:
assets_by_campaign: dict[tuple[str, str], dict[str, FileAsset]] = {}
version_blobs_by_campaign: dict[tuple[str, str], dict[str, tuple[FileVersion, FileBlob]]] = {}
for job_id, attachments in fallback_attachments_by_job.items():
job = job_by_id[job_id]
campaign_key = (job.tenant_id, job.campaign_id)
by_key = assets_by_campaign.get(campaign_key)
if by_key is None:
assets = list_assets_for_user(
session,
tenant_id=job.tenant_id,
user_id="",
campaign_id=job.campaign_id,
is_admin=True,
)
by_key = {}
for asset in assets:
by_key[asset.display_path.strip("/")] = asset
by_key.setdefault(asset.filename, asset)
assets_by_campaign[campaign_key] = by_key
version_blobs_by_campaign[campaign_key] = current_versions_and_blobs(session, assets)
version_blobs = version_blobs_by_campaign[campaign_key]
by_key, version_blobs = _fallback_campaign_assets(
session,
job,
campaign_key=campaign_key,
assets_by_campaign=assets_by_campaign,
version_blobs_by_campaign=version_blobs_by_campaign,
)
for attachment in attachments:
matches = attachment.get("matches") if isinstance(attachment.get("matches"), list) else []
for raw in matches:
if not isinstance(raw, str):
continue
asset = next((by_key[key] for key in _candidate_match_keys(raw) if key in by_key), None)
if not asset:
continue
version_blob = version_blobs.get(asset.id)
if not version_blob:
continue
version, blob = version_blob
_add_use(
session,
job,
asset=asset,
version=version,
blob=blob,
filename_used=asset.filename,
stage=stage,
known_keys=known_keys,
)
_record_fallback_attachment(session, job, attachment, by_key=by_key, version_blobs=version_blobs, stage=stage, known_keys=known_keys)
def _fallback_campaign_assets(
session: Session,
job: CampaignJobLike,
*,
campaign_key: tuple[str, str],
assets_by_campaign: dict[tuple[str, str], dict[str, FileAsset]],
version_blobs_by_campaign: dict[tuple[str, str], dict[str, tuple[FileVersion, FileBlob]]],
) -> tuple[dict[str, FileAsset], dict[str, tuple[FileVersion, FileBlob]]]:
by_key = assets_by_campaign.get(campaign_key)
if by_key is None:
assets = list_assets_for_user(session, tenant_id=job.tenant_id, user_id="", campaign_id=job.campaign_id, is_admin=True)
by_key = _asset_lookup_by_path_and_filename(assets)
assets_by_campaign[campaign_key] = by_key
version_blobs_by_campaign[campaign_key] = current_versions_and_blobs(session, assets)
return by_key, version_blobs_by_campaign[campaign_key]
def _asset_lookup_by_path_and_filename(assets: list[FileAsset]) -> dict[str, FileAsset]:
by_key: dict[str, FileAsset] = {}
for asset in assets:
by_key[asset.display_path.strip("/")] = asset
by_key.setdefault(asset.filename, asset)
return by_key
def _record_fallback_attachment(
session: Session,
job: CampaignJobLike,
attachment: dict[str, object],
*,
by_key: dict[str, FileAsset],
version_blobs: dict[str, tuple[FileVersion, FileBlob]],
stage: str,
known_keys: set[AttachmentUseKey],
) -> None:
matches = attachment.get("matches") if isinstance(attachment.get("matches"), list) else []
for raw in matches:
if not isinstance(raw, str):
continue
asset = next((by_key[key] for key in _candidate_match_keys(raw) if key in by_key), None)
if not asset:
continue
version_blob = version_blobs.get(asset.id)
if not version_blob:
continue
version, blob = version_blob
_add_use(
session,
job,
asset=asset,
version=version,
blob=blob,
filename_used=asset.filename,
stage=stage,
known_keys=known_keys,
)
def record_campaign_attachment_uses_for_job(session: Session, job: CampaignJobLike, *, stage: str = "built") -> None:
@@ -1,6 +1,6 @@
from __future__ import annotations
import os
import json
from collections.abc import Mapping
from dataclasses import dataclass, field
from datetime import datetime, timezone
@@ -9,12 +9,30 @@ from importlib import import_module
import mimetypes
from typing import Any
from urllib.parse import quote, unquote, urljoin, urlsplit
import xml.etree.ElementTree as ET
import xml.etree.ElementTree as ET # nosec B405 - typing/element creation only; parsing uses defusedxml below.
from defusedxml import ElementTree as SafeElementTree
import httpx
from govoplan_core.security.outbound_http import (
OutboundHttpError,
validate_outbound_host,
validate_outbound_http_url,
)
from govoplan_files.backend.storage.http_client import ConnectorHttpError, request_connector_bytes
from govoplan_files.backend.storage.connector_deployment import (
ConnectorDeploymentConfigurationError,
connector_ca_bundle_path,
connector_secret_env_value,
validate_connector_tls_metadata,
)
from govoplan_files.backend.storage.connector_profiles import ConnectorProfile
from govoplan_files.backend.storage.sdk_peer_pinning import (
SdkPeerPinningError,
create_pinned_s3_client,
install_pinned_smb_transport,
pinned_smb_connection_cache,
)
class ConnectorBrowseError(RuntimeError):
@@ -53,7 +71,7 @@ class ConnectorBrowseItem:
}
def browse_connector_profile(profile: ConnectorProfile, *, path: str | None = None, library_id: str | None = None) -> list[ConnectorBrowseItem]:
def browse_connector_profile(profile: ConnectorProfile, *, path: str | None = None, library_id: str | None = None, continuation_token: str | None = None) -> list[ConnectorBrowseItem]:
browse_path = normalize_connector_browse_path(path)
static_items = _static_listing(profile, path=browse_path, library_id=library_id)
if static_items is not None:
@@ -66,6 +84,8 @@ def browse_connector_profile(profile: ConnectorProfile, *, path: str | None = No
return _browse_webdav(profile, path=browse_path)
if profile.provider == "smb":
return _browse_smb(profile, path=browse_path)
if profile.provider == "s3":
return _browse_s3(profile, path=browse_path, library_id=library_id, continuation_token=continuation_token)
raise ConnectorBrowseUnsupported(f"Read-only browsing is not implemented for {profile.provider} connector profiles yet")
@@ -168,14 +188,22 @@ def _browse_webdav(profile: ConnectorProfile, *, path: str) -> list[ConnectorBro
</prop>
</propfind>"""
try:
response = httpx.request("PROPFIND", url, headers=headers, content=body, auth=auth, timeout=15.0)
except httpx.HTTPError as exc:
response = request_connector_bytes(
"PROPFIND",
url,
headers=headers,
content=body,
auth=auth,
timeout=15.0,
label="WebDAV browse",
)
except ConnectorHttpError as exc:
raise ConnectorBrowseError(f"Connector browse failed: {exc}") from exc
if response.status_code in {401, 403}:
raise ConnectorBrowseError("Connector credentials were rejected")
if response.status_code not in {200, 207}:
raise ConnectorBrowseError(f"Connector browse failed with HTTP {response.status_code}")
return parse_webdav_multistatus(root_url=root_url, current_path=path, payload=response.text)
return parse_webdav_multistatus(root_url=root_url, current_path=path, payload=response.content)
def _browse_smb(profile: ConnectorProfile, *, path: str) -> list[ConnectorBrowseItem]:
@@ -222,6 +250,277 @@ def _browse_smb(profile: ConnectorProfile, *, path: str) -> list[ConnectorBrowse
return sorted(items, key=lambda item: (item.kind != "folder", item.name.casefold(), item.path.casefold()))
def _browse_s3(profile: ConnectorProfile, *, path: str, library_id: str | None, continuation_token: str | None) -> list[ConnectorBrowseItem]:
client = _s3_client(profile)
try:
return _browse_s3_with_client(
client,
profile=profile,
path=path,
library_id=library_id,
continuation_token=continuation_token,
)
finally:
close = getattr(client, "close", None)
if callable(close):
close()
def _browse_s3_with_client(
client: Any,
*,
profile: ConnectorProfile,
path: str,
library_id: str | None,
continuation_token: str | None,
) -> list[ConnectorBrowseItem]:
bucket = _s3_bucket(profile, library_id)
if not bucket:
try:
payload = client.list_buckets()
except Exception as exc: # pragma: no cover - concrete exception types are dependency-version specific
raise ConnectorBrowseError(f"S3 connector browse failed: {exc}") from exc
buckets = payload.get("Buckets") if isinstance(payload, Mapping) else None
if not isinstance(buckets, list):
raise ConnectorBrowseError("S3 connector returned invalid bucket list")
return sorted(
(
ConnectorBrowseItem(
kind="library",
name=_clean(item.get("Name")) or "",
path=_clean(item.get("Name")) or "",
external_id=_clean(item.get("Name")),
modified_at=_timestamp(item.get("CreationDate")),
metadata={"bucket": _clean(item.get("Name"))},
)
for item in buckets
if isinstance(item, Mapping) and _clean(item.get("Name"))
),
key=lambda item: item.name.casefold(),
)
prefix = _s3_object_key(profile, path, directory=True)
params: dict[str, object] = {
"Bucket": bucket,
"Prefix": prefix,
"Delimiter": "/",
"MaxKeys": _s3_max_keys(profile),
}
next_page_request = _clean(continuation_token) or _metadata_string(profile, "continuation_token")
if next_page_request:
params["ContinuationToken"] = next_page_request
try:
payload = client.list_objects_v2(**params)
except Exception as exc: # pragma: no cover - concrete exception types are dependency-version specific
raise ConnectorBrowseError(f"S3 connector browse failed: {exc}") from exc
if not isinstance(payload, Mapping):
raise ConnectorBrowseError("S3 connector returned invalid object listing")
items = [
*_s3_prefix_items(bucket=bucket, browse_path=path, base_prefix=prefix, prefixes=payload.get("CommonPrefixes")),
*_s3_object_items(bucket=bucket, browse_path=path, base_prefix=prefix, objects=payload.get("Contents")),
]
next_token = _clean(payload.get("NextContinuationToken"))
if next_token and items:
last = items[-1]
items[-1] = ConnectorBrowseItem(
kind=last.kind,
name=last.name,
path=last.path,
external_id=last.external_id,
external_url=last.external_url,
size_bytes=last.size_bytes,
content_type=last.content_type,
modified_at=last.modified_at,
etag=last.etag,
metadata={**dict(last.metadata), "next_continuation_token": next_token, "listing_truncated": True},
)
return sorted(items, key=lambda item: (item.kind != "folder", item.name.casefold(), item.path.casefold()))
def _s3_client(profile: ConnectorProfile) -> Any:
if profile.secret_ref:
raise ConnectorBrowseError("Secret-ref S3 credentials need a runtime secret resolver before live browsing")
if profile.endpoint_url:
try:
endpoint_url = validate_outbound_http_url(
profile.endpoint_url,
label="S3 connector endpoint",
)
except OutboundHttpError as exc:
raise ConnectorBrowseError(str(exc)) from exc
else:
endpoint_url = None
try:
config_module = import_module("botocore.config")
unsigned = import_module("botocore").UNSIGNED
except ImportError as exc:
raise ConnectorBrowseUnsupported("S3 connector browsing requires the optional boto3 dependency") from exc
kwargs: dict[str, object] = {}
if endpoint_url:
kwargs["endpoint_url"] = endpoint_url
region = _metadata_string(profile, "region") or _metadata_string(profile, "aws_region")
if region:
kwargs["region_name"] = region
access_key = profile.username or _metadata_env(profile, "access_key_id_env") or _metadata_string(profile, "access_key_id")
secret_key = _profile_password(profile) or _metadata_env(profile, "secret_access_key_env")
session_token = _profile_token(profile) or _metadata_env(profile, "session_token_env")
if access_key:
kwargs["aws_access_key_id"] = access_key
if secret_key:
kwargs["aws_secret_access_key"] = secret_key
if session_token:
kwargs["aws_session_token"] = session_token
verify = _s3_verify(profile)
if verify is not None:
kwargs["verify"] = verify
addressing_style = _s3_addressing_style(profile)
config_values: dict[str, object] = {
"proxies": {},
"retries": {"mode": "standard", "max_attempts": 4},
}
if addressing_style:
config_values["s3"] = {"addressing_style": addressing_style}
if bool(access_key) != bool(secret_key):
raise ConnectorBrowseError("S3 connectors require both an access key and a secret key")
if not access_key:
config_values["signature_version"] = unsigned
kwargs["config"] = config_module.Config(**config_values)
try:
return create_pinned_s3_client(**kwargs)
except SdkPeerPinningError as exc:
raise ConnectorBrowseError(str(exc)) from exc
except Exception as exc: # pragma: no cover - concrete exception types are dependency-version specific
raise ConnectorBrowseError(f"S3 connector could not be initialized: {exc}") from exc
def _s3_bucket(profile: ConnectorProfile, library_id: str | None = None) -> str | None:
return _metadata_string(profile, "bucket") or _metadata_string(profile, "bucket_name") or _clean(library_id)
def _s3_object_key(profile: ConnectorProfile, path: str, *, directory: bool = False) -> str:
base_prefix = normalize_connector_browse_path(profile.base_path or _metadata_string(profile, "base_prefix"))
browse_path = normalize_connector_browse_path(path)
key = "/".join(part for part in (base_prefix, browse_path) if part)
if directory and key:
return key.rstrip("/") + "/"
return key
def _s3_prefix_items(
*,
bucket: str,
browse_path: str,
base_prefix: str,
prefixes: object,
) -> list[ConnectorBrowseItem]:
if not isinstance(prefixes, list):
return []
items: list[ConnectorBrowseItem] = []
for item in prefixes:
if not isinstance(item, Mapping):
continue
key = _clean(item.get("Prefix"))
if not key:
continue
relative = _s3_relative_key(key, base_prefix=base_prefix)
name = _path_name(relative)
if not name:
continue
path = _join_browse_path(browse_path, name)
items.append(
ConnectorBrowseItem(
kind="folder",
name=name,
path=path,
external_id=f"{bucket}:{key}",
metadata={"bucket": bucket, "key": key},
)
)
return items
def _s3_object_items(
*,
bucket: str,
browse_path: str,
base_prefix: str,
objects: object,
) -> list[ConnectorBrowseItem]:
if not isinstance(objects, list):
return []
items: list[ConnectorBrowseItem] = []
for item in objects:
if not isinstance(item, Mapping):
continue
key = _clean(item.get("Key"))
if not key or key == base_prefix:
continue
relative = _s3_relative_key(key, base_prefix=base_prefix)
name = _path_name(relative)
if not name:
continue
path = _join_browse_path(browse_path, name)
items.append(
ConnectorBrowseItem(
kind="file",
name=name,
path=path,
external_id=f"{bucket}:{key}",
size_bytes=_int(item.get("Size")),
content_type=mimetypes.guess_type(name)[0],
modified_at=_timestamp(item.get("LastModified")),
etag=_clean(item.get("ETag")),
metadata={
"bucket": bucket,
"key": key,
**({"storage_class": item["StorageClass"]} if "StorageClass" in item else {}),
},
)
)
return items
def _s3_relative_key(key: str, *, base_prefix: str) -> str:
clean_key = key.rstrip("/")
clean_base = base_prefix.rstrip("/")
if clean_base and clean_key.startswith(f"{clean_base}/"):
return clean_key[len(clean_base) + 1 :]
return clean_key
def _s3_max_keys(profile: ConnectorProfile) -> int:
configured = _int(profile.metadata.get("max_keys"))
if configured is None:
return 1000
return max(1, min(configured, 1000))
def _s3_verify(profile: ConnectorProfile) -> bool | str | None:
try:
validate_connector_tls_metadata(profile.metadata)
except ConnectorDeploymentConfigurationError as exc:
raise ConnectorBrowseError(str(exc)) from exc
ca_bundle = _metadata_string(profile, "ca_bundle")
if ca_bundle:
try:
return connector_ca_bundle_path(ca_bundle)
except ConnectorDeploymentConfigurationError as exc:
raise ConnectorBrowseError(str(exc)) from exc
if "verify_tls" in profile.metadata:
return _metadata_bool(profile, "verify_tls", default=True)
if "tls_verify" in profile.metadata:
return _metadata_bool(profile, "tls_verify", default=True)
return None
def _s3_addressing_style(profile: ConnectorProfile) -> str | None:
style = _metadata_string(profile, "addressing_style")
if style in {"path", "virtual", "auto"}:
return style
if _metadata_bool(profile, "path_style", default=False):
return "path"
return None
def _seafile_headers(profile: ConnectorProfile) -> dict[str, str]:
token = _seafile_token(profile)
return {"Authorization": f"Token {token}", "Accept": "application/json"}
@@ -261,16 +560,24 @@ def _request_json(
data: Mapping[str, str] | None = None,
) -> object:
try:
response = httpx.request(method, url, headers=dict(headers or {}), params=params, data=data, timeout=15.0)
except httpx.HTTPError as exc:
response = request_connector_bytes(
method,
url,
headers=dict(headers or {}),
params=params,
data=data,
timeout=15.0,
label="Seafile API",
)
except ConnectorHttpError as exc:
raise ConnectorBrowseError(f"Connector browse failed: {exc}") from exc
if response.status_code in {401, 403}:
raise ConnectorBrowseError("Connector credentials were rejected")
if response.status_code not in {200, 201}:
raise ConnectorBrowseError(f"Connector browse failed with HTTP {response.status_code}")
try:
return response.json()
except ValueError as exc:
return json.loads(response.content)
except (UnicodeDecodeError, ValueError) as exc:
raise ConnectorBrowseError("Connector returned invalid JSON") from exc
@@ -444,11 +751,18 @@ def _metadata_bool(profile: ConnectorProfile, key: str, *, default: bool = False
return str(value).strip().casefold() in {"1", "true", "yes", "on"}
def _env_required(name: str, profile_id: str) -> str:
value = _clean(os.environ.get(name))
if not value:
raise ConnectorBrowseError(f"Connector profile {profile_id} is missing required credential environment")
return value
def _metadata_env(profile: ConnectorProfile, key: str) -> str | None:
env_name = _metadata_string(profile, key)
if not env_name:
return None
return _env_required(env_name, profile)
def _env_required(name: str, profile: ConnectorProfile) -> str:
try:
return connector_secret_env_value(name, source_kind=profile.source_kind)
except ConnectorDeploymentConfigurationError as exc:
raise ConnectorBrowseError(f"Connector profile {profile.id}: {exc}") from exc
def normalize_connector_browse_path(value: object) -> str:
@@ -505,6 +819,11 @@ def _smb_location(profile: ConnectorProfile) -> _SmbLocation:
server = _clean(parsed.hostname)
if not server:
raise ConnectorBrowseError("SMB connector endpoint_url must include a server")
port = parsed.port or _int(profile.metadata.get("port")) or 445
try:
validate_outbound_host(server, port=port, label="SMB connector endpoint")
except OutboundHttpError as exc:
raise ConnectorBrowseError(str(exc)) from exc
path_parts = [part for part in unquote(parsed.path or "").strip("/").split("/") if part]
if not path_parts:
raise ConnectorBrowseError("SMB connector endpoint_url must include a share name")
@@ -514,7 +833,7 @@ def _smb_location(profile: ConnectorProfile) -> _SmbLocation:
return _SmbLocation(
server=server,
share=path_parts[0],
port=parsed.port or _int(profile.metadata.get("port")) or 445,
port=port,
root_path=normalize_connector_browse_path("/".join(root_parts)),
)
@@ -529,6 +848,7 @@ def _smb_unc_path(location: _SmbLocation, path: str) -> str:
def _smb_client_kwargs(profile: ConnectorProfile, location: _SmbLocation) -> dict[str, object]:
kwargs: dict[str, object] = {
"port": location.port,
"connection_cache": pinned_smb_connection_cache(),
"require_signing": _metadata_bool(profile, "require_signing", default=True),
"auth_protocol": _metadata_string(profile, "auth_protocol") or "ntlm",
}
@@ -550,7 +870,7 @@ def _profile_password(profile: ConnectorProfile) -> str | None:
if profile.password_value:
return profile.password_value
if profile.password_env:
return _env_required(profile.password_env, profile.id)
return _env_required(profile.password_env, profile)
return None
@@ -558,15 +878,17 @@ def _profile_token(profile: ConnectorProfile) -> str | None:
if profile.token_value:
return profile.token_value
if profile.token_env:
return _env_required(profile.token_env, profile.id)
return _env_required(profile.token_env, profile)
return None
def _smbclient_module() -> Any:
try:
return import_module("smbclient")
return install_pinned_smb_transport(import_module("smbclient"))
except ImportError as exc:
raise ConnectorBrowseUnsupported("SMB connector browsing requires the optional smbprotocol dependency") from exc
except SdkPeerPinningError as exc:
raise ConnectorBrowseUnsupported(str(exc)) from exc
def _smb_entry_stat(entry: object) -> object | None:
@@ -0,0 +1,337 @@
from __future__ import annotations
from dataclasses import dataclass
from sqlalchemy import inspect
from sqlalchemy.orm import Session
from govoplan_core.audit.logging import audit_event
from govoplan_files.backend.db.models import FileConnectorCredential, FileConnectorProfile
@dataclass(frozen=True, slots=True)
class ConnectorCredentialDeletionResult:
changed: bool
affected_profiles: tuple[FileConnectorProfile, ...] = ()
def delete_connector_credential_row(
session: Session,
row: FileConnectorCredential,
*,
deletion_reason: str,
user_id: str | None = None,
api_key_id: str | None = None,
) -> ConnectorCredentialDeletionResult:
"""Scrub a credential tombstone and disable every profile that used it.
``secret_ref`` predates a Files-owned secret-provider contract. It may be
shared or deployment-owned, so it is detached locally and explicitly
audited without ever being passed to a provider delete operation.
"""
dependent_profiles = (
tuple(
session.query(FileConnectorProfile)
.filter(FileConnectorProfile.credential_profile_id == row.id)
.order_by(FileConnectorProfile.id.asc())
.all()
)
if inspect(session.get_bind()).has_table(FileConnectorProfile.__tablename__)
else ()
)
affected_profiles: list[FileConnectorProfile] = []
for profile in dependent_profiles:
if _delete_connector_profile_row(
session,
profile,
deletion_reason="credential_deleted",
user_id=user_id,
api_key_id=api_key_id,
):
affected_profiles.append(profile)
deleted_secret_kinds = _encrypted_secret_kinds(row)
removed_reference_kinds = _credential_reference_kinds(row)
removed_metadata = bool(row.metadata_)
changed = _credential_row_requires_deletion(row)
if changed:
_scrub_credential_row(row, user_id=user_id)
session.add(row)
_audit_credential_deletion(
session,
row,
deletion_reason=deletion_reason,
deleted_secret_kinds=deleted_secret_kinds,
removed_reference_kinds=removed_reference_kinds,
removed_metadata=removed_metadata,
affected_profile_count=len(affected_profiles),
user_id=user_id,
api_key_id=api_key_id,
)
session.flush()
return ConnectorCredentialDeletionResult(
changed=changed or bool(affected_profiles),
affected_profiles=tuple(affected_profiles),
)
def delete_connector_profile_row(
session: Session,
row: FileConnectorProfile,
*,
deletion_reason: str,
user_id: str | None = None,
api_key_id: str | None = None,
) -> bool:
changed = _delete_connector_profile_row(
session,
row,
deletion_reason=deletion_reason,
user_id=user_id,
api_key_id=api_key_id,
)
session.flush()
return changed
def delete_connector_credentials_for_retirement(session: Session) -> int:
"""Scrub and audit stored connector material before destructive retirement.
Legacy external references are detached and audited as non-owned. Files
never sends those arbitrary references to a provider delete operation.
"""
inspector = inspect(session.get_bind())
profiles = (
session.query(FileConnectorProfile).order_by(FileConnectorProfile.id.asc()).all()
if inspector.has_table(FileConnectorProfile.__tablename__)
else []
)
credentials = (
session.query(FileConnectorCredential).order_by(FileConnectorCredential.id.asc()).all()
if inspector.has_table(FileConnectorCredential.__tablename__)
else []
)
deleted = 0
for profile in profiles:
if not _profile_row_has_credential_material(profile):
continue
if _delete_connector_profile_row(
session,
profile,
deletion_reason="module_data_retired",
):
deleted += 1
for credential in credentials:
if not _credential_row_has_material(credential):
continue
result = delete_connector_credential_row(
session,
credential,
deletion_reason="module_data_retired",
)
if result.changed:
deleted += 1
session.flush()
return deleted
def _delete_connector_profile_row(
session: Session,
row: FileConnectorProfile,
*,
deletion_reason: str,
user_id: str | None = None,
api_key_id: str | None = None,
) -> bool:
deleted_secret_kinds = _encrypted_secret_kinds(row)
removed_reference_kinds = _profile_reference_kinds(row)
removed_metadata = bool(row.metadata_)
changed = _profile_row_requires_deletion(row)
if not changed:
return False
_scrub_profile_row(row, user_id=user_id)
session.add(row)
_audit_profile_deletion(
session,
row,
deletion_reason=deletion_reason,
deleted_secret_kinds=deleted_secret_kinds,
removed_reference_kinds=removed_reference_kinds,
removed_metadata=removed_metadata,
user_id=user_id,
api_key_id=api_key_id,
)
return True
def _scrub_credential_row(row: FileConnectorCredential, *, user_id: str | None) -> None:
row.enabled = False
row.credential_mode = "none"
row.username = None
row.password_encrypted = None
row.token_encrypted = None
row.password_env = None
row.token_env = None
row.secret_ref = None
row.metadata_ = {}
row.updated_by_user_id = user_id
def _scrub_profile_row(row: FileConnectorProfile, *, user_id: str | None) -> None:
row.enabled = False
row.credential_profile_id = None
row.credential_mode = "none"
row.username = None
row.password_encrypted = None
row.token_encrypted = None
row.password_env = None
row.token_env = None
row.secret_ref = None
row.metadata_ = {}
row.updated_by_user_id = user_id
def _credential_row_requires_deletion(row: FileConnectorCredential) -> bool:
return bool(row.enabled or _credential_row_has_material(row))
def _profile_row_requires_deletion(row: FileConnectorProfile) -> bool:
return bool(row.enabled or _profile_row_has_credential_material(row))
def _credential_row_has_material(row: FileConnectorCredential) -> bool:
return bool(
row.username
or row.password_encrypted
or row.token_encrypted
or row.password_env
or row.token_env
or row.secret_ref
or row.metadata_
or row.credential_mode not in {"", "none", "anonymous"}
)
def _profile_row_has_credential_material(row: FileConnectorProfile) -> bool:
return bool(
row.credential_profile_id
or row.username
or row.password_encrypted
or row.token_encrypted
or row.password_env
or row.token_env
or row.secret_ref
or row.metadata_
or row.credential_mode not in {"", "none", "anonymous"}
)
def _encrypted_secret_kinds(row: FileConnectorCredential | FileConnectorProfile) -> list[str]:
kinds: list[str] = []
if row.password_encrypted:
kinds.append("password")
if row.token_encrypted:
kinds.append("token")
return kinds
def _credential_reference_kinds(row: FileConnectorCredential | FileConnectorProfile) -> list[str]:
kinds: list[str] = []
if row.password_env:
kinds.append("password_env")
if row.token_env:
kinds.append("token_env")
if row.secret_ref:
kinds.append("unowned_external_secret_ref")
return kinds
def _profile_reference_kinds(row: FileConnectorProfile) -> list[str]:
kinds = _credential_reference_kinds(row)
if row.credential_profile_id:
kinds.append("credential_profile")
return kinds
def _storage_backend(deleted_secret_kinds: list[str], removed_reference_kinds: list[str]) -> str:
if "unowned_external_secret_ref" in removed_reference_kinds:
return "unowned_external_reference_detached"
if deleted_secret_kinds:
return "encrypted_database"
if removed_reference_kinds:
return "reference_only"
return "none"
def _audit_credential_deletion(
session: Session,
row: FileConnectorCredential,
*,
deletion_reason: str,
deleted_secret_kinds: list[str],
removed_reference_kinds: list[str],
removed_metadata: bool,
affected_profile_count: int,
user_id: str | None,
api_key_id: str | None,
) -> None:
audit_event(
session,
tenant_id=row.tenant_id,
scope=_audit_scope(row.scope_type),
user_id=user_id,
api_key_id=api_key_id,
action="files.connector_credential_deleted",
object_type="file_connector_credential",
object_id=row.id,
details={
"scope_type": row.scope_type,
"scope_id": row.scope_id,
"provider": row.provider,
"storage_backend": _storage_backend(deleted_secret_kinds, removed_reference_kinds),
"deleted_secret_kinds": deleted_secret_kinds,
"removed_reference_kinds": removed_reference_kinds,
"removed_metadata": removed_metadata,
"affected_profile_count": affected_profile_count,
"deletion_reason": deletion_reason,
},
)
def _audit_profile_deletion(
session: Session,
row: FileConnectorProfile,
*,
deletion_reason: str,
deleted_secret_kinds: list[str],
removed_reference_kinds: list[str],
removed_metadata: bool,
user_id: str | None,
api_key_id: str | None,
) -> None:
audit_event(
session,
tenant_id=row.tenant_id,
scope=_audit_scope(row.scope_type),
user_id=user_id,
api_key_id=api_key_id,
action="files.connector_profile_deleted",
object_type="file_connector_profile",
object_id=row.id,
details={
"scope_type": row.scope_type,
"scope_id": row.scope_id,
"provider": row.provider,
"storage_backend": _storage_backend(deleted_secret_kinds, removed_reference_kinds),
"deleted_secret_kinds": deleted_secret_kinds,
"removed_reference_kinds": removed_reference_kinds,
"removed_metadata": removed_metadata,
"deletion_reason": deletion_reason,
},
)
def _audit_scope(scope_type: str) -> str:
return "system" if scope_type == "system" else "tenant"
@@ -5,14 +5,26 @@ from dataclasses import dataclass
from typing import Any
from sqlalchemy.orm import Session
from sqlalchemy import inspect
from govoplan_core.core.policy import normalize_policy_scope_type, policy_source_path
from govoplan_core.security.credential_envelopes import (
CredentialAccessContext,
CredentialEnvelope,
CredentialEnvelopeError,
ResolvedCredentialEnvelope,
list_credential_envelopes,
resolve_credential_envelope,
)
from govoplan_core.security.secrets import decrypt_secret, encrypt_secret
from govoplan_files.backend.db.models import FileConnectorCredential
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.connector_deployment import reject_api_controlled_deployment_references
from govoplan_files.backend.storage.connector_policy import ConnectorPolicySource, connector_policy_sources_from_payload
from govoplan_files.backend.storage.connector_profiles import supported_connector_providers
CORE_CREDENTIAL_ENVELOPE_PREFIX = "credential-envelope:"
@dataclass(frozen=True, slots=True)
class ConnectorCredential:
@@ -32,6 +44,7 @@ class ConnectorCredential:
policy_sources: tuple[ConnectorPolicySource, ...] = ()
metadata: Mapping[str, Any] | None = None
source_kind: str = "database"
has_secret: bool = False
@property
def source_path(self) -> str:
@@ -51,9 +64,13 @@ class ConnectorCredential:
@property
def credentials_configured(self) -> bool:
if not self.enabled:
return False
if self.credential_mode.casefold() in {"", "none", "anonymous"}:
return True
return bool(self.secret_ref or self.password_value or self.token_value or self.password_env or self.token_env)
# Environment references are deliberately unavailable to API-managed
# credential rows. Legacy rows remain visible but fail closed.
return bool(self.has_secret or self.secret_ref or self.password_value or self.token_value)
def to_response(self) -> dict[str, Any]:
return {
@@ -80,7 +97,159 @@ def list_database_connector_credentials(
tenant_id: str,
include_disabled: bool = False,
) -> list[ConnectorCredential]:
return [connector_credential_from_row(row) for row in list_connector_credential_rows(session, tenant_id=tenant_id, include_disabled=include_disabled)]
local = [
connector_credential_from_row(row)
for row in list_connector_credential_rows(
session,
tenant_id=tenant_id,
include_disabled=include_disabled,
)
]
return [
*local,
*list_reusable_connector_credentials(
session,
tenant_id=tenant_id,
include_disabled=include_disabled,
),
]
def reusable_credential_reference(credential_id: str) -> str:
return f"{CORE_CREDENTIAL_ENVELOPE_PREFIX}{credential_id}"
def reusable_credential_id(credential_ref: str | None) -> str | None:
if not credential_ref or not credential_ref.startswith(CORE_CREDENTIAL_ENVELOPE_PREFIX):
return None
value = credential_ref.removeprefix(CORE_CREDENTIAL_ENVELOPE_PREFIX).strip()
return value or None
def file_credential_context(
*,
tenant_id: str,
profile_id: str | None = None,
scope_type: str = "tenant",
scope_id: str | None = None,
administrative: bool = False,
) -> CredentialAccessContext:
return CredentialAccessContext(
tenant_id=tenant_id,
user_id=scope_id if scope_type == "user" else None,
group_ids=frozenset({scope_id}) if scope_type == "group" and scope_id else frozenset(),
target_scope_type=scope_type,
target_scope_id=scope_id or tenant_id,
module_id="files",
server_ref=f"files:{profile_id}" if profile_id else None,
administrative=administrative,
)
def list_reusable_connector_credentials(
session: Session,
*,
tenant_id: str,
include_disabled: bool = False,
) -> list[ConnectorCredential]:
if not inspect(session.get_bind()).has_table(CredentialEnvelope.__tablename__):
return []
context = file_credential_context(tenant_id=tenant_id, administrative=True)
return [
connector_credential_from_envelope(row)
for row in list_credential_envelopes(
session,
context=context,
include_inactive=include_disabled,
)
]
def resolve_reusable_connector_credential(
session: Session,
*,
tenant_id: str,
credential_ref: str,
profile_id: str,
scope_type: str,
scope_id: str | None,
) -> ConnectorCredential:
credential_id = reusable_credential_id(credential_ref)
if credential_id is None:
raise FileStorageError("Reusable credential reference is invalid")
try:
resolved = resolve_credential_envelope(
session,
credential_id=credential_id,
context=file_credential_context(
tenant_id=tenant_id,
profile_id=profile_id,
scope_type=scope_type,
scope_id=scope_id,
),
)
except CredentialEnvelopeError as exc:
raise FileStorageError("Reusable credential is unavailable to this file connection") from exc
return connector_credential_from_resolved_envelope(
resolved,
scope_type=scope_type,
scope_id=scope_id,
)
def connector_credential_from_envelope(row: CredentialEnvelope) -> ConnectorCredential:
return ConnectorCredential(
id=reusable_credential_reference(row.id),
label=row.name,
scope_type=row.scope_type,
scope_id=row.scope_id,
enabled=row.is_active,
credential_mode=_envelope_credential_mode(row.credential_kind, row.secret_keys),
username=_clean_public_value(row.public_data, "username"),
source_kind="credential_envelope",
has_secret=bool(row.secret_data_encrypted),
metadata={
"credential_envelope_id": row.id,
"credential_kind": row.credential_kind,
"allowed_modules": list(row.allowed_modules or []),
"allowed_server_refs": list(row.allowed_server_refs or []),
"inherit_to_lower_scopes": bool(row.inherit_to_lower_scopes),
"revision": row.revision,
},
)
def connector_credential_from_resolved_envelope(
row: ResolvedCredentialEnvelope,
*,
scope_type: str,
scope_id: str | None,
) -> ConnectorCredential:
return ConnectorCredential(
id=reusable_credential_reference(row.id),
label=row.name,
scope_type=scope_type,
scope_id=scope_id,
enabled=True,
credential_mode=_envelope_credential_mode(row.credential_kind, row.secret_data),
username=_clean_public_value(row.public_data, "username"),
password_value=_first_secret(row.secret_data, "password", "secret"),
token_value=_first_secret(
row.secret_data,
"access_token",
"bearer_token",
"token",
"api_key",
),
source_kind="credential_envelope",
has_secret=bool(row.secret_data),
metadata={
"credential_envelope_id": row.id,
"credential_kind": row.credential_kind,
"inherit_to_lower_scopes": True,
"revision": row.revision,
},
)
def list_connector_credential_rows(
@@ -182,6 +351,12 @@ def create_connector_credential_row(
policy: Mapping[str, Any] | None = None,
metadata: Mapping[str, Any] | None = None,
) -> FileConnectorCredential:
reject_api_controlled_deployment_references(
password_env=password_env,
token_env=token_env,
secret_ref=secret_ref,
metadata=metadata,
)
clean_id = _normalize_id(credential_id)
if session.get(FileConnectorCredential, clean_id) is not None:
raise FileStorageError(f"Connector credential already exists: {clean_id}")
@@ -231,6 +406,18 @@ def update_connector_credential_row(
clear_password: bool = False,
clear_token: bool = False,
) -> FileConnectorCredential:
reject_api_controlled_deployment_references(
password_env=password_env,
token_env=token_env,
secret_ref=secret_ref,
metadata=metadata,
)
if secret_ref is not None and _clean(secret_ref) != _clean(row.secret_ref):
if _clean(row.secret_ref):
raise FileStorageError(
"An existing external secret reference cannot be replaced or cleared until Files can prove "
"provider ownership and confirm provider-side deletion"
)
if label is not None:
row.label = _normalize_label(label)
if provider is not None:
@@ -265,14 +452,6 @@ def update_connector_credential_row(
return row
def deactivate_connector_credential_row(session: Session, row: FileConnectorCredential, *, user_id: str | None) -> FileConnectorCredential:
row.enabled = False
row.updated_by_user_id = user_id
session.add(row)
session.flush()
return row
def _normalize_scope(*, tenant_id: str, scope_type: str, scope_id: str | None) -> tuple[str, str | None, str | None]:
clean_scope_type = normalize_policy_scope_type(scope_type)
clean_scope_id = _clean(scope_id)
@@ -333,6 +512,27 @@ def _policy_source_response(source: ConnectorPolicySource) -> dict[str, Any]:
}
def _envelope_credential_mode(kind: str, secret_values: Mapping[str, Any] | list[str]) -> str:
keys = {str(key) for key in secret_values}
if kind in {"token", "oauth2", "api_key"} or keys.intersection(
{"access_token", "bearer_token", "token", "api_key"}
):
return "token"
return "basic"
def _clean_public_value(values: Mapping[str, Any] | None, key: str) -> str | None:
return _clean((values or {}).get(key))
def _first_secret(values: Mapping[str, Any], *keys: str) -> str | None:
for key in keys:
value = _clean(values.get(key))
if value is not None:
return value
return None
def _clean(value: object) -> str | None:
if value is None:
return None
@@ -0,0 +1,241 @@
from __future__ import annotations
import os
import re
from collections.abc import Mapping
from pathlib import Path
from typing import Any
_SECRET_ENV_ALLOWLIST = "GOVOPLAN_CONNECTOR_SECRET_ENV_ALLOWLIST" # noqa: S105 # nosec B105 - configuration key.
_CA_BUNDLE_ALLOWLIST = "GOVOPLAN_CONNECTOR_CA_BUNDLE_ALLOWLIST"
_ENV_NAME = re.compile(r"[A-Za-z_][A-Za-z0-9_]*\Z")
_SECRET_ENV_METADATA_KEYS = frozenset(
{
"access_key_id_env",
"secret_access_key_env",
"session_token_env",
}
)
_SECRET_VALUE_METADATA_KEYS = frozenset(
{
"password",
"token",
"access_token",
"auth_token",
"bearer_token",
"refresh_token",
"api_key",
"access_key",
"access_key_id",
"secret_key",
"secret_access_key",
"session_token",
}
)
_DEVELOPMENT_ENVIRONMENTS = frozenset({"dev", "development", "local", "test", "testing"})
class ConnectorDeploymentConfigurationError(ValueError):
"""Raised when connector data crosses a deployment-owned trust boundary."""
def connector_secret_env_value(name: str, *, source_kind: str) -> str:
clean_name = _validate_secret_env_reference(name, source_kind=source_kind)
value = os.environ.get(clean_name)
if value is None or value == "":
raise ConnectorDeploymentConfigurationError("Connector credential environment variable is not configured")
return value
def connector_secret_env_available(name: str | None, *, source_kind: str) -> bool:
if not name:
return False
try:
clean_name = _validate_secret_env_reference(name, source_kind=source_kind)
except ConnectorDeploymentConfigurationError:
return False
return bool(os.environ.get(clean_name))
def validate_deployment_connector_references(
*,
source_kind: str,
password_env: str | None = None,
token_env: str | None = None,
metadata: Mapping[str, Any] | None = None,
) -> None:
"""Validate references in deployment-owned connector configuration."""
for name in (password_env, token_env):
if name:
_validate_secret_env_reference(name, source_kind=source_kind)
for key in _SECRET_ENV_METADATA_KEYS:
name = _clean((metadata or {}).get(key))
if name:
_validate_secret_env_reference(name, source_kind=source_kind)
validate_connector_tls_metadata(metadata)
def reject_api_controlled_deployment_references(
*,
password_env: str | None = None,
token_env: str | None = None,
secret_ref: str | None = None,
metadata: Mapping[str, Any] | None = None,
) -> None:
"""Reject process-secret selectors controlled through tenant-facing APIs."""
if _clean(password_env) or _clean(token_env):
raise ConnectorDeploymentConfigurationError(
"Environment-backed credentials may only be declared in deployment-owned connector configuration"
)
if _clean(secret_ref):
raise ConnectorDeploymentConfigurationError(
"External secret references cannot be managed through the Files API until Files can prove "
"provider ownership and confirm provider-side deletion"
)
for key, value in _metadata_entries(metadata or {}):
clean_key = str(key).strip().casefold()
if _clean(value) and (clean_key in _SECRET_ENV_METADATA_KEYS or clean_key.endswith("_env")):
raise ConnectorDeploymentConfigurationError(
"Environment-backed credentials may only be declared in deployment-owned connector configuration"
)
if _clean(value) and (
clean_key in _SECRET_VALUE_METADATA_KEYS
or clean_key.endswith(("_password", "_secret", "_api_key"))
):
raise ConnectorDeploymentConfigurationError(
"Connector credential values must use the dedicated encrypted credential fields, not metadata"
)
validate_connector_tls_metadata(metadata)
def _metadata_entries(value: object) -> list[tuple[str, object]]:
entries: list[tuple[str, object]] = []
if isinstance(value, Mapping):
for key, item in value.items():
entries.append((str(key), item))
entries.extend(_metadata_entries(item))
elif isinstance(value, (list, tuple)):
for item in value:
entries.extend(_metadata_entries(item))
return entries
def connector_ca_bundle_path(value: str) -> str:
clean_value = _clean(value)
if not clean_value:
raise ConnectorDeploymentConfigurationError("Connector CA bundle path is empty")
candidate = Path(clean_value)
if not candidate.is_absolute():
raise ConnectorDeploymentConfigurationError("Connector CA bundle paths must be absolute")
try:
resolved = candidate.resolve(strict=True)
except OSError as exc:
raise ConnectorDeploymentConfigurationError("Connector CA bundle path does not exist") from exc
allowed = _allowed_ca_bundle_paths()
if resolved not in allowed:
raise ConnectorDeploymentConfigurationError(
f"Connector CA bundle path is not listed in {_CA_BUNDLE_ALLOWLIST}"
)
if not resolved.is_file():
raise ConnectorDeploymentConfigurationError("Connector CA bundle path must be a regular file")
return str(resolved)
def validate_connector_tls_metadata(metadata: Mapping[str, Any] | None) -> None:
values = metadata or {}
ca_bundle = _clean(values.get("ca_bundle"))
if ca_bundle:
connector_ca_bundle_path(ca_bundle)
for key in ("verify_tls", "tls_verify"):
if key in values and not _as_bool(values.get(key), default=True) and not _development_runtime():
raise ConnectorDeploymentConfigurationError(
"Connector TLS certificate verification may only be disabled in dev/test environments"
)
def connector_effective_endpoint_url(
*,
provider: str | None,
endpoint_url: str | None,
metadata: Mapping[str, Any] | None,
) -> str | None:
"""Return the endpoint that connector I/O will actually use."""
clean_provider = (provider or "").strip().casefold()
values = metadata or {}
webdav_url = _clean(values.get("webdav_endpoint_url"))
browse_protocol = (_clean(values.get("browse_protocol")) or "").casefold()
if webdav_url and (clean_provider in {"seafile", "webdav", "nextcloud"} or browse_protocol == "webdav"):
return webdav_url
return _clean(endpoint_url)
def _validate_secret_env_reference(name: str, *, source_kind: str) -> str:
if source_kind.strip().casefold() != "settings":
raise ConnectorDeploymentConfigurationError(
"Environment-backed credentials may only be used by deployment-owned connector configuration"
)
clean_name = name.strip()
if not _ENV_NAME.fullmatch(clean_name):
raise ConnectorDeploymentConfigurationError("Connector credential environment variable name is invalid")
if clean_name not in _allowed_secret_env_names():
raise ConnectorDeploymentConfigurationError(
f"Connector credential environment variable is not listed in {_SECRET_ENV_ALLOWLIST}"
)
return clean_name
def _allowed_secret_env_names() -> frozenset[str]:
raw = os.environ.get(_SECRET_ENV_ALLOWLIST, "")
names = frozenset(item.strip() for item in raw.split(",") if item.strip())
invalid = sorted(name for name in names if not _ENV_NAME.fullmatch(name))
if invalid:
raise ConnectorDeploymentConfigurationError(f"{_SECRET_ENV_ALLOWLIST} contains an invalid variable name")
return names
def _allowed_ca_bundle_paths() -> frozenset[Path]:
raw = os.environ.get(_CA_BUNDLE_ALLOWLIST, "")
paths: set[Path] = set()
for item in (part.strip() for part in raw.split(",")):
if not item:
continue
path = Path(item)
if not path.is_absolute():
raise ConnectorDeploymentConfigurationError(f"{_CA_BUNDLE_ALLOWLIST} requires absolute paths")
try:
paths.add(path.resolve(strict=True))
except OSError as exc:
raise ConnectorDeploymentConfigurationError(f"{_CA_BUNDLE_ALLOWLIST} contains a missing path") from exc
return frozenset(paths)
def _development_runtime() -> bool:
app_env = os.environ.get("APP_ENV", "dev").strip().casefold()
install_profile = os.environ.get("GOVOPLAN_INSTALL_PROFILE", "").strip().casefold()
return app_env in _DEVELOPMENT_ENVIRONMENTS and (
not install_profile or install_profile in _DEVELOPMENT_ENVIRONMENTS
)
def _as_bool(value: object, *, default: bool) -> bool:
if value is None:
return default
if isinstance(value, bool):
return value
clean = str(value).strip().casefold()
if clean in {"1", "true", "yes", "on"}:
return True
if clean in {"0", "false", "no", "off"}:
return False
raise ConnectorDeploymentConfigurationError("Connector TLS verification setting must be true or false")
def _clean(value: object) -> str | None:
if value is None:
return None
clean = str(value).strip()
return clean or None
@@ -4,7 +4,7 @@ from dataclasses import dataclass, field
import mimetypes
from typing import Any
import httpx
from govoplan_core.security.outbound_http import response_limit
from govoplan_files.backend.storage.connector_browse import (
ConnectorBrowseError,
@@ -21,12 +21,16 @@ from govoplan_files.backend.storage.connector_browse import (
_smb_stat_size,
_smb_unc_path,
_smbclient_module,
_s3_bucket,
_s3_client,
_s3_object_key,
_seafile_headers,
_seafile_url,
_webdav_url,
normalize_connector_browse_path,
)
from govoplan_files.backend.storage.connector_profiles import ConnectorProfile
from govoplan_files.backend.storage.http_client import ConnectorHttpError, request_connector_bytes
from govoplan_files.backend.storage.paths import filename_from_path
@@ -56,6 +60,7 @@ def read_connector_file(
path: str,
max_bytes: int,
) -> ConnectorDownloadedFile:
max_bytes = min(max_bytes, response_limit("file"))
if profile.provider == "seafile":
if _metadata_string(profile, "browse_protocol") == "webdav" or _metadata_string(profile, "webdav_endpoint_url"):
return _read_webdav_file(profile, path=path, max_bytes=max_bytes)
@@ -64,6 +69,8 @@ def read_connector_file(
return _read_webdav_file(profile, path=path, max_bytes=max_bytes)
if profile.provider == "smb":
return _read_smb_file(profile, path=path, max_bytes=max_bytes)
if profile.provider == "s3":
return _read_s3_file(profile, library_id=library_id, path=path, max_bytes=max_bytes)
raise ConnectorImportUnsupported(f"Connector file import is not implemented for {profile.provider} profiles yet")
@@ -97,8 +104,15 @@ def _read_seafile_file(profile: ConnectorProfile, *, library_id: str, path: str,
if not isinstance(download_url, str) or not download_url.strip():
raise ConnectorImportError("Seafile did not return a file download URL")
try:
response = httpx.request("GET", download_url, timeout=30.0)
except httpx.HTTPError as exc:
response = request_connector_bytes(
"GET",
download_url,
timeout=30.0,
kind="file",
max_bytes=max_bytes,
label="Seafile file download",
)
except ConnectorHttpError as exc:
raise ConnectorImportError(f"Seafile file download failed: {exc}") from exc
if response.status_code != 200:
raise ConnectorImportError(f"Seafile file download failed with HTTP {response.status_code}")
@@ -144,8 +158,17 @@ def _read_webdav_file(profile: ConnectorProfile, *, path: str, max_bytes: int) -
elif profile.credential_mode.casefold() not in {"", "none", "anonymous"} and profile.secret_ref:
raise ConnectorImportError("Secret-ref WebDAV credentials need a runtime secret resolver before live import")
try:
response = httpx.request("GET", url, headers=headers, auth=auth, timeout=30.0)
except httpx.HTTPError as exc:
response = request_connector_bytes(
"GET",
url,
headers=headers,
auth=auth,
timeout=30.0,
kind="file",
max_bytes=max_bytes,
label="WebDAV file download",
)
except ConnectorHttpError as exc:
raise ConnectorImportError(f"WebDAV file download failed: {exc}") from exc
if response.status_code in {401, 403}:
raise ConnectorImportError("Connector credentials were rejected")
@@ -171,12 +194,12 @@ def _read_webdav_file(profile: ConnectorProfile, *, path: str, max_bytes: int) -
def _read_smb_file(profile: ConnectorProfile, *, path: str, max_bytes: int) -> ConnectorDownloadedFile:
location = _smb_location(profile)
file_path = normalize_connector_browse_path(path)
if not file_path:
raise ConnectorImportError("SMB import requires a file path")
unc_path = _smb_unc_path(location, file_path)
try:
location = _smb_location(profile)
unc_path = _smb_unc_path(location, file_path)
smbclient = _smbclient_module()
kwargs = _smb_client_kwargs(profile, location)
stat_result = smbclient.stat(unc_path, **kwargs)
@@ -213,6 +236,119 @@ def _read_smb_file(profile: ConnectorProfile, *, path: str, max_bytes: int) -> C
)
def _s3_import_client(profile: ConnectorProfile) -> Any:
try:
return _s3_client(profile)
except ConnectorBrowseUnsupported as exc:
raise ConnectorImportUnsupported(str(exc)) from exc
except ConnectorBrowseError as exc:
raise ConnectorImportError(str(exc)) from exc
def _s3_object_detail(client: Any, *, bucket: str, key: str, max_bytes: int) -> dict[str, Any]:
try:
detail = client.head_object(Bucket=bucket, Key=key)
except Exception as exc: # pragma: no cover - concrete exception types are dependency-version specific
raise ConnectorImportError(f"S3 object metadata lookup failed: {exc}") from exc
if not isinstance(detail, dict):
raise ConnectorImportError("S3 connector returned invalid object metadata")
size = _int(detail.get("ContentLength"))
if size is not None and size > max_bytes:
raise ConnectorImportError(f"S3 object exceeds limit of {max_bytes} bytes")
return detail
def _download_s3_object(
client: Any,
*,
bucket: str,
key: str,
version_id: str | None,
max_bytes: int,
) -> tuple[Any, bytes]:
request: dict[str, object] = {"Bucket": bucket, "Key": key}
if version_id:
request["VersionId"] = version_id
try:
response = client.get_object(**request)
body = response.get("Body")
data = body.read(max_bytes + 1) if hasattr(body, "read") else bytes(response.get("Body") or b"")
except Exception as exc: # pragma: no cover - concrete exception types are dependency-version specific
raise ConnectorImportError(f"S3 object download failed: {exc}") from exc
if len(data) > max_bytes:
raise ConnectorImportError(f"S3 object exceeds limit of {max_bytes} bytes")
return response, data
def _s3_download_metadata(
detail: dict[str, Any],
*,
bucket: str,
key: str,
version_id: str | None,
etag: str | None,
size: int,
) -> dict[str, Any]:
metadata: dict[str, Any] = {
"bucket": bucket,
"key": key,
"version_id": version_id,
"etag": etag,
"size": size,
}
for source_key, target_key in (
("ChecksumSHA256", "checksum_sha256"),
("ChecksumCRC32", "checksum_crc32"),
("StorageClass", "storage_class"),
):
if source_key in detail:
metadata[target_key] = detail[source_key]
return metadata
def _read_s3_file(profile: ConnectorProfile, *, library_id: str, path: str, max_bytes: int) -> ConnectorDownloadedFile:
bucket = _s3_bucket(profile, library_id)
if not bucket:
raise ConnectorImportError("S3 import requires a bucket in library_id or profile metadata")
key = _s3_object_key(profile, path)
if not key:
raise ConnectorImportError("S3 import requires an object key")
client = _s3_import_client(profile)
try:
detail = _s3_object_detail(client, bucket=bucket, key=key, max_bytes=max_bytes)
version_id = _clean(detail.get("VersionId"))
response, data = _download_s3_object(
client,
bucket=bucket,
key=key,
version_id=version_id,
max_bytes=max_bytes,
)
finally:
close = getattr(client, "close", None)
if callable(close):
close()
content_type = _clean(response.get("ContentType") if isinstance(response, dict) else None) or _clean(detail.get("ContentType")) or mimetypes.guess_type(key)[0]
etag = _clean(response.get("ETag") if isinstance(response, dict) else None) or _clean(detail.get("ETag"))
filename = filename_from_path(key)
return ConnectorDownloadedFile(
filename=filename,
data=data,
content_type=content_type,
revision=version_id or etag or _clean(detail.get("LastModified")),
external_id=f"{bucket}:{key}",
external_url=f"s3://{bucket}/{key}",
metadata=_s3_download_metadata(
detail,
bucket=bucket,
key=key,
version_id=version_id,
etag=etag,
size=len(data),
),
)
def _int(value: object) -> int | None:
if value is None or value == "":
return None
@@ -246,11 +246,11 @@ def _applied_fields(policy: Mapping[str, Any]) -> tuple[str, ...]:
def _matches_field(request: ConnectorAccessRequest, field: str, patterns: list[str]) -> bool:
if field == "connectors":
return _matches_exact(request.connector_id, patterns)
return _matches_reference(request.connector_id, patterns)
if field == "credentials":
return _matches_exact(request.credential_id, patterns)
return _matches_reference(request.credential_id, patterns)
if field == "providers":
return _matches_exact(request.provider, patterns)
return _matches_reference(request.provider, patterns)
if field == "external_ids":
return _matches_glob(request.external_id, patterns)
if field == "external_paths":
@@ -260,11 +260,11 @@ def _matches_field(request: ConnectorAccessRequest, field: str, patterns: list[s
return False
def _matches_exact(value: str | None, patterns: list[str]) -> bool:
def _matches_reference(value: str | None, patterns: list[str]) -> bool:
if value is None:
return False
clean = value.casefold()
return any(pattern == "*" or clean == pattern.casefold() for pattern in patterns)
return any(fnmatchcase(clean, pattern.casefold()) for pattern in patterns)
def _matches_glob(value: str | None, patterns: list[str]) -> bool:
@@ -1,6 +1,6 @@
from __future__ import annotations
from collections.abc import Mapping
from collections.abc import Callable, Mapping
from typing import Any
from sqlalchemy.orm import Session
@@ -9,7 +9,14 @@ from govoplan_core.core.policy import normalize_policy_scope_type
from govoplan_core.security.secrets import decrypt_secret, encrypt_secret
from govoplan_files.backend.db.models import FileConnectorCredential, FileConnectorProfile
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.connector_credential_store import connector_credential_from_row, credential_rows_by_id
from govoplan_files.backend.storage.connector_deployment import reject_api_controlled_deployment_references
from govoplan_files.backend.storage.connector_credential_store import (
ConnectorCredential,
connector_credential_from_row,
credential_rows_by_id,
resolve_reusable_connector_credential,
reusable_credential_id,
)
from govoplan_files.backend.storage.connector_policy import connector_policy_sources_from_payload
from govoplan_files.backend.storage.connector_profiles import ConnectorProfile, supported_connector_providers
@@ -19,7 +26,26 @@ def list_database_connector_profiles(
*,
tenant_id: str,
include_disabled: bool = False,
row_visible: Callable[[FileConnectorProfile], bool] | None = None,
) -> list[ConnectorProfile]:
profiles, _profile_ids = select_database_connector_profiles(
session,
tenant_id=tenant_id,
include_disabled=include_disabled,
row_visible=row_visible,
)
return profiles
def select_database_connector_profiles(
session: Session,
*,
tenant_id: str,
include_disabled: bool = False,
row_visible: Callable[[FileConnectorProfile], bool] | None = None,
) -> tuple[list[ConnectorProfile], set[str]]:
"""Return visible profiles and every database id that shadows settings."""
query = session.query(FileConnectorProfile).filter(
(FileConnectorProfile.scope_type == "system")
| (FileConnectorProfile.tenant_id == tenant_id)
@@ -27,9 +53,33 @@ def list_database_connector_profiles(
if not include_disabled:
query = query.filter(FileConnectorProfile.enabled.is_(True))
rows = query.order_by(FileConnectorProfile.scope_type.asc(), FileConnectorProfile.label.asc()).all()
credential_ids = {_clean(row.credential_profile_id) for row in rows if _clean(row.credential_profile_id)}
profile_ids = {row.id for row in rows}
if row_visible is not None:
rows = [row for row in rows if row_visible(row)]
credential_ids = {
_clean(row.credential_profile_id)
for row in rows
if _clean(row.credential_profile_id) and not reusable_credential_id(row.credential_profile_id)
}
credentials = credential_rows_by_id(session, tenant_id=tenant_id, credential_ids={item for item in credential_ids if item}, include_disabled=include_disabled)
return [connector_profile_from_row(row, credential_row=credentials.get(row.credential_profile_id or "")) for row in rows]
return (
[
connector_profile_from_row(
row,
credential_row=(
_resolve_profile_reusable_credential(
session,
tenant_id=tenant_id,
row=row,
)
if reusable_credential_id(row.credential_profile_id)
else credentials.get(row.credential_profile_id or "")
),
)
for row in rows
],
profile_ids,
)
def list_connector_profile_rows(
@@ -47,7 +97,10 @@ def list_connector_profile_rows(
return query.order_by(FileConnectorProfile.scope_type.asc(), FileConnectorProfile.label.asc()).all()
def connector_profile_from_row(row: FileConnectorProfile, credential_row: FileConnectorCredential | None = None) -> ConnectorProfile:
def connector_profile_from_row(
row: FileConnectorProfile,
credential_row: FileConnectorCredential | ConnectorCredential | None = None,
) -> ConnectorProfile:
policy_sources = []
if row.policy:
policy_sources = connector_policy_sources_from_payload({
@@ -56,9 +109,18 @@ def connector_profile_from_row(row: FileConnectorProfile, credential_row: FileCo
"label": row.label,
"policy": row.policy,
})
credential = connector_credential_from_row(credential_row) if credential_row is not None else None
credential = (
credential_row
if isinstance(credential_row, ConnectorCredential)
else connector_credential_from_row(credential_row)
if credential_row is not None
else None
)
if credential:
policy_sources.extend(credential.policy_sources)
metadata = dict(row.metadata_ or {})
if reusable_credential_id(row.credential_profile_id) and credential is None:
metadata["credential_unavailable"] = True
return ConnectorProfile(
id=row.id,
label=row.label,
@@ -79,11 +141,30 @@ def connector_profile_from_row(row: FileConnectorProfile, credential_row: FileCo
token_value=credential.token_value if credential else decrypt_secret(row.token_encrypted),
capabilities=tuple(_string_list(row.capabilities)),
policy_sources=tuple(policy_sources),
metadata=dict(row.metadata_ or {}),
metadata=metadata,
source_kind="database",
)
def _resolve_profile_reusable_credential(
session: Session,
*,
tenant_id: str,
row: FileConnectorProfile,
) -> ConnectorCredential | None:
try:
return resolve_reusable_connector_credential(
session,
tenant_id=tenant_id,
credential_ref=row.credential_profile_id or "",
profile_id=row.id,
scope_type=row.scope_type,
scope_id=row.scope_id,
)
except FileStorageError:
return None
def get_connector_profile_row(
session: Session,
*,
@@ -124,6 +205,12 @@ def create_connector_profile_row(
policy: Mapping[str, Any] | None = None,
metadata: Mapping[str, Any] | None = None,
) -> FileConnectorProfile:
reject_api_controlled_deployment_references(
password_env=password_env,
token_env=token_env,
secret_ref=secret_ref,
metadata=metadata,
)
clean_id = _normalize_profile_id(profile_id)
if session.get(FileConnectorProfile, clean_id) is not None:
raise FileStorageError(f"Connector profile already exists: {clean_id}")
@@ -181,6 +268,80 @@ def update_connector_profile_row(
clear_password: bool = False,
clear_token: bool = False,
) -> FileConnectorProfile:
_validate_profile_update_references(
row,
password_env=password_env,
token_env=token_env,
secret_ref=secret_ref,
metadata=metadata,
)
_update_profile_connection_fields(
row,
label=label,
provider=provider,
endpoint_url=endpoint_url,
base_path=base_path,
enabled=enabled,
credential_profile_id=credential_profile_id,
credential_mode=credential_mode,
)
_update_profile_credential_fields(
row,
username=username,
password=password,
token=token,
password_env=password_env,
token_env=token_env,
secret_ref=secret_ref,
clear_password=clear_password,
clear_token=clear_token,
)
_update_profile_governance_fields(
row,
capabilities=capabilities,
policy=policy,
metadata=metadata,
)
row.updated_by_user_id = user_id
session.add(row)
session.flush()
return row
def _validate_profile_update_references(
row: FileConnectorProfile,
*,
password_env: str | None,
token_env: str | None,
secret_ref: str | None,
metadata: Mapping[str, Any] | None,
) -> None:
reject_api_controlled_deployment_references(
password_env=password_env,
token_env=token_env,
secret_ref=secret_ref,
metadata=metadata,
)
if secret_ref is None or _clean(secret_ref) == _clean(row.secret_ref):
return
if _clean(row.secret_ref):
raise FileStorageError(
"An existing external secret reference cannot be replaced or cleared until Files can prove "
"provider ownership and confirm provider-side deletion"
)
def _update_profile_connection_fields(
row: FileConnectorProfile,
*,
label: str | None,
provider: str | None,
endpoint_url: str | None,
base_path: str | None,
enabled: bool | None,
credential_profile_id: str | None,
credential_mode: str | None,
) -> None:
if label is not None:
row.label = _normalize_label(label)
if provider is not None:
@@ -195,6 +356,20 @@ def update_connector_profile_row(
row.credential_profile_id = _clean(credential_profile_id)
if credential_mode is not None:
row.credential_mode = _normalize_credential_mode(credential_mode)
def _update_profile_credential_fields(
row: FileConnectorProfile,
*,
username: str | None,
password: str | None,
token: str | None,
password_env: str | None,
token_env: str | None,
secret_ref: str | None,
clear_password: bool,
clear_token: bool,
) -> None:
if username is not None:
row.username = _clean(username)
if password is not None:
@@ -211,24 +386,21 @@ def update_connector_profile_row(
row.token_env = _clean(token_env)
if secret_ref is not None:
row.secret_ref = _clean(secret_ref)
def _update_profile_governance_fields(
row: FileConnectorProfile,
*,
capabilities: list[str] | None,
policy: Mapping[str, Any] | None,
metadata: Mapping[str, Any] | None,
) -> None:
if capabilities is not None:
row.capabilities = _string_list(capabilities)
if policy is not None:
row.policy = dict(policy)
if metadata is not None:
row.metadata_ = dict(metadata)
row.updated_by_user_id = user_id
session.add(row)
session.flush()
return row
def deactivate_connector_profile_row(session: Session, row: FileConnectorProfile, *, user_id: str | None) -> FileConnectorProfile:
row.enabled = False
row.updated_by_user_id = user_id
session.add(row)
session.flush()
return row
def _normalize_scope(*, tenant_id: str, scope_type: str, scope_id: str | None) -> tuple[str, str | None, str | None]:
@@ -7,13 +7,26 @@ from dataclasses import dataclass, field
from typing import Any
from govoplan_core.core.policy import normalize_policy_scope_type, policy_source_path
from govoplan_files.backend.storage.connector_deployment import (
connector_secret_env_available,
validate_deployment_connector_references,
)
from govoplan_files.backend.storage.connector_policy import ConnectorPolicySource, connector_policy_sources_from_payload
_ENV_JSON_KEYS = ("GOVOPLAN_FILES_CONNECTOR_PROFILES_JSON", "FILES_CONNECTOR_PROFILES_JSON")
_ENV_FILE_KEYS = ("GOVOPLAN_FILES_CONNECTOR_PROFILES_FILE", "FILES_CONNECTOR_PROFILES_FILE")
_SUPPORTED_PROVIDERS = {"seafile", "nextcloud", "webdav", "smb", "nfs", "dms", "generic"}
_INLINE_SECRET_FIELDS = ("password", "token", "api_key", "access_key", "secret_key")
_SUPPORTED_PROVIDERS = {"seafile", "nextcloud", "webdav", "smb", "s3", "sharepoint", "onedrive", "nfs", "dms", "generic"}
_INLINE_SECRET_FIELDS = (
"password",
"token",
"api_key",
"access_key",
"access_key_id",
"secret_key",
"secret_access_key",
"session_token",
)
def supported_connector_providers() -> set[str]:
@@ -67,15 +80,17 @@ class ConnectorProfile:
@property
def credentials_configured(self) -> bool:
if not self.enabled:
return False
mode = self.credential_mode.casefold()
if mode in {"", "none", "anonymous"}:
return True
if self.secret_ref or self.has_inline_secret or self.password_value or self.token_value:
return True
if self.token_env:
return bool(os.environ.get(self.token_env))
return connector_secret_env_available(self.token_env, source_kind=self.source_kind)
if self.password_env:
return bool(os.environ.get(self.password_env))
return connector_secret_env_available(self.password_env, source_kind=self.source_kind)
return False
def to_response(self) -> dict[str, Any]:
@@ -97,7 +112,7 @@ class ConnectorProfile:
"username": self.username,
"capabilities": list(self.capabilities),
"policy_sources": [_policy_source_response(source) for source in self.policy_sources],
"metadata": dict(self.metadata),
"metadata": _response_metadata(self.metadata),
"source_kind": self.source_kind,
}
@@ -123,45 +138,21 @@ def connector_profiles_from_payload(value: object) -> list[ConnectorProfile]:
def _profile_from_mapping(value: Mapping[str, Any]) -> ConnectorProfile:
profile_id = _clean(value.get("id") or value.get("connector_id") or value.get("name"))
if not profile_id:
raise ValueError("Connector profiles require id")
provider = (_clean(value.get("provider") or value.get("type")) or "generic").casefold()
if provider not in _SUPPORTED_PROVIDERS:
raise ValueError(f"Unsupported connector provider: {provider}")
scope_type = normalize_policy_scope_type(str(value.get("scope_type") or "system"))
scope_id = _clean(value.get("scope_id"))
if scope_type == "system":
scope_id = None
elif not scope_id:
raise ValueError(f"{scope_type} connector profiles require scope_id")
credentials = value.get("credentials")
credentials = credentials if isinstance(credentials, Mapping) else {}
profile_id = _profile_id_from_mapping(value)
provider = _provider_from_mapping(value)
scope_type, scope_id = _scope_from_mapping(value)
credentials = _credentials_from_mapping(value)
mode = _clean(value.get("credential_mode") or value.get("auth_type") or credentials.get("mode") or credentials.get("type")) or "none"
profile = ConnectorProfile(
id=profile_id,
label=_clean(value.get("label") or value.get("name")) or profile_id,
profile = _profile_from_normalized_mapping(
value,
profile_id=profile_id,
provider=provider,
endpoint_url=_clean(value.get("endpoint_url") or value.get("base_url") or value.get("url")),
base_path=_clean(value.get("base_path") or value.get("path") or value.get("root")),
enabled=_bool(value.get("enabled"), default=True),
scope_type=scope_type,
scope_id=scope_id,
credential_profile_id=_clean(value.get("credential_profile_id") or value.get("credential_id")),
credential_profile_label=_clean(value.get("credential_profile_label") or value.get("credential_label")),
credential_mode=mode,
username=_clean(value.get("username") or credentials.get("username")),
password_env=_clean(value.get("password_env") or credentials.get("password_env")),
token_env=_clean(value.get("token_env") or credentials.get("token_env")),
secret_ref=_clean(value.get("secret_ref") or credentials.get("secret_ref")),
password_value=_clean(value.get("password") or credentials.get("password")),
token_value=_clean(value.get("token") or credentials.get("token")),
has_inline_secret=any(_clean(value.get(field) or credentials.get(field)) for field in _INLINE_SECRET_FIELDS),
capabilities=tuple(_string_list(value.get("capabilities"))),
metadata=_public_metadata(value.get("metadata")),
credentials=credentials,
)
return ConnectorProfile(
result = ConnectorProfile(
id=profile.id,
label=profile.label,
provider=profile.provider,
@@ -185,6 +176,76 @@ def _profile_from_mapping(value: Mapping[str, Any]) -> ConnectorProfile:
metadata=profile.metadata,
source_kind="settings",
)
validate_deployment_connector_references(
source_kind=result.source_kind,
password_env=result.password_env,
token_env=result.token_env,
metadata=result.metadata,
)
return result
def _profile_id_from_mapping(value: Mapping[str, Any]) -> str:
profile_id = _clean(value.get("id") or value.get("connector_id") or value.get("name"))
if not profile_id:
raise ValueError("Connector profiles require id")
return profile_id
def _provider_from_mapping(value: Mapping[str, Any]) -> str:
provider = (_clean(value.get("provider") or value.get("type")) or "generic").casefold()
if provider not in _SUPPORTED_PROVIDERS:
raise ValueError(f"Unsupported connector provider: {provider}")
return provider
def _scope_from_mapping(value: Mapping[str, Any]) -> tuple[str, str | None]:
scope_type = normalize_policy_scope_type(str(value.get("scope_type") or "system"))
scope_id = _clean(value.get("scope_id"))
if scope_type == "system":
return scope_type, None
if not scope_id:
raise ValueError(f"{scope_type} connector profiles require scope_id")
return scope_type, scope_id
def _credentials_from_mapping(value: Mapping[str, Any]) -> Mapping[str, Any]:
credentials = value.get("credentials")
return credentials if isinstance(credentials, Mapping) else {}
def _profile_from_normalized_mapping(
value: Mapping[str, Any],
*,
profile_id: str,
provider: str,
scope_type: str,
scope_id: str | None,
credential_mode: str,
credentials: Mapping[str, Any],
) -> ConnectorProfile:
return ConnectorProfile(
id=profile_id,
label=_clean(value.get("label") or value.get("name")) or profile_id,
provider=provider,
endpoint_url=_clean(value.get("endpoint_url") or value.get("base_url") or value.get("url")),
base_path=_clean(value.get("base_path") or value.get("path") or value.get("root")),
enabled=_bool(value.get("enabled"), default=True),
scope_type=scope_type,
scope_id=scope_id,
credential_profile_id=_clean(value.get("credential_profile_id") or value.get("credential_id")),
credential_profile_label=_clean(value.get("credential_profile_label") or value.get("credential_label")),
credential_mode=credential_mode,
username=_clean(value.get("username") or credentials.get("username")),
password_env=_clean(value.get("password_env") or credentials.get("password_env")),
token_env=_clean(value.get("token_env") or credentials.get("token_env")),
secret_ref=_clean(value.get("secret_ref") or credentials.get("secret_ref")),
password_value=_clean(value.get("password") or credentials.get("password")),
token_value=_clean(value.get("token") or credentials.get("token")),
has_inline_secret=any(_clean(value.get(field) or credentials.get(field)) for field in _INLINE_SECRET_FIELDS),
capabilities=tuple(_string_list(value.get("capabilities"))),
metadata=_public_metadata(value.get("metadata")),
)
def _raw_profiles_payload(settings: object | None) -> object:
@@ -243,6 +304,16 @@ def _public_metadata(value: object) -> Mapping[str, Any]:
}
def _response_metadata(value: Mapping[str, Any]) -> dict[str, Any]:
return {
str(key): item
for key, item in value.items()
if str(key).strip().casefold() not in _INLINE_SECRET_FIELDS
and not str(key).strip().casefold().endswith("_env")
and str(key).strip().casefold() != "ca_bundle"
}
def _string_list(value: object) -> list[str]:
if value is None:
return []
@@ -110,7 +110,52 @@ def connector_provider_descriptors() -> tuple[ConnectorProviderDescriptor, ...]:
conflict_strategy="Managed import/sync uses existing files conflict_strategy handling after download.",
preview_strategy="Previews are generated from the frozen managed file after import or sync, not directly from the share.",
audit_events=("files.connector.imported", "files.connector.synced", "files.connector.accessed"),
notes="Requires the optional smbprotocol dependency and an smb://server[:port]/share[/path] profile endpoint.",
notes="Initial sessions, reconnects, aliases, and DFS referral targets use the Files-owned pinned smbprotocol transport and the deployment-wide private-network policy.",
),
ConnectorProviderDescriptor(
provider="s3",
label="S3-compatible object storage",
protocol="s3-api",
implemented=True,
browse_supported=True,
import_supported=True,
optional_dependency="boto3",
permission_model=governed_permissions,
sync_strategy="On-demand bucket/prefix browse and object import/sync; sync updates managed versions and never mutates the remote bucket.",
conflict_strategy="Managed import/sync uses existing files conflict_strategy handling after object download.",
preview_strategy="Previews are generated from the frozen managed file after import or sync, not directly from the bucket.",
audit_events=("files.connector.imported", "files.connector.synced", "files.connector.accessed"),
notes="Every botocore connection pool uses the Files-owned pinned transport, including retries, redirects, endpoint discovery, and virtual-host aliases; outbound proxies and ambient credential discovery are disabled.",
),
ConnectorProviderDescriptor(
provider="sharepoint",
label="SharePoint",
protocol="microsoft-graph/sharepoint",
implemented=False,
browse_supported=False,
import_supported=False,
optional_dependency=None,
permission_model=governed_permissions,
sync_strategy="Planned read-only browse/import through deployment-managed Microsoft Graph or SharePoint credentials.",
conflict_strategy=freeze_model,
preview_strategy="Will preview from frozen managed file after import, not directly from SharePoint.",
audit_events=("files.connector.imported", "files.connector.synced", "files.connector.accessed"),
notes="Provider/profile key is reserved; live browsing/import still needs Graph/SharePoint authentication and paging implementation.",
),
ConnectorProviderDescriptor(
provider="onedrive",
label="OneDrive",
protocol="microsoft-graph/onedrive",
implemented=False,
browse_supported=False,
import_supported=False,
optional_dependency=None,
permission_model=governed_permissions,
sync_strategy="Planned read-only browse/import through deployment-managed Microsoft Graph credentials.",
conflict_strategy=freeze_model,
preview_strategy="Will preview from frozen managed file after import, not directly from OneDrive.",
audit_events=("files.connector.imported", "files.connector.synced", "files.connector.accessed"),
notes="Provider/profile key is reserved; live browsing/import still needs Graph drive selection, OAuth/app registration, and paging implementation.",
),
ConnectorProviderDescriptor(
provider="nfs",
@@ -0,0 +1,213 @@
from __future__ import annotations
from collections.abc import Callable, Iterable
from dataclasses import replace
from sqlalchemy.orm import Session
from govoplan_files.backend.storage.connector_deployment import (
connector_effective_endpoint_url,
)
from govoplan_files.backend.storage.connector_profile_store import (
select_database_connector_profiles,
)
from govoplan_files.backend.storage.connector_profiles import (
ConnectorProfile,
connector_profiles_from_settings,
)
from govoplan_files.backend.storage.connector_policy import (
ConnectorAccessRequest,
connector_policy_decision,
)
from govoplan_files.backend.storage.connector_policy_store import (
effective_connector_policy_sources,
)
from govoplan_files.backend.storage.connector_providers import (
connector_provider_descriptors,
)
CampaignVisibility = Callable[[str], bool]
def visible_connector_profiles_for_actor(
session: Session,
*,
tenant_id: str,
user_id: str,
group_ids: Iterable[str],
settings: object | None,
provider: str | None = None,
campaign_id: str | None = None,
campaign_visible: CampaignVisibility | None = None,
include_disabled: bool = False,
include_admin_scopes: bool = False,
include_effective_policy: bool = True,
) -> list[ConnectorProfile]:
"""Return the same actor-filtered connector set used by API and docs surfaces."""
provider_norm = provider.strip().casefold() if provider else None
actor_group_ids = {str(group_id) for group_id in group_ids if str(group_id)}
database_profiles, database_profile_ids = select_database_connector_profiles(
session,
tenant_id=tenant_id,
include_disabled=include_disabled,
row_visible=lambda row: (
include_admin_scopes
and row.scope_type in {"user", "group", "campaign"}
) or _scope_visible_to_actor(
scope_type=row.scope_type,
scope_id=row.scope_id,
tenant_id=tenant_id,
user_id=user_id,
group_ids=actor_group_ids,
campaign_id=campaign_id,
campaign_visible=campaign_visible,
),
)
configured_profiles = connector_profiles_from_settings(settings)
profiles_by_id: dict[str, ConnectorProfile] = {}
for profile in database_profiles:
if profile.id not in profiles_by_id:
profiles_by_id[profile.id] = profile
for profile in configured_profiles:
if profile.id not in database_profile_ids and profile.id not in profiles_by_id:
profiles_by_id[profile.id] = profile
visible: list[ConnectorProfile] = []
for profile in profiles_by_id.values():
if not (profile.enabled or include_disabled):
continue
if provider_norm is not None and profile.provider != provider_norm:
continue
admin_scope_visible = include_admin_scopes and profile.scope_type in {
"user",
"group",
"campaign",
}
if not admin_scope_visible and not _profile_visible_to_actor(
profile,
tenant_id=tenant_id,
user_id=user_id,
group_ids=actor_group_ids,
campaign_id=campaign_id,
campaign_visible=campaign_visible,
):
continue
visible.append(
_with_effective_connector_policy(
session, tenant_id=tenant_id, profile=profile
)
if include_effective_policy
else profile
)
return visible
def connector_profile_usable_for_import(profile: ConnectorProfile) -> bool:
"""Whether a visible profile can safely offer the current live browse/import task."""
effective_endpoint_url = connector_effective_endpoint_url(
provider=profile.provider,
endpoint_url=profile.endpoint_url,
metadata=profile.metadata,
)
if (
not profile.enabled
or not effective_endpoint_url
or not profile.credentials_configured
or bool(profile.secret_ref)
):
return False
descriptor = next(
(
item
for item in connector_provider_descriptors()
if item.provider == profile.provider
),
None,
)
if descriptor is None:
return False
if not (
descriptor.implemented
and descriptor.installed
and descriptor.browse_supported
and descriptor.import_supported
):
return False
# Prove that the initial root browse performed by the current Files UI is
# policy-allowed. A later selected remote path/item is checked again.
return connector_policy_decision(
ConnectorAccessRequest(
connector_id=profile.id,
credential_id=profile.credential_profile_id,
provider=profile.provider,
external_path="",
external_url=effective_endpoint_url,
operation="import",
),
profile.policy_sources,
).allowed
def _profile_visible_to_actor(
profile: ConnectorProfile,
*,
tenant_id: str,
user_id: str,
group_ids: set[str],
campaign_id: str | None,
campaign_visible: CampaignVisibility | None,
) -> bool:
return _scope_visible_to_actor(
scope_type=profile.scope_type,
scope_id=profile.scope_id,
tenant_id=tenant_id,
user_id=user_id,
group_ids=group_ids,
campaign_id=campaign_id,
campaign_visible=campaign_visible,
)
def _scope_visible_to_actor(
*,
scope_type: str,
scope_id: str | None,
tenant_id: str,
user_id: str,
group_ids: set[str],
campaign_id: str | None,
campaign_visible: CampaignVisibility | None,
) -> bool:
if scope_type == "system":
return True
if scope_type == "tenant":
return scope_id == tenant_id
if scope_type == "user":
return scope_id == user_id
if scope_type == "group":
return bool(scope_id and scope_id in group_ids)
if scope_type != "campaign" or not scope_id:
return False
if campaign_id and scope_id != campaign_id:
return False
return bool(campaign_visible and campaign_visible(scope_id))
def _with_effective_connector_policy(
session: Session,
*,
tenant_id: str,
profile: ConnectorProfile,
) -> ConnectorProfile:
sources = effective_connector_policy_sources(
session,
tenant_id=tenant_id,
scope_type=profile.scope_type,
scope_id=profile.scope_id,
)
if not sources:
return profile
return replace(profile, policy_sources=tuple([*sources, *profile.policy_sources]))
@@ -0,0 +1,93 @@
from __future__ import annotations
from govoplan_core.core.encryption import (
ContentProtectionRequest,
ContentUnprotectionRequest,
ProtectedContent,
encryption_content_cipher,
)
from govoplan_files.backend.runtime import get_registry
from govoplan_files.backend.storage.common import FileStorageError
FILES_PROTECTION_PROFILE = "files-server-envelope-v1"
def protect_blob_content(
session: object,
*,
tenant_id: str,
blob_id: str,
vault_id: str,
ciphertext_ref: str,
plaintext: bytes,
actor_id: str,
content_type: str | None,
) -> ProtectedContent:
capability = encryption_content_cipher(get_registry())
if capability is None:
raise FileStorageError(
"File encryption was requested, but the Encryption module is unavailable."
)
try:
return capability.protect_content(
session,
request=ContentProtectionRequest(
tenant_id=tenant_id,
owner_module="files",
resource_type="file_blob",
resource_id=blob_id,
profile_id=FILES_PROTECTION_PROFILE,
vault_id=vault_id,
ciphertext_ref=ciphertext_ref,
plaintext=plaintext,
policy_decision_ref="files:explicit-vault-selection:v1",
idempotency_key=f"file-blob:{blob_id}:content:v1",
actor_id=actor_id,
metadata={
"content_type": content_type or "application/octet-stream",
},
),
)
except Exception as exc:
raise FileStorageError(
"Managed file content could not be protected by the configured vault."
) from exc
def unprotect_blob_content(
session: object,
*,
tenant_id: str,
blob_id: str,
envelope_id: str,
ciphertext: bytes,
) -> bytes:
capability = encryption_content_cipher(get_registry())
if capability is None:
raise FileStorageError(
"This file is encrypted and cannot be read while Encryption is unavailable."
)
try:
return capability.unprotect_content(
session,
request=ContentUnprotectionRequest(
tenant_id=tenant_id,
owner_module="files",
resource_type="file_blob",
resource_id=blob_id,
envelope_id=envelope_id,
ciphertext=ciphertext,
),
)
except Exception as exc:
raise FileStorageError(
"Managed file content could not be opened with its protection envelope."
) from exc
__all__ = [
"FILES_PROTECTION_PROFILE",
"protect_blob_content",
"unprotect_blob_content",
]
+337 -24
View File
@@ -2,21 +2,32 @@ from __future__ import annotations
import hashlib
import mimetypes
from datetime import datetime
from pathlib import PurePosixPath
from typing import Any, Iterable
from uuid import uuid4
from sqlalchemy import and_, or_
from sqlalchemy import and_, exists, func, or_
from sqlalchemy.orm import Session
from govoplan_core.core.campaigns import CAPABILITY_CAMPAIGNS_ACCESS, CampaignAccessProvider
from govoplan_files.backend.db.models import CampaignAttachmentUse, FileAsset, FileBlob, FileShare, FileVersion
from govoplan_files.backend.runtime import get_registry, settings
from govoplan_files.backend.storage.access import ensure_owner_access, ensure_share_target_exists, user_group_ids
from govoplan_files.backend.storage.backends import StorageBackendError, get_storage_backend
from govoplan_files.backend.storage.backends import (
StorageBackendError,
StorageObjectMissing,
get_storage_backend,
)
from govoplan_files.backend.storage.common import FileConflictResolution, FileStorageError, UploadedStoredFile, utcnow
from govoplan_files.backend.storage.paths import filename_from_path, join_folder_filename, normalize_folder, normalize_logical_path, safe_storage_component
from govoplan_files.backend.storage.paths import filename_from_path, join_folder_filename, normalize_folder, normalize_logical_path
from govoplan_files.backend.storage.provenance import source_provenance_from_metadata
from govoplan_files.backend.storage.recovery import begin_blob_write_recovery
from govoplan_files.backend.storage.integrity import (
QUARANTINED_BLOB_STATUSES,
read_verified_blob_bytes,
)
from govoplan_files.backend.storage.share_state import effective_file_share_clause
def _campaign_access_provider() -> CampaignAccessProvider:
@@ -51,8 +62,9 @@ def _storage_backend_name() -> str:
return settings.file_storage_backend.lower().strip()
def _storage_key(*, tenant_id: str, checksum: str, filename: str) -> str:
return f"tenants/{tenant_id}/files/{checksum[:2]}/{uuid4().hex}-{safe_storage_component(filename)}"
def _storage_key(*, tenant_id: str, checksum: str) -> str:
# Object locators remain opaque so recovery evidence never persists names.
return f"tenants/{tenant_id}/files/{checksum[:2]}/{uuid4().hex}.blob"
def _get_or_create_blob(
@@ -62,40 +74,143 @@ def _get_or_create_blob(
data: bytes,
filename: str,
content_type: str | None,
actor_id: str,
encryption_vault_id: str | None = None,
) -> FileBlob:
checksum = hashlib.sha256(data).hexdigest()
size = len(data)
vault_id = str(encryption_vault_id or "").strip() or None
protection_discriminator = f"vault:{vault_id}" if vault_id else "plaintext"
blob = (
session.query(FileBlob)
.filter(FileBlob.tenant_id == tenant_id, FileBlob.checksum_sha256 == checksum, FileBlob.size_bytes == size)
.filter(FileBlob.tenant_id == tenant_id, FileBlob.checksum_sha256 == checksum, FileBlob.size_bytes == size, FileBlob.protection_discriminator == protection_discriminator)
.one_or_none()
)
if blob:
backend = get_storage_backend()
if not backend.exists(blob.storage_key):
repair_required = (
blob.integrity_status in QUARANTINED_BLOB_STATUSES
or blob.quarantined_at is not None
)
try:
backend.stat(blob.storage_key)
except StorageObjectMissing:
repair_required = True
except StorageBackendError as exc:
raise FileStorageError(str(exc)) from exc
if repair_required:
repair_token = hashlib.sha256(
repr(
(
blob.integrity_status,
blob.integrity_checked_at,
blob.quarantined_at,
blob.storage_checksum_sha256,
)
).encode("utf-8")
).hexdigest()
recovery = begin_blob_write_recovery(
session,
backend=backend,
tenant_id=tenant_id,
blob_id=blob.id,
storage_key=blob.storage_key,
semantic_checksum_sha256=checksum,
semantic_size_bytes=size,
protection_discriminator=protection_discriminator,
created_new=False,
repair_token=repair_token,
)
stored_data = data
expected_envelope_id = blob.encryption_envelope_id
if vault_id:
from govoplan_files.backend.storage.content_protection import protect_blob_content
protected = protect_blob_content(
session,
tenant_id=tenant_id,
blob_id=blob.id,
vault_id=vault_id,
ciphertext_ref=blob.storage_key,
plaintext=data,
actor_id=actor_id,
content_type=content_type,
)
if blob.encryption_envelope_id not in {None, protected.envelope.envelope_id}:
raise FileStorageError("The existing encrypted blob has another protection envelope.")
stored_data = protected.ciphertext
blob.encryption_envelope_id = protected.envelope.envelope_id
expected_envelope_id = protected.envelope.envelope_id
blob.storage_checksum_sha256 = hashlib.sha256(stored_data).hexdigest()
blob.storage_size_bytes = len(stored_data)
recovery.prepare_stored_bytes(
stored_data,
envelope_id=expected_envelope_id,
)
try:
backend.put_bytes(blob.storage_key, data, content_type=content_type)
backend.put_bytes(blob.storage_key, stored_data, content_type="application/octet-stream" if vault_id else content_type)
except StorageBackendError as exc:
raise FileStorageError(str(exc)) from exc
blob.integrity_status = "verified"
blob.integrity_checked_at = utcnow()
blob.integrity_failure = None
blob.quarantined_at = None
blob.ref_count += 1
session.add(blob)
return blob
storage_key = _storage_key(tenant_id=tenant_id, checksum=checksum, filename=filename)
blob_id = str(uuid4())
storage_key = _storage_key(tenant_id=tenant_id, checksum=checksum)
backend = get_storage_backend()
recovery = begin_blob_write_recovery(
session,
backend=backend,
tenant_id=tenant_id,
blob_id=blob_id,
storage_key=storage_key,
semantic_checksum_sha256=checksum,
semantic_size_bytes=size,
protection_discriminator=protection_discriminator,
created_new=True,
)
stored_data = data
envelope_id = None
if vault_id:
from govoplan_files.backend.storage.content_protection import protect_blob_content
protected = protect_blob_content(
session,
tenant_id=tenant_id,
blob_id=blob_id,
vault_id=vault_id,
ciphertext_ref=storage_key,
plaintext=data,
actor_id=actor_id,
content_type=content_type,
)
stored_data = protected.ciphertext
envelope_id = protected.envelope.envelope_id
recovery.prepare_stored_bytes(stored_data, envelope_id=envelope_id)
try:
backend.put_bytes(storage_key, data, content_type=content_type)
backend.put_bytes(storage_key, stored_data, content_type="application/octet-stream" if vault_id else content_type)
except StorageBackendError as exc:
raise FileStorageError(str(exc)) from exc
blob = FileBlob(
id=blob_id,
tenant_id=tenant_id,
storage_backend=_storage_backend_name(),
storage_bucket=_storage_bucket_name(),
storage_key=storage_key,
checksum_sha256=checksum,
size_bytes=size,
protection_discriminator=protection_discriminator,
encryption_envelope_id=envelope_id,
storage_checksum_sha256=hashlib.sha256(stored_data).hexdigest() if vault_id else None,
storage_size_bytes=len(stored_data) if vault_id else None,
content_type=content_type,
ref_count=1,
integrity_status="verified",
integrity_checked_at=utcnow(),
)
session.add(blob)
session.flush()
@@ -120,6 +235,7 @@ def create_file_asset(
conflict_strategy: str = "reject",
conflict_resolutions: Iterable[FileConflictResolution] | None = None,
is_admin: bool = False,
encryption_vault_id: str | None = None,
) -> UploadedStoredFile:
owner_type = owner_type.lower().strip()
ensure_owner_access(session, tenant_id=tenant_id, owner_type=owner_type, owner_id=owner_id, user_id=user_id, is_admin=is_admin)
@@ -147,7 +263,7 @@ def create_file_asset(
elif action == "rename":
logical_path = _next_available_logical_path(session, tenant_id=tenant_id, owner_type=owner_type, owner_id=owner_id, desired_path=logical_path)
blob = _get_or_create_blob(session, tenant_id=tenant_id, data=data, filename=safe_filename, content_type=content_type)
blob = _get_or_create_blob(session, tenant_id=tenant_id, data=data, filename=safe_filename, content_type=content_type, actor_id=user_id, encryption_vault_id=encryption_vault_id)
asset = FileAsset(
tenant_id=tenant_id,
owner_type=owner_type,
@@ -279,6 +395,7 @@ def update_file_asset_content(
data: bytes,
content_type: str | None,
metadata: dict[str, Any],
encryption_vault_id: str | None = None,
) -> tuple[UploadedStoredFile, str]:
if asset.tenant_id != tenant_id or asset.deleted_at is not None:
raise FileStorageError("File not found")
@@ -289,10 +406,23 @@ def update_file_asset_content(
checksum = hashlib.sha256(data).hexdigest()
asset.metadata_ = metadata
session.add(asset)
if current_blob.checksum_sha256 == checksum and current_blob.size_bytes == len(data):
inherited_vault_id = encryption_vault_id
if inherited_vault_id is None and current_blob.encryption_envelope_id:
prefix = "vault:"
if current_blob.protection_discriminator.startswith(prefix):
inherited_vault_id = current_blob.protection_discriminator[len(prefix) :]
inherited_vault_id = str(inherited_vault_id or "").strip() or None
target_protection = (
f"vault:{inherited_vault_id}" if inherited_vault_id else "plaintext"
)
if (
current_blob.checksum_sha256 == checksum
and current_blob.size_bytes == len(data)
and current_blob.protection_discriminator == target_protection
):
return UploadedStoredFile(asset=asset, version=current_version, blob=current_blob), "unchanged"
blob = _get_or_create_blob(session, tenant_id=tenant_id, data=data, filename=safe_filename, content_type=content_type)
blob = _get_or_create_blob(session, tenant_id=tenant_id, data=data, filename=safe_filename, content_type=content_type, actor_id=user_id, encryption_vault_id=inherited_vault_id)
version = FileVersion(
tenant_id=tenant_id,
file_asset_id=asset.id,
@@ -328,7 +458,7 @@ def get_asset_for_user(session: Session, *, tenant_id: str, user_id: str, asset_
.filter(
FileShare.tenant_id == tenant_id,
FileShare.file_asset_id == asset.id,
FileShare.revoked_at.is_(None),
effective_file_share_clause(),
FileShare.permission.in_(permission_values),
or_(
(FileShare.target_type == "user") & (FileShare.target_id == user_id),
@@ -343,6 +473,32 @@ def get_asset_for_user(session: Session, *, tenant_id: str, user_id: str, asset_
return asset
def get_asset_for_share_management(
session: Session,
*,
tenant_id: str,
user_id: str,
asset_id: str,
is_admin: bool = False,
) -> FileAsset:
asset = session.get(FileAsset, asset_id)
if not asset or asset.tenant_id != tenant_id or asset.deleted_at is not None:
raise FileStorageError("File not found")
if is_admin:
return asset
owns_asset = (
asset.owner_type == "user"
and asset.owner_user_id == user_id
) or (
asset.owner_type == "group"
and asset.owner_group_id
in user_group_ids(session, tenant_id=tenant_id, user_id=user_id)
)
if not owns_asset:
raise FileStorageError("Only file owners and administrators can manage shares")
return asset
def list_assets_for_user(
session: Session,
*,
@@ -352,6 +508,8 @@ def list_assets_for_user(
owner_id: str | None = None,
campaign_id: str | None = None,
path_prefix: str | None = None,
campaign_usage: str | None = None,
audit_relevant: bool | None = None,
include_deleted: bool = False,
is_admin: bool = False,
) -> list[FileAsset]:
@@ -363,6 +521,8 @@ def list_assets_for_user(
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
campaign_usage=campaign_usage,
audit_relevant=audit_relevant,
include_deleted=include_deleted,
is_admin=is_admin,
)
@@ -378,6 +538,8 @@ def list_assets_for_user_window(
owner_id: str | None = None,
campaign_id: str | None = None,
path_prefix: str | None = None,
campaign_usage: str | None = None,
audit_relevant: bool | None = None,
include_deleted: bool = False,
is_admin: bool = False,
page_size: int,
@@ -393,6 +555,8 @@ def list_assets_for_user_window(
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
campaign_usage=campaign_usage,
audit_relevant=audit_relevant,
include_deleted=include_deleted,
is_admin=is_admin,
)
@@ -417,6 +581,8 @@ def _asset_visibility_query_for_user(
owner_id: str | None = None,
campaign_id: str | None = None,
path_prefix: str | None = None,
campaign_usage: str | None = None,
audit_relevant: bool | None = None,
include_deleted: bool = False,
is_admin: bool = False,
):
@@ -430,30 +596,97 @@ def _asset_visibility_query_for_user(
if owner_type == "group" and owner_id:
query = query.filter(FileAsset.owner_group_id == owner_id)
if campaign_id:
query = query.join(FileShare, FileShare.file_asset_id == FileAsset.id).filter(
campaign_share = exists().where(
FileShare.tenant_id == tenant_id,
FileShare.file_asset_id == FileAsset.id,
FileShare.target_type == "campaign",
FileShare.target_id == campaign_id,
FileShare.revoked_at.is_(None),
effective_file_share_clause(),
)
query = query.filter(campaign_share)
elif not is_admin and not owner_type:
group_ids = user_group_ids(session, tenant_id=tenant_id, user_id=user_id)
query = query.outerjoin(FileShare, FileShare.file_asset_id == FileAsset.id).filter(
active_share = exists().where(
FileShare.tenant_id == tenant_id,
FileShare.file_asset_id == FileAsset.id,
effective_file_share_clause(),
or_(
(FileShare.target_type == "user") & (FileShare.target_id == user_id),
(FileShare.target_type == "group") & (FileShare.target_id.in_(group_ids)),
(FileShare.target_type == "tenant") & (FileShare.target_id == tenant_id),
)
)
query = query.filter(
or_(
(FileAsset.owner_type == "user") & (FileAsset.owner_user_id == user_id),
(FileAsset.owner_type == "group") & (FileAsset.owner_group_id.in_(group_ids)),
(FileShare.revoked_at.is_(None)) & (FileShare.target_type == "user") & (FileShare.target_id == user_id),
(FileShare.revoked_at.is_(None)) & (FileShare.target_type == "group") & (FileShare.target_id.in_(group_ids)),
(FileShare.revoked_at.is_(None)) & (FileShare.target_type == "tenant") & (FileShare.target_id == tenant_id),
active_share,
)
)
if path_prefix:
prefix = normalize_folder(path_prefix)
if prefix:
query = query.filter(FileAsset.display_path.like(f"{prefix}/%"))
campaign_share_exists = exists().where(
FileShare.tenant_id == tenant_id,
FileShare.file_asset_id == FileAsset.id,
FileShare.target_type == "campaign",
effective_file_share_clause(),
)
campaign_use_exists = exists().where(
CampaignAttachmentUse.tenant_id == tenant_id,
CampaignAttachmentUse.file_asset_id == FileAsset.id,
)
if campaign_usage == "linked":
query = query.filter(or_(campaign_share_exists, campaign_use_exists))
elif campaign_usage == "unlinked":
query = query.filter(~or_(campaign_share_exists, campaign_use_exists))
if audit_relevant is not None:
sent_use_exists = exists().where(
CampaignAttachmentUse.tenant_id == tenant_id,
CampaignAttachmentUse.file_asset_id == FileAsset.id,
CampaignAttachmentUse.use_stage == "sent",
)
query = query.filter(sent_use_exists if audit_relevant else ~sent_use_exists)
return query
def count_assets_for_user(
session: Session,
*,
tenant_id: str,
user_id: str,
owner_type: str | None = None,
owner_id: str | None = None,
campaign_id: str | None = None,
path_prefix: str | None = None,
campaign_usage: str | None = None,
audit_relevant: bool | None = None,
include_deleted: bool = False,
is_admin: bool = False,
) -> int:
query = _asset_visibility_query_for_user(
session,
tenant_id=tenant_id,
user_id=user_id,
owner_type=owner_type,
owner_id=owner_id,
campaign_id=campaign_id,
path_prefix=path_prefix,
campaign_usage=campaign_usage,
audit_relevant=audit_relevant,
include_deleted=include_deleted,
is_admin=is_admin,
)
return int(
query.order_by(None)
.with_entities(func.count(func.distinct(FileAsset.id)))
.scalar()
or 0
)
def current_version_and_blob(session: Session, asset: FileAsset) -> tuple[FileVersion, FileBlob]:
if not asset.current_version_id:
raise FileStorageError("File has no current version")
@@ -495,10 +728,7 @@ def current_versions_and_blobs(session: Session, assets: Iterable[FileAsset]) ->
def read_asset_bytes(session: Session, asset: FileAsset) -> tuple[bytes, FileVersion, FileBlob]:
version, blob = current_version_and_blob(session, asset)
backend = get_storage_backend()
try:
return backend.get_bytes(blob.storage_key), version, blob
except StorageBackendError as exc:
raise FileStorageError(str(exc)) from exc
return read_verified_blob_bytes(blob, backend=backend), version, blob
def share_file(
@@ -510,6 +740,7 @@ def share_file(
target_id: str,
permission: str,
user_id: str,
expires_at: datetime | None = None,
) -> FileShare:
target_type = target_type.lower().strip()
permission = permission.lower().strip()
@@ -519,6 +750,7 @@ def share_file(
raise FileStorageError("Unsupported share target")
if permission not in {"read", "write", "manage"}:
raise FileStorageError("Unsupported file permission")
_validate_share_expiry(expires_at)
if target_type in {"user", "group", "tenant"}:
ensure_share_target_exists(tenant_id=tenant_id, target_type=target_type, target_id=target_id)
if target_type == "campaign":
@@ -536,6 +768,7 @@ def share_file(
)
if existing:
existing.permission = permission
existing.expires_at = expires_at
session.add(existing)
return existing
share = FileShare(
@@ -545,6 +778,7 @@ def share_file(
target_id=target_id,
permission=permission,
created_by_user_id=user_id,
expires_at=expires_at,
)
session.add(share)
return share
@@ -559,6 +793,7 @@ def share_files(
target_id: str,
permission: str,
user_id: str,
expires_at: datetime | None = None,
) -> list[FileShare]:
target_type = target_type.lower().strip()
permission = permission.lower().strip()
@@ -566,6 +801,7 @@ def share_files(
raise FileStorageError("Unsupported share target")
if permission not in {"read", "write", "manage"}:
raise FileStorageError("Unsupported file permission")
_validate_share_expiry(expires_at)
if target_type in {"user", "group", "tenant"}:
ensure_share_target_exists(tenant_id=tenant_id, target_type=target_type, target_id=target_id)
if target_type == "campaign":
@@ -593,6 +829,7 @@ def share_files(
existing = existing_by_asset.get(asset.id)
if existing:
existing.permission = permission
existing.expires_at = expires_at
session.add(existing)
shares.append(existing)
continue
@@ -603,12 +840,88 @@ def share_files(
target_id=target_id,
permission=permission,
created_by_user_id=user_id,
expires_at=expires_at,
)
session.add(share)
shares.append(share)
return shares
def list_file_shares(
session: Session,
*,
tenant_id: str,
asset_id: str,
include_inactive: bool = False,
) -> list[FileShare]:
query = session.query(FileShare).filter(
FileShare.tenant_id == tenant_id,
FileShare.file_asset_id == asset_id,
)
if not include_inactive:
query = query.filter(effective_file_share_clause())
return query.order_by(FileShare.created_at.desc(), FileShare.id.desc()).all()
def revoke_file_share(
session: Session,
*,
tenant_id: str,
asset_id: str,
share_id: str,
user_id: str,
) -> tuple[FileShare, bool]:
share = (
session.query(FileShare)
.filter(
FileShare.id == share_id,
FileShare.tenant_id == tenant_id,
FileShare.file_asset_id == asset_id,
)
.one_or_none()
)
if share is None:
raise FileStorageError("File share not found")
if share.revoked_at is not None:
return share, False
share.revoked_at = utcnow()
share.revoked_by_user_id = user_id
session.add(share)
return share, True
def current_file_share_for_target(
session: Session,
*,
tenant_id: str,
asset_id: str,
target_type: str,
target_id: str,
) -> FileShare | None:
return (
session.query(FileShare)
.filter(
FileShare.tenant_id == tenant_id,
FileShare.file_asset_id == asset_id,
FileShare.target_type == target_type.lower().strip(),
FileShare.target_id == target_id,
FileShare.revoked_at.is_(None),
)
.order_by(FileShare.created_at.desc(), FileShare.id.desc())
.first()
)
def _validate_share_expiry(expires_at: datetime | None) -> None:
if expires_at is None:
return
from govoplan_files.backend.storage.share_state import file_share_is_active
candidate = FileShare(expires_at=expires_at)
if not file_share_is_active(candidate):
raise FileStorageError("File share expiry must be in the future")
def soft_delete_assets(session: Session, assets: Iterable[FileAsset]) -> int:
count = 0
@@ -0,0 +1,247 @@
from __future__ import annotations
import socket
import ssl
import select
from collections.abc import Mapping
from contextlib import contextmanager
from dataclasses import dataclass
from typing import Any, Iterator
import httpcore
import httpx
from govoplan_core.security.outbound_http import (
OutboundHttpError,
bounded_chunks_bytes,
create_outbound_connection,
response_limit,
validate_outbound_http_url,
)
class ConnectorHttpError(RuntimeError):
pass
@dataclass(frozen=True, slots=True)
class ConnectorHttpResponse:
status_code: int
headers: Mapping[str, str]
content: bytes
_HTTPCORE_TRANSPORT_ERRORS = (
httpcore.TimeoutException,
httpcore.NetworkError,
httpcore.ProtocolError,
httpcore.ProxyError,
httpcore.UnsupportedProtocol,
)
class _SocketNetworkStream(httpcore.NetworkStream):
"""Public httpcore NetworkStream adapter around an approved socket."""
def __init__(self, sock: socket.socket) -> None:
self._socket = sock
def read(self, max_bytes: int, timeout: float | None = None) -> bytes:
try:
self._socket.settimeout(timeout)
return self._socket.recv(max_bytes)
except socket.timeout as exc:
raise httpcore.ReadTimeout(str(exc)) from exc
except OSError as exc:
raise httpcore.ReadError(str(exc)) from exc
def write(self, buffer: bytes, timeout: float | None = None) -> None:
try:
self._socket.settimeout(timeout)
self._socket.sendall(buffer)
except socket.timeout as exc:
raise httpcore.WriteTimeout(str(exc)) from exc
except OSError as exc:
raise httpcore.WriteError(str(exc)) from exc
def close(self) -> None:
self._socket.close()
def start_tls(
self,
ssl_context: ssl.SSLContext,
server_hostname: str | None = None,
timeout: float | None = None,
) -> httpcore.NetworkStream:
try:
self._socket.settimeout(timeout)
tls_socket = ssl_context.wrap_socket(self._socket, server_hostname=server_hostname)
except socket.timeout as exc:
self.close()
raise httpcore.ConnectTimeout(str(exc)) from exc
except OSError as exc:
self.close()
raise httpcore.ConnectError(str(exc)) from exc
return _SocketNetworkStream(tls_socket)
def get_extra_info(self, info: str) -> Any:
if info == "ssl_object" and isinstance(self._socket, ssl.SSLSocket):
return self._socket
if info == "client_addr":
return self._socket.getsockname()
if info == "server_addr":
return self._socket.getpeername()
if info == "socket":
return self._socket
if info == "is_readable":
try:
return bool(select.select([self._socket], [], [], 0)[0])
except (OSError, ValueError):
return True
return None
class _OutboundPolicyNetworkBackend(httpcore.NetworkBackend):
def connect_tcp(
self,
host: str,
port: int,
timeout: float | None = None,
local_address: str | None = None,
socket_options: Any = None,
) -> httpcore.NetworkStream:
source_address = None if local_address is None else (local_address, 0)
try:
sock = create_outbound_connection(
host,
port,
timeout=timeout,
source_address=source_address,
socket_options=socket_options,
label="File connector HTTP endpoint",
)
except socket.timeout as exc:
raise httpcore.ConnectTimeout(str(exc)) from exc
except (OSError, OutboundHttpError) as exc:
raise httpcore.ConnectError(str(exc)) from exc
return _SocketNetworkStream(sock)
def connect_unix_socket(self, path: str, timeout: float | None = None, socket_options: Any = None): # type: ignore[no-untyped-def]
del path, timeout, socket_options
raise httpcore.ConnectError("Unix sockets are not supported for file connectors")
class _HttpcoreResponseStream(httpx.SyncByteStream):
def __init__(self, stream: Any, *, request: httpx.Request) -> None:
self._stream = stream
self._request = request
def __iter__(self): # type: ignore[no-untyped-def]
try:
yield from self._stream
except _HTTPCORE_TRANSPORT_ERRORS as exc:
raise httpx.TransportError(str(exc), request=self._request) from exc
def close(self) -> None:
if hasattr(self._stream, "close"):
self._stream.close()
class _OutboundPolicyHTTPTransport(httpx.BaseTransport):
def __init__(self) -> None:
self._connection_pool = httpcore.ConnectionPool(
ssl_context=httpx.create_ssl_context(verify=True, trust_env=False),
max_connections=100,
max_keepalive_connections=20,
keepalive_expiry=5.0,
network_backend=_OutboundPolicyNetworkBackend(),
)
def handle_request(self, request: httpx.Request) -> httpx.Response:
core_request = httpcore.Request(
method=request.method,
url=httpcore.URL(
scheme=request.url.raw_scheme,
host=request.url.raw_host,
port=request.url.port,
target=request.url.raw_path,
),
headers=request.headers.raw,
content=request.stream,
extensions=request.extensions,
)
try:
response = self._connection_pool.handle_request(core_request)
except _HTTPCORE_TRANSPORT_ERRORS as exc:
raise httpx.TransportError(str(exc), request=request) from exc
return httpx.Response(
status_code=response.status,
headers=response.headers,
stream=_HttpcoreResponseStream(response.stream, request=request),
extensions=response.extensions,
)
def close(self) -> None:
self._connection_pool.close()
_CONNECTOR_HTTP_CLIENT = httpx.Client(
transport=_OutboundPolicyHTTPTransport(),
follow_redirects=False,
timeout=15.0,
)
@contextmanager
def _stream_connector_request(method: str, url: str, **kwargs: Any) -> Iterator[httpx.Response]:
timeout = kwargs.pop("timeout", 15.0)
with _CONNECTOR_HTTP_CLIENT.stream(
method,
url,
timeout=timeout,
**kwargs,
) as response:
yield response
def request_connector_bytes(
method: str,
url: str,
*,
headers: Mapping[str, str] | None = None,
params: Mapping[str, str] | None = None,
data: Mapping[str, str] | bytes | str | None = None,
content: bytes | str | None = None,
auth: Any = None,
timeout: float = 15.0,
kind: str = "structured",
max_bytes: int | None = None,
label: str = "File connector",
) -> ConnectorHttpResponse:
try:
validated_url = validate_outbound_http_url(url, label=f"{label} URL")
effective_limit = response_limit(kind) if max_bytes is None else min(int(max_bytes), response_limit(kind))
with _stream_connector_request(
method,
validated_url,
headers=dict(headers or {}),
params=params,
data=data,
content=content,
auth=auth,
timeout=timeout,
) as response:
body = bounded_chunks_bytes(
response.iter_bytes(),
headers=response.headers,
max_bytes=effective_limit,
kind=kind,
label=f"{label} response",
)
return ConnectorHttpResponse(
status_code=response.status_code,
headers=dict(response.headers),
content=body,
)
except (httpx.HTTPError, OutboundHttpError, ValueError) as exc:
raise ConnectorHttpError(str(exc)) from exc
@@ -0,0 +1,550 @@
from __future__ import annotations
import hashlib
from dataclasses import dataclass
from sqlalchemy.orm import Session, object_session
from govoplan_files.backend.db.models import (
FileBlob,
FileIntegrityFinding,
FileIntegrityScan,
)
from govoplan_files.backend.storage.backends import (
StorageBackend,
StorageBackendError,
StorageObjectMissing,
get_storage_backend,
)
from govoplan_files.backend.storage.recovery import (
begin_orphan_cleanup_recovery,
)
from govoplan_files.backend.storage.common import FileStorageError, utcnow
TERMINAL_SCAN_STATUSES = {"completed", "cancelled"}
QUARANTINED_BLOB_STATUSES = {
"missing",
"size_mismatch",
"checksum_mismatch",
}
@dataclass(frozen=True, slots=True)
class BlobInspection:
valid: bool
kind: str
observed_size_bytes: int | None = None
observed_checksum_sha256: str | None = None
@dataclass(frozen=True, slots=True)
class IntegrityActionResult:
action: str
changed: bool
dry_run: bool
finding: FileIntegrityFinding
inspection: BlobInspection | None = None
def storage_prefix_for_tenant(tenant_id: str) -> str:
return f"tenants/{tenant_id}/files/"
def create_integrity_scan(
session: Session,
*,
tenant_id: str,
user_id: str,
verify_checksums: bool = True,
batch_size: int = 100,
backend: StorageBackend | None = None,
) -> FileIntegrityScan:
active_backend = backend or get_storage_backend()
scan = FileIntegrityScan(
tenant_id=tenant_id,
storage_backend=active_backend.name,
storage_prefix=storage_prefix_for_tenant(tenant_id),
verify_checksums=verify_checksums,
batch_size=max(1, min(int(batch_size), 1000)),
created_by_user_id=user_id,
)
session.add(scan)
session.flush()
return scan
def run_integrity_scan_batch(
session: Session,
scan: FileIntegrityScan,
*,
backend: StorageBackend | None = None,
) -> FileIntegrityScan:
if scan.status in TERMINAL_SCAN_STATUSES:
return scan
active_backend = backend or get_storage_backend()
if active_backend.name != scan.storage_backend:
raise FileStorageError(
"The configured storage backend changed after this integrity scan started"
)
if scan.started_at is None:
scan.started_at = utcnow()
scan.status = "running"
scan.last_error = None
if scan.phase == "blobs":
_scan_blob_batch(session, scan, active_backend)
elif scan.phase == "objects":
_scan_object_batch(session, scan, active_backend)
else:
scan.phase = "completed"
scan.status = "completed"
scan.completed_at = utcnow()
scan.revision += 1
session.add(scan)
return scan
def mark_integrity_scan_failed(
scan: FileIntegrityScan,
*,
error: Exception,
) -> None:
scan.status = "failed"
scan.last_error = type(error).__name__[:255]
scan.revision += 1
def inspect_blob(
blob: FileBlob,
*,
backend: StorageBackend,
verify_checksum: bool = True,
) -> BlobInspection:
expected_size = blob.storage_size_bytes if blob.storage_size_bytes is not None else blob.size_bytes
expected_checksum = blob.storage_checksum_sha256 if blob.storage_checksum_sha256 is not None else blob.checksum_sha256
try:
info = backend.stat(blob.storage_key)
except StorageObjectMissing:
return BlobInspection(valid=False, kind="missing")
if info.size_bytes != expected_size:
return BlobInspection(
valid=False,
kind="size_mismatch",
observed_size_bytes=info.size_bytes,
)
if not verify_checksum:
return BlobInspection(
valid=True,
kind="verified",
observed_size_bytes=info.size_bytes,
)
digest = hashlib.sha256()
observed_size = 0
for chunk in backend.iter_bytes(blob.storage_key):
observed_size += len(chunk)
digest.update(chunk)
observed_checksum = digest.hexdigest()
if observed_size != expected_size:
return BlobInspection(
valid=False,
kind="size_mismatch",
observed_size_bytes=observed_size,
observed_checksum_sha256=observed_checksum,
)
if observed_checksum != expected_checksum:
return BlobInspection(
valid=False,
kind="checksum_mismatch",
observed_size_bytes=observed_size,
observed_checksum_sha256=observed_checksum,
)
return BlobInspection(
valid=True,
kind="verified",
observed_size_bytes=observed_size,
observed_checksum_sha256=observed_checksum,
)
def apply_blob_inspection(
session: Session,
blob: FileBlob,
inspection: BlobInspection,
*,
checked_at=None,
) -> None:
timestamp = checked_at or utcnow()
blob.integrity_checked_at = timestamp
if inspection.valid:
blob.integrity_status = "verified"
blob.integrity_failure = None
blob.quarantined_at = None
else:
blob.integrity_status = inspection.kind
blob.integrity_failure = inspection.kind
blob.quarantined_at = blob.quarantined_at or timestamp
session.add(blob)
def ensure_blob_is_readable(blob: FileBlob) -> None:
if blob.integrity_status in QUARANTINED_BLOB_STATUSES or blob.quarantined_at:
raise FileStorageError(
"Managed file content is quarantined because integrity verification failed"
)
def read_verified_blob_bytes(
blob: FileBlob,
*,
backend: StorageBackend,
) -> bytes:
ensure_blob_is_readable(blob)
try:
data = backend.get_bytes(blob.storage_key)
except StorageBackendError as exc:
raise FileStorageError("Managed file content is not available") from exc
expected_storage_size = blob.storage_size_bytes if blob.storage_size_bytes is not None else blob.size_bytes
expected_storage_checksum = blob.storage_checksum_sha256 if blob.storage_checksum_sha256 is not None else blob.checksum_sha256
if len(data) != expected_storage_size:
raise FileStorageError(
"Managed file content failed its recorded size verification"
)
if hashlib.sha256(data).hexdigest() != expected_storage_checksum:
raise FileStorageError(
"Managed file content failed its recorded checksum verification"
)
if blob.encryption_envelope_id:
session = object_session(blob)
if session is None:
raise FileStorageError("Encrypted managed content requires an attached database session")
from govoplan_files.backend.storage.content_protection import unprotect_blob_content
data = unprotect_blob_content(
session,
tenant_id=blob.tenant_id,
blob_id=blob.id,
envelope_id=blob.encryption_envelope_id,
ciphertext=data,
)
if len(data) != blob.size_bytes:
raise FileStorageError("Decrypted managed file content failed its recorded size verification")
if hashlib.sha256(data).hexdigest() != blob.checksum_sha256:
raise FileStorageError("Decrypted managed file content failed its recorded checksum verification")
return data
def recheck_integrity_finding(
session: Session,
finding: FileIntegrityFinding,
*,
user_id: str,
dry_run: bool,
backend: StorageBackend | None = None,
) -> IntegrityActionResult:
if finding.kind == "orphan_object" or not finding.blob_id:
raise FileStorageError("Only blob integrity findings can be rechecked")
blob = session.get(FileBlob, finding.blob_id)
if blob is None or blob.tenant_id != finding.tenant_id:
raise FileStorageError("Integrity finding blob no longer exists")
active_backend = backend or get_storage_backend()
scan = session.get(FileIntegrityScan, finding.scan_id)
if scan is None or scan.storage_backend != active_backend.name:
raise FileStorageError(
"The configured storage backend does not match the integrity finding"
)
inspection = inspect_blob(
blob,
backend=active_backend,
verify_checksum=True,
)
changed = False
if not dry_run:
previous = (
blob.integrity_status,
blob.quarantined_at,
finding.state,
)
apply_blob_inspection(session, blob, inspection)
_update_finding_from_inspection(finding, inspection)
if inspection.valid:
finding.state = "resolved"
finding.resolved_at = utcnow()
finding.resolved_by_user_id = user_id
session.add(finding)
finding.revision += 1
changed = previous != (
blob.integrity_status,
blob.quarantined_at,
finding.state,
)
return IntegrityActionResult(
action="recheck",
changed=changed,
dry_run=dry_run,
finding=finding,
inspection=inspection,
)
def cleanup_orphan_finding(
session: Session,
finding: FileIntegrityFinding,
*,
user_id: str,
dry_run: bool,
backend: StorageBackend | None = None,
) -> IntegrityActionResult:
if finding.kind != "orphan_object":
raise FileStorageError("Only orphan-object findings can be cleaned up")
scan = session.get(FileIntegrityScan, finding.scan_id)
if scan is None or scan.tenant_id != finding.tenant_id:
raise FileStorageError("Integrity scan not found")
if not finding.storage_key.startswith(scan.storage_prefix):
raise FileStorageError("Orphan cleanup is outside the scan storage scope")
referenced = (
session.query(FileBlob.id)
.filter(
FileBlob.tenant_id == finding.tenant_id,
FileBlob.storage_backend == scan.storage_backend,
FileBlob.storage_key == finding.storage_key,
)
.first()
)
if referenced:
if not dry_run and finding.state != "resolved":
finding.state = "resolved"
finding.resolved_at = utcnow()
finding.resolved_by_user_id = user_id
finding.revision += 1
session.add(finding)
return IntegrityActionResult(
action="retained_referenced",
changed=True,
dry_run=False,
finding=finding,
)
return IntegrityActionResult(
action="retained_referenced",
changed=False,
dry_run=dry_run,
finding=finding,
)
if finding.state == "deleted":
return IntegrityActionResult(
action="already_deleted",
changed=False,
dry_run=dry_run,
finding=finding,
)
if dry_run:
return IntegrityActionResult(
action="would_delete",
changed=False,
dry_run=True,
finding=finding,
)
active_backend = backend or get_storage_backend()
if active_backend.name != scan.storage_backend:
raise FileStorageError(
"The configured storage backend does not match the integrity finding"
)
begin_orphan_cleanup_recovery(
session,
finding,
backend=active_backend,
user_id=user_id,
)
try:
active_backend.stat(finding.storage_key)
except StorageObjectMissing:
action = "already_absent"
else:
active_backend.delete(finding.storage_key)
action = "deleted"
finding.state = "deleted"
finding.resolved_at = utcnow()
finding.resolved_by_user_id = user_id
finding.revision += 1
session.add(finding)
return IntegrityActionResult(
action=action,
changed=True,
dry_run=False,
finding=finding,
)
def _scan_blob_batch(
session: Session,
scan: FileIntegrityScan,
backend: StorageBackend,
) -> None:
query = session.query(FileBlob).filter(
FileBlob.tenant_id == scan.tenant_id,
FileBlob.storage_backend == scan.storage_backend,
)
if scan.blob_cursor:
query = query.filter(FileBlob.id > scan.blob_cursor)
blobs = query.order_by(FileBlob.id.asc()).limit(scan.batch_size).all()
for blob in blobs:
inspection = inspect_blob(
blob,
backend=backend,
verify_checksum=scan.verify_checksums,
)
apply_blob_inspection(session, blob, inspection)
scan.scanned_blob_count += 1
if inspection.valid:
scan.verified_blob_count += 1
_resolve_scan_blob_findings(session, scan, blob)
else:
scan.quarantined_blob_count += 1
_record_blob_finding(session, scan, blob, inspection)
scan.blob_cursor = blob.id
if len(blobs) < scan.batch_size:
scan.phase = "objects"
scan.object_cursor = None
def _scan_object_batch(
session: Session,
scan: FileIntegrityScan,
backend: StorageBackend,
) -> None:
page = backend.list_objects(
prefix=scan.storage_prefix,
after=scan.object_cursor,
limit=scan.batch_size,
)
keys = [item.key for item in page.objects]
referenced_keys = {
row[0]
for row in session.query(FileBlob.storage_key)
.filter(
FileBlob.tenant_id == scan.tenant_id,
FileBlob.storage_backend == scan.storage_backend,
FileBlob.storage_key.in_(keys),
)
.all()
} if keys else set()
for item in page.objects:
scan.scanned_object_count += 1
if item.key in referenced_keys:
continue
scan.orphan_object_count += 1
_record_orphan_finding(
session,
scan,
storage_key=item.key,
size_bytes=item.size_bytes,
)
scan.object_cursor = page.next_cursor
if page.next_cursor is None:
scan.phase = "completed"
scan.status = "completed"
scan.completed_at = utcnow()
def _record_blob_finding(
session: Session,
scan: FileIntegrityScan,
blob: FileBlob,
inspection: BlobInspection,
) -> FileIntegrityFinding:
finding = (
session.query(FileIntegrityFinding)
.filter(
FileIntegrityFinding.scan_id == scan.id,
FileIntegrityFinding.blob_id == blob.id,
FileIntegrityFinding.kind == inspection.kind,
)
.one_or_none()
)
if finding is None:
finding = FileIntegrityFinding(
scan_id=scan.id,
tenant_id=scan.tenant_id,
kind=inspection.kind,
blob_id=blob.id,
storage_key=blob.storage_key,
expected_size_bytes=blob.storage_size_bytes if blob.storage_size_bytes is not None else blob.size_bytes,
expected_checksum_sha256=blob.storage_checksum_sha256 if blob.storage_checksum_sha256 is not None else blob.checksum_sha256,
)
else:
finding.revision += 1
_update_finding_from_inspection(finding, inspection)
session.add(finding)
return finding
def _record_orphan_finding(
session: Session,
scan: FileIntegrityScan,
*,
storage_key: str,
size_bytes: int,
) -> FileIntegrityFinding:
finding = (
session.query(FileIntegrityFinding)
.filter(
FileIntegrityFinding.scan_id == scan.id,
FileIntegrityFinding.kind == "orphan_object",
FileIntegrityFinding.storage_key == storage_key,
)
.one_or_none()
)
if finding is None:
finding = FileIntegrityFinding(
scan_id=scan.id,
tenant_id=scan.tenant_id,
kind="orphan_object",
storage_key=storage_key,
observed_size_bytes=size_bytes,
)
session.add(finding)
return finding
def _resolve_scan_blob_findings(
session: Session,
scan: FileIntegrityScan,
blob: FileBlob,
) -> None:
for finding in (
session.query(FileIntegrityFinding)
.filter(
FileIntegrityFinding.scan_id == scan.id,
FileIntegrityFinding.blob_id == blob.id,
FileIntegrityFinding.state == "open",
)
.all()
):
finding.state = "resolved"
finding.resolved_at = utcnow()
finding.revision += 1
session.add(finding)
def _update_finding_from_inspection(
finding: FileIntegrityFinding,
inspection: BlobInspection,
) -> None:
finding.observed_size_bytes = inspection.observed_size_bytes
finding.observed_checksum_sha256 = inspection.observed_checksum_sha256
__all__ = [
"BlobInspection",
"IntegrityActionResult",
"QUARANTINED_BLOB_STATUSES",
"apply_blob_inspection",
"cleanup_orphan_finding",
"create_integrity_scan",
"ensure_blob_is_readable",
"inspect_blob",
"mark_integrity_scan_failed",
"read_verified_blob_bytes",
"recheck_integrity_finding",
"run_integrity_scan_batch",
"storage_prefix_for_tenant",
]
@@ -0,0 +1,691 @@
from __future__ import annotations
from dataclasses import dataclass
from datetime import UTC, datetime
import hashlib
from typing import Protocol
from sqlalchemy import event
from sqlalchemy.orm import Session
from govoplan_core.core.recovery import (
RecoveryGuaranteeError,
RecoveryMode,
RecoveryPlan,
RecoveryStatus,
)
from govoplan_core.core.recovery_runtime import (
DurableRecoveryOperation,
RecoveryOperationBusy,
RecoveryOperationStateConflict,
begin_durable_recovery_operation,
)
from govoplan_core.core.runtime_coordination import process_runtime_identity
from govoplan_core.db.session import get_database
from govoplan_files.backend.db.models import (
FileBlob,
FileIntegrityFinding,
FileIntegrityScan,
)
from govoplan_files.backend.storage.backends import (
StorageBackend,
StorageBackendError,
StorageObjectMissing,
)
from govoplan_files.backend.storage.common import FileStorageError
_PENDING_EFFECTS_KEY = "govoplan_files_pending_recovery_effects"
_HOOKS_INSTALLED_KEY = "govoplan_files_recovery_hooks_installed"
_ROLLBACK_ERRORS_KEY = "govoplan_files_recovery_rollback_errors"
class _PendingEffect(Protocol):
def settle(self, *, committed: bool) -> None: ...
@dataclass(slots=True)
class PendingBlobWrite:
operation: DurableRecoveryOperation
backend: StorageBackend
tenant_id: str
blob_id: str
storage_key: str
semantic_checksum_sha256: str
semantic_size_bytes: int
protection_discriminator: str
created_new: bool
expected_storage_checksum_sha256: str | None = None
expected_storage_size_bytes: int | None = None
expected_envelope_id: str | None = None
def prepare_stored_bytes(
self,
data: bytes,
*,
envelope_id: str | None,
) -> None:
"""Retain process-local verification evidence without recording content."""
self.expected_storage_checksum_sha256 = hashlib.sha256(data).hexdigest()
self.expected_storage_size_bytes = len(data)
self.expected_envelope_id = envelope_id
def settle(self, *, committed: bool) -> None:
evidence = _blob_write_evidence(self)
if _blob_write_complete(evidence):
self.operation.succeed(evidence=evidence)
return
if self.created_new and evidence.get("database_blob_present") is False:
self._settle_unreferenced_new_object(evidence)
return
if not self.created_new and _object_matches(evidence):
if _forward_complete_blob_repair(self):
completed = _blob_write_evidence(self)
if _blob_write_complete(completed):
self.operation.succeed(evidence=completed)
return
if evidence.get("database_blob_present") is True and not _object_matches(
evidence
):
_quarantine_blob_after_failed_verification(self, evidence)
evidence = _blob_write_evidence(self)
status = (
RecoveryStatus.OUTCOME_UNKNOWN
if not evidence.get("verified")
else RecoveryStatus.RECOVERY_REQUIRED
)
transaction = "committed" if committed else "rolled back"
self.operation.unresolved(
status=status,
summary=f"Managed Files blob write remained unresolved after the database transaction {transaction}",
evidence=evidence,
failure_summary=(
"The managed object and Files blob metadata require reconciliation"
),
)
def _settle_unreferenced_new_object(self, evidence: dict[str, object]) -> None:
object_present = evidence.get("object_present")
if object_present is False:
self.operation.reject(
summary="The Files blob write left no durable object or database row",
evidence=evidence,
)
return
if object_present is not True:
self.operation.unresolved(
status=RecoveryStatus.OUTCOME_UNKNOWN,
summary="The unreferenced Files object could not be probed",
evidence=evidence,
failure_summary="Object storage availability prevented upload compensation",
)
return
try:
self.backend.delete(self.storage_key)
except (StorageBackendError, OSError):
self.operation.unresolved(
status=RecoveryStatus.RECOVERY_REQUIRED,
summary="An unreferenced Files object could not be compensated",
evidence=evidence,
failure_summary="Delete the unreferenced managed object after verifying that no FileBlob references it",
)
return
recovered = _blob_write_evidence(self)
if (
recovered.get("verified") is True
and recovered.get("database_blob_present") is False
and recovered.get("object_present") is False
):
self.operation.compensate(
failure_summary="The Files database transaction did not retain the new blob",
failure_evidence=evidence,
recovery_evidence=recovered,
)
return
self.operation.unresolved(
status=RecoveryStatus.RECOVERY_REQUIRED,
summary="Files upload compensation could not be verified",
evidence=recovered,
failure_summary="The unreferenced managed object requires operator reconciliation",
)
@dataclass(slots=True)
class PendingOrphanCleanup:
operation: DurableRecoveryOperation
backend: StorageBackend
finding_id: str
tenant_id: str
storage_key: str
resolved_by_user_id: str
def settle(self, *, committed: bool) -> None:
del committed
evidence = _orphan_cleanup_evidence(self)
if _orphan_cleanup_complete(evidence):
self.operation.succeed(evidence=evidence)
return
if (
evidence.get("verified") is True
and evidence.get("object_present") is True
and evidence.get("finding_deleted") is False
):
self.operation.reject(
summary="The orphan object was retained and the cleanup finding stayed open",
evidence=evidence,
)
return
if evidence.get("object_present") is False:
if _forward_complete_orphan_finding(self):
completed = _orphan_cleanup_evidence(self)
if _orphan_cleanup_complete(completed):
self.operation.succeed(evidence=completed)
return
status = (
RecoveryStatus.OUTCOME_UNKNOWN
if not evidence.get("verified")
else RecoveryStatus.RECOVERY_REQUIRED
)
self.operation.unresolved(
status=status,
summary="Files orphan cleanup requires reconciliation",
evidence=evidence,
failure_summary="Recheck the object and integrity-finding state before another cleanup attempt",
)
def begin_blob_write_recovery(
session: Session,
*,
backend: StorageBackend,
tenant_id: str,
blob_id: str,
storage_key: str,
semantic_checksum_sha256: str,
semantic_size_bytes: int,
protection_discriminator: str,
created_new: bool,
repair_token: str | None = None,
) -> PendingBlobWrite:
disposition = "create" if created_new else "repair"
state_token = repair_token or blob_id
key_digest = hashlib.sha256(storage_key.encode("utf-8")).hexdigest()
idempotency_key = (
f"files-blob-{disposition}:{blob_id}:{state_token[:48]}"
)
try:
started = begin_durable_recovery_operation(
get_database().SessionLocal,
identity=process_runtime_identity(),
module_id="files",
operation_type=f"blob-{disposition}",
idempotency_key=idempotency_key,
request={
"tenant_id": tenant_id,
"blob_id": blob_id,
"storage_key": storage_key if created_new else None,
"storage_key_sha256": key_digest,
"semantic_checksum_sha256": semantic_checksum_sha256,
"semantic_size_bytes": semantic_size_bytes,
"protection_discriminator": protection_discriminator,
"disposition": disposition,
},
recovery_plan=RecoveryPlan(
mode=(
RecoveryMode.COMPENSATION
if created_new
else RecoveryMode.FORWARD_RECOVERY
),
preconditions=(
"the caller has Files write authority for the target owner",
"the storage key belongs to the tenant Files namespace",
"the request records digests rather than file contents",
),
compensation_steps=(
"verify that no FileBlob references the newly reserved key",
"delete only that unreferenced key and verify absence",
)
if created_new
else (),
forward_recovery_steps=(
"verify the expected bytes at the existing blob key",
"forward-complete matching integrity metadata or quarantine the blob",
)
if not created_new
else (),
verification_steps=(
"reload FileBlob metadata through an independent session",
"stream and hash the managed object independently",
),
),
precondition_evidence={
"blob_id": blob_id,
"storage_key_sha256": key_digest,
"semantic_checksum_sha256": semantic_checksum_sha256,
"semantic_size_bytes": semantic_size_bytes,
"created_new": created_new,
},
lease_resource_key=(
f"files:blob:{tenant_id}:{blob_id}"
),
lease_ttl_seconds=15 * 60,
resource_type="file_blob",
resource_id=blob_id,
metadata={
"resources": ["postgresql", "object-storage"],
"storage_backend": backend.name,
},
)
except (RecoveryOperationBusy, RecoveryOperationStateConflict) as exc:
raise FileStorageError(
"This managed blob is already owned by another recovery operation"
) from exc
except (RecoveryGuaranteeError, RuntimeError) as exc:
raise FileStorageError(
"The Files recovery ledger is unavailable; no object was written"
) from exc
if started.replayed or started.operation is None:
raise FileStorageError(
"The matching Files blob operation was already completed; reload before retrying"
)
pending = PendingBlobWrite(
operation=started.operation,
backend=backend,
tenant_id=tenant_id,
blob_id=blob_id,
storage_key=storage_key,
semantic_checksum_sha256=semantic_checksum_sha256,
semantic_size_bytes=semantic_size_bytes,
protection_discriminator=protection_discriminator,
created_new=created_new,
)
_register_pending_effect(session, pending)
return pending
def begin_orphan_cleanup_recovery(
session: Session,
finding: FileIntegrityFinding,
*,
backend: StorageBackend,
user_id: str,
) -> PendingOrphanCleanup:
key_digest = hashlib.sha256(finding.storage_key.encode("utf-8")).hexdigest()
try:
started = begin_durable_recovery_operation(
get_database().SessionLocal,
identity=process_runtime_identity(),
module_id="files",
operation_type="integrity-orphan-cleanup",
idempotency_key=f"files-orphan-cleanup:{finding.id}",
request={
"tenant_id": finding.tenant_id,
"finding_id": finding.id,
"storage_key_sha256": key_digest,
},
recovery_plan=RecoveryPlan(
mode=RecoveryMode.FORWARD_RECOVERY,
preconditions=(
"the integrity finding identifies an unreferenced object",
"the object key remains inside the completed scan scope",
"a fresh database reference check found no FileBlob",
),
forward_recovery_steps=(
"verify object absence independently",
"mark the durable finding deleted only after absence is proven",
),
verification_steps=(
"reload the finding and its scan through an independent session",
"probe the original storage key through the configured backend",
),
),
precondition_evidence={
"finding_id": finding.id,
"scan_id": finding.scan_id,
"storage_key_sha256": key_digest,
"finding_state": finding.state,
},
lease_resource_key=(
f"files:orphan-cleanup:{finding.tenant_id}:{key_digest[:40]}"
),
lease_ttl_seconds=15 * 60,
resource_type="file_integrity_finding",
resource_id=finding.id,
metadata={
"resources": ["postgresql", "object-storage"],
"storage_backend": backend.name,
},
)
except (RecoveryOperationBusy, RecoveryOperationStateConflict) as exc:
raise FileStorageError(
"This orphan object is already owned by another recovery operation"
) from exc
except (RecoveryGuaranteeError, RuntimeError) as exc:
raise FileStorageError(
"The Files recovery ledger is unavailable; no object was deleted"
) from exc
if started.replayed or started.operation is None:
raise FileStorageError(
"This orphan cleanup was already completed; reload the finding"
)
pending = PendingOrphanCleanup(
operation=started.operation,
backend=backend,
finding_id=finding.id,
tenant_id=finding.tenant_id,
storage_key=finding.storage_key,
resolved_by_user_id=user_id,
)
_register_pending_effect(session, pending)
return pending
def _register_pending_effect(session: Session, effect: _PendingEffect) -> None:
if not session.in_transaction():
session.begin()
pending = session.info.setdefault(_PENDING_EFFECTS_KEY, [])
pending.append(effect)
if session.info.get(_HOOKS_INSTALLED_KEY):
return
event.listen(session, "after_commit", _after_session_commit)
event.listen(session, "after_rollback", _after_session_rollback)
session.info[_HOOKS_INSTALLED_KEY] = True
def _after_session_commit(session: Session) -> None:
_settle_pending_effects(session, committed=True)
def _after_session_rollback(session: Session) -> None:
try:
_settle_pending_effects(session, committed=False)
except RecoveryGuaranteeError as exc:
session.info.setdefault(_ROLLBACK_ERRORS_KEY, []).append(str(exc))
def _settle_pending_effects(session: Session, *, committed: bool) -> None:
pending = list(session.info.pop(_PENDING_EFFECTS_KEY, []))
failures: list[Exception] = []
for effect in pending:
try:
effect.settle(committed=committed)
except Exception as exc: # preserve every effect's chance to settle
failures.append(exc)
operation = getattr(effect, "operation", None)
if operation is not None and not operation.closed:
try:
operation.release_unresolved()
except Exception:
pass
if failures:
raise RecoveryGuaranteeError(
f"{len(failures)} Files recovery operation(s) could not be finalized"
) from failures[0]
def _blob_write_evidence(effect: PendingBlobWrite) -> dict[str, object]:
database_blob_present: bool | None
database_matches: bool | None
database_integrity_verified: bool | None
try:
with effect.operation.session_factory() as session:
blob = session.get(FileBlob, effect.blob_id)
database_blob_present = blob is not None
database_matches = bool(
blob is not None
and blob.tenant_id == effect.tenant_id
and blob.storage_key == effect.storage_key
and blob.checksum_sha256 == effect.semantic_checksum_sha256
and blob.size_bytes == effect.semantic_size_bytes
and blob.protection_discriminator
== effect.protection_discriminator
and blob.encryption_envelope_id == effect.expected_envelope_id
and (
effect.expected_storage_checksum_sha256
== (
blob.storage_checksum_sha256
or blob.checksum_sha256
)
)
) if blob is not None else False
database_integrity_verified = bool(
blob is not None
and blob.integrity_status == "verified"
and blob.quarantined_at is None
) if blob is not None else False
except Exception:
database_blob_present = None
database_matches = None
database_integrity_verified = None
object_present: bool | None
observed_size: int | None = None
observed_checksum: str | None = None
try:
digest = hashlib.sha256()
observed_size = 0
for chunk in effect.backend.iter_bytes(effect.storage_key):
observed_size += len(chunk)
digest.update(chunk)
observed_checksum = digest.hexdigest()
object_present = True
except StorageObjectMissing:
object_present = False
except (StorageBackendError, OSError):
object_present = None
object_size_matches = (
observed_size == effect.expected_storage_size_bytes
if object_present is True and effect.expected_storage_size_bytes is not None
else False if object_present is False else None
)
object_checksum_matches = (
observed_checksum == effect.expected_storage_checksum_sha256
if object_present is True
and effect.expected_storage_checksum_sha256 is not None
else False if object_present is False else None
)
verified = database_blob_present is not None and object_present is not None
return {
"verified": verified,
"checks": {
"database_reloaded": database_blob_present is not None,
"object_probed": object_present is not None,
"database_matches_request": database_matches,
"database_integrity_verified": database_integrity_verified,
"object_size_matches": object_size_matches,
"object_checksum_matches": object_checksum_matches,
},
"blob_id": effect.blob_id,
"database_blob_present": database_blob_present,
"database_matches_request": database_matches,
"database_integrity_verified": database_integrity_verified,
"object_present": object_present,
"observed_size_bytes": observed_size,
"observed_checksum_sha256": observed_checksum,
"expected_storage_size_bytes": effect.expected_storage_size_bytes,
"expected_storage_checksum_sha256": effect.expected_storage_checksum_sha256,
}
def _blob_write_complete(evidence: dict[str, object]) -> bool:
return bool(
evidence.get("verified") is True
and evidence.get("database_matches_request") is True
and evidence.get("database_integrity_verified") is True
and _object_matches(evidence)
)
def _object_matches(evidence: dict[str, object]) -> bool:
checks = evidence.get("checks")
return bool(
isinstance(checks, dict)
and evidence.get("object_present") is True
and checks.get("object_size_matches") is True
and checks.get("object_checksum_matches") is True
)
def _forward_complete_blob_repair(effect: PendingBlobWrite) -> bool:
try:
with effect.operation.session_factory() as session:
blob = session.get(FileBlob, effect.blob_id)
if (
blob is None
or blob.tenant_id != effect.tenant_id
or blob.storage_key != effect.storage_key
or blob.checksum_sha256 != effect.semantic_checksum_sha256
or blob.size_bytes != effect.semantic_size_bytes
or blob.protection_discriminator
!= effect.protection_discriminator
or blob.encryption_envelope_id != effect.expected_envelope_id
):
return False
blob.storage_checksum_sha256 = (
effect.expected_storage_checksum_sha256
if effect.expected_envelope_id
else None
)
blob.storage_size_bytes = (
effect.expected_storage_size_bytes
if effect.expected_envelope_id
else None
)
blob.integrity_status = "verified"
blob.integrity_checked_at = datetime.now(UTC)
blob.integrity_failure = None
blob.quarantined_at = None
session.add(blob)
session.commit()
return True
except Exception:
return False
def _quarantine_blob_after_failed_verification(
effect: PendingBlobWrite,
evidence: dict[str, object],
) -> None:
try:
with effect.operation.session_factory() as session:
blob = session.get(FileBlob, effect.blob_id)
if blob is None or blob.storage_key != effect.storage_key:
return
blob.integrity_status = (
"missing"
if evidence.get("object_present") is False
else "checksum_mismatch"
)
blob.integrity_checked_at = datetime.now(UTC)
blob.integrity_failure = "recovery_verification_failed"
blob.quarantined_at = datetime.now(UTC)
session.add(blob)
session.commit()
except Exception:
return
def _orphan_cleanup_evidence(effect: PendingOrphanCleanup) -> dict[str, object]:
try:
object_present: bool | None = effect.backend.exists(effect.storage_key)
except (StorageBackendError, OSError):
object_present = None
finding_present: bool | None
finding_deleted: bool | None
still_unreferenced: bool | None
try:
with effect.operation.session_factory() as session:
finding = session.get(FileIntegrityFinding, effect.finding_id)
finding_present = finding is not None
finding_deleted = bool(
finding is not None and finding.state == "deleted"
) if finding is not None else False
referenced = (
session.query(FileBlob.id)
.filter(
FileBlob.tenant_id == effect.tenant_id,
FileBlob.storage_key == effect.storage_key,
)
.first()
)
still_unreferenced = referenced is None
except Exception:
finding_present = None
finding_deleted = None
still_unreferenced = None
verified = object_present is not None and finding_present is not None
return {
"verified": verified,
"checks": {
"database_reloaded": finding_present is not None,
"object_probed": object_present is not None,
"finding_marked_deleted": finding_deleted,
"object_absent": (
not object_present if object_present is not None else None
),
"still_unreferenced": still_unreferenced,
},
"finding_id": effect.finding_id,
"finding_present": finding_present,
"finding_deleted": finding_deleted,
"object_present": object_present,
"still_unreferenced": still_unreferenced,
}
def _orphan_cleanup_complete(evidence: dict[str, object]) -> bool:
return bool(
evidence.get("verified") is True
and evidence.get("finding_deleted") is True
and evidence.get("object_present") is False
and evidence.get("still_unreferenced") is True
)
def _forward_complete_orphan_finding(effect: PendingOrphanCleanup) -> bool:
try:
with effect.operation.session_factory() as session:
finding = session.get(FileIntegrityFinding, effect.finding_id)
if (
finding is None
or finding.tenant_id != effect.tenant_id
or finding.storage_key != effect.storage_key
):
return False
scan = session.get(FileIntegrityScan, finding.scan_id)
if scan is None or not finding.storage_key.startswith(
scan.storage_prefix
):
return False
referenced = (
session.query(FileBlob.id)
.filter(
FileBlob.tenant_id == effect.tenant_id,
FileBlob.storage_key == effect.storage_key,
)
.first()
)
if referenced:
return False
finding.state = "deleted"
finding.resolved_at = datetime.now(UTC)
finding.resolved_by_user_id = effect.resolved_by_user_id
session.add(finding)
session.commit()
return True
except Exception:
return False
__all__ = [
"PendingBlobWrite",
"PendingOrphanCleanup",
"begin_blob_write_recovery",
"begin_orphan_cleanup_recovery",
]
@@ -0,0 +1,331 @@
from __future__ import annotations
import inspect
import socket
import threading
from functools import lru_cache
from importlib import import_module
from typing import Any
from govoplan_core.security.outbound_http import (
OutboundHttpError,
create_outbound_connection,
)
class SdkPeerPinningError(RuntimeError):
"""Raised when an optional SDK cannot be bound to the pinned transport."""
_BOTOCORE_CLIENT_CREATION_LOCK = threading.Lock()
def create_pinned_s3_client(**kwargs: Any) -> Any:
"""Construct an S3 client whose first and every later socket is pinned.
Botocore does not expose the HTTP-session class through boto3's public
client API. Its endpoint creator does expose that seam, so this function
replaces the creator only for the bounded client-construction operation.
The lock avoids an unsafe interleaving with another GovOPlaN client
construction. Any unrelated botocore client created during the short
replacement window also receives the stricter transport.
"""
try:
boto3 = import_module("boto3")
botocore_args = import_module("botocore.args")
except ImportError as exc:
raise SdkPeerPinningError(
"S3 connector browsing requires the optional boto3 dependency"
) from exc
pinned_session_cls = _botocore_transport_types()[0]
original_creator = getattr(botocore_args, "EndpointCreator", None)
if original_creator is None or not hasattr(original_creator, "create_endpoint"):
raise SdkPeerPinningError(
"The installed botocore release does not expose the endpoint transport seam required for peer pinning"
)
class PinnedEndpointCreator(original_creator): # type: ignore[misc, valid-type]
def create_endpoint(self, *args: Any, **endpoint_kwargs: Any) -> Any:
endpoint_kwargs["http_session_cls"] = pinned_session_cls
return super().create_endpoint(*args, **endpoint_kwargs)
with _BOTOCORE_CLIENT_CREATION_LOCK:
current_creator = getattr(botocore_args, "EndpointCreator", None)
if current_creator is not original_creator:
raise SdkPeerPinningError(
"Botocore endpoint construction changed concurrently; refusing to create an unproven S3 client"
)
botocore_args.EndpointCreator = PinnedEndpointCreator
try:
session_type = getattr(getattr(boto3, "session", None), "Session", None)
if session_type is None:
raise SdkPeerPinningError(
"The installed boto3 release does not expose its isolated session constructor"
)
client = session_type().client("s3", **kwargs)
finally:
if getattr(botocore_args, "EndpointCreator", None) is PinnedEndpointCreator:
botocore_args.EndpointCreator = original_creator
http_session = getattr(getattr(client, "_endpoint", None), "http_session", None)
if http_session is None or not isinstance(http_session, pinned_session_cls):
close = getattr(client, "close", None)
if callable(close):
close()
raise SdkPeerPinningError(
"Botocore did not install the required pinned HTTP transport; the S3 client was discarded"
)
return client
@lru_cache(maxsize=1)
def _botocore_transport_types() -> tuple[type[Any], type[Any], type[Any]]:
try:
awsrequest = import_module("botocore.awsrequest")
httpsession = import_module("botocore.httpsession")
urllib3_exceptions = import_module("urllib3.exceptions")
except ImportError as exc:
raise SdkPeerPinningError(
"S3 connector browsing requires a compatible botocore HTTP transport"
) from exc
required = (
"AWSHTTPConnection",
"AWSHTTPSConnection",
"AWSHTTPConnectionPool",
"AWSHTTPSConnectionPool",
)
if any(not hasattr(awsrequest, name) for name in required) or not hasattr(
httpsession, "URLLib3Session"
):
raise SdkPeerPinningError(
"The installed botocore release is missing the connection classes required for peer pinning"
)
def pinned_new_connection(connection: Any) -> socket.socket:
hostname = str(getattr(connection, "_dns_host", "") or "").strip()
port = int(getattr(connection, "port", 0) or 0)
if not hostname or not port:
raise urllib3_exceptions.NewConnectionError(
connection, "Pinned S3 connection is missing its target authority"
)
try:
return create_outbound_connection(
hostname,
port,
timeout=getattr(connection, "timeout", None),
source_address=getattr(connection, "source_address", None),
socket_options=getattr(connection, "socket_options", None),
label="S3 connector peer",
)
except socket.timeout as exc:
raise urllib3_exceptions.ConnectTimeoutError(
connection,
f"Connection to {hostname} timed out while selecting an approved peer",
) from exc
except (OSError, OutboundHttpError, ValueError) as exc:
raise urllib3_exceptions.NewConnectionError(
connection,
f"S3 connector peer was rejected: {exc}",
) from exc
class PinnedAWSHTTPConnection(awsrequest.AWSHTTPConnection):
_new_conn = pinned_new_connection
class PinnedAWSHTTPSConnection(awsrequest.AWSHTTPSConnection):
_new_conn = pinned_new_connection
class PinnedAWSHTTPConnectionPool(awsrequest.AWSHTTPConnectionPool):
ConnectionCls = PinnedAWSHTTPConnection
class PinnedAWSHTTPSConnectionPool(awsrequest.AWSHTTPSConnectionPool):
ConnectionCls = PinnedAWSHTTPSConnection
class PinnedURLLib3Session(httpsession.URLLib3Session):
def __init__(self, *args: Any, **kwargs: Any) -> None:
# A proxy would select the final target outside this process. Until
# proxy peer delegation is modeled, connector traffic is direct.
kwargs["proxies"] = {}
super().__init__(*args, **kwargs)
self._pool_classes_by_scheme = {
"http": PinnedAWSHTTPConnectionPool,
"https": PinnedAWSHTTPSConnectionPool,
}
manager = getattr(self, "_manager", None)
if manager is None or not hasattr(manager, "pool_classes_by_scheme"):
raise SdkPeerPinningError(
"The installed botocore pool manager cannot enforce pinned connection classes"
)
manager.pool_classes_by_scheme = self._pool_classes_by_scheme
return PinnedURLLib3Session, PinnedAWSHTTPConnection, PinnedAWSHTTPSConnection
@lru_cache(maxsize=1)
def pinned_smb_connection_cache() -> dict[str, Any]:
"""Return the Files-owned cache; unproven process-global sessions are never reused."""
return {}
def install_pinned_smb_transport(smbclient: Any) -> Any:
"""Install a process-wide fail-closed smbclient session factory.
smbclient routes initial connections, reconnects, and DFS targets through
``smbclient._pool.register_session``. Replacing that single factory with a
behavior-compatible implementation ensures every target uses a socket
selected by Core's deployment-wide outbound policy.
"""
try:
pool = import_module("smbclient._pool")
connection_module = import_module("smbprotocol.connection")
session_module = import_module("smbprotocol.session")
transport_module = import_module("smbprotocol.transport")
except ImportError as exc:
raise SdkPeerPinningError(
"SMB connector browsing requires the optional smbprotocol dependency"
) from exc
current = getattr(pool, "register_session", None)
if getattr(current, "__govoplan_peer_pinned__", False):
expected_tcp = getattr(current, "__govoplan_pinned_tcp__", None)
if expected_tcp is None or getattr(connection_module, "Tcp", None) is not expected_tcp:
raise SdkPeerPinningError(
"The installed SMB transport changed after peer pinning; refusing to reuse the session factory"
)
return smbclient
required_parameters = {
"server",
"username",
"password",
"port",
"encrypt",
"connection_timeout",
"connection_cache",
"auth_protocol",
"require_signing",
}
if current is None or not required_parameters.issubset(
inspect.signature(current).parameters
):
raise SdkPeerPinningError(
"The installed smbprotocol release does not expose the session seam required for peer pinning"
)
class PinnedTcp(transport_module.Tcp):
def connect(self) -> None:
with self._sock_lock:
if self.connected:
return
try:
self._sock = create_outbound_connection(
self.server,
int(self.port),
timeout=self.timeout,
label="SMB connector peer",
)
except (OSError, OutboundHttpError, ValueError) as exc:
raise ValueError(
f"SMB connector peer '{self.server}:{self.port}' was rejected: {exc}"
) from exc
self._sock.settimeout(None)
self.connected = True
# Connection.connect() instantiates the module-level Tcp symbol on every
# reconnect. Replacing it process-wide is deliberate: an SMB connection
# created by another code path must become stricter, never bypass Files'
# peer boundary.
if not hasattr(connection_module, "Tcp"):
raise SdkPeerPinningError(
"The installed smbprotocol release cannot install the pinned TCP transport"
)
connection_module.Tcp = PinnedTcp
def pinned_register_session(
server: str,
username: str | None = None,
password: str | None = None,
port: int = 445,
encrypt: bool | None = None,
connection_timeout: float = 60,
connection_cache: dict[str, Any] | None = None,
auth_protocol: str = "negotiate",
require_signing: bool = True,
) -> Any:
cache = pinned_smb_connection_cache() if connection_cache is None else connection_cache
connection_key = f"{server.lower()}:{port}"
connection = cache.get(connection_key)
transport = getattr(connection, "transport", None)
if connection is not None and not isinstance(transport, PinnedTcp):
disconnect = getattr(connection, "disconnect", None)
if callable(disconnect):
try:
disconnect(close=True)
except Exception:
pass
cache.pop(connection_key, None)
connection = None
if connection is None or not getattr(connection.transport, "connected", False):
connection = connection_module.Connection(
pool.ClientConfig().client_guid,
server,
port,
require_signing=require_signing,
)
connection.transport = PinnedTcp(server, port)
connection.connect(timeout=connection_timeout)
if not isinstance(connection.transport, PinnedTcp):
disconnect = getattr(connection, "disconnect", None)
if callable(disconnect):
try:
disconnect(close=True)
except Exception:
pass
raise SdkPeerPinningError(
"smbprotocol replaced the required pinned TCP transport during connection setup"
)
cache[connection_key] = connection
session = next(
(
item
for item in connection.session_table.values()
if username is None or item.username == username
),
None,
)
if session is None:
session = session_module.Session(
connection,
username=username,
password=password,
require_encryption=(encrypt is True),
auth_protocol=auth_protocol,
)
session.connect()
elif encrypt is not None:
if session.encrypt_data and not encrypt:
raise ValueError(
"Cannot disable encryption on an already negotiated session."
)
if not session.encrypt_data and encrypt:
session.encrypt = True
return session
pinned_register_session.__govoplan_peer_pinned__ = True # type: ignore[attr-defined]
pinned_register_session.__govoplan_pinned_tcp__ = PinnedTcp # type: ignore[attr-defined]
pool.register_session = pinned_register_session
if hasattr(smbclient, "register_session"):
smbclient.register_session = pinned_register_session
return smbclient
__all__ = [
"SdkPeerPinningError",
"create_pinned_s3_client",
"install_pinned_smb_transport",
"pinned_smb_connection_cache",
]
@@ -15,21 +15,6 @@ from govoplan_files.backend.storage.common import (
utcnow,
)
from govoplan_files.backend.storage.files import (
_active_asset_at_path,
_active_asset_exists,
_asset_owner_id,
_asset_query_for_owner,
_candidate_renamed_path,
_copy_asset_to_path,
_get_or_create_blob,
_next_available_logical_path,
_normalize_conflict_strategy,
_resolution_by_path,
_soft_delete_conflicting_asset,
_split_logical_path,
_storage_backend_name,
_storage_bucket_name,
_storage_key,
asset_is_audit_relevant,
create_file_asset,
current_version_and_blob,
@@ -42,10 +27,6 @@ from govoplan_files.backend.storage.files import (
soft_delete_assets,
)
from govoplan_files.backend.storage.folders import (
_active_folder_exists,
_ensure_target_folder_hierarchy,
_folder_query_for_owner,
_owner_filter,
create_folder,
list_folders_for_user,
soft_delete_folder,
@@ -0,0 +1,34 @@
from __future__ import annotations
from datetime import datetime, timezone
from sqlalchemy import and_, or_
from govoplan_files.backend.db.models import FileShare
from govoplan_files.backend.storage.common import utcnow
def effective_file_share_clause(*, at: datetime | None = None):
checked_at = at or utcnow()
return and_(
FileShare.revoked_at.is_(None),
or_(FileShare.expires_at.is_(None), FileShare.expires_at > checked_at),
)
def file_share_is_active(share: FileShare, *, at: datetime | None = None) -> bool:
if share.revoked_at is not None:
return False
if share.expires_at is None:
return True
checked_at = _aware_utc(at or utcnow())
return _aware_utc(share.expires_at) > checked_at
def _aware_utc(value: datetime) -> datetime:
if value.tzinfo is None:
return value.replace(tzinfo=timezone.utc)
return value.astimezone(timezone.utc)
__all__ = ["effective_file_share_clause", "file_share_is_active"]
File diff suppressed because it is too large Load Diff
+126 -2
View File
@@ -1,15 +1,18 @@
from __future__ import annotations
import unittest
from datetime import datetime, timedelta, timezone
from sqlalchemy import create_engine
from sqlalchemy import create_engine, text
from sqlalchemy.orm import sessionmaker
from govoplan_access.backend.db.models import Account, Group, User
from govoplan_core.core.access import PrincipalRef
from govoplan_core.core.change_sequence import ChangeSequenceEntry
from govoplan_core.db.base import Base
from govoplan_files.backend.capabilities import FilesAccessService, virtual_folder_resource_id
from govoplan_files.backend.db.models import FileAsset, FileFolder, FileShare
from govoplan_files.backend.storage.files import count_assets_for_user, list_assets_for_user
TENANT_ID = "tenant-1"
@@ -21,6 +24,7 @@ GROUP_ID = "group-1"
class FilesAccessProviderTests(unittest.TestCase):
def test_file_access_provider_explains_owner_share_admin_and_missing_resources(self) -> None:
session = _session()
self.addCleanup(_close_session, session)
_seed_access_subjects(session)
owned = FileAsset(id="file-owned", tenant_id=TENANT_ID, owner_type="user", owner_user_id=USER_ID, display_path="owned.pdf", filename="owned.pdf")
shared = FileAsset(id="file-shared", tenant_id=TENANT_ID, owner_type="user", owner_user_id=OTHER_USER_ID, display_path="shared.pdf", filename="shared.pdf")
@@ -46,6 +50,7 @@ class FilesAccessProviderTests(unittest.TestCase):
def test_file_access_provider_explains_persisted_and_virtual_folders(self) -> None:
session = _session()
self.addCleanup(_close_session, session)
_seed_access_subjects(session)
persisted = FileFolder(id="folder-persisted", tenant_id=TENANT_ID, owner_type="group", owner_group_id=GROUP_ID, path="records")
virtual_child = FileAsset(
@@ -73,13 +78,132 @@ class FilesAccessProviderTests(unittest.TestCase):
self.assertTrue(any(item.kind == "owner" and item.id == GROUP_ID for item in virtual_items))
self.assertEqual("files.not_found", missing_items[0].source)
def test_file_property_filters_cover_campaign_and_audit_usage(self) -> None:
session = _session()
self.addCleanup(_close_session, session)
_seed_access_subjects(session)
assets = [
FileAsset(
id=f"file-{index}",
tenant_id=TENANT_ID,
owner_type="user",
owner_user_id=USER_ID,
display_path=f"{index}.pdf",
filename=f"{index}.pdf",
)
for index in range(1, 4)
]
session.add_all([
*assets,
FileShare(
id="share-campaign",
tenant_id=TENANT_ID,
file_asset_id=assets[0].id,
target_type="campaign",
target_id="campaign-1",
permission="read",
),
FileShare(
id="share-campaign-expired",
tenant_id=TENANT_ID,
file_asset_id=assets[2].id,
target_type="campaign",
target_id="campaign-1",
permission="read",
expires_at=datetime.now(timezone.utc) - timedelta(seconds=1),
),
])
session.commit()
session.execute(
text(
"INSERT INTO campaign_attachment_uses "
"(id, tenant_id, file_asset_id, use_stage) "
"VALUES (:id, :tenant_id, :file_asset_id, :use_stage)"
),
{
"id": "attachment-use-1",
"tenant_id": TENANT_ID,
"file_asset_id": assets[1].id,
"use_stage": "sent",
},
)
session.commit()
linked = list_assets_for_user(
session,
tenant_id=TENANT_ID,
user_id=USER_ID,
owner_type="user",
owner_id=USER_ID,
campaign_usage="linked",
)
unlinked = list_assets_for_user(
session,
tenant_id=TENANT_ID,
user_id=USER_ID,
owner_type="user",
owner_id=USER_ID,
campaign_usage="unlinked",
)
audit_relevant = list_assets_for_user(
session,
tenant_id=TENANT_ID,
user_id=USER_ID,
owner_type="user",
owner_id=USER_ID,
audit_relevant=True,
)
self.assertEqual([asset.id for asset in linked], ["file-1", "file-2"])
self.assertEqual([asset.id for asset in unlinked], ["file-3"])
self.assertEqual([asset.id for asset in audit_relevant], ["file-2"])
self.assertEqual(
count_assets_for_user(
session,
tenant_id=TENANT_ID,
user_id=USER_ID,
owner_type="user",
owner_id=USER_ID,
campaign_usage="unlinked",
),
1,
)
def _session():
engine = create_engine("sqlite:///:memory:", future=True)
Base.metadata.create_all(bind=engine, tables=[Account.__table__, User.__table__, Group.__table__, FileAsset.__table__, FileFolder.__table__, FileShare.__table__])
Base.metadata.create_all(
bind=engine,
tables=[
Account.__table__,
User.__table__,
Group.__table__,
ChangeSequenceEntry.__table__,
FileAsset.__table__,
FileFolder.__table__,
FileShare.__table__,
],
)
with engine.begin() as connection:
connection.execute(
text(
"CREATE TABLE campaign_attachment_uses ("
"id VARCHAR(36) PRIMARY KEY, "
"tenant_id VARCHAR(36) NOT NULL, "
"file_asset_id VARCHAR(36) NOT NULL, "
"use_stage VARCHAR(20) NOT NULL"
")"
)
)
return sessionmaker(bind=engine, future=True)()
def _close_session(session) -> None:
engine = session.get_bind()
session.close()
engine.dispose()
def _seed_access_subjects(session) -> None:
session.add_all([
Account(id="account-1", email="one@example.test", normalized_email="one@example.test"),
+101
View File
@@ -0,0 +1,101 @@
from __future__ import annotations
import io
import hashlib
import unittest
import zipfile
from types import SimpleNamespace
from unittest.mock import patch
from fastapi import UploadFile
from govoplan_files.backend.routes.uploads import (
_validate_archive_preview_token,
preview_archive_upload,
)
from govoplan_files.backend.storage.common import FileStorageError
def _archive_bytes() -> bytes:
target = io.BytesIO()
with zipfile.ZipFile(target, "w", compression=zipfile.ZIP_DEFLATED) as archive:
archive.writestr("folder/report.txt", b"report")
return target.getvalue()
class ArchivePreviewTokenTests(unittest.TestCase):
def setUp(self) -> None:
self.settings = SimpleNamespace(
file_upload_zip_max_bytes=10 * 1024 * 1024,
file_archive_max_entries=10_000,
file_archive_max_expanded_bytes=2 * 1024 * 1024 * 1024,
file_archive_max_expansion_ratio=100,
file_archive_preview_ttl_seconds=30 * 60,
)
self.principal = SimpleNamespace(
tenant_id="tenant-1",
user=SimpleNamespace(id="user-1"),
)
def test_preview_token_is_bound_to_actor_archive_and_destination(self) -> None:
archive_data = _archive_bytes()
upload = UploadFile(
file=io.BytesIO(archive_data),
filename="monthly.zip",
)
with patch(
"govoplan_files.backend.routes.uploads.settings",
self.settings,
):
preview = preview_archive_upload(
file=upload,
owner_type="user",
owner_id="user-1",
path="incoming",
campaign_id=None,
password=None,
principal=self.principal,
)
_validate_archive_preview_token(
preview.preview_token,
tenant_id="tenant-1",
user_id="user-1",
owner_type="user",
owner_id="user-1",
path="incoming",
campaign_id=None,
archive_format="zip",
archive_sha256=hashlib.sha256(archive_data).hexdigest(),
)
with self.assertRaisesRegex(
FileStorageError, "does not match this upload destination"
):
_validate_archive_preview_token(
preview.preview_token,
tenant_id="tenant-1",
user_id="user-1",
owner_type="user",
owner_id="user-1",
path="elsewhere",
campaign_id=None,
archive_format="zip",
archive_sha256=hashlib.sha256(archive_data).hexdigest(),
)
with self.assertRaisesRegex(
FileStorageError, "contents changed"
):
_validate_archive_preview_token(
preview.preview_token,
tenant_id="tenant-1",
user_id="user-1",
owner_type="user",
owner_id="user-1",
path="incoming",
campaign_id=None,
archive_format="zip",
archive_sha256="0" * 64,
)
if __name__ == "__main__":
unittest.main()
+170
View File
@@ -0,0 +1,170 @@
from __future__ import annotations
import io
import tarfile
import unittest
import zipfile
from types import SimpleNamespace
from unittest.mock import patch
import pyzipper
from govoplan_files.backend.storage.archives import (
ArchivePasswordError,
extract_archive_upload,
inspect_archive,
)
from govoplan_files.backend.storage.common import FileStorageError
def _zip_bytes(entries: dict[str, bytes]) -> bytes:
target = io.BytesIO()
with zipfile.ZipFile(target, "w", compression=zipfile.ZIP_DEFLATED) as archive:
for path, data in entries.items():
archive.writestr(path, data)
return target.getvalue()
def _tar_bytes(entries: dict[str, bytes], mode: str = "w:gz") -> bytes:
target = io.BytesIO()
with tarfile.open(fileobj=target, mode=mode) as archive:
for path, data in entries.items():
info = tarfile.TarInfo(path)
info.size = len(data)
archive.addfile(info, io.BytesIO(data))
return target.getvalue()
def _encrypted_zip_bytes(password: str) -> bytes:
target = io.BytesIO()
with pyzipper.AESZipFile(
target,
"w",
compression=zipfile.ZIP_DEFLATED,
encryption=pyzipper.WZ_AES,
) as archive:
archive.setpassword(password.encode("utf-8"))
archive.writestr("secure/report.txt", b"classified")
return target.getvalue()
class ArchiveInspectionTests(unittest.TestCase):
def test_zip_preview_derives_folders_and_sizes(self) -> None:
inspection = inspect_archive(
_zip_bytes(
{
"department/one.txt": b"one",
"department/nested/two.csv": b"two",
}
),
filename="monthly.zip",
)
self.assertEqual("zip", inspection.archive_format)
self.assertEqual(2, inspection.file_count)
self.assertEqual(2, inspection.directory_count)
self.assertEqual(6, inspection.expanded_size_bytes)
self.assertEqual(
[
"department",
"department/nested",
"department/nested/two.csv",
"department/one.txt",
],
[entry.path for entry in inspection.entries],
)
def test_tar_compression_variants_are_inspected(self) -> None:
for filename, mode in (
("monthly.tar", "w"),
("monthly.tar.gz", "w:gz"),
("monthly.tar.bz2", "w:bz2"),
("monthly.tar.xz", "w:xz"),
):
with self.subTest(filename=filename):
inspection = inspect_archive(
_tar_bytes({"report.txt": b"report"}, mode=mode),
filename=filename,
)
self.assertEqual(1, inspection.file_count)
self.assertEqual(filename.removeprefix("monthly."), inspection.archive_format)
def test_unsafe_member_path_is_rejected(self) -> None:
with self.assertRaisesRegex(FileStorageError, "Unsafe archive member"):
inspect_archive(
_zip_bytes({"../escape.txt": b"no"}),
filename="unsafe.zip",
)
def test_archive_bomb_ratio_is_rejected(self) -> None:
with self.assertRaisesRegex(FileStorageError, "expansion ratio"):
inspect_archive(
_zip_bytes({"zeros.bin": b"\0" * 50_000}),
filename="bomb.zip",
max_expansion_ratio=2,
)
def test_encrypted_zip_reports_and_verifies_password(self) -> None:
archive = _encrypted_zip_bytes("correct horse")
missing = inspect_archive(archive, filename="secure.zip")
self.assertTrue(missing.requires_password)
self.assertFalse(missing.password_verified)
verified = inspect_archive(
archive,
filename="secure.zip",
password="correct horse",
)
self.assertTrue(verified.requires_password)
self.assertTrue(verified.password_verified)
with self.assertRaisesRegex(ArchivePasswordError, "incorrect"):
inspect_archive(
archive,
filename="secure.zip",
password="wrong",
)
def test_folder_selection_extracts_only_descendants(self) -> None:
archive = _zip_bytes(
{
"selected/one.txt": b"one",
"selected/two.txt": b"two",
"other/three.txt": b"three",
}
)
stored = SimpleNamespace(
asset=SimpleNamespace(id="asset"),
version=SimpleNamespace(id="version"),
blob=SimpleNamespace(id="blob"),
)
with patch(
"govoplan_files.backend.storage.archives.create_file_asset",
return_value=stored,
) as create_asset:
result = extract_archive_upload(
object(), # type: ignore[arg-type]
tenant_id="tenant-1",
owner_type="user",
owner_id="user-1",
user_id="user-1",
archive_data=archive,
filename="selection.zip",
folder="imports",
campaign_id=None,
selected_paths=["selected"],
)
self.assertEqual(2, len(result))
self.assertEqual(
{"imports/selected/one.txt", "imports/selected/two.txt"},
{
call.kwargs["display_path"]
for call in create_asset.call_args_list
},
)
if __name__ == "__main__":
unittest.main()
+89
View File
@@ -0,0 +1,89 @@
from __future__ import annotations
from types import SimpleNamespace
import unittest
from unittest.mock import patch
from govoplan_core.auth import ApiPrincipal
from govoplan_core.core.access import PrincipalRef
from govoplan_core.core.files import ManagedArtifactStore, ManagedArtifactWriteRequest
from govoplan_files.backend.capabilities import FilesArtifactStore
def principal() -> ApiPrincipal:
return ApiPrincipal(
principal=PrincipalRef(
account_id="account-1",
membership_id="membership-1",
tenant_id="tenant-1",
scopes=frozenset({"files:file:upload"}),
),
account=object(),
user=SimpleNamespace(id="user-1"),
)
class _Session:
def query(self):
raise AssertionError("not used")
def flush(self):
raise AssertionError("not used")
def stored():
return SimpleNamespace(
asset=SimpleNamespace(
id="file-1",
display_path="Generated/Templates/result.html",
owner_type="user",
),
version=SimpleNamespace(
id="version-1",
filename_at_upload="result.html",
content_type="text/html",
size_bytes=6,
checksum_sha256="0" * 64,
),
)
class FilesArtifactStoreTests(unittest.TestCase):
def test_plain_write_uses_files_authority_and_returns_provider_neutral_ref(self) -> None:
service = FilesArtifactStore()
self.assertIsInstance(service, ManagedArtifactStore)
request = ManagedArtifactWriteRequest(
filename="result.html",
payload=b"result",
content_type="text/html",
folder="Generated/Templates",
metadata={"producer_module": "templates"},
)
with patch("govoplan_files.backend.capabilities.create_file_asset", return_value=stored()) as create:
result = service.store_artifact(_Session(), principal(), request=request)
self.assertEqual("file-1", result.file_asset_id)
self.assertEqual("version-1", result.file_version_id)
self.assertEqual("tenant-1", create.call_args.kwargs["tenant_id"])
self.assertEqual("user-1", create.call_args.kwargs["owner_id"])
self.assertEqual(b"result", create.call_args.kwargs["data"])
def test_idempotent_write_uses_source_provenance_without_payload_metadata(self) -> None:
request = ManagedArtifactWriteRequest(
filename="result.html",
payload=b"secret body",
content_type="text/html",
idempotency_key="render-1",
metadata={"producer_module": "templates", "output_sha256": "a" * 64},
)
with patch(
"govoplan_files.backend.capabilities.sync_file_asset_from_source",
return_value=(stored(), "unchanged", None),
) as sync:
FilesArtifactStore().store_artifact(_Session(), principal(), request=request)
metadata = sync.call_args.kwargs["metadata"]
self.assertEqual("render-1", metadata["source_provenance"]["external_id"])
self.assertNotIn("secret body", str(metadata))
if __name__ == "__main__":
unittest.main()
+278
View File
@@ -0,0 +1,278 @@
from __future__ import annotations
import unittest
from types import SimpleNamespace
from unittest.mock import patch
from sqlalchemy import create_engine, inspect
from sqlalchemy.orm import sessionmaker
from govoplan_access.backend.db import models as access_models # noqa: F401 - resolve Files user foreign keys
from govoplan_core.core.change_sequence import ChangeSequenceEntry
from govoplan_core.db.base import Base
from govoplan_core.security.secrets import encrypt_secret
from govoplan_files.backend.db.models import FileConnectorCredential, FileConnectorProfile
from govoplan_files.backend.routes.connector_profiles import deactivate_connector_profile
from govoplan_files.backend.routes.connector_settings import deactivate_connector_credential
class Principal:
tenant_id = "tenant-1"
user = SimpleNamespace(id="user-1")
api_key = None
def has(self, scope: str) -> bool:
return scope == "files:file:admin"
class ConnectorCredentialDeletionTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite:///:memory:")
Base.metadata.create_all(
self.engine,
tables=[
FileConnectorCredential.__table__,
FileConnectorProfile.__table__,
ChangeSequenceEntry.__table__,
],
)
self.session = sessionmaker(bind=self.engine)()
self.principal = Principal()
def tearDown(self) -> None:
self.session.close()
Base.metadata.drop_all(bind=self.engine)
self.engine.dispose()
def test_delete_credential_scrubs_material_audits_and_disables_dependents(self) -> None:
credential = self._credential()
profile = self._profile(credential_profile_id=credential.id)
self.session.add_all([credential, profile])
self.session.commit()
with patch(
"govoplan_files.backend.storage.connector_credential_deletion.audit_event"
) as audit:
response = deactivate_connector_credential(
credential.id,
session=self.session,
principal=self.principal, # type: ignore[arg-type]
)
self.assertFalse(response.enabled)
self.assertFalse(response.credentials_configured)
self.assertIsNone(response.credential_secret_source)
self.session.refresh(credential)
self.session.refresh(profile)
self._assert_credential_scrubbed(credential)
self._assert_profile_scrubbed(profile)
calls = {call.kwargs["action"]: call.kwargs for call in audit.call_args_list}
self.assertEqual(
{"files.connector_profile_deleted", "files.connector_credential_deleted"},
set(calls),
)
credential_audit = calls["files.connector_credential_deleted"]
self.assertEqual("encrypted_database", credential_audit["details"]["storage_backend"])
self.assertEqual(["password", "token"], credential_audit["details"]["deleted_secret_kinds"])
self.assertEqual(1, credential_audit["details"]["affected_profile_count"])
self.assertNotIn("do-not-audit", repr(audit.call_args_list))
changes = self.session.query(ChangeSequenceEntry).order_by(ChangeSequenceEntry.id.asc()).all()
self.assertEqual([credential.id, profile.id], [change.resource_id for change in changes])
self.assertEqual(["deleted", "updated"], [change.operation for change in changes])
def test_delete_profile_scrubs_inline_credentials_and_audits(self) -> None:
profile = self._profile(credential_profile_id="shared-credential")
self.session.add(profile)
self.session.commit()
with patch(
"govoplan_files.backend.storage.connector_credential_deletion.audit_event"
) as audit:
response = deactivate_connector_profile(
profile.id,
session=self.session,
principal=self.principal, # type: ignore[arg-type]
)
self.assertFalse(response.enabled)
self.assertFalse(response.credentials_configured)
self.assertIsNone(response.credential_profile_id)
self.session.refresh(profile)
self._assert_profile_scrubbed(profile)
audit.assert_called_once()
call = audit.call_args.kwargs
self.assertEqual("files.connector_profile_deleted", call["action"])
self.assertEqual("api_delete", call["details"]["deletion_reason"])
self.assertIn("credential_profile", call["details"]["removed_reference_kinds"])
self.assertNotIn("do-not-audit", repr(call))
def test_delete_rolls_back_when_audit_fails(self) -> None:
credential = self._credential()
self.session.add(credential)
self.session.commit()
original_password = credential.password_encrypted
original_token = credential.token_encrypted
with patch(
"govoplan_files.backend.storage.connector_credential_deletion.audit_event",
side_effect=RuntimeError("audit unavailable"),
), self.assertRaisesRegex(RuntimeError, "audit unavailable"):
deactivate_connector_credential(
credential.id,
session=self.session,
principal=self.principal, # type: ignore[arg-type]
)
persisted = self.session.get(FileConnectorCredential, credential.id)
assert persisted is not None
self.assertTrue(persisted.enabled)
self.assertEqual(original_password, persisted.password_encrypted)
self.assertEqual(original_token, persisted.token_encrypted)
self.assertEqual(0, self.session.query(ChangeSequenceEntry).count())
def test_delete_detaches_unowned_legacy_secret_reference_without_claiming_provider_deletion(self) -> None:
credential = self._credential(secret_ref="vault:tenant-1:files:credential")
self.session.add(credential)
self.session.commit()
with patch(
"govoplan_files.backend.storage.connector_credential_deletion.audit_event"
) as audit:
response = deactivate_connector_credential(
credential.id,
session=self.session,
principal=self.principal, # type: ignore[arg-type]
)
self.assertFalse(response.enabled)
persisted = self.session.get(FileConnectorCredential, credential.id)
assert persisted is not None
self._assert_credential_scrubbed(persisted)
audit.assert_called_once()
details = audit.call_args.kwargs["details"]
self.assertEqual("unowned_external_reference_detached", details["storage_backend"])
self.assertIn("unowned_external_secret_ref", details["removed_reference_kinds"])
self.assertNotIn("vault:tenant-1:files:credential", repr(audit.call_args))
def test_retirement_scrubs_and_audits_before_tables_are_dropped(self) -> None:
from govoplan_files.backend.manifest import manifest
self.session.add_all([self._credential(), self._profile()])
self.session.commit()
retirement_provider = manifest.migration_spec.retirement_provider
assert retirement_provider is not None
plan = retirement_provider(self.session, "files")
assert plan.destroy_data_executor is not None
with patch(
"govoplan_files.backend.storage.connector_credential_deletion.audit_event"
) as audit:
plan.destroy_data_executor(self.session, "files")
# The retirement executor deliberately enlists secret scrubbing, audit
# writes, and DDL in the caller-owned transaction. Commit that unit of
# work before inspecting through a separate Engine connection.
self.session.commit()
self.assertEqual(2, audit.call_count)
self.assertEqual(
{"module_data_retired"},
{call.kwargs["details"]["deletion_reason"] for call in audit.call_args_list},
)
self.assertNotIn("do-not-audit", repr(audit.call_args_list))
self.assertFalse(inspect(self.engine).has_table(FileConnectorCredential.__tablename__))
self.assertFalse(inspect(self.engine).has_table(FileConnectorProfile.__tablename__))
def test_retirement_detaches_unowned_external_reference_before_drop(self) -> None:
from govoplan_files.backend.manifest import manifest
self.session.add(self._credential(secret_ref="vault:tenant-1:files:credential"))
self.session.commit()
retirement_provider = manifest.migration_spec.retirement_provider
assert retirement_provider is not None
plan = retirement_provider(self.session, "files")
assert plan.destroy_data_executor is not None
with patch(
"govoplan_files.backend.storage.connector_credential_deletion.audit_event"
) as audit:
plan.destroy_data_executor(self.session, "files")
self.session.commit()
audit.assert_called_once()
details = audit.call_args.kwargs["details"]
self.assertIn("unowned_external_secret_ref", details["removed_reference_kinds"])
self.assertNotIn("vault:tenant-1:files:credential", repr(audit.call_args))
self.assertFalse(inspect(self.engine).has_table(FileConnectorCredential.__tablename__))
self.assertFalse(inspect(self.engine).has_table(FileConnectorProfile.__tablename__))
@staticmethod
def _credential(*, secret_ref: str | None = None) -> FileConnectorCredential:
return FileConnectorCredential(
id="credential-1",
tenant_id="tenant-1",
scope_type="tenant",
scope_id="tenant-1",
label="Credential",
provider="webdav",
enabled=True,
credential_mode="basic",
username="connector-user",
password_encrypted=encrypt_secret("do-not-audit-password"),
token_encrypted=encrypt_secret("do-not-audit-token"),
password_env="DEPLOYMENT_PASSWORD",
token_env="DEPLOYMENT_TOKEN",
secret_ref=secret_ref,
policy={},
metadata_={"private_hint": "do-not-audit-metadata"},
)
@staticmethod
def _profile(*, credential_profile_id: str | None = None) -> FileConnectorProfile:
return FileConnectorProfile(
id="profile-1",
tenant_id="tenant-1",
scope_type="tenant",
scope_id="tenant-1",
label="Profile",
provider="webdav",
endpoint_url="https://dav.example.test",
enabled=True,
credential_profile_id=credential_profile_id,
credential_mode="basic",
username="connector-user",
password_encrypted=encrypt_secret("do-not-audit-profile-password"),
token_encrypted=encrypt_secret("do-not-audit-profile-token"),
password_env="DEPLOYMENT_PROFILE_PASSWORD",
token_env="DEPLOYMENT_PROFILE_TOKEN",
policy={},
metadata_={"private_hint": "do-not-audit-profile-metadata"},
)
def _assert_credential_scrubbed(self, row: FileConnectorCredential) -> None:
self.assertFalse(row.enabled)
self.assertEqual("none", row.credential_mode)
self.assertIsNone(row.username)
self.assertIsNone(row.password_encrypted)
self.assertIsNone(row.token_encrypted)
self.assertIsNone(row.password_env)
self.assertIsNone(row.token_env)
self.assertIsNone(row.secret_ref)
self.assertEqual({}, row.metadata_)
def _assert_profile_scrubbed(self, row: FileConnectorProfile) -> None:
self.assertFalse(row.enabled)
self.assertIsNone(row.credential_profile_id)
self.assertEqual("none", row.credential_mode)
self.assertIsNone(row.username)
self.assertIsNone(row.password_encrypted)
self.assertIsNone(row.token_encrypted)
self.assertIsNone(row.password_env)
self.assertIsNone(row.token_env)
self.assertIsNone(row.secret_ref)
self.assertEqual({}, row.metadata_)
if __name__ == "__main__":
unittest.main()
+227
View File
@@ -0,0 +1,227 @@
from __future__ import annotations
import os
import tempfile
import unittest
from pathlib import Path
from types import SimpleNamespace
from unittest.mock import patch
from unittest.mock import MagicMock
from fastapi import HTTPException
from govoplan_files.backend.routes.connector_settings import discover_connector_endpoint
from govoplan_files.backend.schemas import FileConnectorDiscoveryRequest
from govoplan_files.backend.storage.connector_browse import ConnectorBrowseError, _profile_password, _s3_verify
from govoplan_files.backend.storage.connector_deployment import (
ConnectorDeploymentConfigurationError,
connector_effective_endpoint_url,
connector_secret_env_value,
reject_api_controlled_deployment_references,
)
from govoplan_files.backend.storage.connector_profiles import ConnectorProfile
from govoplan_files.backend.storage.connector_profile_store import create_connector_profile_row
class ConnectorDeploymentBoundaryTests(unittest.TestCase):
def test_database_profile_cannot_read_arbitrary_process_environment(self) -> None:
profile = ConnectorProfile(
id="tenant-webdav",
label="Tenant WebDAV",
provider="webdav",
password_env="MASTER_KEY_B64",
source_kind="database",
)
with patch.dict(
os.environ,
{
"MASTER_KEY_B64": "must-not-leave-process",
"GOVOPLAN_CONNECTOR_SECRET_ENV_ALLOWLIST": "MASTER_KEY_B64",
},
clear=False,
), self.assertRaisesRegex(ConnectorBrowseError, "deployment-owned"):
_profile_password(profile)
def test_deployment_profile_requires_exact_secret_allowlist(self) -> None:
with patch.dict(
os.environ,
{
"GOVOPLAN_FILES_WEBDAV_PASSWORD": "deployment-secret",
"GOVOPLAN_CONNECTOR_SECRET_ENV_ALLOWLIST": "GOVOPLAN_FILES_WEBDAV_PASSWORD",
},
clear=False,
):
self.assertEqual(
"deployment-secret",
connector_secret_env_value(
"GOVOPLAN_FILES_WEBDAV_PASSWORD",
source_kind="settings",
),
)
with self.assertRaisesRegex(ConnectorDeploymentConfigurationError, "not listed"):
connector_secret_env_value("MASTER_KEY_B64", source_kind="settings")
def test_api_profiles_cannot_select_secret_environment_names(self) -> None:
with self.assertRaisesRegex(ConnectorDeploymentConfigurationError, "deployment-owned"):
reject_api_controlled_deployment_references(password_env="DATABASE_URL")
with self.assertRaisesRegex(ConnectorDeploymentConfigurationError, "deployment-owned"):
reject_api_controlled_deployment_references(metadata={"secret_access_key_env": "MASTER_KEY_B64"})
session = MagicMock()
with self.assertRaisesRegex(ConnectorDeploymentConfigurationError, "deployment-owned"):
create_connector_profile_row(
session,
tenant_id="tenant-1",
user_id="user-1",
profile_id="unsafe",
label="Unsafe",
provider="webdav",
scope_type="tenant",
password_env="DATABASE_URL",
)
session.get.assert_not_called()
def test_api_profiles_cannot_create_unowned_external_secret_references(self) -> None:
with self.assertRaisesRegex(ConnectorDeploymentConfigurationError, "provider ownership"):
reject_api_controlled_deployment_references(secret_ref="vault:tenant-1:files")
session = MagicMock()
with self.assertRaisesRegex(ConnectorDeploymentConfigurationError, "provider ownership"):
create_connector_profile_row(
session,
tenant_id="tenant-1",
user_id="user-1",
profile_id="unsafe-secret-ref",
label="Unsafe secret reference",
provider="webdav",
scope_type="tenant",
credential_mode="secret_ref",
secret_ref="vault:tenant-1:files",
)
session.get.assert_not_called()
def test_api_profiles_cannot_hide_plaintext_credentials_in_metadata(self) -> None:
for metadata in (
{"secret_access_key": "plaintext-secret"},
{"credentials": {"password": "plaintext-secret"}},
{"provider": {"refresh_token": "plaintext-secret"}},
):
with self.subTest(metadata=metadata), self.assertRaisesRegex(
ConnectorDeploymentConfigurationError,
"dedicated encrypted credential fields",
):
reject_api_controlled_deployment_references(metadata=metadata)
def test_profile_responses_hide_environment_and_local_ca_references(self) -> None:
profile = ConnectorProfile(
id="deployment-s3",
label="Deployment S3",
provider="s3",
metadata={
"bucket": "documents",
"secret_access_key_env": "GOVOPLAN_FILES_S3_SECRET",
"ca_bundle": "/etc/govoplan/connector-ca.pem",
},
)
self.assertEqual({"bucket": "documents"}, profile.to_response()["metadata"])
def test_ca_bundle_must_be_an_exact_deployment_allowlisted_file(self) -> None:
with tempfile.TemporaryDirectory() as temp_dir:
allowed = Path(temp_dir, "connector-ca.pem")
other = Path(temp_dir, "other-ca.pem")
allowed.write_text("test CA", encoding="utf-8")
other.write_text("other CA", encoding="utf-8")
with patch.dict(
os.environ,
{
"APP_ENV": "production",
"GOVOPLAN_CONNECTOR_CA_BUNDLE_ALLOWLIST": str(allowed),
},
clear=False,
):
profile = ConnectorProfile(
id="s3",
label="S3",
provider="s3",
metadata={"ca_bundle": str(allowed)},
)
self.assertEqual(str(allowed.resolve()), _s3_verify(profile))
with self.assertRaisesRegex(ConnectorDeploymentConfigurationError, "not listed"):
reject_api_controlled_deployment_references(metadata={"ca_bundle": str(other)})
def test_tls_verification_can_only_be_disabled_in_development(self) -> None:
with patch.dict(os.environ, {"APP_ENV": "production"}, clear=False), self.assertRaisesRegex(
ConnectorDeploymentConfigurationError,
"dev/test",
):
reject_api_controlled_deployment_references(metadata={"verify_tls": False})
with patch.dict(os.environ, {"APP_ENV": "test"}, clear=False):
reject_api_controlled_deployment_references(metadata={"verify_tls": False})
def test_effective_endpoint_uses_webdav_override(self) -> None:
self.assertEqual(
"https://dav.example.test/root",
connector_effective_endpoint_url(
provider="seafile",
endpoint_url="https://seafile.example.test",
metadata={"webdav_endpoint_url": "https://dav.example.test/root"},
),
)
class ConnectorDiscoveryBoundaryTests(unittest.TestCase):
@staticmethod
def _principal() -> SimpleNamespace:
return SimpleNamespace(tenant_id="tenant-1", user=SimpleNamespace(id="user-1"), api_key=None)
def test_discovery_rejects_environment_credentials_before_io(self) -> None:
payload = FileConnectorDiscoveryRequest(
provider="webdav",
endpoint_url="https://dav.example.test",
credential_mode="basic",
credentials={"username": "admin", "password_env": "MASTER_KEY_B64"},
)
with patch("govoplan_files.backend.routes.connector_settings.browse_connector_profile") as browse, self.assertRaises(
HTTPException
) as raised:
discover_connector_endpoint(payload, session=object(), principal=self._principal()) # type: ignore[arg-type]
self.assertEqual(400, raised.exception.status_code)
browse.assert_not_called()
def test_discovery_applies_policy_and_audit_before_each_io_candidate(self) -> None:
payload = FileConnectorDiscoveryRequest(
provider="webdav",
endpoint_url="https://dav.example.test/root",
metadata={
"webdav_endpoint_url": "https://bypass.example.test",
"static_listing": {"": []},
},
)
events: list[str] = []
def ensure(*_args: object, **kwargs: object) -> None:
self.assertEqual("https://dav.example.test/root/", kwargs["endpoint_url"])
events.append("policy")
def audit(*_args: object, **_kwargs: object) -> None:
events.append("audit")
def browse(profile: ConnectorProfile, **_kwargs: object) -> list[object]:
self.assertEqual({}, profile.metadata)
events.append("io")
return []
with patch("govoplan_files.backend.routes.connector_settings._ensure_connector_configuration_allowed", side_effect=ensure), patch(
"govoplan_files.backend.routes.connector_settings._audit_connector_discovery_attempt",
side_effect=audit,
), patch("govoplan_files.backend.routes.connector_settings.browse_connector_profile", side_effect=browse):
response = discover_connector_endpoint(payload, session=object(), principal=self._principal()) # type: ignore[arg-type]
self.assertEqual("usable", response.status)
self.assertEqual(["policy", "audit", "io"], events)
self.assertEqual({"discovered_by": "webdav-propfind"}, response.metadata)
if __name__ == "__main__":
unittest.main()
+66
View File
@@ -0,0 +1,66 @@
from __future__ import annotations
import unittest
from govoplan_files.backend.storage.connector_policy import (
ConnectorAccessRequest,
ConnectorPolicySource,
connector_policy_decision,
)
class ConnectorPolicyPatternTests(unittest.TestCase):
def test_resource_reference_allow_patterns_match(self) -> None:
decision = connector_policy_decision(
ConnectorAccessRequest(
connector_id="dev-smb",
credential_id="credential-primary",
provider="smb",
),
(
ConnectorPolicySource(
scope_type="tenant",
scope_id="tenant-1",
label="Tenant",
policy={
"allow": {
"connectors": ["dev-*"],
"credentials": ["credential-*"],
"providers": ["s?b"],
}
},
),
),
)
self.assertTrue(decision.allowed)
def test_resource_reference_deny_patterns_take_precedence(self) -> None:
decision = connector_policy_decision(
ConnectorAccessRequest(
connector_id="dev-smb",
credential_id="credential-legacy",
provider="smb",
),
(
ConnectorPolicySource(
scope_type="tenant",
scope_id="tenant-1",
label="Tenant",
policy={
"allow": {"connectors": ["dev-*"]},
"deny": {"credentials": ["*-legacy"]},
},
),
),
)
self.assertFalse(decision.allowed)
self.assertEqual(
("connector_policy_denylist",),
decision.requirements,
)
if __name__ == "__main__":
unittest.main()
+228
View File
@@ -0,0 +1,228 @@
from __future__ import annotations
import json
import unittest
from unittest.mock import patch
from sqlalchemy import create_engine
from sqlalchemy.orm import sessionmaker
from govoplan_access.backend.db import models as access_models # noqa: F401 - resolve Files user foreign keys
from govoplan_core.db.base import Base
from govoplan_core.security.credential_envelopes import CredentialEnvelope
from govoplan_core.security.secrets import decrypt_secret, encrypt_secret
from govoplan_files.backend.db.models import FileConnectorProfile
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.connector_profile_store import (
list_database_connector_profiles,
update_connector_profile_row,
)
class ConnectorProfileStoreUpdateTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite:///:memory:")
Base.metadata.create_all(
self.engine,
tables=[
FileConnectorProfile.__table__,
CredentialEnvelope.__table__,
],
)
self.session = sessionmaker(bind=self.engine)()
self.row = FileConnectorProfile(
id="profile-1",
tenant_id="tenant-1",
scope_type="tenant",
scope_id="tenant-1",
label="Original profile",
provider="webdav",
endpoint_url="https://dav.example.test/original",
base_path="original",
enabled=True,
credential_mode="basic",
username="original-user",
password_encrypted=encrypt_secret("original-password"),
token_encrypted=encrypt_secret("original-token"),
capabilities=["browse"],
policy={"allow": {"providers": ["webdav"]}},
metadata_={"original": True},
created_by_user_id=None,
updated_by_user_id=None,
)
self.session.add(self.row)
self.session.commit()
def tearDown(self) -> None:
self.session.close()
Base.metadata.drop_all(bind=self.engine)
self.engine.dispose()
def test_update_preserves_normalization_and_flush_only_transaction_boundary(self) -> None:
updated = update_connector_profile_row(
self.session,
self.row,
user_id="user-2",
label=" Updated profile ",
provider="NEXTCLOUD",
endpoint_url=" https://cloud.example.test/dav ",
base_path=" shared/reports ",
enabled=False,
credential_profile_id=" credential-2 ",
credential_mode="TOKEN",
username=" updated-user ",
password="updated-password",
token="updated-token",
capabilities=["browse", " import ", ""],
policy={"deny": {"external_paths": ["private"]}},
metadata={"department": "reports"},
)
self.assertIs(self.row, updated)
self.assertEqual("Updated profile", updated.label)
self.assertEqual("nextcloud", updated.provider)
self.assertEqual("https://cloud.example.test/dav", updated.endpoint_url)
self.assertEqual("shared/reports", updated.base_path)
self.assertFalse(updated.enabled)
self.assertEqual("credential-2", updated.credential_profile_id)
self.assertEqual("token", updated.credential_mode)
self.assertEqual("updated-user", updated.username)
self.assertEqual("updated-password", decrypt_secret(updated.password_encrypted))
self.assertEqual("updated-token", decrypt_secret(updated.token_encrypted))
self.assertEqual(["browse", "import"], updated.capabilities)
self.assertEqual({"deny": {"external_paths": ["private"]}}, updated.policy)
self.assertEqual({"department": "reports"}, updated.metadata_)
self.assertEqual("user-2", updated.updated_by_user_id)
self.session.rollback()
persisted = self.session.get(FileConnectorProfile, self.row.id)
assert persisted is not None
self.assertEqual("Original profile", persisted.label)
self.assertEqual("original-password", decrypt_secret(persisted.password_encrypted))
self.assertTrue(persisted.enabled)
def test_secret_replacement_wins_over_clear_while_clear_removes_an_omitted_token(self) -> None:
update_connector_profile_row(
self.session,
self.row,
user_id="user-2",
password="replacement-password",
clear_password=True,
clear_token=True,
)
self.assertEqual("replacement-password", decrypt_secret(self.row.password_encrypted))
self.assertIsNone(self.row.token_encrypted)
self.assertEqual("https://dav.example.test/original", self.row.endpoint_url)
self.assertEqual("original-user", self.row.username)
def test_legacy_external_secret_reference_cannot_be_cleared_or_partially_mutate_the_row(self) -> None:
self.row.secret_ref = "vault:tenant-1:files:profile"
self.session.commit()
with self.assertRaisesRegex(FileStorageError, "provider-side deletion"):
update_connector_profile_row(
self.session,
self.row,
user_id="user-2",
label="Must not be applied",
secret_ref="",
)
self.assertEqual("Original profile", self.row.label)
self.assertEqual("vault:tenant-1:files:profile", self.row.secret_ref)
def test_flush_failure_remains_rollback_safe(self) -> None:
with patch.object(self.session, "flush", side_effect=RuntimeError("database unavailable")), self.assertRaisesRegex(
RuntimeError,
"database unavailable",
):
update_connector_profile_row(
self.session,
self.row,
user_id="user-2",
label="Uncommitted profile",
password="uncommitted-password",
)
self.session.rollback()
persisted = self.session.get(FileConnectorProfile, self.row.id)
assert persisted is not None
self.assertEqual("Original profile", persisted.label)
self.assertEqual("original-password", decrypt_secret(persisted.password_encrypted))
def test_visibility_filter_runs_before_connector_secrets_are_decrypted(self) -> None:
invisible = FileConnectorProfile(
id="invisible-profile",
tenant_id="tenant-1",
scope_type="user",
scope_id="another-user",
label="Invisible profile",
provider="webdav",
endpoint_url="https://invisible.example.test",
enabled=True,
credential_mode="basic",
password_encrypted="not-a-valid-encrypted-secret",
capabilities=["browse"],
policy={},
metadata_={},
)
self.session.add(invisible)
self.session.commit()
profiles = list_database_connector_profiles(
self.session,
tenant_id="tenant-1",
row_visible=lambda row: row.id == self.row.id,
)
self.assertEqual([self.row.id], [profile.id for profile in profiles])
self.assertEqual("original-password", profiles[0].password_value)
def test_reusable_credential_resolves_and_revocation_keeps_profile_visible(self) -> None:
credential = CredentialEnvelope(
id="shared-credential",
tenant_id="tenant-1",
scope_type="tenant",
scope_id="tenant-1",
name="Shared WebDAV login",
credential_kind="username_password",
public_data={"username": "ada"},
secret_data_encrypted=encrypt_secret(json.dumps({"password": "secret"})),
secret_keys=["password"],
allowed_modules=["files"],
allowed_server_refs=[],
inherit_to_lower_scopes=True,
is_active=True,
revision="revision-1",
)
self.row.credential_profile_id = "credential-envelope:shared-credential"
self.row.username = None
self.row.password_encrypted = None
self.row.token_encrypted = None
self.session.add(credential)
self.session.commit()
profile = list_database_connector_profiles(
self.session,
tenant_id="tenant-1",
)[0]
self.assertEqual("ada", profile.username)
self.assertEqual("secret", profile.password_value)
self.assertTrue(profile.credentials_configured)
credential.is_active = False
self.session.commit()
profile = list_database_connector_profiles(
self.session,
tenant_id="tenant-1",
)[0]
self.assertIsNone(profile.password_value)
self.assertFalse(profile.credentials_configured)
self.assertTrue(profile.metadata["credential_unavailable"])
if __name__ == "__main__":
unittest.main()
+233
View File
@@ -0,0 +1,233 @@
from __future__ import annotations
import unittest
from datetime import UTC, datetime
from unittest.mock import patch
from govoplan_files.backend.storage.connector_browse import ConnectorBrowseError, _smb_location, browse_connector_profile
from govoplan_files.backend.storage.connector_imports import read_connector_file
from govoplan_files.backend.storage.connector_profiles import ConnectorProfile, connector_profiles_from_payload
from govoplan_files.backend.storage.connector_providers import connector_provider_descriptors
class FakeBody:
def __init__(self, data: bytes) -> None:
self.data = data
def read(self, _limit: int) -> bytes:
return self.data
class FakeS3Client:
def __init__(self) -> None:
self.list_objects_request: dict[str, object] | None = None
self.head_request: dict[str, object] | None = None
self.get_request: dict[str, object] | None = None
self.closed = False
def close(self) -> None:
self.closed = True
def list_objects_v2(self, **kwargs: object) -> dict[str, object]:
self.list_objects_request = dict(kwargs)
return {
"CommonPrefixes": [{"Prefix": "root/reports/"}],
"Contents": [
{
"Key": "root/report.xlsx",
"Size": 123,
"LastModified": datetime(2026, 7, 12, 10, 0, tzinfo=UTC),
"ETag": '"etag-1"',
"StorageClass": "STANDARD",
}
],
"NextContinuationToken": "next-page",
}
def list_buckets(self) -> dict[str, object]:
return {"Buckets": [{"Name": "archive", "CreationDate": datetime(2026, 7, 12, 9, 0, tzinfo=UTC)}]}
def head_object(self, **kwargs: object) -> dict[str, object]:
self.head_request = dict(kwargs)
return {
"ContentLength": 13,
"ContentType": "text/plain",
"ETag": '"etag-2"',
"VersionId": "version-1",
"ChecksumSHA256": "checksum",
}
def get_object(self, **kwargs: object) -> dict[str, object]:
self.get_request = dict(kwargs)
return {"Body": FakeBody(b"hello s3\n"), "ContentType": "text/plain", "ETag": '"etag-2"'}
def s3_profile(**overrides: object) -> ConnectorProfile:
values = {
"id": "dev-s3",
"label": "Dev S3",
"provider": "s3",
"endpoint_url": "http://127.0.0.1:9000",
"base_path": "root",
"credential_mode": "basic",
"username": "access-key",
"password_value": "secret-key",
"metadata": {"bucket": "govoplan", "region": "eu-central-1", "path_style": True},
}
values.update(overrides)
return ConnectorProfile(**values)
class ConnectorProviderTests(unittest.TestCase):
def test_smb_browse_uses_the_files_owned_pinned_connection_cache(self) -> None:
profile = ConnectorProfile(
id="smb",
label="SMB",
provider="smb",
endpoint_url="smb://files.example.test/share",
)
with patch.dict(
"os.environ",
{"APP_ENV": "production", "GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS": "false"},
), patch(
"govoplan_core.security.outbound_http.socket.getaddrinfo",
return_value=[(2, 1, 6, "", ("93.184.216.34", 445))],
), patch("govoplan_files.backend.storage.connector_browse._smbclient_module") as sdk:
sdk.return_value.scandir.return_value.__enter__.return_value = iter(())
self.assertEqual([], browse_connector_profile(profile, path=""))
kwargs = sdk.return_value.scandir.call_args.kwargs
self.assertIsInstance(kwargs["connection_cache"], dict)
self.assertTrue(kwargs["require_signing"])
def test_smb_endpoint_preflight_applies_private_network_policy(self) -> None:
profile = ConnectorProfile(id="smb", label="SMB", provider="smb", endpoint_url="smb://10.0.0.5/share")
with patch.dict(
"os.environ",
{"APP_ENV": "production", "GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS": "false"},
), patch(
"govoplan_core.security.outbound_http.socket.getaddrinfo",
return_value=[(2, 1, 6, "", ("10.0.0.5", 445))],
), self.assertRaisesRegex(ConnectorBrowseError, "non-public network"):
_smb_location(profile)
def test_smb_import_uses_the_same_pinned_connection_cache(self) -> None:
profile = ConnectorProfile(id="smb", label="SMB", provider="smb", endpoint_url="smb://10.0.0.5/share")
with patch.dict(
"os.environ",
{"APP_ENV": "production", "GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS": "true"},
), patch(
"govoplan_core.security.outbound_http.socket.getaddrinfo",
return_value=[(2, 1, 6, "", ("10.0.0.5", 445))],
), patch("govoplan_files.backend.storage.connector_imports._smbclient_module") as sdk:
sdk.return_value.stat.return_value.st_size = 4
sdk.return_value.open_file.return_value.__enter__.return_value.read.return_value = b"test"
downloaded = read_connector_file(profile, library_id="", path="notice.txt", max_bytes=1024)
self.assertEqual(b"test", downloaded.data)
self.assertIsInstance(sdk.return_value.stat.call_args.kwargs["connection_cache"], dict)
def test_provider_descriptors_include_s3_and_reserved_microsoft_providers(self) -> None:
descriptors = {descriptor.provider: descriptor for descriptor in connector_provider_descriptors()}
self.assertTrue(descriptors["s3"].implemented)
self.assertTrue(descriptors["s3"].browse_supported)
self.assertEqual("boto3", descriptors["s3"].optional_dependency)
self.assertFalse(descriptors["sharepoint"].implemented)
self.assertFalse(descriptors["onedrive"].implemented)
def test_profile_payload_accepts_new_providers_and_redacts_secret_metadata(self) -> None:
profiles = connector_profiles_from_payload(
{
"profiles": [
{
"id": "dev-s3",
"label": "Dev S3",
"provider": "s3",
"username": "access-key",
"password": "secret-key",
"metadata": {"bucket": "govoplan", "secret_access_key": "do-not-return"},
},
{"id": "sharepoint", "provider": "sharepoint"},
{"id": "onedrive", "provider": "onedrive"},
]
}
)
by_id = {profile.id: profile for profile in profiles}
self.assertEqual("s3", by_id["dev-s3"].provider)
self.assertTrue(by_id["dev-s3"].credentials_configured)
self.assertEqual({"bucket": "govoplan"}, dict(by_id["dev-s3"].metadata))
self.assertEqual("sharepoint", by_id["sharepoint"].provider)
self.assertEqual("onedrive", by_id["onedrive"].provider)
def test_s3_browse_lists_prefixes_and_objects_with_provenance_metadata(self) -> None:
client = FakeS3Client()
with patch("govoplan_files.backend.storage.connector_browse._s3_client", return_value=client):
items = browse_connector_profile(s3_profile(), path="")
self.assertEqual({"Bucket": "govoplan", "Prefix": "root/", "Delimiter": "/", "MaxKeys": 1000}, client.list_objects_request)
self.assertEqual(["folder", "file"], [item.kind for item in items])
self.assertEqual("reports", items[0].path)
self.assertEqual("report.xlsx", items[1].path)
self.assertEqual("govoplan:root/report.xlsx", items[1].external_id)
self.assertEqual("root/report.xlsx", items[1].metadata["key"])
self.assertEqual("next-page", items[1].metadata["next_continuation_token"])
self.assertTrue(client.closed)
def test_s3_browse_lists_buckets_when_profile_has_no_bucket(self) -> None:
client = FakeS3Client()
with patch("govoplan_files.backend.storage.connector_browse._s3_client", return_value=client):
items = browse_connector_profile(s3_profile(metadata={}), path="")
self.assertEqual(["archive"], [item.path for item in items])
def test_s3_client_is_constructed_through_the_pinned_transport(self) -> None:
with patch.dict(
"os.environ",
{"APP_ENV": "production", "GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS": "true"},
), patch(
"govoplan_core.security.outbound_http.socket.getaddrinfo",
return_value=[(2, 1, 6, "", ("127.0.0.1", 9000))],
), patch(
"govoplan_files.backend.storage.connector_browse.create_pinned_s3_client",
return_value=FakeS3Client(),
) as factory:
browse_connector_profile(s3_profile(), path="")
kwargs = factory.call_args.kwargs
self.assertEqual("http://127.0.0.1:9000", kwargs["endpoint_url"])
self.assertEqual("access-key", kwargs["aws_access_key_id"])
self.assertEqual("secret-key", kwargs["aws_secret_access_key"])
self.assertEqual({}, kwargs["config"].proxies)
def test_s3_endpoint_discovery_uses_the_same_pinned_transport(self) -> None:
with patch(
"govoplan_files.backend.storage.connector_browse.create_pinned_s3_client",
return_value=FakeS3Client(),
) as factory:
browse_connector_profile(s3_profile(endpoint_url=None), path="")
self.assertNotIn("endpoint_url", factory.call_args.kwargs)
def test_s3_import_downloads_object_and_preserves_remote_identity(self) -> None:
client = FakeS3Client()
with patch("govoplan_files.backend.storage.connector_imports._s3_client", return_value=client):
downloaded = read_connector_file(s3_profile(), library_id="", path="report.txt", max_bytes=1024)
self.assertEqual({"Bucket": "govoplan", "Key": "root/report.txt"}, client.head_request)
self.assertEqual({"Bucket": "govoplan", "Key": "root/report.txt", "VersionId": "version-1"}, client.get_request)
self.assertEqual("report.txt", downloaded.filename)
self.assertEqual(b"hello s3\n", downloaded.data)
self.assertEqual("version-1", downloaded.revision)
self.assertEqual("govoplan:root/report.txt", downloaded.external_id)
self.assertEqual("s3://govoplan/root/report.txt", downloaded.external_url)
self.assertEqual("checksum", downloaded.metadata["checksum_sha256"])
self.assertTrue(client.closed)
if __name__ == "__main__":
unittest.main()
+272
View File
@@ -0,0 +1,272 @@
from __future__ import annotations
import unittest
from dataclasses import replace
from unittest.mock import patch
from govoplan_files.backend.storage.connector_policy import ConnectorPolicySource
from govoplan_files.backend.storage.connector_profiles import ConnectorProfile
from govoplan_files.backend.storage.connector_visibility import (
connector_profile_usable_for_import,
visible_connector_profiles_for_actor,
)
from govoplan_files.backend.storage.connector_providers import (
connector_provider_descriptors,
)
def _profile(
profile_id: str,
*,
scope_type: str = "system",
scope_id: str | None = None,
provider: str = "webdav",
enabled: bool = True,
) -> ConnectorProfile:
return ConnectorProfile(
id=profile_id,
label=profile_id,
provider=provider,
scope_type=scope_type,
scope_id=scope_id,
endpoint_url=f"https://{profile_id}.example.invalid",
enabled=enabled,
credential_mode="anonymous",
)
class ConnectorVisibilityTests(unittest.TestCase):
@patch(
"govoplan_files.backend.storage.connector_visibility.effective_connector_policy_sources",
return_value=[],
)
@patch(
"govoplan_files.backend.storage.connector_visibility.connector_profiles_from_settings"
)
@patch(
"govoplan_files.backend.storage.connector_visibility.select_database_connector_profiles"
)
def test_profiles_are_filtered_to_actor_scopes_and_database_definition_wins(
self,
database_profiles,
configured_profiles,
_policy_sources,
) -> None:
profiles = [
_profile("system"),
_profile("tenant", scope_type="tenant", scope_id="tenant-1"),
_profile("other-tenant", scope_type="tenant", scope_id="tenant-2"),
_profile("user", scope_type="user", scope_id="user-1"),
_profile("other-user", scope_type="user", scope_id="user-2"),
_profile("group", scope_type="group", scope_id="group-1"),
_profile("campaign", scope_type="campaign", scope_id="campaign-1"),
_profile("disabled", enabled=False),
_profile("duplicate", provider="webdav"),
]
database_profiles.return_value = (profiles, {profile.id for profile in profiles})
configured_profiles.return_value = [_profile("duplicate", provider="nextcloud")]
visible = visible_connector_profiles_for_actor(
object(), # type: ignore[arg-type]
tenant_id="tenant-1",
user_id="user-1",
group_ids={"group-1"},
settings=object(),
campaign_visible=lambda campaign_id: campaign_id == "campaign-1",
)
self.assertEqual(
{"system", "tenant", "user", "group", "campaign", "duplicate"},
{profile.id for profile in visible},
)
duplicate = next(profile for profile in visible if profile.id == "duplicate")
self.assertEqual("webdav", duplicate.provider)
@patch(
"govoplan_files.backend.storage.connector_visibility.effective_connector_policy_sources",
return_value=[],
)
@patch(
"govoplan_files.backend.storage.connector_visibility.connector_profiles_from_settings",
return_value=[],
)
@patch(
"govoplan_files.backend.storage.connector_visibility.select_database_connector_profiles"
)
def test_campaign_and_admin_visibility_remain_explicit(
self,
database_profiles,
_configured_profiles,
_policy_sources,
) -> None:
profiles = [
_profile("campaign-a", scope_type="campaign", scope_id="campaign-a"),
_profile("campaign-b", scope_type="campaign", scope_id="campaign-b"),
_profile("other-user", scope_type="user", scope_id="user-2"),
]
database_profiles.return_value = (profiles, {profile.id for profile in profiles})
campaign_only = visible_connector_profiles_for_actor(
object(), # type: ignore[arg-type]
tenant_id="tenant-1",
user_id="user-1",
group_ids=(),
settings=None,
campaign_id="campaign-a",
campaign_visible=lambda _campaign_id: True,
)
self.assertEqual(["campaign-a"], [profile.id for profile in campaign_only])
admin = visible_connector_profiles_for_actor(
object(), # type: ignore[arg-type]
tenant_id="tenant-1",
user_id="user-1",
group_ids=(),
settings=None,
campaign_visible=None,
include_admin_scopes=True,
)
self.assertEqual(
{"campaign-a", "campaign-b", "other-user"},
{profile.id for profile in admin},
)
@patch(
"govoplan_files.backend.storage.connector_visibility.connector_profiles_from_settings",
return_value=[],
)
@patch(
"govoplan_files.backend.storage.connector_visibility.select_database_connector_profiles"
)
@patch(
"govoplan_files.backend.storage.connector_visibility.effective_connector_policy_sources"
)
def test_effective_policy_is_attached_without_mutating_profile(
self,
policy_sources,
database_profiles,
_configured_profiles,
) -> None:
profile = _profile("profile")
source = ConnectorPolicySource(
scope_type="system",
label="System",
policy={"allow": {"providers": ["webdav"]}},
)
database_profiles.return_value = ([profile], {profile.id})
policy_sources.return_value = [source]
visible = visible_connector_profiles_for_actor(
object(), # type: ignore[arg-type]
tenant_id="tenant-1",
user_id="user-1",
group_ids=(),
settings=None,
)
self.assertEqual((), profile.policy_sources)
self.assertEqual((source,), visible[0].policy_sources)
def test_import_usability_requires_safe_provider_credentials_endpoint_and_identity_policy(
self,
) -> None:
self.assertTrue(connector_profile_usable_for_import(_profile("webdav")))
descriptors = {
item.provider: item for item in connector_provider_descriptors()
}
for provider in ("s3", "smb"):
self.assertEqual(
descriptors[provider].installed,
connector_profile_usable_for_import(
_profile(provider, provider=provider)
),
)
self.assertFalse(
connector_profile_usable_for_import(
_profile("sharepoint", provider="sharepoint")
)
)
self.assertFalse(
connector_profile_usable_for_import(
ConnectorProfile(
id="no-endpoint",
label="No endpoint",
provider="webdav",
credential_mode="anonymous",
)
)
)
self.assertFalse(
connector_profile_usable_for_import(
ConnectorProfile(
id="no-credentials",
label="No credentials",
provider="webdav",
endpoint_url="https://files.example.invalid",
credential_mode="basic",
)
)
)
self.assertTrue(
connector_profile_usable_for_import(
ConnectorProfile(
id="metadata-endpoint",
label="Metadata endpoint",
provider="webdav",
credential_mode="anonymous",
metadata={"webdav_endpoint_url": "https://files.example.invalid"},
)
)
)
self.assertFalse(
connector_profile_usable_for_import(
ConnectorProfile(
id="unresolved-secret-reference",
label="Unresolved secret reference",
provider="webdav",
endpoint_url="https://files.example.invalid",
credential_mode="basic",
secret_ref="vault://files/webdav",
)
)
)
denied = replace(
_profile("denied"),
policy_sources=(
ConnectorPolicySource(
scope_type="system",
label="System",
policy={"deny": {"providers": ["webdav"]}},
),
),
)
self.assertFalse(connector_profile_usable_for_import(denied))
path_limited = replace(
_profile("path-limited"),
policy_sources=(
ConnectorPolicySource(
scope_type="system",
label="System",
policy={"allow": {"external_paths": ["/approved/*"]}},
),
),
)
self.assertFalse(connector_profile_usable_for_import(path_limited))
endpoint_allowed = replace(
_profile("endpoint-allowed"),
policy_sources=(
ConnectorPolicySource(
scope_type="system",
label="System",
policy={"allow": {"external_urls": ["https://*.example.invalid"]}},
),
),
)
self.assertTrue(connector_profile_usable_for_import(endpoint_allowed))
if __name__ == "__main__":
unittest.main()
+149
View File
@@ -0,0 +1,149 @@
from __future__ import annotations
from datetime import UTC, datetime
import hashlib
import unittest
from sqlalchemy import create_engine
from sqlalchemy.orm import Session
from govoplan_core.core.encryption import (
CAPABILITY_ENCRYPTION_CONTENT_CIPHER,
ContentProtectionEnvelope,
ProtectedContent,
)
from govoplan_files.backend.db.models import FileBlob
from govoplan_files.backend.runtime import configure_runtime
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.content_protection import protect_blob_content
from govoplan_files.backend.storage.integrity import read_verified_blob_bytes
class Registry:
def __init__(self, capability=None) -> None:
self.capability_value = capability
def has_capability(self, name: str) -> bool:
return (
name == CAPABILITY_ENCRYPTION_CONTENT_CIPHER
and self.capability_value is not None
)
def capability(self, _name: str):
return self.capability_value
class Cipher:
def protect_content(self, _session, *, request):
ciphertext = b"protected:" + request.plaintext
envelope = ContentProtectionEnvelope(
envelope_id="envelope-file-1",
tenant_id=request.tenant_id,
owner_module=request.owner_module,
resource_type=request.resource_type,
resource_id=request.resource_id,
profile_kind="server_envelope",
profile_id=request.profile_id,
provider_id="test",
vault_id=request.vault_id,
key_version=1,
algorithm_suite="AES-256-GCM",
ciphertext_ref=request.ciphertext_ref,
ciphertext_digest=f"sha256:{hashlib.sha256(ciphertext).hexdigest()}",
authenticated_context_digest="sha256:context",
state="active",
created_at=datetime.now(tz=UTC),
wrapped_key_refs=("wrapped:1",),
)
return ProtectedContent(envelope=envelope, ciphertext=ciphertext)
def unprotect_content(self, _session, *, request):
if not request.ciphertext.startswith(b"protected:"):
raise ValueError("invalid fixture ciphertext")
return request.ciphertext.removeprefix(b"protected:")
def execute_rewrap(self, *_args, **_kwargs):
raise NotImplementedError
def prepare_reencryption(self, *_args, **_kwargs):
raise NotImplementedError
class Backend:
def __init__(self, data: bytes) -> None:
self.data = data
def get_bytes(self, _key: str) -> bytes:
return self.data
class FileContentProtectionTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite+pysqlite:///:memory:")
FileBlob.__table__.create(self.engine)
self.session = Session(self.engine)
def tearDown(self) -> None:
self.session.close()
self.engine.dispose()
def test_encrypted_blob_is_opened_after_stored_object_integrity_check(self) -> None:
configure_runtime(registry=Registry(Cipher()), settings=object())
plaintext = b"monthly records"
protected = protect_blob_content(
self.session,
tenant_id="tenant-1",
blob_id="blob-1",
vault_id="vault-1",
ciphertext_ref="tenants/tenant-1/files/blob-1",
plaintext=plaintext,
actor_id="account-1",
content_type="text/plain",
)
blob = FileBlob(
id="blob-1",
tenant_id="tenant-1",
storage_backend="test",
storage_key="tenants/tenant-1/files/blob-1",
checksum_sha256=hashlib.sha256(plaintext).hexdigest(),
size_bytes=len(plaintext),
protection_discriminator="vault:vault-1",
encryption_envelope_id=protected.envelope.envelope_id,
storage_checksum_sha256=hashlib.sha256(protected.ciphertext).hexdigest(),
storage_size_bytes=len(protected.ciphertext),
content_type="text/plain",
ref_count=1,
integrity_status="verified",
)
self.session.add(blob)
self.session.flush()
self.assertEqual(
plaintext,
read_verified_blob_bytes(blob, backend=Backend(protected.ciphertext)),
)
def test_encrypted_blob_fails_closed_without_encryption_capability(self) -> None:
configure_runtime(registry=Registry(), settings=object())
ciphertext = b"protected:monthly records"
blob = FileBlob(
id="blob-1",
tenant_id="tenant-1",
storage_backend="test",
storage_key="tenants/tenant-1/files/blob-1",
checksum_sha256=hashlib.sha256(b"monthly records").hexdigest(),
size_bytes=len(b"monthly records"),
protection_discriminator="vault:vault-1",
encryption_envelope_id="envelope-file-1",
storage_checksum_sha256=hashlib.sha256(ciphertext).hexdigest(),
storage_size_bytes=len(ciphertext),
ref_count=1,
integrity_status="verified",
)
self.session.add(blob)
self.session.flush()
with self.assertRaisesRegex(FileStorageError, "Encryption is unavailable"):
read_verified_blob_bytes(blob, backend=Backend(ciphertext))
if __name__ == "__main__":
unittest.main()
+250
View File
@@ -0,0 +1,250 @@
from __future__ import annotations
import unittest
from types import SimpleNamespace
from unittest.mock import patch
from sqlalchemy.orm import Session
from govoplan_core.core.modules import DocumentationContext
from govoplan_files.backend.documentation import documentation_topics
from govoplan_files.backend.storage.connector_profiles import ConnectorProfile
class _Principal:
tenant_id = "tenant-1"
user = SimpleNamespace(id="user-1")
group_ids = frozenset({"group-1"})
def __init__(self, scopes: set[str]) -> None:
self.scopes = frozenset(scopes)
def has(self, scope: str) -> bool:
return scope in self.scopes
class FilesRuntimeDocumentationTests(unittest.TestCase):
def setUp(self) -> None:
self.session = Session()
def tearDown(self) -> None:
self.session.close()
def context(
self,
scopes: set[str],
*,
settings: object | None = None,
documentation_type: str = "user",
) -> DocumentationContext:
return DocumentationContext(
registry=object(),
principal=_Principal(scopes),
settings=settings,
session=self.session,
documentation_type=documentation_type, # type: ignore[arg-type]
)
def test_upload_tasks_state_exact_current_safe_limits(self) -> None:
settings = SimpleNamespace(
file_upload_max_bytes=7 * 1024 * 1024,
file_upload_zip_max_bytes=19 * 1024 * 1024,
file_archive_max_expanded_bytes=31 * 1024 * 1024,
file_archive_max_entries=321,
file_archive_max_expansion_ratio=42,
file_archive_preview_ttl_seconds=15 * 60,
)
with patch(
"govoplan_files.backend.documentation.visible_connector_profiles_for_actor",
return_value=[],
):
topics = {
topic.id: topic
for topic in documentation_topics(
self.context(
{"files:file:read", "files:file:upload"}, settings=settings
)
)
}
upload = topics["files.workflow.upload-managed-files"]
self.assertIn("7 MiB (7,340,032 bytes)", upload.body)
self.assertEqual(["files.list"], upload.metadata["help_contexts"])
self.assertEqual(
{"files:file:read", "files:file:upload"},
set(upload.conditions[0].required_scopes),
)
archive = topics["files.workflow.upload-and-unpack-zip"]
self.assertIn("19 MiB (19,922,944 bytes)", archive.body)
self.assertIn("31 MiB (32,505,856 bytes)", archive.body)
self.assertIn("7 MiB (7,340,032 bytes)", archive.body)
self.assertIn("321", archive.body)
self.assertIn("42:1", archive.body)
self.assertIn("15 minutes", archive.body)
self.assertIn("passwords remain request-only", archive.body)
self.assertIn("Actual extracted bytes are counted", archive.body)
def test_connector_task_requires_authority_and_a_visible_usable_profile(
self,
) -> None:
usable = ConnectorProfile(
id="private-profile-id",
label="Private profile label",
provider="webdav",
scope_type="tenant",
scope_id="tenant-1",
endpoint_url="https://private.example.invalid/secret-root",
base_path="/secret-root",
credential_mode="anonymous",
)
context = self.context({"files:file:read", "files:file:upload"})
with patch(
"govoplan_files.backend.documentation.visible_connector_profiles_for_actor",
return_value=[usable],
) as visible:
topics = {topic.id: topic for topic in documentation_topics(context)}
self.assertIn("files.workflow.import-managed-snapshot", topics)
self.assertNotIn("files.connector-import-unavailable", topics)
task = topics["files.workflow.import-managed-snapshot"]
self.assertEqual("configured", task.layer)
self.assertEqual(
["files.list", "files.connector-import"], task.metadata["help_contexts"]
)
self.assertIn("re-authorized", task.body)
self.assertNotIn("private-profile-id", repr(task))
self.assertNotIn("private.example.invalid", repr(task))
self.assertNotIn("secret-root", repr(task))
visible.assert_called_once()
self.assertEqual(("group-1",), tuple(visible.call_args.kwargs["group_ids"]))
without_authority = {
topic.id: topic
for topic in documentation_topics(self.context({"files:file:read"}))
}
self.assertNotIn("files.workflow.import-managed-snapshot", without_authority)
self.assertIn(
"both permission to view Files and permission to upload",
without_authority["files.connector-import-unavailable"].body,
)
def test_connector_limitation_is_precise_but_never_exposes_profile_internals(
self,
) -> None:
blocked = ConnectorProfile(
id="sensitive-profile-id",
label="Sensitive label",
provider="s3",
scope_type="tenant",
scope_id="tenant-1",
endpoint_url="https://internal.example.invalid/private",
base_path="/classified",
credential_mode="basic",
password_value="credential-secret",
secret_ref="runtime/connector-secret",
)
with patch(
"govoplan_files.backend.documentation.visible_connector_profiles_for_actor",
return_value=[blocked],
):
topics = {
topic.id: topic
for topic in documentation_topics(
self.context({"files:file:read", "files:file:upload"})
)
}
self.assertNotIn("files.workflow.import-managed-snapshot", topics)
limitation = topics["files.connector-import-unavailable"]
self.assertEqual("available", limitation.layer)
self.assertIn("pinning-safe provider", limitation.body)
payload = repr(limitation)
for secret in (
"sensitive-profile-id",
"Sensitive label",
"internal.example.invalid",
"classified",
"credential-secret",
):
self.assertNotIn(secret, payload)
def test_files_admin_connector_groups_do_not_widen_campaign_membership(self) -> None:
usable = ConnectorProfile(
id="group-profile",
label="Group profile",
provider="webdav",
scope_type="group",
scope_id="admin-visible-group",
endpoint_url="https://files.example.invalid",
credential_mode="anonymous",
)
def groups_for_actor(_session, *, include_admin_groups, **_kwargs):
return ["member-group", "admin-visible-group"] if include_admin_groups else ["member-group"]
with (
patch(
"govoplan_files.backend.documentation.user_group_ids",
side_effect=groups_for_actor,
),
patch(
"govoplan_files.backend.documentation._campaign_visibility",
return_value=None,
) as campaign_visibility,
patch(
"govoplan_files.backend.documentation.visible_connector_profiles_for_actor",
return_value=[usable],
) as visible,
):
topics = {
topic.id: topic
for topic in documentation_topics(
self.context(
{
"files:file:read",
"files:file:upload",
"files:file:admin",
}
)
)
}
self.assertIn("files.workflow.import-managed-snapshot", topics)
self.assertEqual(
("member-group", "admin-visible-group"),
tuple(visible.call_args.kwargs["group_ids"]),
)
self.assertEqual(
("member-group",),
tuple(campaign_visibility.call_args.kwargs["group_ids"]),
)
def test_connector_evaluation_errors_fail_closed_without_error_details(
self,
) -> None:
with patch(
"govoplan_files.backend.documentation.visible_connector_profiles_for_actor",
side_effect=RuntimeError("private endpoint /root credential-id"),
):
topics = {
topic.id: topic
for topic in documentation_topics(
self.context({"files:file:read", "files:file:upload"})
)
}
self.assertNotIn("files.workflow.import-managed-snapshot", topics)
limitation = topics["files.connector-import-unavailable"]
self.assertIn("could not be safely evaluated", limitation.body)
self.assertNotIn("private endpoint", repr(limitation))
self.assertNotIn("credential-id", repr(limitation))
def test_runtime_provider_only_emits_user_documentation(self) -> None:
self.assertEqual(
(), documentation_topics(self.context(set(), documentation_type="admin"))
)
if __name__ == "__main__":
unittest.main()
+132
View File
@@ -0,0 +1,132 @@
from __future__ import annotations
import unittest
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from threading import Thread
from unittest.mock import MagicMock, patch
import httpcore
from govoplan_core.security.outbound_http import outbound_http_policy, validate_outbound_http_url
from govoplan_files.backend.storage.http_client import (
ConnectorHttpError,
_OutboundPolicyNetworkBackend,
_SocketNetworkStream,
request_connector_bytes,
)
class _TestHandler(BaseHTTPRequestHandler):
def do_GET(self) -> None: # noqa: N802 - stdlib handler contract
body = b"connector-ok"
self.send_response(200)
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def log_message(self, _format: str, *_args: object) -> None:
return
class ConnectorHttpClientTests(unittest.TestCase):
def test_public_transport_adapter_performs_a_real_http_request(self) -> None:
server = ThreadingHTTPServer(("127.0.0.1", 0), _TestHandler)
thread = Thread(target=server.serve_forever, daemon=True)
thread.start()
try:
with patch.dict(
"os.environ",
{"APP_ENV": "test", "GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS": "true"},
clear=False,
):
result = request_connector_bytes(
"GET",
f"http://127.0.0.1:{server.server_port}/object",
)
finally:
server.shutdown()
server.server_close()
thread.join(timeout=2)
self.assertEqual(200, result.status_code)
self.assertEqual(b"connector-ok", result.content)
def test_connector_responses_are_streamed_without_redirects(self) -> None:
response = MagicMock()
response.status_code = 200
response.headers = {"content-length": "6"}
response.iter_bytes.return_value = iter((b"abc", b"def"))
context = MagicMock()
context.__enter__.return_value = response
with patch(
"govoplan_core.security.outbound_http.socket.getaddrinfo",
return_value=[(2, 1, 6, "", ("93.184.216.34", 443))],
), patch("govoplan_files.backend.storage.http_client._stream_connector_request", return_value=context):
result = request_connector_bytes("GET", "https://example.test/object", max_bytes=10)
self.assertEqual(b"abcdef", result.content)
def test_connector_response_limit_is_enforced_while_streaming(self) -> None:
response = MagicMock()
response.status_code = 200
response.headers = {}
response.iter_bytes.return_value = iter((b"12345", b"67890", b"!"))
context = MagicMock()
context.__enter__.return_value = response
with patch(
"govoplan_core.security.outbound_http.socket.getaddrinfo",
return_value=[(2, 1, 6, "", ("93.184.216.34", 443))],
), patch(
"govoplan_files.backend.storage.http_client._stream_connector_request",
return_value=context,
), self.assertRaisesRegex(
ConnectorHttpError,
"configured limit",
):
request_connector_bytes("GET", "https://example.test/object", max_bytes=10)
def test_private_destination_is_blocked_before_http_connection(self) -> None:
with patch.dict(
"os.environ",
{"APP_ENV": "production", "GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS": "false"},
), patch(
"govoplan_core.security.outbound_http.socket.getaddrinfo",
return_value=[(2, 1, 6, "", ("10.0.0.5", 443))],
), patch("govoplan_files.backend.storage.http_client._stream_connector_request") as stream, self.assertRaisesRegex(
ConnectorHttpError,
"non-public network",
):
request_connector_bytes("GET", "https://connector.example.test/object")
stream.assert_not_called()
def test_connection_backend_rejects_private_rebinding_before_socket_open(self) -> None:
policy = outbound_http_policy({"APP_ENV": "production"})
public = [(2, 1, 6, "", ("93.184.216.34", 443))]
private = [(2, 1, 6, "", ("127.0.0.1", 443))]
with patch.dict(
"os.environ",
{"APP_ENV": "production", "GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS": "false"},
), patch(
"govoplan_core.security.outbound_http.socket.getaddrinfo",
side_effect=(public, private),
), patch("govoplan_core.security.outbound_http.socket.socket") as socket_factory:
validate_outbound_http_url("https://connector.example.test/object", policy=policy)
with self.assertRaises(httpcore.ConnectError):
_OutboundPolicyNetworkBackend().connect_tcp("connector.example.test", 443)
socket_factory.assert_not_called()
def test_network_backend_returns_public_network_stream_adapter(self) -> None:
sock = MagicMock()
with patch(
"govoplan_files.backend.storage.http_client.create_outbound_connection",
return_value=sock,
):
stream = _OutboundPolicyNetworkBackend().connect_tcp("connector.example.test", 443)
self.assertIsInstance(stream, _SocketNetworkStream)
self.assertIs(stream.get_extra_info("socket"), sock)
if __name__ == "__main__":
unittest.main()
+211
View File
@@ -0,0 +1,211 @@
from __future__ import annotations
import hashlib
import tempfile
import unittest
from pathlib import Path
from unittest.mock import patch
from fastapi import HTTPException
from sqlalchemy import create_engine
from sqlalchemy.orm import sessionmaker
from govoplan_access.backend.db.models import Account, Group, User
from govoplan_core.db.base import Base
from govoplan_files.backend.db.models import (
FileBlob,
FileIntegrityFinding,
FileIntegrityScan,
)
from govoplan_files.backend.routes.integrity import _assert_expected_revision
from govoplan_files.backend.storage.backends import LocalFilesystemStorageBackend
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.integrity import (
cleanup_orphan_finding,
create_integrity_scan,
read_verified_blob_bytes,
recheck_integrity_finding,
run_integrity_scan_batch,
)
TENANT_ID = "tenant-1"
USER_ID = "user-1"
class IntegrityReconciliationTests(unittest.TestCase):
def setUp(self) -> None:
self.temporary_directory = tempfile.TemporaryDirectory()
self.addCleanup(self.temporary_directory.cleanup)
self.backend = LocalFilesystemStorageBackend(
Path(self.temporary_directory.name)
)
self.engine = create_engine("sqlite:///:memory:", future=True)
Base.metadata.create_all(
bind=self.engine,
tables=[
Account.__table__,
User.__table__,
Group.__table__,
FileBlob.__table__,
FileIntegrityScan.__table__,
FileIntegrityFinding.__table__,
],
)
self.session = sessionmaker(bind=self.engine, future=True)()
self.addCleanup(self._close)
def _close(self) -> None:
self.session.close()
self.engine.dispose()
def test_scan_is_bounded_resumable_and_reconciles_both_orphan_directions(
self,
) -> None:
valid_data = b"valid"
restored_data = b"restore-me"
expected_corrupt_data = b"expected"
stored_corrupt_data = b"corrupt!"
valid = _blob("blob-1", "valid.bin", valid_data)
missing = _blob("blob-2", "missing.bin", restored_data)
corrupt = _blob("blob-3", "corrupt.bin", expected_corrupt_data)
self.session.add_all([valid, missing, corrupt])
self.session.commit()
self.backend.put_bytes(valid.storage_key, valid_data)
self.backend.put_bytes(corrupt.storage_key, stored_corrupt_data)
orphan_key = f"tenants/{TENANT_ID}/files/orphan.bin"
self.backend.put_bytes(orphan_key, b"orphan")
scan = create_integrity_scan(
self.session,
tenant_id=TENANT_ID,
user_id=USER_ID,
batch_size=1,
backend=self.backend,
)
self.session.commit()
self.assertEqual(1, scan.revision)
invocations = 0
while scan.status != "completed":
run_integrity_scan_batch(
self.session,
scan,
backend=self.backend,
)
self.session.commit()
invocations += 1
scan = self.session.get(FileIntegrityScan, scan.id)
self.assertIsNotNone(scan)
self.session.expire_all()
if invocations > 12:
self.fail("Integrity scan did not complete")
self.assertGreater(invocations, 3)
self.assertEqual(1 + invocations, scan.revision)
self.assertEqual(3, scan.scanned_blob_count)
self.assertEqual(1, scan.verified_blob_count)
self.assertEqual(2, scan.quarantined_blob_count)
self.assertEqual(3, scan.scanned_object_count)
self.assertEqual(1, scan.orphan_object_count)
findings = (
self.session.query(FileIntegrityFinding)
.filter(FileIntegrityFinding.scan_id == scan.id)
.all()
)
self.assertEqual(
{"missing", "checksum_mismatch", "orphan_object"},
{finding.kind for finding in findings},
)
missing = self.session.get(FileBlob, missing.id)
corrupt = self.session.get(FileBlob, corrupt.id)
self.assertEqual("missing", missing.integrity_status)
self.assertEqual("checksum_mismatch", corrupt.integrity_status)
with self.assertRaisesRegex(FileStorageError, "quarantined"):
read_verified_blob_bytes(missing, backend=self.backend)
missing_finding = next(
finding for finding in findings if finding.kind == "missing"
)
self.backend.put_bytes(missing.storage_key, restored_data)
preview = recheck_integrity_finding(
self.session,
missing_finding,
user_id=USER_ID,
dry_run=True,
backend=self.backend,
)
self.assertTrue(preview.inspection.valid)
self.assertEqual("open", missing_finding.state)
repaired = recheck_integrity_finding(
self.session,
missing_finding,
user_id=USER_ID,
dry_run=False,
backend=self.backend,
)
self.session.commit()
self.assertTrue(repaired.changed)
self.assertEqual("resolved", missing_finding.state)
self.assertEqual(
restored_data,
read_verified_blob_bytes(missing, backend=self.backend),
)
orphan_finding = next(
finding for finding in findings if finding.kind == "orphan_object"
)
preview_cleanup = cleanup_orphan_finding(
self.session,
orphan_finding,
user_id=USER_ID,
dry_run=True,
backend=self.backend,
)
self.assertEqual("would_delete", preview_cleanup.action)
self.assertTrue(self.backend.exists(orphan_key))
with patch(
"govoplan_files.backend.storage.integrity.begin_orphan_cleanup_recovery"
):
cleanup = cleanup_orphan_finding(
self.session,
orphan_finding,
user_id=USER_ID,
dry_run=False,
backend=self.backend,
)
repeated = cleanup_orphan_finding(
self.session,
orphan_finding,
user_id=USER_ID,
dry_run=False,
backend=self.backend,
)
self.assertEqual("deleted", cleanup.action)
self.assertFalse(self.backend.exists(orphan_key))
self.assertEqual("already_deleted", repeated.action)
self.assertFalse(repeated.changed)
def test_stale_integrity_action_revision_is_rejected(self) -> None:
with self.assertRaises(HTTPException) as captured:
_assert_expected_revision(3, 2)
self.assertEqual(409, captured.exception.status_code)
self.assertIn("reload", str(captured.exception.detail).lower())
def _blob(blob_id: str, filename: str, expected_data: bytes) -> FileBlob:
return FileBlob(
id=blob_id,
tenant_id=TENANT_ID,
storage_backend="local",
storage_key=f"tenants/{TENANT_ID}/files/{filename}",
checksum_sha256=hashlib.sha256(expected_data).hexdigest(),
size_bytes=len(expected_data),
ref_count=1,
)
if __name__ == "__main__":
unittest.main()
+239
View File
@@ -0,0 +1,239 @@
from __future__ import annotations
import unittest
STATIC_TOPIC_IDS = {
"files.search.managed-content",
"files.workflow.organize-managed-files",
"files.workflow.find-and-download-files",
"files.workflow.share-managed-files",
"files.workflow.delete-managed-files",
"files.governed-connectors-and-provenance",
"files.reference.integrity-recovery-and-fail-closed-transports",
"files.reference.shared-storage-profile",
"files.reference.generated-artifact-store",
"files.reference.snapshot-provenance-and-capabilities",
"files.assurance.process-and-release-readiness",
}
RUNTIME_TOPIC_IDS = {
"files.workflow.upload-managed-files",
"files.workflow.upload-and-unpack-zip",
"files.workflow.import-managed-snapshot",
"files.connector-import-unavailable",
}
HANDBOOK_HREF = "govoplan-files/docs/FILES_HANDBOOK.md"
class FilesManifestDocumentationTests(unittest.TestCase):
@classmethod
def setUpClass(cls) -> None:
from govoplan_files.backend.manifest import manifest
cls.manifest = manifest
cls.topics = {topic.id: topic for topic in manifest.documentation}
def topic(self, topic_id: str):
return self.topics[topic_id]
def test_static_topics_have_role_scope_module_and_link_contracts(self) -> None:
self.assertEqual(STATIC_TOPIC_IDS, set(self.topics))
self.assertEqual(
{"admin", "user"},
{
item
for topic in self.topics.values()
for item in topic.documentation_types
},
)
for topic in self.topics.values():
with self.subTest(topic=topic.id):
self.assertTrue(topic.audience)
self.assertTrue(topic.conditions)
self.assertTrue(
any(
"files" in condition.required_modules
for condition in topic.conditions
)
)
self.assertTrue(
any(
condition.any_scopes or condition.required_scopes
for condition in topic.conditions
)
)
self.assertIn("runtime", {link.kind for link in topic.links})
self.assertIn(
HANDBOOK_HREF,
{link.href for link in topic.links if link.kind == "repository"},
)
self.assertIn(
"documentation_topics",
{provider.__name__ for provider in self.manifest.documentation_providers},
)
def test_user_tasks_are_independently_authorized_and_contextual(self) -> None:
expected_scopes = {
"files.workflow.organize-managed-files": {
"files:file:read",
"files:file:organize",
},
"files.workflow.find-and-download-files": {
"files:file:read",
"files:file:download",
},
"files.workflow.share-managed-files": {
"files:file:read",
"files:file:share",
},
"files.workflow.delete-managed-files": {
"files:file:read",
"files:file:delete",
},
}
for topic_id, scopes in expected_scopes.items():
with self.subTest(topic=topic_id):
topic = self.topic(topic_id)
self.assertEqual(("user",), topic.documentation_types)
self.assertEqual("workflow", topic.metadata["kind"])
self.assertEqual(scopes, set(topic.conditions[0].required_scopes))
self.assertEqual(["files.list"], topic.metadata["help_contexts"])
for key in ("prerequisites", "steps", "outcome", "verification"):
self.assertTrue(topic.metadata[key])
def test_share_and_delete_tasks_state_current_boundaries(self) -> None:
share = self.topic("files.workflow.share-managed-files")
self.assertEqual("available", share.layer)
self.assertIn("Expired and revoked grants stop authorizing", share.body)
self.assertIn("revocation is idempotent", share.body)
self.assertIn(
"/api/v1/files/{file_id}/shares", {link.href for link in share.links}
)
delete = self.topic("files.workflow.delete-managed-files")
self.assertIn("Soft-delete", delete.summary)
self.assertIn("no self-service restore or hard-purge", delete.body)
self.assertTrue(
any("not a hard purge" in item for item in delete.metadata["limitations"])
)
def test_admin_topic_covers_policy_redaction_and_atomic_credential_deletion(
self,
) -> None:
topic = self.topic("files.governed-connectors-and-provenance")
self.assertEqual(("admin",), topic.documentation_types)
self.assertEqual("reference", topic.metadata["kind"])
self.assertEqual("File connections", topic.metadata["screen"])
self.assertTrue(topic.metadata["section"])
self.assertEqual(
{
"files.connectors",
"files.connector.credentials",
"files.connector.policy",
},
set(topic.metadata["help_contexts"]),
)
self.assertIn("deny rules win", topic.body)
self.assertIn("redact secret values", topic.body)
self.assertIn("same transaction", topic.body)
self.assertTrue(
any(
"non-owned external references" in item
for item in topic.metadata["security_invariants"]
)
)
self.assertIn(
"/api/v1/files/connectors/policies/tenant",
{link.href for link in topic.links},
)
self.assertIn(
"/api/v1/files/connectors/credentials", {link.href for link in topic.links}
)
def test_operator_topic_covers_recovery_and_pinned_s3_smb(self) -> None:
topic = self.topic(
"files.reference.integrity-recovery-and-fail-closed-transports"
)
self.assertEqual(("admin",), topic.documentation_types)
self.assertEqual("reference", topic.metadata["kind"])
self.assertIn("S3", topic.body)
self.assertIn("SMB", topic.body)
self.assertIn("fail closed", topic.body)
self.assertIn("DFS referrals", topic.body)
self.assertIn("ambient credential discovery", topic.body)
self.assertIn("does not remove backend blob objects", topic.body)
self.assertIn("bounded resumable integrity scan", topic.body)
self.assertIn("quarantined", topic.body)
self.assertIn("MASTER_KEY_B64", topic.metadata["recovery_unit"])
self.assertIn("Encryption envelope and wrapped-key rows", topic.metadata["recovery_unit"])
self.assertIn("lease-fenced Core recovery", topic.body)
self.assertIn("Ops", topic.body)
self.assertTrue(topic.metadata["verification"])
self.assertIn(
"/api/v1/files/integrity/scans",
{link.href for link in topic.links},
)
self.assertIn(
"/admin?section=tenant-file-integrity",
{link.href for link in topic.links},
)
self.assertIn(
"files.admin.tenant-integrity",
{surface.id for surface in self.manifest.frontend.view_surfaces},
)
self.assertIn(
"GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS", topic.configuration_keys
)
def test_integrator_topic_exposes_capability_and_provenance_boundary(self) -> None:
topic = self.topic("files.reference.snapshot-provenance-and-capabilities")
self.assertEqual(("admin", "user"), topic.documentation_types)
self.assertEqual("reference", topic.metadata["kind"])
self.assertEqual(
["files.access@0.1.6", "files.campaign_attachments@0.1.6"],
topic.metadata["provided_interfaces"],
)
self.assertIn("revision", topic.metadata["provenance_fields"])
self.assertIn("exact asset, version, blob, checksum", topic.body)
self.assertIn("collaboration", topic.body)
def test_process_and_release_assurance_exposes_gates_evidence_and_limits(
self,
) -> None:
topic = self.topic("files.assurance.process-and-release-readiness")
self.assertEqual(("admin", "user"), topic.documentation_types)
self.assertEqual("workflow", topic.metadata["kind"])
self.assertEqual(["files.list"], topic.metadata["help_contexts"])
self.assertIn("process_owner", topic.audience)
self.assertIn("release_manager", topic.audience)
self.assertIn("versions align", topic.body)
self.assertIn("Share grant, change, expiry, and revocation", topic.body)
self.assertIn("no enforced retention or legal hold", topic.body)
for key in ("prerequisites", "steps", "outcome", "verification"):
self.assertTrue(topic.metadata[key])
self.assertTrue(
any("meta-repository security" in step for step in topic.metadata["steps"])
)
self.assertTrue(any("fails closed" in step for step in topic.metadata["steps"]))
self.assertTrue(any("SHA-256" in step for step in topic.metadata["steps"]))
def test_internal_related_topic_ids_resolve(self) -> None:
known_ids = STATIC_TOPIC_IDS | RUNTIME_TOPIC_IDS
for topic in self.topics.values():
for related_id in topic.metadata.get("related_topic_ids", []):
if related_id.startswith("files."):
self.assertIn(
related_id,
known_ids,
f"{topic.id} refers to missing topic {related_id}",
)
if __name__ == "__main__":
unittest.main()
+42
View File
@@ -0,0 +1,42 @@
from __future__ import annotations
from unittest.mock import patch
from govoplan_files.backend.operational_checks import managed_storage_roundtrip_check
from govoplan_files.backend.storage.backends import LocalFilesystemStorageBackend
def test_managed_storage_roundtrip_removes_probe(tmp_path) -> None:
backend = LocalFilesystemStorageBackend(tmp_path)
with patch(
"govoplan_files.backend.operational_checks.get_storage_backend",
return_value=backend,
):
result = managed_storage_roundtrip_check()
assert result.state == "ok"
assert result.metrics["backend"] == "local"
assert list(tmp_path.rglob("*.bin")) == []
def test_managed_storage_roundtrip_fails_closed() -> None:
class BrokenBackend:
name = "broken"
def put_bytes(self, *_args, **_kwargs):
raise OSError("private provider detail")
def delete(self, *_args, **_kwargs):
return None
with patch(
"govoplan_files.backend.operational_checks.get_storage_backend",
return_value=BrokenBackend(),
):
result = managed_storage_roundtrip_check()
assert result.state == "error"
assert result.readiness_critical is True
assert "private provider detail" not in result.detail
+97
View File
@@ -0,0 +1,97 @@
from __future__ import annotations
import unittest
from sqlalchemy import create_engine
from sqlalchemy.orm import sessionmaker
from govoplan_access.backend.db import models as access_models # noqa: F401
from govoplan_core.core.provider_governance import ExternalProviderStateContext
from govoplan_core.db.base import Base
from govoplan_files.backend.db.models import FileConnectorProfile, FileConnectorSpace
from govoplan_files.backend.manifest import manifest
from govoplan_files.backend.provider_state import (
REMOTE_STORAGE_PROVIDER_ID,
remote_storage_provider_states,
)
class FilesProviderStateTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite+pysqlite:///:memory:", future=True)
Base.metadata.create_all(
self.engine,
tables=(FileConnectorProfile.__table__, FileConnectorSpace.__table__),
)
self.session = sessionmaker(bind=self.engine, expire_on_commit=False)()
self.profile = FileConnectorProfile(
id="profile-1",
tenant_id="tenant-1",
scope_type="tenant",
scope_id="tenant-1",
label="Remote WebDAV",
provider="webdav",
endpoint_url="https://files.example.test/dav/",
enabled=True,
)
self.session.add_all(
(
self.profile,
FileConnectorSpace(
id="space-1",
tenant_id="tenant-1",
owner_type="tenant",
label="Documents",
connector_profile_id=self.profile.id,
provider="webdav",
remote_path="/documents",
read_only=True,
is_active=True,
),
)
)
self.session.commit()
def tearDown(self) -> None:
self.session.close()
self.engine.dispose()
def test_state_does_not_overstate_live_health_or_expose_endpoint(self) -> None:
state = remote_storage_provider_states(
ExternalProviderStateContext(session=self.session, tenant_id="tenant-1")
)[0]
self.assertEqual("external_mirror", state.authority_mode)
self.assertEqual("unknown", state.health)
self.assertEqual("attention", state.recovery)
self.assertEqual(1, state.metrics["active_spaces"])
self.assertNotIn("files.example.test", str(state.to_dict()))
self.assertEqual(
(),
remote_storage_provider_states(
ExternalProviderStateContext(
session=self.session,
tenant_id="tenant-2",
)
),
)
def test_unimplemented_profile_is_fail_closed_and_manifest_registered(self) -> None:
self.profile.provider = "sharepoint"
self.session.flush()
state = remote_storage_provider_states(
ExternalProviderStateContext(session=self.session, tenant_id="tenant-1")
)[0]
self.assertEqual("error", state.health)
self.assertEqual("unsupported", state.recovery)
self.assertEqual(REMOTE_STORAGE_PROVIDER_ID, manifest.external_providers[0].id)
self.assertEqual(
REMOTE_STORAGE_PROVIDER_ID,
manifest.external_provider_state_providers[0].provider_id,
)
if __name__ == "__main__":
unittest.main()
+113
View File
@@ -0,0 +1,113 @@
from __future__ import annotations
import unittest
from collections import Counter
from inspect import signature
from govoplan_files.backend.router import router
from govoplan_files.backend.routes.assets import router as assets_router
from govoplan_files.backend.routes.connector_io import router as connector_io_router
from govoplan_files.backend.routes.connector_profiles import router as connector_profiles_router
from govoplan_files.backend.routes.connector_settings import router as connector_settings_router
from govoplan_files.backend.routes.folders import router as folders_router
from govoplan_files.backend.routes.integrity import router as integrity_router
from govoplan_files.backend.routes.listing import router as listing_router
from govoplan_files.backend.routes.shares import router as shares_router
from govoplan_files.backend.routes.spaces import router as spaces_router
from govoplan_files.backend.routes.transfers import router as transfers_router
from govoplan_files.backend.routes.uploads import router as uploads_router
class FilesRouterContractTests(unittest.TestCase):
@staticmethod
def _operation_keys(candidate_router) -> list[tuple[str, str]]:
return [
(method, route.path)
for route in candidate_router.routes
for method in sorted(route.methods or ())
]
def test_composed_router_contains_every_workflow_operation_once(self) -> None:
workflow_routers = (
spaces_router,
folders_router,
integrity_router,
listing_router,
uploads_router,
connector_settings_router,
connector_io_router,
connector_profiles_router,
assets_router,
shares_router,
transfers_router,
)
expected = [
operation
for workflow_router in workflow_routers
for operation in self._operation_keys(workflow_router)
]
actual = self._operation_keys(router)
self.assertEqual(expected, actual)
self.assertEqual(52, len(actual))
self.assertFalse(
[operation for operation, count in Counter(actual).items() if count > 1]
)
def test_archive_preview_and_confirmation_routes_are_exposed(self) -> None:
routes = {
(tuple(sorted(route.methods or ())), route.path)
for route in router.routes
}
self.assertIn((("POST",), "/files/archive-preview"), routes)
self.assertIn((("POST",), "/files/archive-confirm"), routes)
def test_connector_routes_keep_existing_api_paths(self) -> None:
routes = {(tuple(sorted(route.methods or ())), route.path) for route in router.routes}
expected = {
(("GET",), "/files/connectors/providers"),
(("GET",), "/files/connectors/settings/delta"),
(("GET",), "/files/connectors/profiles"),
(("POST",), "/files/connectors/profiles"),
(("GET",), "/files/connectors/profiles/{profile_id}/browse"),
(("POST",), "/files/connectors/profiles/{profile_id}/import"),
(("POST",), "/files/connectors/profiles/{profile_id}/sync"),
(("GET",), "/files/connectors/credentials"),
(("POST",), "/files/connectors/credentials"),
(("GET",), "/files/connector-spaces"),
(("POST",), "/files/connector-spaces"),
}
self.assertTrue(expected.issubset(routes))
def test_bulk_organize_routes_keep_existing_api_paths(self) -> None:
routes = {(tuple(sorted(route.methods or ())), route.path) for route in router.routes}
self.assertIn((("POST",), "/files/bulk-rename"), routes)
self.assertIn((("POST",), "/files/transfer"), routes)
def test_share_lifecycle_routes_are_exposed(self) -> None:
routes = {(tuple(sorted(route.methods or ())), route.path) for route in router.routes}
self.assertIn((("GET",), "/files/{file_id}/shares"), routes)
self.assertIn((("POST",), "/files/{file_id}/shares"), routes)
self.assertIn((("DELETE",), "/files/{file_id}/shares/{share_id}"), routes)
self.assertIn((("GET",), "/files/{file_id}/share-target-options"), routes)
def test_file_listing_exposes_structured_property_filters(self) -> None:
route = next(
route
for route in listing_router.routes
if route.path == "/files" and "GET" in (route.methods or ())
)
parameters = signature(route.endpoint).parameters
self.assertIn("campaign_usage", parameters)
self.assertIn("audit_relevant", parameters)
self.assertIn("total", route.response_model.model_fields)
if __name__ == "__main__":
unittest.main()
+294
View File
@@ -0,0 +1,294 @@
from __future__ import annotations
import socket
import sys
import threading
import types
import unittest
from importlib.util import find_spec
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from unittest.mock import patch
from govoplan_core.security.outbound_http import (
OutboundHttpBlocked,
create_outbound_connection as core_create_outbound_connection,
)
from govoplan_files.backend.storage.sdk_peer_pinning import (
SdkPeerPinningError,
_botocore_transport_types,
create_pinned_s3_client,
install_pinned_smb_transport,
pinned_smb_connection_cache,
)
class _Socket:
def __init__(self) -> None:
self.connected_to: object | None = None
def settimeout(self, _value: object) -> None:
return None
def setsockopt(self, *_args: object) -> None:
return None
def connect(self, value: object) -> None:
self.connected_to = value
def close(self) -> None:
return None
class _S3Handler(BaseHTTPRequestHandler):
def do_GET(self) -> None: # noqa: N802 - stdlib handler contract
body = (
b'<?xml version="1.0" encoding="UTF-8"?>'
b'<ListAllMyBucketsResult xmlns="http://s3.amazonaws.com/doc/2006-03-01/">'
b'<Owner><ID>govoplan</ID><DisplayName>GovOPlaN</DisplayName></Owner>'
b'<Buckets><Bucket><Name>evidence</Name><CreationDate>2026-08-04T00:00:00Z</CreationDate>'
b'</Bucket></Buckets></ListAllMyBucketsResult>'
)
self.send_response(200)
self.send_header("Content-Type", "application/xml")
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def log_message(self, _format: str, *_args: object) -> None:
return None
@unittest.skipUnless(find_spec("botocore") and find_spec("boto3"), "boto3 extra is not installed")
class S3PeerPinningTests(unittest.TestCase):
def test_client_is_born_with_the_pinned_http_session(self) -> None:
client = create_pinned_s3_client(
endpoint_url="http://127.0.0.1:9000",
region_name="eu-central-1",
aws_access_key_id="access",
aws_secret_access_key="secret",
)
try:
session_type, _, _ = _botocore_transport_types()
self.assertIsInstance(client._endpoint.http_session, session_type)
self.assertEqual({}, client._endpoint.http_session._proxy_config._proxies)
finally:
client.close()
def test_each_connection_attempt_resolves_and_pins_again(self) -> None:
_, http_connection, _ = _botocore_transport_types()
connection = http_connection(host="objects.example.test", port=80)
sockets = [_Socket(), _Socket()]
with patch(
"govoplan_files.backend.storage.sdk_peer_pinning.create_outbound_connection",
side_effect=sockets,
) as connector:
self.assertIs(sockets[0], connection._new_conn())
self.assertIs(sockets[1], connection._new_conn())
self.assertEqual(2, connector.call_count)
self.assertEqual("objects.example.test", connector.call_args.args[0])
def test_real_sdk_request_uses_the_pinned_socket_path(self) -> None:
server = ThreadingHTTPServer(("127.0.0.1", 0), _S3Handler)
thread = threading.Thread(target=server.serve_forever, daemon=True)
thread.start()
endpoint = f"http://127.0.0.1:{server.server_port}"
try:
with patch.dict(
"os.environ",
{"APP_ENV": "production", "GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS": "true"},
), patch(
"govoplan_files.backend.storage.sdk_peer_pinning.create_outbound_connection",
wraps=core_create_outbound_connection,
) as connector:
client = create_pinned_s3_client(
endpoint_url=endpoint,
region_name="eu-central-1",
aws_access_key_id="access",
aws_secret_access_key="secret",
)
try:
response = client.list_buckets()
finally:
client.close()
finally:
server.shutdown()
server.server_close()
thread.join(timeout=2)
self.assertEqual("evidence", response["Buckets"][0]["Name"])
self.assertEqual("127.0.0.1", connector.call_args.args[0])
def test_every_sdk_selected_origin_uses_a_pinned_pool(self) -> None:
session_type, http_connection, https_connection = _botocore_transport_types()
session = session_type(proxies={})
try:
first = session._manager.connection_from_url("http://one.example.test/root")
redirected = session._manager.connection_from_url("https://two.example.test/root")
self.assertIs(first.ConnectionCls, http_connection)
self.assertIs(redirected.ConnectionCls, https_connection)
finally:
session.close()
def test_tls_connection_retains_the_configured_authority_for_sni(self) -> None:
_, _, https_connection = _botocore_transport_types()
connection = https_connection(host="objects.example.test", port=443)
with patch(
"govoplan_files.backend.storage.sdk_peer_pinning.create_outbound_connection",
return_value=_Socket(),
) as connector:
connection._new_conn()
self.assertEqual("objects.example.test", connection.host)
self.assertEqual("objects.example.test", connection._dns_host)
self.assertEqual("objects.example.test", connector.call_args.args[0])
def test_mixed_answer_and_peer_change_fail_closed(self) -> None:
_, http_connection, _ = _botocore_transport_types()
connection = http_connection(host="objects.example.test", port=80)
records = [
[(socket.AF_INET, socket.SOCK_STREAM, 6, "", ("93.184.216.34", 80))],
[
(socket.AF_INET, socket.SOCK_STREAM, 6, "", ("93.184.216.34", 80)),
(socket.AF_INET, socket.SOCK_STREAM, 6, "", ("10.0.0.5", 80)),
],
]
with patch.dict(
"os.environ",
{"APP_ENV": "production", "GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS": "false"},
), patch(
"govoplan_core.security.outbound_http.socket.getaddrinfo",
side_effect=records,
), patch(
"govoplan_core.security.outbound_http.socket.socket",
return_value=_Socket(),
):
connection._new_conn()
with self.assertRaisesRegex(Exception, "non-public network"):
connection._new_conn()
class _FakeTcp:
def __init__(self, server: str, port: int, timeout: float | None = None) -> None:
self.server = server
self.port = port
self.timeout = timeout
self.connected = False
self._sock = None
self._sock_lock = threading.Lock()
class _FakeConnection:
def __init__(self, _guid: object, server: str, port: int, *, require_signing: bool) -> None:
self.server = server
self.port = port
self.require_signing = require_signing
self.transport = _FakeTcp(server, port)
self.session_table: dict[str, object] = {}
def connect(self, *, timeout: float) -> None:
self.transport.timeout = timeout
self.transport.connect()
class _FakeSession:
def __init__(
self,
connection: _FakeConnection,
*,
username: str | None,
password: str | None,
require_encryption: bool,
auth_protocol: str,
) -> None:
self.connection = connection
self.username = username
self.password = password
self.encrypt_data = require_encryption
self.auth_protocol = auth_protocol
self.encrypt = require_encryption
def connect(self) -> None:
self.connection.session_table[self.username or "anonymous"] = self
class SmbPeerPinningTests(unittest.TestCase):
def setUp(self) -> None:
pinned_smb_connection_cache.cache_clear()
def _modules(self) -> tuple[types.ModuleType, dict[str, types.ModuleType]]:
smbclient = types.ModuleType("smbclient")
pool = types.ModuleType("smbclient._pool")
def original_register_session(
server: str,
username: str | None = None,
password: str | None = None,
port: int = 445,
encrypt: bool | None = None,
connection_timeout: float = 60,
connection_cache: dict[str, object] | None = None,
auth_protocol: str = "negotiate",
require_signing: bool = True,
) -> None:
del server, username, password, port, encrypt, connection_timeout
del connection_cache, auth_protocol, require_signing
pool.register_session = original_register_session
pool.ClientConfig = lambda: types.SimpleNamespace(client_guid="guid")
connection = types.ModuleType("smbprotocol.connection")
connection.Connection = _FakeConnection
connection.Tcp = _FakeTcp
session = types.ModuleType("smbprotocol.session")
session.Session = _FakeSession
transport = types.ModuleType("smbprotocol.transport")
transport.Tcp = _FakeTcp
return smbclient, {
"smbclient._pool": pool,
"smbprotocol.connection": connection,
"smbprotocol.session": session,
"smbprotocol.transport": transport,
}
def test_initial_reconnect_and_referral_hosts_are_each_pinned(self) -> None:
smbclient, modules = self._modules()
sockets = [_Socket(), _Socket(), _Socket()]
with patch.dict(sys.modules, modules), patch(
"govoplan_files.backend.storage.sdk_peer_pinning.create_outbound_connection",
side_effect=sockets,
) as connector:
install_pinned_smb_transport(smbclient)
register = modules["smbclient._pool"].register_session
cache: dict[str, object] = {}
first = register("files.example.test", connection_cache=cache)
first.connection.transport.connected = False
register("files.example.test", connection_cache=cache)
register("dfs-target.example.test", connection_cache=cache)
self.assertEqual(
["files.example.test", "files.example.test", "dfs-target.example.test"],
[call.args[0] for call in connector.call_args_list],
)
def test_referral_to_disallowed_peer_fails_before_session_creation(self) -> None:
smbclient, modules = self._modules()
with patch.dict(sys.modules, modules), patch(
"govoplan_files.backend.storage.sdk_peer_pinning.create_outbound_connection",
side_effect=OutboundHttpBlocked("non-public network"),
):
install_pinned_smb_transport(smbclient)
register = modules["smbclient._pool"].register_session
with self.assertRaisesRegex(ValueError, "non-public network"):
register("private-referral.example.test", connection_cache={})
def test_transport_tampering_after_installation_fails_closed(self) -> None:
smbclient, modules = self._modules()
with patch.dict(sys.modules, modules):
install_pinned_smb_transport(smbclient)
modules["smbprotocol.connection"].Tcp = _FakeTcp
with self.assertRaisesRegex(SdkPeerPinningError, "changed after peer pinning"):
install_pinned_smb_transport(smbclient)
if __name__ == "__main__":
unittest.main()
+145
View File
@@ -0,0 +1,145 @@
from __future__ import annotations
from types import SimpleNamespace
import unittest
from sqlalchemy import create_engine
from sqlalchemy.orm import Session
from govoplan_access.backend.db.models import Account, Group, User
from govoplan_core.auth import ApiPrincipal
from govoplan_core.core.access import PrincipalRef
from govoplan_core.core.change_sequence import ChangeSequenceEntry
from govoplan_core.core.events import EventObjectRef, EventTenantRef, PlatformEvent
from govoplan_core.core.search import (
SearchAuthorizationRequest,
SearchBackfillRequest,
SearchResourceReference,
)
from govoplan_core.db.base import Base
from govoplan_files.backend.db.models import FileAsset, FileFolder, FileShare
from govoplan_files.backend.search_source import (
FilesSearchSource,
PROVIDER_ID,
)
class FilesSearchSourceTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite://")
Base.metadata.create_all(
self.engine,
tables=(
Account.__table__,
User.__table__,
Group.__table__,
FileAsset.__table__,
FileFolder.__table__,
FileShare.__table__,
ChangeSequenceEntry.__table__,
),
)
self.session = Session(self.engine)
self.session.add_all(
(
Account(
id="account-1",
email="one@example.test",
normalized_email="one@example.test",
),
User(
id="user-1",
tenant_id="tenant-1",
account_id="account-1",
email="one@example.test",
),
FileAsset(
id="file-1",
tenant_id="tenant-1",
owner_type="user",
owner_user_id="user-1",
display_path="records/permit.pdf",
filename="permit.pdf",
description="Monthly permit evidence",
),
FileAsset(
id="file-other",
tenant_id="tenant-2",
owner_type="user",
display_path="other.pdf",
filename="other.pdf",
),
)
)
self.session.commit()
self.source = FilesSearchSource()
def tearDown(self) -> None:
self.session.close()
self.engine.dispose()
def test_backfill_and_authorization_are_tenant_bounded(self) -> None:
page = self.source.backfill(
self.session,
request=SearchBackfillRequest(
tenant_id="tenant-1",
provider_id=PROVIDER_ID,
resource_type="file",
rebuild_id="rebuild-1",
),
)
self.assertEqual(("file-1",), tuple(doc.resource_id for doc in page.documents))
reference = SearchResourceReference(
tenant_id="tenant-1",
module_id="files",
resource_type="file",
resource_id="file-1",
)
request = SearchAuthorizationRequest(reference=reference, source_revision="1")
self.assertTrue(
self.source.authorize(
self.session,
_principal({"files:file:read"}),
requests=(request,),
)[reference.key]
)
self.assertFalse(
self.source.authorize(
self.session,
_principal(set()),
requests=(request,),
)[reference.key]
)
def test_committed_file_event_produces_authoritative_upsert(self) -> None:
event = PlatformEvent(
type="files.file.updated",
module_id="files",
tenant=EventTenantRef(id="tenant-1"),
resource=EventObjectRef(type="file", id="file-1"),
)
changes = self.source.index_changes_for_event(
self.session,
event=event,
delivery_key="delivery-1",
)
self.assertEqual(1, len(changes))
self.assertEqual("upsert", changes[0].kind)
self.assertEqual(event.event_id, changes[0].cursor)
def _principal(scopes: set[str]) -> ApiPrincipal:
return ApiPrincipal(
principal=PrincipalRef(
account_id="account-1",
membership_id="user-1",
tenant_id="tenant-1",
scopes=frozenset(scopes),
),
account=SimpleNamespace(id="account-1"),
user=SimpleNamespace(id="user-1"),
)
if __name__ == "__main__":
unittest.main()
+237
View File
@@ -0,0 +1,237 @@
from __future__ import annotations
import unittest
from datetime import timedelta
from unittest.mock import patch
from sqlalchemy import create_engine
from sqlalchemy.dialects import postgresql
from sqlalchemy.orm import sessionmaker
from govoplan_access.backend.db.models import Account, Group, User
from govoplan_core.core.change_sequence import ChangeSequenceEntry
from govoplan_core.db.base import Base
from govoplan_files.backend.db.models import FileAsset, FileShare
from govoplan_files.backend.storage.common import FileStorageError, utcnow
from govoplan_files.backend.storage.files import (
_asset_visibility_query_for_user,
get_asset_for_user,
list_file_shares,
revoke_file_share,
share_file,
)
TENANT_ID = "tenant-1"
OWNER_ID = "owner-1"
RECIPIENT_ID = "recipient-1"
GROUP_ID = "group-1"
class FileShareLifecycleTests(unittest.TestCase):
def setUp(self) -> None:
self.engine = create_engine("sqlite:///:memory:", future=True)
Base.metadata.create_all(
bind=self.engine,
tables=[
Account.__table__,
User.__table__,
Group.__table__,
ChangeSequenceEntry.__table__,
FileAsset.__table__,
FileShare.__table__,
],
)
self.session = sessionmaker(bind=self.engine, future=True)()
self.asset = FileAsset(
id="file-1",
tenant_id=TENANT_ID,
owner_type="user",
owner_user_id=OWNER_ID,
display_path="shared.pdf",
filename="shared.pdf",
)
self.session.add(self.asset)
self.session.commit()
def tearDown(self) -> None:
self.session.close()
self.engine.dispose()
@patch(
"govoplan_files.backend.storage.files.user_group_ids",
return_value=[],
)
def test_expired_share_stops_access_immediately(self, _groups) -> None:
self.session.add(
FileShare(
id="expired-share",
tenant_id=TENANT_ID,
file_asset_id=self.asset.id,
target_type="user",
target_id=RECIPIENT_ID,
permission="read",
expires_at=utcnow() - timedelta(seconds=1),
)
)
self.session.commit()
with self.assertRaisesRegex(FileStorageError, "No access"):
get_asset_for_user(
self.session,
tenant_id=TENANT_ID,
user_id=RECIPIENT_ID,
asset_id=self.asset.id,
)
self.assertEqual(
[],
list_file_shares(
self.session,
tenant_id=TENANT_ID,
asset_id=self.asset.id,
),
)
@patch(
"govoplan_files.backend.storage.files.user_group_ids",
return_value=[GROUP_ID],
)
def test_independent_active_grant_survives_other_expiry(self, _groups) -> None:
self.session.add_all(
[
FileShare(
id="expired-user-share",
tenant_id=TENANT_ID,
file_asset_id=self.asset.id,
target_type="user",
target_id=RECIPIENT_ID,
permission="read",
expires_at=utcnow() - timedelta(seconds=1),
),
FileShare(
id="active-group-share",
tenant_id=TENANT_ID,
file_asset_id=self.asset.id,
target_type="group",
target_id=GROUP_ID,
permission="read",
expires_at=utcnow() + timedelta(hours=1),
),
]
)
self.session.commit()
result = get_asset_for_user(
self.session,
tenant_id=TENANT_ID,
user_id=RECIPIENT_ID,
asset_id=self.asset.id,
)
self.assertEqual(self.asset.id, result.id)
self.assertEqual(
["active-group-share"],
[
share.id
for share in list_file_shares(
self.session,
tenant_id=TENANT_ID,
asset_id=self.asset.id,
)
],
)
@patch(
"govoplan_files.backend.storage.files.ensure_share_target_exists",
return_value=None,
)
def test_revoke_is_idempotent_and_future_expiry_is_persisted(
self, _target_exists
) -> None:
expiry = utcnow() + timedelta(days=1)
share = share_file(
self.session,
tenant_id=TENANT_ID,
asset=self.asset,
target_type="user",
target_id=RECIPIENT_ID,
permission="read",
user_id=OWNER_ID,
expires_at=expiry,
)
self.session.commit()
revoked, first_changed = revoke_file_share(
self.session,
tenant_id=TENANT_ID,
asset_id=self.asset.id,
share_id=share.id,
user_id=OWNER_ID,
)
self.session.commit()
repeated, second_changed = revoke_file_share(
self.session,
tenant_id=TENANT_ID,
asset_id=self.asset.id,
share_id=share.id,
user_id=OWNER_ID,
)
self.assertTrue(first_changed)
self.assertFalse(second_changed)
self.assertEqual(OWNER_ID, revoked.revoked_by_user_id)
self.assertEqual(revoked.revoked_at, repeated.revoked_at)
self.assertEqual([], list_file_shares(
self.session,
tenant_id=TENANT_ID,
asset_id=self.asset.id,
))
self.assertEqual(
[share.id],
[
item.id
for item in list_file_shares(
self.session,
tenant_id=TENANT_ID,
asset_id=self.asset.id,
include_inactive=True,
)
],
)
@patch(
"govoplan_files.backend.storage.files.ensure_share_target_exists",
return_value=None,
)
def test_past_expiry_is_rejected(self, _target_exists) -> None:
with self.assertRaisesRegex(FileStorageError, "future"):
share_file(
self.session,
tenant_id=TENANT_ID,
asset=self.asset,
target_type="user",
target_id=RECIPIENT_ID,
permission="read",
user_id=OWNER_ID,
expires_at=utcnow() - timedelta(seconds=1),
)
@patch(
"govoplan_files.backend.storage.files.user_group_ids",
return_value=[],
)
def test_visibility_query_is_postgresql_json_safe(self, _groups) -> None:
query = _asset_visibility_query_for_user(
self.session,
tenant_id=TENANT_ID,
user_id=RECIPIENT_ID,
)
compiled = str(query.statement.compile(dialect=postgresql.dialect()))
self.assertNotIn("SELECT DISTINCT", compiled.upper())
self.assertIn("EXISTS", compiled.upper())
if __name__ == "__main__":
unittest.main()
+127
View File
@@ -0,0 +1,127 @@
from __future__ import annotations
import io
from types import ModuleType
import sys
import unittest
from unittest.mock import MagicMock, PropertyMock, patch
from govoplan_files.backend.storage.backends import S3StorageBackend, StorageBackendError
def _backend(
*,
endpoint_url: str = "https://objects.example.test",
deployment_managed: bool = False,
) -> S3StorageBackend:
return S3StorageBackend(
bucket="files",
endpoint_url=endpoint_url,
region_name="test",
access_key_id="access",
secret_access_key="secret",
deployment_managed=deployment_managed,
)
class S3StorageBackendTests(unittest.TestCase):
def test_sdk_transport_fails_closed_in_public_and_private_modes(self) -> None:
for allow_private, address in ((False, "93.184.216.34"), (True, "10.0.0.5")):
with self.subTest(allow_private=allow_private), patch.dict(
"os.environ",
{
"APP_ENV": "production",
"GOVOPLAN_CONNECTOR_ALLOW_PRIVATE_NETWORKS": str(allow_private).lower(),
},
), patch(
"govoplan_core.security.outbound_http.socket.getaddrinfo",
return_value=[(2, 1, 6, "", (address, 443))],
), self.assertRaisesRegex(StorageBackendError, "until that transport supports.*DNS/IP pinning"):
_backend().client
def test_installer_managed_garage_uses_exact_endpoint_and_path_style(self) -> None:
boto3 = ModuleType("boto3")
boto3.client = MagicMock(return_value=object())
botocore = ModuleType("botocore")
botocore.__path__ = []
botocore_config = ModuleType("botocore.config")
class Config:
def __init__(self, **values):
self.values = values
botocore_config.Config = Config
with patch.dict(
sys.modules,
{
"boto3": boto3,
"botocore": botocore,
"botocore.config": botocore_config,
},
):
_backend(
endpoint_url="http://garage:3900",
deployment_managed=True,
).client
_args, kwargs = boto3.client.call_args
self.assertEqual("http://garage:3900", kwargs["endpoint_url"])
self.assertEqual(
{"s3": {"addressing_style": "path"}},
kwargs["config"].values,
)
def test_installer_managed_garage_rejects_any_other_endpoint(self) -> None:
with self.assertRaisesRegex(
StorageBackendError,
"restricted to http://garage:3900",
):
_backend(
endpoint_url="http://other-s3:3900",
deployment_managed=True,
).client
def test_get_bytes_rejects_declared_oversize_object_without_reading(self) -> None:
body = MagicMock()
client = MagicMock()
client.get_object.return_value = {"ContentLength": 6, "Body": body}
with patch.dict("os.environ", {"GOVOPLAN_CONNECTOR_MAX_FILE_TRANSFER_BYTES": "5"}), patch.object(
S3StorageBackend,
"client",
new_callable=PropertyMock,
return_value=client,
), self.assertRaisesRegex(StorageBackendError, "deployment limit"):
_backend().get_bytes("large.bin")
body.read.assert_not_called()
body.close.assert_called_once_with()
def test_iter_bytes_rejects_undeclared_oversize_object(self) -> None:
client = MagicMock()
client.get_object.return_value = {"Body": io.BytesIO(b"123456")}
with patch.dict("os.environ", {"GOVOPLAN_CONNECTOR_MAX_FILE_TRANSFER_BYTES": "5"}), patch.object(
S3StorageBackend,
"client",
new_callable=PropertyMock,
return_value=client,
), self.assertRaisesRegex(StorageBackendError, "deployment limit"):
list(_backend().iter_bytes("large.bin", chunk_size=3))
def test_iter_bytes_closes_streaming_body_when_consumer_stops_early(self) -> None:
body = MagicMock()
body.read.side_effect = (b"123", b"456", b"")
client = MagicMock()
client.get_object.return_value = {"ContentLength": 6, "Body": body}
with patch.dict("os.environ", {"GOVOPLAN_CONNECTOR_MAX_FILE_TRANSFER_BYTES": "10"}), patch.object(
S3StorageBackend,
"client",
new_callable=PropertyMock,
return_value=client,
):
chunks = _backend().iter_bytes("object.bin", chunk_size=3)
self.assertEqual(b"123", next(chunks))
chunks.close()
body.close.assert_called_once_with()
if __name__ == "__main__":
unittest.main()
+326
View File
@@ -0,0 +1,326 @@
from __future__ import annotations
import hashlib
from pathlib import Path
import tempfile
import unittest
from unittest.mock import patch
from sqlalchemy import create_engine
from sqlalchemy.orm import sessionmaker
from govoplan_access.backend.db.models import Account, Group, User
from govoplan_core.core.recovery import (
RecoveryCheckpoint,
RecoveryOperation,
RecoveryStatus,
)
from govoplan_core.core.runtime_coordination import (
DistributedLease,
RuntimeIdentity,
bind_process_runtime_identity,
)
from govoplan_core.db.base import Base
from govoplan_core.db.session import configure_database, reset_database
from govoplan_files.backend.db.models import (
FileBlob,
FileIntegrityFinding,
FileIntegrityScan,
)
from govoplan_files.backend.storage.backends import (
LocalFilesystemStorageBackend,
)
from govoplan_files.backend.storage.common import FileStorageError
from govoplan_files.backend.storage.files import _get_or_create_blob
from govoplan_files.backend.storage.integrity import cleanup_orphan_finding
from govoplan_files.backend.storage.recovery import begin_blob_write_recovery
TENANT_ID = "tenant-1"
USER_ID = "user-1"
class StorageRecoveryTests(unittest.TestCase):
def setUp(self) -> None:
self.temporary_directory = tempfile.TemporaryDirectory()
self.addCleanup(self.temporary_directory.cleanup)
root = Path(self.temporary_directory.name)
self.backend = LocalFilesystemStorageBackend(root / "objects")
database_path = root / "recovery.sqlite3"
self.engine = create_engine(f"sqlite:///{database_path}", future=True)
Base.metadata.create_all(
bind=self.engine,
tables=[
Account.__table__,
User.__table__,
Group.__table__,
DistributedLease.__table__,
RecoveryOperation.__table__,
RecoveryCheckpoint.__table__,
FileBlob.__table__,
FileIntegrityScan.__table__,
FileIntegrityFinding.__table__,
],
)
configure_database(
f"sqlite:///{database_path}",
engine=self.engine,
dispose_previous=True,
)
bind_process_runtime_identity(
RuntimeIdentity(
installation_id="files-recovery-test",
node_id="node-1",
incarnation="incarnation-1",
role="api",
software_version="test",
composition_hash="a" * 64,
)
)
self.Session = sessionmaker(
bind=self.engine,
expire_on_commit=False,
future=True,
)
self.enterContext(
patch(
"govoplan_files.backend.storage.files._storage_backend_name",
return_value=self.backend.name,
)
)
self.enterContext(
patch(
"govoplan_files.backend.storage.files._storage_bucket_name",
return_value="",
)
)
self.session = self.Session()
self.addCleanup(self._cleanup_runtime)
def _cleanup_runtime(self) -> None:
self.session.close()
bind_process_runtime_identity(None)
reset_database()
self.engine.dispose()
def test_committed_blob_write_is_independently_verified(self) -> None:
observed_running_operation: list[bool] = []
put_bytes = self.backend.put_bytes
class ObservingBackend:
name = self.backend.name
def __getattr__(backend_self, name):
return getattr(self.backend, name)
def put_bytes(backend_self, key, data, *, content_type=None):
with self.Session() as evidence_session:
operation = evidence_session.query(RecoveryOperation).one()
observed_running_operation.append(
operation.status == RecoveryStatus.RUNNING.value
and len(operation.request_sha256) == 64
)
put_bytes(key, data, content_type=content_type)
observing_backend = ObservingBackend()
with patch(
"govoplan_files.backend.storage.files.get_storage_backend",
return_value=observing_backend,
):
blob = _get_or_create_blob(
self.session,
tenant_id=TENANT_ID,
data=b"durable",
filename="private-name.txt",
content_type="text/plain",
actor_id=USER_ID,
)
self.session.commit()
operation = self._only_operation()
self.assertEqual(RecoveryStatus.SUCCEEDED.value, operation.status)
self.assertEqual([True], observed_running_operation)
self.assertTrue(self.backend.exists(blob.storage_key))
self.assertNotIn("private-name", blob.storage_key)
self.assertEqual(".blob", Path(blob.storage_key).suffix)
def test_rolled_back_blob_write_is_compensated(self) -> None:
with patch(
"govoplan_files.backend.storage.files.get_storage_backend",
return_value=self.backend,
):
blob = _get_or_create_blob(
self.session,
tenant_id=TENANT_ID,
data=b"rollback",
filename="rollback.txt",
content_type="text/plain",
actor_id=USER_ID,
)
storage_key = blob.storage_key
self.assertTrue(self.backend.exists(storage_key))
self.session.rollback()
operation = self._only_operation()
self.assertEqual(RecoveryStatus.RECOVERED.value, operation.status)
self.assertFalse(self.backend.exists(storage_key))
with self.Session() as evidence_session:
self.assertIsNone(evidence_session.get(FileBlob, blob.id))
def test_post_write_tamper_is_quarantined_and_recovery_required(self) -> None:
with patch(
"govoplan_files.backend.storage.files.get_storage_backend",
return_value=self.backend,
):
blob = _get_or_create_blob(
self.session,
tenant_id=TENANT_ID,
data=b"expected",
filename="evidence.bin",
content_type="application/octet-stream",
actor_id=USER_ID,
)
self.backend.put_bytes(blob.storage_key, b"tampered")
self.session.commit()
operation = self._only_operation()
self.assertEqual(
RecoveryStatus.RECOVERY_REQUIRED.value,
operation.status,
)
with self.Session() as evidence_session:
persisted = evidence_session.get(FileBlob, blob.id)
self.assertIsNotNone(persisted)
self.assertEqual("checksum_mismatch", persisted.integrity_status)
self.assertIsNotNone(persisted.quarantined_at)
def test_missing_optional_encryption_fails_before_object_effect(self) -> None:
with patch(
"govoplan_files.backend.storage.files.get_storage_backend",
return_value=self.backend,
), patch(
"govoplan_files.backend.storage.content_protection.encryption_content_cipher",
return_value=None,
), self.assertRaisesRegex(FileStorageError, "Encryption module"):
_get_or_create_blob(
self.session,
tenant_id=TENANT_ID,
data=b"protected",
filename="protected.bin",
content_type="application/octet-stream",
actor_id=USER_ID,
encryption_vault_id="vault-1",
)
self.session.rollback()
operation = self._only_operation()
self.assertEqual(RecoveryStatus.REJECTED.value, operation.status)
objects = self.backend.list_objects(
prefix=f"tenants/{TENANT_ID}/files/",
limit=10,
)
self.assertEqual((), objects.objects)
def test_orphan_cleanup_forward_completes_after_business_rollback(self) -> None:
key = f"tenants/{TENANT_ID}/files/orphan.bin"
self.backend.put_bytes(key, b"orphan")
scan = FileIntegrityScan(
id="scan-1",
tenant_id=TENANT_ID,
storage_backend=self.backend.name,
storage_prefix=f"tenants/{TENANT_ID}/files/",
status="completed",
)
finding = FileIntegrityFinding(
id="finding-1",
scan_id=scan.id,
tenant_id=TENANT_ID,
kind="orphan_object",
state="open",
storage_key=key,
observed_size_bytes=6,
observed_checksum_sha256=hashlib.sha256(b"orphan").hexdigest(),
)
self.session.add_all([scan, finding])
self.session.commit()
cleanup_orphan_finding(
self.session,
finding,
user_id=USER_ID,
dry_run=False,
backend=self.backend,
)
self.assertFalse(self.backend.exists(key))
self.session.rollback()
operation = self._only_operation()
self.assertEqual(RecoveryStatus.SUCCEEDED.value, operation.status)
with self.Session() as evidence_session:
persisted = evidence_session.get(FileIntegrityFinding, finding.id)
self.assertIsNotNone(persisted)
self.assertEqual("deleted", persisted.state)
def test_missing_runtime_identity_blocks_before_object_write(self) -> None:
bind_process_runtime_identity(None)
with patch(
"govoplan_files.backend.storage.files.get_storage_backend",
return_value=self.backend,
), self.assertRaisesRegex(FileStorageError, "recovery ledger"):
_get_or_create_blob(
self.session,
tenant_id=TENANT_ID,
data=b"blocked",
filename="blocked.bin",
content_type="application/octet-stream",
actor_id=USER_ID,
)
self.session.rollback()
objects = self.backend.list_objects(
prefix=f"tenants/{TENANT_ID}/files/",
limit=10,
)
self.assertEqual((), objects.objects)
def test_blob_fence_blocks_a_competing_runtime_before_effect(self) -> None:
checksum = hashlib.sha256(b"fenced").hexdigest()
begin_blob_write_recovery(
self.session,
backend=self.backend,
tenant_id=TENANT_ID,
blob_id="blob-fenced",
storage_key=f"tenants/{TENANT_ID}/files/fenced.blob",
semantic_checksum_sha256=checksum,
semantic_size_bytes=6,
protection_discriminator="plaintext",
created_new=True,
)
with self.Session() as competing_session, self.assertRaisesRegex(
FileStorageError,
"already owned",
):
begin_blob_write_recovery(
competing_session,
backend=self.backend,
tenant_id=TENANT_ID,
blob_id="blob-fenced",
storage_key=f"tenants/{TENANT_ID}/files/fenced.blob",
semantic_checksum_sha256=checksum,
semantic_size_bytes=6,
protection_discriminator="plaintext",
created_new=True,
)
self.session.rollback()
self.assertEqual(RecoveryStatus.REJECTED.value, self._only_operation().status)
def _only_operation(self) -> RecoveryOperation:
with self.Session() as session:
operations = session.query(RecoveryOperation).all()
self.assertEqual(1, len(operations))
session.expunge(operations[0])
return operations[0]
if __name__ == "__main__":
unittest.main()
+108
View File
@@ -0,0 +1,108 @@
from __future__ import annotations
import unittest
from sqlalchemy import create_engine, event
from sqlalchemy.orm import Session
from govoplan_access.backend.db.models import Account, Group, User
from govoplan_core.core.change_sequence import ChangeSequenceEntry
from govoplan_core.db.base import Base
from govoplan_files.backend.db.models import (
FileAsset,
FileConnectorCredential,
FileConnectorPolicy,
FileConnectorProfile,
FileConnectorSpace,
)
from govoplan_files.backend.manifest import _tenant_summary_batch
class FilesTenantSummaryBatchTests(unittest.TestCase):
def test_batch_summary_uses_one_grouped_query_per_owned_table(self) -> None:
engine = create_engine("sqlite+pysqlite:///:memory:")
Base.metadata.create_all(
engine,
tables=[
Account.__table__,
User.__table__,
Group.__table__,
FileAsset.__table__,
FileConnectorCredential.__table__,
FileConnectorPolicy.__table__,
FileConnectorProfile.__table__,
FileConnectorSpace.__table__,
ChangeSequenceEntry.__table__,
],
)
try:
with Session(engine) as session:
session.add_all(
[
FileAsset(
id="file-1",
tenant_id="tenant-1",
owner_type="tenant",
display_path="/one.txt",
filename="one.txt",
),
FileConnectorCredential(
id="credential-1",
tenant_id="tenant-1",
label="Credential",
),
FileConnectorPolicy(
id="policy-1",
tenant_id="tenant-1",
scope_type="tenant",
),
FileConnectorProfile(
id="profile-1",
tenant_id="tenant-1",
label="Profile",
provider="webdav",
),
FileConnectorSpace(
id="space-1",
tenant_id="tenant-1",
owner_type="tenant",
label="Space",
connector_profile_id="profile-1",
provider="webdav",
),
]
)
session.commit()
query_count = 0
def count_query(*_args: object) -> None:
nonlocal query_count
query_count += 1
event.listen(engine, "before_cursor_execute", count_query)
try:
counts = _tenant_summary_batch(
session,
["tenant-1", "tenant-empty"],
)
finally:
event.remove(engine, "before_cursor_execute", count_query)
self.assertEqual(5, query_count)
self.assertEqual(
{
"files": 1,
"connector_credentials": 1,
"connector_policies": 1,
"connector_profiles": 1,
"connector_spaces": 1,
},
counts["tenant-1"],
)
self.assertEqual(0, counts["tenant-empty"]["files"])
finally:
engine.dispose()
if __name__ == "__main__":
unittest.main()
+67
View File
@@ -0,0 +1,67 @@
from __future__ import annotations
import unittest
from govoplan_files.backend.storage.common import FileConflictResolution, FileStorageError
from govoplan_files.backend.storage.transfers import _normalize_rename_request, _normalize_transfer_request
class TransferHelperTests(unittest.TestCase):
def test_transfer_normalization_deduplicates_folder_paths_and_conflicts(self) -> None:
normalized = _normalize_transfer_request(
operation=" COPY ",
source_owner_type=" USER ",
target_owner_type=" GROUP ",
folder_paths=["Reports", "Reports/2026", "", "./Archive"],
target_folder=" Target ",
conflict_strategy="rename",
conflict_resolutions=[FileConflictResolution(target_path="Target/a.txt", action="skip")],
)
self.assertEqual("copy", normalized.operation)
self.assertEqual("user", normalized.source_owner_type)
self.assertEqual("group", normalized.target_owner_type)
self.assertEqual(["Reports", "Reports/2026", "Archive"], normalized.source_folder_paths)
self.assertEqual("Target", normalized.target_folder)
self.assertEqual("rename", normalized.conflict_strategy)
self.assertEqual("skip", normalized.conflict_resolution_map["Target/a.txt"].action)
def test_transfer_normalization_rejects_unknown_operation(self) -> None:
with self.assertRaisesRegex(FileStorageError, "Unsupported transfer operation"):
_normalize_transfer_request(
operation="link",
source_owner_type="user",
target_owner_type="group",
folder_paths=[],
target_folder="",
conflict_strategy="reject",
conflict_resolutions=None,
)
def test_rename_normalization_collapses_nested_folder_roots(self) -> None:
normalized = _normalize_rename_request(
file_ids=["file-1", "file-1", "file-2"],
folder_paths=["Reports", "Reports/2026", "Archive"],
owner_type=" USER ",
owner_id="owner-1",
mode=" prefix ",
)
self.assertEqual("prefix", normalized.mode)
self.assertEqual(["file-1", "file-2"], normalized.selected_file_ids)
self.assertEqual(["Archive", "Reports"], normalized.selected_folder_roots)
self.assertEqual("user", normalized.owner_type)
def test_rename_normalization_rejects_invalid_direct_multi_select(self) -> None:
with self.assertRaisesRegex(FileStorageError, "Direct rename requires exactly one selected item"):
_normalize_rename_request(
file_ids=["file-1", "file-2"],
folder_paths=[],
owner_type="user",
owner_id="owner-1",
mode="direct",
)
if __name__ == "__main__":
unittest.main()
+12 -7
View File
@@ -1,6 +1,6 @@
{
"name": "@govoplan/files-webui",
"version": "0.1.8",
"version": "0.1.16",
"private": true,
"type": "module",
"main": "src/index.ts",
@@ -13,15 +13,20 @@
},
"./styles/file-manager.css": "./src/styles/file-manager.css"
},
"scripts": {
"test:file-drop-target": "node scripts/test-file-drop-target-structure.mjs",
"test:file-property-filters": "node scripts/test-file-property-filters-structure.mjs",
"test:interface-pattern-language": "node scripts/test-interface-pattern-language.mjs"
},
"peerDependencies": {
"@vitejs/plugin-react": "^4.3.4",
"vite": "^6.0.6",
"@vitejs/plugin-react": "^5.2.0",
"vite": "^7.3.6",
"typescript": "^5.7.2",
"react": "^19.0.0",
"react-dom": "^19.0.0",
"react-router-dom": "^7.1.1",
"react": ">=19.2.7 <20",
"react-dom": ">=19.2.7 <20",
"react-router": ">=8.3.0 <9",
"lucide-react": "^1.23.0",
"@govoplan/core-webui": "^0.1.8"
"@govoplan/core-webui": "^0.1.16"
},
"peerDependenciesMeta": {
"@govoplan/core-webui": {
@@ -0,0 +1,35 @@
import { readFileSync } from "node:fs";
const source = readFileSync(new URL("../src/features/files/FilesPage.tsx", import.meta.url), "utf8");
const normalized = source.replace(/\s+/g, " ");
function assertIncludes(fragment, message) {
if (!normalized.includes(fragment.replace(/\s+/g, " "))) throw new Error(message);
}
assertIncludes(
"const target = options.target ?? currentActionTarget();",
"uploads must prefer the explicit target supplied by the initiating interaction"
);
assertIncludes(
"const targetSpace = target ? findSpace(target.spaceId) : null;",
"the upload owner must be resolved from the selected target"
);
assertIncludes(
"owner_type: targetSpace.owner_type, owner_id: targetSpace.owner_id, path: target.folderPath,",
"the upload request must use the selected target owner and folder"
);
assertIncludes(
"async function uploadExternalFilesToTarget(fileList: FileList | File[], target: FileActionTarget) { await handleFilesUpload(fileList, { target }); }",
"external drops must pass their concrete target without asynchronous dialog-state mutation"
);
assertIncludes(
"await uploadExternalFilesToTarget(event.dataTransfer.files, target);",
"drop handling must forward the target on which the files were dropped"
);
assertIncludes(
"onFiles={(files) => handleFilesUpload(files, { target: activeDialogTarget || undefined })}",
"the dialog picker/drop zone must snapshot its selected target for upload"
);
console.log("Files upload target routing structure is intact.");
@@ -0,0 +1,36 @@
import { readFileSync } from "node:fs";
const page = readFileSync(new URL("../src/features/files/FilesPage.tsx", import.meta.url), "utf8").replace(/\s+/g, " ");
const api = readFileSync(new URL("../src/api/files.ts", import.meta.url), "utf8").replace(/\s+/g, " ");
function assertIncludes(source, fragment, message) {
if (!source.includes(fragment.replace(/\s+/g, " "))) throw new Error(message);
}
assertIncludes(
api,
"campaign_usage?: FileCampaignUsageFilter; audit_relevant?: boolean;",
"the Files API must expose structured campaign and audit filters"
);
assertIncludes(
api,
"do { const response = await listFiles",
"property filtering must traverse every server result page"
);
assertIncludes(
page,
"if (propertyFiltersActive && propertyFilterResults) return propertyFilterResults;",
"server-filtered rows must replace, not narrow, the currently loaded explorer window"
);
assertIncludes(
page,
'<option value="unlinked">Not linked or used</option>',
"the high-priority unlinked campaign-file filter must remain available"
);
assertIncludes(
page,
"setPropertyFilterTotal(response.total);",
"the UI must retain the exact server-side filtered count"
);
console.log("Files property filters cover the complete server result window.");
@@ -0,0 +1,56 @@
import assert from "node:assert/strict";
import { readFileSync } from "node:fs";
import { fileURLToPath } from "node:url";
function read(relativePath) {
return readFileSync(fileURLToPath(new URL(relativePath, import.meta.url)), "utf8");
}
const connector = read("../src/features/files/FileConnectorSettingsPanel.tsx");
const integrity = read("../src/features/files/FileIntegrityPanel.tsx");
const filesApi = read("../src/api/files.ts");
const filesPage = read("../src/features/files/FilesPage.tsx");
const moduleSource = read("../src/module.ts");
const styles = read("../src/styles/file-manager.css");
const migration = read("../../docs/INTERFACE_PATTERN_MIGRATION.md");
assert.match(connector, /ActionBlockerHint,[\s\S]*AdvancedOptionsPanel,[\s\S]*DocumentationHelpLink/);
assert.match(connector, /ConnectionTree,[\s\S]*ConfirmDialog,[\s\S]*Dialog,[\s\S]*LoadingFrame/);
assert.match(connector, /topicId: "files\.governed-connectors-and-provenance"/);
assert.match(connector, /documentation=\{CONNECTOR_DOCUMENTATION\}/);
assert.match(connector, /disabledReason=\{profileSaveDisabledReason\}/);
assert.match(connector, /disabledReason=\{credentialSaveDisabledReason\}/);
assert.match(connector, /disabledReason: mutationBlocker/);
assert.match(connector, /<AdvancedOptionsPanel title="Connection metadata JSON"/);
assert.match(connector, /<AdvancedOptionsPanel title="Credential metadata JSON"/);
assert.doesNotMatch(connector, /window\.(?:alert|confirm)\(/);
assert.match(integrity, /File integrity/);
assert.match(filesApi, /expected_revision/);
assert.match(integrity, /cleanupFileIntegrityFinding\(settings, finding, true\)/);
assert.match(integrity, /<ConfirmDialog[\s\S]*Delete unreferenced storage object/);
assert.match(integrity, /files\.reference\.integrity-recovery-and-fail-closed-transports/);
assert.match(moduleSource, /files\.admin\.tenant-integrity/);
assert.doesNotMatch(integrity, /window\.(?:alert|confirm)\(/);
assert.match(filesPage, /DocumentationHelpLink/);
assert.match(filesPage, /topicId: "files\.workflow\.organize-managed-files"/);
assert.match(filesPage, /disabledReason=\{uploadBlocker\}/);
assert.match(filesPage, /disabledReason=\{deleteBlocker\}/);
assert.match(filesPage, /<ConfirmDialog[\s\S]*tone="danger"/);
assert.match(filesPage, /className="workspace-data-page module-entry-page file-manager-page file-manager-fullscreen files-page"/);
assert.doesNotMatch(filesPage, /window\.(?:alert|confirm)\(/);
assert.doesNotMatch(`${connector}\n${integrity}\n${filesPage}\n${moduleSource}`, /@govoplan\/(?:campaign|mail|docs)-webui|govoplan_(?:campaign|mail|docs)/);
assert.match(moduleSource, /"files\.connectors"/);
assert.match(moduleSource, /"files\.fileExplorer"/);
assert.match(styles, /@media \(max-width: 1050px\)[\s\S]*\.files-page \.file-manager-shell[\s\S]*grid-template-columns: 1fr/);
assert.match(styles, /@media \(max-width: 760px\)[\s\S]*\.file-connector-profile-row[\s\S]*grid-template-columns: 1fr/);
for (const archetype of ["Directory/explorer", "Adaptive create/edit", "Effective-policy editor", "Dashboard widget"]) {
assert.match(migration, new RegExp(archetype.replace("/", "\\/")));
}
assert.match(migration, /Shared `Dialog` owns focus entry, Escape handling and focus return/);
assert.match(migration, /Secret values are[\s\S]*never returned for rendering/);
console.log("Files surfaces satisfy the recorded interface pattern-language contract.");

Some files were not shown because too many files have changed in this diff Show More