feat: expose institutional architecture and runtime state
This commit is contained in:
@@ -9,9 +9,11 @@ from typing import Any
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
from fastapi import APIRouter, Depends, HTTPException, Request, status
|
||||
from pydantic import BaseModel, Field
|
||||
from sqlalchemy import text
|
||||
from sqlalchemy.exc import SQLAlchemyError
|
||||
|
||||
from govoplan_core.audit.logging import audit_event
|
||||
from govoplan_core.auth import ApiPrincipal, require_any_scope
|
||||
from govoplan_core.core.maintenance import saved_maintenance_mode
|
||||
from govoplan_core.core.module_installer import (
|
||||
@@ -20,7 +22,20 @@ from govoplan_core.core.module_installer import (
|
||||
read_module_installer_run,
|
||||
)
|
||||
from govoplan_core.core.operations import OperationalCheckProviderRegistration
|
||||
from govoplan_core.core.recovery import (
|
||||
RecoveryOperation,
|
||||
)
|
||||
from govoplan_core.core.registry import PlatformRegistry
|
||||
from govoplan_core.core.provider_governance import (
|
||||
ExternalProviderStateContext,
|
||||
collect_external_provider_states,
|
||||
)
|
||||
from govoplan_core.core.runtime_coordination import (
|
||||
RuntimeCoordinationError,
|
||||
cancel_runtime_node_drain,
|
||||
list_runtime_nodes,
|
||||
request_runtime_node_drain,
|
||||
)
|
||||
from govoplan_core.db.session import get_database
|
||||
from govoplan_core.settings import settings as core_settings
|
||||
|
||||
@@ -33,6 +48,10 @@ _worker_check_cache: tuple[float, dict[str, Any]] | None = None
|
||||
_worker_check_cache_lock = threading.Lock()
|
||||
|
||||
|
||||
class RuntimeDrainRequest(BaseModel):
|
||||
reason: str = Field(default="operator request", min_length=1, max_length=500)
|
||||
|
||||
|
||||
@router.get("/status")
|
||||
def ops_status(
|
||||
request: Request,
|
||||
@@ -64,6 +83,91 @@ def run_ops_checks(
|
||||
return _ops_status_payload(request, force_module_checks=True)
|
||||
|
||||
|
||||
@router.get("/runtime/nodes")
|
||||
def runtime_nodes(
|
||||
principal: ApiPrincipal = Depends(require_any_scope(*OPS_READ_SCOPES)),
|
||||
) -> dict[str, Any]:
|
||||
del principal
|
||||
return _runtime_cluster_status()
|
||||
|
||||
|
||||
@router.post("/runtime/nodes/{node_id}/drain")
|
||||
def drain_runtime_node(
|
||||
node_id: str,
|
||||
body: RuntimeDrainRequest,
|
||||
principal: ApiPrincipal = Depends(require_any_scope(*OPS_RUN_SCOPES)),
|
||||
) -> dict[str, Any]:
|
||||
try:
|
||||
with get_database().SessionLocal() as session:
|
||||
node = request_runtime_node_drain(
|
||||
session,
|
||||
installation_id=core_settings.installation_id,
|
||||
node_id=node_id,
|
||||
reason=body.reason,
|
||||
)
|
||||
audit_event(
|
||||
session,
|
||||
tenant_id=None,
|
||||
scope="system",
|
||||
user_id=getattr(principal.user, "id", None),
|
||||
api_key_id=principal.api_key_id,
|
||||
action="ops.runtime_node.drain_requested",
|
||||
object_type="runtime_node",
|
||||
object_id=node.node_id,
|
||||
details={
|
||||
"installation_id": core_settings.installation_id,
|
||||
"reason": node.drain_reason,
|
||||
"incarnation": node.incarnation,
|
||||
},
|
||||
)
|
||||
session.commit()
|
||||
return {
|
||||
"node_id": node.node_id,
|
||||
"state": node.state,
|
||||
"drain_reason": node.drain_reason,
|
||||
}
|
||||
except RuntimeCoordinationError as exc:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_404_NOT_FOUND,
|
||||
detail=str(exc),
|
||||
) from exc
|
||||
|
||||
|
||||
@router.delete("/runtime/nodes/{node_id}/drain")
|
||||
def cancel_runtime_drain(
|
||||
node_id: str,
|
||||
principal: ApiPrincipal = Depends(require_any_scope(*OPS_RUN_SCOPES)),
|
||||
) -> dict[str, Any]:
|
||||
try:
|
||||
with get_database().SessionLocal() as session:
|
||||
node = cancel_runtime_node_drain(
|
||||
session,
|
||||
installation_id=core_settings.installation_id,
|
||||
node_id=node_id,
|
||||
)
|
||||
audit_event(
|
||||
session,
|
||||
tenant_id=None,
|
||||
scope="system",
|
||||
user_id=getattr(principal.user, "id", None),
|
||||
api_key_id=principal.api_key_id,
|
||||
action="ops.runtime_node.drain_cancelled",
|
||||
object_type="runtime_node",
|
||||
object_id=node.node_id,
|
||||
details={
|
||||
"installation_id": core_settings.installation_id,
|
||||
"incarnation": node.incarnation,
|
||||
},
|
||||
)
|
||||
session.commit()
|
||||
return {"node_id": node.node_id, "state": node.state}
|
||||
except RuntimeCoordinationError as exc:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_404_NOT_FOUND,
|
||||
detail=str(exc),
|
||||
) from exc
|
||||
|
||||
|
||||
def _ops_status_payload(
|
||||
request: Request,
|
||||
*,
|
||||
@@ -75,6 +179,7 @@ def _ops_status_payload(
|
||||
redis_check = _redis_check()
|
||||
current_profile = _current_profile()
|
||||
worker_check = _worker_check()
|
||||
runtime_cluster = _runtime_cluster_status()
|
||||
module_checks = _module_operational_checks(
|
||||
registry,
|
||||
force=force_module_checks,
|
||||
@@ -85,13 +190,17 @@ def _ops_status_payload(
|
||||
_maintenance_check(maintenance_mode),
|
||||
redis_check,
|
||||
worker_check,
|
||||
_runtime_cluster_check(runtime_cluster),
|
||||
_storage_check(),
|
||||
_backup_restore_check(),
|
||||
_deployment_security_check(current_profile),
|
||||
*module_checks,
|
||||
]
|
||||
readiness = _readiness(checks, maintenance_mode)
|
||||
governance = _governance_inventory(registry)
|
||||
governance = _governance_inventory(
|
||||
registry,
|
||||
external_provider_states=_external_provider_states(registry),
|
||||
)
|
||||
return {
|
||||
"summary": {
|
||||
"app_env": core_settings.app_env,
|
||||
@@ -105,15 +214,137 @@ def _ops_status_payload(
|
||||
"file_storage_backend": core_settings.file_storage_backend,
|
||||
"worker_metrics": worker_check.get("metrics", {}),
|
||||
"operational_probe_count": len(module_checks),
|
||||
"runtime_node_count": len(runtime_cluster.get("nodes", [])),
|
||||
"recovery_required_count": runtime_cluster.get("recovery", {}).get("requires_attention", 0),
|
||||
},
|
||||
"readiness": readiness,
|
||||
"checks": checks,
|
||||
"governance": governance,
|
||||
"deployment_profiles": _deployment_profiles(current_profile),
|
||||
"sizing": _sizing_assumptions(),
|
||||
"runtime_cluster": runtime_cluster,
|
||||
}
|
||||
|
||||
|
||||
def _runtime_cluster_status() -> dict[str, Any]:
|
||||
try:
|
||||
with get_database().SessionLocal() as session:
|
||||
nodes = list_runtime_nodes(
|
||||
session,
|
||||
installation_id=core_settings.installation_id,
|
||||
stale_after_seconds=core_settings.runtime_stale_after_seconds,
|
||||
)
|
||||
operations = (
|
||||
session.query(RecoveryOperation)
|
||||
.filter(
|
||||
RecoveryOperation.installation_id
|
||||
== core_settings.installation_id
|
||||
)
|
||||
.order_by(RecoveryOperation.updated_at.desc())
|
||||
.limit(50)
|
||||
.all()
|
||||
)
|
||||
except SQLAlchemyError as exc:
|
||||
return {
|
||||
"available": False,
|
||||
"detail": f"Runtime coordination query failed: {exc}",
|
||||
"state_profile": core_settings.state_profile,
|
||||
"nodes": [],
|
||||
"recovery": {"operations": [], "requires_attention": 0},
|
||||
}
|
||||
active_nodes = [
|
||||
node
|
||||
for node in nodes
|
||||
if node["state"] != "stopped" and not node["stale"]
|
||||
]
|
||||
active_api = sum(node["role"] == "api" for node in active_nodes)
|
||||
active_workers = sum(node["role"] == "worker" for node in active_nodes)
|
||||
operation_payloads = [
|
||||
{
|
||||
"id": operation.id,
|
||||
"module_id": operation.module_id,
|
||||
"operation_type": operation.operation_type,
|
||||
"resource_type": operation.resource_type,
|
||||
"resource_id": operation.resource_id,
|
||||
"mode": operation.mode,
|
||||
"status": operation.status,
|
||||
"checkpoint_count": operation.checkpoint_count,
|
||||
"evidence_head_sha256": operation.evidence_head_sha256,
|
||||
"failure_summary": operation.failure_summary,
|
||||
"updated_at": operation.updated_at.isoformat(),
|
||||
}
|
||||
for operation in operations
|
||||
]
|
||||
requires_attention = sum(
|
||||
operation["status"]
|
||||
in {
|
||||
"outcome_unknown",
|
||||
"recovery_required",
|
||||
"recovering",
|
||||
"manual_intervention",
|
||||
}
|
||||
for operation in operation_payloads
|
||||
)
|
||||
return {
|
||||
"available": True,
|
||||
"detail": "Shared runtime directory is available.",
|
||||
"installation_id": core_settings.installation_id,
|
||||
"state_profile": core_settings.state_profile,
|
||||
"expected": {
|
||||
"api": core_settings.runtime_expected_api_replicas,
|
||||
"worker": core_settings.runtime_expected_worker_replicas,
|
||||
},
|
||||
"active": {"api": active_api, "worker": active_workers},
|
||||
"nodes": nodes,
|
||||
"recovery": {
|
||||
"operations": operation_payloads,
|
||||
"requires_attention": requires_attention,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _runtime_cluster_check(cluster: dict[str, Any]) -> dict[str, Any]:
|
||||
if not cluster.get("available"):
|
||||
return _check(
|
||||
"runtime_cluster",
|
||||
"Runtime cluster",
|
||||
"error",
|
||||
str(cluster.get("detail") or "Runtime directory unavailable."),
|
||||
readiness_critical=True,
|
||||
)
|
||||
nodes = cluster.get("nodes") or []
|
||||
stale = [node for node in nodes if node.get("stale")]
|
||||
expected = cluster.get("expected") or {}
|
||||
active = cluster.get("active") or {}
|
||||
missing_api = max(0, int(expected.get("api") or 0) - int(active.get("api") or 0))
|
||||
missing_workers = max(0, int(expected.get("worker") or 0) - int(active.get("worker") or 0))
|
||||
state = "ok"
|
||||
detail = f"{len(nodes)} runtime node record(s); {len(stale)} stale."
|
||||
readiness_critical = False
|
||||
if missing_api or missing_workers:
|
||||
state = "error" if cluster.get("state_profile") == "shared" else "warning"
|
||||
readiness_critical = state == "error"
|
||||
detail += f" Missing expected replicas: api={missing_api}, worker={missing_workers}."
|
||||
elif stale:
|
||||
state = "warning"
|
||||
return _check(
|
||||
"runtime_cluster",
|
||||
"Runtime cluster",
|
||||
state,
|
||||
detail,
|
||||
readiness_critical=readiness_critical,
|
||||
metrics={
|
||||
"nodes": len(nodes),
|
||||
"stale": len(stale),
|
||||
"active_api": int(active.get("api") or 0),
|
||||
"active_workers": int(active.get("worker") or 0),
|
||||
"recovery_required": int(
|
||||
cluster.get("recovery", {}).get("requires_attention") or 0
|
||||
),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def _registry(request: Request) -> PlatformRegistry:
|
||||
registry = getattr(request.app.state, "govoplan_registry", None)
|
||||
if not isinstance(registry, PlatformRegistry):
|
||||
@@ -121,10 +352,27 @@ def _registry(request: Request) -> PlatformRegistry:
|
||||
return registry
|
||||
|
||||
|
||||
def _governance_inventory(registry: PlatformRegistry) -> dict[str, Any]:
|
||||
def _governance_inventory(
|
||||
registry: PlatformRegistry,
|
||||
*,
|
||||
external_provider_states: Mapping[str, Mapping[str, object]] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
modules: list[dict[str, Any]] = []
|
||||
for manifest in registry.manifests():
|
||||
capability_names = tuple(sorted(manifest.capability_factories))
|
||||
architecture = (
|
||||
manifest.architecture.to_dict() if manifest.architecture else None
|
||||
)
|
||||
external_providers = tuple(
|
||||
{
|
||||
**declaration.to_dict(),
|
||||
"runtime_state": dict(
|
||||
(external_provider_states or {}).get(declaration.id, {})
|
||||
)
|
||||
or None,
|
||||
}
|
||||
for declaration in manifest.external_providers
|
||||
)
|
||||
access_control_count = (
|
||||
len(manifest.resource_acl_providers)
|
||||
+ len(manifest.ownership_providers)
|
||||
@@ -144,6 +392,9 @@ def _governance_inventory(registry: PlatformRegistry) -> dict[str, Any]:
|
||||
"access_control_count": access_control_count,
|
||||
"search_provider_count": len(manifest.search_providers) + len(manifest.search_sources),
|
||||
"migration_managed": manifest.migration_spec is not None,
|
||||
"architecture": architecture,
|
||||
"external_provider_count": len(external_providers),
|
||||
"external_providers": external_providers,
|
||||
})
|
||||
return {
|
||||
"summary": {
|
||||
@@ -159,11 +410,62 @@ def _governance_inventory(registry: PlatformRegistry) -> dict[str, Any]:
|
||||
"access_control_count": sum(item["access_control_count"] for item in modules),
|
||||
"search_provider_count": sum(item["search_provider_count"] for item in modules),
|
||||
"migration_module_count": sum(bool(item["migration_managed"]) for item in modules),
|
||||
"architecture_declared_module_count": sum(
|
||||
bool(item["architecture"]) for item in modules
|
||||
),
|
||||
"external_provider_count": sum(
|
||||
item["external_provider_count"] for item in modules
|
||||
),
|
||||
"configured_external_provider_count": sum(
|
||||
bool(
|
||||
provider.get("runtime_state", {}).get("configured")
|
||||
if isinstance(provider.get("runtime_state"), Mapping)
|
||||
else False
|
||||
)
|
||||
for item in modules
|
||||
for provider in item["external_providers"]
|
||||
),
|
||||
"provider_attention_count": sum(
|
||||
(
|
||||
str(provider.get("runtime_state", {}).get("health"))
|
||||
in {"warning", "error", "unknown"}
|
||||
or str(provider.get("runtime_state", {}).get("recovery"))
|
||||
== "attention"
|
||||
)
|
||||
if isinstance(provider.get("runtime_state"), Mapping)
|
||||
else False
|
||||
for item in modules
|
||||
for provider in item["external_providers"]
|
||||
),
|
||||
"supported_module_count": sum(
|
||||
bool(item["architecture"])
|
||||
and item["architecture"]["maturity"] in {"supported", "lts"}
|
||||
for item in modules
|
||||
),
|
||||
},
|
||||
"modules": modules,
|
||||
}
|
||||
|
||||
|
||||
def _external_provider_states(
|
||||
registry: PlatformRegistry,
|
||||
) -> dict[str, dict[str, object]]:
|
||||
registrations = registry.external_provider_state_providers()
|
||||
if not registrations:
|
||||
return {}
|
||||
try:
|
||||
with get_database().session() as session:
|
||||
return collect_external_provider_states(
|
||||
registrations,
|
||||
ExternalProviderStateContext(session=session),
|
||||
)
|
||||
except Exception: # noqa: BLE001 - preserve Ops inventory during DB outages.
|
||||
return collect_external_provider_states(
|
||||
registrations,
|
||||
ExternalProviderStateContext(session=None),
|
||||
)
|
||||
|
||||
|
||||
def _database_status() -> dict[str, Any]:
|
||||
try:
|
||||
with get_database().session() as session:
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from govoplan_core.core.access import CAPABILITY_AUTH_PERMISSION_EVALUATOR, CAPABILITY_AUTH_PRINCIPAL_RESOLVER
|
||||
from govoplan_core.core.modules import FrontendModule, FrontendRoute, ModuleContext, ModuleManifest, NavItem, PermissionDefinition, RoleTemplate
|
||||
from govoplan_core.core.modules import DocumentationCondition, DocumentationTopic, FrontendModule, FrontendRoute, ModuleContext, ModuleManifest, NavItem, PermissionDefinition, RoleTemplate
|
||||
from govoplan_core.core.provider_governance import ModuleArchitectureDeclaration, ModuleArchitectureDocumentation, ModuleMaturityEvidence
|
||||
from govoplan_core.core.views import ViewSurface
|
||||
|
||||
OPS_READ_SCOPE = "ops:operations:read"
|
||||
@@ -9,6 +10,33 @@ OPS_READ_SCOPES = (OPS_READ_SCOPE, "system:settings:read", "admin:settings:read"
|
||||
OPS_RUN_SCOPE = "ops:operations:run"
|
||||
OPS_RUN_SCOPES = (OPS_RUN_SCOPE, "system:settings:write")
|
||||
|
||||
ARCHITECTURE = ModuleArchitectureDeclaration(
|
||||
layer="runtime_meta",
|
||||
kind="operations",
|
||||
maturity="vertical_slice",
|
||||
evidence=(
|
||||
ModuleMaturityEvidence(
|
||||
kind="test",
|
||||
reference="tests/test_governance_inventory.py",
|
||||
summary="Tests safe operational projection of module governance declarations.",
|
||||
),
|
||||
ModuleMaturityEvidence(
|
||||
kind="documentation",
|
||||
reference="docs/SCALABILITY_PROFILES.md",
|
||||
summary="Documents operational topology and scaling posture.",
|
||||
),
|
||||
),
|
||||
known_limits=(
|
||||
"Provider health observations depend on module-owned operational probes and may be unavailable until configured.",
|
||||
),
|
||||
owned_concepts=("operations status projection", "bounded operational probes"),
|
||||
non_owned_concepts=("domain repair", "external provider credentials"),
|
||||
documentation=ModuleArchitectureDocumentation(
|
||||
recovery=("docs/SCALABILITY_PROFILES.md",),
|
||||
operations=("docs/SCALABILITY_PROFILES.md",),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _permission(scope: str, label: str, description: str) -> PermissionDefinition:
|
||||
module_id, resource, action = scope.split(":", 2)
|
||||
@@ -65,6 +93,48 @@ manifest = ModuleManifest(
|
||||
level="system",
|
||||
),
|
||||
),
|
||||
documentation=(
|
||||
DocumentationTopic(
|
||||
id="ops.health-governance-and-sizing",
|
||||
title="Inspect platform health and deployment posture",
|
||||
summary="Ops combines module-owned health checks with deployment profile, governance inventory, worker assumptions, and sizing guidance.",
|
||||
body="Read-only status distinguishes configured capabilities from healthy integrations. Authorized operators can run bounded probes; a probe must not perform unbounded business work or silently repair data. Use readiness and worker results when diagnosing a node, and use the deployment profile and sizing assumptions when planning horizontal capacity.",
|
||||
documentation_types=("admin", "user"),
|
||||
audience=("operator", "system_admin"),
|
||||
related_modules=("audit", "docs", "notifications"),
|
||||
metadata={"kind": "reference"},
|
||||
),
|
||||
DocumentationTopic(
|
||||
id="ops.runtime-coordination-and-recovery",
|
||||
title="Drain runtime nodes and inspect recovery evidence",
|
||||
summary="Ops projects shared runtime heartbeats, replica gaps, drain controls, and recovery states that require operator attention.",
|
||||
body="Use the runtime table to identify stale or composition-skewed API and worker replicas. Drain before replacement so API readiness closes and workers stop taking new queue work; cancellation is available while the node is still draining. The recovery table reports durable Core recovery operations, but a recorded forward-recovery or manual-intervention state is not an automatic database restore.",
|
||||
documentation_types=("admin", "user"),
|
||||
audience=("operator", "system_admin"),
|
||||
conditions=(
|
||||
DocumentationCondition(
|
||||
required_modules=("ops",),
|
||||
any_scopes=OPS_READ_SCOPES,
|
||||
),
|
||||
),
|
||||
related_modules=("audit", "notifications"),
|
||||
metadata={
|
||||
"kind": "workflow",
|
||||
"route": "/ops",
|
||||
"screen": "Runtime cluster and recovery evidence",
|
||||
"steps": [
|
||||
"Compare active non-stale nodes with the configured API and worker replica expectations.",
|
||||
"Request drain and wait for the node to report draining before replacing it.",
|
||||
"Inspect every recovery-required, outcome-unknown, or manual-intervention operation and follow its recorded recovery mode.",
|
||||
"Verify replacement composition, readiness, queue consumers, and recovery evidence before closing the operation.",
|
||||
],
|
||||
"limitations": [
|
||||
"Drain is observed on the runtime heartbeat interval and does not forcibly terminate active work.",
|
||||
"Ops does not create database backups or make an unsafe post-migration rollback reversible.",
|
||||
],
|
||||
},
|
||||
),
|
||||
),
|
||||
route_factory=_route_factory,
|
||||
nav_items=(NavItem(path="/ops", label="Ops", icon="activity", required_any=OPS_READ_SCOPES, order=890),),
|
||||
frontend=FrontendModule(
|
||||
@@ -76,6 +146,7 @@ manifest = ModuleManifest(
|
||||
ViewSurface(id="ops.widget.health", module_id="ops", kind="section", label="Operations health widget", order=100),
|
||||
),
|
||||
),
|
||||
architecture=ARCHITECTURE,
|
||||
)
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user