Release govoplan-connectors v0.1.26: bound spreadsheet traversal
This commit is contained in:
+1
-1
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|||||||
|
|
||||||
[project]
|
[project]
|
||||||
name = "govoplan-connectors"
|
name = "govoplan-connectors"
|
||||||
version = "0.1.25"
|
version = "0.1.26"
|
||||||
description = "Governed connector catalogue and tabular source capabilities for GovOPlaN."
|
description = "Governed connector catalogue and tabular source capabilities for GovOPlaN."
|
||||||
readme = "README.md"
|
readme = "README.md"
|
||||||
requires-python = ">=3.12"
|
requires-python = ">=3.12"
|
||||||
|
|||||||
@@ -39,7 +39,9 @@ _TRANSLATIONS = {
|
|||||||
"title": "Gesteuerte tabellarische Quellen",
|
"title": "Gesteuerte tabellarische Quellen",
|
||||||
"summary": "Anbieterneutrale Quellenerkennung und begrenzte Lesezugriffe für Dataflow bereitstellen.",
|
"summary": "Anbieterneutrale Quellenerkennung und begrenzte Lesezugriffe für Dataflow bereitstellen.",
|
||||||
"body": (
|
"body": (
|
||||||
"Connectors verantwortet Quellkonfiguration, Zugriffsprüfung, Schemaerkennung, Fingerabdrücke und begrenzte Lesevorgänge. Dataflow speichert nur undurchsichtige Quellreferenzen und erwartete Fingerabdrücke. Jede Quelle weist ihren Live-, Cache-, Datei- oder statischen Modus, einen strukturierten Zustand und unterstützte Projektion, Filterung, Aggregation, Sortierung und Seitennavigation aus. Unveränderliche JSON- und CSV-Snapshots bleiben verfügbar. Verwaltete CSV- und XLSX-Quellen nutzen optional Files, binden eine exakt autorisierte Version, erzwingen Archiv- und Entpackgrenzen und übernehmen neuere Versionen erst nach ausdrücklicher Aktualisierung. Der PostgreSQL-Adapter nutzt eine aktive gesteuerte Konfiguration und eine eingegrenzte Core-Zugangsdatenhülle, liest nur einfache Schema- und Tabellenkennungen und blockiert bei Konfigurations-, Zugangsdaten- oder Schemadrift bis zur geprüften Aktualisierung. Zugangsdaten, Endpunkte, Speicherschlüssel und interne Dateiinhalte werden nie über die Quelle offengelegt."
|
"Connectors verantwortet Quellkonfiguration, Zugriffsprüfung, Schemaerkennung, Fingerabdrücke und begrenzte Lesevorgänge. Dataflow speichert nur undurchsichtige Quellreferenzen und erwartete Fingerabdrücke. Jede Quelle weist ihren Live-, Cache-, Datei- oder statischen Modus, einen strukturierten Zustand und unterstützte Projektion, Filterung, Aggregation, Sortierung und Seitennavigation aus. Unveränderliche JSON- und CSV-Snapshots bleiben verfügbar. Verwaltete CSV- und XLSX-Quellen nutzen optional Files, binden eine exakt autorisierte Version, erzwingen Archiv- und Entpackgrenzen und übernehmen neuere Versionen erst nach ausdrücklicher Aktualisierung. "
|
||||||
|
"XLSX-Lesevorgänge prüfen die tatsächlichen Koordinaten des ausgewählten Arbeitsblatts vor dem Aufbau des Zellrasters: höchstens 500 Spalten und 10.000 Zeilenpositionen nach der Kopfzeile einschließlich leerer Zwischenräume. Unzuverlässige Dimensionsangaben vergrößern weder das Raster noch verbergen sie Zellen; übergroße oder widersprüchliche Koordinaten führen zu einem Validierungsfehler statt zu still abgeschnittenen Daten. "
|
||||||
|
"Der PostgreSQL-Adapter nutzt eine aktive gesteuerte Konfiguration und eine eingegrenzte Core-Zugangsdatenhülle, liest nur einfache Schema- und Tabellenkennungen und blockiert bei Konfigurations-, Zugangsdaten- oder Schemadrift bis zur geprüften Aktualisierung. Zugangsdaten, Endpunkte, Speicherschlüssel und interne Dateiinhalte werden nie über die Quelle offengelegt."
|
||||||
),
|
),
|
||||||
},
|
},
|
||||||
"connectors.sanctions-snapshots": {
|
"connectors.sanctions-snapshots": {
|
||||||
|
|||||||
@@ -126,7 +126,7 @@ from govoplan_connectors.backend.german_documentation import (
|
|||||||
|
|
||||||
|
|
||||||
MODULE_ID = "connectors"
|
MODULE_ID = "connectors"
|
||||||
MODULE_VERSION = "0.1.25"
|
MODULE_VERSION = "0.1.26"
|
||||||
TABULAR_SOURCE_INTERFACE_VERSION = "0.1.0"
|
TABULAR_SOURCE_INTERFACE_VERSION = "0.1.0"
|
||||||
DATASOURCE_ORIGIN_INTERFACE_VERSION = "0.1.0"
|
DATASOURCE_ORIGIN_INTERFACE_VERSION = "0.1.0"
|
||||||
SANCTIONS_SNAPSHOT_INTERFACE_VERSION = "1.0.0"
|
SANCTIONS_SNAPSHOT_INTERFACE_VERSION = "1.0.0"
|
||||||
@@ -1115,7 +1115,9 @@ manifest = ModuleManifest(
|
|||||||
"Immutable JSON/CSV snapshots remain available. Managed CSV/XLSX "
|
"Immutable JSON/CSV snapshots remain available. Managed CSV/XLSX "
|
||||||
"sources use the optional Files capability, pin an exact authorized "
|
"sources use the optional Files capability, pin an exact authorized "
|
||||||
"version, apply archive and expansion limits, and require explicit "
|
"version, apply archive and expansion limits, and require explicit "
|
||||||
"refresh before adopting a newer version. The PostgreSQL adapter uses "
|
"refresh before adopting a newer version. XLSX reads validate actual selected-worksheet coordinates before grid allocation: "
|
||||||
|
"at most 500 columns and 10,000 row positions after the header, including blank gaps. Unreliable declared dimensions "
|
||||||
|
"neither expand the grid nor conceal cells; oversized or inconsistent coordinates fail validation instead of truncating data. The PostgreSQL adapter uses "
|
||||||
"an active governed configuration and scoped Core credential envelope, "
|
"an active governed configuration and scoped Core credential envelope, "
|
||||||
"reflects simple schema/table identifiers, runs read-only bounded "
|
"reflects simple schema/table identifiers, runs read-only bounded "
|
||||||
"projection and pagination, and blocks configuration, credential, or "
|
"projection and pagination, and blocks configuration, credential, or "
|
||||||
|
|||||||
@@ -12,7 +12,10 @@ from decimal import Decimal
|
|||||||
from io import BytesIO
|
from io import BytesIO
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
|
from defusedxml import ElementTree as SafeET
|
||||||
|
from defusedxml.common import DefusedXmlException
|
||||||
from openpyxl import load_workbook
|
from openpyxl import load_workbook
|
||||||
|
from openpyxl.utils.cell import column_index_from_string
|
||||||
from sqlalchemy import (
|
from sqlalchemy import (
|
||||||
JSON,
|
JSON,
|
||||||
BigInteger,
|
BigInteger,
|
||||||
@@ -595,7 +598,12 @@ def _parse_xlsx(
|
|||||||
f"Managed XLSX worksheet {selected_name!r} was not found."
|
f"Managed XLSX worksheet {selected_name!r} was not found."
|
||||||
)
|
)
|
||||||
worksheet = workbook[selected_name]
|
worksheet = workbook[selected_name]
|
||||||
iterator = worksheet.iter_rows(values_only=True)
|
maximum_column = _validate_xlsx_worksheet(worksheet)
|
||||||
|
# Declared dimensions are not authoritative: they can inflate sparse
|
||||||
|
# rows/columns or conceal actual cells. Validate the XML coordinates
|
||||||
|
# before openpyxl synthesizes any missing cells, then ignore dimensions.
|
||||||
|
worksheet.reset_dimensions()
|
||||||
|
iterator = worksheet.iter_rows(max_col=maximum_column, values_only=True)
|
||||||
try:
|
try:
|
||||||
raw_headers = next(iterator)
|
raw_headers = next(iterator)
|
||||||
except StopIteration as exc:
|
except StopIteration as exc:
|
||||||
@@ -604,10 +612,10 @@ def _parse_xlsx(
|
|||||||
) from exc
|
) from exc
|
||||||
headers = _xlsx_headers(raw_headers)
|
headers = _xlsx_headers(raw_headers)
|
||||||
rows: list[Mapping[str, object]] = []
|
rows: list[Mapping[str, object]] = []
|
||||||
for values in iterator:
|
for row_number, values in enumerate(iterator, start=1):
|
||||||
if len(rows) >= MAX_FILE_ROWS:
|
if row_number > MAX_FILE_ROWS:
|
||||||
raise TabularSourceValidationError(
|
raise TabularSourceValidationError(
|
||||||
f"Managed XLSX worksheets are limited to {MAX_FILE_ROWS:,} data rows."
|
f"Managed XLSX worksheets are limited to {MAX_FILE_ROWS:,} row positions after the header, including blank gaps."
|
||||||
)
|
)
|
||||||
normalized = tuple(values[: len(headers)])
|
normalized = tuple(values[: len(headers)])
|
||||||
if all(value in (None, "") for value in normalized):
|
if all(value in (None, "") for value in normalized):
|
||||||
@@ -625,6 +633,60 @@ def _parse_xlsx(
|
|||||||
workbook.close()
|
workbook.close()
|
||||||
|
|
||||||
|
|
||||||
|
def _validate_xlsx_worksheet(worksheet: Any) -> int:
|
||||||
|
"""Bound actual coordinates before read-only openpyxl allocates row tuples.
|
||||||
|
|
||||||
|
The pinned openpyxl 3.x read-only source handle is streamed and closed here;
|
||||||
|
no untrusted dimensions or filesystem paths are used to allocate a grid.
|
||||||
|
"""
|
||||||
|
maximum_column = 1
|
||||||
|
row_position = 0
|
||||||
|
rows_seen = 0
|
||||||
|
column_position = 0
|
||||||
|
cells_seen = 0
|
||||||
|
try:
|
||||||
|
with worksheet._get_source() as source:
|
||||||
|
for event, element in SafeET.iterparse(source, events=("start", "end"), forbid_dtd=True):
|
||||||
|
tag = element.tag.rsplit("}", 1)[-1]
|
||||||
|
if event == "end":
|
||||||
|
element.clear()
|
||||||
|
continue
|
||||||
|
if tag == "row":
|
||||||
|
rows_seen += 1
|
||||||
|
raw_row = element.get("r", str(row_position + 1))
|
||||||
|
if len(raw_row) > 7 or not raw_row.isascii() or not raw_row.isdigit():
|
||||||
|
raise TabularSourceValidationError("Managed XLSX content has an invalid row coordinate.")
|
||||||
|
row_position = int(raw_row)
|
||||||
|
if not 1 <= row_position <= MAX_FILE_ROWS + 1 or rows_seen > MAX_FILE_ROWS + 1:
|
||||||
|
raise TabularSourceValidationError(
|
||||||
|
f"Managed XLSX worksheets are limited to {MAX_FILE_ROWS:,} row positions after the header, including blank gaps."
|
||||||
|
)
|
||||||
|
column_position = 0
|
||||||
|
cells_seen = 0
|
||||||
|
elif tag == "c":
|
||||||
|
cells_seen += 1
|
||||||
|
reference = element.get("r")
|
||||||
|
if reference is not None:
|
||||||
|
match = re.fullmatch(r"([A-Za-z]{1,3})([1-9][0-9]{0,6})", reference)
|
||||||
|
if match is None:
|
||||||
|
raise TabularSourceValidationError("Managed XLSX content has an invalid cell coordinate.")
|
||||||
|
column_position = column_index_from_string(match.group(1))
|
||||||
|
if int(match.group(2)) != row_position:
|
||||||
|
raise TabularSourceValidationError("Managed XLSX cell and row coordinates do not agree.")
|
||||||
|
else:
|
||||||
|
column_position += 1
|
||||||
|
if column_position > MAX_FILE_COLUMNS or cells_seen > MAX_FILE_COLUMNS:
|
||||||
|
raise TabularSourceValidationError(
|
||||||
|
f"Managed XLSX worksheets are limited to {MAX_FILE_COLUMNS:,} columns."
|
||||||
|
)
|
||||||
|
maximum_column = max(maximum_column, column_position)
|
||||||
|
return maximum_column
|
||||||
|
except (DefusedXmlException, SafeET.ParseError, ValueError) as exc:
|
||||||
|
if isinstance(exc, TabularSourceValidationError):
|
||||||
|
raise
|
||||||
|
raise TabularSourceValidationError("Managed XLSX worksheet XML could not be safely parsed.") from exc
|
||||||
|
|
||||||
|
|
||||||
def _validate_xlsx_archive(payload: bytes) -> None:
|
def _validate_xlsx_archive(payload: bytes) -> None:
|
||||||
try:
|
try:
|
||||||
with zipfile.ZipFile(BytesIO(payload)) as archive:
|
with zipfile.ZipFile(BytesIO(payload)) as archive:
|
||||||
|
|||||||
@@ -0,0 +1,70 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import re
|
||||||
|
import unittest
|
||||||
|
from io import BytesIO
|
||||||
|
from unittest.mock import patch
|
||||||
|
from zipfile import ZipFile
|
||||||
|
|
||||||
|
from openpyxl import Workbook
|
||||||
|
|
||||||
|
from govoplan_core.core.tabular_sources import TabularSourceValidationError
|
||||||
|
from govoplan_connectors.backend import tabular_adapters as adapters
|
||||||
|
|
||||||
|
|
||||||
|
def workbook_bytes(*, dimension: str | None = None, last_cell: str = "A2") -> bytes:
|
||||||
|
workbook = Workbook()
|
||||||
|
workbook.active.append(["name"])
|
||||||
|
workbook.active.append(["Ada"])
|
||||||
|
output = BytesIO()
|
||||||
|
workbook.save(output)
|
||||||
|
workbook.close()
|
||||||
|
result = BytesIO()
|
||||||
|
with ZipFile(BytesIO(output.getvalue())) as original, ZipFile(result, "w") as modified:
|
||||||
|
for entry in original.infolist():
|
||||||
|
content = original.read(entry)
|
||||||
|
if entry.filename == "xl/worksheets/sheet1.xml":
|
||||||
|
xml = content.decode()
|
||||||
|
if dimension is not None:
|
||||||
|
xml = re.sub(r'<dimension ref="[^"]+"\s*/>', f'<dimension ref="{dimension}"/>', xml)
|
||||||
|
xml = xml.replace('r="A2"', f'r="{last_cell}"')
|
||||||
|
xml = xml.replace('<row r="2">', f'<row r="{re.search(r"[0-9]+$", last_cell).group()}">')
|
||||||
|
content = xml.encode()
|
||||||
|
modified.writestr(entry, content)
|
||||||
|
return result.getvalue()
|
||||||
|
|
||||||
|
|
||||||
|
def parse(payload: bytes):
|
||||||
|
return adapters.parse_managed_tabular_content(payload, filename="fixture.xlsx", content_type=None, delimiter=",", sheet_name=None)
|
||||||
|
|
||||||
|
|
||||||
|
class XlsxSafetyBoundsTests(unittest.TestCase):
|
||||||
|
def test_sparse_rows_cannot_bypass_limit_by_not_counting_as_data(self):
|
||||||
|
with patch.object(adapters, "MAX_FILE_ROWS", 3):
|
||||||
|
with self.assertRaisesRegex(TabularSourceValidationError, "row"):
|
||||||
|
parse(workbook_bytes(last_cell="A6"))
|
||||||
|
|
||||||
|
def test_forged_small_dimensions_cannot_hide_far_away_cells(self):
|
||||||
|
with patch.object(adapters, "MAX_FILE_ROWS", 3):
|
||||||
|
with self.assertRaisesRegex(TabularSourceValidationError, "row"):
|
||||||
|
parse(workbook_bytes(dimension="A1:A2", last_cell="A1000000"))
|
||||||
|
|
||||||
|
def test_forged_small_dimensions_cannot_hide_out_of_range_columns(self):
|
||||||
|
with self.assertRaisesRegex(TabularSourceValidationError, "column"):
|
||||||
|
parse(workbook_bytes(dimension="A1:A2", last_cell="XFD2"))
|
||||||
|
|
||||||
|
def test_declared_dimensions_do_not_expand_or_truncate_real_rows(self):
|
||||||
|
for dimension in ("A1:XFD1048576", "A1:A1"):
|
||||||
|
with self.subTest(dimension=dimension):
|
||||||
|
rows, sheet = parse(workbook_bytes(dimension=dimension))
|
||||||
|
self.assertEqual(({"name": "Ada"},), rows)
|
||||||
|
self.assertEqual("Sheet", sheet)
|
||||||
|
|
||||||
|
def test_small_blank_gaps_and_exact_limit_remain_usable(self):
|
||||||
|
with patch.object(adapters, "MAX_FILE_ROWS", 3):
|
||||||
|
rows, _sheet = parse(workbook_bytes(last_cell="A4"))
|
||||||
|
self.assertEqual(({"name": "Ada"},), rows)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@govoplan/connectors-webui",
|
"name": "@govoplan/connectors-webui",
|
||||||
"version": "0.1.25",
|
"version": "0.1.26",
|
||||||
"private": true,
|
"private": true,
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"main": "src/index.ts",
|
"main": "src/index.ts",
|
||||||
|
|||||||
Reference in New Issue
Block a user