From 49da0f9df7e3d28c0248da6181ed4903bf7b0196 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?T=C3=B5nis=20Ormisson?= Date: Mon, 7 Sep 2026 17:15:43 +0300 Subject: [PATCH 1/4] Make reads and SAV export database-read-only --- README.md | 7 +- docs/sav-profile.md | 2 +- src/openstatspec/api.py | 4 +- src/openstatspec/spss/sav.py | 292 +-------------------- src/openstatspec/sql/capabilities.py | 22 +- src/openstatspec/sql/wide.py | 302 +--------------------- src/openstatspec/sql/workflow.py | 10 +- tests/test_catalog_lifecycle.py | 56 ---- tests/test_catalog_persistence_review.py | 90 ------- tests/test_dolt_conformance.py | 44 +--- tests/test_loss_reports.py | 17 +- tests/test_normative_catalog.py | 18 +- tests/test_read_only_export.py | 237 +++++++++++++++++ tests/test_review_transaction_boundary.py | 82 ------ tests/test_sql_profiles.py | 8 +- tests/test_strict_catalog_review.py | 7 +- 16 files changed, 294 insertions(+), 904 deletions(-) create mode 100644 tests/test_read_only_export.py diff --git a/README.md b/README.md index 2b9027b..d52ec78 100644 --- a/README.md +++ b/README.md @@ -132,8 +132,11 @@ the separate Transformation Workflow is unsupported. Run `openstatspec capabilities` before an integration to inspect the machine-readable feature matrix. Export is deliberately strict: if known dictionary semantics cannot be reproduced, it stops until you pass the exact -diagnostic code with `--allow-loss`. This avoids silent loss while making an -intentional lossy export auditable. +diagnostic code with `--allow-loss`. Diagnostics are returned to the caller, not +persisted. Reads, validation, and SAV/ZSAV export never write to the database, +including on failure. Export returns no `operation_id` and needs only read +permissions; Dolt reads do not require a write-conformance declaration. Imports +and transformations still require an exact matching Dolt write declaration. The matrix is also available to Python callers as `openstatspec.capability_matrix()`. It distinguishes supported semantics from diff --git a/docs/sav-profile.md b/docs/sav-profile.md index 25e5a65..814c213 100755 --- a/docs/sav-profile.md +++ b/docs/sav-profile.md @@ -49,7 +49,7 @@ equivalence of supported values, order, and dictionary metadata. The `openstatspec capabilities` command and the `openstatspec.capability_matrix()` function are the machine-readable declaration -of this boundary. It records the pinned source commit and installed engine version, and the adapter stores the same identity in every import and export operation record. The matrix deliberately distinguishes a supported feature from a feature that the underlying engine cannot observe or write faithfully. +of this boundary. It records the pinned source commit and installed engine version, and the adapter stores the same identity in import operation records. Export is database-read-only: no operation, fidelity, recovery, or Dolt-history records are written, even on failure. Loss diagnostics are returned to the caller. Dolt export identifies the server and validates the catalog and data without requiring write-conformance declarations; write entry points retain that requirement. The matrix deliberately distinguishes a supported feature from a feature that the underlying engine cannot observe or write faithfully. | SPSS semantic | Pinned pyspssio status | Behaviour | | --- | --- | --- | diff --git a/src/openstatspec/api.py b/src/openstatspec/api.py index b7bf584..540982d 100644 --- a/src/openstatspec/api.py +++ b/src/openstatspec/api.py @@ -50,7 +50,7 @@ def capability_matrix( supported means that the adapter has a tested faithful path. unobservable means that pyspssio's public reader API cannot expose the source semantic. fail-closed-on-export means an imported/catalogued value - blocks export unless the documented audited loss route exists; a plain + blocks export unless the documented explicit loss opt-in exists; a plain fail-closed feature has no faithful writer route at all. """ declaration = { @@ -174,7 +174,7 @@ def export_sav( dolt_conformance_source: DoltConformanceSource | None = None, **options: Any, ) -> Mapping[str, Any]: - """Export one database-resident conforming dataset to SAV/ZSAV.""" + """Export to SAV/ZSAV without database writes or an operation record.""" return result(export_dataset( database_url=database_url, dataset_id=dataset_id, destination=destination, diff --git a/src/openstatspec/spss/sav.py b/src/openstatspec/spss/sav.py index 111d2d3..b914dda 100644 --- a/src/openstatspec/spss/sav.py +++ b/src/openstatspec/spss/sav.py @@ -50,19 +50,12 @@ UnsupportedOperationError, safe_error_identity as _export_error_identity, ) -from ..sql.capabilities import effective_profile from ..sql.dolt_conformance import DoltConformanceSource from ..sql.wide import ( create_wide_dataset, - fail_export_operation, - finish_export_operation, physical_name, - read_export_operation_state, read_fidelity_events, read_wide_dataset, - record_export_backup_retained, - record_export_cleanup_failure, - record_export_operation, validate_spss_catalog, ) @@ -305,14 +298,6 @@ def _path_reference(path: Path, *, role: str) -> dict[str, str]: } -def _path_reference_text(path: Path, *, role: str) -> str: - """Serialize a redacted path identity for legacy text audit columns.""" - return json.dumps( - _path_reference(path, role=role), - sort_keys=True, separators=(",", ":"), - ) - - def _path_entry_exists(path: Path) -> bool: """Report directory entries without following a possibly dangling symlink.""" return path.exists() or path.is_symlink() @@ -471,8 +456,7 @@ def _restore_export_destination( def _raise_export_cleanup_failed( *, original_error: Exception, cleanup_error: Exception, phase: str, destination: Path, backup: Path, staged: Path, - had_previous: bool, database_url: str, operation_id: str | None = None, - dolt_conformance_source: DoltConformanceSource | None = None, + had_previous: bool, ) -> None: inventory = { "destination": _path_reference(destination, role="destination"), @@ -508,33 +492,6 @@ def _raise_export_cleanup_failed( "previous_destination_existed": had_previous, "durable_backup_survives_staging_cleanup": _path_entry_exists(backup), } - cleanup_audit_operation_id = None - cleanup_audit_fault = None - try: - cleanup_audit_operation_id = record_export_cleanup_failure( - database_url=database_url, - destination=_path_reference_text( - destination, role="destination", - ), - original_error=original_error, cleanup_error=cleanup_error, - residual_object_inventory=inventory, - deterministic_recovery_evidence=recovery, - operation_id=operation_id, - dolt_conformance_source=dolt_conformance_source, - ) - except Exception as audit_error: - cleanup_audit_fault = _export_error_identity( - audit_error, phase="cleanup_failed_audit", - ) - exception_recovery = { - **recovery, - "cleanup_failed_audit_persisted": cleanup_audit_fault is None, - "cleanup_failed_audit_operation_id": cleanup_audit_operation_id, - "terminal_reporting": ( - "catalog_and_exception" if cleanup_audit_fault is None - else "out_of_band_exception" - ), - } raise ExportRecoveryError( "cleanup_failed", "Export failed and the destination's prior state could not be restored.", @@ -547,72 +504,15 @@ def _raise_export_cleanup_failed( cleanup_error, phase="export_destination_restore", ), "residual_object_inventory": inventory, - "deterministic_recovery_evidence": exception_recovery, - "audit_fault": cleanup_audit_fault, + "deterministic_recovery_evidence": recovery, "success_forbidden": True, }, ) from cleanup_error -def _mark_export_failed_after_restore( - *, database_url: str, operation_id: str, error: Exception, phase: str, - destination: Path, backup: Path, had_previous: bool, - dolt_conformance_source: DoltConformanceSource | None = None, -) -> None: - """Close a running audit after the old destination has been restored.""" - failure = { - "phase": phase, - "cause": _export_error_identity(error, phase=f"export_{phase}"), - "destination": _path_reference(destination, role="destination"), - "durable_backup": _path_reference(backup, role="durable_backup"), - "previous_destination_existed": had_previous, - "destination_restored": True, - } - try: - fail_export_operation( - database_url=database_url, operation_id=operation_id, - failure_details=failure, - dolt_conformance_source=dolt_conformance_source, - ) - except Exception as audit_error: - raise ExportRecoveryError( - "failure_audit_failed", - "The destination was restored, but the running export audit could not be closed.", - details={ - "subcode": "export_failure_audit_failed", - "original_cause": failure["cause"], - "audit_fault": _export_error_identity( - audit_error, phase="export_failure_audit", - ), - "residual_object_inventory": { - "destination": _path_reference( - destination, role="destination", - ), - "destination_exists": destination.exists(), - "backup": _path_reference( - backup, role="durable_backup", - ), - "backup_exists": _path_entry_exists(backup), - "operation_id": operation_id, - }, - "deterministic_recovery_evidence": { - "procedure_id": "openstatspec.export-audit-reconciliation.v1", - "action_id": operation_id, - "phase": phase, - "destination_restored": True, - "operation_terminal_state_verified": False, - "terminal_reporting": "out_of_band_exception", - }, - "success_forbidden": True, - }, - ) from audit_error - - @contextmanager def _export_staging_directory( - *, destination: Path, database_url: str, - publication_state: dict[str, Any], - dolt_conformance_source: DoltConformanceSource | None = None, + *, destination: Path, publication_state: dict[str, Any], ): """Compensate a published file if staging-directory cleanup fails.""" try: @@ -627,7 +527,6 @@ def _export_staging_directory( backup = publication_state["backup"] staged = publication_state["staged"] had_previous = publication_state["had_previous"] - operation_id = publication_state["operation_id"] try: _restore_export_destination( destination=destination, @@ -644,20 +543,7 @@ def _export_staging_directory( backup=backup, staged=staged, had_previous=had_previous, - database_url=database_url, - operation_id=operation_id, - dolt_conformance_source=dolt_conformance_source, ) - _mark_export_failed_after_restore( - database_url=database_url, - operation_id=operation_id, - error=staging_error, - phase="staging_cleanup", - destination=destination, - backup=backup, - had_previous=had_previous, - dolt_conformance_source=dolt_conformance_source, - ) raise @@ -670,9 +556,6 @@ def export_sav_dataset( destination_path = Path(destination) if destination_path.suffix.lower() not in {".sav", ".zsav"}: raise UnsupportedOperationError("Export destinations must use the .sav or .zsav extension.") - effective_profile( - database_url, dolt_conformance_source=dolt_conformance_source, - ) dataset, variables, rows = read_wide_dataset( database_url=database_url, dataset_id=dataset_id, dolt_conformance_source=dolt_conformance_source, @@ -712,15 +595,9 @@ def export_sav_dataset( ], columns=[variable["source_name"] for variable in variables], ) - accepted_events = tuple( - event for event in loss_report if event["code"] in allow_loss - ) - operation_id = None publication_state: dict[str, Any] = {"published": False} with _export_staging_directory( - destination=destination_path, database_url=database_url, - publication_state=publication_state, - dolt_conformance_source=dolt_conformance_source, + destination=destination_path, publication_state=publication_state, ) as export_directory: staged_destination = Path(export_directory) / destination_path.name _write_with_dictionary_bridge( @@ -731,45 +608,6 @@ def export_sav_dataset( backup = _reserve_export_backup(destination_path) if not had_previous: backup.unlink() - try: - operation_id = record_export_operation( - database_url=database_url, - dataset_id=dataset_id, - destination=_path_reference_text( - destination_path, role="destination", - ), - allowed_fidelity_events=accepted_events, - operation_details={ - "engine": engine_identity(), "legacy_locale": legacy_locale, - "recovery": { - "procedure_id": "openstatspec.export-destination-restore.v1", - "phase": "prepared", - "destination": _path_reference( - destination_path, role="destination", - ), - "durable_backup": _path_reference( - backup, role="durable_backup", - ), - "previous_destination_existed": had_previous, - "publication_finalized": False, - }, - }, - terminal=False, - dolt_conformance_source=dolt_conformance_source, - ) - except Exception as audit_error: - try: - backup.unlink(missing_ok=True) - except Exception as cleanup_error: - _raise_export_cleanup_failed( - original_error=audit_error, cleanup_error=cleanup_error, - phase="audit_start_placeholder_cleanup", - destination=destination_path, backup=backup, - staged=staged_destination, had_previous=had_previous, - database_url=database_url, - dolt_conformance_source=dolt_conformance_source, - ) - raise backup_installed = False try: _publish_staged_destination( @@ -784,7 +622,6 @@ def export_sav_dataset( "published": True, "backup": backup, "staged": staged_destination, - "operation_id": operation_id, }) except Exception as publish_error: had_previous = publication_state.get("had_previous", had_previous) @@ -804,8 +641,6 @@ def export_sav_dataset( original_error=publish_error, cleanup_error=cleanup_error, phase="publish", destination=destination_path, backup=backup, staged=staged_destination, had_previous=had_previous, - database_url=database_url, operation_id=operation_id, - dolt_conformance_source=dolt_conformance_source, ) else: try: @@ -818,133 +653,17 @@ def export_sav_dataset( destination=destination_path, backup=backup, staged=staged_destination, had_previous=had_previous, - database_url=database_url, operation_id=operation_id, - dolt_conformance_source=dolt_conformance_source, ) - _mark_export_failed_after_restore( - database_url=database_url, operation_id=operation_id, - error=publish_error, phase="publish", destination=destination_path, - backup=backup, had_previous=had_previous, - dolt_conformance_source=dolt_conformance_source, - ) raise - assert operation_id is not None - finalization_state = None - try: - finish_export_operation( - database_url=database_url, operation_id=operation_id, - dolt_conformance_source=dolt_conformance_source, - ) - except Exception as finalization_error: - state_read_fault = None - try: - finalization_state = read_export_operation_state( - database_url=database_url, operation_id=operation_id, - dolt_conformance_source=dolt_conformance_source, - ) - except Exception as state_error: - state_read_fault = _export_error_identity( - state_error, phase="export_finalization_state_read", - ) - if ( - finalization_state is not None - and finalization_state["classification"] == "succeeded" - ): - pass - elif ( - finalization_state is not None - and finalization_state["classification"] == "running" - ): - try: - _restore_export_destination( - destination=destination_path, backup=backup, - had_previous=had_previous, - expected_identity=publication_state["published_identity"], - ) - except Exception as cleanup_error: - _raise_export_cleanup_failed( - original_error=finalization_error, - cleanup_error=cleanup_error, - phase="audit_finalization", - destination=destination_path, - backup=backup, - staged=staged_destination, - had_previous=had_previous, - database_url=database_url, - operation_id=operation_id, - dolt_conformance_source=dolt_conformance_source, - ) - _mark_export_failed_after_restore( - database_url=database_url, operation_id=operation_id, - error=finalization_error, phase="audit_finalization", - destination=destination_path, backup=backup, - had_previous=had_previous, - dolt_conformance_source=dolt_conformance_source, - ) - raise - else: - raise ExportRecoveryError( - "audit_finalization_ambiguous", - "The published export and durable backup were preserved because " - "the operation catalogs do not prove whether finalization committed.", - details={ - "subcode": "export_finalization_commit_ambiguous", - "operation_id": operation_id, - "finalization_cause": _export_error_identity( - finalization_error, phase="export_audit_finalization", - ), - "state_read_fault": state_read_fault, - "observed_operation_state": finalization_state, - "residual_object_inventory": { - "destination": _path_reference( - destination_path, role="published_destination", - ), - "destination_exists": destination_path.exists(), - "backup": _path_reference( - backup, role="durable_backup", - ), - "backup_exists": _path_entry_exists(backup), - }, - "deterministic_recovery_evidence": { - "procedure_id": "openstatspec.export-audit-reconciliation.v1", - "operation_terminal_state_verified": False, - "automatic_filesystem_recovery_performed": False, - "published_file_preserved": destination_path.exists(), - "durable_backup_preserved": _path_entry_exists(backup), - "manual_reconciliation_required": True, - "terminal_reporting": "out_of_band_exception", - }, - "success_forbidden": True, - }, - ) from finalization_error if _path_entry_exists(backup): try: backup.unlink() except Exception as cleanup_error: - audit_fault = None - try: - record_export_backup_retained( - database_url=database_url, operation_id=operation_id, - destination=_path_reference_text( - destination_path, role="destination", - ), - backup=_path_reference_text( - backup, role="durable_backup", - ), - cleanup_error=cleanup_error, - dolt_conformance_source=dolt_conformance_source, - ) - except Exception as warning_error: - audit_fault = _export_error_identity( - warning_error, phase="backup_retained_warning_audit", - ) raise ExportRecoveryError( "backup_retained", "The export succeeded, but its durable prior-file backup could not be removed.", details={ "subcode": "post_success_backup_retained", - "operation_id": operation_id, - "operation_status": "succeeded", "destination": _path_reference( destination_path, role="destination", ), @@ -954,8 +673,6 @@ def export_sav_dataset( "cleanup_fault": _export_error_identity( cleanup_error, phase="post_success_backup_disposal", ), - "warning_audit_persisted": audit_fault is None, - "audit_fault": audit_fault, "success_forbidden": False, }, ) @@ -963,7 +680,6 @@ def export_sav_dataset( return { "dataset_id": dataset_id, "destination": str(destination_path), - "operation_id": operation_id, "loss_report": loss_report, } diff --git a/src/openstatspec/sql/capabilities.py b/src/openstatspec/sql/capabilities.py index 5d50537..97fcd73 100644 --- a/src/openstatspec/sql/capabilities.py +++ b/src/openstatspec/sql/capabilities.py @@ -290,7 +290,23 @@ def effective_profile( *, dolt_conformance_source: DoltConformanceSource | None = None, ) -> tuple[SqlProfile, dict[str, Any]]: - """Resolve and enforce the profile used by import preflight.""" + """Resolve and enforce the write profile, including exact Dolt conformance.""" + return _effective_profile(database_url, dolt_conformance_source, for_write=True) + + +def read_profile( + database_url: str, + *, + dolt_conformance_source: DoltConformanceSource | None = None, +) -> tuple[SqlProfile, dict[str, Any]]: + """Identify a read connection without requiring evidence for Dolt writes.""" + return _effective_profile(database_url, dolt_conformance_source, for_write=False) + + +def _effective_profile( + database_url: str, dolt_conformance_source: DoltConformanceSource | None, + *, for_write: bool, +) -> tuple[SqlProfile, dict[str, Any]]: source = _conformance_source(dolt_conformance_source) configured = validate_connection_url(database_url) active = active_connection(database_url, dolt_conformance_source=source) @@ -305,6 +321,10 @@ def effective_profile( raise UnsupportedOperationError( "Dolt requires an explicit mysql+pymysql URL." ) + if active["profile"] == "dolt" and not for_write: + # Existing data is validated against the adapter's data model, not a + # write-conformance envelope or INSERT statement/packet limits. + return DOLT, active dolt_declaration = None if active["profile"] == "dolt": dolt_declaration = source.require_exact_match( diff --git a/src/openstatspec/sql/wide.py b/src/openstatspec/sql/wide.py index da488df..50f8503 100644 --- a/src/openstatspec/sql/wide.py +++ b/src/openstatspec/sql/wide.py @@ -14,11 +14,11 @@ from sqlalchemy import ( BigInteger, Column, Float, MetaData, Table, Text, create_engine, insert, - inspect, or_, select, update, + inspect, or_, select, ) from sqlalchemy.dialects import mysql, postgresql, sqlite from ..core import UnsupportedOperationError -from .capabilities import active_connection, dolt_operational_write_enabled, effective_profile +from .capabilities import active_connection, dolt_operational_write_enabled, effective_profile, read_profile from .profiles import preflight, statement_payload_bytes, validate_connection_url from .normative import ( CATALOG_CONTRACT_ID, @@ -1040,7 +1040,7 @@ def read_wide_dataset( ) -> tuple[dict[str, Any], list[dict[str, Any]], list[dict[str, Any]]]: """Read an export descriptor from the normative catalog without mutation.""" if profile is None: - profile, _active = effective_profile( + profile, _active = read_profile( database_url, dolt_conformance_source=dolt_conformance_source, ) engine = create_engine(database_url) @@ -1374,7 +1374,7 @@ def read_fidelity_events( dolt_conformance_source: Any | None = None, ) -> tuple[dict[str, Any], ...]: """Read fidelity diagnostics, optionally limited to one lifecycle direction.""" - effective_profile( + read_profile( database_url, dolt_conformance_source=dolt_conformance_source, ) engine = create_engine(database_url) @@ -1405,302 +1405,11 @@ def read_fidelity_events( return tuple(result) -@contextmanager -def _bound_export_audit_transaction( - *, database_url: str, dolt_conformance_source: Any | None, phase: str, -): - """Open one verified transaction bound to the effective Dolt working set.""" - validate_connection_url(database_url) - require_persistent_database_url(database_url) - profile, active = effective_profile( - database_url, dolt_conformance_source=dolt_conformance_source, - ) - engine = create_engine(database_url) - normative = normative_catalog(MetaData()) - with _bound_catalog_transaction( - engine=engine, profile_name=profile.name, active=active, - audit_relations={normative.operation.name, normative.fidelity_event.name}, - phase=phase, - ) as connection: - require_verified_catalog(connection) - yield connection, normative - - -def record_export_operation( - *, database_url: str, dataset_id: str, destination: str, - allowed_fidelity_events: Iterable[Mapping[str, Any]], - operation_details: Mapping[str, Any] | None = None, - terminal: bool = True, - dolt_conformance_source: Any | None = None, -) -> str: - """Persist a completed export only in the normative audit catalog.""" - operation_details = dict(operation_details or {}) - operation_id = str(uuid4()) - events = tuple(allowed_fidelity_events) - with _bound_export_audit_transaction( - database_url=database_url, dolt_conformance_source=dolt_conformance_source, - phase="export audit creation", - ) as (connection, normative): - _verify_normative_catalog(connection, normative) - dataset = _resolve_normative_dataset(connection, normative, dataset_id) - completed_at = datetime.now(UTC).replace(tzinfo=None) - status = "succeeded" if terminal else "started" - record_normative_operation( - connection, normative, operation_id=operation_id, - operation_kind="export", status=status, source_format=None, - started_at=completed_at, completed_at=completed_at if terminal else None, - ) - record_normative_fidelity_events( - connection, normative, operation_id=operation_id, - dataset_id=str(dataset["dataset_id"]), direction="export", - events=( - *( - { - **event, - "severity": event.get("severity", "warning"), - "details": { - **event.get("details", {}), - "accepted_by_user": True, - }, - } - for event in events - ), - *(({ - "code": "operation-engine-identity", - "detail": "Export engine identity recorded for audit.", - "severity": "info", - "source_item": destination, - "details": operation_details, - },) if operation_details else ()), - { - "code": "export-operation-dataset-binding", - "detail": "Export operation dataset identity recorded for audit.", - "severity": "info", - "source_item": destination, - "details": {}, - }, - ), - ) - return operation_id - - -def _export_operation_row( - connection: Any, normative: Any, operation_id: str, -) -> Mapping[str, Any]: - row = connection.execute( - select(normative.operation).where( - normative.operation.c.operation_id == operation_id - ) - ).mappings().one_or_none() - if row is None or row["operation_kind"] != "export": - raise UnsupportedOperationError("The export operation does not exist.") - return row - - -def _export_operation_dataset_id( - connection: Any, normative: Any, operation_id: str, -) -> str: - dataset_ids = set(connection.execute( - select(normative.fidelity_event.c.dataset_id) - .where(normative.fidelity_event.c.operation_id == operation_id) - .where(normative.fidelity_event.c.dataset_id.is_not(None)) - ).scalars()) - if len(dataset_ids) != 1: - raise UnsupportedOperationError( - "The export operation is not bound to exactly one dataset." - ) - dataset_id = str(dataset_ids.pop()) - if connection.execute( - select(normative.dataset.c.dataset_id).where( - normative.dataset.c.dataset_id == dataset_id - ) - ).scalar_one_or_none() is None: - raise UnsupportedOperationError( - "The export operation dataset no longer exists." - ) - return dataset_id - - -def finish_export_operation( - *, database_url: str, operation_id: str, - dolt_conformance_source: Any | None = None, -) -> None: - """Mark a started normative export operation as succeeded.""" - with _bound_export_audit_transaction( - database_url=database_url, dolt_conformance_source=dolt_conformance_source, - phase="export audit finalization", - ) as (connection, normative): - _verify_normative_catalog(connection, normative) - row = _export_operation_row(connection, normative, operation_id) - if row["status"] != "started": - raise UnsupportedOperationError( - "Only a started export operation can be finalized." - ) - finish_normative_operation( - connection, normative, operation_id=operation_id, status="succeeded", - ) - - -def read_export_operation_state( - *, database_url: str, operation_id: str, - dolt_conformance_source: Any | None = None, -) -> dict[str, Any]: - """Read the singular normative export-operation state.""" - effective_profile( - database_url, dolt_conformance_source=dolt_conformance_source, - ) - normative = normative_catalog(MetaData()) - with create_engine(database_url).connect() as connection: - require_verified_catalog(connection) - row = _export_operation_row(connection, normative, operation_id) - classification = { - "started": "running", - "succeeded": "succeeded", - "failed": "failed", - }.get(str(row["status"]), "ambiguous") - return { - "operation_id": operation_id, - "normative": { - "operation_kind": row["operation_kind"], - "status": row["status"], - }, - "classification": classification, - } - - -def fail_export_operation( - *, database_url: str, operation_id: str, - failure_details: Mapping[str, Any], - dolt_conformance_source: Any | None = None, -) -> None: - """Close a started normative export after filesystem compensation.""" - event = { - "code": "export_failed", - "detail": "Export publication or finalization failed after audit start.", - "severity": "error", - "details": dict(failure_details), - } - with _bound_export_audit_transaction( - database_url=database_url, dolt_conformance_source=dolt_conformance_source, - phase="export audit failure", - ) as (connection, normative): - _verify_normative_catalog(connection, normative) - row = _export_operation_row(connection, normative, operation_id) - if row["status"] != "started": - raise UnsupportedOperationError( - "Only a started export operation can be failed." - ) - dataset_id = _export_operation_dataset_id( - connection, normative, operation_id, - ) - finish_normative_operation( - connection, normative, operation_id=operation_id, status="failed", - ) - record_normative_fidelity_events( - connection, normative, operation_id=operation_id, - dataset_id=dataset_id, direction="export", events=(event,), - ) - - -def record_export_backup_retained( - *, database_url: str, operation_id: str, destination: str, backup: str, - cleanup_error: Exception, - dolt_conformance_source: Any | None = None, -) -> None: - """Append a warning to a successfully finalized normative export.""" - with _bound_export_audit_transaction( - database_url=database_url, dolt_conformance_source=dolt_conformance_source, - phase="export backup-retention audit", - ) as (connection, normative): - _verify_normative_catalog(connection, normative) - row = _export_operation_row(connection, normative, operation_id) - if row["status"] != "succeeded": - raise UnsupportedOperationError( - "A retained backup warning requires a succeeded export." - ) - dataset_id = _export_operation_dataset_id( - connection, normative, operation_id, - ) - record_normative_fidelity_events( - connection, normative, operation_id=operation_id, - dataset_id=dataset_id, direction="export", events=({ - "code": "backup_retained", - "detail": "A successful export retained its durable prior-file backup.", - "severity": "warning", - "source_item": destination, - "details": { - "durable_backup": backup, - "cleanup_error_type": type(cleanup_error).__name__, - }, - },), - ) - - -def record_export_cleanup_failure( - *, database_url: str, destination: str, original_error: Exception, - cleanup_error: Exception, - residual_object_inventory: Mapping[str, Any], - deterministic_recovery_evidence: Mapping[str, Any], - operation_id: str | None = None, - dolt_conformance_source: Any | None = None, -) -> str: - """Persist terminal cleanup failure in the normative audit catalog.""" - operation_id = operation_id or str(uuid4()) - event = { - "code": "cleanup_failed", - "detail": "Export destination recovery failed; out-of-band review is required.", - "severity": "error", - "source_item": destination, - "details": { - "original_error_type": type(original_error).__name__, - "cleanup_error_type": type(cleanup_error).__name__, - "residual_object_inventory": dict(residual_object_inventory), - "deterministic_recovery_evidence": dict( - deterministic_recovery_evidence - ), - }, - } - with _bound_export_audit_transaction( - database_url=database_url, dolt_conformance_source=dolt_conformance_source, - phase="export cleanup-failure audit", - ) as (connection, normative): - _verify_normative_catalog(connection, normative) - row = connection.execute( - select(normative.operation).where( - normative.operation.c.operation_id == operation_id - ) - ).mappings().one_or_none() - dataset_id = None - if row is None: - failed_at = datetime.now(UTC).replace(tzinfo=None) - record_normative_operation( - connection, normative, operation_id=operation_id, - operation_kind="export", status="failed", source_format=None, - started_at=failed_at, completed_at=failed_at, - ) - else: - if row["operation_kind"] != "export" or row["status"] != "started": - raise UnsupportedOperationError( - "Existing export operation cannot transition to cleanup failure." - ) - dataset_id = _export_operation_dataset_id( - connection, normative, operation_id, - ) - finish_normative_operation( - connection, normative, operation_id=operation_id, status="failed", - ) - record_normative_fidelity_events( - connection, normative, operation_id=operation_id, - dataset_id=dataset_id, direction="export", events=(event,), - ) - return operation_id - - def validate_wide_dataset( *, database_url: str, dataset_id: str, dolt_conformance_source: Any | None = None, ) -> dict[str, Any]: - profile, _active = effective_profile( + profile, _active = read_profile( database_url, dolt_conformance_source=dolt_conformance_source, ) dataset, variables, rows = read_wide_dataset( @@ -1750,4 +1459,3 @@ def validate_wide_dataset( if [row["__case_ordinal"] for row in rows] != list(range(1, len(rows) + 1)): raise ValueError("Case ordinals are not contiguous source order.") return {"dataset_id": dataset["dataset_id"], "valid": True, "case_count": len(rows), "variable_count": len(variables)} - diff --git a/src/openstatspec/sql/workflow.py b/src/openstatspec/sql/workflow.py index 8d96702..86c313b 100644 --- a/src/openstatspec/sql/workflow.py +++ b/src/openstatspec/sql/workflow.py @@ -2490,13 +2490,17 @@ def reconcile_physical_removals( return {"reconciled": len(results), "results": results} def validate_derived_dataset(*, database_url: str, derived_dataset_id: str) -> dict[str, Any]: - """Validate profile ownership, physical shape, row ordinals, and run lineage.""" + """Validate without creating or migrating catalog state.""" + from .catalog_api import _assert_workflow + derived_id = _uuid(derived_dataset_id, "derived_dataset_id") profile = validate_connection_url(database_url) engine = _workflow_engine(database_url, profile.name) tables = workflow_catalog(MetaData()) - with engine.begin() as connection: - create_workflow_catalog(connection, tables) + with engine.connect() as connection: + _assert_core_identity(connection, core_catalog(MetaData())) + _assert_workflow(connection, tables) + _validate_workflow_schema(connection, tables) dataset = connection.execute( select(tables.derived_dataset) .where(tables.derived_dataset.c.derived_dataset_id == derived_id) diff --git a/tests/test_catalog_lifecycle.py b/tests/test_catalog_lifecycle.py index 1c34809..73ba11a 100644 --- a/tests/test_catalog_lifecycle.py +++ b/tests/test_catalog_lifecycle.py @@ -8,10 +8,6 @@ from openstatspec.sql.wide import ( _bounded_batches, create_wide_dataset, - finish_export_operation, - read_export_operation_state, - record_export_cleanup_failure, - record_export_operation, ) @@ -154,55 +150,3 @@ def test_duplicate_import_preserves_existing_dataset(tmp_path): ).fetchall() == [("sample", 1)] assert connection.execute("select name from data_sample").fetchall() == [("ok",)] connection.close() - - -def test_normative_export_operation_transitions_to_success(tmp_path): - path = tmp_path / "export.sqlite" - database = f"sqlite:///{path}" - openstatspec.initialize_catalog(database_url=database) - _create_dataset(database) - - operation_id = record_export_operation( - database_url=database, - dataset_id="sample", - destination="output.sav", - allowed_fidelity_events=(), - terminal=False, - ) - assert read_export_operation_state( - database_url=database, operation_id=operation_id, - )["classification"] == "running" - - finish_export_operation( - database_url=database, operation_id=operation_id, - ) - assert read_export_operation_state( - database_url=database, operation_id=operation_id, - )["classification"] == "succeeded" - - -def test_cleanup_failure_is_recorded_without_compatibility_catalog(tmp_path): - path = tmp_path / "cleanup.sqlite" - database = f"sqlite:///{path}" - openstatspec.initialize_catalog(database_url=database) - - operation_id = record_export_cleanup_failure( - database_url=database, - destination="output.sav", - original_error=RuntimeError("export failed"), - cleanup_error=RuntimeError("restore failed"), - residual_object_inventory={"backup": True}, - deterministic_recovery_evidence={"procedure_id": "test"}, - ) - - connection = sqlite3.connect(path) - assert connection.execute( - "select status from operation where operation_id = ?", - (operation_id,), - ).fetchone() == ("failed",) - assert connection.execute( - "select event_code from fidelity_event where operation_id = ?", - (operation_id,), - ).fetchone() == ("cleanup_failed",) - assert "dataset_catalog" not in _table_names(path) - connection.close() diff --git a/tests/test_catalog_persistence_review.py b/tests/test_catalog_persistence_review.py index c1d50f5..a311def 100644 --- a/tests/test_catalog_persistence_review.py +++ b/tests/test_catalog_persistence_review.py @@ -1,6 +1,5 @@ """Regression tests for persistent URLs and operation-bound audit writes.""" -import json import sqlite3 from contextlib import contextmanager @@ -162,92 +161,3 @@ def inject_drift(**kwargs): assert connection.execute("select count(*) from operation").fetchone() == (0,) assert connection.execute("select count(*) from fidelity_event").fetchone() == (0,) connection.close() - - -def test_export_audit_mutations_route_through_binding_guard(tmp_path, monkeypatch): - database_url = f"sqlite:///{tmp_path / 'bound-export.sqlite'}" - _create(database_url) - real_bound_transaction = wide._bound_catalog_transaction - phases = [] - - @contextmanager - def observe(**kwargs): - phases.append(kwargs["phase"]) - with real_bound_transaction(**kwargs) as connection: - yield connection - - monkeypatch.setattr(wide, "_bound_catalog_transaction", observe) - failed_id = wide.record_export_operation( - database_url=database_url, - dataset_id="sample", - destination="failed.sav", - allowed_fidelity_events=(), - terminal=False, - ) - wide.fail_export_operation( - database_url=database_url, - operation_id=failed_id, - failure_details={"reason": "test"}, - ) - succeeded_id = wide.record_export_operation( - database_url=database_url, - dataset_id="sample", - destination="succeeded.sav", - allowed_fidelity_events=(), - terminal=False, - ) - wide.finish_export_operation( - database_url=database_url, operation_id=succeeded_id, - ) - wide.record_export_backup_retained( - database_url=database_url, - operation_id=succeeded_id, - destination="succeeded.sav", - backup="succeeded.sav.backup", - cleanup_error=RuntimeError("test"), - ) - wide.record_export_cleanup_failure( - database_url=database_url, - destination="cleanup.sav", - original_error=RuntimeError("export"), - cleanup_error=RuntimeError("cleanup"), - residual_object_inventory={}, - deterministic_recovery_evidence={}, - ) - - assert phases == [ - "export audit creation", - "export audit failure", - "export audit creation", - "export audit finalization", - "export backup-retention audit", - "export cleanup-failure audit", - ] - - -def test_export_engine_identity_is_persisted_as_non_loss_audit_metadata(tmp_path): - path = tmp_path / "export-audit.sqlite" - database_url = f"sqlite:///{path}" - _create(database_url) - engine_identity = {"name": "pyspssio", "commit": "export-commit"} - - operation_id = wide.record_export_operation( - database_url=database_url, - dataset_id="sample", - destination="out.sav", - allowed_fidelity_events=(), - operation_details={"engine": engine_identity}, - ) - - connection = sqlite3.connect(path) - row = connection.execute( - "select severity, event_code, source_item, detail_json " - "from fidelity_event where operation_id = ?", - (operation_id,), - ).fetchone() - connection.close() - assert row[:3] == ("info", "operation-engine-identity", "out.sav") - assert json.loads(row[3])["engine"] == engine_identity - assert wide.read_fidelity_events( - database_url=database_url, dataset_id="sample", - ) == () diff --git a/tests/test_dolt_conformance.py b/tests/test_dolt_conformance.py index b4e0330..0e57a79 100644 --- a/tests/test_dolt_conformance.py +++ b/tests/test_dolt_conformance.py @@ -196,11 +196,6 @@ def test_every_sql_mutation_entrypoint_accepts_explicit_conformance_source() -> mutation_entrypoints = ( wide.initialize_wide_catalog, wide.create_wide_dataset, - wide.record_export_cleanup_failure, - wide.record_export_operation, - wide.finish_export_operation, - wide.fail_export_operation, - wide.record_export_backup_retained, ) for entrypoint in mutation_entrypoints: parameter = inspect.signature(entrypoint).parameters[ @@ -236,7 +231,7 @@ def stop_after_read_preflight(**kwargs: object) -> tuple[object, object, object] calls.append(kwargs["dolt_conformance_source"]) raise UnsupportedOperationError("stop after propagation check") - monkeypatch.setattr(wide, "effective_profile", capture_effective_profile) + monkeypatch.setattr(wide, "read_profile", capture_effective_profile) monkeypatch.setattr(wide, "read_wide_dataset", stop_after_read_preflight) with pytest.raises(UnsupportedOperationError, match="propagation check"): wide.validate_wide_dataset( @@ -343,40 +338,3 @@ def stop_after_read(**kwargs: object) -> tuple[object, object, object]: dolt_conformance_source=sentinel, ) assert calls == [sentinel] - - -def test_export_recovery_helpers_propagate_explicit_source( - monkeypatch: pytest.MonkeyPatch, tmp_path: Path, -) -> None: - sentinel = object() - calls: list[tuple[str, object]] = [] - - def capture_cleanup(**kwargs: object) -> str: - calls.append(("cleanup", kwargs["dolt_conformance_source"])) - return "cleanup-operation" - - def capture_failure(**kwargs: object) -> None: - calls.append(("failure", kwargs["dolt_conformance_source"])) - - monkeypatch.setattr( - sav_module, "record_export_cleanup_failure", capture_cleanup, - ) - monkeypatch.setattr(sav_module, "fail_export_operation", capture_failure) - destination = tmp_path / "destination.sav" - backup = tmp_path / "backup.sav" - staged = tmp_path / "staged.sav" - with pytest.raises(sav_module.ExportRecoveryError, match="cleanup_failed"): - sav_module._raise_export_cleanup_failed( - original_error=RuntimeError("synthetic export failure"), - cleanup_error=RuntimeError("synthetic cleanup failure"), - phase="synthetic", destination=destination, backup=backup, - staged=staged, had_previous=False, database_url="sqlite://", - dolt_conformance_source=sentinel, - ) - sav_module._mark_export_failed_after_restore( - database_url="sqlite://", operation_id="synthetic-operation", - error=RuntimeError("synthetic export failure"), phase="synthetic", - destination=destination, backup=backup, had_previous=False, - dolt_conformance_source=sentinel, - ) - assert calls == [("cleanup", sentinel), ("failure", sentinel)] diff --git a/tests/test_loss_reports.py b/tests/test_loss_reports.py index d30b32c..b8bd446 100755 --- a/tests/test_loss_reports.py +++ b/tests/test_loss_reports.py @@ -40,24 +40,11 @@ def test_persisted_import_fidelity_events_require_consent_after_reopen(tmp_path) exported = openstatspec.export_sav(database_url=database, dataset_id="persisted", destination=approved, allow_loss=_REQUIRED_ENGINE_LOSS) assert approved.exists() assert connection.execute( - "select operation_kind, status from operation where operation_id = ?", (exported["operation_id"],) - ).fetchone() == ("export", "succeeded") + "select count(*) from operation where operation_kind = 'export'" + ).fetchone() == (0,) assert {diagnostic.code for diagnostic in exported.diagnostics} == set(_REQUIRED_ENGINE_LOSS) -def test_loss_allowed_export_persists_accepted_diagnostics(tmp_path) -> None: - database_path = tmp_path / "accepted-loss.sqlite" - database = f"sqlite:///{database_path}" - source = tmp_path / "source.sav" - destination = tmp_path / "accepted.sav" - pyspssio.write_sav(str(source), pd.DataFrame({"answer": [1.0]})) - openstatspec.import_sav(source, database_url=database, dataset_id="accepted") - result = openstatspec.export_sav(database_url=database, dataset_id="accepted", destination=destination, allow_loss=_REQUIRED_ENGINE_LOSS) - - connection = sqlite3.connect(database_path) - rows = connection.execute("select direction, severity, event_code, detail_json from fidelity_event where operation_id = ? and severity != 'info' order by event_code", (result["operation_id"],)).fetchall() - assert [(row[0], row[1], row[2]) for row in rows] == [("export", "warning", code) for code in _REQUIRED_ENGINE_LOSS] - assert all('"accepted_by_user": true' in row[3] for row in rows) def test_non_utf8_source_encoding_is_explicit_export_loss(tmp_path) -> None: diff --git a/tests/test_normative_catalog.py b/tests/test_normative_catalog.py index 6f98071..f2d20a0 100644 --- a/tests/test_normative_catalog.py +++ b/tests/test_normative_catalog.py @@ -5,7 +5,7 @@ import openstatspec import pytest -from openstatspec.sql.wide import create_wide_dataset, record_export_operation +from openstatspec.sql.wide import create_wide_dataset from openstatspec.sql.normative import catalog as normative_catalog, create as create_normative from sqlalchemy import MetaData, create_engine from sqlalchemy.dialects.mysql import dialect as mysql_dialect @@ -154,19 +154,3 @@ def test_import_writes_complete_normative_catalog(tmp_path): assert event[:5] == (dataset[0], "import", "warning", "fixture-warning", "answer") assert json.loads(event[5]) == {"message": "Synthetic diagnostic", "test": True} assert event[6] - - export_id = record_export_operation( - database_url=database_url, dataset_id="wave_1", destination="out.sav", - allowed_fidelity_events=[{ - "code": "export-warning", "detail": "Accepted", "source_item": "comment", - }], - ) - assert connection.execute( - "select operation_kind, status from operation where operation_id = ?", (export_id,), - ).fetchone() == ("export", "succeeded") - export_event = connection.execute( - "select dataset_id, direction, event_code, source_item, detail_json " - "from fidelity_event where operation_id = ?", (export_id,), - ).fetchone() - assert export_event[:4] == (dataset[0], "export", "export-warning", "comment") - assert json.loads(export_event[4])["accepted_by_user"] is True diff --git a/tests/test_read_only_export.py b/tests/test_read_only_export.py new file mode 100644 index 0000000..246493e --- /dev/null +++ b/tests/test_read_only_export.py @@ -0,0 +1,237 @@ +"""Read/export must work without database write privileges or write declarations. + +Optional live check against a disposable local Dolt server: +OPENSTATSPEC_TEST_DOLT_ADMIN_URL=mysql+pymysql://root@127.0.0.1:PORT/ \ + python -m pytest tests/test_read_only_export.py +The fixture creates and removes its own database and SELECT-only user. +""" +import os +import sqlite3 +from uuid import uuid4 + +import pandas as pd +import pyspssio +import pytest +from sqlalchemy import MetaData, Table, create_engine, event, select +from sqlalchemy.dialects.mysql import LONGTEXT +from sqlalchemy.engine import Engine, make_url + +import openstatspec +from openstatspec.core import UnsupportedOperationError +from openstatspec.sql import capabilities, wide +from openstatspec.sql.normative import catalog +from openstatspec.spss import sav + + +@pytest.fixture(params=["sqlite", "dolt"]) +def seeded_catalog(request, tmp_path): + admin_url = os.environ.get("OPENSTATSPEC_TEST_DOLT_ADMIN_URL") + if request.param == "dolt" and not admin_url: + pytest.skip("Live Dolt requires OPENSTATSPEC_TEST_DOLT_ADMIN_URL") + source = tmp_path / "source.sav" + pyspssio.write_sav( + str(source), pd.DataFrame({"answer": [1.0, 2.0], "name": ["one", "two"]}), + var_labels={"answer": "Answer"}, + var_value_labels={"answer": {1.0: "Yes", 2.0: "No"}}, + var_missing_values={"answer": {"values": [2.0]}}, + ) + path = tmp_path / "source.sqlite" + url = f"sqlite:///{path}" + openstatspec.import_sav(source, database_url=url, dataset_id="sample") + source_engine = create_engine(url) + logical = catalog(MetaData()) + with source_engine.connect() as connection: + dataset = connection.execute(select(logical.dataset)).mappings().one() + if request.param == "sqlite": + def snapshot(): + with sqlite3.connect(path) as connection: + return tuple(connection.iterdump()) + + def read_only(connection, _record): + if isinstance(connection, sqlite3.Connection): + connection.execute("PRAGMA query_only = ON") + + event.listen(Engine, "connect", read_only) + try: + yield url, snapshot, source + finally: + event.remove(Engine, "connect", read_only) + source_engine.dispose() + return + + name = "export_" + uuid4().hex + user = "reader_" + uuid4().hex[:16] + admin = create_engine(admin_url, isolation_level="AUTOCOMMIT") + target = None + try: + with admin.connect() as connection: + connection.exec_driver_sql(f"CREATE DATABASE `{name}`") + connection.exec_driver_sql(f"CREATE USER '{user}'@'%%' IDENTIFIED BY 'readonly'") + connection.exec_driver_sql(f"GRANT SELECT ON `{name}`.* TO '{user}'@'%%'") + target = create_engine(make_url(admin_url).set(database=name)) + # Seed existing data with administrative SQL, not a bypass of the + # adapter's write guard. The exporter only receives the SELECT user. + logical.catalog_identity.metadata.create_all(target) + physical = Table(dataset["physical_table_name"], MetaData(), autoload_with=source_engine) + physical.c.name.type = LONGTEXT() + physical.create(target) + with source_engine.connect() as origin, target.begin() as destination: + for table in [*logical.catalog_identity.metadata.sorted_tables, physical]: + rows = [dict(row) for row in origin.execute(select(table)).mappings()] + if rows: + destination.execute(table.insert(), rows) + with target.connect() as connection: + connection.exec_driver_sql("CALL DOLT_ADD('.')") + connection.exec_driver_sql( + "CALL DOLT_COMMIT('-m', 'existing dataset', '--author', 'Test ')" + ) + connection.commit() + + def snapshot(): + with target.connect() as connection: + return ( + tuple(connection.exec_driver_sql("SELECT * FROM dolt_status")), + tuple(connection.exec_driver_sql("SELECT commit_hash FROM dolt_log")), + tuple( + (table.name, tuple(connection.execute(select(table)))) + for table in [*logical.catalog_identity.metadata.sorted_tables, physical] + ), + ) + + reader_url = make_url(admin_url).set(database=name, username=user, password="readonly") + with admin.connect() as connection: + grants = tuple(connection.exec_driver_sql(f"SHOW GRANTS FOR '{user}'@'%%'").scalars()) + assert any("GRANT SELECT ON" in grant for grant in grants) + assert all(grant.startswith(("GRANT USAGE ON", "GRANT SELECT ON")) for grant in grants) + yield reader_url.render_as_string(hide_password=False), snapshot, source + finally: + if target is not None: + target.dispose() + source_engine.dispose() + with admin.connect() as connection: + connection.exec_driver_sql(f"DROP DATABASE IF EXISTS `{name}`") + connection.exec_driver_sql(f"DROP USER IF EXISTS '{user}'@'%%'") + admin.dispose() + + +@pytest.fixture +def read_catalog(seeded_catalog): + forbidden = [] + + def check(_connection, _cursor, statement, _parameters, _context, _many): + if statement.lstrip().split()[0].upper() not in {"SELECT", "SHOW", "PRAGMA", "DESCRIBE"}: + forbidden.append(statement) + raise AssertionError("Read operation attempted a non-read SQL statement") + + event.listen(Engine, "before_cursor_execute", check) + try: + yield seeded_catalog + finally: + event.remove(Engine, "before_cursor_execute", check) + assert forbidden == [] + + +@pytest.mark.parametrize("suffix", [".sav", ".zsav"]) +def test_export_and_reads_leave_no_database_trace(read_catalog, tmp_path, suffix): + url, snapshot, source = read_catalog + before = snapshot() + if make_url(url).get_backend_name() == "mysql": + # This is the real server identity and the empty packaged registry. + with pytest.raises(UnsupportedOperationError): + wide.initialize_wide_catalog(database_url=url) + output = tmp_path / ("export" + suffix) + result = openstatspec.export_sav(database_url=url, dataset_id="sample", destination=output) + assert "operation_id" not in result + expected, expected_meta = pyspssio.read_sav(str(source), include_user_missing=True) + actual, actual_meta = pyspssio.read_sav(str(output), include_user_missing=True) + pd.testing.assert_frame_equal(actual, expected) + for key in ("var_labels", "var_value_labels", "var_missing_values", "var_formats", "var_measure_levels"): + assert actual_meta[key] == expected_meta[key] + openstatspec.validate(database_url=url, dataset_id="sample") + wide.read_fidelity_events(database_url=url, dataset_id="sample") + openstatspec.inspect(source) + assert snapshot() == before + assert not list(tmp_path.glob(".*.previous")) + assert not list(tmp_path.glob(".*.staging.*")) + + +@pytest.mark.parametrize("failure", ["loss", "writer", "publish", "restore", "staging", "backup"]) +def test_failed_export_never_writes_database(read_catalog, tmp_path, monkeypatch, failure): + url, snapshot, _source = read_catalog + before = snapshot() + output = tmp_path / "existing.sav" + output.write_bytes(b"previous file") + + def fail(*_args, **_kwargs): + raise OSError("injected failure") + + if failure == "loss": + monkeypatch.setattr(sav, "_export_loss_report", lambda *_a, **_k: ( + {"code": "test-loss", "detail": "Requires consent", "details": {}}, + )) + elif failure == "writer": + monkeypatch.setattr(sav, "_write_with_dictionary_bridge", fail) + elif failure in {"publish", "restore"}: + monkeypatch.setattr(sav.os, "link", fail) + if failure == "restore": + monkeypatch.setattr(sav, "_restore_export_destination", fail) + elif failure == "staging": + cleanup = sav.TemporaryDirectory.cleanup + + def fail_cleanup(self): + cleanup(self) + raise OSError("injected failure") + + monkeypatch.setattr(sav.TemporaryDirectory, "cleanup", fail_cleanup) + else: + unlink = type(output).unlink + + def fail_backup(self, *args, **kwargs): + if self.suffix == ".previous": + raise OSError("injected failure") + return unlink(self, *args, **kwargs) + + monkeypatch.setattr(type(output), "unlink", fail_backup) + with pytest.raises((OSError, UnsupportedOperationError)): + openstatspec.export_sav(database_url=url, dataset_id="sample", destination=output) + assert snapshot() == before + if failure in {"restore", "backup"}: + # If filesystem recovery itself fails, preserve the previous bytes in + # the existing safety backup and report the error only to the caller. + assert next(tmp_path.glob(".*.previous")).read_bytes() == b"previous file" + else: + assert output.read_bytes() == b"previous file" + if failure == "loss": + result = openstatspec.export_sav( + database_url=url, dataset_id="sample", destination=output, allow_loss=["test-loss"], + ) + assert {d.code for d in result.diagnostics} == {"test-loss"} + assert snapshot() == before + + +def test_dolt_reads_do_not_enable_undeclared_writes(monkeypatch): + url = "mysql+pymysql://reader@host/dataset" + active = { + "profile": "dolt", "raw_product_version": "2.3.0", "claimed_supported": False, + } + monkeypatch.setattr(capabilities, "active_connection", lambda *_a, **_k: active) + assert capabilities.read_profile(url)[0].name == "dolt" + with pytest.raises(UnsupportedOperationError): + capabilities.effective_profile(url) + with pytest.raises(UnsupportedOperationError): + wide.initialize_wide_catalog(database_url=url) + with pytest.raises(UnsupportedOperationError): + wide.create_wide_dataset( + database_url=url, dataset_id="blocked", source_name="source.sav", + source_format="SAV", variables=[], rows=[], + ) + + +def test_derived_validation_does_not_initialize_catalog(tmp_path): + url = f"sqlite:///{tmp_path / 'catalog.sqlite'}" + openstatspec.initialize_catalog(database_url=url) + with sqlite3.connect(tmp_path / "catalog.sqlite") as connection: + before = tuple(connection.iterdump()) + with pytest.raises(UnsupportedOperationError): + openstatspec.validate_derived(database_url=url, derived_dataset_id=str(uuid4())) + assert tuple(connection.iterdump()) == before diff --git a/tests/test_review_transaction_boundary.py b/tests/test_review_transaction_boundary.py index 2fe5165..2dd8f9a 100644 --- a/tests/test_review_transaction_boundary.py +++ b/tests/test_review_transaction_boundary.py @@ -1,29 +1,10 @@ """Regression tests for bound commit checks and export dataset linkage.""" -import sqlite3 - from sqlalchemy import create_engine from openstatspec.sql import wide -def _variables(): - return [{ - "ordinal": 1, - "source_name": "name", - "physical_name": "name", - "storage_kind": "string", - "string_width": 8, - "label": "", - "format": "A8", - "measure": "nominal", - "alignment": "left", - "display_width": 8, - "value_labels": "{}", - "missing_ranges": "[]", - }] - - def test_dolt_completion_identity_is_checked_before_transaction_commit( monkeypatch, ): @@ -144,66 +125,3 @@ def test_dolt_completion_rejects_unrelated_working_set_changes(): with pytest.raises(UnsupportedOperationError, match="unrelated working-set"): wide._require_dolt_success_identity(before, after, phase="export audit") - - -def test_export_lifecycle_events_remain_linked_to_the_dataset(tmp_path): - path = tmp_path / "export-events.sqlite" - database_url = f"sqlite:///{path}" - wide.create_wide_dataset( - database_url=database_url, - dataset_id="sample", - source_name="sample.sav", - source_format="SAV", - rows=[{"name": "ok"}], - variables=_variables(), - ) - - failed_id = wide.record_export_operation( - database_url=database_url, - dataset_id="sample", - destination="failed.sav", - allowed_fidelity_events=(), - terminal=False, - ) - wide.fail_export_operation( - database_url=database_url, - operation_id=failed_id, - failure_details={"reason": "test"}, - ) - succeeded_id = wide.record_export_operation( - database_url=database_url, - dataset_id="sample", - destination="succeeded.sav", - allowed_fidelity_events=(), - terminal=False, - ) - wide.finish_export_operation( - database_url=database_url, - operation_id=succeeded_id, - ) - wide.record_export_backup_retained( - database_url=database_url, - operation_id=succeeded_id, - destination="succeeded.sav", - backup="succeeded.sav.backup", - cleanup_error=RuntimeError("test"), - ) - - events = wide.read_fidelity_events( - database_url=database_url, - dataset_id="sample", - ) - assert {event["code"] for event in events} == { - "backup_retained", - "export_failed", - } - assert wide.read_fidelity_events( - database_url=database_url, dataset_id="sample", direction="import", - ) == () - connection = sqlite3.connect(path) - assert connection.execute( - "select count(*) from fidelity_event " - "where operation_id in (?, ?) and dataset_id is null", - (failed_id, succeeded_id), - ).fetchone() == (0,) - connection.close() diff --git a/tests/test_sql_profiles.py b/tests/test_sql_profiles.py index c1499aa..9591a2e 100755 --- a/tests/test_sql_profiles.py +++ b/tests/test_sql_profiles.py @@ -462,7 +462,7 @@ def test_validate_identity_failure_happens_before_catalog_access(monkeypatch) -> def fail_identity(_url, **_kwargs): raise UnsupportedOperationError("identity unavailable") - monkeypatch.setattr(wide, "effective_profile", fail_identity) + monkeypatch.setattr(wide, "read_profile", fail_identity) monkeypatch.setattr( wide, "create_engine", lambda _url: pytest.fail("database access continued after identity failure"), @@ -480,10 +480,10 @@ def test_export_identity_failure_happens_before_read_or_destination(monkeypatch, def fail_identity(_url, **_kwargs): raise UnsupportedOperationError("identity unavailable") - monkeypatch.setattr(sav, "effective_profile", fail_identity) + monkeypatch.setattr(wide, "read_profile", fail_identity) monkeypatch.setattr( - sav, "read_wide_dataset", - lambda **_kwargs: pytest.fail("catalog read continued after identity failure"), + wide, "create_engine", + lambda _url: pytest.fail("catalog read continued after identity failure"), ) with pytest.raises(UnsupportedOperationError, match="identity unavailable"): diff --git a/tests/test_strict_catalog_review.py b/tests/test_strict_catalog_review.py index cb36a93..5f57db5 100644 --- a/tests/test_strict_catalog_review.py +++ b/tests/test_strict_catalog_review.py @@ -3,6 +3,8 @@ import json import sqlite3 +from openstatspec import export_sav + import pytest from sqlalchemy import MetaData, create_engine @@ -87,11 +89,10 @@ def test_read_and_export_reject_obsolete_catalog_relations(tmp_path): with pytest.raises(UnsupportedOperationError, match="obsolete"): wide.read_wide_dataset(database_url=database, dataset_id="sample") with pytest.raises(UnsupportedOperationError, match="obsolete"): - wide.record_export_operation( + export_sav( database_url=database, dataset_id="sample", - destination="sample.sav", - allowed_fidelity_events=(), + destination=tmp_path / "sample.sav", ) From 72b20197c076ddbf701271e7eb7ea23f1fd0b4e1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?T=C3=B5nis=20Ormisson?= Date: Mon, 7 Sep 2026 17:31:02 +0300 Subject: [PATCH 2/4] Enable exact-version default Dolt writes --- .github/workflows/ci.yml | 2 +- README.md | 24 ++-- docs/release-readiness.md | 42 +++++- src/openstatspec/sql/capabilities.py | 67 ++++++---- src/openstatspec/sql/dolt_conformance.py | 7 +- src/openstatspec/sql/inplace_transform.py | 3 + src/openstatspec/sql/wide.py | 14 +- tests/test_dolt_conformance.py | 48 ++++++- tests/test_read_only_export.py | 1 + tests/test_review_transaction_boundary.py | 13 ++ tests/test_sql_profiles.py | 55 ++++---- tests/test_sql_services.py | 156 ++++++++++++++++++++-- 12 files changed, 341 insertions(+), 91 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 282edd2..5d408a5 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -143,7 +143,7 @@ jobs: steps: *sql-test-steps dolt-integration: - name: Dolt ${{ matrix.dolt }} read-only / fail-closed integration + name: Dolt ${{ matrix.dolt }} default-write integration runs-on: ubuntu-latest strategy: fail-fast: false diff --git a/README.md b/README.md index d52ec78..58c9a0c 100644 --- a/README.md +++ b/README.md @@ -96,17 +96,19 @@ SAV/ZSAV export for the semantics exposed by that engine. SQLite is the local reference path. PostgreSQL, MySQL, MariaDB, and Dolt are each covered by separate service-backed CI conformance checks. Dolt support is an independent core profile for the -canonical stable range `>=2.2.2,<2.3.0`; earlier patches, other families, -noncanonical versions, and unknown MySQL-wire products fail closed. +exact write-supported versions 2.2.2 and 2.2.3; all other versions (including +2.3.0), noncanonical versions, and unknown MySQL-wire products fail closed for writes. The supported family claims are broader than the deliberately exact CI evidence points: PostgreSQL 17.x/18.x is exercised at 17.10/18.4, MySQL 8.4.x/9.7.x at 8.4.11/9.7.2, and MariaDB 11.4.x/11.8.x/12.3.x at 11.4.12/11.8.8/12.3.2. Each service job checks the normalized live server version against its exact matrix entry before that run can count as evidence. -Dolt claims the conservative 2.2.x range `>=2.2.2,<2.3.0`; its full service -suite is exercised independently at exact versions 2.2.2 and 2.2.3 using -immutable container-image digests. +Dolt's default-write service checks run independently at exact versions 2.2.2 +and 2.2.3 using immutable container-image digests. Release/CI owns this evidence; +normal callers need no declaration or evidence files. The optional explicit +`DoltConformanceSource` override remains strict and does not fall back to the +default policy if its declarations are missing, invalid, or mismatched. | Engine/profile | Runtime supported policy | Exact CI-tested versions | | --- | --- | --- | @@ -114,7 +116,7 @@ immutable container-image digests. | PostgreSQL | 17.x and 18.x | 17.10 and 18.4 | | MySQL | 8.4.x and 9.7.x | 8.4.11 and 9.7.2 | | MariaDB | 11.4.x, 11.8.x, and 12.3.x | 11.4.12, 11.8.8, and 12.3.2 | -| Dolt | 2.2.x with `>=2.2.2,<2.3.0` | 2.2.2 and 2.2.3 | +| Dolt writes | Exactly 2.2.2 and 2.2.3 | 2.2.2 and 2.2.3 | Microsoft SQL Server (MSSQL) remains roadmap-only and is not a supported runtime profile; see the specification's [MSSQL roadmap](https://github.com/OpenStatSpec/specification/blob/main/docs/mssql-dialect-roadmap.md). @@ -124,7 +126,7 @@ Use these explicit SQLAlchemy URLs: - SQLite: `sqlite:///dataset.sqlite` - PostgreSQL: `postgresql+psycopg://user:password@host/database` - MySQL/MariaDB: `mysql+pymysql://user:password@host/database` -- Dolt `>=2.2.2,<2.3.0`: `mysql+pymysql://user:password@host/database` (detected by server identity) +- Dolt 2.2.2 or 2.2.3: `mysql+pymysql://user:password@host/database` (detected by server identity) The Dolt core profile supports strict wide-table import, validation, and export; the separate Transformation Workflow is unsupported. @@ -135,8 +137,12 @@ dictionary semantics cannot be reproduced, it stops until you pass the exact diagnostic code with `--allow-loss`. Diagnostics are returned to the caller, not persisted. Reads, validation, and SAV/ZSAV export never write to the database, including on failure. Export returns no `operation_id` and needs only read -permissions; Dolt reads do not require a write-conformance declaration. Imports -and transformations still require an exact matching Dolt write declaration. +permissions; Dolt reads have no write-version gate. Imports and transformations +use the packaged exact-version policy by default. Active driver/identity checks, +limit preflight, and the clean expected branch/HEAD guard for in-place apply +remain mandatory. Dolt's reported limits are adapter safety budgets, narrowed +by the active packet limit, not proven native server ceilings; full boundary +conformance is not claimed. The matrix is also available to Python callers as `openstatspec.capability_matrix()`. It distinguishes supported semantics from diff --git a/docs/release-readiness.md b/docs/release-readiness.md index 658ab4d..e68fd26 100644 --- a/docs/release-readiness.md +++ b/docs/release-readiness.md @@ -20,15 +20,17 @@ database contract. | PostgreSQL | `postgresql+psycopg://…` | 17.x and 18.x | 17.10 and 18.4 | | MySQL | `mysql+pymysql://…` | 8.4.x and 9.7.x | 8.4.11 and 9.7.2 | | MariaDB | `mysql+pymysql://…` | 11.4.x, 11.8.x, and 12.3.x | 11.4.12, 11.8.8, and 12.3.2 | -| Dolt | `mysql+pymysql://…` | 2.2.x with `>=2.2.2,<2.3.0` | 2.2.2 and 2.2.3 | +| Dolt writes | `mysql+pymysql://…` | Exactly 2.2.2 and 2.2.3 | 2.2.2 and 2.2.3 | Separate service matrices test the shared MySQL/MariaDB profile contract while retaining distinct active-server identities and version claims. Dolt is an -independent core profile that accepts only canonical stable versions in -`>=2.2.2,<2.3.0`; the optional Transformation Workflow remains SQLite-only. This does not -claim coverage for every server configuration. The machine-readable capability -declaration reports both the theoretical profile boundaries and effective -limits observed from the active connection. +independent core profile with a packaged exact-version write policy; normal +callers supply no conformance files. An explicit external declaration source +remains a strict override. Read-only validation/export has no write-version gate. +The optional Transformation Workflow remains SQLite-only. This does not claim +coverage for every server configuration or proven Dolt native limit ceilings. +Dolt capabilities report adapter safety budgets and active packet constraints; +full boundary conformance remains pending. Every server service job compares the normalized live product version with its exact matrix entry, so a moved or mismatched image cannot substantiate the CI declaration. SQLite's optional Transformation Workflow retains its narrower @@ -102,6 +104,27 @@ implemented openstatspec.frontends.spss package. Stata and SAS remain empty source-tree placeholders and must expose no compiler, apply API, CLI choice, capability claim, or implied support. +## Default Dolt write verification + +Local release verification used the exact 2.2.2 and 2.2.3 image digests in +`.github/workflows/ci.yml`, isolated ports 13482/13483, and disposable `/tmp` +database directories. With `PYTHONPATH=src:../pyspssio`, +`OPENSTATSPEC_SPECIFICATION_DIR=/tmp/oss-php-spec-cd8f198`, and each matching +`OPENSTATSPEC_DOLT_URL` / `OPENSTATSPEC_EXPECTED_DOLT_VERSION`: + +- `../python/.venv/bin/python -m pytest`: **366 passed, 49 skipped per pin**. +- CI selection `-m 'services and not candidate_evidence'`: **7 passed, + 41 skipped per pin**; other database services were not configured locally. +- Dolt 2.3.0 on port 13387: read-only export and unknown-version write rejection + checks passed (**19 passed**) using disposable fixture databases/users. +- `python -m compileall -q src tests` and `git diff --check` passed. + +Live failures exposed and now cover two previously dormant write blockers: +owned table additions being misclassified as unrelated diffs, and Dolt retaining +`@@autocommit=1` despite the driver's setting. Writes now explicitly begin their +transaction. The candidate storage/identifier/column smoke probe also passed on +both pins; it does **not** establish full value/row/statement limit conformance. + ## Maintainer checks before tagging 1. Publish the pinned `openstatspec-pyspssio==0.5.1.post2` engine distribution @@ -110,7 +133,12 @@ capability claim, or implied support. 2. Run `python -m pytest -m "not services"`. 3. Confirm the GitHub Actions matrix is green for exact PostgreSQL 17.10/18.4, MySQL 8.4.11/9.7.2, MariaDB 11.4.12/11.8.8/12.3.2, and exact Dolt - 2.2.2/2.2.3 service evidence from the immutable image pins in CI. + 2.2.2/2.2.3 service evidence from the immutable image pins in CI. The Dolt + jobs must perform default import/validate/SAV+ZSAV export, in-place recode + and labels with unchanged dataset/table counts and HEAD, injected data, + catalog, and audit rollback checks, and failed-import cleanup. Release/CI + owns this evidence, not per-user runtime declaration files. Candidate limit + probes remain non-claiming and separate from these write gates. 4. Build with `python -m build` and install the generated wheel in a clean environment. 5. Confirm `openstatspec capabilities` reflects the intended support boundary. diff --git a/src/openstatspec/sql/capabilities.py b/src/openstatspec/sql/capabilities.py index 97fcd73..60daafa 100644 --- a/src/openstatspec/sql/capabilities.py +++ b/src/openstatspec/sql/capabilities.py @@ -24,8 +24,10 @@ DOLT_WRITE_CONFORMANCE = { "declaration_schema_id": "openstatspec-dolt-adapter-declaration-v1", - "write_enabled": False, - "status": "adapter_owned_concrete_declarations_required", + "write_enabled": True, + "status": "packaged_exact_version_policy", + "declaration_count": 0, + "declarations_available": False, } @@ -82,7 +84,7 @@ def _dolt_driver_eligible(database_url: str) -> bool: def dolt_operational_write_enabled( active: Mapping[str, Any], *, declaration_matched: bool, ) -> bool: - """Require both exact declaration and claimed active driver for writes.""" + """Require an exact write-policy match and the claimed active driver.""" return declaration_matched and bool(active.get("driver_eligible", False)) @@ -91,7 +93,12 @@ def _dolt_write_enabled( *, active_product_version: str | None = None, ) -> bool: - conformance = _conformance_source(source) + if source is None: + return ( + True if active_product_version is None + else active_product_version.strip() in SERVER_POLICIES["dolt"]["claimed"] + ) + conformance = source if active_product_version is None: return bool(conformance.status()["write_enabled"]) conformance.require_exact_match( @@ -115,8 +122,8 @@ def _dolt_write_enabled( "ci": ["MariaDB 11.4.12", "MariaDB 11.8.8", "MariaDB 12.3.2"], }, "dolt": { - "claimed": [], - "ci": [], + "claimed": ["2.2.2", "2.2.3"], + "ci": ["2.2.2", "2.2.3"], }, "postgresql": { "claimed": ["PostgreSQL 17.x", "PostgreSQL 18.x"], @@ -131,13 +138,13 @@ def profile_declarations( dolt_conformance_source: DoltConformanceSource | None = None, ) -> dict[str, dict[str, Any]]: """Declare every profile and, optionally, one active connection.""" - source = _conformance_source(dolt_conformance_source) + source = dolt_conformance_source active = ( active_connection(database_url, dolt_conformance_source=source) if database_url else None ) dolt_declaration = None - if active and active["profile"] == "dolt": + if source is not None and active and active["profile"] == "dolt": try: dolt_declaration = source.require_exact_match( active_product_version=active["raw_product_version"], @@ -290,7 +297,7 @@ def effective_profile( *, dolt_conformance_source: DoltConformanceSource | None = None, ) -> tuple[SqlProfile, dict[str, Any]]: - """Resolve and enforce the write profile, including exact Dolt conformance.""" + """Enforce the write profile and exact Dolt policy or explicit declaration.""" return _effective_profile(database_url, dolt_conformance_source, for_write=True) @@ -307,7 +314,7 @@ def _effective_profile( database_url: str, dolt_conformance_source: DoltConformanceSource | None, *, for_write: bool, ) -> tuple[SqlProfile, dict[str, Any]]: - source = _conformance_source(dolt_conformance_source) + source = dolt_conformance_source configured = validate_connection_url(database_url) active = active_connection(database_url, dolt_conformance_source=source) if configured is not MYSQL and active["profile"] != configured.name: @@ -326,7 +333,7 @@ def _effective_profile( # write-conformance envelope or INSERT statement/packet limits. return DOLT, active dolt_declaration = None - if active["profile"] == "dolt": + if active["profile"] == "dolt" and source is not None: dolt_declaration = source.require_exact_match( active_product_version=active["raw_product_version"], specification_commit=_bound_specification_commit(), @@ -392,13 +399,18 @@ def _profile( dolt_declaration: Mapping[str, Any] | None = None, ) -> dict[str, Any]: dolt_envelope = name == "dolt" + default_dolt = dolt_envelope and dolt_conformance_source is None + dolt_supported = bool( + default_dolt and active and active["profile"] == "dolt" + and server_version_supported("dolt", active["raw_product_version"]) + ) identifier = { "value": profile.identifier_limit, "unit": "characters" if name in {"mysql", "mariadb"} else "bytes", "source": ( "MySQL/MariaDB native identifier limit" if name in {"mysql", "mariadb"} else "PostgreSQL NAMEDATALEN minus one native byte limit" if name == "postgresql" - else "OpenStatSpec Dolt adapter envelope pending pinned live boundary evidence" + else "OpenStatSpec Dolt adapter safety envelope (not a proven server ceiling)" if dolt_envelope else "OpenStatSpec profile boundary; SQLite has no fixed native identifier limit" ), @@ -416,7 +428,7 @@ def _profile( if dolt_envelope: adapter_envelope = { **profile_limits, - "limit_basis": "proposed_adapter_envelope", + "limit_basis": "adapter_safety_envelope", "evidence_status": "pending_pinned_live_conformance", } theoretical = { @@ -435,15 +447,16 @@ def _profile( effective = None status = "not_connected" if active and active["profile"] == name and ( - not dolt_envelope or dolt_declaration is not None + not dolt_envelope or dolt_declaration is not None or dolt_supported ): effective = ( dolt_effective_limits(dolt_declaration) if dolt_envelope and dolt_declaration is not None + else dict(adapter_envelope) if default_dolt else dict(theoretical) ) default_source = ( - "proposed adapter envelope pending pinned live conformance" + "adapter safety envelope; full boundary conformance not claimed" if dolt_envelope else "profile theoretical engine ceiling" ) sources = { @@ -482,7 +495,9 @@ def _profile( effective["maximum_value_bytes"] = min( effective["maximum_value_bytes"], payload, ) - effective["maximum_statement_bytes"] = payload + effective["maximum_statement_bytes"] = min( + effective["maximum_statement_bytes"] or payload, payload, + ) sources["maximum_value_bytes"] = "active @@max_allowed_packet worst-case payload" sources["maximum_statement_bytes"] = "active @@max_allowed_packet worst-case payload" status = "active_connection_mixed" @@ -494,20 +509,25 @@ def _profile( policy = SERVER_POLICIES[name] dolt_declarations = ( _validated_dolt_declarations(dolt_conformance_source) - if dolt_envelope else () + if dolt_envelope and not default_dolt else () ) dolt_claimed_versions = sorted({ version for declaration in dolt_declarations - for version in declaration["claimed_product_versions"] + for version in declaration.get( + "claimed_product_versions", [declaration["active_product_version"]], + ) }) dolt_tested_versions = sorted({ version for declaration in dolt_declarations - for version in declaration["tested_product_versions"] + for version in declaration.get( + "tested_product_versions", [declaration["active_product_version"]], + ) }) dolt_status = ( - _conformance_source(dolt_conformance_source).status() + DOLT_WRITE_CONFORMANCE if default_dolt + else _conformance_source(dolt_conformance_source).status() if dolt_envelope else None ) return { @@ -520,10 +540,10 @@ def _profile( "specification_commit": SPECIFICATION_COMMIT, "driver": "psycopg" if name == "postgresql" else "PyMySQL" if name in MYSQL_WIRE_PROFILES else "sqlite3", "claimed_server_versions": ( - dolt_claimed_versions if dolt_envelope else policy["claimed"] + dolt_claimed_versions if dolt_envelope and not default_dolt else policy["claimed"] ), "ci_tested_server_versions": ( - dolt_tested_versions if dolt_envelope else policy["ci"] + dolt_tested_versions if dolt_envelope and not default_dolt else policy["ci"] ), "write_conformance": ( { @@ -543,7 +563,7 @@ def _profile( ), "operational_write_enabled": ( dolt_operational_write_enabled( - active, declaration_matched=dolt_declaration is not None, + active, declaration_matched=dolt_declaration is not None or dolt_supported, ) if dolt_envelope and active and active["profile"] == name else ( @@ -599,7 +619,6 @@ def server_version_supported( "postgresql": {(17,), (18,)}, "mysql": {(8, 4), (9, 7)}, "mariadb": {(11, 4), (11, 8), (12, 3)}, - "dolt": {(2, 2, 2)}, }[profile] width = len(next(iter(allowed))) return version[:width] in allowed diff --git a/src/openstatspec/sql/dolt_conformance.py b/src/openstatspec/sql/dolt_conformance.py index 71375ff..6e9659c 100644 --- a/src/openstatspec/sql/dolt_conformance.py +++ b/src/openstatspec/sql/dolt_conformance.py @@ -1,4 +1,7 @@ -"""Adapter-owned validation for language-neutral Dolt declarations.""" +"""Optional strict validation for external, adapter-owned Dolt declarations. + +Default writes use the exact-version policy in capabilities, not this registry. +""" from __future__ import annotations @@ -280,7 +283,7 @@ class DoltConformanceSource: @classmethod def packaged(cls) -> "DoltConformanceSource": - """Return the empty built-in registry; no Dolt write claim is packaged.""" + """Return the legacy empty registry, not the default write-version policy.""" return cls() @classmethod diff --git a/src/openstatspec/sql/inplace_transform.py b/src/openstatspec/sql/inplace_transform.py index 0f458be..608ed81 100644 --- a/src/openstatspec/sql/inplace_transform.py +++ b/src/openstatspec/sql/inplace_transform.py @@ -984,6 +984,9 @@ def _run_in_place_submission( # catalog mutations roll back together before the write lock # is released. connection.exec_driver_sql("BEGIN") + if profile.name == "dolt": + # Dolt may retain @@autocommit=1 despite PyMySQL's setting. + connection.exec_driver_sql("BEGIN") require_verified_catalog(connection) if profile.name == "dolt": if not expected_branch or not expected_head: diff --git a/src/openstatspec/sql/wide.py b/src/openstatspec/sql/wide.py index 50f8503..adf424b 100644 --- a/src/openstatspec/sql/wide.py +++ b/src/openstatspec/sql/wide.py @@ -510,12 +510,19 @@ def _capture_dolt_state( "diff_summaries": {}, } - def allowed_audit_data_diff(row: Mapping[str, Any]) -> bool: + def allowed_owned_diff(row: Mapping[str, Any]) -> bool: values = dict(row) from_name = str(values.get("from_table_name") or "") to_name = str(values.get("to_table_name") or "") data_change = str(values.get("data_change")).strip().lower() schema_change = str(values.get("schema_change")).strip().lower() + # Initialization/import may create their own tables without a Dolt + # commit. Do not classify those HEAD-to-WORKING additions as unrelated. + if ( + not from_name and to_name in audit_relations + and str(values.get("diff_type") or "").strip().lower() == "added" + ): + return True return ( from_name == to_name and from_name in audit_relations @@ -545,7 +552,7 @@ def allowed_audit_data_diff(row: Mapping[str, Any]) -> bool: unrelated["diff_summaries"][label] = _dolt_evidence_block( ( row for row in rows - if not allowed_audit_data_diff(row) + if not allowed_owned_diff(row) ), expected_keys=expected_keys, ) @@ -619,6 +626,9 @@ def _bound_catalog_transaction( connection.rollback() try: with connection.begin(): + if profile_name == "dolt": + # Dolt may retain @@autocommit=1 despite PyMySQL's setting. + connection.exec_driver_sql("BEGIN") yield connection after = _capture_dolt_state( connection, profile_name=profile_name, diff --git a/tests/test_dolt_conformance.py b/tests/test_dolt_conformance.py index 0e57a79..7474a0c 100644 --- a/tests/test_dolt_conformance.py +++ b/tests/test_dolt_conformance.py @@ -39,7 +39,7 @@ def test_directory_source_is_explicit_and_invalid_root_fails_closed( def test_directory_source_validates_adapter_owned_declaration_and_evidence( - tmp_path: Path, + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: declaration_directory = tmp_path / "sql/dolt-adapter-declarations" evidence = declaration_directory / "evidence/result.json" @@ -93,6 +93,27 @@ def effective(value: object, unit: str) -> list[dict[str, object]]: ) loaded = DoltConformanceSource.from_directory(tmp_path).validated_declarations() assert loaded == (declaration,) + source = DoltConformanceSource.from_directory(tmp_path) + monkeypatch.setattr(capability_module, "SPECIFICATION_COMMIT", "a" * 40) + monkeypatch.setattr( + capability_module, "active_connection", + lambda *_args, **_kwargs: { + "profile": "dolt", "raw_product_version": "2.2.2", + "claimed_supported": True, "driver_eligible": True, + "observed": {"max_allowed_packet": 1_073_741_824}, + }, + ) + profile, _ = capability_module.effective_profile( + "mysql+pymysql://example.invalid/catalog", dolt_conformance_source=source, + ) + assert profile.max_source_variables == 1016 + assert profile.max_statement_bytes == 1_000_000 + assert capability_module.profile_declarations( + dolt_conformance_source=source, + )["dolt"]["claimed_server_versions"] == ["2.2.2"] + assert not capability_module.server_version_supported( + "dolt", "2.2.3", dolt_conformance_source=source, + ) def test_specification_is_not_a_python_runtime_dependency() -> None: @@ -107,11 +128,26 @@ def test_specification_is_not_a_python_runtime_dependency() -> None: for dependency in dependencies ) declaration = capability_module.profile_declarations()["dolt"] - assert declaration["operational_write_enabled"] is False - assert declaration["write_conformance"]["write_enabled"] is False - assert declaration["write_conformance"]["status"] == "blocked_no_concrete_declarations" - assert declaration["claimed_server_versions"] == [] - assert declaration["ci_tested_server_versions"] == [] + assert declaration["operational_write_enabled"] is True + assert declaration["write_conformance"]["write_enabled"] is True + assert declaration["write_conformance"]["status"] == "packaged_exact_version_policy" + assert declaration["claimed_server_versions"] == ["2.2.2", "2.2.3"] + assert declaration["ci_tested_server_versions"] == ["2.2.2", "2.2.3"] + + +def test_explicit_empty_source_overrides_supported_default(monkeypatch) -> None: + monkeypatch.setattr( + capability_module, "active_connection", + lambda *_args, **_kwargs: { + "profile": "dolt", "raw_product_version": "2.2.2", + "claimed_supported": True, + }, + ) + with pytest.raises(UnsupportedOperationError, match="no concrete declarations"): + capability_module.effective_profile( + "mysql+pymysql://example.invalid/catalog", + dolt_conformance_source=DoltConformanceSource.packaged(), + ) def test_exact_match_binds_active_product_adapter_and_specification( diff --git a/tests/test_read_only_export.py b/tests/test_read_only_export.py index 246493e..d46d56c 100644 --- a/tests/test_read_only_export.py +++ b/tests/test_read_only_export.py @@ -213,6 +213,7 @@ def test_dolt_reads_do_not_enable_undeclared_writes(monkeypatch): url = "mysql+pymysql://reader@host/dataset" active = { "profile": "dolt", "raw_product_version": "2.3.0", "claimed_supported": False, + "server_version": "2.3.0", } monkeypatch.setattr(capabilities, "active_connection", lambda *_a, **_k: active) assert capabilities.read_profile(url)[0].name == "dolt" diff --git a/tests/test_review_transaction_boundary.py b/tests/test_review_transaction_boundary.py index 2dd8f9a..00dd877 100644 --- a/tests/test_review_transaction_boundary.py +++ b/tests/test_review_transaction_boundary.py @@ -80,6 +80,19 @@ def _dolt_summary(**changes): return result +def test_dolt_classifier_allows_owned_table_creation(): + added = _dolt_summary( + from_table_name="", diff_type="added", schema_change=1, + ) + for owned in (True, False): + state = wide._capture_dolt_state( + _DoltDiffConnection(added), profile_name="dolt", + audit_relations={"operation"} if owned else set(), + ) + for evidence in state["unrelated_working_set"]["diff_summaries"].values(): + assert evidence["rows"] == ([] if owned else [added]) + + def test_dolt_classifier_allows_only_same_table_audit_data_changes(): allowed = wide._capture_dolt_state( _DoltDiffConnection(_dolt_summary()), diff --git a/tests/test_sql_profiles.py b/tests/test_sql_profiles.py index 9591a2e..8e07253 100755 --- a/tests/test_sql_profiles.py +++ b/tests/test_sql_profiles.py @@ -121,8 +121,8 @@ def test_claimed_server_version_policy_is_explicit( "2.2.03", "garbage", ], ) -def test_dolt_without_concrete_declaration_is_not_claimed(version: str) -> None: - assert server_version_supported("dolt", version) is False +def test_dolt_default_policy_requires_a_supported_exact_version(version: str) -> None: + assert server_version_supported("dolt", version) is (version.strip() in {"2.2.2", "2.2.3"}) @pytest.mark.services @@ -148,15 +148,15 @@ def test_active_server_identity_matches_claimed_ci_profile( @pytest.mark.services -def test_live_dolt_identity_is_observed_but_not_claimed_without_declarations() -> None: +def test_live_dolt_identity_matches_default_write_policy() -> None: database_url = os.environ.get("OPENSTATSPEC_DOLT_URL") if not database_url: pytest.skip("OPENSTATSPEC_DOLT_URL is not configured") active = active_connection(database_url) assert active["profile"] == "dolt" - assert active["claimed_supported"] is False - assert active["matched_claim"] is None + assert active["claimed_supported"] is True + assert active["matched_claim"] == active["raw_product_version"] @pytest.mark.services @@ -220,8 +220,8 @@ def test_server_policy_distinguishes_claimed_families_from_exact_ci_evidence() - assert declarations["postgresql"]["ci_tested_server_versions"] == [ "PostgreSQL 17.10", "PostgreSQL 18.4", ] - assert declarations["dolt"]["claimed_server_versions"] == [] - assert declarations["dolt"]["ci_tested_server_versions"] == [] + assert declarations["dolt"]["claimed_server_versions"] == ["2.2.2", "2.2.3"] + assert declarations["dolt"]["ci_tested_server_versions"] == ["2.2.2", "2.2.3"] class _ProbeResult: @@ -293,8 +293,8 @@ def test_dolt_identity_is_exact_and_publishes_wire_and_product_versions(monkeypa assert active["identity_source"] == ( "SELECT @@version, @@version_comment, DOLT_VERSION(), ACTIVE_BRANCH()" ) - assert active["claimed_supported"] is False - assert active["matched_claim"] is None + assert active["claimed_supported"] is True + assert active["matched_claim"] == "2.2.3" assert active["working_set_binding"]["active_branch"] == "main" assert connection.calls == [ "select @@version", "select @@version_comment", "select DOLT_VERSION()", @@ -388,29 +388,27 @@ def test_non_dolt_products_never_call_the_dolt_function(monkeypatch) -> None: assert "select DOLT_VERSION()" not in connection.calls -def test_effective_profile_fails_closed_without_dolt_declarations(monkeypatch) -> None: - active = { - "profile": "dolt", - "raw_product_version": "2.2.3", - } - monkeypatch.setattr( - capabilities, "active_connection", lambda _url, **_kwargs: active, - ) - - with pytest.raises(UnsupportedOperationError, match="no concrete declarations"): - effective_profile("mysql+pymysql://user@host/database") +def test_effective_profile_uses_default_dolt_limits(monkeypatch) -> None: + _mock_mysql_probes(monkeypatch) + profile, active = effective_profile("mysql+pymysql://user@host/database") + assert active["claimed_supported"] is True + assert profile.name == "dolt" + assert profile.max_source_variables == DOLT.max_source_variables + assert profile.identifier_limit == DOLT.identifier_limit + assert profile.max_statement_bytes == (1_073_741_824 - 131_072) // 2 + assert profile.max_text_value_bytes == profile.max_statement_bytes -def test_dolt_declaration_is_fail_closed_without_concrete_evidence() -> None: +def test_dolt_default_policy_does_not_claim_proven_server_limits() -> None: declaration = capabilities.profile_declarations()["dolt"] - assert declaration["claimed_server_versions"] == [] - assert declaration["ci_tested_server_versions"] == [] - assert declaration["operational_write_enabled"] is False + assert declaration["claimed_server_versions"] == ["2.2.2", "2.2.3"] + assert declaration["ci_tested_server_versions"] == ["2.2.2", "2.2.3"] + assert declaration["operational_write_enabled"] is True assert declaration["effective_limits"] is None assert declaration["effective_limits_status"] == "not_connected" assert declaration["write_conformance"]["declaration_count"] == 0 - assert declaration["write_conformance"]["write_enabled"] is False - assert declaration["adapter_envelope"]["limit_basis"] == "proposed_adapter_envelope" + assert declaration["write_conformance"]["write_enabled"] is True + assert declaration["adapter_envelope"]["limit_basis"] == "adapter_safety_envelope" assert declaration["theoretical_limits"]["limit_basis"] == ( "server_limits_not_claimed" ) @@ -553,7 +551,7 @@ def test_wide_string_column_uses_effective_dolt_storage() -> None: def test_capability_inspection_reports_unbound_dolt_as_disabled(monkeypatch) -> None: active = { "profile": "dolt", - "raw_product_version": "2.2.3", + "raw_product_version": "2.3.0", "observed": {"max_allowed_packet": 1_073_741_824}, } monkeypatch.setattr( @@ -582,6 +580,9 @@ def __enter__(self): def __exit__(self, *_args): return None + def exec_driver_sql(self, statement): + assert statement == "BEGIN" + def rollback(self): self.rollback_called = True diff --git a/tests/test_sql_services.py b/tests/test_sql_services.py index a442c47..1537435 100755 --- a/tests/test_sql_services.py +++ b/tests/test_sql_services.py @@ -1,4 +1,4 @@ -"""Real-service conformance checks plus Dolt fail-closed and candidate probes.""" +"""Real-service writes, round trips, failure boundaries, and candidate probes.""" import os from uuid import uuid4 @@ -11,7 +11,7 @@ import openstatspec from openstatspec.core import UnsupportedOperationError -from openstatspec.sql.dolt_conformance import DoltConformanceSource +from openstatspec.sql import inplace_transform, wide from openstatspec.sql.normative import ( catalog as normative_catalog, delete_dataset_representation, @@ -42,12 +42,16 @@ def source_sav(tmp_path): @pytest.mark.parametrize( ("environment_name", "dataset_id"), - [("OPENSTATSPEC_POSTGRES_URL", "profile_pg"), ("OPENSTATSPEC_MYSQL_URL", "profile_mysql"), ("OPENSTATSPEC_MARIADB_URL", "profile_mariadb")], + [("OPENSTATSPEC_POSTGRES_URL", "profile_pg"), ("OPENSTATSPEC_MYSQL_URL", "profile_mysql"), ("OPENSTATSPEC_MARIADB_URL", "profile_mariadb"), ("OPENSTATSPEC_DOLT_URL", "profile_dolt")], ) def test_live_profile_import_validate_and_export(environment_name, dataset_id, source_sav, tmp_path): database_url = os.environ.get(environment_name) if not database_url: pytest.skip(f"{environment_name} is not configured") + before = ( + openstatspec.dolt_state_snapshot(database_url=database_url)["state"] + if environment_name == "OPENSTATSPEC_DOLT_URL" else None + ) openstatspec.initialize_catalog(database_url=database_url) runtime_dataset_id = f"{dataset_id}_{uuid4().hex[:8]}" imported = openstatspec.import_sav( @@ -74,11 +78,15 @@ def test_live_profile_import_validate_and_export(environment_name, dataset_id, s assert frame["name"].tolist() == ["Ada", ""] assert metadata["var_labels"] == {"age": "Age", "name": "Name"} assert metadata["var_value_labels"] == {"age": {34.0: "thirty-four"}} + if before is not None: + after = openstatspec.dolt_state_snapshot(database_url=database_url)["state"] + assert after["head"] == before["head"] + assert after["active_branch"] == before["active_branch"] @pytest.mark.parametrize( ("environment_name", "dataset_id"), - [("OPENSTATSPEC_POSTGRES_URL", "semantics_pg"), ("OPENSTATSPEC_MYSQL_URL", "semantics_mysql"), ("OPENSTATSPEC_MARIADB_URL", "semantics_mariadb")], + [("OPENSTATSPEC_POSTGRES_URL", "semantics_pg"), ("OPENSTATSPEC_MYSQL_URL", "semantics_mysql"), ("OPENSTATSPEC_MARIADB_URL", "semantics_mariadb"), ("OPENSTATSPEC_DOLT_URL", "semantics_dolt")], ) @pytest.mark.parametrize("suffix", [".sav", ".zsav"]) def test_live_profile_preserves_supported_sav_semantics(environment_name, dataset_id, suffix, tmp_path): @@ -103,23 +111,18 @@ def test_live_profile_preserves_supported_sav_semantics(environment_name, datase ) assert compare_sav_semantics(source, destination) == {"equivalent": True, "differences": []} -def test_live_dolt_is_read_only_and_rejects_writes_without_declarations( +def test_live_unknown_dolt_is_read_only_and_rejects_default_writes( tmp_path, ) -> None: - database_url = os.environ.get("OPENSTATSPEC_DOLT_URL") + database_url = os.environ.get("OPENSTATSPEC_UNSUPPORTED_DOLT_URL") if not database_url: - pytest.skip("OPENSTATSPEC_DOLT_URL is not configured") - - status = DoltConformanceSource.packaged().status() - assert status["status"] == "blocked_no_concrete_declarations" - assert status["declaration_count"] == 0 - assert status["write_enabled"] is False + pytest.skip("OPENSTATSPEC_UNSUPPORTED_DOLT_URL is not configured") before = openstatspec.dolt_state_snapshot(database_url=database_url) assert before["read_only"] is True assert before["operational_write_enabled"] is False - rejection = "no concrete declarations; write rejected before mutation" + rejection = "not claimed supported" with pytest.raises(UnsupportedOperationError, match=rejection): openstatspec.initialize_catalog(database_url=database_url) after_initialize = openstatspec.dolt_state_snapshot(database_url=database_url) @@ -138,6 +141,133 @@ def test_live_dolt_is_read_only_and_rejects_writes_without_declarations( assert after_import["state"] == before["state"] +def test_live_dolt_default_in_place_atomicity(source_sav, tmp_path, monkeypatch): + database_url = os.environ.get("OPENSTATSPEC_DOLT_URL") + if not database_url: + pytest.skip("OPENSTATSPEC_DOLT_URL is not configured") + imported = openstatspec.import_sav( + source_sav, database_url=database_url, dataset_id="apply_" + uuid4().hex[:8], + ) + openstatspec.install_in_place_transformation_schema(database_url=database_url) + engine = create_engine(database_url) + # The caller, not the adapter, commits the imported baseline before apply. + with engine.begin() as connection: + connection.exec_driver_sql("CALL DOLT_ADD('.')") + connection.exec_driver_sql( + "CALL DOLT_COMMIT('-m', 'test baseline', '--author', 'Test ')" + ) + before = openstatspec.dolt_state_snapshot(database_url=database_url)["state"] + assert before["status"]["rows"] == [] + tables = inspect_database(engine).get_table_names() + + def contents(): + with engine.connect() as connection: + return { + table: tuple(connection.exec_driver_sql( + "SELECT * FROM " + connection.dialect.identifier_preparer.quote(table) + )) for table in tables + } + + original = contents() + schema = openstatspec.VariableSchema(( + openstatspec.VariableDefinition("age", "numeric"), + openstatspec.VariableDefinition("name", "string", declared_string_width=12), + )) + plan = openstatspec.compile_spss_syntax( + "RECODE age (34 = 35). VARIABLE LABELS age 'Updated age'. " + "VALUE LABELS age 35 'thirty-five'.", schema, + ).plan + arguments = dict( + database_url=database_url, dataset_id=imported["dataset_id"], + plan=plan, actor="service-test", expected_branch=before["active_branch"], + expected_head=before["head"], + ) + for key, value, code in ( + ("expected_branch", "wrong-branch", "dolt_branch_mismatch"), + ("expected_head", "wrong-head", "dolt_head_mismatch"), + ("expected_head", None, "dolt_context_required"), + ): + with pytest.raises(openstatspec.TransformationError) as caught: + openstatspec.apply_transformation_plan_in_place(**{**arguments, key: value}) + assert caught.value.code == code + create_plan = openstatspec.compile_spss_syntax( + "RECODE age (34 = 35) INTO new_age.", schema, + ).plan + with pytest.raises(openstatspec.TransformationError) as caught: + openstatspec.apply_transformation_plan_in_place(**{**arguments, "plan": create_plan}) + assert caught.value.code == "schema_change_not_atomic" + for boundary in ("data", "catalog", "audit"): + with monkeypatch.context() as patch: + def fail(phase): + if phase == boundary: + raise RuntimeError("injected " + phase) + patch.setattr(inplace_transform, "_failure_boundary", fail) + with pytest.raises(RuntimeError, match="injected " + boundary): + openstatspec.apply_transformation_plan_in_place(**arguments) + assert contents() == original + assert inspect_database(engine).get_table_names() == tables + assert openstatspec.dolt_state_snapshot(database_url=database_url)["state"] == before + + result = openstatspec.apply_transformation_plan_in_place(**arguments) + assert result["dataset_id"] == imported["dataset_id"] + assert result["physical_table_name"] == imported["data_table"] + assert result["dolt_commit_performed"] is False + assert inspect_database(engine).get_table_names() == tables + assert contents()["dataset"] == original["dataset"] + after = openstatspec.dolt_state_snapshot(database_url=database_url)["state"] + assert after["head"] == before["head"] + assert after["active_branch"] == before["active_branch"] + assert after["status"]["rows"] + with pytest.raises(openstatspec.TransformationError) as caught: + openstatspec.apply_transformation_plan_in_place(**arguments) + assert caught.value.code == "dolt_working_set_dirty" + assert openstatspec.validate(database_url=database_url, dataset_id=imported["dataset_id"])["valid"] + destination = tmp_path / "recoded.sav" + openstatspec.export_sav( + database_url=database_url, dataset_id=imported["dataset_id"], destination=destination, + ) + frame, metadata = pyspssio.read_sav(str(destination), include_user_missing=True) + assert frame["age"].iloc[0] == 35.0 + assert pd.isna(frame["age"].iloc[1]) + assert metadata["var_labels"]["age"] == "Updated age" + assert metadata["var_value_labels"]["age"] == {35.0: "thirty-five"} + assert openstatspec.dolt_state_snapshot(database_url=database_url)["state"] == after + engine.dispose() + + +def test_live_dolt_default_import_failure_cleanup(source_sav, monkeypatch): + database_url = os.environ.get("OPENSTATSPEC_DOLT_URL") + if not database_url: + pytest.skip("OPENSTATSPEC_DOLT_URL is not configured") + openstatspec.initialize_catalog(database_url=database_url) + engine = create_engine(database_url) + tables = inspect_database(engine).get_table_names() + with engine.connect() as connection: + datasets = tuple(connection.exec_driver_sql("SELECT * FROM dataset")) + before = openstatspec.dolt_state_snapshot(database_url=database_url)["state"] + store = wide.store_normative_dataset + + def fail(*args, **kwargs): + store(*args, **kwargs) + raise RuntimeError("injected import failure") + + monkeypatch.setattr(wide, "store_normative_dataset", fail) + with pytest.raises(RuntimeError, match="injected import failure"): + openstatspec.import_sav( + source_sav, database_url=database_url, dataset_id="failed_" + uuid4().hex[:8], + ) + assert inspect_database(engine).get_table_names() == tables + with engine.connect() as connection: + assert tuple(connection.exec_driver_sql("SELECT * FROM dataset")) == datasets + assert connection.exec_driver_sql( + "SELECT COUNT(*) FROM operation WHERE status = 'failed'" + ).scalar_one() >= 1 + after = openstatspec.dolt_state_snapshot(database_url=database_url)["state"] + assert after["head"] == before["head"] + assert after["active_branch"] == before["active_branch"] + engine.dispose() + + @pytest.mark.candidate_evidence def test_live_dolt_candidate_limit_probe_smoke() -> None: database_url = os.environ.get("OPENSTATSPEC_DOLT_URL") From 6a6af2b6e034b52f5256c6bf2c601669a3f79b6f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?T=C3=B5nis=20Ormisson?= Date: Mon, 7 Sep 2026 17:56:23 +0300 Subject: [PATCH 3/4] Keep read checks independent of write limits and database creation --- src/openstatspec/api.py | 4 ++++ src/openstatspec/sql/capabilities.py | 7 +++++- src/openstatspec/sql/catalog_api.py | 3 +++ src/openstatspec/sql/database_urls.py | 17 ++++++++++++++ src/openstatspec/sql/wide.py | 4 +++- src/openstatspec/sql/workflow.py | 2 ++ tests/test_read_only_export.py | 34 +++++++++++++++++++++++---- 7 files changed, 64 insertions(+), 7 deletions(-) diff --git a/src/openstatspec/api.py b/src/openstatspec/api.py index 540982d..4e6405e 100644 --- a/src/openstatspec/api.py +++ b/src/openstatspec/api.py @@ -33,6 +33,7 @@ install_in_place_transformation_schema as _install_in_place_schema, ) from .transform import TransformationPlan +from .sql.database_urls import require_existing_database_url from .sql.capabilities import ( SPECIFICATION_COMMIT, SPECIFICATION_RELEASE, SPECIFICATION_STATUS, active_connection, catalog_binding, @@ -53,8 +54,11 @@ def capability_matrix( blocks export unless the documented explicit loss opt-in exists; a plain fail-closed feature has no faithful writer route at all. """ + if database_url: + require_existing_database_url(database_url) declaration = { "specification": "OpenStatSpec", + "database_io_policy": "openstatspec-database-io-v1", "specification_status": SPECIFICATION_STATUS, "specification_release": SPECIFICATION_RELEASE, "specification_commit": SPECIFICATION_COMMIT, diff --git a/src/openstatspec/sql/capabilities.py b/src/openstatspec/sql/capabilities.py index 60daafa..57b1741 100644 --- a/src/openstatspec/sql/capabilities.py +++ b/src/openstatspec/sql/capabilities.py @@ -3,6 +3,7 @@ from __future__ import annotations import re +import sys from dataclasses import replace from pathlib import Path from typing import Any, Mapping @@ -12,6 +13,7 @@ from .dolt_conformance import DoltConformanceSource, effective_limits as dolt_effective_limits from .normative import catalog +from .database_urls import require_existing_database_url from .profiles import DOLT, MYSQL, POSTGRESQL, SQLITE, MYSQL_WIRE_PROFILES, SqlProfile from .profiles import validate_connection_url from ..core import UnsupportedOperationError @@ -138,6 +140,8 @@ def profile_declarations( dolt_conformance_source: DoltConformanceSource | None = None, ) -> dict[str, dict[str, Any]]: """Declare every profile and, optionally, one active connection.""" + if database_url: + require_existing_database_url(database_url) source = dolt_conformance_source active = ( active_connection(database_url, dolt_conformance_source=source) @@ -307,6 +311,7 @@ def read_profile( dolt_conformance_source: DoltConformanceSource | None = None, ) -> tuple[SqlProfile, dict[str, Any]]: """Identify a read connection without requiring evidence for Dolt writes.""" + require_existing_database_url(database_url) return _effective_profile(database_url, dolt_conformance_source, for_write=False) @@ -331,7 +336,7 @@ def _effective_profile( if active["profile"] == "dolt" and not for_write: # Existing data is validated against the adapter's data model, not a # write-conformance envelope or INSERT statement/packet limits. - return DOLT, active + return replace(DOLT, max_source_variables=sys.maxsize), active dolt_declaration = None if active["profile"] == "dolt" and source is not None: dolt_declaration = source.require_exact_match( diff --git a/src/openstatspec/sql/catalog_api.py b/src/openstatspec/sql/catalog_api.py index 48c0296..ff29579 100644 --- a/src/openstatspec/sql/catalog_api.py +++ b/src/openstatspec/sql/catalog_api.py @@ -10,6 +10,7 @@ CATALOG_CONTRACT_ID, CATALOG_SCHEMA_VERSION, catalog as core_catalog, ) from .profiles import validate_connection_url +from .database_urls import require_existing_database_url from .workflow import ( PROFILE_ID, PROFILE_SCHEMA_VERSION, TransformationError, workflow_catalog, ) @@ -55,6 +56,7 @@ def _assert_workflow(connection: Any, tables: Any) -> None: def catalog_datasets(*, database_url: str, kind: str | None = None) -> dict[str, Any]: """List source and/or derived datasets without mutating either catalog.""" + require_existing_database_url(database_url) if kind not in {None, "core", "derived"}: raise ValueError("kind must be core, derived, or None.") engine = create_engine(database_url) @@ -96,6 +98,7 @@ def catalog_datasets(*, database_url: str, kind: str | None = None) -> dict[str, def catalog_dataset(*, database_url: str, dataset_id: str, kind: str) -> dict[str, Any]: """Return one public catalog record, variables, weight, and derived lineage.""" + require_existing_database_url(database_url) dataset_id = _uuid(dataset_id) if kind not in {"core", "derived"}: raise ValueError("kind must be core or derived.") diff --git a/src/openstatspec/sql/database_urls.py b/src/openstatspec/sql/database_urls.py index 467a915..ff44e1f 100644 --- a/src/openstatspec/sql/database_urls.py +++ b/src/openstatspec/sql/database_urls.py @@ -1,10 +1,27 @@ """Database URL invariants shared by persistent catalog operations.""" +from pathlib import Path +from urllib.parse import unquote + from sqlalchemy.engine import make_url from ..core import UnsupportedOperationError +def require_existing_database_url(database_url: str) -> None: + """Prevent SQLite's implicit file creation during a read operation.""" + url = make_url(database_url) + if url.get_backend_name() != "sqlite": + return + database = url.database or "" + if database in {"", ":memory:", "file::memory:"} or url.query.get("mode") == "memory": + return + if str(url.query.get("uri", "")).lower() == "true" and database.startswith("file:"): + database = unquote(database[5:]) + if not Path(database).is_file(): + raise UnsupportedOperationError("Read operations require an existing SQLite database file.") + + def require_persistent_database_url(database_url: str) -> None: """Reject SQLite URLs whose catalog disappears with a connection or engine.""" parsed_url = make_url(database_url) diff --git a/src/openstatspec/sql/wide.py b/src/openstatspec/sql/wide.py index adf424b..7f0d128 100644 --- a/src/openstatspec/sql/wide.py +++ b/src/openstatspec/sql/wide.py @@ -32,7 +32,7 @@ store_imported_dataset as store_normative_dataset, ) from .catalog_verification import verify_catalog_relations -from .database_urls import require_persistent_database_url +from .database_urls import require_existing_database_url, require_persistent_database_url _IDENTIFIER = re.compile(r"[^a-zA-Z0-9_]+") @@ -648,6 +648,7 @@ def dolt_state_snapshot( *, database_url: str, dolt_conformance_source: Any | None = None, ) -> dict[str, Any]: """Return read-only branch, HEAD, status, and diff evidence for Dolt.""" + require_existing_database_url(database_url) validate_connection_url(database_url) active = active_connection( database_url, dolt_conformance_source=dolt_conformance_source, @@ -1049,6 +1050,7 @@ def read_wide_dataset( dolt_conformance_source: Any | None = None, ) -> tuple[dict[str, Any], list[dict[str, Any]], list[dict[str, Any]]]: """Read an export descriptor from the normative catalog without mutation.""" + require_existing_database_url(database_url) if profile is None: profile, _active = read_profile( database_url, dolt_conformance_source=dolt_conformance_source, diff --git a/src/openstatspec/sql/workflow.py b/src/openstatspec/sql/workflow.py index 86c313b..7fbb154 100644 --- a/src/openstatspec/sql/workflow.py +++ b/src/openstatspec/sql/workflow.py @@ -2492,7 +2492,9 @@ def reconcile_physical_removals( def validate_derived_dataset(*, database_url: str, derived_dataset_id: str) -> dict[str, Any]: """Validate without creating or migrating catalog state.""" from .catalog_api import _assert_workflow + from .database_urls import require_existing_database_url + require_existing_database_url(database_url) derived_id = _uuid(derived_dataset_id, "derived_dataset_id") profile = validate_connection_url(database_url) engine = _workflow_engine(database_url, profile.name) diff --git a/tests/test_read_only_export.py b/tests/test_read_only_export.py index d46d56c..95d8184 100644 --- a/tests/test_read_only_export.py +++ b/tests/test_read_only_export.py @@ -135,10 +135,6 @@ def check(_connection, _cursor, statement, _parameters, _context, _many): def test_export_and_reads_leave_no_database_trace(read_catalog, tmp_path, suffix): url, snapshot, source = read_catalog before = snapshot() - if make_url(url).get_backend_name() == "mysql": - # This is the real server identity and the empty packaged registry. - with pytest.raises(UnsupportedOperationError): - wide.initialize_wide_catalog(database_url=url) output = tmp_path / ("export" + suffix) result = openstatspec.export_sav(database_url=url, dataset_id="sample", destination=output) assert "operation_id" not in result @@ -216,7 +212,11 @@ def test_dolt_reads_do_not_enable_undeclared_writes(monkeypatch): "server_version": "2.3.0", } monkeypatch.setattr(capabilities, "active_connection", lambda *_a, **_k: active) - assert capabilities.read_profile(url)[0].name == "dolt" + profile = capabilities.read_profile(url)[0] + assert profile.name == "dolt" + # A table imported under a larger explicit write envelope stays readable. + from openstatspec.sql.profiles import preflight + preflight(profile, 1_016) with pytest.raises(UnsupportedOperationError): capabilities.effective_profile(url) with pytest.raises(UnsupportedOperationError): @@ -228,6 +228,30 @@ def test_dolt_reads_do_not_enable_undeclared_writes(monkeypatch): ) +@pytest.mark.parametrize("operation", ["export", "validate", "capabilities", "list", "get", "derived", "dolt_state"]) +@pytest.mark.parametrize("uri", [False, True]) +def test_reads_do_not_create_a_missing_sqlite_file(tmp_path, operation, uri): + path = tmp_path / "absent.sqlite" + url = f"sqlite:///file:{path}?uri=true" if uri else f"sqlite:///{path}" + calls = { + "export": lambda: openstatspec.export_sav(database_url=url, dataset_id="sample", destination=tmp_path / "out.sav"), + "validate": lambda: openstatspec.validate(database_url=url, dataset_id="sample"), + "capabilities": lambda: openstatspec.capabilities(url), + "list": lambda: openstatspec.list_datasets(database_url=url), + "get": lambda: openstatspec.get_dataset(database_url=url, dataset_id=str(uuid4()), kind="core"), + "derived": lambda: openstatspec.validate_derived(database_url=url, derived_dataset_id=str(uuid4())), + "dolt_state": lambda: openstatspec.dolt_state_snapshot(database_url=url), + } + with pytest.raises(UnsupportedOperationError, match="existing SQLite"): + calls[operation]() + assert not path.exists() + assert not (tmp_path / "out.sav").exists() + + +def test_database_io_policy_is_explicit(): + assert openstatspec.capabilities()["database_io_policy"] == "openstatspec-database-io-v1" + + def test_derived_validation_does_not_initialize_catalog(tmp_path): url = f"sqlite:///{tmp_path / 'catalog.sqlite'}" openstatspec.initialize_catalog(database_url=url) From 6431067fbb54cba0ad129ac2934ad17104167677 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?T=C3=B5nis=20Ormisson?= Date: Mon, 7 Sep 2026 18:46:36 +0300 Subject: [PATCH 4/4] Prepare Python 0.8.0 release metadata and CI --- .github/workflows/ci.yml | 13 +++++-- .github/workflows/release.yml | 6 ++-- CHANGELOG.md | 24 +++++++++++++ README.md | 7 +++- docs/release-readiness.md | 52 ++++++++++++++++++++++++---- pyproject.toml | 2 +- src/openstatspec/api.py | 2 ++ src/openstatspec/sql/capabilities.py | 6 ++-- tests/test_cli.py | 9 +++-- tests/test_sql_profiles.py | 8 ++--- 10 files changed, 105 insertions(+), 24 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 5d408a5..652a610 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -20,7 +20,7 @@ jobs: uses: actions/checkout@v7 with: &specification-checkout repository: OpenStatSpec/specification - ref: cd8f198c68b849eb8ed018a894670a0904c2181d + ref: 864e84479f554b8ee250ffed44c4dfb963750d4a path: openstatspec-specification - name: Checkout required SPSS engine uses: actions/checkout@v7 @@ -53,9 +53,9 @@ jobs: - run: python -m pip install --upgrade pip build - run: python -m build - run: python -m venv /tmp/openstatspec-wheel-smoke - - run: /tmp/openstatspec-wheel-smoke/bin/python -m pip install ./openstatspec-pyspssio - run: /tmp/openstatspec-wheel-smoke/bin/python -m pip install dist/*.whl - - run: /tmp/openstatspec-wheel-smoke/bin/openstatspec capabilities + - run: env -u PYTHONPATH /tmp/openstatspec-wheel-smoke/bin/openstatspec capabilities + working-directory: /tmp postgres-integration: name: PostgreSQL ${{ matrix.postgres }} integration @@ -155,6 +155,7 @@ jobs: image: dolthub/dolt:2.2.3@sha256:1a651e738a2e7275b9c69fec8c5f42c6e23954314405c412f4dc9287ee7c0080 env: OPENSTATSPEC_DOLT_URL: mysql+pymysql://openstatspec:openstatspec@127.0.0.1:13308/openstatspec + OPENSTATSPEC_TEST_DOLT_ADMIN_URL: mysql+pymysql://openstatspec_test_admin:ci-admin@127.0.0.1:13308/ OPENSTATSPEC_EXPECTED_DOLT_VERSION: ${{ matrix.dolt }} OPENSTATSPEC_DOLT_IMAGE: ${{ matrix.image }} steps: @@ -186,6 +187,11 @@ jobs: --workdir /var/lib/dolt \ "$OPENSTATSPEC_DOLT_IMAGE" sql \ -q "CREATE USER 'openstatspec'@'%' IDENTIFIED BY 'openstatspec'; GRANT ALL PRIVILEGES ON openstatspec.* TO 'openstatspec'@'%'" + docker run --rm \ + --volume "$RUNNER_TEMP/openstatspec-dolt:/var/lib/dolt" \ + --workdir /var/lib/dolt \ + "$OPENSTATSPEC_DOLT_IMAGE" sql \ + -q "CREATE USER 'openstatspec_test_admin'@'%' IDENTIFIED BY 'ci-admin'; GRANT ALL PRIVILEGES ON *.* TO 'openstatspec_test_admin'@'%' WITH GRANT OPTION" docker run --detach --name openstatspec-dolt \ --publish 127.0.0.1:13308:3306 \ --volume "$RUNNER_TEMP/openstatspec-dolt:/var/lib/dolt" \ @@ -218,6 +224,7 @@ jobs: break PY - run: python -m pytest -m "services and not candidate_evidence" + - run: python -m pytest tests/test_read_only_export.py - name: Show Dolt logs on failure and stop server if: always() run: | diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 44c0436..33f212d 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -48,7 +48,7 @@ jobs: uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: repository: OpenStatSpec/specification - ref: cd8f198c68b849eb8ed018a894670a0904c2181d + ref: 864e84479f554b8ee250ffed44c4dfb963750d4a path: openstatspec-specification - name: Checkout specification package source uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 @@ -159,9 +159,9 @@ jobs: - run: python -m pip install -e ".[dev]" - run: python -m pytest -m "not services" - run: python -m venv /tmp/openstatspec-release-smoke - - run: /tmp/openstatspec-release-smoke/bin/python -m pip install ./openstatspec-pyspssio - run: /tmp/openstatspec-release-smoke/bin/python -m pip install dist/openstatspec/openstatspec-*.whl - - run: /tmp/openstatspec-release-smoke/bin/openstatspec capabilities + - run: env -u PYTHONPATH /tmp/openstatspec-release-smoke/bin/openstatspec capabilities + working-directory: /tmp - name: Smoke-test specification wheel run: | python -m venv /tmp/openstatspec-specification-release-smoke diff --git a/CHANGELOG.md b/CHANGELOG.md index ce9a3a2..bb7c919 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,30 @@ All notable changes to this reference implementation are documented here. _No unreleased changes._ +## 0.8.0 - 2026-09-07 + +### Changed + +- Pin released specification `v0.5.0` at exact commit + `864e84479f554b8ee250ffed44c4dfb963750d4a`, with normative status `released`. + Capabilities identify adapter version 0.8.0 and explicitly select SAV/ZSAV 1.0 + with `database_io_policy=openstatspec-database-io-v1`. +- Reads, inspection, validation, and SAV/ZSAV export no longer write database + operation/fidelity audit records, including on failure. Export diagnostics + and loss consent remain operation-scoped; export results omit `operation_id`. +- Default Dolt writes use the packaged tested exact-version policy for 2.2.2 + and 2.2.3 without user-supplied evidence files. Unknown/untested versions + remain blocked before mutation; explicit external overrides remain strict. + Safety budgets are not claimed as proven native server limits. +- Dolt reads no longer require write-version declarations or the write + variable-count ceiling. Missing SQLite files are rejected without creation. +- Exact Dolt 2.2.2/2.2.3 CI explicitly runs SELECT-only export success/failure + tests using a separate fixture-admin account; normal service permissions + and adapter permissions are unchanged. Wheel smoke installs resolve the + required engine from PyPI rather than a source checkout. + +No optional Transformation Workflow 0.3 implementation is claimed. + ## 0.7.1 - 2026-08-12 ### Changed diff --git a/README.md b/README.md index 58c9a0c..79b089b 100644 --- a/README.md +++ b/README.md @@ -4,6 +4,10 @@ The reference Python implementation of the OpenStatSpec specification. This package implements the specification; it does not define or extend it. The normative model lives in the `OpenStatSpec/specification` repository. +Python 0.8.0 pins released specification `v0.5.0` at +`864e84479f554b8ee250ffed44c4dfb963750d4a` and selects SAV/ZSAV 1.0 with +`database_io_policy=openstatspec-database-io-v1`. This does not claim +implementation of the optional Transformation Workflow 0.3 profile. ## Boundaries @@ -137,7 +141,8 @@ dictionary semantics cannot be reproduced, it stops until you pass the exact diagnostic code with `--allow-loss`. Diagnostics are returned to the caller, not persisted. Reads, validation, and SAV/ZSAV export never write to the database, including on failure. Export returns no `operation_id` and needs only read -permissions; Dolt reads have no write-version gate. Imports and transformations +permissions; Dolt reads have no write-version or write-variable-count gate. +Read operations reject missing SQLite files without creating them. Imports and transformations use the packaged exact-version policy by default. Active driver/identity checks, limit preflight, and the clean expected branch/HEAD guard for in-place apply remain mandatory. Dolt's reported limits are adapter safety budgets, narrowed diff --git a/docs/release-readiness.md b/docs/release-readiness.md index e68fd26..e79ff84 100644 --- a/docs/release-readiness.md +++ b/docs/release-readiness.md @@ -1,8 +1,15 @@ -# 0.7.1 release readiness +# 0.8.0 release readiness This page records the expected release contract, not a publication event. Creating a version tag remains a separate maintainer action. +The release selects SAV/ZSAV 1.0 with the optional Database I/O Execution +Policy `openstatspec-database-io-v1`. Reads and exports, including failures, +write no database audit records; export results omit `operation_id` and return +diagnostics to the caller. Reads must not create missing SQLite files or apply +Dolt write-variable-count ceilings. This release does not claim implementation +of the optional Transformation Workflow 0.3 profile. + ## Supported workflow For a supported unencrypted SAV or ZSAV source, the adapter imports one source @@ -104,7 +111,7 @@ implemented openstatspec.frontends.spss package. Stata and SAS remain empty source-tree placeholders and must expose no compiler, apply API, CLI choice, capability claim, or implied support. -## Default Dolt write verification +## Prior default Dolt write verification (before the 0.8.0 pin) Local release verification used the exact 2.2.2 and 2.2.3 image digests in `.github/workflows/ci.yml`, isolated ports 13482/13483, and disposable `/tmp` @@ -125,6 +132,30 @@ owned table additions being misclassified as unrelated diffs, and Dolt retaining transaction. The candidate storage/identifier/column smoke probe also passed on both pins; it does **not** establish full value/row/statement limit conformance. +## 0.8.0 local verification + +Using `../python/.venv/bin/python`, `PYTHONPATH=src:../pyspssio`, and +`OPENSTATSPEC_SPECIFICATION_DIR=/tmp/oss-python-spec-v050-5dSTna` (an archive +of exact `864e84479f554b8ee250ffed44c4dfb963750d4a`): + +- `python -m pytest -ra`: **381 passed, 49 skipped**, with + `OPENSTATSPEC_TEST_DOLT_ADMIN_URL=mysql+pymysql://root@127.0.0.1:13387/` + exercising live Dolt 2.3.0 through disposable SELECT-only users. +- Both immutable CI images, exact Dolt 2.2.2/2.2.3, bootstrapped with the + unchanged service user and separate global fixture admin on ports 13582/13583: + `python -m pytest -m 'services and not candidate_evidence'`: **7 passed, + 41 skipped, 382 deselected per pin**; explicit + `python -m pytest tests/test_read_only_export.py`: **33 passed per pin**. +- `python -m compileall -q src tests`, `git diff --check`, wheel/sdist build, + and `python -m twine check /tmp/oss-python-v080-dist/*` passed. +- A clean wheel install resolved the required engine from PyPI. Isolated + imports came only from the smoke environment's `site-packages`; installed + capabilities reported adapter 0.8.0, the selected policy, released v0.5.0 + provenance, and exactly 2.2.2/2.2.3 for default Dolt writes. + +Other database services were not configured locally. This is local evidence, +not a claim that the updated remote CI or release publication has completed. + ## Maintainer checks before tagging 1. Publish the pinned `openstatspec-pyspssio==0.5.1.post2` engine distribution @@ -139,13 +170,22 @@ both pins; it does **not** establish full value/row/statement limit conformance. catalog, and audit rollback checks, and failed-import cleanup. Release/CI owns this evidence, not per-user runtime declaration files. Candidate limit probes remain non-claiming and separate from these write gates. + Each Dolt job must also run `python -m pytest tests/test_read_only_export.py` + explicitly, outside the `services` marker filter. Bootstrap a separate + `openstatspec_test_admin` on the isolated CI server with global database/user + creation and grant privileges, and set `OPENSTATSPEC_TEST_DOLT_ADMIN_URL`. + The fixture creates its own database and SELECT-only reader and removes + both afterward; the normal service user remains unchanged. 4. Build with `python -m build` and install the generated wheel in a clean - environment. + environment, resolving `openstatspec-pyspssio==0.5.1.post2` from PyPI. + Run the installed CLI outside the source checkout with `PYTHONPATH` unset; + verify adapter version `0.8.0` and the selected database I/O policy. 5. Confirm `openstatspec capabilities` reflects the intended support boundary. 6. Confirm CI, release fixtures, and capabilities use the published OpenStatSpec - specification `v0.3.0` at exact commit - `cd8f198c68b849eb8ed018a894670a0904c2181d`, publish - `specification_status=stable`, and set `specification_release` to `v0.3.0`. + specification `v0.5.0` at exact commit + `864e84479f554b8ee250ffed44c4dfb963750d4a`, publish + `specification_status=released`, and set `specification_release` to `v0.5.0`. + Use an exact checkout via `OPENSTATSPEC_SPECIFICATION_DIR` for local tests. 7. Review this document, the README, and CHANGELOG for accurate scope. The tag-triggered release workflow repeats the non-service test suite, builds diff --git a/pyproject.toml b/pyproject.toml index 4558fd3..862a7dc 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "openstatspec" -version = "0.7.1" +version = "0.8.0" description = "Reference adapter for the OpenStatSpec relational contract" readme = "README.md" requires-python = ">=3.11" diff --git a/src/openstatspec/api.py b/src/openstatspec/api.py index 4e6405e..ff54612 100644 --- a/src/openstatspec/api.py +++ b/src/openstatspec/api.py @@ -34,6 +34,7 @@ ) from .transform import TransformationPlan from .sql.database_urls import require_existing_database_url +from .sql.dolt_conformance import ADAPTER_VERSION from .sql.capabilities import ( SPECIFICATION_COMMIT, SPECIFICATION_RELEASE, SPECIFICATION_STATUS, active_connection, catalog_binding, @@ -58,6 +59,7 @@ def capability_matrix( require_existing_database_url(database_url) declaration = { "specification": "OpenStatSpec", + "adapter_version": ADAPTER_VERSION, "database_io_policy": "openstatspec-database-io-v1", "specification_status": SPECIFICATION_STATUS, "specification_release": SPECIFICATION_RELEASE, diff --git a/src/openstatspec/sql/capabilities.py b/src/openstatspec/sql/capabilities.py index 57b1741..e6978d7 100644 --- a/src/openstatspec/sql/capabilities.py +++ b/src/openstatspec/sql/capabilities.py @@ -20,9 +20,9 @@ # Release/build automation binds the adapter to the exact language-neutral # specification revision used for normative fixtures and profile semantics. -SPECIFICATION_COMMIT = "cd8f198c68b849eb8ed018a894670a0904c2181d" -SPECIFICATION_STATUS = "stable" -SPECIFICATION_RELEASE: str | None = "v0.3.0" +SPECIFICATION_COMMIT = "864e84479f554b8ee250ffed44c4dfb963750d4a" +SPECIFICATION_STATUS = "released" +SPECIFICATION_RELEASE: str | None = "v0.5.0" DOLT_WRITE_CONFORMANCE = { "declaration_schema_id": "openstatspec-dolt-adapter-declaration-v1", diff --git a/tests/test_cli.py b/tests/test_cli.py index 720df61..3c0fdeb 100755 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -42,9 +42,11 @@ def test_cli_import_inspect_validate_and_export_emit_json(tmp_path, capsys) -> N def test_capability_matrix_is_public_and_cli_matches_engine_boundary(capsys) -> None: matrix = openstatspec.capability_matrix() - assert matrix["specification_status"] == "stable" - assert matrix["specification_release"] == "v0.3.0" - assert matrix["specification_commit"] == "cd8f198c68b849eb8ed018a894670a0904c2181d" + assert matrix["adapter_version"] == "0.8.0" + assert matrix["database_io_policy"] == "openstatspec-database-io-v1" + assert matrix["specification_status"] == "released" + assert matrix["specification_release"] == "v0.5.0" + assert matrix["specification_commit"] == "864e84479f554b8ee250ffed44c4dfb963750d4a" assert matrix["directions"] == ["import", "export", "semantic_round_trip"] assert matrix["active_connection"] is None assert matrix["engine"]["package"] == "openstatspec-pyspssio" @@ -82,6 +84,7 @@ def test_capability_matrix_is_public_and_cli_matches_engine_boundary(capsys) -> def test_capability_matrix_reports_active_sqlite_limits(tmp_path) -> None: database_url = f"sqlite:///{tmp_path / 'capabilities.sqlite'}" + sqlite3.connect(tmp_path / "capabilities.sqlite").close() matrix = openstatspec.capability_matrix(database_url=database_url) active = matrix["active_connection"] assert active["profile"] == "sqlite" diff --git a/tests/test_sql_profiles.py b/tests/test_sql_profiles.py index 8e07253..3b1ca73 100755 --- a/tests/test_sql_profiles.py +++ b/tests/test_sql_profiles.py @@ -29,13 +29,13 @@ def test_profile_detection_tracks_supported_dialect_urls() -> None: assert profile_for_url("mariadb+mariadbconnector://user@host/database") is MYSQL assert profile_for_url("mysql+pymysql://user@host/dolt_database") is MYSQL -def test_profile_declarations_publish_stable_specification_provenance() -> None: +def test_profile_declarations_publish_released_specification_provenance() -> None: for declaration in capabilities.profile_declarations().values(): - assert declaration["specification_status"] == "stable" - assert declaration["specification_release"] == "v0.3.0" + assert declaration["specification_status"] == "released" + assert declaration["specification_release"] == "v0.5.0" assert ( declaration["specification_commit"] - == "cd8f198c68b849eb8ed018a894670a0904c2181d" + == "864e84479f554b8ee250ffed44c4dfb963750d4a" ) def test_profile_preflight_fails_without_transforming_a_wide_dataset() -> None: