From d919c97c3160fe88167ceca567f2c7caeac4a0d0 Mon Sep 17 00:00:00 2001 From: Nico Ritschel Date: Fri, 2 Oct 2026 06:07:40 -0700 Subject: [PATCH 1/3] Support current Ossie source profiles and lossless interchange --- pyproject.toml | 2 +- sidemantic/adapters/ossie.py | 6 + sidemantic/cli.py | 9 +- sidemantic/formats.py | 9 +- sidemantic/interchange/ossie/__init__.py | 10 + sidemantic/interchange/ossie/documents.py | 49 +- .../ossie/expression_validation.py | 6 +- sidemantic/interchange/ossie/identifier.py | 15 + sidemantic/interchange/ossie/lowering.py | 32 +- sidemantic/interchange/ossie/parser.py | 127 +++-- sidemantic/interchange/ossie/profiles.py | 45 +- .../logical/0.2.0.dev0-b6c702e/schema.json | 374 +++++++++++++++ .../interchange/ossie/schemas/manifest.json | 54 +++ .../ontology/0.2.0.dev0-b6c702e/schema.json | 341 +++++++++++++ .../ontology/0.2.0.dev0-b6c702e/upstream.json | 314 ++++++++++++ .../interchange/ossie/semantic_validation.py | 293 +++++++++++- sidemantic/interchange/ossie/serialization.py | 32 +- sidemantic/interchange/ossie/synthesis.py | 81 +++- sidemantic/interchange/ossie/validation.py | 62 ++- sidemantic/loaders.py | 95 ++-- tests/adapters/osi/test_ossie_adapter.py | 6 +- .../interchange/ossie/test_current_schema.py | 167 +++++++ .../ossie/test_ontology_semantics.py | 230 +++++++++ tests/interchange/ossie/test_ossie_parser.py | 49 +- .../ossie/test_schema_validation.py | 11 + .../ossie/test_semantic_validation.py | 16 +- tests/interchange/ossie/test_serialization.py | 28 ++ tests/interchange/ossie/test_synthesis.py | 102 +++- .../ossie/test_temporal_conformance.py | 3 +- .../ossie/test_upstream_validator_gate.py | 69 +++ tests/ossie-fixtures/upstream/README.md | 6 + .../b6c702e/core-spec/ossie-schema.json | 374 +++++++++++++++ .../upstream/b6c702e/ontology/ontology.json | 314 ++++++++++++ .../upstream/b6c702e/validation/validate.py | 452 ++++++++++++++++++ tests/test_cli_contract.py | 69 ++- tests/test_formats.py | 66 +++ 36 files changed, 3748 insertions(+), 170 deletions(-) create mode 100644 sidemantic/interchange/ossie/schemas/logical/0.2.0.dev0-b6c702e/schema.json create mode 100644 sidemantic/interchange/ossie/schemas/ontology/0.2.0.dev0-b6c702e/schema.json create mode 100644 sidemantic/interchange/ossie/schemas/ontology/0.2.0.dev0-b6c702e/upstream.json create mode 100644 tests/interchange/ossie/test_current_schema.py create mode 100644 tests/interchange/ossie/test_ontology_semantics.py create mode 100644 tests/ossie-fixtures/upstream/b6c702e/core-spec/ossie-schema.json create mode 100644 tests/ossie-fixtures/upstream/b6c702e/ontology/ontology.json create mode 100644 tests/ossie-fixtures/upstream/b6c702e/validation/validate.py diff --git a/pyproject.toml b/pyproject.toml index 375df5c29..528de6ca4 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -199,7 +199,7 @@ force-exclude = true extend-exclude = [ "sidemantic/adapters/malloy_grammar", # ANTLR-generated files "sidemantic/adapters/holistics_grammar", # ANTLR-generated files - "tests/ossie-fixtures/upstream/validation/validate.py", # Exact pinned Apache validator fixture + "tests/ossie-fixtures/upstream/**/validation/validate.py", # Exact pinned Apache validator fixtures ] [tool.ruff.lint] diff --git a/sidemantic/adapters/ossie.py b/sidemantic/adapters/ossie.py index 0252f3a75..e7eabe17d 100644 --- a/sidemantic/adapters/ossie.py +++ b/sidemantic/adapters/ossie.py @@ -32,6 +32,7 @@ sort_diagnostics, synthesize_ossie_document, ) +from sidemantic.interchange.ossie.profiles import CURRENT_OSSIE_SCHEMA_COMMIT _GENERATED_DIRECTORIES = frozenset({"dbt_packages", "target"}) @@ -75,6 +76,7 @@ def __init__( export_scope_name: str | None = None, expression_dialect: str | None = None, schema_version: str = "0.2.0.dev0", + schema_revision: str | None = None, serialization: OssieSerialization | str | None = None, ) -> None: if not target_dialect.strip(): @@ -84,8 +86,10 @@ def __init__( self._export_scope_name = export_scope_name self._expression_dialect = expression_dialect self._schema_version = schema_version + self._schema_revision = schema_revision self._serialization = OssieSerialization(serialization) if serialization is not None else None self._parse_options = OssieParseOptions( + schema_revision=schema_revision, consumer_profile=consumer_profile, import_policy=import_policy, source_dialect=source_dialect, @@ -212,6 +216,7 @@ def export( scope_name: str | None = None, expression_dialect: str | None = None, schema_version: str | None = None, + schema_revision: str | None = None, serialization: OssieSerialization | str | None = None, portable_only: bool = False, ) -> None: @@ -234,6 +239,7 @@ def export( scope_name=selected_scope, expression_dialect=selected_dialect, schema_version=schema_version or self._schema_version, + schema_revision=schema_revision or self._schema_revision or CURRENT_OSSIE_SCHEMA_COMMIT, serialization=output_serialization, consumer_profile=self._parse_options.consumer_profile, portable_only=portable_only, diff --git a/sidemantic/cli.py b/sidemantic/cli.py index bef5fd5d1..6de8af3d1 100644 --- a/sidemantic/cli.py +++ b/sidemantic/cli.py @@ -1540,6 +1540,11 @@ def convert( "--ossie-schema-version", help="Explicit pinned Ossie schema version for synthesized output", ), + ossie_schema_revision: str = typer.Option( + None, + "--ossie-schema-revision", + help="Pinned Ossie draft revision for synthesized output (commit SHA)", + ), ossie_consumer_profile: str = typer.Option( None, "--ossie-consumer-profile", @@ -1603,7 +1608,7 @@ def convert( from sidemantic.fidelity import capture_import_report source_adapter_options = None - if source_format != "auto" and get_semantic_format(source_format, operation="import").name == "ossie": + if source_format == "auto" or get_semantic_format(source_format, operation="import").name == "ossie": source_adapter_options = {} if ossie_scope is not None: source_adapter_options["scope_id"] = ossie_scope @@ -1632,6 +1637,8 @@ def convert( target_adapter_options = {"consumer_profile": ossie_consumer_profile} if ossie_schema_version is not None: target_export_options["schema_version"] = ossie_schema_version + if ossie_schema_revision is not None: + target_export_options["schema_revision"] = ossie_schema_revision with progress(f"Converting semantic definitions to {target_format}"): with capture_import_report() as fidelity_report: diff --git a/sidemantic/formats.py b/sidemantic/formats.py index 5915e81b2..5d1cecf4d 100644 --- a/sidemantic/formats.py +++ b/sidemantic/formats.py @@ -261,23 +261,22 @@ def load_semantic_source( File inputs are always exact: auto-discovery parses only the named file and never scans its siblings. Directory inputs retain the existing project-wide - discovery behavior. + discovery behavior. In auto mode, adapter options apply only to detected + Ossie sources; other format options require an explicit source format. """ source_path = Path(source) if not source_path.exists(): raise FileNotFoundError(f"Semantic source does not exist: {source_path}") if source_format.strip().lower() == "auto": - if adapter_options: - raise ValueError("adapter_options require an explicit source_format") from sidemantic.core.semantic_layer import SemanticLayer from sidemantic.loaders import load_from_directory, load_from_file layer = SemanticLayer() if source_path.is_file(): - load_from_file(layer, source_path) + load_from_file(layer, source_path, ossie_adapter_options=adapter_options) else: - load_from_directory(layer, source_path) + load_from_directory(layer, source_path, ossie_adapter_options=adapter_options) return layer.graph spec = get_semantic_format(source_format, operation="import") diff --git a/sidemantic/interchange/ossie/__init__.py b/sidemantic/interchange/ossie/__init__.py index 84dbddad8..205b3c81e 100644 --- a/sidemantic/interchange/ossie/__init__.py +++ b/sidemantic/interchange/ossie/__init__.py @@ -18,6 +18,8 @@ OssieOntologyDocument, UnsupportedOssieDocument, freeze_json, + is_logical_document_data, + logical_model_entries, thaw_json, ) from sidemantic.interchange.ossie.lowering import OssieLoweringResult, lower_ossie_document @@ -27,9 +29,12 @@ parse_ossie_document, ) from sidemantic.interchange.ossie.profiles import ( + CURRENT_OSSIE_SCHEMA_COMMIT, DBT_1_12_0_1_0_ALIAS, DBT_1_12_0_1_1, + LEGACY_OSSIE_SCHEMA_COMMIT, OSSIE_CORE_0_1_1, + OSSIE_CORE_0_2_0_CURRENT, OSSIE_CORE_0_2_0_DEV0, OSSIE_PROFILES, OssieConsumerProfile, @@ -59,10 +64,13 @@ ) __all__ = [ + "CURRENT_OSSIE_SCHEMA_COMMIT", + "LEGACY_OSSIE_SCHEMA_COMMIT", "DBT_1_12_0_1_0_ALIAS", "DBT_1_12_0_1_1", "OSSIE_CORE_0_1_1", "OSSIE_CORE_0_2_0_DEV0", + "OSSIE_CORE_0_2_0_CURRENT", "OSSIE_PROFILES", "FrozenJSONValue", "FrozenJSONObject", @@ -94,6 +102,8 @@ "UnsupportedOssieDocument", "diagnostic_sort_key", "freeze_json", + "is_logical_document_data", + "logical_model_entries", "lower_ossie_document", "parse_ossie_document", "resolve_ossie_profile", diff --git a/sidemantic/interchange/ossie/documents.py b/sidemantic/interchange/ossie/documents.py index abf6a227c..e7157651a 100644 --- a/sidemantic/interchange/ossie/documents.py +++ b/sidemantic/interchange/ossie/documents.py @@ -19,6 +19,29 @@ ParsedJSONValue: TypeAlias = JSONScalar | Mapping[str, object] | list[object] | tuple[object, ...] +def is_logical_document_data(data: Mapping[str, object]) -> bool: + """Recognize current flat models and the earlier model-array envelope.""" + + return ( + "semantic_model" in data + or "datasets" in data + or ("name" in data and "ontology" not in data and "ontology_mappings" not in data) + ) + + +def logical_model_entries(data: Mapping[str, object]) -> tuple[tuple[str, object], ...]: + """Expose model values and their source pointers without rewriting source data.""" + + if "semantic_model" in data: + models = data["semantic_model"] + if isinstance(models, (list, tuple)): + return tuple((f"/semantic_model/{index}", model) for index, model in enumerate(models)) + return () + if is_logical_document_data(data): + return (("", data),) + return () + + @dataclass(frozen=True, slots=True) class FrozenJSONObject(Mapping[str, "FrozenJSONValue"]): """An insertion-ordered, deeply immutable JSON object.""" @@ -129,6 +152,7 @@ class _OssieDocumentBase: canonical_data: ParsedJSONValue | FrozenJSONObject serialization: OssieSerialization source: OssieDocumentSource | None = None + schema_revision: str | None = None _known_root_fields: ClassVar[frozenset[str]] = frozenset() @@ -170,7 +194,21 @@ def to_parsed_data(self) -> object: class OssieLogicalDocument(_OssieDocumentBase): """A logical-layer Ossie document, before validation or runtime lowering.""" - _known_root_fields: ClassVar[frozenset[str]] = frozenset({"version", "dialects", "vendors", "semantic_model"}) + _known_root_fields: ClassVar[frozenset[str]] = frozenset( + { + "version", + "dialects", + "vendors", + "semantic_model", + "name", + "description", + "ai_context", + "datasets", + "relationships", + "metrics", + "custom_extensions", + } + ) def __post_init__(self) -> None: super(OssieLogicalDocument, self).__post_init__() @@ -183,8 +221,11 @@ def semantic_model_value(self) -> FrozenJSONValue | None: @property def semantic_models(self) -> tuple[FrozenJSONValue, ...]: - value = self.semantic_model_value - return value if isinstance(value, tuple) else () + return tuple(model for _, model in self.semantic_model_entries) + + @property + def semantic_model_entries(self) -> tuple[tuple[str, FrozenJSONValue], ...]: + return logical_model_entries(self.canonical_data) @dataclass(frozen=True, slots=True, kw_only=True) @@ -192,7 +233,7 @@ class OssieOntologyDocument(_OssieDocumentBase): """An ontology-layer Ossie document, preserved without reasoning semantics.""" _known_root_fields: ClassVar[frozenset[str]] = frozenset( - {"version", "name", "description", "ai_context", "ontology", "ontology_mappings"} + {"version", "name", "description", "ai_context", "requires", "ontology", "ontology_mappings", "prefixes"} ) def __post_init__(self) -> None: diff --git a/sidemantic/interchange/ossie/expression_validation.py b/sidemantic/interchange/ossie/expression_validation.py index 3296e3e6d..98016f191 100644 --- a/sidemantic/interchange/ossie/expression_validation.py +++ b/sidemantic/interchange/ossie/expression_validation.py @@ -33,7 +33,7 @@ ) -def scalar_sql_expression_error(expression: str, *, sqlglot_dialect: str | None) -> str | None: +def scalar_sql_expression_error(expression: str, *, sqlglot_dialect: str | None, row_level: bool = False) -> str | None: """Return why SQL is not one scalar Ossie expression, otherwise ``None``. Parsing uses the selected executable dialect. The structural gate rejects @@ -49,7 +49,11 @@ def scalar_sql_expression_error(expression: str, *, sqlglot_dialect: str | None) return "expression must contain exactly one SQL expression" root = parsed[0] + if isinstance(root, (exp.Alias, exp.Aliases, exp.Star)): + return "expression must be a scalar value, without a projection alias or wildcard" for node in root.walk(): if isinstance(node, _FORBIDDEN_NODE_TYPES): return f"Ossie expressions cannot contain {type(node).__name__}" + if row_level and isinstance(node, exp.AggFunc) and node.find_ancestor(exp.Window) is None: + return "Ossie dataset fields are row-level expressions and cannot contain aggregates" return None diff --git a/sidemantic/interchange/ossie/identifier.py b/sidemantic/interchange/ossie/identifier.py index 22aa738e6..04ac9ebb9 100644 --- a/sidemantic/interchange/ossie/identifier.py +++ b/sidemantic/interchange/ossie/identifier.py @@ -5,6 +5,21 @@ OSSIE_IDENTIFIER_MAX_LENGTH = 128 +def identifier_syntax_valid(identifier: str) -> bool: + """Accept regular names or nonempty ANSI delimited names with escaped quotes.""" + + if identifier.startswith('"'): + if len(identifier) < 3 or not identifier.endswith('"'): + return False + body = identifier[1:-1] + return "\x00" not in body and '"' not in body.replace('""', "") + return ( + bool(identifier) + and (identifier[0].isalpha() or identifier[0] == "_") + and all(character.isalnum() or character == "_" for character in identifier) + ) + + def is_quoted_identifier(identifier: str) -> bool: """Return whether *identifier* uses Ossie's ANSI double-quote form.""" diff --git a/sidemantic/interchange/ossie/lowering.py b/sidemantic/interchange/ossie/lowering.py index 88681274f..078a68140 100644 --- a/sidemantic/interchange/ossie/lowering.py +++ b/sidemantic/interchange/ossie/lowering.py @@ -24,7 +24,7 @@ OssieSourceLocation, sort_diagnostics, ) -from sidemantic.interchange.ossie.documents import OssieLogicalDocument, OssieOntologyDocument +from sidemantic.interchange.ossie.documents import OssieLogicalDocument, OssieOntologyDocument, logical_model_entries from sidemantic.interchange.ossie.expression_validation import scalar_sql_expression_error from sidemantic.interchange.ossie.identifier import identifier_within_limit, normalize_identifier from sidemantic.interchange.ossie.parser import OssieParseResult @@ -178,10 +178,15 @@ def _sql_expression_error(expression: str, target_dialect: str) -> str | None: def _classify_source(source: str, source_dialect: str | None) -> tuple[str, str] | None: dialect = _SQLGLOT_DIALECTS.get(_normalize_dialect(source_dialect)) if source_dialect else None try: - parsed = sqlglot.parse_one(source, read=dialect) + statements = sqlglot.parse(source, read=dialect) + if len(statements) != 1: + return None + parsed = statements[0] except sqlglot.errors.ParseError: parsed = None if isinstance(parsed, (exp.Query, exp.Subquery)): + if any(not select.expressions for select in parsed.find_all(exp.Select)): + return None return "query", source try: @@ -256,6 +261,7 @@ def _lower_scope( *, scope_id: str, scope_index: int, + scope_pointer: str, document_id: str, target_dialect: str, diagnostics: list[OssieDiagnostic], @@ -272,7 +278,7 @@ def _lower_scope( registration_token = set_current_layer(None) try: for dataset_index, dataset, dataset_name in _unique_named_items(dataset_values): - pointer = f"/semantic_model/{scope_index}/datasets/{dataset_index}" + pointer = f"{scope_pointer}/datasets/{dataset_index}" source = dataset.get("source") if not isinstance(source, str) or not source.strip(): diagnostics.append( @@ -400,7 +406,7 @@ def _lower_scope( relationships = _array(semantic_model.get("relationships")) if runtime_override is None else () for relationship_index, relationship, edge_id in _unique_named_items(relationships): - pointer = f"/semantic_model/{scope_index}/relationships/{relationship_index}" + pointer = f"{scope_pointer}/relationships/{relationship_index}" from_name = _name(relationship.get("from")) to_name = _name(relationship.get("to")) from_columns = _array(relationship.get("from_columns")) @@ -463,7 +469,7 @@ def _lower_scope( metrics = _array(semantic_model.get("metrics")) if runtime_override is None else () for metric_index, metric, metric_name in _unique_named_items(metrics): - pointer = f"/semantic_model/{scope_index}/metrics/{metric_index}" + pointer = f"{scope_pointer}/metrics/{metric_index}" selected = _expression_for_target(metric.get("expression"), target_dialect) if selected is None: diagnostics.append( @@ -618,19 +624,16 @@ def lower_ossie_document( lowering_diagnostics=tuple(diagnostics), ) - parsed = document.to_parsed_data() - root = _mapping(parsed) - semantic_models = _array(root.get("semantic_model")) if root else None named_models = [ - (index, model, model_name) - for index, value in enumerate(semantic_models or ()) + (index, pointer, model, model_name) + for index, (pointer, value) in enumerate(logical_model_entries(document.to_parsed_data())) if (model := _mapping(value)) is not None if (model_name := _name(model.get("name"))) is not None ] - name_counts = Counter(normalize_identifier(name) for _, _, name in named_models if identifier_within_limit(name)) + name_counts = Counter(normalize_identifier(name) for _, _, _, name in named_models if identifier_within_limit(name)) document_id = _document_id(parse_result) scopes = [] - for index, semantic_model, name in named_models: + for index, scope_pointer, semantic_model, name in named_models: if not identifier_within_limit(name): continue scope_id = name if name_counts[normalize_identifier(name)] == 1 else f"{name}@{index}" @@ -644,7 +647,7 @@ def lower_ossie_document( parse_result, code="ossie.lowering.runtime_extension_invalid", message=f"Sidemantic runtime extension cannot be restored safely: {exc}", - pointer=f"/semantic_model/{index}/custom_extensions", + pointer=f"{scope_pointer}/custom_extensions", scope=scope_id, ) ) @@ -657,7 +660,7 @@ def lower_ossie_document( severity=OssieDiagnosticSeverity.WARNING, code="ossie.lowering.runtime_extension_restored", message="Restored native Sidemantic runtime semantics; other consumers require Sidemantic extension support.", - json_pointer=f"/semantic_model/{index}/custom_extensions", + json_pointer=f"{scope_pointer}/custom_extensions", scope=scope_id, source=_source_location(parse_result), profile=parse_result.profile, @@ -669,6 +672,7 @@ def lower_ossie_document( semantic_model, scope_id=scope_id, scope_index=index, + scope_pointer=scope_pointer, document_id=document_id, target_dialect=selected_target, diagnostics=diagnostics, diff --git a/sidemantic/interchange/ossie/parser.py b/sidemantic/interchange/ossie/parser.py index b93416c01..d4f35090a 100644 --- a/sidemantic/interchange/ossie/parser.py +++ b/sidemantic/interchange/ossie/parser.py @@ -4,7 +4,7 @@ import json from collections.abc import Mapping -from dataclasses import dataclass +from dataclasses import dataclass, replace from pathlib import PurePosixPath from urllib.parse import urlsplit @@ -22,6 +22,7 @@ OssieLogicalDocument, OssieOntologyDocument, UnsupportedOssieDocument, + is_logical_document_data, ) from sidemantic.interchange.ossie.profiles import ( OssieConsumerProfile, @@ -32,10 +33,15 @@ OssieProfileError, OssieSerialization, ) -from sidemantic.interchange.ossie.validation import SchemaValidationResult, validate_ossie_schema +from sidemantic.interchange.ossie.validation import ( + SchemaValidationResult, + detect_schema_profile, + validate_ossie_schema, +) _MAX_SOURCE_BYTES = 16 * 1024 * 1024 _MAX_NESTING_DEPTH = 256 +_MAX_EXPANDED_NODES = 100_000 try: # pragma: no cover - depends on how PyYAML was built from yaml import CSafeLoader as _SafeLoader @@ -54,6 +60,7 @@ class OssieParseOptions: target_dialect: str | None = None preservation_policy: OssiePreservationPolicy = OssiePreservationPolicy.CANONICAL_DATA validate_schema: bool = False + schema_revision: str | None = None def __post_init__(self) -> None: try: @@ -71,6 +78,8 @@ def __post_init__(self) -> None: raise OssieProfileError(f"{field_name} must be a non-empty string when provided") if not isinstance(self.validate_schema, bool): raise OssieProfileError("validate_schema must be a boolean") + if self.schema_revision is not None and not isinstance(self.schema_revision, str): + raise OssieProfileError("schema_revision must be a pinned commit string") @dataclass(frozen=True, slots=True) @@ -121,28 +130,52 @@ class _NonFiniteJSONNumberError(ValueError): pass +class _ParserLimitError(ValueError): + pass + + class _UniqueKeySafeLoader(_SafeLoader): - """Safe YAML loader that rejects duplicate mapping keys at every depth.""" - - def construct_mapping(self, node: yaml.MappingNode, deep: bool = False) -> dict[object, object]: - self.flatten_mapping(node) - mapping: dict[object, object] = {} - for key_node, value_node in node.value: - key = self.construct_object(key_node, deep=deep) - try: - duplicate = key in mapping - except TypeError as exc: - mark = key_node.start_mark - raise _DuplicateKeyError( - "", - line=mark.line + 1, - column=mark.column + 1, - ) from exc - if duplicate: - mark = key_node.start_mark - raise _DuplicateKeyError(key, line=mark.line + 1, column=mark.column + 1) - mapping[key] = self.construct_object(value_node, deep=deep) - return mapping + """Reject duplicate authored keys while retaining YAML merge overrides.""" + + def construct_document(self, node: yaml.Node) -> object: + # Bound alias expansion before SafeConstructor flattens merge mappings. + stack = [(node, 0)] + count = 0 + while stack: + child, depth = stack.pop() + count += 1 + if count > _MAX_EXPANDED_NODES or depth > _MAX_NESTING_DEPTH: + raise _ParserLimitError("YAML exceeds the expanded-node or nesting parser limit") + if isinstance(child, yaml.MappingNode): + stack.extend((value, depth + 1) for pair in child.value for value in pair) + elif isinstance(child, yaml.SequenceNode): + stack.extend((value, depth + 1) for value in child.value) + self._check_unique_keys(node, set()) + return super().construct_document(node) + + def _check_unique_keys(self, node: yaml.Node, visited: set[int]) -> None: + if id(node) in visited: + return + visited.add(id(node)) + if isinstance(node, yaml.MappingNode): + seen: set[object] = set() + merge_key = object() + for key_node, value_node in node.value: + if key_node.tag == "tag:yaml.org,2002:merge": + key = merge_key + elif isinstance(key_node, yaml.ScalarNode): + key = self.construct_object(key_node, deep=True) + else: + mark = key_node.start_mark + raise _DuplicateKeyError("", line=mark.line + 1, column=mark.column + 1) + if key in seen: + mark = key_node.start_mark + raise _DuplicateKeyError(key, line=mark.line + 1, column=mark.column + 1) + seen.add(key) + self._check_unique_keys(value_node, visited) + elif isinstance(node, yaml.SequenceNode): + for child in node.value: + self._check_unique_keys(child, visited) def _unique_json_object(pairs: list[tuple[str, object]]) -> dict[str, object]: @@ -210,7 +243,7 @@ def _infer_serialization(source_bytes: bytes, identifier: str) -> OssieSerializa return OssieSerialization.JSON try: json.loads(text) - except (json.JSONDecodeError, UnicodeDecodeError): + except (ValueError, RecursionError): return OssieSerialization.YAML return OssieSerialization.JSON @@ -243,7 +276,7 @@ def _classify_document( "ossie.document.root_type", ) - has_logical_root = "semantic_model" in parsed + has_logical_root = is_logical_document_data(parsed) has_ontology_root = "ontology" in parsed or "ontology_mappings" in parsed if has_logical_root and has_ontology_root: reason = "The document mixes logical semantic_model and ontology root families" @@ -267,7 +300,7 @@ def _classify_document( None, ) - reason = "The document contains none of semantic_model, ontology, or ontology_mappings" + reason = "The document contains neither a logical semantic model nor ontology data" return ( UnsupportedOssieDocument( canonical_data=parsed, @@ -305,6 +338,7 @@ def _resolve_options( return None, diagnostic try: + detected = detect_schema_profile(document.to_parsed_data()) options = OssieOptions( schema_version=version, serialization=serialization, @@ -313,6 +347,13 @@ def _resolve_options( source_dialect=parse_options.source_dialect, target_dialect=parse_options.target_dialect, preservation_policy=parse_options.preservation_policy, + schema_revision=( + parse_options.schema_revision + if parse_options.schema_revision is not None + else detected.source_commit + if detected and detected.schema_revision + else None + ), ) except OssieProfileError as exc: return ( @@ -469,27 +510,44 @@ def parse_ossie_document( ) ) return OssieParseResult(document, parse_options, None, None, tuple(diagnostics)) - except RecursionError: + except (RecursionError, _ParserLimitError): document = UnsupportedOssieDocument( canonical_data=None, serialization=serialization, source=source, - reason="The source exceeds the parser nesting budget", + reason="The source exceeds the parser expansion or nesting budget", ) diagnostics.append( _diagnostic( code="ossie.parse.limit", - message=f"Apache Ossie source exceeds the {_MAX_NESTING_DEPTH}-level nesting limit", + message=( + f"Apache Ossie source exceeds the {_MAX_NESTING_DEPTH}-level nesting " + f"or {_MAX_EXPANDED_NODES}-node expansion limit" + ), identifier=source_identifier, ) ) return OssieParseResult(document, parse_options, None, None, tuple(diagnostics)) + except ValueError as exc: + document = UnsupportedOssieDocument( + canonical_data=None, + serialization=serialization, + source=source, + reason="Parsed scalar exceeds the supported data model", + ) + diagnostics.append( + _diagnostic(code="ossie.parse.non_json_value", message=str(exc), identifier=source_identifier) + ) + return OssieParseResult(document, parse_options, None, None, tuple(diagnostics)) + stack = [(parsed, 0)] nesting_exceeded = False + expanded_nodes = 0 while stack: value, depth = stack.pop() - if depth > _MAX_NESTING_DEPTH: + expanded_nodes += 1 + if depth > _MAX_NESTING_DEPTH or expanded_nodes > _MAX_EXPANDED_NODES: nesting_exceeded = True break if isinstance(value, dict): @@ -501,12 +559,15 @@ def parse_ossie_document( canonical_data=None, serialization=serialization, source=source, - reason="The source exceeds the parser nesting budget", + reason="The source exceeds the parser expansion or nesting budget", ) diagnostics.append( _diagnostic( code="ossie.parse.limit", - message=f"Apache Ossie source exceeds the {_MAX_NESTING_DEPTH}-level nesting limit", + message=( + f"Apache Ossie source exceeds the {_MAX_NESTING_DEPTH}-level nesting " + f"or {_MAX_EXPANDED_NODES}-node expansion limit" + ), identifier=source_identifier, ) ) @@ -551,6 +612,8 @@ def parse_ossie_document( ) if profile_diagnostic is not None: diagnostics.append(profile_diagnostic) + if resolved_options is not None and resolved_options.profile.schema_revision: + document = replace(document, schema_revision=resolved_options.profile.schema_revision) schema_validation = None if parse_options.validate_schema: diff --git a/sidemantic/interchange/ossie/profiles.py b/sidemantic/interchange/ossie/profiles.py index 2da048055..9b86a5be7 100644 --- a/sidemantic/interchange/ossie/profiles.py +++ b/sidemantic/interchange/ossie/profiles.py @@ -5,6 +5,9 @@ from dataclasses import dataclass from enum import Enum +CURRENT_OSSIE_SCHEMA_COMMIT = "b6c702ed1c07e91382a69e870c875cbd19570828" +LEGACY_OSSIE_SCHEMA_COMMIT = "831f48e582731cf1ee2e65380ca5abf8157869c7" + class OssieProfileError(ValueError): """Raised when an Ossie option combination does not identify a supported profile.""" @@ -46,6 +49,7 @@ class OssieProfile: consumer_profile: OssieConsumerProfile upstream_schema_version: str | None compatibility_alias_for: str | None = None + schema_revision: str | None = None def __post_init__(self) -> None: object.__setattr__(self, "consumer_profile", OssieConsumerProfile(self.consumer_profile)) @@ -58,7 +62,8 @@ def __post_init__(self) -> None: @property def identifier(self) -> str: - return f"{self.consumer_profile.value}:{self.schema_version}" + revision = f"@{self.schema_revision}" if self.schema_revision else "" + return f"{self.consumer_profile.value}:{self.schema_version}{revision}" @property def is_compatibility_alias(self) -> bool: @@ -85,6 +90,12 @@ def validation_schema_version(self) -> str: consumer_profile=OssieConsumerProfile.OSSIE_CORE, upstream_schema_version="0.2.0.dev0", ) +OSSIE_CORE_0_2_0_CURRENT = OssieProfile( + schema_version="0.2.0.dev0", + consumer_profile=OssieConsumerProfile.OSSIE_CORE, + upstream_schema_version="0.2.0.dev0", + schema_revision=CURRENT_OSSIE_SCHEMA_COMMIT, +) DBT_1_12_0_1_0_ALIAS = OssieProfile( schema_version="0.1.0", consumer_profile=OssieConsumerProfile.DBT_1_12, @@ -102,11 +113,20 @@ def validation_schema_version(self) -> str: OSSIE_CORE_0_2_0_DEV0, DBT_1_12_0_1_0_ALIAS, DBT_1_12_0_1_1, + OSSIE_CORE_0_2_0_CURRENT, ) -_PROFILE_INDEX = {(profile.schema_version, profile.consumer_profile): profile for profile in OSSIE_PROFILES} - - -def resolve_ossie_profile(schema_version: str, consumer_profile: OssieConsumerProfile | str) -> OssieProfile: +_PROFILE_INDEX = { + (profile.schema_version, profile.consumer_profile): profile + for profile in OSSIE_PROFILES + if profile.schema_revision is None +} + + +def resolve_ossie_profile( + schema_version: str, + consumer_profile: OssieConsumerProfile | str, + schema_revision: str | None = None, +) -> OssieProfile: """Resolve a supported profile without treating serialization as a version selector.""" try: @@ -114,6 +134,16 @@ def resolve_ossie_profile(schema_version: str, consumer_profile: OssieConsumerPr except ValueError as exc: raise OssieProfileError(f"Unsupported Ossie consumer profile: {consumer_profile!r}") from exc + if schema_revision is not None: + if not isinstance(schema_revision, str): + raise OssieProfileError("schema_revision must be a pinned commit string") + if schema_version != "0.2.0.dev0" or normalized_consumer is not OssieConsumerProfile.OSSIE_CORE: + raise OssieProfileError("A schema revision is supported only for ossie-core 0.2.0.dev0") + if schema_revision in {CURRENT_OSSIE_SCHEMA_COMMIT, CURRENT_OSSIE_SCHEMA_COMMIT[:7]}: + return OSSIE_CORE_0_2_0_CURRENT + if schema_revision not in {LEGACY_OSSIE_SCHEMA_COMMIT, LEGACY_OSSIE_SCHEMA_COMMIT[:7]}: + raise OssieProfileError(f"Unsupported Ossie schema revision: {schema_revision!r}") + profile = _PROFILE_INDEX.get((schema_version, normalized_consumer)) if profile is not None: return profile @@ -142,6 +172,7 @@ class OssieOptions: source_dialect: str | None = None target_dialect: str | None = None preservation_policy: OssiePreservationPolicy = OssiePreservationPolicy.CANONICAL_DATA + schema_revision: str | None = None def __post_init__(self) -> None: try: @@ -160,8 +191,8 @@ def __post_init__(self) -> None: if value is not None and (not isinstance(value, str) or not value.strip()): raise OssieProfileError(f"{field_name} must be a non-empty string when provided") - resolve_ossie_profile(self.schema_version, self.consumer_profile) + resolve_ossie_profile(self.schema_version, self.consumer_profile, self.schema_revision) @property def profile(self) -> OssieProfile: - return resolve_ossie_profile(self.schema_version, self.consumer_profile) + return resolve_ossie_profile(self.schema_version, self.consumer_profile, self.schema_revision) diff --git a/sidemantic/interchange/ossie/schemas/logical/0.2.0.dev0-b6c702e/schema.json b/sidemantic/interchange/ossie/schemas/logical/0.2.0.dev0-b6c702e/schema.json new file mode 100644 index 000000000..a8d2fcbdc --- /dev/null +++ b/sidemantic/interchange/ossie/schemas/logical/0.2.0.dev0-b6c702e/schema.json @@ -0,0 +1,374 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/apache/ossie/core-spec/ossie-schema.json", + "title": "Apache Ossie Core Metadata Specification", + "description": "JSON Schema for validating a single Apache Ossie semantic model document", + "type": "object", + "properties": { + "version": { + "type": "string", + "const": "0.2.0.dev0", + "description": "Apache Ossie specification version" + }, + "name": { + "$ref": "#/$defs/SemanticModel/properties/name" + }, + "description": { + "$ref": "#/$defs/SemanticModel/properties/description" + }, + "ai_context": { + "$ref": "#/$defs/SemanticModel/properties/ai_context" + }, + "datasets": { + "$ref": "#/$defs/SemanticModel/properties/datasets" + }, + "relationships": { + "$ref": "#/$defs/SemanticModel/properties/relationships" + }, + "metrics": { + "$ref": "#/$defs/SemanticModel/properties/metrics" + }, + "custom_extensions": { + "$ref": "#/$defs/SemanticModel/properties/custom_extensions" + } + }, + "required": ["version", "name", "datasets"], + "additionalProperties": false, + "$defs": { + "Dialect": { + "type": "string", + "enum": ["ANSI_SQL", "SNOWFLAKE", "MDX", "TABLEAU", "DATABRICKS", "MAQL", "BIGQUERY", "SIGMA", "THOUGHTSPOT", "DAX", "OSSIE_SQL_2026"], + "description": "Supported SQL and expression language dialects" + }, + "Vendor": { + "type": "string", + "examples": ["COMMON", "SNOWFLAKE", "SALESFORCE", "DBT", "DATABRICKS", "GOODDATA", "WISDOM", "POWER_BI"], + "description": "Vendor name for custom extensions. Any string value is accepted." + }, + "AIContext": { + "description": "Additional context for AI tools", + "oneOf": [ + { + "type": "string" + }, + { + "type": "object", + "properties": { + "instructions": { + "type": "string", + "description": "Instructions for AI on how to use this entity" + }, + "synonyms": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Alternative names and terms" + }, + "examples": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Sample questions or use cases" + } + }, + "additionalProperties": true + } + ] + }, + "CustomExtension": { + "type": "object", + "description": "Vendor-specific attributes for extensibility", + "properties": { + "vendor_name": { + "$ref": "#/$defs/Vendor" + }, + "data": { + "type": "string", + "description": "JSON string containing vendor-specific data" + } + }, + "required": ["vendor_name", "data"], + "additionalProperties": false + }, + "DialectExpression": { + "type": "object", + "description": "Expression in a specific dialect", + "properties": { + "dialect": { + "$ref": "#/$defs/Dialect" + }, + "expression": { + "type": "string", + "description": "SQL or dialect-specific expression" + } + }, + "required": ["dialect", "expression"], + "additionalProperties": false + }, + "Expression": { + "type": "object", + "description": "Expression definition with multi-dialect support", + "properties": { + "dialects": { + "type": "array", + "items": { + "$ref": "#/$defs/DialectExpression" + }, + "minItems": 1 + } + }, + "required": ["dialects"], + "additionalProperties": false + }, + "DataType": { + "type": "string", + "enum": [ + "String", + "Integer", + "Decimal", + "Float", + "Boolean", + "Date", + "Time", + "DateTime", + "DateTimeTz", + "Opaque" + ], + "description": "Logical data type for fields and metrics, independent of role (e.g. dimension vs fact) and physical representation. `Decimal` is exact base-10 with unspecified precision and scale; `Float` is approximate. `DateTime` has no timezone or offset, while `DateTimeTz` identifies an instant using offset or timezone context but does not guarantee preservation of a named timezone. Omit `datatype` when unknown; use `Opaque` plus `custom_extensions` for a known type outside the portable vocabulary." + }, + "Dimension": { + "type": "object", + "description": "Dimension metadata", + "properties": { + "is_time": { + "type": "boolean", + "description": "Temporal-role marker. When true, consumers that distinguish time dimensions (e.g. for time-series analysis or temporal filtering) should treat this field as a time dimension. This is a *role* flag, independent of the field's data type: a field with `is_time: true` may carry any `datatype` (e.g. `Integer` for a year grain, `String` for a month name, as well as temporal data types). When `is_time` is unset, it defaults to `true` if `datatype` is one of `Date`, `Time`, `DateTime`, or `DateTimeTz`, and `false` otherwise. Set `is_time: false` explicitly to opt a temporal-typed column (such as an audit timestamp) out of time-dimension treatment." + } + }, + "additionalProperties": false + }, + "Field": { + "type": "object", + "description": "Row-level attribute for grouping, filtering, and metric expressions", + "properties": { + "name": { + "type": "string", + "minLength": 1, + "description": "Unique identifier for the field within the dataset" + }, + "expression": { + "$ref": "#/$defs/Expression" + }, + "dimension": { + "$ref": "#/$defs/Dimension" + }, + "label": { + "type": "string", + "description": "Label for categorization" + }, + "description": { + "type": "string", + "description": "Human-readable description" + }, + "datatype": { + "$ref": "#/$defs/DataType" + }, + "ai_context": { + "$ref": "#/$defs/AIContext" + }, + "custom_extensions": { + "type": "array", + "items": { + "$ref": "#/$defs/CustomExtension" + } + } + }, + "required": ["name", "expression"], + "additionalProperties": false + }, + "Dataset": { + "type": "object", + "description": "Logical dataset representing a business entity (fact or dimension table)", + "properties": { + "name": { + "type": "string", + "minLength": 1, + "description": "Unique identifier for the dataset" + }, + "source": { + "type": "string", + "minLength": 1, + "description": "Reference to underlying physical table/view (database.schema.table) or query" + }, + "primary_key": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Primary key columns (single or composite)" + }, + "unique_keys": { + "type": "array", + "items": { + "type": "array", + "items": { + "type": "string" + } + }, + "description": "Array of unique key definitions (each can be single or composite)" + }, + "description": { + "type": "string", + "description": "Human-readable description" + }, + "ai_context": { + "$ref": "#/$defs/AIContext" + }, + "fields": { + "type": "array", + "items": { + "$ref": "#/$defs/Field" + } + }, + "custom_extensions": { + "type": "array", + "items": { + "$ref": "#/$defs/CustomExtension" + } + } + }, + "required": ["name", "source"], + "additionalProperties": false + }, + "Relationship": { + "type": "object", + "description": "Foreign key relationship between datasets", + "properties": { + "name": { + "type": "string", + "minLength": 1, + "description": "Unique identifier for the relationship" + }, + "from": { + "type": "string", + "minLength": 1, + "description": "Dataset on the many side of the relationship" + }, + "to": { + "type": "string", + "minLength": 1, + "description": "Dataset on the one side of the relationship" + }, + "from_columns": { + "type": "array", + "items": { + "type": "string" + }, + "minItems": 1, + "description": "Foreign key columns in the 'from' dataset" + }, + "to_columns": { + "type": "array", + "items": { + "type": "string" + }, + "minItems": 1, + "description": "Primary/unique key columns in the 'to' dataset" + }, + "ai_context": { + "$ref": "#/$defs/AIContext" + }, + "custom_extensions": { + "type": "array", + "items": { + "$ref": "#/$defs/CustomExtension" + } + } + }, + "required": ["name", "from", "to", "from_columns", "to_columns"], + "additionalProperties": false + }, + "Metric": { + "type": "object", + "description": "Quantitative measure defined on business data", + "properties": { + "name": { + "type": "string", + "minLength": 1, + "description": "Unique identifier for the metric" + }, + "expression": { + "$ref": "#/$defs/Expression" + }, + "description": { + "type": "string", + "description": "Human-readable description of what the metric measures" + }, + "datatype": { + "$ref": "#/$defs/DataType" + }, + "ai_context": { + "$ref": "#/$defs/AIContext" + }, + "custom_extensions": { + "type": "array", + "items": { + "$ref": "#/$defs/CustomExtension" + } + } + }, + "required": ["name", "expression"], + "additionalProperties": false + }, + "SemanticModel": { + "type": "object", + "description": "Top-level container representing a complete semantic model", + "properties": { + "name": { + "type": "string", + "minLength": 1, + "description": "Unique identifier for the semantic model" + }, + "description": { + "type": "string", + "description": "Human-readable description" + }, + "ai_context": { + "$ref": "#/$defs/AIContext" + }, + "datasets": { + "type": "array", + "items": { + "$ref": "#/$defs/Dataset" + }, + "minItems": 1, + "description": "Collection of logical datasets" + }, + "relationships": { + "type": "array", + "items": { + "$ref": "#/$defs/Relationship" + }, + "description": "Defines how datasets are connected" + }, + "metrics": { + "type": "array", + "items": { + "$ref": "#/$defs/Metric" + }, + "description": "Quantifiable measures spanning datasets" + }, + "custom_extensions": { + "type": "array", + "items": { + "$ref": "#/$defs/CustomExtension" + } + } + }, + "required": ["name", "datasets"], + "additionalProperties": false + } + } +} diff --git a/sidemantic/interchange/ossie/schemas/manifest.json b/sidemantic/interchange/ossie/schemas/manifest.json index 4258223a0..478085e1c 100644 --- a/sidemantic/interchange/ossie/schemas/manifest.json +++ b/sidemantic/interchange/ossie/schemas/manifest.json @@ -70,6 +70,60 @@ "reason": "Resolve ontology references against the pinned logical schema without network access." } ] + }, + "logical-0.2.0.dev0-b6c702e": { + "document_kind": "logical", + "version": "0.2.0.dev0", + "schema_revision": "b6c702e", + "schema_dialect": "https://json-schema.org/draft/2020-12/schema", + "resource_uri": "urn:sidemantic:ossie:schema:logical:0.2.0.dev0:b6c702e", + "runtime_path": "logical/0.2.0.dev0-b6c702e/schema.json", + "runtime_sha256": "5b9cf15d31057e2b7363194b1c254a6b669fc412b3f8f402d4efbdcfa25fcc87", + "upstream_path": "logical/0.2.0.dev0-b6c702e/schema.json", + "upstream_sha256": "5b9cf15d31057e2b7363194b1c254a6b669fc412b3f8f402d4efbdcfa25fcc87", + "source": { + "repository": "https://github.com/apache/ossie", + "commit": "b6c702ed1c07e91382a69e870c875cbd19570828", + "path": "core-spec/ossie-schema.json", + "url": "https://raw.githubusercontent.com/apache/ossie/b6c702ed1c07e91382a69e870c875cbd19570828/core-spec/ossie-schema.json", + "license": "Apache-2.0" + }, + "transformations": [] + }, + "ontology-0.2.0.dev0-b6c702e": { + "document_kind": "ontology", + "version": "0.2.0.dev0", + "schema_revision": "b6c702e", + "schema_dialect": "https://json-schema.org/draft/2020-12/schema", + "resource_uri": "urn:sidemantic:ossie:schema:ontology:0.2.0.dev0:b6c702e", + "runtime_path": "ontology/0.2.0.dev0-b6c702e/schema.json", + "runtime_sha256": "a17df18b10aab95b4e890c8cecaba3fc3ee9a0cfc70352be32f9b55ea2c37bd5", + "upstream_path": "ontology/0.2.0.dev0-b6c702e/upstream.json", + "upstream_sha256": "0a742fc41b0999511084ea42f5070ceee19be3b542d81936c89e75b6800231a0", + "source": { + "repository": "https://github.com/apache/ossie", + "commit": "b6c702ed1c07e91382a69e870c875cbd19570828", + "path": "ontology/ontology.json", + "url": "https://raw.githubusercontent.com/apache/ossie/b6c702ed1c07e91382a69e870c875cbd19570828/ontology/ontology.json", + "license": "Apache-2.0" + }, + "transformations": [ + { + "operation": "replace_id", + "from": "https://github.com/apache/ossie/ontology/ontology.json", + "to": "urn:sidemantic:ossie:schema:ontology:0.2.0.dev0:b6c702e", + "reason": "Use a revision-specific local ontology resource identity for offline validation." + }, + { + "operation": "replace_reference_base", + "from": "https://raw.githubusercontent.com/apache/ossie/main/core-spec/ossie-schema.json", + "to": "urn:sidemantic:ossie:schema:logical:0.2.0.dev0:b6c702e", + "reason": "Resolve ontology references against the same pinned logical schema revision without network access." + } + ], + "dependencies": [ + "logical-0.2.0.dev0-b6c702e" + ] } } } diff --git a/sidemantic/interchange/ossie/schemas/ontology/0.2.0.dev0-b6c702e/schema.json b/sidemantic/interchange/ossie/schemas/ontology/0.2.0.dev0-b6c702e/schema.json new file mode 100644 index 000000000..af452716c --- /dev/null +++ b/sidemantic/interchange/ossie/schemas/ontology/0.2.0.dev0-b6c702e/schema.json @@ -0,0 +1,341 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "urn:sidemantic:ossie:schema:ontology:0.2.0.dev0:b6c702e", + "title": "Apache Ossie Ontology Metadata Specification", + "description": "JSON Schema for validating Apache Ossie ontology definitions", + "type": "object", + "properties": { + "version": { + "type": "string", + "const": "0.2.0.dev0", + "description": "Ontology specification version" + }, + "name": { + "type": "string", + "description": "Unique identifier for the ontology" + }, + "description": { + "type": "string", + "description": "Human-readable description" + }, + "ai_context": { + "$ref": "urn:sidemantic:ossie:schema:logical:0.2.0.dev0:b6c702e#/$defs/AIContext" + }, + "requires": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that constrain the population of this ontology" + }, + "ontology": { + "type": "array", + "items": { + "$ref": "#/$defs/OntologyComponent" + }, + "minItems": 1, + "description": "Components that define the concepts and relationships in this ontology" + }, + "ontology_mappings": { + "type": "array", + "description": "Collection of ontology maps from logical models", + "items": { + "$ref": "#/$defs/OntologyMap" + } + }, + "prefixes": { + "type": "object", + "description": "Maps namespace prefixes to the IRIs they abbreviate, enabling QName expansion (e.g., 'foaf' -> 'http://xmlns.com/foaf/0.1/').", + "additionalProperties": { + "type": "string", + "format": "iri" + } + } + }, + "required": [ + "version", + "name", + "ontology" + ], + "additionalProperties": false, + "$defs": { + "OntologyComponent": { + "type": "object", + "description": "Ontology component that defines a single concept and any relationships that are keyed primarily by that concept", + "properties": { + "concept": { + "type": "string", + "description": "Unique name of the concept defined by this component" + }, + "type": { + "$ref": "#/$defs/ConceptType" + }, + "description": { + "type": "string", + "description": "Human-readable description of the concept" + }, + "extends": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Indicates that this concept extends one or more other concepts" + }, + "derived_by": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that define how this concept is derived" + }, + "identify_by": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Names of relationships to use as the preferred identifier of this concept" + }, + "requires": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that constrain the population of this concept" + }, + "relationships": { + "type": "array", + "items": { + "$ref": "#/$defs/Relationship" + }, + "description": "Defines relationships that pertain primarily to the concept defined in this component" + }, + "iri": { + "type": "string", + "description": "Optional global identifier for this concept, expressed as a full IRI or as a QName (prefix:local) resolved against the ontology-level 'prefixes' map." + } + }, + "required": [ + "concept", + "type" + ], + "additionalProperties": false + }, + "Expression": { + "type": "string", + "description": "ANSI SQL expression" + }, + "Relationship": { + "type": "object", + "description": "Relationship between concepts in the ontology", + "properties": { + "name": { + "type": "string", + "description": "Name of the relationship" + }, + "description": { + "type": "string", + "description": "Human-readable description of the relationship" + }, + "roles": { + "type": "array", + "items": { + "$ref": "#/$defs/Role" + }, + "description": "Additional roles in this relationship" + }, + "multiplicity": { + "$ref": "#/$defs/Multiplicity" + }, + "derived_by": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that define how this concept is derived" + }, + "requires": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that constrain the population of this relationship" + }, + "verbalizes": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Natural language expressions that verbalize this relationship" + }, + "iri": { + "type": "string", + "description": "Optional global identifier for this relationship, expressed as a full IRI or as a QName (prefix:local) resolved against the ontology-level 'prefixes' map." + } + }, + "required": [ + "name", + "verbalizes" + ], + "additionalProperties": false + }, + "ConceptMapping": { + "type": "object", + "description": "Mappings from logical model constructs to some ontology component", + "properties": { + "concept": { + "type": "string", + "description": "Name of the concept whose part of the ontology we are mapping to" + }, + "object_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/ObjectMapping" + }, + "description": "Mappings from logical constructs that populate the concept in this component. Valid only when the concept is an entity type" + }, + "link_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/LinkMapping" + }, + "description": "Mappings from logical model relationships to ontology relationships pertaining to the mapped concept" + } + }, + "required": [ + "concept" + ], + "additionalProperties": false + }, + "ConceptType": { + "type": "string", + "enum": [ + "EntityType", + "ValueType" + ], + "description": "A concept is either an entity type or a value type" + }, + "ReferentMapping": { + "type": "object", + "description": "Mapping from logical model constructs to a relationship used to references some entity type in the ontology", + "properties": { + "relationship": { + "type": "string", + "description": "Name of referent relationship" + }, + "expression": { + "$ref": "#/$defs/Expression" + }, + "referent_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/ReferentMapping" + } + } + }, + "required": [ + "relationship" + ], + "additionalProperties": false + }, + "Role": { + "type": "object", + "description": "Role in some relationship (the container)", + "properties": { + "concept": { + "type": "string", + "description": "Name of the concept playing this role" + }, + "name": { + "type": "string", + "description": "Optional name of this role, used when the same concept plays multiple roles in the same relationship" + } + }, + "required": [ + "concept" + ], + "additionalProperties": false + }, + "ObjectMapping": { + "type": "object", + "description": "Pattern of logical-level expressions for identifying objects of some concept using the values in one or more fields", + "properties": { + "concept": { + "type": "string", + "description": "Name of the concept whose objects we are mapping to" + }, + "referent_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/ReferentMapping" + }, + "description": "Maps logical-model constructs to referent relationships of this entity type" + }, + "expression": { + "$ref": "#/$defs/Expression" + } + }, + "additionalProperties": false + }, + "LinkMapping": { + "type": "object", + "description": "Mapping from logical schema to the links of relationships in the ontology", + "properties": { + "relationship": { + "type": "string", + "description": "Name of relationship being populated by this mapping node" + }, + "object_mapping": { + "$ref": "#/$defs/ObjectMapping" + }, + "children": { + "type": "array", + "items": { + "$ref": "#/$defs/LinkMapping" + }, + "description": "Relationship maps at the next level in this hierarchy" + } + }, + "required": [ + "object_mapping" + ], + "additionalProperties": false + }, + "Multiplicity": { + "type": "string", + "enum": [ + "ManyToOne", + "OneToOne" + ], + "description": "Relationship multiplicity" + }, + "OntologyMap": { + "type": "object", + "description": "Map from the constructs of some logical model to some ontology", + "properties": { + "name": { + "type": "string", + "description": "Name of this ontology map" + }, + "description": { + "type": "string", + "description": "Human-readable description of this ontology map" + }, + "semantic_model": { + "$ref": "urn:sidemantic:ossie:schema:logical:0.2.0.dev0:b6c702e" + }, + "concept_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/ConceptMapping" + }, + "description": "Maps logical model constructs to some concept and its relationships in the ontology" + } + }, + "required": [ + "semantic_model", + "concept_mappings" + ], + "additionalProperties": false + } + } +} diff --git a/sidemantic/interchange/ossie/schemas/ontology/0.2.0.dev0-b6c702e/upstream.json b/sidemantic/interchange/ossie/schemas/ontology/0.2.0.dev0-b6c702e/upstream.json new file mode 100644 index 000000000..68e16b838 --- /dev/null +++ b/sidemantic/interchange/ossie/schemas/ontology/0.2.0.dev0-b6c702e/upstream.json @@ -0,0 +1,314 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/apache/ossie/ontology/ontology.json", + "title": "Apache Ossie Ontology Metadata Specification", + "description": "JSON Schema for validating Apache Ossie ontology definitions", + "type": "object", + "properties": { + "version": { + "type": "string", + "const": "0.2.0.dev0", + "description": "Ontology specification version" + }, + "name": { + "type": "string", + "description": "Unique identifier for the ontology" + }, + "description": { + "type": "string", + "description": "Human-readable description" + }, + "ai_context": { + "$ref": "https://raw.githubusercontent.com/apache/ossie/main/core-spec/ossie-schema.json#/$defs/AIContext" + }, + "requires": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that constrain the population of this ontology" + }, + "ontology": { + "type": "array", + "items": { + "$ref": "#/$defs/OntologyComponent" + }, + "minItems": 1, + "description": "Components that define the concepts and relationships in this ontology" + }, + "ontology_mappings": { + "type": "array", + "description": "Collection of ontology maps from logical models", + "items": { + "$ref": "#/$defs/OntologyMap" + } + }, + "prefixes": { + "type": "object", + "description": "Maps namespace prefixes to the IRIs they abbreviate, enabling QName expansion (e.g., 'foaf' -> 'http://xmlns.com/foaf/0.1/').", + "additionalProperties": { + "type": "string", + "format": "iri" + } + } + }, + "required": ["version", "name", "ontology"], + "additionalProperties": false, + "$defs": { + "OntologyComponent": { + "type": "object", + "description": "Ontology component that defines a single concept and any relationships that are keyed primarily by that concept", + "properties": { + "concept": { + "type": "string", + "description": "Unique name of the concept defined by this component" + }, + "type": { + "$ref": "#/$defs/ConceptType" + }, + "description": { + "type": "string", + "description": "Human-readable description of the concept" + }, + "extends": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Indicates that this concept extends one or more other concepts" + }, + "derived_by": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that define how this concept is derived" + }, + "identify_by": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Names of relationships to use as the preferred identifier of this concept" + }, + "requires": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that constrain the population of this concept" + }, + "relationships": { + "type": "array", + "items": { + "$ref": "#/$defs/Relationship" + }, + "description": "Defines relationships that pertain primarily to the concept defined in this component" + }, + "iri": { + "type": "string", + "description": "Optional global identifier for this concept, expressed as a full IRI or as a QName (prefix:local) resolved against the ontology-level 'prefixes' map." + } + }, + "required": ["concept", "type"], + "additionalProperties": false + }, + "Expression": { + "type": "string", + "description": "ANSI SQL expression" + }, + "Relationship": { + "type": "object", + "description": "Relationship between concepts in the ontology", + "properties": { + "name": { + "type": "string", + "description": "Name of the relationship" + }, + "description": { + "type": "string", + "description": "Human-readable description of the relationship" + }, + "roles": { + "type": "array", + "items": { + "$ref": "#/$defs/Role" + }, + "description": "Additional roles in this relationship" + }, + "multiplicity": { + "$ref": "#/$defs/Multiplicity" + }, + "derived_by": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that define how this concept is derived" + }, + "requires": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that constrain the population of this relationship" + }, + "verbalizes": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Natural language expressions that verbalize this relationship" + }, + "iri": { + "type": "string", + "description": "Optional global identifier for this relationship, expressed as a full IRI or as a QName (prefix:local) resolved against the ontology-level 'prefixes' map." + } + }, + "required": ["name", "verbalizes"], + "additionalProperties": false + }, + "ConceptMapping": { + "type": "object", + "description": "Mappings from logical model constructs to some ontology component", + "properties": { + "concept": { + "type": "string", + "description": "Name of the concept whose part of the ontology we are mapping to" + }, + "object_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/ObjectMapping" + }, + "description": "Mappings from logical constructs that populate the concept in this component. Valid only when the concept is an entity type" + }, + "link_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/LinkMapping" + }, + "description": "Mappings from logical model relationships to ontology relationships pertaining to the mapped concept" + } + }, + "required": ["concept"], + "additionalProperties": false + }, + "ConceptType": { + "type": "string", + "enum": [ "EntityType", "ValueType" ], + "description": "A concept is either an entity type or a value type" + }, + "ReferentMapping": { + "type": "object", + "description": "Mapping from logical model constructs to a relationship used to references some entity type in the ontology", + "properties": { + "relationship": { + "type": "string", + "description": "Name of referent relationship" + }, + "expression": { + "$ref": "#/$defs/Expression" + }, + "referent_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/ReferentMapping" + } + } + }, + "required": ["relationship"], + "additionalProperties": false + }, + "Role": { + "type": "object", + "description": "Role in some relationship (the container)", + "properties": { + "concept": { + "type": "string", + "description": "Name of the concept playing this role" + }, + "name": { + "type": "string", + "description": "Optional name of this role, used when the same concept plays multiple roles in the same relationship" + } + }, + "required": ["concept"], + "additionalProperties": false + }, + "ObjectMapping": { + "type": "object", + "description": "Pattern of logical-level expressions for identifying objects of some concept using the values in one or more fields", + "properties": { + "concept": { + "type": "string", + "description": "Name of the concept whose objects we are mapping to" + }, + "referent_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/ReferentMapping" + }, + "description": "Maps logical-model constructs to referent relationships of this entity type" + }, + "expression": { + "$ref": "#/$defs/Expression" + } + }, + "additionalProperties": false + }, + "LinkMapping": { + "type": "object", + "description": "Mapping from logical schema to the links of relationships in the ontology", + "properties": { + "relationship": { + "type": "string", + "description": "Name of relationship being populated by this mapping node" + }, + "object_mapping": { + "$ref": "#/$defs/ObjectMapping" + }, + "children": { + "type": "array", + "items": { + "$ref": "#/$defs/LinkMapping" + }, + "description": "Relationship maps at the next level in this hierarchy" + } + }, + "required": ["object_mapping"], + "additionalProperties": false + }, + "Multiplicity": { + "type": "string", + "enum": [ "ManyToOne", "OneToOne" ], + "description": "Relationship multiplicity" + }, + "OntologyMap": { + "type": "object", + "description": "Map from the constructs of some logical model to some ontology", + "properties": { + "name": { + "type": "string", + "description": "Name of this ontology map" + }, + "description": { + "type": "string", + "description": "Human-readable description of this ontology map" + }, + "semantic_model": { + "$ref": "https://raw.githubusercontent.com/apache/ossie/main/core-spec/ossie-schema.json" + }, + "concept_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/ConceptMapping" + }, + "description": "Maps logical model constructs to some concept and its relationships in the ontology" + } + }, + "required": ["semantic_model", "concept_mappings"], + "additionalProperties": false + } + } +} diff --git a/sidemantic/interchange/ossie/semantic_validation.py b/sidemantic/interchange/ossie/semantic_validation.py index d19259533..8ab61ee35 100644 --- a/sidemantic/interchange/ossie/semantic_validation.py +++ b/sidemantic/interchange/ossie/semantic_validation.py @@ -8,6 +8,7 @@ from __future__ import annotations +import re from collections import Counter from collections.abc import Mapping, Sequence from dataclasses import dataclass @@ -22,10 +23,12 @@ from sidemantic.interchange.ossie.documents import ( OssieLogicalDocument, OssieOntologyDocument, + is_logical_document_data, ) from sidemantic.interchange.ossie.identifier import ( OSSIE_IDENTIFIER_MAX_LENGTH, identifier_length, + identifier_syntax_valid, identifier_within_limit, normalize_identifier, ) @@ -39,6 +42,8 @@ SemanticDocumentKind = Literal["logical", "ontology", "unsupported"] SemanticFailureStage = Literal["semantic"] JSONObject: TypeAlias = Mapping[str, object] +_BUILTIN_VALUE_CONCEPTS = frozenset({"Boolean", "Date", "DateTime", "Decimal", "Float", "Integer", "String"}) +_BUILTIN_CONCEPTS = _BUILTIN_VALUE_CONCEPTS | {"Any"} @dataclass(frozen=True, slots=True) @@ -121,7 +126,14 @@ def _profile_for( if not isinstance(version, str): return None try: - return resolve_ossie_profile(version, OssieConsumerProfile.OSSIE_CORE) + from sidemantic.interchange.ossie.validation import detect_schema_profile + + schema = detect_schema_profile(document) + return resolve_ossie_profile( + version, + OssieConsumerProfile.OSSIE_CORE, + schema.source_commit if schema and schema.schema_revision else None, + ) except OssieProfileError: return None @@ -204,6 +216,14 @@ def validate_identifier_length( *, scope: str | None = None, ) -> bool: + if not identifier_syntax_valid(identifier): + self.emit( + "ossie.semantic.identifier.invalid", + "Expected an ANSI regular identifier or a non-empty double-quoted identifier with doubled interior quotes.", + json_pointer, + scope=scope, + ) + return False length = identifier_length(identifier) if length <= OSSIE_IDENTIFIER_MAX_LENGTH: return True @@ -657,12 +677,41 @@ def validate_declared_keys( def validate_ontology(self, document: JSONObject) -> None: ontology = _array(document.get("ontology")) - concept_names = { - concept - for value in ontology or () - if (component := _mapping(value)) is not None - if (concept := _name(component.get("concept"))) is not None - } + self.ontology_concepts: dict[str, JSONObject] = {} + self.ontology_relationships: dict[tuple[str, str], JSONObject] = {} + concept_pointers: dict[str, str] = {} + for index, value in enumerate(ontology or ()): + component = _mapping(value) + if component is None or (concept := _name(component.get("concept"))) is None: + continue + pointer = _pointer("/ontology", index) + if concept in self.ontology_concepts or concept in _BUILTIN_CONCEPTS: + self.emit( + "ossie.semantic.ontology.concept_duplicate", + f"Duplicate concept {concept!r}.", + _pointer(pointer, "concept"), + ) + continue + self.ontology_concepts[concept] = component + concept_pointers[concept] = pointer + for relation_index, relation_value in enumerate(_array(component.get("relationships")) or ()): + relation = _mapping(relation_value) + if relation is None or (name := _name(relation.get("name"))) is None: + continue + key = (concept, name) + if key in self.ontology_relationships: + self.emit( + "ossie.semantic.ontology.relationship_duplicate", + f"Duplicate relationship {concept}.{name}.", + _pointer(pointer, "relationships", relation_index, "name"), + ) + else: + self.ontology_relationships[key] = relation + concept_names = set(self.ontology_concepts) | _BUILTIN_CONCEPTS + for prefix, iri in (_mapping(document.get("prefixes")) or {}).items(): + self.validate_ontology_iri(iri, _pointer("/prefixes", prefix), {}) + for concept, component in self.ontology_concepts.items(): + self.validate_ontology_component(concept, component, concept_pointers[concept], concept_names, document) ontology_mappings = _array(document.get("ontology_mappings")) embedded_models: list[object] = [] embedded_model_pointers: list[str] = [] @@ -688,24 +737,154 @@ def validate_ontology(self, document: JSONObject) -> None: concept_names=concept_names, ) object_mappings = _array(concept_mapping.get("object_mappings")) + link_mappings = _array(concept_mapping.get("link_mappings")) + if "object_mappings" not in concept_mapping and "link_mappings" not in concept_mapping: + self.emit( + "ossie.semantic.ontology.mapping_empty", + "A concept mapping requires object_mappings or link_mappings.", + concept_mapping_pointer, + ) for object_index, object_value in enumerate(object_mappings or ()): self.validate_object_mapping( object_value, pointer=_pointer(concept_mapping_pointer, "object_mappings", object_index), concept_names=concept_names, + owner=_name(concept_mapping.get("concept")), ) - link_mappings = _array(concept_mapping.get("link_mappings")) for link_index, link_value in enumerate(link_mappings or ()): self.validate_link_mapping( link_value, pointer=_pointer(concept_mapping_pointer, "link_mappings", link_index), concept_names=concept_names, + owner=_name(concept_mapping.get("concept")), ) # Ontology maps embed complete logical SemanticModel objects. Validate # each as an isolated scope while keeping ontology itself preservation-only. self.validate_embedded_semantic_models(embedded_models, embedded_model_pointers) + def ontology_supertypes(self, concept: str) -> set[str]: + """Find declared ancestors without interpreting population constraints.""" + found: set[str] = set() + pending = [concept] + while pending: + name = pending.pop() + if name in found: + continue + found.add(name) + component = self.ontology_concepts.get(name) + if component: + pending.extend(parent for parent in _array(component.get("extends")) or () if isinstance(parent, str)) + return found + + def ontology_relationship(self, owner: str | None, name: object) -> JSONObject | None: + if not isinstance(name, str): + return None + if "." in name: + concept, relationship = name.rsplit(".", 1) + if owner and concept not in self.ontology_supertypes(owner): + return None + return self.ontology_relationships.get((concept, relationship)) + candidates = [owner, *sorted(self.ontology_supertypes(owner) - {owner})] if owner else () + for concept in candidates: + if relation := self.ontology_relationships.get((concept, name)): + return relation + return None + + def validate_ontology_component( + self, concept: str, component: JSONObject, pointer: str, concept_names: set[str], document: JSONObject + ) -> None: + for index, parent in enumerate(_array(component.get("extends")) or ()): + reference_pointer = _pointer(pointer, "extends", index) + self.validate_concept_reference(parent, pointer=reference_pointer, concept_names=concept_names) + if isinstance(parent, str) and parent in concept_names: + parent_type = ( + "ValueType" + if parent in _BUILTIN_VALUE_CONCEPTS + else "EntityType" + if parent == "Any" + else self.ontology_concepts[parent].get("type") + ) + if parent_type != component.get("type"): + self.emit( + "ossie.semantic.ontology.supertype_kind", + "Entity and value concepts cannot extend each other.", + reference_pointer, + ) + if component.get("type") == "ValueType" and not self.ontology_supertypes(concept) & _BUILTIN_VALUE_CONCEPTS: + self.emit( + "ossie.semantic.ontology.value_base_missing", + "A value concept must extend a built-in value type directly or indirectly.", + _pointer(pointer, "extends"), + ) + self.validate_ontology_iri(component.get("iri"), _pointer(pointer, "iri"), document) + for index, value in enumerate(_array(component.get("relationships")) or ()): + relation = _mapping(value) + if relation is None: + continue + relation_pointer = _pointer(pointer, "relationships", index) + roles = _array(relation.get("roles")) or () + role_names = {concept} + for role_index, role_value in enumerate(roles): + role = _mapping(role_value) + if role is None: + continue + role_pointer = _pointer(relation_pointer, "roles", role_index) + self.validate_concept_reference( + role.get("concept"), pointer=_pointer(role_pointer, "concept"), concept_names=concept_names + ) + role_name = _name(role.get("name")) or _name(role.get("concept")) + if role_name and role_name in role_names: + self.emit( + "ossie.semantic.ontology.role_duplicate", + f"Role {role_name!r} requires a distinguishing name.", + role_pointer, + ) + if role_name: + role_names.add(role_name) + if relation.get("multiplicity") == "OneToOne" and len(roles) != 1: + self.emit( + "ossie.semantic.ontology.multiplicity_arity", + "OneToOne multiplicity requires a binary relationship.", + _pointer(relation_pointer, "multiplicity"), + ) + self.validate_ontology_iri(relation.get("iri"), _pointer(relation_pointer, "iri"), document) + for index, name in enumerate(_array(component.get("identify_by")) or ()): + reference_pointer = _pointer(pointer, "identify_by", index) + relation = self.ontology_relationship(concept, name) + if relation is None: + self.emit( + "ossie.semantic.ontology.relationship_unknown", + f"Unknown identifying relationship {name!r}.", + reference_pointer, + ) + elif len(_array(relation.get("roles")) or ()) != 1: + self.emit( + "ossie.semantic.ontology.identifier_arity", + "An identifying relationship must be binary.", + reference_pointer, + ) + + def validate_ontology_iri(self, value: object, pointer: str, document: JSONObject) -> None: + if value is None or not isinstance(value, str): + return + prefixes = _mapping(document.get("prefixes")) or {} + # An undeclared QName and an absolute opaque IRI share prefix:local + # syntax. Do not reject a valid custom IRI scheme merely because it is + # absent from the namespace map. + prefix, separator, local = value.partition(":") + invalid_characters = any( + character.isspace() or ord(character) < 32 or character in '<>"{}|\\^`' for character in value + ) + if separator and local and not invalid_characters and not re.search(r"%(?![0-9A-Fa-f]{2})", value): + if prefix in prefixes or re.fullmatch(r"[A-Za-z][A-Za-z0-9+.-]*", prefix): + return + self.emit( + "ossie.semantic.ontology.iri_invalid", + "An IRI must be absolute or use a QName prefix declared in prefixes.", + pointer, + ) + def validate_concept_reference( self, concept: object, @@ -716,7 +895,7 @@ def validate_concept_reference( if isinstance(concept, str) and concept and concept not in concept_names: self.emit( "ossie.semantic.ontology.concept_unknown", - f"Ontology mapping references unknown concept {concept!r}.", + f"Ontology references unknown concept {concept!r}.", pointer, ) @@ -726,6 +905,7 @@ def validate_object_mapping( *, pointer: str, concept_names: set[str], + owner: str | None = None, ) -> None: object_mapping = _mapping(value) if object_mapping is None: @@ -736,6 +916,41 @@ def validate_object_mapping( pointer=_pointer(pointer, "concept"), concept_names=concept_names, ) + owner = _name(object_mapping.get("concept")) or owner + self.validate_referent_mappings(object_mapping, pointer, owner) + + def validate_referent_mappings(self, value: JSONObject, pointer: str, owner: str | None) -> None: + referents = _array(value.get("referent_mappings")) or () + if "expression" not in value and "referent_mappings" not in value: + self.emit( + "ossie.semantic.ontology.object_mapping_empty", + "An object or referent mapping requires an expression or referent_mappings.", + pointer, + ) + for index, child in enumerate(referents): + referent = _mapping(child) + if referent is None: + continue + referent_pointer = _pointer(pointer, "referent_mappings", index) + relation = self.ontology_relationship(owner, referent.get("relationship")) + target = None + if relation is None: + self.emit( + "ossie.semantic.ontology.relationship_unknown", + f"Unknown referent relationship {referent.get('relationship')!r}.", + _pointer(referent_pointer, "relationship"), + ) + else: + roles = _array(relation.get("roles")) or () + if len(roles) != 1: + self.emit( + "ossie.semantic.ontology.identifier_arity", + "A referent relationship must be binary.", + _pointer(referent_pointer, "relationship"), + ) + elif role := _mapping(roles[0]): + target = _name(role.get("concept")) + self.validate_referent_mappings(referent, referent_pointer, target) def validate_link_mapping( self, @@ -743,22 +958,58 @@ def validate_link_mapping( *, pointer: str, concept_names: set[str], + owner: str | None = None, + depth: int = 1, ) -> None: link_mapping = _mapping(value) if link_mapping is None: return + relation = self.ontology_relationship(owner, link_mapping.get("relationship")) + object_owner = owner if depth == 1 else None + if depth > 1: + # Intermediate nodes may omit relationship and concept. Descendant + # relationships still identify the concept in this tuple position. + role_concepts: set[str] = set() + pending = [link_mapping] + while pending: + node = pending.pop() + mapped = self.ontology_relationship(owner, node.get("relationship")) + roles = _array(mapped.get("roles")) if mapped else None + if roles and depth <= len(roles) + 1: + role = _mapping(roles[depth - 2]) + if role and (role_concept := _name(role.get("concept"))): + role_concepts.add(role_concept) + pending.extend(child for child in _array(node.get("children")) or () if isinstance(child, Mapping)) + if len(role_concepts) == 1: + object_owner = next(iter(role_concepts)) if "object_mapping" in link_mapping: self.validate_object_mapping( link_mapping.get("object_mapping"), pointer=_pointer(pointer, "object_mapping"), concept_names=concept_names, + owner=object_owner, ) + if "relationship" in link_mapping: + if relation is None: + self.emit( + "ossie.semantic.ontology.relationship_unknown", + f"Unknown mapped relationship {link_mapping['relationship']!r}.", + _pointer(pointer, "relationship"), + ) + elif len(_array(relation.get("roles")) or ()) + 1 != depth: + self.emit( + "ossie.semantic.ontology.link_arity", + "Link mapping depth must equal the mapped relationship's arity.", + _pointer(pointer, "relationship"), + ) children = _array(link_mapping.get("children")) for child_index, child in enumerate(children or ()): self.validate_link_mapping( child, pointer=_pointer(pointer, "children", child_index), concept_names=concept_names, + owner=owner, + depth=depth + 1, ) def validate_embedded_semantic_models( @@ -828,7 +1079,7 @@ def validate_ossie_semantics( source = OssieSourceLocation(identifier=document.source.identifier) elif isinstance(document, Mapping): canonical_data = document - has_logical_root = "semantic_model" in canonical_data + has_logical_root = is_logical_document_data(canonical_data) has_ontology_root = "ontology" in canonical_data or "ontology_mappings" in canonical_data if has_logical_root and not has_ontology_root: document_kind = "logical" @@ -839,12 +1090,26 @@ def validate_ossie_semantics( else: raise TypeError("Semantic validation requires a parsed mapping or a supported Ossie document") + if ( + profile is None + and isinstance(document, (OssieLogicalDocument, OssieOntologyDocument)) + and document.schema_revision + ): + profile = resolve_ossie_profile(document.version, OssieConsumerProfile.OSSIE_CORE, document.schema_revision) validator = _SemanticValidator(profile=_profile_for(canonical_data, profile), source=source) if document_kind == "logical": - validator.validate_semantic_models( - _array(canonical_data.get("semantic_model")), - parent_pointer="/semantic_model", - ) + if "semantic_model" in canonical_data: + validator.validate_semantic_models( + _array(canonical_data.get("semantic_model")), + parent_pointer="/semantic_model", + ) + else: + name = _name(canonical_data.get("name")) + if name: + validator.validate_identifier_length(name, "/name") + scope = name or "semantic_model" + validator.checked_scopes.append(scope) + validator.validate_scope(canonical_data, "", scope) elif document_kind == "ontology": validator.validate_ontology(canonical_data) diff --git a/sidemantic/interchange/ossie/serialization.py b/sidemantic/interchange/ossie/serialization.py index 0e896c8ec..c231003b5 100644 --- a/sidemantic/interchange/ossie/serialization.py +++ b/sidemantic/interchange/ossie/serialization.py @@ -77,10 +77,11 @@ def _diagnostic( def _schema_profile_name(document: OssieDocument) -> str | None: if document.version is None: return None + suffix = f"-{document.schema_revision[:7]}" if document.schema_revision else "" if isinstance(document, OssieLogicalDocument): - return f"logical-{document.version}" + return f"logical-{document.version}{suffix}" if isinstance(document, OssieOntologyDocument): - return f"ontology-{document.version}" + return f"ontology-{document.version}{suffix}" return None @@ -120,7 +121,9 @@ def _validate_canonical_data( from sidemantic.interchange.ossie.validation import validate_ossie_schema validation_profile = profile - if validation_profile is None and document.version != "0.1.0": + if validation_profile is None and ( + document.schema_revision or document.version not in {"0.1.0", "0.1.1", "0.2.0.dev0"} + ): validation_profile = profile_name validation = validate_ossie_schema( canonical_data, @@ -147,9 +150,14 @@ def _exact_source_matches( options=OssieParseOptions( serialization=document.serialization, consumer_profile=consumer_profile or OssieConsumerProfile.OSSIE_CORE, + schema_revision=document.schema_revision, ), ) - return parsed.valid and parsed.document.to_parsed_data() == document.to_parsed_data() + # Python equality conflates JSON booleans and numbers (True == 1), including + # inside nested collections. Canonical JSON preserves their scalar types. + return parsed.valid and _canonical_json(parsed.document.to_parsed_data()) == _canonical_json( + document.to_parsed_data() + ) def _canonical_json(canonical_data: object) -> bytes: @@ -222,14 +230,14 @@ def serialize_ossie_document( ) source = document.source - if ( - exact_source - and output_serialization is document.serialization - and source is not None - and source.original_bytes is not None - ): + if exact_source: exact_source_consumer = profile.consumer_profile if profile is not None else consumer_profile - if _exact_source_matches(document, source.original_bytes, exact_source_consumer): + if ( + output_serialization is document.serialization + and source is not None + and source.original_bytes is not None + and _exact_source_matches(document, source.original_bytes, exact_source_consumer) + ): return OssieSerializationResult( data=source.original_bytes, serialization=output_serialization, @@ -240,7 +248,7 @@ def serialize_ossie_document( _diagnostic( document, code="ossie.serialization.exact_source_mismatch", - message="Retained source bytes no longer match canonical data; canonical serialization was used", + message="Exact source bytes are unavailable, use a different serialization, or no longer match canonical data; canonical serialization was used", severity=OssieDiagnosticSeverity.WARNING, ) ) diff --git a/sidemantic/interchange/ossie/synthesis.py b/sidemantic/interchange/ossie/synthesis.py index a686ddffc..86d09e79e 100644 --- a/sidemantic/interchange/ossie/synthesis.py +++ b/sidemantic/interchange/ossie/synthesis.py @@ -28,6 +28,7 @@ from sidemantic.interchange.ossie.expression_validation import scalar_sql_expression_error from sidemantic.interchange.ossie.identifier import normalize_identifier from sidemantic.interchange.ossie.profiles import ( + CURRENT_OSSIE_SCHEMA_COMMIT, OssieConsumerProfile, OssieProfileError, OssieSerialization, @@ -95,6 +96,23 @@ def _scalar_expression_error(text: str, dialect: str) -> str | None: return scalar_sql_expression_error(text, sqlglot_dialect=_SQLGLOT_DIALECTS[dialect]) +def _field_expression(text: str, dialect: str) -> str: + """Remove native owner qualifiers from dataset-local SQL, not literals.""" + try: + tokens = sqlglot.tokenize(text, read=_SQLGLOT_DIALECTS[dialect]) + except sqlglot.errors.SqlglotError: + return text # The expression validator reports malformed SQL below. + for index in reversed(range(len(tokens) - 3)): + first, last = tokens[index], tokens[index + 3] + if ( + first.token_type is sqlglot.TokenType.L_BRACE + and last.token_type is sqlglot.TokenType.DOT + and text[first.start : tokens[index + 2].end + 1] == "{model}" + ): + text = text[: first.start] + text[last.end + 1 :] + return text + + def _metric_expression(metric: Metric) -> str | None: if metric.type not in {None, "ratio", "derived"}: return None @@ -255,7 +273,9 @@ def check(item: BaseModel, handled: set[str], location: str) -> None: # Lowering adds these source-location annotations to every core model. # They describe the input document, not native state to synthesize. source_annotations = ( - {"metadata"} if not set(model.metadata or {}) - {"ossie_source_kind", "ossie_pointer"} else set() + {"metadata"} + if not set(model.metadata or {}) - {"ossie_source_kind", "ossie_pointer", "ossie_source_name"} + else set() ) check( model, @@ -274,9 +294,13 @@ def check(item: BaseModel, handled: set[str], location: str) -> None: f"Model {model.name!r}", ) for dimension in model.dimensions: + # Imported identifiers are already bound to runtime names. Their + # original spelling is archival provenance, not executable state. + source_annotations = {"metadata"} if not set(dimension.metadata or {}) - {"ossie_source_name"} else set() check( dimension, - {"name", "type", "sql", "logical_data_type", "declared_is_time", "description", "label", "public"}, + {"name", "type", "sql", "logical_data_type", "declared_is_time", "description", "label", "public"} + | source_annotations, f"Field {model.name}.{dimension.name}", ) for relationship in model.relationships: @@ -290,7 +314,8 @@ def check(item: BaseModel, handled: set[str], location: str) -> None: # with model source annotations, these do not add native semantics. source_annotations = ( {"metadata"} - if not set(metric.metadata or {}) - {"ossie_expression_dialect", "ossie_target_dialect"} + if not set(metric.metadata or {}) + - {"ossie_expression_dialect", "ossie_target_dialect", "ossie_source_name"} else set() ) check( @@ -362,8 +387,10 @@ def _runtime_expression_diagnostics(graph: SemanticGraph, dialect: str) -> list[ return diagnostics -def _dataset(model: Model, dialect: str, index: int, diagnostics: list[OssieDiagnostic]) -> dict[str, object] | None: - pointer = f"/semantic_model/0/datasets/{index}" +def _dataset( + model: Model, dialect: str, index: int, diagnostics: list[OssieDiagnostic], scope_pointer: str +) -> dict[str, object] | None: + pointer = f"{scope_pointer}/datasets/{index}" if ( model.extends or model.invariant_filters @@ -437,7 +464,8 @@ def _dataset(model: Model, dialect: str, index: int, diagnostics: list[OssieDiag ) ) continue - expression_error = _scalar_expression_error(dimension.sql_expr, dialect) + expression = _field_expression(dimension.sql_expr, dialect) + expression_error = _scalar_expression_error(expression, dialect) if expression_error is not None: diagnostics.append( _error( @@ -449,7 +477,7 @@ def _dataset(model: Model, dialect: str, index: int, diagnostics: list[OssieDiag continue field: dict[str, object] = { "name": dimension.name, - "expression": _expression(dimension.sql_expr, dialect), + "expression": _expression(expression, dialect), } if dimension.logical_data_type is not None: if dimension.logical_data_type not in _DATA_TYPES: @@ -468,6 +496,9 @@ def _dataset(model: Model, dialect: str, index: int, diagnostics: list[OssieDiag # Preserve runtime time-role semantics when datatype omission would # otherwise make Ossie default this field to non-time. field["dimension"] = {"is_time": True} + elif dimension.type != "time" and dimension.logical_data_type in {"Date", "Time", "DateTime", "DateTimeTz"}: + # A native non-time role must override Ossie's temporal-type default. + field["dimension"] = {"is_time": False} if dimension.description is not None: field["description"] = dimension.description if dimension.label is not None: @@ -478,13 +509,15 @@ def _dataset(model: Model, dialect: str, index: int, diagnostics: list[OssieDiag return dataset -def _relationships(models: dict[str, Model], diagnostics: list[OssieDiagnostic]) -> list[dict[str, object]]: +def _relationships( + models: dict[str, Model], diagnostics: list[OssieDiagnostic], scope_pointer: str +) -> list[dict[str, object]]: result: list[dict[str, object]] = [] used_names: set[str] = set() relationship_index = 0 for from_model in models.values(): for relationship in from_model.relationships: - pointer = f"/semantic_model/0/relationships/{relationship_index}" + pointer = f"{scope_pointer}/relationships/{relationship_index}" relationship_index += 1 if relationship.target_model is not None: diagnostics.append( @@ -667,6 +700,7 @@ def synthesize_ossie_document( scope_name: str, expression_dialect: str, schema_version: str = "0.2.0.dev0", + schema_revision: str | None = CURRENT_OSSIE_SCHEMA_COMMIT, serialization: OssieSerialization | str = OssieSerialization.YAML, consumer_profile: OssieConsumerProfile | str = OssieConsumerProfile.OSSIE_CORE, portable_only: bool = False, @@ -686,20 +720,24 @@ def synthesize_ossie_document( diagnostics: list[OssieDiagnostic] = [] try: - profile = resolve_ossie_profile(schema_version, consumer_profile) + profile = resolve_ossie_profile( + schema_version, consumer_profile, schema_revision if schema_version == "0.2.0.dev0" else None + ) except OssieProfileError as exc: return OssieSynthesisResult( document=None, diagnostics=(_error("ossie.synthesis.profile_unsupported", str(exc), "/version"),), ) + current_shape = profile.schema_revision == CURRENT_OSSIE_SCHEMA_COMMIT + scope_pointer = "" if current_shape else "/semantic_model/0" models = dict(graph.models) datasets = [ dataset for index, model in enumerate(models.values()) - if (dataset := _dataset(model, dialect, index, diagnostics)) is not None + if (dataset := _dataset(model, dialect, index, diagnostics, scope_pointer)) is not None ] semantic_model: dict[str, object] = {"name": scope_name, "datasets": datasets} - relationships = _relationships(models, diagnostics) + relationships = _relationships(models, diagnostics, scope_pointer) generated_fields: dict[str, dict[str, str]] = {} metrics = _metrics(graph, models, dialect, diagnostics, generated_fields) for dataset in datasets: @@ -711,7 +749,11 @@ def synthesize_ossie_document( semantic_model["relationships"] = relationships if metrics: semantic_model["metrics"] = metrics - data = {"version": schema_version, "semantic_model": [semantic_model]} + data = ( + {"version": schema_version, **semantic_model} + if current_shape + else {"version": schema_version, "semantic_model": [semantic_model]} + ) diagnostics.extend(_unrepresented_state_diagnostics(graph)) extension_codes = { @@ -743,13 +785,20 @@ def synthesize_ossie_document( } ], } + if current_shape: + # The extension fingerprints the exact containing core object. + semantic_model = {"version": schema_version, **semantic_model} try: extension = encode_runtime_extension(graph, semantic_model, expression_dialect=dialect) except (TypeError, ValueError) as exc: diagnostics.append(_error("ossie.synthesis.runtime_extension_invalid", str(exc))) else: semantic_model["custom_extensions"] = [extension] - data = {"version": schema_version, "vendors": ["SIDEMANTIC"], "semantic_model": [semantic_model]} + data = ( + semantic_model + if current_shape + else {"version": schema_version, "vendors": ["SIDEMANTIC"], "semantic_model": [semantic_model]} + ) diagnostics = [ _warning( "ossie.synthesis.runtime_extension_required", @@ -765,7 +814,9 @@ def synthesize_ossie_document( if any(diagnostic.severity is OssieDiagnosticSeverity.ERROR for diagnostic in diagnostics): return OssieSynthesisResult(document=None, diagnostics=tuple(diagnostics)) - document = OssieLogicalDocument(canonical_data=data, serialization=serialization) + document = OssieLogicalDocument( + canonical_data=data, serialization=serialization, schema_revision=profile.schema_revision + ) return OssieSynthesisResult(document=document, diagnostics=tuple(diagnostics)) diff --git a/sidemantic/interchange/ossie/validation.py b/sidemantic/interchange/ossie/validation.py index 8411e913e..4b82f57c2 100644 --- a/sidemantic/interchange/ossie/validation.py +++ b/sidemantic/interchange/ossie/validation.py @@ -24,6 +24,7 @@ sort_diagnostics, ) from sidemantic.interchange.ossie.profiles import ( + CURRENT_OSSIE_SCHEMA_COMMIT, OssieConsumerProfile, OssieProfile, resolve_ossie_profile, @@ -59,6 +60,7 @@ class SchemaProfile: source_url: str transformations: tuple[Mapping[str, str], ...] dependencies: tuple[str, ...] + schema_revision: str | None = None @dataclass(frozen=True, slots=True) @@ -141,6 +143,7 @@ def _profile_from_record(name: str, record: Mapping[str, Any]) -> SchemaProfile: source_url=source["url"], transformations=tuple(record.get("transformations", ())), dependencies=tuple(record.get("dependencies", ())), + schema_revision=record.get("schema_revision"), ) @@ -214,6 +217,10 @@ def _verify_bundle_integrity(profile: SchemaProfile) -> Mapping[str, Any]: upstream_bytes, upstream = _read_verified_asset(profile, upstream=True) expected_name = f"{profile.document_kind}-{profile.version}" + expected_resource_uri = f"urn:sidemantic:ossie:schema:{profile.document_kind}:{profile.version}" + if profile.schema_revision: + expected_name += f"-{profile.schema_revision}" + expected_resource_uri += f":{profile.schema_revision}" repository_prefix = "https://github.com/" repository_slug = profile.source_repository.removeprefix(repository_prefix).rstrip("/") expected_source_url = ( @@ -228,7 +235,8 @@ def _verify_bundle_integrity(profile: SchemaProfile) -> Mapping[str, Any]: or profile.source_url != expected_source_url or runtime.get("$schema") != profile.schema_dialect or not isinstance(runtime.get("$id"), str) - or profile.resource_uri != f"urn:sidemantic:ossie:schema:{profile.document_kind}:{profile.version}" + or profile.resource_uri != expected_resource_uri + or (profile.schema_revision and profile.schema_revision != profile.source_commit[:7]) or version_schema.get("const") != profile.version or len(profile.dependencies) != len(set(profile.dependencies)) or any(dependency == profile.name for dependency in profile.dependencies) @@ -282,15 +290,43 @@ def detect_schema_profile(document: Any) -> SchemaProfile | None: if version != "0.2.0.dev0": return None - has_logical_root = "semantic_model" in document + from sidemantic.interchange.ossie.documents import is_logical_document_data + + has_logical_root = is_logical_document_data(document) has_ontology_root = "ontology" in document or "ontology_mappings" in document if has_logical_root == has_ontology_root: return None if has_logical_root: - return get_schema_profile("logical-0.2.0.dev0") + suffix = "" if "semantic_model" in document else f"-{CURRENT_OSSIE_SCHEMA_COMMIT[:7]}" + return get_schema_profile(f"logical-0.2.0.dev0{suffix}") + if _current_ontology_shape(document): + return get_schema_profile(f"ontology-0.2.0.dev0-{CURRENT_OSSIE_SCHEMA_COMMIT[:7]}") return get_schema_profile("ontology-0.2.0.dev0") +def _current_ontology_shape(document: Mapping[str, object]) -> bool: + if "prefixes" in document: + return True + ontology = document.get("ontology") + for concept in ontology if isinstance(ontology, (list, tuple)) else (): + if not isinstance(concept, Mapping): + continue + if "iri" in concept: + return True + relationships = concept.get("relationships") + if isinstance(relationships, (list, tuple)) and any( + isinstance(relationship, Mapping) and "iri" in relationship for relationship in relationships + ): + return True + mappings = document.get("ontology_mappings") + return isinstance(mappings, (list, tuple)) and any( + isinstance(mapping, Mapping) + and isinstance(mapping.get("semantic_model"), Mapping) + and "version" in mapping["semantic_model"] + for mapping in mappings + ) + + def _json_pointer(path: Sequence[JsonPathPart]) -> str: def escape(part: JsonPathPart) -> str: return str(part).replace("~", "~0").replace("/", "~1") @@ -335,7 +371,11 @@ def _diagnostic( schema: OssieSchemaProvenance | None = None if profile is not None: if ossie_profile is None: - ossie_profile = resolve_ossie_profile(profile.version, OssieConsumerProfile.OSSIE_CORE) + ossie_profile = resolve_ossie_profile( + profile.version, + OssieConsumerProfile.OSSIE_CORE, + profile.source_commit if profile.schema_revision else None, + ) schema = OssieSchemaProvenance( schema_id=profile.resource_uri, version=profile.version, @@ -452,9 +492,13 @@ def validate_ossie_schema( if contract_profile.is_compatibility_alias: resolved_profile = get_schema_profile(f"logical-{contract_profile.validation_schema_version}") else: - resolved_profile = detect_schema_profile(document) or get_schema_profile( - f"logical-{contract_profile.validation_schema_version}" + family = ( + "ontology" + if isinstance(document, Mapping) and ("ontology" in document or "ontology_mappings" in document) + else "logical" ) + suffix = f"-{contract_profile.schema_revision[:7]}" if contract_profile.schema_revision else "" + resolved_profile = get_schema_profile(f"{family}-{contract_profile.validation_schema_version}{suffix}") except KeyError: resolved_profile = None elif isinstance(profile, SchemaProfile): @@ -494,7 +538,11 @@ def validate_ossie_schema( resolved_profile = get_schema_profile("logical-0.1.1") elif explicit_consumer is not None: try: - contract_profile = resolve_ossie_profile(document_version, explicit_consumer) + contract_profile = resolve_ossie_profile( + document_version, + explicit_consumer, + resolved_profile.source_commit if resolved_profile and resolved_profile.schema_revision else None, + ) except ValueError as exc: diagnostic = _diagnostic( code="ossie.schema.profile_context_mismatch", diff --git a/sidemantic/loaders.py b/sidemantic/loaders.py index 55dabdfc1..872f2c38c 100644 --- a/sidemantic/loaders.py +++ b/sidemantic/loaders.py @@ -5,6 +5,7 @@ import runpy import sys import warnings +from collections.abc import Mapping from pathlib import Path from typing import TYPE_CHECKING @@ -173,6 +174,7 @@ def load_from_directory( strict: bool = True, only_file: "Path | None" = None, ossie_scope_id: str | None = None, + ossie_adapter_options: Mapping[str, object] | None = None, ) -> None: """Load all semantic layer definitions from a directory. @@ -210,6 +212,27 @@ def load_from_directory( from sidemantic.adapters.tmdl import TMDLAdapter from sidemantic.adapters.yardstick import YardstickAdapter + ossie_options = dict(ossie_adapter_options or {}) + unknown_options = ossie_options.keys() - { + "scope_id", + "target_dialect", + "source_dialect", + "consumer_profile", + "import_policy", + "preserve_source", + "schema_revision", + } + if unknown_options: + raise ValueError(f"adapter_options require an explicit source_format: {', '.join(sorted(unknown_options))}") + if ossie_scope_id is not None: + if "scope_id" in ossie_options and ossie_options["scope_id"] != ossie_scope_id: + raise ValueError("Conflicting Ossie scope selections") + ossie_options["scope_id"] = ossie_scope_id + ossie_options.setdefault("target_dialect", layer.dialect or "duckdb") + + def ossie_adapter(consumer_profile: str = "ossie-core") -> OssieAdapter: + return OssieAdapter(**{"consumer_profile": consumer_profile, **ossie_options}) + directory = Path(directory) if not directory.exists(): raise ValueError(f"Directory {directory} does not exist") @@ -413,18 +436,15 @@ def load_from_directory( else: # Sidemantic SQL files (pure SQL or with YAML frontmatter) adapter = SidemanticAdapter() + elif file_path.name.lower().endswith((".ossie.json", ".ossie.yaml", ".ossie.yml")): + # An explicit format suffix also routes malformed documents to the + # validated parser, so they cannot silently become an empty graph. + if _is_generated_artifact(file_path, directory): + continue + adapter = ossie_adapter() elif suffix == ".json": content = file_path.read_text() - if file_path.name.lower().endswith(".ossie.json"): - # An explicit format suffix opts into Ossie discovery anywhere - # in the source tree, including malformed files without markers. - if _is_generated_artifact(file_path, directory): - continue - adapter = OssieAdapter( - target_dialect=layer.dialect or "duckdb", - scope_id=ossie_scope_id, - ) - elif '"ldm"' in content and '"datasets"' in content: + if '"ldm"' in content and '"datasets"' in content: adapter = GoodDataAdapter() elif '"projectModel"' in content: adapter = GoodDataAdapter() @@ -433,9 +453,9 @@ def load_from_directory( elif '"datasets"' in content and ('"dataSourceTableId"' in content or '"data_source_table_id"' in content): adapter = GoodDataAdapter() elif ( - '"semantic_model"' in content - and '"datasets"' in content - and _is_under_osi_tree(file_path, directory) + '"datasets"' in content + and ('"semantic_model"' in content or ('"version"' in content and '"name"' in content)) + and (only_file is not None or _is_under_osi_tree(file_path, directory)) and not _is_generated_artifact(file_path, directory) ): # Released-spec OSI profile (dbt OSI consumer) ships as JSON in an @@ -455,11 +475,10 @@ def load_from_directory( _handle_parse_error(file_path, e, strict=strict) continue if is_osi: - adapter = OssieAdapter( - target_dialect=layer.dialect or "duckdb", - consumer_profile="dbt-1.12", - scope_id=ossie_scope_id, - ) + import json + + consumer_profile = "dbt-1.12" if json.loads(content).get("version") == "0.1.0" else "ossie-core" + adapter = ossie_adapter(consumer_profile) else: import json import re @@ -507,13 +526,9 @@ def load_from_directory( pass elif _yaml_has_top_level_key(yaml_data, "semantic_models"): adapter = MetricFlowAdapter() - elif _yaml_has_top_level_key(yaml_data, "semantic_model") and _contains_yaml_key(yaml_data, "datasets"): + elif _looks_like_ossie_mapping(yaml_data): consumer_profile = "dbt-1.12" if yaml_data.get("version") == "0.1.0" else "ossie-core" - adapter = OssieAdapter( - target_dialect=layer.dialect or "duckdb", - consumer_profile=consumer_profile, - scope_id=ossie_scope_id, - ) + adapter = ossie_adapter(consumer_profile) elif _yaml_has_top_level_key(yaml_data, "cubes") or ( _yaml_has_top_level_key(yaml_data, "views") and _contains_yaml_key(yaml_data, "measures") ): @@ -678,6 +693,7 @@ def load_from_file( *, strict: bool = True, ossie_scope_id: str | None = None, + ossie_adapter_options: Mapping[str, object] | None = None, ) -> None: """Load semantic definitions from a single file, ignoring sibling files. @@ -696,7 +712,14 @@ def load_from_file( file = Path(file) if not file.is_file(): raise ValueError(f"File {file} does not exist") - load_from_directory(layer, file.parent, strict=strict, only_file=file, ossie_scope_id=ossie_scope_id) + load_from_directory( + layer, + file.parent, + strict=strict, + only_file=file, + ossie_scope_id=ossie_scope_id, + ossie_adapter_options=ossie_adapter_options, + ) def _load_graphene_project( @@ -811,8 +834,19 @@ def _load_yaml_mapping(content: str) -> dict: return data if isinstance(data, dict) else {} +def _looks_like_ossie_mapping(data: object) -> bool: + """Identify both Ossie logical document shapes without matching other formats.""" + if not isinstance(data, dict): + return False + return ("semantic_model" in data and _contains_yaml_key(data, "datasets")) or { + "version", + "name", + "datasets", + }.issubset(data) + + def _looks_like_osi_json(content: str) -> bool: - """Return True for a released-spec OSI JSON document (dbt OSI consumer). + """Return True for a released or current Ossie JSON document. Released OSI ships as JSON with a top-level ``semantic_model`` list whose entries contain ``datasets``. This mirrors the YAML OSI detection and avoids @@ -828,14 +862,7 @@ def _looks_like_osi_json(content: str) -> bool: data = json.loads(content) except json.JSONDecodeError as e: raise ValueError(f"Invalid JSON: {e}") from e - if not isinstance(data, dict) or "semantic_model" not in data: - return False - models = data.get("semantic_model") - if isinstance(models, dict): - models = [models] - if not isinstance(models, list): - return False - return any(isinstance(model, dict) and "datasets" in model for model in models) + return _looks_like_ossie_mapping(data) # Directories that hold generated/compiled artifacts rather than source models. diff --git a/tests/adapters/osi/test_ossie_adapter.py b/tests/adapters/osi/test_ossie_adapter.py index 8aceb5088..efaaf334a 100644 --- a/tests/adapters/osi/test_ossie_adapter.py +++ b/tests/adapters/osi/test_ossie_adapter.py @@ -57,6 +57,8 @@ def test_historical_osi_class_spelling_routes_to_canonical_adapter(tmp_path: Pat source: analytics.customers concept_mappings: - concept: Customer + object_mappings: + - expression: customers.id """ @@ -208,7 +210,7 @@ def test_graph_export_requires_explicit_scope_and_dialect(tmp_path: Path) -> Non output = tmp_path / "model.yaml" OssieAdapter(export_scope_name="commerce", expression_dialect="BIGQUERY").export(graph, output) data = yaml.safe_load(output.read_text()) - dialect = data["semantic_model"][0]["datasets"][0]["fields"][0]["expression"]["dialects"][0] + dialect = data["datasets"][0]["fields"][0]["expression"]["dialects"][0] assert dialect == {"dialect": "BIGQUERY", "expression": "id"} @@ -238,7 +240,7 @@ def test_filtered_metric_export_roundtrip_executes(tmp_path: Path, adapter_class ) output = tmp_path / "model.yaml" adapter_class(export_scope_name="commerce", expression_dialect="ANSI_SQL").export(graph, output, portable_only=True) - assert not yaml.safe_load(output.read_text())["semantic_model"][0].get("custom_extensions") + assert not yaml.safe_load(output.read_text()).get("custom_extensions") layer = SemanticLayer() layer.graph = adapter_class().parse(output) layer.adapter.conn.execute("create table orders(amount integer, status varchar)") diff --git a/tests/interchange/ossie/test_current_schema.py b/tests/interchange/ossie/test_current_schema.py new file mode 100644 index 000000000..63c9fd76c --- /dev/null +++ b/tests/interchange/ossie/test_current_schema.py @@ -0,0 +1,167 @@ +"""Regression coverage for both immutable 0.2 development schema snapshots.""" + +from __future__ import annotations + +import json +from copy import deepcopy + +import pytest + +from sidemantic.interchange.ossie import ( + CURRENT_OSSIE_SCHEMA_COMMIT, + LEGACY_OSSIE_SCHEMA_COMMIT, + OssieParseOptions, + is_logical_document_data, + logical_model_entries, + parse_ossie_document, + serialize_ossie_document, + validate_ossie_semantics, +) +from sidemantic.interchange.ossie.validation import validate_ossie_schema + + +def _model(): + return { + "version": "0.2.0.dev0", + "name": "sales", + "datasets": [{"name": "orders", "source": "orders"}], + } + + +def _ontology(): + return { + "version": "0.2.0.dev0", + "name": "business", + "prefixes": {"biz": "https://example.org/business/"}, + "ontology": [{"concept": "Person", "type": "EntityType", "iri": "biz:Person"}], + "ontology_mappings": [{"semantic_model": _model(), "concept_mappings": []}], + } + + +@pytest.mark.parametrize("current", [False, True]) +def test_snapshot_selection_preserves_shape_provenance_and_bytes(current): + model = _model() + document = model if current else {"version": model.pop("version"), "semantic_model": [model]} + raw = json.dumps(document, indent=3).encode() + b"\n\n" + parsed = parse_ossie_document( + raw, options=OssieParseOptions(validate_schema=True, preservation_policy="source-bytes") + ) + + assert parsed.valid + assert parsed.schema_validation.schema_commit == ( + CURRENT_OSSIE_SCHEMA_COMMIT if current else LEGACY_OSSIE_SCHEMA_COMMIT + ) + assert parsed.document.to_parsed_data() == document + assert parsed.document.semantic_model_entries[0][0] == ("" if current else "/semantic_model/0") + assert len(parsed.document.semantic_models) == 1 + serialized = serialize_ossie_document(parsed.document, "json", exact_source=True) + assert serialized.exact_source_reused + assert serialized.data == raw + assert is_logical_document_data(document) + assert logical_model_entries(document)[0][0] == ("" if current else "/semantic_model/0") + + +@pytest.mark.parametrize("revision", [CURRENT_OSSIE_SCHEMA_COMMIT, LEGACY_OSSIE_SCHEMA_COMMIT]) +def test_explicit_revision_selects_a_pin_instead_of_silently_reinterpreting_shape(revision): + parsed = parse_ossie_document( + json.dumps(_model()).encode(), + options=OssieParseOptions(validate_schema=True, schema_revision=revision), + ) + assert parsed.valid is (revision == CURRENT_OSSIE_SCHEMA_COMMIT) + assert parsed.schema_validation.schema_commit == revision + + +def test_current_schema_has_distinct_profile_identity_and_root_diagnostics(): + model = _model() + model["datasets"][0]["source"] = "" + parsed = parse_ossie_document(json.dumps(model).encode(), options=OssieParseOptions(validate_schema=True)) + assert not parsed.valid + assert parsed.profile.schema_revision == CURRENT_OSSIE_SCHEMA_COMMIT + assert CURRENT_OSSIE_SCHEMA_COMMIT in parsed.profile.identifier + assert any(d.json_pointer == "/datasets/0/source" for d in parsed.diagnostics) + assert all(d.schema.commit == CURRENT_OSSIE_SCHEMA_COMMIT for d in parsed.diagnostics) + + +def test_flat_semantic_errors_keep_root_pointers(): + model = _model() + model["datasets"][0]["primary_key"] = ["missing"] + result = validate_ossie_semantics(model) + assert not result.valid + assert result.checked_scopes == ("sales",) + assert any(d.json_pointer == "/datasets/0/primary_key/0" for d in result.diagnostics) + assert all(d.profile.schema_revision == CURRENT_OSSIE_SCHEMA_COMMIT for d in result.diagnostics) + + +@pytest.mark.parametrize("dialect", ["DAX", "OSSIE_SQL_2026"]) +def test_new_dialects_are_current_schema_only(dialect): + model = _model() + model["datasets"][0]["fields"] = [ + {"name": "amount", "expression": {"dialects": [{"dialect": dialect, "expression": "amount"}]}} + ] + assert validate_ossie_schema(model).valid + legacy = {"version": model.pop("version"), "semantic_model": [model]} + result = validate_ossie_schema(legacy) + assert not result.valid + assert any(d.code == "ossie.schema.enum" for d in result.diagnostics) + + +def test_current_ontology_resolves_complete_core_references_offline(monkeypatch): + import socket + + def no_network(*args, **kwargs): + pytest.fail("Ontology validation attempted network access") + + monkeypatch.setattr(socket, "create_connection", no_network) + ontology = _ontology() + result = validate_ossie_schema(ontology) + assert result.valid + assert result.schema_commit == CURRENT_OSSIE_SCHEMA_COMMIT + assert validate_ossie_semantics(ontology).valid + del ontology["ontology_mappings"][0]["semantic_model"]["version"] + result = validate_ossie_schema(ontology) + assert not result.valid + assert any(d.json_pointer == "/ontology_mappings/0/semantic_model" for d in result.diagnostics) + + +def test_explicit_current_ontology_pin_survives_ambiguous_source_shape(): + ontology = _ontology() + del ontology["prefixes"] + del ontology["ontology"][0]["iri"] + del ontology["ontology_mappings"] + raw = json.dumps(ontology).encode() + result = parse_ossie_document( + raw, + options=OssieParseOptions( + schema_revision=CURRENT_OSSIE_SCHEMA_COMMIT, validate_schema=True, preservation_policy="source-bytes" + ), + ) + assert result.valid + assert result.document.schema_revision == CURRENT_OSSIE_SCHEMA_COMMIT + assert result.schema_validation.schema_commit == CURRENT_OSSIE_SCHEMA_COMMIT + serialized = serialize_ossie_document(result.document, "json", exact_source=True) + assert serialized.exact_source_reused + assert serialized.data == raw + + +def test_invalid_flat_document_is_classified_then_schema_validated(): + result = parse_ossie_document( + b'{"version":"0.2.0.dev0","name":"sales"}', options=OssieParseOptions(validate_schema=True) + ) + assert not result.valid + assert [d.code for d in result.diagnostics] == ["ossie.schema.required"] + + +def test_canonical_current_source_round_trip_keeps_unknown_ai_data(): + model = _model() + model["ai_context"] = {"custom": {"flag": True, "label": "café"}} + parsed = parse_ossie_document(json.dumps(model).encode()) + detached = deepcopy(parsed.document.to_parsed_data()) + detached["ai_context"]["custom"]["flag"] = False + assert parsed.document.to_parsed_data() == model + for serialization in ("json", "yaml"): + emitted = serialize_ossie_document(parsed.document, serialization) + reparsed = parse_ossie_document( + emitted.data, options=OssieParseOptions(serialization=serialization, validate_schema=True) + ) + assert reparsed.valid + assert reparsed.document.to_parsed_data() == model diff --git a/tests/interchange/ossie/test_ontology_semantics.py b/tests/interchange/ossie/test_ontology_semantics.py new file mode 100644 index 000000000..7c255002f --- /dev/null +++ b/tests/interchange/ossie/test_ontology_semantics.py @@ -0,0 +1,230 @@ +from __future__ import annotations + +from copy import deepcopy + +import pytest + +from sidemantic.interchange.ossie import validate_ossie_semantics +from sidemantic.interchange.ossie.validation import validate_ossie_schema + + +def _ontology(): + return { + "version": "0.2.0.dev0", + "name": "business", + "ontology": [ + { + "concept": "Person", + "type": "EntityType", + "identify_by": ["id"], + "relationships": [ + {"name": "id", "roles": [{"concept": "String"}], "multiplicity": "OneToOne", "verbalizes": []} + ], + }, + {"concept": "Employee", "type": "EntityType", "extends": ["Person"]}, + {"concept": "Salary", "type": "ValueType", "extends": ["Decimal"]}, + ], + "ontology_mappings": [ + { + "semantic_model": {"name": "sales", "datasets": [{"name": "orders", "source": "orders"}]}, + "concept_mappings": [ + {"concept": "Person", "object_mappings": [{"concept": "String", "expression": "orders.id"}]} + ], + } + ], + } + + +def _codes(document): + return {diagnostic.code for diagnostic in validate_ossie_semantics(document).diagnostics} + + +@pytest.mark.parametrize("builtin", ["Any", "Boolean", "Date", "DateTime", "Decimal", "Float", "Integer", "String"]) +def test_builtin_concepts_need_no_declaration_in_mapping_or_roles(builtin): + document = _ontology() + document["ontology_mappings"][0]["concept_mappings"][0]["object_mappings"][0]["concept"] = builtin + document["ontology"][0]["relationships"][0]["roles"][0]["concept"] = builtin + assert validate_ossie_schema(document).valid + assert validate_ossie_semantics(document).valid + + +def test_unknown_supertype_and_role_concepts_are_reported_with_source_pointers(): + document = _ontology() + document["ontology"][1]["extends"] = ["Missing"] + document["ontology"][0]["relationships"][0]["roles"][0]["concept"] = "Missing" + result = validate_ossie_semantics(document) + assert {d.json_pointer for d in result.diagnostics if d.code == "ossie.semantic.ontology.concept_unknown"} == { + "/ontology/1/extends/0", + "/ontology/0/relationships/0/roles/0/concept", + } + + +def test_duplicate_concepts_and_relationships_are_rejected(): + document = _ontology() + document["ontology"].append(deepcopy(document["ontology"][0])) + document["ontology"][0]["relationships"].append(deepcopy(document["ontology"][0]["relationships"][0])) + assert {"ossie.semantic.ontology.concept_duplicate", "ossie.semantic.ontology.relationship_duplicate"} <= _codes( + document + ) + + +@pytest.mark.parametrize("parent", ["Person", "Missing"]) +def test_value_concepts_require_a_reachable_builtin_value_base(parent): + document = _ontology() + document["ontology"][2]["extends"] = [parent] + assert "ossie.semantic.ontology.value_base_missing" in _codes(document) + + +def test_indirect_value_ancestry_is_valid_and_cycles_terminate(): + document = _ontology() + document["ontology"].append({"concept": "NetSalary", "type": "ValueType", "extends": ["Salary"]}) + assert validate_ossie_semantics(document).valid + document["ontology"][2]["extends"] = ["NetSalary"] + assert "ossie.semantic.ontology.value_base_missing" in _codes(document) + + +def test_entity_cannot_extend_value_concept(): + document = _ontology() + document["ontology"][1]["extends"] = ["String"] + assert "ossie.semantic.ontology.supertype_kind" in _codes(document) + + +def test_repeated_role_concepts_need_distinguishing_names(): + document = _ontology() + relationship = document["ontology"][0]["relationships"][0] + relationship["roles"] = [{"concept": "Person"}] + assert "ossie.semantic.ontology.role_duplicate" in _codes(document) + relationship["roles"][0]["name"] = "other" + assert validate_ossie_semantics(document).valid + + +def test_identifying_and_one_to_one_relationships_must_be_binary(): + document = _ontology() + document["ontology"][0]["relationships"][0]["roles"] = [] + assert {"ossie.semantic.ontology.identifier_arity", "ossie.semantic.ontology.multiplicity_arity"} <= _codes( + document + ) + + +def test_unknown_identifying_relationship_is_rejected(): + document = _ontology() + document["ontology"][0]["identify_by"] = ["missing"] + assert "ossie.semantic.ontology.relationship_unknown" in _codes(document) + + +@pytest.mark.parametrize( + "mapping", + [ + {"object_mappings": [{}]}, + {"object_mappings": [{"referent_mappings": [{"relationship": "id"}]}]}, + {}, + ], +) +def test_mapping_requires_its_expression_or_nested_mapping_structure(mapping): + document = _ontology() + document["ontology_mappings"][0]["concept_mappings"] = [{"concept": "Person", **mapping}] + assert _codes(document) & {"ossie.semantic.ontology.object_mapping_empty", "ossie.semantic.ontology.mapping_empty"} + + +def test_nested_referent_relationships_resolve_in_target_concept(): + document = _ontology() + document["ontology"].append( + { + "concept": "Account", + "type": "EntityType", + "identify_by": ["owner"], + "relationships": [{"name": "owner", "roles": [{"concept": "Person"}], "verbalizes": []}], + } + ) + mapping = { + "concept": "Account", + "object_mappings": [ + { + "referent_mappings": [ + {"relationship": "owner", "referent_mappings": [{"relationship": "id", "expression": "orders.id"}]} + ] + } + ], + } + document["ontology_mappings"][0]["concept_mappings"] = [mapping] + assert validate_ossie_semantics(document).valid + mapping["object_mappings"][0]["referent_mappings"][0]["referent_mappings"][0]["relationship"] = "missing" + assert "ossie.semantic.ontology.relationship_unknown" in _codes(document) + + +def test_link_mapping_relationship_and_arity_are_checked(): + document = _ontology() + mapping = { + "concept": "Person", + "link_mappings": [ + { + "object_mapping": {"expression": "orders.id"}, + "children": [ + {"relationship": "id", "object_mapping": {"concept": "String", "expression": "orders.id"}} + ], + } + ], + } + document["ontology_mappings"][0]["concept_mappings"] = [mapping] + assert validate_ossie_semantics(document).valid + mapping["link_mappings"][0]["relationship"] = "id" + assert "ossie.semantic.ontology.link_arity" in _codes(document) + mapping["link_mappings"][0]["relationship"] = "missing" + assert "ossie.semantic.ontology.relationship_unknown" in _codes(document) + + +@pytest.mark.parametrize("iri", ["biz:Person", "https://example.org/人", "urn:business:Person", "custom-scheme:Person"]) +def test_current_iri_metadata_accepts_qnames_and_absolute_schemes(iri): + document = _ontology() + document.pop("ontology_mappings") + document["prefixes"] = {"biz": "https://example.org/business/"} + document["ontology"][0]["iri"] = iri + assert validate_ossie_schema(document).valid + assert validate_ossie_semantics(document).valid + + +@pytest.mark.parametrize("iri", ["relative/path", "https://example.org/bad space", "https://example.org/%zz"]) +def test_malformed_iri_metadata_is_rejected(iri): + document = _ontology() + document["ontology"][0]["iri"] = iri + assert "ossie.semantic.ontology.iri_invalid" in _codes(document) + + +def test_intermediate_link_concept_is_inferred_from_descendant_relationship(): + document = _ontology() + document["ontology"].append( + { + "concept": "Company", + "type": "EntityType", + "identify_by": ["name"], + "relationships": [{"name": "name", "roles": [{"concept": "String"}], "verbalizes": []}], + } + ) + document["ontology"][0]["relationships"].append( + { + "name": "earns_at", + "roles": [{"concept": "Company"}, {"concept": "Salary"}], + "verbalizes": [], + } + ) + document["ontology_mappings"][0]["concept_mappings"] = [ + { + "concept": "Person", + "link_mappings": [ + { + "object_mapping": {"expression": "orders.id"}, + "children": [ + { + "object_mapping": { + "referent_mappings": [{"relationship": "name", "expression": "orders.company"}] + }, + "children": [ + {"relationship": "earns_at", "object_mapping": {"expression": "orders.salary"}} + ], + } + ], + } + ], + } + ] + assert validate_ossie_semantics(document).valid diff --git a/tests/interchange/ossie/test_ossie_parser.py b/tests/interchange/ossie/test_ossie_parser.py index 49e193084..db214f1cc 100644 --- a/tests/interchange/ossie/test_ossie_parser.py +++ b/tests/interchange/ossie/test_ossie_parser.py @@ -112,7 +112,7 @@ def test_classifies_supported_document_families(root, expected_type): "ossie.document.family_mixed", "mixes logical", ), - (b'{"version":"0.2.0.dev0","name":"unknown"}', "ossie.document.family_missing", "none of"), + (b'{"version":"0.2.0.dev0","unknown":true}', "ossie.document.family_missing", "neither"), (b'["version", "0.2.0.dev0"]', "ossie.document.root_type", "root must be an object"), ], ) @@ -297,3 +297,50 @@ def test_parser_enforces_a_bounded_input_budget(): assert isinstance(result.document, UnsupportedOssieDocument) assert diagnostic_codes(result) == ["ossie.parse.limit"] + + +def test_yaml_merge_overrides_are_not_duplicate_authored_keys(): + raw = b"""version: 0.2.0.dev0 +semantic_model: +- name: sales + datasets: + - &base {name: orders, source: orders} + - <<: *base + name: returns + source: returns +""" + result = parse_ossie_document(raw, options=OssieParseOptions(validate_schema=True)) + assert result.valid + assert [dataset["name"] for dataset in result.document.semantic_models[0]["datasets"]] == ["orders", "returns"] + + +@pytest.mark.parametrize("merged", [False, True]) +def test_duplicate_explicit_keys_are_rejected_even_with_merge(merged): + prefix = "base: &base {name: base}\n" if merged else "" + merge = " <<: *base\n" if merged else "" + raw = (prefix + "mapping:\n" + merge + " name: first\n name: second\n").encode() + result = parse_ossie_document(raw) + assert diagnostic_codes(result) == ["ossie.parse.duplicate_key"] + + +@pytest.mark.parametrize("merge", [False, True]) +def test_yaml_alias_expansion_is_bounded_before_construction(merge): + if merge: + lines = ["a0: &a0 {name: source}"] + for index in range(1, 9): + aliases = ", ".join([f"*a{index - 1}"] * 10) + lines.append(f"a{index}: &a{index} {{<<: [{aliases}]}}") + else: + lines = ["a0: &a0 [0]"] + for index in range(1, 9): + aliases = ", ".join([f"*a{index - 1}"] * 10) + lines.append(f"a{index}: &a{index} [{aliases}]") + result = parse_ossie_document("\n".join(lines).encode()) + assert diagnostic_codes(result) == ["ossie.parse.limit"] + + +@pytest.mark.parametrize("serialization", ["json", "yaml", None]) +def test_oversized_integer_returns_diagnostic(serialization): + raw = b"9" * 4500 + result = parse_ossie_document(raw, options=OssieParseOptions(serialization=serialization)) + assert diagnostic_codes(result) == ["ossie.parse.non_json_value"] diff --git a/tests/interchange/ossie/test_schema_validation.py b/tests/interchange/ossie/test_schema_validation.py index de083d141..35143c722 100644 --- a/tests/interchange/ossie/test_schema_validation.py +++ b/tests/interchange/ossie/test_schema_validation.py @@ -101,6 +101,17 @@ def test_schema_profiles_have_exact_pins_and_integrity() -> None: ), } + expected["logical-0.2.0.dev0-b6c702e"] = ( + "b6c702ed1c07e91382a69e870c875cbd19570828", + "5b9cf15d31057e2b7363194b1c254a6b669fc412b3f8f402d4efbdcfa25fcc87", + "5b9cf15d31057e2b7363194b1c254a6b669fc412b3f8f402d4efbdcfa25fcc87", + ) + expected["ontology-0.2.0.dev0-b6c702e"] = ( + "b6c702ed1c07e91382a69e870c875cbd19570828", + "a17df18b10aab95b4e890c8cecaba3fc3ee9a0cfc70352be32f9b55ea2c37bd5", + "0a742fc41b0999511084ea42f5070ceee19be3b542d81936c89e75b6800231a0", + ) + profiles = {profile.name: profile for profile in validation.available_schema_profiles()} assert set(profiles) == set(expected) diff --git a/tests/interchange/ossie/test_semantic_validation.py b/tests/interchange/ossie/test_semantic_validation.py index cb249fbf7..3b7b04862 100644 --- a/tests/interchange/ossie/test_semantic_validation.py +++ b/tests/interchange/ossie/test_semantic_validation.py @@ -668,7 +668,7 @@ def test_explicit_consumer_profile_carries_into_semantic_diagnostics() -> None: assert result.diagnostics[0].profile is DBT_1_12_0_1_0_ALIAS -def test_ontology_validation_checks_explicit_mapping_references_only() -> None: +def test_ontology_validation_checks_explicit_mapping_references() -> None: ontology = OssieOntologyDocument( canonical_data={ "version": "0.2.0.dev0", @@ -686,11 +686,11 @@ def test_ontology_validation_checks_explicit_mapping_references_only() -> None: "concept_mappings": [ { "concept": "MissingConcept", - "object_mappings": [{"concept": "Order"}], + "object_mappings": [{"concept": "Order", "expression": "1"}], "link_mappings": [ { - "object_mapping": {"concept": "AlsoMissing"}, - "children": [{"object_mapping": {"concept": "Order"}}], + "object_mapping": {"concept": "AlsoMissing", "expression": "1"}, + "children": [{"object_mapping": {"concept": "Order", "expression": "1"}}], } ], } @@ -728,7 +728,7 @@ def test_ontology_embedded_models_are_isolated_logical_scopes() -> None: "ontology_mappings": [ { "semantic_model": _semantic_model( - "orders-scope", + "orders_scope", datasets=[_dataset("orders")], relationships=[ { @@ -744,7 +744,7 @@ def test_ontology_embedded_models_are_isolated_logical_scopes() -> None: }, { "semantic_model": _semantic_model( - "customers-scope", + "customers_scope", datasets=[_dataset("customers")], ), "concept_mappings": [], @@ -754,12 +754,12 @@ def test_ontology_embedded_models_are_isolated_logical_scopes() -> None: result = validate_ossie_semantics(ontology) - assert result.checked_scopes == ("orders-scope", "customers-scope") + assert result.checked_scopes == ("orders_scope", "customers_scope") assert _contracts(result) == [ ( "ossie.semantic.relationship.to_dataset_unknown", "/ontology_mappings/0/semantic_model/relationships/0/to", - "orders-scope", + "orders_scope", ) ] diff --git a/tests/interchange/ossie/test_serialization.py b/tests/interchange/ossie/test_serialization.py index 4f746f511..b209aa879 100644 --- a/tests/interchange/ossie/test_serialization.py +++ b/tests/interchange/ossie/test_serialization.py @@ -85,6 +85,7 @@ def test_cross_serialization_is_canonical_and_does_not_reuse_source() -> None: assert json.loads(result.data) == LOGICAL_DATA assert result.data.endswith(b"\n") assert not result.exact_source_reused + assert [item.code for item in result.diagnostics] == ["ossie.serialization.exact_source_mismatch"] def test_canonical_json_and_yaml_are_deterministic_unicode_safe_and_tag_free() -> None: @@ -196,3 +197,30 @@ def test_dbt_alias_serialization_rejects_missing_or_wrong_context() -> None: assert _diagnostic_codes(missing.value) == ["ossie.schema.profile_context_required"] assert _diagnostic_codes(wrong.value) == ["ossie.schema.profile_context_mismatch"] + + +@pytest.mark.parametrize(("original_value", "current_value"), [(True, 1), (False, 0), (1, True), (1, 1.0)]) +@pytest.mark.parametrize("serialization", ["json", "yaml"]) +def test_exact_source_comparison_preserves_nested_scalar_types(original_value, current_value, serialization): + data = json.loads(json.dumps(LOGICAL_DATA)) + data["semantic_model"][0]["ai_context"] = {"nested": [original_value]} + original = (json.dumps(data) if serialization == "json" else yaml.safe_dump(data)).encode() + data["semantic_model"][0]["ai_context"] = {"nested": [current_value]} + document = OssieLogicalDocument( + canonical_data=data, + serialization=serialization, + source=OssieDocumentSource(original_bytes=original), + ) + result = serialize_ossie_document(document, serialization, exact_source=True) + assert not result.exact_source_reused + restored = json.loads(result.data) if serialization == "json" else yaml.safe_load(result.data) + assert type(restored["semantic_model"][0]["ai_context"]["nested"][0]) is type(current_value) + assert [item.code for item in result.diagnostics] == ["ossie.serialization.exact_source_mismatch"] + + +@pytest.mark.parametrize("source", [None, OssieDocumentSource(identifier="unretained.yaml")]) +def test_exact_source_request_warns_when_bytes_were_not_retained(source): + result = serialize_ossie_document(_logical_document(source=source), "yaml", exact_source=True) + assert not result.exact_source_reused + assert yaml.safe_load(result.data) == LOGICAL_DATA + assert [item.code for item in result.diagnostics] == ["ossie.serialization.exact_source_mismatch"] diff --git a/tests/interchange/ossie/test_synthesis.py b/tests/interchange/ossie/test_synthesis.py index 853381388..02f7bf6b6 100644 --- a/tests/interchange/ossie/test_synthesis.py +++ b/tests/interchange/ossie/test_synthesis.py @@ -78,7 +78,7 @@ def test_synthesis_preserves_types_time_false_keys_and_edge_identity() -> None: assert result.valid data = result.document.to_parsed_data() - semantic_model = data["semantic_model"][0] + semantic_model = data orders = semantic_model["datasets"][0] loaded_at = orders["fields"][2] relationship = semantic_model["relationships"][0] @@ -261,7 +261,7 @@ def test_synthesis_accepts_reordered_unique_key_without_reordering_join_pairs(ke result = synthesize_ossie_document(graph, scope_name="commerce", expression_dialect="ANSI_SQL") assert result.valid, result.diagnostics - relationship = result.document.to_parsed_data()["semantic_model"][0]["relationships"][0] + relationship = result.document.to_parsed_data()["relationships"][0] assert relationship["from_columns"] == ["region", "customer_id"] assert relationship["to_columns"] == ["region", "id"] @@ -314,7 +314,7 @@ def test_synthesis_preserves_quoted_column_case_and_existing_qualifiers() -> Non result = synthesize_ossie_document(graph, scope_name="commerce", expression_dialect="ANSI_SQL") assert result.valid, result.diagnostics - expression = result.document.to_parsed_data()["semantic_model"][0]["metrics"][0]["expression"] + expression = result.document.to_parsed_data()["metrics"][0]["expression"] assert expression["dialects"][0]["expression"] == 'SUM(COALESCE(orders."Amount", 0) * orders."Quantity")' @@ -326,7 +326,7 @@ def test_synthesis_preserves_model_metric_references_in_derived_formulas(express result = synthesize_ossie_document(graph, scope_name="commerce", expression_dialect="ANSI_SQL") assert result.valid, result.diagnostics - metrics = result.document.to_parsed_data()["semantic_model"][0]["metrics"] + metrics = result.document.to_parsed_data()["metrics"] assert metrics[1]["expression"]["dialects"][0]["expression"] == "revenue * 2" @@ -338,7 +338,7 @@ def test_synthesis_retains_model_binding_for_columnless_aggregates(options: dict result = synthesize_ossie_document(graph, scope_name="commerce", expression_dialect="ANSI_SQL") assert result.valid, result.diagnostics - scope = result.document.to_parsed_data()["semantic_model"][0] + scope = result.document.to_parsed_data() assert "orders.__sidemantic_row" in scope["metrics"][0]["expression"]["dialects"][0]["expression"] @@ -397,7 +397,7 @@ def test_synthesis_restores_native_metric_semantics_through_extension(options): assert result.valid, result.diagnostics assert any(d.code == "ossie.synthesis.runtime_extension_required" for d in result.diagnostics) document = result.document.to_parsed_data() - assert all("analytics.orders" not in d["source"] for d in document["semantic_model"][0]["datasets"]) + assert all("analytics.orders" not in d["source"] for d in document["datasets"]) lowered = lower_ossie_document(parse_ossie_document(json.dumps(document).encode()), target_dialect="duckdb") assert lowered.valid, lowered.diagnostics assert lowered.catalog["commerce"].graph.models["orders"].metrics[0] == graph.models["orders"].metrics[0] @@ -614,3 +614,93 @@ def test_core_metric_dialect_provenance_is_not_native_semantics(): refused = synthesize_ossie_document(graph, scope_name="commerce", expression_dialect="ANSI_SQL", portable_only=True) assert not refused.valid assert any("metadata" in item.message for item in refused.diagnostics) + + +@pytest.mark.parametrize("dialect", ["ANSI_SQL", "BIGQUERY", "DATABRICKS", "SNOWFLAKE"]) +def test_portable_field_owner_placeholder_executes_without_native_template_expansion(dialect): + graph = SemanticGraph() + graph.add_model( + Model( + name="orders", + sql="SELECT 10 AS amount", + dimensions=[Dimension(name="amount", type="numeric", sql='COALESCE({model}."amount", 0) * 2')], + ) + ) + result = synthesize_ossie_document(graph, scope_name="commerce", expression_dialect=dialect, portable_only=True) + assert result.valid, result.diagnostics + scope = result.document.to_parsed_data() + expression = scope["datasets"][0]["fields"][0]["expression"]["dialects"][0]["expression"] + # Execute exported SQL directly: reimport alone would mask leaked templates. + layer = SemanticLayer(auto_register=False) + assert layer.adapter.conn.execute(f"SELECT {expression} FROM (SELECT 10 AS amount) AS source").fetchall() == [(20,)] + + +def test_field_owner_placeholder_normalization_preserves_sql_literals(): + graph = SemanticGraph() + graph.add_model( + Model( + name="orders", + sql="SELECT 'value' AS label", + dimensions=[Dimension(name="label", type="categorical", sql="{model}.label || '{model}.literal'")], + ) + ) + result = synthesize_ossie_document(graph, scope_name="commerce", expression_dialect="ANSI_SQL", portable_only=True) + assert result.valid, result.diagnostics + expression = result.document.to_parsed_data()["datasets"][0]["fields"][0]["expression"]["dialects"][0]["expression"] + layer = SemanticLayer(auto_register=False) + assert layer.adapter.conn.execute(f"SELECT {expression} FROM (SELECT 'value' AS label) AS source").fetchall() == [ + ("value{model}.literal",) + ] + + +@pytest.mark.parametrize("datatype", ["Date", "Time", "DateTime", "DateTimeTz"]) +def test_synthesis_preserves_native_non_time_role_for_temporal_datatype(datatype): + graph = SemanticGraph() + graph.add_model( + Model( + name="orders", + sql="SELECT TIMESTAMP '2026-01-02' AS ts", + dimensions=[Dimension(name="ts", type="categorical", logical_data_type=datatype)], + ) + ) + result = synthesize_ossie_document(graph, scope_name="commerce", expression_dialect="ANSI_SQL", portable_only=True) + assert result.valid, result.diagnostics + data = result.document.to_parsed_data() + field = data["datasets"][0]["fields"][0] + assert field["dimension"] == {"is_time": False} + lowered = lower_ossie_document(parse_ossie_document(json.dumps(data).encode()), target_dialect="duckdb") + assert lowered.valid, lowered.diagnostics + restored = lowered.catalog["commerce"].graph.models["orders"].dimensions[0] + assert restored.type == "categorical" + assert restored.logical_data_type == datatype + native = SemanticLayer(auto_register=False) + native.graph = graph + imported = SemanticLayer.from_catalog(lowered.catalog, auto_register=False) + assert imported.query(dimensions=["orders.ts"]).fetchall() == native.query(dimensions=["orders.ts"]).fetchall() + + +def test_explicit_legacy_revision_preserves_enveloped_export_contract(): + from sidemantic.interchange.ossie.profiles import LEGACY_OSSIE_SCHEMA_COMMIT + + result = synthesize_ossie_document( + _graph(), + scope_name="commerce", + expression_dialect="ANSI_SQL", + schema_revision=LEGACY_OSSIE_SCHEMA_COMMIT, + ) + assert result.valid, result.diagnostics + data = result.document.to_parsed_data() + assert data["semantic_model"][0]["name"] == "commerce" + assert "datasets" not in data + lowered = lower_ossie_document(parse_ossie_document(json.dumps(data).encode()), target_dialect="duckdb") + assert lowered.valid, lowered.diagnostics + + +def test_stable_version_export_keeps_enveloped_shape(): + graph = SemanticGraph() + graph.add_model(Model(name="orders", table="orders", dimensions=[Dimension(name="amount", type="numeric")])) + result = synthesize_ossie_document( + graph, scope_name="commerce", expression_dialect="ANSI_SQL", schema_version="0.1.1" + ) + assert result.valid, result.diagnostics + assert result.document.to_parsed_data()["semantic_model"][0]["name"] == "commerce" diff --git a/tests/interchange/ossie/test_temporal_conformance.py b/tests/interchange/ossie/test_temporal_conformance.py index f4ad760ea..04f3506a0 100644 --- a/tests/interchange/ossie/test_temporal_conformance.py +++ b/tests/interchange/ossie/test_temporal_conformance.py @@ -68,8 +68,7 @@ def test_complete_datatype_and_time_role_matrix_survives_lowering_and_synthesis( assert synthesized.valid synthesized_fields = { - field["name"]: field - for field in synthesized.document.to_parsed_data()["semantic_model"][0]["datasets"][0]["fields"] + field["name"]: field for field in synthesized.document.to_parsed_data()["datasets"][0]["fields"] } for name, (_, data_type, declared_is_time) in expected.items(): assert synthesized_fields[name]["datatype"] == data_type diff --git a/tests/interchange/ossie/test_upstream_validator_gate.py b/tests/interchange/ossie/test_upstream_validator_gate.py index 28319f0c0..39335170b 100644 --- a/tests/interchange/ossie/test_upstream_validator_gate.py +++ b/tests/interchange/ossie/test_upstream_validator_gate.py @@ -14,7 +14,11 @@ from sidemantic.core.model import Model from sidemantic.core.semantic_graph import SemanticGraph from sidemantic.interchange.ossie import ( + CURRENT_OSSIE_SCHEMA_COMMIT, + LEGACY_OSSIE_SCHEMA_COMMIT, + OssieParseOptions, OssieSerialization, + parse_ossie_document, require_synthesized_document, serialize_ossie_document, synthesize_ossie_document, @@ -96,6 +100,7 @@ def test_canonical_sidemantic_export_passes_exact_pinned_apache_validator( scope_name="commerce", expression_dialect="ANSI_SQL", schema_version=schema_version, + schema_revision=LEGACY_OSSIE_SCHEMA_COMMIT if schema_version == "0.2.0.dev0" else None, serialization=serialization, ) document = require_synthesized_document(synthesis) @@ -129,3 +134,67 @@ def test_gate_uses_local_validator_and_schema_paths_only(tmp_path: Path) -> None assert completed.returncode == 0, completed.stdout + completed.stderr assert str(_VALIDATOR) not in completed.stdout assert yaml.safe_load(output.read_text())["version"] == "0.1.1" + + +def _run_current_validator(output: Path, *, ontology: bool = False) -> subprocess.CompletedProcess[str]: + root = _UPSTREAM_ROOT / CURRENT_OSSIE_SCHEMA_COMMIT[:7] + schema = root / ("ontology/ontology.json" if ontology else "core-spec/ossie-schema.json") + return subprocess.run( + [sys.executable, str(root / "validation/validate.py"), str(output), "--schema", str(schema)], + cwd=root, + capture_output=True, + text=True, + check=False, + ) + + +@pytest.mark.parametrize("serialization", [OssieSerialization.JSON, OssieSerialization.YAML]) +def test_current_flat_export_passes_current_pinned_apache_validator(tmp_path, serialization): + synthesis = synthesize_ossie_document( + _graph(include_datatypes=True), scope_name="commerce", expression_dialect="ANSI_SQL" + ) + document = require_synthesized_document(synthesis) + assert "semantic_model" not in document.to_parsed_data() + output = tmp_path / f"current.{serialization.value}" + output.write_bytes(serialize_ossie_document(document, serialization).data) + completed = _run_current_validator(output) + assert completed.returncode == 0, completed.stdout + completed.stderr + + +def test_current_ontology_source_passes_current_offline_validator(tmp_path): + raw = json.dumps( + { + "version": "0.2.0.dev0", + "name": "business", + "prefixes": {"biz": "https://example.org/"}, + "ontology": [{"concept": "Person", "type": "EntityType", "iri": "biz:Person"}], + "ontology_mappings": [ + { + "semantic_model": { + "version": "0.2.0.dev0", + "name": "sales", + "datasets": [{"name": "orders", "source": "orders"}], + }, + "concept_mappings": [], + } + ], + } + ).encode() + parsed = parse_ossie_document(raw, options=OssieParseOptions(validate_schema=True)) + assert parsed.valid + output = tmp_path / "ontology.yaml" + output.write_bytes(serialize_ossie_document(parsed.document, "yaml").data) + completed = _run_current_validator(output, ontology=True) + assert completed.returncode == 0, completed.stdout + completed.stderr + + +def test_current_validator_rejects_legacy_wrapper_and_invalid_current_model(tmp_path): + output = tmp_path / "invalid.json" + for document in ( + {"version": "0.2.0.dev0", "semantic_model": []}, + {"version": "0.2.0.dev0", "name": "sales", "datasets": []}, + ): + output.write_text(json.dumps(document)) + completed = _run_current_validator(output) + assert completed.returncode != 0 + assert "Validation FAILED" in completed.stdout diff --git a/tests/ossie-fixtures/upstream/README.md b/tests/ossie-fixtures/upstream/README.md index aa30c7a22..0cabf4124 100644 --- a/tests/ossie-fixtures/upstream/README.md +++ b/tests/ossie-fixtures/upstream/README.md @@ -26,3 +26,9 @@ compatibility alias is intentionally not covered by this gate. The gate invokes the vendored script as a subprocess with only these local validator, schema, and generated-output paths. It checks both successful canonical Sidemantic exports and a deliberately invalid input. + +`b6c702e/` retains an additional unmodified validator and core/ontology schemas +from Apache Ossie commit `b6c702ed1c07e91382a69e870c875cbd19570828`. +Its flat-root document contract and complete embedded ontology models are tested +separately from the earlier array-envelope snapshot. Its local directory layout +matches upstream so ontology references resolve to its own pinned core schema. diff --git a/tests/ossie-fixtures/upstream/b6c702e/core-spec/ossie-schema.json b/tests/ossie-fixtures/upstream/b6c702e/core-spec/ossie-schema.json new file mode 100644 index 000000000..a8d2fcbdc --- /dev/null +++ b/tests/ossie-fixtures/upstream/b6c702e/core-spec/ossie-schema.json @@ -0,0 +1,374 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/apache/ossie/core-spec/ossie-schema.json", + "title": "Apache Ossie Core Metadata Specification", + "description": "JSON Schema for validating a single Apache Ossie semantic model document", + "type": "object", + "properties": { + "version": { + "type": "string", + "const": "0.2.0.dev0", + "description": "Apache Ossie specification version" + }, + "name": { + "$ref": "#/$defs/SemanticModel/properties/name" + }, + "description": { + "$ref": "#/$defs/SemanticModel/properties/description" + }, + "ai_context": { + "$ref": "#/$defs/SemanticModel/properties/ai_context" + }, + "datasets": { + "$ref": "#/$defs/SemanticModel/properties/datasets" + }, + "relationships": { + "$ref": "#/$defs/SemanticModel/properties/relationships" + }, + "metrics": { + "$ref": "#/$defs/SemanticModel/properties/metrics" + }, + "custom_extensions": { + "$ref": "#/$defs/SemanticModel/properties/custom_extensions" + } + }, + "required": ["version", "name", "datasets"], + "additionalProperties": false, + "$defs": { + "Dialect": { + "type": "string", + "enum": ["ANSI_SQL", "SNOWFLAKE", "MDX", "TABLEAU", "DATABRICKS", "MAQL", "BIGQUERY", "SIGMA", "THOUGHTSPOT", "DAX", "OSSIE_SQL_2026"], + "description": "Supported SQL and expression language dialects" + }, + "Vendor": { + "type": "string", + "examples": ["COMMON", "SNOWFLAKE", "SALESFORCE", "DBT", "DATABRICKS", "GOODDATA", "WISDOM", "POWER_BI"], + "description": "Vendor name for custom extensions. Any string value is accepted." + }, + "AIContext": { + "description": "Additional context for AI tools", + "oneOf": [ + { + "type": "string" + }, + { + "type": "object", + "properties": { + "instructions": { + "type": "string", + "description": "Instructions for AI on how to use this entity" + }, + "synonyms": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Alternative names and terms" + }, + "examples": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Sample questions or use cases" + } + }, + "additionalProperties": true + } + ] + }, + "CustomExtension": { + "type": "object", + "description": "Vendor-specific attributes for extensibility", + "properties": { + "vendor_name": { + "$ref": "#/$defs/Vendor" + }, + "data": { + "type": "string", + "description": "JSON string containing vendor-specific data" + } + }, + "required": ["vendor_name", "data"], + "additionalProperties": false + }, + "DialectExpression": { + "type": "object", + "description": "Expression in a specific dialect", + "properties": { + "dialect": { + "$ref": "#/$defs/Dialect" + }, + "expression": { + "type": "string", + "description": "SQL or dialect-specific expression" + } + }, + "required": ["dialect", "expression"], + "additionalProperties": false + }, + "Expression": { + "type": "object", + "description": "Expression definition with multi-dialect support", + "properties": { + "dialects": { + "type": "array", + "items": { + "$ref": "#/$defs/DialectExpression" + }, + "minItems": 1 + } + }, + "required": ["dialects"], + "additionalProperties": false + }, + "DataType": { + "type": "string", + "enum": [ + "String", + "Integer", + "Decimal", + "Float", + "Boolean", + "Date", + "Time", + "DateTime", + "DateTimeTz", + "Opaque" + ], + "description": "Logical data type for fields and metrics, independent of role (e.g. dimension vs fact) and physical representation. `Decimal` is exact base-10 with unspecified precision and scale; `Float` is approximate. `DateTime` has no timezone or offset, while `DateTimeTz` identifies an instant using offset or timezone context but does not guarantee preservation of a named timezone. Omit `datatype` when unknown; use `Opaque` plus `custom_extensions` for a known type outside the portable vocabulary." + }, + "Dimension": { + "type": "object", + "description": "Dimension metadata", + "properties": { + "is_time": { + "type": "boolean", + "description": "Temporal-role marker. When true, consumers that distinguish time dimensions (e.g. for time-series analysis or temporal filtering) should treat this field as a time dimension. This is a *role* flag, independent of the field's data type: a field with `is_time: true` may carry any `datatype` (e.g. `Integer` for a year grain, `String` for a month name, as well as temporal data types). When `is_time` is unset, it defaults to `true` if `datatype` is one of `Date`, `Time`, `DateTime`, or `DateTimeTz`, and `false` otherwise. Set `is_time: false` explicitly to opt a temporal-typed column (such as an audit timestamp) out of time-dimension treatment." + } + }, + "additionalProperties": false + }, + "Field": { + "type": "object", + "description": "Row-level attribute for grouping, filtering, and metric expressions", + "properties": { + "name": { + "type": "string", + "minLength": 1, + "description": "Unique identifier for the field within the dataset" + }, + "expression": { + "$ref": "#/$defs/Expression" + }, + "dimension": { + "$ref": "#/$defs/Dimension" + }, + "label": { + "type": "string", + "description": "Label for categorization" + }, + "description": { + "type": "string", + "description": "Human-readable description" + }, + "datatype": { + "$ref": "#/$defs/DataType" + }, + "ai_context": { + "$ref": "#/$defs/AIContext" + }, + "custom_extensions": { + "type": "array", + "items": { + "$ref": "#/$defs/CustomExtension" + } + } + }, + "required": ["name", "expression"], + "additionalProperties": false + }, + "Dataset": { + "type": "object", + "description": "Logical dataset representing a business entity (fact or dimension table)", + "properties": { + "name": { + "type": "string", + "minLength": 1, + "description": "Unique identifier for the dataset" + }, + "source": { + "type": "string", + "minLength": 1, + "description": "Reference to underlying physical table/view (database.schema.table) or query" + }, + "primary_key": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Primary key columns (single or composite)" + }, + "unique_keys": { + "type": "array", + "items": { + "type": "array", + "items": { + "type": "string" + } + }, + "description": "Array of unique key definitions (each can be single or composite)" + }, + "description": { + "type": "string", + "description": "Human-readable description" + }, + "ai_context": { + "$ref": "#/$defs/AIContext" + }, + "fields": { + "type": "array", + "items": { + "$ref": "#/$defs/Field" + } + }, + "custom_extensions": { + "type": "array", + "items": { + "$ref": "#/$defs/CustomExtension" + } + } + }, + "required": ["name", "source"], + "additionalProperties": false + }, + "Relationship": { + "type": "object", + "description": "Foreign key relationship between datasets", + "properties": { + "name": { + "type": "string", + "minLength": 1, + "description": "Unique identifier for the relationship" + }, + "from": { + "type": "string", + "minLength": 1, + "description": "Dataset on the many side of the relationship" + }, + "to": { + "type": "string", + "minLength": 1, + "description": "Dataset on the one side of the relationship" + }, + "from_columns": { + "type": "array", + "items": { + "type": "string" + }, + "minItems": 1, + "description": "Foreign key columns in the 'from' dataset" + }, + "to_columns": { + "type": "array", + "items": { + "type": "string" + }, + "minItems": 1, + "description": "Primary/unique key columns in the 'to' dataset" + }, + "ai_context": { + "$ref": "#/$defs/AIContext" + }, + "custom_extensions": { + "type": "array", + "items": { + "$ref": "#/$defs/CustomExtension" + } + } + }, + "required": ["name", "from", "to", "from_columns", "to_columns"], + "additionalProperties": false + }, + "Metric": { + "type": "object", + "description": "Quantitative measure defined on business data", + "properties": { + "name": { + "type": "string", + "minLength": 1, + "description": "Unique identifier for the metric" + }, + "expression": { + "$ref": "#/$defs/Expression" + }, + "description": { + "type": "string", + "description": "Human-readable description of what the metric measures" + }, + "datatype": { + "$ref": "#/$defs/DataType" + }, + "ai_context": { + "$ref": "#/$defs/AIContext" + }, + "custom_extensions": { + "type": "array", + "items": { + "$ref": "#/$defs/CustomExtension" + } + } + }, + "required": ["name", "expression"], + "additionalProperties": false + }, + "SemanticModel": { + "type": "object", + "description": "Top-level container representing a complete semantic model", + "properties": { + "name": { + "type": "string", + "minLength": 1, + "description": "Unique identifier for the semantic model" + }, + "description": { + "type": "string", + "description": "Human-readable description" + }, + "ai_context": { + "$ref": "#/$defs/AIContext" + }, + "datasets": { + "type": "array", + "items": { + "$ref": "#/$defs/Dataset" + }, + "minItems": 1, + "description": "Collection of logical datasets" + }, + "relationships": { + "type": "array", + "items": { + "$ref": "#/$defs/Relationship" + }, + "description": "Defines how datasets are connected" + }, + "metrics": { + "type": "array", + "items": { + "$ref": "#/$defs/Metric" + }, + "description": "Quantifiable measures spanning datasets" + }, + "custom_extensions": { + "type": "array", + "items": { + "$ref": "#/$defs/CustomExtension" + } + } + }, + "required": ["name", "datasets"], + "additionalProperties": false + } + } +} diff --git a/tests/ossie-fixtures/upstream/b6c702e/ontology/ontology.json b/tests/ossie-fixtures/upstream/b6c702e/ontology/ontology.json new file mode 100644 index 000000000..68e16b838 --- /dev/null +++ b/tests/ossie-fixtures/upstream/b6c702e/ontology/ontology.json @@ -0,0 +1,314 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/apache/ossie/ontology/ontology.json", + "title": "Apache Ossie Ontology Metadata Specification", + "description": "JSON Schema for validating Apache Ossie ontology definitions", + "type": "object", + "properties": { + "version": { + "type": "string", + "const": "0.2.0.dev0", + "description": "Ontology specification version" + }, + "name": { + "type": "string", + "description": "Unique identifier for the ontology" + }, + "description": { + "type": "string", + "description": "Human-readable description" + }, + "ai_context": { + "$ref": "https://raw.githubusercontent.com/apache/ossie/main/core-spec/ossie-schema.json#/$defs/AIContext" + }, + "requires": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that constrain the population of this ontology" + }, + "ontology": { + "type": "array", + "items": { + "$ref": "#/$defs/OntologyComponent" + }, + "minItems": 1, + "description": "Components that define the concepts and relationships in this ontology" + }, + "ontology_mappings": { + "type": "array", + "description": "Collection of ontology maps from logical models", + "items": { + "$ref": "#/$defs/OntologyMap" + } + }, + "prefixes": { + "type": "object", + "description": "Maps namespace prefixes to the IRIs they abbreviate, enabling QName expansion (e.g., 'foaf' -> 'http://xmlns.com/foaf/0.1/').", + "additionalProperties": { + "type": "string", + "format": "iri" + } + } + }, + "required": ["version", "name", "ontology"], + "additionalProperties": false, + "$defs": { + "OntologyComponent": { + "type": "object", + "description": "Ontology component that defines a single concept and any relationships that are keyed primarily by that concept", + "properties": { + "concept": { + "type": "string", + "description": "Unique name of the concept defined by this component" + }, + "type": { + "$ref": "#/$defs/ConceptType" + }, + "description": { + "type": "string", + "description": "Human-readable description of the concept" + }, + "extends": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Indicates that this concept extends one or more other concepts" + }, + "derived_by": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that define how this concept is derived" + }, + "identify_by": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Names of relationships to use as the preferred identifier of this concept" + }, + "requires": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that constrain the population of this concept" + }, + "relationships": { + "type": "array", + "items": { + "$ref": "#/$defs/Relationship" + }, + "description": "Defines relationships that pertain primarily to the concept defined in this component" + }, + "iri": { + "type": "string", + "description": "Optional global identifier for this concept, expressed as a full IRI or as a QName (prefix:local) resolved against the ontology-level 'prefixes' map." + } + }, + "required": ["concept", "type"], + "additionalProperties": false + }, + "Expression": { + "type": "string", + "description": "ANSI SQL expression" + }, + "Relationship": { + "type": "object", + "description": "Relationship between concepts in the ontology", + "properties": { + "name": { + "type": "string", + "description": "Name of the relationship" + }, + "description": { + "type": "string", + "description": "Human-readable description of the relationship" + }, + "roles": { + "type": "array", + "items": { + "$ref": "#/$defs/Role" + }, + "description": "Additional roles in this relationship" + }, + "multiplicity": { + "$ref": "#/$defs/Multiplicity" + }, + "derived_by": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that define how this concept is derived" + }, + "requires": { + "type": "array", + "items": { + "$ref": "#/$defs/Expression" + }, + "description": "Expressions that constrain the population of this relationship" + }, + "verbalizes": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Natural language expressions that verbalize this relationship" + }, + "iri": { + "type": "string", + "description": "Optional global identifier for this relationship, expressed as a full IRI or as a QName (prefix:local) resolved against the ontology-level 'prefixes' map." + } + }, + "required": ["name", "verbalizes"], + "additionalProperties": false + }, + "ConceptMapping": { + "type": "object", + "description": "Mappings from logical model constructs to some ontology component", + "properties": { + "concept": { + "type": "string", + "description": "Name of the concept whose part of the ontology we are mapping to" + }, + "object_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/ObjectMapping" + }, + "description": "Mappings from logical constructs that populate the concept in this component. Valid only when the concept is an entity type" + }, + "link_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/LinkMapping" + }, + "description": "Mappings from logical model relationships to ontology relationships pertaining to the mapped concept" + } + }, + "required": ["concept"], + "additionalProperties": false + }, + "ConceptType": { + "type": "string", + "enum": [ "EntityType", "ValueType" ], + "description": "A concept is either an entity type or a value type" + }, + "ReferentMapping": { + "type": "object", + "description": "Mapping from logical model constructs to a relationship used to references some entity type in the ontology", + "properties": { + "relationship": { + "type": "string", + "description": "Name of referent relationship" + }, + "expression": { + "$ref": "#/$defs/Expression" + }, + "referent_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/ReferentMapping" + } + } + }, + "required": ["relationship"], + "additionalProperties": false + }, + "Role": { + "type": "object", + "description": "Role in some relationship (the container)", + "properties": { + "concept": { + "type": "string", + "description": "Name of the concept playing this role" + }, + "name": { + "type": "string", + "description": "Optional name of this role, used when the same concept plays multiple roles in the same relationship" + } + }, + "required": ["concept"], + "additionalProperties": false + }, + "ObjectMapping": { + "type": "object", + "description": "Pattern of logical-level expressions for identifying objects of some concept using the values in one or more fields", + "properties": { + "concept": { + "type": "string", + "description": "Name of the concept whose objects we are mapping to" + }, + "referent_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/ReferentMapping" + }, + "description": "Maps logical-model constructs to referent relationships of this entity type" + }, + "expression": { + "$ref": "#/$defs/Expression" + } + }, + "additionalProperties": false + }, + "LinkMapping": { + "type": "object", + "description": "Mapping from logical schema to the links of relationships in the ontology", + "properties": { + "relationship": { + "type": "string", + "description": "Name of relationship being populated by this mapping node" + }, + "object_mapping": { + "$ref": "#/$defs/ObjectMapping" + }, + "children": { + "type": "array", + "items": { + "$ref": "#/$defs/LinkMapping" + }, + "description": "Relationship maps at the next level in this hierarchy" + } + }, + "required": ["object_mapping"], + "additionalProperties": false + }, + "Multiplicity": { + "type": "string", + "enum": [ "ManyToOne", "OneToOne" ], + "description": "Relationship multiplicity" + }, + "OntologyMap": { + "type": "object", + "description": "Map from the constructs of some logical model to some ontology", + "properties": { + "name": { + "type": "string", + "description": "Name of this ontology map" + }, + "description": { + "type": "string", + "description": "Human-readable description of this ontology map" + }, + "semantic_model": { + "$ref": "https://raw.githubusercontent.com/apache/ossie/main/core-spec/ossie-schema.json" + }, + "concept_mappings": { + "type": "array", + "items": { + "$ref": "#/$defs/ConceptMapping" + }, + "description": "Maps logical model constructs to some concept and its relationships in the ontology" + } + }, + "required": ["semantic_model", "concept_mappings"], + "additionalProperties": false + } + } +} diff --git a/tests/ossie-fixtures/upstream/b6c702e/validation/validate.py b/tests/ossie-fixtures/upstream/b6c702e/validation/validate.py new file mode 100644 index 000000000..b51982d91 --- /dev/null +++ b/tests/ossie-fixtures/upstream/b6c702e/validation/validate.py @@ -0,0 +1,452 @@ +#!/usr/bin/env python3 +# +# /// script +# requires-python = ">=3.11" +# dependencies = [ +# "jsonschema>=4.26.0", +# "pyyaml>=6.0.3", +# "sqlglot>=30.12.0", +# ] +# /// + +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +""" +Ossie Semantic Model Validator + +Validates Ossie YAML or JSON documents containing one semantic model against: +1. JSON Schema (structure, types, enums) +2. Unique names (datasets, fields, metrics, relationships) +3. Valid relationship references +4. Relationship column arity (from_columns and to_columns lengths match) +5. SQL syntax (using sqlglot) + +Usage: + python validation/validate.py + python validation/validate.py --schema ontology/ontology.json + python validation/validate.py examples/tpcds_semantic_model.yaml +""" + +import json +import sys +from collections.abc import Hashable +from pathlib import Path + +try: + import yaml + from jsonschema import Draft202012Validator + from referencing import Registry, Resource + from referencing.exceptions import Unresolvable + from yaml.constructor import ConstructorError +except ImportError: + print("Missing dependencies. Install with:") + print(" pip install pyyaml jsonschema") + sys.exit(1) + +try: + import sqlglot + from sqlglot.errors import ParseError, TokenError + SQLGLOT_AVAILABLE = True +except ImportError: + SQLGLOT_AVAILABLE = False + +# Map Ossie dialects to sqlglot dialects +DIALECT_MAP = { + "ANSI_SQL": None, # sqlglot default + "OSSIE_SQL_2026": None, # ANSI-SQL-compatible; parse with the sqlglot default + "SNOWFLAKE": "snowflake", + "DATABRICKS": "databricks", + "BIGQUERY": "bigquery", + "MDX": None, # Not supported by sqlglot, skip validation + "TABLEAU": None, # Not supported by sqlglot, skip validation + "MAQL": None, # Not supported by sqlglot, skip validation + "SIGMA": None, # Sigma's spreadsheet-style formula language, not SQL; skip validation + "THOUGHTSPOT": None, # Not supported by sqlglot, skip validation + "DAX": None, # Not supported by sqlglot, skip validation +} + +# Dialects that sqlglot cannot parse +SKIP_SQL_VALIDATION = {"MDX", "TABLEAU", "MAQL", "SIGMA", "THOUGHTSPOT", "DAX"} + + +class ValidationWarning(str): + """An explicitly nonfatal diagnostic, compatible with existing string callers.""" + + +class UniqueKeyLoader(yaml.SafeLoader): + """Safe YAML loader that rejects duplicate explicit mapping keys.""" + + MERGE_TAG = "tag:yaml.org,2002:merge" + + def construct_document(self, node: yaml.Node): + # Validate the composed node graph before SafeConstructor touches it: + # flatten_mapping() rewrites mapping nodes in place while expanding "<<", + # so a check that runs during construction sees merged, not authored, keys. + self._check_unique_keys(node, set()) + return super().construct_document(node) + + def _check_unique_keys(self, node: yaml.Node, visited: set) -> None: + if id(node) in visited: # alias or recursive anchor: check the node once + return + visited.add(id(node)) + + if isinstance(node, yaml.MappingNode): + seen = set() + merge_key = object() + for key_node, value_node in node.value: + if key_node.tag == self.MERGE_TAG: + key, display_key = merge_key, "<<" + elif isinstance(key_node, yaml.ScalarNode): + key = display_key = self.construct_object(key_node, deep=True) + else: + # Collection keys are unhashable under SafeLoader. Reject them + # here rather than constructing them, which would run + # flatten_mapping() on the graph this walk must not disturb. + raise ConstructorError( + "while constructing a mapping", + node.start_mark, + "found an unhashable key", + key_node.start_mark, + ) + + if not isinstance(key, Hashable): + raise ConstructorError( + "while constructing a mapping", + node.start_mark, + "found an unhashable key", + key_node.start_mark, + ) + + if key in seen: + raise ConstructorError( + "while constructing a mapping", + node.start_mark, + f"found duplicate key {display_key!r}", + key_node.start_mark, + ) + seen.add(key) + + self._check_unique_keys(key_node, visited) + self._check_unique_keys(value_node, visited) + elif isinstance(node, yaml.SequenceNode): + for child in node.value: + self._check_unique_keys(child, visited) + + +def validate_schema(data: dict, schema: dict) -> list[str]: + """Validate against JSON Schema, resolving core references locally.""" + core_path = Path(__file__).parent.parent / "core-spec" / "ossie-schema.json" + core = json.loads(core_path.read_text()) + resource = Resource.from_contents(core) + # Ontology references use the raw URL; also register the canonical schema ID. + registry = Registry().with_resources([ + (core["$id"], resource), + ( + "https://raw.githubusercontent.com/apache/ossie/main/core-spec/ossie-schema.json", + resource, + ), + ]) + validator = Draft202012Validator(schema, registry=registry) + errors = [] + try: + for error in validator.iter_errors(data): + path = " -> ".join(str(p) for p in error.absolute_path) if error.absolute_path else "(root)" + errors.append(f"[Schema] {path}: {error.message}") + except Unresolvable as error: + errors.append(f"[Schema] Cannot resolve schema reference: {error.ref}") + return errors + + +def find_duplicates(items: list[str]) -> list[str]: + """Find duplicate items in a list.""" + seen = set() + duplicates = [] + for item in items: + if item in seen: + duplicates.append(item) + seen.add(item) + return duplicates + + +def validate_unique_names(data: dict) -> list[str]: + """Validate unique names for datasets, fields, metrics, relationships.""" + if not isinstance(data, dict) or "datasets" not in data: + return [] + + model = data + errors = [] + + model_name = model.get("name", "") + + # Check unique dataset names + dataset_names = [d.get("name") for d in model.get("datasets", []) if d.get("name")] + for dup in find_duplicates(dataset_names): + errors.append(f"[Unique] Duplicate dataset name '{dup}' in model '{model_name}'") + + # Check unique field names within each dataset + for dataset in model.get("datasets", []): + dataset_name = dataset.get("name", "") + field_names = [f.get("name") for f in dataset.get("fields", []) if f.get("name")] + for dup in find_duplicates(field_names): + errors.append(f"[Unique] Duplicate field name '{dup}' in dataset '{dataset_name}'") + + # Check unique metric names + metric_names = [m.get("name") for m in model.get("metrics", []) if m.get("name")] + for dup in find_duplicates(metric_names): + errors.append(f"[Unique] Duplicate metric name '{dup}' in model '{model_name}'") + + # Check unique relationship names + rel_names = [r.get("name") for r in model.get("relationships", []) if r.get("name")] + for dup in find_duplicates(rel_names): + errors.append(f"[Unique] Duplicate relationship name '{dup}' in model '{model_name}'") + + return errors + + +def validate_references(data: dict) -> list[str]: + """Validate that relationships reference existing datasets and that + to_columns covers a declared key of the 'to' dataset.""" + if not isinstance(data, dict) or "datasets" not in data: + return [] + + model = data + errors = [] + + model_name = model.get("name", "") + datasets = {d.get("name"): d for d in model.get("datasets", []) if d.get("name")} + + for rel in model.get("relationships", []): + rel_name = rel.get("name", "") + from_ds = rel.get("from") + to_ds = rel.get("to") + + if from_ds and from_ds not in datasets: + errors.append(f"[Reference] Relationship '{rel_name}' in model '{model_name}' references unknown dataset '{from_ds}'") + if to_ds and to_ds not in datasets: + errors.append(f"[Reference] Relationship '{rel_name}' in model '{model_name}' references unknown dataset '{to_ds}'") + + # The spec defines to_columns as "Primary/unique key columns in the + # 'to' dataset". Coverage (superset of a key) still guarantees the + # many-to-one join, and declared keys may be incomplete since + # primary_key and unique_keys are optional — so accept any + # to_columns that covers a declared key, report a warning rather + # than an error, and skip datasets that declare no keys. + # Shape guards keep semantic checks from crashing on documents + # that already fail schema validation. + dataset = datasets.get(to_ds) + to_columns = rel.get("to_columns") + if dataset and isinstance(to_columns, list) and to_columns: + candidate_keys = [dataset.get("primary_key")] + list(dataset.get("unique_keys") or []) + declared_keys = [k for k in candidate_keys if isinstance(k, list) and k] + to_column_set = set(to_columns) + if declared_keys and not any(set(key) <= to_column_set for key in declared_keys): + errors.append(ValidationWarning( + f"[Reference] Warning: Relationship '{rel_name}' in model '{model_name}': to_columns {to_columns} does not cover the primary key or a unique key of dataset '{to_ds}'" + )) + + return errors + + +def validate_relationship_column_arity(data: dict) -> list[str]: + """Validate that from_columns and to_columns have the same length. + + The spec requires the two arrays to correspond positionally, so their + lengths must match. JSON Schema cannot express this, so it is checked here. + """ + if not isinstance(data, dict) or "datasets" not in data: + return [] + + model = data + errors = [] + + model_name = model.get("name", "") + + for rel in model.get("relationships", []): + rel_name = rel.get("name", "") + from_columns = rel.get("from_columns") + to_columns = rel.get("to_columns") + + # Skip anything that already failed schema validation. + if not isinstance(from_columns, list) or not isinstance(to_columns, list): + continue + + if len(from_columns) != len(to_columns): + errors.append( + f"[Arity] Relationship '{rel_name}' in model '{model_name}': " + f"from_columns ({len(from_columns)}) and " + f"to_columns ({len(to_columns)}) must have the same number of columns" + ) + + return errors + + +def validate_sql_expression(expr: str, dialect: str, context: str) -> str | None: + """Validate a single SQL expression. Returns error message or None if valid.""" + if not SQLGLOT_AVAILABLE: + return None + + if dialect in SKIP_SQL_VALIDATION: + return None + + sqlglot_dialect = DIALECT_MAP.get(dialect) + + try: + # Try parsing as expression first (for field expressions like "column_name") + sqlglot.parse_one(expr, dialect=sqlglot_dialect) + return None + except (ParseError, TokenError, RecursionError): + # A bare column reference fails to parse alone; retry it wrapped in + # SELECT below. RecursionError (deeply nested input) is included so the + # retry reports it instead of crashing, while genuine errors such as a + # non-string expr raising TypeError still surface. + pass + + try: + # Try wrapping in SELECT for simple column references + sqlglot.parse_one(f"SELECT {expr}", dialect=sqlglot_dialect) + return None + except (ParseError, TokenError) as e: + return f"[SQL] {context}: {str(e).split(chr(10))[0]}" + except RecursionError: + # Deeply nested input exhausts the recursion limit rather than raising a + # parser error; report it instead of letting it abort validation. + return f"[SQL] {context}: expression is too deeply nested to parse" + + +def validate_sql(data: dict) -> list[str]: + """Validate SQL expressions in fields and metrics.""" + # Only core semantic model documents contain a root datasets property. + if not isinstance(data, dict) or "datasets" not in data: + return [] + + if not SQLGLOT_AVAILABLE: + return [ValidationWarning( + "[SQL] Warning: sqlglot not installed, skipping SQL validation. Install with: pip install sqlglot" + )] + + model = data + errors = [] + + model_name = model.get("name", "") + + # Validate field expressions + for dataset in model.get("datasets", []): + dataset_name = dataset.get("name", "") + for field in dataset.get("fields", []): + field_name = field.get("name", "") + expression = field.get("expression", {}) + for dialect_expr in expression.get("dialects", []): + dialect = dialect_expr.get("dialect", "ANSI_SQL") + expr = dialect_expr.get("expression", "") + if expr: + context = f"Field '{dataset_name}.{field_name}' in model '{model_name}' ({dialect})" + error = validate_sql_expression(expr, dialect, context) + if error: + errors.append(error) + + # Validate metric expressions + for metric in model.get("metrics", []): + metric_name = metric.get("name", "") + expression = metric.get("expression", {}) + for dialect_expr in expression.get("dialects", []): + dialect = dialect_expr.get("dialect", "ANSI_SQL") + expr = dialect_expr.get("expression", "") + if expr: + context = f"Metric '{metric_name}' in model '{model_name}' ({dialect})" + error = validate_sql_expression(expr, dialect, context) + if error: + errors.append(error) + + return errors + + +def main(): + if len(sys.argv) < 2: + print(__doc__) + sys.exit(1) + + args = sys.argv[1:] + yaml_path = Path(args[0]) + + schema_path = Path(__file__).parent.parent / "core-spec" / "ossie-schema.json" + if len(args) > 1: + if len(args) == 3 and args[1] == "--schema": + schema_path = Path(args[2]) + else: + print("Usage: python validation/validate.py [--schema ]") + sys.exit(1) + + if not yaml_path.exists(): + print(f"Error: File not found: {yaml_path}") + sys.exit(1) + + if not schema_path.exists(): + print(f"Error: Schema not found: {schema_path}") + sys.exit(1) + + # Load files + with open(schema_path) as f: + schema = json.load(f) + + with open(yaml_path) as f: + try: + data = yaml.load(f, Loader=UniqueKeyLoader) + except yaml.YAMLError as e: + print(f"Error: Invalid YAML: {e}") + sys.exit(1) + except RecursionError: + # Deeply nested input surfaces as RecursionError, not YAMLError. + print("Error: Invalid YAML: input is too deeply nested to parse") + sys.exit(1) + + # Run validations + errors = [] + errors.extend(validate_schema(data, schema)) + + # Semantic checks rely on valid structure; let schema validation report + # malformed inputs (including legacy arrays) without traversing them. + if not errors and isinstance(data, dict) and "datasets" in data: + errors.extend(validate_unique_names(data)) + errors.extend(validate_references(data)) + errors.extend(validate_relationship_column_arity(data)) + errors.extend(validate_sql(data)) + + # Report results + if errors: + # Severity must not depend on user-controlled text in a diagnostic. + warnings = [e for e in errors if isinstance(e, ValidationWarning)] + actual_errors = [e for e in errors if not isinstance(e, ValidationWarning)] + + for warning in warnings: + print(f" {warning}") + + if actual_errors: + print(f"\nValidation FAILED with {len(actual_errors)} error(s):\n") + for error in actual_errors: + print(f" {error}") + sys.exit(1) + else: + print(f"Validation PASSED: {yaml_path.name}") + sys.exit(0) + else: + print(f"Validation PASSED: {yaml_path.name}") + sys.exit(0) + + +if __name__ == "__main__": + main() diff --git a/tests/test_cli_contract.py b/tests/test_cli_contract.py index 1af89ccd8..4c8391af1 100644 --- a/tests/test_cli_contract.py +++ b/tests/test_cli_contract.py @@ -614,7 +614,7 @@ def test_convert_native_metric_extension_roundtrip_and_portable_refusal(tmp_path with pytest.warns(UserWarning, match="requires Sidemantic runtime extension"): converted = runner.invoke(app, arguments) assert converted.exit_code == 0, converted.output - assert json.loads(output.read_text())["semantic_model"][0]["custom_extensions"] + assert json.loads(output.read_text())["custom_extensions"] reimported = runner.invoke( app, ["convert", str(output), "--from", target_format, "--to", "sidemantic", "--output", str(restored)], @@ -675,7 +675,8 @@ def test_convert_to_dbt_alias_carries_explicit_consumer_profile(tmp_path: Path): assert json.loads(output.read_text())["version"] == "0.1.0" -def test_convert_from_ossie_selects_scope_and_target_dialect(tmp_path: Path): +@pytest.mark.parametrize("source_format", ["ossie", "auto"]) +def test_convert_from_ossie_selects_scope_and_target_dialect(tmp_path: Path, source_format: str): source = tmp_path / "source.ossie.yaml" output = tmp_path / "output.yml" source.write_text( @@ -689,6 +690,12 @@ def test_convert_from_ossie_selects_scope_and_target_dialect(tmp_path: Path): datasets: - name: orders source: marketing.orders + fields: + - name: integer_amount + expression: + dialects: + - {dialect: ANSI_SQL, expression: 'CAST(amount AS BIGINT)'} + - {dialect: BIGQUERY, expression: 'SAFE_CAST(amount AS INT64)'} """ ) @@ -698,7 +705,7 @@ def test_convert_from_ossie_selects_scope_and_target_dialect(tmp_path: Path): "convert", str(source), "--from", - "ossie", + source_format, "--to", "sidemantic", "--output", @@ -706,12 +713,66 @@ def test_convert_from_ossie_selects_scope_and_target_dialect(tmp_path: Path): "--ossie-scope", "marketing", "--ossie-target-dialect", - "duckdb", + "bigquery", ], ) assert result.exit_code == 0, result.output assert "marketing.orders" in output.read_text() + assert "INT64" in output.read_text() + + +@pytest.mark.parametrize("extension", [".yaml", ".json"]) +def test_convert_auto_current_ossie_does_not_produce_empty_graph(tmp_path: Path, extension: str): + source = tmp_path / f"current{extension}" + source.write_text( + json.dumps( + { + "version": "0.2.0.dev0", + "name": "commerce", + "datasets": [{"name": "orders", "source": "analytics.orders"}], + } + ) + ) + output = tmp_path / "native.yml" + + result = runner.invoke(app, ["convert", str(source), "--output", str(output)]) + + assert result.exit_code == 0, result.output + assert "analytics.orders" in output.read_text() + + +def test_convert_ossie_can_export_pinned_legacy_draft(tmp_path: Path): + source = tmp_path / "native.yml" + source.write_text("""models: + - name: orders + table: analytics.orders + primary_key: id + dimensions: + - {name: id, type: numeric} +""") + output = tmp_path / "legacy.json" + + result = runner.invoke( + app, + [ + "convert", + str(source), + "--to", + "ossie", + "--output", + str(output), + "--ossie-scope", + "commerce", + "--ossie-expression-dialect", + "ANSI_SQL", + "--ossie-schema-revision", + "831f48e582731cf1ee2e65380ca5abf8157869c7", + ], + ) + + assert result.exit_code == 0, result.output + assert json.loads(output.read_text())["semantic_model"][0]["name"] == "commerce" def test_info_and_validate_can_select_a_multi_scope_ossie_document(tmp_path: Path): diff --git a/tests/test_formats.py b/tests/test_formats.py index faf8d808d..eb31c1731 100644 --- a/tests/test_formats.py +++ b/tests/test_formats.py @@ -1,6 +1,8 @@ +import json from pathlib import Path import pytest +import yaml from sidemantic.formats import ( OutputKind, @@ -107,6 +109,70 @@ def test_explicit_ossie_format_uses_scoped_validated_importer(tmp_path: Path): assert graph.get_model("orders").table == "analytics.orders" +@pytest.mark.parametrize("extension", [".yaml", ".json", ".ossie.yaml", ".ossie.json"]) +def test_auto_loads_current_flat_ossie_file(tmp_path: Path, extension: str): + source = tmp_path / f"current{extension}" + data = { + "version": "0.2.0.dev0", + "name": "commerce", + "datasets": [{"name": "orders", "source": "analytics.orders"}], + } + source.write_text(json.dumps(data) if extension.endswith("json") else yaml.safe_dump(data)) + (tmp_path / "sibling.yml").write_text(_native_model("sibling")) + + graph = load_semantic_source(source) + + assert set(graph.models) == {"orders"} + assert graph.get_model("orders").table == "analytics.orders" + + +@pytest.mark.parametrize("extension", [".ossie.json", ".ossie.yaml", ".ossie.yml"]) +def test_auto_load_rejects_malformed_explicit_ossie_file(tmp_path: Path, extension: str): + source = tmp_path / f"invalid{extension}" + source.write_text("{}") + + with pytest.raises(ValueError, match="invalid"): + load_semantic_source(source) + + +@pytest.mark.parametrize("directory", [False, True]) +def test_auto_ossie_options_select_scope_and_execution_dialect(tmp_path: Path, directory: bool): + source = tmp_path / "model.ossie.yaml" + source.write_text("""version: 0.2.0.dev0 +semantic_model: + - name: finance + datasets: + - {name: orders, source: finance.orders} + - name: marketing + datasets: + - name: orders + source: marketing.orders + fields: + - name: integer_amount + expression: + dialects: + - {dialect: ANSI_SQL, expression: 'CAST(amount AS BIGINT)'} + - {dialect: BIGQUERY, expression: 'SAFE_CAST(amount AS INT64)'} +""") + + graph = load_semantic_source( + tmp_path if directory else source, + adapter_options={"scope_id": "marketing", "target_dialect": "bigquery"}, + ) + + orders = graph.get_model("orders") + assert orders.table == "marketing.orders" + assert "INT64" in orders.dimensions[0].sql + + +def test_auto_rejects_options_for_unknown_adapter(tmp_path: Path): + source = tmp_path / "native.yml" + source.write_text(_native_model("orders")) + + with pytest.raises(ValueError, match="explicit source_format"): + load_semantic_source(source, adapter_options={"unknown": True}) + + def test_ossie_graph_export_requires_and_accepts_explicit_synthesis_options(tmp_path: Path): source = tmp_path / "source.yml" output = tmp_path / "output.json" From 539d067670156e2a0085965966d973002318d174 Mon Sep 17 00:00:00 2001 From: Nico Ritschel Date: Fri, 2 Oct 2026 06:28:49 -0700 Subject: [PATCH 2/3] Fix Ossie discovery across source profiles --- sidemantic/loaders.py | 63 ++++++++++------------------ tests/core/test_directory_loaders.py | 46 ++++++++++++++++++-- 2 files changed, 66 insertions(+), 43 deletions(-) diff --git a/sidemantic/loaders.py b/sidemantic/loaders.py index 872f2c38c..7c4e86ee8 100644 --- a/sidemantic/loaders.py +++ b/sidemantic/loaders.py @@ -441,7 +441,14 @@ def ossie_adapter(consumer_profile: str = "ossie-core") -> OssieAdapter: # validated parser, so they cannot silently become an empty graph. if _is_generated_artifact(file_path, directory): continue - adapter = ossie_adapter() + try: + # JSON is also valid YAML; this probe only selects the consumer. + # The Ossie parser still owns format and schema validation. + ossie_data = _load_yaml_mapping(file_path.read_text()) + except yaml.YAMLError: + ossie_data = {} + consumer_profile = "dbt-1.12" if ossie_data.get("version") == "0.1.0" else "ossie-core" + adapter = ossie_adapter(consumer_profile) elif suffix == ".json": content = file_path.read_text() if '"ldm"' in content and '"datasets"' in content: @@ -455,29 +462,25 @@ def ossie_adapter(consumer_profile: str = "ossie-core") -> OssieAdapter: elif ( '"datasets"' in content and ('"semantic_model"' in content or ('"version"' in content and '"name"' in content)) - and (only_file is not None or _is_under_osi_tree(file_path, directory)) and not _is_generated_artifact(file_path, directory) ): - # Released-spec OSI profile (dbt OSI consumer) ships as JSON in an - # OSI/ directory at the project root. Mirror the YAML detection - # (semantic_model + datasets), but only inside that OSI/ tree: - # dbt's OSI consumer scans only ``/OSI/``, so an - # archived or scratch OSI .json elsewhere under the project must - # not add stale models or collide with the real sources. - # Skip dbt-generated copies (e.g. target/osi_document.json) so a - # `dbt compile` artifact never shadows the real OSI/ sources. + import json + + legacy_location = only_file is not None or _is_under_osi_tree(file_path, directory) try: - is_osi = _looks_like_osi_json(content) - except ValueError as e: - # The file textually looks like OSI (semantic_model + datasets) - # but is malformed JSON. Surface it as a parse error instead of - # silently skipping, mirroring the malformed-YAML handling above. - _handle_parse_error(file_path, e, strict=strict) + ossie_data = json.loads(content) + except ValueError as exc: + # Keep archived legacy envelopes outside OSI/ ignored, but + # surface malformed current sources just as YAML does. + if legacy_location or '"semantic_model"' not in content: + _handle_parse_error(file_path, exc, strict=strict) continue - if is_osi: - import json - - consumer_profile = "dbt-1.12" if json.loads(content).get("version") == "0.1.0" else "ossie-core" + if _looks_like_ossie_mapping(ossie_data) and ( + legacy_location or {"version", "name", "datasets"}.issubset(ossie_data) + ): + # Only released/legacy envelopes follow dbt's OSI/ source + # restriction. Current flat documents have no such layout. + consumer_profile = "dbt-1.12" if ossie_data.get("version") == "0.1.0" else "ossie-core" adapter = ossie_adapter(consumer_profile) else: import json @@ -845,26 +848,6 @@ def _looks_like_ossie_mapping(data: object) -> bool: }.issubset(data) -def _looks_like_osi_json(content: str) -> bool: - """Return True for a released or current Ossie JSON document. - - Released OSI ships as JSON with a top-level ``semantic_model`` list whose - entries contain ``datasets``. This mirrors the YAML OSI detection and avoids - routing unrelated JSON (e.g. GoodData) to the OSI adapter. - - Raises ``ValueError`` when ``content`` is not valid JSON so callers that have - already confirmed the OSI text markers can surface a parse error instead of - silently skipping a malformed OSI document. - """ - import json - - try: - data = json.loads(content) - except json.JSONDecodeError as e: - raise ValueError(f"Invalid JSON: {e}") from e - return _looks_like_ossie_mapping(data) - - # Directories that hold generated/compiled artifacts rather than source models. # dbt writes a copy of the OSI document to ``target/`` on ``dbt compile``; routing # those to the OSI adapter would resurrect deleted or stale models, so skip them. diff --git a/tests/core/test_directory_loaders.py b/tests/core/test_directory_loaders.py index 0914ac8b4..97d9026b2 100644 --- a/tests/core/test_directory_loaders.py +++ b/tests/core/test_directory_loaders.py @@ -517,11 +517,51 @@ def test_load_from_directory_detects_explicit_ossie_json_outside_osi_tree(tmp_pa assert layer.graph.models["orders"]._source_format == "Ossie" +@pytest.mark.parametrize("suffix", ["yaml", "yml", "json"]) +def test_load_from_directory_detects_dbt_profile_for_explicit_ossie_files(tmp_path, suffix): + import json + + import yaml + + document = { + "version": "0.1.0", + "semantic_model": [{"name": "analytics", "datasets": [{"name": "orders", "source": "analytics.orders"}]}], + } + content = json.dumps(document) if suffix == "json" else yaml.safe_dump(document) + (tmp_path / f"orders.ossie.{suffix}").write_text(content) + + layer = SemanticLayer() + load_from_directory(layer, tmp_path) + + assert list(layer.graph.models) == ["orders"] + assert layer.graph.models["orders"].table == "analytics.orders" + + +@pytest.mark.parametrize("relative_path", ["commerce.json", "models/commerce.json"]) +def test_load_from_directory_detects_current_flat_ossie_json(tmp_path, relative_path): + from sidemantic.validation_runner import validate_directory + + source = tmp_path / relative_path + source.parent.mkdir(parents=True, exist_ok=True) + source.write_text( + '{"version": "0.2.0.dev0", "name": "analytics", "datasets": [{"name": "orders", "source": "analytics.orders"}]}' + ) + + layer = SemanticLayer() + load_from_directory(layer, tmp_path) + + assert list(layer.graph.models) == ["orders"] + assert layer.graph.models["orders"].table == "analytics.orders" + assert layer.graph.models["orders"]._source_format == "Ossie" + assert validate_directory(tmp_path).passed + + +@pytest.mark.parametrize("suffix", ["yaml", "yml", "json"]) @pytest.mark.parametrize("content", ['{"version":', "{}"]) -def test_load_from_directory_surfaces_invalid_explicit_ossie_json(tmp_path, content): - (tmp_path / "invalid.ossie.json").write_text(content) +def test_load_from_directory_surfaces_invalid_explicit_ossie_files(tmp_path, content, suffix): + (tmp_path / f"invalid.ossie.{suffix}").write_text(content) - with pytest.raises(ValueError, match=r"invalid\.ossie\.json"): + with pytest.raises(ValueError, match=rf"invalid\.ossie\.{suffix}"): load_from_directory(SemanticLayer(), tmp_path) layer = SemanticLayer() From 1a7681aee01d26c35881a791571d932ebc52ee04 Mon Sep 17 00:00:00 2001 From: Nico Ritschel Date: Fri, 2 Oct 2026 06:42:54 -0700 Subject: [PATCH 3/3] Refresh man page for Ossie revision option --- sidemantic/man/sidemantic.1 | 3 +++ 1 file changed, 3 insertions(+) diff --git a/sidemantic/man/sidemantic.1 b/sidemantic/man/sidemantic.1 index fd0bdf747..9834aa92f 100644 --- a/sidemantic/man/sidemantic.1 +++ b/sidemantic/man/sidemantic.1 @@ -202,6 +202,9 @@ Exact dialect label for SQL synthesized into an Ossie output \fB\-\-ossie\-schema\-version TEXT\fR Explicit pinned Ossie schema version for synthesized output .TP +\fB\-\-ossie\-schema\-revision TEXT\fR +Pinned Ossie draft revision for synthesized output (commit SHA) +.TP \fB\-\-ossie\-consumer\-profile TEXT\fR Ossie import or export consumer profile (ossie\-core or dbt\-1.12) .TP