From ee589ee4f6f80963d2bb592a6a0cc5ed4e3a6c08 Mon Sep 17 00:00:00 2001 From: Ralfo Becher Date: Sat, 26 Sep 2026 23:52:05 +0200 Subject: [PATCH] fix(orionbelt): export metric expressions that compute what OrionBelt computes Other Ossie consumers read a metric's SQL, not the ORIONBELT extension. That SQL referenced the physical table instead of the dataset, dropped measure filters and totals, left {[Name]} references to counts and metrics unresolved, and exported period-over-period as prev.value. Export (new _portable.py renderer): - reference ""."", always double-quoted so reserved words parse - filters as AGG(CASE WHEN ... END), totals as the grand-total window SUM(SUM(x)) OVER () (exact SUM/COUNT ratio for AVG), defaultValue as COALESCE, count_distinct as COUNT(DISTINCT ...) - inline synthesized counts and metric-on-metric references - cumulative and window metrics order and partition by each dimension at its timeGrain, cast back to a temporal resultType, as the query groups it - leave out period-over-period, grain/filterContext/anchor measures, nested windows and reference cycles with a warning, kept whole in the model extension (obml_unexported) - tag aggregation: measure as DATABRICKS - carry each column's OBML name and each measure/metric's OBML definition Import: - restore definitions, left-out entities and column names, so OBML -> Ossie -> OBML returns the model it was given instead of re-parsing filters and totals out of the SQL New tests run every exported metric in DuckDB, grouped as OrionBelt groups it (skipped when duckdb or sqlglot is not installed). --- converters/orionbelt/README.md | 19 + .../orionbelt/ossie_obml_mapping_analysis.md | 20 +- .../src/ossie_orionbelt/_portable.py | 445 +++++++++++ .../src/ossie_orionbelt/obml_to_ossie.py | 692 ++++-------------- .../src/ossie_orionbelt/ossie_to_obml.py | 74 ++ .../tests/test_ossie_converter_cumulative.py | 3 + .../tests/test_ossie_converter_pop.py | 172 ++--- .../tests/test_ossie_converter_properties.py | 17 +- ...st_ossie_converter_roundtrip_robustness.py | 8 +- .../tests/test_ossie_portable_expressions.py | 415 +++++++++++ .../orionbelt/tests/test_ossie_v02_compat.py | 11 +- 11 files changed, 1205 insertions(+), 671 deletions(-) create mode 100644 converters/orionbelt/src/ossie_orionbelt/_portable.py create mode 100644 converters/orionbelt/tests/test_ossie_portable_expressions.py diff --git a/converters/orionbelt/README.md b/converters/orionbelt/README.md index fb947c95..daefed37 100644 --- a/converters/orionbelt/README.md +++ b/converters/orionbelt/README.md @@ -92,6 +92,25 @@ assert result.valid obml_again = OssietoOBML(ossie).convert() ``` +## Metric expressions + +Other Ossie consumers read a metric's SQL, not the `ORIONBELT` extension, so the +OBML to Ossie export writes SQL that computes what OrionBelt computes: + +- columns are referenced as `"".""`, always double-quoted; +- measure filters become `AGG(CASE WHEN THEN END)`, totals the + grand-total window `SUM(SUM(x)) OVER ()`, and `defaultValue` a `COALESCE`; +- synthesized counts and metric-on-metric `{[Name]}` references are inlined; +- cumulative and window metrics order and partition by each dimension at its + `timeGrain`, as the query groups it. + +A measure or metric with no faithful single expression (period-over-period, +`grain`, `filterContext`, `anchor`, nested windows, reference cycles) is left out +of the Ossie metrics with a warning and kept whole in the model-level extension +(`obml_unexported`). Every exported measure and metric also carries its OBML +definition, and every field its OBML column name, so Ossie to OBML restores the +original model instead of re-parsing the SQL. + ## Vendor extensions Ossie `custom_extensions` carry vendor-tagged payloads. This converter: diff --git a/converters/orionbelt/ossie_obml_mapping_analysis.md b/converters/orionbelt/ossie_obml_mapping_analysis.md index 4fd90d3c..d4829c01 100644 --- a/converters/orionbelt/ossie_obml_mapping_analysis.md +++ b/converters/orionbelt/ossie_obml_mapping_analysis.md @@ -46,7 +46,7 @@ - **Ossie** uses snake_case codes everywhere (`name: "store_sales"`) - **OBML** supports dual naming — a display name as the dictionary key and a `code` for the physical SQL reference -During Ossie → OBML conversion, field names are used directly as both the display name and code. During OBML → Ossie conversion, the `code` value becomes the Ossie field `name`. +During Ossie → OBML conversion, field names are used directly as both the display name and code. During OBML → Ossie conversion, the `code` value becomes the Ossie field `name`, and the OBML column name rides in the field's `custom_extensions` (`obml_column_name`) so the reverse trip restores it. Metric expressions reference columns as `"".""`: the OBML data object name and the column code, always double-quoted so reserved words parse and names match exactly. Cumulative and window metrics order and partition by each dimension exactly as the query groups it: truncated to its `timeGrain` (`DATE_TRUNC`) and cast back to a temporal `resultType`. A reference cycle among metrics is left out with a warning. ### 2.2 Relationship Placement @@ -136,8 +136,9 @@ These OBML features have no direct Ossie equivalent. Where possible, metadata is - Dynamic date filters (`dynamicDate`, `dynamicDateRange`) — not yet preserved - `timeGrain` on dimensions — preserved in field `custom_extensions` (`obml_time_grain`) - Dimension `format` — preserved in field `custom_extensions` (`obml_dimension_format`) -- Measure filters — preserved in metric `custom_extensions` (`obml_filters`) -- Measure `total` — preserved in metric `custom_extensions` (`obml_total`) +- Measure filters — written into the expression as `AGG(CASE WHEN THEN END)`, and preserved in metric `custom_extensions` (`obml_filters`) +- Measure `total` — written into the expression as the grand-total window OrionBelt computes (`SUM(SUM(x)) OVER ()`; exact `SUM/COUNT` ratio for `AVG`), and preserved in metric `custom_extensions` (`obml_total`) +- Measure `grain`, `filterContext` and `anchor`, period-over-period metrics, and cumulative or window metrics over a window (a `total` measure or another cumulative or window metric; window calls cannot nest): these depend on the query, so they have no faithful single expression. They are left out of the Ossie metrics with a warning and kept whole in the model-level `custom_extensions` (`obml_unexported`); the reverse trip restores them - Measure `format` — preserved in metric `custom_extensions` (`obml_format`) - Measure `delimiter` — preserved in metric `custom_extensions` (`obml_delimiter`) - Measure `withinGroup` — preserved in metric `custom_extensions` (`obml_within_group`) @@ -149,7 +150,7 @@ These OBML features have no direct Ossie equivalent. Where possible, metadata is - **`primary_key`** — natively represented: Ossie's dataset-level `primary_key` array maps to per-column `primaryKey: true` on OBML columns (`DataObjectColumn.primaryKey`), and back to the dataset array on export. - **`unique_keys`** — no native OBML equivalent; round-trips via an `Ossie`-vendor `customExtension` (`obml_unique_keys`). -- **Multi-dialect expressions** — on import the converter reads the first available SQL dialect in the order `ANSI_SQL`, `OSSIE_SQL_2026`, `SNOWFLAKE`, `DATABRICKS`; non-SQL dialects (`MDX`, `TABLEAU`, `MAQL`, `SIGMA`, `THOUGHTSPOT`, `DAX`) are not parsed. A metric with no SQL-parseable dialect, or an expression OBML cannot decompose, is preserved verbatim (`obml_unconverted_metrics`) with a `LOSSY:` warning rather than dropped. On export, OBML measures/metrics emit `ANSI_SQL`. +- **Multi-dialect expressions** — on import the converter reads the first available SQL dialect in the order `ANSI_SQL`, `OSSIE_SQL_2026`, `SNOWFLAKE`, `DATABRICKS`; non-SQL dialects (`MDX`, `TABLEAU`, `MAQL`, `SIGMA`, `THOUGHTSPOT`, `DAX`) are not parsed. A metric with no SQL-parseable dialect, or an expression OBML cannot decompose, is preserved verbatim (`obml_unconverted_metrics`) with a `LOSSY:` warning rather than dropped. On export, OBML measures/metrics emit `ANSI_SQL`, except `aggregation: measure`, whose `MEASURE("")` call is tagged `DATABRICKS`. - **`ai_context`** — preserved losslessly via `customExtensions` (see Section 2.4). - **`custom_extensions`** — mapped to OBML `customExtensions`. @@ -169,11 +170,12 @@ These OBML features have no direct Ossie equivalent. Where possible, metadata is 1. Combine `database.schema.code` into the Ossie `source` string 2. Convert columns to fields with `ANSI_SQL` dialect expressions 3. Extract inline joins into global relationships with generated names -4. Convert measures to Ossie metrics with SQL expressions -5. Expand metric templates by substituting measure SQL into `{[Name]}` references -6. Map OBML dimension metadata into `field.dimension.is_time` flags -7. Preserve secondary join info in relationship `ai_context` -8. Store OBML-specific type info in `custom_extensions` with `vendor_name: "COMMON"` +4. Convert measures to Ossie metrics with SQL expressions that compute what OrionBelt computes (filters, totals, defaults spelled out) +5. Expand metric templates by substituting measure, synthesized-count and metric SQL into `{[Name]}` references +6. Leave out what has no faithful expression, with a warning; each exported measure or metric also carries its full OBML definition (`obml_definition`), which the reverse trip restores instead of re-parsing the SQL +7. Map OBML dimension metadata into `field.dimension.is_time` flags +8. Preserve secondary join info in relationship `ai_context` +9. Store OBML-specific type info in `custom_extensions` with `vendor_name: "COMMON"` ## 4. Validation diff --git a/converters/orionbelt/src/ossie_orionbelt/_portable.py b/converters/orionbelt/src/ossie_orionbelt/_portable.py new file mode 100644 index 00000000..2a54c641 --- /dev/null +++ b/converters/orionbelt/src/ossie_orionbelt/_portable.py @@ -0,0 +1,445 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +"""Portable Apache Ossie SQL for OBML measures and metrics. + +An exported metric expression is what every other Ossie consumer reads, so it +has to compute what OrionBelt computes. OBML carries semantics in structure +(measure ``filters``, ``total``, synthesized counts, metric-on-metric +references) that the expression must spell out: + +* column references are ``.``: the OBML data object name and + the column's physical code, which are the Ossie dataset and field names; +* measure ``filters`` become ``AGG(CASE WHEN THEN END)``, the + spec's portable filtered aggregation, mirroring the compiler's own rendering; +* ``total: true`` becomes the grand-total window the compiler emits, e.g. + ``SUM(SUM(x)) OVER ()``; +* ``{[Name]}`` references resolve to the referenced measure, synthesized count + or metric, inlined. + +What has no faithful single-expression form raises :class:`NotPortableError` so the +caller can leave it out of the Ossie document instead of guessing. +""" + +from __future__ import annotations + +import re +from typing import Any + +_MEASURE_REF = re.compile(r"\{\[([^\]]+)\]\}") +_COLUMN_REF = re.compile(r"\{\[([^\]]+)\]\.\[([^\]]+)\]\}") + +_DEFAULT_COUNT_PATTERN = "{object} Count" + +# Measure options whose value depends on the query (its grouping or filters), +# which a standalone Ossie metric expression cannot see. +_QUERY_DEPENDENT_OPTIONS = { + "grain": "a grain override is evaluated relative to the query's dimensions", + "filterContext": "a filter context changes which query filters apply", + "anchor": "an anchored expression spans independent facts", +} + +# How the compiler re-aggregates a ``total: true`` measure over the grouped +# result (``compiler/total_wrap.py``); anything not listed re-aggregates by SUM. +_TOTAL_REAGG = {"min": "MIN", "max": "MAX"} + +# A time-grain dimension declaring one of these ``resultType``s is cast back to +# it after truncation, as the compiler does (``compiler/resolution.py``), so the +# query groups by the cast expression. +_TEMPORAL_CASTS = {"date": "DATE", "timestamp": "TIMESTAMP", "time": "TIME"} + +_CUMULATIVE_FUNCS = {"sum", "avg", "min", "max", "count"} +_GRAIN_TO_DATE = {"year", "quarter", "month", "week"} +_WINDOW_FUNCS = { + "rank", + "dense_rank", + "row_number", + "ntile", + "lag", + "lead", + "first_value", + "last_value", +} + +_COMPARISONS = { + "equals": "=", + "notequals": "<>", + "gt": ">", + "gte": ">=", + "lt": "<", + "lte": "<=", +} + + +class NotPortableError(Exception): + """An OBML measure or metric has no faithful Apache Ossie expression.""" + + +def sql_ident(name: str) -> str: + """Render *name* as a double-quoted identifier. + + Always quoted: a plain-looking name can still be a reserved word (a data + object called ``Order``), and a quoted identifier matches the Ossie dataset + or field name exactly, case included. + """ + return '"' + name.replace('"', '""') + '"' + + +def _string_literal(value: str) -> str: + return "'" + value.replace("'", "''") + "'" + + +def _scalar_literal(value: Any) -> str: + if value is None: + return "NULL" + if isinstance(value, bool): + return "TRUE" if value else "FALSE" + if isinstance(value, int | float): + return str(value) + return _string_literal(str(value)) + + +def _filter_literal(fv: dict[str, Any]) -> str: + """Render a typed OBML ``FilterValue`` as the literal the compiler compares against.""" + if fv.get("isNull"): + return "NULL" + data_type = fv.get("dataType") + if data_type == "int": + return _scalar_literal(fv.get("valueInt")) + if data_type == "float": + return _scalar_literal(fv.get("valueFloat")) + if data_type == "boolean": + return _scalar_literal(fv.get("valueBoolean")) + if data_type == "date" and fv.get("valueDate") is not None: + return f"DATE {_string_literal(str(fv['valueDate']))}" + if data_type == "timestamp" and fv.get("valueDate") is not None: + return f"TIMESTAMP {_string_literal(str(fv['valueDate']))}" + return _scalar_literal(fv.get("valueString")) + + +def _filter_text(fv: dict[str, Any]) -> str: + """The raw text of a filter value, for LIKE patterns.""" + for key in ("valueString", "valueInt", "valueFloat", "valueDate", "valueBoolean"): + if fv.get(key) is not None: + return str(fv[key]) + return "" + + +def _like(col: str, pattern_parts: tuple[str, str], value: str, negated: bool) -> str: + """``col [NOT] LIKE`` with *value* matched literally between the wildcards.""" + escaped = value.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_") + op = "NOT LIKE" if negated else "LIKE" + pattern = _string_literal(pattern_parts[0] + escaped + pattern_parts[1]) + escape = " ESCAPE '\\'" if escaped != value else "" + return f"{col} {op} {pattern}{escape}" + + +def synthesized_counts(obml: dict[str, Any]) -> dict[str, str]: + """Map each synthesized count measure's name to its data object. + + Mirrors ``orionbelt.models.synthesis``: ``exposeCounts`` and per-object + ``countable`` opt out, nested objects never get one, the name resolves + ``countLabel`` > ``countLabelPattern`` > ``"{object} Count"``, and a + declared measure of the same name wins. + """ + if obml.get("exposeCounts") is False: + return {} + pattern = obml.get("countLabelPattern") or _DEFAULT_COUNT_PATTERN + declared = obml.get("measures") or {} + counts: dict[str, str] = {} + for key, obj in (obml.get("dataObjects") or {}).items(): + if obj.get("countable") is False or obj.get("nestedIn"): + continue + name = (obj.get("countLabel") or pattern).replace("{object}", key) + if name not in declared and name not in counts: + counts[name] = key + return counts + + +class PortableRenderer: + """Render OBML measures and metrics as portable Ossie SQL expressions.""" + + def __init__(self, obml: dict[str, Any]) -> None: + self.data_objects: dict[str, Any] = obml.get("dataObjects") or {} + self.dimensions: dict[str, Any] = obml.get("dimensions") or {} + self.measures: dict[str, Any] = obml.get("measures") or {} + self.metrics: dict[str, Any] = obml.get("metrics") or {} + self.counts = synthesized_counts(obml) + self._resolving: set[str] = set() + + # ── references ──────────────────────────────────────────────────── + + def column(self, data_object: str, column: str) -> str: + """``.`` for an OBML column.""" + obj = self.data_objects.get(data_object) + if obj is None: + raise NotPortableError(f"references unknown data object '{data_object}'") + col = (obj.get("columns") or {}).get(column, {}) + code = col.get("code", column.lower().replace(" ", "_")) + return f"{sql_ident(data_object)}.{sql_ident(code)}" + + def _column_ref(self, ref: dict[str, Any]) -> str: + return self.column(ref.get("dataObject", ""), ref.get("column", "")) + + def _dimension(self, name: str) -> str: + """The dimension exactly as the query groups by it. + + A ``timeGrain`` truncates the column, and a temporal ``resultType`` casts + the truncation back to that type, as the compiler renders it; ordering + by anything else is not a grouped expression. + """ + dim = self.dimensions.get(name) + if dim is None: + raise NotPortableError(f"references unknown dimension '{name}'") + column = self.column(dim.get("dataObject", ""), dim.get("column", "")) + grain = dim.get("timeGrain") + if not grain: + return column + truncated = f"DATE_TRUNC({_string_literal(grain)}, {column})" + cast = _TEMPORAL_CASTS.get(str(dim.get("resultType", "")).lower()) + return f"CAST({truncated} AS {cast})" if cast else truncated + + def _expression(self, template: str) -> str: + return _COLUMN_REF.sub(lambda m: self.column(m.group(1), m.group(2)), template) + + # ── measures ────────────────────────────────────────────────────── + + def measure(self, name: str) -> str: + """The expression for a declared or synthesized measure.""" + if name in self.measures: + return self._declared_measure(name, self.measures[name]) + if name in self.counts: + return self._count(self.counts[name]) + raise NotPortableError(f"references unknown measure '{name}'") + + def _count(self, data_object: str) -> str: + obj = self.data_objects[data_object] + pk = [c for c, col in (obj.get("columns") or {}).items() if col.get("primaryKey")] + if len(pk) != 1: + raise NotPortableError( + f"the row count of '{data_object}' needs a single-column primary key " + f"to anchor COUNT on" + ) + return f"COUNT({self.column(data_object, pk[0])})" + + def _declared_measure(self, name: str, m: dict[str, Any]) -> str: + for option, why in _QUERY_DEPENDENT_OPTIONS.items(): + if m.get(option): + raise NotPortableError(f"measure '{name}' uses '{option}': {why}") + agg = str(m.get("aggregation", "sum")).lower() + if agg == "measure": + raise NotPortableError(f"measure '{name}' is resolved by a Databricks Metric View") + + if m.get("columns"): + args = [self._column_ref(c) for c in m["columns"]] + elif m.get("expression"): + args = [self._expression(m["expression"])] + else: + raise NotPortableError(f"measure '{name}' has no columns or expression") + + if m.get("filters"): + condition = self._filters(m["filters"]) + args = [f"CASE WHEN {condition} THEN {a} END" for a in args] + + distinct = "DISTINCT " if m.get("distinct") or agg == "count_distinct" else "" + func = "COUNT" if agg == "count_distinct" else agg.upper() + if agg == "listagg" and m.get("delimiter") is not None: + args.append(_string_literal(str(m["delimiter"]))) + within = "" + if m.get("withinGroup"): + wg = m["withinGroup"] + direction = "DESC" if str(wg.get("order", "ASC")).upper() == "DESC" else "ASC" + within = f" WITHIN GROUP (ORDER BY {self._column_ref(wg['column'])} {direction})" + sql = f"{func}({distinct}{', '.join(args)}){within}" + + if m.get("total"): + if agg == "avg": + arg = args[0] + sql = f"(SUM(SUM({arg})) OVER () / SUM(COUNT({arg})) OVER ())" + else: + sql = f"{_TOTAL_REAGG.get(agg, 'SUM')}({sql}) OVER ()" + if m.get("defaultValue") is not None: + sql = f"COALESCE({sql}, {_scalar_literal(m['defaultValue'])})" + return sql + + # ── measure filters (mirrors compiler/filters.py) ───────────────── + + def _filters(self, items: list[Any]) -> str: + return " AND ".join(self._filter_item(i, top=True) for i in items) + + def _filter_item(self, item: dict[str, Any], top: bool = False) -> str: + if "filters" in item: + logic = " OR " if str(item.get("logic", "and")).lower() == "or" else " AND " + inner = logic.join(self._filter_item(i) for i in item["filters"]) + if not inner: + raise NotPortableError("an empty measure filter group") + return f"NOT ({inner})" if item.get("negated") else f"({inner})" + leaf = self._filter_leaf(item) + return leaf if top else f"({leaf})" + + def _filter_leaf(self, f: dict[str, Any]) -> str: + col = self._column_ref(f.get("column") or {}) + op = str(f.get("operator", "")).lower() + values = f.get("values") or [] + first = _filter_literal(values[0]) if values else "NULL" + if op in _COMPARISONS: + return f"{col} {_COMPARISONS[op]} {first}" + if op in ("inlist", "notinlist"): + keyword = "NOT IN" if op == "notinlist" else "IN" + return f"{col} {keyword} ({', '.join(_filter_literal(v) for v in values)})" + if op == "set": + return f"{col} IS NOT NULL" + if op == "notset": + return f"{col} IS NULL" + text = _filter_text(values[0]) if values else "" + if op in ("contains", "notcontains"): + return _like(col, ("%", "%"), text, negated=op == "notcontains") + if op == "starts_with": + return _like(col, ("", "%"), text, negated=False) + if op == "ends_with": + return _like(col, ("%", ""), text, negated=False) + if op in ("like", "notlike"): + keyword = "NOT LIKE" if op == "notlike" else "LIKE" + return f"{col} {keyword} {_string_literal(text)}" + if op in ("between", "notbetween"): + if len(values) < 2: + return f"{col} {'<>' if op == 'notbetween' else '='} {first}" + keyword = "NOT BETWEEN" if op == "notbetween" else "BETWEEN" + return f"{col} {keyword} {first} AND {_filter_literal(values[1])}" + raise NotPortableError(f"measure filter operator '{op}' has no portable form") + + # ── metrics ─────────────────────────────────────────────────────── + + def metric(self, name: str) -> str: + """The expression for an OBML metric.""" + if name in self._resolving: + raise NotPortableError(f"metric '{name}' references itself") + self._resolving.add(name) + try: + return self._metric(name, self.metrics[name]) + finally: + self._resolving.discard(name) + + def _has_window(self, name: str, visiting: frozenset[str] = frozenset()) -> bool: + """Whether the expression for *name* already contains a window call. + + Raises :class:`NotPortableError` on a reference cycle, direct or not. + """ + if name in visiting: + raise NotPortableError(f"metric '{name}' is part of a reference cycle") + if name in self.metrics: + met = self.metrics[name] + if met.get("type") in ("cumulative", "window"): + return True + refs = _MEASURE_REF.findall(met.get("expression") or "") + return any(self._has_window(r, visiting | {name}) for r in refs) + return bool(self.measures.get(name, {}).get("total")) + + def _windowed(self, name: str, ref: str) -> str: + """The SQL for *ref*, which a window of metric *name* aggregates over. + + Window calls cannot nest, so a reference that already carries one (a + grand total, or a cumulative or window metric) needs another query + layer that one expression does not have. + """ + if self._has_window(ref): + raise NotPortableError( + f"metric '{name}' applies a window over '{ref}', which is itself a window; " + f"nested window calls need another query layer" + ) + return self._reference(ref) + + def _reference(self, name: str) -> str: + if name in self.metrics: + return f"({self.metric(name)})" + return self.measure(name) + + def _metric(self, name: str, met: dict[str, Any]) -> str: + kind = met.get("type", "derived") + if kind == "cumulative": + return self._cumulative(name, met) + if kind == "window": + return self._window(name, met) + if kind == "period_over_period": + raise NotPortableError( + f"metric '{name}' compares against a period shifted on a date spine, " + f"which one expression cannot reproduce when periods are missing" + ) + template = met.get("expression") + if not template: + raise NotPortableError(f"metric '{name}' has no expression") + return _MEASURE_REF.sub(lambda m: self._reference(m.group(1)), template) + + def _partitions(self, met: dict[str, Any]) -> list[str]: + return [self._dimension(d) for d in met.get("partitionBy") or []] + + def _cumulative(self, name: str, met: dict[str, Any]) -> str: + func = str(met.get("cumulativeType", "sum")).lower() + if func not in _CUMULATIVE_FUNCS or not met.get("measure"): + raise NotPortableError(f"metric '{name}' is an incomplete cumulative metric") + inner = self._windowed(name, met["measure"]) + time = self._dimension(met.get("timeDimension", "")) + partitions = self._partitions(met) + frame = "ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW" + grain = met.get("grainToDate") + if grain: + if grain not in _GRAIN_TO_DATE: + raise NotPortableError(f"metric '{name}' has unknown grainToDate '{grain}'") + partitions.insert(0, f"DATE_TRUNC({_string_literal(grain)}, {time})") + elif met.get("window") is not None: + frame = f"ROWS BETWEEN {int(met['window']) - 1} PRECEDING AND CURRENT ROW" + partition = f"PARTITION BY {', '.join(partitions)} " if partitions else "" + return f"{func.upper()}({inner}) OVER ({partition}ORDER BY {time} {frame})" + + def _window(self, name: str, met: dict[str, Any]) -> str: + """Mirror ``compiler/window_wrap.py``: ranking orders by the measure, else by time.""" + func = str(met.get("windowFunction", "")).lower() + if func not in _WINDOW_FUNCS: + raise NotPortableError(f"metric '{name}' has unknown windowFunction '{func}'") + direction = "DESC" if str(met.get("orderDirection", "desc")).lower() == "desc" else "ASC" + inner = self._windowed(name, met["measure"]) if met.get("measure") else None + time = self._dimension(met["timeDimension"]) if met.get("timeDimension") else None + + args: list[str] = [] + order: str | None = None + if func in ("lag", "lead"): + if inner is None or time is None: + raise NotPortableError(f"metric '{name}' needs a measure and a timeDimension") + args = [inner, str(int(met.get("offset") or 1))] + if met.get("defaultValue") is not None: + args.append(_scalar_literal(met["defaultValue"])) + order = f"{time} ASC" + elif func in ("first_value", "last_value"): + if inner is None: + raise NotPortableError(f"metric '{name}' needs a measure") + args = [inner] + order = f"{time} {direction}" if time else None + else: + if func == "ntile": + if met.get("buckets") is None: + raise NotPortableError(f"metric '{name}' needs buckets") + args = [str(int(met["buckets"]))] + key = inner or time + order = f"{key} {direction}" if key else None + + clauses = [] + partitions = self._partitions(met) + if partitions: + clauses.append(f"PARTITION BY {', '.join(partitions)}") + if order: + clauses.append(f"ORDER BY {order}") + return f"{func.upper()}({', '.join(args)}) OVER ({' '.join(clauses)})" diff --git a/converters/orionbelt/src/ossie_orionbelt/obml_to_ossie.py b/converters/orionbelt/src/ossie_orionbelt/obml_to_ossie.py index 179acd1f..dd9e5f87 100644 --- a/converters/orionbelt/src/ossie_orionbelt/obml_to_ossie.py +++ b/converters/orionbelt/src/ossie_orionbelt/obml_to_ossie.py @@ -24,7 +24,6 @@ from __future__ import annotations import json -import re from typing import Any from ossie_orionbelt._common import ( @@ -34,6 +33,7 @@ _VENDOR_OBML, OBML_TO_OSSIE_TYPE, ) +from ossie_orionbelt._portable import NotPortableError, PortableRenderer class OBMLtoOssie: @@ -51,11 +51,15 @@ def __init__( self.model_description = model_description self.ai_instructions = ai_instructions self.warnings: list[str] = [] + # Measures and metrics with no faithful Ossie expression, by kind + # ("measures" / "metrics"); they ride in the model-level extension. + self.unexported: dict[str, dict[str, Any]] = {} def convert(self) -> dict: - # Reset warnings so a second convert() call on the same instance does - # not duplicate them. + # Reset per-conversion state so a second convert() call on the same + # instance does not duplicate warnings or left-out entities. self.warnings = [] + self.unexported = {} ossie: dict[str, Any] = {"version": _OSSIE_VERSION} @@ -74,7 +78,7 @@ def convert(self) -> dict: all_relationships.extend(rels) # ── Metrics (OBML measures + metrics → Ossie metrics) ──────── - ossie_metrics = self._convert_measures_and_metrics(obml_measures, obml_metrics, data_objects) + ossie_metrics = self._convert_measures_and_metrics(obml_measures, obml_metrics) # Re-emit Ossie metrics that OBML could not represent and that the import # path preserved verbatim (vendor Ossie, ``obml_unconverted_metrics``). @@ -129,6 +133,8 @@ def convert(self) -> dict: count_label_pattern = self.obml.get("countLabelPattern") if count_label_pattern is not None: roundtrip_data["obml_count_label_pattern"] = count_label_pattern + if self.unexported: + roundtrip_data["obml_unexported"] = self.unexported sem_model["custom_extensions"] = [ { "vendor_name": _VENDOR_OBML, @@ -387,6 +393,11 @@ def _convert_column( "data_type": ossie_type, "obml_abstract_type": abstract_type, } + # The Ossie field name is the physical code; keep the OBML column name + # so the reverse trip restores it (measure filters and model filters + # refer to columns by that name). + if col_name != code: + ext_data["obml_column_name"] = col_name # Preserve OBML-only column properties if col_obj.get("sqlType"): ext_data["obml_sql_type"] = col_obj["sqlType"] @@ -533,43 +544,75 @@ def _merge_restored_metrics(self, ossie_metrics: list[dict]) -> None: existing.add(name) ossie_metrics.append(restored) - def _convert_measures_and_metrics( - self, obml_measures: dict, obml_metrics: dict, data_objects: dict - ) -> list: - """Convert OBML measures and metrics to Ossie metrics.""" + def _convert_measures_and_metrics(self, obml_measures: dict, obml_metrics: dict) -> list: + """Convert OBML measures and metrics to Ossie metrics. + + The expression is what other Ossie consumers read, so it comes from + :class:`PortableRenderer` and computes what OrionBelt computes. A measure + or metric with no faithful expression is left out of the document and + kept whole in ``self.unexported`` for the model-level extension; the + reverse conversion restores it from there. + """ + renderer = PortableRenderer(self.obml) ossie_metrics = [] - # Convert each OBML measure to an Ossie metric - for measure_name, measure_obj in obml_measures.items(): - ossie_metric = self._convert_measure(measure_name, measure_obj, data_objects) - if ossie_metric: - self._carry_foreign_to_ossie_metric(measure_obj, ossie_metric) - ossie_metrics.append(ossie_metric) - - # Convert OBML metrics (which reference measures) to Ossie metrics - for metric_name, metric_obj in obml_metrics.items(): - if metric_obj.get("type") == "cumulative": - ossie_metric = self._convert_obml_cumulative_metric( - metric_name, metric_obj, obml_measures, data_objects - ) - elif metric_obj.get("type") == "period_over_period": - ossie_metric = self._convert_obml_pop_metric( - metric_name, metric_obj, obml_measures, data_objects - ) - elif metric_obj.get("type") == "window": - ossie_metric = self._convert_obml_window_metric( - metric_name, metric_obj, obml_measures, data_objects - ) + for name, measure in obml_measures.items(): + if str(measure.get("aggregation", "")).lower() == "measure": + ossie_metric = self._convert_delegated_measure(name, measure) else: - ossie_metric = self._convert_obml_metric( - metric_name, metric_obj, obml_measures, data_objects - ) - if ossie_metric: - self._carry_foreign_to_ossie_metric(metric_obj, ossie_metric) - ossie_metrics.append(ossie_metric) + try: + sql = renderer.measure(name) + except NotPortableError as exc: + self._leave_out("measures", name, measure, str(exc)) + continue + ossie_metric = self._convert_measure(name, measure, sql) + self._finish_ossie_metric("measure", measure, ossie_metric) + ossie_metrics.append(ossie_metric) + + for name, metric in obml_metrics.items(): + try: + sql = renderer.metric(name) + except NotPortableError as exc: + self._leave_out("metrics", name, metric, str(exc)) + continue + ossie_metric = self._convert_metric(name, metric, sql) + self._finish_ossie_metric("metric", metric, ossie_metric) + ossie_metrics.append(ossie_metric) return ossie_metrics + def _leave_out(self, kind: str, name: str, definition: dict, reason: str) -> None: + """Keep a non-portable measure or metric for the model-level extension.""" + self.unexported.setdefault(kind, {})[name] = definition + self.warnings.append( + f"{kind[:-1].capitalize()} '{name}' is not exported as an Ossie metric " + f"because {reason}. It is kept in the {_VENDOR_OBML} extension for the reverse " + f"conversion." + ) + + def _finish_ossie_metric(self, kind: str, obml_obj: dict, ossie_metric: dict) -> None: + """Attach what every exported measure or metric carries besides its SQL.""" + definition = {k: v for k, v in obml_obj.items() if k != "customExtensions"} + self._merge_obml_extension(ossie_metric, "obml_definition", definition) + self._merge_obml_extension(ossie_metric, "obml_definition_kind", kind) + self._carry_foreign_to_ossie_metric(obml_obj, ossie_metric) + + @staticmethod + def _merge_obml_extension(ossie_metric: dict, key: str, value: Any) -> None: + """Set *key* in the metric's single OBML-vendor extension, creating it if absent. + + The reverse direction reads only the first OBML-vendor extension it + finds, so every value goes into that one payload. + """ + exts = ossie_metric.setdefault("custom_extensions", []) + for ext in exts: + if ext.get("vendor_name") == _VENDOR_OBML: + data = json.loads(ext.get("data") or "{}") + data[key] = value + ext["data"] = json.dumps(data) + return + exts.append({"vendor_name": _VENDOR_OBML, "data": json.dumps({key: value})}) + def _carry_foreign_to_ossie_metric(self, obml_obj: dict, ossie_metric: dict) -> None: """Re-emit third-party vendor extensions on an OBML measure/metric to the Ossie metric, dropping the key again if nothing foreign was added.""" @@ -579,109 +622,36 @@ def _carry_foreign_to_ossie_metric(self, obml_obj: dict, ossie_metric: dict) -> if not ossie_metric["custom_extensions"]: del ossie_metric["custom_extensions"] - def _convert_measure(self, name: str, measure: dict, data_objects: dict) -> dict | None: - """Convert an OBML measure to an Ossie metric.""" - - columns = measure.get("columns", []) - agg = measure.get("aggregation", "sum").upper() - distinct = measure.get("distinct", False) - obml_synonyms = measure.get("synonyms", []) - - # Build ai_context with synonyms (name + native OBML synonyms) - ai_synonyms = [name] + [s for s in obml_synonyms if s != name] - ai_ctx: dict[str, Any] = {"synonyms": ai_synonyms} if ai_synonyms else {} - - # ``aggregation: measure`` delegates resolution to the engine - # (Databricks Metric View) — there is no source column to read - # and no ANSI_SQL expression to emit. Ossie has no first-class - # concept for engine-delegated aggregation, so we serialize the - # measure as an Ossie metric whose expression is the literal - # ``MEASURE("