24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129 | def convert(self, source, datasets: dict, **options) -> list[OsiSemanticModel]: # noqa: C901
"""Convert a ViewDefinition to OSI semantic models.
Args:
source: ViewDefinition with from_item and joins
datasets: dict mapping datasource_id -> DomoDataset (or None)
**options: Unused in Phase 1.0
Returns:
list[OsiSemanticModel], one per participating dataset
"""
models: dict[str, OsiSemanticModel] = {}
base_ds_id = source.from_item.dataset_id
# Collect join key columns per dataset_id, tracking which side they're on.
# Base-table left-side keys are "primary"; joined-table right-side keys are "foreign".
primary_keys: dict[str, set[str]] = {} # dataset_id → primary (left) join cols
foreign_keys: dict[str, set[str]] = {} # dataset_id → foreign (right) join cols
for join in source.joins or []:
if join.left_column:
primary_keys.setdefault(base_ds_id, set()).add(join.left_column)
if join.right_column and join.dataset_id:
foreign_keys.setdefault(join.dataset_id, set()).add(join.right_column)
def _build_model(dataset_id: str, alias: str | None = None) -> OsiSemanticModel:
dataset = datasets.get(dataset_id)
ar_map = self._get_ai_readiness_map(dataset)
join_keys = primary_keys.get(dataset_id, set()) | foreign_keys.get(
dataset_id, set()
)
entities = []
dimensions = []
metrics = []
schema_columns = None
if dataset and dataset.Schema and dataset.Schema.columns:
schema_columns = dataset.Schema.columns
if schema_columns:
for col in schema_columns:
col_name = col.name if hasattr(col, "name") else str(col)
col_type = col.type if hasattr(col, "type") else ""
classification = self._classify_column(
col_name, col_type, join_keys, ar_map
)
if classification == "entity":
is_primary = col_name in primary_keys.get(dataset_id, set())
entities.append(
self._build_entity(col_name, is_primary=is_primary)
)
elif classification in ("dimension", "time_dimension"):
dimensions.append(
self._build_dimension(
col_name, col_type, ar_map.get(col_name)
)
)
elif classification == "metric":
metrics.append(
self._build_metric_from_ai_readiness(
col_name, col_type, ar_map.get(col_name)
)
)
elif join_keys:
# No schema available — at minimum expose the join keys as entities
for jk in sorted(primary_keys.get(dataset_id, set())):
entities.append(self._build_entity(jk, is_primary=True))
for jk in sorted(foreign_keys.get(dataset_id, set())):
entities.append(self._build_entity(jk, is_primary=False))
name = (dataset.name if dataset else None) or alias or dataset_id
description = ""
if dataset and dataset.AI_Readiness:
description = dataset.AI_Readiness.unit_of_analysis or ""
return OsiSemanticModel(
name=name,
source_dataset_id=dataset_id,
description=description,
entities=entities,
dimensions=dimensions,
metrics=metrics,
domo_extensions={"dataset_id": dataset_id},
)
# Build base model
models[base_ds_id] = _build_model(base_ds_id, source.from_item.alias)
# Build a model for each join target and attach relationship to base
for join in source.joins or []:
joined_ds_id = join.dataset_id
if joined_ds_id not in models:
models[joined_ds_id] = _build_model(joined_ds_id)
relationship = OsiRelationship(
from_model=models[base_ds_id].name,
to_model=models[joined_ds_id].name,
from_field=join.left_column or "",
to_field=join.right_column or "",
join_type=join.join_type or "LEFT",
cardinality=None, # DSV schema does not expose cardinality
)
models[base_ds_id].relationships.append(relationship)
return list(models.values())
|