Skip to content

column_relationship_erd_converter

column_relationship_erd_converter

ColumnRelationshipERConverter — convert ColumnRelationships to Mermaid ER diagrams.

Takes a set of ColumnRelationship objects (from dataflow actions or dataset views) and produces a MermaidERDiagram showing column-level lineage as entity-relationship diagrams.

Supports tracing a specific column upstream through the relationship graph, enabling the "where does this column come from?" use case.

ColumnRelationshipERConverter dataclass

ColumnRelationshipERConverter()

Convert ColumnRelationships to a Mermaid ER diagram.

Groups column relationships by entity pair, builds entities with columns, and creates relationships with cardinality derived from transformation types.

Supports tracing a specific column upstream through the relationship graph — enabling the "where does this column come from?" use case.

Example

from crew_dcs.classes.subentity import ColumnRelationship rels = { ... ColumnRelationship(from_entity_id="ds-1", from_column="ID", ... to_entity_id="meta-1", to_column="ID"), ... ColumnRelationship(from_entity_id="meta-1", from_column="ID", ... to_entity_id="gb-1", to_column="ID", ... transformation_type=ColumnTransformationType.AGGREGATION), ... } diagram = ColumnRelationshipERConverter.convert(rels, entity_names={"ds-1": "Source", "meta-1": "Metadata", "gb-1": "GroupBy"}) print(diagram.to_string())

convert staticmethod

convert(
    column_relationships: (
        set[ColumnRelationship] | list[ColumnRelationship]
    ),
    *,
    entity_names: dict[str, str] | None = None,
    title: str | None = None,
    trace_from_entity: str | None = None,
    trace_from_column: str | None = None
) -> MermaidERDiagram

Convert column relationships to a Mermaid ER diagram.

Parameters:

Name Type Description Default
column_relationships set[ColumnRelationship] | list[ColumnRelationship]

Set or list of ColumnRelationship objects

required
entity_names dict[str, str] | None

Optional mapping of entity_id → display name. Falls back to entity_id if not provided.

None
title str | None

Optional diagram title

None
trace_from_entity str | None

If provided, trace only the upstream chain from this entity (e.g., a card's dataset ID)

None
trace_from_column str | None

If provided alongside trace_from_entity, trace only the upstream chain for this specific column (e.g., a beast mode or dataset column)

None

Returns:

Type Description
MermaidERDiagram

MermaidERDiagram with entities and relationships

Example — trace a specific column: >>> # "Where does the 'Revenue' column in this card come from?" >>> diagram = ColumnRelationshipERConverter.convert( ... all_rels, ... entity_names=names, ... trace_from_entity="card-dataset-id", ... trace_from_column="Revenue", ... )

Source code in src/crew_dcs/integrations/graphs/mermaid/column_relationship_erd_converter.py
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
@staticmethod
def convert(  # noqa: C901
    column_relationships: set[ColumnRelationship] | list[ColumnRelationship],
    *,
    entity_names: dict[str, str] | None = None,
    title: str | None = None,
    trace_from_entity: str | None = None,
    trace_from_column: str | None = None,
) -> MermaidERDiagram:
    """Convert column relationships to a Mermaid ER diagram.

    Args:
        column_relationships: Set or list of ColumnRelationship objects
        entity_names: Optional mapping of entity_id → display name.
                     Falls back to entity_id if not provided.
        title: Optional diagram title
        trace_from_entity: If provided, trace only the upstream chain
                          from this entity (e.g., a card's dataset ID)
        trace_from_column: If provided alongside trace_from_entity,
                          trace only the upstream chain for this specific
                          column (e.g., a beast mode or dataset column)

    Returns:
        MermaidERDiagram with entities and relationships

    Example — trace a specific column:
        >>> # "Where does the 'Revenue' column in this card come from?"
        >>> diagram = ColumnRelationshipERConverter.convert(
        ...     all_rels,
        ...     entity_names=names,
        ...     trace_from_entity="card-dataset-id",
        ...     trace_from_column="Revenue",
        ... )
    """
    if not column_relationships:
        return MermaidERDiagram(title=title)

    # If tracing, filter to only the upstream chain
    if trace_from_entity:
        column_relationships = _trace_upstream(
            column_relationships,
            start_entity_id=trace_from_entity,
            start_column=trace_from_column,
        )
        if not column_relationships:
            return MermaidERDiagram(title=title)

    entity_names = entity_names or {}

    # Step 1: Collect all entity IDs and their columns
    entity_columns: dict[str, set[str]] = defaultdict(set)
    entity_key_columns: dict[str, dict[str, str]] = defaultdict(
        dict
    )  # entity_id → {col_name → key_type}

    for rel in column_relationships:
        # Track columns on both sides
        if rel.from_column:
            entity_columns[rel.from_entity_id].add(rel.from_column)
        if rel.to_column:
            entity_columns[rel.to_entity_id].add(rel.to_column)

        # Mark FK on columns involved in JOINs
        if rel.transformation_type == ColumnTransformationType.JOIN:
            if rel.from_column:
                entity_key_columns[rel.from_entity_id][rel.from_column] = "FK"
            if rel.to_column:
                entity_key_columns[rel.to_entity_id][rel.to_column] = "FK"

    # Step 2: Build entities
    diagram = MermaidERDiagram(title=title)

    for entity_id, columns in entity_columns.items():
        name = entity_names.get(entity_id, entity_id)
        entity = MermaidEREntity(id=entity_id, name=name)
        for col_name in sorted(columns):
            key = entity_key_columns.get(entity_id, {}).get(col_name)
            entity.add_column(MermaidERColumn(name=col_name, key=key))
        diagram.add_entity(entity)

    # Step 3: Group relationships by entity pair
    pair_rels: dict[tuple[str, str], list[ColumnRelationship]] = defaultdict(list)
    for rel in column_relationships:
        pair_key = (rel.from_entity_id, rel.to_entity_id)
        pair_rels[pair_key].append(rel)

    # Step 4: Build relationships
    for (from_id, to_id), rels in pair_rels.items():
        from_entity = diagram.find_entity(from_id)
        to_entity = diagram.find_entity(to_id)
        if not from_entity or not to_entity:
            continue

        # Determine strongest cardinality from all column relationships
        strongest_card = CARDINALITY_ONE_TO_ONE
        column_labels: list[str] = []

        for rel in rels:
            # Cardinality from transformation type
            type_name = (
                rel.transformation_type.value
                if rel.transformation_type
                else "IDENTITY"
            )
            card = TRANSFORMATION_CARDINALITY.get(type_name, CARDINALITY_ONE_TO_ONE)
            strongest_card = _stronger_cardinality(strongest_card, card)

            # Build column mapping label
            from_col = rel.from_column or "?"
            to_col = rel.to_column or "(new)"
            label_part = from_col if from_col == to_col else f"{from_col}→{to_col}"

            # Add transformation type for non-identity
            if type_name != "IDENTITY":
                label_part += f" [{type_name}]"

            # Add expression if present (beast modes, formulas)
            if rel.expression:
                label_part += f" ({rel.expression})"

            column_labels.append(label_part)

        # Determine if identifying (solid) or non-identifying (dashed)
        # INDIRECT relationships are non-identifying
        has_indirect = any(r.directness == ColumnDirectness.INDIRECT for r in rels)

        # Build label
        label = ", ".join(column_labels) if column_labels else None

        er_rel = MermaidERRelationship(
            from_entity=from_entity,
            to_entity=to_entity,
            cardinality=strongest_card,
            label=label,
            identifying=not has_indirect,
        )
        diagram.add_relationship(er_rel)

    return diagram