Skip to content

convenience

convenience

generate_schema_from_dataframe

generate_schema_from_dataframe(
    df: DataFrame,
    upsert_key: str | None = None,
    column_types: dict[str, str] | None = None,
) -> dict

Generate a Domo schema dict from a pandas DataFrame.

Maps pandas dtypes to Domo schema column types: object/bool -> STRING, integer -> LONG, float -> DOUBLE, datetime -> DATETIME.

Parameters:

Name Type Description Default
df DataFrame

pandas DataFrame to analyze

required
upsert_key str | None

Column name to mark as the identity/upsert column

None
column_types dict[str, str] | None

Override mapping {column_name: "STRING"|"LONG"|"DOUBLE"|"DATE"|"DATETIME"}

None

Returns:

Type Description
dict

{"columns": [{"name": "...", "type": "...", "upsertKey": bool}, ...]}

Source code in src/crew_dcs/routes/dataset/convenience.py
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
def generate_schema_from_dataframe(
    df: pd.DataFrame,
    upsert_key: str | None = None,
    column_types: dict[str, str] | None = None,
) -> dict:
    """Generate a Domo schema dict from a pandas DataFrame.

    Maps pandas dtypes to Domo schema column types:
    object/bool -> STRING, integer -> LONG, float -> DOUBLE, datetime -> DATETIME.

    Args:
        df: pandas DataFrame to analyze
        upsert_key: Column name to mark as the identity/upsert column
        column_types: Override mapping {column_name: "STRING"|"LONG"|"DOUBLE"|"DATE"|"DATETIME"}

    Returns:
        {"columns": [{"name": "...", "type": "...", "upsertKey": bool}, ...]}
    """
    column_types = column_types or {}

    columns = []
    for col_name in df.columns:
        col_type = column_types.get(col_name) or _detect_domo_type(df[col_name])
        columns.append(
            {
                "name": col_name,
                "type": _DOMO_TYPE_MAP.get(col_type, col_type),
                "upsertKey": col_name == upsert_key,
            }
        )

    return {"columns": columns}

upload_dataframe async

upload_dataframe(
    auth: DomoAuth,
    dataset_name: str,
    df: DataFrame,
    *,
    schema: dict | None = None,
    upsert_key: str | None = None,
    column_types: dict[str, str] | None = None,
    update_method: str = "REPLACE",
    is_index: bool = True,
    is_create_if_not_exists: bool = True,
    context: RouteContext | None = None,
    **context_kwargs
) -> ResponseGetData

Upload a DataFrame to Domo in one call.

If the dataset doesn't exist, creates it with an auto-detected schema. If it does exist, amends the schema for any new DataFrame columns before uploading.

Parameters:

Name Type Description Default
auth DomoAuth

Authentication object for API requests

required
dataset_name str

Name of the dataset to find or create

required
df DataFrame

DataFrame to upload

required
schema dict | None

Explicit schema override used when creating a new dataset

None
upsert_key str | None

Column name to mark as the identity/upsert column

None
column_types dict[str, str] | None

Override auto-detected types for specific columns

None
update_method str

"REPLACE" or "APPEND"

'REPLACE'
is_index bool

Whether to index the dataset after uploading

True
is_create_if_not_exists bool

Create the dataset if no match is found by name

True
context RouteContext | None

RouteContext for request configuration

None

Returns:

Type Description
ResponseGetData

ResponseGetData from the upload commit, with dataset_id set.

Raises:

Type Description
DatasetNotFoundError

If no dataset matches and is_create_if_not_exists is False

Source code in src/crew_dcs/routes/dataset/convenience.py
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
@gd.route_function
@log_call(
    level_name="route",
    config=LogDecoratorConfig(
        entity_extractor=DomoEntityExtractor(),
        result_processor=DomoEntityResultProcessor(),
    ),
)
async def upload_dataframe(
    auth: DomoAuth,
    dataset_name: str,
    df: pd.DataFrame,
    *,
    schema: dict | None = None,
    upsert_key: str | None = None,
    column_types: dict[str, str] | None = None,
    update_method: str = "REPLACE",
    is_index: bool = True,
    is_create_if_not_exists: bool = True,
    context: RouteContext | None = None,
    **context_kwargs,
) -> rgd.ResponseGetData:
    """Upload a DataFrame to Domo in one call.

    If the dataset doesn't exist, creates it with an auto-detected schema.
    If it does exist, amends the schema for any new DataFrame columns before
    uploading.

    Args:
        auth: Authentication object for API requests
        dataset_name: Name of the dataset to find or create
        df: DataFrame to upload
        schema: Explicit schema override used when creating a new dataset
        upsert_key: Column name to mark as the identity/upsert column
        column_types: Override auto-detected types for specific columns
        update_method: "REPLACE" or "APPEND"
        is_index: Whether to index the dataset after uploading
        is_create_if_not_exists: Create the dataset if no match is found by name
        context: RouteContext for request configuration

    Returns:
        ResponseGetData from the upload commit, with `dataset_id` set.

    Raises:
        DatasetNotFoundError: If no dataset matches and is_create_if_not_exists is False
    """
    context = RouteContext.build_context(context=context, **context_kwargs)

    dataset_id, search_res = await _find_dataset_id_by_name(
        auth=auth, dataset_name=dataset_name, context=context
    )

    if dataset_id is None:
        if not is_create_if_not_exists:
            raise DatasetNotFoundError(dataset_id=dataset_name, res=search_res)

        create_schema = schema or generate_schema_from_dataframe(
            df, upsert_key=upsert_key, column_types=column_types
        )
        create_res = await create(
            auth=auth,
            dataset_name=dataset_name,
            schema=create_schema,
            context=context,
        )
        dataset_id = create_res.response.get("id") or create_res.response.get(
            "dataSource", {}
        ).get("dataSourceId")
    else:
        schema_res = await get_schema(auth=auth, dataset_id=dataset_id, context=context)
        existing_columns = schema_res.response.get("tables", [{}])[0].get("columns", [])
        existing_names = {col.get("name") for col in existing_columns}
        new_column_names = [name for name in df.columns if name not in existing_names]

        needs_upsert_update = upsert_key is not None and not any(
            col.get("name") == upsert_key and col.get("upsertKey")
            for col in existing_columns
        )

        if new_column_names or needs_upsert_update:
            merged_columns = [
                {
                    "id": col.get("id"),
                    "name": col.get("name"),
                    "type": col.get("type"),
                    "upsertKey": col.get("name") == upsert_key
                    if upsert_key is not None
                    else col.get("upsertKey", False),
                }
                for col in existing_columns
            ]
            for name in new_column_names:
                col_type = (column_types or {}).get(name) or _detect_domo_type(df[name])
                merged_columns.append(
                    {
                        "name": name,
                        "type": _DOMO_TYPE_MAP.get(col_type, col_type),
                        "upsertKey": name == upsert_key,
                    }
                )

            await alter_schema(
                auth=auth,
                dataset_id=dataset_id,
                schema_obj={"columns": merged_columns},
                context=context,
            )

    stage_1_res = await upload_dataset_stage_1(
        auth=auth, dataset_id=dataset_id, context=context
    )
    upload_id = stage_1_res.response

    await upload_dataset_stage_2_df(
        auth=auth,
        dataset_id=dataset_id,
        upload_id=upload_id,
        upload_df=df,
        context=context,
    )

    res = await upload_dataset_stage_3(
        auth=auth,
        dataset_id=dataset_id,
        upload_id=upload_id,
        update_method=update_method,
        is_index=False,
        context=context,
    )

    if is_index:
        res = await index_dataset(auth=auth, dataset_id=dataset_id, context=context)

    res.dataset_id = dataset_id

    return res