项目文件夹

文件
wehub-resource-sync 593b94c120
pytest / Unit Tests (push) Has been cancelled
pytest / Integration (integration_tests_a) (push) Has been cancelled
pytest / Integration (integration_tests_b) (push) Has been cancelled
pytest / Integration (integration_tests_c) (push) Has been cancelled
pytest / Integration (integration_tests_d) (push) Has been cancelled
pytest / Integration (integration_tests_e) (push) Has been cancelled
pytest / Integration (integration_tests_f) (push) Has been cancelled
pytest / Integration (integration_tests_g) (push) Has been cancelled
pytest / Integration (integration_tests_h) (push) Has been cancelled
pytest / Integration (integration_tests_i) (push) Has been cancelled
pytest / Integration (integration_tests_j) (push) Has been cancelled
pytest / Distributed (distributed_a) (push) Has been cancelled
pytest / Distributed (distributed_b) (push) Has been cancelled
pytest / Distributed (distributed_c) (push) Has been cancelled
pytest / Distributed (distributed_d) (push) Has been cancelled
pytest / Distributed (distributed_e) (push) Has been cancelled
pytest / Distributed (distributed_f) (push) Has been cancelled
pytest / Minimal Install (push) Has been cancelled
pytest / Event File (push) Has been cancelled
pytest (slow) / py-slow (push) Has been cancelled
Publish JSON Schema / publish-schema (push) Has been cancelled
chore: import upstream snapshot with attribution
2026-07-13 12:49:20 +08:00

35 行
1.8 KiB
Python

import ludwig
from ludwig.constants import DEFAULTS, INPUT_FEATURES, OUTPUT_FEATURES, PREPROCESSING, PROC_COLUMN, TYPE
from ludwig.data.cache.types import CacheableDataset
from ludwig.types import ModelConfigDict
from ludwig.utils.data_utils import hash_dict
def calculate_checksum(original_dataset: CacheableDataset, config: ModelConfigDict):
"""Calculates a checksum for a dataset and model config.
The checksum is used to determine if the dataset and model config have changed since the last time the model was
trained. If either has changed, a different checksum will be produced which will lead to a cache miss and force
preprocessing to be performed again.
"""
features = config.get(INPUT_FEATURES, []) + config.get(OUTPUT_FEATURES, []) + config.get("features", [])
info = {
"ludwig_version": ludwig.globals.LUDWIG_VERSION,
"dataset_checksum": original_dataset.checksum,
"global_preprocessing": config.get(PREPROCESSING, {}),
"global_defaults": config.get(DEFAULTS, {}),
# PROC_COLUMN contains both the feature name and the feature hash that is computed
# based on each feature's preprocessing parameters and the feature's type.
# creating a sorted list out of the dict because hash_dict requires all values
# of the dict to be ordered object to ensure the creation fo the same hash
"feature_proc_columns": sorted({feature[PROC_COLUMN] for feature in features}),
"feature_types": [feature[TYPE] for feature in features],
"feature_preprocessing": [feature.get(PREPROCESSING, {}) for feature in features],
}
# LLM-specific params
if "prompt" in config:
info["prompt"] = config["prompt"]
return hash_dict(info, max_length=None).decode("ascii")