Add a bundle-backed config: dataset as smoke.hdb via git-lfs

Exercises the provider bundle path — verify sha256, unpack to the
content-addressed cache, symlink in at mount="data" so the config metadata_csv
path resolves. The loose-file config stays alongside it as the plain case.
This commit is contained in:
ryan
2026-08-15 08:45:00 +02:00
parent 4e815df327
commit 79c109613d
3 changed files with 95 additions and 0 deletions
+91
View File
@@ -0,0 +1,91 @@
{
"_notes": [
"Bundle-backed variant of tabular_smoke: the dataset arrives as smoke.hdb",
"rather than as loose files. mount='data' places the bundle's smoke/labels.csv",
"at data/smoke/labels.csv, which is what metadata_csv below expects."
],
"run_name": "tabular_bundle",
"num_classes": 2,
"split_identity_level": 1,
"eval_stage": "cd_fuse",
"save_predictions": true,
"seed": 1234,
"folds": 3,
"fold_seed": 100,
"output_root": "results",
"data": {
"module": "hypertower_core.profiles.generic",
"args": {
"metadata_csv": "data/smoke/labels.csv",
"id_column": "sample_id",
"label_col": "diagnosis",
"target_type": "classification",
"group_column": "patient_id",
"cat_cols": [
"site"
],
"exclude_cols": [
"notes"
]
}
},
"towers": [
{
"name": "cd",
"module": "hypertower_core.components.towers.clinical_tower",
"class": "ClinicalEncoder",
"data_source": "matrix",
"args": {
"hidden_dim": 64
}
}
],
"stages": [
{
"name": "cd_warm",
"type": "warm",
"tower": "cd",
"head_name": "cd_aux",
"level": "sample",
"epochs": 8
},
{
"name": "cd_aux",
"type": "head",
"input": "cd",
"train_with": "cd_fuse"
},
{
"name": "cd_fuse",
"type": "fusion",
"module": "hypertower_core.components.bridges.mono_bridge",
"class": "MonoBridge",
"inputs": [
"cd"
],
"level": "sample",
"epochs": 10,
"train_towers": true,
"args": {
"use_ln": false
}
}
],
"training": {
"lr": 0.001,
"batch_size": 32,
"tune_binary_threshold": true
},
"data_bundles": [
{
"file": "smoke.hdb",
"encrypted": false,
"mount": "data",
"unpacked_mb": 1,
"sha256": "b01dbf19fe3639a1b75fb80d75b31158f78dc42bec3767fe1caf94afd5b8fda5"
}
],
"dataset": {
"n_samples": 400
}
}