From 79c109613de83dcb8d3eb4fe9d2c88b8c591ee8f Mon Sep 17 00:00:00 2001 From: ryan Date: Sat, 15 Aug 2026 08:45:00 +0200 Subject: [PATCH] Add a bundle-backed config: dataset as smoke.hdb via git-lfs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Exercises the provider bundle path — verify sha256, unpack to the content-addressed cache, symlink in at mount="data" so the config metadata_csv path resolves. The loose-file config stays alongside it as the plain case. --- .gitattributes | 4 ++ configs/tabular_bundle.hcf | 91 +++++++++++++++++++++++++++++++++++++ smoke.hdb | Bin 0 -> 129 bytes 3 files changed, 95 insertions(+) create mode 100644 .gitattributes create mode 100644 configs/tabular_bundle.hcf create mode 100644 smoke.hdb diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..bfc1f66 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,4 @@ +*.hdb filter=lfs diff=lfs merge=lfs -text +*.hrb filter=lfs diff=lfs merge=lfs -text +*.hdb binary +*.hrb binary diff --git a/configs/tabular_bundle.hcf b/configs/tabular_bundle.hcf new file mode 100644 index 0000000..9015721 --- /dev/null +++ b/configs/tabular_bundle.hcf @@ -0,0 +1,91 @@ +{ + "_notes": [ + "Bundle-backed variant of tabular_smoke: the dataset arrives as smoke.hdb", + "rather than as loose files. mount='data' places the bundle's smoke/labels.csv", + "at data/smoke/labels.csv, which is what metadata_csv below expects." + ], + "run_name": "tabular_bundle", + "num_classes": 2, + "split_identity_level": 1, + "eval_stage": "cd_fuse", + "save_predictions": true, + "seed": 1234, + "folds": 3, + "fold_seed": 100, + "output_root": "results", + "data": { + "module": "hypertower_core.profiles.generic", + "args": { + "metadata_csv": "data/smoke/labels.csv", + "id_column": "sample_id", + "label_col": "diagnosis", + "target_type": "classification", + "group_column": "patient_id", + "cat_cols": [ + "site" + ], + "exclude_cols": [ + "notes" + ] + } + }, + "towers": [ + { + "name": "cd", + "module": "hypertower_core.components.towers.clinical_tower", + "class": "ClinicalEncoder", + "data_source": "matrix", + "args": { + "hidden_dim": 64 + } + } + ], + "stages": [ + { + "name": "cd_warm", + "type": "warm", + "tower": "cd", + "head_name": "cd_aux", + "level": "sample", + "epochs": 8 + }, + { + "name": "cd_aux", + "type": "head", + "input": "cd", + "train_with": "cd_fuse" + }, + { + "name": "cd_fuse", + "type": "fusion", + "module": "hypertower_core.components.bridges.mono_bridge", + "class": "MonoBridge", + "inputs": [ + "cd" + ], + "level": "sample", + "epochs": 10, + "train_towers": true, + "args": { + "use_ln": false + } + } + ], + "training": { + "lr": 0.001, + "batch_size": 32, + "tune_binary_threshold": true + }, + "data_bundles": [ + { + "file": "smoke.hdb", + "encrypted": false, + "mount": "data", + "unpacked_mb": 1, + "sha256": "b01dbf19fe3639a1b75fb80d75b31158f78dc42bec3767fe1caf94afd5b8fda5" + } + ], + "dataset": { + "n_samples": 400 + } +} \ No newline at end of file diff --git a/smoke.hdb b/smoke.hdb new file mode 100644 index 0000000000000000000000000000000000000000..e69442c5a34434f5d088fddfb74a5dfc7eeb24c7 GIT binary patch literal 129 zcmWN?OA^8$3;@tQr{DsXrVu{84FMv|sB}#2!qe;9ysN!s%$M%xdB|?eeVn%k%ksZ} zXesk)iQpysGrdfw3Qv_d6@#Laq}GEhLKxYmGV0WfliwYZI1vB3v!P%