conda create --name sd python==3.12
conda activate sd
pip install pandas kagglehub[pandas-datasets] optunaInstall git LFS first if you don't have it:
https://gist.github.com/pourmand1376/bc48a407f781d6decae316a5cfa7d8ab
chmod +x acquisition.sh
./acquisition.sh
python handler.py moma # check if handler worksHow to use inside a model pipeline:
sys.path.append(path_to_struggletab_codebase)
from handler import UnifiedDataLoader; from eval import evaluate
train_df = loader.get_train_data()
synthetic_data_frame = your_full_model_pipeline() # run your model training/sampling code
result = evaluate(dataset_name, synthetic_data_frame)- Honeypot from Mike Sconzo, SecRepo
- Stroke from Kaggle / Federico Soriano Palacios
- Museum of Modern Art (MoMA) Collection, v2026-06-02
- CERN from Thomas McCauley, CERN
- High Frequency Crypto Limit Order Book Data shared on Kaggle by Martin Søgaard Nielsen
- Olist from André Sionek, Olist