{ "_note": "val_R values are honest cross-subject holdouts: per-subject z-scoring applied, 15% of subjects per dataset held out before training, zero subject overlap with training data.", "models": { "text2roi_whisper_v4": { "file": "text2roi_whisper_v4.pt", "description": "Audio → 56-ROI MLP projector. Input: Whisper-large-v3 encoder states (1280d), per-TR. Honest cross-subject holdout val_R=0.217; actual held-out eval on 23 subjects = R=0.257 (+4.2% vs TRIBE v2, Meta AI Algonauts 2025 winner). Single shared model, no per-subject fine-tuning.", "architecture": "Linear(1280→1024) → GELU → Dropout → LayerNorm → Linear(1024→512) → GELU → Dropout → Linear(512→56)", "in_dim": 1280, "hidden_dim": 1024, "out_dim": 56, "embedding_model": "openai/whisper-large-v3", "best_val_mean_pearson_r": 0.217, "holdout_mean_pearson_r": 0.257, "holdout_n_subjects": 23, "holdout_n_trs": 33038, "vs_tribe_v2_delta": 0.042, "r_above_010": "53/56", "r_above_020": "41/56", "r_above_030": "20/56", "top_rois": { "ACC": 0.438, "STG": 0.422, "Thalamus": 0.408, "V1": 0.393, "LP_R": 0.380 }, "epochs": 120, "loss": "pearson_r", "training_data": "real fMRI (CNeuroMod Friends, narratives, LPP, HCP, language fMRI), fsaverage5+subcortical 28,444-voxel space, 2.73M TRs total", "eval_honesty": "per-subject z-scoring + per-subject train/val split (15% subjects withheld per dataset, zero overlap)", "roi_schema": "UNIVERSAL_ROI_NAMES_56", "recommended_for": "audio stimuli — speech, music, naturalistic audio" }, "text2roi_combined_v4": { "file": "text2roi_combined_v4.pt", "description": "Text+Audio → 56-ROI MLP projector. Input: concatenated whisper (1280d) | qwen3 (2560d) = 3840d total. For text-only inference, zero-pad the whisper slot. Trained with modality dropout (0.3) so text-only, audio-only, and combined all work at inference. Honest cross-subject val_R=0.192.", "architecture": "Linear(3840→1024) → GELU → Dropout → LayerNorm → Linear(1024→512) → GELU → Dropout → Linear(512→56)", "in_dim": 3840, "hidden_dim": 1024, "out_dim": 56, "embedding_model": "Qwen/Qwen3-Embedding-4B", "whisper_model": "openai/whisper-large-v3", "best_val_mean_pearson_r": 0.192, "modality_dropout": 0.3, "epochs": 120, "loss": "pearson_r", "training_data": "real fMRI, 422k train TRs, 65k val TRs (files with both whisper+qwen3 features)", "eval_honesty": "per-subject z-scoring + per-subject train/val split", "roi_schema": "UNIVERSAL_ROI_NAMES_56", "recommended_for": "text inputs (zero-pad whisper slot) or combined text+audio" }, "text2roi_qwen3_v8": { "file": "text2roi_qwen3_v8.pt", "description": "Text → 56-ROI MLP projector. Input: Qwen3-Embedding-4B (2560d). Honest cross-subject val_R=0.115. Cross-dataset text generalization is an active area of improvement — use text2roi_combined_v4 for best text results.", "architecture": "Linear(2560→1024) → GELU → Dropout → LayerNorm → Linear(1024→512) → GELU → Dropout → Linear(512→56)", "in_dim": 2560, "hidden_dim": 1024, "out_dim": 56, "embedding_model": "Qwen/Qwen3-Embedding-4B", "best_val_mean_pearson_r": 0.115, "epochs": 120, "loss": "pearson_r + anchor_ranking", "anchor_loss_weight": 1.0, "training_data": "real fMRI (narratives, CNeuroMod, LPP, HCP, language fMRI, Cowen-Keltner), 704k train TRs", "eval_honesty": "per-subject z-scoring + per-subject train/val split", "roi_schema": "UNIVERSAL_ROI_NAMES_56", "recommended_for": "text-only inference (experimental; combined_v4 preferred)" } }, "roi_schema": { "name": "UNIVERSAL_ROI_NAMES_56", "n_rois": 56, "atlas": "HCP MMP1.0", "space": "fsaverage5 + subcortical", "rois": [ "V1","V2","V3","V4","V3A","V3B","LO1","LO2","MT","MST","V7","IPS1", "FFA-1","FFA-2","PPA","RSC","OFA","EBA","IPS2","IPS3","IPS4","IPS5","SPL1", "hIP1","hIP2","hIP3","dlPFC","vlPFC","OFC","ACC","mPFC","FP1","FP2", "IFG","IFGorb","STG","STS","MTG","AG","PCC","mPFC_dmn","LP_L","LP_R", "HPC_L","HPC_R","AI","dACC","sgACC","vmPFC", "Amygdala_L","Amygdala_R","Caudate_L","Caudate_R","Putamen_L","Putamen_R","Thalamus" ] } }