{ "provenance": { "commit": "fb1be803c04a109f15aba152f6660a7b8463be74", "branch": "feat/toothfairy4-pipeline", "config": { "name": "toothfairy4-prototype-v1", "seed": 2026, "folds": 5, "preprocess": { "spacing_mm": [ 0.5, 0.5, 0.5 ], "shape_zyx": [ 112, 192, 192 ], "clip_percentiles": [ 0.5, 99.5 ], "orientation": "LPS", "dtype": "float16" }, "prototypes": { "max_prototypes": 1024, "min_support": 2, "embedder": "tfidf", "linkage_threshold": 0.62, "max_sentence_words": 60, "assign_threshold": 0.45, "tooth_aware": true }, "encoder": { "backbone": "slice2d", "timm_model": "convnext_tiny", "width": 32, "dropout": 0.2, "epochs": 40, "batch_size": 4, "accumulate": 2, "learning_rate": 0.0002, "weight_decay": 0.02, "warmup_ratio": 0.1, "label_smoothing": 0.02, "focal_gamma_negative": 2.0, "focal_gamma_positive": 0.0, "ema_decay": 0.999, "amp": true, "num_workers": 8, "min_head_support": 0, "seed": 2026 }, "llm": { "base_model": "Qwen/Qwen2.5-1.5B-Instruct", "enabled": false, "lora_rank": 32, "lora_alpha": 64, "epochs": 3, "batch_size": 4, "learning_rate": 0.0001, "max_length": 1536, "max_new_tokens": 512 }, "decode": { "strategy": "prototype", "clinical_weight": 0.8, "captioning_weight": 0.2, "calibration_rounds": 6, "refine_top": 250, "min_sentences": 4, "max_sentences": 16, "mbr_candidates": 24 } } }, "dataset": { "cases": 622, "prototypes": 989, "assign_threshold": 0.45, "tooth_aware": true }, "training": { "folds": 5, "fold_val_map": [ 0.08832909003921618, 0.06671829082485603, 0.06935018761915886, 0.08048351787180753, 0.06401017932076238 ], "mean_val_map": 0.0737782531351602 }, "calibration": { "objective": { "final": 0.35358663390798667, "clinical": 0.3862701303619008, "captioning": 0.22285264809233002, "logical_precision": 0.4928832064684155, "logical_recall": 0.3175767155686247, "bleu_4": 0.14492126109494552, "meteor": 0.3007840350897145, "mean_sentences": 12.257234726688102 }, "baseline_fixed_threshold": { "final": 0.3000755511911487, "clinical": 0.3236022472744878, "captioning": 0.20596876685779228, "logical_precision": 0.3846958663879885, "logical_recall": 0.2792538876064256, "bleu_4": 0.12718854623405096, "meteor": 0.2847489874815336, "mean_sentences": 11.591639871382636 }, "rounds": 6 }, "evaluation": { "final": 0.3544252290767211, "clinical": 0.38607203617714003, "captioning": 0.22783800067504528, "logical_precision": 0.49096604984868647, "logical_recall": 0.3181086871959718, "bleu_4": 0.14926612750396082, "meteor": 0.3064098738461298, "num_cases": 622, "clinical_source": "lexical-surrogate" }, "evaluation_radfact_lite": null, "ablation": { "corpus prior (no imaging)": { "final": 0.35751793649563435, "clinical": 0.39374008539376365, "captioning": 0.21262934090311725, "logical_precision": 0.5385852090032154, "logical_recall": 0.31029144605574627, "bleu_4": 0.11701313497207318, "meteor": 0.30824554683416133, "num_cases": 622 }, "linear, 122 features": { "final": 0.3544252290767211, "clinical": 0.38607203617714003, "captioning": 0.22783800067504528, "logical_precision": 0.49096604984868647, "logical_recall": 0.3181086871959718, "bleu_4": 0.14926612750396082, "meteor": 0.3064098738461298, "num_cases": 622 }, "fine-tuned encoder (29M params)": { "final": 0.3403329756224751, "clinical": 0.3752387592094134, "captioning": 0.20070984127472175, "logical_precision": 0.5481078407123423, "logical_recall": 0.2852675303807301, "bleu_4": 0.13306690758262765, "meteor": 0.26835277496681587, "num_cases": 622 } }, "caveats": [ "Clinical scores are the repository's offline RadFact surrogate unless 'evaluation_radfact_lite' is populated. The surrogate ranks decoder variants; it is not the challenge metric.", "BLEU-4 and METEOR are exact reimplementations of the grader's local implementations and match NLTK to machine precision.", "Scores are out-of-fold on the public training release, not the hidden 50-case test set, and the folds are stratified rather than leave-one-centre-out, so they measure in-domain performance." ] }