"""Unit tests for the reshape executor's core extraction logic. These use a synthetic ITT-style grid (no external files), so they run in CI and pin the transform: canonical Sample_ID construction, raw timepoint extraction, and %Basal computation against the basal (T0) value. """ from app.core.reshape_executor import ( _canonical_variable, _parse_row_label_id, _process_columns_sheet, _process_rows_sheet, ) from app.models.mapping import GroupAssignment, ReshapePlan, SheetReshapePlan # An ITT-like sheet: header (sample IDs) in row 3, TIME labels in column 1, # values in rows 4-6, and a side block from column 7 that must be ignored. GRID = [ ["", "", "", "", "", "", "", "", ""], # r0 ["", "", "", "", "", "", "", "", ""], # r1 ["", "", "", "", "", "", "", "% data", "858"], # r2 (side block header) ["", "TIME", "858", "861", "", "", "", "", ""], # r3 sample IDs ["", "0", "200", "100", "", "", "", "15", "x"], # r4 basal ["", "15", "300", "150", "", "", "", "30", "y"], # r5 ["", "30", "100", "250", "", "", "", "45", "z"], # r6 ] def _plan() -> ReshapePlan: return ReshapePlan( sheets=[SheetReshapePlan( logical_name="ITT.xlsx::KO M HFD", samples_axis="columns", sample_id_row=3, variable_label_index=1, value_first=4, value_last=6, group=GroupAssignment(genotype="KO", sex="Male", diet="HFD"), variable_kind="timepoint", )], ) def test_canonical_ids_and_metadata(): values, metadata = _process_columns_sheet(GRID, _plan().sheets[0], _plan()) assert set(values.keys()) == {"KO-M-HFD-858", "KO-M-HFD-861"} assert metadata["KO-M-HFD-858"] == {"Genotype": "KO", "Sex": "Male", "Diet": "HFD"} def test_raw_timepoint_values(): values, _ = _process_columns_sheet(GRID, _plan().sheets[0], _plan()) s = values["KO-M-HFD-858"] assert s["T0"] == 200.0 assert s["T15"] == 300.0 assert s["T30"] == 100.0 def test_percent_basal_against_t0(): values, _ = _process_columns_sheet(GRID, _plan().sheets[0], _plan()) s = values["KO-M-HFD-858"] # 300/200*100 = 150 ; 100/200*100 = 50 ; no %Basal T0 row assert s["%Basal T15"] == 150.0 assert s["%Basal T30"] == 50.0 assert "%Basal T0" not in s def test_side_block_columns_ignored(): values, _ = _process_columns_sheet(GRID, _plan().sheets[0], _plan()) # Only the two real animals: the "% data" side block (col 7+) is not a sample. assert len(values) == 2 # --------------------------------------------------------------------------- # Samples-in-rows (insulin-style): row labels are samples, columns are measures. # Header uses raw aliases; a blank + junk row follows the real data. # --------------------------------------------------------------------------- ROWS_GRID = [ ["", "pg/ml", "80X", "ug / ml", "DNA", "ng Insulin/ug DNA"], # r0 header (aliases) ["Wt F Chow 1", "614", "49139", "0.049", "0.045", "1091"], # r1 sample ["KO M Chow 2", "3652", "292190", "0.292", "0.081", "1942"], # r2 sample ["", "", "", "", "", ""], # r3 blank -> data stops ["Mean", "2133", "", "", "", ""], # r4 junk summary ] def _rows_plan() -> ReshapePlan: return ReshapePlan(sheets=[SheetReshapePlan( logical_name="Insulin.xls::WT F Chow HFD Total Ins", samples_axis="rows", sample_id_col=0, variable_label_index=0, group=GroupAssignment(), variable_kind="generic", )]) def test_rows_label_parser(): assert _parse_row_label_id("Wt F Chow 1", _rows_plan(), GroupAssignment()) == ( "WT-F-C-1", "WT", "Female", "Chow", ) assert _parse_row_label_id("KO M HFD 12", _rows_plan(), GroupAssignment()) == ( "KO-M-HFD-12", "KO", "Male", "HFD", ) # A junk cell doesn't parse (used to detect where the data block ends). assert _parse_row_label_id("Mean", _rows_plan(), GroupAssignment()) is None def test_variable_alias_mapping(): assert _canonical_variable("80X") == "80-Fold" assert _canonical_variable("80Fold") == "80-Fold" assert _canonical_variable("ug / ml") == "ug/ml" assert _canonical_variable("ug Insulin") == "ug/ml" assert _canonical_variable("DNA") == "DNA (ug)" assert _canonical_variable("ng Insulin/ug DNA") == "ng Ins/ug DNA" assert _canonical_variable("") is None def test_rows_axis_ids_aliases_and_metadata(): values, metadata = _process_rows_sheet(ROWS_GRID, _rows_plan().sheets[0], _rows_plan()) assert set(values.keys()) == {"WT-F-C-1", "KO-M-C-2"} # Aliased headers land under canonical variable names. assert values["WT-F-C-1"]["pg/ml"] == 614.0 assert values["WT-F-C-1"]["80-Fold"] == 49139.0 assert values["WT-F-C-1"]["ug/ml"] == 0.049 assert values["WT-F-C-1"]["DNA (ug)"] == 0.045 assert values["WT-F-C-1"]["ng Ins/ug DNA"] == 1091.0 # Metadata derived from the row label, not a sheet name. assert metadata["WT-F-C-1"] == {"Genotype": "WT", "Sex": "Female", "Diet": "Chow"} def test_rows_axis_junk_below_blank_ignored(): values, _ = _process_rows_sheet(ROWS_GRID, _rows_plan().sheets[0], _rows_plan()) # The "Mean" summary row after the blank line must not become a sample. assert len(values) == 2