import%20marimo%0A%0A__generated_with%20%3D%20%220.23.5%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%207%20%E2%80%94%20Phase%202%20unblinded%20test%20set%20analysis%0A%0A%20%20%20%20The%20challenge%20has%20finished.%20%20The%20final%2C%20fully%20unblinded%20**Phase%202**%20test%20set%0A%20%20%20%20(%60pxr-challenge_TEST_PHASE_2_UNBLINDED.csv%60%2C%20260%20compounds)%20is%20now%20available.%0A%20%20%20%20This%20notebook%20is%20the%20post-mortem%3A%20it%20mirrors%20the%20structure%20of%0A%20%20%20%20%605_unblinded_analysis.py%60%20(the%20Phase%201%20post-mortem)%20but%20focuses%20on%20the%20Phase%202%0A%20%20%20%20labels%20and%20evaluates%20**every%20submission%20we%20produced**%20%E2%80%94%20not%20just%20the%20ensemble%0A%20%20%20%20components.%0A%0A%20%20%20%20It%20answers%20three%20questions%3A%0A%0A%20%20%20%201.%20**Dataset%20profile**%20%E2%80%94%20how%20does%20the%20Phase%202%20pEC50%20distribution%20compare%20with%0A%20%20%20%20%20%20%20the%20training%20set%20*and*%20with%20the%20Phase%201%20test%20set%3F%20%20Do%20the%20three%20sets%20probe%0A%20%20%20%20%20%20%20the%20same%20activity%20range%3F%0A%20%20%20%202.%20**Activity%20cliffs**%20%E2%80%94%20how%20many%20structurally%20similar%20(train%2C%20Phase%202)%20compound%0A%20%20%20%20%20%20%20pairs%20have%20very%20different%20potency%3F%20%20These%20are%20the%20hardest%20cases%20for%20any%20model.%0A%20%20%20%203.%20**Model%20ranking**%20%E2%80%94%20recomputing%20MAE%20(the%20competition%20ranking%20metric)%20alongside%0A%20%20%20%20%20%20%20RMSE%20%2F%20R%C2%B2%20%2F%20Pearson%20*r*%20%2F%20Spearman%20%CF%81%20for%20*all*%20submissions%20against%20the%20Phase%202%0A%20%20%20%20%20%20%20truth%3A%20which%20submissions%20worked%20best%20by%20MAE%2C%20which%20worked%20worst%2C%20and%20%E2%80%94%20crucially%0A%20%20%20%20%20%20%20%E2%80%94%20would%20any%20submission%20have%20beaten%20our%20final%20one%0A%20%20%20%20%20%20%20(%606_ens_default_augfilt_submission.csv%60)%3F%0A%0A%20%20%20%20**Inputs%3A**%0A%20%20%20%20-%20Phase%202%20unblinded%20CSV%3A%20%60data%2Fraw%2F20260703%2Fpxr-challenge_TEST_PHASE_2_UNBLINDED.csv%60.%0A%20%20%20%20-%20Phase%201%20unblinded%20CSV%20(cached%20by%20notebook%205)%3A%20%60data%2Fraw%2F20260528%2Fdose_response_test_unblinded.csv%60.%0A%20%20%20%20-%20Every%20submission%20CSV%20in%20%60submissions%2F%60%20ending%20in%20%60_submission.csv%60.%0A%20%20%20%20-%20The%20external%20competitor%20blend%20%60data%2Fraw%2F20260703%2Frank9_gashaw_submission_blend_751510.csv%60%0A%20%20%20%20%20%20(shown%20for%20reference%20only%20%E2%80%94%20it%20is%20somebody%20else's%20model%2C%20not%20ours).%0A%20%20%20%20-%20Processed%20training%20data%3A%20%60data%2Fprocessed%2Fall_compounds_activity_data.csv%60.%0A%20%20%20%20-%20Pre-generated%20MMP%20database%3A%20%60data%2Fprocessed%2Fall_compounds_mmp.mmp.csv.gz%60.%0A%0A%20%20%20%20Every%20submission%20covers%20the%20full%20513-compound%20test%20set%20(253%20Phase%201%20%2B%20260%0A%20%20%20%20Phase%202)%2C%20so%20unlike%20notebook%205%20we%20do%20**not**%20need%20to%20regenerate%20any%20predictions%20%E2%80%94%0A%20%20%20%20we%20simply%20slice%20each%20submission%20to%20the%20260%20Phase%202%20compounds%20and%20score%20it.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20glob%0A%20%20%20%20import%20math%0A%20%20%20%20from%20pathlib%20import%20Path%0A%20%20%20%20from%20typing%20import%20Optional%0A%0A%20%20%20%20import%20altair%20as%20alt%0A%20%20%20%20import%20marimo%20as%20mo%0A%20%20%20%20import%20matplotlib.pyplot%20as%20plt%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20polars%20as%20pl%0A%20%20%20%20from%20scipy.stats%20import%20gaussian_kde%2C%20pearsonr%2C%20spearmanr%0A%20%20%20%20from%20sklearn.metrics%20import%20mean_absolute_error%2C%20mean_squared_error%2C%20r2_score%0A%0A%20%20%20%20from%20rdkit%20import%20Chem%2C%20RDLogger%0A%20%20%20%20from%20rdkit.Chem%20import%20rdDepictor%0A%20%20%20%20from%20rdkit.Chem.Draw%20import%20rdMolDraw2D%0A%0A%20%20%20%20RDLogger.DisableLog(%22rdApp.*%22)%0A%0A%20%20%20%20%23%20Directory%20where%20every%20figure%20produced%20by%20this%20notebook%20is%20written.%0A%20%20%20%20PLOT_DIR%20%3D%20Path(%22..%2Fplots%2F7_unblinded_phase2_analysis%22)%0A%20%20%20%20PLOT_DIR.mkdir(parents%3DTrue%2C%20exist_ok%3DTrue)%0A%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20Chem%2C%0A%20%20%20%20%20%20%20%20Optional%2C%0A%20%20%20%20%20%20%20%20PLOT_DIR%2C%0A%20%20%20%20%20%20%20%20Path%2C%0A%20%20%20%20%20%20%20%20alt%2C%0A%20%20%20%20%20%20%20%20gaussian_kde%2C%0A%20%20%20%20%20%20%20%20math%2C%0A%20%20%20%20%20%20%20%20mean_absolute_error%2C%0A%20%20%20%20%20%20%20%20mean_squared_error%2C%0A%20%20%20%20%20%20%20%20mo%2C%0A%20%20%20%20%20%20%20%20np%2C%0A%20%20%20%20%20%20%20%20pearsonr%2C%0A%20%20%20%20%20%20%20%20pl%2C%0A%20%20%20%20%20%20%20%20plt%2C%0A%20%20%20%20%20%20%20%20r2_score%2C%0A%20%20%20%20%20%20%20%20rdDepictor%2C%0A%20%20%20%20%20%20%20%20rdMolDraw2D%2C%0A%20%20%20%20%20%20%20%20spearmanr%2C%0A%20%20%20%20)%0A%0A%0A%40app.cell%0Adef%20_(Chem%2C%20Optional%2C%20rdDepictor%2C%20rdMolDraw2D)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Molecule%20drawing%20%2F%20identifier%20helpers%20(reused%20from%20notebook%205)%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%0A%20%20%20%20def%20smi_to_svg(smi%3A%20str%2C%20width%3A%20int%20%3D%20280%2C%20height%3A%20int%20%3D%20200)%20-%3E%20str%3A%0A%20%20%20%20%20%20%20%20%22%22%22Render%20a%20SMILES%20string%20as%20an%20SVG%20image%20via%20RDKit%20MolDraw2DSVG.%22%22%22%0A%20%20%20%20%20%20%20%20mol%20%3D%20Chem.MolFromSmiles(smi)%0A%20%20%20%20%20%20%20%20if%20mol%20is%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20%22%22%0A%20%20%20%20%20%20%20%20rdDepictor.Compute2DCoords(mol)%0A%20%20%20%20%20%20%20%20drawer%20%3D%20rdMolDraw2D.MolDraw2DSVG(width%2C%20height)%0A%20%20%20%20%20%20%20%20drawer.DrawMolecule(mol)%0A%20%20%20%20%20%20%20%20drawer.FinishDrawing()%0A%20%20%20%20%20%20%20%20return%20drawer.GetDrawingText()%0A%0A%20%20%20%20def%20strip_xml_decl(svg%3A%20str)%20-%3E%20str%3A%0A%20%20%20%20%20%20%20%20%22%22%22Remove%20the%20%3C%3Fxml%20...%20%3F%3E%20declaration%20so%20SVG%20embeds%20cleanly%20in%20HTML.%22%22%22%0A%20%20%20%20%20%20%20%20return%20svg.split(%22%3F%3E%22%2C%201)%5B-1%5D.strip()%20if%20%22%3F%3E%22%20in%20svg%20else%20svg%0A%0A%20%20%20%20def%20smi_to_inchikey(smi%3A%20str)%20-%3E%20Optional%5Bstr%5D%3A%0A%20%20%20%20%20%20%20%20%22%22%22Return%20the%20InChIKey%20for%20*smi*%2C%20or%20None%20if%20the%20SMILES%20cannot%20be%20parsed.%22%22%22%0A%20%20%20%20%20%20%20%20mol%20%3D%20Chem.MolFromSmiles(smi)%0A%20%20%20%20%20%20%20%20return%20Chem.MolToInchiKey(mol)%20if%20mol%20else%20None%0A%0A%20%20%20%20def%20smi_to_inchi(smi%3A%20str)%20-%3E%20Optional%5Bstr%5D%3A%0A%20%20%20%20%20%20%20%20%22%22%22Return%20the%20InChI%20for%20*smi*%2C%20or%20None%20if%20the%20SMILES%20cannot%20be%20parsed.%22%22%22%0A%20%20%20%20%20%20%20%20mol%20%3D%20Chem.MolFromSmiles(smi)%0A%20%20%20%20%20%20%20%20return%20Chem.MolToInchi(mol)%20if%20mol%20else%20None%0A%0A%20%20%20%20return%20smi_to_inchi%2C%20smi_to_inchikey%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Load%20unblinded%20test%20sets%0A%0A%20%20%20%20Both%20the%20Phase%201%20and%20Phase%202%20unblinded%20CSVs%20share%20the%20same%20column%20layout.%20%20We%0A%20%20%20%20standardise%20the%20verbose%20column%20names%20(same%20renaming%20used%20in%20notebook%205)%20and%20add%0A%20%20%20%20InChIKey%20%2F%20InChI%20identifiers%20so%20we%20can%20join%20against%20the%20processed%20training%20data.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(Path%2C%20pl%2C%20smi_to_inchi%2C%20smi_to_inchikey)%3A%0A%20%20%20%20%23%20Long%20assay%20column%20names%20are%20identical%20between%20Phase%201%20and%20Phase%202%20%E2%80%94%20factor%20the%0A%20%20%20%20%23%20rename%20map%20out%20so%20both%20sets%20are%20standardised%20the%20same%20way.%0A%20%20%20%20_RENAME_MAP%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22pEC50_std.error%20(-log10(molarity))%22%3A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22pEC50_se%22%2C%0A%20%20%20%20%20%20%20%20%22pEC50_ci.lower%20(-log10(molarity))%22%3A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22pEC50_ci_lower%22%2C%0A%20%20%20%20%20%20%20%20%22pEC50_ci.upper%20(-log10(molarity))%22%3A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22pEC50_ci_upper%22%2C%0A%20%20%20%20%20%20%20%20%22Emax_estimate%20(log2FC%20vs.%20baseline)%22%3A%20%20%20%20%20%20%20%20%20%20%20%20%20%22Emax%22%2C%0A%20%20%20%20%20%20%20%20%22Emax_std.error%20(log2FC%20vs.%20baseline)%22%3A%20%20%20%20%20%20%20%20%20%20%20%20%22Emax_se%22%2C%0A%20%20%20%20%20%20%20%20%22Emax_ci.lower%20(log2FC%20vs.%20baseline)%22%3A%20%20%20%20%20%20%20%20%20%20%20%20%20%22Emax_ci_lower%22%2C%0A%20%20%20%20%20%20%20%20%22Emax_ci.upper%20(log2FC%20vs.%20baseline)%22%3A%20%20%20%20%20%20%20%20%20%20%20%20%20%22Emax_ci_upper%22%2C%0A%20%20%20%20%20%20%20%20%22Emax.vs.pos.ctrl_estimate%20(dimensionless)%22%3A%20%20%20%20%20%20%20%22Emax_vs_ctrl%22%2C%0A%20%20%20%20%20%20%20%20%22Emax.vs.pos.ctrl_std.error%20(dimensionless)%22%3A%20%20%20%20%20%20%22Emax_vs_ctrl_se%22%2C%0A%20%20%20%20%20%20%20%20%22Emax.vs.pos.ctrl_ci.lower%20(dimensionless)%22%3A%20%20%20%20%20%20%20%22Emax_vs_ctrl_ci_lower%22%2C%0A%20%20%20%20%20%20%20%20%22Emax.vs.pos.ctrl_ci.upper%20(dimensionless)%22%3A%20%20%20%20%20%20%20%22Emax_vs_ctrl_ci_upper%22%2C%0A%20%20%20%20%7D%0A%0A%20%20%20%20def%20load_unblinded(path%3A%20Path)%20-%3E%20pl.DataFrame%3A%0A%20%20%20%20%20%20%20%20%22%22%22Read%20an%20unblinded%20CSV%2C%20add%20identifiers%2C%20and%20standardise%20column%20names.%22%22%22%0A%20%20%20%20%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.read_csv(path)%0A%20%20%20%20%20%20%20%20%20%20%20%20.rename(%7B%22SMILES%22%3A%20%22smiles%22%7D)%0A%20%20%20%20%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22smiles%22).map_elements(smi_to_inchikey%2C%20return_dtype%3Dpl.Utf8).alias(%22inchikey%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22smiles%22).map_elements(smi_to_inchi%2C%20%20%20%20return_dtype%3Dpl.Utf8).alias(%22inchi%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20.rename(_RENAME_MAP)%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%23%20Phase%202%20is%20the%20focus%20of%20this%20notebook%3B%20Phase%201%20is%20loaded%20only%20for%20the%0A%20%20%20%20%23%20three-way%20distribution%20comparison.%0A%20%20%20%20PHASE2_PATH%20%3D%20Path(%22..%2Fdata%2Fraw%2F20260703%2Fpxr-challenge_TEST_PHASE_2_UNBLINDED.csv%22)%0A%20%20%20%20PHASE1_PATH%20%3D%20Path(%22..%2Fdata%2Fraw%2F20260528%2Fdose_response_test_unblinded.csv%22)%0A%0A%20%20%20%20unblinded%3A%20pl.DataFrame%20%3D%20load_unblinded(PHASE2_PATH)%20%20%20%20%20%20%23%20Phase%202%20(main)%0A%20%20%20%20unblinded_p1%3A%20pl.DataFrame%20%3D%20load_unblinded(PHASE1_PATH)%20%20%20%23%20Phase%201%20(reference)%0A%0A%20%20%20%20print(%22Phase%202%20shape%3A%22%2C%20unblinded.shape)%0A%20%20%20%20print(%22Phase%201%20shape%3A%22%2C%20unblinded_p1.shape)%0A%20%20%20%20unblinded%0A%20%20%20%20return%20unblinded%2C%20unblinded_p1%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Dataset%20overview%20%E2%80%94%20three-way%20activity%20distribution%0A%0A%20%20%20%20The%20dose-response%20pEC50%20describes%20potency%20on%20a%20%E2%88%92log10(molarity)%20scale%3A%20higher%0A%20%20%20%20values%20mean%20more%20potent%20compounds.%20%20We%20overlay%20three%20distributions%20on%20a%20single%0A%20%20%20%20axis%20%E2%80%94%20the%20**training%20set**%2C%20the%20**Phase%201**%20unblinded%20test%20set%2C%20and%20the%0A%20%20%20%20**Phase%202**%20unblinded%20test%20set%20%E2%80%94%20to%20check%20whether%20all%20three%20probe%20the%20same%0A%20%20%20%20activity%20range.%0A%0A%20%20%20%20A%20compound%20is%20typically%20classified%20as%20a%20**hit**%20when%20pEC50%20%E2%89%A5%206%20(EC%E2%82%85%E2%82%80%20%E2%89%A4%201%20%C2%B5M)%0A%20%20%20%20together%20with%20a%20positive%20Emax%3B%20the%20dashed%20line%20marks%20the%20pEC50%20%3D%206%20threshold.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20PLOT_DIR%2C%0A%20%20%20%20gaussian_kde%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20np%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20plt%2C%0A%20%20%20%20unblinded%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20unblinded_p1%3A%20%22pl.DataFrame%22%2C%0A)%3A%0A%20%20%20%20%23%20Load%20training%20pEC50%20values%20for%20comparison.%0A%20%20%20%20all_compounds%20%3D%20pl.read_csv(%22..%2Fdata%2Fprocessed%2Fall_compounds_activity_data.csv%22)%0A%0A%20%20%20%20train_pec50%20%3D%20(%0A%20%20%20%20%20%20%20%20all_compounds%0A%20%20%20%20%20%20%20%20.filter(pl.col(%22in_dose_response%22)%20%26%20pl.col(%22pEC50_dr%22).is_not_null())%0A%20%20%20%20%20%20%20%20.get_column(%22pEC50_dr%22)%0A%20%20%20%20%20%20%20%20.to_numpy()%0A%20%20%20%20)%0A%20%20%20%20p1_pec50%20%3D%20unblinded_p1.get_column(%22pEC50%22).to_numpy()%0A%20%20%20%20p2_pec50%20%3D%20unblinded.get_column(%22pEC50%22).to_numpy()%0A%0A%20%20%20%20%23%20Common%20x-axis%20spanning%20all%20three%20distributions.%0A%20%20%20%20_lo%20%3D%20min(train_pec50.min()%2C%20p1_pec50.min()%2C%20p2_pec50.min())%20-%200.3%0A%20%20%20%20_hi%20%3D%20max(train_pec50.max()%2C%20p1_pec50.max()%2C%20p2_pec50.max())%20%2B%200.3%0A%20%20%20%20x_range%20%3D%20np.linspace(_lo%2C%20_hi%2C%20400)%0A%0A%20%20%20%20%23%20(label%2C%20values%2C%20colour)%20for%20each%20series.%0A%20%20%20%20_series%20%3D%20%5B%0A%20%20%20%20%20%20%20%20(f%22Training%20(n%3D%7Blen(train_pec50)%3A%2C%7D)%22%2C%20%20%20%20%20%20%20%20train_pec50%2C%20%22%234e79a7%22)%2C%0A%20%20%20%20%20%20%20%20(f%22Phase%201%20%E2%80%94%20unblinded%20(n%3D%7Blen(p1_pec50)%3A%2C%7D)%22%2C%20p1_pec50%2C%20%20%20%20%22%2359a14f%22)%2C%0A%20%20%20%20%20%20%20%20(f%22Phase%202%20%E2%80%94%20unblinded%20(n%3D%7Blen(p2_pec50)%3A%2C%7D)%22%2C%20p2_pec50%2C%20%20%20%20%22%23e15759%22)%2C%0A%20%20%20%20%5D%0A%0A%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20fig_dist%2C%20ax_dist%20%3D%20plt.subplots(figsize%3D(7.5%2C%204.8)%2C%20dpi%3D150)%0A%20%20%20%20%20%20%20%20for%20_lbl%2C%20_vals%2C%20_col%20in%20_series%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_kde%20%3D%20gaussian_kde(_vals%2C%20bw_method%3D%22scott%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20ax_dist.plot(x_range%2C%20_kde(x_range)%2C%20color%3D_col%2C%20linewidth%3D2%2C%20label%3D_lbl)%0A%20%20%20%20%20%20%20%20%20%20%20%20ax_dist.fill_between(x_range%2C%20_kde(x_range)%2C%20alpha%3D0.12%2C%20color%3D_col)%0A%20%20%20%20%20%20%20%20ax_dist.axvline(6.0%2C%20color%3D%22black%22%2C%20linestyle%3D%22--%22%2C%20linewidth%3D1.2%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20label%3D%22Hit%20threshold%20(pEC50%20%3D%206)%22)%0A%20%20%20%20%20%20%20%20ax_dist.set_xlabel(%22pEC50%20%20%5B%E2%88%92log%E2%82%81%E2%82%80(M)%5D%22%2C%20fontsize%3D12)%0A%20%20%20%20%20%20%20%20ax_dist.set_ylabel(%22Density%22%2C%20fontsize%3D12)%0A%20%20%20%20%20%20%20%20ax_dist.set_title(%22pEC50%20distribution%20%E2%80%94%20training%20vs.%20Phase%201%20vs.%20Phase%202%22%2C%20fontsize%3D13)%0A%20%20%20%20%20%20%20%20ax_dist.legend(fontsize%3D10%2C%20frameon%3DTrue%2C%20framealpha%3D0.9)%0A%20%20%20%20%20%20%20%20fig_dist.tight_layout()%0A%20%20%20%20%20%20%20%20fig_dist.savefig(PLOT_DIR%20%2F%20%22pec50_distribution_train_p1_p2.png%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20mo.center(mo.as_html(fig_dist))%0A%20%20%20%20return%20all_compounds%2C%20p1_pec50%2C%20p2_pec50%2C%20train_pec50%0A%0A%0A%40app.cell%0Adef%20_(mo%2C%20np%2C%20p1_pec50%2C%20p2_pec50%2C%20train_pec50)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Per-set%20summary%20statistics%20table%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20def%20_stats(vals%3A%20%22np.ndarray%22)%20-%3E%20dict%3A%0A%20%20%20%20%20%20%20%20%22%22%22Summary%20statistics%20for%20one%20pEC50%20array%2C%20incl.%20hit%20fraction%20(pEC50%20%E2%89%A5%206).%22%22%22%0A%20%20%20%20%20%20%20%20return%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22n%22%3A%20%20%20%20%20%20%20len(vals)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22mean%22%3A%20%20%20%20float(np.mean(vals))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22median%22%3A%20%20float(np.median(vals))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22std%22%3A%20%20%20%20%20float(np.std(vals))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22min%22%3A%20%20%20%20%20float(np.min(vals))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22max%22%3A%20%20%20%20%20float(np.max(vals))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22pct_hit%22%3A%20100.0%20*%20float(np.mean(vals%20%3E%3D%206.0))%2C%0A%20%20%20%20%20%20%20%20%7D%0A%0A%20%20%20%20_t%20%3D%20_stats(train_pec50)%0A%20%20%20%20_p1%20%3D%20_stats(p1_pec50)%0A%20%20%20%20_p2%20%3D%20_stats(p2_pec50)%0A%0A%20%20%20%20mo.md(f%22%22%22%0A%20%20%20%20%23%23%23%20Distribution%20summary%0A%0A%20%20%20%20%7C%20Statistic%20%7C%20Training%20%7C%20Phase%201%20%7C%20Phase%202%20%7C%0A%20%20%20%20%7C---%7C---%7C---%7C---%7C%0A%20%20%20%20%7C%20n%20compounds%20%7C%20%7B_t%5B'n'%5D%3A%2C%7D%20%7C%20%7B_p1%5B'n'%5D%3A%2C%7D%20%7C%20%7B_p2%5B'n'%5D%3A%2C%7D%20%7C%0A%20%20%20%20%7C%20Mean%20pEC50%20%7C%20%7B_t%5B'mean'%5D%3A.3f%7D%20%7C%20%7B_p1%5B'mean'%5D%3A.3f%7D%20%7C%20%7B_p2%5B'mean'%5D%3A.3f%7D%20%7C%0A%20%20%20%20%7C%20Median%20pEC50%20%7C%20%7B_t%5B'median'%5D%3A.3f%7D%20%7C%20%7B_p1%5B'median'%5D%3A.3f%7D%20%7C%20%7B_p2%5B'median'%5D%3A.3f%7D%20%7C%0A%20%20%20%20%7C%20Std%20pEC50%20%7C%20%7B_t%5B'std'%5D%3A.3f%7D%20%7C%20%7B_p1%5B'std'%5D%3A.3f%7D%20%7C%20%7B_p2%5B'std'%5D%3A.3f%7D%20%7C%0A%20%20%20%20%7C%20Min%20pEC50%20%7C%20%7B_t%5B'min'%5D%3A.3f%7D%20%7C%20%7B_p1%5B'min'%5D%3A.3f%7D%20%7C%20%7B_p2%5B'min'%5D%3A.3f%7D%20%7C%0A%20%20%20%20%7C%20Max%20pEC50%20%7C%20%7B_t%5B'max'%5D%3A.3f%7D%20%7C%20%7B_p1%5B'max'%5D%3A.3f%7D%20%7C%20%7B_p2%5B'max'%5D%3A.3f%7D%20%7C%0A%20%20%20%20%7C%20Hit%20fraction%20(pEC50%20%E2%89%A5%206)%20%7C%20%7B_t%5B'pct_hit'%5D%3A.1f%7D%25%20%7C%20%7B_p1%5B'pct_hit'%5D%3A.1f%7D%25%20%7C%20%7B_p2%5B'pct_hit'%5D%3A.1f%7D%25%20%7C%0A%0A%20%20%20%20**Reading%20the%20table%3A**%20a%20large%20gap%20between%20the%20training%20mean%2Fmedian%20and%20the%0A%20%20%20%20Phase%202%20mean%2Fmedian%2C%20or%20a%20much%20smaller%20Phase%202%20hit%20fraction%2C%20would%20signal%20that%0A%20%20%20%20the%20final%20test%20set%20is%20shifted%20toward%20weaker%20compounds%20than%20the%20models%20were%0A%20%20%20%20trained%20on%20%E2%80%94%20a%20covariate%20shift%20that%20inflates%20test%20error%20relative%20to%0A%20%20%20%20cross-validation.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Final%20submission%20%E2%80%94%20reference%20predictions%0A%0A%20%20%20%20The%20activity-cliff%20scatter%20below%20is%20coloured%20by%20the%20prediction%20error%20of%20our%0A%20%20%20%20**final%20submitted%20model**%2C%20%606_ens_default_augfilt_submission.csv%60.%20%20We%20load%20it%0A%20%20%20%20once%20here%20and%20join%20it%20to%20the%20Phase%202%20truth%20so%20the%20residual%20is%20available%20to%20the%0A%20%20%20%20structural%20analysis%3B%20the%20full%20model%20comparison%20follows%20later.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(Path%2C%20pl%2C%20unblinded%3A%20%22pl.DataFrame%22)%3A%0A%20%20%20%20%23%20Ground-truth%20table%3A%20Molecule%20Name%20%E2%86%92%20true%20Phase%202%20pEC50.%0A%20%20%20%20truth%3A%20pl.DataFrame%20%3D%20unblinded.select(%5B%22Molecule%20Name%22%2C%20%22pEC50%22%5D).rename(%0A%20%20%20%20%20%20%20%20%7B%22pEC50%22%3A%20%22pEC50_true%22%7D%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Path%20to%20the%20model%20we%20actually%20submitted%20as%20our%20final%20entry.%0A%20%20%20%20FINAL_SUBMISSION%20%3D%20Path(%22..%2Fsubmissions%2F6_ens_default_augfilt_submission.csv%22)%0A%0A%20%20%20%20%23%20Residuals%20of%20the%20final%20submission%20on%20the%20260%20Phase%202%20compounds.%0A%20%20%20%20final_resid%3A%20pl.DataFrame%20%3D%20(%0A%20%20%20%20%20%20%20%20pl.read_csv(FINAL_SUBMISSION)%0A%20%20%20%20%20%20%20%20.select(%5B%22Molecule%20Name%22%2C%20%22pEC50%22%5D)%0A%20%20%20%20%20%20%20%20.rename(%7B%22pEC50%22%3A%20%22final_pred%22%7D)%0A%20%20%20%20%20%20%20%20.join(truth%2C%20on%3D%22Molecule%20Name%22%2C%20how%3D%22inner%22)%0A%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22final_pred%22)%20-%20pl.col(%22pEC50_true%22)).alias(%22final_residual%22)%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20)%0A%0A%20%20%20%20print(%22Final%20submission%20residuals%20joined%20for%22%2C%20final_resid.shape%5B0%5D%2C%20%22compounds%22)%0A%20%20%20%20return%20final_resid%2C%20truth%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Nearest-neighbour%20similarity%20to%20training%20set%20and%20activity%20cliffs%0A%0A%20%20%20%20Before%20evaluating%20model%20performance%2C%20we%20examine%20structural%20similarity%20between%20each%0A%20%20%20%20**Phase%202**%20test%20compound%20and%20its%20closest%20analogue%20in%20the%20dose-response%20training%0A%20%20%20%20set%2C%20then%20ask%20whether%20structural%20similarity%20predicts%20activity%20similarity.%0A%0A%20%20%20%20An%20**activity%20cliff**%20is%20a%20pair%20of%20structurally%20similar%20compounds%20with%20substantially%0A%20%20%20%20different%20potency%20%E2%80%94%20they%20are%20notoriously%20difficult%20for%20ML%20models%20because%20a%20small%0A%20%20%20%20structural%20change%20produces%20a%20large%20activity%20change.%0A%0A%20%20%20%20Definitions%20used%20here%20(identical%20to%20notebook%205)%3A%0A%0A%20%20%20%20%7C%20Criterion%20%7C%20Threshold%20%7C%0A%20%20%20%20%7C---%7C---%7C%0A%20%20%20%20%7C%20Structurally%20similar%20%7C%20ECFP4%20Tanimoto%20%E2%89%A5%200.4%20%7C%0A%20%20%20%20%7C%20Activity%20cliff%20%7C%20Similar%20**and**%20%5C%7C%CE%94pEC50%5C%7C%20%E2%89%A5%201.0%20log%20unit%20%7C%0A%0A%20%20%20%20For%20each%20Phase%202%20compound%20we%20report%20its%20nearest%20neighbour%20(NN)%20in%20the%20training%20set%2C%0A%20%20%20%20its%20Tanimoto%20similarity%2C%20and%20the%20activity%20difference%20between%20the%20two.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20Chem%2C%0A%20%20%20%20all_compounds%2C%0A%20%20%20%20np%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20unblinded%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20unblinded_p1%3A%20%22pl.DataFrame%22%2C%0A)%3A%0A%20%20%20%20from%20rdkit.Chem%20import%20AllChem%0A%20%20%20%20from%20rdkit%20import%20DataStructs%20as%20_DataStructs%0A%0A%20%20%20%20%23%20Training%20set%20restricted%20to%20compounds%20with%20a%20measured%20pEC50.%0A%20%20%20%20_train_df%20%3D%20(%0A%20%20%20%20%20%20%20%20all_compounds%0A%20%20%20%20%20%20%20%20.filter(pl.col(%22in_dose_response%22)%20%26%20pl.col(%22pEC50_dr%22).is_not_null())%0A%20%20%20%20%20%20%20%20.unique(subset%3D%5B%22inchikey%22%5D)%0A%20%20%20%20%20%20%20%20.select(%5B%22inchikey%22%2C%20%22smiles%22%2C%20%22molecule_names%22%2C%20%22pEC50_dr%22%5D)%0A%20%20%20%20)%0A%0A%20%20%20%20def%20_ecfp4(smi%3A%20str)%3A%0A%20%20%20%20%20%20%20%20%22%22%22Return%20ECFP4%20BitVect%20(radius%202%2C%201024%20bits)%20for%20a%20SMILES%20string.%22%22%22%0A%20%20%20%20%20%20%20%20mol%20%3D%20Chem.MolFromSmiles(smi)%0A%20%20%20%20%20%20%20%20if%20mol%20is%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20None%0A%20%20%20%20%20%20%20%20return%20AllChem.GetMorganFingerprintAsBitVect(mol%2C%20radius%3D2%2C%20nBits%3D1024)%0A%0A%20%20%20%20%23%20Precompute%20the%20training-set%20fingerprints%20once%20%E2%80%94%20reused%20for%20both%20test%20phases.%0A%20%20%20%20_train_fps%20%20%20%20%3D%20%5B_ecfp4(s)%20for%20s%20in%20_train_df%5B%22smiles%22%5D.to_list()%5D%0A%20%20%20%20_train_ids%20%20%20%20%3D%20_train_df%5B%22inchikey%22%5D.to_list()%0A%20%20%20%20_train_smiles%20%3D%20_train_df%5B%22smiles%22%5D.to_list()%0A%20%20%20%20_train_names%20%20%3D%20_train_df%5B%22molecule_names%22%5D.to_list()%0A%20%20%20%20_train_pec50s%20%3D%20_train_df%5B%22pEC50_dr%22%5D.to_list()%0A%0A%20%20%20%20def%20build_nn_cliff(test_df%3A%20pl.DataFrame)%20-%3E%20pl.DataFrame%3A%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20For%20every%20compound%20in%20*test_df*%2C%20find%20its%20nearest%20ECFP4%20training%20neighbour%20and%0A%20%20%20%20%20%20%20%20classify%20the%20pair%20as%20an%20activity%20cliff%2C%20similar%2Fconcordant%2C%20or%20dissimilar.%0A%0A%20%20%20%20%20%20%20%20Same%20thresholds%20as%20notebook%205%3A%20similar%20%3D%20Tanimoto%20%E2%89%A5%200.4%3B%20cliff%20%3D%20similar%20and%0A%20%20%20%20%20%20%20%20%7C%CE%94pEC50%7C%20%E2%89%A5%201.0.%20%20Works%20for%20any%20test%20frame%20with%20the%20standard%20unblinded%20columns.%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20_test_fps%20%3D%20%5B_ecfp4(s)%20for%20s%20in%20test_df%5B%22smiles%22%5D.to_list()%5D%0A%20%20%20%20%20%20%20%20_max_sims%3A%20list%5Bfloat%5D%20%3D%20%5B%5D%0A%20%20%20%20%20%20%20%20_nn_idx%3A%20%20%20list%5Bint%5D%20%20%20%3D%20%5B%5D%0A%20%20%20%20%20%20%20%20for%20_fp%20in%20_test_fps%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20_fp%20is%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_max_sims.append(float(%22nan%22))%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_nn_idx.append(0)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20continue%0A%20%20%20%20%20%20%20%20%20%20%20%20_sims%20%3D%20_DataStructs.BulkTanimotoSimilarity(_fp%2C%20_train_fps)%0A%20%20%20%20%20%20%20%20%20%20%20%20_best%20%3D%20int(np.argmax(_sims))%0A%20%20%20%20%20%20%20%20%20%20%20%20_max_sims.append(float(_sims%5B_best%5D))%0A%20%20%20%20%20%20%20%20%20%20%20%20_nn_idx.append(_best)%0A%0A%20%20%20%20%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20test_df%0A%20%20%20%20%20%20%20%20%20%20%20%20.select(%5B%22Molecule%20Name%22%2C%20%22smiles%22%2C%20%22inchikey%22%2C%20%22pEC50%22%2C%20%22Emax%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.Series(%22nn_sim%22%2C%20%20%20%20%20%20%20%20%20_max_sims%2C%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20dtype%3Dpl.Float32)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.Series(%22nn_inchikey%22%2C%20%20%20%20%5B_train_ids%5Bi%5D%20%20%20%20for%20i%20in%20_nn_idx%5D%2C%20%20dtype%3Dpl.Utf8)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.Series(%22nn_smiles%22%2C%20%20%20%20%20%20%5B_train_smiles%5Bi%5D%20for%20i%20in%20_nn_idx%5D%2C%20%20dtype%3Dpl.Utf8)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.Series(%22nn_name%22%2C%20%20%20%20%20%20%20%20%5B_train_names%5Bi%5D%20%20for%20i%20in%20_nn_idx%5D%2C%20%20dtype%3Dpl.Utf8)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.Series(%22nn_pEC50_train%22%2C%20%5B_train_pec50s%5Bi%5D%20for%20i%20in%20_nn_idx%5D%2C%20%20dtype%3Dpl.Float64)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22pEC50%22)%20-%20pl.col(%22nn_pEC50_train%22)).alias(%22delta_pEC50%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22pEC50%22)%20-%20pl.col(%22nn_pEC50_train%22)).abs().alias(%22abs_delta_pEC50%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.when(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22nn_sim%22)%20%3E%3D%200.4)%20%26%20(pl.col(%22abs_delta_pEC50%22)%20%3E%3D%201.0)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.then(pl.lit(%22Activity%20cliff%22))%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.when(pl.col(%22nn_sim%22)%20%3E%3D%200.4)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.then(pl.lit(%22Similar%20%2F%20concordant%22))%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.otherwise(pl.lit(%22Dissimilar%22))%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.alias(%22pair_class%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%23%20Phase%202%20is%20the%20focus%20(drives%20the%20scatter%20%2B%20interactive%20browser)%3B%20Phase%201%20is%0A%20%20%20%20%23%20computed%20only%20so%20the%20summary%20table%20can%20show%20it%20as%20a%20comparison%20column.%0A%20%20%20%20nn_cliff_df%3A%20%20%20%20pl.DataFrame%20%3D%20build_nn_cliff(unblinded)%0A%20%20%20%20nn_cliff_df_p1%3A%20pl.DataFrame%20%3D%20build_nn_cliff(unblinded_p1)%0A%0A%20%20%20%20nn_cliff_df%0A%20%20%20%20return%20nn_cliff_df%2C%20nn_cliff_df_p1%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20PLOT_DIR%2C%0A%20%20%20%20final_resid%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20nn_cliff_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20np%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20plt%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Static%20scatter%20%E2%80%94%20coloured%20by%20our%20final%20submission's%20residual%20error%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_cliff_with_res%20%3D%20nn_cliff_df.join(%0A%20%20%20%20%20%20%20%20final_resid.select(%5B%22Molecule%20Name%22%2C%20%22final_residual%22%5D)%2C%0A%20%20%20%20%20%20%20%20on%3D%22Molecule%20Name%22%2C%20how%3D%22left%22%2C%0A%20%20%20%20)%0A%0A%20%20%20%20_residuals%20%3D%20_cliff_with_res%5B%22final_residual%22%5D.to_numpy()%0A%20%20%20%20_abs_max%20%3D%20max(abs(np.nanmin(_residuals))%2C%20abs(np.nanmax(_residuals)))%0A%0A%20%20%20%20%23%20Marker%20style%20per%20pair_class%20so%20structural%20context%20is%20preserved.%0A%20%20%20%20_marker_map%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22Activity%20cliff%22%3A%20%20%20%20%20%20%20(%22%5E%22%2C%2055%2C%201.0)%2C%0A%20%20%20%20%20%20%20%20%22Similar%20%2F%20concordant%22%3A%20(%22o%22%2C%2030%2C%200.80)%2C%0A%20%20%20%20%20%20%20%20%22Dissimilar%22%3A%20%20%20%20%20%20%20%20%20%20%20(%22s%22%2C%2025%2C%200.65)%2C%0A%20%20%20%20%7D%0A%0A%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20_fig_cliff%2C%20_ax_cliff%20%3D%20plt.subplots(figsize%3D(7.5%2C%205.5)%2C%20dpi%3D150)%0A%0A%20%20%20%20%20%20%20%20for%20_cls%2C%20(_mkr%2C%20_sz%2C%20_alpha)%20in%20_marker_map.items()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_sub%20%3D%20_cliff_with_res.filter(pl.col(%22pair_class%22)%20%3D%3D%20_cls)%0A%20%20%20%20%20%20%20%20%20%20%20%20_sc%20%3D%20_ax_cliff.scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_sub%5B%22nn_sim%22%5D.to_numpy()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_sub%5B%22abs_delta_pEC50%22%5D.to_numpy()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20c%3D_sub%5B%22final_residual%22%5D.to_numpy()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20cmap%3D%22RdBu_r%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20vmin%3D-_abs_max%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20vmax%3D_abs_max%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20marker%3D_mkr%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20s%3D_sz%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alpha%3D_alpha%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20edgecolors%3D%22none%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20label%3D_cls%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20_cb%20%3D%20_fig_cliff.colorbar(_sc%2C%20ax%3D_ax_cliff%2C%20fraction%3D0.035%2C%20pad%3D0.02)%0A%20%20%20%20%20%20%20%20_cb.set_label(%22Final%20submission%20residual%20%20(pred%20%E2%88%92%20true%20pEC50)%22%2C%20fontsize%3D10)%0A%0A%20%20%20%20%20%20%20%20_ax_cliff.axvline(0.4%2C%20color%3D%22%23555%22%2C%20linestyle%3D%22--%22%2C%20linewidth%3D1%2C%20alpha%3D0.7%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20label%3D%22Similarity%20threshold%20(0.4)%22)%0A%20%20%20%20%20%20%20%20_ax_cliff.axhline(1.0%2C%20color%3D%22%23555%22%2C%20linestyle%3D%22%3A%22%2C%20%20linewidth%3D1%2C%20alpha%3D0.7%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20label%3D%22%7C%CE%94pEC50%7C%20threshold%20(1.0)%22)%0A%0A%20%20%20%20%20%20%20%20_n_cliff_static%20%3D%20nn_cliff_df.filter(pl.col(%22pair_class%22)%20%3D%3D%20%22Activity%20cliff%22).shape%5B0%5D%0A%20%20%20%20%20%20%20%20_ax_cliff.set_xlabel(%22ECFP4%20Tanimoto%20similarity%20to%20nearest%20training%20neighbour%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax_cliff.set_ylabel(%22%7C%CE%94pEC50%7C%20%20(test%20pEC50%20%E2%88%92%20nearest%20train%20pEC50)%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax_cliff.set_title(%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22Phase%202%20activity%20cliff%20analysis%20%20%E2%80%94%20%20%7B_n_cliff_static%7D%20cliff(s)%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Colour%20%3D%20final-model%20residual%20(red%20%3D%20overpredicted%2C%20blue%20%3D%20underpredicted)%3B%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%22markers%3A%20%E2%96%B3%20cliff%2C%20%E2%97%8B%20similar%2Fconcordant%2C%20%E2%96%A1%20dissimilar%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D10%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20_ax_cliff.legend(fontsize%3D8%2C%20frameon%3DTrue%2C%20framealpha%3D0.9%2C%20loc%3D%22upper%20left%22)%0A%20%20%20%20%20%20%20%20_ax_cliff.set_xlim(-0.02%2C%201.02)%0A%20%20%20%20%20%20%20%20_fig_cliff.tight_layout()%0A%20%20%20%20%20%20%20%20_fig_cliff.savefig(%0A%20%20%20%20%20%20%20%20%20%20%20%20PLOT_DIR%20%2F%20%22activity_cliffs_scatter.png%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20dpi%3D300%2C%20bbox_inches%3D%22tight%22%2C%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20mo.center(mo.as_html(_fig_cliff))%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo%2C%20nn_cliff_df%3A%20%22pl.DataFrame%22%2C%20nn_cliff_df_p1%3A%20%22pl.DataFrame%22%2C%20pl)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Summary%20statistics%20%E2%80%94%20Phase%201%20vs%20Phase%202%20side%20by%20side%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20def%20_nn_summary(df%3A%20pl.DataFrame)%20-%3E%20dict%3A%0A%20%20%20%20%20%20%20%20%22%22%22Nearest-neighbour%20cliff%20summary%20counts%2Fstats%20for%20one%20test-set%20frame.%22%22%22%0A%20%20%20%20%20%20%20%20_n%20%3D%20df.shape%5B0%5D%0A%20%20%20%20%20%20%20%20return%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22n%22%3A%20%20%20%20%20%20%20%20%20_n%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22n_similar%22%3A%20df.filter(pl.col(%22nn_sim%22)%20%3E%3D%200.4).shape%5B0%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22n_cliffs%22%3A%20%20df.filter(pl.col(%22pair_class%22)%20%3D%3D%20%22Activity%20cliff%22).shape%5B0%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22n_concord%22%3A%20df.filter(pl.col(%22pair_class%22)%20%3D%3D%20%22Similar%20%2F%20concordant%22).shape%5B0%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22n_dissim%22%3A%20%20df.filter(pl.col(%22pair_class%22)%20%3D%3D%20%22Dissimilar%22).shape%5B0%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22sim_mean%22%3A%20%20float(df%5B%22nn_sim%22%5D.mean())%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22sim_med%22%3A%20%20%20float(df%5B%22nn_sim%22%5D.median())%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22delt_mean%22%3A%20float(df%5B%22abs_delta_pEC50%22%5D.mean())%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22delt_med%22%3A%20%20float(df%5B%22abs_delta_pEC50%22%5D.median())%2C%0A%20%20%20%20%20%20%20%20%7D%0A%0A%20%20%20%20def%20_cnt_pct(k%3A%20str%2C%20s%3A%20dict)%20-%3E%20str%3A%0A%20%20%20%20%20%20%20%20%22%22%22Render%20a%20'count%20(pct%25)'%20cell%20for%20count-style%20metrics.%22%22%22%0A%20%20%20%20%20%20%20%20return%20f%22%7Bs%5Bk%5D%7D%20(%7B100%20*%20s%5Bk%5D%20%2F%20s%5B'n'%5D%3A.1f%7D%25)%22%0A%0A%20%20%20%20_p1%20%3D%20_nn_summary(nn_cliff_df_p1)%0A%20%20%20%20_p2%20%3D%20_nn_summary(nn_cliff_df)%0A%0A%20%20%20%20mo.md(f%22%22%22%0A%20%20%20%20%23%23%23%20Activity%20cliff%20summary%20%E2%80%94%20nearest-neighbour%20view%0A%0A%20%20%20%20Phase%201%20is%20shown%20alongside%20Phase%202%20for%20comparison%20(both%20measured%20against%20the%20same%0A%20%20%20%20dose-response%20training%20set%2C%20identical%20thresholds).%0A%0A%20%20%20%20%7C%20Metric%20%7C%20Phase%201%20%7C%20Phase%202%20%7C%0A%20%20%20%20%7C---%7C---%7C---%7C%0A%20%20%20%20%7C%20Test%20compounds%20analysed%20%7C%20%7B_p1%5B'n'%5D%7D%20%7C%20%7B_p2%5B'n'%5D%7D%20%7C%0A%20%20%20%20%7C%20Similar%20pairs%20(Tanimoto%20%E2%89%A5%200.4)%20%7C%20%7B_cnt_pct('n_similar'%2C%20_p1)%7D%20%7C%20%7B_cnt_pct('n_similar'%2C%20_p2)%7D%20%7C%0A%20%20%20%20%7C%20**Activity%20cliffs**%20(similar%20%2B%20%5C%5C%7C%CE%94pEC50%5C%5C%7C%20%E2%89%A5%201.0)%20%7C%20**%7B_cnt_pct('n_cliffs'%2C%20_p1)%7D**%20%7C%20**%7B_cnt_pct('n_cliffs'%2C%20_p2)%7D**%20%7C%0A%20%20%20%20%7C%20Similar%20%2F%20concordant%20%7C%20%7B_cnt_pct('n_concord'%2C%20_p1)%7D%20%7C%20%7B_cnt_pct('n_concord'%2C%20_p2)%7D%20%7C%0A%20%20%20%20%7C%20Dissimilar%20(Tanimoto%20%3C%200.4)%20%7C%20%7B_cnt_pct('n_dissim'%2C%20_p1)%7D%20%7C%20%7B_cnt_pct('n_dissim'%2C%20_p2)%7D%20%7C%0A%20%20%20%20%7C%20Mean%20Tanimoto%20sim%20to%20NN%20%7C%20%7B_p1%5B'sim_mean'%5D%3A.3f%7D%20%7C%20%7B_p2%5B'sim_mean'%5D%3A.3f%7D%20%7C%0A%20%20%20%20%7C%20Median%20Tanimoto%20sim%20to%20NN%20%7C%20%7B_p1%5B'sim_med'%5D%3A.3f%7D%20%7C%20%7B_p2%5B'sim_med'%5D%3A.3f%7D%20%7C%0A%20%20%20%20%7C%20Mean%20%5C%5C%7C%CE%94pEC50%5C%5C%7C%20(all%20pairs)%20%7C%20%7B_p1%5B'delt_mean'%5D%3A.3f%7D%20%7C%20%7B_p2%5B'delt_mean'%5D%3A.3f%7D%20%7C%0A%20%20%20%20%7C%20Median%20%5C%5C%7C%CE%94pEC50%5C%5C%7C%20(all%20pairs)%20%7C%20%7B_p1%5B'delt_med'%5D%3A.3f%7D%20%7C%20%7B_p2%5B'delt_med'%5D%3A.3f%7D%20%7C%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%23%20Interactive%20browser%20%E2%80%94%20Phase%202%20compound%20and%20nearest%20training%20neighbour%0A%0A%20%20%20%20Hover%20over%20a%20point%20in%20the%20scatter%20to%20display%20both%20compound%20structures%20and%0A%20%20%20%20their%20activity%20data%20side%20by%20side.%20%20Red%20points%20(activity%20cliffs)%20are%20the%0A%20%20%20%20most%20informative%3A%20structurally%20similar%20to%20a%20training%20compound%20but%20with%0A%20%20%20%20substantially%20different%20potency%20%E2%80%94%20a%20direct%20challenge%20for%20ML%20models.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(alt%2C%20mo%2C%20nn_cliff_df%3A%20%22pl.DataFrame%22)%3A%0A%20%20%20%20_cliff_color_scale%20%3D%20alt.Scale(%0A%20%20%20%20%20%20%20%20domain%3D%5B%22Activity%20cliff%22%2C%20%22Similar%20%2F%20concordant%22%2C%20%22Dissimilar%22%5D%2C%0A%20%20%20%20%20%20%20%20range%3D%5B%22%23e15759%22%2C%20%22%2376b7b2%22%2C%20%22%23b0b0b0%22%5D%2C%0A%20%20%20%20)%0A%0A%20%20%20%20_cliff_scatter%20%3D%20(%0A%20%20%20%20%20%20%20%20alt.Chart(nn_cliff_df)%0A%20%20%20%20%20%20%20%20.mark_circle(opacity%3D0.85%2C%20size%3D70)%0A%20%20%20%20%20%20%20%20.encode(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dalt.X(%22nn_sim%3AQ%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20title%3D%22Tanimoto%20similarity%20to%20nearest%20training%20compound%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20scale%3Dalt.Scale(domain%3D%5B0%2C%201%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20axis%3Dalt.Axis(titleFontSize%3D11))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Dalt.Y(%22abs_delta_pEC50%3AQ%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20title%3D%22%7C%CE%94pEC50%7C%20%20(test%20%E2%88%92%20nearest%20train)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20axis%3Dalt.Axis(titleFontSize%3D11))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20color%3Dalt.Color(%22pair_class%3AN%22%2C%20scale%3D_cliff_color_scale%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20legend%3Dalt.Legend(title%3D%22Pair%20class%22))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20tooltip%3D%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22Molecule%20Name%3AN%22%2C%20%20%20title%3D%22Test%20compound%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22pEC50%3AQ%22%2C%20%20%20%20%20%20%20%20%20%20%20title%3D%22Test%20pEC50%22%2C%20%20%20%20%20format%3D%22.3f%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22nn_name%3AN%22%2C%20%20%20%20%20%20%20%20%20title%3D%22NN%20train%20cmpd%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22nn_pEC50_train%3AQ%22%2C%20%20title%3D%22Train%20NN%20pEC50%22%2C%20format%3D%22.3f%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22nn_sim%3AQ%22%2C%20%20%20%20%20%20%20%20%20%20title%3D%22Tanimoto%22%2C%20%20%20%20%20%20%20format%3D%22.3f%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22abs_delta_pEC50%3AQ%22%2C%20title%3D%22%7C%CE%94pEC50%7C%22%2C%20%20%20%20%20%20%20format%3D%22.3f%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22pair_class%3AN%22%2C%20%20%20%20%20%20title%3D%22Class%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20.properties(title%3D%22Nearest-neighbour%20similarity%20vs.%20prediction%20gap%22%2C%20width%3D600%2C%20height%3D400)%0A%20%20%20%20%20%20%20%20.configure_title(fontSize%3D12)%0A%20%20%20%20)%0A%0A%20%20%20%20mo.center(mo.as_html(_cliff_scatter))%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Matched%20molecular%20pair%20(MMP)%20analysis%20%E2%80%94%20training%20vs.%20Phase%202%20test%20set%0A%0A%20%20%20%20Matched%20molecular%20pairs%20are%20compound%20pairs%20that%20differ%20by%20exactly%20one%20structural%0A%20%20%20%20transformation%20at%20a%20single%20site%20while%20sharing%20a%20common%20scaffold%20(the%20**constant**%0A%20%20%20%20fragment).%20%20Because%20the%20change%20is%20chemically%20precise%2C%20MMP%20analysis%20lets%20us%0A%20%20%20%20attribute%20an%20activity%20difference%20directly%20to%20a%20specific%20structural%20modification.%0A%0A%20%20%20%20The%20MMP%20database%20was%20pre-generated%20across%20the%20full%20compound%20collection.%20%20Here%20we%0A%20%20%20%20focus%20only%20on%20**cross-set%20pairs**%3A%20one%20compound%20from%20the%20dose-response%20training%20set%0A%20%20%20%20and%20one%20from%20the%20Phase%202%20unblinded%20test%20set.%20%20This%20directly%20reveals%20how%20often%0A%20%20%20%20structural%20similarity%20(at%20the%20MMP%20level)%20corresponds%20to%20activity%20similarity%20%E2%80%94%20and%2C%0A%20%20%20%20crucially%2C%20how%20often%20it%20does%20not%20(activity%20cliffs).%0A%0A%20%20%20%20Activity%20cliff%20criterion%3A%20**%7C%CE%94pEC50%7C%20%E2%89%A5%201.0**%20log%20unit%20between%20the%20two%20MMP%20partners.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20all_compounds%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20unblinded%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20unblinded_p1%3A%20%22pl.DataFrame%22%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Load%20MMP%20database%20and%20orient%20cross-set%20pairs%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_mmp%20%3D%20pl.read_csv(%0A%20%20%20%20%20%20%20%20%22..%2Fdata%2Fprocessed%2Fall_compounds_mmp.mmp.csv.gz%22%2C%0A%20%20%20%20%20%20%20%20separator%3D%22%5Ct%22%2C%0A%20%20%20%20%20%20%20%20has_header%3DFalse%2C%0A%20%20%20%20%20%20%20%20new_columns%3D%5B%22smiles1%22%2C%20%22smiles2%22%2C%20%22inchikey1%22%2C%20%22inchikey2%22%2C%20%22transformation%22%2C%20%22constant%22%5D%2C%0A%20%20%20%20)%0A%0A%20%20%20%20_train_act%20%3D%20(%0A%20%20%20%20%20%20%20%20all_compounds%0A%20%20%20%20%20%20%20%20.filter(pl.col(%22in_dose_response%22)%20%26%20pl.col(%22pEC50_dr%22).is_not_null())%0A%20%20%20%20%20%20%20%20.unique(%22inchikey%22)%0A%20%20%20%20%20%20%20%20.select(%5B%22inchikey%22%2C%20%22smiles%22%2C%20%22molecule_names%22%2C%20%22pEC50_dr%22%5D)%0A%20%20%20%20)%0A%20%20%20%20_train_iks%20%3D%20set(_train_act%5B%22inchikey%22%5D.to_list())%0A%0A%20%20%20%20def%20build_mmp_cross(test_df%3A%20pl.DataFrame)%20-%3E%20pl.DataFrame%3A%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20Extract%20cross-set%20matched%20molecular%20pairs%20%E2%80%94%20one%20training%20compound%2C%20one%20test%0A%20%20%20%20%20%20%20%20compound%20%E2%80%94%20for%20*test_df*%2C%20and%20classify%20each%20pair%20as%20an%20activity%20cliff%0A%20%20%20%20%20%20%20%20(%7C%CE%94pEC50%7C%20%E2%89%A5%201.0)%20or%20concordant.%20%20Reusable%20across%20test%20phases.%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20_test_act%20%3D%20test_df.select(%5B%22Molecule%20Name%22%2C%20%22smiles%22%2C%20%22inchikey%22%2C%20%22pEC50%22%5D)%0A%20%20%20%20%20%20%20%20_test_iks%20%3D%20set(_test_act%5B%22inchikey%22%5D.to_list())%0A%0A%20%20%20%20%20%20%20%20%23%20Orient%20so%20ik_train%20is%20always%20the%20training%20compound%2C%20ik_test%20the%20test%20compound.%0A%20%20%20%20%20%20%20%20_a%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20_mmp%0A%20%20%20%20%20%20%20%20%20%20%20%20.filter(pl.col(%22inchikey1%22).is_in(_train_iks)%20%26%20pl.col(%22inchikey2%22).is_in(_test_iks))%0A%20%20%20%20%20%20%20%20%20%20%20%20.select(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22inchikey1%22).alias(%22ik_train%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22inchikey2%22).alias(%22ik_test%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22transformation%22%2C%20%22constant%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D)%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20_b%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20_mmp%0A%20%20%20%20%20%20%20%20%20%20%20%20.filter(pl.col(%22inchikey2%22).is_in(_train_iks)%20%26%20pl.col(%22inchikey1%22).is_in(_test_iks))%0A%20%20%20%20%20%20%20%20%20%20%20%20.select(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22inchikey2%22).alias(%22ik_train%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22inchikey1%22).alias(%22ik_test%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22transformation%22%2C%20%22constant%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D)%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.concat(%5B_a%2C%20_b%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20.join(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_train_act.rename(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22inchikey%22%3A%20%20%20%20%20%20%20%22ik_train%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22smiles%22%3A%20%20%20%20%20%20%20%20%20%22smiles_train%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22molecule_names%22%3A%20%22name_train%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22pEC50_dr%22%3A%20%20%20%20%20%20%20%22pEC50_train%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20on%3D%22ik_train%22%2C%20how%3D%22left%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20.join(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_test_act.rename(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22inchikey%22%3A%20%20%20%20%20%20%22ik_test%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22smiles%22%3A%20%20%20%20%20%20%20%20%22smiles_test%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Molecule%20Name%22%3A%20%22name_test%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22pEC50%22%3A%20%20%20%20%20%20%20%20%20%22pEC50_test%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20on%3D%22ik_test%22%2C%20how%3D%22left%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20.filter(pl.col(%22pEC50_train%22).is_not_null()%20%26%20pl.col(%22pEC50_test%22).is_not_null())%0A%20%20%20%20%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22pEC50_test%22)%20-%20pl.col(%22pEC50_train%22)).alias(%22delta_pEC50%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22pEC50_test%22)%20-%20pl.col(%22pEC50_train%22)).abs().alias(%22abs_delta_pEC50%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.when(pl.col(%22abs_delta_pEC50%22)%20%3E%3D%201.0)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.then(pl.lit(%22Activity%20cliff%22))%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.otherwise(pl.lit(%22Concordant%22))%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.alias(%22pair_class%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%23%20Phase%202%20drives%20the%20figures%20below%3B%20Phase%201%20is%20built%20only%20for%20the%20summary%20column.%0A%20%20%20%20mmp_cross_df%3A%20%20%20%20pl.DataFrame%20%3D%20build_mmp_cross(unblinded)%0A%20%20%20%20mmp_cross_df_p1%3A%20pl.DataFrame%20%3D%20build_mmp_cross(unblinded_p1)%0A%0A%20%20%20%20%23%20Cluster-level%20summary%3A%20one%20row%20per%20unique%20scaffold%20constant%20(Phase%202).%0A%20%20%20%20mmp_cluster_df%3A%20pl.DataFrame%20%3D%20(%0A%20%20%20%20%20%20%20%20mmp_cross_df%0A%20%20%20%20%20%20%20%20.group_by(%22constant%22)%0A%20%20%20%20%20%20%20%20.agg(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.len().alias(%22n_pairs%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22abs_delta_pEC50%22).max().alias(%22max_abs_delta%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22abs_delta_pEC50%22).mean().alias(%22mean_abs_delta%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22abs_delta_pEC50%22).median().alias(%22median_abs_delta%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22abs_delta_pEC50%22)%20%3E%3D%201.0).sum().alias(%22n_cliffs%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22ik_test%22).n_unique().alias(%22n_test_cmpds%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22ik_train%22).n_unique().alias(%22n_train_cmpds%22)%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22n_cliffs%22)%20%2F%20pl.col(%22n_pairs%22)%20*%20100).round(1).alias(%22pct_cliffs%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20Readable%20label%3A%20constant%20fragment%20truncated%20to%2040%20chars.%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22constant%22).str.slice(0%2C%2040).alias(%22constant_label%22)%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20.sort(%22max_abs_delta%22%2C%20descending%3DTrue)%0A%20%20%20%20)%0A%0A%20%20%20%20print(f%22Cross-set%20MMPs%3A%20%7Bmmp_cross_df.shape%5B0%5D%7D%22)%0A%20%20%20%20print(f%22Activity%20cliffs%3A%20%7Bmmp_cross_df.filter(pl.col('pair_class')%3D%3D'Activity%20cliff').shape%5B0%5D%7D%22)%0A%20%20%20%20print(f%22Unique%20scaffold%20clusters%3A%20%7Bmmp_cluster_df.shape%5B0%5D%7D%22)%0A%20%20%20%20print(f%22Clusters%20with%20%E2%89%A51%20cliff%3A%20%7Bmmp_cluster_df.filter(pl.col('n_cliffs')%3E%3D1).shape%5B0%5D%7D%22)%0A%20%20%20%20mmp_cross_df%0A%20%20%20%20return%20mmp_cluster_df%2C%20mmp_cross_df%2C%20mmp_cross_df_p1%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20PLOT_DIR%2C%0A%20%20%20%20gaussian_kde%2C%0A%20%20%20%20mmp_cluster_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20mmp_cross_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20np%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20plt%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Two-panel%20overview%20figure%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_cliff%20%20%20%3D%20mmp_cross_df.filter(pl.col(%22pair_class%22)%20%3D%3D%20%22Activity%20cliff%22)%5B%22abs_delta_pEC50%22%5D.to_numpy()%0A%20%20%20%20_concord%20%3D%20mmp_cross_df.filter(pl.col(%22pair_class%22)%20%3D%3D%20%22Concordant%22)%5B%22abs_delta_pEC50%22%5D.to_numpy()%0A%20%20%20%20_all%20%20%20%20%20%3D%20mmp_cross_df%5B%22abs_delta_pEC50%22%5D.to_numpy()%0A%0A%20%20%20%20%23%20Top%2020%20clusters%20for%20the%20bar%20panel%20(sorted%20by%20max%20%7C%CE%94pEC50%7C).%0A%20%20%20%20_top20%20%3D%20mmp_cluster_df.head(20)%0A%0A%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20_fig_mmp%2C%20(_ax_kde%2C%20_ax_bar)%20%3D%20plt.subplots(%0A%20%20%20%20%20%20%20%20%20%20%20%201%2C%202%2C%20figsize%3D(13%2C%205)%2C%20dpi%3D150%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20gridspec_kw%3D%7B%22width_ratios%22%3A%20%5B1%2C%201.4%5D%7D%2C%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20%23%20%E2%94%80%E2%94%80%20Left%20panel%3A%20KDE%20of%20%7C%CE%94pEC50%7C%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%20%20%20%20_x%20%3D%20np.linspace(0%2C%20_all.max()%20%2B%200.3%2C%20400)%0A%20%20%20%20%20%20%20%20_kde_all%20%20%20%3D%20gaussian_kde(_all%2C%20%20%20%20bw_method%3D%22scott%22)%0A%20%20%20%20%20%20%20%20_ax_kde.plot(_x%2C%20_kde_all(_x)%2C%20%20%20%20color%3D%22%23555%22%2C%20linewidth%3D2%2C%20label%3Df%22All%20MMPs%20(n%3D%7Blen(_all)%7D)%22)%0A%20%20%20%20%20%20%20%20_ax_kde.fill_between(_x%2C%20_kde_all(_x)%2C%20alpha%3D0.10%2C%20color%3D%22%23555%22)%0A%0A%20%20%20%20%20%20%20%20if%20len(_cliff)%20%3E%3D%202%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_kde_cliff%20%3D%20gaussian_kde(_cliff%2C%20bw_method%3D%22scott%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_kde.plot(_x%2C%20_kde_cliff(_x)%2C%20color%3D%22%23e15759%22%2C%20linewidth%3D2%2C%20label%3Df%22Cliffs%20(n%3D%7Blen(_cliff)%7D)%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_kde.fill_between(_x%2C%20_kde_cliff(_x)%2C%20alpha%3D0.15%2C%20color%3D%22%23e15759%22)%0A%0A%20%20%20%20%20%20%20%20if%20len(_concord)%20%3E%3D%202%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_kde_con%20%3D%20gaussian_kde(_concord%2C%20bw_method%3D%22scott%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_kde.plot(_x%2C%20_kde_con(_x)%2C%20color%3D%22%2376b7b2%22%2C%20linewidth%3D2%2C%20label%3Df%22Concordant%20(n%3D%7Blen(_concord)%7D)%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_kde.fill_between(_x%2C%20_kde_con(_x)%2C%20alpha%3D0.15%2C%20color%3D%22%2376b7b2%22)%0A%0A%20%20%20%20%20%20%20%20_ax_kde.axvline(1.0%2C%20color%3D%22black%22%2C%20linestyle%3D%22--%22%2C%20linewidth%3D1%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20label%3D%22Cliff%20threshold%20(1.0)%22)%0A%20%20%20%20%20%20%20%20_ax_kde.set_xlabel(%22%7C%CE%94pEC50%7C%20%20(test%20%E2%88%92%20train)%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax_kde.set_ylabel(%22Density%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax_kde.set_xlim(0%2C%20_all.max()%20%2B%200.2)%0A%20%20%20%20%20%20%20%20_ax_kde.legend(fontsize%3D9%2C%20frameon%3DTrue)%0A%20%20%20%20%20%20%20%20_ax_kde.set_title(%22%7C%CE%94pEC50%7C%20distribution%20%E2%80%94%20all%20cross-set%20MMPs%20(Phase%202)%22%2C%20fontsize%3D11)%0A%0A%20%20%20%20%20%20%20%20%23%20%E2%94%80%E2%94%80%20Right%20panel%3A%20top%2020%20clusters%2C%20bar%20%3D%20max%20%7C%CE%94pEC50%7C%2C%20coloured%20by%20pct_cliffs%20%E2%94%80%E2%94%80%0A%20%20%20%20%20%20%20%20_cluster_labels%20%3D%20%5Bf%22C%7Bi%2B1%7D%22%20for%20i%20in%20range(len(_top20))%5D%0A%20%20%20%20%20%20%20%20_max_deltas%20%20%20%20%20%3D%20_top20%5B%22max_abs_delta%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_pct_cliffs%20%20%20%20%20%3D%20_top20%5B%22pct_cliffs%22%5D.to_numpy()%0A%0A%20%20%20%20%20%20%20%20_cmap%20%20%20%3D%20plt.cm.RdYlGn_r%0A%20%20%20%20%20%20%20%20_norm%20%20%20%3D%20plt.Normalize(vmin%3D0%2C%20vmax%3D100)%0A%20%20%20%20%20%20%20%20_colors%20%3D%20%5B_cmap(_norm(v))%20for%20v%20in%20_pct_cliffs%5D%0A%0A%20%20%20%20%20%20%20%20_bars%20%3D%20_ax_bar.barh(%0A%20%20%20%20%20%20%20%20%20%20%20%20_cluster_labels%5B%3A%3A-1%5D%2C%20_max_deltas%5B%3A%3A-1%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20color%3D_colors%5B%3A%3A-1%5D%2C%20edgecolor%3D%22white%22%2C%20linewidth%3D0.5%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20_ax_bar.axvline(1.0%2C%20color%3D%22black%22%2C%20linestyle%3D%22--%22%2C%20linewidth%3D1%2C%20alpha%3D0.7)%0A%0A%20%20%20%20%20%20%20%20%23%20Annotate%20bars%20with%20n_pairs%20and%20cliff%20counts.%0A%20%20%20%20%20%20%20%20for%20_i%2C%20(_bar%2C%20_row)%20in%20enumerate(zip(_bars%2C%20list(_top20.iter_rows(named%3DTrue))%5B%3A%3A-1%5D))%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_bar.text(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_bar.get_width()%20%2B%200.05%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_bar.get_y()%20%2B%20_bar.get_height()%20%2F%202%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22n%3D%7B_row%5B'n_pairs'%5D%7D%20%7C%20%7B_row%5B'n_cliffs'%5D%7D%20cliffs%20(%7B_row%5B'pct_cliffs'%5D%3A.0f%7D%25)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20va%3D%22center%22%2C%20fontsize%3D7.5%2C%20color%3D%22%23333%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20_sm%20%3D%20plt.cm.ScalarMappable(cmap%3D_cmap%2C%20norm%3D_norm)%0A%20%20%20%20%20%20%20%20_sm.set_array(%5B%5D)%0A%20%20%20%20%20%20%20%20_cbar%20%3D%20_fig_mmp.colorbar(_sm%2C%20ax%3D_ax_bar%2C%20fraction%3D0.03%2C%20pad%3D0.02)%0A%20%20%20%20%20%20%20%20_cbar.set_label(%22%25%20pairs%20that%20are%20cliffs%22%2C%20fontsize%3D9)%0A%0A%20%20%20%20%20%20%20%20_ax_bar.set_xlabel(%22Max%20%7C%CE%94pEC50%7C%20in%20cluster%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax_bar.set_title(%22Top%2020%20MMP%20scaffold%20clusters%5Cn(sorted%20by%20max%20%7C%CE%94pEC50%7C)%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax_bar.set_xlim(0%2C%20_max_deltas.max()%20%2B%201.2)%0A%0A%20%20%20%20%20%20%20%20_fig_mmp.tight_layout()%0A%20%20%20%20%20%20%20%20_fig_mmp.savefig(%0A%20%20%20%20%20%20%20%20%20%20%20%20PLOT_DIR%20%2F%20%22mmp_activity_cliffs.png%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20dpi%3D300%2C%20bbox_inches%3D%22tight%22%2C%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20mo.center(mo.as_html(_fig_mmp))%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20mmp_cluster_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20mmp_cross_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20mmp_cross_df_p1%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20pl%2C%0A)%3A%0A%20%20%20%20def%20_mmp_pair_summary(df%3A%20pl.DataFrame)%20-%3E%20dict%3A%0A%20%20%20%20%20%20%20%20%22%22%22Pair-level%20MMP%20cliff%20summary%20for%20one%20cross-set%20frame%20(phase-agnostic).%22%22%22%0A%20%20%20%20%20%20%20%20_n%20%3D%20df.shape%5B0%5D%0A%20%20%20%20%20%20%20%20_ad%20%3D%20df%5B%22abs_delta_pEC50%22%5D%0A%20%20%20%20%20%20%20%20return%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22n%22%3A%20%20%20%20%20%20%20%20_n%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22n_cliffs%22%3A%20df.filter(pl.col(%22pair_class%22)%20%3D%3D%20%22Activity%20cliff%22).shape%5B0%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22n_concord%22%3A%20df.filter(pl.col(%22pair_class%22)%20%3D%3D%20%22Concordant%22).shape%5B0%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22q50%22%3A%20float(_ad.quantile(0.50))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22q75%22%3A%20float(_ad.quantile(0.75))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22q90%22%3A%20float(_ad.quantile(0.90))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22q95%22%3A%20float(_ad.quantile(0.95))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22max%22%3A%20float(_ad.max())%2C%0A%20%20%20%20%20%20%20%20%7D%0A%0A%20%20%20%20_p1%20%3D%20_mmp_pair_summary(mmp_cross_df_p1)%0A%20%20%20%20_p2%20%3D%20_mmp_pair_summary(mmp_cross_df)%0A%0A%20%20%20%20%23%20Scaffold-cluster%20figures%20are%20Phase%202%20only%20(clusters%20not%20computed%20for%20Phase%201).%0A%20%20%20%20_n_clusters%20%20%20%20%20%20%20%3D%20mmp_cluster_df.shape%5B0%5D%0A%20%20%20%20_n_cliff_clusters%20%3D%20mmp_cluster_df.filter(pl.col(%22n_cliffs%22)%20%3E%3D%201).shape%5B0%5D%0A%0A%20%20%20%20def%20_cliff_pct(k%3A%20str%2C%20s%3A%20dict)%20-%3E%20str%3A%0A%20%20%20%20%20%20%20%20return%20f%22%7Bs%5Bk%5D%7D%20(%7B100%20*%20s%5Bk%5D%20%2F%20s%5B'n'%5D%3A.1f%7D%25)%22%0A%0A%20%20%20%20mo.md(f%22%22%22%0A%20%20%20%20%23%23%23%20MMP%20summary%0A%0A%20%20%20%20Phase%201%20is%20shown%20alongside%20Phase%202%20for%20comparison%20(both%20cross-set%20against%20the%20same%0A%20%20%20%20training%20set).%20%20The%20scaffold-cluster%20rows%20are%20Phase%202%20only.%0A%0A%20%20%20%20%7C%20Metric%20%7C%20Phase%201%20%7C%20Phase%202%20%7C%0A%20%20%20%20%7C---%7C---%7C---%7C%0A%20%20%20%20%7C%20Total%20cross-set%20MMP%20pairs%20%7C%20%7B_p1%5B'n'%5D%7D%20%7C%20%7B_p2%5B'n'%5D%7D%20%7C%0A%20%20%20%20%7C%20**Activity%20cliffs**%20(%5C%5C%7C%CE%94pEC50%5C%5C%7C%20%E2%89%A5%201.0)%20%7C%20**%7B_cliff_pct('n_cliffs'%2C%20_p1)%7D**%20%7C%20**%7B_cliff_pct('n_cliffs'%2C%20_p2)%7D**%20%7C%0A%20%20%20%20%7C%20Concordant%20pairs%20%7C%20%7B_cliff_pct('n_concord'%2C%20_p1)%7D%20%7C%20%7B_cliff_pct('n_concord'%2C%20_p2)%7D%20%7C%0A%20%20%20%20%7C%20Median%20%5C%5C%7C%CE%94pEC50%5C%5C%7C%20%7C%20%7B_p1%5B'q50'%5D%3A.2f%7D%20%7C%20%7B_p2%5B'q50'%5D%3A.2f%7D%20%7C%0A%20%20%20%20%7C%2075th%20percentile%20%5C%5C%7C%CE%94pEC50%5C%5C%7C%20%7C%20%7B_p1%5B'q75'%5D%3A.2f%7D%20%7C%20%7B_p2%5B'q75'%5D%3A.2f%7D%20%7C%0A%20%20%20%20%7C%2090th%20percentile%20%5C%5C%7C%CE%94pEC50%5C%5C%7C%20%7C%20%7B_p1%5B'q90'%5D%3A.2f%7D%20%7C%20%7B_p2%5B'q90'%5D%3A.2f%7D%20%7C%0A%20%20%20%20%7C%2095th%20percentile%20%5C%5C%7C%CE%94pEC50%5C%5C%7C%20%7C%20%7B_p1%5B'q95'%5D%3A.2f%7D%20%7C%20%7B_p2%5B'q95'%5D%3A.2f%7D%20%7C%0A%20%20%20%20%7C%20Maximum%20%5C%5C%7C%CE%94pEC50%5C%5C%7C%20%7C%20%7B_p1%5B'max'%5D%3A.2f%7D%20%7C%20%7B_p2%5B'max'%5D%3A.2f%7D%20%7C%0A%20%20%20%20%7C%20Unique%20scaffold%20clusters%20(Phase%202)%20%7C%20%E2%80%94%20%7C%20%7B_n_clusters%7D%20%7C%0A%20%20%20%20%7C%20Clusters%20with%20%E2%89%A5%201%20cliff%20(Phase%202)%20%7C%20%E2%80%94%20%7C%20%7B_n_cliff_clusters%7D%20(%7B100*_n_cliff_clusters%2F_n_clusters%3A.1f%7D%25)%20%7C%0A%0A%20%20%20%20**Interpretation%3A**%20a%20high%20cliff%20rate%20here%20means%20the%20structural%20change%20between%20each%0A%20%20%20%20MMP%20partner%20leads%20to%20a%20greater%20than%201%20log-unit%20shift%20in%20potency%20%E2%80%94%20activity-sensitive%0A%20%20%20%20structural%20space%20that%20the%20training%20data%20does%20not%20resolve%20well%2C%20and%20the%20most%20likely%0A%20%20%20%20source%20of%20large%20Phase%202%20prediction%20errors.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Model%20evaluation%20on%20the%20Phase%202%20test%20set%0A%0A%20%20%20%20%23%23%23%20Load%20every%20submission%0A%0A%20%20%20%20Unlike%20notebook%205%2C%20which%20reconstructed%20the%20ensemble%20from%20its%20component%20models%2C%0A%20%20%20%20here%20we%20evaluate%20**every%20submission%20CSV%20we%20ever%20produced**%20(all%20files%20in%0A%20%20%20%20%60submissions%2F%60%20ending%20in%20%60_submission.csv%60)%20directly%20against%20the%20Phase%202%20truth.%0A%20%20%20%20Because%20each%20submission%20covers%20the%20full%20513-compound%20test%20set%2C%20scoring%20is%20just%20an%0A%20%20%20%20inner%20join%20to%20the%20260%20Phase%202%20compounds%20%E2%80%94%20no%20prediction%20regeneration%20is%20needed.%0A%0A%20%20%20%20The%20submissions%20span%20the%20whole%20modelling%20arc%3A%0A%0A%20%20%20%20%7C%20Prefix%20%7C%20Notebook%20%7C%20Family%20%7C%0A%20%20%20%20%7C---%7C---%7C---%7C%0A%20%20%20%20%7C%20%602_%60%20%7C%20Baselines%20%7C%20CheMeleon%20single%20model%20%7C%0A%20%20%20%20%7C%20%603_%60%20%7C%20First%20optimisation%20%7C%20Chemprop%20%2F%20RF%20%2F%20early%20ensembles%20%7C%0A%20%20%20%20%7C%20%604_%60%20%7C%20Second%20optimisation%20%7C%20HPO'd%20components%20%2B%20tuned%20ensembles%20%7C%0A%20%20%20%20%7C%20%605_%60%20%7C%20Phase%201%20regen%20%7C%20Regenerated%20ensemble%20%7C%0A%20%20%20%20%7C%20%606_%60%20%7C%20Final%20optimisation%20%7C%20Augment%2Ffilter%20%2B%20calibrated%20ensembles%20%7C%0A%0A%20%20%20%20Our%20**final%20submitted%20entry**%20is%20%606_ens_default_augfilt_submission.csv%60%20%E2%80%94%20it%20is%0A%20%20%20%20flagged%20separately%20in%20every%20table%20and%20chart%20below.%20%20The%20external%20competitor%20blend%0A%20%20%20%20%60rank9_gashaw_submission_blend_751510.csv%60%20is%20scored%20too%2C%20but%20marked%20*(external)*%3A%0A%20%20%20%20it%20is%20not%20one%20of%20our%20models%20and%20is%20shown%20only%20for%20reference.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(Path%2C%20pl%2C%20truth%3A%20%22pl.DataFrame%22)%3A%0A%20%20%20%20SUBMISSION_DIR%20%3D%20Path(%22..%2Fsubmissions%22)%0A%20%20%20%20FINAL_NAME%20%3D%20%226_ens_default_augfilt_submission.csv%22%0A%20%20%20%20EXTERNAL_PATH%20%3D%20Path(%22..%2Fdata%2Fraw%2F20260703%2Frank9_gashaw_submission_blend_751510.csv%22)%0A%0A%20%20%20%20def%20_pretty_label(filename%3A%20str)%20-%3E%20str%3A%0A%20%20%20%20%20%20%20%20%22%22%22Turn%20a%20submission%20filename%20into%20a%20compact%2C%20human-readable%20model%20label.%22%22%22%0A%20%20%20%20%20%20%20%20stem%20%3D%20filename.removesuffix(%22_submission.csv%22).removesuffix(%22.csv%22)%0A%20%20%20%20%20%20%20%20return%20stem%0A%0A%20%20%20%20def%20load_and_score(path%3A%20Path%2C%20label%3A%20str%2C%20is_final%3A%20bool%2C%20is_external%3A%20bool)%20-%3E%20pl.DataFrame%3A%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20Load%20a%20submission%20CSV%20(columns%3A%20SMILES%2C%20Molecule%20Name%2C%20pEC50)%20and%20join%20with%0A%20%20%20%20%20%20%20%20the%20Phase%202%20ground%20truth%2C%20attaching%20per-row%20error%20columns%20and%20metadata.%0A%0A%20%20%20%20%20%20%20%20Returns%20a%20long-format%20frame%20with%3A%0A%20%20%20%20%20%20%20%20%20%20model_name%2C%20is_final%2C%20is_external%2C%20Molecule%20Name%2C%20pEC50_pred%2C%20pEC50_true%2C%0A%20%20%20%20%20%20%20%20%20%20error%2C%20abs_error.%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20df%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.read_csv(path)%0A%20%20%20%20%20%20%20%20%20%20%20%20.select(%5B%22Molecule%20Name%22%2C%20%22pEC50%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20.rename(%7B%22pEC50%22%3A%20%22pEC50_pred%22%7D)%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20df.join(truth%2C%20on%3D%22Molecule%20Name%22%2C%20how%3D%22inner%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.lit(label).alias(%22model_name%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.lit(is_final).alias(%22is_final%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.lit(is_external).alias(%22is_external%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22pEC50_pred%22)%20-%20pl.col(%22pEC50_true%22)).alias(%22error%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22pEC50_pred%22)%20-%20pl.col(%22pEC50_true%22)).abs().alias(%22abs_error%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%23%20Every%20local%20submission%2C%20sorted%20by%20filename%20(chronological%20by%20notebook%20prefix).%0A%20%20%20%20_sub_paths%20%3D%20sorted(SUBMISSION_DIR.glob(%22*_submission.csv%22))%0A%0A%20%20%20%20_frames%20%3D%20%5B%0A%20%20%20%20%20%20%20%20load_and_score(%0A%20%20%20%20%20%20%20%20%20%20%20%20_p%2C%20_pretty_label(_p.name)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20is_final%3D(_p.name%20%3D%3D%20FINAL_NAME)%2C%20is_external%3DFalse%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20for%20_p%20in%20_sub_paths%0A%20%20%20%20%5D%0A%20%20%20%20%23%20External%20reference%20blend.%0A%20%20%20%20_frames.append(%0A%20%20%20%20%20%20%20%20load_and_score(%0A%20%20%20%20%20%20%20%20%20%20%20%20EXTERNAL_PATH%2C%20%22rank9_gashaw_blend%20(external)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20is_final%3DFalse%2C%20is_external%3DTrue%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20)%0A%0A%20%20%20%20all_predictions%3A%20pl.DataFrame%20%3D%20pl.concat(_frames)%0A%0A%20%20%20%20print(f%22Loaded%20%7Blen(_sub_paths)%7D%20local%20submissions%20%2B%201%20external%2C%20%22%0A%20%20%20%20%20%20%20%20%20%20f%22%7Ball_predictions.shape%5B0%5D%7D%20total%20rows%20%22%0A%20%20%20%20%20%20%20%20%20%20f%22(%7Ball_predictions%5B'Molecule%20Name'%5D.n_unique()%7D%20unique%20Phase%202%20compounds)%22)%0A%20%20%20%20all_predictions%0A%20%20%20%20return%20EXTERNAL_PATH%2C%20FINAL_NAME%2C%20SUBMISSION_DIR%2C%20all_predictions%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%23%20Aggregate%20metrics%20per%20submission%0A%0A%20%20%20%20Five%20standard%20regression%20metrics%20per%20submission%2C%20plus%20signed%20bias%3A%0A%0A%20%20%20%20%7C%20Metric%20%7C%20Description%20%7C%0A%20%20%20%20%7C---%7C---%7C%0A%20%20%20%20%7C%20**RMSE**%20%7C%20Root%20mean%20squared%20error%20(lower%20%3D%20better)%20%7C%0A%20%20%20%20%7C%20**MAE**%20%7C%20Mean%20absolute%20error%20(lower%20%3D%20better)%20%7C%0A%20%20%20%20%7C%20**R%C2%B2**%20%7C%20Coefficient%20of%20determination%20(higher%20%3D%20better)%20%7C%0A%20%20%20%20%7C%20**Pearson%20r**%20%7C%20Linear%20correlation%20between%20predicted%20and%20true%20pEC50%20%7C%0A%20%20%20%20%7C%20**Spearman%20%CF%81**%20%7C%20Rank%20correlation%20(robust%20to%20outliers)%20%7C%0A%20%20%20%20%7C%20**bias**%20%7C%20Mean%20signed%20error%20(pred%20%E2%88%92%20true)%3B%20%3E0%20%3D%20overpredicts%20%7C%0A%0A%20%20%20%20The%20table%20is%20sorted%20by%20**MAE**%20ascending%20%E2%80%94%20the%20metric%20the%20challenge%20leaderboard%0A%20%20%20%20used%20to%20rank%20submissions%20%E2%80%94%20so%20the%20best%20submission%20appears%20first.%20%20The%20%60rank%60%0A%20%20%20%20column%20is%20the%20MAE%20position%20among%20our%20own%20submissions%20(the%20external%20blend%20is%0A%20%20%20%20excluded%20from%20ranking).%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20all_predictions%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20math%2C%0A%20%20%20%20mean_absolute_error%2C%0A%20%20%20%20mean_squared_error%2C%0A%20%20%20%20np%2C%0A%20%20%20%20pearsonr%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20r2_score%2C%0A%20%20%20%20spearmanr%2C%0A)%3A%0A%20%20%20%20def%20compute_metrics(group%3A%20pl.DataFrame)%20-%3E%20dict%3A%0A%20%20%20%20%20%20%20%20%22%22%22Compute%20regression%20metrics%20for%20one%20submission's%20predictions.%22%22%22%0A%20%20%20%20%20%20%20%20y_true%20%3D%20group.get_column(%22pEC50_true%22).to_numpy()%0A%20%20%20%20%20%20%20%20y_pred%20%3D%20group.get_column(%22pEC50_pred%22).to_numpy()%0A%20%20%20%20%20%20%20%20rmse%20%3D%20math.sqrt(mean_squared_error(y_true%2C%20y_pred))%0A%20%20%20%20%20%20%20%20mae%20%20%3D%20mean_absolute_error(y_true%2C%20y_pred)%0A%20%20%20%20%20%20%20%20r2%20%20%20%3D%20r2_score(y_true%2C%20y_pred)%0A%20%20%20%20%20%20%20%20r%2C%20_%20%3D%20pearsonr(y_true%2C%20y_pred)%0A%20%20%20%20%20%20%20%20rho%2C%20_%20%3D%20spearmanr(y_true%2C%20y_pred)%0A%20%20%20%20%20%20%20%20return%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22model_name%22%3A%20%20%20group.get_column(%22model_name%22)%5B0%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22is_final%22%3A%20%20%20%20%20bool(group.get_column(%22is_final%22)%5B0%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22is_external%22%3A%20%20bool(group.get_column(%22is_external%22)%5B0%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22n%22%3A%20%20%20%20%20%20%20%20%20%20%20%20len(y_true)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22RMSE%22%3A%20%20%20%20%20%20%20%20%20round(rmse%2C%204)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22MAE%22%3A%20%20%20%20%20%20%20%20%20%20round(mae%2C%20%204)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22R2%22%3A%20%20%20%20%20%20%20%20%20%20%20round(r2%2C%20%20%204)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Pearson_r%22%3A%20%20%20%20round(r%2C%20%20%20%204)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Spearman_rho%22%3A%20round(rho%2C%20%204)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22bias%22%3A%20%20%20%20%20%20%20%20%20round(float(np.mean(y_pred%20-%20y_true))%2C%204)%2C%0A%20%20%20%20%20%20%20%20%7D%0A%0A%20%20%20%20metrics_rows%20%3D%20%5B%0A%20%20%20%20%20%20%20%20compute_metrics(grp)%0A%20%20%20%20%20%20%20%20for%20grp%20in%20all_predictions.partition_by(%22model_name%22%2C%20maintain_order%3DTrue)%0A%20%20%20%20%5D%0A%0A%20%20%20%20metrics_df%3A%20pl.DataFrame%20%3D%20(%0A%20%20%20%20%20%20%20%20pl.DataFrame(metrics_rows)%0A%20%20%20%20%20%20%20%20%23%20MAE%20is%20the%20competition%20ranking%20metric%2C%20so%20sort%20and%20rank%20by%20MAE.%0A%20%20%20%20%20%20%20%20.sort(%22MAE%22)%0A%20%20%20%20%20%20%20%20%23%20Rank%20only%20among%20our%20own%20submissions%20(external%20blend%20gets%20a%20null%20rank).%0A%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.when(~pl.col(%22is_external%22))%0A%20%20%20%20%20%20%20%20%20%20%20%20.then(pl.col(%22MAE%22).rank(%22ordinal%22).over(pl.col(%22is_external%22)))%0A%20%20%20%20%20%20%20%20%20%20%20%20.otherwise(None)%0A%20%20%20%20%20%20%20%20%20%20%20%20.cast(pl.Int32)%0A%20%20%20%20%20%20%20%20%20%20%20%20.alias(%22rank%22)%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20.select(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22rank%22%2C%20%22model_name%22%2C%20%22is_final%22%2C%20%22is_external%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22MAE%22%2C%20%22RMSE%22%2C%20%22R2%22%2C%20%22Pearson_r%22%2C%20%22Spearman_rho%22%2C%20%22bias%22%2C%20%22n%22%2C%0A%20%20%20%20%20%20%20%20%5D)%0A%20%20%20%20)%0A%0A%20%20%20%20metrics_df%0A%20%20%20%20return%20(metrics_df%2C)%0A%0A%0A%40app.cell%0Adef%20_(FINAL_NAME%2C%20metrics_df%3A%20%22pl.DataFrame%22%2C%20mo%2C%20pl)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Headline%20comparison%3A%20final%20submission%20vs.%20best-possible%20submission%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_final_label%20%3D%20FINAL_NAME.removesuffix(%22_submission.csv%22)%0A%20%20%20%20_own%20%3D%20metrics_df.filter(~pl.col(%22is_external%22))%0A%0A%20%20%20%20_final_row%20%3D%20_own.filter(pl.col(%22model_name%22)%20%3D%3D%20_final_label).row(0%2C%20named%3DTrue)%0A%20%20%20%20_best_row%20%20%3D%20_own.sort(%22MAE%22).row(0%2C%20named%3DTrue)%0A%20%20%20%20_worst_row%20%3D%20_own.sort(%22MAE%22%2C%20descending%3DTrue).row(0%2C%20named%3DTrue)%0A%20%20%20%20_ext_row%20%20%20%3D%20metrics_df.filter(pl.col(%22is_external%22)).row(0%2C%20named%3DTrue)%0A%0A%20%20%20%20_n_own%20%20%20%20%20%20%3D%20_own.shape%5B0%5D%0A%20%20%20%20_n_better%20%20%20%3D%20_own.filter(pl.col(%22MAE%22)%20%3C%20_final_row%5B%22MAE%22%5D).shape%5B0%5D%0A%20%20%20%20_final_beats_best%20%3D%20_final_row%5B%22model_name%22%5D%20%3D%3D%20_best_row%5B%22model_name%22%5D%0A%0A%20%20%20%20_verdict%20%3D%20(%0A%20%20%20%20%20%20%20%20%22Our%20final%20submission%20**was%20the%20best**%20local%20submission%20by%20MAE%20(the%20competition%20%22%0A%20%20%20%20%20%20%20%20%22ranking%20metric)%20%E2%80%94%20no%20other%20submission%20would%20have%20scored%20better.%22%0A%20%20%20%20%20%20%20%20if%20_final_beats_best%20else%0A%20%20%20%20%20%20%20%20f%22**%7B_n_better%7D%20submission(s)%20would%20have%20beaten%20our%20final%20entry**%20by%20MAE%20(the%20%22%0A%20%20%20%20%20%20%20%20f%22competition%20ranking%20metric).%20The%20best%2C%20%60%7B_best_row%5B'model_name'%5D%7D%60%2C%20achieves%20%22%0A%20%20%20%20%20%20%20%20f%22MAE%3D%7B_best_row%5B'MAE'%5D%3A.4f%7D%20vs.%20our%20%7B_final_row%5B'MAE'%5D%3A.4f%7D%20%22%0A%20%20%20%20%20%20%20%20f%22(%CE%94%20%3D%20%7B_final_row%5B'MAE'%5D%20-%20_best_row%5B'MAE'%5D%3A%2B.4f%7D).%22%0A%20%20%20%20)%0A%0A%20%20%20%20mo.md(f%22%22%22%0A%20%20%20%20%23%23%23%20Headline%20result%20%E2%80%94%20did%20we%20submit%20the%20best%20model%3F%0A%0A%20%20%20%20Ranking%20metric%3A%20**MAE**%20(as%20used%20by%20the%20competition%20leaderboard).%0A%0A%20%20%20%20%7C%20Submission%20%7C%20MAE%20%7C%20RMSE%20%7C%20R%C2%B2%20%7C%20Spearman%20%CF%81%20%7C%20Rank%20(of%20%7B_n_own%7D)%20%7C%0A%20%20%20%20%7C---%7C---%7C---%7C---%7C---%7C---%7C%0A%20%20%20%20%7C%20**Final%20%E2%80%94%20%60%7B_final_row%5B'model_name'%5D%7D%60**%20%7C%20**%7B_final_row%5B'MAE'%5D%3A.4f%7D**%20%7C%20%7B_final_row%5B'RMSE'%5D%3A.4f%7D%20%7C%20%7B_final_row%5B'R2'%5D%3A.4f%7D%20%7C%20%7B_final_row%5B'Spearman_rho'%5D%3A.4f%7D%20%7C%20**%7B_final_row%5B'rank'%5D%7D**%20%7C%0A%20%20%20%20%7C%20Best%20local%20%E2%80%94%20%60%7B_best_row%5B'model_name'%5D%7D%60%20%7C%20%7B_best_row%5B'MAE'%5D%3A.4f%7D%20%7C%20%7B_best_row%5B'RMSE'%5D%3A.4f%7D%20%7C%20%7B_best_row%5B'R2'%5D%3A.4f%7D%20%7C%20%7B_best_row%5B'Spearman_rho'%5D%3A.4f%7D%20%7C%20%7B_best_row%5B'rank'%5D%7D%20%7C%0A%20%20%20%20%7C%20Worst%20local%20%E2%80%94%20%60%7B_worst_row%5B'model_name'%5D%7D%60%20%7C%20%7B_worst_row%5B'MAE'%5D%3A.4f%7D%20%7C%20%7B_worst_row%5B'RMSE'%5D%3A.4f%7D%20%7C%20%7B_worst_row%5B'R2'%5D%3A.4f%7D%20%7C%20%7B_worst_row%5B'Spearman_rho'%5D%3A.4f%7D%20%7C%20%7B_worst_row%5B'rank'%5D%7D%20%7C%0A%20%20%20%20%7C%20External%20%E2%80%94%20%60%7B_ext_row%5B'model_name'%5D%7D%60%20%7C%20%7B_ext_row%5B'MAE'%5D%3A.4f%7D%20%7C%20%7B_ext_row%5B'RMSE'%5D%3A.4f%7D%20%7C%20%7B_ext_row%5B'R2'%5D%3A.4f%7D%20%7C%20%7B_ext_row%5B'Spearman_rho'%5D%3A.4f%7D%20%7C%20%E2%80%94%20%7C%0A%0A%20%20%20%20%7B_verdict%7D%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(alt%2C%20metrics_df%3A%20%22pl.DataFrame%22%2C%20mo%2C%20pl)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Ranking%20bar%20chart%20%E2%80%94%20final%20entry%20highlighted%2C%20external%20shown%20greyed%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_chart_df%20%3D%20metrics_df.with_columns(%0A%20%20%20%20%20%20%20%20pl.when(pl.col(%22is_final%22)).then(pl.lit(%22Final%20submission%22))%0A%20%20%20%20%20%20%20%20.when(pl.col(%22is_external%22)).then(pl.lit(%22External%22))%0A%20%20%20%20%20%20%20%20.otherwise(pl.lit(%22Other%20submission%22))%0A%20%20%20%20%20%20%20%20.alias(%22kind%22)%0A%20%20%20%20)%0A%0A%20%20%20%20_color_scale%20%3D%20alt.Scale(%0A%20%20%20%20%20%20%20%20domain%3D%5B%22Final%20submission%22%2C%20%22Other%20submission%22%2C%20%22External%22%5D%2C%0A%20%20%20%20%20%20%20%20range%3D%5B%22%23e15759%22%2C%20%22%234e79a7%22%2C%20%22%23b0b0b0%22%5D%2C%0A%20%20%20%20)%0A%0A%20%20%20%20_bar%20%3D%20(%0A%20%20%20%20%20%20%20%20alt.Chart(_chart_df)%0A%20%20%20%20%20%20%20%20.mark_bar()%0A%20%20%20%20%20%20%20%20.encode(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dalt.X(%22MAE%3AQ%22%2C%20title%3D%22MAE%20%20(lower%20%3D%20better)%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Dalt.Y(%22model_name%3AN%22%2C%20sort%3Dalt.SortField(%22MAE%22%2C%20order%3D%22ascending%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20title%3DNone%2C%20axis%3Dalt.Axis(labelLimit%3D340%2C%20labelFontSize%3D9))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20color%3Dalt.Color(%22kind%3AN%22%2C%20scale%3D_color_scale%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20legend%3Dalt.Legend(title%3DNone%2C%20orient%3D%22bottom%22))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20tooltip%3D%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22rank%3AQ%22%2C%20%20%20%20%20%20%20%20%20title%3D%22Rank%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22model_name%3AN%22%2C%20%20%20title%3D%22Submission%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22MAE%3AQ%22%2C%20%20%20%20%20%20%20%20%20%20title%3D%22MAE%22%2C%20%20%20%20%20%20%20%20%20format%3D%22.4f%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22RMSE%3AQ%22%2C%20%20%20%20%20%20%20%20%20title%3D%22RMSE%22%2C%20%20%20%20%20%20%20%20format%3D%22.4f%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22R2%3AQ%22%2C%20%20%20%20%20%20%20%20%20%20%20title%3D%22R%C2%B2%22%2C%20%20%20%20%20%20%20%20%20%20format%3D%22.4f%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22Pearson_r%3AQ%22%2C%20%20%20%20title%3D%22Pearson%20r%22%2C%20%20%20format%3D%22.4f%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22Spearman_rho%3AQ%22%2C%20title%3D%22Spearman%20%CF%81%22%2C%20format%3D%22.4f%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alt.Tooltip(%22bias%3AQ%22%2C%20%20%20%20%20%20%20%20%20title%3D%22Bias%22%2C%20%20%20%20%20%20%20%20format%3D%22%2B.4f%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20.properties(%0A%20%20%20%20%20%20%20%20%20%20%20%20title%3D%22All%20submissions%20ranked%20by%20MAE%20on%20the%20Phase%202%20test%20set%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20width%3D520%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20height%3D430%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20.configure_title(fontSize%3D13)%0A%20%20%20%20)%0A%0A%20%20%20%20mo.ui.altair_chart(_bar)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Head-to-head%20statistical%20comparison%20(paired%20bootstrap%20%2B%20Holm-Bonferroni)%0A%0A%20%20%20%20A%20raw%20MAE%20gap%20between%20two%20submissions%20may%20be%20real%20or%20may%20be%20noise%20from%20the%20finite%0A%20%20%20%20test%20set.%20%20The%20challenge%20organisers%20describe%20their%20significance%20test%20in%0A%20%20%20%20%5B*Peak%20performance%20or%20just%20noise%3F*%5D(https%3A%2F%2Fopenadmet.ghost.io%2Fpeak-performance-or-just-noise%2F)%3B%0A%20%20%20%20we%20reproduce%20it%20here%20for%20our%20final%20submission%20vs.%20the%20external%20Gashaw%20blend.%0A%0A%20%20%20%20**Method%20(as%20described%20by%20the%20organisers)%3A**%0A%0A%20%20%20%201.%20**Paired%20bootstrap.**%20%20Draw%20%60n_boot%20%3D%201000%60%20pseudo-test%20sets%20by%20sampling%20the%0A%20%20%20%20%20%20%20test%20compounds%20*with%20replacement*.%20%20Crucially%2C%20the%20**same**%20resampled%20indices%0A%20%20%20%20%20%20%20are%20used%20for%20both%20participants%20in%20each%20iteration%2C%20so%20the%20comparison%20is%20paired%20%E2%80%94%0A%20%20%20%20%20%20%20%22since%20each%20bootstrap%20iteration%20uses%20the%20same%20samples%20for%20each%20user%2C%20we%20can%0A%20%20%20%20%20%20%20directly%20compare%20participants'%20performance.%22%0A%20%20%20%202.%20**Per-iteration%20metric%20and%20delta.**%20%20Compute%20each%20user's%20MAE%20on%20the%20pseudo-test%0A%20%20%20%20%20%20%20set%2C%20then%20the%20delta%20%CE%94%20%3D%20MAE(A)%20%E2%88%92%20MAE(B).%20%20This%20yields%20a%20bootstrap%20distribution%0A%20%20%20%20%20%20%20of%20%CE%94.%0A%20%20%20%203.%20**Confidence%20interval.**%20%20The%2095%25%20CI%20is%20the%202.5th%E2%80%9397.5th%20percentiles%20of%20the%20%CE%94%0A%20%20%20%20%20%20%20distribution.%20%20If%20%CE%94%20%3D%200%20falls%20**outside**%20the%20CI%20the%20difference%20is%20significant.%0A%20%20%20%204.%20**Two-tailed%20p-value.**%20%20p%20%3D%202%20%C2%B7%20min(%20P(%CE%94%20%3E%200)%2C%20P(%CE%94%20%3C%200)%20)%20%E2%80%94%20twice%20the%20smaller%0A%20%20%20%20%20%20%20tail%20fraction%20(the%20proportion%20of%20bootstrap%20iterations%20in%20which%20the%20worse%0A%20%20%20%20%20%20%20participant%20actually%20came%20out%20ahead).%0A%20%20%20%205.%20**Holm-Bonferroni%20(HB)%20correction.**%20%20Across%20all%20*M*%20pairwise%20comparisons%20in%20the%0A%20%20%20%20%20%20%20challenge%2C%20p-values%20are%20sorted%20ascending%3B%20the%20entry%20at%20rank%20*r*%20is%20compared%20to%0A%20%20%20%20%20%20%20the%20adjusted%20threshold%20**%CE%B1%20%2F%20(M%20%E2%88%92%20r%20%2B%201)**.%20%20A%20comparison%20is%20significant%20only%20if%0A%20%20%20%20%20%20%20its%20p-value%20is%20below%20its%20HB-adjusted%20threshold.%0A%0A%20%20%20%20The%20MAE%20point%20estimates%20come%20from%20the%20Phase%202%20test%20set%3B%20the%20reported%20*mean%20MAE%20%C2%B1%20SD*%0A%20%20%20%20is%20the%20mean%20and%20standard%20deviation%20of%20each%20user's%20MAE%20across%20the%201000%20bootstrap%0A%20%20%20%20iterations%2C%20which%20is%20why%20it%20differs%20slightly%20from%20the%20single-number%20MAE%20above.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(all_predictions%3A%20%22pl.DataFrame%22%2C%20np%2C%20pl)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Head-to-head%20configuration%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20Labels%20as%20they%20appear%20in%20%60all_predictions.model_name%60.%0A%20%20%20%20HH_USER_A%20%3D%20%22rank9_gashaw_blend%20(external)%22%20%20%20%23%20%22Gashaw%22%20in%20the%20organisers'%20UI%0A%20%20%20%20HH_USER_B%20%3D%20%226_ens_default_augfilt%22%20%20%20%20%20%20%20%20%20%20%20%23%20our%20final%20submission%20(%22adlvdl%22)%0A%20%20%20%20HH_LABEL_A%20%3D%20%22Gashaw%22%0A%20%20%20%20HH_LABEL_B%20%3D%20%22adlvdl%22%0A%0A%20%20%20%20N_BOOT%20%3D%201000%20%20%20%20%20%20%20%20%20%20%23%20organisers'%20standard%3A%201%2C000%20bootstrap%20iterations%0A%20%20%20%20ALPHA%20%3D%200.05%20%20%20%20%20%20%20%20%20%20%20%23%20nominal%20significance%20level%0A%20%20%20%20BOOT_SEED%20%3D%200%20%20%20%20%20%20%20%20%20%20%23%20fixed%20seed%3B%20reproduces%20the%20organisers'%20reported%20p%20%3D%200.0260%0A%0A%20%20%20%20%23%20Holm-Bonferroni%20context%20from%20the%20*full%20challenge*%20leaderboard.%20%20These%20two%20numbers%0A%20%20%20%20%23%20come%20from%20the%20organisers'%20ranking%20of%20every%20pairwise%20comparison%2C%20so%20they%20cannot%20be%0A%20%20%20%20%23%20recomputed%20from%20our%20local%20data%20alone%20%E2%80%94%20they%20are%20entered%20to%20match%20the%20reported%20UI.%0A%20%20%20%20HB_ADJUSTMENT_RANK%20%3D%202453%20%20%20%20%20%23%20this%20pair's%20rank%20among%20all%20sorted%20p-values%0A%20%20%20%20HB_TOTAL_COMPARISONS%20%3D%2095%20*%2094%20%2F%2F%202%20%20%20%23%20M%20%3D%20C(95%2C%202)%20%3D%204465%20pairwise%20comparisons%0A%20%20%20%20%23%20%20%20(95%20participants%20on%20the%20leaderboard).%20%20HB%20adjusted%20threshold%20%3D%0A%20%20%20%20%23%20%20%20ALPHA%20%2F%20(M%20-%20rank%20%2B%201)%20%3D%200.05%20%2F%202013%20%E2%89%88%200.0000249%2C%20which%20rounds%20to%200.0000%20in%20the%0A%20%20%20%20%23%20%20%20organisers'%20displayed%20table.%0A%0A%20%20%20%20def%20_abs_err(user%3A%20str)%20-%3E%20%22np.ndarray%22%3A%0A%20%20%20%20%20%20%20%20%22%22%22Per-compound%20absolute%20error%20for%20*user*%2C%20aligned%20by%20Molecule%20Name%20order.%22%22%22%0A%20%20%20%20%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20all_predictions%0A%20%20%20%20%20%20%20%20%20%20%20%20.filter(pl.col(%22model_name%22)%20%3D%3D%20user)%0A%20%20%20%20%20%20%20%20%20%20%20%20.sort(%22Molecule%20Name%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20.get_column(%22abs_error%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20.to_numpy()%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%23%20Align%20both%20users%20on%20the%20same%20compound%20ordering%20so%20the%20pairing%20is%20valid.%0A%20%20%20%20_names_a%20%3D%20(%0A%20%20%20%20%20%20%20%20all_predictions.filter(pl.col(%22model_name%22)%20%3D%3D%20HH_USER_A)%0A%20%20%20%20%20%20%20%20.sort(%22Molecule%20Name%22).get_column(%22Molecule%20Name%22).to_list()%0A%20%20%20%20)%0A%20%20%20%20_names_b%20%3D%20(%0A%20%20%20%20%20%20%20%20all_predictions.filter(pl.col(%22model_name%22)%20%3D%3D%20HH_USER_B)%0A%20%20%20%20%20%20%20%20.sort(%22Molecule%20Name%22).get_column(%22Molecule%20Name%22).to_list()%0A%20%20%20%20)%0A%20%20%20%20assert%20_names_a%20%3D%3D%20_names_b%2C%20%22Users%20must%20cover%20the%20same%20compounds%20for%20a%20paired%20test%22%0A%0A%20%20%20%20ae_a%20%3D%20_abs_err(HH_USER_A)%0A%20%20%20%20ae_b%20%3D%20_abs_err(HH_USER_B)%0A%20%20%20%20n_cmpds%20%3D%20len(ae_a)%0A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Paired%20bootstrap%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_rng%20%3D%20np.random.default_rng(BOOT_SEED)%0A%20%20%20%20boot_mae_a%20%3D%20np.empty(N_BOOT)%0A%20%20%20%20boot_mae_b%20%3D%20np.empty(N_BOOT)%0A%20%20%20%20for%20_i%20in%20range(N_BOOT)%3A%0A%20%20%20%20%20%20%20%20_idx%20%3D%20_rng.integers(0%2C%20n_cmpds%2C%20n_cmpds)%20%20%20%23%20shared%20indices%20%E2%86%92%20paired%20sample%0A%20%20%20%20%20%20%20%20boot_mae_a%5B_i%5D%20%3D%20ae_a%5B_idx%5D.mean()%0A%20%20%20%20%20%20%20%20boot_mae_b%5B_i%5D%20%3D%20ae_b%5B_idx%5D.mean()%0A%0A%20%20%20%20boot_delta%20%3D%20boot_mae_a%20-%20boot_mae_b%20%20%20%20%20%20%20%20%20%20%20%20%23%20%CE%94%20%3D%20MAE(A)%20%E2%88%92%20MAE(B)%0A%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20ALPHA%2C%0A%20%20%20%20%20%20%20%20HB_ADJUSTMENT_RANK%2C%0A%20%20%20%20%20%20%20%20HB_TOTAL_COMPARISONS%2C%0A%20%20%20%20%20%20%20%20HH_LABEL_A%2C%0A%20%20%20%20%20%20%20%20HH_LABEL_B%2C%0A%20%20%20%20%20%20%20%20N_BOOT%2C%0A%20%20%20%20%20%20%20%20boot_delta%2C%0A%20%20%20%20%20%20%20%20boot_mae_a%2C%0A%20%20%20%20%20%20%20%20boot_mae_b%2C%0A%20%20%20%20)%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20ALPHA%2C%0A%20%20%20%20HB_ADJUSTMENT_RANK%2C%0A%20%20%20%20HB_TOTAL_COMPARISONS%2C%0A%20%20%20%20HH_LABEL_A%2C%0A%20%20%20%20HH_LABEL_B%2C%0A%20%20%20%20N_BOOT%2C%0A%20%20%20%20boot_delta%2C%0A%20%20%20%20boot_mae_a%2C%0A%20%20%20%20boot_mae_b%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20np%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Statistics%20derived%20from%20the%20bootstrap%20distribution%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_mean_a%2C%20_sd_a%20%3D%20float(boot_mae_a.mean())%2C%20float(boot_mae_a.std(ddof%3D1))%0A%20%20%20%20_mean_b%2C%20_sd_b%20%3D%20float(boot_mae_b.mean())%2C%20float(boot_mae_b.std(ddof%3D1))%0A%0A%20%20%20%20_ci_lo%20%3D%20float(np.quantile(boot_delta%2C%200.025))%0A%20%20%20%20_ci_hi%20%3D%20float(np.quantile(boot_delta%2C%200.975))%0A%0A%20%20%20%20%23%20Two-tailed%20p-value%3A%20twice%20the%20smaller%20tail%20fraction.%0A%20%20%20%20_p_gt%20%3D%20float(np.mean(boot_delta%20%3E%200))%0A%20%20%20%20_p_lt%20%3D%20float(np.mean(boot_delta%20%3C%200))%0A%20%20%20%20_p_value%20%3D%202.0%20*%20min(_p_gt%2C%20_p_lt)%0A%0A%20%20%20%20%23%20Holm-Bonferroni%20adjusted%20threshold%20for%20this%20pair's%20rank.%0A%20%20%20%20_hb_threshold%20%3D%20ALPHA%20%2F%20(HB_TOTAL_COMPARISONS%20-%20HB_ADJUSTMENT_RANK%20%2B%201)%0A%0A%20%20%20%20%23%20Significance%3A%20p-value%20must%20clear%20the%20HB-adjusted%20threshold.%20%20(Equivalently%2C%20%CE%94%20%3D%200%0A%20%20%20%20%23%20would%20have%20to%20sit%20outside%20a%20CI%20at%20the%20corrected%20level.)%0A%20%20%20%20_significant%20%3D%20_p_value%20%3C%20_hb_threshold%0A%0A%20%20%20%20%23%20Store%20the%20summary%20so%20downstream%20cells%20%2F%20tables%20can%20reuse%20it.%0A%20%20%20%20hh_summary%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22user_a%22%3A%20HH_LABEL_A%2C%20%22user_b%22%3A%20HH_LABEL_B%2C%0A%20%20%20%20%20%20%20%20%22mean_a%22%3A%20_mean_a%2C%20%22sd_a%22%3A%20_sd_a%2C%0A%20%20%20%20%20%20%20%20%22mean_b%22%3A%20_mean_b%2C%20%22sd_b%22%3A%20_sd_b%2C%0A%20%20%20%20%20%20%20%20%22ci_lo%22%3A%20_ci_lo%2C%20%22ci_hi%22%3A%20_ci_hi%2C%0A%20%20%20%20%20%20%20%20%22p_value%22%3A%20_p_value%2C%0A%20%20%20%20%20%20%20%20%22hb_threshold%22%3A%20_hb_threshold%2C%0A%20%20%20%20%20%20%20%20%22significant%22%3A%20_significant%2C%0A%20%20%20%20%7D%0A%0A%20%20%20%20_sig_cell%20%3D%20(%0A%20%20%20%20%20%20%20%20%22%3Cb%20style%3D'color%3A%232e7d32'%3ETRUE%3C%2Fb%3E%22%20if%20_significant%0A%20%20%20%20%20%20%20%20else%20%22%3Cb%20style%3D'color%3A%23c62828'%3EFALSE%3C%2Fb%3E%22%0A%20%20%20%20)%0A%0A%20%20%20%20mo.md(f%22%22%22%0A%20%20%20%20%23%23%23%20Head-to-head%20statistical%20summary%0A%0A%20%20%20%20%7C%20%7C%20%7C%0A%20%20%20%20%7C---%7C---%7C%0A%20%20%20%20%7C%20Track%20%7C%20Activity%20%7C%0A%20%20%20%20%7C%20User%20A%20%7C%20**%7BHH_LABEL_A%7D**%20%7C%0A%20%20%20%20%7C%20User%20B%20%7C%20**%7BHH_LABEL_B%7D**%20%7C%0A%20%20%20%20%7C%20Evaluation%20Metric%20%7C%20MAE%20%7C%0A%20%20%20%20%7C%20User%20A%20mean%20MAE%20%C2%B1%20SD%20%7C%20%7B_mean_a%3A.4f%7D%20%C2%B1%20%7B_sd_a%3A.4f%7D%20%7C%0A%20%20%20%20%7C%20User%20B%20mean%20MAE%20%C2%B1%20SD%20%7C%20%7B_mean_b%3A.4f%7D%20%C2%B1%20%7B_sd_b%3A.4f%7D%20%7C%0A%20%20%20%20%7C%20Nominal%20%CE%B1%20%7C%20%7BALPHA%7D%20%7C%0A%20%20%20%20%7C%20HB%20Adjustment%20Rank%20%7C%20%7BHB_ADJUSTMENT_RANK%7D%20%7C%0A%20%20%20%20%7C%20HB%20Adjusted%20Threshold%20%7C%20%7B_hb_threshold%3A.4f%7D%20%7C%0A%20%20%20%20%7C%20Observed%20p-value%20%7C%20%7B_p_value%3A.4f%7D%20%7C%0A%20%20%20%20%7C%2095%25%20CI%20of%20%CE%94%20(A%20%E2%88%92%20B)%20%7C%20%5B%7B_ci_lo%3A.4f%7D%2C%20%7B_ci_hi%3A.4f%7D%5D%20%7C%0A%20%20%20%20%7C%20Statistically%20Significant%20%7C%20%7B_sig_cell%7D%20%7C%0A%0A%20%20%20%20**Reading%20it%3A**%20%7BHH_LABEL_A%7D%20has%20the%20lower%20mean%20MAE%2C%20so%20its%20model%20is%20nominally%0A%20%20%20%20better.%20%20The%20two-tailed%20p-value%20(%7B_p_value%3A.4f%7D)%20sits%20above%20the%20Holm-Bonferroni%0A%20%20%20%20adjusted%20threshold%20(%7B_hb_threshold%3A.4f%7D)%2C%20so%20after%20correcting%20for%20the%20%7BHB_TOTAL_COMPARISONS%3A%2C%7D%0A%20%20%20%20pairwise%20comparisons%20in%20the%20challenge%20the%20difference%20is%20**not%20statistically%0A%20%20%20%20significant**%20%E2%80%94%20%7BN_BOOT%3A%2C%7D%20resamples%20cannot%20rule%20out%20that%20the%20gap%20is%20test-set%20noise.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%20(hh_summary%2C)%0A%0A%0A%40app.cell%0Adef%20_(HH_LABEL_A%2C%20HH_LABEL_B%2C%20PLOT_DIR%2C%20boot_delta%2C%20hh_summary%2C%20mo%2C%20np%2C%20plt)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Delta%20distribution%20plot%20%E2%80%94%20matches%20the%20organisers'%20UI%20figure%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_above%20%3D%20boot_delta%20%3E%200%20%20%20%20%20%23%20%CE%94%20%3E%200%20%E2%86%92%20User%20A%20(Gashaw)%20has%20the%20HIGHER%20(worse)%20MAE%0A%20%20%20%20_below%20%3D%20~_above%0A%0A%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20_fig_hh%2C%20_ax_hh%20%3D%20plt.subplots(figsize%3D(7.5%2C%205)%2C%20dpi%3D150)%0A%0A%20%20%20%20%20%20%20%20_bins%20%3D%20np.linspace(boot_delta.min()%2C%20max(boot_delta.max()%2C%200.02)%2C%2055)%0A%20%20%20%20%20%20%20%20_ax_hh.hist(boot_delta%5B_below%5D%2C%20bins%3D_bins%2C%20color%3D%22%23d9915b%22%2C%20alpha%3D0.9%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20label%3Df%22%CE%94%20%3C%200%20(%7BHH_LABEL_B%7D%20higher%20MAE)%22)%0A%20%20%20%20%20%20%20%20_ax_hh.hist(boot_delta%5B_above%5D%2C%20bins%3D_bins%2C%20color%3D%22%236b83c9%22%2C%20alpha%3D0.9%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20label%3Df%22%CE%94%20%3E%200%20(%7BHH_LABEL_A%7D%20higher%20MAE)%22)%0A%0A%20%20%20%20%20%20%20%20_ax_hh.axvline(0.0%2C%20color%3D%22black%22%2C%20linestyle%3D%22%3A%22%2C%20linewidth%3D1.5)%0A%20%20%20%20%20%20%20%20_ax_hh.text(0.001%2C%20_ax_hh.get_ylim()%5B1%5D%20*%200.94%2C%20%22%CE%94%20%3D%200%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D10%2C%20va%3D%22top%22)%0A%0A%20%20%20%20%20%20%20%20_ax_hh.set_xlabel(%22%CE%94%20MAE%20%20(A%20%E2%88%92%20B)%22%2C%20fontsize%3D12)%0A%20%20%20%20%20%20%20%20_ax_hh.set_ylabel(%22Bootstrap%20Samples%22%2C%20fontsize%3D12)%0A%20%20%20%20%20%20%20%20_ax_hh.set_title(%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22Bootstrap%20%CE%94%20(%7BHH_LABEL_A%7D%20%E2%88%92%20%7BHH_LABEL_B%7D)%3A%20MAE%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22p%20%3D%20%7Bhh_summary%5B'p_value'%5D%3A.4f%7D%20%20%C2%B7%20%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20f%2295%25%20CI%20%5B%7Bhh_summary%5B'ci_lo'%5D%3A.3f%7D%2C%20%7Bhh_summary%5B'ci_hi'%5D%3A.3f%7D%5D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D12%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20_ax_hh.legend(fontsize%3D10%2C%20frameon%3DTrue%2C%20framealpha%3D0.9%2C%20loc%3D%22upper%20left%22)%0A%20%20%20%20%20%20%20%20_fig_hh.tight_layout()%0A%20%20%20%20%20%20%20%20_fig_hh.savefig(PLOT_DIR%20%2F%20%22head_to_head_delta_distribution.png%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20mo.center(mo.as_html(_fig_hh))%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Are%20our%20own%20submissions%20statistically%20distinguishable%3F%20%E2%80%94%20Tier%201%20analysis%0A%0A%20%20%20%20We%20now%20apply%20the%20**same%20paired-bootstrap%20%2B%20Holm-Bonferroni**%20procedure%20to%20*our%20own*%0A%20%20%20%20submissions%20(the%20external%20Gashaw%20blend%20is%20excluded).%20%20Two%20questions%3A%0A%0A%20%20%20%201.%20**Is%20any%20other%20submission%20statistically%20better%20than%20our%20final%20submission%3F**%0A%20%20%20%20%20%20%20If%20not%2C%20our%20final%20entry%20was%20a%20defensible%20choice%20even%20if%20a%20lower-MAE%20submission%0A%20%20%20%20%20%20%20existed%20%E2%80%94%20the%20gap%20could%20plausibly%20be%20test-set%20noise.%0A%20%20%20%202.%20**Which%20submissions%20form%20%22Tier%201%22%3F**%20%20Here%20Tier%201%20is%20anchored%20on%20**our%20final%0A%20%20%20%20%20%20%20submitted%20entry**%20(%606_ens_default_augfilt%60)%20rather%20than%20the%20best-MAE%20submission%20%E2%80%94%0A%20%20%20%20%20%20%20i.e.%20the%20set%20of%20submissions%20that%20are%20**not%20significantly%20different%20from%20our%0A%20%20%20%20%20%20%20final%20entry**.%20%20Everything%20*significantly%20better%20or%20worse*%20than%20the%20final%20entry%0A%20%20%20%20%20%20%20falls%20outside%20Tier%201.%0A%0A%20%20%20%20**Method%3A**%0A%0A%20%20%20%20-%20Run%20one%20paired%20bootstrap%20(1000%20iterations%2C%20shared%20resampled%20compound%20indices)%20and%0A%20%20%20%20%20%20cache%20each%20submission's%20per-iteration%20MAE.%0A%20%20%20%20-%20For%20**every%20pair**%20of%20submissions%20compute%20the%20two-tailed%20p-value%0A%20%20%20%20%20%20p%20%3D%202%C2%B7min(P(%CE%94%3E0)%2C%20P(%CE%94%3C0))%20%E2%80%94%20that%20is%20*M%20%3D%20C(n%2C%202)*%20comparisons.%0A%20%20%20%20-%20Sort%20all%20p-values%20ascending%20and%20apply%20Holm-Bonferroni%3A%20the%20comparison%20at%20rank%20*r*%0A%20%20%20%20%20%20is%20significant%20only%20if%20its%20p-value%20%3C%20%CE%B1%20%2F%20(M%20%E2%88%92%20r%20%2B%201)%2C%20**and**%20every%20comparison%0A%20%20%20%20%20%20ahead%20of%20it%20in%20the%20sorted%20list%20was%20also%20significant%20(the%20HB%20step-down%20rule).%0A%20%20%20%20-%20**Tier%201**%20membership%20%3D%20a%20submission%20whose%20*final-vs-it*%20comparison%20is%20**not**%0A%20%20%20%20%20%20significant%20under%20HB%20(plus%20our%20final%20submission%20itself).%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20ALPHA%2C%0A%20%20%20%20FINAL_NAME%2C%0A%20%20%20%20N_BOOT%2C%0A%20%20%20%20all_predictions%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20metrics_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20np%2C%0A%20%20%20%20pl%2C%0A)%3A%0A%20%20%20%20import%20itertools%20as%20_itertools%0A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Own%20submissions%20only%2C%20ordered%20best%20%E2%86%92%20worst%20by%20MAE%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_own_order%20%3D%20(%0A%20%20%20%20%20%20%20%20metrics_df.filter(~pl.col(%22is_external%22)).sort(%22MAE%22).get_column(%22model_name%22).to_list()%0A%20%20%20%20)%0A%20%20%20%20%23%20Tier%201%20anchor%20%3D%20our%20final%20submitted%20entry%20(not%20necessarily%20the%20best-MAE%20one).%0A%20%20%20%20TIER_BEST%20%3D%20FINAL_NAME.removesuffix(%22_submission.csv%22)%0A%0A%20%20%20%20%23%20Reference%20compound%20ordering%20(sorted%20Molecule%20Name)%20shared%20by%20all%20submissions.%0A%20%20%20%20_ref_names%20%3D%20(%0A%20%20%20%20%20%20%20%20all_predictions.filter(pl.col(%22model_name%22)%20%3D%3D%20TIER_BEST)%0A%20%20%20%20%20%20%20%20.sort(%22Molecule%20Name%22).get_column(%22Molecule%20Name%22).to_list()%0A%20%20%20%20)%0A%0A%20%20%20%20def%20_abs_err_for(model%3A%20str)%20-%3E%20%22np.ndarray%22%3A%0A%20%20%20%20%20%20%20%20%22%22%22Per-compound%20absolute%20error%20for%20*model*%2C%20aligned%20to%20the%20shared%20ordering.%22%22%22%0A%20%20%20%20%20%20%20%20_df%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20all_predictions.filter(pl.col(%22model_name%22)%20%3D%3D%20model)%0A%20%20%20%20%20%20%20%20%20%20%20%20.sort(%22Molecule%20Name%22)%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20assert%20_df.get_column(%22Molecule%20Name%22).to_list()%20%3D%3D%20_ref_names%2C%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22%7Bmodel%7D%20does%20not%20cover%20the%20same%20compounds%20as%20the%20reference%22%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20return%20_df.get_column(%22abs_error%22).to_numpy()%0A%0A%20%20%20%20_ae%20%3D%20%7Bm%3A%20_abs_err_for(m)%20for%20m%20in%20_own_order%7D%0A%20%20%20%20_n_cmpds%20%3D%20len(_ref_names)%0A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20One%20shared%20paired%20bootstrap%3A%20cache%20each%20submission's%20per-iteration%20MAE%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20Reusing%20the%20SAME%20resample%20indices%20across%20all%20submissions%20keeps%20every%20pairwise%0A%20%20%20%20%23%20comparison%20paired%20and%20mutually%20consistent.%0A%20%20%20%20TIER_SEED%20%3D%200%0A%20%20%20%20_rng%20%3D%20np.random.default_rng(TIER_SEED)%0A%20%20%20%20_boot_idx%20%3D%20%5B_rng.integers(0%2C%20_n_cmpds%2C%20_n_cmpds)%20for%20_%20in%20range(N_BOOT)%5D%0A%0A%20%20%20%20boot_mae%3A%20dict%5Bstr%2C%20%22np.ndarray%22%5D%20%3D%20%7B%0A%20%20%20%20%20%20%20%20m%3A%20np.array(%5B_ae%5Bm%5D%5B_idx%5D.mean()%20for%20_idx%20in%20_boot_idx%5D)%20for%20m%20in%20_own_order%0A%20%20%20%20%7D%0A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20All%20pairwise%20two-tailed%20p-values%20(M%20%3D%20C(n%2C%202))%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20def%20_two_tailed_p(delta%3A%20%22np.ndarray%22)%20-%3E%20float%3A%0A%20%20%20%20%20%20%20%20return%202.0%20*%20min(float(np.mean(delta%20%3E%200))%2C%20float(np.mean(delta%20%3C%200)))%0A%0A%20%20%20%20_pairs%20%3D%20list(_itertools.combinations(_own_order%2C%202))%0A%20%20%20%20_pair_rows%20%3D%20%5B%5D%0A%20%20%20%20for%20_a%2C%20_b%20in%20_pairs%3A%0A%20%20%20%20%20%20%20%20_delta%20%3D%20boot_mae%5B_a%5D%20-%20boot_mae%5B_b%5D%0A%20%20%20%20%20%20%20%20_pair_rows.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22model_a%22%3A%20_a%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22model_b%22%3A%20_b%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22mean_mae_a%22%3A%20float(boot_mae%5B_a%5D.mean())%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22mean_mae_b%22%3A%20float(boot_mae%5B_b%5D.mean())%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22delta_mean%22%3A%20float(_delta.mean())%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22ci_lo%22%3A%20float(np.quantile(_delta%2C%200.025))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22ci_hi%22%3A%20float(np.quantile(_delta%2C%200.975))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22p_value%22%3A%20_two_tailed_p(_delta)%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%0A%20%20%20%20_M%20%3D%20len(_pairs)%0A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Holm-Bonferroni%20step-down%20across%20all%20pairwise%20p-values%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20pairwise_df%3A%20pl.DataFrame%20%3D%20(%0A%20%20%20%20%20%20%20%20pl.DataFrame(_pair_rows)%0A%20%20%20%20%20%20%20%20.sort(%22p_value%22)%0A%20%20%20%20%20%20%20%20.with_row_index(%22hb_rank%22%2C%20offset%3D1)%0A%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20(pl.lit(ALPHA)%20%2F%20(pl.lit(_M)%20-%20pl.col(%22hb_rank%22)%20%2B%201)).alias(%22hb_threshold%22)%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20)%0A%20%20%20%20%23%20Raw%20per-comparison%20rejection%2C%20then%20enforce%20the%20step-down%20(once%20a%20test%20fails%2C%20all%0A%20%20%20%20%23%20higher-ranked%20tests%20fail%20too).%0A%20%20%20%20_raw_reject%20%3D%20(pairwise_df%5B%22p_value%22%5D%20%3C%20pairwise_df%5B%22hb_threshold%22%5D).to_list()%0A%20%20%20%20_stepdown%3A%20list%5Bbool%5D%20%3D%20%5B%5D%0A%20%20%20%20_still_ok%20%3D%20True%0A%20%20%20%20for%20_r%20in%20_raw_reject%3A%0A%20%20%20%20%20%20%20%20_still_ok%20%3D%20_still_ok%20and%20_r%0A%20%20%20%20%20%20%20%20_stepdown.append(_still_ok)%0A%0A%20%20%20%20pairwise_df%20%3D%20pairwise_df.with_columns(%0A%20%20%20%20%20%20%20%20pl.Series(%22significant%22%2C%20_stepdown%2C%20dtype%3Dpl.Boolean)%0A%20%20%20%20)%0A%0A%20%20%20%20TIER_TOTAL_COMPARISONS%20%3D%20_M%0A%20%20%20%20return%20TIER_BEST%2C%20TIER_TOTAL_COMPARISONS%2C%20boot_mae%2C%20pairwise_df%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20TIER_BEST%2C%0A%20%20%20%20boot_mae%3A%20dict%5Bstr%2C%20%22np.ndarray%22%5D%2C%0A%20%20%20%20metrics_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20pairwise_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20pl%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Extract%20every%20%22final%20vs.%20other%22%20comparison%20to%20assign%20Tier%201%20membership%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20def%20_best_vs(model%3A%20str)%20-%3E%20dict%3A%0A%20%20%20%20%20%20%20%20%22%22%22Return%20the%20pairwise-row%20stats%20for%20the%20(final%2C%20model)%20comparison%2C%20oriented%0A%20%20%20%20%20%20%20%20as%20%CE%94%20%3D%20MAE(model)%20%E2%88%92%20MAE(final)%20so%20a%20positive%20%CE%94%20means%20'worse%20than%20final'.%22%22%22%0A%20%20%20%20%20%20%20%20_row%20%3D%20pairwise_df.filter(%0A%20%20%20%20%20%20%20%20%20%20%20%20((pl.col(%22model_a%22)%20%3D%3D%20TIER_BEST)%20%26%20(pl.col(%22model_b%22)%20%3D%3D%20model))%0A%20%20%20%20%20%20%20%20%20%20%20%20%7C%20((pl.col(%22model_a%22)%20%3D%3D%20model)%20%26%20(pl.col(%22model_b%22)%20%3D%3D%20TIER_BEST))%0A%20%20%20%20%20%20%20%20).row(0%2C%20named%3DTrue)%0A%20%20%20%20%20%20%20%20%23%20Orient%20%CE%94%20so%20it%20is%20(model%20%E2%88%92%20best).%0A%20%20%20%20%20%20%20%20if%20_row%5B%22model_a%22%5D%20%3D%3D%20TIER_BEST%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_dmean%2C%20_lo%2C%20_hi%20%3D%20-_row%5B%22delta_mean%22%5D%2C%20-_row%5B%22ci_hi%22%5D%2C%20-_row%5B%22ci_lo%22%5D%0A%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_dmean%2C%20_lo%2C%20_hi%20%3D%20_row%5B%22delta_mean%22%5D%2C%20_row%5B%22ci_lo%22%5D%2C%20_row%5B%22ci_hi%22%5D%0A%20%20%20%20%20%20%20%20return%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22p_value%22%3A%20_row%5B%22p_value%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22hb_rank%22%3A%20_row%5B%22hb_rank%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22hb_threshold%22%3A%20_row%5B%22hb_threshold%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22significant%22%3A%20_row%5B%22significant%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22delta_vs_best%22%3A%20_dmean%2C%20%22ci_lo%22%3A%20_lo%2C%20%22ci_hi%22%3A%20_hi%2C%0A%20%20%20%20%20%20%20%20%7D%0A%0A%20%20%20%20_own_order%20%3D%20(%0A%20%20%20%20%20%20%20%20metrics_df.filter(~pl.col(%22is_external%22)).sort(%22MAE%22).get_column(%22model_name%22).to_list()%0A%20%20%20%20)%0A%0A%20%20%20%20_rows%20%3D%20%5B%5D%0A%20%20%20%20for%20_m%20in%20_own_order%3A%0A%20%20%20%20%20%20%20%20_mean_mae%20%3D%20float(boot_mae%5B_m%5D.mean())%0A%20%20%20%20%20%20%20%20if%20_m%20%3D%3D%20TIER_BEST%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_rows.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22model_name%22%3A%20_m%2C%20%22mean_mae%22%3A%20_mean_mae%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22delta_vs_best%22%3A%200.0%2C%20%22ci_lo%22%3A%200.0%2C%20%22ci_hi%22%3A%200.0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22p_value%22%3A%20None%2C%20%22hb_rank%22%3A%20None%2C%20%22hb_threshold%22%3A%20None%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22sig_vs_best%22%3A%20False%2C%20%22tier%22%3A%20%22Tier%201%20(final)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_bv%20%3D%20_best_vs(_m)%0A%20%20%20%20%20%20%20%20%20%20%20%20_in_tier1%20%3D%20not%20_bv%5B%22significant%22%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20_rows.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22model_name%22%3A%20_m%2C%20%22mean_mae%22%3A%20_mean_mae%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22delta_vs_best%22%3A%20_bv%5B%22delta_vs_best%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22ci_lo%22%3A%20_bv%5B%22ci_lo%22%5D%2C%20%22ci_hi%22%3A%20_bv%5B%22ci_hi%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22p_value%22%3A%20_bv%5B%22p_value%22%5D%2C%20%22hb_rank%22%3A%20_bv%5B%22hb_rank%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22hb_threshold%22%3A%20_bv%5B%22hb_threshold%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22sig_vs_best%22%3A%20_bv%5B%22significant%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22tier%22%3A%20%22Tier%201%22%20if%20_in_tier1%20else%20%22Outside%20Tier%201%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D)%0A%0A%20%20%20%20tier_df%3A%20pl.DataFrame%20%3D%20pl.DataFrame(_rows)%0A%20%20%20%20tier_df%0A%20%20%20%20return%20(tier_df%2C)%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20FINAL_NAME%2C%0A%20%20%20%20TIER_BEST%2C%0A%20%20%20%20TIER_TOTAL_COMPARISONS%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20pairwise_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20tier_df%3A%20%22pl.DataFrame%22%2C%0A)%3A%0A%20%20%20%20_final_label%20%3D%20FINAL_NAME.removesuffix(%22_submission.csv%22)%0A%0A%20%20%20%20%23%20Best-MAE-vs-final%20comparison%20(best-MAE%20submission%20among%20our%20own%20entries).%0A%20%20%20%20_best_mae_label%20%3D%20(%0A%20%20%20%20%20%20%20%20tier_df.filter(pl.col(%22model_name%22)%20!%3D%20_final_label)%0A%20%20%20%20%20%20%20%20.sort(%22mean_mae%22).get_column(%22model_name%22).head(1).to_list()%0A%20%20%20%20)%0A%20%20%20%20_best_mae_label%20%3D%20_best_mae_label%5B0%5D%20if%20_best_mae_label%20else%20_final_label%0A%0A%20%20%20%20_n_tier1%20%3D%20tier_df.filter(pl.col(%22tier%22).str.starts_with(%22Tier%201%22)).shape%5B0%5D%0A%20%20%20%20_n_out%20%20%20%3D%20tier_df.filter(pl.col(%22tier%22)%20%3D%3D%20%22Outside%20Tier%201%22).shape%5B0%5D%0A%20%20%20%20_outside%20%3D%20tier_df.filter(pl.col(%22tier%22)%20%3D%3D%20%22Outside%20Tier%201%22)%0A%0A%20%20%20%20if%20_best_mae_label%20%3D%3D%20_final_label%3A%0A%20%20%20%20%20%20%20%20_bf_verdict%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22Our%20final%20submission%20%60%7B_final_label%7D%60%20**is**%20the%20best-MAE%20submission%20among%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%22our%20own%20entries%2C%20so%20the%20best-vs-final%20question%20is%20moot.%22%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20_bf%20%3D%20pairwise_df.filter(%0A%20%20%20%20%20%20%20%20%20%20%20%20((pl.col(%22model_a%22)%20%3D%3D%20TIER_BEST)%20%26%20(pl.col(%22model_b%22)%20%3D%3D%20_best_mae_label))%0A%20%20%20%20%20%20%20%20%20%20%20%20%7C%20((pl.col(%22model_a%22)%20%3D%3D%20_best_mae_label)%20%26%20(pl.col(%22model_b%22)%20%3D%3D%20TIER_BEST))%0A%20%20%20%20%20%20%20%20).row(0%2C%20named%3DTrue)%0A%20%20%20%20%20%20%20%20_bf_sig%20%3D%20_bf%5B%22significant%22%5D%0A%20%20%20%20%20%20%20%20_bf_verdict%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22Final%20(%60%7B_final_label%7D%60)%20vs.%20best-MAE%20(%60%7B_best_mae_label%7D%60)%3A%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22p%20%3D%20%7B_bf%5B'p_value'%5D%3A.4f%7D%2C%20HB%20rank%20%7B_bf%5B'hb_rank'%5D%7D%20of%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22%7BTIER_TOTAL_COMPARISONS%7D%2C%20adjusted%20threshold%20%7B_bf%5B'hb_threshold'%5D%3A.4f%7D%20%E2%86%92%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%2B%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22**significantly%20different**%20%E2%80%94%20the%20best-MAE%20submission%20is%20statistically%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22better%20than%20our%20final%20one.%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20_bf_sig%20else%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22**not%20significantly%20different**%20%E2%80%94%20although%20the%20best-MAE%20submission%20has%20a%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22lower%20MAE%2C%20the%20gap%20is%20within%20bootstrap%20noise%2C%20so%20our%20final%20submission%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22is%20statistically%20indistinguishable%20from%20the%20best-MAE%20one.%22%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20_out_list%20%3D%20(%0A%20%20%20%20%20%20%20%20%22%2C%20%22.join(f%22%60%7Br%5B'model_name'%5D%7D%60%20(p%3D%7Br%5B'p_value'%5D%3A.4f%7D)%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20r%20in%20_outside.iter_rows(named%3DTrue))%0A%20%20%20%20%20%20%20%20if%20_n_out%20%3E%200%20else%20%22%E2%80%94%20none%20%E2%80%94%22%0A%20%20%20%20)%0A%0A%20%20%20%20mo.md(f%22%22%22%0A%20%20%20%20%23%23%23%20Tier%201%20verdict%0A%0A%20%20%20%20**Final%20submission%20(anchor)%3A**%20%60%7BTIER_BEST%7D%60%0A%20%20%20%20**Total%20pairwise%20comparisons%20(M)%3A**%20%7BTIER_TOTAL_COMPARISONS%7D%0A%0A%20%20%20%20%7B_bf_verdict%7D%0A%0A%20%20%20%20**Tier%201**%20(%7B_n_tier1%7D%20of%20%7Btier_df.shape%5B0%5D%7D%20submissions%20%E2%80%94%20not%20significantly%20different%0A%20%20%20%20from%20our%20final%20submission)%3A%20every%20submission%20except%20the%20%7B_n_out%7D%20below.%0A%0A%20%20%20%20**Outside%20Tier%201**%20(%7B_n_out%7D%20%E2%80%94%20significantly%20different%20from%20our%20final%20submission%20under%0A%20%20%20%20Holm-Bonferroni)%3A%20%7B_out_list%7D%0A%0A%20%20%20%20%3E%20**Why%20some%20near-final%20submissions%20fall%20outside%20Tier%201%20while%20more%20distant%20ones%20stay%20in.**%0A%20%20%20%20%3E%20The%20test%20is%20*paired*%3A%20significance%20depends%20on%20the%20variance%20of%20%CE%94%20%3D%20MAE(model)%20%E2%88%92%20MAE(final)%2C%0A%20%20%20%20%3E%20not%20on%20the%20raw%20MAE%20gap.%20%20Submissions%20that%20are%20structurally%20similar%20to%20our%20final%20entry%20(e.g.%0A%20%20%20%20%3E%20the%20other%20tuned%20ensembles)%20are%20highly%20correlated%20with%20it%2C%20so%20their%20%CE%94%20has%20a%20very%20small%0A%20%20%20%20%3E%20CI%20%E2%80%94%20even%20a%20tiny%2C%20consistent%20gap%20becomes%20significant.%20%20A%20dissimilar%20submission%20(a%20bare%0A%20%20%20%20%3E%20baseline%2C%20a%20single-model)%20has%20a%20much%20wider%20%CE%94%20CI%2C%20so%20a%20*larger*%20MAE%20gap%20can%20still%20overlap%0A%20%20%20%20%3E%20zero.%20%20The%20forest%20plot%20below%20makes%20this%20explicit%3A%20red%20points%20have%20narrow%20CIs%20that%20clear%0A%20%20%20%20%3E%20zero%3B%20blue%20points%20have%20CIs%20that%20straddle%20it.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo%2C%20pl%2C%20tier_df%3A%20%22pl.DataFrame%22)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Full%20Tier%20table%2C%20sorted%20best%20%E2%86%92%20worst%20by%20bootstrap-mean%20MAE%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_disp%20%3D%20(%0A%20%20%20%20%20%20%20%20tier_df%0A%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22mean_mae%22).round(4)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22delta_vs_best%22).round(4)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.when(pl.col(%22p_value%22).is_null()).then(None)%0A%20%20%20%20%20%20%20%20%20%20%20%20.otherwise(pl.col(%22p_value%22).round(4)).alias(%22p_value%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.when(pl.col(%22hb_threshold%22).is_null()).then(None)%0A%20%20%20%20%20%20%20%20%20%20%20%20.otherwise(pl.col(%22hb_threshold%22).round(4)).alias(%22hb_threshold%22)%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20.select(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22model_name%22%2C%20%22mean_mae%22%2C%20%22delta_vs_best%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22p_value%22%2C%20%22hb_rank%22%2C%20%22hb_threshold%22%2C%20%22tier%22%2C%0A%20%20%20%20%20%20%20%20%5D)%0A%20%20%20%20)%0A%0A%20%20%20%20_hdr%20%3D%20%22%7C%20Submission%20%7C%20Mean%20MAE%20%7C%20%CE%94%20vs%20final%20%7C%20p%20(vs%20final)%20%7C%20HB%20rank%20%7C%20HB%20thresh%20%7C%20Tier%20%7C%5Cn%22%0A%20%20%20%20_hdr%20%2B%3D%20%22%7C---%7C---%7C---%7C---%7C---%7C---%7C---%7C%5Cn%22%0A%20%20%20%20_body%20%3D%20%22%22%0A%20%20%20%20for%20_r%20in%20_disp.iter_rows(named%3DTrue)%3A%0A%20%20%20%20%20%20%20%20_p%20%3D%20%22%E2%80%94%22%20if%20_r%5B%22p_value%22%5D%20is%20None%20else%20f%22%7B_r%5B'p_value'%5D%3A.4f%7D%22%0A%20%20%20%20%20%20%20%20_hbr%20%3D%20%22%E2%80%94%22%20if%20_r%5B%22hb_rank%22%5D%20is%20None%20else%20str(_r%5B%22hb_rank%22%5D)%0A%20%20%20%20%20%20%20%20_hbt%20%3D%20%22%E2%80%94%22%20if%20_r%5B%22hb_threshold%22%5D%20is%20None%20else%20f%22%7B_r%5B'hb_threshold'%5D%3A.4f%7D%22%0A%20%20%20%20%20%20%20%20_tier_cell%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22**%7B_r%5B'tier'%5D%7D**%22%20if%20_r%5B%22tier%22%5D%20%3D%3D%20%22Outside%20Tier%201%22%20else%20_r%5B%22tier%22%5D%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20_body%20%2B%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22%7C%20%60%7B_r%5B'model_name'%5D%7D%60%20%7C%20%7B_r%5B'mean_mae'%5D%3A.4f%7D%20%7C%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22%7B_r%5B'delta_vs_best'%5D%3A%2B.4f%7D%20%7C%20%7B_p%7D%20%7C%20%7B_hbr%7D%20%7C%20%7B_hbt%7D%20%7C%20%7B_tier_cell%7D%20%7C%5Cn%22%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20mo.md(%22%23%23%23%20Per-submission%20Tier%20table%5Cn%5Cn%22%20%2B%20_hdr%20%2B%20_body)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(PLOT_DIR%2C%20TIER_BEST%2C%20mo%2C%20np%2C%20plt%2C%20tier_df%3A%20%22pl.DataFrame%22)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Forest%20plot%3A%20%CE%94%20MAE%20vs%20final%20submission%2C%20with%2095%25%20CIs%2C%20coloured%20by%20Tier%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20Sort%20worst%20%E2%86%92%20best%20so%20the%20best%20sits%20at%20the%20top%20of%20the%20axis.%0A%20%20%20%20_plot_df%20%3D%20tier_df.sort(%22mean_mae%22%2C%20descending%3DTrue)%0A%20%20%20%20_labels%20%3D%20_plot_df%5B%22model_name%22%5D.to_list()%0A%20%20%20%20_delta%20%20%3D%20_plot_df%5B%22delta_vs_best%22%5D.to_numpy()%0A%20%20%20%20_lo%20%20%20%20%20%3D%20_plot_df%5B%22ci_lo%22%5D.to_numpy()%0A%20%20%20%20_hi%20%20%20%20%20%3D%20_plot_df%5B%22ci_hi%22%5D.to_numpy()%0A%20%20%20%20_tiers%20%20%3D%20_plot_df%5B%22tier%22%5D.to_list()%0A%0A%20%20%20%20_tier_color%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22Tier%201%20(final)%22%3A%20%22%232e7d32%22%2C%0A%20%20%20%20%20%20%20%20%22Tier%201%22%3A%20%20%20%20%20%20%20%20%20%22%234e79a7%22%2C%0A%20%20%20%20%20%20%20%20%22Outside%20Tier%201%22%3A%20%22%23e15759%22%2C%0A%20%20%20%20%7D%0A%20%20%20%20_y%20%3D%20np.arange(len(_labels))%0A%0A%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20_fig_t%2C%20_ax_t%20%3D%20plt.subplots(figsize%3D(9%2C%20max(4.5%2C%200.45%20*%20len(_labels)))%2C%20dpi%3D150)%0A%0A%20%20%20%20%20%20%20%20for%20_i%20in%20range(len(_labels))%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_col%20%3D%20_tier_color%5B_tiers%5B_i%5D%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20Error%20bar%20spans%20the%2095%25%20CI%20of%20%CE%94(model%20%E2%88%92%20final)%3B%20point%20is%20the%20mean%20%CE%94.%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_t.errorbar(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_delta%5B_i%5D%2C%20_y%5B_i%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20xerr%3D%5B%5B_delta%5B_i%5D%20-%20_lo%5B_i%5D%5D%2C%20%5B_hi%5B_i%5D%20-%20_delta%5B_i%5D%5D%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20fmt%3D%22o%22%2C%20color%3D_col%2C%20ecolor%3D_col%2C%20elinewidth%3D1.6%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20capsize%3D3%2C%20markersize%3D6%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20_ax_t.axvline(0.0%2C%20color%3D%22black%22%2C%20linestyle%3D%22--%22%2C%20linewidth%3D1.2)%0A%20%20%20%20%20%20%20%20_ax_t.text(0.0%2C%20len(_labels)%20-%200.4%2C%20%22%20%20%CE%94%20%3D%200%20(%3D%20final)%22%2C%20fontsize%3D9%2C%20va%3D%22top%22)%0A%0A%20%20%20%20%20%20%20%20_ax_t.set_yticks(_y)%0A%20%20%20%20%20%20%20%20_ax_t.set_yticklabels(_labels%2C%20fontsize%3D8)%0A%20%20%20%20%20%20%20%20_ax_t.set_xlabel(%22%CE%94%20MAE%20vs.%20final%20submission%20%20(model%20%E2%88%92%20final)%3B%2095%25%20bootstrap%20CI%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax_t.set_title(%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22Tier%201%20analysis%20%E2%80%94%20distance%20from%20final%20submission%20(%60%7BTIER_BEST%7D%60)%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Blue%2Fgreen%20%3D%20Tier%201%20(not%20sig.%20different%20from%20final)%3B%20red%20%3D%20outside%20Tier%201%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D11%2C%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20%23%20Legend%20via%20proxy%20handles.%0A%20%20%20%20%20%20%20%20from%20matplotlib.lines%20import%20Line2D%20as%20_Line2D%0A%20%20%20%20%20%20%20%20_handles%20%3D%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20_Line2D(%5B0%5D%2C%20%5B0%5D%2C%20marker%3D%22o%22%2C%20color%3D%22w%22%2C%20markerfacecolor%3D_c%2C%20markersize%3D8%2C%20label%3D_lbl)%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20_lbl%2C%20_c%20in%20_tier_color.items()%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20%20%20%20%20_ax_t.legend(handles%3D_handles%2C%20fontsize%3D9%2C%20frameon%3DTrue%2C%20framealpha%3D0.9%2C%20loc%3D%22lower%20right%22)%0A%20%20%20%20%20%20%20%20_fig_t.tight_layout()%0A%20%20%20%20%20%20%20%20_fig_t.savefig(PLOT_DIR%20%2F%20%22tier1_forest_plot.png%22%2C%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20mo.center(mo.as_html(_fig_t))%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%23%20All-vs-all%20significance%20heatmap%0A%0A%20%20%20%20The%20forest%20plot%20only%20compares%20each%20submission%20to%20the%20*best*%20one.%20%20This%20heatmap%20shows%0A%20%20%20%20**every%20pair**%20at%20once.%20%20Submissions%20are%20ordered%20best%20%E2%86%92%20worst%20by%20MAE%20(top-left%20%3D%20best).%0A%0A%20%20%20%20-%20**Colour**%20%3D%20signed%20%CE%94%20MAE%20(row%20%E2%88%92%20column)%20on%20a%20diverging%20scale%3A%20**red**%20means%20the%0A%20%20%20%20%20%20row%20submission%20has%20the%20*higher*%20(worse)%20MAE%20than%20the%20column%20submission%2C%20**blue**%0A%20%20%20%20%20%20means%20the%20row%20is%20*better*.%20%20White%20%E2%89%88%20no%20difference.%0A%20%20%20%20-%20**A%20dot%20(%C2%B7)**%20marks%20pairs%20whose%20difference%20is%20**statistically%20significant**%20under%0A%20%20%20%20%20%20the%20Holm-Bonferroni%20step-down%20rule%20(the%20paired-bootstrap%20test%20above).%20%20Blank%20cells%0A%20%20%20%20%20%20are%20pairs%20that%20are%20*not*%20significantly%20different%20%E2%80%94%20i.e.%20statistically%20tied.%0A%0A%20%20%20%20Reading%20a%20row%20left-to-right%20tells%20you%20which%20submissions%20that%20model%20beats%20(blue)%20or%0A%20%20%20%20loses%20to%20(red)%2C%20and%20which%20of%20those%20gaps%20are%20real%20(dotted)%20versus%20noise%20(plain).%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20PLOT_DIR%2C%0A%20%20%20%20metrics_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20np%2C%0A%20%20%20%20pairwise_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20plt%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Build%20symmetric%20%CE%94-MAE%20and%20significance%20matrices%20over%20own%20submissions%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_order%20%3D%20(%0A%20%20%20%20%20%20%20%20metrics_df.filter(~pl.col(%22is_external%22)).sort(%22MAE%22).get_column(%22model_name%22).to_list()%0A%20%20%20%20)%0A%20%20%20%20_n%20%3D%20len(_order)%0A%20%20%20%20_pos%20%3D%20%7Bm%3A%20i%20for%20i%2C%20m%20in%20enumerate(_order)%7D%0A%0A%20%20%20%20_delta_mat%20%3D%20np.full((_n%2C%20_n)%2C%20np.nan)%20%20%20%23%20signed%20%CE%94%20MAE%20(row%20%E2%88%92%20col)%0A%20%20%20%20_sig_mat%20%20%20%3D%20np.zeros((_n%2C%20_n)%2C%20dtype%3Dbool)%0A%0A%20%20%20%20for%20_row%20in%20pairwise_df.iter_rows(named%3DTrue)%3A%0A%20%20%20%20%20%20%20%20_ia%2C%20_ib%20%3D%20_pos%5B_row%5B%22model_a%22%5D%5D%2C%20_pos%5B_row%5B%22model_b%22%5D%5D%0A%20%20%20%20%20%20%20%20%23%20pairwise_df%20stores%20delta_mean%20%3D%20MAE(a)%20%E2%88%92%20MAE(b).%0A%20%20%20%20%20%20%20%20_delta_mat%5B_ia%2C%20_ib%5D%20%3D%20_row%5B%22delta_mean%22%5D%0A%20%20%20%20%20%20%20%20_delta_mat%5B_ib%2C%20_ia%5D%20%3D%20-_row%5B%22delta_mean%22%5D%0A%20%20%20%20%20%20%20%20_sig_mat%5B_ia%2C%20_ib%5D%20%3D%20_row%5B%22significant%22%5D%0A%20%20%20%20%20%20%20%20_sig_mat%5B_ib%2C%20_ia%5D%20%3D%20_row%5B%22significant%22%5D%0A%0A%20%20%20%20_abs_max%20%3D%20float(np.nanmax(np.abs(_delta_mat)))%0A%0A%20%20%20%20%23%20Short%20axis%20labels%20%E2%80%94%20keep%20the%20notebook%2Fmodel%20prefix%2C%20drop%20the%20long%20tail.%0A%20%20%20%20_short%20%3D%20%5Bm.replace(%22_submission%22%2C%20%22%22)%20for%20m%20in%20_order%5D%0A%0A%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20_fig_hm%2C%20_ax_hm%20%3D%20plt.subplots(figsize%3D(10%2C%208.5)%2C%20dpi%3D150)%0A%20%20%20%20%20%20%20%20_ax_hm.grid(False)%0A%0A%20%20%20%20%20%20%20%20%23%20Diverging%20colour%3A%20red%20%3D%20row%20worse%20(higher%20MAE)%2C%20blue%20%3D%20row%20better.%0A%20%20%20%20%20%20%20%20_im%20%3D%20_ax_hm.imshow(%0A%20%20%20%20%20%20%20%20%20%20%20%20_delta_mat%2C%20cmap%3D%22RdBu_r%22%2C%20vmin%3D-_abs_max%2C%20vmax%3D_abs_max%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20aspect%3D%22equal%22%2C%20interpolation%3D%22nearest%22%2C%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20%23%20Grey%20the%20diagonal%20(self-comparison).%0A%20%20%20%20%20%20%20%20for%20_i%20in%20range(_n)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_hm.add_patch(plt.Rectangle((_i%20-%200.5%2C%20_i%20-%200.5)%2C%201%2C%201%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20facecolor%3D%22%23dddddd%22%2C%20edgecolor%3D%22none%22%2C%20zorder%3D2))%0A%0A%20%20%20%20%20%20%20%20%23%20Mark%20statistically%20significant%20off-diagonal%20pairs%20with%20a%20dot.%0A%20%20%20%20%20%20%20%20_sig_y%2C%20_sig_x%20%3D%20np.where(_sig_mat)%0A%20%20%20%20%20%20%20%20_ax_hm.scatter(_sig_x%2C%20_sig_y%2C%20marker%3D%22.%22%2C%20s%3D45%2C%20color%3D%22black%22%2C%20zorder%3D3)%0A%0A%20%20%20%20%20%20%20%20_ax_hm.set_xticks(range(_n))%0A%20%20%20%20%20%20%20%20_ax_hm.set_yticks(range(_n))%0A%20%20%20%20%20%20%20%20_ax_hm.set_xticklabels(_short%2C%20fontsize%3D7%2C%20rotation%3D90)%0A%20%20%20%20%20%20%20%20_ax_hm.set_yticklabels(_short%2C%20fontsize%3D7)%0A%20%20%20%20%20%20%20%20_ax_hm.set_xlabel(%22Column%20submission%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax_hm.set_ylabel(%22Row%20submission%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax_hm.set_title(%0A%20%20%20%20%20%20%20%20%20%20%20%20%22All-vs-all%20submission%20comparison%20(Phase%202)%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Colour%20%3D%20%CE%94%20MAE%20(row%20%E2%88%92%20column)%3B%20red%20%3D%20row%20worse.%20%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Dot%20%3D%20significantly%20different%20(Holm-Bonferroni)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D11%2C%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20_cb%20%3D%20_fig_hm.colorbar(_im%2C%20ax%3D_ax_hm%2C%20fraction%3D0.046%2C%20pad%3D0.03)%0A%20%20%20%20%20%20%20%20_cb.set_label(%22%CE%94%20MAE%20%20(row%20%E2%88%92%20column)%22%2C%20fontsize%3D10)%0A%0A%20%20%20%20%20%20%20%20_fig_hm.tight_layout()%0A%20%20%20%20%20%20%20%20_fig_hm.savefig(PLOT_DIR%20%2F%20%22all_vs_all_significance_heatmap.png%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20%23%20Count%20of%20significant%20pairs%20(upper%20triangle%20only)%20for%20a%20caption.%0A%20%20%20%20_n_sig_pairs%20%3D%20int(_sig_mat%5Bnp.triu_indices(_n%2C%20k%3D1)%5D.sum())%0A%20%20%20%20_n_pairs%20%3D%20_n%20*%20(_n%20-%201)%20%2F%2F%202%0A%20%20%20%20_caption%20%3D%20mo.md(%0A%20%20%20%20%20%20%20%20f%22**%7B_n_sig_pairs%7D%20of%20%7B_n_pairs%7D**%20submission%20pairs%20are%20significantly%20different%3B%20%22%0A%20%20%20%20%20%20%20%20f%22the%20remaining%20%7B_n_pairs%20-%20_n_sig_pairs%7D%20are%20statistical%20ties.%22%0A%20%20%20%20)%0A%0A%20%20%20%20mo.vstack(%5Bmo.center(mo.as_html(_fig_hm))%2C%20_caption%5D)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%23%20Predicted%20vs.%20observed%20scatter%20%E2%80%94%20best%2C%20final%2C%20and%20worst%20submissions%0A%0A%20%20%20%20Rather%20than%20plotting%20all%20~18%20submissions%2C%20we%20contrast%20three%3A%20the%20**best**%20local%0A%20%20%20%20submission%20by%20MAE%2C%20our%20**final**%20submission%2C%20and%20the%20**worst**%20local%20submission.%0A%20%20%20%20The%20dashed%20diagonal%20is%20the%20perfect-prediction%20line%3B%20points%20are%20coloured%20by%0A%20%20%20%20absolute%20error%20(redder%20%3D%20larger%20mistake).%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20PLOT_DIR%2C%0A%20%20%20%20all_predictions%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20mean_absolute_error%2C%0A%20%20%20%20metrics_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20np%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20plt%2C%0A)%3A%0A%20%20%20%20%23%20Choose%20the%20three%20submissions%20to%20contrast%20(all%20from%20our%20own%2C%20not%20external).%0A%20%20%20%20_own%20%3D%20metrics_df.filter(~pl.col(%22is_external%22)).sort(%22MAE%22)%0A%20%20%20%20_best_name%20%20%3D%20_own.row(0%2C%20named%3DTrue)%5B%22model_name%22%5D%0A%20%20%20%20_final_name%20%3D%20_own.filter(pl.col(%22is_final%22)).row(0%2C%20named%3DTrue)%5B%22model_name%22%5D%0A%20%20%20%20_worst_name%20%3D%20_own.sort(%22MAE%22%2C%20descending%3DTrue).row(0%2C%20named%3DTrue)%5B%22model_name%22%5D%0A%0A%20%20%20%20%23%20Preserve%20order%20best%20%E2%86%92%20final%20%E2%86%92%20worst%20but%20de-duplicate%20if%20final%20%3D%3D%20best%2Fworst.%0A%20%20%20%20_panel_specs%20%3D%20%5B(%22Best%22%2C%20_best_name)%2C%20(%22Final%22%2C%20_final_name)%2C%20(%22Worst%22%2C%20_worst_name)%5D%0A%20%20%20%20_seen%3A%20set%5Bstr%5D%20%3D%20set()%0A%20%20%20%20_panels%20%3D%20%5B%5D%0A%20%20%20%20for%20_tag%2C%20_name%20in%20_panel_specs%3A%0A%20%20%20%20%20%20%20%20if%20_name%20not%20in%20_seen%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_panels.append((_tag%2C%20_name))%0A%20%20%20%20%20%20%20%20%20%20%20%20_seen.add(_name)%0A%0A%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20_fig%2C%20_axes%20%3D%20plt.subplots(1%2C%20len(_panels)%2C%20figsize%3D(len(_panels)%20*%204.6%2C%204.6)%2C%20dpi%3D130)%0A%20%20%20%20%20%20%20%20_axes%20%3D%20np.atleast_1d(_axes)%0A%0A%20%20%20%20%20%20%20%20for%20_ax%2C%20(_tag%2C%20_name)%20in%20zip(_axes%2C%20_panels)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_grp%20%3D%20all_predictions.filter(pl.col(%22model_name%22)%20%3D%3D%20_name)%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_true%20%3D%20_grp.get_column(%22pEC50_true%22).to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_pred%20%3D%20_grp.get_column(%22pEC50_pred%22).to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_err%20%20%20%20%3D%20np.abs(_y_pred%20-%20_y_true)%0A%20%20%20%20%20%20%20%20%20%20%20%20_mae%20%3D%20mean_absolute_error(_y_true%2C%20_y_pred)%0A%20%20%20%20%20%20%20%20%20%20%20%20_rmse%20%3D%20float(np.sqrt(np.mean((_y_pred%20-%20_y_true)%20**%202)))%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20_sc%20%3D%20_ax.scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_y_pred%2C%20_y_true%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20c%3D_err%2C%20cmap%3D%22RdYlGn_r%22%2C%20vmin%3D0%2C%20vmax%3D1.5%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20s%3D28%2C%20alpha%3D0.78%2C%20edgecolors%3D%22none%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20_lims%20%3D%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20min(_y_true.min()%2C%20_y_pred.min())%20-%200.2%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20max(_y_true.max()%2C%20_y_pred.max())%20%2B%200.2%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.plot(_lims%2C%20_lims%2C%20%22k--%22%2C%20linewidth%3D0.8%2C%20zorder%3D0)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.set_xlim(_lims)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.set_ylim(_lims)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.set_xlabel(%22Predicted%20pEC50%22%2C%20fontsize%3D9)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.set_ylabel(%22True%20pEC50%22%2C%20fontsize%3D9)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.set_title(f%22%7B_tag%7D%3A%20%7B_name%7D%5CnMAE%3D%7B_mae%3A.3f%7D%20%20RMSE%3D%7B_rmse%3A.3f%7D%22%2C%20fontsize%3D8%2C%20pad%3D4)%0A%20%20%20%20%20%20%20%20%20%20%20%20plt.colorbar(_sc%2C%20ax%3D_ax%2C%20label%3D%22%7Cerror%7C%22%2C%20fraction%3D0.046%2C%20pad%3D0.04)%0A%0A%20%20%20%20%20%20%20%20_fig.suptitle(%22Predicted%20vs.%20true%20pEC50%20%E2%80%94%20Phase%202%20(best%20%2F%20final%20%2F%20worst)%22%2C%20fontsize%3D13%2C%20y%3D1.02)%0A%20%20%20%20%20%20%20%20_fig.tight_layout()%0A%20%20%20%20%20%20%20%20_fig.savefig(PLOT_DIR%20%2F%20%22pred_vs_true_best_final_worst.png%22%2C%20dpi%3D200%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20mo.center(mo.as_html(_fig))%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%23%20Where%20did%20the%20final%20submission%20win%20or%20lose%20vs.%20the%20best%20alternative%3F%0A%0A%20%20%20%20We%20line%20up%20the%20per-compound%20absolute%20error%20of%20the%20**final**%20submission%20against%0A%20%20%20%20the%20**best**%20local%20submission.%20%20Points%20below%20the%20diagonal%20are%20compounds%20the%20best%0A%20%20%20%20submission%20predicted%20better%20than%20our%20final%20one%3B%20points%20above%20are%20where%20the%20final%0A%20%20%20%20submission%20was%20better.%20%20If%20the%20final%20was%20already%20the%20best%2C%20the%20two%20frames%20are%20the%0A%20%20%20%20same%20model%20and%20this%20panel%20is%20skipped.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20PLOT_DIR%2C%0A%20%20%20%20all_predictions%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20metrics_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20nn_cliff_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20plt%2C%0A)%3A%0A%20%20%20%20_own%20%3D%20metrics_df.filter(~pl.col(%22is_external%22)).sort(%22MAE%22)%0A%20%20%20%20_best_name%20%20%3D%20_own.row(0%2C%20named%3DTrue)%5B%22model_name%22%5D%0A%20%20%20%20_final_name%20%3D%20_own.filter(pl.col(%22is_final%22)).row(0%2C%20named%3DTrue)%5B%22model_name%22%5D%0A%0A%20%20%20%20if%20_best_name%20%3D%3D%20_final_name%3A%0A%20%20%20%20%20%20%20%20_head_delta%20%3D%20mo.md(%0A%20%20%20%20%20%20%20%20%20%20%20%20%22**Our%20final%20submission%20was%20already%20the%20best%20local%20submission%20%E2%80%94%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%22no%20better%20alternative%20exists%20to%20compare%20against.**%22%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20_final_e%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20all_predictions.filter(pl.col(%22model_name%22)%20%3D%3D%20_final_name)%0A%20%20%20%20%20%20%20%20%20%20%20%20.select(%5B%22Molecule%20Name%22%2C%20%22abs_error%22%5D).rename(%7B%22abs_error%22%3A%20%22final_abs_err%22%7D)%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20_best_e%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20all_predictions.filter(pl.col(%22model_name%22)%20%3D%3D%20_best_name)%0A%20%20%20%20%20%20%20%20%20%20%20%20.select(%5B%22Molecule%20Name%22%2C%20%22abs_error%22%5D).rename(%7B%22abs_error%22%3A%20%22best_abs_err%22%7D)%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%23%20Attach%20pair_class%20so%20we%20can%20see%20whether%20cliffs%20drive%20the%20difference.%0A%20%20%20%20%20%20%20%20_cmp%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20_final_e.join(_best_e%2C%20on%3D%22Molecule%20Name%22%2C%20how%3D%22inner%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20.join(nn_cliff_df.select(%5B%22Molecule%20Name%22%2C%20%22pair_class%22%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20on%3D%22Molecule%20Name%22%2C%20how%3D%22left%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22final_abs_err%22)%20-%20pl.col(%22best_abs_err%22)).alias(%22delta_abs_err%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20_n_final_better%20%3D%20_cmp.filter(pl.col(%22delta_abs_err%22)%20%3C%20-1e-9).shape%5B0%5D%0A%20%20%20%20%20%20%20%20_n_best_better%20%20%3D%20_cmp.filter(pl.col(%22delta_abs_err%22)%20%3E%201e-9).shape%5B0%5D%0A%0A%20%20%20%20%20%20%20%20_cls_colors%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Activity%20cliff%22%3A%20%22%23e15759%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Similar%20%2F%20concordant%22%3A%20%22%2376b7b2%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Dissimilar%22%3A%20%22%23b0b0b0%22%2C%0A%20%20%20%20%20%20%20%20%7D%0A%0A%20%20%20%20%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_fig_d%2C%20_ax_d%20%3D%20plt.subplots(figsize%3D(6.2%2C%206.0)%2C%20dpi%3D130)%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20_cls%2C%20_col%20in%20_cls_colors.items()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_s%20%3D%20_cmp.filter(pl.col(%22pair_class%22)%20%3D%3D%20_cls)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20_s.shape%5B0%5D%20%3D%3D%200%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20continue%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_ax_d.scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_s%5B%22best_abs_err%22%5D.to_numpy()%2C%20_s%5B%22final_abs_err%22%5D.to_numpy()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20c%3D_col%2C%20s%3D34%2C%20alpha%3D0.8%2C%20edgecolors%3D%22none%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20label%3Df%22%7B_cls%7D%20(n%3D%7B_s.shape%5B0%5D%7D)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20_lim%20%3D%20float(max(_cmp%5B%22best_abs_err%22%5D.max()%2C%20_cmp%5B%22final_abs_err%22%5D.max()))%20%2B%200.1%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_d.plot(%5B0%2C%20_lim%5D%2C%20%5B0%2C%20_lim%5D%2C%20%22k--%22%2C%20linewidth%3D0.9%2C%20zorder%3D0)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_d.set_xlabel(f%22Best%20submission%20%7Cerror%7C%5Cn(%7B_best_name%7D)%22%2C%20fontsize%3D9)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_d.set_ylabel(f%22Final%20submission%20%7Cerror%7C%5Cn(%7B_final_name%7D)%22%2C%20fontsize%3D9)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_d.set_title(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Per-compound%20error%3A%20final%20vs.%20best%20submission%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22Above%20diagonal%20%3D%20best%20wins%20(%7B_n_best_better%7D)%3B%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22below%20%3D%20final%20wins%20(%7B_n_final_better%7D)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D9%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_d.set_xlim(-0.05%2C%20_lim)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_d.set_ylim(-0.05%2C%20_lim)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_d.legend(fontsize%3D8%2C%20frameon%3DTrue%2C%20framealpha%3D0.9)%0A%20%20%20%20%20%20%20%20%20%20%20%20_fig_d.tight_layout()%0A%20%20%20%20%20%20%20%20%20%20%20%20_fig_d.savefig(PLOT_DIR%20%2F%20%22final_vs_best_per_compound.png%22%2C%20dpi%3D200%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20%20%20%20%20_head_delta%20%3D%20mo.center(mo.as_html(_fig_d))%0A%0A%20%20%20%20_head_delta%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%23%20Error%20concentration%20%E2%80%94%20do%20cliffs%20explain%20the%20misses%3F%0A%0A%20%20%20%20We%20split%20the%20Phase%202%20compounds%20by%20their%20nearest-neighbour%20class%20(activity%20cliff%20%2F%0A%20%20%20%20similar-concordant%20%2F%20dissimilar)%20and%20report%20the%20final%20submission's%20mean%20absolute%0A%20%20%20%20error%20in%20each%20bucket.%20%20If%20cliffs%20and%20dissimilar%20compounds%20dominate%20the%20error%2C%20that%0A%20%20%20%20confirms%20the%20failures%20are%20driven%20by%20structural%20novelty%20rather%20than%20model%20tuning.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20FINAL_NAME%2C%0A%20%20%20%20all_predictions%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20nn_cliff_df%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20pl%2C%0A)%3A%0A%20%20%20%20_final_label%20%3D%20FINAL_NAME.removesuffix(%22_submission.csv%22)%0A%0A%20%20%20%20_final_errs%20%3D%20(%0A%20%20%20%20%20%20%20%20all_predictions.filter(pl.col(%22model_name%22)%20%3D%3D%20_final_label)%0A%20%20%20%20%20%20%20%20.select(%5B%22Molecule%20Name%22%2C%20%22abs_error%22%5D)%0A%20%20%20%20%20%20%20%20.join(nn_cliff_df.select(%5B%22Molecule%20Name%22%2C%20%22pair_class%22%2C%20%22nn_sim%22%2C%20%22abs_delta_pEC50%22%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20on%3D%22Molecule%20Name%22%2C%20how%3D%22left%22)%0A%20%20%20%20)%0A%0A%20%20%20%20_by_class%20%3D%20(%0A%20%20%20%20%20%20%20%20_final_errs%0A%20%20%20%20%20%20%20%20.group_by(%22pair_class%22)%0A%20%20%20%20%20%20%20%20.agg(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.len().alias(%22n%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22abs_error%22).mean().alias(%22mean_abs_error%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22abs_error%22).median().alias(%22median_abs_error%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22abs_error%22)%20%3E%201.0).sum().alias(%22n_bad%22)%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20.with_columns((pl.col(%22n_bad%22)%20%2F%20pl.col(%22n%22)%20*%20100).round(1).alias(%22pct_bad%22))%0A%20%20%20%20%20%20%20%20.sort(%22mean_abs_error%22%2C%20descending%3DTrue)%0A%20%20%20%20)%0A%0A%20%20%20%20_overall_mae%20%3D%20float(_final_errs%5B%22abs_error%22%5D.mean())%0A%0A%20%20%20%20_rows%20%3D%20%22%5Cn%22.join(%0A%20%20%20%20%20%20%20%20f%22%7C%20%7Br%5B'pair_class'%5D%7D%20%7C%20%7Br%5B'n'%5D%7D%20%7C%20%7Br%5B'mean_abs_error'%5D%3A.3f%7D%20%7C%20%22%0A%20%20%20%20%20%20%20%20f%22%7Br%5B'median_abs_error'%5D%3A.3f%7D%20%7C%20%7Br%5B'n_bad'%5D%7D%20(%7Br%5B'pct_bad'%5D%3A.1f%7D%25)%20%7C%22%0A%20%20%20%20%20%20%20%20for%20r%20in%20_by_class.iter_rows(named%3DTrue)%0A%20%20%20%20)%0A%0A%20%20%20%20mo.md(f%22%22%22%0A%20%20%20%20%23%23%23%20Final%20submission%20error%20by%20structural%20class%0A%0A%20%20%20%20Overall%20mean%20%5C%5C%7Cerror%5C%5C%7C%20on%20Phase%202%3A%20**%7B_overall_mae%3A.3f%7D**%20pEC50%20units.%0A%0A%20%20%20%20%7C%20NN%20class%20%7C%20n%20%7C%20Mean%20%5C%5C%7Cerror%5C%5C%7C%20%7C%20Median%20%5C%5C%7Cerror%5C%5C%7C%20%7C%20%5C%5C%7Cerror%5C%5C%7C%20%3E%201%20%7C%0A%20%20%20%20%7C---%7C---%7C---%7C---%7C---%7C%0A%20%20%20%20%7B_rows%7D%0A%0A%20%20%20%20Buckets%20with%20a%20higher%20mean%20%5C%5C%7Cerror%5C%5C%7C%20than%20the%20overall%20average%20are%20where%20the%20final%0A%20%20%20%20model%20struggled%20most%20%E2%80%94%20typically%20activity%20cliffs%20and%20structurally%20dissimilar%0A%20%20%20%20compounds%2C%20i.e.%20exactly%20the%20regions%20the%20cliff%20analysis%20above%20flagged%20as%0A%20%20%20%20activity-sensitive%20or%20out-of-domain.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%23%20Systematic%20bias%20by%20activity%20bin%0A%0A%20%20%20%20For%20each%20submission%20we%20check%20whether%20predictions%20are%20systematically%20shifted%20within%0A%20%20%20%20activity%20ranges.%20%20A%20model%20that%20is%20accurate%20overall%20but%20biased%20in%20the%20hit%20region%0A%20%20%20%20(pEC50%20%E2%89%A5%206)%20is%20dangerous%20for%20virtual%20screening%20because%20it%20mis-ranks%20the%20compounds%0A%20%20%20%20that%20matter%20most.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(all_predictions%3A%20%22pl.DataFrame%22%2C%20pl)%3A%0A%20%20%20%20%23%20Bin%20compounds%20by%20true%20pEC50%20into%20four%20activity%20ranges.%0A%20%20%20%20binned%3A%20pl.DataFrame%20%3D%20all_predictions.with_columns(%0A%20%20%20%20%20%20%20%20pl.when(pl.col(%22pEC50_true%22)%20%3E%3D%206.0)%0A%20%20%20%20%20%20%20%20.then(pl.lit(%22%3E6%20(hit%20zone)%22))%0A%20%20%20%20%20%20%20%20.when(pl.col(%22pEC50_true%22)%20%3E%3D%205.0)%0A%20%20%20%20%20%20%20%20.then(pl.lit(%225%E2%80%936%20(moderate)%22))%0A%20%20%20%20%20%20%20%20.when(pl.col(%22pEC50_true%22)%20%3E%3D%204.0)%0A%20%20%20%20%20%20%20%20.then(pl.lit(%224%E2%80%935%20(weak)%22))%0A%20%20%20%20%20%20%20%20.otherwise(pl.lit(%22%3C4%20(inactive)%22))%0A%20%20%20%20%20%20%20%20.alias(%22pec50_bin%22)%0A%20%20%20%20)%0A%0A%20%20%20%20bias_by_bin%3A%20pl.DataFrame%20%3D%20(%0A%20%20%20%20%20%20%20%20binned%0A%20%20%20%20%20%20%20%20.group_by(%5B%22model_name%22%2C%20%22pec50_bin%22%5D)%0A%20%20%20%20%20%20%20%20.agg(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22error%22).mean().alias(%22mean_error%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22abs_error%22).mean().alias(%22mean_abs_error%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.len().alias(%22n%22)%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20.sort(%5B%22model_name%22%2C%20%22pec50_bin%22%5D)%0A%20%20%20%20)%0A%0A%20%20%20%20bias_by_bin%0A%20%20%20%20return%20(bias_by_bin%2C)%0A%0A%0A%40app.cell%0Adef%20_(PLOT_DIR%2C%20bias_by_bin%3A%20%22pl.DataFrame%22%2C%20mo%2C%20np%2C%20plt)%3A%0A%20%20%20%20_bin_order%20%3D%20%5B%22%3C4%20(inactive)%22%2C%20%224%E2%80%935%20(weak)%22%2C%20%225%E2%80%936%20(moderate)%22%2C%20%22%3E6%20(hit%20zone)%22%5D%0A%0A%20%20%20%20%23%20Model%20%C3%97%20bin%20matrix%20of%20mean%20signed%20errors.%0A%20%20%20%20_model_names%20%3D%20sorted(bias_by_bin.get_column(%22model_name%22).unique().to_list())%0A%20%20%20%20_matrix%20%3D%20np.full((len(_model_names)%2C%20len(_bin_order))%2C%20np.nan)%0A%20%20%20%20_n_matrix%20%3D%20np.full((len(_model_names)%2C%20len(_bin_order))%2C%200)%0A%0A%20%20%20%20for%20_row%20in%20bias_by_bin.iter_rows(named%3DTrue)%3A%0A%20%20%20%20%20%20%20%20_mi%20%3D%20_model_names.index(_row%5B%22model_name%22%5D)%0A%20%20%20%20%20%20%20%20if%20_row%5B%22pec50_bin%22%5D%20in%20_bin_order%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_bi%20%3D%20_bin_order.index(_row%5B%22pec50_bin%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20_matrix%5B_mi%2C%20_bi%5D%20%3D%20_row%5B%22mean_error%22%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20_n_matrix%5B_mi%2C%20_bi%5D%20%3D%20_row%5B%22n%22%5D%0A%0A%20%20%20%20_abs_max%20%3D%20float(np.nanmax(np.abs(_matrix)))%0A%0A%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20_fig_bias%2C%20_ax_bias%20%3D%20plt.subplots(figsize%3D(7.5%2C%20max(4.5%2C%200.42%20*%20len(_model_names)))%2C%20dpi%3D150)%0A%20%20%20%20%20%20%20%20_ax_bias.grid(False)%0A%20%20%20%20%20%20%20%20_im%20%3D%20_ax_bias.imshow(%0A%20%20%20%20%20%20%20%20%20%20%20%20_matrix%2C%20cmap%3D%22RdBu_r%22%2C%20vmin%3D-_abs_max%2C%20vmax%3D_abs_max%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20aspect%3D%22auto%22%2C%20interpolation%3D%22nearest%22%2C%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20for%20_mi%20in%20range(len(_model_names))%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20_bi%20in%20range(len(_bin_order))%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_val%20%3D%20_matrix%5B_mi%2C%20_bi%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_n%20%3D%20_n_matrix%5B_mi%2C%20_bi%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20not%20np.isnan(_val)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_txt_col%20%3D%20%22white%22%20if%20abs(_val)%20%3E%200.6%20*%20_abs_max%20else%20%22black%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_ax_bias.text(_bi%2C%20_mi%2C%20f%22%7B_val%3A%2B.2f%7D%5Cnn%3D%7B_n%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20ha%3D%22center%22%2C%20va%3D%22center%22%2C%20fontsize%3D6.5%2C%20color%3D_txt_col)%0A%0A%20%20%20%20%20%20%20%20_ax_bias.set_xticks(range(len(_bin_order)))%0A%20%20%20%20%20%20%20%20_ax_bias.set_xticklabels(_bin_order%2C%20fontsize%3D9%2C%20rotation%3D-20%2C%20ha%3D%22left%22)%0A%20%20%20%20%20%20%20%20_ax_bias.set_yticks(range(len(_model_names)))%0A%20%20%20%20%20%20%20%20_ax_bias.set_yticklabels(_model_names%2C%20fontsize%3D7)%0A%20%20%20%20%20%20%20%20_ax_bias.set_xlabel(%22pEC50%20bin%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax_bias.set_title(%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Per-bin%20mean%20prediction%20error%20%E2%80%94%20all%20submissions%20(Phase%202)%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%22(red%20%3D%20overpredict%2C%20blue%20%3D%20underpredict)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D11%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20_cb%20%3D%20_fig_bias.colorbar(_im%2C%20ax%3D_ax_bias%2C%20fraction%3D0.03%2C%20pad%3D0.03)%0A%20%20%20%20%20%20%20%20_cb.set_label(%22Mean%20error%20(pred%20%E2%88%92%20true)%22%2C%20fontsize%3D10)%0A%20%20%20%20%20%20%20%20_fig_bias.tight_layout()%0A%20%20%20%20%20%20%20%20_fig_bias.savefig(PLOT_DIR%20%2F%20%22bias_heatmap.png%22%2C%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20mo.center(mo.as_html(_fig_bias))%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Phase%201%20vs.%20Phase%202%20%E2%80%94%20does%20MAE%20generalise%20across%20test%20sets%3F%0A%0A%20%20%20%20Every%20submission%20CSV%20covers%20the%20**full%20513-compound%20test%20set**%20(253%20Phase%201%20%2B%0A%20%20%20%20260%20Phase%202)%2C%20even%20though%20the%20analysis%20above%20only%20scores%20the%20Phase%202%20half.%20%20Since%0A%20%20%20%20the%20Phase%201%20unblinded%20truth%20(%60unblinded_p1%60)%20is%20already%20loaded%2C%20we%20can%20score%20every%0A%20%20%20%20submission%20against%20**both**%20test%20sets%20and%20compare%20MAE%20side%20by%20side.%20%20This%20tells%20us%0A%20%20%20%20whether%20a%20submission's%20ranking%20is%20an%20artefact%20of%20the%20(smaller)%20Phase%202%20set%20or%0A%20%20%20%20reflects%20a%20genuinely%20stable%20model.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20EXTERNAL_PATH%2C%0A%20%20%20%20FINAL_NAME%2C%0A%20%20%20%20SUBMISSION_DIR%2C%0A%20%20%20%20all_predictions%3A%20%22pl.DataFrame%22%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20unblinded_p1%3A%20%22pl.DataFrame%22%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Ground%20truth%20for%20Phase%201%20(mirrors%20%60truth%60%20above%2C%20which%20is%20Phase%202-only)%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20truth_p1%3A%20pl.DataFrame%20%3D%20unblinded_p1.select(%5B%22Molecule%20Name%22%2C%20%22pEC50%22%5D).rename(%0A%20%20%20%20%20%20%20%20%7B%22pEC50%22%3A%20%22pEC50_true%22%7D%0A%20%20%20%20)%0A%0A%20%20%20%20def%20_load_and_score_p1(path%3A%20%22Path%22%2C%20label%3A%20str)%20-%3E%20pl.DataFrame%3A%0A%20%20%20%20%20%20%20%20%22%22%22Score%20one%20submission%20CSV%20against%20the%20Phase%201%20truth%20(same%20shape%20as%0A%20%20%20%20%20%20%20%20%60load_and_score%60%2C%20but%20joined%20against%20%60truth_p1%60%20instead%20of%20Phase%202%20%60truth%60).%22%22%22%0A%20%20%20%20%20%20%20%20df%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.read_csv(path)%0A%20%20%20%20%20%20%20%20%20%20%20%20.select(%5B%22Molecule%20Name%22%2C%20%22pEC50%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20.rename(%7B%22pEC50%22%3A%20%22pEC50_pred%22%7D)%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20return%20df.join(truth_p1%2C%20on%3D%22Molecule%20Name%22%2C%20how%3D%22inner%22).with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.lit(label).alias(%22model_name%22)%2C%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20_sub_paths%20%3D%20sorted(SUBMISSION_DIR.glob(%22*_submission.csv%22))%0A%20%20%20%20_frames_p1%20%3D%20%5B%0A%20%20%20%20%20%20%20%20_load_and_score_p1(_p%2C%20_p.name.removesuffix(%22_submission.csv%22).removesuffix(%22.csv%22))%0A%20%20%20%20%20%20%20%20for%20_p%20in%20_sub_paths%0A%20%20%20%20%5D%0A%20%20%20%20_frames_p1.append(_load_and_score_p1(EXTERNAL_PATH%2C%20%22rank9_gashaw_blend%20(external)%22))%0A%0A%20%20%20%20predictions_p1%3A%20pl.DataFrame%20%3D%20pl.concat(_frames_p1)%0A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20MAE%20per%20submission%2C%20Phase%201%20vs%20Phase%202%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_mae_p1%20%3D%20(%0A%20%20%20%20%20%20%20%20predictions_p1%0A%20%20%20%20%20%20%20%20.group_by(%22model_name%22)%0A%20%20%20%20%20%20%20%20.agg((pl.col(%22pEC50_pred%22)%20-%20pl.col(%22pEC50_true%22)).abs().mean().alias(%22MAE_phase1%22))%0A%20%20%20%20)%0A%20%20%20%20_mae_p2%20%3D%20all_predictions.select(%5B%22model_name%22%2C%20%22is_final%22%2C%20%22is_external%22%5D).unique(%0A%20%20%20%20%20%20%20%20%22model_name%22%0A%20%20%20%20).join(%0A%20%20%20%20%20%20%20%20all_predictions.group_by(%22model_name%22).agg(pl.col(%22abs_error%22).mean().alias(%22MAE_phase2%22))%2C%0A%20%20%20%20%20%20%20%20on%3D%22model_name%22%2C%0A%20%20%20%20)%0A%0A%20%20%20%20phase_mae_df%3A%20pl.DataFrame%20%3D%20(%0A%20%20%20%20%20%20%20%20_mae_p2.join(_mae_p1%2C%20on%3D%22model_name%22%2C%20how%3D%22inner%22)%0A%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22MAE_phase2%22)%20-%20pl.col(%22MAE_phase1%22)).alias(%22delta_mae%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22model_name%22).eq(FINAL_NAME.removesuffix(%22_submission.csv%22)).alias(%22is_final_row%22)%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20.sort(%22MAE_phase2%22)%0A%20%20%20%20)%0A%0A%20%20%20%20phase_mae_df%0A%20%20%20%20return%20(phase_mae_df%2C)%0A%0A%0A%40app.cell%0Adef%20_(mo%2C%20phase_mae_df%3A%20%22pl.DataFrame%22%2C%20pl)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Rank%20agreement%20between%20the%20two%20test%20sets%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_own%20%3D%20phase_mae_df.filter(~pl.col(%22is_external%22))%0A%20%20%20%20_rank_p2%20%3D%20_own.sort(%22MAE_phase2%22).with_row_index(%22rank_phase2%22%2C%20offset%3D1)%0A%20%20%20%20_rank_p1%20%3D%20_own.sort(%22MAE_phase1%22).with_row_index(%22rank_phase1%22%2C%20offset%3D1)%0A%20%20%20%20_ranks%20%3D%20_rank_p2.select(%5B%22model_name%22%2C%20%22rank_phase2%22%5D).join(%0A%20%20%20%20%20%20%20%20_rank_p1.select(%5B%22model_name%22%2C%20%22rank_phase1%22%5D)%2C%20on%3D%22model_name%22%0A%20%20%20%20)%0A%20%20%20%20_spearman_rank%20%3D%20_ranks.select(pl.corr(%22rank_phase1%22%2C%20%22rank_phase2%22%2C%20method%3D%22spearman%22)).item()%0A%0A%20%20%20%20_mean_gap%20%3D%20float(_own.get_column(%22delta_mae%22).abs().mean())%0A%20%20%20%20_worst_row%20%3D%20_own.sort(pl.col(%22delta_mae%22).abs()%2C%20descending%3DTrue).row(0%2C%20named%3DTrue)%0A%0A%20%20%20%20mo.md(f%22%22%22%0A%20%20%20%20%23%23%23%20Summary%0A%0A%20%20%20%20-%20**Spearman%20rank%20correlation**%20between%20Phase%201%20MAE%20rank%20and%20Phase%202%20MAE%20rank%0A%20%20%20%20%20%20(our%20own%20submissions%20only)%3A%20**%7B_spearman_rank%3A.3f%7D**%20%E2%80%94%20%7B%22the%20two%20test%20sets%20agree%20closely%20on%20which%20submissions%20are%20best.%22%20if%20_spearman_rank%20%3E%200.7%20else%20%22the%20two%20test%20sets%20disagree%20substantially%20on%20which%20submissions%20are%20best.%22%7D%0A%20%20%20%20-%20**Mean%20%7CMAE%20gap%7C**%20across%20submissions%3A%20%7B_mean_gap%3A.4f%7D%20pEC50%20units.%0A%20%20%20%20-%20**Largest%20gap%3A**%20%60%7B_worst_row%5B'model_name'%5D%7D%60%20%E2%80%94%20MAE(Phase%201)%20%3D%20%7B_worst_row%5B'MAE_phase1'%5D%3A.4f%7D%20vs.%0A%20%20%20%20%20%20MAE(Phase%202)%20%3D%20%7B_worst_row%5B'MAE_phase2'%5D%3A.4f%7D%20(%CE%94%20%3D%20%7B_worst_row%5B'delta_mae'%5D%3A%2B.4f%7D).%0A%0A%20%20%20%20A%20high%20rank%20correlation%20means%20Phase%202%20was%20not%20a%20fluke%20%E2%80%94%20the%20submissions%20that%20did%0A%20%20%20%20well%20there%20also%20did%20well%20on%20the%20independent%20Phase%201%20set.%20%20A%20submission%20with%20a%0A%20%20%20%20Phase%202%20MAE%20much%20lower%20than%20its%20Phase%201%20MAE%20may%20simply%20have%20gotten%20lucky%20on%20the%0A%20%20%20%20smaller%20Phase%202%20set%20rather%20than%20being%20a%20genuinely%20better%20model.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(PLOT_DIR%2C%20mo%2C%20phase_mae_df%3A%20%22pl.DataFrame%22%2C%20plt)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Static%20scatter%3A%20MAE(Phase%201)%20vs%20MAE(Phase%202)%2C%20coloured%20by%20notebook%20family%20%E2%94%80%E2%94%80%0A%20%20%20%20_family_by_prefix%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%222%22%3A%20%222_%20%E2%80%94%20Baselines%22%2C%0A%20%20%20%20%20%20%20%20%223%22%3A%20%223_%20%E2%80%94%20First%20optimisation%22%2C%0A%20%20%20%20%20%20%20%20%224%22%3A%20%224_%20%E2%80%94%20Second%20optimisation%22%2C%0A%20%20%20%20%20%20%20%20%225%22%3A%20%225_%20%E2%80%94%20Phase%201%20regen%22%2C%0A%20%20%20%20%20%20%20%20%226%22%3A%20%226_%20%E2%80%94%20Final%20optimisation%22%2C%0A%20%20%20%20%7D%0A%20%20%20%20_family_color%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%222_%20%E2%80%94%20Baselines%22%3A%20%20%20%20%20%20%20%20%20%20%20%20%22%2359a14f%22%2C%0A%20%20%20%20%20%20%20%20%223_%20%E2%80%94%20First%20optimisation%22%3A%20%20%20%22%234e79a7%22%2C%0A%20%20%20%20%20%20%20%20%224_%20%E2%80%94%20Second%20optimisation%22%3A%20%20%22%23f28e2b%22%2C%0A%20%20%20%20%20%20%20%20%225_%20%E2%80%94%20Phase%201%20regen%22%3A%20%20%20%20%20%20%20%20%22%23b07aa1%22%2C%0A%20%20%20%20%20%20%20%20%226_%20%E2%80%94%20Final%20optimisation%22%3A%20%20%20%22%23e15759%22%2C%0A%20%20%20%20%20%20%20%20%22External%22%3A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22%23b0b0b0%22%2C%0A%20%20%20%20%7D%0A%0A%20%20%20%20def%20_family_of(model_name%3A%20str)%20-%3E%20str%3A%0A%20%20%20%20%20%20%20%20if%20%22external%22%20in%20model_name%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20%22External%22%0A%20%20%20%20%20%20%20%20return%20_family_by_prefix.get(model_name.split(%22_%22%2C%201)%5B0%5D%2C%20%22External%22)%0A%0A%20%20%20%20_rows%20%3D%20phase_mae_df.iter_rows(named%3DTrue)%0A%20%20%20%20_lo%20%3D%20min(phase_mae_df%5B%22MAE_phase1%22%5D.min()%2C%20phase_mae_df%5B%22MAE_phase2%22%5D.min())%20-%200.02%0A%20%20%20%20_hi%20%3D%20max(phase_mae_df%5B%22MAE_phase1%22%5D.max()%2C%20phase_mae_df%5B%22MAE_phase2%22%5D.max())%20%2B%200.02%0A%0A%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20_fig_pmae%2C%20_ax_pmae%20%3D%20plt.subplots(figsize%3D(6.5%2C%206.5)%2C%20dpi%3D150)%0A%0A%20%20%20%20%20%20%20%20_ax_pmae.plot(%5B_lo%2C%20_hi%5D%2C%20%5B_lo%2C%20_hi%5D%2C%20linestyle%3D%22--%22%2C%20color%3D%22grey%22%2C%20linewidth%3D1.2%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20label%3D%22Equal%20MAE%22%2C%20zorder%3D1)%0A%0A%20%20%20%20%20%20%20%20for%20_r%20in%20_rows%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_fam%20%3D%20_family_of(_r%5B%22model_name%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20_edge%20%3D%20%22black%22%20if%20_r%5B%22is_final%22%5D%20else%20%22none%22%0A%20%20%20%20%20%20%20%20%20%20%20%20_lw%20%3D%201.4%20if%20_r%5B%22is_final%22%5D%20else%200%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax_pmae.scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_r%5B%22MAE_phase1%22%5D%2C%20_r%5B%22MAE_phase2%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20s%3D110%2C%20color%3D_family_color%5B_fam%5D%2C%20edgecolor%3D_edge%2C%20linewidth%3D_lw%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20alpha%3D0.85%2C%20zorder%3D2%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20_ax_pmae.set_xlim(_lo%2C%20_hi)%0A%20%20%20%20%20%20%20%20_ax_pmae.set_ylim(_lo%2C%20_hi)%0A%20%20%20%20%20%20%20%20_ax_pmae.set_xlabel(%22MAE%20%E2%80%94%20Phase%201%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax_pmae.set_ylabel(%22MAE%20%E2%80%94%20Phase%202%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax_pmae.set_title(%0A%20%20%20%20%20%20%20%20%20%20%20%20%22MAE%20per%20submission%20%E2%80%94%20Phase%201%20vs.%20Phase%202%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%22(coloured%20by%20notebook%20family%3B%20black%20outline%20%3D%20our%20final%20submission)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D11%2C%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20%23%20Legend%20via%20proxy%20handles%2C%20ordered%202_%20%E2%86%92%206_%2C%20then%20External%2C%20then%20the%20diagonal.%0A%20%20%20%20%20%20%20%20from%20matplotlib.lines%20import%20Line2D%20as%20_Line2D_pmae%0A%20%20%20%20%20%20%20%20_legend_order%20%3D%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%222_%20%E2%80%94%20Baselines%22%2C%20%223_%20%E2%80%94%20First%20optimisation%22%2C%20%224_%20%E2%80%94%20Second%20optimisation%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%225_%20%E2%80%94%20Phase%201%20regen%22%2C%20%226_%20%E2%80%94%20Final%20optimisation%22%2C%20%22External%22%2C%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20%20%20%20%20_handles%20%3D%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20_Line2D_pmae(%5B0%5D%2C%20%5B0%5D%2C%20marker%3D%22o%22%2C%20color%3D%22w%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20markerfacecolor%3D_family_color%5B_lbl%5D%2C%20markersize%3D9%2C%20label%3D_lbl)%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20_lbl%20in%20_legend_order%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20%20%20%20%20_handles.append(%0A%20%20%20%20%20%20%20%20%20%20%20%20_Line2D_pmae(%5B0%5D%2C%20%5B0%5D%2C%20linestyle%3D%22--%22%2C%20color%3D%22grey%22%2C%20label%3D%22Equal%20MAE%22)%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20_ax_pmae.legend(handles%3D_handles%2C%20fontsize%3D8%2C%20frameon%3DTrue%2C%20framealpha%3D0.9%2C%20loc%3D%22upper%20left%22)%0A%20%20%20%20%20%20%20%20_fig_pmae.tight_layout()%0A%20%20%20%20%20%20%20%20_fig_pmae.savefig(PLOT_DIR%20%2F%20%22phase1_vs_phase2_mae_scatter.png%22%2C%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20mo.center(mo.as_html(_fig_pmae))%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
9b9e56668bd77613f4b663c00b2a886d