import%20marimo%0A%0A__generated_with%20%3D%20%220.23.5%22%0Aapp%20%3D%20marimo.App(width%3D%22medium%22)%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%206%20%E2%80%94%20ML%20optimization%203%3A%20calibration%2C%20reweighting%20and%20extra%20data%0A%0A%20%20%20%20The%20unblinded%20analysis%20(notebook%205%20%2F%20the%0A%20%20%20%20%5Bblog%20post%5D(https%3A%2F%2Fgithub.com%2Fadlvdl%2Fpxr_challenge))%20exposed%20two%20correctable%0A%20%20%20%20failure%20modes%20in%20the%20Phase-1%20models%3A%0A%0A%20%20%20%201.%20**Regression%20to%20the%20mean%20at%20the%20activity%20extremes.**%20Every%20model%20overpredicts%0A%20%20%20%20%20%20%20inactive%20compounds%20and%20underpredicts%20the%20rare%20potent%20ones.%20On%20the%20unblinded%0A%20%20%20%20%20%20%20set%20the%20hit%20zone%20(pEC50%20%3E%206)%20was%20underpredicted%20by%20up%20to%20a%20full%20log%20unit%20%E2%80%94%0A%20%20%20%20%20%20%20precisely%20the%20region%20where%20a%20screening%20decision%20is%20made.%0A%20%20%20%202.%20**Unused%20auxiliary%20assay%20data.**%20Beyond%20the%20dose-response%20labels%2C%20OpenADMET%0A%20%20%20%20%20%20%20released%20additional%20measurements%20that%20other%20participants%20found%20useful.%0A%0A%20%20%20%20This%20notebook%20tests%20three%20concrete%20remedies%20the%20blog%20post%20proposed%20as%20next%20steps%3A%0A%0A%20%20%20%20%7C%20Analysis%20%7C%20Idea%20%7C%20Targets%20failure%20%7C%0A%20%20%20%20%7C---%7C---%7C---%7C%0A%20%20%20%20%7C%20**1%20%E2%80%94%20Post-hoc%20calibration**%20%7C%20De-shrink%20predictions%20(linear%20%2B%20isotonic)%20%7C%20%231%20%7C%0A%20%20%20%20%7C%20**2%20%E2%80%94%20Loss%2Fsample%20reweighting**%20%7C%20Up-weight%20extreme-activity%20compounds%20in%20training%20%7C%20%231%20%7C%0A%20%20%20%20%7C%20**3%20%E2%80%94%20Oversampling%20extremes**%20%7C%20Duplicate%20extreme%20compounds%20(works%20for%20every%20model)%20%7C%20%231%20%7C%0A%20%20%20%20%7C%20**4%20%E2%80%94%20Extra%20assay%20data**%20%7C%20Augment%20training%20with%20the%2096-compound%20*semi-pure*%20set%20%7C%20%232%20%7C%0A%20%20%20%20%7C%20**5%20%E2%80%94%20Filtering%20training%20data**%20%7C%20Remove%20unreliable%20points%20(counter%20%2F%20cliffs%20%2F%20kNN%20noise)%20%7C%20label%20noise%20%7C%0A%0A%20%20%20%20**Protocol.**%20Strategies%20are%20compared%20with%20the%20same%20**5%C3%975%20cross-validation**%20used%0A%20%20%20%20throughout%20notebooks%202%E2%80%934%20(random%20split%2C%20seed%20%3D%2042).%20To%20avoid%20biasing%20decisions%0A%20%20%20%20toward%20the%20now-known%20Phase-1%20labels%2C%20**only%20the%20single%20best%20strategy%20chosen%20by%20CV%0A%20%20%20%20is%20applied%20to%20the%20unblinded%20test%20set**%20at%20the%20very%20end.%0A%0A%20%20%20%20%3E%20Uncertainty%20estimation%20%E2%80%94%20the%20third%20idea%20in%20the%20blog's%20*Next%20steps*%20%E2%80%94%20is%20left%20for%0A%20%20%20%20%3E%20a%20later%20notebook%3A%20it%20does%20not%20directly%20improve%20the%20single-point%20predictions%20the%0A%20%20%20%20%3E%20submission%20requires.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%20Imports%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20os%0A%20%20%20%20%23%20Let%20PyTorch's%20libomp%20and%20sklearn's%20libomp%20coexist%20in%20one%20process%20(needed%0A%20%20%20%20%23%20because%20Chemprop's%20device%20check%20imports%20torch%20while%20RF%2FXGBoost%20run%20in-process).%0A%20%20%20%20os.environ.setdefault(%22KMP_DUPLICATE_LIB_OK%22%2C%20%22TRUE%22)%0A%20%20%20%20%23%20Force%20the%20non-interactive%20Agg%20backend%3A%20figures%20are%20only%20saved%20%2F%20rendered%20to%0A%20%20%20%20%23%20HTML%2C%20never%20shown%20in%20a%20GUI.%20Avoids%20the%20macOS%20backend%20import%20error%20under%0A%20%20%20%20%23%20headless%20execution%20(and%20is%20exactly%20what%20marimo%20needs%20to%20render%20figures).%0A%20%20%20%20os.environ.setdefault(%22MPLBACKEND%22%2C%20%22Agg%22)%0A%0A%20%20%20%20import%20gc%0A%20%20%20%20import%20gzip%0A%20%20%20%20import%20json%0A%20%20%20%20import%20math%0A%20%20%20%20import%20shutil%0A%20%20%20%20import%20subprocess%0A%20%20%20%20import%20sys%0A%20%20%20%20import%20tempfile%0A%20%20%20%20import%20warnings%0A%20%20%20%20from%20pathlib%20import%20Path%0A%20%20%20%20from%20typing%20import%20Iterator%2C%20Optional%0A%0A%20%20%20%20import%20marimo%20as%20mo%0A%20%20%20%20import%20matplotlib.pyplot%20as%20plt%0A%20%20%20%20%23%20Initialise%20the%20matplotlib%20backend%20%2B%20ft2font%20C%20extension%20NOW%2C%20before%20the%20heavy%0A%20%20%20%20%23%20native%20libraries%20(torch%2Fsmurff%2Fskfp)%20load%20%E2%80%94%20otherwise%20a%20shared-library%20symbol%0A%20%20%20%20%23%20clash%20breaks%20ft2font's%20import%20at%20the%20first%20figure.%0A%20%20%20%20plt.close(plt.figure())%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20pandas%20as%20pd%0A%20%20%20%20import%20pingouin%20as%20pg%0A%20%20%20%20import%20polars%20as%20pl%0A%20%20%20%20import%20seaborn%20as%20sns%0A%20%20%20%20from%20tqdm.auto%20import%20tqdm%0A%0A%20%20%20%20from%20scipy.stats%20import%20spearmanr%0A%20%20%20%20from%20statsmodels.stats.libqsturng%20import%20psturng%2C%20qsturng%0A%0A%20%20%20%20from%20sklearn.ensemble%20import%20RandomForestClassifier%2C%20RandomForestRegressor%0A%20%20%20%20from%20sklearn.isotonic%20import%20IsotonicRegression%0A%20%20%20%20from%20sklearn.linear_model%20import%20LinearRegression%0A%20%20%20%20from%20sklearn.metrics%20import%20(%0A%20%20%20%20%20%20%20%20mean_absolute_error%2C%0A%20%20%20%20%20%20%20%20mean_squared_error%2C%0A%20%20%20%20%20%20%20%20precision_score%2C%0A%20%20%20%20%20%20%20%20r2_score%2C%0A%20%20%20%20%20%20%20%20recall_score%2C%0A%20%20%20%20)%0A%20%20%20%20from%20sklearn.model_selection._split%20import%20_BaseKFold%20as%20BaseKFold%0A%0A%20%20%20%20import%20xgboost%20as%20xgb%0A%20%20%20%20import%20torch%0A%20%20%20%20import%20smurff%0A%20%20%20%20import%20scipy.sparse%20as%20sp%0A%0A%20%20%20%20from%20rdkit%20import%20Chem%2C%20RDLogger%0A%20%20%20%20from%20skfp.fingerprints%20import%20(%0A%20%20%20%20%20%20%20%20ECFPFingerprint%2C%0A%20%20%20%20%20%20%20%20MACCSFingerprint%2C%0A%20%20%20%20%20%20%20%20MordredFingerprint%2C%0A%20%20%20%20%20%20%20%20MQNsFingerprint%2C%0A%20%20%20%20)%0A%20%20%20%20from%20skfp.preprocessing%20import%20ConformerGenerator%2C%20MolFromSmilesTransformer%0A%0A%20%20%20%20RDLogger.DisableLog(%22rdApp.*%22)%0A%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20BaseKFold%2C%0A%20%20%20%20%20%20%20%20Chem%2C%0A%20%20%20%20%20%20%20%20ConformerGenerator%2C%0A%20%20%20%20%20%20%20%20ECFPFingerprint%2C%0A%20%20%20%20%20%20%20%20IsotonicRegression%2C%0A%20%20%20%20%20%20%20%20Iterator%2C%0A%20%20%20%20%20%20%20%20LinearRegression%2C%0A%20%20%20%20%20%20%20%20MACCSFingerprint%2C%0A%20%20%20%20%20%20%20%20MQNsFingerprint%2C%0A%20%20%20%20%20%20%20%20MolFromSmilesTransformer%2C%0A%20%20%20%20%20%20%20%20MordredFingerprint%2C%0A%20%20%20%20%20%20%20%20Optional%2C%0A%20%20%20%20%20%20%20%20Path%2C%0A%20%20%20%20%20%20%20%20RandomForestClassifier%2C%0A%20%20%20%20%20%20%20%20RandomForestRegressor%2C%0A%20%20%20%20%20%20%20%20gc%2C%0A%20%20%20%20%20%20%20%20gzip%2C%0A%20%20%20%20%20%20%20%20json%2C%0A%20%20%20%20%20%20%20%20math%2C%0A%20%20%20%20%20%20%20%20mean_absolute_error%2C%0A%20%20%20%20%20%20%20%20mean_squared_error%2C%0A%20%20%20%20%20%20%20%20mo%2C%0A%20%20%20%20%20%20%20%20np%2C%0A%20%20%20%20%20%20%20%20pd%2C%0A%20%20%20%20%20%20%20%20pg%2C%0A%20%20%20%20%20%20%20%20pl%2C%0A%20%20%20%20%20%20%20%20plt%2C%0A%20%20%20%20%20%20%20%20precision_score%2C%0A%20%20%20%20%20%20%20%20psturng%2C%0A%20%20%20%20%20%20%20%20qsturng%2C%0A%20%20%20%20%20%20%20%20r2_score%2C%0A%20%20%20%20%20%20%20%20recall_score%2C%0A%20%20%20%20%20%20%20%20shutil%2C%0A%20%20%20%20%20%20%20%20smurff%2C%0A%20%20%20%20%20%20%20%20sns%2C%0A%20%20%20%20%20%20%20%20sp%2C%0A%20%20%20%20%20%20%20%20spearmanr%2C%0A%20%20%20%20%20%20%20%20subprocess%2C%0A%20%20%20%20%20%20%20%20sys%2C%0A%20%20%20%20%20%20%20%20tempfile%2C%0A%20%20%20%20%20%20%20%20torch%2C%0A%20%20%20%20%20%20%20%20tqdm%2C%0A%20%20%20%20%20%20%20%20warnings%2C%0A%20%20%20%20%20%20%20%20xgb%2C%0A%20%20%20%20)%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Shared%20infrastructure%0A%0A%20%20%20%20The%20model%20classes%2C%20fingerprint%20helpers%2C%20cross-validation%20splitter%20and%20the%0A%20%20%20%20Tukey-HSD%20multiple-comparison%20plots%20are%20copied%20verbatim%20from%20notebook%204%20so%20the%0A%20%20%20%205%C3%975%20CV%20results%20are%20directly%20comparable.%20Two%20small%20extensions%20are%20made%3A%0A%0A%20%20%20%20-%20%60RandomForestModel.train%60%20and%20%60BoostedTreesModel.train%60%20accept%20an%20optional%0A%20%20%20%20%20%20%60sample_weight%60%2C%20needed%20for%20the%20reweighting%20analysis.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(Path%2C%20json%2C%20np%2C%20subprocess%2C%20sys%2C%20tempfile)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20CheMeleon%20embedding%20subprocess%20script%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20CheMeleon%20(PyTorch)%20must%20run%20in%20an%20isolated%20subprocess%20to%20avoid%20the%20OpenMP%0A%20%20%20%20%23%20runtime%20collision%20between%20PyTorch's%20libkmp%20and%20sklearn%20in%20the%20parent%20process.%0A%20%20%20%20%23%20The%20script%20is%20written%20once%20here%20and%20reused%20by%20all%20analysis%20cells.%0A%20%20%20%20_CHEMELEON_SCRIPT%20%3D%20%22%5Cn%22.join(%5B%0A%20%20%20%20%20%20%20%20%22import%20os%2C%20json%2C%20sys%2C%20numpy%20as%20np%22%2C%0A%20%20%20%20%20%20%20%20%22os.environ%5B'KMP_DUPLICATE_LIB_OK'%5D%20%3D%20'TRUE'%22%2C%0A%20%20%20%20%20%20%20%20%22from%20pathlib%20import%20Path%22%2C%0A%20%20%20%20%20%20%20%20%22import%20torch%22%2C%0A%20%20%20%20%20%20%20%20%22from%20chemprop%20import%20featurizers%22%2C%0A%20%20%20%20%20%20%20%20%22from%20chemprop%20import%20nn%20as%20chemnn%22%2C%0A%20%20%20%20%20%20%20%20%22from%20chemprop.models%20import%20MPNN%22%2C%0A%20%20%20%20%20%20%20%20%22from%20chemprop.nn%20import%20RegressionFFN%22%2C%0A%20%20%20%20%20%20%20%20%22from%20chemprop.data%20import%20BatchMolGraph%22%2C%0A%20%20%20%20%20%20%20%20%22from%20rdkit.Chem%20import%20MolFromSmiles%22%2C%0A%20%20%20%20%20%20%20%20%22smiles_file%2C%20out_train%2C%20out_test%20%3D%20sys.argv%5B1%5D%2C%20sys.argv%5B2%5D%2C%20sys.argv%5B3%5D%22%2C%0A%20%20%20%20%20%20%20%20%22with%20open(smiles_file)%20as%20f%3A%22%2C%0A%20%20%20%20%20%20%20%20%22%20%20%20%20data%20%3D%20json.load(f)%22%2C%0A%20%20%20%20%20%20%20%20%22mp_path%20%3D%20Path.home()%20%2F%20'.chemprop'%20%2F%20'chemeleon_mp.pt'%22%2C%0A%20%20%20%20%20%20%20%20%22ckpt%20%3D%20torch.load(mp_path%2C%20weights_only%3DTrue)%22%2C%0A%20%20%20%20%20%20%20%20%22mp%20%3D%20chemnn.BondMessagePassing(**ckpt%5B'hyper_parameters'%5D)%22%2C%0A%20%20%20%20%20%20%20%20%22mp.load_state_dict(ckpt%5B'state_dict'%5D)%22%2C%0A%20%20%20%20%20%20%20%20%22model%20%3D%20MPNN(mp%2C%20chemnn.MeanAggregation()%2C%20RegressionFFN(input_dim%3Dmp.output_dim))%22%2C%0A%20%20%20%20%20%20%20%20%22model.eval()%22%2C%0A%20%20%20%20%20%20%20%20%22feat%20%3D%20featurizers.SimpleMoleculeMolGraphFeaturizer()%22%2C%0A%20%20%20%20%20%20%20%20%22def%20embed(smiles)%3A%22%2C%0A%20%20%20%20%20%20%20%20%22%20%20%20%20bmg%20%3D%20BatchMolGraph(%5Bfeat(MolFromSmiles(s))%20for%20s%20in%20smiles%5D)%22%2C%0A%20%20%20%20%20%20%20%20%22%20%20%20%20with%20torch.no_grad()%3A%22%2C%0A%20%20%20%20%20%20%20%20%22%20%20%20%20%20%20%20%20return%20model.fingerprint(bmg).numpy(force%3DTrue)%22%2C%0A%20%20%20%20%20%20%20%20%22np.save(out_train%2C%20embed(data%5B'train'%5D))%22%2C%0A%20%20%20%20%20%20%20%20%22np.save(out_test%2C%20%20embed(data%5B'test'%5D))%22%2C%0A%20%20%20%20%5D)%0A%20%20%20%20_CHEMELEON_SCRIPT_PATH%20%3D%20Path(tempfile.gettempdir())%20%2F%20%22chemeleon_embed.py%22%0A%20%20%20%20_CHEMELEON_SCRIPT_PATH.write_text(_CHEMELEON_SCRIPT)%0A%0A%20%20%20%20def%20chemeleon_embed(%0A%20%20%20%20%20%20%20%20smiles_train%3A%20list%5Bstr%5D%2C%0A%20%20%20%20%20%20%20%20smiles_test%3A%20list%5Bstr%5D%2C%0A%20%20%20%20%20%20%20%20prefix%3A%20str%20%3D%20%22chemeleon%22%2C%0A%20%20%20%20)%20-%3E%20tuple%5Bnp.ndarray%2C%20np.ndarray%5D%3A%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20Generate%20CheMeleon%20embeddings%20for%20train%20and%20test%20SMILES%20in%20an%20isolated%0A%20%20%20%20%20%20%20%20subprocess%2C%20returning%20(X_train%2C%20X_test)%20as%20float32%20numpy%20arrays.%0A%0A%20%20%20%20%20%20%20%20Args%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20smiles_train%3A%20SMILES%20strings%20for%20the%20training%20set.%0A%20%20%20%20%20%20%20%20%20%20%20%20smiles_test%3A%20%20SMILES%20strings%20for%20the%20test%20set.%0A%20%20%20%20%20%20%20%20%20%20%20%20prefix%3A%20Prefix%20for%20the%20temp%20files%20written%20by%20this%20call.%0A%0A%20%20%20%20%20%20%20%20Returns%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20Tuple%20(X_train%2C%20X_test)%20of%20shape%20(n_train%2C%202048)%20and%20(n_test%2C%202048).%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20tmp%20%3D%20Path(tempfile.gettempdir())%0A%20%20%20%20%20%20%20%20smi_file%20%20%20%3D%20tmp%20%2F%20f%22%7Bprefix%7D_smiles.json%22%0A%20%20%20%20%20%20%20%20train_file%20%3D%20tmp%20%2F%20f%22%7Bprefix%7D_train%22%0A%20%20%20%20%20%20%20%20test_file%20%20%3D%20tmp%20%2F%20f%22%7Bprefix%7D_test%22%0A%20%20%20%20%20%20%20%20smi_file.write_text(json.dumps(%7B%22train%22%3A%20smiles_train%2C%20%22test%22%3A%20smiles_test%7D))%0A%20%20%20%20%20%20%20%20result%20%3D%20subprocess.run(%0A%20%20%20%20%20%20%20%20%20%20%20%20%5Bsys.executable%2C%20str(_CHEMELEON_SCRIPT_PATH)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20str(smi_file)%2C%20str(train_file)%2C%20str(test_file)%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20capture_output%3DTrue%2C%20text%3DTrue%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20if%20result.returncode%20!%3D%200%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20raise%20RuntimeError(f%22CheMeleon%20subprocess%20failed%3A%5Cn%7Bresult.stderr%7D%22)%0A%20%20%20%20%20%20%20%20X_train%20%3D%20np.load(str(train_file)%20%2B%20%22.npy%22)%0A%20%20%20%20%20%20%20%20X_test%20%20%3D%20np.load(str(test_file)%20%20%2B%20%22.npy%22)%0A%20%20%20%20%20%20%20%20for%20p%20in%20%5Bsmi_file%2C%20Path(str(train_file)%20%2B%20%22.npy%22)%2C%20Path(str(test_file)%20%2B%20%22.npy%22)%5D%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20p.unlink(missing_ok%3DTrue)%0A%20%20%20%20%20%20%20%20return%20X_train%2C%20X_test%0A%0A%20%20%20%20return%20(chemeleon_embed%2C)%0A%0A%0A%40app.cell%0Adef%20_(np%2C%20pl)%3A%0A%20%20%20%20def%20extract_fp_matrix(df%3A%20pl.DataFrame%2C%20fp_col%3A%20str)%20-%3E%20np.ndarray%3A%0A%20%20%20%20%20%20%20%20%22%22%22Extract%20a%202-D%20float32%20feature%20matrix%20from%20a%20fingerprint%20column.%22%22%22%0A%20%20%20%20%20%20%20%20return%20np.stack(df%5Bfp_col%5D.to_list()).astype(np.float32)%0A%0A%20%20%20%20return%20(extract_fp_matrix%2C)%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20ConformerGenerator%2C%0A%20%20%20%20ECFPFingerprint%2C%0A%20%20%20%20MACCSFingerprint%2C%0A%20%20%20%20MQNsFingerprint%2C%0A%20%20%20%20MolFromSmilesTransformer%2C%0A%20%20%20%20MordredFingerprint%2C%0A%20%20%20%20pl%2C%0A)%3A%0A%20%20%20%20_fp_dict%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22ecfp%22%3A%20ECFPFingerprint%2C%0A%20%20%20%20%20%20%20%20%22morgan%22%3A%20ECFPFingerprint%2C%0A%20%20%20%20%20%20%20%20%22maccs%22%3A%20MACCSFingerprint%2C%0A%20%20%20%20%20%20%20%20%22mordred%22%3A%20MordredFingerprint%2C%0A%20%20%20%20%20%20%20%20%22mqn%22%3A%20MQNsFingerprint%2C%0A%20%20%20%20%7D%0A%0A%20%20%20%20def%20generate_fingerprint(df%3A%20pl.DataFrame%2C%20fingerprint_type%3A%20str%2C%20**kwargs)%20-%3E%20pl.DataFrame%3A%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20Generate%20molecular%20fingerprints%20and%20add%20them%20as%20a%20new%20column%20to%20the%20DataFrame.%0A%0A%20%20%20%20%20%20%20%20CheMeleon%20embeddings%20are%20not%20handled%20here%20%E2%80%94%20use%20the%20%60chemeleon_embed%60%0A%20%20%20%20%20%20%20%20function%20which%20runs%20inference%20in%20an%20isolated%20subprocess.%0A%0A%20%20%20%20%20%20%20%20Args%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20df%3A%20Polars%20DataFrame%20containing%20a%20%22smiles%22%20column.%0A%20%20%20%20%20%20%20%20%20%20%20%20fingerprint_type%3A%20One%20of%20the%20keys%20in%20%60_fp_dict%60.%0A%20%20%20%20%20%20%20%20%20%20%20%20**kwargs%3A%20Forwarded%20to%20the%20fingerprint%20class%20constructor.%0A%0A%20%20%20%20%20%20%20%20Returns%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20DataFrame%20with%20an%20added%20column%20named%20after%20%60fingerprint_type%60.%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20if%20fingerprint_type%20not%20in%20_fp_dict%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20raise%20ValueError(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22Fingerprint%20type%20not%20recognized%3A%20%7Bfingerprint_type!r%7D.%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22Valid%20values%3A%20%7Blist(_fp_dict.keys())%7D%22%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20smiles_list%20%3D%20df.get_column(%22smiles%22).to_list()%0A%20%20%20%20%20%20%20%20fp_func%20%3D%20_fp_dict%5Bfingerprint_type%5D(**kwargs)%0A%0A%20%20%20%20%20%20%20%20if%20fp_func.requires_conformers%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20mol_from_smiles%20%3D%20MolFromSmilesTransformer()%0A%20%20%20%20%20%20%20%20%20%20%20%20conf_gen%20%3D%20ConformerGenerator()%0A%20%20%20%20%20%20%20%20%20%20%20%20mols_list%20%3D%20conf_gen.transform(mol_from_smiles.transform(smiles_list))%0A%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20mols_list%20%3D%20smiles_list%0A%0A%20%20%20%20%20%20%20%20fps%20%3D%20fp_func.transform(mols_list)%0A%20%20%20%20%20%20%20%20return%20df.with_columns(pl.Series(values%3Dfps%2C%20name%3Dfingerprint_type))%0A%0A%20%20%20%20return%20(generate_fingerprint%2C)%0A%0A%0A%40app.cell%0Adef%20_(RandomForestClassifier%2C%20RandomForestRegressor%2C%20np)%3A%0A%20%20%20%20class%20RandomForestModel%3A%0A%20%20%20%20%20%20%20%20%22%22%22Scikit-learn%20Random%20Forest%20with%20a%20unified%20fit%2Fpredict%20interface.%22%22%22%0A%0A%20%20%20%20%20%20%20%20def%20__init__(%0A%20%20%20%20%20%20%20%20%20%20%20%20self%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pred_type%3A%20str%20%3D%20%22regression%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20n_estimators%3A%20int%20%3D%20500%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20max_depth%3A%20int%20%7C%20None%20%3D%20None%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20min_samples_split%3A%20int%20%3D%202%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20min_samples_leaf%3A%20int%20%3D%201%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20max_features%3A%20str%20%7C%20float%20%3D%20%22sqrt%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20random_state%3A%20int%20%7C%20None%20%3D%2042%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20n_jobs%3A%20int%20%3D%20-1%2C%0A%20%20%20%20%20%20%20%20)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20self.pred_type%20%3D%20pred_type%0A%20%20%20%20%20%20%20%20%20%20%20%20common%20%3D%20dict(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20n_estimators%3Dn_estimators%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20max_depth%3Dmax_depth%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20min_samples_split%3Dmin_samples_split%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20min_samples_leaf%3Dmin_samples_leaf%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20max_features%3Dmax_features%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20random_state%3Drandom_state%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20n_jobs%3Dn_jobs%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20pred_type%20%3D%3D%20%22regression%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20self.model%20%3D%20RandomForestRegressor(**common)%0A%20%20%20%20%20%20%20%20%20%20%20%20elif%20pred_type%20%3D%3D%20%22classification%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20self.model%20%3D%20RandomForestClassifier(**common)%0A%20%20%20%20%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20raise%20ValueError(%22pred_type%20must%20be%20'classification'%20or%20'regression'%22)%0A%0A%20%20%20%20%20%20%20%20def%20train(%0A%20%20%20%20%20%20%20%20%20%20%20%20self%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20X_train%3A%20np.ndarray%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y_train%3A%20np.ndarray%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20sample_weight%3A%20np.ndarray%20%7C%20None%20%3D%20None%2C%0A%20%20%20%20%20%20%20%20)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%22%22Fit%20the%20forest%2C%20optionally%20weighting%20individual%20training%20samples.%22%22%22%0A%20%20%20%20%20%20%20%20%20%20%20%20self.model.fit(X_train%2C%20y_train%2C%20sample_weight%3Dsample_weight)%0A%0A%20%20%20%20%20%20%20%20def%20predict(self%2C%20X_test%3A%20np.ndarray)%20-%3E%20np.ndarray%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20self.pred_type%20%3D%3D%20%22classification%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20return%20self.model.predict_proba(X_test)%5B%3A%2C%201%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20self.model.predict(X_test)%0A%0A%20%20%20%20return%20(RandomForestModel%2C)%0A%0A%0A%40app.cell%0Adef%20_(np%2C%20xgb)%3A%0A%20%20%20%20class%20BoostedTreesModel%3A%0A%20%20%20%20%20%20%20%20%22%22%22XGBoost%20gradient-boosted%20trees%20with%20a%20unified%20fit%2Fpredict%20interface.%22%22%22%0A%0A%20%20%20%20%20%20%20%20def%20__init__(%0A%20%20%20%20%20%20%20%20%20%20%20%20self%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pred_type%3A%20str%20%3D%20%22regression%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20n_estimators%3A%20int%20%3D%20500%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20max_depth%3A%20int%20%3D%206%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20learning_rate%3A%20float%20%3D%200.05%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20subsample%3A%20float%20%3D%200.8%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20colsample_bytree%3A%20float%20%3D%200.8%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20min_child_weight%3A%20float%20%3D%201.0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20reg_alpha%3A%20float%20%3D%200.0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20reg_lambda%3A%20float%20%3D%201.0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20early_stopping_rounds%3A%20int%20%3D%2030%2C%0A%20%20%20%20%20%20%20%20)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20self.pred_type%20%3D%20pred_type%0A%20%20%20%20%20%20%20%20%20%20%20%20common%20%3D%20dict(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20n_estimators%3Dn_estimators%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20max_depth%3Dmax_depth%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20learning_rate%3Dlearning_rate%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20subsample%3Dsubsample%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20colsample_bytree%3Dcolsample_bytree%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20min_child_weight%3Dmin_child_weight%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20reg_alpha%3Dreg_alpha%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20reg_lambda%3Dreg_lambda%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20early_stopping_rounds%3Dearly_stopping_rounds%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20tree_method%3D%22hist%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20n_jobs%3D-1%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20pred_type%20%3D%3D%20%22regression%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20self.model%20%3D%20xgb.XGBRegressor(**common)%0A%20%20%20%20%20%20%20%20%20%20%20%20elif%20pred_type%20%3D%3D%20%22classification%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20self.model%20%3D%20xgb.XGBClassifier(**common)%0A%20%20%20%20%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20raise%20ValueError(%22pred_type%20must%20be%20'classification'%20or%20'regression'%22)%0A%0A%20%20%20%20%20%20%20%20def%20train(%0A%20%20%20%20%20%20%20%20%20%20%20%20self%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20X_train%3A%20np.ndarray%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y_train%3A%20np.ndarray%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20X_val%3A%20np.ndarray%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y_val%3A%20np.ndarray%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20sample_weight%3A%20np.ndarray%20%7C%20None%20%3D%20None%2C%0A%20%20%20%20%20%20%20%20)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%22%22Fit%20with%20an%20evaluation%20set%20for%20early%20stopping%2C%20optionally%20weighted.%22%22%22%0A%20%20%20%20%20%20%20%20%20%20%20%20self.model.fit(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20X_train%2C%20y_train%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20sample_weight%3Dsample_weight%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20eval_set%3D%5B(X_val%2C%20y_val)%5D%2C%20verbose%3DFalse%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20def%20predict(self%2C%20X_test%3A%20np.ndarray)%20-%3E%20np.ndarray%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20self.pred_type%20%3D%3D%20%22classification%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20return%20self.model.predict_proba(X_test)%5B%3A%2C%201%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20self.model.predict(X_test)%0A%0A%20%20%20%20return%20(BoostedTreesModel%2C)%0A%0A%0A%40app.cell%0Adef%20_(Optional%2C%20Path%2C%20np%2C%20pl%2C%20shutil%2C%20subprocess%2C%20sys%2C%20tempfile%2C%20torch)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Chemprop%20(scratch%20D-MPNN)%20via%20CLI%20subprocess%2C%20with%20sample%20weights%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20Copied%20from%20notebook%202's%20baseline%20ChempropModel%20so%20the%20%60uniform%60%20reference%0A%20%20%20%20%23%20(notebook%202's%20chemprop%20CV)%20and%20the%20reweighted%20runs%20here%20share%20an%20identical%0A%20%20%20%20%23%20architecture.%20Training%20runs%20in%20a%20subprocess%20(the%20chemprop%20CLI)%20to%20keep%0A%20%20%20%20%23%20PyTorch%20out%20of%20the%20notebook%20kernel.%20The%20only%20extension%20is%20optional%20per-sample%0A%20%20%20%20%23%20%60sample_weight%60%2C%20written%20as%20a%20%60weight%60%20column%20and%20passed%20to%20%60chemprop%20train%20-w%60%0A%20%20%20%20%23%20%E2%80%94%20Chemprop's%20native%20per-datapoint%20loss%20weighting.%0A%20%20%20%20_CHEMPROP_BIN%20%3D%20Path(sys.executable).parent%20%2F%20%22chemprop%22%0A%20%20%20%20_CHEMPROP_LOG%20%3D%20Path(%22..%2Flogs%2F6_chemprop_cli.log%22)%0A%20%20%20%20_CHEMPROP_LOG.parent.mkdir(parents%3DTrue%2C%20exist_ok%3DTrue)%0A%20%20%20%20_CHEMPROP_MODEL_DIR%20%3D%20Path(tempfile.gettempdir())%20%2F%20%22chemprop_reweight_model%22%0A%0A%20%20%20%20def%20_chemprop_device()%20-%3E%20str%3A%0A%20%20%20%20%20%20%20%20%22%22%22Best%20available%20accelerator%20for%20the%20chemprop%20CLI.%22%22%22%0A%20%20%20%20%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20%22cuda%22%20if%20torch.cuda.is_available()%0A%20%20%20%20%20%20%20%20%20%20%20%20else%20%22mps%22%20if%20torch.backends.mps.is_available()%0A%20%20%20%20%20%20%20%20%20%20%20%20else%20%22cpu%22%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20def%20_write_chemprop_csv(%0A%20%20%20%20%20%20%20%20smiles%3A%20list%5Bstr%5D%2C%0A%20%20%20%20%20%20%20%20targets%3A%20%22np.ndarray%20%7C%20None%22%2C%0A%20%20%20%20%20%20%20%20path%3A%20Path%2C%0A%20%20%20%20%20%20%20%20target_col%3A%20str%2C%0A%20%20%20%20%20%20%20%20weights%3A%20%22np.ndarray%20%7C%20None%22%20%3D%20None%2C%0A%20%20%20%20)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%22%22%22Write%20a%20chemprop%20input%20CSV%20with%20smiles%2C%20optional%20target%20and%20weight%20columns.%22%22%22%0A%20%20%20%20%20%20%20%20cols%3A%20dict%20%3D%20%7B%22smiles%22%3A%20smiles%7D%0A%20%20%20%20%20%20%20%20if%20targets%20is%20not%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20cols%5Btarget_col%5D%20%3D%20targets.flatten().tolist()%0A%20%20%20%20%20%20%20%20if%20weights%20is%20not%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20cols%5B%22weight%22%5D%20%3D%20weights.flatten().tolist()%0A%20%20%20%20%20%20%20%20pl.DataFrame(cols).write_csv(path)%0A%0A%20%20%20%20def%20_run_chemprop_cli(args%3A%20list%5Bstr%5D)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%22%22%22Run%20the%20chemprop%20CLI%2C%20logging%20to%20file%3B%20raise%20with%20a%20tail%20on%20failure.%22%22%22%0A%20%20%20%20%20%20%20%20cmd%20%3D%20%5Bstr(_CHEMPROP_BIN)%5D%20%2B%20args%0A%20%20%20%20%20%20%20%20with%20open(_CHEMPROP_LOG%2C%20%22a%22)%20as%20log%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20log.write(f%22%5Cn%7B'%3D'%20*%2060%7D%5CnCMD%3A%20%7B'%20'.join(cmd)%7D%5Cn%7B'%3D'%20*%2060%7D%5Cn%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20result%20%3D%20subprocess.run(cmd%2C%20stdout%3Dlog%2C%20stderr%3Dlog%2C%20text%3DTrue)%0A%20%20%20%20%20%20%20%20if%20result.returncode%20!%3D%200%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20print(%22%5Cn%22.join(_CHEMPROP_LOG.read_text().splitlines()%5B-30%3A%5D))%0A%20%20%20%20%20%20%20%20%20%20%20%20raise%20RuntimeError(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22chemprop%20CLI%20failed%20(exit%20%7Bresult.returncode%7D).%20Log%3A%20%7B_CHEMPROP_LOG%7D%22)%0A%0A%20%20%20%20class%20ChempropModel%3A%0A%20%20%20%20%20%20%20%20%22%22%22Chemprop%20D-MPNN%20trained%20from%20scratch%20via%20the%20CLI%2C%20with%20optional%20weights.%22%22%22%0A%0A%20%20%20%20%20%20%20%20def%20__init__(%0A%20%20%20%20%20%20%20%20%20%20%20%20self%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pred_type%3A%20str%20%3D%20%22regression%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20model_dir%3A%20Path%20%3D%20_CHEMPROP_MODEL_DIR%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20epochs%3A%20int%20%3D%2050%2C%0A%20%20%20%20%20%20%20%20)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20pred_type%20not%20in%20(%22regression%22%2C%20%22classification%22)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20raise%20ValueError(%22pred_type%20must%20be%20'regression'%20or%20'classification'%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20self.pred_type%20%3D%20pred_type%0A%20%20%20%20%20%20%20%20%20%20%20%20self.model_dir%20%3D%20model_dir%0A%20%20%20%20%20%20%20%20%20%20%20%20self.epochs%20%3D%20epochs%0A%20%20%20%20%20%20%20%20%20%20%20%20self.target_col%3A%20Optional%5Bstr%5D%20%3D%20None%0A%0A%20%20%20%20%20%20%20%20def%20train(%0A%20%20%20%20%20%20%20%20%20%20%20%20self%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20X_train%3A%20list%5Bstr%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y_train%3A%20np.ndarray%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20X_val%3A%20list%5Bstr%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y_val%3A%20np.ndarray%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20target_col%3A%20str%20%3D%20%22target%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20sample_weight%3A%20np.ndarray%20%7C%20None%20%3D%20None%2C%0A%20%20%20%20%20%20%20%20)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20%20%20%20%20Train%20via%20%60chemprop%20train%60.%20When%20%60sample_weight%60%20is%20given%20it%20is%20written%0A%20%20%20%20%20%20%20%20%20%20%20%20as%20a%20%60weight%60%20column%20and%20passed%20with%20%60-w%60%3B%20the%20validation%20set%20is%20given%0A%20%20%20%20%20%20%20%20%20%20%20%20unit%20weights%20so%20early%20stopping%20tracks%20an%20unweighted%20val%20loss.%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20%20%20%20%20self.target_col%20%3D%20target_col%0A%20%20%20%20%20%20%20%20%20%20%20%20tmp%20%3D%20Path(tempfile.gettempdir())%0A%20%20%20%20%20%20%20%20%20%20%20%20train_csv%20%3D%20tmp%20%2F%20%226_chemprop_train.csv%22%0A%20%20%20%20%20%20%20%20%20%20%20%20val_csv%20%3D%20tmp%20%2F%20%226_chemprop_val.csv%22%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20_write_chemprop_csv(X_train%2C%20y_train%2C%20train_csv%2C%20target_col%2C%20sample_weight)%0A%20%20%20%20%20%20%20%20%20%20%20%20_val_w%20%3D%20np.ones(len(X_val))%20if%20sample_weight%20is%20not%20None%20else%20None%0A%20%20%20%20%20%20%20%20%20%20%20%20_write_chemprop_csv(X_val%2C%20y_val%2C%20val_csv%2C%20target_col%2C%20_val_w)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20self.model_dir.exists()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20shutil.rmtree(self.model_dir)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20task_type%20%3D%20%22regression%22%20if%20self.pred_type%20%3D%3D%20%22regression%22%20else%20%22binary%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20Pass%20val_csv%20twice%20(val%20%2B%20dummy%20test)%20so%20the%20CLI%20tracks%20val_loss.%0A%20%20%20%20%20%20%20%20%20%20%20%20args%20%3D%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22train%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--data-path%22%2C%20str(train_csv)%2C%20str(val_csv)%2C%20str(val_csv)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--smiles-columns%22%2C%20%22smiles%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--target-columns%22%2C%20target_col%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--task-type%22%2C%20task_type%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--accelerator%22%2C%20_chemprop_device()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--epochs%22%2C%20str(self.epochs)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--save-dir%22%2C%20str(self.model_dir)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20sample_weight%20is%20not%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20args%20%2B%3D%20%5B%22-w%22%2C%20%22weight%22%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20_run_chemprop_cli(args)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20train_csv.unlink(missing_ok%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20val_csv.unlink(missing_ok%3DTrue)%0A%0A%20%20%20%20%20%20%20%20def%20predict(self%2C%20X_test%3A%20list%5Bstr%5D)%20-%3E%20np.ndarray%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%22%22Run%20inference%20via%20%60chemprop%20predict%60.%22%22%22%0A%20%20%20%20%20%20%20%20%20%20%20%20tmp%20%3D%20Path(tempfile.gettempdir())%0A%20%20%20%20%20%20%20%20%20%20%20%20test_csv%20%3D%20tmp%20%2F%20%226_chemprop_test.csv%22%0A%20%20%20%20%20%20%20%20%20%20%20%20pred_csv%20%3D%20tmp%20%2F%20%226_chemprop_preds.csv%22%0A%20%20%20%20%20%20%20%20%20%20%20%20model_pt%20%3D%20self.model_dir%20%2F%20%22model_0%22%20%2F%20%22best.pt%22%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20_write_chemprop_csv(X_test%2C%20None%2C%20test_csv%2C%20self.target_col)%0A%20%20%20%20%20%20%20%20%20%20%20%20_run_chemprop_cli(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22predict%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--test-path%22%2C%20str(test_csv)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--model-path%22%2C%20str(model_pt)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--preds-path%22%2C%20str(pred_csv)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20preds%20%3D%20pl.read_csv(pred_csv)%5Bself.target_col%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20test_csv.unlink(missing_ok%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20pred_csv.unlink(missing_ok%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20preds.flatten()%0A%0A%20%20%20%20return%20(ChempropModel%2C)%0A%0A%0A%40app.cell%0Adef%20_(np%2C%20smurff%2C%20sp%2C%20tempfile)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Macau%20(Bayesian%20matrix%20factorization%2C%20smurff)%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20Copied%20from%20notebook%204.%20Uses%20the%20fingerprint%20matrix%20as%20row%20side%20information.%0A%20%20%20%20%23%20Default%20params%20match%20the%20%60macau_chemeleon%60%20baseline%20reused%20as%20the%201%C3%97%20row.%0A%20%20%20%20class%20MacauModel%3A%0A%20%20%20%20%20%20%20%20%22%22%22Bayesian%20matrix%20factorization%20(Macau)%20with%20fingerprint%20side%20information.%22%22%22%0A%0A%20%20%20%20%20%20%20%20def%20__init__(%0A%20%20%20%20%20%20%20%20%20%20%20%20self%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20num_latent%3A%20int%20%3D%2016%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20burnin%3A%20int%20%3D%20100%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20nsamples%3A%20int%20%3D%20200%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20univariate%3A%20bool%20%3D%20False%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20direct%3A%20bool%20%3D%20True%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20num_threads%3A%20int%20%7C%20None%20%3D%20None%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20seed%3A%20int%20%3D%2042%2C%0A%20%20%20%20%20%20%20%20)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20self.num_latent%20%3D%20num_latent%0A%20%20%20%20%20%20%20%20%20%20%20%20self.burnin%20%3D%20burnin%0A%20%20%20%20%20%20%20%20%20%20%20%20self.nsamples%20%3D%20nsamples%0A%20%20%20%20%20%20%20%20%20%20%20%20self.univariate%20%3D%20univariate%0A%20%20%20%20%20%20%20%20%20%20%20%20self.direct%20%3D%20direct%0A%20%20%20%20%20%20%20%20%20%20%20%20self.num_threads%20%3D%20num_threads%0A%20%20%20%20%20%20%20%20%20%20%20%20self.seed%20%3D%20seed%0A%20%20%20%20%20%20%20%20%20%20%20%20self._predict_session%20%3D%20None%0A%0A%20%20%20%20%20%20%20%20def%20train(self%2C%20X_train%3A%20np.ndarray%2C%20y_train%3A%20np.ndarray)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%22%22Fit%20via%20Gibbs%20sampling%3B%20y_train%20is%20a%201-D%20target%20turned%20into%20an%20(n%C3%971)%20matrix.%22%22%22%0A%20%20%20%20%20%20%20%20%20%20%20%20n%20%3D%20len(y_train)%0A%20%20%20%20%20%20%20%20%20%20%20%20_vals%20%3D%20y_train.flatten().astype(np.float64)%0A%20%20%20%20%20%20%20%20%20%20%20%20_mask%20%3D%20~np.isnan(_vals)%0A%20%20%20%20%20%20%20%20%20%20%20%20Y_train%20%3D%20sp.coo_matrix(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(_vals%5B_mask%5D%2C%20(np.where(_mask)%5B0%5D%2C%20np.zeros(_mask.sum()%2C%20dtype%3Dint)))%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20shape%3D(n%2C%201)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20with%20tempfile.TemporaryDirectory()%20as%20tmpdir%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20import%20os%2C%20logging%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20save_name%20%3D%20os.path.join(tmpdir%2C%20%22smurff_model.hdf5%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_smurff_logger%20%3D%20logging.getLogger(%22smurff%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_root_logger%20%3D%20logging.getLogger()%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_prev_s%2C%20_prev_r%20%3D%20_smurff_logger.level%2C%20_root_logger.level%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_smurff_logger.setLevel(logging.ERROR)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_root_logger.setLevel(logging.ERROR)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20try%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20session%20%3D%20smurff.MacauSession(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20Ytrain%3DY_train%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20side_info%3D%5BX_train.astype(np.float64)%2C%20None%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20num_latent%3Dself.num_latent%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20burnin%3Dself.burnin%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20nsamples%3Dself.nsamples%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20univariate%3Dself.univariate%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20direct%3Dself.direct%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20num_threads%3Dself.num_threads%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20seed%3Dself.seed%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20verbose%3D0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20save_name%3Dsave_name%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20save_freq%3D1%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20session.init()%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20while%20session.step()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pass%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20finally%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_smurff_logger.setLevel(_prev_s)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_root_logger.setLevel(_prev_r)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20self._predict_session%20%3D%20session.makePredictSession()%0A%0A%20%20%20%20%20%20%20%20def%20predict(self%2C%20X_test%3A%20np.ndarray)%20-%3E%20np.ndarray%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%22%22Average%20posterior%20Gibbs%20samples%20for%20the%20test%20compounds.%22%22%22%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20self._predict_session%20is%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20raise%20RuntimeError(%22Call%20train()%20before%20predict().%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20sample_arrays%20%3D%20self._predict_session.predict(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(X_test.astype(np.float64)%2C%20slice(None)))%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20np.mean(np.array(sample_arrays)%2C%20axis%3D0).flatten()%0A%0A%20%20%20%20return%20(MacauModel%2C)%0A%0A%0A%40app.cell%0Adef%20_(Path%2C%20np%2C%20subprocess%2C%20sys%2C%20tempfile)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20TabPFN%20predictor%20via%20CPU%20subprocess%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20TabPFN%20(torch)%20runs%20in%20an%20isolated%20subprocess%20on%20CPU%20%E2%80%94%20matching%20notebook%204's%0A%20%20%20%20%23%20baseline%20(n_estimators%3D8%2C%20ignore_pretraining_limits%3DTrue)%20so%20the%20reused%201%C3%97%0A%20%20%20%20%23%20%60tabpfn_chemeleon%60%20row%20is%20comparable.%20CPU%20avoids%20the%20MPS%20allocator%20ceiling.%0A%20%20%20%20_TABPFN_SCRIPT%20%3D%20Path(tempfile.gettempdir())%20%2F%20%226_tabpfn.py%22%0A%20%20%20%20_TABPFN_SCRIPT.write_text(%22%5Cn%22.join(%5B%0A%20%20%20%20%20%20%20%20%22import%20os%2C%20sys%2C%20numpy%20as%20np%22%2C%0A%20%20%20%20%20%20%20%20%22os.environ%5B'KMP_DUPLICATE_LIB_OK'%5D%20%3D%20'TRUE'%22%2C%0A%20%20%20%20%20%20%20%20%22from%20dotenv%20import%20load_dotenv%3B%20from%20pathlib%20import%20Path%22%2C%0A%20%20%20%20%20%20%20%20%22load_dotenv(Path('.env'))%22%2C%0A%20%20%20%20%20%20%20%20%22import%20torch%22%2C%0A%20%20%20%20%20%20%20%20%22torch.set_num_threads(max(1%2C%20(os.cpu_count()%20or%201)%20-%201))%22%2C%0A%20%20%20%20%20%20%20%20%22from%20tabpfn%20import%20TabPFNRegressor%22%2C%0A%20%20%20%20%20%20%20%20%22X_train%20%3D%20np.load(sys.argv%5B1%5D)%3B%20y_train%20%3D%20np.load(sys.argv%5B2%5D)%22%2C%0A%20%20%20%20%20%20%20%20%22X_test%20%3D%20np.load(sys.argv%5B3%5D)%3B%20out%20%3D%20sys.argv%5B4%5D%22%2C%0A%20%20%20%20%20%20%20%20%22model%20%3D%20TabPFNRegressor(n_estimators%3D8%2C%20ignore_pretraining_limits%3DTrue%2C%20device%3D'cpu')%22%2C%0A%20%20%20%20%20%20%20%20%22model.fit(X_train%2C%20y_train)%22%2C%0A%20%20%20%20%20%20%20%20%22np.save(out%2C%20model.predict(X_test))%22%2C%0A%20%20%20%20%5D))%0A%0A%20%20%20%20def%20tabpfn_predict(%0A%20%20%20%20%20%20%20%20X_train%3A%20np.ndarray%2C%20y_train%3A%20np.ndarray%2C%20X_test%3A%20np.ndarray%2C%0A%20%20%20%20)%20-%3E%20np.ndarray%3A%0A%20%20%20%20%20%20%20%20%22%22%22Fit%20TabPFN%20on%20(X_train%2C%20y_train)%20and%20predict%20X_test%2C%20via%20CPU%20subprocess.%22%22%22%0A%20%20%20%20%20%20%20%20tmp%20%3D%20Path(tempfile.gettempdir())%0A%20%20%20%20%20%20%20%20f_xtr%2C%20f_ytr%2C%20f_xte%20%3D%20tmp%20%2F%20%226_tf_Xtr.npy%22%2C%20tmp%20%2F%20%226_tf_ytr.npy%22%2C%20tmp%20%2F%20%226_tf_Xte.npy%22%0A%20%20%20%20%20%20%20%20f_out%20%3D%20tmp%20%2F%20%226_tf_preds%22%0A%20%20%20%20%20%20%20%20np.save(str(f_xtr)%2C%20X_train)%0A%20%20%20%20%20%20%20%20np.save(str(f_ytr)%2C%20y_train)%0A%20%20%20%20%20%20%20%20np.save(str(f_xte)%2C%20X_test)%0A%20%20%20%20%20%20%20%20res%20%3D%20subprocess.run(%0A%20%20%20%20%20%20%20%20%20%20%20%20%5Bsys.executable%2C%20str(_TABPFN_SCRIPT)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20str(f_xtr)%2C%20str(f_ytr)%2C%20str(f_xte)%2C%20str(f_out)%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20capture_output%3DTrue%2C%20text%3DTrue%2C%20cwd%3Dstr(Path(%22..%2F%22).resolve())%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20if%20res.returncode%20!%3D%200%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20raise%20RuntimeError(f%22TabPFN%20subprocess%20failed%3A%5Cn%7Bres.stderr%7D%22)%0A%20%20%20%20%20%20%20%20preds%20%3D%20np.load(str(f_out)%20%2B%20%22.npy%22)%0A%20%20%20%20%20%20%20%20for%20p%20in%20%5Bf_xtr%2C%20f_ytr%2C%20f_xte%2C%20Path(str(f_out)%20%2B%20%22.npy%22)%5D%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20p.unlink(missing_ok%3DTrue)%0A%20%20%20%20%20%20%20%20return%20preds%0A%0A%20%20%20%20return%20(tabpfn_predict%2C)%0A%0A%0A%40app.cell%0Adef%20_(BaseKFold%2C%20Iterator%2C%20Optional%2C%20np%2C%20pl)%3A%0A%20%20%20%20def%20split_dataset_random(%0A%20%20%20%20%20%20%20%20df%3A%20pl.DataFrame%2C%20p_test%3A%20float%20%3D%200.2%2C%20seed%3A%20int%20%3D%2042%2C%0A%20%20%20%20)%20-%3E%20tuple%5Bpl.DataFrame%2C%20pl.DataFrame%5D%3A%0A%20%20%20%20%20%20%20%20%22%22%22Randomly%20split%20a%20DataFrame%20into%20(train%2C%20test)%20subsets.%22%22%22%0A%20%20%20%20%20%20%20%20rng%20%3D%20np.random.default_rng(seed)%0A%20%20%20%20%20%20%20%20idx%20%3D%20rng.permutation(df.shape%5B0%5D)%0A%20%20%20%20%20%20%20%20n_test%20%3D%20int(len(idx)%20*%20p_test)%0A%20%20%20%20%20%20%20%20test_idx%2C%20train_idx%20%3D%20idx%5B%3An_test%5D%2C%20idx%5Bn_test%3A%5D%0A%20%20%20%20%20%20%20%20return%20df%5Btrain_idx%5D.clone()%2C%20df%5Btest_idx%5D.clone()%0A%0A%20%20%20%20class%20GroupKFoldShuffle(BaseKFold)%3A%0A%20%20%20%20%20%20%20%20%22%22%22GroupKFold%20that%20shuffles%20groups%20before%20splitting%20(reproducible).%22%22%22%0A%0A%20%20%20%20%20%20%20%20def%20__init__(%0A%20%20%20%20%20%20%20%20%20%20%20%20self%2C%20n_splits%3A%20int%20%3D%205%2C%20*%2C%20shuffle%3A%20bool%20%3D%20False%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20random_state%3A%20Optional%5Bint%5D%20%3D%20None%2C%0A%20%20%20%20%20%20%20%20)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20super().__init__(n_splits%3Dn_splits%2C%20shuffle%3Dshuffle%2C%20random_state%3Drandom_state)%0A%0A%20%20%20%20%20%20%20%20def%20split(self%2C%20X%2C%20y%3DNone%2C%20groups%3DNone)%20-%3E%20Iterator%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20unique_groups%20%3D%20np.unique(groups)%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20self.shuffle%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20rng%20%3D%20np.random.default_rng(self.random_state)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20unique_groups%20%3D%20rng.permutation(unique_groups)%0A%20%20%20%20%20%20%20%20%20%20%20%20split_groups%20%3D%20np.array_split(unique_groups%2C%20self.n_splits)%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20test_group_ids%20in%20split_groups%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20test_mask%20%3D%20np.isin(groups%2C%20test_group_ids)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20yield%20np.where(~test_mask)%5B0%5D%2C%20np.where(test_mask)%5B0%5D%0A%0A%20%20%20%20return%20GroupKFoldShuffle%2C%20split_dataset_random%0A%0A%0A%40app.cell%0Adef%20_(GroupKFoldShuffle%2C%20Iterator%2C%20pl%2C%20split_dataset_random)%3A%0A%20%20%20%20def%20generate_cv_splits_random(%0A%20%20%20%20%20%20%20%20df%3A%20pl.DataFrame%2C%0A%20%20%20%20%20%20%20%20n_outer%3A%20int%20%3D%205%2C%0A%20%20%20%20%20%20%20%20n_inner%3A%20int%20%3D%205%2C%0A%20%20%20%20%20%20%20%20seed%3A%20int%20%3D%2042%2C%0A%20%20%20%20%20%20%20%20p_val%3A%20float%20%3D%200%2C%0A%20%20%20%20)%20-%3E%20Iterator%3A%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20Generate%20nested%205%C3%975%20CV%20splits%20with%20a%20random%20per-molecule%20assignment.%0A%0A%20%20%20%20%20%20%20%20Yields%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20(fold_index%2C%20outer_index%2C%20inner_index%2C%20train_df%2C%20val_df%2C%20test_df).%0A%20%20%20%20%20%20%20%20%20%20%20%20%60val_df%60%20is%20None%20when%20p_val%20%3D%3D%200.%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20for%20i%20in%20range(n_outer)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20kf%20%3D%20GroupKFoldShuffle(n_splits%3Dn_inner%2C%20random_state%3Dseed%20%2B%20i%2C%20shuffle%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20groups%20%3D%20list(range(df.shape%5B0%5D))%20%20%23%20each%20molecule%20is%20its%20own%20group%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20j%2C%20(train_idx%2C%20test_idx)%20in%20enumerate(kf.split(df%2C%20groups%3Dgroups))%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20fold%20%3D%20i%20*%20n_inner%20%2B%20j%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20train%20%3D%20df%5Btrain_idx%5D.clone()%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20test%20%3D%20df%5Btest_idx%5D.clone()%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20val%20%3D%20None%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20p_val%20%3E%200%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20train%2C%20val%20%3D%20split_dataset_random(train%2C%20p_test%3Dp_val%2C%20seed%3Dseed%20%2B%20fold)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20yield%20fold%2C%20i%2C%20j%2C%20train%2C%20val%2C%20test%0A%0A%20%20%20%20return%20(generate_cv_splits_random%2C)%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20mean_absolute_error%2C%0A%20%20%20%20mean_squared_error%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20precision_score%2C%0A%20%20%20%20r2_score%2C%0A%20%20%20%20recall_score%2C%0A%20%20%20%20spearmanr%2C%0A%20%20%20%20warnings%2C%0A)%3A%0A%20%20%20%20def%20calc_regression_metrics(%0A%20%20%20%20%20%20%20%20df%3A%20pl.DataFrame%2C%0A%20%20%20%20%20%20%20%20cycle_col%3A%20str%2C%0A%20%20%20%20%20%20%20%20val_col%3A%20str%2C%0A%20%20%20%20%20%20%20%20pred_col%3A%20str%2C%0A%20%20%20%20%20%20%20%20thresh%3A%20float%2C%0A%20%20%20%20)%20-%3E%20pl.DataFrame%3A%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20Per%20(cv_cycle%2C%20method%2C%20split)%20regression%20metrics%3A%20MAE%2C%20MSE%2C%20R%C2%B2%2C%20%CF%81%2C%20prec%2C%20recall.%0A%0A%20%20%20%20%20%20%20%20Precision%2Frecall%20use%20a%20binary%20hit%20label%20derived%20from%20%60thresh%60.%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20df_in%20%3D%20df.filter(pl.col(pred_col).is_not_nan()%20%26%20pl.col(pred_col).is_not_null())%0A%20%20%20%20%20%20%20%20df_in%20%3D%20df_in.with_columns(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(val_col)%20%3E%20thresh).alias(%22true_class%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(pred_col)%20%3E%20thresh).alias(%22pred_class%22)%2C%0A%20%20%20%20%20%20%20%20%5D)%0A%20%20%20%20%20%20%20%20assert%20df_in%5B%22true_class%22%5D.n_unique()%20%3D%3D%202%2C%20%22Binary%20classification%20requires%20two%20classes%22%0A%0A%20%20%20%20%20%20%20%20metric_list%3A%20list%5Bdict%5D%20%3D%20%5B%5D%0A%20%20%20%20%20%20%20%20for%20group_keys%2C%20group_df%20in%20df_in.group_by(%5Bcycle_col%2C%20%22method%22%2C%20%22split%22%5D)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20cycle%2C%20method%2C%20split%20%3D%20group_keys%0A%20%20%20%20%20%20%20%20%20%20%20%20y_true%20%3D%20group_df%5Bval_col%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20y_pred%20%3D%20group_df%5Bpred_col%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20with%20warnings.catch_warnings()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20warnings.simplefilter(%22ignore%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20rho%2C%20_%20%3D%20spearmanr(y_true%2C%20y_pred)%0A%20%20%20%20%20%20%20%20%20%20%20%20metric_list.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22cv_cycle%22%3A%20cycle%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22method%22%3A%20method%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22split%22%3A%20split%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22mae%22%3A%20mean_absolute_error(y_true%2C%20y_pred)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22mse%22%3A%20mean_squared_error(y_true%2C%20y_pred)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22r2%22%3A%20r2_score(y_true%2C%20y_pred)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22rho%22%3A%20float(rho)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22prec%22%3A%20precision_score(group_df%5B%22true_class%22%5D.to_numpy()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20group_df%5B%22pred_class%22%5D.to_numpy())%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22recall%22%3A%20recall_score(group_df%5B%22true_class%22%5D.to_numpy()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20group_df%5B%22pred_class%22%5D.to_numpy())%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20return%20pl.DataFrame(metric_list)%0A%0A%20%20%20%20return%20(calc_regression_metrics%2C)%0A%0A%0A%40app.cell%0Adef%20_(Optional%2C%20np%2C%20pd%2C%20pg%2C%20pl%2C%20psturng%2C%20qsturng%2C%20warnings)%3A%0A%20%20%20%20def%20rm_tukey_hsd(%0A%20%20%20%20%20%20%20%20df%3A%20pl.DataFrame%2C%0A%20%20%20%20%20%20%20%20metric%3A%20str%2C%0A%20%20%20%20%20%20%20%20group_col%3A%20str%2C%0A%20%20%20%20%20%20%20%20alpha%3A%20float%20%3D%200.05%2C%0A%20%20%20%20%20%20%20%20sort%3A%20bool%20%3D%20False%2C%0A%20%20%20%20%20%20%20%20direction_dict%3A%20Optional%5Bdict%5D%20%3D%20None%2C%0A%20%20%20%20)%20-%3E%20tuple%3A%0A%20%20%20%20%20%20%20%20%22%22%22Repeated-measures%20Tukey%20HSD%20across%20CV%20folds%20(subject%20%3D%20cv_cycle).%22%22%22%0A%20%20%20%20%20%20%20%20df_pd%20%3D%20df.to_pandas()%0A%0A%20%20%20%20%20%20%20%20if%20sort%20and%20direction_dict%20and%20metric%20in%20direction_dict%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20ascending%20%3D%20direction_dict%5Bmetric%5D%20%3D%3D%20%22minimize%22%0A%20%20%20%20%20%20%20%20%20%20%20%20df_means%20%3D%20df_pd.groupby(group_col).mean(numeric_only%3DTrue).sort_values(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20metric%2C%20ascending%3Dascending)%0A%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20df_means%20%3D%20df_pd.groupby(group_col).mean(numeric_only%3DTrue)%0A%0A%20%20%20%20%20%20%20%20with%20warnings.catch_warnings()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20warnings.filterwarnings(%22ignore%22%2C%20category%3DRuntimeWarning)%0A%20%20%20%20%20%20%20%20%20%20%20%20aov%20%3D%20pg.rm_anova(dv%3Dmetric%2C%20within%3Dgroup_col%2C%20subject%3D%22cv_cycle%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20data%3Ddf_pd%2C%20detailed%3DTrue)%0A%20%20%20%20%20%20%20%20mse%20%3D%20aov.loc%5B1%2C%20%22MS%22%5D%0A%20%20%20%20%20%20%20%20df_resid%20%3D%20aov.loc%5B1%2C%20%22DF%22%5D%0A%0A%20%20%20%20%20%20%20%20methods%20%3D%20df_means.index%0A%20%20%20%20%20%20%20%20n_groups%20%3D%20len(methods)%0A%20%20%20%20%20%20%20%20n_per_group%20%3D%20df_pd%5Bgroup_col%5D.value_counts().mean()%0A%20%20%20%20%20%20%20%20tukey_se%20%3D%20np.sqrt(2%20*%20mse%20%2F%20n_per_group)%0A%20%20%20%20%20%20%20%20q%20%3D%20qsturng(1%20-%20alpha%2C%20n_groups%2C%20df_resid)%0A%0A%20%20%20%20%20%20%20%20num_comparisons%20%3D%20n_groups%20*%20(n_groups%20-%201)%20%2F%2F%202%0A%20%20%20%20%20%20%20%20result_tab%20%3D%20pd.DataFrame(index%3Drange(num_comparisons)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20columns%3D%5B%22group1%22%2C%20%22group2%22%2C%20%22meandiff%22%2C%20%22lower%22%2C%20%22upper%22%2C%20%22p-adj%22%5D)%0A%20%20%20%20%20%20%20%20df_means_diff%20%3D%20pd.DataFrame(index%3Dmethods%2C%20columns%3Dmethods%2C%20data%3D0.0)%0A%20%20%20%20%20%20%20%20pc%20%3D%20pd.DataFrame(index%3Dmethods%2C%20columns%3Dmethods%2C%20data%3D1.0)%0A%0A%20%20%20%20%20%20%20%20row_idx%20%3D%200%0A%20%20%20%20%20%20%20%20for%20i%2C%20m1%20in%20enumerate(methods)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20j%2C%20m2%20in%20enumerate(methods)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20i%20%3C%20j%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20g1%20%3D%20df_pd%5Bdf_pd%5Bgroup_col%5D%20%3D%3D%20m1%5D%5Bmetric%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20g2%20%3D%20df_pd%5Bdf_pd%5Bgroup_col%5D%20%3D%3D%20m2%5D%5Bmetric%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20mean_diff%20%3D%20g1.mean()%20-%20g2.mean()%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20studentized%20%3D%20np.abs(mean_diff)%20%2F%20tukey_se%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20adjusted_p%20%3D%20psturng(studentized%20*%20np.sqrt(2)%2C%20n_groups%2C%20df_resid)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20isinstance(adjusted_p%2C%20np.ndarray)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20adjusted_p%20%3D%20adjusted_p%5B0%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20lower%20%3D%20mean_diff%20-%20(q%20%2F%20np.sqrt(2)%20*%20tukey_se)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20upper%20%3D%20mean_diff%20%2B%20(q%20%2F%20np.sqrt(2)%20*%20tukey_se)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20result_tab.loc%5Brow_idx%5D%20%3D%20%5Bm1%2C%20m2%2C%20mean_diff%2C%20lower%2C%20upper%2C%20adjusted_p%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pc.loc%5Bm1%2C%20m2%5D%20%3D%20pc.loc%5Bm2%2C%20m1%5D%20%3D%20adjusted_p%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20df_means_diff.loc%5Bm1%2C%20m2%5D%20%3D%20mean_diff%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20df_means_diff.loc%5Bm2%2C%20m1%5D%20%3D%20-mean_diff%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20row_idx%20%2B%3D%201%0A%0A%20%20%20%20%20%20%20%20df_means_diff%20%3D%20df_means_diff.astype(float)%0A%20%20%20%20%20%20%20%20result_tab%5B%22group1_mean%22%5D%20%3D%20result_tab%5B%22group1%22%5D.map(df_means%5Bmetric%5D)%0A%20%20%20%20%20%20%20%20result_tab%5B%22group2_mean%22%5D%20%3D%20result_tab%5B%22group2%22%5D.map(df_means%5Bmetric%5D)%0A%20%20%20%20%20%20%20%20result_tab.index%20%3D%20result_tab%5B%22group1%22%5D%20%2B%20%22%20-%20%22%20%2B%20result_tab%5B%22group2%22%5D%0A%20%20%20%20%20%20%20%20return%20result_tab%2C%20df_means%2C%20df_means_diff%2C%20pc%0A%0A%20%20%20%20return%20(rm_tukey_hsd%2C)%0A%0A%0A%40app.cell%0Adef%20_(Optional%2C%20Path%2C%20math%2C%20np%2C%20plt%2C%20rm_tukey_hsd%2C%20sns)%3A%0A%20%20%20%20def%20mcs_plot(pc%2C%20effect_size%2C%20means%2C%20labels%3DTrue%2C%20cmap%3DNone%2C%20ax%3DNone%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20show_diff%3DTrue%2C%20cell_text_size%3D16%2C%20axis_text_size%3D12%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20show_cbar%3DTrue%2C%20reverse_cmap%3DFalse%2C%20vlim%3DNone%2C%20**kwargs)%3A%0A%20%20%20%20%20%20%20%20%22%22%22Multiple-comparison-of-means%20heatmap%20(Tukey%20HSD).%22%22%22%0A%20%20%20%20%20%20%20%20for%20key%20in%20%5B%22cbar%22%2C%20%22vmin%22%2C%20%22vmax%22%2C%20%22center%22%5D%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20kwargs.pop(key%2C%20None)%0A%20%20%20%20%20%20%20%20cmap%20%3D%20cmap%20or%20%22coolwarm%22%0A%20%20%20%20%20%20%20%20if%20reverse_cmap%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20cmap%20%3D%20cmap%20%2B%20%22_r%22%0A%0A%20%20%20%20%20%20%20%20significance%20%3D%20pc.copy().astype(object)%0A%20%20%20%20%20%20%20%20significance%5B(pc%20%3C%200.001)%20%26%20(pc%20%3E%3D%200)%5D%20%3D%20%22***%22%0A%20%20%20%20%20%20%20%20significance%5B(pc%20%3C%200.01)%20%26%20(pc%20%3E%3D%200.001)%5D%20%3D%20%22**%22%0A%20%20%20%20%20%20%20%20significance%5B(pc%20%3C%200.05)%20%26%20(pc%20%3E%3D%200.01)%5D%20%3D%20%22*%22%0A%20%20%20%20%20%20%20%20significance%5B(pc%20%3E%3D%200.05)%5D%20%3D%20%22%22%0A%20%20%20%20%20%20%20%20np.fill_diagonal(significance.values%2C%20%22%22)%0A%20%20%20%20%20%20%20%20annotations%20%3D%20effect_size.round(2).astype(str)%20%2B%20significance%20if%20show_diff%20else%20significance%0A%0A%20%20%20%20%20%20%20%20hax%20%3D%20sns.heatmap(%0A%20%20%20%20%20%20%20%20%20%20%20%20effect_size%2C%20cmap%3Dcmap%2C%20annot%3Dannotations%2C%20fmt%3D%22%22%2C%20cbar%3Dshow_cbar%2C%20ax%3Dax%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20annot_kws%3D%7B%22size%22%3A%20cell_text_size%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20vmin%3D-2%20*%20vlim%20if%20vlim%20else%20None%2C%20vmax%3D2%20*%20vlim%20if%20vlim%20else%20None%2C%20**kwargs%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20if%20labels%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20label_list%20%3D%20list(means.index)%0A%20%20%20%20%20%20%20%20%20%20%20%20hax.set_xticklabels(%5Bf%22%7Bx%7D%5Cn%7Bmeans.loc%5Bx%5D%3A.3f%7D%22%20for%20x%20in%20label_list%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20size%3Daxis_text_size%2C%20ha%3D%22center%22%2C%20va%3D%22top%22%2C%20rotation%3D0)%0A%20%20%20%20%20%20%20%20%20%20%20%20hax.set_yticklabels(%5Bf%22%7Bx%7D%5Cn%7Bmeans.loc%5Bx%5D%3A.3f%7D%5Cn%22%20for%20x%20in%20label_list%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20size%3Daxis_text_size%2C%20ha%3D%22center%22%2C%20va%3D%22center%22%2C%20rotation%3D90)%0A%20%20%20%20%20%20%20%20hax.set_xlabel(%22%22)%0A%20%20%20%20%20%20%20%20hax.set_ylabel(%22%22)%0A%20%20%20%20%20%20%20%20return%20hax%0A%0A%20%20%20%20def%20make_mcs_plot_grid(%0A%20%20%20%20%20%20%20%20df%2C%20stats%3A%20list%5Bstr%5D%2C%20group_col%3A%20str%2C%20alpha%3A%20float%20%3D%200.05%2C%0A%20%20%20%20%20%20%20%20figsize%3A%20tuple%20%3D%20(10%2C%208)%2C%20direction_dict%3A%20dict%20%7C%20None%20%3D%20None%2C%0A%20%20%20%20%20%20%20%20effect_dict%3A%20dict%20%7C%20None%20%3D%20None%2C%20show_diff%3A%20bool%20%3D%20True%2C%0A%20%20%20%20%20%20%20%20cell_text_size%3A%20int%20%3D%2014%2C%20axis_text_size%3A%20int%20%3D%2011%2C%20title_text_size%3A%20int%20%3D%2015%2C%0A%20%20%20%20%20%20%20%20sort_axes%3A%20bool%20%3D%20True%2C%20save_path%3A%20Optional%5BPath%5D%20%3D%20None%2C%0A%20%20%20%20)%20-%3E%20%22plt.Figure%22%3A%0A%20%20%20%20%20%20%20%20%22%22%22Grid%20of%20Tukey-HSD%20MCS%20heatmaps%2C%20one%20panel%20per%20metric.%22%22%22%0A%20%20%20%20%20%20%20%20direction_dict%20%3D%20dict(direction_dict%20or%20%7B%7D)%0A%20%20%20%20%20%20%20%20effect_dict%20%3D%20dict(effect_dict%20or%20%7B%7D)%0A%20%20%20%20%20%20%20%20for%20key%20in%20%5B%22r2%22%2C%20%22rho%22%2C%20%22prec%22%2C%20%22recall%22%2C%20%22mae%22%2C%20%22mse%22%5D%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20direction_dict.setdefault(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20key%2C%20%22maximize%22%20if%20key%20in%20%5B%22r2%22%2C%20%22rho%22%2C%20%22prec%22%2C%20%22recall%22%5D%20else%20%22minimize%22)%0A%20%20%20%20%20%20%20%20for%20key%20in%20%5B%22r2%22%2C%20%22rho%22%2C%20%22prec%22%2C%20%22recall%22%5D%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20effect_dict.setdefault(key%2C%200.1)%0A%20%20%20%20%20%20%20%20effect_dict.setdefault(%22mae%22%2C%200.05)%0A%20%20%20%20%20%20%20%20effect_dict.setdefault(%22mse%22%2C%200.1)%0A%0A%20%20%20%20%20%20%20%20ncol%20%3D%201%20if%20len(stats)%20%3D%3D%201%20else%20(2%20if%20len(stats)%20%3D%3D%204%20else%203)%0A%20%20%20%20%20%20%20%20nrow%20%3D%20math.ceil(len(stats)%20%2F%20ncol)%0A%20%20%20%20%20%20%20%20fig%2C%20ax%20%3D%20plt.subplots(nrow%2C%20ncol%2C%20figsize%3Dfigsize%2C%20squeeze%3DFalse)%0A%0A%20%20%20%20%20%20%20%20for%20i%2C%20stat%20in%20enumerate(stats)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20stat%20%3D%20stat.lower()%0A%20%20%20%20%20%20%20%20%20%20%20%20_%2C%20df_means%2C%20df_means_diff%2C%20pc%20%3D%20rm_tukey_hsd(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20df%2C%20stat%2C%20group_col%2C%20alpha%2C%20sort_axes%2C%20direction_dict)%0A%20%20%20%20%20%20%20%20%20%20%20%20mcs_plot(pc%2C%20effect_size%3Ddf_means_diff%2C%20means%3Ddf_means%5Bstat%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20show_diff%3Dshow_diff%2C%20ax%3Dax%5Bi%20%2F%2F%20ncol%2C%20i%20%25%20ncol%5D%2C%20cbar%3DTrue%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20cell_text_size%3Dcell_text_size%2C%20axis_text_size%3Daxis_text_size%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20reverse_cmap%3D(direction_dict.get(stat)%20%3D%3D%20%22minimize%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20vlim%3Deffect_dict.get(stat))%0A%20%20%20%20%20%20%20%20%20%20%20%20ax%5Bi%20%2F%2F%20ncol%2C%20i%20%25%20ncol%5D.set_title(stat.upper()%2C%20fontsize%3Dtitle_text_size)%0A%0A%20%20%20%20%20%20%20%20for%20i%20in%20range(len(stats)%2C%20nrow%20*%20ncol)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20ax%5Bi%20%2F%2F%20ncol%2C%20i%20%25%20ncol%5D.set_visible(False)%0A%20%20%20%20%20%20%20%20fig.tight_layout()%0A%20%20%20%20%20%20%20%20if%20save_path%20is%20not%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20fig.savefig(save_path%2C%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%20%20%20%20%20%20%20%20return%20fig%0A%0A%20%20%20%20return%20make_mcs_plot_grid%2C%20mcs_plot%0A%0A%0A%40app.cell%0Adef%20_(np%2C%20pl%2C%20plt)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Activity-bin%20helpers%20(shared%20by%20all%20analyses)%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20The%20same%20four%20pEC50%20bins%20used%20in%20the%20unblinded%20analysis%20(notebook%205).%0A%20%20%20%20BIN_ORDER%20%3D%20%5B%22%3C4%20(inactive)%22%2C%20%224%E2%80%935%20(weak)%22%2C%20%225%E2%80%936%20(moderate)%22%2C%20%22%3E6%20(hit%20zone)%22%5D%0A%0A%20%20%20%20def%20activity_bin_expr(col%3A%20str%20%3D%20%22y_true%22)%20-%3E%20pl.Expr%3A%0A%20%20%20%20%20%20%20%20%22%22%22Polars%20expression%20assigning%20each%20row%20to%20one%20of%20the%20four%20pEC50%20bins.%22%22%22%0A%20%20%20%20%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.when(pl.col(col)%20%3E%3D%206.0).then(pl.lit(%22%3E6%20(hit%20zone)%22))%0A%20%20%20%20%20%20%20%20%20%20%20%20.when(pl.col(col)%20%3E%3D%205.0).then(pl.lit(%225%E2%80%936%20(moderate)%22))%0A%20%20%20%20%20%20%20%20%20%20%20%20.when(pl.col(col)%20%3E%3D%204.0).then(pl.lit(%224%E2%80%935%20(weak)%22))%0A%20%20%20%20%20%20%20%20%20%20%20%20.otherwise(pl.lit(%22%3C4%20(inactive)%22))%0A%20%20%20%20%20%20%20%20%20%20%20%20.alias(%22pec50_bin%22)%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20def%20bias_by_bin_table(%0A%20%20%20%20%20%20%20%20df%3A%20pl.DataFrame%2C%20method_col%3A%20str%2C%20true_col%3A%20str%2C%20pred_col%3A%20str%2C%0A%20%20%20%20)%20-%3E%20pl.DataFrame%3A%0A%20%20%20%20%20%20%20%20%22%22%22Mean%20signed%20error%20(pred%20%E2%88%92%20true)%20per%20method%20%C3%97%20activity%20bin.%22%22%22%0A%20%20%20%20%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20df.with_columns(activity_bin_expr(true_col))%0A%20%20%20%20%20%20%20%20%20%20%20%20.group_by(%5Bmethod_col%2C%20%22pec50_bin%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20.agg(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(pred_col)%20-%20pl.col(true_col)).mean().alias(%22mean_error%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(pred_col)%20-%20pl.col(true_col)).abs().mean().alias(%22mean_abs_error%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.len().alias(%22n%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20def%20plot_bias_heatmap(%0A%20%20%20%20%20%20%20%20bias_df%3A%20pl.DataFrame%2C%20method_order%3A%20list%5Bstr%5D%2C%20title%3A%20str%2C%20save_path%2C%0A%20%20%20%20)%20-%3E%20%22plt.Figure%22%3A%0A%20%20%20%20%20%20%20%20%22%22%22Heatmap%20of%20mean%20signed%20error%3A%20rows%20%3D%20methods%2C%20columns%20%3D%20activity%20bins.%22%22%22%0A%20%20%20%20%20%20%20%20matrix%20%3D%20np.full((len(method_order)%2C%20len(BIN_ORDER))%2C%20np.nan)%0A%20%20%20%20%20%20%20%20for%20row%20in%20bias_df.iter_rows(named%3DTrue)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20row%5B%22method%22%5D%20in%20method_order%20and%20row%5B%22pec50_bin%22%5D%20in%20BIN_ORDER%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20mi%20%3D%20method_order.index(row%5B%22method%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20bi%20%3D%20BIN_ORDER.index(row%5B%22pec50_bin%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20matrix%5Bmi%2C%20bi%5D%20%3D%20row%5B%22mean_error%22%5D%0A%20%20%20%20%20%20%20%20abs_max%20%3D%20float(np.nanmax(np.abs(matrix)))%0A%0A%20%20%20%20%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20fig%2C%20ax%20%3D%20plt.subplots(figsize%3D(7%2C%200.7%20*%20len(method_order)%20%2B%201.8)%2C%20dpi%3D150)%0A%20%20%20%20%20%20%20%20%20%20%20%20ax.grid(False)%0A%20%20%20%20%20%20%20%20%20%20%20%20im%20%3D%20ax.imshow(matrix%2C%20cmap%3D%22RdBu_r%22%2C%20vmin%3D-abs_max%2C%20vmax%3Dabs_max%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20aspect%3D%22auto%22%2C%20interpolation%3D%22nearest%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20mi%20in%20range(len(method_order))%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20bi%20in%20range(len(BIN_ORDER))%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20val%20%3D%20matrix%5Bmi%2C%20bi%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20not%20np.isnan(val)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20col%20%3D%20%22white%22%20if%20abs(val)%20%3E%200.6%20*%20abs_max%20else%20%22black%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20ax.text(bi%2C%20mi%2C%20f%22%7Bval%3A%2B.3f%7D%22%2C%20ha%3D%22center%22%2C%20va%3D%22center%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D9%2C%20color%3Dcol)%0A%20%20%20%20%20%20%20%20%20%20%20%20ax.set_xticks(range(len(BIN_ORDER)))%0A%20%20%20%20%20%20%20%20%20%20%20%20ax.set_xticklabels(BIN_ORDER%2C%20fontsize%3D9%2C%20rotation%3D-20%2C%20ha%3D%22left%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20ax.set_yticks(range(len(method_order)))%0A%20%20%20%20%20%20%20%20%20%20%20%20ax.set_yticklabels(method_order%2C%20fontsize%3D10)%0A%20%20%20%20%20%20%20%20%20%20%20%20ax.set_title(title%2C%20fontsize%3D12)%0A%20%20%20%20%20%20%20%20%20%20%20%20cb%20%3D%20fig.colorbar(im%2C%20ax%3Dax%2C%20fraction%3D0.04%2C%20pad%3D0.03)%0A%20%20%20%20%20%20%20%20%20%20%20%20cb.set_label(%22Mean%20error%20(pred%20%E2%88%92%20true)%22%2C%20fontsize%3D9)%0A%20%20%20%20%20%20%20%20%20%20%20%20fig.tight_layout()%0A%20%20%20%20%20%20%20%20%20%20%20%20fig.savefig(save_path%2C%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%20%20%20%20%20%20%20%20return%20fig%0A%0A%20%20%20%20return%20BIN_ORDER%2C%20bias_by_bin_table%2C%20plot_bias_heatmap%0A%0A%0A%40app.cell%0Adef%20_(Path%2C%20gc%2C%20pl)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Dose-response%20training%20set%20%E2%80%94%20the%20single%20shared%20target%20for%20every%20analysis%20%E2%94%80%E2%94%80%0A%20%20%20%20DR_TRAIN%20%3D%20(%0A%20%20%20%20%20%20%20%20pl.read_csv(%22..%2Fdata%2Fprocessed%2Fall_compounds_activity_data.csv%22)%0A%20%20%20%20%20%20%20%20.filter(pl.col(%22pEC50_dr%22).is_not_null())%0A%20%20%20%20%20%20%20%20.select(%5B%22smiles%22%2C%20%22inchikey%22%2C%20%22molecule_names%22%2C%20%22pEC50_dr%22%5D)%0A%20%20%20%20)%0A%20%20%20%20PLOTS_DIR%20%3D%20Path(%22..%2Fplots%2F6_ml_optimization_3%22)%0A%20%20%20%20PLOTS_DIR.mkdir(parents%3DTrue%2C%20exist_ok%3DTrue)%0A%20%20%20%20PRED_DIR%20%3D%20Path(%22..%2Fpredictions%22)%0A%20%20%20%20gc.collect()%0A%20%20%20%20print(f%22DR%20training%20compounds%3A%20%7BDR_TRAIN.shape%5B0%5D%7D%22)%0A%20%20%20%20DR_TRAIN.head()%0A%20%20%20%20return%20DR_TRAIN%2C%20PLOTS_DIR%2C%20PRED_DIR%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%20Analysis%201%20%E2%80%94%20Post-hoc%20calibration%20(de-shrinking)%0A%0A%20%20%20%20The%20cheapest%20fix%20for%20regression-to-the-mean%20is%20a%20one-dimensional%20map%20applied%0A%20%20%20%20*after*%20training%20that%20pulls%20predictions%20back%20toward%20the%20truth%3A%20low%20predictions%0A%20%20%20%20down%2C%20high%20predictions%20up.%20Two%20variants%20are%20tested%2C%20exactly%20as%20proposed%20in%20the%0A%20%20%20%20blog%20post%3A%0A%0A%20%20%20%20-%20**Linear%20de-shrink**%20%E2%80%94%20%60LinearRegression%60%20of%20true%20on%20predicted%20pEC50.%20A%20fitted%0A%20%20%20%20%20%20slope%20%3E%201%20expands%20the%20prediction%20range.%0A%20%20%20%20-%20**Isotonic**%20%E2%80%94%20%60IsotonicRegression%60%2C%20a%20flexible%20monotonic%20map%20that%20can%20correct%0A%20%20%20%20%20%20curvature%20the%20linear%20fit%20cannot.%0A%0A%20%20%20%20**Leakage-free%20evaluation.**%20Calibration%20is%20model-agnostic%2C%20so%20it%20reuses%20the%0A%20%20%20%20out-of-fold%205%C3%975%20CV%20predictions%20already%20saved%20in%20notebook%204%20%E2%80%94%20no%20retraining.%20For%0A%20%20%20%20each%20test%20fold%20the%20calibrator%20is%20fit%20on%20the%20*sibling*%20inner%20folds%20of%20the%20same%0A%20%20%20%20outer%20repeat%20(a%20disjoint%20set%20of%20compounds)%20and%20then%20applied%20to%20the%20held-out%0A%20%20%20%20fold.%20This%20keeps%20the%20compounds%20used%20to%20fit%20the%20calibrator%20separate%20from%20those%0A%20%20%20%20used%20to%20score%20it%2C%20so%20the%20reported%20numbers%20are%20honest.%0A%0A%20%20%20%20Calibration%20is%20evaluated%20for%20all%20six%20HPO%20component%20models%20**and**%20the%20submitted%0A%20%20%20%20ensemble%20(%60cp5%C2%B7ch5%C2%B7rf0%C2%B7xg%E2%85%93%C2%B7mc1%C2%B7tf5%60)%2C%20reconstructed%20from%20the%20per-model%20OOF%0A%20%20%20%20predictions.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(PRED_DIR%2C%20pl)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Load%20the%20saved%205%C3%975%20CV%20out-of-fold%20predictions%20for%20all%20six%20models%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20def%20_load_oof(path%3A%20str%2C%20methods%3A%20list%5Bstr%5D)%20-%3E%20pl.DataFrame%3A%0A%20%20%20%20%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.read_csv(PRED_DIR%20%2F%20path)%0A%20%20%20%20%20%20%20%20%20%20%20%20.filter(pl.col(%22method%22).is_in(methods))%0A%20%20%20%20%20%20%20%20%20%20%20%20.select(%5B%22inchikey%22%2C%20%22molecule_names%22%2C%20%22fold%22%2C%20%22outer_fold%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22inner_fold%22%2C%20%22method%22%2C%20%22y_true%22%2C%20%22y_pred%22%5D)%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20oof_components%20%3D%20pl.concat(%5B%0A%20%20%20%20%20%20%20%20_load_oof(%224_hpo_best_5x5cv.csv.gz%22%2C%20%5B%22chemprop_hpo%22%2C%20%22chemeleon_hpo%22%5D)%2C%0A%20%20%20%20%20%20%20%20_load_oof(%224_hpo_a5_best_5x5cv.csv.gz%22%2C%20%5B%22macau_che_hpo%22%2C%20%22rf_mordred_hpo%22%2C%20%22xgb_mordred_hpo%22%5D)%2C%0A%20%20%20%20%20%20%20%20_load_oof(%224_fp_model_comparison_2.csv.gz%22%2C%20%5B%22tabpfn_chemeleon%22%5D)%2C%0A%20%20%20%20%5D)%0A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Reconstruct%20the%20submitted%20ensemble%20OOF%3A%20weighted%20average%20per%20compound%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20Weights%20from%20notebook%204's%20best%20submission%20(rf%20weight%20is%200%2C%20so%20rf%20is%20excluded).%0A%20%20%20%20_ENS_WEIGHTS%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22chemprop_hpo%22%3A%205.0%2C%20%22chemeleon_hpo%22%3A%205.0%2C%0A%20%20%20%20%20%20%20%20%22xgb_mordred_hpo%22%3A%201.0%20%2F%203.0%2C%20%22macau_che_hpo%22%3A%201.0%2C%20%22tabpfn_chemeleon%22%3A%205.0%2C%0A%20%20%20%20%7D%0A%20%20%20%20_wide%20%3D%20(%0A%20%20%20%20%20%20%20%20oof_components%0A%20%20%20%20%20%20%20%20.filter(pl.col(%22method%22).is_in(list(_ENS_WEIGHTS)))%0A%20%20%20%20%20%20%20%20.pivot(values%3D%22y_pred%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20index%3D%5B%22inchikey%22%2C%20%22molecule_names%22%2C%20%22fold%22%2C%20%22outer_fold%22%2C%20%22inner_fold%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20on%3D%22method%22%2C%20aggregate_function%3D%22first%22)%0A%20%20%20%20)%0A%20%20%20%20_truth%20%3D%20oof_components.select(%5B%22inchikey%22%2C%20%22outer_fold%22%2C%20%22y_true%22%5D).unique(%0A%20%20%20%20%20%20%20%20subset%3D%5B%22inchikey%22%2C%20%22outer_fold%22%5D)%0A%20%20%20%20_num%20%3D%20sum(w%20*%20pl.col(m)%20for%20m%2C%20w%20in%20_ENS_WEIGHTS.items())%0A%20%20%20%20_den%20%3D%20sum(_ENS_WEIGHTS.values())%0A%20%20%20%20ensemble_oof%20%3D%20(%0A%20%20%20%20%20%20%20%20_wide.join(_truth%2C%20on%3D%5B%22inchikey%22%2C%20%22outer_fold%22%5D%2C%20how%3D%22left%22)%0A%20%20%20%20%20%20%20%20.with_columns((_num%20%2F%20_den).alias(%22y_pred%22)%2C%20pl.lit(%22ensemble%22).alias(%22method%22))%0A%20%20%20%20%20%20%20%20.select(%5B%22inchikey%22%2C%20%22molecule_names%22%2C%20%22fold%22%2C%20%22outer_fold%22%2C%20%22inner_fold%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22method%22%2C%20%22y_true%22%2C%20%22y_pred%22%5D)%0A%20%20%20%20)%0A%0A%20%20%20%20cv_oof_all%20%3D%20pl.concat(%5Boof_components%2C%20ensemble_oof%5D)%0A%20%20%20%20print(%22OOF%20rows%20per%20method%3A%22)%0A%20%20%20%20print(cv_oof_all.group_by(%22method%22).len().sort(%22method%22))%0A%20%20%20%20return%20cv_oof_all%2C%20ensemble_oof%0A%0A%0A%40app.cell%0Adef%20_(IsotonicRegression%2C%20LinearRegression%2C%20np%2C%20pl)%3A%0A%20%20%20%20def%20crossfit_calibrate(df_method%3A%20pl.DataFrame%2C%20kind%3A%20str)%20-%3E%20pl.DataFrame%3A%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20Apply%20leakage-free%20cross-fit%20calibration%20to%20one%20method's%20OOF%20predictions.%0A%0A%20%20%20%20%20%20%20%20For%20each%20of%20the%2025%20folds%20the%20calibrator%20is%20fit%20on%20the%20other%20four%20inner%0A%20%20%20%20%20%20%20%20folds%20of%20the%20same%20outer%20repeat%20(disjoint%20compounds)%20and%20applied%20to%20the%0A%20%20%20%20%20%20%20%20held-out%20fold.%0A%0A%20%20%20%20%20%20%20%20Args%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20df_method%3A%20OOF%20rows%20for%20a%20single%20method%20(cols%3A%20fold%2C%20outer_fold%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20inner_fold%2C%20y_true%2C%20y_pred%2C%20...).%0A%20%20%20%20%20%20%20%20%20%20%20%20kind%3A%20%22raw%22%20(identity)%2C%20%22linear%22%20or%20%22isotonic%22.%0A%0A%20%20%20%20%20%20%20%20Returns%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20The%20input%20rows%20with%20an%20added%20%60y_cal%60%20column.%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20parts%3A%20list%5Bpl.DataFrame%5D%20%3D%20%5B%5D%0A%20%20%20%20%20%20%20%20for%20fold%20in%20range(25)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20outer%20%3D%20fold%20%2F%2F%205%0A%20%20%20%20%20%20%20%20%20%20%20%20fit%20%3D%20df_method.filter((pl.col(%22outer_fold%22)%20%3D%3D%20outer)%20%26%20(pl.col(%22fold%22)%20!%3D%20fold))%0A%20%20%20%20%20%20%20%20%20%20%20%20test%20%3D%20df_method.filter(pl.col(%22fold%22)%20%3D%3D%20fold)%0A%20%20%20%20%20%20%20%20%20%20%20%20xp_fit%20%3D%20fit%5B%22y_pred%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20yt_fit%20%3D%20fit%5B%22y_true%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20xp_test%20%3D%20test%5B%22y_pred%22%5D.to_numpy()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20kind%20%3D%3D%20%22linear%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20cal%20%3D%20LinearRegression().fit(xp_fit.reshape(-1%2C%201)%2C%20yt_fit)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20y_cal%20%3D%20cal.predict(xp_test.reshape(-1%2C%201))%0A%20%20%20%20%20%20%20%20%20%20%20%20elif%20kind%20%3D%3D%20%22isotonic%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20cal%20%3D%20IsotonicRegression(out_of_bounds%3D%22clip%22).fit(xp_fit%2C%20yt_fit)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20y_cal%20%3D%20cal.predict(xp_test)%0A%20%20%20%20%20%20%20%20%20%20%20%20elif%20kind%20%3D%3D%20%22raw%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20y_cal%20%3D%20xp_test%0A%20%20%20%20%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20raise%20ValueError(f%22Unknown%20calibration%20kind%3A%20%7Bkind!r%7D%22)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20parts.append(test.with_columns(pl.Series(%22y_cal%22%2C%20np.asarray(y_cal))))%0A%20%20%20%20%20%20%20%20return%20pl.concat(parts)%0A%0A%20%20%20%20return%20(crossfit_calibrate%2C)%0A%0A%0A%40app.cell%0Adef%20_(crossfit_calibrate%2C%20cv_oof_all%2C%20pl)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Run%20all%20three%20calibration%20variants%20for%20every%20method%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_records%3A%20list%5Bpl.DataFrame%5D%20%3D%20%5B%5D%0A%20%20%20%20for%20_method%20in%20cv_oof_all%5B%22method%22%5D.unique().to_list()%3A%0A%20%20%20%20%20%20%20%20_dfm%20%3D%20cv_oof_all.filter(pl.col(%22method%22)%20%3D%3D%20_method)%0A%20%20%20%20%20%20%20%20for%20_kind%20in%20%5B%22raw%22%2C%20%22linear%22%2C%20%22isotonic%22%5D%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_cal%20%3D%20crossfit_calibrate(_dfm%2C%20_kind).with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.lit(_method).alias(%22base_model%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.lit(_kind).alias(%22calibration%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.format(%22%7B%7D__%7B%7D%22%2C%20pl.lit(_method)%2C%20pl.lit(_kind)).alias(%22method_cal%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20_records.append(_cal)%0A%0A%20%20%20%20calib_cv%20%3D%20pl.concat(_records)%0A%20%20%20%20print(f%22Calibrated%20CV%20rows%3A%20%7Bcalib_cv.shape%5B0%5D%3A%2C%7D%20%22%0A%20%20%20%20%20%20%20%20%20%20f%22(%7Bcalib_cv%5B'base_model'%5D.n_unique()%7D%20models%20%C3%97%203%20variants)%22)%0A%20%20%20%20calib_cv%0A%20%20%20%20return%20(calib_cv%2C)%0A%0A%0A%40app.cell%0Adef%20_(calib_cv%2C%20mean_absolute_error%2C%20mo%2C%20pl)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Summary%3A%20overall%20%2B%20extreme-zone%20metrics%20per%20model%20%C3%97%20calibration%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20def%20_zone_mae(df%3A%20pl.DataFrame%2C%20lo%3A%20float%20%7C%20None%2C%20hi%3A%20float%20%7C%20None)%20-%3E%20float%3A%0A%20%20%20%20%20%20%20%20sub%20%3D%20df%0A%20%20%20%20%20%20%20%20if%20lo%20is%20not%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20sub%20%3D%20sub.filter(pl.col(%22y_true%22)%20%3E%3D%20lo)%0A%20%20%20%20%20%20%20%20if%20hi%20is%20not%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20sub%20%3D%20sub.filter(pl.col(%22y_true%22)%20%3C%20hi)%0A%20%20%20%20%20%20%20%20if%20sub.shape%5B0%5D%20%3D%3D%200%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20float(%22nan%22)%0A%20%20%20%20%20%20%20%20return%20mean_absolute_error(sub%5B%22y_true%22%5D.to_numpy()%2C%20sub%5B%22y_cal%22%5D.to_numpy())%0A%0A%20%20%20%20_rows%20%3D%20%5B%5D%0A%20%20%20%20for%20(_base%2C%20_cal)%2C%20_grp%20in%20calib_cv.group_by(%5B%22base_model%22%2C%20%22calibration%22%5D)%3A%0A%20%20%20%20%20%20%20%20_yt%20%3D%20_grp%5B%22y_true%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_yc%20%3D%20_grp%5B%22y_cal%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_hit%20%3D%20_grp.filter(pl.col(%22y_true%22)%20%3E%3D%206.0)%0A%20%20%20%20%20%20%20%20_inact%20%3D%20_grp.filter(pl.col(%22y_true%22)%20%3C%204.0)%0A%20%20%20%20%20%20%20%20_rows.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22base_model%22%3A%20_base%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22calibration%22%3A%20_cal%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22MAE%22%3A%20round(mean_absolute_error(_yt%2C%20_yc)%2C%204)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22hitzone_MAE%22%3A%20round(_zone_mae(_grp%2C%206.0%2C%20None)%2C%203)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22hitzone_bias%22%3A%20round(float((_hit%5B%22y_cal%22%5D%20-%20_hit%5B%22y_true%22%5D).mean())%2C%203)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22inactive_bias%22%3A%20round(float((_inact%5B%22y_cal%22%5D%20-%20_inact%5B%22y_true%22%5D).mean())%2C%203)%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%0A%20%20%20%20calib_summary%20%3D%20(%0A%20%20%20%20%20%20%20%20pl.DataFrame(_rows)%0A%20%20%20%20%20%20%20%20.with_columns(pl.col(%22calibration%22).cast(pl.Enum(%5B%22raw%22%2C%20%22linear%22%2C%20%22isotonic%22%5D)))%0A%20%20%20%20%20%20%20%20.sort(%5B%22base_model%22%2C%20%22calibration%22%5D)%0A%20%20%20%20)%0A%0A%20%20%20%20mo.vstack(%5B%0A%20%20%20%20%20%20%20%20mo.md(%22%23%23%23%20Calibration%20summary%20%E2%80%94%20overall%20MAE%20and%20extreme-zone%20behaviour%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22%60hitzone%60%20%3D%20pEC50%20%E2%89%A5%206%3B%20%60inactive%60%20%3D%20pEC50%20%3C%204.%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Bias%20is%20mean%20signed%20error%20(pred%20%E2%88%92%20true).%22)%2C%0A%20%20%20%20%20%20%20%20mo.ui.table(calib_summary.to_pandas()%2C%20selection%3DNone%2C%20pagination%3DFalse)%2C%0A%20%20%20%20%5D)%0A%20%20%20%20return%20(calib_summary%2C)%0A%0A%0A%40app.cell%0Adef%20_(PLOTS_DIR%2C%20calib_summary%2C%20mo%2C%20np%2C%20plt)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Visual%20comparison%20of%20calibration%20effect%20across%20models%20and%20ensemble%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_models%20%3D%20calib_summary%5B%22base_model%22%5D.unique().sort().to_list()%0A%20%20%20%20_calibrations%20%3D%20%5B%22raw%22%2C%20%22linear%22%2C%20%22isotonic%22%5D%0A%20%20%20%20_metrics%20%3D%20%5B%22MAE%22%2C%20%22hitzone_MAE%22%2C%20%22hitzone_bias%22%2C%20%22inactive_bias%22%5D%0A%20%20%20%20_titles%20%3D%20%5B%22Overall%20MAE%22%2C%20%22Hit-zone%20MAE%20(pEC50%20%E2%89%A5%206)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Hit-zone%20bias%20(pred%20%E2%88%92%20true)%22%2C%20%22Inactive%20bias%20(pred%20%E2%88%92%20true)%22%5D%0A%20%20%20%20_colors%20%3D%20%7B%22raw%22%3A%20%22%234e79a7%22%2C%20%22linear%22%3A%20%22%23e15759%22%2C%20%22isotonic%22%3A%20%22%2359a14f%22%7D%0A%0A%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20fig_cal%2C%20axes_cal%20%3D%20plt.subplots(2%2C%202%2C%20figsize%3D(14%2C%209)%2C%20dpi%3D140)%0A%20%20%20%20%20%20%20%20x_pos%20%3D%20np.arange(len(_models))%0A%20%20%20%20%20%20%20%20bar_w%20%3D%200.25%0A%0A%20%20%20%20%20%20%20%20for%20ax%2C%20metric%2C%20title%20in%20zip(axes_cal.flatten()%2C%20_metrics%2C%20_titles)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20k%2C%20cal%20in%20enumerate(_calibrations)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20vals%20%3D%20%5B%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20model%20in%20_models%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20row%20%3D%20calib_summary.filter(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(calib_summary%5B%22base_model%22%5D%20%3D%3D%20model)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%26%20(calib_summary%5B%22calibration%22%5D%20%3D%3D%20cal)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20vals.append(float(row%5Bmetric%5D%5B0%5D)%20if%20row.shape%5B0%5D%20else%20float(%22nan%22))%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20offset%20%3D%20(k%20-%201)%20*%20bar_w%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20ax.bar(x_pos%20%2B%20offset%2C%20vals%2C%20bar_w%2C%20label%3Dcal%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20color%3D_colors%5Bcal%5D%2C%20edgecolor%3D%22white%22%2C%20linewidth%3D0.5)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20metric%20in%20(%22hitzone_bias%22%2C%20%22inactive_bias%22)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20ax.axhline(0%2C%20color%3D%22black%22%2C%20linewidth%3D0.9%2C%20linestyle%3D%22--%22%2C%20zorder%3D0)%0A%20%20%20%20%20%20%20%20%20%20%20%20ax.set_xticks(x_pos)%0A%20%20%20%20%20%20%20%20%20%20%20%20ax.set_xticklabels(_models%2C%20fontsize%3D8%2C%20rotation%3D30%2C%20ha%3D%22right%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20ax.set_title(title%2C%20fontsize%3D11)%0A%0A%20%20%20%20%20%20%20%20handles%2C%20labels%20%3D%20axes_cal%5B0%2C%200%5D.get_legend_handles_labels()%0A%20%20%20%20%20%20%20%20fig_cal.legend(handles%2C%20labels%2C%20loc%3D%22upper%20center%22%2C%20ncol%3D3%2C%20fontsize%3D10%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20title%3D%22calibration%22%2C%20title_fontsize%3D10%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20framealpha%3D1.0%2C%20facecolor%3D%22white%22%2C%20edgecolor%3D%22%23cccccc%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20bbox_to_anchor%3D(0.5%2C%201.02))%0A%20%20%20%20%20%20%20%20fig_cal.suptitle(%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Effect%20of%20post-hoc%20calibration%20across%20models%20and%20ensemble%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D13%2C%20y%3D1.06)%0A%20%20%20%20%20%20%20%20fig_cal.tight_layout()%0A%20%20%20%20%20%20%20%20fig_cal.savefig(PLOTS_DIR%20%2F%20%22analysis1_calibration_comparison.png%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20mo.vstack(%5B%0A%20%20%20%20%20%20%20%20mo.md(%22%23%23%23%20Calibration%20comparison%20across%20all%20models%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Grouped%20bars%20show%20the%20raw%20(uncalibrated)%2C%20linear%2C%20and%20isotonic%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22calibration%20variants%20for%20each%20base%20model.%20The%20bias%20panels%20show%20signed%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22error%3A%20positive%20%3D%20overprediction%2C%20negative%20%3D%20underprediction.%20Ideal%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22bias%20is%20zero.%22)%2C%0A%20%20%20%20%20%20%20%20mo.as_html(fig_cal)%2C%0A%20%20%20%20%5D)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20PLOTS_DIR%2C%0A%20%20%20%20calc_regression_metrics%2C%0A%20%20%20%20calib_cv%2C%0A%20%20%20%20make_mcs_plot_grid%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20pl%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20MCS%20(Tukey%20HSD)%20on%20the%20ensemble%3A%20raw%20vs%20linear%20vs%20isotonic%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_ens%20%3D%20(%0A%20%20%20%20%20%20%20%20calib_cv%0A%20%20%20%20%20%20%20%20.filter(pl.col(%22base_model%22)%20%3D%3D%20%22ensemble%22)%0A%20%20%20%20%20%20%20%20.select(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22fold%22).alias(%22cv_cycle%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22calibration%22).alias(%22method%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.lit(%22random%22).alias(%22split%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22y_true%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22y_cal%22).alias(%22y_pred%22)%2C%0A%20%20%20%20%20%20%20%20%5D)%0A%20%20%20%20)%0A%20%20%20%20_metrics%20%3D%20calc_regression_metrics(_ens%2C%20%22cv_cycle%22%2C%20%22y_true%22%2C%20%22y_pred%22%2C%20thresh%3D4.0)%0A%0A%20%20%20%20_fig%20%3D%20make_mcs_plot_grid(%0A%20%20%20%20%20%20%20%20_metrics%2C%20stats%3D%5B%22mae%22%5D%2C%20group_col%3D%22method%22%2C%0A%20%20%20%20%20%20%20%20figsize%3D(7%2C%206)%2C%20effect_dict%3D%7B%22mae%22%3A%200.02%7D%2C%0A%20%20%20%20%20%20%20%20save_path%3DPLOTS_DIR%20%2F%20%22analysis1_calibration_mcs_mae.png%22%2C%0A%20%20%20%20)%0A%20%20%20%20mo.vstack(%5B%0A%20%20%20%20%20%20%20%20mo.md(%22%23%23%23%20Ensemble%20%E2%80%94%20does%20calibration%20significantly%20change%20CV%20MAE%3F%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Tukey%20HSD%20across%20the%2025%20folds.%20Cells%20show%20the%20mean%20MAE%20difference%3B%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22%60*%60%20marks%20p%20%3C%200.05.%22)%2C%0A%20%20%20%20%20%20%20%20mo.as_html(_fig)%2C%0A%20%20%20%20%5D)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(PLOTS_DIR%2C%20bias_by_bin_table%2C%20calib_cv%2C%20mo%2C%20pl%2C%20plot_bias_heatmap)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Per-bin%20signed-error%20heatmap%20for%20the%20ensemble%20(raw%20vs%20linear%20vs%20isotonic)%20%E2%94%80%0A%20%20%20%20_ens%20%3D%20calib_cv.filter(pl.col(%22base_model%22)%20%3D%3D%20%22ensemble%22).with_columns(%0A%20%20%20%20%20%20%20%20pl.col(%22calibration%22).alias(%22method%22))%0A%20%20%20%20_bias%20%3D%20bias_by_bin_table(_ens%2C%20%22method%22%2C%20%22y_true%22%2C%20%22y_cal%22)%0A%20%20%20%20_fig%20%3D%20plot_bias_heatmap(%0A%20%20%20%20%20%20%20%20_bias%2C%20method_order%3D%5B%22raw%22%2C%20%22linear%22%2C%20%22isotonic%22%5D%2C%0A%20%20%20%20%20%20%20%20title%3D%22Ensemble%20mean%20signed%20error%20by%20activity%20bin%5Cn(red%20%3D%20overpredict%2C%20blue%20%3D%20underpredict)%22%2C%0A%20%20%20%20%20%20%20%20save_path%3DPLOTS_DIR%20%2F%20%22analysis1_calibration_bias_heatmap.png%22%2C%0A%20%20%20%20)%0A%20%20%20%20mo.vstack(%5B%0A%20%20%20%20%20%20%20%20mo.md(%22%23%23%23%20Where%20does%20calibration%20act%3F%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22If%20de-shrinking%20works%2C%20the%20%60%3C4%60%20column%20should%20move%20toward%200%20from%20the%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22right%20(less%20overprediction)%20and%20the%20%60%3E6%60%20column%20toward%200%20from%20the%20left%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22(less%20underprediction).%22)%2C%0A%20%20%20%20%20%20%20%20mo.as_html(_fig)%2C%0A%20%20%20%20%5D)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20While%20the%20methods%20we%20used%20worked%20in%20that%20the%20bias%20is%20closer%20to%20zero%20in%20all%20cases%2C%20the%20effect%20is%20very%20small.%0A%20%20%20%20The%20isotonic%20method%20was%20slightly%20better%20than%20linear.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%20Analysis%202%20%E2%80%94%20Loss%20%2F%20sample%20reweighting%0A%0A%20%20%20%20Calibration%20acts%20after%20the%20fact%3B%20reweighting%20changes%20what%20the%20model%20learns.%20By%0A%20%20%20%20up-weighting%20compounds%20at%20the%20activity%20extremes%2C%20the%20model%20is%20pushed%20to%20commit%0A%20%20%20%20to%20potent%20and%20inactive%20compounds%20instead%20of%20hedging%20toward%20the%20dense%20middle%20of%0A%20%20%20%20the%20distribution.%0A%0A%20%20%20%20Three%20weighting%20schemes%20are%20compared%3A%0A%0A%20%20%20%20%7C%20Scheme%20%7C%20Weight%20for%20a%20compound%20with%20label%20*y*%20%7C%0A%20%20%20%20%7C---%7C---%7C%0A%20%20%20%20%7C%20%60uniform%60%20%7C%201%20(baseline)%20%7C%0A%20%20%20%20%7C%20%60invdensity%60%20%7C%20%E2%88%9D%201%20%2F%20(population%20of%20*y*'s%20pEC50%20bin)%20%E2%80%94%20up-weights%20rare%20regions%20%7C%0A%20%20%20%20%7C%20%60distance%60%20%7C%201%20%2B%20%5C%7C*y*%20%E2%88%92%20median(*y*)%5C%7C%20%E2%80%94%20grows%20linearly%20toward%20the%20extremes%20%7C%0A%0A%20%20%20%20**Which%20models%20can%20be%20reweighted%3F**%20Only%20models%20with%20a%20per-sample%20loss%20weight%3A%0A%0A%20%20%20%20%7C%20Model%20%7C%20Reweighting%20%7C%20Mechanism%20%7C%0A%20%20%20%20%7C---%7C---%7C---%7C%0A%20%20%20%20%7C%20XGBoost%20%C2%B7%20Mordred%20%7C%20%E2%9C%93%20%7C%20sklearn%20%60sample_weight%60%20%7C%0A%20%20%20%20%7C%20RF%20%C2%B7%20CheMeleon%20%7C%20%E2%9C%93%20%7C%20sklearn%20%60sample_weight%60%20%7C%0A%20%20%20%20%7C%20Chemprop%20%C2%B7%20scratch%20%7C%20%E2%9C%93%20%7C%20%60MoleculeDatapoint.weight%60%20%2F%20%60chemprop%20train%20-w%60%20%7C%0A%20%20%20%20%7C%20TabPFN%20%7C%20%E2%9C%97%20%7C%20in-context%20model%20%E2%80%94%20no%20trainable%20loss%20to%20weight%20%7C%0A%20%20%20%20%7C%20Macau%20(smurff)%20%7C%20%E2%9C%97%20%7C%20no%20per-observation%20weight%20in%20the%20API%20%7C%0A%0A%20%20%20%20So%20**TabPFN%20and%20Macau%20are%20out**%3B%20Chemprop%20is%20added%20as%20the%20deep-model%20case.%20To%20save%0A%20%20%20%20compute%2C%20only%20%60invdensity%60%20and%20%60distance%60%20are%20trained%20for%20Chemprop%20here%20%E2%80%94%20its%0A%20%20%20%20%60uniform%60%20baseline%20is%20taken%20from%20notebook%202%20(identical%20folds%20and%20architecture).%0A%0A%20%20%20%20All%20weights%20are%20normalised%20to%20mean%201%20so%20the%20effective%20regularisation%20strength%20is%0A%20%20%20%20unchanged.%20Same%205%C3%975%20CV%20splits%20as%20everywhere%20else%20(seed%20%3D%2042%2C%2010%20%25%20inner%0A%20%20%20%20validation%20for%20early%20stopping).%20Tree-model%20features%20are%20pre-computed%20once%20and%0A%20%20%20%20indexed%20per%20fold%20so%20the%20only%20thing%20that%20varies%20between%20schemes%20is%20the%20weighting.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20Chem%2C%0A%20%20%20%20DR_TRAIN%2C%0A%20%20%20%20PRED_DIR%2C%0A%20%20%20%20chemeleon_embed%2C%0A%20%20%20%20extract_fp_matrix%2C%0A%20%20%20%20generate_fingerprint%2C%0A%20%20%20%20np%2C%0A%20%20%20%20pl%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Pre-compute%20Mordred%20%2B%20CheMeleon%20features%20for%20every%20compound%20we%20will%20need%20%E2%94%80%E2%94%80%0A%20%20%20%20%23%20The%20feature%20of%20a%20molecule%20never%20changes%20between%20folds%2C%20so%20we%20embed%2Ffingerprint%0A%20%20%20%20%23%20the%20union%20of%20(DR%20training%20set%20%2B%20semi-pure%20set)%20exactly%20once%20and%20cache%20to%20disk.%0A%20%20%20%20_SEMIPURE_RAW%20%3D%20pl.read_csv(%0A%20%20%20%20%20%20%20%20%22..%2Fdata%2Fraw%2F20260619%2Fpxr-challenge_96-compound-uscale-semi-pure_TRAIN.csv%22)%0A%0A%20%20%20%20def%20_ik(smi%3A%20str)%20-%3E%20str%20%7C%20None%3A%0A%20%20%20%20%20%20%20%20m%20%3D%20Chem.MolFromSmiles(smi)%0A%20%20%20%20%20%20%20%20return%20Chem.MolToInchiKey(m)%20if%20m%20else%20None%0A%0A%20%20%20%20semipure%20%3D%20(%0A%20%20%20%20%20%20%20%20_SEMIPURE_RAW%0A%20%20%20%20%20%20%20%20.rename(%7B%22SMILES%22%3A%20%22smiles%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Corrected%20Semi-Pure%20pEC50%20(log)%22%3A%20%22pEC50_corrected%22%7D)%0A%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22pEC50_corrected%22).cast(pl.Float64%2C%20strict%3DFalse)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22smiles%22).map_elements(_ik%2C%20return_dtype%3Dpl.Utf8).alias(%22inchikey%22)%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20.filter(pl.col(%22pEC50_corrected%22).is_not_null()%20%26%20pl.col(%22inchikey%22).is_not_null())%0A%20%20%20%20%20%20%20%20.select(%5B%22smiles%22%2C%20%22inchikey%22%2C%20%22pEC50_corrected%22%5D)%0A%20%20%20%20%20%20%20%20.unique(subset%3D%5B%22inchikey%22%5D%2C%20keep%3D%22first%22)%0A%20%20%20%20)%0A%0A%20%20%20%20_feat_df%20%3D%20(%0A%20%20%20%20%20%20%20%20pl.concat(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20DR_TRAIN.select(%5B%22inchikey%22%2C%20%22smiles%22%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20semipure.select(%5B%22inchikey%22%2C%20%22smiles%22%5D)%2C%0A%20%20%20%20%20%20%20%20%5D)%0A%20%20%20%20%20%20%20%20.unique(subset%3D%5B%22inchikey%22%5D%2C%20keep%3D%22first%22)%0A%20%20%20%20)%0A%0A%20%20%20%20FEATURE_CACHE_PATH%20%3D%20PRED_DIR%20%2F%20%226_feature_cache.npz%22%0A%20%20%20%20if%20FEATURE_CACHE_PATH.exists()%3A%0A%20%20%20%20%20%20%20%20_cache%20%3D%20np.load(FEATURE_CACHE_PATH%2C%20allow_pickle%3DTrue)%0A%20%20%20%20%20%20%20%20feature_cache%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22iks%22%3A%20list(_cache%5B%22iks%22%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22mordred%22%3A%20_cache%5B%22mordred%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22chemeleon%22%3A%20_cache%5B%22chemeleon%22%5D%2C%0A%20%20%20%20%20%20%20%20%7D%0A%20%20%20%20%20%20%20%20print(f%22Loaded%20feature%20cache%3A%20%7BFEATURE_CACHE_PATH.name%7D%22)%0A%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20print(f%22Computing%20features%20for%20%7B_feat_df.shape%5B0%5D%7D%20unique%20compounds%20%E2%80%A6%22)%0A%20%20%20%20%20%20%20%20_che%2C%20_%20%3D%20chemeleon_embed(_feat_df%5B%22smiles%22%5D.to_list()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_feat_df%5B%22smiles%22%5D.to_list()%5B%3A1%5D%2C%20prefix%3D%22a2cache%22)%0A%20%20%20%20%20%20%20%20_mor_df%20%3D%20generate_fingerprint(_feat_df%2C%20%22mordred%22)%0A%20%20%20%20%20%20%20%20_mor%20%3D%20extract_fp_matrix(_mor_df%2C%20%22mordred%22)%0A%20%20%20%20%20%20%20%20%23%20Drop%20descriptor%20columns%20that%20are%20NaN%20for%20any%20compound%20(consistent%20mask).%0A%20%20%20%20%20%20%20%20_mask%20%3D%20~np.isnan(_mor).any(axis%3D0)%0A%20%20%20%20%20%20%20%20_mor%20%3D%20_mor%5B%3A%2C%20_mask%5D%0A%20%20%20%20%20%20%20%20feature_cache%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22iks%22%3A%20_feat_df%5B%22inchikey%22%5D.to_list()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22mordred%22%3A%20_mor.astype(np.float32)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22chemeleon%22%3A%20_che.astype(np.float32)%2C%0A%20%20%20%20%20%20%20%20%7D%0A%20%20%20%20%20%20%20%20np.savez_compressed(%0A%20%20%20%20%20%20%20%20%20%20%20%20FEATURE_CACHE_PATH%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20iks%3Dnp.array(feature_cache%5B%22iks%22%5D%2C%20dtype%3Dobject)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mordred%3Dfeature_cache%5B%22mordred%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20chemeleon%3Dfeature_cache%5B%22chemeleon%22%5D%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20print(f%22Cached%20%E2%86%92%20%7BFEATURE_CACHE_PATH.name%7D%20%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22(mordred%20%7Bfeature_cache%5B'mordred'%5D.shape%7D%2C%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22chemeleon%20%7Bfeature_cache%5B'chemeleon'%5D.shape%7D)%22)%0A%0A%20%20%20%20_ik2idx%20%3D%20%7Bik%3A%20i%20for%20i%2C%20ik%20in%20enumerate(feature_cache%5B%22iks%22%5D)%7D%0A%0A%20%20%20%20def%20gather_X(inchikeys%3A%20list%5Bstr%5D%2C%20kind%3A%20str)%20-%3E%20np.ndarray%3A%0A%20%20%20%20%20%20%20%20%22%22%22Return%20the%20cached%20feature%20matrix%20(kind%20%3D%20'mordred'%7C'chemeleon')%20for%20rows.%22%22%22%0A%20%20%20%20%20%20%20%20idx%20%3D%20%5B_ik2idx%5Bik%5D%20for%20ik%20in%20inchikeys%5D%0A%20%20%20%20%20%20%20%20return%20feature_cache%5Bkind%5D%5Bidx%5D%0A%0A%20%20%20%20print(f%22semi-pure%20usable%20compounds%3A%20%7Bsemipure.shape%5B0%5D%7D%22)%0A%20%20%20%20return%20gather_X%2C%20semipure%0A%0A%0A%40app.cell%0Adef%20_(np)%3A%0A%20%20%20%20def%20compute_sample_weights(%0A%20%20%20%20%20%20%20%20y%3A%20np.ndarray%2C%20scheme%3A%20str%2C%20n_bins%3A%20int%20%3D%2020%2C%20alpha%3A%20float%20%3D%201.0%2C%0A%20%20%20%20)%20-%3E%20np.ndarray%3A%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20Per-sample%20training%20weights%20for%20a%20reweighting%20scheme%20(mean-normalised%20to%201).%0A%0A%20%20%20%20%20%20%20%20Args%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3A%20Training%20labels%20(pEC50).%0A%20%20%20%20%20%20%20%20%20%20%20%20scheme%3A%20%22uniform%22%2C%20%22invdensity%22%20or%20%22distance%22.%0A%20%20%20%20%20%20%20%20%20%20%20%20n_bins%3A%20Number%20of%20histogram%20bins%20for%20inverse-density%20weighting.%0A%20%20%20%20%20%20%20%20%20%20%20%20alpha%3A%20Slope%20for%20the%20distance-from-median%20scheme.%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20if%20scheme%20%3D%3D%20%22uniform%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20w%20%3D%20np.ones_like(y%2C%20dtype%3Dfloat)%0A%20%20%20%20%20%20%20%20elif%20scheme%20%3D%3D%20%22invdensity%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20counts%2C%20edges%20%3D%20np.histogram(y%2C%20bins%3Dn_bins)%0A%20%20%20%20%20%20%20%20%20%20%20%20idx%20%3D%20np.clip(np.digitize(y%2C%20edges%5B1%3A-1%5D)%2C%200%2C%20n_bins%20-%201)%0A%20%20%20%20%20%20%20%20%20%20%20%20w%20%3D%201.0%20%2F%20np.maximum(counts%5Bidx%5D%2C%201).astype(float)%0A%20%20%20%20%20%20%20%20elif%20scheme%20%3D%3D%20%22distance%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20w%20%3D%201.0%20%2B%20alpha%20*%20np.abs(y%20-%20np.median(y))%0A%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20raise%20ValueError(f%22Unknown%20weighting%20scheme%3A%20%7Bscheme!r%7D%22)%0A%20%20%20%20%20%20%20%20return%20w%20*%20(len(w)%20%2F%20w.sum())%0A%0A%20%20%20%20return%20(compute_sample_weights%2C)%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20BoostedTreesModel%2C%0A%20%20%20%20DR_TRAIN%2C%0A%20%20%20%20PRED_DIR%2C%0A%20%20%20%20RandomForestModel%2C%0A%20%20%20%20compute_sample_weights%2C%0A%20%20%20%20gather_X%2C%0A%20%20%20%20gc%2C%0A%20%20%20%20generate_cv_splits_random%2C%0A%20%20%20%20gzip%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20tqdm%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%205%C3%975%20CV%3A%20two%20models%20%C3%97%20three%20weighting%20schemes%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_OUT%20%3D%20PRED_DIR%20%2F%20%226_reweighting_cv.csv.gz%22%0A%20%20%20%20_SCHEMES%20%3D%20%5B%22uniform%22%2C%20%22invdensity%22%2C%20%22distance%22%5D%0A%20%20%20%20_MODELS%20%3D%20%5B(%22xgb_mordred%22%2C%20%22mordred%22)%2C%20(%22rf_chemeleon%22%2C%20%22chemeleon%22)%5D%0A%0A%20%20%20%20if%20_OUT.exists()%3A%0A%20%20%20%20%20%20%20%20reweight_cv%20%3D%20pl.read_csv(_OUT)%0A%20%20%20%20%20%20%20%20print(f%22Found%20%7B_OUT.name%7D%20%E2%80%94%20skipping%20training%20(%7Breweight_cv.shape%5B0%5D%3A%2C%7D%20rows).%22)%0A%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20_records%3A%20list%5Bdict%5D%20%3D%20%5B%5D%0A%20%20%20%20%20%20%20%20for%20_fold%2C%20_outer%2C%20_inner%2C%20_tr%2C%20_va%2C%20_te%20in%20tqdm(%0A%20%20%20%20%20%20%20%20%20%20%20%20generate_cv_splits_random(DR_TRAIN%2C%20n_outer%3D5%2C%20n_inner%3D5%2C%20seed%3D42%2C%20p_val%3D0.1)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20total%3D25%2C%20desc%3D%22reweighting%20CV%22%2C%20unit%3D%22fold%22%2C%0A%20%20%20%20%20%20%20%20)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_tr%20%3D%20_tr%5B%22pEC50_dr%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_va%20%3D%20_va%5B%22pEC50_dr%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_te%20%3D%20_te%5B%22pEC50_dr%22%5D.to_numpy()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20_model_key%2C%20_kind%20in%20_MODELS%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_X_tr%20%3D%20gather_X(_tr%5B%22inchikey%22%5D.to_list()%2C%20_kind)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_X_va%20%3D%20gather_X(_va%5B%22inchikey%22%5D.to_list()%2C%20_kind)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_X_te%20%3D%20gather_X(_te%5B%22inchikey%22%5D.to_list()%2C%20_kind)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20_scheme%20in%20_SCHEMES%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_w%20%3D%20compute_sample_weights(_y_tr%2C%20_scheme)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20_model_key%20%3D%3D%20%22xgb_mordred%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m%20%3D%20BoostedTreesModel(pred_type%3D%22regression%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m.train(_X_tr%2C%20_y_tr%2C%20_X_va%2C%20_y_va%2C%20sample_weight%3D_w)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m%20%3D%20RandomForestModel(pred_type%3D%22regression%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m.train(_X_tr%2C%20_y_tr%2C%20sample_weight%3D_w)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_pred%20%3D%20_m.predict(_X_te)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20del%20_m%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20gc.collect()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20_ik%2C%20_yt%2C%20_yp%20in%20zip(_te%5B%22inchikey%22%5D.to_list()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_y_te.tolist()%2C%20_pred.tolist())%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_records.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22inchikey%22%3A%20_ik%2C%20%22fold%22%3A%20_fold%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22model%22%3A%20_model_key%2C%20%22scheme%22%3A%20_scheme%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22method%22%3A%20f%22%7B_model_key%7D_%7B_scheme%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22y_true%22%3A%20_yt%2C%20%22y_pred%22%3A%20_yp%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7D)%0A%0A%20%20%20%20%20%20%20%20reweight_cv%20%3D%20pl.DataFrame(_records)%0A%20%20%20%20%20%20%20%20with%20gzip.open(_OUT%2C%20%22wb%22)%20as%20_f%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20reweight_cv.write_csv(_f)%0A%20%20%20%20%20%20%20%20print(f%22Wrote%20%7Breweight_cv.shape%5B0%5D%3A%2C%7D%20rows%20%E2%86%92%20%7B_OUT.name%7D%22)%0A%20%20%20%20return%20(reweight_cv%2C)%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20ChempropModel%2C%0A%20%20%20%20DR_TRAIN%2C%0A%20%20%20%20PRED_DIR%2C%0A%20%20%20%20compute_sample_weights%2C%0A%20%20%20%20generate_cv_splits_random%2C%0A%20%20%20%20gzip%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20tqdm%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Reweighting%20on%20a%20scratch%20Chemprop%20D-MPNN%20(slow%20%E2%80%94%20checkpointed%20per%20fold)%20%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20Chemprop%20is%20the%20one%20deep%20model%20that%20supports%20per-sample%20loss%20weights%20natively.%0A%20%20%20%20%23%20Only%20the%20non-trivial%20schemes%20are%20trained%20here%3B%20the%20%60uniform%60%20reference%20is%20the%0A%20%20%20%20%23%20scratch-chemprop%205%C3%975%20CV%20from%20notebook%202%20(identical%20splits%2C%20identical%0A%20%20%20%20%23%20architecture)%2C%20loaded%20in%20the%20summary%20cell%20below.%0A%20%20%20%20_OUT%20%3D%20PRED_DIR%20%2F%20%226_reweighting_chemprop_cv.csv.gz%22%0A%20%20%20%20_CKPT%20%3D%20_OUT.with_suffix(%22.ckpt.gz%22)%0A%20%20%20%20_SCHEMES%20%3D%20%5B%22invdensity%22%2C%20%22distance%22%5D%0A%0A%20%20%20%20if%20_OUT.exists()%3A%0A%20%20%20%20%20%20%20%20reweight_chemprop_cv%20%3D%20pl.read_csv(_OUT)%0A%20%20%20%20%20%20%20%20print(f%22Found%20%7B_OUT.name%7D%20%E2%80%94%20skipping%20Chemprop%20training%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22(%7Breweight_chemprop_cv.shape%5B0%5D%3A%2C%7D%20rows).%22)%0A%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20if%20_CKPT.exists()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_records%20%3D%20pl.read_csv(_CKPT).to_dicts()%0A%20%20%20%20%20%20%20%20%20%20%20%20_done%20%3D%20%7B(r%5B%22fold%22%5D%2C%20r%5B%22scheme%22%5D)%20for%20r%20in%20_records%7D%0A%20%20%20%20%20%20%20%20%20%20%20%20print(f%22Resuming%20Chemprop%20reweighting%20from%20checkpoint%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22(%7Blen(_records)%3A%2C%7D%20rows%2C%20%7Blen(_done)%7D%20fold%C3%97scheme%20done).%22)%0A%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_records%2C%20_done%20%3D%20%5B%5D%2C%20set()%0A%0A%20%20%20%20%20%20%20%20for%20_fold%2C%20_outer%2C%20_inner%2C%20_tr%2C%20_va%2C%20_te%20in%20tqdm(%0A%20%20%20%20%20%20%20%20%20%20%20%20generate_cv_splits_random(DR_TRAIN%2C%20n_outer%3D5%2C%20n_inner%3D5%2C%20seed%3D42%2C%20p_val%3D0.1)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20total%3D25%2C%20desc%3D%22Chemprop%20reweighting%20CV%22%2C%20unit%3D%22fold%22%2C%0A%20%20%20%20%20%20%20%20)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_tr%20%3D%20_tr%5B%22pEC50_dr%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_va%20%3D%20_va%5B%22pEC50_dr%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_te%20%3D%20_te%5B%22pEC50_dr%22%5D.to_numpy()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20_scheme%20in%20_SCHEMES%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20(_fold%2C%20_scheme)%20in%20_done%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20continue%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_w%20%3D%20compute_sample_weights(_y_tr%2C%20_scheme)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m%20%3D%20ChempropModel(pred_type%3D%22regression%22%2C%20epochs%3D50)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m.train(_tr%5B%22smiles%22%5D.to_list()%2C%20_y_tr%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_va%5B%22smiles%22%5D.to_list()%2C%20_y_va%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20target_col%3D%22pEC50_dr%22%2C%20sample_weight%3D_w)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_pred%20%3D%20_m.predict(_te%5B%22smiles%22%5D.to_list())%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20del%20_m%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20_ik%2C%20_yt%2C%20_yp%20in%20zip(_te%5B%22inchikey%22%5D.to_list()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_y_te.tolist()%2C%20_pred.tolist())%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_records.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22inchikey%22%3A%20_ik%2C%20%22fold%22%3A%20_fold%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22model%22%3A%20%22chemprop_scratch%22%2C%20%22scheme%22%3A%20_scheme%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22method%22%3A%20f%22chemprop_scratch_%7B_scheme%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22y_true%22%3A%20_yt%2C%20%22y_pred%22%3A%20_yp%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%23%20Checkpoint%20after%20every%20(fold%2C%20scheme)%20so%20the%20long%20run%20can%20resume.%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20with%20gzip.open(_CKPT%2C%20%22wb%22)%20as%20_f%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.DataFrame(_records).write_csv(_f)%0A%0A%20%20%20%20%20%20%20%20reweight_chemprop_cv%20%3D%20pl.DataFrame(_records)%0A%20%20%20%20%20%20%20%20with%20gzip.open(_OUT%2C%20%22wb%22)%20as%20_f%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20reweight_chemprop_cv.write_csv(_f)%0A%20%20%20%20%20%20%20%20_CKPT.unlink(missing_ok%3DTrue)%0A%20%20%20%20%20%20%20%20print(f%22Wrote%20%7Breweight_chemprop_cv.shape%5B0%5D%3A%2C%7D%20rows%20%E2%86%92%20%7B_OUT.name%7D%22)%0A%20%20%20%20return%20(reweight_chemprop_cv%2C)%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20BIN_ORDER%2C%0A%20%20%20%20PLOTS_DIR%2C%0A%20%20%20%20PRED_DIR%2C%0A%20%20%20%20bias_by_bin_table%2C%0A%20%20%20%20calc_regression_metrics%2C%0A%20%20%20%20mcs_plot%2C%0A%20%20%20%20mean_absolute_error%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20np%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20plt%2C%0A%20%20%20%20reweight_chemprop_cv%2C%0A%20%20%20%20reweight_cv%2C%0A%20%20%20%20rm_tukey_hsd%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Combine%20tree%20models%20(here)%20%2B%20Chemprop%20(here)%20%2B%20Chemprop%20uniform%20(nb%202)%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20The%20scratch-chemprop%20%60uniform%60%20baseline%20is%20taken%20from%20notebook%202's%205%C3%975%20CV%2C%0A%20%20%20%20%23%20which%20used%20identical%20folds%20(seed%2042)%20and%20an%20identical%20architecture%2C%20so%20it%20is%20a%0A%20%20%20%20%23%20fair%20reference%20for%20the%20reweighted%20chemprop%20runs.%0A%20%20%20%20_chemprop_uniform%20%3D%20(%0A%20%20%20%20%20%20%20%20pl.read_csv(PRED_DIR%20%2F%20%222_ml_baseline_5x5cv_random_predictions.csv.gz%22)%0A%20%20%20%20%20%20%20%20.filter(pl.col(%22model%22)%20%3D%3D%20%22chemprop%22)%0A%20%20%20%20%20%20%20%20.select(%5B%22inchikey%22%2C%20%22fold%22%2C%20%22y_true%22%2C%20%22y_pred%22%5D)%0A%20%20%20%20%20%20%20%20.with_columns(pl.lit(%22chemprop_scratch_uniform%22).alias(%22method%22))%0A%20%20%20%20)%0A%20%20%20%20_cols%20%3D%20%5B%22inchikey%22%2C%20%22fold%22%2C%20%22method%22%2C%20%22y_true%22%2C%20%22y_pred%22%5D%0A%20%20%20%20reweight_all%20%3D%20pl.concat(%5B%0A%20%20%20%20%20%20%20%20reweight_cv.select(_cols)%2C%0A%20%20%20%20%20%20%20%20reweight_chemprop_cv.select(_cols)%2C%0A%20%20%20%20%20%20%20%20_chemprop_uniform.select(_cols)%2C%0A%20%20%20%20%5D)%0A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Summary%20table%2C%20MCS%20and%20per-bin%20bias%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_rows%20%3D%20%5B%5D%0A%20%20%20%20for%20(_method%2C)%2C%20_grp%20in%20reweight_all.group_by(%5B%22method%22%5D)%3A%0A%20%20%20%20%20%20%20%20_yt%20%3D%20_grp%5B%22y_true%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_yp%20%3D%20_grp%5B%22y_pred%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_hit%20%3D%20_grp.filter(pl.col(%22y_true%22)%20%3E%3D%206.0)%0A%20%20%20%20%20%20%20%20_inact%20%3D%20_grp.filter(pl.col(%22y_true%22)%20%3C%204.0)%0A%20%20%20%20%20%20%20%20_rows.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22method%22%3A%20_method%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22MAE%22%3A%20round(mean_absolute_error(_yt%2C%20_yp)%2C%204)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22hitzone_MAE%22%3A%20round(mean_absolute_error(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_hit%5B%22y_true%22%5D.to_numpy()%2C%20_hit%5B%22y_pred%22%5D.to_numpy())%2C%203)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22hitzone_bias%22%3A%20round(float((_hit%5B%22y_pred%22%5D%20-%20_hit%5B%22y_true%22%5D).mean())%2C%203)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22inactive_bias%22%3A%20round(float((_inact%5B%22y_pred%22%5D%20-%20_inact%5B%22y_true%22%5D).mean())%2C%203)%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20reweight_summary%20%3D%20pl.DataFrame(_rows).sort(%22MAE%22)%0A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20MCS%20heatmaps%3A%20one%20subplot%20per%20model%2C%20scheme-only%20labels%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_MODEL_TITLES%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22xgb_mordred%22%3A%20%22XGBoost%20%C2%B7%20Mordred%22%2C%0A%20%20%20%20%20%20%20%20%22rf_chemeleon%22%3A%20%22RF%20%C2%B7%20CheMeleon%22%2C%0A%20%20%20%20%20%20%20%20%22chemprop_scratch%22%3A%20%22Chemprop%20%C2%B7%20scratch%22%2C%0A%20%20%20%20%7D%0A%20%20%20%20_SCHEMES%20%3D%20%5B%22uniform%22%2C%20%22distance%22%2C%20%22invdensity%22%5D%0A%0A%20%20%20%20_scheme_df%20%3D%20reweight_all.with_columns(%0A%20%20%20%20%20%20%20%20pl.col(%22method%22).str.extract(r%22(uniform%7Cinvdensity%7Cdistance)%24%22).alias(%22scheme%22)%2C%0A%20%20%20%20%20%20%20%20pl.col(%22method%22).str.replace(r%22_(uniform%7Cinvdensity%7Cdistance)%24%22%2C%20%22%22).alias(%22model%22)%2C%0A%20%20%20%20)%0A%0A%20%20%20%20fig_mcs%2C%20axes_mcs%20%3D%20plt.subplots(1%2C%203%2C%20figsize%3D(18%2C%205.5)%2C%20dpi%3D140)%0A%20%20%20%20for%20_ax%2C%20(_model_key%2C%20_title)%20in%20zip(axes_mcs%2C%20_MODEL_TITLES.items())%3A%0A%20%20%20%20%20%20%20%20_sub%20%3D%20_scheme_df.filter(pl.col(%22model%22)%20%3D%3D%20_model_key)%0A%20%20%20%20%20%20%20%20_met%20%3D%20calc_regression_metrics(%0A%20%20%20%20%20%20%20%20%20%20%20%20_sub.select(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22fold%22).alias(%22cv_cycle%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22scheme%22).alias(%22method%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.lit(%22random%22).alias(%22split%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22y_true%22%2C%20%22y_pred%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22cv_cycle%22%2C%20%22y_true%22%2C%20%22y_pred%22%2C%20thresh%3D4.0%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20_%2C%20_means%2C%20_diffs%2C%20_pc%20%3D%20rm_tukey_hsd(%0A%20%20%20%20%20%20%20%20%20%20%20%20_met%2C%20%22mae%22%2C%20%22method%22%2C%200.05%2C%20True%2C%20%7B%22mae%22%3A%20%22minimize%22%7D)%0A%20%20%20%20%20%20%20%20mcs_plot(_pc%2C%20effect_size%3D_diffs%2C%20means%3D_means%5B%22mae%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20show_diff%3DTrue%2C%20ax%3D_ax%2C%20cbar%3DTrue%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20cell_text_size%3D14%2C%20axis_text_size%3D11%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20reverse_cmap%3DTrue%2C%20vlim%3D0.03)%0A%20%20%20%20%20%20%20%20_ax.set_title(_title%2C%20fontsize%3D13)%0A%20%20%20%20fig_mcs.suptitle(%22MAE%20%E2%80%94%20Tukey%20HSD%20per%20model%20(reweighting%20schemes)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D14%2C%20y%3D1.02)%0A%20%20%20%20fig_mcs.tight_layout()%0A%20%20%20%20fig_mcs.savefig(PLOTS_DIR%20%2F%20%22analysis2_reweighting_mcs_mae.png%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Per-bin%20signed-error%20heatmaps%3A%20one%20subplot%20per%20model%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_bias_all%20%3D%20bias_by_bin_table(_scheme_df%2C%20%22method%22%2C%20%22y_true%22%2C%20%22y_pred%22)%0A%20%20%20%20_bias_all%20%3D%20_bias_all.with_columns(%0A%20%20%20%20%20%20%20%20pl.col(%22method%22).str.extract(r%22(uniform%7Cinvdensity%7Cdistance)%24%22).alias(%22scheme%22)%2C%0A%20%20%20%20%20%20%20%20pl.col(%22method%22).str.replace(r%22_(uniform%7Cinvdensity%7Cdistance)%24%22%2C%20%22%22).alias(%22model%22)%2C%0A%20%20%20%20)%0A%0A%20%20%20%20fig_bias%2C%20axes_bias%20%3D%20plt.subplots(%0A%20%20%20%20%20%20%20%201%2C%203%2C%20figsize%3D(18%2C%203.8)%2C%20dpi%3D150)%0A%20%20%20%20for%20_ax%2C%20(_model_key%2C%20_title)%20in%20zip(axes_bias%2C%20_MODEL_TITLES.items())%3A%0A%20%20%20%20%20%20%20%20_sub%20%3D%20_bias_all.filter(pl.col(%22model%22)%20%3D%3D%20_model_key)%0A%20%20%20%20%20%20%20%20_matrix%20%3D%20np.full((len(_SCHEMES)%2C%20len(BIN_ORDER))%2C%20np.nan)%0A%20%20%20%20%20%20%20%20for%20_row%20in%20_sub.iter_rows(named%3DTrue)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20_row%5B%22scheme%22%5D%20in%20_SCHEMES%20and%20_row%5B%22pec50_bin%22%5D%20in%20BIN_ORDER%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_mi%20%3D%20_SCHEMES.index(_row%5B%22scheme%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_bi%20%3D%20BIN_ORDER.index(_row%5B%22pec50_bin%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_matrix%5B_mi%2C%20_bi%5D%20%3D%20_row%5B%22mean_error%22%5D%0A%20%20%20%20%20%20%20%20_abs_max%20%3D%20float(np.nanmax(np.abs(_matrix)))%0A%20%20%20%20%20%20%20%20_ax.grid(False)%0A%20%20%20%20%20%20%20%20_im%20%3D%20_ax.imshow(_matrix%2C%20cmap%3D%22RdBu_r%22%2C%20vmin%3D-_abs_max%2C%20vmax%3D_abs_max%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20aspect%3D%22auto%22%2C%20interpolation%3D%22nearest%22)%0A%20%20%20%20%20%20%20%20for%20_mi%20in%20range(len(_SCHEMES))%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20_bi%20in%20range(len(BIN_ORDER))%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_val%20%3D%20_matrix%5B_mi%2C%20_bi%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20not%20np.isnan(_val)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_col%20%3D%20%22white%22%20if%20abs(_val)%20%3E%200.6%20*%20_abs_max%20else%20%22black%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_ax.text(_bi%2C%20_mi%2C%20f%22%7B_val%3A%2B.3f%7D%22%2C%20ha%3D%22center%22%2C%20va%3D%22center%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D9%2C%20color%3D_col)%0A%20%20%20%20%20%20%20%20_ax.set_xticks(range(len(BIN_ORDER)))%0A%20%20%20%20%20%20%20%20_ax.set_xticklabels(BIN_ORDER%2C%20fontsize%3D8%2C%20rotation%3D-20%2C%20ha%3D%22left%22)%0A%20%20%20%20%20%20%20%20_ax.set_yticks(range(len(_SCHEMES)))%0A%20%20%20%20%20%20%20%20if%20_ax%20is%20axes_bias%5B0%5D%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.set_yticklabels(_SCHEMES%2C%20fontsize%3D10)%0A%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.set_yticklabels(%5B%5D)%0A%20%20%20%20%20%20%20%20_ax.set_title(_title%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20fig_bias.colorbar(_im%2C%20ax%3D_ax%2C%20fraction%3D0.04%2C%20pad%3D0.03).set_label(%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Mean%20error%20(pred%20%E2%88%92%20true)%22%2C%20fontsize%3D8)%0A%20%20%20%20fig_bias.suptitle(%0A%20%20%20%20%20%20%20%20%22Reweighting%20%E2%80%94%20mean%20signed%20error%20by%20activity%20bin%5Cn%22%0A%20%20%20%20%20%20%20%20%22(red%20%3D%20overpredict%2C%20blue%20%3D%20underpredict)%22%2C%0A%20%20%20%20%20%20%20%20fontsize%3D12%2C%20y%3D1.06)%0A%20%20%20%20fig_bias.tight_layout()%0A%20%20%20%20fig_bias.savefig(PLOTS_DIR%20%2F%20%22analysis2_reweighting_bias_heatmap.png%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20mo.vstack(%5B%0A%20%20%20%20%20%20%20%20mo.md(%22%23%23%23%20Reweighting%20results%20%E2%80%94%20XGBoost%C2%B7Mordred%2C%20RF%C2%B7CheMeleon%2C%20Chemprop%C2%B7scratch%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Inverse-density%20and%20distance%20weighting%20trade%20overall%20MAE%20for%20reduced%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22bias%20at%20the%20extremes%20%E2%80%94%20watch%20the%20%60%3E6%60%20and%20%60%3C4%60%20columns%20versus%20%60uniform%60.%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Chemprop%20is%20the%20one%20deep%20model%20that%20supports%20loss%20weighting%3B%20its%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22%60uniform%60%20row%20is%20notebook%202's%20baseline%20(same%20folds%2C%20same%20architecture).%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Chemprop%20is%20stochastic%2C%20so%20small%20differences%20may%20reflect%20seed%20noise.%22)%2C%0A%20%20%20%20%20%20%20%20mo.ui.table(reweight_summary.to_pandas()%2C%20selection%3DNone%2C%20pagination%3DFalse)%2C%0A%20%20%20%20%20%20%20%20mo.as_html(fig_mcs)%2C%0A%20%20%20%20%20%20%20%20mo.as_html(fig_bias)%2C%0A%20%20%20%20%5D)%0A%20%20%20%20return%20(reweight_summary%2C)%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%20Analysis%203%20%E2%80%94%20Oversampling%20the%20activity%20extremes%0A%0A%20%20%20%20Reweighting%20(Analysis%202)%20only%20works%20for%20models%20with%20a%20per-sample%20loss%20weight.%0A%20%20%20%20**Oversampling**%20achieves%20the%20same%20goal%20at%20the%20*data*%20level%20%E2%80%94%20duplicating%20the%0A%20%20%20%20rare%20extreme%20compounds%20in%20the%20training%20set%20%E2%80%94%20and%20therefore%20works%20for%20**every**%0A%20%20%20%20model%2C%20including%20TabPFN%20and%20Macau%20which%20have%20no%20loss%20weight%20to%20set.%0A%0A%20%20%20%20The%20extremes%20are%20the%20**active**%20tail%20(pEC50%20%3E%205.5%2C%20%E2%89%88%209%20%25%20of%20the%20set)%20and%20the%0A%20%20%20%20**inactive**%20tail%20(pEC50%20%3C%203.5%2C%20%E2%89%88%2023%20%25)%3B%20the%20dense%20middle%20(3.5%E2%80%935.5)%20is%20left%0A%20%20%20%20untouched.%20Each%20extreme%20compound%20in%20the%20*training*%20fold%20is%20replicated%20%C3%97k%3A%0A%0A%20%20%20%20%7C%20k%20%7C%20extreme%20share%20of%20training%20set%20%7C%0A%20%20%20%20%7C---%7C---%7C%0A%20%20%20%20%7C%201%20(baseline)%20%7C%2032%20%25%20%7C%0A%20%20%20%20%7C%202%20%7C%2049%20%25%20%7C%0A%20%20%20%20%7C%203%20%7C%2059%20%25%20%7C%0A%20%20%20%20%7C%205%20%7C%2071%20%25%20%7C%0A%0A%20%20%20%20Four%20models%20are%20tested%20%E2%80%94%20**XGBoost%C2%B7Mordred%2C%20Macau%C2%B7CheMeleon%2C%20TabPFN%C2%B7CheMeleon%2C%0A%20%20%20%20Chemprop%C2%B7scratch**.%20To%20save%20compute%2C%20the%20**1%C3%97%20baseline%20is%20reused**%20from%20earlier%0A%20%20%20%20config-matched%205%C3%975%20CV%20runs%20(XGBoost%20from%20Analysis%202%2C%20Macau%2FTabPFN%20from%20notebook%0A%20%20%20%204%2C%20Chemprop%20from%20notebook%202)%3B%20only%20k%20%3D%202%2C%203%2C%205%20are%20trained%20here.%20Oversampling%20is%0A%20%20%20%20applied%20to%20the%20training%20split%20only%2C%20so%20the%20test%20folds%20are%20unchanged%20and%20every%0A%20%20%20%20comparison%20stays%20on%20the%20same%20dose-response%20compounds.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(np)%3A%0A%20%20%20%20def%20oversample_extremes(%0A%20%20%20%20%20%20%20%20y%3A%20np.ndarray%2C%20factor%3A%20int%2C%20hi%3A%20float%20%3D%205.5%2C%20lo%3A%20float%20%3D%203.5%2C%0A%20%20%20%20)%20-%3E%20np.ndarray%3A%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20Row%20indices%20that%20replicate%20each%20extreme-activity%20compound%20%60factor%60%20times.%0A%0A%20%20%20%20%20%20%20%20Args%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3A%20Training%20labels%20(pEC50).%0A%20%20%20%20%20%20%20%20%20%20%20%20factor%3A%20Replication%20count%20for%20extreme%20rows%20(1%20%3D%20no%20oversampling).%0A%20%20%20%20%20%20%20%20%20%20%20%20hi%3A%20Active-tail%20threshold%20(compounds%20with%20y%20%3E%20hi%20are%20extreme).%0A%20%20%20%20%20%20%20%20%20%20%20%20lo%3A%20Inactive-tail%20threshold%20(compounds%20with%20y%20%3C%20lo%20are%20extreme).%0A%0A%20%20%20%20%20%20%20%20Returns%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20Index%20array%20into%20the%20training%20arrays%3A%20all%20rows%20once%2C%20plus%20(factor%20%E2%88%92%201)%0A%20%20%20%20%20%20%20%20%20%20%20%20extra%20copies%20of%20the%20extreme%20rows.%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20base%20%3D%20np.arange(len(y))%0A%20%20%20%20%20%20%20%20if%20factor%20%3C%3D%201%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20base%0A%20%20%20%20%20%20%20%20extreme%20%3D%20np.where((y%20%3E%20hi)%20%7C%20(y%20%3C%20lo))%5B0%5D%0A%20%20%20%20%20%20%20%20extra%20%3D%20np.repeat(extreme%2C%20factor%20-%201)%0A%20%20%20%20%20%20%20%20return%20np.concatenate(%5Bbase%2C%20extra%5D)%0A%0A%20%20%20%20return%20(oversample_extremes%2C)%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20BoostedTreesModel%2C%0A%20%20%20%20ChempropModel%2C%0A%20%20%20%20DR_TRAIN%2C%0A%20%20%20%20MacauModel%2C%0A%20%20%20%20PRED_DIR%2C%0A%20%20%20%20gather_X%2C%0A%20%20%20%20gc%2C%0A%20%20%20%20generate_cv_splits_random%2C%0A%20%20%20%20gzip%2C%0A%20%20%20%20oversample_extremes%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20tabpfn_predict%2C%0A%20%20%20%20tqdm%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%205%C3%975%20CV%3A%20oversample%20the%20extremes%20%C3%97%7B2%2C3%2C5%7D%20for%20four%20model%20families%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20The%201%C3%97%20baseline%20is%20reused%20in%20the%20summary%20cell%2C%20so%20only%20k%3E1%20is%20trained%20here.%0A%20%20%20%20%23%20The%20slow%20models%20are%20Chemprop%20(CLI)%20and%20TabPFN%20(CPU%20subprocess)%3B%20the%20loop%20is%0A%20%20%20%20%23%20checkpointed%20per%20(fold%2C%20model%2C%20factor)%20so%20it%20survives%20interruptions.%0A%20%20%20%20_OUT%20%3D%20PRED_DIR%20%2F%20%226_oversampling_cv.csv.gz%22%0A%20%20%20%20_CKPT%20%3D%20_OUT.with_suffix(%22.ckpt.gz%22)%0A%20%20%20%20_FACTORS%20%3D%20%5B2%2C%203%2C%205%5D%0A%0A%20%20%20%20if%20_OUT.exists()%3A%0A%20%20%20%20%20%20%20%20oversampling_cv%20%3D%20pl.read_csv(_OUT)%0A%20%20%20%20%20%20%20%20print(f%22Found%20%7B_OUT.name%7D%20%E2%80%94%20skipping%20(%7Boversampling_cv.shape%5B0%5D%3A%2C%7D%20rows).%22)%0A%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20if%20_CKPT.exists()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_records%20%3D%20pl.read_csv(_CKPT).to_dicts()%0A%20%20%20%20%20%20%20%20%20%20%20%20_done%20%3D%20%7B(r%5B%22fold%22%5D%2C%20r%5B%22method%22%5D)%20for%20r%20in%20_records%7D%0A%20%20%20%20%20%20%20%20%20%20%20%20print(f%22Resuming%20oversampling%20CV%20from%20checkpoint%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22(%7Blen(_records)%3A%2C%7D%20rows%2C%20%7Blen(_done)%7D%20fold%C3%97method%20done).%22)%0A%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_records%2C%20_done%20%3D%20%5B%5D%2C%20set()%0A%0A%20%20%20%20%20%20%20%20for%20_fold%2C%20_outer%2C%20_inner%2C%20_tr%2C%20_va%2C%20_te%20in%20tqdm(%0A%20%20%20%20%20%20%20%20%20%20%20%20generate_cv_splits_random(DR_TRAIN%2C%20n_outer%3D5%2C%20n_inner%3D5%2C%20seed%3D42%2C%20p_val%3D0.1)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20total%3D25%2C%20desc%3D%22oversampling%20CV%22%2C%20unit%3D%22fold%22%2C%0A%20%20%20%20%20%20%20%20)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_tr%20%3D%20_tr%5B%22pEC50_dr%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_va%20%3D%20_va%5B%22pEC50_dr%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_te%20%3D%20_te%5B%22pEC50_dr%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_te_iks%20%3D%20_te%5B%22inchikey%22%5D.to_list()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20Features%20gathered%20once%20per%20fold%20(rows%20duplicated%20per%20factor%20via%20_idx).%0A%20%20%20%20%20%20%20%20%20%20%20%20_Xtr_mor%20%3D%20gather_X(_tr%5B%22inchikey%22%5D.to_list()%2C%20%22mordred%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_Xva_mor%20%3D%20gather_X(_va%5B%22inchikey%22%5D.to_list()%2C%20%22mordred%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_Xte_mor%20%3D%20gather_X(_te_iks%2C%20%22mordred%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_Xtr_che%20%3D%20gather_X(_tr%5B%22inchikey%22%5D.to_list()%2C%20%22chemeleon%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_Xte_che%20%3D%20gather_X(_te_iks%2C%20%22chemeleon%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_smi_tr%20%3D%20_tr%5B%22smiles%22%5D.to_list()%0A%20%20%20%20%20%20%20%20%20%20%20%20_smi_va%20%3D%20_va%5B%22smiles%22%5D.to_list()%0A%20%20%20%20%20%20%20%20%20%20%20%20_smi_te%20%3D%20_te%5B%22smiles%22%5D.to_list()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20def%20_save(model_key%3A%20str%2C%20factor%3A%20int%2C%20preds)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20_ik%2C%20_yt%2C%20_yp%20in%20zip(_te_iks%2C%20_y_te.tolist()%2C%20list(preds))%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_records.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22inchikey%22%3A%20_ik%2C%20%22fold%22%3A%20_fold%2C%20%22model%22%3A%20model_key%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22factor%22%3A%20factor%2C%20%22method%22%3A%20f%22%7Bmodel_key%7D_os%7Bfactor%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22y_true%22%3A%20_yt%2C%20%22y_pred%22%3A%20float(_yp)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20with%20gzip.open(_CKPT%2C%20%22wb%22)%20as%20_f%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.DataFrame(_records).write_csv(_f)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20_factor%20in%20_FACTORS%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_idx%20%3D%20oversample_extremes(_y_tr%2C%20_factor)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_yo%20%3D%20_y_tr%5B_idx%5D%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20(_fold%2C%20f%22xgb_mordred_os%7B_factor%7D%22)%20not%20in%20_done%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m%20%3D%20BoostedTreesModel(pred_type%3D%22regression%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m.train(_Xtr_mor%5B_idx%5D%2C%20_yo%2C%20_Xva_mor%2C%20_y_va)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_save(%22xgb_mordred%22%2C%20_factor%2C%20_m.predict(_Xte_mor))%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20del%20_m%3B%20gc.collect()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20(_fold%2C%20f%22macau_chemeleon_os%7B_factor%7D%22)%20not%20in%20_done%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m%20%3D%20MacauModel(seed%3D42)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m.train(_Xtr_che%5B_idx%5D%2C%20_yo)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_save(%22macau_chemeleon%22%2C%20_factor%2C%20_m.predict(_Xte_che))%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20del%20_m%3B%20gc.collect()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20(_fold%2C%20f%22tabpfn_chemeleon_os%7B_factor%7D%22)%20not%20in%20_done%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_save(%22tabpfn_chemeleon%22%2C%20_factor%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20tabpfn_predict(_Xtr_che%5B_idx%5D%2C%20_yo%2C%20_Xte_che))%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20(_fold%2C%20f%22chemprop_scratch_os%7B_factor%7D%22)%20not%20in%20_done%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m%20%3D%20ChempropModel(pred_type%3D%22regression%22%2C%20epochs%3D50)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m.train(%5B_smi_tr%5Bi%5D%20for%20i%20in%20_idx%5D%2C%20_yo%2C%20_smi_va%2C%20_y_va%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20target_col%3D%22pEC50_dr%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_save(%22chemprop_scratch%22%2C%20_factor%2C%20_m.predict(_smi_te))%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20del%20_m%3B%20gc.collect()%0A%0A%20%20%20%20%20%20%20%20oversampling_cv%20%3D%20pl.DataFrame(_records)%0A%20%20%20%20%20%20%20%20with%20gzip.open(_OUT%2C%20%22wb%22)%20as%20_f%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20oversampling_cv.write_csv(_f)%0A%20%20%20%20%20%20%20%20_CKPT.unlink(missing_ok%3DTrue)%0A%20%20%20%20%20%20%20%20print(f%22Wrote%20%7Boversampling_cv.shape%5B0%5D%3A%2C%7D%20rows%20%E2%86%92%20%7B_OUT.name%7D%22)%0A%20%20%20%20return%20(oversampling_cv%2C)%0A%0A%0A%40app.cell%0Adef%20_(PLOTS_DIR%2C%20PRED_DIR%2C%20mean_absolute_error%2C%20mo%2C%20oversampling_cv%2C%20pl%2C%20plt)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Combine%20reused%201%C3%97%20baselines%20with%20the%20oversampled%20runs%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20Note%3A%20notebook%202's%20file%20labels%20models%20in%20a%20%60model%60%20column%3B%20the%20others%20use%0A%20%20%20%20%23%20%60method%60%20%E2%80%94%20so%20the%20filter%20column%20is%20passed%20explicitly.%0A%20%20%20%20def%20_baseline(path%3A%20str%2C%20col%3A%20str%2C%20val%3A%20str%2C%20model_key%3A%20str)%20-%3E%20pl.DataFrame%3A%0A%20%20%20%20%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.read_csv(PRED_DIR%20%2F%20path)%0A%20%20%20%20%20%20%20%20%20%20%20%20.filter(pl.col(col)%20%3D%3D%20val)%0A%20%20%20%20%20%20%20%20%20%20%20%20.select(%5B%22fold%22%2C%20%22y_true%22%2C%20%22y_pred%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20.with_columns(pl.lit(model_key).alias(%22model%22)%2C%20pl.lit(1).cast(pl.Int64).alias(%22factor%22))%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20_base%20%3D%20pl.concat(%5B%0A%20%20%20%20%20%20%20%20_baseline(%226_reweighting_cv.csv.gz%22%2C%20%22method%22%2C%20%22xgb_mordred_uniform%22%2C%20%22xgb_mordred%22)%2C%0A%20%20%20%20%20%20%20%20_baseline(%224_fp_model_comparison_1.csv.gz%22%2C%20%22method%22%2C%20%22macau_chemeleon%22%2C%20%22macau_chemeleon%22)%2C%0A%20%20%20%20%20%20%20%20_baseline(%224_fp_model_comparison_2.csv.gz%22%2C%20%22method%22%2C%20%22tabpfn_chemeleon%22%2C%20%22tabpfn_chemeleon%22)%2C%0A%20%20%20%20%20%20%20%20_baseline(%222_ml_baseline_5x5cv_random_predictions.csv.gz%22%2C%20%22model%22%2C%20%22chemprop%22%2C%20%22chemprop_scratch%22)%2C%0A%20%20%20%20%5D)%0A%20%20%20%20_cols%20%3D%20%5B%22fold%22%2C%20%22model%22%2C%20%22factor%22%2C%20%22y_true%22%2C%20%22y_pred%22%5D%0A%20%20%20%20os_all%20%3D%20pl.concat(%5B%0A%20%20%20%20%20%20%20%20_base.select(_cols)%2C%0A%20%20%20%20%20%20%20%20oversampling_cv.select(_cols)%2C%0A%20%20%20%20%5D).with_columns(pl.format(%22%7B%7D_os%7B%7D%22%2C%20pl.col(%22model%22)%2C%20pl.col(%22factor%22)).alias(%22method%22))%0A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Pooled%20metrics%20per%20(model%2C%20factor)%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_rows%20%3D%20%5B%5D%0A%20%20%20%20for%20(_model%2C%20_factor)%2C%20_grp%20in%20os_all.group_by(%5B%22model%22%2C%20%22factor%22%5D)%3A%0A%20%20%20%20%20%20%20%20_yt%20%3D%20_grp%5B%22y_true%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_yp%20%3D%20_grp%5B%22y_pred%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_hit%20%3D%20_grp.filter(pl.col(%22y_true%22)%20%3E%205.5)%0A%20%20%20%20%20%20%20%20_inact%20%3D%20_grp.filter(pl.col(%22y_true%22)%20%3C%203.5)%0A%20%20%20%20%20%20%20%20_rows.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22model%22%3A%20_model%2C%20%22factor%22%3A%20_factor%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22method%22%3A%20f%22%7B_model%7D_os%7B_factor%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22MAE%22%3A%20round(mean_absolute_error(_yt%2C%20_yp)%2C%204)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22hitzone_bias%22%3A%20round(float((_hit%5B%22y_pred%22%5D%20-%20_hit%5B%22y_true%22%5D).mean())%2C%203)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22inactive_bias%22%3A%20round(float((_inact%5B%22y_pred%22%5D%20-%20_inact%5B%22y_true%22%5D).mean())%2C%203)%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20oversampling_summary%20%3D%20pl.DataFrame(_rows).sort(%5B%22model%22%2C%20%22factor%22%5D)%0A%0A%20%20%20%20%23%20Per-fold%20MAE%20(mean%20%C2%B1%20std%20across%20folds)%20for%20the%20MAE-vs-factor%20curve.%0A%20%20%20%20_per_fold%20%3D%20%5B%5D%0A%20%20%20%20for%20(_model%2C%20_factor%2C%20_fold)%2C%20_grp%20in%20os_all.group_by(%5B%22model%22%2C%20%22factor%22%2C%20%22fold%22%5D)%3A%0A%20%20%20%20%20%20%20%20_per_fold.append(%7B%22model%22%3A%20_model%2C%20%22factor%22%3A%20_factor%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22mae%22%3A%20mean_absolute_error(_grp%5B%22y_true%22%5D.to_numpy()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_grp%5B%22y_pred%22%5D.to_numpy())%7D)%0A%20%20%20%20_pf%20%3D%20(%0A%20%20%20%20%20%20%20%20pl.DataFrame(_per_fold)%0A%20%20%20%20%20%20%20%20.group_by(%5B%22model%22%2C%20%22factor%22%5D)%0A%20%20%20%20%20%20%20%20.agg(pl.col(%22mae%22).mean().alias(%22mae_mean%22)%2C%20pl.col(%22mae%22).std().alias(%22mae_std%22))%0A%20%20%20%20)%0A%0A%20%20%20%20_MODELS%20%3D%20%5B%22xgb_mordred%22%2C%20%22macau_chemeleon%22%2C%20%22tabpfn_chemeleon%22%2C%20%22chemprop_scratch%22%5D%0A%20%20%20%20_COLORS%20%3D%20%7B%22xgb_mordred%22%3A%20%22%234e79a7%22%2C%20%22macau_chemeleon%22%3A%20%22%23e15759%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22tabpfn_chemeleon%22%3A%20%22%2359a14f%22%2C%20%22chemprop_scratch%22%3A%20%22%23b07aa1%22%7D%0A%0A%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20_fig%2C%20(_ax1%2C%20_ax2)%20%3D%20plt.subplots(1%2C%202%2C%20figsize%3D(13%2C%205)%2C%20dpi%3D140)%0A%20%20%20%20%20%20%20%20for%20_model%20in%20_MODELS%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_s%20%3D%20_pf.filter(pl.col(%22model%22)%20%3D%3D%20_model).sort(%22factor%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax1.errorbar(_s%5B%22factor%22%5D.to_numpy()%2C%20_s%5B%22mae_mean%22%5D.to_numpy()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20yerr%3D_s%5B%22mae_std%22%5D.to_numpy()%2C%20marker%3D%22o%22%2C%20capsize%3D3%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20color%3D_COLORS%5B_model%5D%2C%20label%3D_model)%0A%20%20%20%20%20%20%20%20%20%20%20%20_b%20%3D%20oversampling_summary.filter(pl.col(%22model%22)%20%3D%3D%20_model).sort(%22factor%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax2.plot(_b%5B%22factor%22%5D.to_numpy()%2C%20_b%5B%22hitzone_bias%22%5D.to_numpy()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20marker%3D%22o%22%2C%20color%3D_COLORS%5B_model%5D%2C%20label%3Df%22%7B_model%7D%20(hit)%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax2.plot(_b%5B%22factor%22%5D.to_numpy()%2C%20_b%5B%22inactive_bias%22%5D.to_numpy()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20marker%3D%22s%22%2C%20linestyle%3D%22--%22%2C%20color%3D_COLORS%5B_model%5D%2C%20alpha%3D0.6)%0A%20%20%20%20%20%20%20%20_ax1.set_xlabel(%22oversampling%20factor%20k%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax1.set_ylabel(%22CV%20MAE%20(mean%20%C2%B1%20std%20over%20folds)%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax1.set_title(%22Overall%20MAE%20vs%20oversampling%22%2C%20fontsize%3D12)%0A%20%20%20%20%20%20%20%20_ax1.set_xticks(%5B1%2C%202%2C%203%2C%205%5D)%0A%20%20%20%20%20%20%20%20_ax1.legend(fontsize%3D9)%0A%20%20%20%20%20%20%20%20_ax2.axhline(0%2C%20color%3D%22black%22%2C%20linewidth%3D0.9%2C%20linestyle%3D%22--%22%2C%20zorder%3D0)%0A%20%20%20%20%20%20%20%20_ax2.set_xlabel(%22oversampling%20factor%20k%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax2.set_ylabel(%22signed%20bias%20(pred%20%E2%88%92%20true)%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax2.set_title(%22Extreme-zone%20bias%20vs%20oversampling%5Cn(%E2%97%8B%20hit%20%3E%205.5%2C%20%E2%96%A1%20inactive%20%3C%203.5)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D12)%0A%20%20%20%20%20%20%20%20_ax2.set_xticks(%5B1%2C%202%2C%203%2C%205%5D)%0A%20%20%20%20%20%20%20%20_ax2.legend(fontsize%3D8%2C%20ncol%3D2)%0A%20%20%20%20%20%20%20%20_fig.tight_layout()%0A%20%20%20%20%20%20%20%20_fig.savefig(PLOTS_DIR%20%2F%20%22analysis3_oversampling_curves.png%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20mo.vstack(%5B%0A%20%20%20%20%20%20%20%20mo.md(%22%23%23%23%20Oversampling%20results%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Left%3A%20does%20oversampling%20lower%20overall%20MAE%3F%20Right%3A%20does%20it%20pull%20the%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22hit-zone%20bias%20(%E2%97%8B)%20up%20toward%200%20and%20the%20inactive%20bias%20(%E2%96%A1)%20down%20toward%200%3F%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22k%20%3D%201%20is%20the%20reused%20baseline.%22)%2C%0A%20%20%20%20%20%20%20%20mo.ui.table(oversampling_summary.to_pandas()%2C%20selection%3DNone%2C%20pagination%3DFalse)%2C%0A%20%20%20%20%20%20%20%20mo.as_html(_fig)%2C%0A%20%20%20%20%5D)%0A%20%20%20%20return%20(oversampling_summary%2C)%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20While%20oversampling%20reduces%20bias%20in%20the%20extremes%20of%20the%20distribution%20in%20a%20more%20impactful%20way%20than%20previous%20methods%2C%20in%20general%20all%20models%20perform%20much%20worse.%20tabPFN%20see%20some%20of%20the%20biggest%20decreases%20in%20performance%20and%20also%20see%20an%20increase%20in%20the%20bias%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%20Analysis%204%20%E2%80%94%20Extra%20assay%20data%20(96-compound%20semi-pure%20set)%0A%0A%20%20%20%20OpenADMET%20released%20a%20small%20but%20high-quality%20dataset%3A%2096%20compounds%20re-measured%0A%20%20%20%20from%20**semi-pure**%20microscale%20synthesis%20with%20purity-corrected%20pEC50%20values%0A%20%20%20%20(%E2%89%88%2094%20usable%2C%20spanning%20pEC50%204.0%E2%80%937.0%20%E2%80%94%20exactly%20the%20moderate-to-hit%20range%20where%0A%20%20%20%20the%20models%20struggle).%20Only%20**5%20of%20these%20overlap%20the%20dose-response%20training%20set**%2C%0A%20%20%20%20so%20they%20are%20almost%20entirely%20new%20measurements.%0A%0A%20%20%20%20Two%20questions%3A%0A%0A%20%20%20%201.%20**Label-noise%20audit**%20%E2%80%94%20for%20the%20handful%20of%20overlapping%20compounds%2C%20how%20far%20is%0A%20%20%20%20%20%20%20the%20original%20training%20pEC50%20from%20the%20corrected%20semi-pure%20value%3F%20This%20bounds%0A%20%20%20%20%20%20%20how%20much%20of%20the%20model%20error%20is%20irreducible%20label%20noise.%0A%20%20%20%202.%20**Augmentation**%20%E2%80%94%20does%20adding%20these%20compounds%20to%20the%20training%20fold%20improve%0A%20%20%20%20%20%20%205%C3%975%20CV%20on%20the%20dose-response%20set%3F%20The%20semi-pure%20compounds%20are%20added%20to%20the%0A%20%20%20%20%20%20%20*training*%20split%20only%20(and%20never%20when%20they%20collide%20with%20a%20test-fold%20compound)%2C%0A%20%20%20%20%20%20%20so%20evaluation%20stays%20on%20the%20dose-response%20compounds%20and%20folds%20stay%20comparable.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(DR_TRAIN%2C%20PLOTS_DIR%2C%20mo%2C%20pl%2C%20plt%2C%20semipure)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Label-noise%20audit%20on%20the%20overlap%20with%20the%20DR%20training%20set%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_overlap%20%3D%20(%0A%20%20%20%20%20%20%20%20semipure.join(%0A%20%20%20%20%20%20%20%20%20%20%20%20DR_TRAIN.select(%5B%22inchikey%22%2C%20%22pEC50_dr%22%2C%20%22molecule_names%22%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20on%3D%22inchikey%22%2C%20how%3D%22inner%22)%0A%20%20%20%20%20%20%20%20.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22pEC50_corrected%22)%20-%20pl.col(%22pEC50_dr%22)).alias(%22delta%22))%0A%20%20%20%20%20%20%20%20.sort(pl.col(%22delta%22).abs()%2C%20descending%3DTrue)%0A%20%20%20%20)%0A%0A%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20_fig%2C%20_ax%20%3D%20plt.subplots(figsize%3D(5.5%2C%205.5)%2C%20dpi%3D150)%0A%20%20%20%20%20%20%20%20_x%20%3D%20_overlap%5B%22pEC50_dr%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_y%20%3D%20_overlap%5B%22pEC50_corrected%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_ax.scatter(_x%2C%20_y%2C%20s%3D70%2C%20color%3D%22%234e79a7%22%2C%20edgecolors%3D%22white%22%2C%20zorder%3D3)%0A%20%20%20%20%20%20%20%20_lims%20%3D%20%5Bmin(_x.min()%2C%20_y.min())%20-%200.3%2C%20max(_x.max()%2C%20_y.max())%20%2B%200.3%5D%0A%20%20%20%20%20%20%20%20_ax.plot(_lims%2C%20_lims%2C%20%22k--%22%2C%20linewidth%3D1%2C%20zorder%3D0%2C%20label%3D%22identity%22)%0A%20%20%20%20%20%20%20%20for%20_r%20in%20_overlap.iter_rows(named%3DTrue)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.annotate(f%22%7B_r%5B'delta'%5D%3A%2B.2f%7D%22%2C%20(_r%5B%22pEC50_dr%22%5D%2C%20_r%5B%22pEC50_corrected%22%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D8%2C%20xytext%3D(4%2C%204)%2C%20textcoords%3D%22offset%20points%22)%0A%20%20%20%20%20%20%20%20_ax.set_xlim(_lims)%3B%20_ax.set_ylim(_lims)%0A%20%20%20%20%20%20%20%20_ax.set_xlabel(%22Original%20training%20pEC50%20(pEC50_dr)%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax.set_ylabel(%22Corrected%20semi-pure%20pEC50%22%2C%20fontsize%3D11)%0A%20%20%20%20%20%20%20%20_ax.set_title(f%22Label-noise%20audit%20%E2%80%94%20%7B_overlap.shape%5B0%5D%7D%20overlapping%20compounds%22%2C%20fontsize%3D12)%0A%20%20%20%20%20%20%20%20_ax.legend(fontsize%3D10)%0A%20%20%20%20%20%20%20%20_fig.tight_layout()%0A%20%20%20%20%20%20%20%20_fig.savefig(PLOTS_DIR%20%2F%20%22analysis4_label_noise.png%22%2C%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20_mad%20%3D%20float(_overlap%5B%22delta%22%5D.abs().mean())%20if%20_overlap.shape%5B0%5D%20else%20float(%22nan%22)%0A%20%20%20%20mo.vstack(%5B%0A%20%20%20%20%20%20%20%20mo.md(f%22%23%23%23%20Label-noise%20audit%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22Mean%20absolute%20discrepancy%20between%20original%20and%20corrected%20pEC50%3A%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22**%7B_mad%3A.2f%7D%20log%20units**%20across%20%7B_overlap.shape%5B0%5D%7D%20overlapping%20compounds.%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22For%20reference%2C%20the%20best%20ensemble%20CV%20MAE%20is%20%E2%89%88%200.47%20%E2%80%94%20so%20part%20of%20the%20model%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22error%20is%20simply%20measurement%20noise%20in%20the%20labels.%22)%2C%0A%20%20%20%20%20%20%20%20mo.as_html(_fig)%2C%0A%20%20%20%20%5D)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20BoostedTreesModel%2C%0A%20%20%20%20DR_TRAIN%2C%0A%20%20%20%20PRED_DIR%2C%0A%20%20%20%20RandomForestModel%2C%0A%20%20%20%20gather_X%2C%0A%20%20%20%20gc%2C%0A%20%20%20%20generate_cv_splits_random%2C%0A%20%20%20%20gzip%2C%0A%20%20%20%20np%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20semipure%2C%0A%20%20%20%20tqdm%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%205%C3%975%20CV%3A%20baseline%20vs%20%2Bsemi-pure%20augmentation%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_OUT%20%3D%20PRED_DIR%20%2F%20%226_extradata_cv.csv.gz%22%0A%20%20%20%20_MODELS%20%3D%20%5B(%22xgb_mordred%22%2C%20%22mordred%22)%2C%20(%22rf_chemeleon%22%2C%20%22chemeleon%22)%5D%0A%20%20%20%20_sp_iks%20%3D%20semipure%5B%22inchikey%22%5D.to_list()%0A%20%20%20%20_sp_y%20%3D%20semipure%5B%22pEC50_corrected%22%5D.to_numpy()%0A%0A%20%20%20%20if%20_OUT.exists()%3A%0A%20%20%20%20%20%20%20%20extradata_cv%20%3D%20pl.read_csv(_OUT)%0A%20%20%20%20%20%20%20%20print(f%22Found%20%7B_OUT.name%7D%20%E2%80%94%20skipping%20training%20(%7Bextradata_cv.shape%5B0%5D%3A%2C%7D%20rows).%22)%0A%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20_records%3A%20list%5Bdict%5D%20%3D%20%5B%5D%0A%20%20%20%20%20%20%20%20for%20_fold%2C%20_outer%2C%20_inner%2C%20_tr%2C%20_va%2C%20_te%20in%20tqdm(%0A%20%20%20%20%20%20%20%20%20%20%20%20generate_cv_splits_random(DR_TRAIN%2C%20n_outer%3D5%2C%20n_inner%3D5%2C%20seed%3D42%2C%20p_val%3D0.1)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20total%3D25%2C%20desc%3D%22extra-data%20CV%22%2C%20unit%3D%22fold%22%2C%0A%20%20%20%20%20%20%20%20)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_tr%20%3D%20_tr%5B%22pEC50_dr%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_va%20%3D%20_va%5B%22pEC50_dr%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_te%20%3D%20_te%5B%22pEC50_dr%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_te_iks%20%3D%20set(_te%5B%22inchikey%22%5D.to_list())%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20Semi-pure%20rows%20usable%20this%20fold%3A%20exclude%20any%20colliding%20with%20the%20test%20fold.%0A%20%20%20%20%20%20%20%20%20%20%20%20_keep%20%3D%20%5Bi%20for%20i%2C%20ik%20in%20enumerate(_sp_iks)%20if%20ik%20not%20in%20_te_iks%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20_sp_keep_iks%20%3D%20%5B_sp_iks%5Bi%5D%20for%20i%20in%20_keep%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20_sp_keep_y%20%3D%20_sp_y%5B_keep%5D%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20_model_key%2C%20_kind%20in%20_MODELS%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_X_tr%20%3D%20gather_X(_tr%5B%22inchikey%22%5D.to_list()%2C%20_kind)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_X_va%20%3D%20gather_X(_va%5B%22inchikey%22%5D.to_list()%2C%20_kind)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_X_te%20%3D%20gather_X(_te%5B%22inchikey%22%5D.to_list()%2C%20_kind)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_X_sp%20%3D%20gather_X(_sp_keep_iks%2C%20_kind)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20_scenario%20in%20%5B%22base%22%2C%20%22%2Bsemipure%22%5D%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20_scenario%20%3D%3D%20%22base%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_Xt%2C%20_yt%20%3D%20_X_tr%2C%20_y_tr%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_Xt%20%3D%20np.vstack(%5B_X_tr%2C%20_X_sp%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_yt%20%3D%20np.concatenate(%5B_y_tr%2C%20_sp_keep_y%5D)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20_model_key%20%3D%3D%20%22xgb_mordred%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m%20%3D%20BoostedTreesModel(pred_type%3D%22regression%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m.train(_Xt%2C%20_yt%2C%20_X_va%2C%20_y_va)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m%20%3D%20RandomForestModel(pred_type%3D%22regression%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m.train(_Xt%2C%20_yt)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_pred%20%3D%20_m.predict(_X_te)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20del%20_m%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20gc.collect()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20_ik%2C%20_ytrue%2C%20_yp%20in%20zip(_te%5B%22inchikey%22%5D.to_list()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_y_te.tolist()%2C%20_pred.tolist())%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_records.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22inchikey%22%3A%20_ik%2C%20%22fold%22%3A%20_fold%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22model%22%3A%20_model_key%2C%20%22scenario%22%3A%20_scenario%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22method%22%3A%20f%22%7B_model_key%7D_%7B_scenario%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22y_true%22%3A%20_ytrue%2C%20%22y_pred%22%3A%20_yp%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7D)%0A%0A%20%20%20%20%20%20%20%20extradata_cv%20%3D%20pl.DataFrame(_records)%0A%20%20%20%20%20%20%20%20with%20gzip.open(_OUT%2C%20%22wb%22)%20as%20_f%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20extradata_cv.write_csv(_f)%0A%20%20%20%20%20%20%20%20print(f%22Wrote%20%7Bextradata_cv.shape%5B0%5D%3A%2C%7D%20rows%20%E2%86%92%20%7B_OUT.name%7D%22)%0A%20%20%20%20return%20(extradata_cv%2C)%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20PLOTS_DIR%2C%0A%20%20%20%20calc_regression_metrics%2C%0A%20%20%20%20extradata_cv%2C%0A%20%20%20%20make_mcs_plot_grid%2C%0A%20%20%20%20mean_absolute_error%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20pl%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Summary%20%2B%20MCS%20for%20the%20augmentation%20experiment%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_rows%20%3D%20%5B%5D%0A%20%20%20%20for%20(_method%2C)%2C%20_grp%20in%20extradata_cv.group_by(%5B%22method%22%5D)%3A%0A%20%20%20%20%20%20%20%20_yt%20%3D%20_grp%5B%22y_true%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_yp%20%3D%20_grp%5B%22y_pred%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_hit%20%3D%20_grp.filter(pl.col(%22y_true%22)%20%3E%3D%206.0)%0A%20%20%20%20%20%20%20%20_rows.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22method%22%3A%20_method%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22MAE%22%3A%20round(mean_absolute_error(_yt%2C%20_yp)%2C%204)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22hitzone_MAE%22%3A%20round(mean_absolute_error(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_hit%5B%22y_true%22%5D.to_numpy()%2C%20_hit%5B%22y_pred%22%5D.to_numpy())%2C%203)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22hitzone_bias%22%3A%20round(float((_hit%5B%22y_pred%22%5D%20-%20_hit%5B%22y_true%22%5D).mean())%2C%203)%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20extradata_summary%20%3D%20pl.DataFrame(_rows).sort(%22MAE%22)%0A%0A%20%20%20%20_metrics%20%3D%20calc_regression_metrics(%0A%20%20%20%20%20%20%20%20extradata_cv.select(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22fold%22).alias(%22cv_cycle%22)%2C%20%22method%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.lit(%22random%22).alias(%22split%22)%2C%20%22y_true%22%2C%20%22y_pred%22%5D)%2C%0A%20%20%20%20%20%20%20%20%22cv_cycle%22%2C%20%22y_true%22%2C%20%22y_pred%22%2C%20thresh%3D4.0)%0A%20%20%20%20_fig%20%3D%20make_mcs_plot_grid(%0A%20%20%20%20%20%20%20%20_metrics%2C%20stats%3D%5B%22mae%22%5D%2C%20group_col%3D%22method%22%2C%0A%20%20%20%20%20%20%20%20figsize%3D(10%2C%209)%2C%20effect_dict%3D%7B%22mae%22%3A%200.02%7D%2C%0A%20%20%20%20%20%20%20%20cell_text_size%3D11%2C%20axis_text_size%3D9%2C%20title_text_size%3D13%2C%0A%20%20%20%20%20%20%20%20save_path%3DPLOTS_DIR%20%2F%20%22analysis4_extradata_mcs_mae.png%22)%0A%0A%20%20%20%20mo.vstack(%5B%0A%20%20%20%20%20%20%20%20mo.md(%22%23%23%23%20Augmentation%20results%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22%60%2Bsemipure%60%20adds%20%E2%89%88%2090%20high-quality%20compounds%20to%20each%20training%20fold.%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22With%20only%20a%20~2%20%25%20increase%20in%20training%20size%2C%20any%20gain%20is%20expected%20to%20be%20small.%22)%2C%0A%20%20%20%20%20%20%20%20mo.ui.table(extradata_summary.to_pandas()%2C%20selection%3DNone%2C%20pagination%3DFalse)%2C%0A%20%20%20%20%20%20%20%20mo.as_html(_fig)%2C%0A%20%20%20%20%5D)%0A%20%20%20%20return%20(extradata_summary%2C)%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%20Analysis%205%20%E2%80%94%20Filtering%20(cleaning)%20the%20training%20set%0A%0A%20%20%20%20The%20previous%20analyses%20*add*%20or%20*reweight*%20data.%20This%20one%20*removes*%20likely%0A%20%20%20%20unreliable%20training%20points%20and%20asks%20whether%20a%20cleaner%20set%20generalises%20better.%0A%20%20%20%20Under%20the%20same%205%C3%975%20CV%20the%20**training%20fold%20is%20filtered**%20(the%20test%20fold%20is%20left%0A%20%20%20%20intact)%20and%20compared%20against%20(a)%20the%20full-data%20baseline%20and%20(b)%20a%20**size-matched%0A%20%20%20%20random-removal%20control**%20%E2%80%94%20so%20we%20can%20tell%20genuine%20cleaning%20from%20the%20mere%20effect%0A%20%20%20%20of%20training%20on%20less%20data.%0A%0A%20%20%20%20Three%20filters%20are%20compared%2C%20each%20removing%20%E2%89%88%2010%20%25%20of%20the%20training%20fold%3A%0A%0A%20%20%20%20%7C%20Filter%20%7C%20Rule%20%7C%20Rationale%20%7C%0A%20%20%20%20%7C---%7C---%7C---%7C%0A%20%20%20%20%7C%20%60counter%60%20%7C%20drop%20where%20counter-screen%20pEC50%20%E2%89%A5%20PXR%20pEC50%20%7C%20the%20hit%20is%20not%20PXR-specific%20%7C%0A%20%20%20%20%7C%20%60cliff%60%20%7C%20ECFP4%20Tanimoto%20%E2%89%A5%200.4%20%26%20%5C%7C%CE%94pEC50%5C%7C%20%E2%89%A5%201.0%20%E2%86%92%20drop%20the%20member%20nearer%20the%20mean%20%7C%20remove%20contradictory%20SAR%2C%20keep%20the%20informative%20extreme%20%7C%0A%20%20%20%20%7C%20%60knn%60%20%7C%20drop%20the%2010%20%25%20whose%20pEC50%20deviates%20most%20from%20their%205%20nearest%20ECFP4%20neighbours%20%7C%20local%20label-noise%20removal%20%7C%0A%0A%20%20%20%20Every%20filter%20is%20paired%20with%20a%20%60rand_*%60%20control%20that%20drops%20the%20**same%20number**%20of%0A%20%20%20%20random%20rows.%20Two%20models%20are%20tested%20%E2%80%94%20XGBoost%C2%B7Mordred%2C%20Macau%C2%B7CheMeleon%20%E2%80%94%20with%0A%20%20%20%20the%201%C3%97%20(full-data)%20baseline%20reused%20from%20earlier%20config-matched%20runs.%20Filtering%0A%20%20%20%20is%20computed%20inside%20each%20fold%20from%20the%20training%20compounds%20only%2C%20so%20there%20is%20no%0A%20%20%20%20test-set%20leakage.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(Chem%2C%20np%2C%20pl)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Training%20table%20with%20counter%20data%20%2B%20ECFP4%20fingerprints%20%2B%20filter%20functions%20%E2%94%80%E2%94%80%0A%20%20%20%20from%20rdkit.Chem%20import%20AllChem%20as%20_AllChem%0A%20%20%20%20from%20rdkit%20import%20DataStructs%20as%20_DataStructs%0A%0A%20%20%20%20%23%20Same%20rows%2Forder%20as%20DR_TRAIN%20(so%20folds%20match%20the%20reused%20baselines)%20plus%20counter.%0A%20%20%20%20filter_train%20%3D%20(%0A%20%20%20%20%20%20%20%20pl.read_csv(%22..%2Fdata%2Fprocessed%2Fall_compounds_activity_data.csv%22)%0A%20%20%20%20%20%20%20%20.filter(pl.col(%22pEC50_dr%22).is_not_null())%0A%20%20%20%20%20%20%20%20.select(%5B%22smiles%22%2C%20%22inchikey%22%2C%20%22molecule_names%22%2C%20%22pEC50_dr%22%2C%20%22pEC50_counter%22%5D)%0A%20%20%20%20)%0A%0A%20%20%20%20def%20_ecfp4(smi%3A%20str)%3A%0A%20%20%20%20%20%20%20%20mol%20%3D%20Chem.MolFromSmiles(smi)%0A%20%20%20%20%20%20%20%20return%20_AllChem.GetMorganFingerprintAsBitVect(mol%2C%202%2C%20nBits%3D2048)%20if%20mol%20else%20None%0A%0A%20%20%20%20ecfp4_by_ik%20%3D%20%7B%0A%20%20%20%20%20%20%20%20ik%3A%20_ecfp4(s)%0A%20%20%20%20%20%20%20%20for%20ik%2C%20s%20in%20zip(filter_train%5B%22inchikey%22%5D.to_list()%2C%20filter_train%5B%22smiles%22%5D.to_list())%0A%20%20%20%20%7D%0A%20%20%20%20print(f%22ECFP4%20cached%20for%20%7Blen(ecfp4_by_ik)%7D%20compounds%22)%0A%0A%20%20%20%20def%20counter_remove(tr%3A%20pl.DataFrame)%20-%3E%20set%3A%0A%20%20%20%20%20%20%20%20%22%22%22Positions%20of%20compounds%20whose%20counter%20pEC50%20%E2%89%A5%20their%20PXR%20pEC50%20(non-selective).%22%22%22%0A%20%20%20%20%20%20%20%20_rm%20%3D%20tr.with_columns(%0A%20%20%20%20%20%20%20%20%20%20%20%20(pl.col(%22pEC50_counter%22).is_not_null()%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%26%20(pl.col(%22pEC50_counter%22)%20%3E%3D%20pl.col(%22pEC50_dr%22))).alias(%22_rm%22))%0A%20%20%20%20%20%20%20%20return%20set(np.where(_rm%5B%22_rm%22%5D.to_numpy())%5B0%5D.tolist())%0A%0A%20%20%20%20def%20cliff_remove(fps%3A%20list%2C%20y%3A%20np.ndarray%2C%20tan%3A%20float%20%3D%200.4%2C%20dthr%3A%20float%20%3D%201.0)%20-%3E%20set%3A%0A%20%20%20%20%20%20%20%20%22%22%22Activity-cliff%20members%20to%20drop%20(the%20one%20nearer%20the%20global%20mean%20of%20each%20pair).%22%22%22%0A%20%20%20%20%20%20%20%20mean%20%3D%20float(y.mean())%0A%20%20%20%20%20%20%20%20remove%3A%20set%20%3D%20set()%0A%20%20%20%20%20%20%20%20for%20i%20in%20range(len(fps))%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20fps%5Bi%5D%20is%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20continue%0A%20%20%20%20%20%20%20%20%20%20%20%20sims%20%3D%20_DataStructs.BulkTanimotoSimilarity(fps%5Bi%5D%2C%20fps%5Bi%20%2B%201%3A%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20off%2C%20s%20in%20enumerate(sims)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20j%20%3D%20i%20%2B%201%20%2B%20off%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20s%20%3E%3D%20tan%20and%20abs(y%5Bi%5D%20-%20y%5Bj%5D)%20%3E%3D%20dthr%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20remove.add(i%20if%20abs(y%5Bi%5D%20-%20mean)%20%3C%20abs(y%5Bj%5D%20-%20mean)%20else%20j)%0A%20%20%20%20%20%20%20%20return%20remove%0A%0A%20%20%20%20def%20knn_remove(fps%3A%20list%2C%20y%3A%20np.ndarray%2C%20k%3A%20int%20%3D%205%2C%20q%3A%20float%20%3D%200.10)%20-%3E%20set%3A%0A%20%20%20%20%20%20%20%20%22%22%22Drop%20the%20top-q%20fraction%20by%20deviation%20from%20their%20k%20nearest%20ECFP4%20neighbours.%22%22%22%0A%20%20%20%20%20%20%20%20n%20%3D%20len(fps)%0A%20%20%20%20%20%20%20%20dev%20%3D%20np.zeros(n)%0A%20%20%20%20%20%20%20%20for%20i%20in%20range(n)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20fps%5Bi%5D%20is%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20dev%5Bi%5D%20%3D%20-1.0%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20continue%0A%20%20%20%20%20%20%20%20%20%20%20%20sims%20%3D%20np.array(_DataStructs.BulkTanimotoSimilarity(fps%5Bi%5D%2C%20fps))%0A%20%20%20%20%20%20%20%20%20%20%20%20sims%5Bi%5D%20%3D%20-1.0%0A%20%20%20%20%20%20%20%20%20%20%20%20nn%20%3D%20np.argpartition(sims%2C%20-k)%5B-k%3A%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20dev%5Bi%5D%20%3D%20abs(y%5Bi%5D%20-%20y%5Bnn%5D.mean())%0A%20%20%20%20%20%20%20%20n_remove%20%3D%20int(round(q%20*%20n))%0A%20%20%20%20%20%20%20%20return%20set(np.argsort(dev)%5B-n_remove%3A%5D.tolist())%0A%0A%20%20%20%20return%20cliff_remove%2C%20counter_remove%2C%20ecfp4_by_ik%2C%20filter_train%2C%20knn_remove%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20BoostedTreesModel%2C%0A%20%20%20%20MacauModel%2C%0A%20%20%20%20PRED_DIR%2C%0A%20%20%20%20cliff_remove%2C%0A%20%20%20%20counter_remove%2C%0A%20%20%20%20ecfp4_by_ik%2C%0A%20%20%20%20filter_train%2C%0A%20%20%20%20gather_X%2C%0A%20%20%20%20gc%2C%0A%20%20%20%20generate_cv_splits_random%2C%0A%20%20%20%20gzip%2C%0A%20%20%20%20knn_remove%2C%0A%20%20%20%20np%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20tqdm%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%205%C3%975%20CV%3A%20filter%20the%20training%20fold%2C%20two%20models%2C%20filter%20%2B%20matched%20random%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%201%C3%97%20(full-data)%20baseline%20is%20reused%20in%20the%20summary%20cell%3B%20only%20filtered%2Frandom%0A%20%20%20%20%23%20variants%20are%20trained%20here.%20Checkpointed%20per%20(fold%2C%20method).%0A%20%20%20%20_OUT%20%3D%20PRED_DIR%20%2F%20%226_filtering_cv.csv.gz%22%0A%20%20%20%20_CKPT%20%3D%20_OUT.with_suffix(%22.ckpt.gz%22)%0A%20%20%20%20_FILTERS%20%3D%20%5B%22counter%22%2C%20%22cliff%22%2C%20%22knn%22%5D%0A%20%20%20%20_MODELS%20%3D%20%5B%0A%20%20%20%20%20%20%20%20(%22xgb_mordred%22%2C%20%22mordred%22)%2C%0A%20%20%20%20%20%20%20%20(%22macau_chemeleon%22%2C%20%22chemeleon%22)%2C%0A%20%20%20%20%5D%0A%0A%20%20%20%20if%20_OUT.exists()%3A%0A%20%20%20%20%20%20%20%20filtering_cv%20%3D%20pl.read_csv(_OUT)%0A%20%20%20%20%20%20%20%20print(f%22Found%20%7B_OUT.name%7D%20%E2%80%94%20skipping%20(%7Bfiltering_cv.shape%5B0%5D%3A%2C%7D%20rows).%22)%0A%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20if%20_CKPT.exists()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_records%20%3D%20pl.read_csv(_CKPT).to_dicts()%0A%20%20%20%20%20%20%20%20%20%20%20%20_done%20%3D%20%7B(r%5B%22fold%22%5D%2C%20r%5B%22method%22%5D)%20for%20r%20in%20_records%7D%0A%20%20%20%20%20%20%20%20%20%20%20%20print(f%22Resuming%20filtering%20CV%20(%7Blen(_records)%3A%2C%7D%20rows%2C%20%7Blen(_done)%7D%20done).%22)%0A%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_records%2C%20_done%20%3D%20%5B%5D%2C%20set()%0A%0A%20%20%20%20%20%20%20%20for%20_fold%2C%20_outer%2C%20_inner%2C%20_tr%2C%20_va%2C%20_te%20in%20tqdm(%0A%20%20%20%20%20%20%20%20%20%20%20%20generate_cv_splits_random(filter_train%2C%20n_outer%3D5%2C%20n_inner%3D5%2C%20seed%3D42%2C%20p_val%3D0.1)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20total%3D25%2C%20desc%3D%22filtering%20CV%22%2C%20unit%3D%22fold%22%2C%0A%20%20%20%20%20%20%20%20)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_tr%20%3D%20_tr%5B%22pEC50_dr%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_va%20%3D%20_va%5B%22pEC50_dr%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_y_te%20%3D%20_te%5B%22pEC50_dr%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_tr_iks%20%3D%20_tr%5B%22inchikey%22%5D.to_list()%0A%20%20%20%20%20%20%20%20%20%20%20%20_te_iks%20%3D%20_te%5B%22inchikey%22%5D.to_list()%0A%20%20%20%20%20%20%20%20%20%20%20%20_n_tr%20%3D%20len(_tr_iks)%0A%20%20%20%20%20%20%20%20%20%20%20%20_fps%20%3D%20%5Becfp4_by_ik%5Bik%5D%20for%20ik%20in%20_tr_iks%5D%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20Remove-sets%20for%20the%20three%20filters%20and%20their%20size-matched%20random%20twins.%0A%20%20%20%20%20%20%20%20%20%20%20%20_removes%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22counter%22%3A%20counter_remove(_tr)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22cliff%22%3A%20cliff_remove(_fps%2C%20_y_tr)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22knn%22%3A%20knn_remove(_fps%2C%20_y_tr)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%0A%20%20%20%20%20%20%20%20%20%20%20%20_rng%20%3D%20np.random.default_rng(1000%20%2B%20_fold)%0A%20%20%20%20%20%20%20%20%20%20%20%20_variants%3A%20dict%5Bstr%2C%20set%5D%20%3D%20%7B%7D%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20_f%20in%20_FILTERS%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_variants%5B_f%5D%20%3D%20_removes%5B_f%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_variants%5Bf%22rand_%7B_f%7D%22%5D%20%3D%20set(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_rng.choice(_n_tr%2C%20size%3Dlen(_removes%5B_f%5D)%2C%20replace%3DFalse).tolist())%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20Features%20gathered%20once%20per%20fold%3B%20rows%20subset%20via%20_kept.%0A%20%20%20%20%20%20%20%20%20%20%20%20_Xtr_mor%20%3D%20gather_X(_tr_iks%2C%20%22mordred%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_Xva_mor%20%3D%20gather_X(_va%5B%22inchikey%22%5D.to_list()%2C%20%22mordred%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_Xte_mor%20%3D%20gather_X(_te_iks%2C%20%22mordred%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_Xtr_che%20%3D%20gather_X(_tr_iks%2C%20%22chemeleon%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_Xte_che%20%3D%20gather_X(_te_iks%2C%20%22chemeleon%22)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20def%20_save(method%3A%20str%2C%20preds)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20_ik%2C%20_yt%2C%20_yp%20in%20zip(_te_iks%2C%20_y_te.tolist()%2C%20list(preds))%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_records.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22inchikey%22%3A%20_ik%2C%20%22fold%22%3A%20_fold%2C%20%22method%22%3A%20method%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22y_true%22%3A%20_yt%2C%20%22y_pred%22%3A%20float(_yp)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20with%20gzip.open(_CKPT%2C%20%22wb%22)%20as%20_f%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.DataFrame(_records).write_csv(_f)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20_variant%2C%20_rm%20in%20_variants.items()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_kept%20%3D%20np.array(%5Bp%20for%20p%20in%20range(_n_tr)%20if%20p%20not%20in%20_rm%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_yk%20%3D%20_y_tr%5B_kept%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20_model_key%2C%20_kind%20in%20_MODELS%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_method%20%3D%20f%22%7B_model_key%7D_%7B_variant%7D%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20(_fold%2C%20_method)%20in%20_done%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20continue%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20_model_key%20%3D%3D%20%22xgb_mordred%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m%20%3D%20BoostedTreesModel(pred_type%3D%22regression%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m.train(_Xtr_mor%5B_kept%5D%2C%20_yk%2C%20_Xva_mor%2C%20_y_va)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_save(_method%2C%20_m.predict(_Xte_mor))%3B%20del%20_m%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20elif%20_model_key%20%3D%3D%20%22macau_chemeleon%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m%20%3D%20MacauModel(seed%3D42)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_m.train(_Xtr_che%5B_kept%5D%2C%20_yk)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_save(_method%2C%20_m.predict(_Xte_che))%3B%20del%20_m%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20gc.collect()%0A%0A%20%20%20%20%20%20%20%20filtering_cv%20%3D%20pl.DataFrame(_records)%0A%20%20%20%20%20%20%20%20with%20gzip.open(_OUT%2C%20%22wb%22)%20as%20_f%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20filtering_cv.write_csv(_f)%0A%20%20%20%20%20%20%20%20_CKPT.unlink(missing_ok%3DTrue)%0A%20%20%20%20%20%20%20%20print(f%22Wrote%20%7Bfiltering_cv.shape%5B0%5D%3A%2C%7D%20rows%20%E2%86%92%20%7B_OUT.name%7D%22)%0A%20%20%20%20return%20(filtering_cv%2C)%0A%0A%0A%40app.cell%0Adef%20_(PLOTS_DIR%2C%20PRED_DIR%2C%20filtering_cv%2C%20mean_absolute_error%2C%20mo%2C%20np%2C%20pl%2C%20plt)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Combine%20reused%20full-data%20baselines%20with%20the%20filtered%20%2F%20random%20variants%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20notebook%202%20labels%20models%20in%20a%20%60model%60%20column%3B%20the%20others%20use%20%60method%60.%0A%20%20%20%20def%20_baseline(path%3A%20str%2C%20col%3A%20str%2C%20val%3A%20str%2C%20model_key%3A%20str)%20-%3E%20pl.DataFrame%3A%0A%20%20%20%20%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.read_csv(PRED_DIR%20%2F%20path)%0A%20%20%20%20%20%20%20%20%20%20%20%20.filter(pl.col(col)%20%3D%3D%20val)%0A%20%20%20%20%20%20%20%20%20%20%20%20.select(%5B%22fold%22%2C%20%22y_true%22%2C%20%22y_pred%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20.with_columns(pl.format(%22%7B%7D_baseline%22%2C%20pl.lit(model_key)).alias(%22method%22))%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20_base%20%3D%20pl.concat(%5B%0A%20%20%20%20%20%20%20%20_baseline(%226_reweighting_cv.csv.gz%22%2C%20%22method%22%2C%20%22xgb_mordred_uniform%22%2C%20%22xgb_mordred%22)%2C%0A%20%20%20%20%20%20%20%20_baseline(%224_fp_model_comparison_1.csv.gz%22%2C%20%22method%22%2C%20%22macau_chemeleon%22%2C%20%22macau_chemeleon%22)%2C%0A%20%20%20%20%5D)%0A%20%20%20%20_cols%20%3D%20%5B%22fold%22%2C%20%22method%22%2C%20%22y_true%22%2C%20%22y_pred%22%5D%0A%20%20%20%20filt_all%20%3D%20pl.concat(%5B_base.select(_cols)%2C%20filtering_cv.select(_cols)%5D)%0A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Summary%3A%20MAE%20per%20method%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_rows%20%3D%20%5B%5D%0A%20%20%20%20for%20(_method%2C)%2C%20_grp%20in%20filt_all.group_by(%5B%22method%22%5D)%3A%0A%20%20%20%20%20%20%20%20_rows.append(%7B%22method%22%3A%20_method%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22MAE%22%3A%20round(mean_absolute_error(_grp%5B%22y_true%22%5D.to_numpy()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_grp%5B%22y_pred%22%5D.to_numpy())%2C%204)%7D)%0A%20%20%20%20filtering_summary%20%3D%20pl.DataFrame(_rows).sort(%22method%22)%0A%0A%20%20%20%20_MODELS%20%3D%20%5B%22xgb_mordred%22%2C%20%22macau_chemeleon%22%5D%0A%20%20%20%20_FILTERS%20%3D%20%5B%22counter%22%2C%20%22cliff%22%2C%20%22knn%22%5D%0A%0A%20%20%20%20def%20_mae(method%3A%20str)%20-%3E%20float%3A%0A%20%20%20%20%20%20%20%20_r%20%3D%20filtering_summary.filter(pl.col(%22method%22)%20%3D%3D%20method)%0A%20%20%20%20%20%20%20%20return%20float(_r%5B%22MAE%22%5D%5B0%5D)%20if%20_r.shape%5B0%5D%20else%20float(%22nan%22)%0A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Per-model%20%CE%94MAE%20vs%20baseline%3A%20filter%20bar%20next%20to%20its%20matched%20random%20twin%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20_fig%2C%20_axes%20%3D%20plt.subplots(1%2C%202%2C%20figsize%3D(12%2C%205)%2C%20dpi%3D140)%0A%20%20%20%20%20%20%20%20for%20_ax%2C%20_model%20in%20zip(_axes.flatten()%2C%20_MODELS)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_base_mae%20%3D%20_mae(f%22%7B_model%7D_baseline%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_x%20%3D%20np.arange(len(_FILTERS))%0A%20%20%20%20%20%20%20%20%20%20%20%20_filt_d%20%3D%20%5B_mae(f%22%7B_model%7D_%7Bf%7D%22)%20-%20_base_mae%20for%20f%20in%20_FILTERS%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20_rand_d%20%3D%20%5B_mae(f%22%7B_model%7D_rand_%7Bf%7D%22)%20-%20_base_mae%20for%20f%20in%20_FILTERS%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.bar(_x%20-%200.2%2C%20_filt_d%2C%200.4%2C%20label%3D%22filter%22%2C%20color%3D%22%234e79a7%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.bar(_x%20%2B%200.2%2C%20_rand_d%2C%200.4%2C%20label%3D%22random%20(matched%20N)%22%2C%20color%3D%22%23bab0ac%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.axhline(0%2C%20color%3D%22black%22%2C%20linewidth%3D0.9)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.set_xticks(_x)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.set_xticklabels(_FILTERS)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.set_ylabel(%22%CE%94MAE%20vs%20full-data%20baseline%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.set_title(f%22%7B_model%7D%20%20(baseline%20MAE%3D%7B_base_mae%3A.3f%7D)%22%2C%20fontsize%3D10)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.legend(fontsize%3D8)%0A%20%20%20%20%20%20%20%20_fig.suptitle(%22Training-set%20filtering%20%E2%80%94%20%CE%94MAE%20vs%20baseline%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22(negative%20%3D%20improvement%3B%20beat%20the%20grey%20random%20control%20to%20count)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20fontsize%3D12%2C%20y%3D1.01)%0A%20%20%20%20%20%20%20%20_fig.tight_layout()%0A%20%20%20%20%20%20%20%20_fig.savefig(PLOTS_DIR%20%2F%20%22analysis5_filtering_delta_mae.png%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20mo.vstack(%5B%0A%20%20%20%20%20%20%20%20mo.md(%22%23%23%23%20Filtering%20results%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22A%20filter%20is%20worthwhile%20only%20if%20its%20blue%20bar%20is%20**below%200**%20(beats%20the%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22full-data%20baseline)%20**and%20below%20its%20grey%20random%20twin**%20(the%20gain%20is%20from%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22cleaning%2C%20not%20from%20a%20smaller%20training%20set).%22)%2C%0A%20%20%20%20%20%20%20%20mo.ui.table(filtering_summary.to_pandas()%2C%20selection%3DNone%2C%20pagination%3DFalse)%2C%0A%20%20%20%20%20%20%20%20mo.as_html(_fig)%2C%0A%20%20%20%20%5D)%0A%20%20%20%20return%20(filtering_summary%2C)%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%20Strategy%20selection%20and%20application%20to%20the%20unblinded%20test%20set%0A%0A%20%20%20%20The%20five%20analyses%20are%20compared%20on%20a%20common%20footing%3A%20the%20CV%20MAE%20**improvement%20of%0A%20%20%20%20each%20strategy%20over%20its%20own%20baseline**.%20Comparing%20absolute%20MAE%20across%20analyses%0A%20%20%20%20would%20be%20unfair%2C%20because%20the%20calibration%20strategy%20operates%20on%20the%20strong%20ensemble%0A%20%20%20%20while%20reweighting%2C%20oversampling%2C%20augmentation%20and%20filtering%20operate%20on%20weaker%0A%20%20%20%20single%20component%20models.%0A%0A%20%20%20%20**What%20can%20actually%20be%20applied%20to%20the%20submission%3F**%20The%20submitted%20model%20is%20the%0A%20%20%20%20ensemble.%20Of%20the%20three%20ideas%2C%20only%20**calibration**%20operates%20directly%20on%20it%20%E2%80%94%0A%20%20%20%20reweighting%20and%20augmentation%20modify%20individual%20components%20that%20are%20each%20weaker%0A%20%20%20%20than%20the%20ensemble%2C%20so%20they%20are%20CV%20experiments%20that%20would%20only%20pay%20off%20by%0A%20%20%20%20*rebuilding*%20the%20ensemble%20from%20reweighted%2Faugmented%20components%20(future%20work).%0A%0A%20%20%20%20Therefore%2C%20honouring%20the%20rule%20that%20the%20unblinded%20labels%20must%20not%20drive%20model%0A%20%20%20%20selection%2C%20the%20strategy%20applied%20to%20the%20unblinded%20set%20is%20the%20**best%20calibration%0A%20%20%20%20variant%20chosen%20purely%20by%205%C3%975%20CV**%20(which%20may%20be%20%60raw%60%2C%20i.e.%20no%20change).%20Its%20effect%0A%20%20%20%20on%20the%20unblinded%20labels%20is%20then%20reported%20once%2C%20for%20confirmation%20only.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20calib_summary%2C%0A%20%20%20%20extradata_summary%2C%0A%20%20%20%20filtering_summary%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20oversampling_summary%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20reweight_summary%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Assemble%20the%20cross-analysis%20improvement%20table%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20def%20_best_delta(summary%3A%20pl.DataFrame%2C%20baseline_method%3A%20str%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20candidates%3A%20list%5Bstr%5D)%20-%3E%20tuple%5Bstr%2C%20float%2C%20float%5D%3A%0A%20%20%20%20%20%20%20%20base_mae%20%3D%20summary.filter(pl.col(%22method%22)%20%3D%3D%20baseline_method)%5B%22MAE%22%5D%5B0%5D%0A%20%20%20%20%20%20%20%20best_method%2C%20best_mae%20%3D%20baseline_method%2C%20base_mae%0A%20%20%20%20%20%20%20%20for%20cand%20in%20candidates%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20mae%20%3D%20summary.filter(pl.col(%22method%22)%20%3D%3D%20cand)%5B%22MAE%22%5D%5B0%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20mae%20%3C%20best_mae%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20best_method%2C%20best_mae%20%3D%20cand%2C%20mae%0A%20%20%20%20%20%20%20%20return%20best_method%2C%20best_mae%2C%20best_mae%20-%20base_mae%0A%0A%20%20%20%20%23%20Calibration%20%E2%80%94%20evaluate%20on%20the%20ensemble%20(the%20submitted%20model).%0A%20%20%20%20_cal%20%3D%20calib_summary.filter(pl.col(%22base_model%22)%20%3D%3D%20%22ensemble%22)%0A%20%20%20%20_cal_base%20%3D%20_cal.filter(pl.col(%22calibration%22)%20%3D%3D%20%22raw%22)%5B%22MAE%22%5D%5B0%5D%0A%20%20%20%20_cal_best_row%20%3D%20_cal.sort(%22MAE%22).row(0%2C%20named%3DTrue)%0A%20%20%20%20_cal_best%2C%20_cal_mae%2C%20_cal_delta%20%3D%20(%0A%20%20%20%20%20%20%20%20_cal_best_row%5B%22calibration%22%5D%2C%20_cal_best_row%5B%22MAE%22%5D%2C%20_cal_best_row%5B%22MAE%22%5D%20-%20_cal_base)%0A%0A%20%20%20%20%23%20Reweighting%20%E2%80%94%20best%20scheme%20per%20model%20relative%20to%20that%20model's%20uniform%20baseline.%0A%20%20%20%20_rw_xgb%20%3D%20_best_delta(reweight_summary%2C%20%22xgb_mordred_uniform%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%5B%22xgb_mordred_invdensity%22%2C%20%22xgb_mordred_distance%22%5D)%0A%20%20%20%20_rw_rf%20%3D%20_best_delta(reweight_summary%2C%20%22rf_chemeleon_uniform%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%5B%22rf_chemeleon_invdensity%22%2C%20%22rf_chemeleon_distance%22%5D)%0A%20%20%20%20_rw_cp%20%3D%20_best_delta(reweight_summary%2C%20%22chemprop_scratch_uniform%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%5B%22chemprop_scratch_invdensity%22%2C%20%22chemprop_scratch_distance%22%5D)%0A%0A%20%20%20%20%23%20Oversampling%20%E2%80%94%20best%20factor%20(os2%2Fos3%2Fos5)%20vs%20the%20reused%201%C3%97%20baseline%20per%20model.%0A%20%20%20%20def%20_os_delta(model_key%3A%20str)%20-%3E%20tuple%5Bstr%2C%20float%2C%20float%5D%3A%0A%20%20%20%20%20%20%20%20return%20_best_delta(%0A%20%20%20%20%20%20%20%20%20%20%20%20oversampling_summary%2C%20f%22%7Bmodel_key%7D_os1%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5Bf%22%7Bmodel_key%7D_os2%22%2C%20f%22%7Bmodel_key%7D_os3%22%2C%20f%22%7Bmodel_key%7D_os5%22%5D)%0A%20%20%20%20_os_xgb%20%3D%20_os_delta(%22xgb_mordred%22)%0A%20%20%20%20_os_mac%20%3D%20_os_delta(%22macau_chemeleon%22)%0A%20%20%20%20_os_tf%20%3D%20_os_delta(%22tabpfn_chemeleon%22)%0A%20%20%20%20_os_cp%20%3D%20_os_delta(%22chemprop_scratch%22)%0A%0A%20%20%20%20%23%20Augmentation%20%E2%80%94%20%2Bsemipure%20vs%20base%20per%20model.%0A%20%20%20%20_ed_xgb%20%3D%20_best_delta(extradata_summary%2C%20%22xgb_mordred_base%22%2C%20%5B%22xgb_mordred_%2Bsemipure%22%5D)%0A%20%20%20%20_ed_rf%20%3D%20_best_delta(extradata_summary%2C%20%22rf_chemeleon_base%22%2C%20%5B%22rf_chemeleon_%2Bsemipure%22%5D)%0A%0A%20%20%20%20%23%20Filtering%20%E2%80%94%20best%20filter%20(counter%2Fcliff%2Fknn)%20vs%20the%20full-data%20baseline%20per%20model.%0A%20%20%20%20def%20_filt_delta(model_key%3A%20str)%20-%3E%20tuple%5Bstr%2C%20float%2C%20float%5D%3A%0A%20%20%20%20%20%20%20%20return%20_best_delta(%0A%20%20%20%20%20%20%20%20%20%20%20%20filtering_summary%2C%20f%22%7Bmodel_key%7D_baseline%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5Bf%22%7Bmodel_key%7D_counter%22%2C%20f%22%7Bmodel_key%7D_cliff%22%2C%20f%22%7Bmodel_key%7D_knn%22%5D)%0A%20%20%20%20_fl_xgb%20%3D%20_filt_delta(%22xgb_mordred%22)%0A%20%20%20%20_fl_mac%20%3D%20_filt_delta(%22macau_chemeleon%22)%0A%0A%20%20%20%20def%20_row(analysis%3A%20str%2C%20model%3A%20str%2C%20res%3A%20tuple)%20-%3E%20dict%3A%0A%20%20%20%20%20%20%20%20return%20%7B%22analysis%22%3A%20analysis%2C%20%22model%22%3A%20model%2C%20%22best_variant%22%3A%20res%5B0%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22cv_mae%22%3A%20round(res%5B1%5D%2C%204)%2C%20%22delta_mae%22%3A%20round(res%5B2%5D%2C%204)%7D%0A%0A%20%20%20%20selection_table%20%3D%20pl.DataFrame(%5B%0A%20%20%20%20%20%20%20%20%7B%22analysis%22%3A%20%221%20calibration%22%2C%20%22model%22%3A%20%22ensemble%22%2C%20%22best_variant%22%3A%20_cal_best%2C%0A%20%20%20%20%20%20%20%20%20%22cv_mae%22%3A%20round(_cal_mae%2C%204)%2C%20%22delta_mae%22%3A%20round(_cal_delta%2C%204)%7D%2C%0A%20%20%20%20%20%20%20%20_row(%222%20reweighting%22%2C%20%22xgb_mordred%22%2C%20_rw_xgb)%2C%0A%20%20%20%20%20%20%20%20_row(%222%20reweighting%22%2C%20%22rf_chemeleon%22%2C%20_rw_rf)%2C%0A%20%20%20%20%20%20%20%20_row(%222%20reweighting%22%2C%20%22chemprop_scratch%22%2C%20_rw_cp)%2C%0A%20%20%20%20%20%20%20%20_row(%223%20oversampling%22%2C%20%22xgb_mordred%22%2C%20_os_xgb)%2C%0A%20%20%20%20%20%20%20%20_row(%223%20oversampling%22%2C%20%22macau_chemeleon%22%2C%20_os_mac)%2C%0A%20%20%20%20%20%20%20%20_row(%223%20oversampling%22%2C%20%22tabpfn_chemeleon%22%2C%20_os_tf)%2C%0A%20%20%20%20%20%20%20%20_row(%223%20oversampling%22%2C%20%22chemprop_scratch%22%2C%20_os_cp)%2C%0A%20%20%20%20%20%20%20%20_row(%224%20extra%20data%22%2C%20%22xgb_mordred%22%2C%20_ed_xgb)%2C%0A%20%20%20%20%20%20%20%20_row(%224%20extra%20data%22%2C%20%22rf_chemeleon%22%2C%20_ed_rf)%2C%0A%20%20%20%20%20%20%20%20_row(%225%20filtering%22%2C%20%22xgb_mordred%22%2C%20_fl_xgb)%2C%0A%20%20%20%20%20%20%20%20_row(%225%20filtering%22%2C%20%22macau_chemeleon%22%2C%20_fl_mac)%2C%0A%20%20%20%20%5D).sort(%22delta_mae%22)%0A%0A%20%20%20%20%23%20The%20biggest%20CV%20improvement%20of%20any%20strategy%20over%20its%20own%20baseline%20(for%20insight).%0A%20%20%20%20winner%20%3D%20selection_table.row(0%2C%20named%3DTrue)%0A%20%20%20%20%23%20The%20strategy%20actually%20applied%20to%20the%20ensemble%20submission%3A%20best%20calibration%20by%20CV.%0A%20%20%20%20best_calibration%20%3D%20_cal_best%0A%0A%20%20%20%20mo.vstack(%5B%0A%20%20%20%20%20%20%20%20mo.md(%22%23%23%23%20Cross-analysis%20comparison%20%E2%80%94%20CV%20MAE%20improvement%20over%20each%20baseline%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Lower%20%60delta_mae%60%20(more%20negative)%20is%20better.%22)%2C%0A%20%20%20%20%20%20%20%20mo.ui.table(selection_table.to_pandas()%2C%20selection%3DNone%2C%20pagination%3DFalse)%2C%0A%20%20%20%20%20%20%20%20mo.md(f%22**Largest%20CV%20improvement%20(any%20strategy)%3A**%20%60%7Bwinner%5B'analysis'%5D%7D%60%20%E2%86%92%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22%60%7Bwinner%5B'best_variant'%5D%7D%60%20(%CE%94MAE%20%3D%20%7Bwinner%5B'delta_mae'%5D%3A%2B.4f%7D).%5Cn%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22**Applied%20to%20the%20ensemble%20submission%20(best%20calibration%20by%20CV)%3A**%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22%60%7Bbest_calibration%7D%60%20(%CE%94MAE%20%3D%20%7B_cal_delta%3A%2B.4f%7D).%22)%2C%0A%20%20%20%20%5D)%0A%20%20%20%20return%20best_calibration%2C%20winner%0A%0A%0A%40app.cell%0Adef%20_(IsotonicRegression%2C%20LinearRegression%2C%20ensemble_oof%2C%20pl)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Load%20the%20unblinded%20labels%20and%20the%20submitted%20ensemble%20test%20predictions%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20unblinded%20%3D%20(%0A%20%20%20%20%20%20%20%20pl.read_csv(%22..%2Fdata%2Fraw%2F20260528%2Fdose_response_test_unblinded.csv%22)%0A%20%20%20%20%20%20%20%20.select(%5B%22Molecule%20Name%22%2C%20%22pEC50%22%5D)%0A%20%20%20%20%20%20%20%20.rename(%7B%22pEC50%22%3A%20%22pEC50_true%22%7D)%0A%20%20%20%20)%0A%20%20%20%20ens_test%20%3D%20(%0A%20%20%20%20%20%20%20%20pl.read_csv(%22..%2Fsubmissions%2F4_ens_cp5_ch5_rf0_xg13_mc1_tf5_submission.csv%22)%0A%20%20%20%20%20%20%20%20.select(%5B%22Molecule%20Name%22%2C%20%22SMILES%22%2C%20%22pEC50%22%5D)%0A%20%20%20%20%20%20%20%20.rename(%7B%22pEC50%22%3A%20%22pEC50_pred%22%7D)%0A%20%20%20%20)%0A%20%20%20%20unblinded_eval%20%3D%20ens_test.join(unblinded%2C%20on%3D%22Molecule%20Name%22%2C%20how%3D%22inner%22)%0A%0A%20%20%20%20%23%20Fit%20GLOBAL%20calibrators%20on%20all%20ensemble%20OOF%20predictions%20(training%20data%20only).%0A%20%20%20%20_xp%20%3D%20ensemble_oof%5B%22y_pred%22%5D.to_numpy()%0A%20%20%20%20_yt%20%3D%20ensemble_oof%5B%22y_true%22%5D.to_numpy()%0A%20%20%20%20calib_linear%20%3D%20LinearRegression().fit(_xp.reshape(-1%2C%201)%2C%20_yt)%0A%20%20%20%20calib_isotonic%20%3D%20IsotonicRegression(out_of_bounds%3D%22clip%22).fit(_xp%2C%20_yt)%0A%0A%20%20%20%20_raw%20%3D%20unblinded_eval%5B%22pEC50_pred%22%5D.to_numpy()%0A%20%20%20%20unblinded_eval%20%3D%20unblinded_eval.with_columns(%0A%20%20%20%20%20%20%20%20pl.Series(%22pred_linear%22%2C%20calib_linear.predict(_raw.reshape(-1%2C%201)))%2C%0A%20%20%20%20%20%20%20%20pl.Series(%22pred_isotonic%22%2C%20calib_isotonic.predict(_raw))%2C%0A%20%20%20%20)%0A%0A%20%20%20%20print(f%22Unblinded%20compounds%20evaluated%3A%20%7Bunblinded_eval.shape%5B0%5D%7D%22)%0A%20%20%20%20print(f%22Linear%20de-shrink%20slope%3A%20%7Bfloat(calib_linear.coef_%5B0%5D)%3A.3f%7D%20%22%0A%20%20%20%20%20%20%20%20%20%20f%22(%3E1%20expands%20the%20range)%2C%20intercept%20%7Bfloat(calib_linear.intercept_)%3A.3f%7D%22)%0A%20%20%20%20return%20calib_isotonic%2C%20calib_linear%2C%20ens_test%2C%20unblinded_eval%0A%0A%0A%40app.cell%0Adef%20_(PLOTS_DIR%2C%20mean_absolute_error%2C%20mo%2C%20np%2C%20pl%2C%20plt%2C%20unblinded_eval)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Effect%20of%20calibration%20on%20the%20unblinded%20set%20(overall%20%2B%20hit%20zone)%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20def%20_row(name%3A%20str%2C%20col%3A%20str)%20-%3E%20dict%3A%0A%20%20%20%20%20%20%20%20yt%20%3D%20unblinded_eval%5B%22pEC50_true%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20yp%20%3D%20unblinded_eval%5Bcol%5D.to_numpy()%0A%20%20%20%20%20%20%20%20hit%20%3D%20unblinded_eval.filter(pl.col(%22pEC50_true%22)%20%3E%3D%206.0)%0A%20%20%20%20%20%20%20%20inact%20%3D%20unblinded_eval.filter(pl.col(%22pEC50_true%22)%20%3C%204.0)%0A%20%20%20%20%20%20%20%20return%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22variant%22%3A%20name%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22MAE%22%3A%20round(mean_absolute_error(yt%2C%20yp)%2C%204)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22bias%22%3A%20round(float(np.mean(yp%20-%20yt))%2C%204)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22hitzone_MAE%22%3A%20round(mean_absolute_error(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20hit%5B%22pEC50_true%22%5D.to_numpy()%2C%20hit%5Bcol%5D.to_numpy())%2C%203)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22hitzone_bias%22%3A%20round(float((hit%5Bcol%5D%20-%20hit%5B%22pEC50_true%22%5D).mean())%2C%203)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22inactive_bias%22%3A%20round(float((inact%5Bcol%5D%20-%20inact%5B%22pEC50_true%22%5D).mean())%2C%203)%2C%0A%20%20%20%20%20%20%20%20%7D%0A%0A%20%20%20%20unblinded_summary%20%3D%20pl.DataFrame(%5B%0A%20%20%20%20%20%20%20%20_row(%22ensemble%20(raw)%22%2C%20%22pEC50_pred%22)%2C%0A%20%20%20%20%20%20%20%20_row(%22ensemble%20%2B%20linear%22%2C%20%22pred_linear%22)%2C%0A%20%20%20%20%20%20%20%20_row(%22ensemble%20%2B%20isotonic%22%2C%20%22pred_isotonic%22)%2C%0A%20%20%20%20%5D)%0A%0A%20%20%20%20with%20plt.style.context(%22seaborn-v0_8-whitegrid%22)%3A%0A%20%20%20%20%20%20%20%20_fig%2C%20_axes%20%3D%20plt.subplots(1%2C%203%2C%20figsize%3D(13.5%2C%204.6)%2C%20dpi%3D140%2C%20sharex%3DTrue%2C%20sharey%3DTrue)%0A%20%20%20%20%20%20%20%20_yt%20%3D%20unblinded_eval%5B%22pEC50_true%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20for%20_ax%2C%20(_name%2C%20_col)%20in%20zip(_axes%2C%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20(%22Raw%20ensemble%22%2C%20%22pEC50_pred%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20(%22%2B%20linear%20de-shrink%22%2C%20%22pred_linear%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20(%22%2B%20isotonic%22%2C%20%22pred_isotonic%22)%2C%0A%20%20%20%20%20%20%20%20%5D)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_yp%20%3D%20unblinded_eval%5B_col%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20_err%20%3D%20np.abs(_yp%20-%20_yt)%0A%20%20%20%20%20%20%20%20%20%20%20%20_sc%20%3D%20_ax.scatter(_yp%2C%20_yt%2C%20c%3D_err%2C%20cmap%3D%22RdYlGn_r%22%2C%20vmin%3D0%2C%20vmax%3D1.5%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20s%3D22%2C%20alpha%3D0.8%2C%20edgecolors%3D%22none%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_lims%20%3D%20%5Bmin(_yt.min()%2C%20_yp.min())%20-%200.2%2C%20max(_yt.max()%2C%20_yp.max())%20%2B%200.2%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.plot(_lims%2C%20_lims%2C%20%22k--%22%2C%20linewidth%3D0.8%2C%20zorder%3D0)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.axhline(6.0%2C%20color%3D%22%23888%22%2C%20linestyle%3D%22%3A%22%2C%20linewidth%3D0.8)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.set_xlim(_lims)%3B%20_ax.set_ylim(_lims)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.set_xlabel(%22Predicted%20pEC50%22%2C%20fontsize%3D10)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ax.set_title(f%22%7B_name%7D%5CnMAE%3D%7Bmean_absolute_error(_yt%2C%20_yp)%3A.3f%7D%22%2C%20fontsize%3D10)%0A%20%20%20%20%20%20%20%20_axes%5B0%5D.set_ylabel(%22True%20pEC50%20(unblinded)%22%2C%20fontsize%3D10)%0A%20%20%20%20%20%20%20%20_fig.colorbar(_sc%2C%20ax%3D_axes%2C%20label%3D%22%7Cerror%7C%22%2C%20fraction%3D0.025%2C%20pad%3D0.02)%0A%20%20%20%20%20%20%20%20_fig.suptitle(%22Calibration%20on%20the%20unblinded%20test%20set%22%2C%20fontsize%3D13%2C%20y%3D1.02)%0A%20%20%20%20%20%20%20%20_fig.savefig(PLOTS_DIR%20%2F%20%22unblinded_calibration.png%22%2C%20dpi%3D300%2C%20bbox_inches%3D%22tight%22)%0A%0A%20%20%20%20mo.vstack(%5B%0A%20%20%20%20%20%20%20%20mo.md(%22%23%23%23%20Calibration%20on%20the%20unblinded%20set%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22The%20decision-relevant%20question%20is%20the%20**hit%20zone**%3A%20does%20de-shrinking%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22reduce%20the%20underprediction%20of%20genuinely%20potent%20compounds%20without%20hurting%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22overall%20MAE%3F%22)%2C%0A%20%20%20%20%20%20%20%20mo.ui.table(unblinded_summary.to_pandas()%2C%20selection%3DNone%2C%20pagination%3DFalse)%2C%0A%20%20%20%20%20%20%20%20mo.as_html(_fig)%2C%0A%20%20%20%20%5D)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20best_calibration%2C%0A%20%20%20%20calib_isotonic%2C%0A%20%20%20%20calib_linear%2C%0A%20%20%20%20ens_test%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20winner%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Write%20the%20final%20submission%3A%20the%20CV-selected%20calibration%20of%20the%20ensemble%20%E2%94%80%E2%94%80%0A%20%20%20%20%23%20Only%20this%20single%2C%20CV-chosen%20strategy%20is%20applied%20to%20the%20held-out%20test%20set.%0A%20%20%20%20from%20pathlib%20import%20Path%20as%20_Path%0A%20%20%20%20_sub_dir%20%3D%20_Path(%22..%2Fsubmissions%22)%0A%20%20%20%20_sub_dir.mkdir(parents%3DTrue%2C%20exist_ok%3DTrue)%0A%0A%20%20%20%20_raw%20%3D%20ens_test%5B%22pEC50_pred%22%5D.to_numpy()%0A%20%20%20%20if%20best_calibration%20%3D%3D%20%22linear%22%3A%0A%20%20%20%20%20%20%20%20_cal%20%3D%20calib_linear.predict(_raw.reshape(-1%2C%201))%0A%20%20%20%20elif%20best_calibration%20%3D%3D%20%22isotonic%22%3A%0A%20%20%20%20%20%20%20%20_cal%20%3D%20calib_isotonic.predict(_raw)%0A%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20_cal%20%3D%20_raw%0A%0A%20%20%20%20_out%20%3D%20_sub_dir%20%2F%20f%226_ensemble_calibrated_%7Bbest_calibration%7D_submission.csv%22%0A%20%20%20%20ens_test.with_columns(pl.Series(%22pEC50%22%2C%20_cal)).select(%0A%20%20%20%20%20%20%20%20%5B%22SMILES%22%2C%20%22Molecule%20Name%22%2C%20%22pEC50%22%5D).write_csv(_out)%0A%0A%20%20%20%20if%20best_calibration%20%3D%3D%20%22raw%22%3A%0A%20%20%20%20%20%20%20%20_msg%20%3D%20(f%225%C3%975%20CV%20selected%20**%60raw%60**%20%E2%80%94%20no%20calibration%20variant%20beat%20the%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22uncalibrated%20ensemble%2C%20so%20the%20submission%20is%20unchanged%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22(written%20as%20%60%7B_out.name%7D%60%20for%20completeness).%22)%0A%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20_msg%20%3D%20(f%22Applied%20the%20CV-selected%20calibration%20**%60%7Bbest_calibration%7D%60**%20to%20all%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22%7Bens_test.shape%5B0%5D%7D%20test%20compounds%20%E2%86%92%20%60%7B_out.name%7D%60.%22)%0A%0A%20%20%20%20mo.md(f%22%22%22%0A%20%20%20%20%23%23%20Conclusion%0A%0A%20%20%20%20%7B_msg%7D%0A%0A%20%20%20%20The%20largest%20CV%20improvement%20of%20*any*%20strategy%20over%20its%20own%20baseline%20was%0A%20%20%20%20%60%7Bwinner%5B'analysis'%5D%7D%20%2F%20%7Bwinner%5B'best_variant'%5D%7D%60%20(%CE%94MAE%20%3D%20%7Bwinner%5B'delta_mae'%5D%3A%2B.4f%7D)%2C%0A%20%20%20%20but%20it%20acts%20on%20a%20single%20component%20model%20weaker%20than%20the%20ensemble%20and%20so%20is%20not%0A%20%20%20%20used%20for%20the%20submission.%0A%0A%20%20%20%20**What%20the%205%C3%975%20CV%20told%20us%2C%20and%20what%20held%20up%20on%20the%20unblinded%20set%3A**%0A%0A%20%20%20%20-%20Post-hoc%20**de-shrinking**%20is%20the%20cleanest%20lever%20for%20the%20regression-to-the-mean%0A%20%20%20%20%20%20bias%2C%20but%20its%20effect%20on%20overall%20MAE%20is%20small%20%E2%80%94%20the%20bias%20is%20conditional%20on%0A%20%20%20%20%20%20activity%2C%20and%20a%20global%201-D%20map%20can%20only%20partly%20correct%20it.%20Its%20value%20is%0A%20%20%20%20%20%20concentrated%20in%20the%20extreme%20bins%20(notably%20reduced%20hit-zone%20underprediction).%0A%20%20%20%20-%20**Reweighting**%20shifts%20error%20from%20the%20dense%20middle%20to%20the%20extremes%3B%20whether%0A%20%20%20%20%20%20that%20is%20worth%20a%20small%20overall-MAE%20cost%20depends%20on%20whether%20the%20hit%20zone%20is%20the%0A%20%20%20%20%20%20priority%20(for%20screening%2C%20it%20usually%20is).%20The%20natural%20next%20step%20is%20to%20rebuild%0A%20%20%20%20%20%20the%20ensemble%20from%20reweighted%20components.%0A%20%20%20%20-%20**Oversampling**%20is%20the%20data-level%20twin%20of%20reweighting%20and%20is%20the%20only%20one%20of%0A%20%20%20%20%20%20these%20that%20also%20reaches%20TabPFN%20and%20Macau%3B%20the%20MAE%2Fbias-vs-factor%20curves%20show%0A%20%20%20%20%20%20how%20far%20each%20model%20can%20be%20pushed%20before%20duplicate%20extremes%20start%20to%20hurt.%0A%20%20%20%20-%20The%20**semi-pure%20data**%20is%20too%20small%20to%20move%205%C3%975%20CV%20much%2C%20but%20the%20label-noise%0A%20%20%20%20%20%20audit%20is%20the%20more%20useful%20product%3A%20it%20quantifies%20how%20much%20of%20the%20residual%20error%0A%20%20%20%20%20%20is%20irreducible%20measurement%20noise%20rather%20than%20model%20error.%0A%20%20%20%20-%20**Filtering**%20is%20only%20worthwhile%20when%20a%20filter%20beats%20both%20the%20full-data%20baseline%0A%20%20%20%20%20%20*and*%20its%20size-matched%20random%20control%3B%20the%20paired%20bars%20make%20that%20test%20explicit%2C%0A%20%20%20%20%20%20separating%20genuine%20cleaning%20from%20the%20cost%20of%20simply%20training%20on%20fewer%20points.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%20Analysis%206%20%E2%80%94%20Improved%20ensembles%20(augmentation%20%2B%20counter-filtering)%0A%0A%20%20%20%20Analyses%202%E2%80%935%20tested%20reweighting%2C%20oversampling%2C%20augmentation%2C%20and%20filtering%0A%20%20%20%20*individually*%20on%20single-component%20models.%20This%20final%20analysis%20combines%20the%20two%0A%20%20%20%20most%20promising%20data-level%20interventions%20%E2%80%94%0A%0A%20%20%20%201.%20**Augmentation**%20%E2%80%94%20adding%20the%2096-compound%20semi-pure%20set%20to%20training.%0A%20%20%20%202.%20**Counter-filtering**%20%E2%80%94%20removing%20compounds%20whose%20counter-screen%20pEC50%20%E2%89%A5%20PXR%0A%20%20%20%20%20%20%20pEC50%20(non-selective%20hits).%0A%0A%20%20%20%20%E2%80%94%20and%20retrains%20*all%20five%20ensemble%20components*%20(Chemprop%2C%20CheMeleon%2C%20XGBoost%2C%0A%20%20%20%20Macau%2C%20TabPFN)%20on%20the%20cleaned%20%2B%20augmented%20training%20set.%0A%0A%20%20%20%20Two%20ensembles%20are%20built%20with%20the%20same%20model%20mixture%20and%20weights%20as%20the%0A%20%20%20%20submitted%20ensemble%20(%60cp5%C2%B7ch5%C2%B7rf0%C2%B7xg%E2%85%93%C2%B7mc1%C2%B7tf5%60)%3A%0A%0A%20%20%20%20%7C%20Ensemble%20%7C%20Models%20%7C%0A%20%20%20%20%7C---%7C---%7C%0A%20%20%20%20%7C%20**Default**%20%7C%20All%20components%20use%20default%20hyperparameters%20%7C%0A%20%20%20%20%7C%20**HPO**%20%7C%20All%20components%20use%20the%20HPO-tuned%20hyperparameters%20from%20notebook%204%20%7C%0A%0A%20%20%20%20Both%20ensembles%20are%20evaluated%20on%20the%20unblinded%20test%20set%20and%20compared%20against%20the%0A%20%20%20%20original%20submitted%20ensemble.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(Optional%2C%20Path%2C%20np%2C%20pl%2C%20shutil%2C%20subprocess%2C%20sys%2C%20tempfile%2C%20torch)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Chemprop%20model%20classes%20for%20Analysis%206%20(scratch%20%2B%20CheMeleon)%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%23%20Both%20mirror%20notebook%204's%20implementations%3A%20full%20HPO%20parameter%20surface%20passed%0A%20%20%20%20%23%20as%20CLI%20args.%20%20Shared%20helpers%20are%20cell-private%3B%20the%20two%20classes%20are%20exported.%0A%20%20%20%20_A6_BIN%20%3D%20Path(sys.executable).parent%20%2F%20%22chemprop%22%0A%20%20%20%20_A6_LOG%20%3D%20Path(%22..%2Flogs%2F6_a6_chemprop_cli.log%22)%0A%20%20%20%20_A6_LOG.parent.mkdir(parents%3DTrue%2C%20exist_ok%3DTrue)%0A%20%20%20%20_A6_SCRATCH_DIR%20%3D%20Path(tempfile.gettempdir())%20%2F%20%226_a6_scratch_model%22%0A%20%20%20%20_A6_CHEMELEON_DIR%20%3D%20Path(tempfile.gettempdir())%20%2F%20%226_a6_chemeleon_model%22%0A%0A%20%20%20%20def%20_a6_device()%20-%3E%20str%3A%0A%20%20%20%20%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20%22cuda%22%20if%20torch.cuda.is_available()%0A%20%20%20%20%20%20%20%20%20%20%20%20else%20%22mps%22%20if%20torch.backends.mps.is_available()%0A%20%20%20%20%20%20%20%20%20%20%20%20else%20%22cpu%22%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20def%20_a6_write_csv(%0A%20%20%20%20%20%20%20%20smiles%3A%20list%5Bstr%5D%2C%20targets%3A%20%22np.ndarray%20%7C%20None%22%2C%0A%20%20%20%20%20%20%20%20path%3A%20Path%2C%20target_col%3A%20str%2C%0A%20%20%20%20)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20cols%3A%20dict%20%3D%20%7B%22smiles%22%3A%20smiles%7D%0A%20%20%20%20%20%20%20%20if%20targets%20is%20not%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20cols%5Btarget_col%5D%20%3D%20targets.flatten().tolist()%0A%20%20%20%20%20%20%20%20pl.DataFrame(cols).write_csv(path)%0A%0A%20%20%20%20def%20_a6_run_cli(args%3A%20list%5Bstr%5D)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20cmd%20%3D%20%5Bstr(_A6_BIN)%5D%20%2B%20args%0A%20%20%20%20%20%20%20%20with%20open(_A6_LOG%2C%20%22a%22)%20as%20log%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20log.write(f%22%5Cn%7B'%3D'%20*%2060%7D%5CnCMD%3A%20%7B'%20'.join(cmd)%7D%5Cn%7B'%3D'%20*%2060%7D%5Cn%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20result%20%3D%20subprocess.run(cmd%2C%20stdout%3Dlog%2C%20stderr%3Dlog%2C%20text%3DTrue)%0A%20%20%20%20%20%20%20%20if%20result.returncode%20!%3D%200%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20print(%22%5Cn%22.join(_A6_LOG.read_text().splitlines()%5B-30%3A%5D))%0A%20%20%20%20%20%20%20%20%20%20%20%20raise%20RuntimeError(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22chemprop%20CLI%20failed%20(exit%20%7Bresult.returncode%7D).%20Log%3A%20%7B_A6_LOG%7D%22)%0A%0A%20%20%20%20class%20ChempropScratchModel%3A%0A%20%20%20%20%20%20%20%20%22%22%22Chemprop%20D-MPNN%20trained%20from%20scratch%20via%20CLI%2C%20with%20full%20HPO%20params.%22%22%22%0A%0A%20%20%20%20%20%20%20%20def%20__init__(%0A%20%20%20%20%20%20%20%20%20%20%20%20self%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pred_type%3A%20str%20%3D%20%22regression%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20model_dir%3A%20Path%20%3D%20_A6_SCRATCH_DIR%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20epochs%3A%20int%20%3D%2050%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20message_hidden_dim%3A%20int%20%3D%20300%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20depth%3A%20int%20%3D%203%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20dropout%3A%20float%20%3D%200.0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20ffn_hidden_dim%3A%20int%20%3D%20300%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20ffn_num_layers%3A%20int%20%3D%202%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20batch_size%3A%20int%20%3D%2064%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20init_lr%3A%20float%20%3D%201e-4%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20max_lr%3A%20float%20%3D%201e-3%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20final_lr%3A%20float%20%3D%201e-4%2C%0A%20%20%20%20%20%20%20%20)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20pred_type%20not%20in%20(%22regression%22%2C%20%22classification%22)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20raise%20ValueError(%22pred_type%20must%20be%20'regression'%20or%20'classification'%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20self.pred_type%20%3D%20pred_type%0A%20%20%20%20%20%20%20%20%20%20%20%20self.model_dir%20%3D%20model_dir%0A%20%20%20%20%20%20%20%20%20%20%20%20self.epochs%20%3D%20epochs%0A%20%20%20%20%20%20%20%20%20%20%20%20self.message_hidden_dim%20%3D%20message_hidden_dim%0A%20%20%20%20%20%20%20%20%20%20%20%20self.depth%20%3D%20depth%0A%20%20%20%20%20%20%20%20%20%20%20%20self.dropout%20%3D%20dropout%0A%20%20%20%20%20%20%20%20%20%20%20%20self.ffn_hidden_dim%20%3D%20ffn_hidden_dim%0A%20%20%20%20%20%20%20%20%20%20%20%20self.ffn_num_layers%20%3D%20ffn_num_layers%0A%20%20%20%20%20%20%20%20%20%20%20%20self.batch_size%20%3D%20batch_size%0A%20%20%20%20%20%20%20%20%20%20%20%20self.init_lr%20%3D%20init_lr%0A%20%20%20%20%20%20%20%20%20%20%20%20self.max_lr%20%3D%20max_lr%0A%20%20%20%20%20%20%20%20%20%20%20%20self.final_lr%20%3D%20final_lr%0A%20%20%20%20%20%20%20%20%20%20%20%20self.target_col%3A%20Optional%5Bstr%5D%20%3D%20None%0A%0A%20%20%20%20%20%20%20%20def%20_base_train_args(self%2C%20task_type%3A%20str%2C%20target_col%3A%20str)%20-%3E%20list%5Bstr%5D%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--smiles-columns%22%2C%20%22smiles%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--target-columns%22%2C%20target_col%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--task-type%22%2C%20task_type%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--accelerator%22%2C%20_a6_device()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--message-hidden-dim%22%2C%20str(self.message_hidden_dim)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--depth%22%2C%20str(self.depth)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--dropout%22%2C%20str(self.dropout)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--ffn-hidden-dim%22%2C%20str(self.ffn_hidden_dim)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--ffn-num-layers%22%2C%20str(self.ffn_num_layers)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--batch-size%22%2C%20str(self.batch_size)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--init-lr%22%2C%20str(self.init_lr)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--max-lr%22%2C%20str(self.max_lr)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--final-lr%22%2C%20str(self.final_lr)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D%0A%0A%20%20%20%20%20%20%20%20def%20train(%0A%20%20%20%20%20%20%20%20%20%20%20%20self%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20X_train%3A%20list%5Bstr%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y_train%3A%20np.ndarray%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20X_val%3A%20list%5Bstr%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y_val%3A%20np.ndarray%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20target_col%3A%20str%20%3D%20%22target%22%2C%0A%20%20%20%20%20%20%20%20)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20self.target_col%20%3D%20target_col%0A%20%20%20%20%20%20%20%20%20%20%20%20tmp%20%3D%20Path(tempfile.gettempdir())%0A%20%20%20%20%20%20%20%20%20%20%20%20train_csv%20%3D%20tmp%20%2F%20%226_a6_scratch_train.csv%22%0A%20%20%20%20%20%20%20%20%20%20%20%20val_csv%20%3D%20tmp%20%2F%20%226_a6_scratch_val.csv%22%0A%20%20%20%20%20%20%20%20%20%20%20%20_a6_write_csv(X_train%2C%20y_train%2C%20train_csv%2C%20target_col)%0A%20%20%20%20%20%20%20%20%20%20%20%20_a6_write_csv(X_val%2C%20y_val%2C%20val_csv%2C%20target_col)%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20self.model_dir.exists()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20shutil.rmtree(self.model_dir)%0A%20%20%20%20%20%20%20%20%20%20%20%20task_type%20%3D%20%22regression%22%20if%20self.pred_type%20%3D%3D%20%22regression%22%20else%20%22binary%22%0A%20%20%20%20%20%20%20%20%20%20%20%20_a6_run_cli(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22train%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--data-path%22%2C%20str(train_csv)%2C%20str(val_csv)%2C%20str(val_csv)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20*self._base_train_args(task_type%2C%20target_col)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--epochs%22%2C%20str(self.epochs)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--save-dir%22%2C%20str(self.model_dir)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20train_csv.unlink(missing_ok%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20val_csv.unlink(missing_ok%3DTrue)%0A%0A%20%20%20%20%20%20%20%20def%20predict(self%2C%20X_test%3A%20list%5Bstr%5D)%20-%3E%20np.ndarray%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20tmp%20%3D%20Path(tempfile.gettempdir())%0A%20%20%20%20%20%20%20%20%20%20%20%20test_csv%20%3D%20tmp%20%2F%20%226_a6_scratch_test.csv%22%0A%20%20%20%20%20%20%20%20%20%20%20%20pred_csv%20%3D%20tmp%20%2F%20%226_a6_scratch_preds.csv%22%0A%20%20%20%20%20%20%20%20%20%20%20%20model_pt%20%3D%20self.model_dir%20%2F%20%22model_0%22%20%2F%20%22best.pt%22%0A%20%20%20%20%20%20%20%20%20%20%20%20_a6_write_csv(X_test%2C%20None%2C%20test_csv%2C%20self.target_col)%0A%20%20%20%20%20%20%20%20%20%20%20%20_a6_run_cli(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22predict%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--test-path%22%2C%20str(test_csv)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--model-path%22%2C%20str(model_pt)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--preds-path%22%2C%20str(pred_csv)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20preds%20%3D%20pl.read_csv(pred_csv)%5Bself.target_col%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20test_csv.unlink(missing_ok%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20pred_csv.unlink(missing_ok%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20preds.flatten()%0A%0A%20%20%20%20class%20ChempropChemeleonModel%3A%0A%20%20%20%20%20%20%20%20%22%22%22Chemprop%20D-MPNN%20fine-tuned%20from%20CheMeleon%20pretrained%20backbone%20via%20CLI.%22%22%22%0A%0A%20%20%20%20%20%20%20%20def%20__init__(%0A%20%20%20%20%20%20%20%20%20%20%20%20self%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20pred_type%3A%20str%20%3D%20%22regression%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20model_dir%3A%20Path%20%3D%20_A6_CHEMELEON_DIR%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20epochs%3A%20int%20%3D%2050%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20dropout%3A%20float%20%3D%200.0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20ffn_hidden_dim%3A%20int%20%3D%20900%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20ffn_num_layers%3A%20int%20%3D%202%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20batch_size%3A%20int%20%3D%2064%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20init_lr%3A%20float%20%3D%201e-4%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20max_lr%3A%20float%20%3D%201e-3%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20final_lr%3A%20float%20%3D%201e-4%2C%0A%20%20%20%20%20%20%20%20)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20pred_type%20not%20in%20(%22regression%22%2C%20%22classification%22)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20raise%20ValueError(%22pred_type%20must%20be%20'regression'%20or%20'classification'%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20self.pred_type%20%3D%20pred_type%0A%20%20%20%20%20%20%20%20%20%20%20%20self.model_dir%20%3D%20model_dir%0A%20%20%20%20%20%20%20%20%20%20%20%20self.epochs%20%3D%20epochs%0A%20%20%20%20%20%20%20%20%20%20%20%20self.dropout%20%3D%20dropout%0A%20%20%20%20%20%20%20%20%20%20%20%20self.ffn_hidden_dim%20%3D%20ffn_hidden_dim%0A%20%20%20%20%20%20%20%20%20%20%20%20self.ffn_num_layers%20%3D%20ffn_num_layers%0A%20%20%20%20%20%20%20%20%20%20%20%20self.batch_size%20%3D%20batch_size%0A%20%20%20%20%20%20%20%20%20%20%20%20self.init_lr%20%3D%20init_lr%0A%20%20%20%20%20%20%20%20%20%20%20%20self.max_lr%20%3D%20max_lr%0A%20%20%20%20%20%20%20%20%20%20%20%20self.final_lr%20%3D%20final_lr%0A%20%20%20%20%20%20%20%20%20%20%20%20self.target_col%3A%20Optional%5Bstr%5D%20%3D%20None%0A%0A%20%20%20%20%20%20%20%20def%20_base_train_args(self%2C%20task_type%3A%20str%2C%20target_col%3A%20str)%20-%3E%20list%5Bstr%5D%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--smiles-columns%22%2C%20%22smiles%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--target-columns%22%2C%20target_col%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--task-type%22%2C%20task_type%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--accelerator%22%2C%20_a6_device()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--dropout%22%2C%20str(self.dropout)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--ffn-hidden-dim%22%2C%20str(self.ffn_hidden_dim)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--ffn-num-layers%22%2C%20str(self.ffn_num_layers)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--batch-size%22%2C%20str(self.batch_size)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--init-lr%22%2C%20str(self.init_lr)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--max-lr%22%2C%20str(self.max_lr)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--final-lr%22%2C%20str(self.final_lr)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D%0A%0A%20%20%20%20%20%20%20%20def%20train(%0A%20%20%20%20%20%20%20%20%20%20%20%20self%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20X_train%3A%20list%5Bstr%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y_train%3A%20np.ndarray%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20X_val%3A%20list%5Bstr%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y_val%3A%20np.ndarray%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20target_col%3A%20str%20%3D%20%22target%22%2C%0A%20%20%20%20%20%20%20%20)%20-%3E%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20self.target_col%20%3D%20target_col%0A%20%20%20%20%20%20%20%20%20%20%20%20tmp%20%3D%20Path(tempfile.gettempdir())%0A%20%20%20%20%20%20%20%20%20%20%20%20train_csv%20%3D%20tmp%20%2F%20%226_a6_che_train.csv%22%0A%20%20%20%20%20%20%20%20%20%20%20%20val_csv%20%3D%20tmp%20%2F%20%226_a6_che_val.csv%22%0A%20%20%20%20%20%20%20%20%20%20%20%20_a6_write_csv(X_train%2C%20y_train%2C%20train_csv%2C%20target_col)%0A%20%20%20%20%20%20%20%20%20%20%20%20_a6_write_csv(X_val%2C%20y_val%2C%20val_csv%2C%20target_col)%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20self.model_dir.exists()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20shutil.rmtree(self.model_dir)%0A%20%20%20%20%20%20%20%20%20%20%20%20task_type%20%3D%20%22regression%22%20if%20self.pred_type%20%3D%3D%20%22regression%22%20else%20%22binary%22%0A%20%20%20%20%20%20%20%20%20%20%20%20_a6_run_cli(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22train%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--data-path%22%2C%20str(train_csv)%2C%20str(val_csv)%2C%20str(val_csv)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20*self._base_train_args(task_type%2C%20target_col)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--epochs%22%2C%20str(self.epochs)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--from-foundation%22%2C%20%22CHEMELEON%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--save-dir%22%2C%20str(self.model_dir)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20train_csv.unlink(missing_ok%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20val_csv.unlink(missing_ok%3DTrue)%0A%0A%20%20%20%20%20%20%20%20def%20predict(self%2C%20X_test%3A%20list%5Bstr%5D)%20-%3E%20np.ndarray%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20tmp%20%3D%20Path(tempfile.gettempdir())%0A%20%20%20%20%20%20%20%20%20%20%20%20test_csv%20%3D%20tmp%20%2F%20%226_a6_che_test.csv%22%0A%20%20%20%20%20%20%20%20%20%20%20%20pred_csv%20%3D%20tmp%20%2F%20%226_a6_che_preds.csv%22%0A%20%20%20%20%20%20%20%20%20%20%20%20model_pt%20%3D%20self.model_dir%20%2F%20%22model_0%22%20%2F%20%22best.pt%22%0A%20%20%20%20%20%20%20%20%20%20%20%20_a6_write_csv(X_test%2C%20None%2C%20test_csv%2C%20self.target_col)%0A%20%20%20%20%20%20%20%20%20%20%20%20_a6_run_cli(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22predict%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--test-path%22%2C%20str(test_csv)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--model-path%22%2C%20str(model_pt)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22--preds-path%22%2C%20str(pred_csv)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20preds%20%3D%20pl.read_csv(pred_csv)%5Bself.target_col%5D.to_numpy()%0A%20%20%20%20%20%20%20%20%20%20%20%20test_csv.unlink(missing_ok%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20pred_csv.unlink(missing_ok%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20preds.flatten()%0A%0A%20%20%20%20return%20ChempropChemeleonModel%2C%20ChempropScratchModel%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20BoostedTreesModel%2C%0A%20%20%20%20ChempropChemeleonModel%2C%0A%20%20%20%20ChempropScratchModel%2C%0A%20%20%20%20MacauModel%2C%0A%20%20%20%20Path%2C%0A%20%20%20%20chemeleon_embed%2C%0A%20%20%20%20counter_remove%2C%0A%20%20%20%20extract_fp_matrix%2C%0A%20%20%20%20filter_train%2C%0A%20%20%20%20gc%2C%0A%20%20%20%20generate_fingerprint%2C%0A%20%20%20%20mean_absolute_error%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20np%2C%0A%20%20%20%20pl%2C%0A%20%20%20%20semipure%2C%0A%20%20%20%20tabpfn_predict%2C%0A%20%20%20%20unblinded_eval%2C%0A)%3A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Prepare%20the%20augmented%20%2B%20counter-filtered%20training%20set%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_TARGET_COL%20%3D%20%22pEC50_dr%22%0A%20%20%20%20_SEED%20%3D%2042%0A%20%20%20%20_SUB_DIR%20%3D%20Path(%22..%2Fsubmissions%22)%0A%20%20%20%20_SUB_DIR.mkdir(parents%3DTrue%2C%20exist_ok%3DTrue)%0A%0A%20%20%20%20_SUB_DEFAULT%20%3D%20_SUB_DIR%20%2F%20%226_ens_default_augfilt_submission.csv%22%0A%20%20%20%20_SUB_HPO%20%3D%20_SUB_DIR%20%2F%20%226_ens_hpo_augfilt_submission.csv%22%0A%0A%20%20%20%20if%20_SUB_DEFAULT.exists()%20and%20_SUB_HPO.exists()%3A%0A%20%20%20%20%20%20%20%20_ens_subs%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22default%22%3A%20pl.read_csv(_SUB_DEFAULT)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22hpo%22%3A%20pl.read_csv(_SUB_HPO)%2C%0A%20%20%20%20%20%20%20%20%7D%0A%20%20%20%20%20%20%20%20print(f%22Found%20%7B_SUB_DEFAULT.name%7D%20and%20%7B_SUB_HPO.name%7D%20%E2%80%94%20skipping%20Analysis%206%20training.%22)%0A%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%23%20%E2%94%80%E2%94%80%20Ensemble%20weights%20(same%20as%20submitted%3A%20cp5%C2%B7ch5%C2%B7rf0%C2%B7xg%E2%85%93%C2%B7mc1%C2%B7tf5)%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%20%20%20%20_ENS_WEIGHTS%20%3D%20%7B%22cp%22%3A%205.0%2C%20%22ch%22%3A%205.0%2C%20%22xg%22%3A%201.0%20%2F%203.0%2C%20%22mc%22%3A%201.0%2C%20%22tf%22%3A%205.0%7D%0A%20%20%20%20%20%20%20%20_W_TOTAL%20%3D%20sum(_ENS_WEIGHTS.values())%0A%0A%20%20%20%20%20%20%20%20%23%20Counter-filter%3A%20remove%20compounds%20where%20counter%20pEC50%20%3E%3D%20PXR%20pEC50%0A%20%20%20%20%20%20%20%20_rm_idx%20%3D%20counter_remove(filter_train)%0A%20%20%20%20%20%20%20%20_keep_idx%20%3D%20sorted(set(range(filter_train.shape%5B0%5D))%20-%20_rm_idx)%0A%20%20%20%20%20%20%20%20_train_filtered%20%3D%20filter_train%5B_keep_idx%5D.select(%0A%20%20%20%20%20%20%20%20%20%20%20%20%5B%22smiles%22%2C%20%22inchikey%22%2C%20%22molecule_names%22%2C%20_TARGET_COL%5D)%0A%20%20%20%20%20%20%20%20print(f%22Counter-filtered%3A%20%7Bfilter_train.shape%5B0%5D%7D%20%E2%86%92%20%7B_train_filtered.shape%5B0%5D%7D%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22(removed%20%7Blen(_rm_idx)%7D)%22)%0A%0A%20%20%20%20%20%20%20%20%23%20Augment%20with%20semi-pure%20compounds%20(excluding%20any%20already%20in%20training)%0A%20%20%20%20%20%20%20%20_existing_iks%20%3D%20set(_train_filtered%5B%22inchikey%22%5D.to_list())%0A%20%20%20%20%20%20%20%20_sp_new%20%3D%20semipure.filter(~pl.col(%22inchikey%22).is_in(list(_existing_iks)))%0A%20%20%20%20%20%20%20%20_train_aug%20%3D%20pl.concat(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20_train_filtered%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20_sp_new.select(%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22smiles%22%2C%20%22inchikey%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.lit(%22%22).alias(%22molecule_names%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20pl.col(%22pEC50_corrected%22).alias(_TARGET_COL)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D)%2C%0A%20%20%20%20%20%20%20%20%5D)%0A%20%20%20%20%20%20%20%20print(f%22Augmented%3A%20%7B_train_filtered.shape%5B0%5D%7D%20%2B%20%7B_sp_new.shape%5B0%5D%7D%20semi-pure%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20f%22%3D%20%7B_train_aug.shape%5B0%5D%7D%22)%0A%0A%20%20%20%20%20%20%20%20%23%2010%25%20val%20split%20for%20early%20stopping%0A%20%20%20%20%20%20%20%20_rng%20%3D%20np.random.default_rng(_SEED)%0A%20%20%20%20%20%20%20%20_n%20%3D%20len(_train_aug)%0A%20%20%20%20%20%20%20%20_val_idx%20%3D%20_rng.choice(_n%2C%20size%3Dint(_n%20*%200.1)%2C%20replace%3DFalse)%0A%20%20%20%20%20%20%20%20_tr_idx%20%3D%20np.setdiff1d(np.arange(_n)%2C%20_val_idx)%0A%20%20%20%20%20%20%20%20_train_sub%20%3D%20_train_aug%5B_tr_idx.tolist()%5D%0A%20%20%20%20%20%20%20%20_val_sub%20%3D%20_train_aug%5B_val_idx.tolist()%5D%0A%0A%20%20%20%20%20%20%20%20%23%20Test%20set%0A%20%20%20%20%20%20%20%20_test_df%20%3D%20pl.read_csv(%22..%2Fdata%2Fraw%2F20260409%2Fdose_response_test.csv%22)%0A%20%20%20%20%20%20%20%20_test_smiles%20%3D%20_test_df%5B%22SMILES%22%5D.to_list()%0A%20%20%20%20%20%20%20%20_test_names%20%3D%20_test_df%5B%22Molecule%20Name%22%5D.to_list()%0A%0A%20%20%20%20%20%20%20%20%23%20%E2%94%80%E2%94%80%20Features%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%20%20%20%20%23%20Mordred%20fingerprints%0A%20%20%20%20%20%20%20%20_fp_tr%20%3D%20generate_fingerprint(_train_sub%2C%20%22mordred%22)%0A%20%20%20%20%20%20%20%20_fp_va%20%3D%20generate_fingerprint(_val_sub%2C%20%22mordred%22)%0A%20%20%20%20%20%20%20%20_fp_te%20%3D%20generate_fingerprint(%0A%20%20%20%20%20%20%20%20%20%20%20%20pl.DataFrame(%7B%22smiles%22%3A%20_test_smiles%2C%20%22inchikey%22%3A%20_test_names%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22molecule_names%22%3A%20_test_names%7D)%2C%20%22mordred%22)%0A%20%20%20%20%20%20%20%20_Xtr_mor%20%3D%20extract_fp_matrix(_fp_tr%2C%20%22mordred%22)%0A%20%20%20%20%20%20%20%20_Xva_mor%20%3D%20extract_fp_matrix(_fp_va%2C%20%22mordred%22)%0A%20%20%20%20%20%20%20%20_Xte_mor%20%3D%20extract_fp_matrix(_fp_te%2C%20%22mordred%22)%0A%20%20%20%20%20%20%20%20del%20_fp_tr%2C%20_fp_va%2C%20_fp_te%0A%20%20%20%20%20%20%20%20%23%20Drop%20NaN%20columns%20(consistent%20mask%20across%20train%2Fval%2Ftest)%0A%20%20%20%20%20%20%20%20_valid_mor%20%3D%20~np.isnan(_Xtr_mor).any(axis%3D0)%0A%20%20%20%20%20%20%20%20_Xtr_mor%20%3D%20_Xtr_mor%5B%3A%2C%20_valid_mor%5D%0A%20%20%20%20%20%20%20%20_Xva_mor%20%3D%20_Xva_mor%5B%3A%2C%20_valid_mor%5D%0A%20%20%20%20%20%20%20%20_Xte_mor%20%3D%20_Xte_mor%5B%3A%2C%20_valid_mor%5D%0A%0A%20%20%20%20%20%20%20%20%23%20CheMeleon%20embeddings%0A%20%20%20%20%20%20%20%20_Xtr_che_full%2C%20_Xte_che%20%3D%20chemeleon_embed(%0A%20%20%20%20%20%20%20%20%20%20%20%20_train_aug%5B%22smiles%22%5D.to_list()%2C%20_test_smiles%2C%20prefix%3D%22a6_ens%22)%0A%20%20%20%20%20%20%20%20_Xtr_che%20%3D%20_Xtr_che_full%5B_tr_idx%5D%0A%20%20%20%20%20%20%20%20_Xva_che%20%3D%20_Xtr_che_full%5B_val_idx%5D%0A%0A%20%20%20%20%20%20%20%20_y_tr%20%3D%20_train_sub%5B_TARGET_COL%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_y_va%20%3D%20_val_sub%5B_TARGET_COL%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_y_full%20%3D%20_train_aug%5B_TARGET_COL%5D.to_numpy()%0A%0A%20%20%20%20%20%20%20%20%23%20%E2%94%80%E2%94%80%20HPO%20best%20parameters%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%20%20%20%20_CHEMPROP_HPO%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22epochs%22%3A%2050%2C%20%22batch_size%22%3A%2032%2C%20%22dropout%22%3A%200.0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22ffn_hidden_dim%22%3A%201024%2C%20%22ffn_num_layers%22%3A%204%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22max_lr%22%3A%200.00014521767021847913%2C%0A%20%20%20%20%20%20%20%20%7D%0A%20%20%20%20%20%20%20%20_CHEMELEON_HPO%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20k%3A%20v%20for%20k%2C%20v%20in%20_CHEMPROP_HPO.items()%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20k%20not%20in%20(%22message_hidden_dim%22%2C%20%22depth%22)%0A%20%20%20%20%20%20%20%20%7D%0A%20%20%20%20%20%20%20%20_XGB_HPO%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22colsample_bytree%22%3A%200.7280642324424045%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22learning_rate%22%3A%200.019940375676697552%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22max_depth%22%3A%205%2C%20%22n_estimators%22%3A%20800%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22reg_alpha%22%3A%200.006750974160271454%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22subsample%22%3A%200.7328142736138993%2C%0A%20%20%20%20%20%20%20%20%7D%0A%20%20%20%20%20%20%20%20_MACAU_HPO%20%3D%20%7B%22burnin%22%3A%20400%2C%20%22nsamples%22%3A%20700%2C%20%22num_latent%22%3A%208%7D%0A%0A%20%20%20%20%20%20%20%20%23%20%E2%94%80%E2%94%80%20Train%20both%20ensemble%20variants%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20%20%20%20%20_ens_subs%20%3D%20%7B%7D%0A%0A%20%20%20%20%20%20%20%20for%20_variant%2C%20_config%20in%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20(%22default%22%2C%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22cp_kw%22%3A%20%7B%22epochs%22%3A%2050%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22ch_kw%22%3A%20%7B%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22xg_kw%22%3A%20%7B%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22mc_kw%22%3A%20%7B%22seed%22%3A%20_SEED%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20(%22hpo%22%2C%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22cp_kw%22%3A%20_CHEMPROP_HPO%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22ch_kw%22%3A%20_CHEMELEON_HPO%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22xg_kw%22%3A%20_XGB_HPO%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22mc_kw%22%3A%20%7B%22seed%22%3A%20_SEED%2C%20**_MACAU_HPO%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D)%2C%0A%20%20%20%20%20%20%20%20%5D%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20print(f%22%5Cn%7B'%3D'%20*%2050%7D%5CnTraining%20%7B_variant%7D%20ensemble%5Cn%7B'%3D'%20*%2050%7D%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_preds%3A%20dict%5Bstr%2C%20np.ndarray%5D%20%3D%20%7B%7D%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20cp%20%E2%80%94%20Chemprop%20scratch%0A%20%20%20%20%20%20%20%20%20%20%20%20print(f%22%20%20%5B%7B_variant%7D%5D%20cp%20%E2%80%94%20Chemprop%20scratch%20%E2%80%A6%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_cp%20%3D%20ChempropScratchModel(pred_type%3D%22regression%22%2C%20**_config%5B%22cp_kw%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20_cp.train(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_train_sub%5B%22smiles%22%5D.to_list()%2C%20_y_tr%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_val_sub%5B%22smiles%22%5D.to_list()%2C%20_y_va%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20target_col%3D_TARGET_COL%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20_preds%5B%22cp%22%5D%20%3D%20_cp.predict(_test_smiles)%0A%20%20%20%20%20%20%20%20%20%20%20%20del%20_cp%3B%20gc.collect()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20ch%20%E2%80%94%20CheMeleon%20fine-tuned%0A%20%20%20%20%20%20%20%20%20%20%20%20print(f%22%20%20%5B%7B_variant%7D%5D%20ch%20%E2%80%94%20CheMeleon%20%E2%80%A6%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ch%20%3D%20ChempropChemeleonModel(pred_type%3D%22regression%22%2C%20**_config%5B%22ch_kw%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ch.train(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_train_sub%5B%22smiles%22%5D.to_list()%2C%20_y_tr%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_val_sub%5B%22smiles%22%5D.to_list()%2C%20_y_va%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20target_col%3D_TARGET_COL%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20_preds%5B%22ch%22%5D%20%3D%20_ch.predict(_test_smiles)%0A%20%20%20%20%20%20%20%20%20%20%20%20del%20_ch%3B%20gc.collect()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20xg%20%E2%80%94%20XGBoost%20on%20Mordred%0A%20%20%20%20%20%20%20%20%20%20%20%20print(f%22%20%20%5B%7B_variant%7D%5D%20xg%20%E2%80%94%20XGBoost%20Mordred%20%E2%80%A6%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_xg%20%3D%20BoostedTreesModel(pred_type%3D%22regression%22%2C%20**_config%5B%22xg_kw%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20_xg.train(_Xtr_mor%2C%20_y_tr%2C%20_Xva_mor%2C%20_y_va)%0A%20%20%20%20%20%20%20%20%20%20%20%20_preds%5B%22xg%22%5D%20%3D%20_xg.predict(_Xte_mor)%0A%20%20%20%20%20%20%20%20%20%20%20%20del%20_xg%3B%20gc.collect()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20mc%20%E2%80%94%20Macau%20on%20CheMeleon%20FP%20(uses%20full%20training%20set%2C%20no%20early%20stopping)%0A%20%20%20%20%20%20%20%20%20%20%20%20print(f%22%20%20%5B%7B_variant%7D%5D%20mc%20%E2%80%94%20Macau%20CheMeleon%20%E2%80%A6%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_mc%20%3D%20MacauModel(**_config%5B%22mc_kw%22%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20_mc.train(_Xtr_che_full%2C%20_y_full)%0A%20%20%20%20%20%20%20%20%20%20%20%20_preds%5B%22mc%22%5D%20%3D%20_mc.predict(_Xte_che)%0A%20%20%20%20%20%20%20%20%20%20%20%20del%20_mc%3B%20gc.collect()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20tf%20%E2%80%94%20TabPFN%20on%20CheMeleon%20FP%20(no%20HPO%20%E2%80%94%20in-context%20model)%0A%20%20%20%20%20%20%20%20%20%20%20%20print(f%22%20%20%5B%7B_variant%7D%5D%20tf%20%E2%80%94%20TabPFN%20CheMeleon%20%E2%80%A6%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20_preds%5B%22tf%22%5D%20%3D%20tabpfn_predict(_Xtr_che_full%2C%20_y_full%2C%20_Xte_che)%0A%20%20%20%20%20%20%20%20%20%20%20%20gc.collect()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20Compute%20weighted%20ensemble%20and%20save%20submission%0A%20%20%20%20%20%20%20%20%20%20%20%20_ens_pred%20%3D%20sum(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_preds%5Btag%5D%20*%20(w%20%2F%20_W_TOTAL)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20tag%2C%20w%20in%20_ENS_WEIGHTS.items()%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20_sub_df%20%3D%20pl.DataFrame(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22SMILES%22%3A%20_test_smiles%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Molecule%20Name%22%3A%20_test_names%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22pEC50%22%3A%20_ens_pred.tolist()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20%20%20%20%20_sub_path%20%3D%20_SUB_DIR%20%2F%20f%226_ens_%7B_variant%7D_augfilt_submission.csv%22%0A%20%20%20%20%20%20%20%20%20%20%20%20_sub_df.write_csv(_sub_path)%0A%20%20%20%20%20%20%20%20%20%20%20%20_ens_subs%5B_variant%5D%20%3D%20_sub_df%0A%20%20%20%20%20%20%20%20%20%20%20%20print(f%22%20%20%E2%86%92%20%7B_sub_path.name%7D%22)%0A%0A%20%20%20%20%23%20%E2%94%80%E2%94%80%20Evaluate%20on%20unblinded%20test%20set%20%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%E2%94%80%0A%20%20%20%20_eval_rows%20%3D%20%5B%5D%0A%20%20%20%20_yt%20%3D%20unblinded_eval%5B%22pEC50_true%22%5D.to_numpy()%0A%0A%20%20%20%20%23%20Original%20ensemble%0A%20%20%20%20_yp_orig%20%3D%20unblinded_eval%5B%22pEC50_pred%22%5D.to_numpy()%0A%20%20%20%20_hit_orig%20%3D%20unblinded_eval.filter(pl.col(%22pEC50_true%22)%20%3E%3D%206.0)%0A%20%20%20%20_eval_rows.append(%7B%0A%20%20%20%20%20%20%20%20%22ensemble%22%3A%20%22original%20(nb%204)%22%2C%0A%20%20%20%20%20%20%20%20%22MAE%22%3A%20round(mean_absolute_error(_yt%2C%20_yp_orig)%2C%204)%2C%0A%20%20%20%20%20%20%20%20%22hitzone_bias%22%3A%20round(float(%0A%20%20%20%20%20%20%20%20%20%20%20%20(_hit_orig%5B%22pEC50_pred%22%5D%20-%20_hit_orig%5B%22pEC50_true%22%5D).mean())%2C%203)%2C%0A%20%20%20%20%7D)%0A%0A%20%20%20%20for%20_variant%20in%20%5B%22default%22%2C%20%22hpo%22%5D%3A%0A%20%20%20%20%20%20%20%20_pred_df%20%3D%20_ens_subs%5B_variant%5D.select(%5B%22Molecule%20Name%22%2C%20%22pEC50%22%5D).rename(%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%22pEC50%22%3A%20%22pEC50_new%22%7D)%0A%20%20%20%20%20%20%20%20_joined%20%3D%20unblinded_eval.join(_pred_df%2C%20on%3D%22Molecule%20Name%22%2C%20how%3D%22inner%22)%0A%20%20%20%20%20%20%20%20_yt_j%20%3D%20_joined%5B%22pEC50_true%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_yp_j%20%3D%20_joined%5B%22pEC50_new%22%5D.to_numpy()%0A%20%20%20%20%20%20%20%20_hit%20%3D%20_joined.filter(pl.col(%22pEC50_true%22)%20%3E%3D%206.0)%0A%20%20%20%20%20%20%20%20_eval_rows.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22ensemble%22%3A%20f%22aug%2Bfilt%20(%7B_variant%7D)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22MAE%22%3A%20round(mean_absolute_error(_yt_j%2C%20_yp_j)%2C%204)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22hitzone_bias%22%3A%20round(float(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(_hit%5B%22pEC50_new%22%5D%20-%20_hit%5B%22pEC50_true%22%5D).mean())%2C%203)%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%0A%20%20%20%20ens_comparison%20%3D%20pl.DataFrame(_eval_rows)%0A%0A%20%20%20%20mo.vstack(%5B%0A%20%20%20%20%20%20%20%20mo.md(%22%23%23%23%20Improved%20ensemble%20comparison%20on%20unblinded%20test%20set%5Cn%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Both%20new%20ensembles%20use%20the%20same%20weights%20as%20the%20original%20submission%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22(%60cp5%C2%B7ch5%C2%B7xg%E2%85%93%C2%B7mc1%C2%B7tf5%60)%20but%20are%20retrained%20on%20counter-filtered%20%2B%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22semi-pure-augmented%20data.%22)%2C%0A%20%20%20%20%20%20%20%20mo.ui.table(ens_comparison.to_pandas()%2C%20selection%3DNone%2C%20pagination%3DFalse)%2C%0A%20%20%20%20%5D)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
8f3f694750cac74ddfddb34eaea75531