import%20marimo%0A%0A__generated_with%20%3D%20%220.23.14%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%0A%20%20%20%20return%20(mo%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%20Group%2027%20%E2%80%94%20Course-Drop%20Prediction%20(Nova%20Academy)%0A%0A%20%20%20%20**Submitters%3A**%20Rotem%20David%20Semah%20(ID%3A%20%60211396593%60)%20%C2%B7%20Ron%20Drach%20(ID%3A%20%60213915499%60)%0A%0A%20%20%20%20---%0A%0A%20%20%20%20This%20notebook%20follows%20the%20project%20from%20understanding%20the%20data%20through%20preparation%2C%20modelling%2C%20evaluation%2C%20and%20interpretation.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20from%20functools%20import%20wraps%0A%20%20%20%20from%20inspect%20import%20getsource%0A%20%20%20%20import%20warnings%0A%20%20%20%20from%20pathlib%20import%20Path%0A%20%20%20%20from%20textwrap%20import%20dedent%0A%20%20%20%20from%20joblib%20import%20dump%2C%20hash%20as%20joblib_hash%2C%20load%0A%20%20%20%20warnings.filterwarnings('ignore')%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20pandas%20as%20pd%0A%20%20%20%20import%20matplotlib.pyplot%20as%20plt%0A%20%20%20%20from%20matplotlib_inline.backend_inline%20import%20set_matplotlib_formats%0A%20%20%20%20import%20seaborn%20as%20sns%0A%20%20%20%20import%20shap%0A%20%20%20%20from%20catboost%20import%20CatBoostClassifier%0A%20%20%20%20from%20lightgbm%20import%20LGBMClassifier%0A%20%20%20%20from%20scipy.stats%20import%20rankdata%0A%20%20%20%20from%20sklearn.linear_model%20import%20LogisticRegression%0A%20%20%20%20from%20sklearn.metrics%20import%20average_precision_score%2C%20roc_auc_score%2C%20classification_report%2C%20RocCurveDisplay%2C%20PrecisionRecallDisplay%2C%20ConfusionMatrixDisplay%0A%20%20%20%20from%20sklearn.model_selection%20import%20train_test_split%0A%20%20%20%20from%20sklearn.neural_network%20import%20MLPClassifier%0A%20%20%20%20from%20sklearn.preprocessing%20import%20OneHotEncoder%2C%20StandardScaler%0A%20%20%20%20from%20xgboost%20import%20XGBClassifier%0A%20%20%20%20from%20IPython.display%20import%20display%0A%0A%20%20%20%20def%20cache(fn)%3A%0A%20%20%20%20%20%20%20%20cache_dir%20%3D%20Path('.cache%2Fjoblib')%20%2F%20fn.__name__%0A%20%20%20%20%20%20%20%20source_hash%20%3D%20joblib_hash(dedent(getsource(fn)))%0A%0A%20%20%20%20%20%20%20%20%40wraps(fn)%0A%20%20%20%20%20%20%20%20def%20wrapped(*args%2C%20**kwargs)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20path%20%3D%20cache_dir%20%2F%20f'%7Bjoblib_hash((source_hash%2C%20args%2C%20kwargs))%7D.joblib'%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20path.exists()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20return%20load(path)%0A%20%20%20%20%20%20%20%20%20%20%20%20result%20%3D%20fn(*args%2C%20**kwargs)%0A%20%20%20%20%20%20%20%20%20%20%20%20path.parent.mkdir(parents%3DTrue%2C%20exist_ok%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20dump(result%2C%20path)%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20result%0A%20%20%20%20%20%20%20%20return%20wrapped%0A%20%20%20%20sns.set_theme(style%3D'whitegrid'%2C%20palette%3D'colorblind'%2C%20rc%3D%7B'figure.constrained_layout.use'%3A%20False%7D)%0A%20%20%20%20set_matplotlib_formats('png')%0A%20%20%20%20pd.set_option('display.max_columns'%2C%20None)%0A%20%20%20%20TRAIN_PATH%20%3D%20'data%2FTrain_Data.csv'%0A%20%20%20%20TEST_PATH%20%3D%20'data%2FTest_Data_No_Target.csv'%0A%20%20%20%20TARGET%20%3D%20'Dropped_Course'%0A%20%20%20%20SEED%20%3D%2042%0A%0A%20%20%20%20def%20load_raw(path%3A%20str)%20-%3E%20pd.DataFrame%3A%0A%20%20%20%20%20%20%20%20return%20pd.read_csv(path%2C%20parse_dates%3D%5B'Course_Start_Date'%5D)%0A%0A%20%20%20%20def%20show(fig%3DNone)%3A%0A%20%20%20%20%20%20%20%20if%20fig%20is%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20fig%20%3D%20plt.gcf()%0A%20%20%20%20%20%20%20%20display(fig)%0A%20%20%20%20%20%20%20%20plt.close(fig)%0A%0A%20%20%20%20def%20subplot_grid(nrows%3D1%2C%20ncols%3D1%2C%20**kwargs)%3A%0A%20%20%20%20%20%20%20%20kwargs.setdefault('layout'%2C%20'constrained')%0A%20%20%20%20%20%20%20%20kwargs.setdefault('figsize'%2C%20(10%2C%204.375%20*%20nrows))%0A%20%20%20%20%20%20%20%20return%20plt.subplots(nrows%2C%20ncols%2C%20**kwargs)%0A%0A%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20CatBoostClassifier%2C%0A%20%20%20%20%20%20%20%20ConfusionMatrixDisplay%2C%0A%20%20%20%20%20%20%20%20LGBMClassifier%2C%0A%20%20%20%20%20%20%20%20LogisticRegression%2C%0A%20%20%20%20%20%20%20%20MLPClassifier%2C%0A%20%20%20%20%20%20%20%20OneHotEncoder%2C%0A%20%20%20%20%20%20%20%20PrecisionRecallDisplay%2C%0A%20%20%20%20%20%20%20%20RocCurveDisplay%2C%0A%20%20%20%20%20%20%20%20SEED%2C%0A%20%20%20%20%20%20%20%20StandardScaler%2C%0A%20%20%20%20%20%20%20%20TARGET%2C%0A%20%20%20%20%20%20%20%20TEST_PATH%2C%0A%20%20%20%20%20%20%20%20TRAIN_PATH%2C%0A%20%20%20%20%20%20%20%20XGBClassifier%2C%0A%20%20%20%20%20%20%20%20average_precision_score%2C%0A%20%20%20%20%20%20%20%20cache%2C%0A%20%20%20%20%20%20%20%20classification_report%2C%0A%20%20%20%20%20%20%20%20display%2C%0A%20%20%20%20%20%20%20%20load_raw%2C%0A%20%20%20%20%20%20%20%20np%2C%0A%20%20%20%20%20%20%20%20pd%2C%0A%20%20%20%20%20%20%20%20plt%2C%0A%20%20%20%20%20%20%20%20rankdata%2C%0A%20%20%20%20%20%20%20%20roc_auc_score%2C%0A%20%20%20%20%20%20%20%20shap%2C%0A%20%20%20%20%20%20%20%20show%2C%0A%20%20%20%20%20%20%20%20sns%2C%0A%20%20%20%20%20%20%20%20subplot_grid%2C%0A%20%20%20%20%20%20%20%20train_test_split%2C%0A%20%20%20%20)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%201.%20Business%20understanding%0A%0A%20%20%20%20Nova%20Academy%20prepares%20cloud%20environments%2C%20catering%2C%20equipment%2C%20and%20classroom%20capacity%20before%20each%20B2B%20course%20begins.%20A%20cancellation%20therefore%20wastes%20prepared%20resources%20and%20can%20leave%20capacity%20that%20could%20have%20been%20offered%20to%20another%20group.%0A%0A%20%20%20%20Our%20goal%20is%20to%20estimate%20cancellation%20risk%20for%20new%20registrations%20early%20enough%20to%20support%20operational%20decisions.%20The%20assignment%20requires%20a%20continuous%20%60Drop_Probability%60%20output%20and%20evaluates%20its%20ranking%20quality%20with%20ROC-AUC%3B%20the%20minimum%20required%20AUC%20is%200.70.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%202.%20Data%20loading%20%26%20first%20look%0A%0A%20%20%20%20Two%20files%20are%20provided%3A%0A%0A%20%20%20%20-%20%60Train_Data.csv%60%20%E2%80%94%20historical%20registrations%20**with**%20the%20%60Dropped_Course%60%0A%20%20%20%20%20%20label.%0A%20%20%20%20-%20%60Test_Data_No_Target.csv%60%20%E2%80%94%20registrations%20to%20score%2C%20**without**%20the%20label.%0A%0A%20%20%20%20Each%20row%20is%20one%20registration%2C%20identified%20by%20%60Client_ID%60.%20We%20first%20inspect%20inferred%20types%2C%20missingness%2C%20cardinality%2C%20common%20values%2C%20and%20zeros%20before%20deciding%20how%20any%20column%20should%20be%20treated.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(TEST_PATH%2C%20TRAIN_PATH%2C%20display%2C%20load_raw%2C%20pd)%3A%0A%20%20%20%20train_raw%20%3D%20load_raw(TRAIN_PATH)%0A%20%20%20%20test_raw%20%3D%20load_raw(TEST_PATH)%0A%0A%20%20%20%20print(f%22train%3A%20%7Btrain_raw.shape%5B0%5D%3A%2C%7D%20rows%20x%20%7Btrain_raw.shape%5B1%5D%7D%20cols%22)%0A%20%20%20%20print(f%22test%20%3A%20%7Btest_raw.shape%5B0%5D%3A%2C%7D%20rows%20x%20%7Btest_raw.shape%5B1%5D%7D%20cols%22)%0A%0A%20%20%20%20data_dictionary%20%3D%20pd.DataFrame(%7B%0A%20%20%20%20%20%20%20%20%22dtype%22%3A%20train_raw.dtypes.astype(str)%2C%0A%20%20%20%20%20%20%20%20%22n_missing%22%3A%20train_raw.isna().sum()%2C%0A%20%20%20%20%20%20%20%20%22missing_%25%22%3A%20(train_raw.isna().mean()%20*%20100).round(2)%2C%0A%20%20%20%20%20%20%20%20%22n_unique%22%3A%20train_raw.nunique(dropna%3DTrue)%2C%0A%20%20%20%20%20%20%20%20%22n_zero%22%3A%20(train_raw%20%3D%3D%200).sum(numeric_only%3DFalse)%2C%0A%20%20%20%20%20%20%20%20%22most_frequent%22%3A%20train_raw.mode(dropna%3DTrue).iloc%5B0%5D%2C%0A%20%20%20%20%7D)%0A%20%20%20%20display(data_dictionary)%0A%20%20%20%20return%20test_raw%2C%20train_raw%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20**What%20the%20dictionary%20tells%20us.**%0A%0A%20%20%20%20-%20%60Client_ID%60%20is%20unique%20per%20row%0A%20%20%20%20-%20%60Agent_ID%60%20and%20%60Company_ID%60%20were%20inferred%20as%20numeric%20even%20though%20they%20are%20identifiers%2C%20so%20we%20convert%20them%20to%20strings%20after%20this%20first%20inspection.%20%60Company_ID%60%20is%20also%20missing%20for%20most%20rows.%0A%20%20%20%20-%20Several%20text%20fields%20have%20unexpectedly%20high%20cardinality.%20We%20inspect%20their%20raw%20values%20later%20before%20deciding%20whether%20that%20reflects%20real%20variety%20or%20inconsistent%20spelling.%0A%20%20%20%20-%20The%20numeric%20summary%20below%20lets%20us%20look%20for%20suspicious%20ranges%20and%20extreme%20values.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(test_raw%2C%20train_raw)%3A%0A%20%20%20%20for%20id_frame%20in%20(train_raw%2C%20test_raw)%3A%0A%20%20%20%20%20%20%20%20for%20id_col%20in%20(%22Agent_ID%22%2C%20%22Company_ID%22)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20id_frame%5Bid_col%5D%20%3D%20id_frame%5Bid_col%5D.astype(%22string%22)%0A%20%20%20%20train_raw.describe()%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%202.1%20Target%20balance%0A%0A%20%20%20%20We%20first%20check%20whether%20one%20target%20class%20is%20rare%20enough%20to%20require%20special%20treatment%20during%20training%20or%20evaluation.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(TARGET%2C%20display%2C%20pd%2C%20show%2C%20subplot_grid%2C%20train_raw)%3A%0A%20%20%20%20target_counts%20%3D%20train_raw%5BTARGET%5D.value_counts().sort_index()%0A%20%20%20%20target_rate%20%3D%20train_raw%5BTARGET%5D.value_counts(normalize%3DTrue).sort_index()%0A%20%20%20%20balance%20%3D%20pd.DataFrame(%7B%0A%20%20%20%20%20%20%20%20'count'%3A%20target_counts%2C%0A%20%20%20%20%20%20%20%20'rate_%25'%3A%20(target_rate%20*%20100).round(1)%2C%0A%20%20%20%20%7D)%0A%20%20%20%20balance.index%20%3D%20%5B'0%20%3D%20completed'%2C%20'1%20%3D%20dropped'%5D%0A%20%20%20%20display(balance)%0A%0A%20%20%20%20balance_fig%2C%20balance_ax%20%3D%20subplot_grid()%0A%20%20%20%20balance%5B'rate_%25'%5D.plot.bar(ax%3Dbalance_ax)%0A%20%20%20%20balance_ax.set(%0A%20%20%20%20%20%20%20%20title%3D'Course%20outcomes%20in%20the%20training%20data'%2C%20xlabel%3D''%2C%20ylabel%3D'share%20(%25)'%0A%20%20%20%20)%0A%20%20%20%20balance_ax.tick_params(axis%3D'x'%2C%20rotation%3D0)%0A%20%20%20%20show(balance_fig)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20target%20is%20moderately%20balanced%3A%2058.6%25%20completed%20and%2041.4%25%20dropped.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%203.%20Exploratory%20Data%20Analysis%0A%0A%20%20%20%20We%20begin%20with%20the%20target%20and%20date%20coverage%2C%20then%20inspect%20missingness%2C%20categorical%20quality%2C%20and%20numeric%20relationships.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%203.1%20Train%20and%20test%20dates%0A%0A%20%20%20%20We%20plot%20the%20monthly%20drop%20rate%20across%20the%20_training_%20period%20and%20overlay%20where%20training%20ends%20and%20where%20the%20hidden%20test%20window%20ends.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(TARGET%2C%20show%2C%20subplot_grid%2C%20test_raw%2C%20train_raw)%3A%0A%20%20%20%20train_end%20%3D%20train_raw%5B'Course_Start_Date'%5D.max()%0A%20%20%20%20test_start%20%3D%20test_raw%5B'Course_Start_Date'%5D.min()%0A%20%20%20%20test_end%20%3D%20test_raw%5B'Course_Start_Date'%5D.max()%0A%20%20%20%20print(%0A%20%20%20%20%20%20%20%20f%22train%20dates%3A%20%7Btrain_raw%5B'Course_Start_Date'%5D.min().date()%7D%20-%3E%20%7Btrain_end.date()%7D%22%0A%20%20%20%20)%0A%20%20%20%20print(f'test%20%20dates%3A%20%7Btest_start.date()%7D%20-%3E%20%7Btest_end.date()%7D')%0A%20%20%20%20monthly%20%3D%20(%0A%20%20%20%20%20%20%20%20train_raw.set_index('Course_Start_Date').resample('MS')%5BTARGET%5D.mean().mul(100)%0A%20%20%20%20)%0A%0A%20%20%20%20monthly_fig%2C%20monthly_ax%20%3D%20subplot_grid()%0A%20%20%20%20monthly.plot(marker%3D'o'%2C%20ax%3Dmonthly_ax)%0A%20%20%20%20monthly_ax.axhline(%0A%20%20%20%20%20%20%20%20train_raw%5BTARGET%5D.mean()%20*%20100%2C%20linestyle%3D'--'%2C%20label%3D'train%20average'%0A%20%20%20%20)%0A%20%20%20%20monthly_ax.axvline(%0A%20%20%20%20%20%20%20%20train_end%2C%20linestyle%3D'--'%2C%20label%3Df'train%20ends%20(%7Btrain_end.date()%7D)'%0A%20%20%20%20)%0A%20%20%20%20monthly_ax.axvline(test_end%2C%20linestyle%3D'%3A'%2C%20label%3Df'test%20ends%20(%7Btest_end.date()%7D)')%0A%20%20%20%20monthly_ax.set(%0A%20%20%20%20%20%20%20%20xlim%3D(train_raw%5B'Course_Start_Date'%5D.min()%2C%20test_end)%2C%0A%20%20%20%20%20%20%20%20ylabel%3D'drop%20rate%20(%25)'%2C%0A%20%20%20%20%20%20%20%20title%3D'Drop%20rate%20over%20time%20%E2%80%94%20training%20period%20and%20the%20hidden%20test%20horizon'%2C%0A%20%20%20%20)%0A%20%20%20%20monthly_ax.legend()%0A%20%20%20%20show(monthly_fig)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20Training%20covers%20July%202015%20through%20April%202017.%20The%20test%20set%20begins%20at%20the%20end%20of%20that%20period%20and%20continues%20through%20August%202017%2C%20so%20the%20prediction%20task%20is%20temporal%3A%20learn%20from%20earlier%20registrations%20and%20score%20a%20later%20window.%0A%0A%20%20%20%20The%20monthly%20drop%20rate%20also%20changes%20across%20the%20training%20period.%20Because%20a%20random%20split%20would%20mix%20earlier%20and%20later%20regimes%2C%20we%20define%20validation%20chronologically%20and%20later%20compare%20the%20result%20with%20a%20random%20split.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%203.2%20Missing%20values%0A%0A%20%20%20%20We%20compare%20missingness%20in%20train%20and%20test%2C%20then%20ask%20whether%20_the%20fact%20of%20being%20missing_%20is%20itself%20predictive.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(display%2C%20pd%2C%20test_raw%2C%20train_raw)%3A%0A%20%20%20%20missing_compare%20%3D%20pd.DataFrame(%7B%0A%20%20%20%20%20%20%20%20%22train_missing_%25%22%3A%20train_raw.isna().mean().mul(100).round(2)%2C%0A%20%20%20%20%20%20%20%20%22test_missing_%25%22%3A%20test_raw.isna().mean().mul(100).round(2)%2C%0A%20%20%20%20%7D)%0A%20%20%20%20missing_compare%20%3D%20missing_compare%5B%0A%20%20%20%20%20%20%20%20(missing_compare%5B%22train_missing_%25%22%5D%20%3E%200)%0A%20%20%20%20%20%20%20%20%7C%20(missing_compare%5B%22test_missing_%25%22%5D%20%3E%200)%0A%20%20%20%20%5D.sort_values(%22train_missing_%25%22%2C%20ascending%3DFalse)%0A%20%20%20%20display(missing_compare)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20Most%20train%2Ftest%20missingness%20rates%20are%20close.%20We%20next%20check%20whether%20the%20presence%20of%20a%20value%20is%20associated%20with%20the%20target.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(TARGET%2C%20display%2C%20pd%2C%20train_raw)%3A%0A%20%20%20%20missingness_cols%20%3D%20%5B%0A%20%20%20%20%20%20%20%20'Company_ID'%2C%0A%20%20%20%20%20%20%20%20'Agent_ID'%2C%0A%20%20%20%20%20%20%20%20'Registration_Days_Before'%2C%0A%20%20%20%20%20%20%20%20'Physical_Course_Kits'%2C%0A%20%20%20%20%20%20%20%20'Daily_Tuition_Cost'%2C%0A%20%20%20%20%20%20%20%20'Payment_Terms'%2C%0A%20%20%20%20%5D%0A%0A%20%20%20%20missing_summary%20%3D%20(%0A%20%20%20%20%20%20%20%20pd%0A%20%20%20%20%20%20%20%20.concat(%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20col%3A%20train_raw%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.assign(is_missing%3Dtrain_raw%5Bcol%5D.isna())%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.groupby('is_missing')%5BTARGET%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.agg(count%3D'size'%2C%20drop_rate%3D'mean')%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20col%20in%20missingness_cols%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20names%3D%5B'column'%5D%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20.reset_index()%0A%20%20%20%20%20%20%20%20.assign(drop_rate_pct%3Dlambda%20df%3A%20(df%5B'drop_rate'%5D%20*%20100).round(1))%0A%20%20%20%20%20%20%20%20.drop(columns%3D'drop_rate')%0A%20%20%20%20%20%20%20%20.rename(columns%3D%7B'drop_rate_pct'%3A%20'drop_rate_%25'%7D)%0A%20%20%20%20)%0A%20%20%20%20display(missing_summary)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20Rows%20without%20a%20%60Company_ID%60%20have%20a%20noticeably%20higher%20drop%20rate%2C%20and%20%60Agent_ID%60%20presence%20also%20separates%20groups.%20This%20motivates%20explicit%20presence%20flags%20instead%20of%20replacing%20missing%20identifiers%20with%20a%20typical%20value.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%203.3%20Inspecting%20categorical%20values%0A%0A%20%20%20%20Several%20text%20columns%20have%20far%20more%20distinct%20values%20than%20their%20meanings%20suggest%3A%20hundreds%20of%20payment%20terms%2C%20colors%2C%20and%20enrollment%20types%20would%20be%20surprising.%20We%20inspect%20the%20raw%20labels%20before%20deciding%20whether%20the%20cardinality%20is%20real.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(train_raw)%3A%0A%20%20%20%20TEXT_COLS%20%3D%20list(train_raw.select_dtypes(include%3D%5B'object'%5D).columns)%0A%20%20%20%20N_COUNT%20%3D%209%0A%0A%20%20%20%20for%20text_col%20in%20TEXT_COLS%3A%0A%20%20%20%20%20%20%20%20top_values%20%3D%20train_raw%5Btext_col%5D.value_counts(normalize%3DTrue).head(N_COUNT)%0A%20%20%20%20%20%20%20%20cats%20%3D%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20f'%7Bvalue!r%7D%3A%20(%7Bshare%20*%20100%3A.1f%7D%25)'%20for%20value%2C%20share%20in%20top_values.items()%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20%20%20%20%20cats_str%20%3D%20'%5Cn'.join(%0A%20%20%20%20%20%20%20%20%20%20%20%20'%20%7C%20'.join(cats%5Bi%20%3A%20i%20%2B%203%5D)%20for%20i%20in%20range(0%2C%20len(cats)%2C%203)%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20print(%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22%5Cn%7B'%3D'%20*%2080%7D%5Cn%7Btext_col%7D%20(%7Btrain_raw%5Btext_col%5D.nunique()%7D%20unique%20values)%5Cn%5Cn%7Bcats_str%7D%5Cn%22%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20return%20(TEXT_COLS%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20raw%20values%20explain%20much%20of%20the%20inflated%20cardinality.%20Labels%20such%20as%20%60'BLUE'%60%2C%20%60'blue'%60%2C%20and%20%60'%20%20Blue%20%20'%60%20describe%20the%20same%20category%20but%20are%20stored%20separately%3B%20punctuation%20and%20placeholder%20strings%20create%20similar%20splits%20in%20other%20fields.%20Before%20treating%20these%20columns%20as%20genuinely%20high-cardinality%2C%20we%20normalize%20the%20obvious%20formatting%20variants%20and%20measure%20how%20many%20levels%20remain.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(pd)%3A%0A%20%20%20%20%23%20Placeholder%20strings%20that%20mean%20%22missing%22%2C%20in%20any%20casing%2Fpadding%20after%20canonicalisation.%0A%20%20%20%20COMMON_NANS%20%3D%20%7B%0A%20%20%20%20%20%20%20%20''%2C%0A%20%20%20%20%20%20%20%20'-'%2C%0A%20%20%20%20%20%20%20%20'--'%2C%0A%20%20%20%20%20%20%20%20'.'%2C%0A%20%20%20%20%20%20%20%20'%3F'%2C%0A%20%20%20%20%20%20%20%20'na'%2C%0A%20%20%20%20%20%20%20%20'n%2Fa'%2C%0A%20%20%20%20%20%20%20%20'nan'%2C%0A%20%20%20%20%20%20%20%20'none'%2C%0A%20%20%20%20%20%20%20%20'null'%2C%0A%20%20%20%20%20%20%20%20'unknown'%2C%0A%20%20%20%20%20%20%20%20'unknonwn'%2C%0A%20%20%20%20%7D%0A%20%20%20%20COUNTRY_ALIASES%20%3D%20%7B'cn'%3A%20'chn'%7D%20%20%23%20both%20mean%20China%0A%0A%20%20%20%20def%20canonicalize(s%3A%20pd.Series)%20-%3E%20pd.Series%3A%0A%20%20%20%20%20%20%20%20s%20%3D%20s.astype('string').str.strip().str.lower()%0A%20%20%20%20%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20s.str%0A%20%20%20%20%20%20%20%20%20%20%20%20.replace('%5C%5Cband%5C%5Cb'%2C%20'%26'%2C%20regex%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20.str.replace('%5B%5Ea-z0-9%26()%20.%2B-%5D%2B'%2C%20''%2C%20regex%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20.str.replace('%5C%5Cs%2B'%2C%20'%20'%2C%20regex%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20.str.strip()%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20return%20COMMON_NANS%2C%20COUNTRY_ALIASES%2C%20canonicalize%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20We%20normalize%20case%2C%20surrounding%20whitespace%2C%20repeated%20spaces%2C%20and%20injected%20punctuation.%20Placeholder%20labels%20such%20as%20%60Unknown%60%20and%20%60%3F%60%20become%20missing%20values%20rather%20than%20new%20categories.%20The%20same%20deterministic%20cleaning%20function%20will%20be%20applied%20to%20train%20and%20test.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(COMMON_NANS%2C%20COUNTRY_ALIASES%2C%20TEXT_COLS%2C%20canonicalize%2C%20pd)%3A%0A%20%20%20%20CAT_COLS%20%3D%20TEXT_COLS%20%2B%20%5B'Agent_ID'%2C%20'Company_ID'%5D%0A%0A%20%20%20%20def%20normalize_cats(df%3A%20pd.DataFrame)%20-%3E%20pd.DataFrame%3A%0A%20%20%20%20%20%20%20%20%22%22%22Canonicalise%20every%20categorical%2C%20then%20map%20junk%20placeholders%20to%20NaN.%22%22%22%0A%20%20%20%20%20%20%20%20df%20%3D%20df.copy()%0A%20%20%20%20%20%20%20%20for%20col%20in%20CAT_COLS%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20s%20%3D%20canonicalize(df%5Bcol%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20df%5Bcol%5D%20%3D%20s.mask(s.isin(COMMON_NANS))%0A%20%20%20%20%20%20%20%20df%5B'Origin_Country'%5D%20%3D%20df%5B'Origin_Country'%5D.replace(COUNTRY_ALIASES)%0A%20%20%20%20%20%20%20%20return%20df%0A%0A%20%20%20%20return%20(normalize_cats%2C)%0A%0A%0A%40app.cell%0Adef%20_(TEXT_COLS%2C%20display%2C%20normalize_cats%2C%20pd%2C%20train_raw)%3A%0A%20%20%20%20clean_train%20%3D%20normalize_cats(train_raw)%0A%0A%20%20%20%20cardinality_change%20%3D%20(%0A%20%20%20%20%20%20%20%20pd%0A%20%20%20%20%20%20%20%20.DataFrame(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22raw_unique%22%3A%20%7Bc%3A%20train_raw%5Bc%5D.nunique()%20for%20c%20in%20TEXT_COLS%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22clean_unique%22%3A%20%7Bc%3A%20clean_train%5Bc%5D.nunique()%20for%20c%20in%20TEXT_COLS%7D%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20.assign(collapsed%3Dlambda%20t%3A%20t%5B%22raw_unique%22%5D%20-%20t%5B%22clean_unique%22%5D)%0A%20%20%20%20%20%20%20%20.sort_values(%22collapsed%22%2C%20ascending%3DFalse)%0A%20%20%20%20)%0A%20%20%20%20display(cardinality_change)%0A%20%20%20%20return%20(clean_train%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20before%2Fafter%20table%20confirms%20that%20most%20of%20the%20apparent%20variety%20was%20formatting%20noise%3A%20%60Payment_Terms%60%20falls%20from%20236%20raw%20labels%20to%203%20cleaned%20levels%2C%20and%20%60Client_Category%60%20from%20505%20to%207.%20Columns%20that%20were%20already%20consistent%20remain%20unchanged.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%203.4%20Which%20categories%20actually%20relate%20to%20dropping%3F%0A%0A%20%20%20%20We%20start%20with%20business%20fields%20that%20have%20only%20a%20few%20cleaned%20levels%2C%20where%20a%20direct%20plot%20remains%20readable.%20Country%20and%20identifiers%20need%20separate%20treatment%20because%20hundreds%20of%20levels%20would%20make%20the%20same%20plot%20misleading.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(TARGET%2C%20TEXT_COLS%2C%20clean_train%2C%20show%2C%20subplot_grid)%3A%0A%20%20%20%20def%20plot_dropout_by_category(df%2C%20col%2C%20ax%2C%20min_count%3D50%2C%20top_n%3D10)%3A%0A%20%20%20%20%20%20%20%20stats%20%3D%20df.groupby(col%2C%20dropna%3DFalse)%5BTARGET%5D.agg(%0A%20%20%20%20%20%20%20%20%20%20%20%20drop_rate%3D%22mean%22%2C%20count%3D%22size%22%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20stats%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20stats%5Bstats%5B%22count%22%5D%20%3E%3D%20min_count%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20.sort_values(%22count%22%2C%20ascending%3DFalse)%0A%20%20%20%20%20%20%20%20%20%20%20%20.head(top_n)%0A%20%20%20%20%20%20%20%20%20%20%20%20.sort_values(%22drop_rate%22)%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20labels%20%3D%20%5Bf%22%7Bi%7D%5Cn(n%3D%7Bint(r%5B'count'%5D)%7D)%22%20for%20i%2C%20r%20in%20stats.iterrows()%5D%0A%20%20%20%20%20%20%20%20overall%20%3D%20df%5BTARGET%5D.mean()%0A%20%20%20%20%20%20%20%20ax.barh(%0A%20%20%20%20%20%20%20%20%20%20%20%20labels%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20stats%5B%22drop_rate%22%5D%20*%20100%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20color%3D%5B%22C1%22%20if%20rate%20%3E%20overall%20else%20%22C0%22%20for%20rate%20in%20stats%5B%22drop_rate%22%5D%5D%2C%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20ax.axvline(overall%20*%20100%2C%20linestyle%3D%22--%22)%0A%20%20%20%20%20%20%20%20ax.set(%0A%20%20%20%20%20%20%20%20%20%20%20%20xlabel%3D%22Drop%20rate%20(%25)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20title%3Df%22Drop%20rate%20by%5Cn%7Bcol.replace('_'%2C%20'%20')%7D%22%2C%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20category_features%20%3D%20%5Bcol%20for%20col%20in%20TEXT_COLS%20if%20col%20!%3D%20%22Origin_Country%22%5D%0A%20%20%20%20category_rows%20%3D%20(len(category_features)%20%2B%201)%20%2F%2F%202%0A%20%20%20%20category_fig%2C%20category_axes%20%3D%20subplot_grid(%0A%20%20%20%20%20%20%20%20category_rows%2C%202%2C%20figsize%3D(10%2C%203.4%20*%20category_rows)%0A%20%20%20%20)%0A%20%20%20%20for%20category_ax%2C%20category_feature%20in%20zip(category_axes.flat%2C%20category_features)%3A%0A%20%20%20%20%20%20%20%20plot_dropout_by_category(clean_train%2C%20category_feature%2C%20category_ax)%0A%20%20%20%20for%20unused_category_ax%20in%20category_axes.flat%5Blen(category_features)%20%3A%5D%3A%0A%20%20%20%20%20%20%20%20unused_category_ax.set_visible(False)%0A%20%20%20%20show(category_fig)%0A%20%20%20%20return%20(plot_dropout_by_category%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%60Payment_Terms%60%20shows%20the%20strongest%20category-level%20separation%3A%20almost%20all%20prepaid%2C%20non-refundable%20registrations%20dropped%2C%20compared%20with%20roughly%2030%25%20of%20pay-on-start%20registrations.%20Because%20the%20field's%20recording%20time%20is%20unknown%2C%20we%20retain%20it%20provisionally%20and%20treat%20it%20as%20a%20possible%20timing-leakage%20risk.%0A%0A%20%20%20%20%60Welcome_Gift_Type%60%20and%20%60Lanyard_Color%60%20show%20little%20relationship%20with%20dropping.%20The%20%60Assigned_Lab_Config%60%20pattern%20may%20partly%20reflect%20the%20standard%20PC%20being%20the%20default.%0A%0A%20%20%20%20The%20other%20plots%20also%20show%20useful%20separation.%20Direct-website%20and%20dedicated-sales%20registrations%20drop%20less%20often%20than%20reseller%20traffic%2C%20organisational%20enrollment%20is%20lower-risk%20than%20general%20admission%2C%20and%20client%20segments%20differ.%20These%20fields%20are%20therefore%20retained%20as%20descriptive%20predictors%20rather%20than%20causal%20explanations.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%23%20A%20closer%20look%20at%20high-cardinality%20categories%0A%0A%20%20%20%20%60Origin_Country%60%2C%20%60Agent_ID%60%2C%20and%20%60Company_ID%60%20have%20too%20many%20levels%20for%20an%20unfiltered%20chart.%20We%20examine%20country%20first%2C%20keeping%20only%20sufficiently%20large%20groups%3B%20Portugal%20is%20a%20compact%20example%20because%20it%20is%20both%20the%20largest%20country%20group%20and%20far%20from%20the%20overall%20drop%20rate.%20We%20then%20inspect%20agent%20and%20company%20information%20separately.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(TARGET%2C%20clean_train%2C%20display%2C%20pd%2C%20show%2C%20sns%2C%20subplot_grid)%3A%0A%20%20%20%20country_min_n%20%3D%20150%0A%20%20%20%20country_top_n%20%3D%2012%0A%20%20%20%20overall_drop%20%3D%20clean_train%5BTARGET%5D.mean()%0A%20%20%20%20country_stats%20%3D%20(%0A%20%20%20%20%20%20%20%20clean_train%0A%20%20%20%20%20%20%20%20.groupby('Origin_Country'%2C%20dropna%3DFalse)%5BTARGET%5D%0A%20%20%20%20%20%20%20%20.agg(count%3D'size'%2C%20drop_rate%3D'mean')%0A%20%20%20%20%20%20%20%20.assign(%0A%20%20%20%20%20%20%20%20%20%20%20%20drop_rate_pct%3Dlambda%20d%3A%20d%5B'drop_rate'%5D%20*%20100%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20lift_pp%3Dlambda%20d%3A%20(d%5B'drop_rate'%5D%20-%20overall_drop)%20*%20100%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20)%0A%20%20%20%20top_by_size%20%3D%20country_stats.sort_values('count'%2C%20ascending%3DFalse).head(%0A%20%20%20%20%20%20%20%20country_top_n%0A%20%20%20%20)%0A%20%20%20%20extreme_by_lift%20%3D%20(%0A%20%20%20%20%20%20%20%20country_stats%5Bcountry_stats%5B'count'%5D%20%3E%3D%20country_min_n%5D%0A%20%20%20%20%20%20%20%20.iloc%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20lambda%20d%3A%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20d%5B'lift_pp'%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.abs()%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.sort_values(ascending%3DFalse)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.index.map(d.index.get_loc)%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20%20%20%20%20.head(country_top_n)%0A%20%20%20%20)%0A%0A%20%20%20%20def%20plot_country_dropout(stats%2C%20title%2C%20ax)%3A%0A%20%20%20%20%20%20%20%20stats%20%3D%20stats.sort_values('drop_rate_pct')%0A%20%20%20%20%20%20%20%20labels%20%3D%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22%7B(idx%20if%20pd.notna(idx)%20else%20'%3Cmissing%3E')%7D%20(n%3D%7Bint(row%5B'count'%5D)%3A%2C%7D)%22%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20idx%2C%20row%20in%20stats.iterrows()%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20%20%20%20%20below%2C%20above%20%3D%20sns.color_palette(n_colors%3D2)%0A%20%20%20%20%20%20%20%20colors%20%3D%20%5Babove%20if%20lift%20%3E%3D%200%20else%20below%20for%20lift%20in%20stats%5B'lift_pp'%5D%5D%0A%20%20%20%20%20%20%20%20ax.barh(labels%2C%20stats%5B'drop_rate_pct'%5D%2C%20color%3Dcolors)%0A%20%20%20%20%20%20%20%20ax.axvline(%0A%20%20%20%20%20%20%20%20%20%20%20%20overall_drop%20*%20100%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20linestyle%3D'--'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20label%3Df'overall%20(%7Boverall_drop%20*%20100%3A.1f%7D%25)'%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20ax.set(xlabel%3D'drop%20rate%20(%25)'%2C%20title%3Dtitle)%0A%20%20%20%20%20%20%20%20ax.legend()%0A%0A%20%20%20%20plots%20%3D%20%5B%0A%20%20%20%20%20%20%20%20(top_by_size%2C%20f'Drop%20rate%20by%20largest%20%7Bcountry_top_n%7D%20countries')%2C%0A%20%20%20%20%20%20%20%20(extreme_by_lift%2C%20f'Most%20unusual%20country%20drop%20rates%20(n%20%3E%3D%20%7Bcountry_min_n%7D)')%2C%0A%20%20%20%20%5D%0A%20%20%20%20country_fig%2C%20country_axes%20%3D%20subplot_grid(1%2C%202)%0A%20%20%20%20for%20country_ax%2C%20(country_plot_stats%2C%20country_title)%20in%20zip(country_axes%2C%20plots)%3A%0A%20%20%20%20%20%20%20%20plot_country_dropout(country_plot_stats%2C%20country_title%2C%20country_ax)%0A%20%20%20%20show(country_fig)%0A%20%20%20%20display(%0A%20%20%20%20%20%20%20%20country_stats%0A%20%20%20%20%20%20%20%20.sort_values('count'%2C%20ascending%3DFalse)%0A%20%20%20%20%20%20%20%20.head(country_top_n)%5B%5B'count'%2C%20'drop_rate_pct'%2C%20'lift_pp'%5D%5D%0A%20%20%20%20%20%20%20%20.round(2)%0A%20%20%20%20)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20Portugal%20contains%2026%2C429%20registrations%20and%20has%20a%2063.8%25%20drop%20rate%2C%20making%20it%20both%20the%20largest%20country%20group%20and%20the%20clearest%20geographic%20difference.%20We%20use%20it%20to%20investigate%20whether%20country%20overlaps%20with%20agents%2C%20channels%2C%20or%20other%20parts%20of%20the%20acquisition%20process.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(TARGET%2C%20clean_train%2C%20display%2C%20np)%3A%0A%20%20%20%20is_portugal%20%3D%20(%0A%20%20%20%20%20%20%20%20clean_train%5B%22Origin_Country%22%5D.eq(%22prt%22).fillna(False).to_numpy(dtype%3Dbool)%0A%20%20%20%20)%0A%20%20%20%20country_group%20%3D%20np.where(is_portugal%2C%20%22Portugal%22%2C%20%22Other%20countries%22)%0A%0A%20%20%20%20portugal_summary%20%3D%20(%0A%20%20%20%20%20%20%20%20clean_train%0A%20%20%20%20%20%20%20%20.assign(country_group%3Dcountry_group)%0A%20%20%20%20%20%20%20%20.groupby(%22country_group%22)%5BTARGET%5D%0A%20%20%20%20%20%20%20%20.agg(count%3D%22size%22%2C%20drop_rate%3D%22mean%22)%0A%20%20%20%20%20%20%20%20.assign(drop_rate_pct%3Dlambda%20d%3A%20d%5B%22drop_rate%22%5D%20*%20100)%0A%20%20%20%20)%0A%0A%20%20%20%20display(portugal_summary%5B%5B%22count%22%2C%20%22drop_rate_pct%22%5D%5D.round(1))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20Compared%20with%20all%20other%20countries%2C%20Portugal%20remains%20clearly%20different.%20We%20next%20inspect%20the%20identifier%20fields%20as%20categories%2C%20not%20numbers%2C%20to%20see%20whether%20they%20show%20related%20structure.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20TARGET%2C%0A%20%20%20%20clean_train%2C%0A%20%20%20%20display%2C%0A%20%20%20%20plot_dropout_by_category%2C%0A%20%20%20%20show%2C%0A%20%20%20%20subplot_grid%2C%0A%20%20%20%20train_raw%2C%0A)%3A%0A%20%20%20%20company_presence%20%3D%20train_raw.groupby(train_raw%5B'Company_ID'%5D.notna())%5BTARGET%5D.agg(%0A%20%20%20%20%20%20%20%20count%3D'size'%2C%20drop_rate%3D'mean'%0A%20%20%20%20)%0A%20%20%20%20company_presence.index%20%3D%20%5B'no%20company_id'%2C%20'has%20company_id'%5D%0A%0A%20%20%20%20identifier_fig%2C%20identifier_axes%20%3D%20subplot_grid(1%2C%202)%0A%20%20%20%20plot_dropout_by_category(%0A%20%20%20%20%20%20%20%20clean_train%2C%20'Agent_ID'%2C%20identifier_axes%5B0%5D%2C%20min_count%3D150%2C%20top_n%3D12%0A%20%20%20%20)%0A%20%20%20%20identifier_axes%5B1%5D.bar(company_presence.index%2C%20company_presence%5B'drop_rate'%5D%20*%20100)%0A%20%20%20%20identifier_axes%5B1%5D.set(%0A%20%20%20%20%20%20%20%20ylabel%3D'drop%20rate%20(%25)'%2C%20title%3D'Drop%20rate%20by%20Company_ID%20presence'%0A%20%20%20%20)%0A%20%20%20%20show(identifier_fig)%0A%20%20%20%20display(company_presence)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20Frequent%20agents%20have%20different%20drop%20rates%2C%20while%20registrations%20with%20a%20%60Company_ID%60%20drop%20less%20often%20(21.2%25%20versus%2042.5%25).%20These%20relationships%20may%20overlap%20with%20geography%2C%20so%20we%20perform%20a%20small%20check%3A%20does%20knowing%20the%20agent%20improve%20country%20prediction%20over%20always%20guessing%20the%20most%20common%20country%3F%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(SEED%2C%20clean_train%2C%20display%2C%20pd%2C%20train_test_split)%3A%0A%20%20%20%20agent_country_pairs%20%3D%20clean_train%5B%5B%22Agent_ID%22%2C%20%22Origin_Country%22%5D%5D.dropna()%0A%0A%20%20%20%20country_tr%2C%20country_va%20%3D%20train_test_split(%0A%20%20%20%20%20%20%20%20agent_country_pairs%2C%20test_size%3D0.25%2C%20random_state%3DSEED%0A%20%20%20%20)%0A%20%20%20%20majority_country%20%3D%20country_tr%5B%22Origin_Country%22%5D.mode().iat%5B0%5D%0A%20%20%20%20agent_country_map%20%3D%20country_tr.groupby(%22Agent_ID%22)%5B%22Origin_Country%22%5D.agg(%0A%20%20%20%20%20%20%20%20lambda%20s%3A%20s.value_counts().idxmax()%0A%20%20%20%20)%0A%20%20%20%20agent_country_pred%20%3D%20(%0A%20%20%20%20%20%20%20%20country_va%5B%22Agent_ID%22%5D.map(agent_country_map).fillna(majority_country)%0A%20%20%20%20)%0A%0A%20%20%20%20display(%0A%20%20%20%20%20%20%20%20pd.DataFrame(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22check%22%3A%20%5B%22majority%20country%20baseline%22%2C%20%22agent%20modal%20country%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22accuracy%22%3A%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20country_va%5B%22Origin_Country%22%5D.eq(majority_country).mean()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20agent_country_pred.eq(country_va%5B%22Origin_Country%22%5D).mean()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%20%20%20%20%7D).round(3)%0A%20%20%20%20)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20Agent-based%20prediction%20raises%20country%20accuracy%20from%200.391%20to%200.421%2C%20indicating%20modest%20overlap%20between%20the%20two%20fields.%20Both%20are%20included%20using%20the%20compact%20representation%20introduced%20during%20preparation.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%203.5%20Numeric%20features%3A%20summary%2C%20correlation%2C%20and%20suspects%0A%0A%20%20%20%20We%20now%20inspect%20numeric%20ranges%2C%20distributions%2C%20and%20their%20linear%20correlations%20with%20the%20target.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(TARGET%2C%20display%2C%20train_raw)%3A%0A%20%20%20%20ID_LIKE%20%3D%20%5B%22Client_ID%22%2C%20%22Agent_ID%22%2C%20%22Company_ID%22%5D%0A%20%20%20%20num_cols%20%3D%20%5B%0A%20%20%20%20%20%20%20%20c%0A%20%20%20%20%20%20%20%20for%20c%20in%20train_raw.select_dtypes(include%3D%5B%22int64%22%2C%20%22float64%22%5D).columns%0A%20%20%20%20%20%20%20%20if%20c%20not%20in%20ID_LIKE%20%2B%20%5BTARGET%5D%0A%20%20%20%20%5D%0A%0A%20%20%20%20numeric_summary%20%3D%20(%0A%20%20%20%20%20%20%20%20train_raw%5Bnum_cols%5D.agg(%5B'mean'%2C%20'median'%2C%20'std'%2C%20'min'%2C%20'max'%2C%20'skew'%5D).T%0A%20%20%20%20)%0A%20%20%20%20numeric_summary.insert(0%2C%20'missing_%25'%2C%20train_raw%5Bnum_cols%5D.isna().mean()%20*%20100)%0A%20%20%20%20numeric_summary.insert(%0A%20%20%20%20%20%20%20%201%2C%20'corr_target'%2C%20train_raw%5Bnum_cols%5D.corrwith(train_raw%5BTARGET%5D)%0A%20%20%20%20)%0A%20%20%20%20numeric_summary%20%3D%20(%0A%20%20%20%20%20%20%20%20numeric_summary%0A%20%20%20%20%20%20%20%20.round(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20'missing_%25'%3A%201%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'corr_target'%3A%203%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'mean'%3A%202%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'median'%3A%202%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'std'%3A%202%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'min'%3A%202%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'max'%3A%202%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'skew'%3A%202%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20.rename_axis('column')%0A%20%20%20%20%20%20%20%20.reset_index()%0A%20%20%20%20%20%20%20%20.sort_values('corr_target'%2C%20key%3Dabs%2C%20ascending%3DFalse)%0A%20%20%20%20)%0A%20%20%20%20display(numeric_summary)%0A%20%20%20%20return%20(num_cols%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20maximum%20values%20reveal%20several%20likely%20data%20errors%3A%20%60Students_Count%60%20reaches%209999%2C%20and%20%60Practical_Hours%60%20contains%20both%20negative%20values%20and%20values%20up%20to%2010000.%20We%20leave%20the%20raw%20values%20unchanged%20for%20this%20first%20inspection%20and%20decide%20how%20to%20handle%20them%20in%20the%20outlier%20section.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(TARGET%2C%20num_cols%2C%20plt%2C%20show%2C%20sns%2C%20train_raw)%3A%0A%20%20%20%20corr%20%3D%20train_raw%5Bnum_cols%20%2B%20%5BTARGET%5D%5D.corr()%0A%20%20%20%20corr_fig%2C%20corr_ax%20%3D%20plt.subplots(figsize%3D(12%2C%207)%2C%20layout%3D'constrained')%0A%20%20%20%20sns.heatmap(corr%2C%20annot%3DTrue%2C%20fmt%3D'.2f'%2C%20cmap%3D'coolwarm'%2C%20center%3D0%2C%20ax%3Dcorr_ax)%0A%20%20%20%20corr_ax.set_title('Numeric%20correlation%20heatmap%20(incl.%20target)')%0A%20%20%20%20corr_ax.grid(False)%0A%20%20%20%20show(corr_fig)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20No%20raw%20numeric%20feature%20has%20an%20extremely%20strong%20Pearson%20correlation%20with%20the%20target.%20%60Registration_Days_Before%60%20and%20%60Pre_Course_Supports_Tickets%60%20stand%20out%20most%2C%20while%20inter-feature%20correlations%20are%20generally%20modest.%20Because%20Pearson%20correlation%20measures%20linear%20association%20and%20is%20sensitive%20to%20extremes%2C%20we%20next%20use%20binned%20drop%20rates%20to%20inspect%20the%20shape%20of%20the%20strongest%20relationships.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%203.6%20Numeric%20drop-rate%20profiles%0A%0A%20%20%20%20Binning%20a%20couple%20of%20the%20more%20predictive%20numeric%20features%20shows%20_how_%20risk%20moves%20with%20them%20(not%20just%20whether%20they%20correlate%20linearly).%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(TARGET%2C%20pd%2C%20show%2C%20subplot_grid%2C%20train_raw)%3A%0A%20%20%20%20def%20plot_dropout_by_bins(df%2C%20col%2C%20bins%2C%20ax)%3A%0A%20%20%20%20%20%20%20%20tmp%20%3D%20df%5B%5Bcol%2C%20TARGET%5D%5D.dropna().copy()%0A%20%20%20%20%20%20%20%20tmp%5B'bin'%5D%20%3D%20pd.qcut(tmp%5Bcol%5D%2C%20q%3Dbins%2C%20duplicates%3D'drop')%0A%20%20%20%20%20%20%20%20stats%20%3D%20tmp.groupby('bin'%2C%20observed%3DTrue)%5BTARGET%5D.mean().mul(100)%0A%20%20%20%20%20%20%20%20stats.plot.bar(ax%3Dax)%0A%20%20%20%20%20%20%20%20ax.axhline(df%5BTARGET%5D.mean()%20*%20100%2C%20linestyle%3D'--'%2C%20label%3D'mean')%0A%20%20%20%20%20%20%20%20ax.set(ylabel%3D'drop%20rate%20(%25)'%2C%20title%3Df'Drop%20rate%20by%20%7Bcol%7D%20bins')%0A%20%20%20%20%20%20%20%20ax.legend()%0A%20%20%20%20%20%20%20%20ax.tick_params(axis%3D'x'%2C%20labelrotation%3D45)%0A%0A%20%20%20%20numeric_bin_specs%20%3D%20%5B%0A%20%20%20%20%20%20%20%20('Registration_Days_Before'%2C%208)%2C%0A%20%20%20%20%20%20%20%20('Pre_Course_Supports_Tickets'%2C%206)%2C%0A%20%20%20%20%5D%0A%20%20%20%20numeric_bin_fig%2C%20numeric_bin_axes%20%3D%20subplot_grid(1%2C%202)%0A%20%20%20%20for%20numeric_bin_ax%2C%20(numeric_feature%2C%20bins)%20in%20zip(%0A%20%20%20%20%20%20%20%20numeric_bin_axes%2C%20numeric_bin_specs%0A%20%20%20%20)%3A%0A%20%20%20%20%20%20%20%20plot_dropout_by_bins(train_raw%2C%20numeric_feature%2C%20bins%2C%20numeric_bin_ax)%0A%20%20%20%20show(numeric_bin_fig)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20Drop%20rate%20rises%20across%20longer%20registration%20lead%20times%2C%20which%20suggests%20that%20plans%20are%20more%20likely%20to%20change%20when%20courses%20are%20booked%20far%20in%20advance.%20More%20pre-course%20support%20tickets%20are%20associated%20with%20lower%20dropping%2C%20suggesting%20that%20early%20engagement%20may%20reflect%20stronger%20commitment.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%203.7%20EDA%20conclusions%0A%0A%20%20%20%20Several%20observations%20now%20guide%20preparation%20and%20modelling%3A%0A%0A%20%20%20%20-%20Missing%20%60Company_ID%60%2C%20support%20activity%2C%20registration%20channel%2C%20enrollment%20type%2C%20and%20lead%20time%20all%20separate%20groups%20with%20different%20drop%20rates.%20Together%2C%20these%20patterns%20suggest%20a%20broader%20difference%20in%20buyer%20commitment.%0A%20%20%20%20-%20%60Payment_Terms%60%20is%20unusually%20strong%20and%20counter-intuitive.%20We%20retain%20it%20provisionally%2C%20while%20treating%20its%20recording%20time%20as%20an%20unresolved%20limitation.%0A%20%20%20%20-%20Country%20and%20agent%20both%20contain%20signal%20and%20overlap%20slightly.%20Their%20many%20levels%20require%20a%20compact%20encoding%20instead%20of%20a%20large%20one-hot%20expansion.%0A%20%20%20%20-%20The%20later%20test%20window%20and%20changing%20monthly%20rates%20make%20time-aware%20validation%20important.%20We%20therefore%20use%20a%20future%20holdout%20and%20represent%20both%20seasonality%20and%20longer-term%20time.%0A%20%20%20%20-%20Some%20numeric%20values%20are%20clearly%20suspicious%2C%20while%20other%20large%20values%20may%20be%20legitimate%20rare%20cases.%20We%20will%20correct%20only%20the%20values%20for%20which%20we%20have%20evidence%20of%20an%20error.%0A%0A%20%20%20%20These%20conclusions%20support%20comparing%20a%20flexible%20nonlinear%20model%20with%20linear%20and%20neural%20baselines%20on%20the%20same%20future%20holdout.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%204.%20Missing-value%20handling%20%26%20outlier%20analysis%0A%0A%20%20%20%20We%20now%20turn%20the%20EDA%20findings%20into%20reproducible%20preparation%20rules.%20The%20same%20fitted%20rules%20must%20be%20applied%20to%20later%20data%2C%20but%20the%20exact%20missing-value%20treatment%20can%20differ%20by%20model%20family.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%204.1%20Inspecting%20and%20handling%20outliers%0A%0A%20%20%20%20We%20look%20for%20values%20that%20are%20physically%20impossible%20or%20absurdly%20far%20from%20the%20bulk.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(display%2C%20num_cols%2C%20pd%2C%20test_raw%2C%20train_raw)%3A%0A%20%20%20%20def%20sus_report(df%2C%20cols%2C%20max_mult%3D10)%3A%0A%20%20%20%20%20%20%20%20out%20%3D%20%5B%5D%0A%20%20%20%20%20%20%20%20for%20c%20in%20cols%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20s%20%3D%20df%5Bc%5D.dropna()%0A%20%20%20%20%20%20%20%20%20%20%20%20q99%20%3D%20s.quantile(0.99)%0A%20%20%20%20%20%20%20%20%20%20%20%20iqr%20%3D%20s.quantile(0.75)%20-%20s.quantile(0.25)%0A%20%20%20%20%20%20%20%20%20%20%20%20scale%20%3D%20max(q99%2C%20iqr%2C%201.0)%0A%20%20%20%20%20%20%20%20%20%20%20%20why%20%3D%20%5B%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20s.min()%20%3C%200%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20why.append(%22negative%20values%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20s.max()%20%3E%20max_mult%20*%20scale%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20why.append(f%22max%3D%7Bs.max()%3Ag%7D%20%3E%3E%20q99%3D%7Bq99%3Ag%7D%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20why%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20out.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22column%22%3A%20c%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22min%22%3A%20s.min()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22max%22%3A%20s.max()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22q99%22%3A%20round(q99%2C%201)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22why%22%3A%20%22%3B%20%22.join(why)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20return%20pd.DataFrame(out)%0A%0A%20%20%20%20print(%22Suspect%20columns%20%E2%80%94%20TRAIN%22)%0A%20%20%20%20display(sus_report(train_raw%2C%20num_cols))%0A%20%20%20%20print(%22Suspect%20columns%20%E2%80%94%20TEST%22)%0A%20%20%20%20display(sus_report(test_raw%2C%20num_cols))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20test%20set%20introduces%20no%20new%20forms%20of%20corruption%2C%20suggesting%20the%20same%20cleaning%20policy%20can%20be%20safely%20shared.%20Comparing%20the%20maximum%20values%20to%20the%2099th%20percentile%20helps%20identify%20columns%20with%20extreme%20outliers%3A%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(pd%2C%20show%2C%20sns%2C%20test_raw%2C%20train_raw)%3A%0A%20%20%20%20TAIL_CHECK_COLS%20%3D%20%5B%0A%20%20%20%20%20%20%20%20%22Students_Count%22%2C%0A%20%20%20%20%20%20%20%20%22Practical_Hours%22%2C%0A%20%20%20%20%20%20%20%20%22Daily_Tuition_Cost%22%2C%0A%20%20%20%20%20%20%20%20%22Prev_Course_Attended%22%2C%0A%20%20%20%20%20%20%20%20%22Waiting_List_Days%22%2C%0A%20%20%20%20%20%20%20%20%22Registration_Changes%22%2C%0A%20%20%20%20%5D%0A%0A%20%20%20%20tail_long%20%3D%20pd.concat(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20df%5Bcol%5D.dropna().rename('value').to_frame().assign(split%3Dsplit%2C%20column%3Dcol)%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20split%2C%20df%20in%20%5B('train'%2C%20train_raw)%2C%20('test'%2C%20test_raw)%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20col%20in%20TAIL_CHECK_COLS%0A%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%20%20%20%20ignore_index%3DTrue%2C%0A%20%20%20%20)%0A%0A%20%20%20%20grid%20%3D%20sns.catplot(%0A%20%20%20%20%20%20%20%20data%3Dtail_long%2C%0A%20%20%20%20%20%20%20%20x%3D'split'%2C%0A%20%20%20%20%20%20%20%20y%3D'value'%2C%0A%20%20%20%20%20%20%20%20hue%3D'split'%2C%0A%20%20%20%20%20%20%20%20col%3D'column'%2C%0A%20%20%20%20%20%20%20%20col_wrap%3D3%2C%0A%20%20%20%20%20%20%20%20kind%3D'box'%2C%0A%20%20%20%20%20%20%20%20sharey%3DFalse%2C%0A%20%20%20%20%20%20%20%20height%3D3.2%2C%0A%20%20%20%20%20%20%20%20aspect%3D1.05%2C%0A%20%20%20%20%20%20%20%20palette%3D'colorblind'%2C%0A%20%20%20%20%20%20%20%20legend%3DFalse%2C%0A%20%20%20%20%20%20%20%20flierprops%3D%7B'markersize'%3A%203%2C%20'alpha'%3A%200.35%7D%2C%0A%20%20%20%20)%0A%20%20%20%20grid.set_axis_labels(''%2C%20'Raw%20value%20(log-like%20scale)').set_titles('%7Bcol_name%7D')%0A%20%20%20%20for%20tail_column%2C%20tail_ax%20in%20grid.axes_dict.items()%3A%0A%20%20%20%20%20%20%20%20tail_values%20%3D%20tail_long.loc%5Btail_long%5B'column'%5D.eq(tail_column)%2C%20'value'%5D%0A%20%20%20%20%20%20%20%20tail_ax.set_yscale('symlog'%20if%20tail_values.min()%20%3C%200%20else%20'log')%0A%20%20%20%20%20%20%20%20tail_ax.set_title(tail_column.replace('_'%2C%20'%20'))%0A%20%20%20%20grid.figure.suptitle('Train%2Ftest%20tail%20comparison'%2C%20fontsize%3D15)%0A%20%20%20%20grid.figure.set_layout_engine('constrained')%0A%20%20%20%20show(grid.figure)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20box%20plots%20identify%20three%20clear%20data-entry%20errors%2C%20so%20we%20apply%3A%0A%0A%20%20%20%20-%20%60Students_Count%20%3C%3D%2010%60%3A%20the%20values%20beyond%20the%20observed%20low-count%20support%20are%20repeated%20%609999%60%20placeholders%20in%20both%20train%20and%20test.%20The%20cap%20keeps%20those%20rows%20as%20large%20groups%20without%20treating%209999%20as%20a%20real%20count.%0A%20%20%20%20-%20%60Practical_Hours%60%20in%20%60%5B0%2C%2012%5D%60%3A%20negative%20values%20are%20impossible%2C%20and%20%605000%60%2F%6010000%60%20are%20clear%20placeholders.%20A%2012-hour%20upper%20bound%20still%20allows%20a%20long%20practical%20day%20and%20prevents%20corrupted%20placeholder%20values%20from%20distorting%20the%20feature%20space.%0A%20%20%20%20-%20%60Daily_Tuition_Cost%20%3C%3D%20600%60%3A%20train%20has%20a%20single%20%605400%60%20value%2C%20while%20the%20test%20maximum%20is%20510.%20A%20cap%20of%20600%20leaves%20the%20observed%20test%20range%20untouched%20and%20prevents%20one%20corrupted%20training%20value%20from%20dominating%20cost%20calculations.%0A%0A%20%20%20%20Other%20flagged%20count%20columns%20(%60Prev_Course_Dropouts%60%2C%20%60Prev_Course_Attended%60%2C%20%60Registration_Changes%60%2C%20and%20test-side%20%60Waiting_List_Days%60)%20have%20long%20but%20plausible%20tails%20(as%20seen%20in%20the%20box-plots)%2C%20so%20we%20leave%20them%20unchanged%20and%20restrict%20clipping%20to%20the%20three%20apparent%20data-entry%20errors%20above.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(display%2C%20pd%2C%20show%2C%20subplot_grid%2C%20test_raw%2C%20train_raw)%3A%0A%20%20%20%20CAP_RULES%20%3D%20%7B%0A%20%20%20%20%20%20%20%20'Students_Count'%3A%20(None%2C%2010)%2C%0A%20%20%20%20%20%20%20%20'Practical_Hours'%3A%20(0%2C%2012)%2C%0A%20%20%20%20%20%20%20%20'Daily_Tuition_Cost'%3A%20(None%2C%20600)%2C%0A%20%20%20%20%7D%0A%20%20%20%20cap_rows%20%3D%20%5B%5D%0A%20%20%20%20for%20cap_col%2C%20(lower%2C%20upper)%20in%20CAP_RULES.items()%3A%0A%20%20%20%20%20%20%20%20train_capped%20%3D%20train_raw%5Bcap_col%5D.clip(lower%3Dlower%2C%20upper%3Dupper)%0A%20%20%20%20%20%20%20%20test_capped%20%3D%20test_raw%5Bcap_col%5D.clip(lower%3Dlower%2C%20upper%3Dupper)%0A%20%20%20%20%20%20%20%20cap_rows.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20'column'%3A%20cap_col%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'train_rows_affected'%3A%20int(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(train_raw%5Bcap_col%5D.notna()%20%26%20train_capped.ne(train_raw%5Bcap_col%5D)).sum()%0A%20%20%20%20%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'test_rows_affected'%3A%20int(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20(test_raw%5Bcap_col%5D.notna()%20%26%20test_capped.ne(test_raw%5Bcap_col%5D)).sum()%0A%20%20%20%20%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20display(pd.DataFrame(cap_rows))%0A%0A%20%20%20%20cap_fig%2C%20cap_axes%20%3D%20subplot_grid(2%2C%203%2C%20sharey%3D'row'%2C%20figsize%3D(8%2C%205.5))%0A%20%20%20%20for%20cap_index%2C%20(cap_column%2C%20(lower%2C%20upper))%20in%20enumerate(CAP_RULES.items())%3A%0A%20%20%20%20%20%20%20%20before%20%3D%20train_raw%5Bcap_column%5D.dropna()%0A%20%20%20%20%20%20%20%20after%20%3D%20before.clip(lower%3Dlower%2C%20upper%3Dupper)%0A%20%20%20%20%20%20%20%20for%20cap_ax%2C%20cap_values%2C%20cap_label%20in%20zip(%0A%20%20%20%20%20%20%20%20%20%20%20%20cap_axes%5B%3A%2C%20cap_index%5D%2C%20(before%2C%20after)%2C%20('raw'%2C%20'clipped')%0A%20%20%20%20%20%20%20%20)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20cap_ax.hist(cap_values%2C%20bins%3D30)%0A%20%20%20%20%20%20%20%20%20%20%20%20cap_ax.set(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20yscale%3D'log'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20title%3Df'%7Bcap_column%7D%3A%20%7Bcap_label%7D'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20xlabel%3Df'max%3D%7Bcap_values.max()%3Ag%7D'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20cap_axes%5B0%2C%200%5D.set_ylabel('count%20(log)')%0A%20%20%20%20cap_axes%5B1%2C%200%5D.set_ylabel('count%20(log)')%0A%20%20%20%20show(cap_fig)%0A%20%20%20%20return%20(CAP_RULES%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20before%2Fafter%20distributions%20show%20that%20the%20caps%20remove%20isolated%20invalid%20tails%20while%0A%20%20%20%20preserving%20the%20bulk%20of%20each%20feature.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%204.2%20Missing-value%20policy%0A%0A%20%20%20%20One%20missing-value%20policy%20would%20not%20suit%20every%20model%20family%3A%0A%0A%20%20%20%20-%20Categorical%20missingness%20becomes%20an%20explicit%20%60%22missing%22%60%20level%20on%20both%20preprocessing%20paths.%20This%20preserves%20the%20possibility%20that%20absence%20itself%20carries%20information.%0A%20%20%20%20-%20%60Agent_ID%60%20and%20%60Company_ID%60%20also%20receive%20presence%20flags%20because%20EDA%20showed%20a%20clear%20difference%20between%20present%20and%20missing%20groups.%20Their%20high%20cardinality%20is%20handled%20separately%20in%20feature%20engineering.%0A%20%20%20%20-%20Models%20that%20support%20numeric%20missing%20values%20natively%20can%20retain%20%60NaN%60%20and%20learn%20how%20to%20route%20it.%20Models%20that%20require%20a%20complete%20numeric%20matrix%20receive%20medians%20learned%20from%20the%20training%20partition%20only.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%205.%20Feature%20engineering%20%26%20dimensionality%0A%0A%20%20%20%20Based%20on%20the%20EDA%20findings%20and%20domain%20questions%2C%20we%20create%20features%20that%20expose%20relationships%20more%20directly%20or%20represent%20high-cardinality%20fields%20more%20compactly.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%205.1%20The%20engineered%20features%20and%20their%20rationale%0A%0A%20%20%20%20%7C%20Raw%20signal%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%20Engineered%20feature(s)%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%20Reason%20for%20testing%20it%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%0A%20%20%20%20%7C%20-------------------------%20%7C%20------------------------------------------------------------%20%7C%20--------------------------------------------------------------------------------------------------------------------------------------------------------------------%20%7C%0A%20%20%20%20%7C%20%60Course_Start_Date%60%20%20%20%20%20%20%20%7C%20%60start_month%60%2C%20%60start_dow%60%2C%20%60start_week%60%2C%20%60days_since_epoch%60%20%7C%20Month%20and%20ISO%20week%20represent%20seasonality%2C%20weekday%20represents%20scheduling%20patterns%2C%20and%20the%20linear%20index%20represents%20the%20longer-term%20shift%20seen%20in%20Section%203.1.%20%20%20%20%20%20%20%20%20%20%7C%0A%20%20%20%20%7C%20Participant%20counts%20%20%20%20%20%20%20%20%7C%20%60total_participants%60%2C%20%60prof_share%60%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%20Total%20size%20and%20professional%20share%20describe%20group%20composition%20more%20directly%20than%20three%20separate%20counts.%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%0A%20%20%20%20%7C%20Practical%2Ftheory%20hours%20%20%20%20%7C%20%60total_hours%60%2C%20%60practical_share%60%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%20Total%20duration%20and%20hands-on%20share%20distinguish%20courses%20with%20the%20same%20raw%20hour%20count%20but%20different%20structure.%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%0A%20%20%20%20%7C%20Client%20history%20%20%20%20%20%20%20%20%20%20%20%20%7C%20%60prev_drop_rate%20%3D%20dropouts%20%2F%20(attended%20%2B%201)%60%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%20Adds%20a%20relative%20cancellation-history%20signal%20while%20retaining%20both%20raw%20counters.%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%0A%20%20%20%20%7C%20Tuition%20cost%20and%20hours%20%20%20%20%7C%20%60cost_x_days%60%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%20Combines%20price%20and%20course%20length%20so%20the%20model%20can%20consider%20their%20interaction.%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%0A%20%20%20%20%7C%20Requested%20vs%20assigned%20lab%20%7C%20%60got_requested_lab%60%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%20Captures%20whether%20the%20assigned%20lab%20configuration%20matches%20the%20original%20request.%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%0A%20%20%20%20%7C%20Missing%20company%2Fagent%20IDs%20%7C%20%60has_company_id%60%2C%20%60has_agent_id%60%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%20Preserves%20the%20presence%20differences%20observed%20in%20Section%203.2%20even%20when%20a%20raw%20identifier%20is%20removed.%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%0A%20%20%20%20%7C%20Agent%2Fcompany%2Fcountry%20IDs%20%7C%20frequency%20encodings%20and%20native%20categories%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%20Retains%20identity%20and%20commonness%20information%20while%20avoiding%20a%20wide%20dummy%20matrix.%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7C%0A%0A%20%20%20%20We%20retain%20both%20historical%20counters%20and%20add%20%60prev_drop_rate%20%3D%20dropouts%20%2F%20(attended%20%2B%201)%60%20as%20a%20stabilized%20cancellation-intensity%20feature%3A%20the%20%60%2B1%60%20prevents%20division%20by%20zero%20because%20most%20rows%20have%20no%20recorded%20prior%20attendance%2C%20and%20the%20result%20is%20not%20a%20probability%20or%20restricted%20to%20%60%5B0%2C%201%5D%60.%0A%0A%20%20%20%20%60Assigned_Lab_Config%60%20is%20populated%20even%20for%20cancelled%20bookings%2C%20so%20we%20treat%20it%20as%20a%20planned%20assignment%20known%20before%20the%20course%20and%20use%20it%20in%20%60got_requested_lab%60.%20This%20timing%20assumption%20would%20need%20confirmation%20before%20deploying%20the%20model.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%205.2%20Dimensionality%0A%0A%20%20%20%20The%20main%20expansion%20risk%20comes%20from%20identifiers%3A%20%60Agent_ID%60%20has%20204%20cleaned%20levels%20and%20%60Origin_Country%60%20has%20154.%20Different%20model%20families%20therefore%20need%20different%20preparation%20paths.%0A%0A%20%20%20%20Models%20that%20require%20numeric%20inputs%20receive%20rare-level%20grouping%2C%20one-hot%20encoding%2C%20training-median%20imputation%2C%20and%20scaling.%20Boosted-tree%20implementations%20with%20native%20categorical%20support%20can%20work%20with%20category%20labels%20directly%2C%20so%20the%20matrix%20does%20not%20need%20one%20dummy%20column%20per%20agent%20or%20country.%20We%20also%20add%20one%20frequency%20feature%20per%20high-cardinality%20identifier%20and%20remove%20raw%20%60Company_ID%60%2C%20retaining%20only%20its%20frequency%20and%20presence%20flag.%0A%0A%20%20%20%20During%20validation%2C%20frequency%20maps%20are%20learned%20from%20the%20earlier%20training%20partition.%20For%20the%20final%20submission%2C%20we%20compute%20frequencies%20across%20the%20available%20train%20and%20test%20features%2C%20without%20using%20%60Dropped_Course%60.%20This%20transductive%20step%20gives%20each%20identifier%20one%20consistent%20frequency%20at%20scoring%20time.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20CAP_RULES%2C%0A%20%20%20%20TEXT_COLS%2C%0A%20%20%20%20normalize_cats%2C%0A%20%20%20%20np%2C%0A%20%20%20%20num_cols%2C%0A%20%20%20%20pd%2C%0A%20%20%20%20test_raw%2C%0A%20%20%20%20train_raw%2C%0A)%3A%0A%20%20%20%20frequency_cols%20%3D%20('Agent_ID'%2C%20'Company_ID'%2C%20'Origin_Country')%0A%20%20%20%20native_cat_cols%20%3D%20%5Bcol%20for%20col%20in%20TEXT_COLS%20if%20col%20!%3D%20'Assigned_Lab_Config'%5D%20%2B%20%5B%0A%20%20%20%20%20%20%20%20'Agent_ID'%0A%20%20%20%20%5D%0A%0A%20%20%20%20def%20make_freq_maps(*dfs)%3A%0A%20%20%20%20%20%20%20%20%22%22%22Label-free%20frequency%20of%20each%20ID%20value%20across%20the%20supplied%20frames.%22%22%22%0A%20%20%20%20%20%20%20%20combined%20%3D%20pd.concat(%5Bnormalize_cats(d)%20for%20d%20in%20dfs%5D%2C%20ignore_index%3DTrue)%0A%20%20%20%20%20%20%20%20return%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20col%3A%20combined%5Bcol%5D.value_counts(normalize%3DTrue)%20for%20col%20in%20frequency_cols%0A%20%20%20%20%20%20%20%20%7D%0A%0A%20%20%20%20freq_maps%20%3D%20make_freq_maps(train_raw%2C%20test_raw)%0A%0A%20%20%20%20def%20build_features(%0A%20%20%20%20%20%20%20%20df%3A%20pd.DataFrame%2C%0A%20%20%20%20%20%20%20%20freq_maps%3A%20dict%2C%0A%20%20%20%20%20%20%20%20add_time%3A%20bool%20%3D%20True%2C%0A%20%20%20%20%20%20%20%20add_week%3A%20bool%20%3D%20True%2C%0A%20%20%20%20)%20-%3E%20pd.DataFrame%3A%0A%20%20%20%20%20%20%20%20%22%22%22Cleaning%20%2B%20feature%20engineering.%20Identical%20transform%20for%20train%20and%20test.%22%22%22%0A%20%20%20%20%20%20%20%20df%20%3D%20normalize_cats(df)%0A%20%20%20%20%20%20%20%20out%20%3D%20df%5Bnum_cols%5D.copy()%0A%20%20%20%20%20%20%20%20for%20col%2C%20(lower%2C%20upper)%20in%20CAP_RULES.items()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20out%5Bcol%5D%20%3D%20out%5Bcol%5D.clip(lower%3Dlower%2C%20upper%3Dupper)%0A%20%20%20%20%20%20%20%20dates%20%3D%20df%5B'Course_Start_Date'%5D%0A%20%20%20%20%20%20%20%20out%5B'start_month'%5D%20%3D%20dates.dt.month%0A%20%20%20%20%20%20%20%20out%5B'start_dow'%5D%20%3D%20dates.dt.dayofweek%0A%20%20%20%20%20%20%20%20if%20add_week%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20out%5B'start_week'%5D%20%3D%20dates.dt.isocalendar().week.astype(float)%0A%20%20%20%20%20%20%20%20if%20add_time%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20out%5B'days_since_epoch'%5D%20%3D%20(dates%20-%20pd.Timestamp('2015-01-01')).dt.days%0A%20%20%20%20%20%20%20%20total%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20out%5B%5B'Professionals_Count'%2C%20'Students_Count'%2C%20'Observers_Count'%5D%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20.fillna(0)%0A%20%20%20%20%20%20%20%20%20%20%20%20.sum(axis%3D1)%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20out%5B'total_participants'%5D%20%3D%20total%0A%20%20%20%20%20%20%20%20out%5B'prof_share'%5D%20%3D%20out%5B'Professionals_Count'%5D%20%2F%20total.replace(0%2C%20np.nan)%0A%20%20%20%20%20%20%20%20out%5B'total_hours'%5D%20%3D%20out%5B'Practical_Hours'%5D%20%2B%20out%5B'Theory_Hours'%5D%0A%20%20%20%20%20%20%20%20out%5B'practical_share'%5D%20%3D%20out%5B'Practical_Hours'%5D%20%2F%20out%5B'total_hours'%5D.replace(%0A%20%20%20%20%20%20%20%20%20%20%20%200%2C%20np.nan%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20out%5B'cost_x_days'%5D%20%3D%20out%5B'Daily_Tuition_Cost'%5D%20*%20out%5B'total_hours'%5D%0A%20%20%20%20%20%20%20%20out%5B'prev_drop_rate'%5D%20%3D%20out%5B'Prev_Course_Dropouts'%5D%20%2F%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20out%5B'Prev_Course_Attended'%5D%20%2B%201%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20out%5B'kits_per_participant'%5D%20%3D%20out%5B'Physical_Course_Kits'%5D%20%2F%20total.replace(%0A%20%20%20%20%20%20%20%20%20%20%20%200%2C%20np.nan%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20out%5B'tickets_per_participant'%5D%20%3D%20out%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20'Pre_Course_Supports_Tickets'%0A%20%20%20%20%20%20%20%20%5D%20%2F%20total.replace(0%2C%20np.nan)%0A%20%20%20%20%20%20%20%20out%5B'got_requested_lab'%5D%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20df%5B'Requested_Lab_Config'%5D%20%3D%3D%20df%5B'Assigned_Lab_Config'%5D%0A%20%20%20%20%20%20%20%20).astype(float)%0A%20%20%20%20%20%20%20%20for%20col%20in%20('Company_ID'%2C%20'Agent_ID')%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20out%5Bf'has_%7Bcol.lower()%7D'%5D%20%3D%20df%5Bcol%5D.notna().astype(int)%0A%20%20%20%20%20%20%20%20for%20col%20in%20frequency_cols%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20out%5Bf'%7Bcol%7D_freq'%5D%20%3D%20df%5Bcol%5D.map(freq_maps%5Bcol%5D).fillna(0).astype(float)%0A%20%20%20%20%20%20%20%20for%20col%20in%20native_cat_cols%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20out%5Bcol%5D%20%3D%20df%5Bcol%5D.fillna('missing').astype('category')%0A%20%20%20%20%20%20%20%20return%20out%0A%0A%20%20%20%20def%20align_categories(train_X%2C%20*others)%3A%0A%20%20%20%20%20%20%20%20%22%22%22Give%20every%20frame%20identical%20category%20levels%20so%20the%20boosters%20agree.%22%22%22%0A%20%20%20%20%20%20%20%20for%20col%20in%20train_X.select_dtypes('category').columns%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20cats%20%3D%20train_X%5Bcol%5D.cat.categories%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20other%20in%20others%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20cats%20%3D%20cats.union(other%5Bcol%5D.cat.categories)%0A%20%20%20%20%20%20%20%20%20%20%20%20train_X%5Bcol%5D%20%3D%20train_X%5Bcol%5D.cat.set_categories(cats)%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20other%20in%20others%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20other%5Bcol%5D%20%3D%20other%5Bcol%5D.cat.set_categories(cats)%0A%0A%20%20%20%20X_all%20%3D%20build_features(train_raw%2C%20freq_maps)%0A%20%20%20%20cat_cols%20%3D%20X_all.select_dtypes('category').columns%0A%20%20%20%20native_dim%20%3D%20X_all.shape%5B1%5D%0A%20%20%20%20onehot_dim%20%3D%20X_all.drop(columns%3Dcat_cols).shape%5B1%5D%20%2B%20sum(%0A%20%20%20%20%20%20%20%20(X_all%5Bc%5D.nunique(dropna%3DFalse)%20for%20c%20in%20cat_cols)%0A%20%20%20%20)%0A%20%20%20%20print(f'features%20with%20native%20categorical%20handling%20%3A%20%7Bnative_dim%7D')%0A%20%20%20%20print(f'estimated%20dims%20after%20naive%20one-hot%20%20%20%20%20%20%20%20%3A%20%7Bonehot_dim%7D')%0A%20%20%20%20print(f'dummy%20columns%20avoided%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%3A%20%7Bonehot_dim%20-%20native_dim%7D')%0A%20%20%20%20return%20align_categories%2C%20build_features%2C%20make_freq_maps%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20tree%2Fnative-categorical%20path%20contains%2042%20columns.%20A%20naive%20one-hot%20expansion%20of%20the%20same%20fields%20would%20create%20about%20435%20columns%2C%20mostly%20from%20agent%20and%20country%2C%20so%20this%20representation%20avoids%20393%20sparse%20dummy%20columns.%20The%20linear%20and%20neural%20baselines%20use%20one-hot%20encoding%20with%20rare%20levels%20grouped%20into%20%60other%60.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%206.%20Validation%20methodology%0A%0A%20%20%20%20Before%20comparing%20models%2C%20we%20need%20a%20validation%20setup%20that%20resembles%20the%20later%20test%20window.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%206.1%20Adversarial%20validation%20%E2%80%94%20quantifying%20the%20drift%0A%0A%20%20%20%20We%20train%20a%20classifier%20to%20tell%20**test%20rows%20from%20train%20rows**%20using%20the%20features%20(label%20removed%2C%20raw%20date%20and%20%60Client_ID%60%20dropped).%20If%20it%20separates%20them%20well%20above%20AUC%200.5%2C%20the%20feature%20distributions%20have%20genuinely%20drifted.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20SEED%2C%0A%20%20%20%20TARGET%2C%0A%20%20%20%20TEST_PATH%2C%0A%20%20%20%20TRAIN_PATH%2C%0A%20%20%20%20XGBClassifier%2C%0A%20%20%20%20cache%2C%0A%20%20%20%20load_raw%2C%0A%20%20%20%20pd%2C%0A%20%20%20%20roc_auc_score%2C%0A%20%20%20%20train_test_split%2C%0A)%3A%0A%20%20%20%20%40cache%20%0A%20%20%20%20def%20adversarial_validation(tr%2C%20te%2C%20target%2C%20seed)%3A%0A%20%20%20%20%20%20%20%20tr%20%3D%20tr.drop(columns%3D%5Btarget%5D)%0A%20%20%20%20%20%20%20%20combined%20%3D%20pd.concat(%0A%20%20%20%20%20%20%20%20%20%20%20%20%5Btr.assign(is_test%3D0)%2C%20te.assign(is_test%3D1)%5D%2C%20ignore_index%3DTrue%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20y%20%3D%20combined.pop(%22is_test%22)%0A%20%20%20%20%20%20%20%20X%20%3D%20combined.drop(columns%3D%5B%22Client_ID%22%2C%20%22Course_Start_Date%22%5D%2C%20errors%3D%22ignore%22)%0A%20%20%20%20%20%20%20%20for%20c%20in%20X.select_dtypes(include%3D%5B%22object%22%2C%20%22string%22%5D).columns%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20X%5Bc%5D%20%3D%20X%5Bc%5D.astype(%22string%22).str.strip().str.lower().fillna(%22missing%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20X%5Bc%5D%20%3D%20X%5Bc%5D.astype(%22category%22).cat.codes%0A%20%20%20%20%20%20%20%20X%20%3D%20X.fillna(-1)%0A%0A%20%20%20%20%20%20%20%20Xtr%2C%20Xva%2C%20ytr%2C%20yva%20%3D%20train_test_split(%0A%20%20%20%20%20%20%20%20%20%20%20%20X%2C%20y%2C%20test_size%3D0.25%2C%20random_state%3Dseed%2C%20stratify%3Dy%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20m%20%3D%20XGBClassifier(%0A%20%20%20%20%20%20%20%20%20%20%20%20n_estimators%3D300%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20max_depth%3D4%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20learning_rate%3D0.05%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20subsample%3D0.9%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20colsample_bytree%3D0.9%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20eval_metric%3D%22logloss%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20random_state%3Dseed%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20n_jobs%3D-1%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20m.fit(Xtr%2C%20ytr)%0A%20%20%20%20%20%20%20%20a%20%3D%20roc_auc_score(yva%2C%20m.predict_proba(Xva)%5B%3A%2C%201%5D)%0A%20%20%20%20%20%20%20%20top%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20pd%0A%20%20%20%20%20%20%20%20%20%20%20%20.Series(m.feature_importances_%2C%20index%3DX.columns)%0A%20%20%20%20%20%20%20%20%20%20%20%20.sort_values(ascending%3DFalse)%0A%20%20%20%20%20%20%20%20%20%20%20%20.head(8)%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20return%20a%2C%20top%0A%0A%20%20%20%20adversarial_auc%2C%20drift_drivers%20%3D%20adversarial_validation(%0A%20%20%20%20%20%20%20%20load_raw(TRAIN_PATH)%2C%20load_raw(TEST_PATH)%2C%20TARGET%2C%20SEED%0A%20%20%20%20)%0A%20%20%20%20print(%0A%20%20%20%20%20%20%20%20f%22adversarial%20AUC%20(train%20vs%20test)%3A%20%7Badversarial_auc%3A.3f%7D%20%20%22%0A%20%20%20%20%20%20%20%20%22(0.5%3Didentical%2C%201.0%3Dtrivially%20separable)%22%0A%20%20%20%20)%0A%20%20%20%20print(%22%5Cntop%20drift%20drivers%3A%22)%0A%20%20%20%20print(drift_drivers)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20classifier%20reaches%20AUC%200.935%2C%20so%20train%20and%20test%20features%20are%20distinguishable%20even%20after%20removing%20the%20raw%20date.%20The%20strongest%20differences%20include%20tuition%20cost%2C%20client%20history%2C%20registration%20lead%20time%2C%20waiting%20time%2C%20and%20several%20categorical%20fields.%20Together%20with%20the%20changing%20monthly%20drop%20rate%2C%20this%20leads%20us%20to%20evaluate%20models%20on%20a%20later%20time%20window.%0A%0A%20%20%20%20%23%23%206.2%20The%20chronological%20holdout%0A%0A%20%20%20%20We%20use%20%602017-01-01%60%20as%20the%20cutoff%20because%20it%20leaves%20roughly%20four%20months%20for%20validation%2C%20matching%20the%20length%20and%20future-facing%20structure%20of%20the%20hidden%20test%20window.%20All%20model%20and%20feature%20comparisons%20fit%20on%20the%20earlier%20rows%20and%20evaluate%20on%20this%20later%20holdout.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(TARGET%2C%20align_categories%2C%20build_features%2C%20make_freq_maps%2C%20pd%2C%20train_raw)%3A%0A%20%20%20%20CHRONO_CUTOFF%20%3D%20'2017-01-01'%0A%20%20%20%20cutoff%20%3D%20pd.Timestamp(CHRONO_CUTOFF)%0A%20%20%20%20tr_raw%20%3D%20train_raw%5Btrain_raw%5B%22Course_Start_Date%22%5D%20%3C%20cutoff%5D%0A%20%20%20%20va_raw%20%3D%20train_raw%5Btrain_raw%5B%22Course_Start_Date%22%5D%20%3E%3D%20cutoff%5D%0A%20%20%20%20y_tr%20%3D%20tr_raw%5BTARGET%5D.values%0A%20%20%20%20y_va%20%3D%20va_raw%5BTARGET%5D.values%0A%20%20%20%20print(%0A%20%20%20%20%20%20%20%20f%22chrono%20split%20-%3E%20fit%3D%7Blen(tr_raw)%3A%2C%7D%20%20validate%3D%7Blen(va_raw)%3A%2C%7D%20%20%22%0A%20%20%20%20%20%20%20%20f%22(val%20drop%20rate%3D%7Bva_raw%5BTARGET%5D.mean()%3A.3f%7D)%22%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20feature%20matrices%20reused%20across%20all%20experiments%20below.%20For%20chronological%0A%20%20%20%20%23%20validation%2C%20frequency%20maps%20are%20fit%20only%20on%20the%20past%20training%20window.%20The%0A%20%20%20%20%23%20train%2Btest%20map%20above%20is%20reserved%20for%20submission-time%20label-free%20transductive%20scoring.%0A%20%20%20%20freq_maps_chrono%20%3D%20make_freq_maps(tr_raw)%0A%20%20%20%20Xtr_without_time%20%3D%20build_features(%0A%20%20%20%20%20%20%20%20tr_raw%2C%20freq_maps_chrono%2C%20add_time%3DFalse%2C%20add_week%3DTrue%0A%20%20%20%20)%0A%20%20%20%20Xva_without_time%20%3D%20build_features(%0A%20%20%20%20%20%20%20%20va_raw%2C%20freq_maps_chrono%2C%20add_time%3DFalse%2C%20add_week%3DTrue%0A%20%20%20%20)%0A%20%20%20%20align_categories(Xtr_without_time%2C%20Xva_without_time)%0A%20%20%20%20Xtr_t%20%3D%20build_features(tr_raw%2C%20freq_maps_chrono%2C%20add_time%3DTrue%2C%20add_week%3DTrue)%0A%20%20%20%20Xva_t%20%3D%20build_features(va_raw%2C%20freq_maps_chrono%2C%20add_time%3DTrue%2C%20add_week%3DTrue)%0A%20%20%20%20align_categories(Xtr_t%2C%20Xva_t)%0A%20%20%20%20return%20Xtr_t%2C%20Xtr_without_time%2C%20Xva_t%2C%20Xva_without_time%2C%20va_raw%2C%20y_tr%2C%20y_va%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%206.3%20Why%20a%20random%20split%20is%20misleading%0A%0A%20%20%20%20To%20isolate%20the%20effect%20of%20the%20split%20itself%2C%20we%20fit%20the%20same%20fixed%20reference%20XGBoost%20configuration%20once%20on%20the%20chronological%20split%20and%20once%20on%20a%20random%20split%20of%20similar%20size.%20This%20is%20a%20validation%20diagnostic%2C%20not%20the%20model-selection%20result%20used%20later.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20SEED%2C%0A%20%20%20%20TARGET%2C%0A%20%20%20%20XGBClassifier%2C%0A%20%20%20%20Xtr_t%2C%0A%20%20%20%20Xva_t%2C%0A%20%20%20%20align_categories%2C%0A%20%20%20%20build_features%2C%0A%20%20%20%20cache%2C%0A%20%20%20%20display%2C%0A%20%20%20%20make_freq_maps%2C%0A%20%20%20%20pd%2C%0A%20%20%20%20roc_auc_score%2C%0A%20%20%20%20train_raw%2C%0A%20%20%20%20train_test_split%2C%0A%20%20%20%20va_raw%2C%0A%20%20%20%20y_tr%2C%0A%20%20%20%20y_va%2C%0A)%3A%0A%20%20%20%20%40cache%0A%20%20%20%20def%20fit_split_diagnostic(X_train%2C%20y_train%2C%20X_valid%2C%20seed)%3A%0A%20%20%20%20%20%20%20%20model%20%3D%20XGBClassifier(%0A%20%20%20%20%20%20%20%20%20%20%20%20n_estimators%3D300%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20learning_rate%3D0.05%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20max_depth%3D6%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20enable_categorical%3DTrue%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20tree_method%3D'hist'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20eval_metric%3D'auc'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20random_state%3Dseed%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20n_jobs%3D-1%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20model.fit(X_train%2C%20y_train)%0A%20%20%20%20%20%20%20%20return%20model.predict_proba(X_valid)%5B%3A%2C%201%5D%0A%0A%20%20%20%20random_fraction%20%3D%20len(va_raw)%20%2F%20len(train_raw)%0A%20%20%20%20tr_random%2C%20va_random%20%3D%20train_test_split(%0A%20%20%20%20%20%20%20%20train_raw%2C%0A%20%20%20%20%20%20%20%20test_size%3Drandom_fraction%2C%0A%20%20%20%20%20%20%20%20random_state%3DSEED%2C%0A%20%20%20%20%20%20%20%20stratify%3Dtrain_raw%5BTARGET%5D%2C%0A%20%20%20%20)%0A%20%20%20%20random_maps%20%3D%20make_freq_maps(tr_random)%0A%20%20%20%20Xtr_random%20%3D%20build_features(tr_random%2C%20random_maps)%0A%20%20%20%20Xva_random%20%3D%20build_features(va_random%2C%20random_maps)%0A%20%20%20%20align_categories(Xtr_random%2C%20Xva_random)%0A%0A%20%20%20%20split_check%20%3D%20pd.DataFrame(%7B%0A%20%20%20%20%20%20%20%20'split'%3A%20%5B'Chronological%20future%20holdout'%2C%20'Random%20holdout%20(diagnostic)'%5D%2C%0A%20%20%20%20%20%20%20%20'AUC'%3A%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20roc_auc_score(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20y_va%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20fit_split_diagnostic(Xtr_t%2C%20y_tr%2C%20Xva_t%2C%20SEED)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20roc_auc_score(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20va_random%5BTARGET%5D.values%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20fit_split_diagnostic(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20Xtr_random%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20tr_random%5BTARGET%5D.values%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20Xva_random%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20SEED%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%7D)%0A%20%20%20%20display(split_check.round(4))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20displayed%20comparison%20shows%20that%20the%20same%20fixed%20reference%20model%20scores%20materially%20higher%20on%20the%20random%20holdout%20than%20on%20the%20chronological%20future%20window.%20The%20random%20split%20mixes%20older%20and%20newer%20registrations%2C%20so%20it%20produces%20an%20optimistic%20score%20for%20a%20genuinely%20future-facing%20task.%20We%20therefore%20use%20the%20chronological%20holdout%20for%20every%20model%20and%20feature%20decision%20below.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%207.%20Model%20experiments%20%26%20tuning%0A%0A%20%20%20%20The%20assignment%20requires%20at%20least%20three%20models%20and%20hyperparameter%20tuning.%20We%20compare%20one%20linear%20model%2C%20one%20neural%20network%2C%20and%20one%20boosted-tree%20model%20on%20the%20same%20chronological%20holdout.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%207.1%20The%20model%20families%0A%0A%20%20%20%20We%20tune%20one%20important%20capacity%20or%20regularization%20parameter%20for%20each%20family%20and%20compare%20the%20selected%20settings%20on%20the%20same%20holdout.%20The%20strongest%20family%20becomes%20the%20development%20focus%20of%20the%20next%20experiments%3B%20final%20model%20selection%20is%20deferred%20until%20all%20candidates%20are%20evaluated%20in%20Section%208.%0A%0A%20%20%20%20Each%20model%20family%20is%20paired%20with%20its%20appropriate%20preprocessing%20pipeline%3A%20bounded%20one-hot%20and%20scaling%20for%20the%20continuous%20baselines%2C%20and%20native%20categorical%20handling%20for%20the%20tree%20boosters.%0A%0A%20%20%20%20-%20**Logistic%20Regression**%20provides%20an%20interpretable%20linear%20reference.%20Its%20main%20tuning%20parameter%20here%20is%20%60C%60%2C%20the%20inverse%20regularization%20strength.%0A%20%20%20%20-%20**MLP**%20can%20learn%20nonlinear%20combinations%20but%20requires%20a%20complete%2C%20scaled%20numeric%20matrix.%20We%20vary%20the%20number%20of%2064-unit%20hidden%20layers%20while%20keeping%20the%20remaining%20training%20settings%20fixed.%0A%20%20%20%20-%20**XGBoost**%20builds%20trees%20sequentially%20so%20later%20trees%20correct%20earlier%20errors.%20It%20can%20represent%20thresholds%20and%20interactions%20directly%3B%20we%20tune%20tree%20depth%20and%20then%20the%20learning-rate%2Ftree-count%20budget.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(OneHotEncoder%2C%20cache%2C%20pd)%3A%0A%20%20%20%20%40cache%20%0A%20%20%20%20def%20encode_for_continuous_models(%0A%20%20%20%20%20%20%20%20X_tr%3A%20pd.DataFrame%2C%20X_va%3A%20pd.DataFrame%2C%20min_count%3D30%0A%20%20%20%20)%3A%0A%20%20%20%20%20%20%20%20%22%22%22Bounded%20one-hot%20%2B%20median%20imputation%20for%20the%20LR%2FMLP%20baselines.%22%22%22%0A%20%20%20%20%20%20%20%20Xt%2C%20Xv%20%3D%20X_tr.copy()%2C%20X_va.copy()%0A%20%20%20%20%20%20%20%20cat_cols%20%3D%20list(Xt.select_dtypes(%22category%22).columns)%0A%0A%20%20%20%20%20%20%20%20%23%20Collapse%20rare%20levels%20to%20%22other%22%20so%20the%20one-hot%20matrix%20stays%20bounded%20and%20the%0A%20%20%20%20%20%20%20%20%23%20linear%2FMLP%20baselines%20do%20not%20overfit%20categories%20with%20only%20a%20few%20examples.%0A%20%20%20%20%20%20%20%20for%20c%20in%20cat_cols%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20train_s%20%3D%20Xt%5Bc%5D.astype(%22string%22).fillna(%22missing%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20keep%20%3D%20train_s.value_counts()%5Blambda%20s%3A%20s%20%3E%3D%20min_count%5D.index%0A%20%20%20%20%20%20%20%20%20%20%20%20Xt%5Bc%5D%20%3D%20train_s.where(train_s.isin(keep)%2C%20%22other%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20Xv%5Bc%5D%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20Xv%5Bc%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.astype(%22string%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.fillna(%22missing%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20.where(lambda%20s%3A%20s.isin(keep)%2C%20%22other%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20num_cols%20%3D%20%5Bc%20for%20c%20in%20Xt.columns%20if%20c%20not%20in%20cat_cols%5D%0A%20%20%20%20%20%20%20%20medians%20%3D%20Xt%5Bnum_cols%5D.median()%0A%20%20%20%20%20%20%20%20Xt_num%20%3D%20Xt%5Bnum_cols%5D.fillna(medians).reset_index(drop%3DTrue)%0A%20%20%20%20%20%20%20%20Xv_num%20%3D%20Xv%5Bnum_cols%5D.fillna(medians).reset_index(drop%3DTrue)%0A%0A%20%20%20%20%20%20%20%20ohe%20%3D%20OneHotEncoder(handle_unknown%3D%22ignore%22%2C%20sparse_output%3DFalse%2C%20dtype%3Dfloat)%0A%20%20%20%20%20%20%20%20Xt_cat%20%3D%20ohe.fit_transform(Xt%5Bcat_cols%5D)%0A%20%20%20%20%20%20%20%20Xv_cat%20%3D%20ohe.transform(Xv%5Bcat_cols%5D)%0A%20%20%20%20%20%20%20%20names%20%3D%20ohe.get_feature_names_out(cat_cols)%0A%0A%20%20%20%20%20%20%20%20Xt_out%20%3D%20pd.concat(%5BXt_num%2C%20pd.DataFrame(Xt_cat%2C%20columns%3Dnames)%5D%2C%20axis%3D1)%0A%20%20%20%20%20%20%20%20Xv_out%20%3D%20pd.concat(%5BXv_num%2C%20pd.DataFrame(Xv_cat%2C%20columns%3Dnames)%5D%2C%20axis%3D1)%0A%20%20%20%20%20%20%20%20return%20Xt_out%2C%20Xv_out%0A%0A%20%20%20%20return%20(encode_for_continuous_models%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%207.2%20Focused%20hyperparameter%20tuning%0A%0A%20%20%20%20For%20each%20required%20family%2C%20we%20vary%20one%20capacity%20or%20regularization%20parameter%20and%20evaluate%20it%20with%20chronological%20ROC-AUC.%20Logistic%20Regression%20and%20MLP%20select%20the%20best%20validation%20score%3B%20XGBoost%20uses%20a%20parsimonious%20rule%2C%20choosing%20the%20smallest%20depth%20within%200.0005%20AUC%20of%20the%20best%20result.%20Training%20AUC%20is%20shown%20to%20reveal%20when%20extra%20capacity%20improves%20fit%20without%20helping%20the%20future%20holdout.%0A%0A%20%20%20%20For%20the%20MLP%2C%20the%20sweep%20varies%20hidden%20depth%20while%20keeping%20each%20layer%20at%2064%20units.%20For%20XGBoost%2C%20it%20varies%20%60max_depth%60%20while%20holding%20the%20remaining%20settings%20fixed.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20LogisticRegression%2C%0A%20%20%20%20MLPClassifier%2C%0A%20%20%20%20SEED%2C%0A%20%20%20%20StandardScaler%2C%0A%20%20%20%20XGBClassifier%2C%0A%20%20%20%20Xtr_t%2C%0A%20%20%20%20Xva_t%2C%0A%20%20%20%20cache%2C%0A%20%20%20%20encode_for_continuous_models%2C%0A%20%20%20%20pd%2C%0A%20%20%20%20roc_auc_score%2C%0A%20%20%20%20y_tr%2C%0A%20%20%20%20y_va%2C%0A)%3A%0A%20%20%20%20xgb_baseline_params%20%3D%20%7B%0A%20%20%20%20%20%20%20%20'colsample_bytree'%3A%200.8%2C%0A%20%20%20%20%20%20%20%20'enable_categorical'%3A%20True%2C%0A%20%20%20%20%20%20%20%20'tree_method'%3A%20'hist'%2C%0A%20%20%20%20%20%20%20%20'eval_metric'%3A%20'auc'%2C%0A%20%20%20%20%20%20%20%20'n_jobs'%3A%20-1%2C%0A%20%20%20%20%20%20%20%20'min_child_weight'%3A%205%2C%0A%20%20%20%20%20%20%20%20'subsample'%3A%200.9%2C%0A%20%20%20%20%20%20%20%20'reg_alpha'%3A%200.0%2C%0A%20%20%20%20%20%20%20%20'reg_lambda'%3A%201.0%2C%0A%20%20%20%20%7D%0A%0A%20%20%20%20def%20make_xgb(seed%3DSEED%2C%20**params)%3A%0A%20%20%20%20%20%20%20%20return%20XGBClassifier(**params%2C%20random_state%3Dseed)%0A%0A%20%20%20%20Xtr_enc%2C%20Xva_enc%20%3D%20encode_for_continuous_models(Xtr_t%2C%20Xva_t)%0A%20%20%20%20print(f'continuous%20baseline%20features%20after%20bounded%20one-hot%3A%20%7BXtr_enc.shape%5B1%5D%7D')%0A%20%20%20%20%23%20Scale%20the%20continuous%20baselines%20after%20fitting%20the%20encoder%20on%20the%20past%20window.%0A%20%20%20%20scaler%20%3D%20StandardScaler()%0A%20%20%20%20Xtr_scaled%20%3D%20scaler.fit_transform(Xtr_enc)%0A%20%20%20%20Xva_scaled%20%3D%20scaler.transform(Xva_enc)%0A%0A%20%20%20%20%40cache%20%0A%20%20%20%20def%20hyper_tune(%0A%20%20%20%20%20%20%20%20X_train_scaled%2C%0A%20%20%20%20%20%20%20%20X_valid_scaled%2C%0A%20%20%20%20%20%20%20%20X_train_tree%2C%0A%20%20%20%20%20%20%20%20X_valid_tree%2C%0A%20%20%20%20%20%20%20%20y_train%2C%0A%20%20%20%20%20%20%20%20y_valid%2C%0A%20%20%20%20%20%20%20%20seed%2C%0A%20%20%20%20%20%20%20%20xgb_fixed_params%2C%0A%20%20%20%20)%3A%0A%20%20%20%20%20%20%20%20tuning_rows%20%3D%20%5B%5D%0A%20%20%20%20%20%20%20%20validation_predictions%20%3D%20%7B%7D%0A%0A%20%20%20%20%20%20%20%20def%20record_trial(family%2C%20axis%2C%20x%2C%20model%2C%20X_train%2C%20X_valid%2C%20keep%3DFalse)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20train_predictions%20%3D%20model.predict_proba(X_train)%5B%3A%2C%201%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20valid_predictions%20%3D%20model.predict_proba(X_valid)%5B%3A%2C%201%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20tuning_rows.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'family'%3A%20family%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'axis'%3A%20axis%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'x'%3A%20x%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'train_AUC'%3A%20roc_auc_score(y_train%2C%20train_predictions)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'val_AUC'%3A%20roc_auc_score(y_valid%2C%20valid_predictions)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20keep%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20validation_predictions%5B(family%2C%20x)%5D%20%3D%20valid_predictions%0A%0A%20%20%20%20%20%20%20%20for%20C%20in%20(0.001%2C%200.01%2C%200.1%2C%201.0%2C%2010.0%2C%20100.0)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20model%20%3D%20LogisticRegression(C%3DC%2C%20max_iter%3D2000).fit(X_train_scaled%2C%20y_train)%0A%20%20%20%20%20%20%20%20%20%20%20%20record_trial(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'Logistic%20Regression'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'C%20%20%E2%86%92%20(less%20regularisation)'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20C%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20model%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20X_train_scaled%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20X_valid_scaled%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20keep%3DTrue%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20for%20hidden_layers%20in%20(1%2C%202%2C%203%2C%204)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20model%20%3D%20MLPClassifier(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20hidden_layer_sizes%3D(64%2C)%20*%20hidden_layers%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20learning_rate_init%3D0.001%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20max_iter%3D150%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20early_stopping%3DTrue%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20n_iter_no_change%3D10%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20random_state%3Dseed%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20).fit(X_train_scaled%2C%20y_train)%0A%20%20%20%20%20%20%20%20%20%20%20%20record_trial(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'MLP%20neural%20network'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'hidden%20layers%20%E2%86%92%20(more%20depth)'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20hidden_layers%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20model%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20X_train_scaled%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20X_valid_scaled%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20keep%3DTrue%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20for%20depth%20in%20(2%2C%203%2C%204%2C%205%2C%206%2C%208%2C%2010)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20model%20%3D%20make_xgb(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20seed%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20**xgb_fixed_params%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20n_estimators%3D300%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20learning_rate%3D0.05%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20max_depth%3Ddepth%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20).fit(X_train_tree%2C%20y_train)%0A%20%20%20%20%20%20%20%20%20%20%20%20record_trial(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'Gradient-boosted%20trees%20(XGBoost)'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'max_depth%20%E2%86%92%20(more%20capacity)'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20depth%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20model%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20X_train_tree%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20X_valid_tree%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20return%20tuning_rows%2C%20validation_predictions%0A%0A%20%20%20%20tuning_rows%2C%20validation_predictions%20%3D%20hyper_tune(%0A%20%20%20%20%20%20%20%20Xtr_scaled%2C%0A%20%20%20%20%20%20%20%20Xva_scaled%2C%0A%20%20%20%20%20%20%20%20Xtr_t%2C%0A%20%20%20%20%20%20%20%20Xva_t%2C%0A%20%20%20%20%20%20%20%20y_tr%2C%0A%20%20%20%20%20%20%20%20y_va%2C%0A%20%20%20%20%20%20%20%20SEED%2C%0A%20%20%20%20%20%20%20%20xgb_baseline_params%2C%0A%20%20%20%20)%0A%20%20%20%20tuning_raw%20%3D%20pd.DataFrame(tuning_rows)%0A%20%20%20%20return%20make_xgb%2C%20tuning_raw%2C%20validation_predictions%2C%20xgb_baseline_params%0A%0A%0A%40app.cell%0Adef%20_(display%2C%20tuning_raw%2C%20validation_predictions)%3A%0A%20%20%20%20tuning%20%3D%20tuning_raw.copy()%0A%20%20%20%20tuning%5B'selected'%5D%20%3D%20False%0A%20%20%20%20for%20family%2C%20trials%20in%20tuning.groupby('family'%2C%20sort%3DFalse)%3A%0A%20%20%20%20%20%20%20%20if%20family%20%3D%3D%20'Gradient-boosted%20trees%20(XGBoost)'%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20best_auc%20%3D%20trials%5B'val_AUC'%5D.max()%0A%20%20%20%20%20%20%20%20%20%20%20%20eligible%20%3D%20trials%5Btrials%5B'val_AUC'%5D%20%3E%3D%20best_auc%20-%200.0005%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20chosen_index%20%3D%20eligible%5B'x'%5D.idxmin()%0A%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20chosen_index%20%3D%20trials%5B'val_AUC'%5D.idxmax()%0A%20%20%20%20%20%20%20%20tuning.loc%5Bchosen_index%2C%20'selected'%5D%20%3D%20True%0A%20%20%20%20tuning%5B'log_x'%5D%20%3D%20tuning%5B'family'%5D.eq('Logistic%20Regression')%0A%0A%20%20%20%20selected_tuning%20%3D%20tuning.loc%5B%0A%20%20%20%20%20%20%20%20tuning%5B'selected'%5D%2C%20%5B'family'%2C%20'x'%2C%20'train_AUC'%2C%20'val_AUC'%5D%0A%20%20%20%20%5D.copy()%0A%20%20%20%20selected_tuning%5B%5B'train_AUC'%2C%20'val_AUC'%5D%5D%20%3D%20selected_tuning%5B%0A%20%20%20%20%20%20%20%20%5B'train_AUC'%2C%20'val_AUC'%5D%0A%20%20%20%20%5D.round(4)%0A%20%20%20%20display(selected_tuning)%0A%0A%20%20%20%20selected_x%20%3D%20selected_tuning.set_index('family')%5B'x'%5D%0A%20%20%20%20pred_lr%20%3D%20validation_predictions%5B%0A%20%20%20%20%20%20%20%20('Logistic%20Regression'%2C%20selected_x.loc%5B'Logistic%20Regression'%5D)%0A%20%20%20%20%5D%0A%20%20%20%20pred_mlp%20%3D%20validation_predictions%5B%0A%20%20%20%20%20%20%20%20('MLP%20neural%20network'%2C%20selected_x.loc%5B'MLP%20neural%20network'%5D)%0A%20%20%20%20%5D%0A%20%20%20%20selected_depth%20%3D%20int(selected_x.loc%5B'Gradient-boosted%20trees%20(XGBoost)'%5D)%0A%20%20%20%20return%20pred_lr%2C%20pred_mlp%2C%20selected_depth%2C%20tuning%0A%0A%0A%40app.cell%0Adef%20_(show%2C%20sns%2C%20tuning)%3A%0A%20%20%20%20def%20plot_auc_sweep(data%2C%20x%2C%20facet%2C%20title)%3A%0A%20%20%20%20%20%20%20%20curves%20%3D%20data.melt(%0A%20%20%20%20%20%20%20%20%20%20%20%20id_vars%3D%5Bfacet%2C%20x%2C%20'axis'%2C%20'log_x'%2C%20'selected'%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20value_vars%3D%5B'train_AUC'%2C%20'val_AUC'%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20var_name%3D'split'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20value_name%3D'ROC-AUC'%2C%0A%20%20%20%20%20%20%20%20).replace(%7B'split'%3A%20%7B'train_AUC'%3A%20'Train'%2C%20'val_AUC'%3A%20'Validation'%7D%7D)%0A%20%20%20%20%20%20%20%20grid%20%3D%20sns.relplot(%0A%20%20%20%20%20%20%20%20%20%20%20%20data%3Dcurves%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dx%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D'ROC-AUC'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20hue%3D'split'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20style%3D'split'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20kind%3D'line'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20markers%3DTrue%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20dashes%3DFalse%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20facet_kws%3D%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'sharex'%3A%20False%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'sharey'%3A%20facet%20!%3D%20'family'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'legend_out'%3A%20False%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20col%3Dfacet%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20grid.figure.set_size_inches(10%2C%204.375)%0A%20%20%20%20%20%20%20%20grid.set_titles('').set_ylabels('ROC-AUC')%0A%20%20%20%20%20%20%20%20for%20value%2C%20ax%20in%20grid.axes_dict.items()%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20subset%20%3D%20data%5Bdata%5Bfacet%5D.eq(value)%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20ax.set(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20title%3Dstr(value)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20facet%20%3D%3D%20'family'%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20else%20f'%7Bfacet.replace(%22_%22%2C%20%22%20%22)%7D%20%3D%20%7Bvalue%7D'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20xlabel%3Dsubset%5B'axis'%5D.iloc%5B0%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20subset%5B'log_x'%5D.iloc%5B0%5D%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20ax.set_xscale('log')%0A%20%20%20%20%20%20%20%20%20%20%20%20chosen%20%3D%20subset.loc%5Bsubset%5B'selected'%5D%5D.iloc%5B0%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20ax.scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20chosen%5Bx%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20chosen%5B'val_AUC'%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20marker%3D'*'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20s%3D190%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20color%3D'%23E69F00'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20edgecolor%3D'black'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20linewidth%3D0.7%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20zorder%3D5%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20grid.figure.suptitle(title%2C%20y%3D0.98)%0A%20%20%20%20%20%20%20%20grid.figure.subplots_adjust(%0A%20%20%20%20%20%20%20%20%20%20%20%20top%3D0.80%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20wspace%3D0.25%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20show(grid.figure)%0A%0A%20%20%20%20plot_auc_sweep(%0A%20%20%20%20%20%20%20%20tuning%2C%20'x'%2C%20'family'%2C%20'Focused%20tuning%3A%20training%20vs%20validation%20ROC-AUC'%0A%20%20%20%20)%0A%20%20%20%20return%20(plot_auc_sweep%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20gold%20star%20marks%20the%20setting%20selected%20by%20the%20tuning%20rule.%20For%20Logistic%20Regression%20and%20MLP%20this%20is%20the%20maximum%20validation%20AUC.%20For%20XGBoost%20it%20is%20the%20smallest%20depth%20within%200.0005%20of%20the%20best%20validation%20score%2C%20avoiding%20extra%20capacity%20for%20a%20negligible%20gain.%20XGBoost%20is%20the%20strongest%20development%20candidate%20on%20the%20future%20holdout%2C%20so%20its%20selected%20depth%20flows%20into%20the%20remaining%20experiments%20while%20Logistic%20Regression%20and%20MLP%20remain%20in%20the%20final%20comparison.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%207.3%20Improving%20the%20gradient%20model%0A%0A%20%20%20%20%23%23%23%20(a)%20Boosting%20budget%20and%20regularization%0A%0A%20%20%20%20The%20number%20of%20trees%20and%20the%20learning%20rate%20interact%20directly.%20We%20first%20evaluate%20various%20tree%20counts%20across%20two%20learning-rate%20settings%20using%20the%20baseline%20profile%20from%20Section%207.2.%20We%20then%20hold%20the%20selected%20depth%2C%20learning%20rate%2C%20and%20tree%20count%20fixed%20and%20compare%20that%20baseline%20with%20one%20pre-specified%20stronger%20regularization%20profile.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20SEED%2C%0A%20%20%20%20Xtr_t%2C%0A%20%20%20%20Xva_t%2C%0A%20%20%20%20cache%2C%0A%20%20%20%20display%2C%0A%20%20%20%20make_xgb%2C%0A%20%20%20%20pd%2C%0A%20%20%20%20plot_auc_sweep%2C%0A%20%20%20%20roc_auc_score%2C%0A%20%20%20%20selected_depth%2C%0A%20%20%20%20xgb_baseline_params%2C%0A%20%20%20%20y_tr%2C%0A%20%20%20%20y_va%2C%0A)%3A%0A%20%20%20%20%40cache%20%0A%20%20%20%20def%20run_budget_sweep(%0A%20%20%20%20%20%20%20%20X_train%2C%0A%20%20%20%20%20%20%20%20X_valid%2C%0A%20%20%20%20%20%20%20%20y_train%2C%0A%20%20%20%20%20%20%20%20y_valid%2C%0A%20%20%20%20%20%20%20%20seed%2C%0A%20%20%20%20%20%20%20%20depth%2C%0A%20%20%20%20%20%20%20%20xgb_fixed_params%2C%0A%20%20%20%20)%3A%0A%20%20%20%20%20%20%20%20budget_rows%20%3D%20%5B%5D%0A%20%20%20%20%20%20%20%20for%20lr_rate%20in%20(0.1%2C%200.03)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20n%20in%20(50%2C%20100%2C%20200%2C%20400%2C%20700%2C%201000)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20model%20%3D%20make_xgb(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20seed%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20**xgb_fixed_params%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20n_estimators%3Dn%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20learning_rate%3Dlr_rate%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20max_depth%3Ddepth%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20model.fit(X_train%2C%20y_train)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20train_predictions%20%3D%20model.predict_proba(X_train)%5B%3A%2C%201%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20valid_predictions%20%3D%20model.predict_proba(X_valid)%5B%3A%2C%201%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20budget_rows.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'learning_rate'%3A%20lr_rate%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'n_trees'%3A%20n%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'train_AUC'%3A%20roc_auc_score(y_train%2C%20train_predictions)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'val_AUC'%3A%20roc_auc_score(y_valid%2C%20valid_predictions)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20return%20budget_rows%0A%0A%20%20%20%20budget_rows%20%3D%20run_budget_sweep(%0A%20%20%20%20%20%20%20%20Xtr_t%2C%0A%20%20%20%20%20%20%20%20Xva_t%2C%0A%20%20%20%20%20%20%20%20y_tr%2C%0A%20%20%20%20%20%20%20%20y_va%2C%0A%20%20%20%20%20%20%20%20SEED%2C%0A%20%20%20%20%20%20%20%20selected_depth%2C%0A%20%20%20%20%20%20%20%20xgb_baseline_params%2C%0A%20%20%20%20)%0A%20%20%20%20budget%20%3D%20pd.DataFrame(budget_rows)%0A%20%20%20%20budget%5B'axis'%5D%20%3D%20'number%20of%20trees'%0A%20%20%20%20budget%5B'log_x'%5D%20%3D%20False%0A%20%20%20%20budget%5B'selected'%5D%20%3D%20False%0A%20%20%20%20budget.loc%5Bbudget.groupby('learning_rate')%5B'val_AUC'%5D.idxmax()%2C%20'selected'%5D%20%3D%20True%0A%20%20%20%20plot_auc_sweep(%0A%20%20%20%20%20%20%20%20budget%2C%0A%20%20%20%20%20%20%20%20'n_trees'%2C%0A%20%20%20%20%20%20%20%20'learning_rate'%2C%0A%20%20%20%20%20%20%20%20'Boosting%20budget%3A%20train%20vs%20validation%20ROC-AUC'%2C%0A%20%20%20%20)%0A%0A%20%20%20%20budget_best%20%3D%20budget.loc%5B%0A%20%20%20%20%20%20%20%20budget.groupby('learning_rate')%5B'val_AUC'%5D.idxmax()%2C%0A%20%20%20%20%20%20%20%20%5B'learning_rate'%2C%20'n_trees'%2C%20'train_AUC'%2C%20'val_AUC'%5D%2C%0A%20%20%20%20%5D.copy()%0A%20%20%20%20budget_best%5B%5B'train_AUC'%2C%20'val_AUC'%5D%5D%20%3D%20budget_best%5B%5B'train_AUC'%2C%20'val_AUC'%5D%5D.round(%0A%20%20%20%20%20%20%20%204%0A%20%20%20%20)%0A%20%20%20%20display(budget_best)%0A%20%20%20%20selected_budget%20%3D%20budget.loc%5Bbudget%5B'val_AUC'%5D.idxmax()%5D%0A%20%20%20%20selected_learning_rate%20%3D%20float(selected_budget%5B'learning_rate'%5D)%0A%20%20%20%20selected_n_trees%20%3D%20int(selected_budget%5B'n_trees'%5D)%0A%20%20%20%20print(%0A%20%20%20%20%20%20%20%20f%22selected%20XGBoost%20budget%3A%20learning_rate%3D%7Bselected_learning_rate%3Ag%7D%2C%20%22%0A%20%20%20%20%20%20%20%20f%22n_estimators%3D%7Bselected_n_trees%7D%22%0A%20%20%20%20)%0A%20%20%20%20xgb_strong_params%20%3D%20%7B%0A%20%20%20%20%20%20%20%20**xgb_baseline_params%2C%0A%20%20%20%20%20%20%20%20'min_child_weight'%3A%2010%2C%0A%20%20%20%20%20%20%20%20'subsample'%3A%200.8%2C%0A%20%20%20%20%20%20%20%20'reg_alpha'%3A%200.1%2C%0A%20%20%20%20%20%20%20%20'reg_lambda'%3A%203.0%2C%0A%20%20%20%20%7D%0A%20%20%20%20regularization_profiles%20%3D%20%7B%0A%20%20%20%20%20%20%20%20'Baseline'%3A%20xgb_baseline_params%2C%0A%20%20%20%20%20%20%20%20'Stronger'%3A%20xgb_strong_params%2C%0A%20%20%20%20%7D%0A%20%20%20%20strong_model%20%3D%20make_xgb(%0A%20%20%20%20%20%20%20%20SEED%2C%0A%20%20%20%20%20%20%20%20**xgb_strong_params%2C%0A%20%20%20%20%20%20%20%20max_depth%3Dselected_depth%2C%0A%20%20%20%20%20%20%20%20learning_rate%3Dselected_learning_rate%2C%0A%20%20%20%20%20%20%20%20n_estimators%3Dselected_n_trees%2C%0A%20%20%20%20).fit(Xtr_t%2C%20y_tr)%0A%20%20%20%20regularization_comparison%20%3D%20pd.DataFrame(%5B%0A%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20'profile'%3A%20'Baseline'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'train_AUC'%3A%20float(selected_budget%5B'train_AUC'%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'val_AUC'%3A%20float(selected_budget%5B'val_AUC'%5D)%2C%0A%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20'profile'%3A%20'Stronger'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'train_AUC'%3A%20roc_auc_score(y_tr%2C%20strong_model.predict_proba(Xtr_t)%5B%3A%2C%201%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'val_AUC'%3A%20roc_auc_score(y_va%2C%20strong_model.predict_proba(Xva_t)%5B%3A%2C%201%5D)%2C%0A%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%5D)%0A%20%20%20%20selected_profile_name%20%3D%20regularization_comparison.loc%5B%0A%20%20%20%20%20%20%20%20regularization_comparison%5B'val_AUC'%5D.idxmax()%2C%20'profile'%0A%20%20%20%20%5D%0A%20%20%20%20regularization_comparison%5B'selected'%5D%20%3D%20regularization_comparison%5B'profile'%5D.eq(%0A%20%20%20%20%20%20%20%20selected_profile_name%0A%20%20%20%20)%0A%20%20%20%20display(regularization_comparison.round(4))%0A%20%20%20%20print(f%22selected%20XGBoost%20regularization%20profile%3A%20%7Bselected_profile_name%7D%22)%0A%20%20%20%20selected_xgb_fixed_params%20%3D%20regularization_profiles%5Bselected_profile_name%5D%0A%20%20%20%20return%20selected_learning_rate%2C%20selected_n_trees%2C%20selected_xgb_fixed_params%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20baseline%20profile%20selects%20%60learning_rate%3D0.03%60%20and%20700%20trees%20with%20validation%20AUC%200.9135%2C%20compared%20with%200.9125%20for%20the%20best%20%60learning_rate%3D0.1%60%20setting%20at%20200%20trees.%20At%20that%20selected%20structure%2C%20the%20stronger%20profile%20improves%20validation%20AUC%20to%200.9144%2C%20so%20we%20retain%20%60min_child_weight%3D10%60%2C%20%60subsample%3D0.8%60%2C%20%60reg_alpha%3D0.1%60%2C%20and%20%60reg_lambda%3D3.0%60%20for%20the%20selected%20XGBoost%20component.%20The%20training%20curves%20also%20show%20why%20we%20stop%20at%20the%20validation-selected%20budget%3A%20additional%20trees%20can%20continue%20improving%20fit%20after%20future-window%20performance%20has%20stopped%20improving.%0A%0A%20%20%20%20%23%23%23%20(b)%20Does%20adding%20other%20boosters%20help%3F%0A%0A%20%20%20%20LightGBM%20and%20CatBoost%20build%20boosted%20trees%20differently%20from%20XGBoost%2C%20so%20they%20may%20rank%20some%20registrations%20differently.%20We%20add%20them%20with%20fixed%2C%20capacity-aligned%20settings%20and%20test%20the%20ensemble%20itself%3A%20does%20combining%20their%20rankings%20improve%20the%20tuned%20XGBoost%20result%3F%0A%0A%20%20%20%20Because%20ROC-AUC%20depends%20on%20ranking%2C%20we%20convert%20each%20model's%20predictions%20to%20percentile%20ranks%20before%20averaging.%20This%20prevents%20one%20probability%20scale%20from%20dominating%20the%20blend%2C%20but%20the%20resulting%20score%20is%20not%20a%20calibrated%20cancellation%20probability.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20CatBoostClassifier%2C%0A%20%20%20%20LGBMClassifier%2C%0A%20%20%20%20SEED%2C%0A%20%20%20%20XGBClassifier%2C%0A%20%20%20%20cache%2C%0A%20%20%20%20np%2C%0A%20%20%20%20rankdata%2C%0A)%3A%0A%20%20%20%20%40cache%0A%20%20%20%20def%20fit_booster_component(name%2C%20X_train%2C%20y_train%2C%20params)%3A%0A%20%20%20%20%20%20%20%20%22%22%22Fit%20one%20explicitly%20configured%20blend%20component.%22%22%22%0A%20%20%20%20%20%20%20%20if%20name%20%3D%3D%20'cat'%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20cat_columns%20%3D%20X_train.select_dtypes('category').columns.tolist()%0A%20%20%20%20%20%20%20%20%20%20%20%20cat_indices%20%3D%20%5BX_train.columns.get_loc(column)%20for%20column%20in%20cat_columns%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20cat_train%20%3D%20X_train.copy()%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20column%20in%20cat_columns%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20cat_train%5Bcolumn%5D%20%3D%20cat_train%5Bcolumn%5D.astype(str)%0A%20%20%20%20%20%20%20%20%20%20%20%20model%20%3D%20CatBoostClassifier(**params%2C%20cat_features%3Dcat_indices)%0A%20%20%20%20%20%20%20%20%20%20%20%20model.fit(cat_train%2C%20y_train)%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20model%0A%20%20%20%20%20%20%20%20if%20name%20%3D%3D%20'lgbm'%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20model%20%3D%20LGBMClassifier(**params)%0A%20%20%20%20%20%20%20%20%20%20%20%20model.fit(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20X_train%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20y_train%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20categorical_feature%3DX_train.select_dtypes('category').columns.tolist()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20model%0A%20%20%20%20%20%20%20%20model%20%3D%20XGBClassifier(**params)%0A%20%20%20%20%20%20%20%20model.fit(X_train%2C%20y_train)%0A%20%20%20%20%20%20%20%20return%20model%0A%0A%20%20%20%20class%20BoostedRankBlend%3A%0A%20%20%20%20%20%20%20%20%22%22%22Three%20explicit%20boosters%20combined%20by%20batch-wise%20percentile%20ranks.%22%22%22%0A%0A%20%20%20%20%20%20%20%20component_names%20%3D%20('lgbm'%2C%20'xgb'%2C%20'cat')%0A%0A%20%20%20%20%20%20%20%20def%20__init__(%0A%20%20%20%20%20%20%20%20%20%20%20%20self%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20*%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20xgb_depth%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20xgb_learning_rate%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20xgb_n_estimators%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20xgb_fixed_params%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20random_state%3DSEED%2C%0A%20%20%20%20%20%20%20%20)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20self.xgb_depth%20%3D%20xgb_depth%0A%20%20%20%20%20%20%20%20%20%20%20%20self.xgb_learning_rate%20%3D%20xgb_learning_rate%0A%20%20%20%20%20%20%20%20%20%20%20%20self.xgb_n_estimators%20%3D%20xgb_n_estimators%0A%20%20%20%20%20%20%20%20%20%20%20%20self.xgb_fixed_params%20%3D%20dict(xgb_fixed_params)%0A%20%20%20%20%20%20%20%20%20%20%20%20self.random_state%20%3D%20random_state%0A%20%20%20%20%20%20%20%20%20%20%20%20self.component_params%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'lgbm'%3A%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'n_estimators'%3A%20700%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'learning_rate'%3A%200.03%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'num_leaves'%3A%2063%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'min_child_samples'%3A%2040%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'subsample'%3A%200.9%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'subsample_freq'%3A%201%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'colsample_bytree'%3A%200.8%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'reg_lambda'%3A%201.0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'random_state'%3A%20random_state%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'n_jobs'%3A%20-1%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'verbosity'%3A%20-1%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'xgb'%3A%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20**self.xgb_fixed_params%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'random_state'%3A%20random_state%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'max_depth'%3A%20xgb_depth%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'learning_rate'%3A%20xgb_learning_rate%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'n_estimators'%3A%20xgb_n_estimators%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'cat'%3A%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'iterations'%3A%201200%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'learning_rate'%3A%200.05%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'depth'%3A%206%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'l2_leaf_reg'%3A%203.0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'random_seed'%3A%20random_state%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'verbose'%3A%20False%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'allow_writing_files'%3A%20False%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'eval_metric'%3A%20'AUC'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%0A%0A%20%20%20%20%20%20%20%20def%20fit(self%2C%20X%2C%20y)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20self.feature_names_in_%20%3D%20list(X.columns)%0A%20%20%20%20%20%20%20%20%20%20%20%20self.categorical_columns_%20%3D%20X.select_dtypes('category').columns.tolist()%0A%20%20%20%20%20%20%20%20%20%20%20%20self.models_%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20name%3A%20fit_booster_component(name%2C%20X%2C%20y%2C%20self.component_params%5Bname%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20name%20in%20self.component_names%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20self%0A%0A%20%20%20%20%20%20%20%20def%20clone(self)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%22%22Return%20an%20unfitted%20blend%20with%20the%20same%20selected%20configuration.%22%22%22%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20type(self)(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20xgb_depth%3Dself.xgb_depth%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20xgb_learning_rate%3Dself.xgb_learning_rate%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20xgb_n_estimators%3Dself.xgb_n_estimators%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20xgb_fixed_params%3Dself.xgb_fixed_params%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20random_state%3Dself.random_state%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20def%20_prediction_frame(self%2C%20name%2C%20X)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20list(X.columns)%20!%3D%20self.feature_names_in_%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20raise%20ValueError(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'Prediction%20columns%20must%20match%20the%20fitted%20feature%20matrix.'%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20name%20!%3D%20'cat'%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20return%20X%0A%20%20%20%20%20%20%20%20%20%20%20%20cat_frame%20%3D%20X.copy()%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20column%20in%20self.categorical_columns_%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20cat_frame%5Bcolumn%5D%20%3D%20cat_frame%5Bcolumn%5D.astype(str)%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20cat_frame%0A%0A%20%20%20%20%20%20%20%20def%20predict_component(self%2C%20name%2C%20X)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20self.models_%5Bname%5D.predict_proba(self._prediction_frame(name%2C%20X))%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%3A%2C%201%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D%0A%0A%20%20%20%20%20%20%20%20def%20predict_components(self%2C%20X)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20name%3A%20self.predict_component(name%2C%20X)%20for%20name%20in%20self.component_names%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%0A%0A%20%20%20%20%20%20%20%20%40staticmethod%0A%20%20%20%20%20%20%20%20def%20rank_average(component_scores)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%22%22Average%20batch-wise%20percentile%20ranks%2C%20not%20calibrated%20probabilities.%22%22%22%0A%20%20%20%20%20%20%20%20%20%20%20%20scores%20%3D%20list(component_scores.values())%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20np.mean(%5Brankdata(score)%20%2F%20len(score)%20for%20score%20in%20scores%5D%2C%20axis%3D0)%0A%0A%20%20%20%20%20%20%20%20def%20predict_rank_score(self%2C%20X)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%22%22Score%20one%20complete%20evaluation%20batch%20with%20the%20selected%20rank%20blend.%22%22%22%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20self.rank_average(self.predict_components(X))%0A%0A%20%20%20%20%20%20%20%20%40property%0A%20%20%20%20%20%20%20%20def%20xgb_model_(self)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20self.models_%5B'xgb'%5D%0A%0A%20%20%20%20return%20(BoostedRankBlend%2C)%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20BoostedRankBlend%2C%0A%20%20%20%20SEED%2C%0A%20%20%20%20Xtr_t%2C%0A%20%20%20%20Xva_t%2C%0A%20%20%20%20display%2C%0A%20%20%20%20pd%2C%0A%20%20%20%20roc_auc_score%2C%0A%20%20%20%20selected_depth%2C%0A%20%20%20%20selected_learning_rate%2C%0A%20%20%20%20selected_n_trees%2C%0A%20%20%20%20selected_xgb_fixed_params%2C%0A%20%20%20%20y_tr%2C%0A%20%20%20%20y_va%2C%0A)%3A%0A%20%20%20%20selected_blend%20%3D%20BoostedRankBlend(%0A%20%20%20%20%20%20%20%20xgb_depth%3Dselected_depth%2C%0A%20%20%20%20%20%20%20%20xgb_learning_rate%3Dselected_learning_rate%2C%0A%20%20%20%20%20%20%20%20xgb_n_estimators%3Dselected_n_trees%2C%0A%20%20%20%20%20%20%20%20xgb_fixed_params%3Dselected_xgb_fixed_params%2C%0A%20%20%20%20%20%20%20%20random_state%3DSEED%2C%0A%20%20%20%20).fit(Xtr_t%2C%20y_tr)%0A%20%20%20%20pred_t%20%3D%20selected_blend.predict_components(Xva_t)%0A%20%20%20%20blend_t%20%3D%20selected_blend.rank_average(pred_t)%0A%0A%20%20%20%20xgb_auc%20%3D%20roc_auc_score(y_va%2C%20pred_t%5B'xgb'%5D)%0A%20%20%20%20blend_check%20%3D%20(%0A%20%20%20%20%20%20%20%20pd%0A%20%20%20%20%20%20%20%20.DataFrame(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20'model'%3A%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'LightGBM%20(fixed%20setting)'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'XGBoost%20(selected%20configuration)'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'CatBoost%20(fixed%20setting)'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20'Rank-average%20blend%20(LGBM%2BXGB%2BCat)'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'chrono_AUC'%3A%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20roc_auc_score(y_va%2C%20pred_t%5B'lgbm'%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20xgb_auc%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20roc_auc_score(y_va%2C%20pred_t%5B'cat'%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20roc_auc_score(y_va%2C%20blend_t)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20.sort_values('chrono_AUC'%2C%20ascending%3DFalse)%0A%20%20%20%20%20%20%20%20.reset_index(drop%3DTrue)%0A%20%20%20%20)%0A%20%20%20%20blend_check%5B'delta_vs_XGBoost'%5D%20%3D%20(blend_check%5B'chrono_AUC'%5D%20-%20xgb_auc).round(4)%0A%20%20%20%20display(blend_check)%0A%20%20%20%20return%20blend_t%2C%20selected_blend%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20selected%20regularized%20XGBoost%20component%20reaches%20AUC%200.9144%3B%20adding%20the%20fixed%20LightGBM%20and%20CatBoost%20rankings%20raises%20the%20complete%20blend%20to%20about%200.9159.%20We%20therefore%20treat%20the%20fitted%20rank%20blend%20as%20the%20single%20boosted-tree%20candidate%20for%20the%20remaining%20ablations%2C%20evaluation%2C%20and%20final%20refit.%0A%0A%20%20%20%20%23%23%23%20(c)%20Continuous%20time%20index%20on%20the%20complete%20blend%0A%0A%20%20%20%20EDA%20showed%20both%20seasonality%20and%20longer-term%20drift.%20We%20compare%20the%20complete%20blend%20using%20the%20calendar%20features%20alone%20with%20the%20same%20blend%20after%20adding%20%60days_since_epoch%60%2C%20holding%20all%20model%20settings%20fixed.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20Xtr_without_time%2C%0A%20%20%20%20Xva_without_time%2C%0A%20%20%20%20blend_t%2C%0A%20%20%20%20display%2C%0A%20%20%20%20pd%2C%0A%20%20%20%20roc_auc_score%2C%0A%20%20%20%20selected_blend%2C%0A%20%20%20%20y_tr%2C%0A%20%20%20%20y_va%2C%0A)%3A%0A%20%20%20%20without_time_blend%20%3D%20selected_blend.clone().fit(Xtr_without_time%2C%20y_tr)%0A%20%20%20%20blend_without_time%20%3D%20without_time_blend.predict_rank_score(Xva_without_time)%0A%20%20%20%20without_index_auc%20%3D%20roc_auc_score(y_va%2C%20blend_without_time)%0A%20%20%20%20with_index_auc%20%3D%20roc_auc_score(y_va%2C%20blend_t)%0A%20%20%20%20temporal_check%20%3D%20pd.DataFrame(%5B%7B%0A%20%20%20%20%20%20%20%20'without%20continuous%20index%20AUC'%3A%20without_index_auc%2C%0A%20%20%20%20%20%20%20%20'with%20continuous%20index%20AUC'%3A%20with_index_auc%2C%0A%20%20%20%20%20%20%20%20'AUC%20gain'%3A%20with_index_auc%20-%20without_index_auc%2C%0A%20%20%20%20%7D%5D)%0A%20%20%20%20display(temporal_check.round(6))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20Adding%20the%20continuous%20time%20index%20improves%20AUC%20from%200.912548%20to%200.915926%2C%20so%20we%20use%20it%20in%20the%20final%20evaluation.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%208.%20Model%20evaluation%0A%0A%20%20%20%20We%20compare%20the%20tuned%20Logistic%20Regression%2C%20MLP%2C%20and%20boosted-tree%20candidates%20on%20the%20chronological%20holdout.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%208.1%20ROC%20%26%20precision%E2%80%93recall%20curves%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20PrecisionRecallDisplay%2C%0A%20%20%20%20RocCurveDisplay%2C%0A%20%20%20%20blend_t%2C%0A%20%20%20%20pred_lr%2C%0A%20%20%20%20pred_mlp%2C%0A%20%20%20%20show%2C%0A%20%20%20%20subplot_grid%2C%0A%20%20%20%20y_va%2C%0A)%3A%0A%20%20%20%20curve_candidates%20%3D%20%5B%0A%20%20%20%20%20%20%20%20('Logistic%20Regression'%2C%20pred_lr)%2C%0A%20%20%20%20%20%20%20%20('MLP'%2C%20pred_mlp)%2C%0A%20%20%20%20%20%20%20%20('Boosted-tree%20rank%20blend'%2C%20blend_t)%2C%0A%20%20%20%20%5D%0A%20%20%20%20curve_displays%20%3D%20%5B%0A%20%20%20%20%20%20%20%20(RocCurveDisplay%2C%20'ROC%20curve')%2C%0A%20%20%20%20%20%20%20%20(PrecisionRecallDisplay%2C%20'Precision%E2%80%93Recall%20curve')%2C%0A%20%20%20%20%5D%0A%20%20%20%20curve_fig%2C%20curve_axes%20%3D%20subplot_grid(%0A%20%20%20%20%20%20%20%201%2C%202%2C%20figsize%3D(15%2C%205.5)%2C%20layout%3D'compressed'%0A%20%20%20%20)%0A%20%20%20%20for%20curve_ax%2C%20(display_class%2C%20curve_title)%20in%20zip(curve_axes%2C%20curve_displays)%3A%0A%20%20%20%20%20%20%20%20for%20curve_index%2C%20(candidate_name%2C%20candidate_predictions)%20in%20enumerate(%0A%20%20%20%20%20%20%20%20%20%20%20%20curve_candidates%0A%20%20%20%20%20%20%20%20)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20display_class.from_predictions(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20y_va%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20candidate_predictions%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20name%3Dcandidate_name%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20ax%3Dcurve_ax%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20plot_chance_level%3Dcurve_index%20%3D%3D%20len(curve_candidates)%20-%201%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20despine%3DTrue%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20curve_ax.set_title(curve_title)%0A%20%20%20%20show(curve_fig)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20boosted-tree%20blend%20has%20the%20highest%20holdout%20ROC-AUC%20(0.916)%20and%20Average%20Precision%20(0.897)%20in%20this%20comparison.%20Threshold-based%20metrics%20are%20examined%20next.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%208.2%20Confusion%20matrices%20%26%20threshold%20metrics%0A%0A%20%20%20%20A%20confusion%20matrix%20requires%20a%20threshold%2C%20so%20we%20use%200.5%20as%20a%20simple%20reference%20cutoff%20for%20each%20candidate.%20For%20the%20rank-average%20blend%2C%20this%20is%20not%20a%2050%25%20cancellation%20probability%3A%20rank%20averaging%20preserves%20ordering%20but%20discards%20the%20individual%20models'%20probability%20scales.%20Nova%20Academy%20could%20later%20adjust%20the%20cutoff%20according%20to%20the%20relative%20cost%20of%20unnecessary%20follow-up%20and%20missed%20cancellations.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20ConfusionMatrixDisplay%2C%0A%20%20%20%20average_precision_score%2C%0A%20%20%20%20blend_t%2C%0A%20%20%20%20classification_report%2C%0A%20%20%20%20display%2C%0A%20%20%20%20pd%2C%0A%20%20%20%20pred_lr%2C%0A%20%20%20%20pred_mlp%2C%0A%20%20%20%20roc_auc_score%2C%0A%20%20%20%20show%2C%0A%20%20%20%20subplot_grid%2C%0A%20%20%20%20y_va%2C%0A)%3A%0A%20%20%20%20matrix_candidates%20%3D%20%5B%0A%20%20%20%20%20%20%20%20('Logistic%20Regression'%2C%20pred_lr)%2C%0A%20%20%20%20%20%20%20%20('MLP%20neural%20network'%2C%20pred_mlp)%2C%0A%20%20%20%20%20%20%20%20('Boosted-tree%20rank%20blend'%2C%20blend_t)%2C%0A%20%20%20%20%5D%0A%20%20%20%20matrix_fig%2C%20matrix_axes%20%3D%20subplot_grid(1%2C%203%2C%20figsize%3D(16%2C%203.5))%0A%20%20%20%20metric_rows%20%3D%20%5B%5D%0A%20%20%20%20for%20matrix_ax%2C%20(matrix_name%2C%20matrix_predictions)%20in%20zip(%0A%20%20%20%20%20%20%20%20matrix_axes%2C%20matrix_candidates%0A%20%20%20%20)%3A%0A%20%20%20%20%20%20%20%20matrix_labels%20%3D%20(matrix_predictions%20%3E%3D%200.5).astype(int)%0A%20%20%20%20%20%20%20%20matrix_auc%20%3D%20roc_auc_score(y_va%2C%20matrix_predictions)%0A%20%20%20%20%20%20%20%20matrix_ap%20%3D%20average_precision_score(y_va%2C%20matrix_predictions)%0A%20%20%20%20%20%20%20%20matrix_report%20%3D%20classification_report(%0A%20%20%20%20%20%20%20%20%20%20%20%20y_va%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20matrix_labels%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20target_names%3D%5B'completed'%2C%20'dropped'%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20output_dict%3DTrue%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20zero_division%3D0%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20ConfusionMatrixDisplay.from_predictions(%0A%20%20%20%20%20%20%20%20%20%20%20%20y_va%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20matrix_labels%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20display_labels%3D%5B'completed'%2C%20'dropped'%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20values_format%3D'%2Cd'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20cmap%3D'Blues'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20colorbar%3DFalse%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20ax%3Dmatrix_ax%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20matrix_ax.set_title(f'%7Bmatrix_name%7D%5CnROC-AUC%3D%7Bmatrix_auc%3A.3f%7D')%0A%20%20%20%20%20%20%20%20matrix_ax.grid(False)%0A%20%20%20%20%20%20%20%20metric_rows.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20'model'%3A%20matrix_name%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'ROC-AUC'%3A%20matrix_auc%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'Average%20Precision'%3A%20matrix_ap%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'accuracy'%3A%20matrix_report%5B'accuracy'%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'precision%20(dropped)'%3A%20matrix_report%5B'dropped'%5D%5B'precision'%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'recall%20(dropped)'%3A%20matrix_report%5B'dropped'%5D%5B'recall'%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'F1%20(dropped)'%3A%20matrix_report%5B'dropped'%5D%5B'f1-score'%5D%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20matrix_fig.suptitle('Candidate-model%20confusion%20matrices%20at%20a%200.5%20reference%20cutoff')%0A%20%20%20%20show(matrix_fig)%0A%20%20%20%20evaluation_metrics%20%3D%20pd.DataFrame(metric_rows).set_index('model').round(3)%0A%20%20%20%20display(evaluation_metrics)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%208.3%20Model%20family%20comparison%20and%20final%20selection%0A%0A%20%20%20%20-%20Logistic%20Regression%20slightly%20outperformed%20the%20MLP%2C%20suggesting%20that%20non-linearity%20is%20not%20the%20primary%20limitation%20for%20continuous%20models%20on%20this%20dataset.%0A%20%20%20%20-%20The%20Logistic%20Regression%20and%20MLP%20ROC%20and%20Precision%E2%80%93Recall%20curves%20largely%20overlap%2C%20indicating%20that%20the%20two%20models%20produce%20very%20similar%20rankings%20across%20most%20decision%20thresholds.%0A%20%20%20%20-%20The%20blended%20gradient-boosted%20model%20consistently%20outperformed%20both%20the%20linear%20and%20neural-network%20models%20across%20the%20evaluated%20metrics.%0A%0A%20%20%20%20%23%23%23%20Interpreting%20the%20Confusion%20Matrices%20(0.5%20Threshold)%0A%0A%20%20%20%20At%20the%200.5%20reference%20cutoff%2C%20the%20three%20models%20exhibit%20different%20operating%20characteristics.%20The%20boosted-tree%20blend%20identifies%20more%20dropped%20registrations%20and%20misses%20fewer%20of%20them%20than%20the%20continuous%20baselines%2C%20at%20the%20cost%20of%20more%20false%20alarms.%20Precision%20measures%20how%20many%20flagged%20registrations%20were%20actually%20dropped%2C%20recall%20measures%20how%20many%20dropped%20registrations%20were%20found%2C%20F1%20balances%20those%20two%20rates%2C%20and%20accuracy%20summarizes%20all%20correct%20classifications.%0A%0A%20%20%20%20Note%20that%20the%20boosted-tree%20model%20produces%20a%20continuous%20risk%20score%2C%20its%20false-positive%2Ffalse-negative%20trade-off%20can%20be%20adjusted%20by%20selecting%20a%20different%20decision%20threshold.%20The%20confusion%20matrices%20therefore%20illustrate%20one%20operating%20point%20(0.5)%20rather%20than%20an%20inherent%20limitation%20of%20the%20model.%0A%0A%20%20%20%20%23%23%23%20Final%20model%20selection%0A%20%20%20%20The%20blended%20boosted-tree%20model%20achieved%20the%20highest%20AUC.%20Since%20its%20operating%20point%20can%20be%20adjusted%20by%20selecting%20an%20appropriate%20decision%20threshold%2C%20we%20chose%20it%20as%20our%20final%20submission.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%208.4%20Registrations%20near%20the%20illustrative%20threshold%0A%0A%20%20%20%20We%20inspect%20how%20many%20selected-blend%20scores%20fall%20near%20the%200.5%20reference%20cutoff.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(blend_t%2C%20show%2C%20sns%2C%20subplot_grid)%3A%0A%20%20%20%20score_fig%2C%20score_ax%20%3D%20subplot_grid()%0A%20%20%20%20sns.histplot(blend_t%2C%20bins%3D50%2C%20ax%3Dscore_ax)%0A%20%20%20%20score_ax.axvline(0.5%2C%20linestyle%3D'--'%2C%20label%3D'reference%20cutoff')%0A%20%20%20%20score_ax.axvspan(%0A%20%20%20%20%20%20%20%200.4%2C%200.6%2C%20color%3D'orange'%2C%20alpha%3D0.25%2C%20zorder%3D2%2C%20label%3D'near-threshold%20band'%0A%20%20%20%20)%0A%20%20%20%20score_ax.set(%0A%20%20%20%20%20%20%20%20xlabel%3D'Rank-average%20risk%20score'%2C%20title%3D'Selected-blend%20score%20distribution'%0A%20%20%20%20)%0A%20%20%20%20score_ax.legend()%0A%20%20%20%20near_threshold%20%3D%20((blend_t%20%3E%200.4)%20%26%20(blend_t%20%3C%200.6)).mean()%20*%20100%0A%20%20%20%20print(f'share%20of%20holdout%20in%20the%200.40%E2%80%930.60%20band%3A%20%7Bnear_threshold%3A.1f%7D%25')%0A%20%20%20%20show(score_fig)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20Scores%20in%20this%20band%20are%20close%20to%20the%20illustrative%20cutoff%2C%20so%20small%20changes%20in%20the%20cutoff%20can%20change%20their%20binary%20classification.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%209.%20Interpretation%20with%20SHAP%0A%0A%20%20%20%20The%20selected%20blend%20averages%20three%20sets%20of%20prediction%20ranks%20and%20therefore%20has%20no%20single%20fitted%20tree%20structure%20for%20SHAP%20to%20decompose.%20We%20use%20the%20tuned%20XGBoost%20component%20as%20a%20representative%20fitted%20model%20for%20detailed%20interpretation%2C%20then%20compare%20its%20SHAP%20patterns%20with%20the%20earlier%20EDA.%0A%0A%20%20%20%20We%20compute%20TreeSHAP%20values%20on%20a%20fixed%20validation%20sample%20of%20up%20to%2010%2C000%20rows%20to%20keep%20the%20analysis%20reproducible%20and%20the%20runtime%20manageable.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(SEED%2C%20Xva_t%2C%20cache%2C%20np%2C%20roc_auc_score%2C%20selected_blend%2C%20shap%2C%20y_va)%3A%0A%20%20%20%20%40cache%20%0A%20%20%20%20def%20compute_shap_analysis(%0A%20%20%20%20%20%20%20%20shap_model%2C%0A%20%20%20%20%20%20%20%20X_valid%2C%0A%20%20%20%20%20%20%20%20seed%2C%0A%20%20%20%20)%3A%0A%20%20%20%20%20%20%20%20valid_scores%20%3D%20shap_model.predict_proba(X_valid)%5B%3A%2C%201%5D%0A%20%20%20%20%20%20%20%20X_shap%20%3D%20X_valid.sample(min(10000%2C%20len(X_valid))%2C%20random_state%3Dseed)%0A%20%20%20%20%20%20%20%20sample_scores%20%3D%20shap_model.predict_proba(X_shap)%5B%3A%2C%201%5D%0A%20%20%20%20%20%20%20%20explainer%20%3D%20shap.TreeExplainer(shap_model)%0A%20%20%20%20%20%20%20%20shap_values%20%3D%20explainer.shap_values(X_shap)%0A%20%20%20%20%20%20%20%20if%20isinstance(shap_values%2C%20list)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20shap_values%20%3D%20shap_values%5B1%5D%0A%20%20%20%20%20%20%20%20shap_values%20%3D%20np.asarray(shap_values)%0A%20%20%20%20%20%20%20%20if%20shap_values.ndim%20%3D%3D%203%3A%20%20%23%20some%20shap%20versions%20return%20(n%2C%20features%2C%20classes)%0A%20%20%20%20%20%20%20%20%20%20%20%20shap_values%20%3D%20shap_values%5B%3A%2C%20%3A%2C%201%5D%0A%20%20%20%20%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20X_shap%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20sample_scores%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20explainer.expected_value%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20shap_values%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20valid_scores%2C%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20X_shap%2C%20shap_scores%2C%20shap_base%2C%20shap_values%2C%20shap_valid_scores%20%3D%20(%0A%20%20%20%20%20%20%20%20compute_shap_analysis(%0A%20%20%20%20%20%20%20%20%20%20%20%20selected_blend.xgb_model_%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20Xva_t%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20SEED%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20)%0A%20%20%20%20shap_auc%20%3D%20roc_auc_score(y_va%2C%20shap_valid_scores)%0A%20%20%20%20print(f%22XGBoost%2Btime%20chrono%20AUC%3A%20%7Bshap_auc%3A.4f%7D%22)%0A%20%20%20%20return%20X_shap%2C%20shap_base%2C%20shap_scores%2C%20shap_values%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%209.1%20Global%20importance%20(beeswarm%20%2B%20bar)%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(X_shap%2C%20display%2C%20np%2C%20pd%2C%20plt%2C%20shap%2C%20shap_values%2C%20show)%3A%0A%20%20%20%20shap.summary_plot(shap_values%2C%20X_shap%2C%20show%3DFalse%2C%20max_display%3D20)%0A%20%20%20%20plt.title('SHAP%20summary%20(beeswarm)%20%E2%80%94%20XGBoost%2Btime')%0A%20%20%20%20show()%0A%20%20%20%20importance%20%3D%20(%0A%20%20%20%20%20%20%20%20pd%0A%20%20%20%20%20%20%20%20.DataFrame(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20'feature'%3A%20X_shap.columns%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'mean_abs_shap'%3A%20np.abs(shap_values).mean(0)%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20.sort_values('mean_abs_shap'%2C%20ascending%3DFalse)%0A%20%20%20%20%20%20%20%20.reset_index(drop%3DTrue)%0A%20%20%20%20)%0A%20%20%20%20top%20%3D%20importance.head(20)%0A%0A%20%20%20%20importance_fig%2C%20importance_ax%20%3D%20plt.subplots(%0A%20%20%20%20%20%20%20%20figsize%3D(8%2C%207)%2C%20layout%3D'constrained'%0A%20%20%20%20)%0A%20%20%20%20top.sort_values('mean_abs_shap').plot.barh(%0A%20%20%20%20%20%20%20%20x%3D'feature'%2C%20y%3D'mean_abs_shap'%2C%20legend%3DFalse%2C%20ax%3Dimportance_ax%0A%20%20%20%20)%0A%20%20%20%20importance_ax.set(%0A%20%20%20%20%20%20%20%20xlabel%3D'mean%20%7CSHAP%20value%7C'%2C%20title%3D'Top%2020%20features%20by%20SHAP%20importance'%0A%20%20%20%20)%0A%20%20%20%20show(importance_fig)%0A%20%20%20%20display(top)%0A%20%20%20%20return%20(importance%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20strongest%20XGBoost%20contributions%20broadly%20match%20the%20earlier%20exploration.%20The%20leading%20features%20are%20%60Payment_Terms%60%2C%20%60Origin_Country%60%2C%20%60days_since_epoch%60%2C%20%60Agent_ID%60%2C%20%60tickets_per_participant%60%2C%20and%20%60Registration_Days_Before%60.%20Raw%20country%20and%20agent%20identity%20contribute%20more%20than%20their%20frequency%20encodings%2C%20while%20the%20engineered%20ratios%20add%20smaller%20supporting%20signals.%0A%0A%20%20%20%20%23%23%23%20Checking%20the%20suspicious%20%60Payment_Terms%60%20signal%0A%0A%20%20%20%20EDA%20showed%20that%20prepaid%2C%20non-refundable%20registrations%20drop%20unexpectedly%20often%2C%20and%20representative-model%20SHAP%20now%20ranks%20%60Payment_Terms%60%20first.%20To%20measure%20how%20strongly%20the%20selected%20model%20relies%20on%20it%2C%20we%20refit%20all%20three%20blend%20components%20without%20the%20field%20and%20compare%20chronological%20AUC.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20Xtr_t%2C%0A%20%20%20%20Xva_t%2C%0A%20%20%20%20blend_t%2C%0A%20%20%20%20display%2C%0A%20%20%20%20pd%2C%0A%20%20%20%20roc_auc_score%2C%0A%20%20%20%20selected_blend%2C%0A%20%20%20%20y_tr%2C%0A%20%20%20%20y_va%2C%0A)%3A%0A%20%20%20%20Xtr_no_payment%20%3D%20Xtr_t.drop(columns%3D%5B%22Payment_Terms%22%5D)%0A%20%20%20%20Xva_no_payment%20%3D%20Xva_t.drop(columns%3D%5B%22Payment_Terms%22%5D)%0A%20%20%20%20no_payment_blend%20%3D%20selected_blend.clone().fit(Xtr_no_payment%2C%20y_tr)%0A%20%20%20%20pred_blend_no_payment%20%3D%20no_payment_blend.predict_rank_score(Xva_no_payment)%0A%20%20%20%20payment_check%20%3D%20pd.DataFrame(%7B%0A%20%20%20%20%20%20%20%20%22model%22%3A%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Rank-average%20blend%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Rank-average%20blend%2C%20no%20Payment_Terms%22%2C%0A%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%20%20%20%20%22chrono_AUC%22%3A%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20roc_auc_score(y_va%2C%20blend_t)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20roc_auc_score(y_va%2C%20pred_blend_no_payment)%2C%0A%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%7D)%0A%20%20%20%20payment_check%5B%22delta_vs_with_payment%22%5D%20%3D%20(%0A%20%20%20%20%20%20%20%20payment_check%5B%22chrono_AUC%22%5D%20-%20payment_check.loc%5B0%2C%20%22chrono_AUC%22%5D%0A%20%20%20%20)%0A%20%20%20%20display(payment_check)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20Removing%20%60Payment_Terms%60%20from%20every%20blend%20component%20changes%20chronological%20AUC%20from%200.9159%20to%200.9101.%20This%20sensitivity%20result%20shows%20that%20the%20model%20relies%20on%20the%20field%2C%20but%20it%20does%20not%20prove%20that%20the%20field%20is%20safe%20from%20timing%20leakage%3B%20its%20exact%20recording%20time%20still%20needs%20confirmation%20with%20the%20data%20owner.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%209.2%20Direction%20of%20the%20strongest%20non-payment%20signal%0A%0A%20%20%20%20%60Origin_Country%60%20is%20the%20strongest%20feature%20after%20%60Payment_Terms%60%2C%20so%20we%20plot%20the%20average%20SHAP%20contribution%20of%20its%20most%20common%20levels.%20Positive%20values%20push%20XGBoost%20toward%20a%20higher%20cancellation%20score%3B%20negative%20values%20push%20it%20toward%20completion.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(X_shap%2C%20cache%2C%20importance%2C%20pd%2C%20plt%2C%20shap%2C%20shap_values%2C%20show%2C%20sns)%3A%0A%20%20%20%20top_feat%20%3D%20(%0A%20%20%20%20%20%20%20%20importance.loc%5Bimportance%5B'feature'%5D%20!%3D%20'Payment_Terms'%2C%20'feature'%5D.iloc%5B0%5D%0A%20%20%20%20%20%20%20%20if%20importance%5B'feature'%5D.iloc%5B0%5D%20%3D%3D%20'Payment_Terms'%0A%20%20%20%20%20%20%20%20else%20importance%5B'feature'%5D.iloc%5B0%5D%0A%20%20%20%20)%0A%20%20%20%20%40cache%20%0A%20%20%20%20def%20plot_shap_dependence_readable(feature)%3A%0A%20%20%20%20%20%20%20%20max_categories%20%3D%2015%0A%20%20%20%20%20%20%20%20col_idx%20%3D%20list(X_shap.columns).index(feature)%0A%20%20%20%20%20%20%20%20values%20%3D%20X_shap%5Bfeature%5D%0A%20%20%20%20%20%20%20%20is_categorical%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20str(values.dtype)%20%3D%3D%20'category'%0A%20%20%20%20%20%20%20%20%20%20%20%20or%20values.dtype%20%3D%3D%20'object'%0A%20%20%20%20%20%20%20%20%20%20%20%20or%20values.nunique(dropna%3DFalse)%20%3C%3D%20max_categories%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20if%20not%20is_categorical%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20shap.dependence_plot(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20feature%2C%20shap_values%2C%20X_shap%2C%20interaction_index%3DNone%2C%20show%3DFalse%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20plt.title(f'SHAP%20dependence%20%E2%80%94%20%7Bfeature%7D')%0A%20%20%20%20%20%20%20%20%20%20%20%20show()%0A%20%20%20%20%20%20%20%20%20%20%20%20return%0A%20%20%20%20%20%20%20%20labels%20%3D%20values.astype('string').fillna('missing')%0A%20%20%20%20%20%20%20%20keep%20%3D%20labels.value_counts().head(max_categories).index%0A%20%20%20%20%20%20%20%20grouped%20%3D%20pd.DataFrame(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20'level'%3A%20labels.where(labels.isin(keep)%2C%20'other')%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20'shap'%3A%20shap_values%5B%3A%2C%20col_idx%5D%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20summary%20%3D%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20grouped%0A%20%20%20%20%20%20%20%20%20%20%20%20.groupby('level'%2C%20observed%3DTrue)%0A%20%20%20%20%20%20%20%20%20%20%20%20.agg(mean_shap%3D('shap'%2C%20'mean')%2C%20n%3D('shap'%2C%20'size'))%0A%20%20%20%20%20%20%20%20%20%20%20%20.sort_values('mean_shap')%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20fig%2C%20ax%20%3D%20plt.subplots(figsize%3D(8%2C%207)%2C%20layout%3D'constrained')%0A%20%20%20%20%20%20%20%20sns.barplot(data%3Dsummary.reset_index()%2C%20y%3D'level'%2C%20x%3D'mean_shap'%2C%20ax%3Dax)%0A%20%20%20%20%20%20%20%20ax.axvline(0%2C%20color%3D'black'%2C%20linewidth%3D1)%0A%20%20%20%20%20%20%20%20ax.set(%0A%20%20%20%20%20%20%20%20%20%20%20%20title%3Df'Mean%20SHAP%20by%20%7Bfeature%7D%20level%20(top%20%7Bmax_categories%7D%20%2B%20other)'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20xlabel%3D'mean%20SHAP%20contribution'%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20ylabel%3Dfeature%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20show(fig)%0A%0A%20%20%20%20plot_shap_dependence_readable(top_feat)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%209.3%20Explaining%20one%20near-threshold%20registration%0A%0A%20%20%20%20We%20choose%20one%20sampled%20XGBoost%20prediction%20near%200.5%20and%20decompose%20it.%20The%20waterfall%20shows%20which%20features%20pushed%20this%20particular%20score%20upward%20and%20which%20pushed%20it%20downward.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(X_shap%2C%20np%2C%20shap%2C%20shap_base%2C%20shap_scores%2C%20shap_values%2C%20show)%3A%0A%20%20%20%20borderline%20%3D%20np.where((shap_scores%20%3E%200.45)%20%26%20(shap_scores%20%3C%200.55))%5B0%5D%0A%20%20%20%20idx%20%3D%20int(borderline%5B0%5D)%20if%20len(borderline)%20else%200%0A%20%20%20%20base%20%3D%20shap_base%0A%20%20%20%20if%20isinstance(base%2C%20(list%2C%20np.ndarray))%3A%0A%20%20%20%20%20%20%20%20base%20%3D%20np.asarray(base).ravel()%5B-1%5D%0A%0A%20%20%20%20print(%0A%20%20%20%20%20%20%20%20f%22explaining%20order%20at%20sample%20position%20%7Bidx%7D%20%E2%80%94%20model%20P(drop)%3D%7Bshap_scores%5Bidx%5D%3A.3f%7D%22%0A%20%20%20%20)%0A%20%20%20%20explanation%20%3D%20shap.Explanation(%0A%20%20%20%20%20%20%20%20values%3Dshap_values%5Bidx%5D%2C%0A%20%20%20%20%20%20%20%20base_values%3Dbase%2C%0A%20%20%20%20%20%20%20%20data%3DX_shap.iloc%5Bidx%5D.values%2C%0A%20%20%20%20%20%20%20%20feature_names%3Dlist(X_shap.columns)%2C%0A%20%20%20%20)%0A%20%20%20%20shap.plots.waterfall(explanation%2C%20max_display%3D14%2C%20show%3DFalse)%0A%20%20%20%20show()%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20For%20this%20registration%2C%20positive%20and%20negative%20contributions%20nearly%20balance%2C%20producing%20a%20score%20close%20to%20the%20reference%20threshold.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%2010.%20Building%20and%20checking%20the%20final%20submission%0A%0A%20%20%20%20We%20fit%20the%20selected%20blend%20on%20all%20labelled%20rows%2C%20score%20the%20official%20test%20set%2C%20and%20write%20the%20required%20two-column%20CSV.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20TARGET%2C%0A%20%20%20%20align_categories%2C%0A%20%20%20%20build_features%2C%0A%20%20%20%20make_freq_maps%2C%0A%20%20%20%20np%2C%0A%20%20%20%20pd%2C%0A%20%20%20%20selected_blend%2C%0A%20%20%20%20test_raw%2C%0A%20%20%20%20train_raw%2C%0A)%3A%0A%20%20%20%20submission_path%20%3D%20'data%2FGroup_27_Submission.csv'%0A%20%20%20%20submission_maps%20%3D%20make_freq_maps(train_raw%2C%20test_raw)%0A%20%20%20%20X_train_full%20%3D%20build_features(train_raw%2C%20submission_maps)%0A%20%20%20%20X_test%20%3D%20build_features(test_raw%2C%20submission_maps)%0A%20%20%20%20align_categories(X_train_full%2C%20X_test)%0A%20%20%20%20y_full%20%3D%20train_raw%5BTARGET%5D.values%0A%0A%20%20%20%20RUN_SUBMISSION%20%3D%20False%0A%20%20%20%20if%20RUN_SUBMISSION%3A%0A%20%20%20%20%20%20%20%20submission_blend%20%3D%20selected_blend.clone().fit(X_train_full%2C%20y_full)%0A%20%20%20%20%20%20%20%20submission_scores%20%3D%20submission_blend.predict_rank_score(X_test)%0A%20%20%20%20%20%20%20%20submission%20%3D%20pd.DataFrame(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Client_ID%22%3A%20test_raw%5B%22Client_ID%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Drop_Probability%22%3A%20submission_scores%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%20%20%20%20%20%20%20%20score_values%20%3D%20submission%5B'Drop_Probability'%5D.to_numpy()%0A%20%20%20%20%20%20%20%20assert%20list(submission.columns)%20%3D%3D%20%5B'Client_ID'%2C%20'Drop_Probability'%5D%0A%20%20%20%20%20%20%20%20assert%20len(submission)%20%3D%3D%2015866%0A%20%20%20%20%20%20%20%20assert%20submission%5B'Client_ID'%5D.reset_index(drop%3DTrue).equals(%0A%20%20%20%20%20%20%20%20%20%20%20%20test_raw%5B'Client_ID'%5D.reset_index(drop%3DTrue)%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20assert%20submission%5B'Client_ID'%5D.is_unique%0A%20%20%20%20%20%20%20%20assert%20not%20submission.isna().any().any()%0A%20%20%20%20%20%20%20%20assert%20np.isfinite(score_values).all()%0A%20%20%20%20%20%20%20%20assert%20((score_values%20%3E%3D%200)%20%26%20(score_values%20%3C%3D%201)).all()%0A%20%20%20%20%20%20%20%20submission.to_csv(submission_path%2C%20index%3DFalse)%0A%20%20%20%20%20%20%20%20print(f%22validated%20and%20wrote%20%7Bsubmission_path%7D%20(%7Blen(submission)%3A%2C%7D%20rows)%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%2011.%20Conclusions%20%26%20Executive%20Summary%0A%0A%20%20%20%20Nova%20Academy's%20test%20registrations%20occur%20after%20the%20training%20period%2C%20and%20both%20the%20monthly%20target%20rate%20and%20the%20adversarial-validation%20result%20(AUC%200.935)%20show%20temporal%20distribution%20shift.%20Model%20selection%20therefore%20used%20a%20four-month%20chronological%20holdout.%0A%0A%20%20%20%20Cleaning%20reduced%20hundreds%20of%20inconsistent%20text%20labels%20to%20compact%20category%20sets%2C%20while%20missingness%2C%20payment%20terms%2C%20country%2C%20agent%2C%20registration%20timing%2C%20and%20support%20activity%20carried%20predictive%20information.%20Logistic%20Regression%2C%20MLP%2C%20and%20XGBoost%20were%20tuned%20on%20the%20future%20holdout%3B%20XGBoost%20was%20strongest%2C%20and%20a%20controlled%20comparison%20selected%20stronger%20regularization.%20Adding%20fixed%20LightGBM%20and%20CatBoost%20components%20produced%20the%20final%20rank-average%20blend%2C%20which%20reached%20chronological%20ROC-AUC%200.9159%20and%20Average%20Precision%200.897.%0A%0A%20%20%20%20The%20final%20CSV%20contains%20continuous%20rank-average%20scores%20in%20the%20required%20two-column%20format.%20These%20scores%20are%20not%20calibrated%20probabilities%2C%20and%20hidden-test%20performance%20is%20unknown.%0A%0A%20%20%20%20Further%20work%20could%20include%20confirming%20when%20%60Payment_Terms%60%20is%20recorded%20and%20calibrating%20the%20selected%20blend%20score%20for%20cost-based%20operational%20thresholds.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
68ea38d01e4aae65475eac6af850a91a