{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":13439745,"sourceType":"datasetVersion","datasetId":8530502}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### PS-s4e12 - [Regression with an Insurance Dataset](https://www.kaggle.com/competitions/playground-series-s4e12/code?competitionId=84896&sortBy=scoreAscending&excludeNonAccessedDatasources=true)\n\nPlayground Series - Season 4, Episode 12\n\n&nbsp;\n\nPublic solutions:\n\n|[private](https://www.kaggle.com/competitions/playground-series-s4e12/leaderboard)|[public](https://www.kaggle.com/competitions/playground-series-s4e12/leaderboard?tab=public)| &nbsp; | | &nbsp; | &nbsp; | &nbsp; |\n|:-|:-| :-: | :-: | :-: | :-: | :-: |\n| 1.01654 | [1.01471](https://) |&nbsp;v.1&nbsp;| [vertical feature engineering](https://www.kaggle.com/code/nina2025/ps-s5e4-h-blend-fe) | **m**aster | [F&A](https://www.kaggle.com/nina2025) | Georgia |\n| 1.01891 | [1.01719](https://www.kaggle.com/code/masayakawamata/s4e12-l2-xgb?scriptVersionId=256801945) |&nbsp;v.1&nbsp;| [S4E12 . L2_XGB](https://www.kaggle.com/code/masayakawamata/s4e12-l2-xgb/notebook) | expert | [Masaya Kawamata](https://www.kaggle.com/masayakawamata) | Japan |\n| 1.02097 | [1.01909](https://www.kaggle.com/code/cdeotte/first-place-single-model-cv-1-016-lb-1-016?scriptVersionId=215561845) |&nbsp;v.1&nbsp;| [1-st Place - Single Model - [CV 1.016 LB 1.016]](https://www.kaggle.com/code/cdeotte/first-place-single-model-cv-1-016-lb-1-016/notebook) | expert | [Chris Deotte](https://www.kaggle.com/cdeotte) | United States |\n| 1.03057 | [1.02866](https://www.kaggle.com/code/mikhailnaumov/regression-with-an-insurance-cat-lgb-xgb-hgb-ydf?scriptVersionId=214862769) |&nbsp;v.4&nbsp;| [Insurance competition database](https://www.kaggle.com/code/martynovandrey/insurance-competition-database/notebook) | **m**aster | [Martynov Andrey](https://www.kaggle.com/martynovandrey) | World |\n| 1.03009 | [1.02871](https://www.kaggle.com/code/mikhailnaumov/regression-with-an-insurance-cat-lgb-xgb-hgb-ydf?scriptVersionId=214862769) |&nbsp;v.4&nbsp;| [Regr. with an Insu.  CAT+LGB+XGB+HGB+YDF](https://www.kaggle.com/code/mikhailnaumov/regression-with-an-insurance-cat-lgb-xgb-hgb-ydf/notebook) | **m**aster | [Mikhail Naumov](https://www.kaggle.com/mikhailnaumov) | World |\n|||||||\n| | | | **main weights** | **sort** | **correct weights** |\n| 1.01638 | [1.01449](https://www.kaggle.com/code/nina2025/ps-s4e12-h-blend-fe?scriptVersionId=269324587) | v.[1](#blend) |[ +0.80 +0.20 +0.10 -0.05 -0.05 ]|[asc/desc](#blend)|[+0,08,+0.05, -0,01,-0.03,-0.09]|\n| 1.01669 | [1.01480](https://www.kaggle.com/code/nina2025/ps-s4e12-h-blend-fe?scriptVersionId=269327212) | v.[2](#blend) |[ +0.70 +0.20 +0.10 &nbsp;0.00 &nbsp;0.00 ]|[asc/desc](#blend)|[+0,08,+0.05, -0,01,-0.03,-0.09]|\n| 0.01722 | [1.01534](https://www.kaggle.com/code/nina2025/ps-s4e12-h-blend-fe?scriptVersionId=269329260) | v.[3](#blend) |[ +0.60 +0.20 +0.10 +0.05 +0.05 ]|[asc/desc](#blend)|[+0,08,+0.05, -0,01,-0.03,-0.09]|\n| 1.01670 | [1.01480](https://www.kaggle.com/code/nina2025/ps-s4e12-h-blend-fe?scriptVersionId=269331847) | v.[4](#blend) |[ +0.95 +0.25 +0.11 -0.15 -0.16 ]|[asc/desc](#blend)|[+0,08,+0.05, -0,01,-0.03,-0.09]|\n| 1.01638 | [1.01449](https://www.kaggle.com/code/nina2025/ps-s4e12-h-blend-fe?scriptVersionId=269324587) | v.[5,1](#blend) |[ +0.80 +0.20 +0.10 -0.05 -0.05 ]|[asc/desc](#blend)|[+0,08,+0.05, -0,01,-0.03,-0.09]|","metadata":{}},{"cell_type":"code","source":"import os,ast\nimport numpy as np\nimport pandas as pd\n\nfrom bokeh.plotting import figure, gridplot \nfrom bokeh.io import output_file, show, output_notebook\noutput_notebook()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T05:39:51.329129Z","iopub.execute_input":"2025-10-20T05:39:51.329608Z","iopub.status.idle":"2025-10-20T05:39:54.106955Z","shell.execute_reply.started":"2025-10-20T05:39:51.329454Z","shell.execute_reply":"2025-10-20T05:39:54.106041Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### h-blend, Bokeh","metadata":{}},{"cell_type":"code","source":"def color_scheme(dk,color):\n    colors    = ['red','green','blue']\n    clr_silver= ['gold','gray','darkgray','silver','gainsboro']\n    clr_alls  = ['crimson',\"forestgreen\",'mediumblue','gold',\"darkmagenta\",\"silver\"]\n    clr_alls2 = [\"silver\",'gold']\n    clr_alls3 = ['darkmagenta',\"forestgreen\",'mediumblue']\n    clr_alls3b= ['crimson',\"darkgreen\",'mediumblue']\n    clr_alls3c= ['tomato',\"limegreen\",'royalblue']\n    clr_alls4 = ['red',\"forestgreen\",'mediumblue','crimson']\n    clr_alls4m= ['red',\"forestgreen\",'mediumblue',\"darkmagenta\"]\n    clr_alls5 = ['crimson',\"green\",'mediumblue','maroon',\"darkmagenta\"]\n    clr_Red   = [\"crimson\",\"orangered\",\"red\",'tomato','gold',\"firebrick\",]\n    clr_Red3  = [\"crimson\",\"red\",\"tomato\"]\n    clr_Red4  = [\"firebrick\",\"orangered\",\"crimson\",'tomato']\n    clr_Green = [\"green\",\"limegreen\",\"darkgreen\",\"forestgreen\",'lime']\n    clr_Green2= ['olivedrab',\"darkgreen\",\"forestgreen\"]\n    clr_Green3= [\"darkgreen\",\"forestgreen\",\"limegreen\",]\n    clr_Blue  = [\"mediumblue\",\"steelblue\",\"blue\",\"royalblue\",'midnightblue','cyan']\n    clr_Blue3 = ['midnightblue','mediumblue',\"royalblue\"]\n    clr_Blue4 = ['midnightblue',\"royalblue\",\"mediumblue\",\"steelblue\"]\n    clr_Brown = [\"maroon\",\"sienna\",\"sandybrown\",\"chocolate\",'brown',]\n    clr_Brown3= [\"maroon\",\"sienna\",\"sandybrown\"]\n    clr_Brown4= [\"maroon\",\"sienna\",\"chocolate\",\"sandybrown\"]\n    clr_magent= ['darkmagenta','magenta']\n    clr_Two   = ['olivedrab','gold']\n    clr_Two2  = ['crimson','darkgreen']\n    clr_tes3  = ['limegreen',\"magenta\",'red']\n    clr_tes3b = ['darkmagenta',\"magenta\",'red']\n    clr_tes6  = ['limegreen'] + clr_Brown\n    clr_tes7  = ['limegreen'] + clr_Brown4 + [\"magenta\"]+[\"darkmagenta\"]\n    clr_tes8  = clr_Red4 + clr_Blue4\n    clr_tes9  = clr_Red3 + clr_Green3 + clr_Blue3\n    clr_tes10 = clr_Brown + clr_Green\n    clr_tes11 = clr_Brown + ['red','darkmagenta'] + clr_Green\n    l = len(dk['subm'])\n    if color == 'Two2'  : colors = clr_Two2   [0:l]\n    if color == 'Two'   : colors = clr_Two    [0:l]\n    if color == 'alls'  : colors = clr_alls   [0:l]\n    if color == 'alls2' : colors = clr_alls2  [0:l]\n    if color == 'alls3' : colors = clr_alls3  [0:l]\n    if color == 'alls3b': colors = clr_alls3b [0:l]\n    if color == 'alls3c': colors = clr_alls3c [0:l]\n    if color == 'alls4' : colors = clr_alls4  [0:l]\n    if color == 'alls4m': colors = clr_alls4m [0:l]\n    if color == 'alls5' : colors = clr_alls5  [0:l]\n    if color == 'red'   : colors = clr_Red    [0:l]\n    if color == 'red3'  : colors = clr_Red3   [0:l]\n    if color == 'green' : colors = clr_Green  [0:l]\n    if color == 'green2': colors = clr_Green2 [0:l]\n    if color == 'green3': colors = clr_Green3 [0:l]\n    if color == 'blue'  : colors = clr_Blue   [0:l]\n    if color == 'blue3' : colors = clr_Blue3  [0:l]\n    if color == 'brown' : colors = clr_Brown  [0:l]\n    if color == 'brown3': colors = clr_Brown3 [0:l]\n    if color == 'tes3'  : colors = clr_tes3   [0:l]\n    if color == 'tes3b' : colors = clr_tes3b  [0:l]\n    if color == 'tes6'  : colors = clr_tes6   [0:l]\n    if color == 'tes7'  : colors = clr_tes7   [0:l]\n    if color == 'tes8'  : colors = clr_tes8   [0:l]\n    if color == 'tes9'  : colors = clr_tes9   [0:l]\n    if color == 'tes10' : colors = clr_tes10  [0:l]\n    if color == 'tes11' : colors = clr_tes11  [0:l]\n    if color == 'magent': colors = clr_magent [0:l]\n    if color == 'silver': colors = clr_silver [0:l]\n    return colors\n\n\ndef bokeh_show(\n        params,\n        df_cross,\n        colors, \n        show_figures1, \n        show_figures2, wps_fig2,\n        color_cross):\n    \n    def dossier(js,subms,cols):\n        def quant(i,js,subms,cols):\n            return {\"c\" : i, \"q\" : sum([1 for subm in cols[i] if subm == subms[js]])}\n        return {\n            'name' : subms[js],\n            'q_in' : [quant(i,js,subms,cols) for i in range(len(subms))]\n        }\n    alls = pd.read_csv(f'tida_desc.csv')\n    matrix = [ast.literal_eval(str(row.alls)) for row in alls.itertuples()]\n    subms = sorted(matrix[0])\n    cols = [[data[i] for data in matrix] for i in range(len(subms))]\n    df_subms = pd.DataFrame({f'col_{i}': [x[i] for x in matrix] for i in range(len(subms))})\n    dossiers = [dossier(js,subms,cols) for js in range(len(subms))]\n    subm_names = [one_dossier['name'] for one_dossier in dossiers]\n    figures1,qss,i = [],[],0\n    height = 101 if len(colors)==2\\\n        else 134 if len(colors)==3 else (154 if len(colors)==4 else 174)\n    for one_dossier in dossiers: \n        i_col = 'alls. ' + str(one_dossier['q_in'][i]['c'])\n        qs = [one['q'] for one in one_dossier['q_in']]\n        x_names = [name.replace(\"Group\",\"\").replace(\"subm_\",\"\") for name in subm_names]\n        width = 157  if len(colors) == 5\\\n            else (121 if len(colors) == 8\\\n            else (131 if len(colors) == 9\\\n            else (141 if len(colors) == 10\\\n            else (171 if len(colors) == 11 else 111))))\n        f = figure(x_range=x_names,width=width, height=height, title=i_col)\n        f.vbar(x=x_names, width=0.585, top=qs, color=colors)\n        figures1.append(f)\n        qss.append(qs)\n        i+=1\n    grid = gridplot([figures1])\n    output_file('tida_alls.html')\n    if show_figures1 == True: show(grid)\n    sub_wts = params['subwts']\n    main_wts = [subm['weight'] for subm in params['subm']]\n    mms,acc_mass = [],[]\n    for j in range(len(dossiers)):\n        one_dossier = dossiers[j]\n        qs = [one['q'] for one in one_dossier['q_in']]\n        mm = [qs[h] * (main_wts[j] + sub_wts[h]) for h in range(len(sub_wts))]\n        mass = sum(mm)\n        mms.append(mm)\n        acc_mass.append(round(mass))                        #subm_names[::-1]\n    y_names = [name + \" - \" + str(mass) for name,mass in zip(subm_names,acc_mass)]\n    f1 = figure(y_range=y_names, width=313, height=height, title='relations of general masses')\n    f1.hbar(y=y_names, height=0.585, right=acc_mass, left=0, color=colors)\n    output_file('tida_alls2.html')\n    alls = [f'alls.{i}' for i in range(len(dossiers))]\n    subm = [f'sub{i}'   for i in range(len(dossiers))] \n    mmsT  = np.asarray(mms).T\n    data = {'cols' : alls}\n    for i in range(len(dossiers)): data[f'sub{i}'] = mmsT[i,:]\n    f2 = figure(y_range=alls, height=height, width=274, title=\" ( relations of columns masses )\")\n    f2.hbar_stack(subm, y='cols', height=0.585, color=colors, source=data)\n    qssT  = np.asarray(qss).T\n    data = {'cols' : alls}\n    for i in range(len(dossiers)): data[f'sub{i}'] = qssT[i,:]\n    f3 = figure(y_range=alls, height=height, width=215, title=\"ratios in columns\")\n    f3.hbar_stack(subm, y='cols', height=0.585, color=colors, source=data)\n    grid = gridplot([[f3,f2,f1]])\n    show(grid)\n    if show_figures2 == True:\n        def read(params,i):\n            FiN = params[\"path\"] + params[\"subm\"][i][\"name\"] + \".csv\"\n            target_name_back = {'target':params[\"target\"],'pred':params[\"target\"]}\n            return pd.read_csv(FiN).rename(columns=target_name_back)\n        dfs = [read(params,i) for i in range(len(params[\"subm\"]))] + [df_cross]\n        f   = figure(width=800, height=274)\n        f.title.text = 'Click on legend entries to mute the corresponding lines'\n        b,e        = 21000,21244\n        line_x     = [dfs[i][b:e]['id']                     for i in range(len(dfs))]\n        line_y     = [dfs[i][b:e]['Premium Amount'] for i in range(len(dfs))]\n        color      = colors + [color_cross]\n        alpha      = [0.8 for i in range(len(dfs)-1)] + [0.95]\n        lws        = [1.0 for i in range(len(dfs)-1)] + [1.00]\n        legend = subm_names + ['cross']\n        for i in range(len(legend)):\n            f.line(line_x[i], line_y[i], line_width=lws[i], color=color[i], alpha=alpha[i],\n                   muted_color='white',legend_label=legend[i])\n        f.legend.location = \"top_left\"\n        f.legend.click_policy=\"mute\"\n        show(f)\n\n\ndef h_blend(params,color,cross='steelblue',\n            figures1=False,figures2=False,wf2=555,\n            details=False):\n\n    import copy\n\n    color_cross = cross\n\n    dk = copy.deepcopy(params)\n\n    show_details,show_figures1,show_figures2 = details,figures1,figures2\n\n    file_short_names = [subm['name'] for subm in params['subm']]\n    type_sort    = params['type_sort'][0]\n    dk['asc']    = params['type_sort'][1]\n    dk['desc']   = params['type_sort'][2]\n    dk['id']     = params['id_target'][0]\n    dk['target'] = params['id_target'][1]\n# ------------------------------------------------------------------------\n    def read(dk,i):\n        tnm = dk[\"subm\"][i][\"name\"]\n        FiN = dk[\"path\"] + tnm + \".csv\"\n        return pd.read_csv(FiN).rename(columns={\n            'target':tnm, 'pred':tnm, dk[\"target\"]:tnm})\n        \n    def merge(dfs_subm):\n        df_subms = pd.merge(dfs_subm[0],  dfs_subm[1], on=[dk['id']])\n        for i in range(2, len(dk[\"subm\"])): \n            df_subms = pd.merge(df_subms, dfs_subm[i], on=[dk['id']])\n        return df_subms\n        \n    def da(dk,sorting_direction,show_details):\n        \n        df_subms = merge([read(dk,i) for i in range(len(dk[\"subm\"]))])\n        cols = [col for col in df_subms.columns if col != dk['id']]\n        short_name_cols = [c for c in cols]\n        \n        def alls1(x, sd=sorting_direction,cs=cols):\n            reverse = True if sd=='desc' else False\n            tes = {c: x[c] for c in cs}.items()\n            subms_sorted = [t[0] for t in sorted(tes,key=lambda k:k[1],reverse=reverse)]\n            return subms_sorted\n\n        import random\n\n        def alls2(x, sd=sorting_direction,cs=cols):\n            reverse = True if sd=='desc' else False\n            tes = {c: x[c] for c in cs}.items()\n            subms_random = [t[0] for t in tes]\n            random.shuffle(subms_random)\n            return subms_random\n\n        alls = alls1 if type_sort == 'asc/desc' else alls2\n            \n        def summa(x,cs,wts,ic_alls): \n            return sum([x[cs[j]] * (wts[0][j] + wts[1][ic_alls[j]]) for j in range(len(cs))])\n            \n        wts = [[[e['weight'] for e in dk[\"subm\"]], [w for w in dk[\"subwts\" ]]]]\n          \n        def correct(x, cs=cols, wts=wts):\n            i = [x['alls'].index(c) for c in short_name_cols]\n            return summa(x,cs,wts[0],i)\n\n        if len(wts) == 1:\n            correct_sub_weights = [wt for wt in dk[\"subwts\"]]\n            weights = [subm['weight'] for subm in dk[\"subm\"]]\n            def correct(x, cs=cols, w=weights, cw=correct_sub_weights):\n                ic = [x['alls'].index(c) for c in short_name_cols]\n                cS = [x[cols[j]] * (w[j] + cw[ic[j]]) for j in range(len(cols))]\n                return sum(cS)\n                   \n        def amxm(x, cs=cols):\n            list_values = x[cs].to_list()\n            mxm = abs(max(list_values)-min(list_values))\n            return mxm\n\n        if len(wts) > 1:\n            df_subms['mx-m']   = df_subms.apply(lambda x: amxm   (x), axis=1)\n        df_subms['alls']       = df_subms.apply(lambda x: alls   (x), axis=1)\n        df_subms[dk[\"target\"]] = df_subms.apply(lambda x: correct(x), axis=1)\n        schema_rename = { old_nc:new_shnc for old_nc, new_shnc in zip(cols, short_name_cols) }\n        df_subms = df_subms.rename(columns=schema_rename)\n        df_subms = df_subms.rename(columns={dk[\"target\"]:\"ensemble\"})\n        df_subms.insert(loc=1, column=' _ ', value=['   '] * len(df_subms))\n        df_subms[' _ '] = df_subms[' _ '].astype(str)\n        pd.set_option('display.max_rows',100)\n        pd.set_option('display.float_format', '{:.4f}'.format)\n        vcols = [dk['id']]+[' _ '] + short_name_cols + [' _ ']+['alls']+[' _ ']+['ensemble']\n        if len(wts) > 1: vcols.append([' _ '] + ['mx-m'])\n        df_subms = df_subms[vcols]\n        if show_details and sorting_direction=='asc': display(df_subms.head(5))\n        pd.set_option('display.float_format', '{:.5f}'.format)\n        df_subms = df_subms.rename(columns={\"ensemble\":dk[\"target\"]})\n        df_subms.to_csv(f'tida_{sorting_direction}.csv', index=False)\n        return df_subms[[dk['id'],dk['target']]]\n   \n    def ensemble_da(dk,        show_details): \n        dfD    = da(dk,'desc', show_details)\n        dfA    = da(dk,'asc',  show_details)\n        dfA[dk['target']] = dk['desc']*dfD[dk['target']] + dfA[dk['target']]*dk['asc']\n        return dfA\n\n    da = ensemble_da(dk,show_details)\n    colors = color_scheme(dk, color)\n    bokeh_show(dk, da, colors, show_figures1, show_figures2, wf2, color_cross)\n    return  da\n\n\ndef matrix_vs(path,fs_names):\n    def load(path,fs_names):\n        dfs = [pd.read_csv(path + name_subm +'.csv') for name_subm in fs_names]\n        for i in range(len(dfs)):\n            dfs[i] = dfs[i].rename(columns={\n                \"target\": f'{fs_names[i]}',\"Premium Amount\": f'{fs_names[i]}'})\n        dfsm = pd.merge(dfs[0], dfs[1], on=\"id\")\n        for i in range(2,len(dfs)):\n            dfsm = pd.merge(dfsm,dfs[i],on='id')\n        return dfsm\n    def make_list_vs(fs_names):\n        list = []\n        for i in range(0,len(fs_names)-1):\n            for j in range(i+1,len(fs_names)):\n                list.append(fs_names[i] + \"_vs_\" + fs_names[j])\n        return list\n    def get_mvs(dfs, list_vs):\n        def get_abs_distance(x,t1,t2):\n            return abs(x[t1]-x[t2])\n        for vs in list_vs:\n            t = vs.split('_vs_')\n            dfs[vs] = dfs.apply(lambda x: get_abs_distance(x,t[0],t[1]), axis=1)\n        return dfs   \n    def distance_vs(name, st_names, list_vs, dfs):\n        distances = []\n        for st in st_names:\n            vs_between = name + \"_vs_\" + st\n            if vs_between not in list_vs:\n                distances.append(0)\n            else: distances.append(round(dfs[vs_between].sum()))\n        return distances\n    dfs = load(path,fs_names)\n    list_vs = make_list_vs(fs_names)\n    mvs = get_mvs(dfs, list_vs)\n    m1 = pd.DataFrame({'subm':fs_names})\n    m2 = pd.DataFrame({ name :distance_vs(name, fs_names, list_vs, mvs) for name in fs_names})\n    matrix = pd.concat([m1,m2],axis=1)\n    return matrix\n\n\ndef procedure_Cage(FiN_import,n_iter=4,ks1=[1.0054,0.0021],ks2=[1.00037,0.00037]):\n    import seaborn as sns\n    import matplotlib.pyplot as plt\n    import warnings; warnings.filterwarnings('ignore')\n    \n    sub_sample = pd.read_csv('../input/playground-series-s4e12/sample_submission.csv') \n    sub_import = pd.read_csv(FiN_import) \n    per = sub_import['Premium Amount'].values\n    # ..................................................................................................\n    sns.set()\n    plt.figure(figsize=(5, 2))\n    plt.hist(per, bins=80)\n    plt.gca().set_facecolor('mintcream')\n    plt.suptitle('Before | Premium Amount', y=0.96, fontsize=12, c='navy')\n    # ..................................................................................................\n    print('- - - - - - - ',FiN_import)\n    min_per  = np.min(per);  print('Min:',  round(min_per, 7))\n    max_per  = np.max(per);  print('Max:',  round(max_per, 7))\n    mean_per = np.mean(per); print('Mean:', round(mean_per,7))\n    print('-------')\n    R = -0.0\n    guide = mean_per - R\n    # ....................................\n    per1 = [f for f in per if f < guide]\n    per2 = [f for f in per if f > guide]\n    print(len(per1),'-',len(per2))\n    print('-------')\n    N = n_iter\n    for _ in range(N):\n        for i in range(len(per)):\n            per_guide = (per[i] + guide) / 2            \n            if per[i] <= guide:\n                per[i] = (per[i] *ks1[0]) - (per_guide *ks1[1])\n            else:\n                per[i] = (per[i] *ks2[0]) - (per_guide *ks2[1])\n    # .......................................................................\n    sns.set()\n    plt.figure(figsize=(5, 2))\n    plt.hist(per, bins=80)\n    plt.gca().set_facecolor('snow')\n    plt.suptitle('After | Premium Amount', y=0.96, fontsize=11, c='navy')\n    # .......................................................................\n    min_per  = np.min(per);  print('Min:',  round(min_per, 7))\n    max_per  = np.max(per);  print('Max:',  round(max_per, 7))\n    mean_per = np.mean(per); print('Mean:', round(mean_per,7)); \n    # .......................................................................\n    print('- - - - - - - ', 'Cage '+FiN_import, '\\n')\n    # .......................................................................\n    \n    sub_sample['Premium Amount'] = per\n    return sub_sample \n\n\ndef Cage_by_MehranKazeminia(file_name_to_Cage):\n    import seaborn as sns\n    import matplotlib.pyplot as plt\n    import warnings; warnings.filterwarnings('ignore')\n    \n    sub_import = pd.read_csv(file_name_to_Cage)\n    \n    per = sub_import['Premium Amount'].values\n    # ...................................................................................\n    sns.set()\n    plt.figure(figsize=(5, 2))\n    plt.hist(per, bins=80)\n    plt.gca().set_facecolor('mintcream')\n    plt.suptitle('Before | Premium Amount', y=0.96, fontsize=13, c='navy')\n    # ...................................................................................\n    min_per  = np.min (per) ;print('Min:',  round(min_per, 3))\n    max_per  = np.max (per) ;print('Max:',  round(max_per, 3))\n    mean_per = np.mean(per) ;print('Mean:', round(mean_per,3))\n    # ...................................................................................\n    R = 0.0              # Adjusting the R value can increase the accuracy of the guide.\n    guide = mean_per - R\n    # ....................................\n    per1 = [f for f in per if f < guide]\n    per2 = [f for f in per if f > guide]\n    \n    print(len(per1), len(per2))\n    # .......................................................................\n    for i in range(len(per)):\n        \n        per_guide = (per[i] + guide) / 2\n            \n        if per[i] <= guide: per[i] = (per[i]* 1.30) - (per_guide* 0.30)\n        if per[i] >  guide: per[i] = (per[i]* 1.00) - (per_guide* 0.00)\n    # .......................................................................\n    sns.set()\n    plt.figure(figsize=(5, 2))\n    plt.hist(per, bins=80)\n    plt.gca().set_facecolor('snow')\n    plt.suptitle('After | Premium Amount', y=0.96, fontsize=12, c='navy')\n    \n    # .......................................................................\n    min_per  = np.min (per) ;print('Min:',  round(min_per, 3))\n    max_per  = np.max (per) ;print('Max:',  round(max_per, 3))\n    mean_per = np.mean(per) ;print('Mean:', round(mean_per,3))\n    \n    # -------------------------------------------------------------------------\n    # After the after\n    # -------------------------------------------------------------------------\n    \n    for i in range(len(per)): \n        if per[i] < (min_per+7): per[i] = per[i] ** 0.994\n        if per[i] > (max_per-9): per[i] = per[i] ** 1.006\n    # .......................................................................\n    sns.set()\n    plt.figure(figsize=(5, 2))\n    plt.hist(per, bins=80)\n    plt.gca().set_facecolor('ghostwhite')\n    plt.suptitle('After the after | Premium Amount', y=0.96, fontsize=11, c='navy')\n    # .......................................................................\n    min_per  = np.min (per) ;print('Min:',  round(min_per, 3))\n    max_per  = np.max (per) ;print('Max:',  round(max_per, 3))\n    mean_per = np.mean(per) ;print('Mean:', round(mean_per,3))\n    \n    # -------------------------------------------------------------------------\n    \n    df = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\n    \n    df['accident_risk'] = per\n    \n    return df","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2025-10-20T05:44:34.507432Z","iopub.execute_input":"2025-10-20T05:44:34.507759Z","iopub.status.idle":"2025-10-20T05:44:34.581759Z","shell.execute_reply.started":"2025-10-20T05:44:34.507735Z","shell.execute_reply":"2025-10-20T05:44:34.580983Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## blend","metadata":{}},{"cell_type":"code","source":"path ='/kaggle/input/ps-s4e12/submission_'\n\nparams = {\n      'path'     : path,            \n      'id_target': ['id',\"Premium Amount\"],          \n      'type_sort': ['asc/desc',0.30, 0.70],\n      'subwts'   : [ +0.08,+0.05, -0.01,-0.03,-0.09 ],       \n      'subm'     : [ \n         { 'name': f'1.01471', 'weight': +0.80 },\n         { 'name': f'1.01719', 'weight': +0.20 },\n         { 'name': f'1.01909', 'weight': +0.10 },\n         { 'name': f'1.02866', 'weight': -0.05 },\n         { 'name': f'1.02871', 'weight': -0.05 },]\n}\n\ndf_cross = h_blend ( params, color='silver', figures1=True, figures2=True, details=True)\n\nfiles = [subm['name'] for subm in params['subm']]\ndistances = matrix_vs ( path, files )         \ndistances","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T05:44:49.554607Z","iopub.execute_input":"2025-10-20T05:44:49.55497Z","iopub.status.idle":"2025-10-20T05:47:39.090519Z","shell.execute_reply.started":"2025-10-20T05:44:49.554903Z","shell.execute_reply":"2025-10-20T05:47:39.089567Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submit","metadata":{}},{"cell_type":"code","source":"df_cross.to_csv('submission.csv',index=False)\ndf_cross","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}