{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:43:30.231814Z","iopub.execute_input":"2024-12-31T22:43:30.232308Z","iopub.status.idle":"2024-12-31T22:43:30.672024Z","shell.execute_reply.started":"2024-12-31T22:43:30.232262Z","shell.execute_reply":"2024-12-31T22:43:30.670724Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style= \"font-family: Cambria;\n            font-weight:bold;\n            color: white;\n            padding:15px 8px 8px 10px;\n            margin:0px 0px 0px 0px;\n            background:black;\n            border-radius: 0px 14px 14px 14px;\n            border-bottom: 4px solid #aa4;\n            border-right: 4px solid #dd5\"> Imports </h1> ","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\nimport plotly.io as pio\n\n# Kaggle環境でPlotlyをオフラインで使用する設定\npio.renderers.default = 'notebook'\n# pio.renderers.default = 'notebook_connected'\n\n\n\nimport ipywidgets as widgets\nfrom ipywidgets import interact, Layout\nfrom IPython.display import HTML, display\nfrom IPython.display import IFrame\n\nimport matplotlib.pyplot as plt\n\nfrom scipy import optimize\n\nfrom xgboost import XGBRegressor,XGBClassifier, DMatrix\nfrom lightgbm import LGBMRegressor, LGBMClassifier, log_evaluation, early_stopping\nimport lightgbm as lgb\n\nfrom catboost import CatBoostClassifier, CatBoostRegressor, Pool\n\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder,LabelEncoder, StandardScaler, OrdinalEncoder, MinMaxScaler\nfrom sklearn.model_selection import train_test_split, StratifiedKFold, cross_val_score, KFold\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import mean_squared_log_error, mean_squared_error, r2_score, accuracy_score\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.cluster import KMeans\nfrom sklearn.decomposition import PCA\nfrom sklearn.manifold import TSNE\n# from sklearn.metrics import matthews_corrcoef,roc_auc_score, confusion_matrix, ConfusionMatrixDisplay, roc_curve, auc\nfrom sklearn.ensemble import HistGradientBoostingRegressor\n\nimport random\n\nfrom tqdm import tqdm\n\nfrom gc import collect\nfrom colorama import Fore, Style, init;\n\n# import GPy\n\nimport optuna\nimport shap\n\nfrom optuna.samplers import TPESampler\n\n\nfrom scipy import optimize\n\n# ignore wornings\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:43:30.674226Z","iopub.execute_input":"2024-12-31T22:43:30.674845Z","iopub.status.idle":"2024-12-31T22:43:38.965246Z","shell.execute_reply.started":"2024-12-31T22:43:30.674769Z","shell.execute_reply":"2024-12-31T22:43:38.964174Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style= \"font-family: Cambria;\n            font-weight:bold;\n            color: white;\n            padding:15px 8px 8px 10px;\n            margin:0px 0px 0px 0px;\n            background:black;\n            border-radius: 0px 14px 14px 14px;\n            border-bottom: 4px solid #aa4;\n            border-right: 4px solid #dd5\"> Load Data </h1> ","metadata":{}},{"cell_type":"code","source":"# Load the training and test data from CSV files\n\ndf_sample_submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\ndf_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv', index_col=0)\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv', index_col=0)\n# df_original = pd.read_csv('/kaggle/input/depression-surveydataset-for-analysis/final_depression_dataset_1.csv')\n\n\n\n\nprint(f'N_train = {len(df_train)}, N_test = {len(df_test)}')\n\n\ndf_train['train_test'] = 'train'\n# df_original['train_test'] = 'original'\ndf_test['train_test'] = 'test'\n\ndf_train_reduced = df_train.sample(len(df_train)//10)\ndf_test_reduced = df_test.sample(len(df_test)//80)\nprint(f'N_train_reduced = {len(df_train_reduced)}, N_test_reduced = {len(df_test_reduced)}')\n\ntarget_col = 'Premium Amount'\n\ndf_all = pd.concat([\n    df_train,\n    df_test,\n    # df_train_reduced,\n    # df_test_reduced\n])\n\n\n\ndf_all['Policy Start Date'] = pd.to_datetime(df_all['Policy Start Date'])\ndf_all['Policy Start Date'] = pd.to_numeric(df_all['Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:43:38.966722Z","iopub.execute_input":"2024-12-31T22:43:38.967356Z","iopub.status.idle":"2024-12-31T22:43:50.405965Z","shell.execute_reply.started":"2024-12-31T22:43:38.96732Z","shell.execute_reply":"2024-12-31T22:43:50.404828Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <h1 style= \"font-family: Cambria;font-weight:bold;color: white;padding:10px 8px 8px 10px;margin:-50px 0px 0px 0px;background:black;border-radius: 0px 14px 14px 14px;border-bottom: 4px solid #aa4;border-right: 4px solid #dd5\"> EDA </h1> \n\n## <h2 style= \"font-family: Cambria;font-weight:bold;color: white;padding:10px 4px 4px 10px;margin:-40px 0px 0px 0px;background:black;border-radius: 0px 14px 14px 14px;border-bottom: 4px solid #a4a;border-right: 4px solid #d5d\"> define functions </h2> ","metadata":{}},{"cell_type":"code","source":"# define a format for displaying df data.\ndef custom_format(x):\n    if isinstance(x, float):\n        return ('{0:.３f}'.format(x)).rstrip('0').rstrip('.')\n    return x\n\ndef display_short(df, n):    \n    pd.set_option('display.max_colwidth', 10)\n    pd.set_option('display.max_acolumns', None)\n\n    df_disp = []\n    for i, train_test in enumerate(df['train_test'].unique()):\n        tmp = df[df['train_test']==train_test].sample(n)\n        tmp.index = pd.MultiIndex.from_product([[train_test], tmp.index])\n        df_disp.append(tmp)\n        \n    df_disp = pd.concat(df_disp)\n    display(df_disp)\n    pd.set_option('display.max_colwidth', None)\n    pd.set_option('display.max_columns', 20)\n\ndef display_dfinfo(df):\n    df_disp = []\n    for tt in df['train_test'].unique():\n        tmp = df.loc[df['train_test']==tt].describe(\n            percentiles= [0.05, 0.25, 0.50, 0.75, 0.95]\n        ).drop(['Target'], axis= 1, errors = 'ignore')\n\n        tmp.loc['skew'] = df.loc[df['train_test']==tt].select_dtypes(include=[int,float]).skew()\n        tmp.loc['kurtosis'] = df.loc[\n            df['train_test']==tt\n        ].select_dtypes(include=[int,float]).kurtosis()\n\n        tmp.loc['dtype'] = df.loc[df['train_test']==tt].dtypes\n        tmp.loc['NaN count'] = df.loc[df['train_test']==tt].isna().sum(axis=0)\n\n        for col in df.select_dtypes(exclude=[int,float]).columns:\n            tmp.loc[:,col] = 0\n            tmp.loc['count',col] = df.loc[df['train_test']==tt,col].count()\n            tmp.loc['dtype',col] = df.loc[df['train_test']==tt,col].dtype\n            tmp.loc['NaN count',col] = df.loc[df['train_test']==tt,col].isna().sum(axis=0)\n\n        tmp.loc['N unique'] = df.loc[df['train_test']==tt].nunique()\n        tmp.columns = pd.MultiIndex.from_product([tmp.columns,[tt]])\n        df_disp.append(tmp)\n\n\n    df_disp = pd.concat(df_disp, axis=1)\n    df_disp = df_disp[df_disp.columns.get_level_values(0).unique()]\n\n    df_disp = df_disp.T\n    df_disp = df_disp.loc[\n        df.columns,[\n            'count','NaN count','N unique','dtype',\n            'mean', 'min', '5%', '25%', '50%', '75%', '95%', 'max',\n            'std', 'skew', 'kurtosis'\n        ]\n    ]\n    formatter ={}\n    display(\n        df_disp.style.format(formatter = custom_format).\\\n        background_gradient(\n        subset = ['mean', 'min', '5%', '25%', '50%', '75%', '95%', 'max'],\n        cmap = 'Reds',axis=1)\n    )\n\n# histogram        \ndef display_plotdata(num_col1,cat_col):\n    if num_col1 == cat_col:\n        df_plot = df[[num_col1]].copy()\n    else:\n        df_plot = df[[num_col1, cat_col]].copy()\n\n\n    df_plot[cat_col] = df_plot[cat_col].astype('object')\n    try:\n        df_description = df.groupby('train_test').describe(\n            percentiles= [0.05, 0.25, 0.50, 0.75, 0.95]\n        )[num_col1]\n        df_description['skew'] = df.groupby('train_test')[num_col1].skew()\n        df_description['count'] = df_description['count'].astype(int)\n        df_description['nunique'] = df.groupby('train_test')[num_col1].nunique()\n        group_format = dict(\n            [\n                [s, ' {:.0f} '] if s == 'count' else [s, ' {:.4f} ']\n                for s in df_description.columns\n            ]\n        )\n\n        display(\n            df_description.loc[\n                :,\n                [\n                    'count','nunique', 'mean',\n                    'min', '5%', '25%', '50%', '75%', '95%', 'max',\n                    'std','skew'\n                ]\n            ].style.format(\n                formatter = custom_format\n            ).background_gradient(\n                subset = [\n                    'mean', 'min', '5%', '25%', '50%', '75%', '95%', 'max'\n                ],\n                cmap = 'Reds',axis=1\n            )\n        )\n    except:\n        display(\n            df_all.groupby('train_test')[num_col1].\\\n            agg(['count', 'nunique']).loc[['train','test']].\\\n            style.format(formatter = custom_format).\\\n            background_gradient(cmap = 'Blues')\n        )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:43:50.40888Z","iopub.execute_input":"2024-12-31T22:43:50.409258Z","iopub.status.idle":"2024-12-31T22:43:50.429177Z","shell.execute_reply.started":"2024-12-31T22:43:50.409224Z","shell.execute_reply":"2024-12-31T22:43:50.427749Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def hist_df(df, num_col1,cat_col, n_data, displaytype = 'density', plottype = 'sns'):\n    if np.isnan(n_data):\n        df_plot = df.copy()\n    else:\n        df_plot = df.sample(n_data)\n    \n    if plottype == 'plotly':\n        if num_col1 != cat_col:\n            fig=px.histogram(\n                df_plot,\n                x = num_col1, marginal = 'violin', color = cat_col,\n                nbins=50, histnorm = displaytype,\n                barmode = 'relative', opacity = 0.5\n            )\n        else:\n            fig=px.histogram(\n                df_plot,\n                x = num_col1,\n                histnorm = displaytype,\n                barmode = 'relative',\n                opacity = 0.5\n            )\n        fig.update_layout(\n            width=900, height=350,\n            margin=dict(l=0, r=0, b=0, t=20),  # Adjust margins\n            xaxis=dict( title_font=dict(size=20) ),\n            yaxis=dict( title_font=dict(size=20) ),\n            legend=dict( font=dict(size=15) ),\n        )\n        fig.show()\n        # fig.write_html(\"plot.html\")\n        # IFrame(\"plot.html\", width=800, height=600)\n\n\n    elif plottype == 'sns':\n        if num_col1 != cat_col:\n            fig = sns.histplot(\n                df_plot,\n                x = num_col1, hue = cat_col,\n                edgecolor = None,\n                alpha = 0.5,\n            )\n        else:\n            fig=sns.histplot(\n                df_plot,\n                x = num_col1,\n                edgecolor = None,\n                alpha = 0.5\n            )\n            ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:43:50.4308Z","iopub.execute_input":"2024-12-31T22:43:50.431282Z","iopub.status.idle":"2024-12-31T22:43:50.461914Z","shell.execute_reply.started":"2024-12-31T22:43:50.431237Z","shell.execute_reply":"2024-12-31T22:43:50.460924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def scatter_df(df, num_col1, num_col2, cat_col, n_data = 10000, plottype = 'sns'):\n    if np.isnan(n_data):\n        df_plot = df.copy()\n    else:\n        df_plot = df.sample(n_data)\n    df_tmp = df_plot[[num_col1,num_col2, cat_col]]\n    df_tmp.columns = [c+' '*i for i,c in enumerate(df_tmp.columns)]\n\n    if plottype == 'plotly':\n        fig = px.scatter(\n            df_tmp,\n            x = num_col1+'', y = num_col2+' ', color = cat_col+'  ',\n            marginal_x = 'histogram', marginal_y = 'violin', opacity=0.2,\n            color_continuous_scale = px.colors.sequential.Rainbow,\n                trendline='ols'\n        )\n        fig.update_layout(\n            width=750, height=550,\n            margin=dict(l=0, r=0, b=0, t=0),  # Adjust margins\n            xaxis1=dict(domain=[0.1, 0.75]),  # Specifies the area of the x-axis subplot1\n            yaxis1=dict(domain=[0.1, 0.8]),  # Specifies the area of the y-axis subplot1\n            xaxis2=dict(domain=[0.76, 1.0]),  # Specifies the area of the x-axis subplot2\n            yaxis2=dict(domain=[0.1, 0.8]),  # Specifies the area of the y-axis subplot2\n            xaxis3=dict(domain=[0.1, 0.75]),  # Specifies the area of the x-axis subplot3\n            yaxis3=dict(domain=[0.81, 1.0]),  # Specifies the area of the y-axis subplot3\n            xaxis=dict( title_font=dict(size=20) ),\n            yaxis=dict( title_font=dict(size=20) ),\n            legend=dict( font=dict(size=15) ),\n        ) \n        fig.show()\n\n    elif plottype == 'sns':\n        if (df_tmp[cat_col+'  '].dtype==object) or (df_tmp[cat_col+'  '].nunique()<=10):\n\n            g = sns.JointGrid(\n                data = df_tmp,\n                x = num_col1+'', y = num_col2+' ', hue = cat_col+'  ',\n            )\n    \n            # center plot\n            g.plot_joint(sns.scatterplot, color=\"blue\")\n            \n            # mariginal plot\n            g.plot_marginals(sns.kdeplot, color=\"skyblue\", alpha=0.4, fill=True) \n    \n            \n            plt.show()\n        else:\n            fig = sns.scatterplot(\n                data = df_tmp,\n                x = num_col1+'', y = num_col2+' ', hue = cat_col+'  ',\n                palette='rainbow'\n            )\n            plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:43:50.463832Z","iopub.execute_input":"2024-12-31T22:43:50.464187Z","iopub.status.idle":"2024-12-31T22:43:50.480076Z","shell.execute_reply.started":"2024-12-31T22:43:50.464156Z","shell.execute_reply":"2024-12-31T22:43:50.479074Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def violin_df(df, num_col1, num_col2, cat_col, n_data = 10000, plottype = 'sns'):\n    if np.isnan(n_data):\n        df_plot = df.copy()\n    else:\n        df_plot = df.sample(n_data)\n    \n    col_in_num = cat_col in [num_col1, num_col2]\n    if col_in_num:\n        df_tmp = df_plot[[num_col1,num_col2]]\n    else:\n        df_tmp = df_plot[[num_col1,num_col2, cat_col]]\n        df_tmp[cat_col] = df_tmp[cat_col].astype(object)\n\n\n    if plottype == 'plotly':\n        if col_in_num or df_tmp[cat_col].nunique() > 10:\n            fig = px.violin(\n                df_tmp, x = num_col1, y = num_col2,\n            )\n        else:\n            df_tmp = df_plot[[num_col1,num_col2, cat_col]]\n            fig = px.violin(\n                df_tmp, x = num_col1, y = num_col2, color = cat_col,\n            )\n        fig.update_layout(\n            width=950, height=550,\n            margin=dict(l=0, r=0, b=0, t=0),  # Adjust margins\n            xaxis=dict( title_font=dict(size=20) ),\n            yaxis=dict( title_font=dict(size=20) ),\n            legend=dict( font=dict(size=12) ),\n        ) \n        fig.show()\n    elif plottype == 'sns':\n        if col_in_num or df_tmp[cat_col].nunique() > 10:\n            fig = sns.violinplot(\n                df_tmp, x = num_col1, y = num_col2,\n                linewidth=1\n            )\n        else:\n            fig = sns.violinplot(\n                df_tmp, x = num_col1, y = num_col2, hue = cat_col,\n                linewidth=1\n            )\n        plt.show()\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:43:50.481729Z","iopub.execute_input":"2024-12-31T22:43:50.482087Z","iopub.status.idle":"2024-12-31T22:43:50.502969Z","shell.execute_reply.started":"2024-12-31T22:43:50.482056Z","shell.execute_reply":"2024-12-31T22:43:50.50153Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cross_df(df,num_col2, num_col1, cat_col, n_data, plottype = 'sns'):\n    if df[cat_col].dtype == object:\n        df_plot = pd.crosstab(df_all[num_col1],df_all[num_col2])\n        label = cat_col\n    else:\n        df_plot = pd.pivot(\n            df_all.groupby([num_col1, num_col2])[cat_col].mean().reset_index(),\n            columns=num_col1, index=num_col2, values = cat_col\n        )\n        label = 'count'\n    if plottype == 'plotly':\n        fig = px.imshow(\n            df_plot,\n            color_continuous_scale = 'blues',\n            labels={\"color\": cat_col}\n        )   \n        fig.update_layout(\n            width=950, height=550,\n            margin=dict(l=0, r=0, b=0, t=0),  # Adjust margins\n            xaxis=dict( title_font=dict(size=20) ),\n            yaxis=dict( title_font=dict(size=20) ),\n            legend=dict( font=dict(size=12) ),\n        ) \n        fig.show()\n    elif plottype=='sns':\n        display(df_plot)\n        fig = sns.heatmap(\n            df_plot,\n            cmap = 'coolwarm',\n            annot=True,\n        )\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:43:50.504344Z","iopub.execute_input":"2024-12-31T22:43:50.504659Z","iopub.status.idle":"2024-12-31T22:43:50.523714Z","shell.execute_reply.started":"2024-12-31T22:43:50.504627Z","shell.execute_reply":"2024-12-31T22:43:50.522481Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_interact(df, num_col, cat_col, button_width = 100, n_data = np.nan, plottype='sns'):\n    num_col_button1 = widgets.Dropdown(\n        options = num_col,\n        button_style='info',\n        description = 'X'\n    )\n    if plottype == 'plotly':\n        fig = px.imshow(df_all[num_cols].corr(),zmax=1, zmin=-1, color_continuous_scale='rdbu_r')\n        fig.update_layout(\n            width=max(min(len(num_cols)*70,500),300),\n            height=max(min(len(num_cols)*70,500),300),\n            title='Correlation matrix')\n        fig.show()\n    elif plottype=='sns':\n        fig = sns.heatmap(df_all[num_cols].corr(),vmax=1, vmin=-1)\n\n        \n    \n    \n    num_col_button1.style.description_width = '10px'\n    num_col_button1.style.font_size = '1px'\n\n    num_col_button2 = widgets.Dropdown(\n        options = num_col,\n        button_style='primary',\n        description = 'Y',\n        value = num_col[0]\n    )\n    num_col_button2.style.button_width = f'{button_width}px'\n    num_col_button2.style.description_width = '10px'\n\n    cat_col_button = widgets.Dropdown(options = cat_col, button_style='warning',description = 'color')\n    cat_col_button.style.button_width = '80px'\n    cat_col_button.style.description_width = '50px'\n\n    # narrow down the number of samples to be displayed.\n    @interact(\n        num_col1 = num_col_button1, num_col2 = num_col_button2,\n        cat_col = cat_col_button\n    )\n    def plot_df(num_col1, num_col2, cat_col):\n        print(f'type({num_col1}):{df[num_col1].dtype}, type({num_col2}):{df[num_col2].dtype}')\n        \n        if num_col1 == num_col2:\n            hist_df(df, num_col1,cat_col,n_data, plottype = plottype)\n        elif (df[num_col1].dtype != object ) and (df[num_col2].dtype != object):\n            scatter_df(df,num_col1, num_col2, cat_col, n_data, plottype=plottype)\n        elif df[num_col1].dtype != object:\n            violin_df(df,num_col1, num_col2, cat_col, n_data, plottype=plottype)\n        elif df[num_col2].dtype != object:\n            violin_df(df,num_col1, num_col2, cat_col, n_data, plottype=plottype)\n        else:\n            cross_df(df,num_col2, num_col1, cat_col, n_data, plottype=plottype)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:43:50.525121Z","iopub.execute_input":"2024-12-31T22:43:50.525462Z","iopub.status.idle":"2024-12-31T22:43:50.540571Z","shell.execute_reply.started":"2024-12-31T22:43:50.525421Z","shell.execute_reply":"2024-12-31T22:43:50.539568Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <h2 style= \"font-family: Cambria;font-weight:bold;color: white;padding:10px 4px 4px 10px;margin:-40px 0px 0px 0px;background:black;border-radius: 0px 14px 14px 14px;border-bottom: 4px solid #a4a;border-right: 4px solid #d5d\"> Data Description </h2> ","metadata":{}},{"cell_type":"code","source":"display(df_all.head(4))\n\ndisplay_dfinfo(df_all)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:43:50.544007Z","iopub.execute_input":"2024-12-31T22:43:50.544479Z","iopub.status.idle":"2024-12-31T22:44:13.154689Z","shell.execute_reply.started":"2024-12-31T22:43:50.544444Z","shell.execute_reply":"2024-12-31T22:44:13.15358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols = df_all.select_dtypes(include=[float, int]).columns\ndisp_cols = df_all.columns\ncat_col = df_all.columns\n\nplot_interact(df_all, disp_cols, cat_col, 107, n_data=5000, plottype='sns')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:44:13.156Z","iopub.execute_input":"2024-12-31T22:44:13.156307Z","iopub.status.idle":"2024-12-31T22:44:14.820361Z","shell.execute_reply.started":"2024-12-31T22:44:13.156278Z","shell.execute_reply":"2024-12-31T22:44:14.819218Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <h1 style= \"font-family: Cambria;font-weight:bold;color: white;padding:10px 8px 8px 10px;margin:-50px 0px 0px 0px;background:black;border-radius: 0px 14px 14px 14px;border-bottom: 4px solid #aa4;border-right: 4px solid #dd5\"> Preprocessing </h1> ","metadata":{}},{"cell_type":"code","source":"def safe_transform(encoder, labels):\n    known_labels = set(encoder.classes_)\n    return [\n        encoder.transform([label])[0] if label in known_labels else -1 for label in labels\n    ]\n\ndef target_encoder(df, input_col, target_col):\n    tmp = df[[input_col, target_col]]\n    means = df.groupby(input_col)[target_col].mean()\n    for ind in means.index:\n        tmp.loc[tmp[f'{input_col}']==ind, f'{input_col}_te'] = means[ind]\n\n    return tmp[f'{input_col}_te'].values\n\ndef preprocessing(df, num_cols, cat_cols, target_col,train_test = 'train_test'):\n    df_pp = df[num_cols].copy()\n    for i, cat_col in enumerate(cat_cols):\n        print(cat_col, end = ' / ')\n        # target encoding\n        df_pp[f'{cat_col}_te'] = target_encoder(df, cat_col, target_col)\n        \n    print()\n    df_pp[target_col] = df[target_col]\n    df_pp[train_test] = df[train_test]\n    return df_pp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:44:14.82172Z","iopub.execute_input":"2024-12-31T22:44:14.822099Z","iopub.status.idle":"2024-12-31T22:44:14.83097Z","shell.execute_reply.started":"2024-12-31T22:44:14.822067Z","shell.execute_reply":"2024-12-31T22:44:14.829683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_all[target_col] = df_all[target_col].astype(float)\ncat_cols = list(df_all.select_dtypes(include='object').columns)\ncat_cols.remove('train_test')\n\nnum_cols = list(df_all.select_dtypes(include=[int, float]).columns)\n\ndf_all_pp = preprocessing(\n    df_all,\n    num_cols, cat_cols,\n    target_col\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:44:14.832459Z","iopub.execute_input":"2024-12-31T22:44:14.832983Z","iopub.status.idle":"2024-12-31T22:44:23.338701Z","shell.execute_reply.started":"2024-12-31T22:44:14.832934Z","shell.execute_reply":"2024-12-31T22:44:23.337452Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <h2 style= \"font-family: Cambria;font-weight:bold;color: white;padding:10px 4px 4px 10px;margin:-40px 0px 0px 0px;background:black;border-radius: 0px 14px 14px 14px;border-bottom: 4px solid #a4a;border-right: 4px solid #d5d\"> define data</h2> ","metadata":{}},{"cell_type":"code","source":"def adversarial_validation(df_adv):\n#     Return a list of train data indistinguishable from test data\n    xgb = XGBClassifier()\n    X_adv = df_adv.drop('train_test',axis = 1)\n    y_adv = df_adv['train_test'].map({'train':0,'original':0, 'test':1})\n    \n    xgb.fit(X_adv, y_adv)\n    predict_adv = pd.DataFrame(\n        xgb.predict_proba(X_adv.loc[y_adv==0])[:,0], columns=['train'],\n        index = X_adv.index[y_adv==0]\n    )\n    predict_adv.sort_values(by='train',inplace = True)\n    return predict_adv.index","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:44:23.340191Z","iopub.execute_input":"2024-12-31T22:44:23.341002Z","iopub.status.idle":"2024-12-31T22:44:23.347564Z","shell.execute_reply.started":"2024-12-31T22:44:23.340952Z","shell.execute_reply":"2024-12-31T22:44:23.346416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_all_pp[target_col]=df_all_pp[target_col].astype(float)\n\ninput_coaggregatels = list(df_all_pp.columns)\n# input_cols = list(df_all_pp.columns)\n# input_cols.remove(target_col)\ninput_cols = [\n    'Annual Income', 'Credit Score', 'Previous Claims', 'Health Score',\n   'Policy Start Date', 'Age', 'Customer Feedback_te', 'Vehicle Age',\n   'Occupation_te', 'Insurance Duration', 'Number of Dependents',\n   'Marital Status_te', 'Location_te', 'Exercise Frequency_te',\n   'Policy Type_te','train_test'\n]\n\ndf_target = df_all_pp.loc[:, input_cols + [target_col]].copy()\n# df_target = df_all_pp.loc[:, input_cols + [target_col]].copy()\n\ntrain_data = df_target.loc[\n    df_target['train_test'].isin(['train','original'])\n].drop('train_test', axis=1)\ntest_data =  df_target.loc[\n    df_target['train_test']=='test'\n].drop('train_test', axis=1)\n\nvalid_indices = adversarial_validation(df_target.drop(target_col, axis = 1))\nvalid_indices = valid_indices[:round(len(valid_indices)*0.3)]\n\nX_train = train_data.drop(target_col,axis=1).drop(valid_indices)\nX_val = train_data.drop(target_col,axis=1).loc[valid_indices]\nX_test = test_data.drop(target_col, axis = 1)\n\ny_train = train_data[target_col].drop(valid_indices)\ny_val = train_data[target_col].loc[valid_indices]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:45:19.527806Z","iopub.execute_input":"2024-12-31T22:45:19.528402Z","iopub.status.idle":"2024-12-31T22:45:46.144957Z","shell.execute_reply.started":"2024-12-31T22:45:19.52835Z","shell.execute_reply":"2024-12-31T22:45:46.143767Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <h1 style= \"font-family: Cambria;font-weight:bold;color: white;padding:10px 8px 8px 10px;margin:-50px 0px 0px 0px;background:black;border-radius: 0px 14px 14px 14px;border-bottom: 4px solid #aa4;border-right: 4px solid #dd5\"> Models </h1> \n## <h2 style= \"font-family: Cambria;font-weight:bold;color: white;padding:10px 4px 4px 10px;margin:-40px 0px 0px 0px;background:black;border-radius: 0px 14px 14px 14px;border-bottom: 4px solid #a4a;border-right: 4px solid #d5d\"> define functions</h2> ","metadata":{}},{"cell_type":"code","source":"def bayese_objective(X, y, Regressor, metric):\n    def bayese_trial(trial):\n        if Regressor == XGBRegressor:\n            params = {\n                'grow_policy': trial.suggest_categorical('grow_policy', [\"depthwise\", \"lossguide\"]),\n                'n_estimators': trial.suggest_int('n_estimators', 100, 1000),\n                'learning_rate': trial.suggest_float('learning_rate', 0.01, 1.0, log=True),\n                'gamma' : trial.suggest_float('gamma', 1e-9, 0.5),\n                'subsample': trial.suggest_float('subsample', 0.3, 1.0),\n                'colsample_bytree': trial.suggest_float('colsample_bytree', 0.3, 1.0),\n                'max_depth': trial.suggest_int('max_depth', 0, 12),\n                'min_child_weight': trial.suggest_int('min_child_weight', 1, 7),\n                'reg_lambda': trial.suggest_float('reg_lambda', 1e-9, 100.0, log=True),\n                'reg_alpha': trial.suggest_float('reg_alpha', 1e-9, 100.0, log=True),\n\n                'booster':'gbtree',\n                'device':\"cuda\",\n                'verbosity': 0,\n                'tree_method':\"hist\",\n                'eval_metric': metrics['XGB'],\n            }\n        elif Regressor == LGBMRegressor:\n            params = {\n                \"n_estimators\": trial.suggest_int('n_estimators', 50, 1000, step=10),\n                \"learning_rate\": trial.suggest_float('learning_rate', 0.01, 0.5, log=True),\n                \"max_depth\": trial.suggest_int('max_depth', 3, 15),\n                \"min_child_samples\": trial.suggest_int('lgbm_min_child_samples', 1, 20),\n                \"subsample\": trial.suggest_float('subsample', 0.5, 1.0),\n                \"colsample_bytree\": trial.suggest_float('colsample_bytree', 0.5, 1.0),\n                'num_leaves': trial.suggest_int('num_leaves', 2, 256),\n                'random_state': trial.suggest_int('random_state', 0, 1000),\n                'verbose':-1,\n                \n                'metric': metrics['LGBM']\n            }\n        elif Regressor == CatBoostRegressor:\n            params = {\n                \"iterations\": trial.suggest_int('iterations', 50, 1000, step=10),\n                \"learning_rate\": trial.suggest_float('learning_rate', 0.01, 0.5, log=True),\n                \"depth\": trial.suggest_int('depth', 3, 15),\n                \"l2_leaf_reg\": trial.suggest_float('l2_leaf_reg', 1e-3, 1),\n                \"random_state\": trial.suggest_int('random_state', 0, 1000),                \n                \"verbose\": False,\n                'eval_metric': metrics['CatBoost'], \n            }\n        cv = StratifiedKFold(n_splits=4, shuffle=True, random_state=0)\n\n        cv_splits = cv.split(X, y = y)\n        cv_scores = list()\n\n        for train_idx, val_idx in cv_splits:\n            model = Regressor()\n            model.set_params(**params)\n            X_train_fold, X_val_fold = X.iloc[train_idx], X.iloc[val_idx]\n            y_train_fold, y_val_fold = y.iloc[train_idx], y.iloc[val_idx]\n            model.fit(X_train_fold, y_train_fold)\n            \n            # y_val_prob = model.predict_proba(X_val_fold)[:,1]\n            # fpr, tpr, thresholds = roc_curve(y_val_fold, y_val_prob)\n\n            # score = auc(fpr, tpr)\n            y_val_prob = model.predict(X_val_fold)\n            score = mean_squared_log_error(y_val_fold, np.abs(y_val_prob))\n\n            cv_scores.append(score)\n        return np.mean(cv_scores)\n    return bayese_trial","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:45:46.147418Z","iopub.execute_input":"2024-12-31T22:45:46.147791Z","iopub.status.idle":"2024-12-31T22:45:46.161645Z","shell.execute_reply.started":"2024-12-31T22:45:46.147755Z","shell.execute_reply":"2024-12-31T22:45:46.160469Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"memo: metric candidates\n<table>\n    <tr>\n        <td></td><th>Regression</th>\n        <th>Binary Classification</th>\n        <th>Multiclass Classification</th>\n        <th>Ranking</th>\n    </tr><tr><th>XGboost</th>\n        <td>'rmse, 'mae', 'rmsle', 'mape'</td>\n        <td>'logloss', 'error', 'auc', 'aucpr', 'f1'</td>\n        <td>'merror', 'mlogloss', 'auc'</td>\n        <td>'map', 'ndcg', 'auc'</td>\n    </tr>\n    <tr><th>LightGBM</th>\n        <td>\n            'l2',\n            'mean_squared_error', 'l1', 'mean_absolute_error', 'huber',\n        'fair', 'poisson', 'quantile', 'mape', 'rmse', 'rmsle'\n        </td>\n        <td colspan=\"2\">\n            'binary_logloss', 'binary_error', 'auc', 'multi_logloss', 'multi_error',\n        </td>\n        <td>\n            'ndcg', 'map'\n        </td>\n    </tr>\n    <tr><th>CatBoost</th>\n        <td> 'RMSE', 'MAE', 'Quantile', 'MAPE', 'Poisson', 'Tweedie'</td>\n        <td colspan=\"2\">\n            'Accuracy', 'AUC', 'Logloss', 'CrossEntropy',\n            'Precision', 'Recall', 'F1', 'BalancedAccuracy',\n            'Kappa', 'WKappa', 'TotalF1', 'MultiClass', 'MultiClassOneVsAll', 'Custom', \n        </td>\n        <td>\n            'PFound', 'NDCG', 'AverageGain ', 'QuerySoftMax'\n        </td>\n    </tr>\n</table>","metadata":{}},{"cell_type":"markdown","source":"## <h2 style= \"font-family: Cambria;font-weight:bold;color: white;padding:10px 4px 4px 10px;margin:-40px 0px 0px 0px;background:black;border-radius: 0px 14px 14px 14px;border-bottom: 4px solid #a4a;border-right: 4px solid #d5d\"> Run BO or define parameters</h2> \n","metadata":{}},{"cell_type":"code","source":"%%time\nbest_params1 = []\nbest_scores1 = []\n_optim = False\n\nmetrics = {\n    'XGB': 'rmsle',\n    'LGBM': 'mean_squared_error',\n    'CatBoost': 'RMSE'\n}\nmodels = {\n    'XGB':XGBRegressor,\n    'LGBM':LGBMRegressor,\n    'Cat':CatBoostRegressor\n}\nif _optim:# run Bayese Optimization\n    for key, model in models.items():\n        print(f'{key}:')\n        try:\n            study = optuna.create_study(\n                direction = 'minimize',\n                sampler=optuna.samplers.TPESampler(seed=0),\n                study_name=f\"{key}_study\", storage=f\"sqlite:///{key}_study.db\", load_if_exists=True\n            )\n                        \n            study.optimize(\n                bayese_objective(X_train, y_train, model, metrics),\n                n_trials=50, timeout=3600 * 1, n_jobs = -1\n            )\n\n            best_params1.append(study.best_trial.params)\n            best_scores1.append(study.best_trial.value)\n            print(f'{key}:')\n            print('best params:')\n            print(best_params1[-1])\n            print('best scores:')\n            print(best_scores1[-1])\n        except:\n            print(f'{model} failed')\n\nelse:\n    best_params1 = {\n        'XGB':{\n            'grow_policy': 'lossguide', 'n_estimators': 686, 'learning_rate': 0.014904138542290992, 'gamma': 0.4525989478775382, 'subsample': 0.8111031783818401, 'colsample_bytree': 0.9748824881137719, 'max_depth': 8, 'min_child_weight': 4, 'reg_lambda': 1.2637449110901273e-06, 'reg_alpha': 5.0523818332960776e-06,\n            \n            'booster':'gbtree',\n            'device':\"cuda\",\n            'verbosity': 0,\n            'tree_method':\"hist\",\n            'eval_metric': metrics['XGB'],\n        },'LGBM':{\n            'n_estimators': 950, 'learning_rate': 0.011194224630485548, 'max_depth': 10, 'lgbm_min_child_samples': 5, 'subsample': 0.7832246773836278, 'colsample_bytree': 0.9354601600590208, 'num_leaves': 75, 'random_state': 880,\n            'random_state': 42,\n            'verbose':-1,\n            'metric': metrics['LGBM']\n        },'CatBoost':{\n            'iterations': 1000, 'learning_rate': 0.028792249201356347, 'depth': 8, 'l2_leaf_reg': 0.7668788982021637, 'random_state': 645,\n            \n            'random_state': 42,\n            \"verbose\": False,\n            'eval_metric': metrics['CatBoost'],\n        }}\n\n# print(best_params1,'\\n')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:45:46.162888Z","iopub.execute_input":"2024-12-31T22:45:46.163223Z","iopub.status.idle":"2024-12-31T22:45:46.187922Z","shell.execute_reply.started":"2024-12-31T22:45:46.163189Z","shell.execute_reply":"2024-12-31T22:45:46.186667Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <h2 style= \"font-family: Cambria;font-weight:bold;color: white;padding:10px 4px 4px 10px;margin:-40px 0px 0px 0px;background:black;border-radius: 0px 14px 14px 14px;border-bottom: 4px solid #a4a;border-right: 4px solåçid #d5d\"> Fit models</h2> ","metadata":{}},{"cell_type":"code","source":"%%time\nmodels_best = [XGBRegressor(), LGBMRegressor(), CatBoostRegressor()]\npredict_cols = ['XGB','LGBM', 'CatBoost']\n\n\npredict_trains = []\npredict_vals = []\npredict_tests = []\nfor i,model in enumerate(models_best):\n    print(predict_cols[i], end = ' / ')\n    model.set_params(**best_params1[predict_cols[i]])\n    model.fit(X_train, y_train)\n    predict_trains.append(model.predict(X_train).flatten())\n    predict_vals.append(model.predict(X_val).flatten())\n    predict_tests.append(model.predict(X_test).flatten())\nprint('finished')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:45:46.190215Z","iopub.execute_input":"2024-12-31T22:45:46.190565Z","iopub.status.idle":"2024-12-31T22:52:17.155616Z","shell.execute_reply.started":"2024-12-31T22:45:46.190532Z","shell.execute_reply":"2024-12-31T22:52:17.154451Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <h2 style= \"font-family: Cambria;font-weight:bold;color: white;padding:10px 4px 4px 10px;margin:-40px 0px 0px 0px;background:black;border-radius: 0px 14px 14px 14px;border-bottom: 4px solid #a4a;border-right: 4px solåçid #d5d\">Blend models</h2> ","metadata":{}},{"cell_type":"code","source":"predict_trains = pd.DataFrame(\n    np.array(predict_trains).T, columns = predict_cols, index=X_train.index\n)\npredict_vals = pd.DataFrame(\n    np.array(predict_vals).T, columns = predict_cols, index = X_val.index\n)\npredict_tests = pd.DataFrame(\n    np.array(predict_tests).T, columns = predict_cols, index = X_test.index\n)\npredict_cols = predict_cols + ['blend']\n\npredict_trains['blend'] = predict_trains.mean(axis=1)\npredict_vals['blend'] = predict_vals.mean(axis=1)\npredict_tests['blend'] = predict_tests.mean(axis=1)\n\npredict_trains['True'] = y_train\npredict_trains['train_val'] = 'train'\npredict_vals['True'] = y_val\npredict_vals['train_val'] = 'val'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:52:17.15718Z","iopub.execute_input":"2024-12-31T22:52:17.157625Z","iopub.status.idle":"2024-12-31T22:52:17.467291Z","shell.execute_reply.started":"2024-12-31T22:52:17.157576Z","shell.execute_reply":"2024-12-31T22:52:17.466397Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <h2 style= \"font-family: Cambria;font-weight:bold;color: white;padding:10px 4px 4px 10px;margin:-40px 0px 0px 0px;background:black;border-radius: 0px 14px 14px 14px;border-bottom: 4px solid #a4a;border-right: 4px solåçid #d5d\"> Result Plot</h2> ","metadata":{}},{"cell_type":"code","source":"target_max = max(predict_trains.max()[:-1].max(), predict_vals.max()[:-1].max())\ntarget_min = min(predict_trains.min()[:-1].min(), predict_vals.min()[:-1].min())\nfig, ax = plt.subplots(nrows = 2, ncols = len(models_best)+1, figsize = (12,6))\nfor i in range(len(models_best)+1):\n    cc = np.sqrt(mean_squared_log_error(predict_trains['True'], np.abs(predict_trains.iloc[:,i])))\n    ax[0][i].scatter(predict_trains['True'],predict_trains.iloc[:,i], s = 3, alpha=0.02)    \n    ax[0][i].set_title(f'{predict＿cols[i]} Train RMSE={cc:.3}', fontdict={'size':12})\n    ax[0][i].set_xlabel('True'); ax[0][i].set_ylabel('Predict');\n    ax[0][i].set_xlim(target_min, target_max); ax[0][i].set_ylim(target_min, target_max)\n    ax[0][i].set_aspect(1)\n    ax[0][i].grid()\n    \n    cc = np.sqrt(mean_squared_log_error(predict_vals['True'], np.abs(predict_vals.iloc[:,i])))\n    ax[1][i].scatter(predict_vals['True'],predict_vals.iloc[:,i], s = 3, alpha=0.1)\n    ax[1][i].set_title(f'{predict_cols[i]} Valid RMSE={cc:.3}', fontdict={'size':12})\n    ax[1][i].set_xlim(target_min, target_max); ax[1][i].set_ylim(target_min, target_max)\n    ax[1][i].set_xlabel('True'); ax[1][i].set_ylabel('Predict');\n    ax[1][i].set_aspect(1)\n    ax[1][i].grid()\n    \n\nplt.tight_layout()  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:52:17.468619Z","iopub.execute_input":"2024-12-31T22:52:17.469504Z","iopub.status.idle":"2024-12-31T22:52:26.282334Z","shell.execute_reply.started":"2024-12-31T22:52:17.469453Z","shell.execute_reply":"2024-12-31T22:52:26.281221Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <h2 style= \" font-family: Cambria; font-weight:bold; color: white; padding:10px 4px 4px 10px; margin:-40px 0px 0px -3px; background:black; border-radius: 0px 14px 14px 14px; border-bottom: 4px solid #4aa; border-right: 4px solid #5dd\"> Parameter Importances </h2> ","metadata":{}},{"cell_type":"code","source":"shap.initjs()\nmodel_button = widgets.ToggleButtons(\n    options = predict_cols[:-1],\n    button_style='info',description = 'model:'\n)\nmodel_button.style.button_width = f'100px'\nmodel_button.style.description_width = '90px'\n\ntype_button = widgets.ToggleButtons(\n    options = ['dot','bar'],\n    button_style='warning',description = 'type:'\n)\ntype_button.style.button_width = f'100px'\ntype_button.style.description_width = '90px'\n\nmax_disp_slider = widgets.IntSlider(\n    value=7, min=0, max=len(df_train.columns), step=1, \n    description='max_display:', orientation='horizontal'\n)\nmax_disp_slider.style.button_width = f'100px'\nmax_disp_slider.style.description_width = '90px'\n\n\n@interact(model_name = model_button, plot_type = type_button, max_display = max_disp_slider)\ndef plot_re(model_name, plot_type, max_display):\n    df_train = X_train.sample(1000)\n    i = list(predict_cols).index(model_name)\n\n    model = models_best[i]\n        \n    explainer = shap.TreeExplainer(model=model, model_output='raw')\n    shap_values = explainer.shap_values(X=df_train)\n    shap.summary_plot(shap_values, df_train, plot_type=plot_type, max_display=max_display)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:52:26.28356Z","iopub.execute_input":"2024-12-31T22:52:26.283901Z","iopub.status.idle":"2024-12-31T22:53:25.261915Z","shell.execute_reply.started":"2024-12-31T22:52:26.283852Z","shell.execute_reply":"2024-12-31T22:53:25.260879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = X_train.sample(1000)\ni = list(predict_cols).index('XGB')\n\nmodel = models_best[0]\n    \nexplainer = shap.TreeExplainer(model=model, model_output='raw')\nshap_values = explainer.shap_values(X=df_train)\n\n# 特徴量ごとの重要度を計算\nmean_shap_values = np.abs(shap_values).mean(axis=0)\n\n# 上位10個の特徴量を取得\ntop_10_indices = np.argsort(mean_shap_values)[-15:][::-1]\ntop_10_features = X_train.columns[top_10_indices]\ntop_10_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:53:25.263093Z","iopub.execute_input":"2024-12-31T22:53:25.263376Z","iopub.status.idle":"2024-12-31T22:53:25.300407Z","shell.execute_reply.started":"2024-12-31T22:53:25.263347Z","shell.execute_reply":"2024-12-31T22:53:25.299169Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <h1 style= \"font-family: Cambria;font-weight:bold;color: white;padding:10px 8px 8px 10px;margin:-50px 0px 0px 0px;background:black;border-radius: 0px 14px 14px 14px;border-bottom: 4px solid #aa4;border-right: 4px solid #dd5\"> Submit </h1> ","metadata":{"execution":{"iopub.status.busy":"2024-12-31T07:27:22.095586Z","iopub.execute_input":"2024-12-31T07:27:22.096506Z","iopub.status.idle":"2024-12-31T07:27:22.129278Z","shell.execute_reply.started":"2024-12-31T07:27:22.096459Z","shell.execute_reply":"2024-12-31T07:27:22.127832Z"}}},{"cell_type":"code","source":"y_test_predict = predict_tests['blend']\ndf_submit = df_sample_submission.set_index('id').copy()\ndf_submit[target_col] = y_test_predict\ndf_submit.to_csv('submit.csv',index = True)\n\ndisplay(df_submit)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T22:53:25.301184Z","iopub.status.idle":"2024-12-31T22:53:25.301534Z","shell.execute_reply.started":"2024-12-31T22:53:25.301369Z","shell.execute_reply":"2024-12-31T22:53:25.301387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}