{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n# %pip install catboost --upgrade\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nfrom IPython.display import display, HTML\nfrom colorama import Fore, Style\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, QuantileTransformer, OneHotEncoder, LabelEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.model_selection import train_test_split, RandomizedSearchCV, GridSearchCV\nfrom sklearn.metrics import accuracy_score, classification_report, precision_score, recall_score, f1_score, roc_auc_score\nfrom scipy.stats import randint\nfrom lightgbm import LGBMClassifier\nimport lightgbm as lgb\nfrom catboost import CatBoostRegressor\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import KFold\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nfrom sklearn.impute import SimpleImputer\n\nimport lightgbm as lgb\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\nwarnings.filterwarnings('ignore')\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T16:48:56.459032Z","iopub.execute_input":"2024-12-05T16:48:56.459283Z","iopub.status.idle":"2024-12-05T16:49:10.421632Z","shell.execute_reply.started":"2024-12-05T16:48:56.459257Z","shell.execute_reply":"2024-12-05T16:49:10.420759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_tr = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf_ts = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ndf_s = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T16:49:10.423494Z","iopub.execute_input":"2024-12-05T16:49:10.424504Z","iopub.status.idle":"2024-12-05T16:49:18.889655Z","shell.execute_reply.started":"2024-12-05T16:49:10.424473Z","shell.execute_reply":"2024-12-05T16:49:18.888739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_tr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T16:49:18.89095Z","iopub.execute_input":"2024-12-05T16:49:18.89132Z","iopub.status.idle":"2024-12-05T16:49:19.601726Z","shell.execute_reply.started":"2024-12-05T16:49:18.891285Z","shell.execute_reply":"2024-12-05T16:49:19.600862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_ts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T16:49:19.604143Z","iopub.execute_input":"2024-12-05T16:49:19.604799Z","iopub.status.idle":"2024-12-05T16:49:19.625102Z","shell.execute_reply.started":"2024-12-05T16:49:19.604768Z","shell.execute_reply":"2024-12-05T16:49:19.624166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_s","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T16:49:19.626189Z","iopub.execute_input":"2024-12-05T16:49:19.626481Z","iopub.status.idle":"2024-12-05T16:49:19.638747Z","shell.execute_reply.started":"2024-12-05T16:49:19.626441Z","shell.execute_reply":"2024-12-05T16:49:19.637733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"The (Row,Column) is:\\n\", df_tr.shape)\n    \nprint(\"Data type of each column:\\n\", df_tr.dtypes)\n    \nprint(\"The number of null values in each column are:\\n\", df_tr.isnull().sum())\n    \nprint(\"Numeric summary:\\n\", df_tr.describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T16:49:19.639739Z","iopub.execute_input":"2024-12-05T16:49:19.640026Z","iopub.status.idle":"2024-12-05T16:49:20.722064Z","shell.execute_reply.started":"2024-12-05T16:49:19.639988Z","shell.execute_reply":"2024-12-05T16:49:20.721161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def print_value_counts_less_than(df, threshold=50):\n    for col in df.columns:\n        value_counts = df[col].value_counts()\n        if len(value_counts) < threshold:\n            print(f\"Value counts for column '{col}':\")\n            for val, count in value_counts.items():\n                print(f\"    {val}: {count}\")\n            print()\n\nprint_value_counts_less_than(df_tr, threshold=30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T16:49:20.723175Z","iopub.execute_input":"2024-12-05T16:49:20.72346Z","iopub.status.idle":"2024-12-05T16:49:22.029049Z","shell.execute_reply.started":"2024-12-05T16:49:20.723433Z","shell.execute_reply":"2024-12-05T16:49:22.028292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_low_value_count_columns(df):\n    background_color = '#5fa1bc'\n    sns.set_theme(style=\"whitegrid\", rc={\"axes.facecolor\": background_color})\n    \n    low_value_count_columns = [col for col in df.columns if df[col].nunique() < 20]\n    num_plots = len(low_value_count_columns)\n    num_rows = (num_plots + 1) // 2  \n    num_cols = 2 \n    \n    fig, axs = plt.subplots(num_rows, num_cols, figsize=(15, 5 * num_rows))\n    axs = axs.flatten()\n    \n    for i, col in enumerate(low_value_count_columns):\n        p = sns.countplot(y=col, data=df, palette='magma', edgecolor='white', linewidth=2, ax=axs[i], order=df[col].value_counts().index)\n        for container in p.containers:\n            plt.bar_label(container, label_type='center', color=\"black\", fontsize=10, weight='bold', padding=6, position=(0.5, 0.5),\n                          bbox={\"boxstyle\": \"round\", \"pad\": 0.2, \"facecolor\": \"white\", \"edgecolor\": \"black\", \"linewidth\": 2, \"alpha\": 1})\n    \n    plt.tight_layout()\n    plt.show()\nplot_low_value_count_columns(df_tr)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T16:49:22.030066Z","iopub.execute_input":"2024-12-05T16:49:22.030343Z","iopub.status.idle":"2024-12-05T16:49:30.667455Z","shell.execute_reply.started":"2024-12-05T16:49:22.030316Z","shell.execute_reply":"2024-12-05T16:49:30.666587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_tr.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T16:49:30.668432Z","iopub.execute_input":"2024-12-05T16:49:30.668738Z","iopub.status.idle":"2024-12-05T16:49:31.218292Z","shell.execute_reply.started":"2024-12-05T16:49:30.668702Z","shell.execute_reply":"2024-12-05T16:49:31.217388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_columns = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score',\n                     'Vehicle Age', 'Credit Score', 'Insurance Duration']\nnumerical_columns = [col for col in df_tr.select_dtypes(include=['number']).columns if col != 'Target']\n\nfig, axes = plt.subplots(nrows=len(numerical_columns), ncols=1, figsize=(10, len(numerical_columns)*4))\n\naxes = axes.flatten()\n\nfor i, col in enumerate(numerical_columns):\n    sns.histplot(df_tr[col], kde=True, ax=axes[i])\n    axes[i].set_title(f'{col} Distribution')\n    axes[i].tick_params(axis='x', rotation=45)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T16:49:31.220492Z","iopub.execute_input":"2024-12-05T16:49:31.220787Z","iopub.status.idle":"2024-12-05T16:50:17.77624Z","shell.execute_reply.started":"2024-12-05T16:49:31.22076Z","shell.execute_reply":"2024-12-05T16:50:17.775389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_columns = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score',\n                     'Vehicle Age', 'Credit Score', 'Insurance Duration']\nnumerical_columns = [col for col in df_tr.select_dtypes(include=['number']).columns if col != 'Target']\n\nfig, axes = plt.subplots(nrows=len(numerical_columns), ncols=1, figsize=(10, len(numerical_columns)*4))\n\naxes = axes.flatten()\n\nfor i, col in enumerate(numerical_columns):\n    sns.histplot(df_tr[col], kde=True, ax=axes[i])\n    axes[i].set_title(f'{col} Distribution')\n    axes[i].tick_params(axis='x', rotation=45)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T16:50:17.777308Z","iopub.execute_input":"2024-12-05T16:50:17.777592Z","iopub.status.idle":"2024-12-05T16:51:04.060343Z","shell.execute_reply.started":"2024-12-05T16:50:17.777564Z","shell.execute_reply":"2024-12-05T16:51:04.059424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def fill_nulls(train_df, test_df, target_col):\n    # Separate the target column from the train data\n    X_train = train_df.drop(columns=[target_col])\n    y_train = train_df[target_col]\n    \n    # Identify numeric and categorical columns\n    numeric_cols = X_train.select_dtypes(include=['float64', 'int64']).columns\n    categorical_cols = X_train.select_dtypes(include=['object', 'category']).columns\n    \n    # Create imputers\n    iterative_imputer = IterativeImputer(max_iter=10, random_state=42)  # For numeric columns\n    categorical_imputer = SimpleImputer(strategy='constant', fill_value='Unknown')  # For categorical columns\n    \n    # Impute numeric columns\n    X_train[numeric_cols] = iterative_imputer.fit_transform(X_train[numeric_cols])\n    test_df[numeric_cols] = iterative_imputer.transform(test_df[numeric_cols])\n    \n    # Impute categorical columns\n    X_train[categorical_cols] = categorical_imputer.fit_transform(X_train[categorical_cols])\n    test_df[categorical_cols] = categorical_imputer.transform(test_df[categorical_cols])\n    \n    # Reattach the target column to the train dataset\n    train_df_filled = X_train.copy()\n    train_df_filled[target_col] = y_train\n\n    return train_df_filled, test_df\n\n# Apply the function\ndf_tr_filled, df_ts_filled = fill_nulls(df_tr, df_ts, target_col='Premium Amount')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T16:51:37.949089Z","iopub.execute_input":"2024-12-05T16:51:37.949446Z","iopub.status.idle":"2024-12-05T16:51:59.257808Z","shell.execute_reply.started":"2024-12-05T16:51:37.949412Z","shell.execute_reply":"2024-12-05T16:51:59.257057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert_to_category(df):\n    for col in df.columns:\n        if df[col].dtype == 'object':  \n            df[col] = df[col].astype('category')\n    return df\n\ndf_tr_category = convert_to_category(df_tr_filled)\ndf_ts_category = convert_to_category(df_ts_filled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T16:52:16.10235Z","iopub.execute_input":"2024-12-05T16:52:16.103184Z","iopub.status.idle":"2024-12-05T16:52:17.909108Z","shell.execute_reply.started":"2024-12-05T16:52:16.103149Z","shell.execute_reply":"2024-12-05T16:52:17.908234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df_tr_category.drop(columns=['Premium Amount'], axis=1)\ny = df_tr_category['Premium Amount']\ny_log = np.log1p(y)  \ndef rmsle(y_true, y_pred):\n    return np.sqrt(np.mean((np.log1p(y_pred) - np.log1p(y_true))**2))\ncat_features = [X.columns.get_loc(col) for col in X.select_dtypes(include=['category']).columns]\ncatboost_model = CatBoostRegressor(\n    iterations=1000, \n    learning_rate=0.1, \n    depth=6, \n    random_seed=42, \n    task_type=\"GPU\",  \n    devices=\"0\"       \n)\nkf = KFold(n_splits=10, shuffle=True, random_state=42)\nrmsle_scores = []\ntrain_rmsle_scores = []  \nfor fold, (train_idx, val_idx) in enumerate(kf.split(X)):\n    print(f\"Fold {fold + 1}\")\n    X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n    y_train, y_val = y_log.iloc[train_idx], y_log.iloc[val_idx]  \n    catboost_model.fit(\n        X_train, y_train, \n        eval_set=(X_val, y_val), \n        cat_features=cat_features,\n        early_stopping_rounds=50, \n        verbose=0\n    )\n    y_pred_val = catboost_model.predict(X_val)\n    y_pred_train = catboost_model.predict(X_train)\n    fold_rmsle_val = rmsle(np.expm1(y_val), np.maximum(np.expm1(y_pred_val), 0))  \n    fold_rmsle_train = rmsle(np.expm1(y_train), np.maximum(np.expm1(y_pred_train), 0))  \n    \n    rmsle_scores.append(fold_rmsle_val)\n    train_rmsle_scores.append(fold_rmsle_train)\n    \n    print(f\"Fold RMSLE (Validation): {fold_rmsle_val:.4f}\")\n    print(f\"Fold RMSLE (Training): {fold_rmsle_train:.4f}\")\nmean_rmsle = np.mean(rmsle_scores)\nstd_rmsle = np.std(rmsle_scores)\nmean_train_rmsle = np.mean(train_rmsle_scores)\nstd_train_rmsle = np.std(train_rmsle_scores)\nprint(f\"Mean RMSLE (Validation): {mean_rmsle:.4f}\")\nprint(f\"Standard Deviation of RMSLE (Validation): {std_rmsle:.4f}\")\nprint(f\"Mean RMSLE (Training): {mean_train_rmsle:.4f}\")\nprint(f\"Standard Deviation of RMSLE (Training): {std_train_rmsle:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T16:52:17.910706Z","iopub.execute_input":"2024-12-05T16:52:17.911041Z","iopub.status.idle":"2024-12-05T17:06:22.020703Z","shell.execute_reply.started":"2024-12-05T16:52:17.911013Z","shell.execute_reply":"2024-12-05T17:06:22.019722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"L_pred_log = catboost_model.predict(df_ts_category)\n\nL_pred = np.expm1(L_pred_log)\n\nprint(L_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:06:22.02208Z","iopub.execute_input":"2024-12-05T17:06:22.022356Z","iopub.status.idle":"2024-12-05T17:06:27.36957Z","shell.execute_reply.started":"2024-12-05T17:06:22.022328Z","shell.execute_reply":"2024-12-05T17:06:27.368762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_s['Premium Amount'] = L_pred\ndf_s.to_csv('Submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:06:27.370616Z","iopub.execute_input":"2024-12-05T17:06:27.3709Z","iopub.status.idle":"2024-12-05T17:06:28.710342Z","shell.execute_reply.started":"2024-12-05T17:06:27.370874Z","shell.execute_reply":"2024-12-05T17:06:28.709632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_s.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:06:28.711761Z","iopub.execute_input":"2024-12-05T17:06:28.712033Z","iopub.status.idle":"2024-12-05T17:06:28.720325Z","shell.execute_reply.started":"2024-12-05T17:06:28.712007Z","shell.execute_reply":"2024-12-05T17:06:28.719408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}