{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:38:01.687993Z","iopub.execute_input":"2024-12-30T13:38:01.688325Z","iopub.status.idle":"2024-12-30T13:38:01.696548Z","shell.execute_reply.started":"2024-12-30T13:38:01.688299Z","shell.execute_reply":"2024-12-30T13:38:01.695144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:38:04.012956Z","iopub.execute_input":"2024-12-30T13:38:04.013327Z","iopub.status.idle":"2024-12-30T13:38:04.017964Z","shell.execute_reply.started":"2024-12-30T13:38:04.013293Z","shell.execute_reply":"2024-12-30T13:38:04.016589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data=pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ndata.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:38:04.257433Z","iopub.execute_input":"2024-12-30T13:38:04.257855Z","iopub.status.idle":"2024-12-30T13:38:10.115297Z","shell.execute_reply.started":"2024-12-30T13:38:04.257826Z","shell.execute_reply":"2024-12-30T13:38:10.114244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.drop(\"id\",axis=1,inplace=True)\ndata.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:38:10.116537Z","iopub.execute_input":"2024-12-30T13:38:10.116892Z","iopub.status.idle":"2024-12-30T13:38:10.291943Z","shell.execute_reply.started":"2024-12-30T13:38:10.116856Z","shell.execute_reply":"2024-12-30T13:38:10.290608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:38:10.294157Z","iopub.execute_input":"2024-12-30T13:38:10.29459Z","iopub.status.idle":"2024-12-30T13:38:10.301618Z","shell.execute_reply.started":"2024-12-30T13:38:10.294555Z","shell.execute_reply":"2024-12-30T13:38:10.300602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:38:10.302716Z","iopub.execute_input":"2024-12-30T13:38:10.302982Z","iopub.status.idle":"2024-12-30T13:38:10.937702Z","shell.execute_reply.started":"2024-12-30T13:38:10.302959Z","shell.execute_reply":"2024-12-30T13:38:10.936635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.heatmap(data.corr(numeric_only=True),annot=True, fmt=\".2f\", cmap=\"coolwarm\", cbar=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:38:10.938772Z","iopub.execute_input":"2024-12-30T13:38:10.939133Z","iopub.status.idle":"2024-12-30T13:38:12.154796Z","shell.execute_reply.started":"2024-12-30T13:38:10.939097Z","shell.execute_reply":"2024-12-30T13:38:12.153439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.hist(figsize=(12, 10), bins=20)\nplt.suptitle(\"Histogram for Numerical Columns\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:38:12.156168Z","iopub.execute_input":"2024-12-30T13:38:12.156568Z","iopub.status.idle":"2024-12-30T13:38:14.088539Z","shell.execute_reply.started":"2024-12-30T13:38:12.156531Z","shell.execute_reply":"2024-12-30T13:38:14.087532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T12:40:15.160224Z","iopub.execute_input":"2024-12-30T12:40:15.160554Z","iopub.status.idle":"2024-12-30T12:40:15.783609Z","shell.execute_reply.started":"2024-12-30T12:40:15.160496Z","shell.execute_reply":"2024-12-30T12:40:15.782626Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Performing EDA using pivot tables","metadata":{}},{"cell_type":"code","source":"data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:38:33.0844Z","iopub.execute_input":"2024-12-30T13:38:33.084833Z","iopub.status.idle":"2024-12-30T13:38:33.720659Z","shell.execute_reply.started":"2024-12-30T13:38:33.084801Z","shell.execute_reply":"2024-12-30T13:38:33.719444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data[\"Education Level\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:38:47.032581Z","iopub.execute_input":"2024-12-30T13:38:47.032922Z","iopub.status.idle":"2024-12-30T13:38:47.128885Z","shell.execute_reply.started":"2024-12-30T13:38:47.032896Z","shell.execute_reply":"2024-12-30T13:38:47.127732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pivot1=pd.pivot_table(data,index=\"Age\",columns=\"Education Level\",values=\"Premium Amount\",aggfunc=\"count\")\nsns.barplot(pivot1)\nplt.plot()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:38:52.700542Z","iopub.execute_input":"2024-12-30T13:38:52.700887Z","iopub.status.idle":"2024-12-30T13:38:53.143044Z","shell.execute_reply.started":"2024-12-30T13:38:52.700858Z","shell.execute_reply":"2024-12-30T13:38:53.141827Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From above graph it is clearly seen that the High school on an average takes less Premium amount compared to Bachelors Phd Masters ","metadata":{}},{"cell_type":"code","source":"num_col=data.drop(\"Premium Amount\",axis=1).select_dtypes(include=['float64']).columns\ncat_col=data.select_dtypes(include=['object']).columns\n\nfig, axes = plt.subplots(nrows=len(num_col), ncols=1, figsize=(8, len(num_col) * 4))  # Create subplots\n\n# Loop through numerical columns and plot each on a separate subplot\nfor ax, col in zip(axes, num_col):\n    sns.boxplot(x=data[col], ax=ax)  # Plot on the corresponding axis\n    ax.set_title(f\"Boxplot of {col}\")  # Set title for each subplot\n\nplt.tight_layout()  # Adjust layout to avoid overlap\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:40:06.246696Z","iopub.execute_input":"2024-12-30T13:40:06.247057Z","iopub.status.idle":"2024-12-30T13:40:08.500539Z","shell.execute_reply.started":"2024-12-30T13:40:06.24703Z","shell.execute_reply":"2024-12-30T13:40:08.499388Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Function for Calculating and Replacing Outliers","metadata":{}},{"cell_type":"code","source":"# Here we will be calculating the Outliers and then limit with upperbound and lower bound\ndef calculate_iqr_bounds(df, column):\n    Q1 = df[column].quantile(0.25)\n    Q3 = df[column].quantile(0.75)\n    IQR = Q3 - Q1\n    lower_bound = Q1 - 1.5 * IQR\n    upper_bound = Q3 + 1.5 * IQR\n    return lower_bound, upper_bound\n    \ndef replace_outliers(df, column, lower_bound, upper_bound):\n    df[column] = df[column].clip(lower=lower_bound, upper=upper_bound)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:41:20.922021Z","iopub.execute_input":"2024-12-30T13:41:20.922477Z","iopub.status.idle":"2024-12-30T13:41:20.928794Z","shell.execute_reply.started":"2024-12-30T13:41:20.922442Z","shell.execute_reply":"2024-12-30T13:41:20.927386Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# following columns has outlier\n1. Annual Income\n2. Previous Claim","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:41:47.317165Z","iopub.execute_input":"2024-12-30T13:41:47.317589Z","iopub.status.idle":"2024-12-30T13:41:47.979536Z","shell.execute_reply.started":"2024-12-30T13:41:47.317557Z","shell.execute_reply":"2024-12-30T13:41:47.978459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import StandardScaler,OneHotEncoder,MinMaxScaler,MaxAbsScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression,SGDRegressor,Ridge,Lasso,ElasticNet\nfrom sklearn.ensemble import RandomForestRegressor,GradientBoostingRegressor,StackingRegressor\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.feature_selection import RFE\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans\nfrom lightgbm import LGBMRegressor\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import make_scorer\nfrom xgboost import XGBRegressor\nfrom sklearn.svm import SVR","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:41:54.391192Z","iopub.execute_input":"2024-12-30T13:41:54.391552Z","iopub.status.idle":"2024-12-30T13:41:54.398445Z","shell.execute_reply.started":"2024-12-30T13:41:54.391523Z","shell.execute_reply":"2024-12-30T13:41:54.397124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df=data.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:56:40.090204Z","iopub.execute_input":"2024-12-30T13:56:40.090617Z","iopub.status.idle":"2024-12-30T13:56:40.290231Z","shell.execute_reply.started":"2024-12-30T13:56:40.090582Z","shell.execute_reply":"2024-12-30T13:56:40.288992Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Extracting information like day, month and year from column Policy Start Date","metadata":{}},{"cell_type":"code","source":"df[\"Policy Start Date\"]=pd.to_datetime(df['Policy Start Date'])\ndf[\"day\"]=df[\"Policy Start Date\"].dt.day_name()\n# df[\"date\"]=df[\"Policy Start Date\"].dt.day\ndf[\"month\"] = df[\"Policy Start Date\"].dt.month\ndf[\"year\"] = df[\"Policy Start Date\"].dt.year\ndf.drop(\"Policy Start Date\",axis=1,inplace =True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:56:42.512101Z","iopub.execute_input":"2024-12-30T13:56:42.512478Z","iopub.status.idle":"2024-12-30T13:56:43.576692Z","shell.execute_reply.started":"2024-12-30T13:56:42.512448Z","shell.execute_reply":"2024-12-30T13:56:43.575612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df[\"date\"].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:56:46.0328Z","iopub.execute_input":"2024-12-30T13:56:46.033133Z","iopub.status.idle":"2024-12-30T13:56:46.037123Z","shell.execute_reply.started":"2024-12-30T13:56:46.033106Z","shell.execute_reply":"2024-12-30T13:56:46.035795Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model performed good when converted did sin transformation on year.","metadata":{}},{"cell_type":"code","source":"df['year_sin'] = np.sin(2 * np.pi * (df['year'] - 2019) / (2024 - 2019))\ndf.drop(\"year\",axis=1,inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:56:48.08792Z","iopub.execute_input":"2024-12-30T13:56:48.08826Z","iopub.status.idle":"2024-12-30T13:56:48.301002Z","shell.execute_reply.started":"2024-12-30T13:56:48.088231Z","shell.execute_reply":"2024-12-30T13:56:48.299626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_col=df.drop(\"Premium Amount\",axis=1).select_dtypes(include=['float64']).columns\ncat_col=df.select_dtypes(include=['object']).columns\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:56:50.364589Z","iopub.execute_input":"2024-12-30T13:56:50.364955Z","iopub.status.idle":"2024-12-30T13:56:51.180891Z","shell.execute_reply.started":"2024-12-30T13:56:50.364925Z","shell.execute_reply":"2024-12-30T13:56:51.179705Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dividing Data into training and test data","metadata":{}},{"cell_type":"code","source":"X_train,X_test,y_train,y_test=train_test_split(df.drop(\"Premium Amount\",axis=1),df[\"Premium Amount\"],random_state=20,test_size=0.1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:56:52.612661Z","iopub.execute_input":"2024-12-30T13:56:52.613001Z","iopub.status.idle":"2024-12-30T13:56:53.409391Z","shell.execute_reply.started":"2024-12-30T13:56:52.612975Z","shell.execute_reply":"2024-12-30T13:56:53.408466Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Handling Outliers","metadata":{}},{"cell_type":"code","source":"# for balance column\nlower_bound_train_income, upper_bound_train_income = calculate_iqr_bounds(X_train, 'Annual Income')\nX_train = replace_outliers(X_train, 'Annual Income', lower_bound_train_income, upper_bound_train_income)\nX_test=replace_outliers(X_test, 'Annual Income', lower_bound_train_income, upper_bound_train_income)\n# for Previous Claims\nlower_bound_train_previous, upper_bound_train_previous = calculate_iqr_bounds(X_train, 'Previous Claims')\nX_train = replace_outliers(X_train, 'Previous Claims', lower_bound_train_income, upper_bound_train_income)\nX_test=replace_outliers(X_test, 'Previous Claims', lower_bound_train_income, upper_bound_train_income)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:56:55.412847Z","iopub.execute_input":"2024-12-30T13:56:55.413195Z","iopub.status.idle":"2024-12-30T13:56:55.531193Z","shell.execute_reply.started":"2024-12-30T13:56:55.41316Z","shell.execute_reply":"2024-12-30T13:56:55.53026Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Creating a scorer for root_mean_squared_log_error ","metadata":{}},{"cell_type":"code","source":"def rmsle(y_true, y_pred):\n    # Avoid log(0) by clipping predictions to a minimum value\n    y_pred = np.clip(y_pred, 1e-10, None)\n    return np.sqrt(np.mean((np.log1p(y_pred) - np.log1p(y_true)) ** 2))\nrmsle_scorer=make_scorer(rmsle,greater_is_better=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:56:57.69671Z","iopub.execute_input":"2024-12-30T13:56:57.697047Z","iopub.status.idle":"2024-12-30T13:56:57.701738Z","shell.execute_reply.started":"2024-12-30T13:56:57.697022Z","shell.execute_reply":"2024-12-30T13:56:57.700629Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Doing data preprocessing before model training","metadata":{}},{"cell_type":"code","source":"cat_col","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:56:59.87895Z","iopub.execute_input":"2024-12-30T13:56:59.879381Z","iopub.status.idle":"2024-12-30T13:56:59.885842Z","shell.execute_reply.started":"2024-12-30T13:56:59.879346Z","shell.execute_reply":"2024-12-30T13:56:59.884699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_pipe=Pipeline([\n     # (\"impute\",SimpleImputer(strategy=\"constant\",fill_value=\"missing\")),\n    (\"ohe\",OneHotEncoder()),\n    (\"scale\",StandardScaler(with_mean=False))\n])\nnum_pipe=Pipeline([#(\"impute\",SimpleImputer(strategy=\"mean\")),\n                 (\"scale\",StandardScaler(with_mean=False))\n                    \n                  ])\nct=ColumnTransformer([\n    (\"num\",num_pipe,num_col),\n    (\"cat\",cat_pipe,cat_col)])\ntransformer=Pipeline([(\"ct\",ct)\n                      # (\"model\",LGBMRegressor())\n])\ntransformed=transformer.fit_transform(X_train)                               ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:57:01.592049Z","iopub.execute_input":"2024-12-30T13:57:01.592472Z","iopub.status.idle":"2024-12-30T13:57:07.344858Z","shell.execute_reply.started":"2024-12-30T13:57:01.592434Z","shell.execute_reply":"2024-12-30T13:57:07.343775Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Below function to provide parameter to objective","metadata":{}},{"cell_type":"code","source":"def rmsle_objective(y_true, y_pred):\n    y_pred = np.maximum(0, y_pred)  # Ensure non-negativity\n    grad = -1 / (y_pred + 1) * (np.log1p(y_true) - np.log1p(y_pred))\n    hess = 1 / ((y_pred + 1) ** 2)\n    return grad, hess","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:57:52.081762Z","iopub.execute_input":"2024-12-30T13:57:52.082186Z","iopub.status.idle":"2024-12-30T13:57:52.087181Z","shell.execute_reply.started":"2024-12-30T13:57:52.082148Z","shell.execute_reply":"2024-12-30T13:57:52.085948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model=LGBMRegressor(learning_rate=0.2,max_depth=-1,n_estimators=150,reg_alpha=0.5,reg_lambda=0.1,objective=rmsle_objective)\nmodel=LGBMRegressor(objective=rmsle_objective) \n# model=XGBRegressor(objective=rmsle_objective,n_estimators=150,grow_policy='lossguide')\n# Maximum depth of trees\n# }\n# model=GridSearchCV(estimator=LGBMRegressor(),param_grid=param_grid,cv=4,n_jobs=-1,scoring=rmsle_scorer)\nmodel.fit(transformed,y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:57:53.526437Z","iopub.execute_input":"2024-12-30T13:57:53.526835Z","iopub.status.idle":"2024-12-30T13:58:05.913075Z","shell.execute_reply.started":"2024-12-30T13:57:53.526806Z","shell.execute_reply":"2024-12-30T13:58:05.911945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred=model.predict(transformer.transform(X_test))\nrmsle(y_test,y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:58:05.914747Z","iopub.execute_input":"2024-12-30T13:58:05.915173Z","iopub.status.idle":"2024-12-30T13:58:06.678452Z","shell.execute_reply.started":"2024-12-30T13:58:05.915121Z","shell.execute_reply":"2024-12-30T13:58:06.67741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(rmsle(y_train,model.predict(transformed)))\n# model.best_params_\n# output for Grid Search\n# {'learning_rate': 0.2,\n#  'max_depth': -1,\n # 'n_estimators': 150,\n # 'reg_alpha': 0.5, \n#  'reg_lambda': 0.1}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:58:11.01817Z","iopub.execute_input":"2024-12-30T13:58:11.018557Z","iopub.status.idle":"2024-12-30T13:58:11.022206Z","shell.execute_reply.started":"2024-12-30T13:58:11.018528Z","shell.execute_reply":"2024-12-30T13:58:11.021153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# grid.best_params_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:58:13.714458Z","iopub.execute_input":"2024-12-30T13:58:13.714844Z","iopub.status.idle":"2024-12-30T13:58:13.718388Z","shell.execute_reply.started":"2024-12-30T13:58:13.714814Z","shell.execute_reply":"2024-12-30T13:58:13.717338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# clustering=KMeans(n_clusters=2,random_state=42)\n# X_train[\"cluster\"]=clustering.fit_predict(X_train_transformed)\n# # clustering.fit(X_train_transformed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:58:13.901357Z","iopub.execute_input":"2024-12-30T13:58:13.901758Z","iopub.status.idle":"2024-12-30T13:58:13.90573Z","shell.execute_reply.started":"2024-12-30T13:58:13.901727Z","shell.execute_reply":"2024-12-30T13:58:13.904572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# for cluster_label in X_train[\"cluster\"].unique():\n#     cluster_data=X_train[X_train[\"cluster\"]==cluster_label]\n#     plt.scatter(\n#         cluster_data[\"Age\"][:100],cluster_data[\"Annual Income\"][:100],\n#         label=f'Cluster {cluster_label}', alpha=0.6\n#     )\n# plt.title('Cluster Visualization')\n# plt.xlabel('Age')\n# plt.ylabel('Annual Income')\n# plt.legend()\n# plt.grid()\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:58:14.061218Z","iopub.execute_input":"2024-12-30T13:58:14.061588Z","iopub.status.idle":"2024-12-30T13:58:14.06553Z","shell.execute_reply.started":"2024-12-30T13:58:14.061558Z","shell.execute_reply":"2024-12-30T13:58:14.064385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_comp=pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\n# X_comp.drop(\"Policy Start Date\",axis=1,inplace=True)\ntest_id=X_comp[\"id\"]\nX_comp.drop(\"id\",axis=1,inplace=True)\nX_comp[\"Policy Start Date\"]=pd.to_datetime(X_comp['Policy Start Date'])\nX_comp[\"day\"]=X_comp[\"Policy Start Date\"].dt.day_name()\n# df[\"date\"]=df[\"Policy Start Date\"].dt.day\nX_comp[\"month\"] = X_comp[\"Policy Start Date\"].dt.month\nX_comp[\"year\"] = X_comp[\"Policy Start Date\"].dt.year\nX_comp['year_sin'] = np.sin(2 * np.pi * (X_comp['year'] - 2019) / (2024 - 2019))\nX_comp.drop(\"Policy Start Date\",axis=1,inplace =True)\nX_comp.drop(\"year\",axis=1,inplace =True)\n# X_comp.drop(\"month\",axis=1,inplace =True)\n\nX_comp=replace_outliers(X_comp, 'Annual Income', lower_bound_train_income, upper_bound_train_income)\nX_comp=replace_outliers(X_comp, 'Previous Claims', lower_bound_train_previous, upper_bound_train_previous)\ny_com_pred=model.predict(transformer.transform(X_comp))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:58:16.141666Z","iopub.execute_input":"2024-12-30T13:58:16.142032Z","iopub.status.idle":"2024-12-30T13:58:25.565108Z","shell.execute_reply.started":"2024-12-30T13:58:16.142001Z","shell.execute_reply":"2024-12-30T13:58:25.564031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission=pd.DataFrame({\n    \"id\":test_id,\n    \"Premium Amount\":y_com_pred\n    })\nsubmission.to_csv(\"submission.csv\",index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T13:58:25.5666Z","iopub.execute_input":"2024-12-30T13:58:25.566905Z","iopub.status.idle":"2024-12-30T13:58:27.17064Z","shell.execute_reply.started":"2024-12-30T13:58:25.566873Z","shell.execute_reply":"2024-12-30T13:58:27.169209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}