{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install scikit-learn==1.5.2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:30:33.513289Z","iopub.execute_input":"2024-12-09T06:30:33.513674Z","iopub.status.idle":"2024-12-09T06:30:43.594139Z","shell.execute_reply.started":"2024-12-09T06:30:33.513641Z","shell.execute_reply":"2024-12-09T06:30:43.592588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sklearn\nsklearn.__version__","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:30:43.596538Z","iopub.execute_input":"2024-12-09T06:30:43.596924Z","iopub.status.idle":"2024-12-09T06:30:43.604808Z","shell.execute_reply.started":"2024-12-09T06:30:43.596888Z","shell.execute_reply":"2024-12-09T06:30:43.603671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import root_mean_squared_log_error","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:30:43.606348Z","iopub.execute_input":"2024-12-09T06:30:43.606781Z","iopub.status.idle":"2024-12-09T06:30:43.617443Z","shell.execute_reply.started":"2024-12-09T06:30:43.606735Z","shell.execute_reply":"2024-12-09T06:30:43.616478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:30:43.619721Z","iopub.execute_input":"2024-12-09T06:30:43.620061Z","iopub.status.idle":"2024-12-09T06:30:51.06952Z","shell.execute_reply.started":"2024-12-09T06:30:43.620028Z","shell.execute_reply":"2024-12-09T06:30:51.068589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:30:51.071047Z","iopub.execute_input":"2024-12-09T06:30:51.071397Z","iopub.status.idle":"2024-12-09T06:30:51.756296Z","shell.execute_reply.started":"2024-12-09T06:30:51.071366Z","shell.execute_reply":"2024-12-09T06:30:51.755057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_col = [col for col in train_df.columns if train_df[col].dtype == 'float64' or train_df[col].dtype == 'int64']\ncat_col = [col for col in train_df.columns if train_df[col].dtype == 'object']\n\nnum_col.remove('Premium Amount')\ntarget_col = ['Premium Amount']\n\nprint(\"numerical columns:\", num_col)\nprint(\"categorical columns:\", cat_col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:30:51.757565Z","iopub.execute_input":"2024-12-09T06:30:51.757892Z","iopub.status.idle":"2024-12-09T06:30:51.765724Z","shell.execute_reply.started":"2024-12-09T06:30:51.757863Z","shell.execute_reply":"2024-12-09T06:30:51.764716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.histplot(train_df['Premium Amount'], kde=True)\nplt.title('Distribution of Premium Amount')\nplt.xlabel('Premium Amount')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:30:51.766881Z","iopub.execute_input":"2024-12-09T06:30:51.767222Z","iopub.status.idle":"2024-12-09T06:30:58.057992Z","shell.execute_reply.started":"2024-12-09T06:30:51.767162Z","shell.execute_reply":"2024-12-09T06:30:58.056773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(len(num_col), 1, figsize=(15, 5 * len(num_col))) \n\nfor i, col in enumerate(num_col):\n    train_df['Binned'] = pd.qcut(train_df[col], q=5, duplicates='drop')  \n    sns.boxplot(data=train_df, x='Binned', y='Premium Amount', ax=axes[i], palette='viridis') \n    axes[i].set_title(f'Boxplot {col}')\n    axes[i].set_xlabel(f'{col} Binned')  # Gunakan f-string\n    axes[i].set_ylabel('Premium Amount')\n    sns.despine(ax=axes[i])\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:30:58.059454Z","iopub.execute_input":"2024-12-09T06:30:58.059847Z","iopub.status.idle":"2024-12-09T06:31:02.474859Z","shell.execute_reply.started":"2024-12-09T06:30:58.059806Z","shell.execute_reply":"2024-12-09T06:31:02.473791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filtered_columns = [col for col in cat_col if col != 'Policy Start Date']\n\nfig, axes = plt.subplots(len(filtered_columns), 1, figsize=(15, 5 * len(filtered_columns)))  # 1 kolom per plot\n\nfor i, col in enumerate(filtered_columns):\n    sns.countplot(data=train_df, x=col, ax=axes[i], palette='viridis')\n    axes[i].set_title(f'Countplot of {col}')\n    axes[i].set_xlabel(col)\n    axes[i].set_ylabel('Count')\n    axes[i].tick_params(axis='x', rotation=45)  # Rotasi label sumbu X jika terlalu panjang\n    sns.despine(ax=axes[i])\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:31:02.476368Z","iopub.execute_input":"2024-12-09T06:31:02.476751Z","iopub.status.idle":"2024-12-09T06:31:10.926247Z","shell.execute_reply.started":"2024-12-09T06:31:02.47671Z","shell.execute_reply":"2024-12-09T06:31:10.925113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"correlation_matrix = train_df[num_col + ['Premium Amount']].corr()\n\nplt.figure(figsize=(12, 8))\nsns.heatmap(correlation_matrix, annot=True, fmt=\".2f\", cmap='coolwarm', cbar=True)\nplt.title('Correlation Matrix')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:31:10.930333Z","iopub.execute_input":"2024-12-09T06:31:10.930756Z","iopub.status.idle":"2024-12-09T06:31:11.96136Z","shell.execute_reply.started":"2024-12-09T06:31:10.930712Z","shell.execute_reply":"2024-12-09T06:31:11.960256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing = train_df.isnull().sum()\nmissing = missing[missing > 0]\nmissing.sort_values(inplace=True)\nmissing.plot(kind='barh')\nplt.title('Missing Values in Training Data')\nplt.xlabel('Number of Missing Values')\nplt.ylabel('Features')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:31:11.963149Z","iopub.execute_input":"2024-12-09T06:31:11.964069Z","iopub.status.idle":"2024-12-09T06:31:12.880931Z","shell.execute_reply.started":"2024-12-09T06:31:11.964032Z","shell.execute_reply":"2024-12-09T06:31:12.879646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"duplicate = train_df.duplicated().sum()\nprint(f\"Number of duplicate rows: {duplicate}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:31:12.882861Z","iopub.execute_input":"2024-12-09T06:31:12.883366Z","iopub.status.idle":"2024-12-09T06:31:14.740986Z","shell.execute_reply.started":"2024-12-09T06:31:12.88332Z","shell.execute_reply":"2024-12-09T06:31:14.739737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_df.drop(columns=['Premium Amount', 'Policy Start Date'])\ny = train_df['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:31:14.742184Z","iopub.execute_input":"2024-12-09T06:31:14.742515Z","iopub.status.idle":"2024-12-09T06:31:14.965971Z","shell.execute_reply.started":"2024-12-09T06:31:14.742485Z","shell.execute_reply":"2024-12-09T06:31:14.964878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_col.remove('id')\ncat_col.remove('Policy Start Date')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:31:14.967635Z","iopub.execute_input":"2024-12-09T06:31:14.968083Z","iopub.status.idle":"2024-12-09T06:31:14.974234Z","shell.execute_reply.started":"2024-12-09T06:31:14.968027Z","shell.execute_reply":"2024-12-09T06:31:14.973117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder\n\nnum_pipeline = Pipeline([\n    ('imputer', SimpleImputer(strategy=\"mean\")),\n#    ('std_scaler', StandardScaler()),\n])\n\ncat_pipeline = Pipeline([\n    ('imputer', SimpleImputer(strategy=\"constant\", fill_value=\"Unknown\")),\n    ('onehot', OneHotEncoder(handle_unknown='ignore')),\n])\n\npreprocessor = ColumnTransformer([\n    (\"num\", num_pipeline, num_col),\n    (\"cat\", cat_pipeline, cat_col),\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:31:14.975664Z","iopub.execute_input":"2024-12-09T06:31:14.976005Z","iopub.status.idle":"2024-12-09T06:31:14.988373Z","shell.execute_reply.started":"2024-12-09T06:31:14.975961Z","shell.execute_reply":"2024-12-09T06:31:14.987188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_processed = preprocessor.fit_transform(X)\ntest_processed = preprocessor.transform(test_df.drop(columns=['id', 'Policy Start Date']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:31:14.98978Z","iopub.execute_input":"2024-12-09T06:31:14.990236Z","iopub.status.idle":"2024-12-09T06:31:22.660208Z","shell.execute_reply.started":"2024-12-09T06:31:14.990167Z","shell.execute_reply":"2024-12-09T06:31:22.659274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X_processed, y, test_size=0.3, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:31:22.661377Z","iopub.execute_input":"2024-12-09T06:31:22.661692Z","iopub.status.idle":"2024-12-09T06:31:22.911135Z","shell.execute_reply.started":"2024-12-09T06:31:22.661662Z","shell.execute_reply":"2024-12-09T06:31:22.910034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = LinearRegression()\nmodel.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:31:22.912512Z","iopub.execute_input":"2024-12-09T06:31:22.912825Z","iopub.status.idle":"2024-12-09T06:31:24.481465Z","shell.execute_reply.started":"2024-12-09T06:31:22.912796Z","shell.execute_reply":"2024-12-09T06:31:24.480188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(X_test)\nrmsle = root_mean_squared_log_error(y_test, y_pred)\nprint(f\"RMSLE: {rmsle:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:31:24.482806Z","iopub.execute_input":"2024-12-09T06:31:24.483185Z","iopub.status.idle":"2024-12-09T06:31:24.547746Z","shell.execute_reply.started":"2024-12-09T06:31:24.483145Z","shell.execute_reply":"2024-12-09T06:31:24.54383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\nrf = RandomForestRegressor()\n\nparams = {\n    'n_jobs' : [-1],\n    'n_estimators' : [20,30,50,70],\n    'criterion' : ['squared_error', 'absolute_error', 'friedman_mse'],\n    'max_depth' : [None,2,5,7],\n    'random_state': [42]\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:33:42.214134Z","iopub.execute_input":"2024-12-09T06:33:42.214551Z","iopub.status.idle":"2024-12-09T06:33:42.220475Z","shell.execute_reply.started":"2024-12-09T06:33:42.214516Z","shell.execute_reply":"2024-12-09T06:33:42.219125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.experimental import enable_halving_search_cv\nfrom sklearn.model_selection import HalvingGridSearchCV\n\ngrid = HalvingGridSearchCV(rf, params, scoring='neg_root_mean_squared_error', cv=3, verbose=1).fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:37:37.632032Z","iopub.execute_input":"2024-12-09T06:37:37.632459Z","iopub.status.idle":"2024-12-09T06:44:00.24885Z","shell.execute_reply.started":"2024-12-09T06:37:37.632422Z","shell.execute_reply":"2024-12-09T06:44:00.243211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Best parameters:', grid.best_params_)\nprint('Best score:', grid.best_score_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:44:00.488844Z","iopub.status.idle":"2024-12-09T06:44:00.489418Z","shell.execute_reply.started":"2024-12-09T06:44:00.489126Z","shell.execute_reply":"2024-12-09T06:44:00.489155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = grid.predict(X_test)\nrmsle = root_mean_squared_log_error(y_test, y_pred)\nprint(f\"RMSLE: {rmsle:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T06:44:00.412989Z","iopub.execute_input":"2024-12-09T06:44:00.416519Z","iopub.status.idle":"2024-12-09T06:44:00.484789Z","shell.execute_reply.started":"2024-12-09T06:44:00.416458Z","shell.execute_reply":"2024-12-09T06:44:00.481224Z"}},"outputs":[],"execution_count":null}]}