{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Regression with an Insurance Dataset\n\n\n## 1) Load the librairies","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:24.394837Z","iopub.execute_input":"2024-12-20T07:54:24.395188Z","iopub.status.idle":"2024-12-20T07:54:24.804031Z","shell.execute_reply.started":"2024-12-20T07:54:24.395159Z","shell.execute_reply":"2024-12-20T07:54:24.802856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sklearn\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:24.8054Z","iopub.execute_input":"2024-12-20T07:54:24.805955Z","iopub.status.idle":"2024-12-20T07:54:25.790083Z","shell.execute_reply.started":"2024-12-20T07:54:24.805922Z","shell.execute_reply":"2024-12-20T07:54:25.789004Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2) Load the Datasets","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:25.792467Z","iopub.execute_input":"2024-12-20T07:54:25.793095Z","iopub.status.idle":"2024-12-20T07:54:37.783724Z","shell.execute_reply.started":"2024-12-20T07:54:25.793048Z","shell.execute_reply":"2024-12-20T07:54:37.7825Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3) EDA (Exploratory Data Analysis)","metadata":{}},{"cell_type":"markdown","source":"### A) First Step","metadata":{}},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:37.785051Z","iopub.execute_input":"2024-12-20T07:54:37.785352Z","iopub.status.idle":"2024-12-20T07:54:37.837676Z","shell.execute_reply.started":"2024-12-20T07:54:37.785325Z","shell.execute_reply":"2024-12-20T07:54:37.836213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:37.839067Z","iopub.execute_input":"2024-12-20T07:54:37.83938Z","iopub.status.idle":"2024-12-20T07:54:38.504015Z","shell.execute_reply.started":"2024-12-20T07:54:37.83935Z","shell.execute_reply":"2024-12-20T07:54:38.502729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:38.504862Z","iopub.execute_input":"2024-12-20T07:54:38.505254Z","iopub.status.idle":"2024-12-20T07:54:39.233191Z","shell.execute_reply.started":"2024-12-20T07:54:38.505215Z","shell.execute_reply":"2024-12-20T07:54:39.231995Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Observation**: \n\n- With head() method, we can see that there are Nan values includes in the datasets. The dataset is composed of both numbers features and string features.There is a time Feature in the dataset.\n- With info() method, we can see that the dataset is composed of 21 columns and 1200000 rows., there is 9 features of types float, 1 of type Int and 11 of type Object (string).\n- With describe() method, We can see basic stats about the 10 features that are numbers (Strange feature: Previous Claim)","metadata":{}},{"cell_type":"markdown","source":"### b) Second Step\n\n- Display the features that are Nan values and use .ffill() method to replace the gap.","metadata":{}},{"cell_type":"code","source":"df_train.isna().any()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:39.234375Z","iopub.execute_input":"2024-12-20T07:54:39.234782Z","iopub.status.idle":"2024-12-20T07:54:39.850675Z","shell.execute_reply.started":"2024-12-20T07:54:39.234751Z","shell.execute_reply":"2024-12-20T07:54:39.84953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = df_train.ffill()\ndf_train.isna().any()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:39.854083Z","iopub.execute_input":"2024-12-20T07:54:39.854381Z","iopub.status.idle":"2024-12-20T07:54:42.760308Z","shell.execute_reply.started":"2024-12-20T07:54:39.854357Z","shell.execute_reply":"2024-12-20T07:54:42.75914Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- We have erasing the Nan values for the train dataset","metadata":{}},{"cell_type":"markdown","source":"### C) Third Step","metadata":{}},{"cell_type":"markdown","source":"We will display graphics based on intuition so that we can better visualize the data and clean the dataset","metadata":{}},{"cell_type":"code","source":"features = df_train[\"Previous Claims\"]\nfeatures_count = features.value_counts().sort_index()\nfig, ax = plt.subplots()\nax.plot(features_count.index, features_count.values, linewidth = 2.0)\nax.set_title(\"Distribution of Previous Claims\", fontsize=14)\nax.set_xlabel(\"Values of Previous Claims\", fontsize=12)\nax.set_ylabel(\"Number of People\", fontsize=12)\nax.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:42.762689Z","iopub.execute_input":"2024-12-20T07:54:42.76298Z","iopub.status.idle":"2024-12-20T07:54:43.085256Z","shell.execute_reply.started":"2024-12-20T07:54:42.762951Z","shell.execute_reply":"2024-12-20T07:54:43.08405Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Obervation**:\n\n- There are very few individuals with \"Previous Claims\" greater than 4. The third quartile (Q3) is 2, while the mean is 1. Therefore, I propose replacing all values greater than 2, as these are likely outliers, particularly higher values like 5, 6, 7, 8, and 9, which deviate significantly from the rest of the data.","metadata":{}},{"cell_type":"code","source":"features_std = np.where(features > 3, 3, features)\nfeatures_std_sr = pd.Series(features_std, name = \"Previous Claims\")\nfeatures_std_sr.unique()\ndf_train[\"Previous Claims\"] = features_std_sr\ndf_train[\"Previous Claims\"].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:43.08645Z","iopub.execute_input":"2024-12-20T07:54:43.086748Z","iopub.status.idle":"2024-12-20T07:54:43.143644Z","shell.execute_reply.started":"2024-12-20T07:54:43.08672Z","shell.execute_reply":"2024-12-20T07:54:43.142546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train[\"Previous Claims\"].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:43.144901Z","iopub.execute_input":"2024-12-20T07:54:43.14523Z","iopub.status.idle":"2024-12-20T07:54:43.209689Z","shell.execute_reply.started":"2024-12-20T07:54:43.145199Z","shell.execute_reply":"2024-12-20T07:54:43.20851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampled_data = df_train.sample(n=1200000, random_state=42)\nfeatures_x = sampled_data[\"Insurance Duration\"]\nfeatures_y = sampled_data[\"Premium Amount\"]\n\nfig, ax = plt.subplots()\nax.scatter(features_x, features_y, alpha=0.5)  # Scatter plot is better for large data\nax.set_title(\"Premium Amount vs. Insurance Duration (Sampled Data)\")\nax.set_xlabel(\"Insurance Duration\")\nax.set_ylabel(\"Premium Amount\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:43.210706Z","iopub.execute_input":"2024-12-20T07:54:43.211003Z","iopub.status.idle":"2024-12-20T07:54:48.058806Z","shell.execute_reply.started":"2024-12-20T07:54:43.210977Z","shell.execute_reply":"2024-12-20T07:54:48.057507Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Observation:\n\n- There appears to be no significant pattern or relationship between Insurance Duration and Premium Amount. The density of points for the Premium Amount remains largely consistent across different durations, suggesting that Insurance Duration has minimal or no impact on the target variable. Therefore, this feature might not contribute meaningfully to the model and could be considered for removal","metadata":{}},{"cell_type":"markdown","source":"#### 1) Encoding String Like Values","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:48.059888Z","iopub.execute_input":"2024-12-20T07:54:48.060363Z","iopub.status.idle":"2024-12-20T07:54:48.117617Z","shell.execute_reply.started":"2024-12-20T07:54:48.060331Z","shell.execute_reply":"2024-12-20T07:54:48.116331Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Displaying the columns that are Object type.","metadata":{}},{"cell_type":"code","source":"columns_object = [x for x in df_train.columns if df_train[x].dtype == \"object\"]\ncolumns_object","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:48.118924Z","iopub.execute_input":"2024-12-20T07:54:48.119321Z","iopub.status.idle":"2024-12-20T07:54:48.127792Z","shell.execute_reply.started":"2024-12-20T07:54:48.11928Z","shell.execute_reply":"2024-12-20T07:54:48.126587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def encoding_string_values(df, columns_object = None):\n    if columns_object == None:\n        columns_object = [x for x in df_train.columns if df_train[x].dtype == \"object\"]\n    dict_label_encoder = {}\n    for label in columns_object:\n        label_encoder = LabelEncoder()\n        df[label] = label_encoder.fit_transform(df[label])\n        dict_label_encoder[label] = label_encoder\n    return dict_label_encoder, df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:48.128935Z","iopub.execute_input":"2024-12-20T07:54:48.129314Z","iopub.status.idle":"2024-12-20T07:54:48.154045Z","shell.execute_reply.started":"2024-12-20T07:54:48.12928Z","shell.execute_reply":"2024-12-20T07:54:48.152339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dict_label_encoder, df_train = encoding_string_values(df_train)\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:48.155738Z","iopub.execute_input":"2024-12-20T07:54:48.156254Z","iopub.status.idle":"2024-12-20T07:54:51.830606Z","shell.execute_reply.started":"2024-12-20T07:54:48.156203Z","shell.execute_reply":"2024-12-20T07:54:51.829293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dict_label_encoder[\"Marital Status\"].classes_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:51.831726Z","iopub.execute_input":"2024-12-20T07:54:51.83212Z","iopub.status.idle":"2024-12-20T07:54:51.838944Z","shell.execute_reply.started":"2024-12-20T07:54:51.832075Z","shell.execute_reply":"2024-12-20T07:54:51.837721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = df_train[\"Marital Status\"]\nmarital_classes = dict_label_encoder[\"Marital Status\"].classes_\n\nfig,ax = plt.subplots()\nax.hist(features)\nax.set_title(\"Density of Marital Status\")\nax.set_xlabel(\"Marital Status\")\nax.set_ylabel(\"Number of people\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:51.840179Z","iopub.execute_input":"2024-12-20T07:54:51.84055Z","iopub.status.idle":"2024-12-20T07:54:52.204371Z","shell.execute_reply.started":"2024-12-20T07:54:51.840505Z","shell.execute_reply":"2024-12-20T07:54:52.203127Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4) Clean the Data and Prepare the dataset","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:52.20588Z","iopub.execute_input":"2024-12-20T07:54:52.206299Z","iopub.status.idle":"2024-12-20T07:54:52.361629Z","shell.execute_reply.started":"2024-12-20T07:54:52.206255Z","shell.execute_reply":"2024-12-20T07:54:52.360358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if \"id\" in df_train.columns:\n    df_train = df_train.drop(\"id\", axis = 1)\ny = df_train[\"Premium Amount\"]\nX = df_train.drop(columns = [\"Premium Amount\",\"Policy Start Date\"], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:52.362663Z","iopub.execute_input":"2024-12-20T07:54:52.362991Z","iopub.status.idle":"2024-12-20T07:54:52.672629Z","shell.execute_reply.started":"2024-12-20T07:54:52.36296Z","shell.execute_reply":"2024-12-20T07:54:52.671455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if \"id\" in df_test.columns:\n    id_submit = df_test[\"id\"]\n    df_test = df_test.drop(columns = [\"id\",\"Policy Start Date\"], axis = 1)\n\nassert  X.columns.equals(df_test.columns), \"Column mismatch between X and df_test\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:52.673835Z","iopub.execute_input":"2024-12-20T07:54:52.674259Z","iopub.status.idle":"2024-12-20T07:54:52.832551Z","shell.execute_reply.started":"2024-12-20T07:54:52.674214Z","shell.execute_reply":"2024-12-20T07:54:52.831494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X =  StandardScaler().fit_transform(X)\nX_train,X_test, y_train, y_test = train_test_split(X,y, test_size = 0.2, random_state = 42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:52.833605Z","iopub.execute_input":"2024-12-20T07:54:52.833981Z","iopub.status.idle":"2024-12-20T07:54:53.708061Z","shell.execute_reply.started":"2024-12-20T07:54:52.833953Z","shell.execute_reply":"2024-12-20T07:54:53.706962Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5) Train the model","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression,Ridge,Lasso,ElasticNet\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.svm import SVR","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:53.712244Z","iopub.execute_input":"2024-12-20T07:54:53.712558Z","iopub.status.idle":"2024-12-20T07:54:54.033189Z","shell.execute_reply.started":"2024-12-20T07:54:53.712534Z","shell.execute_reply":"2024-12-20T07:54:54.031899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def root_mean_squared_log_error(y_true, y_pred):\n    \"\"\"\n    Calculate Root Mean Squared Logarithmic Error (RMSLE).\n\n    Parameters:\n    y_true : array-like, true values\n    y_pred : array-like, predicted values\n\n    Returns:\n    rmsle : float, the root mean squared logarithmic error\n    \"\"\"\n    log_true = np.log1p(y_true)\n    log_pred = np.log1p(y_pred)\n    return np.sqrt(np.mean((log_true - log_pred) ** 2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:54.034774Z","iopub.execute_input":"2024-12-20T07:54:54.035129Z","iopub.status.idle":"2024-12-20T07:54:54.041101Z","shell.execute_reply.started":"2024-12-20T07:54:54.035097Z","shell.execute_reply":"2024-12-20T07:54:54.039933Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### A) Choosing a model","metadata":{}},{"cell_type":"code","source":"model_title = [\"Linear\", \"Ridge\", \"Lasso\", \"Elastic\", \"Tree\"]\nmodel_linear = [LinearRegression(), Ridge(), Lasso(), ElasticNet(),DecisionTreeRegressor()]\nmodel_list = []\nfor title, model in zip(model_title, model_linear):\n    model.fit(X_train,y_train)\n    y_pred = model.predict(X_test)\n    result = root_mean_squared_log_error(y_test, y_pred)\n    model_list.append([title, model, result])\n    \n\nmodel_list_sorted = sorted(model_list, key = lambda x: x[2])\nfor title, model, result in model_list_sorted:\n    print(f\"Model: {title}, RMSLE: {result:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:54:54.042619Z","iopub.execute_input":"2024-12-20T07:54:54.04313Z","iopub.status.idle":"2024-12-20T07:55:19.341962Z","shell.execute_reply.started":"2024-12-20T07:54:54.043085Z","shell.execute_reply":"2024-12-20T07:55:19.340589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_ch = model_list_sorted[1][1]\nmodel_ch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T07:55:19.343399Z","iopub.execute_input":"2024-12-20T07:55:19.343831Z","iopub.status.idle":"2024-12-20T07:55:19.357312Z","shell.execute_reply.started":"2024-12-20T07:55:19.343784Z","shell.execute_reply":"2024-12-20T07:55:19.355992Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### B) Fine Tuning a model","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV, RandomizedSearchCV\nfrom sklearn.metrics import make_scorer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T08:08:16.955683Z","iopub.execute_input":"2024-12-20T08:08:16.956103Z","iopub.status.idle":"2024-12-20T08:08:16.961274Z","shell.execute_reply.started":"2024-12-20T08:08:16.956072Z","shell.execute_reply":"2024-12-20T08:08:16.959678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_grid = {\n    'alpha': [0.1],  # Regularization strength\n    'fit_intercept': [True],\n    'solver': ['auto', 'svd', 'cholesky', 'lsqr', 'sparse_cg', 'sag', 'saga']\n    }\n\n\ncustom_metrics = make_scorer(root_mean_squared_log_error)\n\n\ngrid_search = GridSearchCV(\n    model_ch, \n    param_grid, \n    scoring=custom_metrics,  \n    cv=5,  \n    verbose=2\n)\n\n\ngrid_search.fit(X_train, y_train)\n\n\nprint(\"Best Parameters:\", grid_search.best_params_)\nprint(\"Best Score:\", grid_search.best_score_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T08:15:24.657298Z","iopub.execute_input":"2024-12-20T08:15:24.65765Z","iopub.status.idle":"2024-12-20T08:19:08.180345Z","shell.execute_reply.started":"2024-12-20T08:15:24.657625Z","shell.execute_reply":"2024-12-20T08:19:08.178149Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_grid = {\n    'alpha': [0.1, 0.5, 1, 5, 10, 50, 100, 500, 1000, 5000, 10000],  # Regularization strength\n    'fit_intercept': [True],\n    'solver': ['lsqr']\n    }\n\n\ncustom_metrics = make_scorer(root_mean_squared_log_error)\n\n\ngrid_search = GridSearchCV(\n    model_ch, \n    param_grid, \n    scoring=custom_metrics,  \n    cv=5,  \n    verbose=2\n)\n\n\ngrid_search.fit(X_train, y_train)\n\n\nprint(\"Best Parameters:\", grid_search.best_params_)\nprint(\"Best Score:\", grid_search.best_score_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T08:24:22.351507Z","iopub.execute_input":"2024-12-20T08:24:22.351855Z","iopub.status.idle":"2024-12-20T08:24:47.753862Z","shell.execute_reply.started":"2024-12-20T08:24:22.351826Z","shell.execute_reply":"2024-12-20T08:24:47.752457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_model = grid_search.best_estimator_\ny_pred = best_model.predict(X_test)\ntest_score = root_mean_squared_log_error(y_test, y_pred)\nprint(\"Score sur les données de test :\", test_score)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T08:29:52.360568Z","iopub.execute_input":"2024-12-20T08:29:52.360964Z","iopub.status.idle":"2024-12-20T08:29:52.425294Z","shell.execute_reply.started":"2024-12-20T08:29:52.360934Z","shell.execute_reply":"2024-12-20T08:29:52.422479Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### C) Features Selections","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_selection import RFE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T08:34:25.04172Z","iopub.execute_input":"2024-12-20T08:34:25.042074Z","iopub.status.idle":"2024-12-20T08:34:25.082629Z","shell.execute_reply.started":"2024-12-20T08:34:25.042048Z","shell.execute_reply":"2024-12-20T08:34:25.081782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selector = RFE(model_ch,n_features_to_select=1, step=1)\nselector = selector.fit(X, y)\nprint(selector.support_)\n\nprint(f\"Ranking Features -> {selector.ranking_}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T08:36:48.099182Z","iopub.execute_input":"2024-12-20T08:36:48.099569Z","iopub.status.idle":"2024-12-20T08:36:51.056962Z","shell.execute_reply.started":"2024-12-20T08:36:48.099537Z","shell.execute_reply":"2024-12-20T08:36:51.053972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_ranking = {}\nfor feature, ranking in zip(df_train.columns, selector.ranking_):\n    features_ranking[feature] = ranking\nsorted_dict = dict(sorted(features_ranking.items(), key=lambda item: item[1]))\nprint(sorted_dict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T08:40:03.217835Z","iopub.execute_input":"2024-12-20T08:40:03.218238Z","iopub.status.idle":"2024-12-20T08:40:03.224995Z","shell.execute_reply.started":"2024-12-20T08:40:03.218207Z","shell.execute_reply":"2024-12-20T08:40:03.223605Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6) Submit the Prediction","metadata":{}},{"cell_type":"markdown","source":"id_submit : var that store the id \n\ndf_test : var that store the test dataset","metadata":{}},{"cell_type":"code","source":"df_test.ffill(inplace = True)\ndf_test.bfill(inplace = True)\ndf_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T08:30:55.59579Z","iopub.execute_input":"2024-12-20T08:30:55.596128Z","iopub.status.idle":"2024-12-20T08:30:57.561973Z","shell.execute_reply.started":"2024-12-20T08:30:55.596099Z","shell.execute_reply":"2024-12-20T08:30:57.560784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dict_label_encoder.items()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T08:30:59.289899Z","iopub.execute_input":"2024-12-20T08:30:59.290253Z","iopub.status.idle":"2024-12-20T08:30:59.301228Z","shell.execute_reply.started":"2024-12-20T08:30:59.290225Z","shell.execute_reply":"2024-12-20T08:30:59.299706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for column , le in dict_label_encoder.items():\n    if column != \"Policy Start Date\":\n        df_test[column] = le.transform(df_test[column])\ndf_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T08:31:01.955269Z","iopub.execute_input":"2024-12-20T08:31:01.955645Z","iopub.status.idle":"2024-12-20T08:31:03.187591Z","shell.execute_reply.started":"2024-12-20T08:31:01.955614Z","shell.execute_reply":"2024-12-20T08:31:03.186453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test = StandardScaler().fit_transform(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T08:31:11.6918Z","iopub.execute_input":"2024-12-20T08:31:11.692164Z","iopub.status.idle":"2024-12-20T08:31:11.980568Z","shell.execute_reply.started":"2024-12-20T08:31:11.692137Z","shell.execute_reply":"2024-12-20T08:31:11.979433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def submit_model_predict(model, df = df_test, id = None, name_file_csv = \"submission.csv\" ,name_column = None):\n    try:\n        result = model.predict(df)\n        submission = pd.DataFrame({\"id\":id,name_column: result})\n        submission.to_csv(name_file_csv, index = False)\n        return True\n    except Exception as e:\n        # Print the exception for debugging purposes\n        print(f\"Error during submission: {e}\")\n        return False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T08:31:41.561178Z","iopub.execute_input":"2024-12-20T08:31:41.561548Z","iopub.status.idle":"2024-12-20T08:31:41.567309Z","shell.execute_reply.started":"2024-12-20T08:31:41.561519Z","shell.execute_reply":"2024-12-20T08:31:41.566116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit_model_predict(best_model, df = df_test, id = id_submit, name_file_csv = \"submission_3.csv\", name_column = \"Premium Amount\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T08:32:01.27425Z","iopub.execute_input":"2024-12-20T08:32:01.274639Z","iopub.status.idle":"2024-12-20T08:32:03.023036Z","shell.execute_reply.started":"2024-12-20T08:32:01.274605Z","shell.execute_reply":"2024-12-20T08:32:03.02202Z"}},"outputs":[],"execution_count":null}]}