{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":9178166,"sourceType":"datasetVersion","datasetId":5547076}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![img](https://www.eoda.de/wp-content/uploads/2020/02/Versicherungen-Data-Science.jpg)\n\n---\n\n\n### Problem definition: Regression\n\n### Target: \n\nContinuous feature. Predict \"Premium Amount\" \n\n---\n","metadata":{}},{"cell_type":"markdown","source":"# 🔧 SETUP - Import libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport missingno as msno\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport matplotlib.gridspec as gridspec\nfrom scipy import stats\n\nimport matplotlib.style as style\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_log_error\nimport pandas as pd\nimport missingno as mnso\nimport numpy as np\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.ensemble import RandomForestRegressor\nfrom scipy.stats import boxcox\nfrom scipy.special import boxcox1p\nfrom sklearn.model_selection import cross_val_score, StratifiedKFold\n\nsns.set_style(\"darkgrid\", {\"grid.color\": \".6\", \"grid.linestyle\": \":\"})\n\ncolor_palette = sns.color_palette(['#D62728', '#9467BD', '#8C564B', '#1F77B4', '#FF7F0E', '#2CA02C'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:01:33.668383Z","iopub.execute_input":"2024-12-31T02:01:33.668818Z","iopub.status.idle":"2024-12-31T02:01:35.125063Z","shell.execute_reply.started":"2024-12-31T02:01:33.668767Z","shell.execute_reply":"2024-12-31T02:01:35.124123Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load data : train,test and original","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\noriginal = pd.read_csv(\"/kaggle/input/insurance-premium-prediction/Insurance Premium Prediction Dataset.csv\")\n\ntrain.shape,test.shape,original.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:01:35.126177Z","iopub.execute_input":"2024-12-31T02:01:35.126622Z","iopub.status.idle":"2024-12-31T02:01:45.083314Z","shell.execute_reply.started":"2024-12-31T02:01:35.126596Z","shell.execute_reply":"2024-12-31T02:01:45.082542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_id = test.id.values\ntest.drop(\"id\",axis  = 1, inplace = True)\ntrain.drop(\"id\", axis  = 1,inplace = True)\n\ntrain = pd.concat([train,original], axis = 0)\n\ntrain.shape,test.shape,original.shape\ntrain_size = train.shape[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:01:45.085053Z","iopub.execute_input":"2024-12-31T02:01:45.085287Z","iopub.status.idle":"2024-12-31T02:01:45.655972Z","shell.execute_reply.started":"2024-12-31T02:01:45.085268Z","shell.execute_reply":"2024-12-31T02:01:45.655229Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Concat (X)_train and test","metadata":{}},{"cell_type":"code","source":"X = train.copy()\nX.dropna(subset=[\"Premium Amount\"], inplace=True)\ny = X.pop(\"Premium Amount\")\n\nfull = pd.concat([X,test], axis = 0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:01:45.657136Z","iopub.execute_input":"2024-12-31T02:01:45.657378Z","iopub.status.idle":"2024-12-31T02:01:47.092619Z","shell.execute_reply.started":"2024-12-31T02:01:45.657357Z","shell.execute_reply":"2024-12-31T02:01:47.091703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Duplicates ?","metadata":{}},{"cell_type":"code","source":"full.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:01:47.093565Z","iopub.execute_input":"2024-12-31T02:01:47.093813Z","iopub.status.idle":"2024-12-31T02:01:50.174563Z","shell.execute_reply.started":"2024-12-31T02:01:47.093792Z","shell.execute_reply":"2024-12-31T02:01:50.17371Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 📊 Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"display(train.info())\ntrain.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:01:50.175499Z","iopub.execute_input":"2024-12-31T02:01:50.175795Z","iopub.status.idle":"2024-12-31T02:01:51.63007Z","shell.execute_reply.started":"2024-12-31T02:01:50.175765Z","shell.execute_reply":"2024-12-31T02:01:51.629363Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🔎 Missing","metadata":{}},{"cell_type":"code","source":"print(\"y missing : \")\nprint(\"%: \",y.isnull().mean())\nprint(\"sum: \", y.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:01:51.630798Z","iopub.execute_input":"2024-12-31T02:01:51.630996Z","iopub.status.idle":"2024-12-31T02:01:51.639375Z","shell.execute_reply.started":"2024-12-31T02:01:51.630979Z","shell.execute_reply":"2024-12-31T02:01:51.638694Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### %","metadata":{}},{"cell_type":"code","source":"def missing_percentage(df):\n    \"\"\"This function takes a DataFrame(df) as input and returns two columns, total missing values and total missing values percentage\"\"\"\n    ## the two following line may seem complicated but its actually very simple. \n    total = df.isnull().sum().sort_values(ascending = False)[df.isnull().sum().sort_values(ascending = False) != 0]\n    percent = round(df.isnull().sum().sort_values(ascending = False)/len(df)*100,2)[round(df.isnull().sum().sort_values(ascending = False)/len(df)*100,2) != 0]\n    return pd.concat([total, percent], axis=1, keys=['Total','Percent'])\n\nmissing_percentage(full)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:01:51.641561Z","iopub.execute_input":"2024-12-31T02:01:51.641827Z","iopub.status.idle":"2024-12-31T02:01:55.56131Z","shell.execute_reply.started":"2024-12-31T02:01:51.641808Z","shell.execute_reply":"2024-12-31T02:01:55.560542Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Missing Matrix","metadata":{}},{"cell_type":"code","source":"msno.matrix(full)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:01:55.562771Z","iopub.execute_input":"2024-12-31T02:01:55.562992Z","iopub.status.idle":"2024-12-31T02:02:07.604631Z","shell.execute_reply.started":"2024-12-31T02:01:55.562973Z","shell.execute_reply":"2024-12-31T02:02:07.603748Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🎯 Target\n\nWe can see that premium amount is right skewed. Therefore we can try to transform our target to see if we transform it to a (more) normal distribution. To do this we apply a log transformation.","metadata":{}},{"cell_type":"code","source":"import joblib\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\ndef plotting_3_chart(df, feature):\n    ## Importing seaborn, matplotlab and scipy modules. \n    style.use('fivethirtyeight')\n    \n    df = df.dropna()\n    ## Creating a customized chart. and giving in figsize and everything. \n    fig = plt.figure(constrained_layout=True, figsize=(12,8))\n    ## creating a grid of 3 cols and 3 rows. \n    grid = gridspec.GridSpec(ncols=3, nrows=3, figure=fig)\n    #gs = fig3.add_gridspec(3, 3)\n\n    ## Customizing the histogram grid. \n    ax1 = fig.add_subplot(grid[0, :2])\n    ## Set the title. \n    ax1.set_title('Histogram')\n    ## plot the histogram. \n    sns.distplot(df.loc[:,feature], norm_hist=True, ax = ax1)\n\n    # customizing the QQ_plot. \n    ax2 = fig.add_subplot(grid[1, :2])\n    ## Set the title. \n    ax2.set_title('QQ_plot')\n    ## Plotting the QQ_Plot. \n    stats.probplot(df.loc[:,feature], plot = ax2)\n\n    ## Customizing the Box Plot. \n    ax3 = fig.add_subplot(grid[:, 2])\n    ## Set title. \n    ax3.set_title('Box Plot')\n    ## Plotting the box plot. \n    sns.boxplot(df.loc[:,feature], orient='v', ax = ax3 );\n\n    #skewness and kurtosis\n    print(\"Skewness: \" + str(df[feature].skew()))\n    print(\"Kurtosis: \" + str(df[feature].kurt()))\n\nplotting_3_chart(train, 'Premium Amount')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:07.605594Z","iopub.execute_input":"2024-12-31T02:02:07.60595Z","iopub.status.idle":"2024-12-31T02:02:13.081947Z","shell.execute_reply.started":"2024-12-31T02:02:07.605913Z","shell.execute_reply":"2024-12-31T02:02:13.081108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 5))\n\ntrain['Premium Amount'].plot(kind='hist', bins=100, ax=ax[0], color='#f8766d')\n\nnp.log1p(train['Premium Amount']).plot(kind='hist', bins=100, ax=ax[1], color='#00bfc4')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:13.08286Z","iopub.execute_input":"2024-12-31T02:02:13.083127Z","iopub.status.idle":"2024-12-31T02:02:14.030014Z","shell.execute_reply.started":"2024-12-31T02:02:13.083096Z","shell.execute_reply":"2024-12-31T02:02:14.029183Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n\n![img](https://media.geeksforgeeks.org/wp-content/uploads/20231101173636/Skewness-copy.webp)\n\n### Analysis of \"Premium Amount\" from the Plots:\n\n1. **Histogram (Top Left)**:\n   - The histogram indicates a **right-skewed distribution** (also known as **positively skewed**). This means that the majority of the data points (Premium Amounts) are clustered towards the lower end, but there are a few extreme values (outliers) pulling the distribution to the right. \n   - The histogram also shows that most premium amounts are concentrated between 0 and 2000, with a noticeable long tail extending towards 5000, indicating that higher values are rarer but still present.\n\n2. **Box Plot (Top Right)**:\n   - The box plot shows a **typical distribution with some outliers**. The median (the line inside the box) is around 3000, suggesting that the central tendency of Premium Amount is higher than the lower range values. \n   - There are **outliers** above 4000, indicating that there are extreme premium amounts in the dataset. The box itself is relatively narrow, which suggests that most of the data is close to the median.\n\n3. **Probability Plot (Bottom)**:\n   - The probability plot compares the observed values of \"Premium Amount\" to a **normal distribution**. The points deviate from the red line, showing that the distribution is not normal. The points in the lower and upper ends of the plot curve away from the theoretical line, suggesting that the data exhibits skewness, especially in the tail areas.\n\n### Skewness and Kurtosis:\n\n- **Skewness**: \n   - The skewness value is **1.2769**, indicating that the distribution of \"Premium Amount\" is positively skewed (right skew). In simpler terms, most of the values are on the lower side, and the tail stretches toward the higher values (premium amounts above 3000-4000). This is consistent with the observation in the histogram and the probability plot.\n\n- **Kurtosis**:\n   - The kurtosis value is **1.6123**, which is lower than the kurtosis of a normal distribution (which is 3). A kurtosis value below 3 suggests that the distribution is **platykurtic** – it has fewer extreme values or outliers than a normal distribution. In this case, although there are some outliers, the distribution does not have an excessive number of extreme values compared to a normal distribution.\n\n### Conclusion:\n\n- The \"Premium Amount\" variable has a **right-skewed distribution**, meaning most customers pay lower premiums, with a few paying much higher premiums.\n- The **skewness** confirms the right-skewed nature of the data.\n- The **kurtosis** indicates a distribution with fewer outliers than a normal distribution, though there are still some noticeable outliers in the higher premium amounts.\n\n---\n\n<!--  nar:\n\n### Análisis de \"Monto Premium\" a partir de los gráficos:\n\n1. **Histograma (arriba a la izquierda)**:\n- El histograma indica una **distribución sesgada hacia la derecha** (también conocida como **sesgada positivamente**). Esto significa que la mayoría de los puntos de datos (Montos Premium) están agrupados hacia el extremo inferior, pero hay algunos valores extremos (valores atípicos) que empujan la distribución hacia la derecha.\n- El histograma también muestra que la mayoría de los montos Premium se concentran entre 0 y 2000, con una cola larga notable que se extiende hacia 5000, lo que indica que los valores más altos son más raros pero aún están presentes.\n\n2. **Gráfico de caja (arriba a la derecha)**:\n- El gráfico de caja muestra una **distribución típica con algunos valores atípicos**. La mediana (la línea dentro del recuadro) es de alrededor de 3000, lo que sugiere que la tendencia central de Monto de la prima es más alta que los valores del rango inferior.\n- Hay **valores atípicos** por encima de 4000, lo que indica que hay montos de prima extremos en el conjunto de datos. El recuadro en sí es relativamente estrecho, lo que sugiere que la mayoría de los datos están cerca de la mediana.\n\n3. **Gráfico de probabilidad (parte inferior)**:\n- El gráfico de probabilidad compara los valores observados de \"Monto de la prima\" con una **distribución normal**. Los puntos se desvían de la línea roja, lo que muestra que la distribución no es normal. Los puntos en los extremos inferior y superior del gráfico se alejan de la línea teórica, lo que sugiere que los datos presentan asimetría, especialmente en las áreas de cola.\n\n### Asimetría y curtosis:\n\n- **Asimetría**:\n- El valor de asimetría es **1,2769**, lo que indica que la distribución de \"Importe de prima\" está sesgada positivamente (sesgo a la derecha). En términos más simples, la mayoría de los valores están en el lado inferior y la cola se extiende hacia los valores más altos (importes de prima superiores a 3000-4000). Esto es coherente con la observación en el histograma y el gráfico de probabilidad.\n\n- **Curtosis**:\n- El valor de curtosis es **1,6123**, que es inferior a la curtosis de una distribución normal (que es 3). Un valor de curtosis inferior a 3 sugiere que la distribución es **platicúrtica**: tiene menos valores extremos o valores atípicos que una distribución normal. En este caso, aunque hay algunos valores atípicos, la distribución no tiene una cantidad excesiva de valores extremos en comparación con una distribución normal.\n\n### Conclusión:\n\n- La variable \"Monto de la prima\" tiene una **distribución sesgada hacia la derecha**, lo que significa que la mayoría de los clientes pagan primas más bajas y unos pocos pagan primas mucho más altas.\n- La **asimetría** confirma la naturaleza sesgada hacia la derecha de los datos.\n- La **curtosis** indica una distribución con menos valores atípicos que una distribución normal, aunque todavía hay algunos valores atípicos notables en los montos de prima más altos.\n\nEste tipo de información es esencial para el **preprocesamiento de datos** (como normalizar o transformar los datos) y comprender el comportamiento de la variable \"Monto de la prima\" en cualquier tarea de modelado, ya que la asimetría puede afectar el rendimiento de ciertos algoritmos de aprendizaje automático.-->\n\n---","metadata":{}},{"cell_type":"markdown","source":"# 🔢 Numerical features: Statistical analysis\n\nWe continu by inspecting the numerical features in the dataset. We again start by checking if there is a statistical significant relation between the target (premium amount) and each numerical feature in the dataset. This time we can use the Spearman rank-order correlation coefficient. ","metadata":{}},{"cell_type":"markdown","source":"### Histograms","metadata":{}},{"cell_type":"code","source":"\nnum_columns = 3 \nnum_rows = len(train.columns) // num_columns + (1 if len(train.columns) % num_columns != 0 else 0)\n\nfig, axes = plt.subplots(num_rows, num_columns, figsize=(15, num_rows * 5))\n\n\naxes = axes.flatten()\n\n\nfor i, col in enumerate(train.select_dtypes(include=['float64', 'int64']).columns):\n    train[col].hist(ax=axes[i], bins=30)\n    axes[i].set_title(f'Histogram of {col}')\n    axes[i].set_xlabel(col)\n    axes[i].set_ylabel('Frec')\n\nfor j in range(i + 1, len(axes)):\n    fig.delaxes(axes[j])\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:14.03085Z","iopub.execute_input":"2024-12-31T02:02:14.031123Z","iopub.status.idle":"2024-12-31T02:02:16.943091Z","shell.execute_reply.started":"2024-12-31T02:02:14.031101Z","shell.execute_reply":"2024-12-31T02:02:16.942148Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Spearman Correlation\n\nhttps://en.wikipedia.org/wiki/Spearman%27s_rank_correlation_coefficient\n\n![img](https://www.gstatic.com/education/formulas2/553212783/es/spearman_s_rank_correlation_coefficient.svg)\n\n\nThe Spearman rank-order correlation coefficient is a nonparametric measure of the monotonicity of the relationship between two datasets. Like other correlation coefficients, this one varies between -1 and +1 with 0 implying no correlation. Correlations of -1 or +1 imply an exact monotonic relationship. Positive correlations imply that as x increases, so does y. Negative correlations imply that as x increases, y decreases.","metadata":{}},{"cell_type":"code","source":"from scipy.stats import spearmanr\n\ndef spearman_corr(df: pd.DataFrame, target: str) -> pd.DataFrame:\n\n    test_num_list = []\n    test_num_cols = []\n    num_cols = df.select_dtypes(include=['int64', 'float64']).columns.drop(target)\n    \n  \n    valid_data = df[[target] + list(num_cols)].dropna()\n    \n    for num_col in num_cols:\n   \n        correlation, p_value = spearmanr(valid_data[num_col], valid_data[target])\n        \n        test_num_list.append([correlation, np.round(p_value, 5)])\n        test_num_cols.append(num_col)\n    \n\n    return pd.DataFrame(test_num_list, index=test_num_cols, columns=['correlation', 'p_value']).sort_values(by=\"correlation\", ascending=False)\n\nspearman_corr(train, \"Premium Amount\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:16.944029Z","iopub.execute_input":"2024-12-31T02:02:16.944275Z","iopub.status.idle":"2024-12-31T02:02:17.940121Z","shell.execute_reply.started":"2024-12-31T02:02:16.944254Z","shell.execute_reply":"2024-12-31T02:02:17.939367Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The result of test is that there are a couple of numerical features which have a p-value smaller than 0.05. This means that for these features there is a statistical significant relation between them and the target premium amount. However, the correlation coefficient for these features is so small that it is really hard to find out where the differences are. Therefore we continu or analysis with a visual inspection of each numerical feature. \n\n<!--  El resultado de la prueba es que hay un par de características numéricas que tienen un valor p menor que 0,05. Esto significa que para estas características (edad, ingresos anuales, puntuación de salud, reclamaciones anteriores y puntuación crediticia) existe una relación estadísticamente significativa entre ellas y el importe de la prima objetivo. Sin embargo, el coeficiente de correlación para estas características es tan pequeño que resulta realmente difícil averiguar dónde están las diferencias. Por lo tanto, continuamos nuestro análisis con una inspección visual de cada característica numérica.-->","metadata":{}},{"cell_type":"markdown","source":"### Pearson Correlation\n\nhttps://en.wikipedia.org/wiki/Pearson_correlation_coefficient\n\n![img](https://www.gstatic.com/education/formulas2/553212783/es/correlation_coefficient_formula.svg)\n\nPearson correlation is used to measure the linear relationship between numerical features. It helps identify if two variables are positively or negatively related, or if they are independent. This is useful for feature selection, multicollinearity detection, and understanding the strength and direction of relationships in the data.","metadata":{}},{"cell_type":"code","source":"## Getting the correlation of all the features with target variable. \nnum_cols = [col for col in train.columns if train[col].dtype != \"O\"]\n(train[num_cols].corr())[\"Premium Amount\"].sort_values(ascending = False)[1:]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:17.940857Z","iopub.execute_input":"2024-12-31T02:02:17.941058Z","iopub.status.idle":"2024-12-31T02:02:18.3488Z","shell.execute_reply.started":"2024-12-31T02:02:17.941041Z","shell.execute_reply":"2024-12-31T02:02:18.347906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr = train[num_cols].corr()\nsns.heatmap(corr, cmap='viridis', annot=False, linewidth=.5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:18.349678Z","iopub.execute_input":"2024-12-31T02:02:18.349959Z","iopub.status.idle":"2024-12-31T02:02:19.187954Z","shell.execute_reply.started":"2024-12-31T02:02:18.349932Z","shell.execute_reply":"2024-12-31T02:02:19.186968Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Interpret the results\n\n----\n\n### **Findings from the Correlation Chart (Pearson)**:\n1. **Colors in the chart**:\n   - Correlations close to 1 (yellow) indicate a strong positive relationship between the variables.\n   - Correlations close to 0 (purple) indicate a weak or no relationship between the variables.\n   \n2. **Variables with high positive correlation**:\n   - **`Previous Claims`** has a strong positive correlation with **`Insurance Duration`**, suggesting that as the insurance duration increases, previous claims are likely to increase as well.\n   \n3. **Weak or no relationships**:\n   - **`Premium Amount`** has very weak or no correlations with many of the variables, indicating that the relationship with other variables is not strong in terms of Pearson correlation.\n   \n4. **Interesting relationships**:\n   - There are some weak correlations between **`Premium Amount`** and **`Previous Claims`** and **`Health Score`**, but they are very small (close to 0).\n\n### **Findings from Spearman Correlations**:\n1. **Variables with significant relationships (very low p-value)**:\n   - **`Previous Claims`** has a correlation of **0.032741** with **`Premium Amount`**, although its p-value is very low (indicating a statistically significant relationship).\n   - **`Health Score`** also shows a small but significant correlation with **`Premium Amount`** (0.005749), with a very low p-value.\n   - **`Annual Income`** has the lowest correlation of **-0.063164** with a very low p-value, indicating that while the relationship is weak, it is statistically significant.\n   \n2. **Variables with no significant relationship**:\n   - **`Vehicle Age`**, **`Insurance Duration`**, and **`Number of Dependents`** have high p-values (greater than 0.1), suggesting that there is no significant relationship with **`Premium Amount`**.\n\n3. **Variables with negative correlations**:\n   - **`Credit Score`** shows a negative correlation of **-0.027947**, with a very low p-value, indicating that as the credit score decreases, the insurance amount is likely to be lower, but the relationship is weak.\n   - **`Annual Income`** has a weak negative correlation of **-0.063164**, suggesting a slight inverse relationship with the insurance amount, although the relationship is not very strong.\n\n### **Conclusion**:\n- Most variables do not have a strong correlation with **`Premium Amount`**, but some, such as **`Previous Claims`** and **`Health Score`**, have significant but weak positive correlations.\n- Negative correlations with **`Credit Score`** and **`Annual Income`** suggest that these variables have a weak inverse relationship with **`Premium Amount`**.\n- Overall, **`Previous Claims`** appears to be the most relevant variable for predicting **`Premium Amount`**, but it remains weak based on the correlation value.\n\n\n\n<!-- Para interpretar los resultados de las correlaciones de Pearson y Spearman en relación con el gráfico de correlación que has mostrado, aquí tienes algunos hallazgos clave:\n\n### **Hallazgos del gráfico de correlación (Pearson)**:\n1. **Colores en el gráfico**:\n   - Las correlaciones cercanas a 1 (amarillo) indican una relación positiva fuerte entre las variables.\n   - Las correlaciones cercanas a 0 (morado) indican una relación débil o nula entre las variables.\n   \n2. **Variables con alta correlación positiva**:\n   - **`Previous Claims`** tiene una fuerte correlación positiva con **`Insurance Duration`**, lo que sugiere que a medida que aumenta la duración del seguro, es probable que aumenten las reclamaciones anteriores.\n   \n3. **Relaciones débiles o nulas**:\n   - **`Premium Amount`** tiene correlaciones muy débiles o nulas con muchas de las variables, lo que indica que la relación con otras variables no es fuerte en términos de Pearson.\n   \n4. **Relaciones interesantes**:\n   - Hay algunas correlaciones débiles en **`Premium Amount`** con **`Previous Claims`** y **`Health Score`**, pero son muy pequeñas (cerca de 0).\n\n### **Hallazgos de las correlaciones de Spearman**:\n1. **Variables con relaciones significativas (p-valor muy bajo)**:\n   - **`Previous Claims`** tiene una correlación de **0.032741** con **`Premium Amount`**, aunque su p-valor es muy bajo (lo que indica una relación estadísticamente significativa).\n   - **`Health Score`** también muestra una pequeña pero significativa correlación con **`Premium Amount`** (0.005749), con un p-valor muy bajo.\n   - **`Annual Income`** tiene la correlación más baja de **-0.063164** con un p-valor muy bajo, indicando que, aunque la relación es débil, es estadísticamente significativa.\n   \n2. **Variables sin relación significativa**:\n   - **`Vehicle Age`**, **`Insurance Duration`**, y **`Number of Dependents`** tienen p-valores altos (mayores a 0.1), lo que sugiere que no hay una relación significativa con **`Premium Amount`**.\n\n3. **Variables con correlaciones negativas**:\n   - **`Credit Score`** muestra una correlación negativa de **-0.027947**, con un p-valor muy bajo, lo que indica que a medida que disminuye el puntaje de crédito, es probable que el monto del seguro sea más bajo, pero la relación es débil.\n   - **`Annual Income`** tiene una correlación negativa débil de **-0.063164**, lo que indica que hay una ligera relación inversa con el monto del seguro, aunque la relación no es muy fuerte.\n\n### **Conclusión**:\n- La mayoría de las variables no tienen una correlación fuerte con **`Premium Amount`**, pero algunas, como **`Previous Claims`** y **`Health Score`**, tienen correlaciones positivas significativas, aunque débiles.\n- Las correlaciones negativas con **`Credit Score`** y **`Annual Income`** sugieren que estas variables tienen una relación inversa débil con el **`Premium Amount`**.\n- En general, **`Previous Claims`** parece ser la variable más relevante para predecir **`Premium Amount`**, pero sigue siendo débil según el valor de la correlación.\n\nEl análisis sugiere que, para mejorar la predicción de **`Premium Amount`**, podrían ser necesarias otras técnicas o variables adicionales para obtener un modelo más robusto. -->","metadata":{}},{"cell_type":"markdown","source":"# 🔠 Categorical features: Statistical analysis\n\nANOVA:\n\nhttps://en.wikipedia.org/wiki/Analysis_of_variance\n\n![img](https://upload.wikimedia.org/wikipedia/commons/thumb/1/13/Example_of_ANOVA_table.jpg/380px-Example_of_ANOVA_table.jpg)\n\nKruskal-Wallis: \n\nhttps://en.wikipedia.org/wiki/Kruskal%E2%80%93Wallis_test\n\n![img](https://wikimedia.org/api/rest_v1/media/math/render/svg/3999eb7c5d7b3e0bbbd6eedfe37db5efb44deceb)\n\nThe goal is the check if there is a statistical significant relationship between the categories and the target 'premium_amount'. A way we can do this is by using a one-way ANOVA test or the Kruskal-Wallis H-test. The one-way ANOVA tests the null hypothesis that two or more groups have the same population mean. The Kruskal-Wallis H-test tests the null hypothesis that the population median of all of the groups are equal. It is a non-parametric version of ANOVA. Because the ANOVA has some underlying assumptions we use both to check for a statistical significant relaton between the categorical features and the target. See the Scipy documentation for more information on both tests [here](https://docs.scipy.org/doc/scipy/reference/generated/scipy.stats.kruskal.html) and [here](https://docs.scipy.org/doc/scipy/reference/generated/scipy.stats.f_oneway.html). ","metadata":{}},{"cell_type":"code","source":"from scipy.stats import f_oneway\nfrom scipy.stats import kruskal\n\ndef cat_kruskal(train_df: pd.DataFrame, target: str, to_drop: str) -> pd.DataFrame:\n    \n    test_cat_list = []\n    test_cat_cols = []\n    global cat_cols\n    cat_cols = train_df.select_dtypes(include=['object', 'category']).columns.drop(to_drop)\n    \n    for col in cat_cols:\n        test_group = train_df.groupby(col)[target].apply(list)\n        \n        f_oneway_result = f_oneway(*test_group)\n        kruskal_result = kruskal(*test_group)\n    \n        test_cat_list.append(\n            [f_oneway_result.statistic, f_oneway_result.pvalue, kruskal_result.statistic, kruskal_result.pvalue]\n        )\n    \n        test_cat_cols.append(\n            col\n        )\n    \n    return pd.DataFrame(test_cat_list, index=test_cat_cols, columns=['anova_statistic','anova_pvalue','kruskal_statistic','kruskal_pvalue'])\n\ncat_kruskal(train, \"Premium Amount\", \"Policy Start Date\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:19.18873Z","iopub.execute_input":"2024-12-31T02:02:19.188986Z","iopub.status.idle":"2024-12-31T02:02:23.501264Z","shell.execute_reply.started":"2024-12-31T02:02:19.188965Z","shell.execute_reply":"2024-12-31T02:02:23.500548Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Based on the the output of the tests there is only one categorical feature (occupation) for which the H0 hypothesis can be rejected. This means that there is a statistical significant difference between premium amount and someones occupation. Maybe we could put this to our benefit in our feature engineering or modelling process.","metadata":{}},{"cell_type":"code","source":"cat_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:23.502078Z","iopub.execute_input":"2024-12-31T02:02:23.502351Z","iopub.status.idle":"2024-12-31T02:02:23.507345Z","shell.execute_reply.started":"2024-12-31T02:02:23.502318Z","shell.execute_reply":"2024-12-31T02:02:23.506548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef plot_cat(train_df: pd.DataFrame, target: str) -> pd.DataFrame:\n    n_rows = len(cat_cols)\n    n_cols = 2\n    \n    for idx, cat_col in enumerate(cat_cols):\n        \n        fix, ax = plt.subplots(1, n_cols, figsize=(15,5))\n    \n        mean_premium_amount = train_df.groupby(cat_col)[target].mean().reset_index()\n    \n        sns.countplot(\n            data=train_df, x=cat_col, ax=ax[0], palette=color_palette\n        )\n        \n        sns.barplot(\n            data=mean_premium_amount, x=cat_col, y=target, ax=ax[1], palette=color_palette\n        )\n        ax[0].set_title('count of categories')\n        ax[1].set_title('mean premium amount by category')\n    \n    plt.tight_layout()\n\nplot_cat(train, \"Premium Amount\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:23.508322Z","iopub.execute_input":"2024-12-31T02:02:23.508648Z","iopub.status.idle":"2024-12-31T02:02:35.093525Z","shell.execute_reply.started":"2024-12-31T02:02:23.508627Z","shell.execute_reply":"2024-12-31T02:02:35.092755Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As the output from the tests already suggested; there isn't really a big difference between the different categorical features. Visual inspection endorsis this.\n\n-----\n\n<!-- \nComo ya lo indica el resultado de las pruebas, no hay una gran diferencia entre las distintas características categóricas. La inspección visual lo confirma. -->","metadata":{}},{"cell_type":"markdown","source":"# ⏳ Preprocessing \n\n#### 1) Breaking down: 'Policy Start Date'\n\nHow ? for example: apply a 'Sine and cosine transformations'\n\nSine and cosine transformations using π (pi) are applied to convert cyclic features (like year, month, day) into numerical representations that preserve their cyclical nature, which is important in machine learning models. Dates such as **month**, **day**, or **year** are cyclic, meaning the start and end values are closer than the middle ones. For example, **January (month 1)** is close to **December (month 12)**. Using sine and cosine, these cyclic features are mapped to continuous feature space, allowing the model to capture cyclic relationships. \n\n- **`Year_sin` and `Year_cos`**: The **year** itself isn't cyclic, but transforming it this way helps capture temporal trends. The formula maps the year to a circular coordinate system, capturing cyclical patterns.\n  \n- **`Month_sin` and `Month_cos`**: **Month** is cyclic since after **December (12)** comes **January (1)**. Using sine and cosine converts it into a periodic representation with a 12-month cycle.\n\n- **`Day_sin` and `Day_cos`**: The **day** is also cyclic, with a maximum of 31 days. The transformation converts days into a cyclic representation.\n\nSine and cosine ensure that extreme values (e.g., the last day of the month and the first) are represented continuously without breaking the cyclic nature. This avoids treating the dates as linear, which wouldn't properly represent the cyclic structure.\n\n\n#### 2) Binning 'Insurance Duration'\n\n#### 3) Create other features","metadata":{}},{"cell_type":"code","source":"train.tail()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:35.094267Z","iopub.execute_input":"2024-12-31T02:02:35.094529Z","iopub.status.idle":"2024-12-31T02:02:35.113066Z","shell.execute_reply.started":"2024-12-31T02:02:35.094508Z","shell.execute_reply":"2024-12-31T02:02:35.112405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_policy_start_date_separated_features(df):\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'], errors='coerce')\n    df['Policy_Start_Year'] = df['Policy Start Date'].dt.year\n    df['Policy_Start_Month'] = df['Policy Start Date'].dt.month\n    df['Policy_Start_Day'] = df['Policy Start Date'].dt.day\n    \n\n    return df\n\ndef circular_features_time(df, date_column='Policy Start Date'):\n    df[date_column] = pd.to_datetime(df[date_column])\n    \n    df['day_of_week'] = df[date_column].dt.dayofweek\n    df['day_sin'] = np.sin(2 * np.pi * df['day_of_week'] / 7)\n    df['day_cos'] = np.cos(2 * np.pi * df['day_of_week'] / 7)\n    \n    df['day_of_year'] = df[date_column].dt.dayofyear\n    df['year_day_sin'] = np.sin(2 * np.pi * df['day_of_year'] / 365)\n    df['year_day_cos'] = np.cos(2 * np.pi * df['day_of_year'] / 365)\n    \n    df['month'] = df[date_column].dt.month\n    df['month_sin'] = np.sin(2 * np.pi * df['month'] / 12)\n    df['month_cos'] = np.cos(2 * np.pi * df['month'] / 12)\n    \n    df[\"seconds_since_1970\"] = df[date_column].astype(\"int64\") // 10**9\n    df = df.drop('Policy Start Date', axis=1)\n\n    df['Annual Income'] = np.log1p(df['Annual Income'])\n\n    return df\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:35.113876Z","iopub.execute_input":"2024-12-31T02:02:35.114133Z","iopub.status.idle":"2024-12-31T02:02:35.129857Z","shell.execute_reply.started":"2024-12-31T02:02:35.114113Z","shell.execute_reply":"2024-12-31T02:02:35.129099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"full = create_policy_start_date_separated_features(full)\n\nfull = circular_features_time(full)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:35.130581Z","iopub.execute_input":"2024-12-31T02:02:35.130841Z","iopub.status.idle":"2024-12-31T02:02:37.277938Z","shell.execute_reply.started":"2024-12-31T02:02:35.130813Z","shell.execute_reply":"2024-12-31T02:02:37.276976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def impute_missing_numerical_data(df):\n    num = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score',\n       'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']\n    for n in num:\n        df[f\"is_{n}_na\"] = df[n].isna()\n        df[n] = df[n].fillna(df[n].mean())\n\n    return df\n\ndef impute_missing_categorical_data(df):\n    cat = ['Marital Status', 'Occupation', 'Customer Feedback']\n    for c in cat:\n        df[f\"is_{c}_na\"] = df[c].isna()\n        df[c] = df[c].fillna(df[c].mode())\n\n    return df\n\nfull = impute_missing_categorical_data(full)\n\nfull = impute_missing_numerical_data(full)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:37.278801Z","iopub.execute_input":"2024-12-31T02:02:37.279116Z","iopub.status.idle":"2024-12-31T02:02:38.642261Z","shell.execute_reply.started":"2024-12-31T02:02:37.279087Z","shell.execute_reply":"2024-12-31T02:02:38.641611Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Scaling numerical features","metadata":{}},{"cell_type":"code","source":"\n\nnum_cols = full.select_dtypes(exclude=['object', 'datetime', 'bool']).columns.tolist()\n\ncat_cols = full.select_dtypes(include='object').columns.tolist()\n\nscaler = StandardScaler()\n\nfull[num_cols] = scaler.fit_transform(full[num_cols])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:38.6457Z","iopub.execute_input":"2024-12-31T02:02:38.645925Z","iopub.status.idle":"2024-12-31T02:02:40.794619Z","shell.execute_reply.started":"2024-12-31T02:02:38.645906Z","shell.execute_reply":"2024-12-31T02:02:40.793938Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Encoding","metadata":{}},{"cell_type":"code","source":"def encode_ordinal(df):\n    educ = {\"High School\":0, \"Bachelor's\":1, \"Master's\":2, \"PhD\":3}\n    policy = {'Basic':0, 'Comprehensive':1, 'Premium':2}\n    exerc = {'Rarely':0, 'Daily':1, 'Weekly':2, 'Monthly': 3}\n    feedback = {'Poor':0, 'Average':1, 'Good':2, \"Unknown\": 0}\n\n    df['Education Level'] = df['Education Level'].map(educ)\n    df['Policy Type'] = df['Policy Type'].map(policy)\n    df['Exercise Frequency'] = df['Exercise Frequency'].map(exerc)\n    df['Customer Feedback'] = df['Customer Feedback'].map(feedback)\n    return df\n\ndef encode_binary(df):\n    df['Gender'] = df['Gender'].map({'Male':0, 'Female':1})\n    df['Smoking Status'] = df['Smoking Status'].map({'Yes':1, 'No':0})\n    return df\n\n\ndef one_hot_dummies(df, categorical):\n    oh = pd.get_dummies(df[categorical])\n    df = df.drop(categorical, axis=1)\n    return pd.concat([df, oh], axis=1)\n\n\nfull = encode_binary(full)\nfull = encode_ordinal(full)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:40.795632Z","iopub.execute_input":"2024-12-31T02:02:40.795834Z","iopub.status.idle":"2024-12-31T02:02:41.547182Z","shell.execute_reply.started":"2024-12-31T02:02:40.795817Z","shell.execute_reply":"2024-12-31T02:02:41.546524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_features_total = full.select_dtypes(exclude='object').columns\nnumerical_features_continuos = full.select_dtypes(exclude=['object', 'int']).columns\nnumerical_features_discrete = full.select_dtypes(exclude=['object', 'float']).columns\ncategorical_features = full.select_dtypes(include='object').columns\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:41.547878Z","iopub.execute_input":"2024-12-31T02:02:41.548087Z","iopub.status.idle":"2024-12-31T02:02:43.049735Z","shell.execute_reply.started":"2024-12-31T02:02:41.548069Z","shell.execute_reply":"2024-12-31T02:02:43.048808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"full = one_hot_dummies(full, categorical_features)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:43.050581Z","iopub.execute_input":"2024-12-31T02:02:43.050825Z","iopub.status.idle":"2024-12-31T02:02:45.103192Z","shell.execute_reply.started":"2024-12-31T02:02:43.050804Z","shell.execute_reply":"2024-12-31T02:02:45.102281Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### New features","metadata":{}},{"cell_type":"code","source":"def add_new_features(df):\n    df['Income to Dependents Ratio'] = df['Annual Income'] / (df['Number of Dependents'].fillna(0) + 1)\n    df['Income_per_Dependent'] = df['Annual Income'] / (df['Number of Dependents'] + 1)\n    df['CreditScore_InsuranceDuration'] = df['Credit Score'] * df['Insurance Duration']\n    df['Health_Risk_Score'] = df['Smoking Status'].apply(lambda x: 1 if x == 'Smoker' else 0) + \\\n                                df['Exercise Frequency'].apply(lambda x: 1 if x == 'Low' else (0.5 if x == 'Medium' else 0)) + \\\n                                (100 - df['Health Score']) / 20\n    df['Credit_Health_Score'] = df['Credit Score'] * df['Health Score']\n    df['Health_Age_Interaction'] = df['Health Score'] * df['Age']\n    return df\n\nfull = add_new_features(full)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:45.104173Z","iopub.execute_input":"2024-12-31T02:02:45.104403Z","iopub.status.idle":"2024-12-31T02:02:46.510121Z","shell.execute_reply.started":"2024-12-31T02:02:45.104385Z","shell.execute_reply":"2024-12-31T02:02:46.50947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1478860\nX_size = X.shape[0]\n\nX_FE_train = full[:X_size]\nX_FE_test = full[X_size:]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:46.510881Z","iopub.execute_input":"2024-12-31T02:02:46.511097Z","iopub.status.idle":"2024-12-31T02:02:46.514796Z","shell.execute_reply.started":"2024-12-31T02:02:46.511078Z","shell.execute_reply":"2024-12-31T02:02:46.514136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X_FE_train.shape[0])\nprint(y.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:46.51564Z","iopub.execute_input":"2024-12-31T02:02:46.515868Z","iopub.status.idle":"2024-12-31T02:02:46.529458Z","shell.execute_reply.started":"2024-12-31T02:02:46.51584Z","shell.execute_reply":"2024-12-31T02:02:46.528792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_log = np.log1p(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:46.530149Z","iopub.execute_input":"2024-12-31T02:02:46.530479Z","iopub.status.idle":"2024-12-31T02:02:46.547291Z","shell.execute_reply.started":"2024-12-31T02:02:46.530432Z","shell.execute_reply":"2024-12-31T02:02:46.546657Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🌐 Modeling","metadata":{}},{"cell_type":"code","source":"def rmsle(y_true, y_pred):\n    \n    log_true = np.log1p(y_true)  # log(1 + y_true)\n    log_pred = np.log1p(y_pred)  # log(1 + y_pred)\n\n    return np.sqrt(mean_squared_error(log_true, log_pred))\n\n#experiment with feauture selection (Annual Income, Policy Type, Health Score)\ndef select_important_features(df, columns = ['Annual Income', 'Policy Type', 'Health Score']):\n    return df[columns]\n\nx_train, x_val, y_train, y_val = train_test_split(X_FE_train, y_log, test_size=0.11, random_state=42)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:02:46.548063Z","iopub.execute_input":"2024-12-31T02:02:46.548323Z","iopub.status.idle":"2024-12-31T02:02:47.115188Z","shell.execute_reply.started":"2024-12-31T02:02:46.548303Z","shell.execute_reply":"2024-12-31T02:02:47.11452Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Optuna ","metadata":{}},{"cell_type":"markdown","source":"# XGBoost","metadata":{}},{"cell_type":"code","source":"import optuna\nfrom sklearn.ensemble import HistGradientBoostingRegressor as hgbc\nfrom sklearn.metrics import make_scorer\nimport xgboost as xgb\n\ndef rmsle_on_original_scale(y_true, y_pred):\n    y_true = np.expm1(y_true)\n    y_pred = np.expm1(y_pred)\n    \n    y_pred = np.maximum(y_pred, np.min(y_true))\n    \n    log_true = np.log1p(y_true)\n    log_pred = np.log1p(y_pred)\n    return np.sqrt(np.mean((log_true - log_pred) ** 2))\n\nrmsle_scorer = make_scorer(rmsle_on_original_scale, greater_is_better=False)\n\n\n# def objective_xgb(trial):\n    # params = {\n    #     'n_estimators': trial.suggest_int('n_estimators', 20, 500),\n    #     'max_depth': int(trial.suggest_float('max_depth', 1, 100, log=True)),\n    #     'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n    #     'booster': trial.suggest_categorical('booster', ['gbtree', 'gblinear', 'dart']),\n    #     'gamma': trial.suggest_float('gamma', 0, 2),\n    #     'max_delta_step': trial.suggest_float('max_delta_step', 0, 10),\n    #     'subsample': trial.suggest_float('subsample', 0, 1)\n    # }\n\n    # params = {        \n    #         \"n_estimators\": trial.suggest_int(\"n_estimators\", 50, 1000, step=100),\n    #         \"max_depth\":trial.suggest_int(\"max_depth\", 4, 10),\n    #         \"min_child_weight\": trial.suggest_int(\"min_child_weight\", 7, 8),\n    #         \"learning_rate\": trial.suggest_float(\"learning_rate\", 1e-4, 1e-1, log=True), \n    #         \"subsample\": trial.suggest_float(\"subsample\", 0.7, 1.0),\n    #         \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.1, 1.0),\n    #         \"reg_alpha\": trial.suggest_float(\"reg_alpha\", 1e-2, 10.),\n    #         \"reg_lambda\": trial.suggest_float(\"reg_lambda\", 1e-2, 10.),\n    #         \"gamma\": trial.suggest_float(\"gamma\", 0.7, 1.0, step=0.1),\n    # }    \n    \n    # xgb_model = xgb.XGBRegressor(device='gpu', objective = 'reg:squarederror', tree_method='gpu_hist', **params)\n\n    # return -1*cross_val_score(xgb_model, x_train, y_train, n_jobs=-1, cv=3, scoring=rmsle_scorer).mean()\n\n# --- START\n    \n# study = optuna.create_study(direction='minimize')\n# study.optimize(objective_xgb, n_trials=30)\n# trial = study.best_trial\n# print('Accuracy: {}'.format(trial.value))\n\n# print(\"Best hyperparameters: {}\".format(trial.params))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:03:04.53531Z","iopub.execute_input":"2024-12-31T02:03:04.535611Z","iopub.status.idle":"2024-12-31T02:03:04.953131Z","shell.execute_reply.started":"2024-12-31T02:03:04.535586Z","shell.execute_reply":"2024-12-31T02:03:04.952426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {'n_estimators': 2000, 'max_depth':20, 'min_child_weight':19, 'learning_rate': 0.029157092015323985, 'subsample': 0.8137283875432285, 'colsample_bytree': 0.9983658272411838, 'reg_alpha': 6.517426484808757, 'reg_lambda': 0.8771067924995002, 'gamma': 0.76}\n# params = study.best_trial.params\n# xgb_model = xgb.XGBRegressor(verbosity = 0, device='gpu', objective = 'reg:squarederror', tree_method='gpu_hist', **params)\n# xgb_model.fit(x_train, y_train, eval_set=[(x_val, y_val)])\n\n\n# Crear el modelo XGBoost\nxgb_model = xgb.XGBRegressor(\n    device='gpu',\n    objective='reg:squarederror',\n    tree_method='gpu_hist',\n    \n    **params\n)\n\n# Entrenar el modelo con early stopping\nxgb_model.fit(\n    x_train, \n    y_train, \n    eval_set=[(x_val, y_val)],         # Conjunto de validación a fin de comparar\n    # eval_set=[(x_train, y_train), (x_val, y_val)],  # Conjunto de entrenamiento y validación\n    early_stopping_rounds=20,         # Detener si no mejora en 50 iteraciones\n    eval_metric='rmse',               # Métrica de evaluación: RMSE\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:03:55.903313Z","iopub.execute_input":"2024-12-31T02:03:55.903672Z","iopub.status.idle":"2024-12-31T02:04:46.482889Z","shell.execute_reply.started":"2024-12-31T02:03:55.903643Z","shell.execute_reply":"2024-12-31T02:04:46.481996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(results.keys())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:06:50.89782Z","iopub.execute_input":"2024-12-31T02:06:50.898269Z","iopub.status.idle":"2024-12-31T02:06:50.902973Z","shell.execute_reply.started":"2024-12-31T02:06:50.898239Z","shell.execute_reply":"2024-12-31T02:06:50.902149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Obtener resultados de evaluación\nresults = xgb_model.evals_result_\n\n# Graficar la curva de entrenamiento\nplt.figure(figsize=(10, 6))\nplt.plot(results['validation_0']['rmse'], label='Entrenamiento')\n# plt.plot(results['validation_1']['rmse'], label='Validación')\nplt.xlabel('Iteraciones')\nplt.ylabel('RMSE')\nplt.title('Curva de Entrenamiento')\nplt.legend()\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:09:59.589179Z","iopub.execute_input":"2024-12-31T02:09:59.589544Z","iopub.status.idle":"2024-12-31T02:09:59.828118Z","shell.execute_reply.started":"2024-12-31T02:09:59.589514Z","shell.execute_reply":"2024-12-31T02:09:59.827267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predicciones = xgb_model.predict(X_FE_test)\npredicciones","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:05:09.33795Z","iopub.execute_input":"2024-12-31T02:05:09.33829Z","iopub.status.idle":"2024-12-31T02:05:12.772215Z","shell.execute_reply.started":"2024-12-31T02:05:09.338259Z","shell.execute_reply":"2024-12-31T02:05:12.771509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.metrics import mean_absolute_percentage_error\n    \n# import numpy as np\n# import xgboost as xgb\n# from sklearn.metrics import mean_absolute_percentage_error\n\n# # Suponiendo que tienes los datos de prueba\n# # x_val: Características de validación\n# # y_val: Valores verdaderos de validación\n\n# # Predicciones con la media\n# mean_pred = np.full_like(y_val, np.mean(y_val), dtype=np.float64)\n\n# # Predicciones con la mediana\n# median_pred = np.full_like(y_val, np.median(y_val), dtype=np.float64)\n\n# # Predicciones con XGBoost (suponiendo que tienes el modelo entrenado 'xgb_model')\n# xgb_pred = xgb_model.predict(x_val)\n\n\n# # Calculando MAPE para cada modelo\n# mape_mean = mean_absolute_percentage_error(y_val, mean_pred)\n# mape_median = mean_absolute_percentage_error(y_val, median_pred)\n# mape_xgb = mean_absolute_percentage_error(y_val, xgb_pred)\n\n# # Imprimir resultados\n# print(f\"MAPE con la media: {mape_mean:.2f}%\")\n# print(f\"MAPE con la mediana: {mape_median:.2f}%\")\n# print(f\"MAPE con XGBoost: {mape_xgb:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:37:27.200367Z","iopub.execute_input":"2024-12-31T02:37:27.200737Z","iopub.status.idle":"2024-12-31T02:37:27.204392Z","shell.execute_reply.started":"2024-12-31T02:37:27.200709Z","shell.execute_reply":"2024-12-31T02:37:27.20351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dffinal = pd.DataFrame()\n\ndffinal[\"id\"] = test_id\ndffinal[\"Premium Amount\"] = np.expm1(predicciones)\n\ndffinal","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:20:11.142272Z","iopub.execute_input":"2024-12-31T02:20:11.142617Z","iopub.status.idle":"2024-12-31T02:20:11.169279Z","shell.execute_reply.started":"2024-12-31T02:20:11.142591Z","shell.execute_reply":"2024-12-31T02:20:11.168487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dffinal[\"Premium Amount\"].hist(bins = 100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:21:05.005272Z","iopub.execute_input":"2024-12-31T02:21:05.00559Z","iopub.status.idle":"2024-12-31T02:21:05.391363Z","shell.execute_reply.started":"2024-12-31T02:21:05.005565Z","shell.execute_reply":"2024-12-31T02:21:05.390469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dffinal.to_csv(\"submission.csv\", index=False, sep=',')  # Si el delimitador es coma","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:36:21.560662Z","iopub.execute_input":"2024-12-31T02:36:21.560987Z","iopub.status.idle":"2024-12-31T02:36:22.553462Z","shell.execute_reply.started":"2024-12-31T02:36:21.560958Z","shell.execute_reply":"2024-12-31T02:36:22.552375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import joblib\n\n# Suponiendo que 'pipeline' es tu modelo entrenado (puede ser un modelo o una pipeline)\njoblib.dump(xgb_model, 'XGBOOST Insurance1.pkl')\n\n# Cargar el modelo guardado\nmodel_loaded = joblib.load('XGBOOST Insurance1.pkl')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T02:31:29.360235Z","iopub.execute_input":"2024-12-31T02:31:29.360573Z","iopub.status.idle":"2024-12-31T02:31:29.675499Z","shell.execute_reply.started":"2024-12-31T02:31:29.360546Z","shell.execute_reply":"2024-12-31T02:31:29.674788Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Interpreting XGBoost\n\n----\n\n### What is XGBoost?\n\n**XGBoost** (Extreme Gradient Boosting) is a highly efficient, scalable machine learning algorithm primarily used for supervised learning tasks like classification and regression. It is based on **gradient boosting**, where models are built in a sequential manner, each new model correcting the errors made by the previous one. XGBoost enhances this process by implementing optimizations such as:\n\n- **Regularization** to prevent overfitting.\n- **Handling missing values** automatically.\n- Efficient **parallelization** for faster computation.\n- **Tree Pruning** to avoid overcomplex trees.\n\nXGBoost is popular due to its high performance and ability to work well with structured/tabular data, making it a go-to model for Kaggle competitions and real-world applications.\n\n### Interpreting XGBoost with SHAP\n\nSHAP (SHapley Additive exPlanations) is a powerful tool to interpret machine learning models, including XGBoost. It calculates feature importance and the impact each feature has on individual predictions. Here's a brief overview of how to use SHAP for interpreting XGBoost:\n\n\nBy using SHAP, you can understand the relationship between features and predictions, making your XGBoost model more interpretable and trustworthy.","metadata":{}},{"cell_type":"code","source":"# import shap\n\n# # Create SHAP TreeExplainer for XGBoost model\n# explainer = shap.TreeExplainer(xgb_model)\n\n# # Get SHAP values for the test set\n# shap_values = explainer.shap_values(X_FE_test)\n\n# # Visualize SHAP summary plot (global interpretability)\n# shap.summary_plot(shap_values, X_FE_test)\n\n# # Visualize SHAP dependence plot (local interpretability)\n# shap.dependence_plot(\"feature_name\", shap_values, X_FE_test)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2024-12-31T02:49:33.374Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Deploying the model \n\nhttps://www.ibm.com/docs/en/oapi/1.3.6?topic=databases-deploying-model\n\nhttps://moez-62905.medium.com/easily-deploy-machine-learning-models-from-the-comfort-of-your-notebook-9068a88f4cf5\n\nhttps://www.geeksforgeeks.org/deploy-a-machine-learning-model-using-streamlit-library/\n\n\nDeploying a model to cloud services like **AWS (Amazon Web Services)** or **Google Cloud Platform (GCP)** involves several steps:\n\n### 1. **AWS Deployment**\nAWS offers several services to deploy machine learning models, including **Amazon SageMaker**, **AWS Lambda**, and **Elastic Beanstalk**.\n\n#### Using **Amazon SageMaker**:\nAmazon SageMaker is a fully managed service for building, training, and deploying machine learning models. Here's how to deploy a model on SageMaker:\n\n- **Step 1: Prepare Your Model**\n   Train your model locally (e.g., using XGBoost) and save it to a file, such as `model.pkl`.\n\n- **Step 2: Upload Your Model to S3**\n   Upload the saved model to an S3 bucket for easy access by SageMaker.\n   ```python\n   import boto3\n   s3 = boto3.client('s3')\n   s3.upload_file('model.pkl', 'my-bucket', 'model/model.pkl')\n   ```\n\n- **Step 3: Create a SageMaker Model**\n   In SageMaker, you need to create a model object to link your model and the container for deployment.\n   ```python\n   import sagemaker\n   from sagemaker import get_execution_role\n   from sagemaker.xgboost import XGBoostModel\n\n   role = get_execution_role()\n\n   model = XGBoostModel(\n       model_data='s3://my-bucket/model/model.pkl',\n       role=role,\n       entry_point='inference.py',  # The inference script\n   )\n   ```\n\n- **Step 4: Deploy the Model**\n   You can deploy the model to an endpoint, where it can be accessed for predictions.\n   ```python\n   predictor = model.deploy(\n       initial_instance_count=1,\n       instance_type='ml.m4.xlarge'\n   )\n   ```\n\n- **Step 5: Get Predictions**\n   Once deployed, you can invoke the model endpoint for predictions.\n   ```python\n   result = predictor.predict(data)\n   print(result)\n   ```\n\n- **Step 6: Clean Up**\n   After using the endpoint, don’t forget to delete it to avoid additional charges.\n   ```python\n   predictor.delete_endpoint()\n   ```\n\n#### Using **AWS Lambda**:\nAWS Lambda can be used to deploy smaller models in a serverless fashion. You need to package your model (e.g., `model.pkl`) and inference code into a Lambda function, and then trigger predictions via HTTP requests using API Gateway.\n\n### 2. **Google Cloud Deployment**\nGoogle Cloud provides several tools for deploying models, with **AI Platform (Vertex AI)** being one of the most common services used for model deployment.\n\n#### Using **Vertex AI (formerly AI Platform)**:\nHere's how you can deploy a model on Google Cloud’s Vertex AI:\n\n- **Step 1: Prepare Your Model**\n   As with AWS, you train and save your model (e.g., `model.pkl`).\n\n- **Step 2: Upload the Model to Google Cloud Storage (GCS)**\n   First, upload your model to a Google Cloud Storage bucket.\n   ```bash\n   gsutil cp model.pkl gs://your-bucket-name/model/model.pkl\n   ```\n\n- **Step 3: Deploy the Model Using Vertex AI**\n   Use Google Cloud’s Vertex AI to deploy the model. You can use the `gcloud` CLI or the Python client to create a model resource and then deploy it.\n\n   - Create a model resource in Vertex AI:\n     ```bash\n     gcloud ai models upload \\\n       --region=us-central1 \\\n       --display-name=my-xgboost-model \\\n       --artifact-uri=gs://your-bucket-name/model/model.pkl\n     ```\n\n   - Deploy the model to an endpoint:\n     ```bash\n     gcloud ai endpoints create --region=us-central1 --display-name=my-xgboost-endpoint\n     ```\n\n- **Step 4: Get Predictions**\n   You can send data to the deployed model using the endpoint for predictions, either through HTTP API requests or using the Python client.\n\n- **Step 5: Clean Up**\n   Like AWS, make sure to delete the endpoint once you're done to avoid unnecessary costs:\n   ```bash\n   gcloud ai endpoints delete --region=us-central1 --endpoint=YOUR_ENDPOINT_ID\n   ```\n\n### Choosing AWS vs Google Cloud:\n- **AWS**: Ideal if you're already using other AWS services or if you're looking for a fully managed solution (SageMaker).\n- **Google Cloud**: Google Cloud's Vertex AI is a great choice for those who want to integrate with Google’s other machine learning tools and have excellent support for TensorFlow and other machine learning frameworks.\n\n### Conclusion:\nBoth AWS and Google Cloud offer scalable and efficient platforms to deploy machine learning models. The choice between AWS and Google Cloud depends on your existing infrastructure, preferences, and specific use case. **SageMaker** on AWS and **Vertex AI** on Google Cloud are two powerful tools to streamline the deployment and management of your machine learning models.\n","metadata":{}}]}