{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:40:53.004365Z","iopub.execute_input":"2024-12-08T08:40:53.004646Z","iopub.status.idle":"2024-12-08T08:40:54.389519Z","shell.execute_reply.started":"2024-12-08T08:40:53.004617Z","shell.execute_reply":"2024-12-08T08:40:54.388474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# File paths\ntrain_path = '/kaggle/input/playground-series-s4e12/train.csv'\ntest_path = '/kaggle/input/playground-series-s4e12/test.csv'\nsubmission_path = '/kaggle/input/playground-series-s4e12/sample_submission.csv'\n\n# Creating DataFrames\ntrain_df = pd.read_csv(train_path)\ntest_df = pd.read_csv(test_path)\nsubmission_df = pd.read_csv(submission_path)\n\n# Displaying the first few rows of each DataFrame\nprint(\"Train DataFrame:\")\nprint(train_df.head(), \"\\n\")\n\nprint(\"Test DataFrame:\")\nprint(test_df.head(), \"\\n\")\n\nprint(\"Sample Submission DataFrame:\")\nprint(submission_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:40:54.391245Z","iopub.execute_input":"2024-12-08T08:40:54.39173Z","iopub.status.idle":"2024-12-08T08:41:03.073929Z","shell.execute_reply.started":"2024-12-08T08:40:54.391691Z","shell.execute_reply":"2024-12-08T08:41:03.073044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# General information\nprint(train_df.info())\nprint(test_df.info())\n\n# Summary statistics\nprint(train_df.describe(include='all'))\nprint(test_df.describe(include='all'))\n\n# Check for missing values\nprint(\"Missing values in Train DataFrame:\")\nprint(train_df.isnull().sum())\n\nprint(\"\\nMissing values in Test DataFrame:\")\nprint(test_df.isnull().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:41:03.075206Z","iopub.execute_input":"2024-12-08T08:41:03.075622Z","iopub.status.idle":"2024-12-08T08:41:08.458886Z","shell.execute_reply.started":"2024-12-08T08:41:03.075579Z","shell.execute_reply":"2024-12-08T08:41:08.457881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Visualizing missing values\nplt.figure(figsize=(12, 6))\nsns.heatmap(train_df.isnull(), cbar=False, cmap='viridis')\nplt.title('Missing Values in Train Data')\nplt.show()\n\nplt.figure(figsize=(12, 6))\nsns.heatmap(test_df.isnull(), cbar=False, cmap='viridis')\nplt.title('Missing Values in Test Data')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:41:08.461601Z","iopub.execute_input":"2024-12-08T08:41:08.462381Z","iopub.status.idle":"2024-12-08T08:41:43.033294Z","shell.execute_reply.started":"2024-12-08T08:41:08.462336Z","shell.execute_reply":"2024-12-08T08:41:43.032394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#ver2\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.experimental import enable_iterative_imputer  # noqa\nfrom sklearn.impute import IterativeImputer\nimport pandas as pd\nimport numpy as np\n\ndef impute_and_encode(train_df, test_df):\n    target_column = 'Premium Amount'\n    if target_column in train_df.columns:\n        target_train = train_df[target_column]\n        train_df = train_df.drop(columns=[target_column])  # Remove target from features\n    else:\n        target_train = None\n    \n    # Handle datetime columns consistently\n    if 'Policy Start Date' in train_df.columns:\n        # Convert 'Policy Start Date' to datetime if it's not already\n        train_df['Policy Start Date'] = pd.to_datetime(train_df['Policy Start Date'], errors='coerce')\n        test_df['Policy Start Date'] = pd.to_datetime(test_df['Policy Start Date'], errors='coerce')\n        \n        # Extract year, month, day, and day of week from the timestamp\n        train_df['Year'] = train_df['Policy Start Date'].dt.year\n        train_df['Month'] = train_df['Policy Start Date'].dt.month\n        train_df['Day'] = train_df['Policy Start Date'].dt.day\n        train_df['DayOfWeek'] = train_df['Policy Start Date'].dt.dayofweek\n\n        test_df['Year'] = test_df['Policy Start Date'].dt.year\n        test_df['Month'] = test_df['Policy Start Date'].dt.month\n        test_df['Day'] = test_df['Policy Start Date'].dt.day\n        test_df['DayOfWeek'] = test_df['Policy Start Date'].dt.dayofweek\n        \n        # Drop the 'Policy Start Date' column\n        train_df = train_df.drop(columns=['Policy Start Date'])\n        test_df = test_df.drop(columns=['Policy Start Date'])\n    \n    # Handle categorical columns and label encoding\n    categorical_columns = [col for col in train_df.columns if train_df[col].dtype == 'object']\n    label_encoder = LabelEncoder()\n    for col in categorical_columns:\n        train_df[col] = label_encoder.fit_transform(train_df[col])\n        test_df[col] = label_encoder.transform(test_df[col])  # Transform test set using train encoder\n\n    # Handle numeric columns for imputation\n    numeric_columns = [col for col in train_df.columns if train_df[col].dtype in ['int64', 'float64']]\n    \n    # Combine train and test numeric columns for imputation\n    combined_numeric_df = pd.concat([train_df[numeric_columns], test_df[numeric_columns]], axis=0)\n    \n    # Initialize and apply IterativeImputer\n    mice_imputer = IterativeImputer(max_iter=10, random_state=42)\n    imputed_data = mice_imputer.fit_transform(combined_numeric_df)\n\n    # Separate the imputed data into train and test sets\n    train_df_imputed2 = imputed_data[:len(train_df), :]\n    test_df_imputed2 = imputed_data[len(train_df):, :]\n\n    # Convert the imputed data to DataFrame\n    train_df_imputed2 = pd.DataFrame(train_df_imputed2, columns=numeric_columns)\n    test_df_imputed2 = pd.DataFrame(test_df_imputed2, columns=numeric_columns)\n    \n    # Reattach categorical columns from the original datasets (without duplicates)\n    train_df_imputed2 = pd.concat([train_df_imputed2, train_df[categorical_columns].reset_index(drop=True)], axis=1)\n    test_df_imputed2 = pd.concat([test_df_imputed2, test_df[categorical_columns].reset_index(drop=True)], axis=1)\n\n    # Add the date-related features (ensure no duplicates by removing old columns)\n    train_df_imputed2['Year'] = train_df['Year'].reset_index(drop=True)\n    train_df_imputed2['Month'] = train_df['Month'].reset_index(drop=True)\n    train_df_imputed2['Day'] = train_df['Day'].reset_index(drop=True)\n    train_df_imputed2['DayOfWeek'] = train_df['DayOfWeek'].reset_index(drop=True)\n    \n    test_df_imputed2['Year'] = test_df['Year'].reset_index(drop=True)\n    test_df_imputed2['Month'] = test_df['Month'].reset_index(drop=True)\n    test_df_imputed2['Day'] = test_df['Day'].reset_index(drop=True)\n    test_df_imputed2['DayOfWeek'] = test_df['DayOfWeek'].reset_index(drop=True)\n\n    # Add the target back to train_df_imputed if it was removed\n    if target_train is not None:\n        train_df_imputed2[target_column] = target_train\n\n    # Ensure no duplicated columns\n    train_df_imputed2 = train_df_imputed2.loc[:, ~train_df_imputed2.columns.duplicated()]\n    test_df_imputed2 = test_df_imputed2.loc[:, ~test_df_imputed2.columns.duplicated()]\n\n    return train_df_imputed2, test_df_imputed2\n\n# Example usage:\ntrain_df_imputed2, test_df_imputed2 = impute_and_encode(train_df, test_df)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:41:43.034254Z","iopub.execute_input":"2024-12-08T08:41:43.034623Z","iopub.status.idle":"2024-12-08T08:43:56.794646Z","shell.execute_reply.started":"2024-12-08T08:41:43.034596Z","shell.execute_reply":"2024-12-08T08:43:56.793868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Visualizing missing values\nplt.figure(figsize=(12, 6))\nsns.heatmap(train_df_imputed2.isnull(), cbar=False, cmap='viridis')\nplt.title('Missing Values in Train Data imputed')\nplt.show()\n\nplt.figure(figsize=(12, 6))\nsns.heatmap(test_df_imputed2.isnull(), cbar=False, cmap='viridis')\nplt.title('Missing Values in Test Data imputed')\nplt.show()\n# Check for missing values\nprint(\"Missing values in Train DataFrame:\")\nprint(train_df_imputed2.isnull().sum())\n\nprint(\"\\nMissing values in Test DataFrame:\")\nprint(test_df_imputed2.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:43:56.796021Z","iopub.execute_input":"2024-12-08T08:43:56.796296Z","iopub.status.idle":"2024-12-08T08:44:33.543113Z","shell.execute_reply.started":"2024-12-08T08:43:56.79627Z","shell.execute_reply":"2024-12-08T08:44:33.542056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom scipy.stats import skew, kurtosis, shapiro, probplot\n\n# Load your dataset\n# Assuming `train_df_imputed2` contains the target variable \"Premium Amount\"\ntarget = train_df_imputed2[\"Premium Amount\"]\n\n# 1. Plot Histogram\nplt.figure(figsize=(10, 5))\nsns.histplot(target, kde=True, bins=30, color='blue')\nplt.title('Histogram of Premium Amount')\nplt.xlabel('Premium Amount')\nplt.ylabel('Frequency')\nplt.show()\n\n# 2. Q-Q Plot\nplt.figure(figsize=(6, 6))\nprobplot(target, dist=\"norm\", plot=plt)\nplt.title('Q-Q Plot for Premium Amount')\nplt.show()\n\n# 3. Descriptive Statistics\nmean = np.mean(target)\nvariance = np.var(target)\nskewness = skew(target)\nkurt = kurtosis(target)\n\nprint(f\"Mean: {mean}\")\nprint(f\"Variance: {variance}\")\nprint(f\"Skewness: {skewness}\")\nprint(f\"Kurtosis: {kurt}\")\n\n# 4. Normality Test (Shapiro-Wilk)\nstat, p_value = shapiro(target)\nif p_value > 0.05:\n    print(\"Shapiro-Wilk Test: Data appears to be normally distributed (p > 0.05).\")\nelse:\n    print(\"Shapiro-Wilk Test: Data is not normally distributed (p <= 0.05).\")\n\n# 5. Box Plot\nplt.figure(figsize=(10, 5))\nsns.boxplot(x=target, color='orange')\nplt.title('Box Plot of Premium Amount')\nplt.xlabel('Premium Amount')\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:44:33.544143Z","iopub.execute_input":"2024-12-08T08:44:33.54442Z","iopub.status.idle":"2024-12-08T08:44:41.151767Z","shell.execute_reply.started":"2024-12-08T08:44:33.544394Z","shell.execute_reply":"2024-12-08T08:44:41.150994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import boxcox\n\n# Box-Cox Transformation (assuming all values are positive)\ntarget_positive = target[target > 0]  # Ensure no non-positive values\ntarget_boxcox, _ = boxcox(target_positive)\n\n# Plot Box-Cox transformed target\nplt.figure(figsize=(10, 5))\nsns.histplot(target_boxcox, kde=True, bins=30, color='purple')\nplt.title('Histogram of Box-Cox Transformed Premium Amount')\nplt.xlabel('Box-Cox Transformed Premium Amount')\nplt.ylabel('Frequency')\nplt.show()\n\n# Recalculate Skewness and Kurtosis for Box-Cox Transformed Data\nboxcox_skewness = skew(target_boxcox)\nboxcox_kurt = kurtosis(target_boxcox)\n\nprint(f\"Box-Cox Skewness: {boxcox_skewness}\")\nprint(f\"Box-Cox Kurtosis: {boxcox_kurt}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:44:41.15269Z","iopub.execute_input":"2024-12-08T08:44:41.152959Z","iopub.status.idle":"2024-12-08T08:44:50.820326Z","shell.execute_reply.started":"2024-12-08T08:44:41.152933Z","shell.execute_reply":"2024-12-08T08:44:50.819421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom scipy.stats import probplot\n\n# Assuming 'target_boxcox' is your Box-Cox transformed data\n\n# 1. Q-Q Plot for Box-Cox Transformed Data\nplt.figure(figsize=(6, 6))\nprobplot(target_boxcox, dist=\"norm\", plot=plt)\nplt.title('Q-Q Plot for Box-Cox Transformed Premium Amount')\nplt.show()\n\n# 2. Histogram for Box-Cox Transformed Data\nplt.figure(figsize=(10, 5))\nsns.histplot(target_boxcox, kde=False, bins=30, color='purple')\nplt.title('Histogram of Box-Cox Transformed Premium Amount')\nplt.xlabel('Box-Cox Transformed Premium Amount')\nplt.ylabel('Frequency')\nplt.show()\n\n# 3. KDE Plot for Box-Cox Transformed Data\nplt.figure(figsize=(10, 5))\nsns.kdeplot(target_boxcox, color='purple', shade=True)\nplt.title('KDE Plot of Box-Cox Transformed Premium Amount')\nplt.xlabel('Box-Cox Transformed Premium Amount')\nplt.ylabel('Density')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:44:50.821417Z","iopub.execute_input":"2024-12-08T08:44:50.821701Z","iopub.status.idle":"2024-12-08T08:44:59.022979Z","shell.execute_reply.started":"2024-12-08T08:44:50.821675Z","shell.execute_reply":"2024-12-08T08:44:59.021935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Assuming 'target' is your variable (e.g., Premium Amount)\nplt.figure(figsize=(10, 5))\nsns.boxplot(x=target, color='orange')\nplt.title('Box Plot of Premium Amount')\nplt.xlabel('Premium Amount')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:44:59.027082Z","iopub.execute_input":"2024-12-08T08:44:59.027567Z","iopub.status.idle":"2024-12-08T08:44:59.334763Z","shell.execute_reply.started":"2024-12-08T08:44:59.027518Z","shell.execute_reply":"2024-12-08T08:44:59.333931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Calculate Z-scores for the target variable\nz_scores = (target - np.mean(target)) / np.std(target)\n\n# Identify outliers (Z-score > 3 or < -3)\noutliers_z = target[np.abs(z_scores) > 3]\n\nprint(\"Outliers based on Z-score:\")\nprint(outliers_z)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:44:59.335951Z","iopub.execute_input":"2024-12-08T08:44:59.336302Z","iopub.status.idle":"2024-12-08T08:44:59.362345Z","shell.execute_reply.started":"2024-12-08T08:44:59.336263Z","shell.execute_reply":"2024-12-08T08:44:59.361539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import boxcox\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom scipy.stats import skew, kurtosis\n\n# Apply Box-Cox transformation to the target variable\ntarget_train = train_df_imputed2[\"Premium Amount\"]\ntarget_positive_train = target_train[target_train > 0]  # Ensure no non-positive values\n\n# Perform the Box-Cox transformation\ntarget_train_boxcox, lambda_boxcox = boxcox(target_positive_train)\n\n# Print the lambda value used for Box-Cox transformation\nprint(f\"Lambda for Box-Cox Transformation: {lambda_boxcox}\")\n\n# Update the target variable in the train_df_imputed2 with the transformed values\ntrain_df_imputed2[\"Premium Amount\"] = target_train.copy()\ntrain_df_imputed2.loc[target_train > 0, \"Premium Amount\"] = target_train_boxcox\n\n# Plot the transformed target variable\nplt.figure(figsize=(10, 5))\nsns.histplot(target_train_boxcox, kde=True, bins=30, color='purple')\nplt.title('Histogram of Box-Cox Transformed Premium Amount')\nplt.xlabel('Box-Cox Transformed Premium Amount')\nplt.ylabel('Frequency')\nplt.show()\n\n# Q-Q Plot for Box-Cox transformed data\nplt.figure(figsize=(6, 6))\nprobplot(target_train_boxcox, dist=\"norm\", plot=plt)\nplt.title('Q-Q Plot for Box-Cox Transformed Premium Amount')\nplt.show()\n\n# Calculate skewness and kurtosis for the Box-Cox transformed target\nboxcox_skewness = skew(target_train_boxcox)\nboxcox_kurt = kurtosis(target_train_boxcox)\n\nprint(f\"Box-Cox Transformed Skewness: {boxcox_skewness}\")\nprint(f\"Box-Cox Transformed Kurtosis: {boxcox_kurt}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:44:59.363372Z","iopub.execute_input":"2024-12-08T08:44:59.363654Z","iopub.status.idle":"2024-12-08T08:45:12.071393Z","shell.execute_reply.started":"2024-12-08T08:44:59.363628Z","shell.execute_reply":"2024-12-08T08:45:12.070443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\ndef plot_histograms(variable):\n    # Print the column names of both datasets\n    print(\"Train Dataset Columns:\")\n    print(train_df_imputed2.columns)\n    print(\"\\nTest Dataset Columns:\")\n    print(test_df_imputed2.columns)\n    \n    # Plot histogram for train dataset\n    plt.figure(figsize=(10, 5))\n    sns.histplot(train_df_imputed2[variable], kde=True, bins=30, color='blue')\n    plt.title(f'Histogram of {variable} (Train Dataset)')\n    plt.xlabel(variable)\n    plt.ylabel('Frequency')\n    plt.show()\n\n    # Plot histogram for test dataset\n    plt.figure(figsize=(10, 5))\n    sns.histplot(test_df_imputed2[variable], kde=True, bins=30, color='green')\n    plt.title(f'Histogram of {variable} (Test Dataset)')\n    plt.xlabel(variable)\n    plt.ylabel('Frequency')\n    plt.show()\n\n# Example: To plot histograms for 'Age' variable\nplot_histograms('Annual Income')\n#Annual income","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:45:12.072498Z","iopub.execute_input":"2024-12-08T08:45:12.072774Z","iopub.status.idle":"2024-12-08T08:45:20.205479Z","shell.execute_reply.started":"2024-12-08T08:45:12.072749Z","shell.execute_reply":"2024-12-08T08:45:20.204685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.model_selection import train_test_split, RandomizedSearchCV\nfrom sklearn.metrics import mean_squared_log_error\nfrom scipy.special import inv_boxcox\nimport numpy as np\nimport pandas as pd\nfrom sklearn.metrics import make_scorer\nimport numpy as np\n\ndef rmsle(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n\nrmsle_scorer = make_scorer(rmsle, greater_is_better=False)\n\n# Prepare your data (replace with your actual dataset)\nX = train_df_imputed2.drop(columns=['Premium Amount', 'id'])\ny = train_df_imputed2['Premium Amount']\n\n# Split the data into train and validation sets\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Define the parameter grid for hyperparameter tuning\nparam_dist = {\n    'objective': ['regression'],\n    'metric': ['rmse'],\n    'learning_rate': [0.025],  # chhose 0.025 look for above that\n    'num_leaves': [256],  # look for 128 around\n    'max_depth': [30],  # 30 looking good\n    'subsample': [0.8],  # sorted\n    'colsample_bytree': [0.8],  #sorted\n    'min_child_samples': [35],  # look for near 35 \n    'reg_alpha': [0.02],  # look for 0.015 0.02 and 0.03\n    'reg_lambda': [0.02],  # same above\n    'n_estimators': [2000]  # Number of boosting rounds\n}\n\n# Randomized search with cross-validation\nrandom_search = RandomizedSearchCV(\n    estimator=lgb.LGBMRegressor(device='gpu', gpu_platform_id=0, gpu_device_id=0),  # The model to search over\n    param_distributions=param_dist,\n    n_iter=50,  # Number of parameter settings to sample\n    scoring= rmsle_scorer ,  # Minimize negative RMSLE\n    cv=3,  # Number of folds in cross-validation\n    verbose=1,\n    random_state=42  # Ensure reproducibility\n)\n\n# Perform the randomized search\nrandom_search.fit(X_train, y_train)\n\n# Get the best parameters and the best model\nbest_params = random_search.best_params_\nbest_model = random_search.best_estimator_\n\nprint(\"Best parameters found:\", best_params)\n\n# Predict on the validation set and apply inverse Box-Cox transformation\ny_pred_valid_transformed = best_model.predict(X_valid)\n\n# Inverse Box-Cox transformation (replace lambda_boxcox with the actual value)\nlambda_boxcox = 0.399232107508173  # Replace this with your actual Box-Cox lambda\ny_pred_valid = inv_boxcox(y_pred_valid_transformed, lambda_boxcox)\n\n# Calculate and print RMSLE\nrmsle_valid = rmsle(y_valid, y_pred_valid)\nprint(f\"RMSLE on Validation Set: {rmsle_valid:.5f}\")\n\n# Predict on the test set using the best model and apply inverse Box-Cox transformation\nX_test = test_df_imputed2.drop(columns=['id'])\ny_pred_test_transformed = best_model.predict(X_test)\ny_pred_test = inv_boxcox(y_pred_test_transformed, lambda_boxcox)\n\n# Prepare the submission file\nsubmission_df = pd.DataFrame({'id': test_df_imputed2['id'], 'Premium Amount': y_pred_test})\n\n# Convert 'id' column to integer type if necessary\nsubmission_df['id'] = submission_df['id'].astype('Int32')\n\n# Save the submission\nsubmission_df.to_csv('submission11.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T11:50:50.753283Z","iopub.execute_input":"2024-12-08T11:50:50.753976Z","iopub.status.idle":"2024-12-08T11:58:10.312759Z","shell.execute_reply.started":"2024-12-08T11:50:50.753943Z","shell.execute_reply":"2024-12-08T11:58:10.312004Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"import lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_log_error\nfrom scipy.special import inv_boxcox\nimport numpy as np\nimport pandas as pd\n\n# Prepare your data (replace with your actual dataset)\nX = train_df_imputed2.drop(columns=['Premium Amount', 'id'])\ny = train_df_imputed2['Premium Amount']\n\n# Split the data into train and validation sets\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Define hyperparameters with a logical structure for regression\nparam_grid = {\n    'objective': 'regression',\n    'metric': 'rmse',\n    'learning_rate': 0.01,  # Valid structure: learning_rate as a list of float values.\n    'num_leaves': 50,\n    'max_depth': 1000,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'min_child_samples': 100,\n    'reg_alpha': 1.0,\n    'reg_lambda': 1.0,\n    'n_estimators': 2000\n}\n# Add early stopping callback\ncallbacks = [\n    lgb.early_stopping(stopping_rounds=50, verbose=True)  # Stops if no improvement after 50 rounds\n]\n\n# Train model with early stopping\ntrain_data = lgb.Dataset(X_train, label=y_train)\nvalid_data = lgb.Dataset(X_valid, label=y_valid)\n\nmodel = lgb.train(\n    params=param_grid,\n    train_set=train_data,\n    num_boost_round=1000,  # Maximum boosting iterations\n    valid_sets=[train_data, valid_data],  # Validation data for monitoring\n    callbacks=callbacks\n)\n\n# Predict on the validation set and apply inverse Box-Cox transformation\ny_pred_valid_transformed = model.predict(X_valid, num_iteration=model.best_iteration)\n\n# Inverse Box-Cox transformation (replace lambda_boxcox with the actual value)\nlambda_boxcox = 0.5  # Replace this with your actual Box-Cox lambda\ny_pred_valid = inv_boxcox(y_pred_valid_transformed, lambda_boxcox)\n\n# Calculate RMSLE for the validation set\nrmsle_valid = np.sqrt(mean_squared_log_error(y_valid, y_pred_valid))\nprint(f\"RMSLE on Validation Set: {rmsle_valid}\")\n\n# Predict on the test set and apply inverse Box-Cox transformation\nX_test = test_df_imputed2.drop(columns=['id'])\ny_pred_test_transformed = model.predict(X_test, num_iteration=model.best_iteration)\ny_pred_test = inv_boxcox(y_pred_test_transformed, lambda_boxcox)\n\n# Prepare the submission file\nsubmission_df = pd.DataFrame({'id': test_df_imputed2['id'], 'Premium Amount': y_pred_test})\n\n# Convert 'id' column to integer type if necessary\nsubmission_df['id'] = submission_df['id'].astype('Int32')\n\n# Save the submission\nsubmission_df.to_csv('submission7.csv', index=False)\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:45:20.207142Z","iopub.execute_input":"2024-12-08T08:45:20.207582Z","iopub.status.idle":"2024-12-08T08:45:20.215059Z","shell.execute_reply.started":"2024-12-08T08:45:20.207534Z","shell.execute_reply":"2024-12-08T08:45:20.214221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Best parameters found: {'subsample': 0.8, 'reg_lambda': 0.01, 'reg_alpha': 0.01, 'objective': 'regression', 'num_leaves': 64, 'n_estimators': 2000, 'min_child_samples': 30, 'metric': 'rmse', 'max_depth': 30, 'learning_rate': 0.02, 'colsample_bytree': 0.8}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:45:44.800971Z","iopub.status.idle":"2024-12-08T08:45:44.801472Z","shell.execute_reply.started":"2024-12-08T08:45:44.801205Z","shell.execute_reply":"2024-12-08T08:45:44.801232Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Best parameters found: {'subsample': 0.7, 'reg_lambda': 0.0, 'reg_alpha': 0.0, 'objective': 'regression', 'num_leaves': 64, 'n_estimators': 2000, 'min_child_samples': 30, 'metric': 'rmse', 'max_depth': 20, 'learning_rate': 0.02, 'colsample_bytree': 0.8}\r\nRMSLE on Validation Set: 3.3204780237894105","metadata":{}}]}