{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:15:59.444618Z","iopub.execute_input":"2024-12-12T23:15:59.445612Z","iopub.status.idle":"2024-12-12T23:16:00.575394Z","shell.execute_reply.started":"2024-12-12T23:15:59.445557Z","shell.execute_reply":"2024-12-12T23:16:00.574343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data=pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:16:00.577335Z","iopub.execute_input":"2024-12-12T23:16:00.577918Z","iopub.status.idle":"2024-12-12T23:16:07.067273Z","shell.execute_reply.started":"2024-12-12T23:16:00.577871Z","shell.execute_reply":"2024-12-12T23:16:07.066226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature engineering implementation\n\n# 1. Import necessary libraries\nimport numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom catboost import CatBoostRegressor, Pool\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.cluster import KMeans\nfrom datetime import datetime\n\n# Copy the dataset\ndata = train_data.copy()\n\n# 2. Create new features\n\n# (1) Annual income per dependent\n# Calculate the annual income divided by the number of dependents to get the income per dependent.\ndata['Income_per_Dependent'] = data['Annual Income'] / (data['Number of Dependents'] + 1)\n\n# (2) Income categories\n# Categorize annual income into Low (< 30000), Medium (30000-60000), and High (>60000).\ndata['Income_Category'] = pd.cut(data['Annual Income'], bins=[0, 30000, 60000, np.inf], labels=['Low', 'Medium', 'High'])\n\n# (3) Health score per age\n# Standardize the health score by age to evaluate health relative to age.\ndata['Health_per_Age'] = data['Health Score'] / data['Age']\n\n# (4) Family type based on marital status and number of dependents\n# Combine marital status and the number of dependents to create a family type category.\ndata['Family_Type'] = data['Marital Status'] + \"_\" + data['Number of Dependents'].fillna(0).astype(int).astype(str)\n\n# (5) Insurance duration categories\n# Group insurance duration into three categories: Short (0-3), Medium (3-7), and Long (>7).\ndata['Insurance_Duration_Category'] = pd.cut(data['Insurance Duration'], bins=[0, 3, 7, np.inf], labels=['Short', 'Medium', 'Long'])\n\n# (6) Lifestyle index combining exercise frequency and smoking status\n# Combine exercise frequency and smoking status into a lifestyle index.\ndata['Lifestyle_Index'] = data['Exercise Frequency'].map({'Rarely': 1, 'Monthly': 2, 'Daily': 3}) * (data['Smoking Status'] == 'No').astype(int)\n\n# (7) Regional indicators (one-hot encoding for location)\n# Create dummy variables for location to encode regional information.\ndata = pd.get_dummies(data, columns=['Location'], prefix='Location')\n\n# (8) Extract time of policy start from 'Policy Start Date'\n# Extract the hour from the policy start date to analyze contract timing.\ndef extract_hour(date_str):\n    try:\n        # Extract the hour from datetime string\n        return int(date_str.split(' ')[1].split(':')[0])\n    except (IndexError, ValueError):\n        # Return a default value (e.g., -1) for invalid formats\n        return -1\n\n# Apply the function to extract hour\ndata['Policy_Hour'] = data['Policy Start Date'].apply(extract_hour)\n\n# (9) Customer grouping using PCA and clustering\n# Select numeric columns and scale them\nnumeric_cols = data.select_dtypes(include=['float64', 'int64']).columns\nscaler = StandardScaler()\nscaled_data = scaler.fit_transform(data[numeric_cols].fillna(0))\n\n# K-means clustering\n# Apply K-means clustering to group customers into five clusters.\nkmeans = KMeans(n_clusters=5, random_state=42)\ndata['Customer_Group'] = kmeans.fit_predict(scaled_data)\n\n# (10) Interaction terms for health and exercise\n# Create interaction terms to evaluate the combined effect of health and exercise.\ndata['Health_Exercise_Interaction'] = data['Health Score'] * data['Exercise Frequency'].map({'Rarely': 1, 'Monthly': 2, 'Daily': 3})\n\n# 3. Prepare data for CatBoost\n# Define the target variable and features\ndata = data.dropna(subset=['Premium Amount'])  # Drop rows where target is missing\nX = data.drop(columns=['id', 'Premium Amount', 'Policy Start Date'])  # Exclude unnecessary columns\ny = data['Premium Amount']\n\n# Handle missing values in categorical features by replacing NaN with 'Unknown'\nfor col in X.select_dtypes(include=['object']).columns:\n    X[col] = X[col].fillna('Unknown')\n\n# Ensure all categorical columns are strings and properly encoded for CatBoost\nfor col in X.select_dtypes(include=['category', 'object']).columns:\n    X[col] = X[col].astype(str)\n\n# Identify categorical features for CatBoost\ncategorical_features = [col for col in X.columns if X[col].dtype == 'object']\n\n# Train-test split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Create CatBoost Pool objects\ntrain_pool = Pool(X_train, y_train, cat_features=categorical_features)\ntest_pool = Pool(X_test, y_test, cat_features=categorical_features)\n\n# 4. Train CatBoost Regressor\ncatboost_model = CatBoostRegressor(\n    iterations=1000,\n    learning_rate=0.05,\n    depth=6,\n    loss_function='RMSE',\n    random_seed=42,\n    verbose=100\n)\n\ncatboost_model.fit(train_pool)\n\n# 5. Evaluate the model using RMSLE\ny_pred = catboost_model.predict(test_pool)\nrmsle = np.sqrt(mean_squared_log_error(y_test, y_pred))\nprint(f\"RMSLE: {rmsle}\")\n\n# Save feature importance\nfeature_importances = catboost_model.get_feature_importance()\nimportant_features = pd.DataFrame({'Feature': X_train.columns, 'Importance': feature_importances})\nimportant_features.sort_values(by='Importance', ascending=False, inplace=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:16:07.068913Z","iopub.execute_input":"2024-12-12T23:16:07.069325Z","iopub.status.idle":"2024-12-12T23:30:19.341857Z","shell.execute_reply.started":"2024-12-12T23:16:07.06928Z","shell.execute_reply":"2024-12-12T23:30:19.340749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"important_features.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:30:19.343852Z","iopub.execute_input":"2024-12-12T23:30:19.344175Z","iopub.status.idle":"2024-12-12T23:30:19.359051Z","shell.execute_reply.started":"2024-12-12T23:30:19.344143Z","shell.execute_reply":"2024-12-12T23:30:19.357942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test=pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:30:19.360289Z","iopub.execute_input":"2024-12-12T23:30:19.360609Z","iopub.status.idle":"2024-12-12T23:30:23.23133Z","shell.execute_reply.started":"2024-12-12T23:30:19.360577Z","shell.execute_reply":"2024-12-12T23:30:23.230468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:30:23.232686Z","iopub.execute_input":"2024-12-12T23:30:23.233141Z","iopub.status.idle":"2024-12-12T23:30:23.666684Z","shell.execute_reply.started":"2024-12-12T23:30:23.233095Z","shell.execute_reply":"2024-12-12T23:30:23.665539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# (1) Annual income per dependent\n# Calculate the annual income divided by the number of dependents to get the income per dependent.\ntest['Income_per_Dependent'] = test['Annual Income'] / (test['Number of Dependents'] + 1)\n\n# (2) Income categories\n# Categorize annual income into Low (< 30000), Medium (30000-60000), and High (>60000).\ntest['Income_Category'] = pd.cut(test['Annual Income'], bins=[0, 30000, 60000, np.inf], labels=['Low', 'Medium', 'High'])\n\n# (3) Health score per age\n# Standardize the health score by age to evaluate health relative to age.\ntest['Health_per_Age'] = test['Health Score'] / test['Age']\n\n# (4) Family type based on marital status and number of dependents\n# Combine marital status and the number of dependents to create a family type category.\ntest['Family_Type'] = test['Marital Status'] + \"_\" + test['Number of Dependents'].fillna(0).astype(int).astype(str)\n\n# (5) Insurance duration categories\n# Group insurance duration into three categories: Short (0-3), Medium (3-7), and Long (>7).\ntest['Insurance_Duration_Category'] = pd.cut(test['Insurance Duration'], bins=[0, 3, 7, np.inf], labels=['Short', 'Medium', 'Long'])\n\n# (6) Lifestyle index combining exercise frequency and smoking status\n# Combine exercise frequency and smoking status into a lifestyle index.\ntest['Lifestyle_Index'] = test['Exercise Frequency'].map({'Rarely': 1, 'Monthly': 2, 'Daily': 3}) * (test['Smoking Status'] == 'No').astype(int)\n\n# (7) Regional indicators (one-hot encoding for location)\n# Create dummy variables for location to encode regional information.\ntest = pd.get_dummies(test, columns=['Location'], prefix='Location')\n\n# (8) Extract time of policy start from 'Policy Start Date'\n# Extract the hour from the policy start date to analyze contract timing.\ndef extract_hour(date_str):\n    try:\n        # Extract the hour from datetime string\n        return int(date_str.split(' ')[1].split(':')[0])\n    except (IndexError, ValueError):\n        # Return a default value (e.g., -1) for invalid formats\n        return -1\n\n# Apply the function to extract hour\ntest['Policy_Hour'] = test['Policy Start Date'].apply(extract_hour)\n\n# (9) Customer grouping using PCA and clustering\n# Select numeric columns and scale them\nnumeric_cols = test.select_dtypes(include=['float64', 'int64']).columns\nscaler = StandardScaler()\nscaled_data = scaler.fit_transform(test[numeric_cols].fillna(0))  # Correctly reference the data using test[numeric_cols]\n\n# K-means clustering\n# Apply K-means clustering to group customers into five clusters.\nkmeans = KMeans(n_clusters=5, random_state=42)\ntest['Customer_Group'] = kmeans.fit_predict(scaled_data)\n\n# (10) Interaction terms for health and exercise\n# Create interaction terms to evaluate the combined effect of health and exercise.\ntest['Health_Exercise_Interaction'] = test['Health Score'] * test['Exercise Frequency'].map({'Rarely': 1, 'Monthly': 2, 'Daily': 3})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:30:23.667893Z","iopub.execute_input":"2024-12-12T23:30:23.668206Z","iopub.status.idle":"2024-12-12T23:30:32.844968Z","shell.execute_reply.started":"2024-12-12T23:30:23.668173Z","shell.execute_reply":"2024-12-12T23:30:32.844097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Exclude unnecessary columns\ntest_X = test.drop(columns=['id', 'Policy Start Date'], errors='ignore')\n\n# Handle missing values in categorical features by replacing NaN with 'Unknown'\nfor col in test_X.select_dtypes(include=['object']).columns:\n    test_X[col] = test_X[col].fillna('Unknown').astype(str)\n\n# Handle missing values in numerical features by replacing NaN with 0\nfor col in test_X.select_dtypes(include=['float64', 'int64']).columns:\n    test_X[col] = test_X[col].fillna(0)\n\n# Ensure test_X has the same features as the training set\nmissing_cols = set(X_train.columns) - set(test_X.columns)\nfor col in missing_cols:\n    test_X[col] = 0  # Add missing columns with default value 0\ntest_X = test_X[X_train.columns]\n\n# Ensure categorical columns are properly encoded\nfor col in categorical_features:\n    if col in test_X.columns:\n        test_X[col] = test_X[col].astype(str)\n\n# Predict\ntry:\n    premium_predictions = catboost_model.predict(test_X)\nexcept Exception as e:\n    print(f\"Error during prediction: {e}\")\n    raise\n\n# Add predictions to the test dataset\ntest['Predicted_Premium_Amount'] = premium_predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:30:32.846151Z","iopub.execute_input":"2024-12-12T23:30:32.846494Z","iopub.status.idle":"2024-12-12T23:30:37.605166Z","shell.execute_reply.started":"2024-12-12T23:30:32.846459Z","shell.execute_reply":"2024-12-12T23:30:37.604194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission=pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:32:01.515526Z","iopub.execute_input":"2024-12-12T23:32:01.51593Z","iopub.status.idle":"2024-12-12T23:32:01.789818Z","shell.execute_reply.started":"2024-12-12T23:32:01.515898Z","shell.execute_reply":"2024-12-12T23:32:01.788944Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:32:12.138563Z","iopub.execute_input":"2024-12-12T23:32:12.139635Z","iopub.status.idle":"2024-12-12T23:32:12.151263Z","shell.execute_reply.started":"2024-12-12T23:32:12.139581Z","shell.execute_reply":"2024-12-12T23:32:12.150289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission['Premium Amount']=premium_predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:33:10.161325Z","iopub.execute_input":"2024-12-12T23:33:10.161743Z","iopub.status.idle":"2024-12-12T23:33:10.167506Z","shell.execute_reply.started":"2024-12-12T23:33:10.161705Z","shell.execute_reply":"2024-12-12T23:33:10.166465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('/kaggle/working/submission.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T23:34:38.293636Z","iopub.execute_input":"2024-12-12T23:34:38.294045Z","iopub.status.idle":"2024-12-12T23:34:40.007042Z","shell.execute_reply.started":"2024-12-12T23:34:38.29401Z","shell.execute_reply":"2024-12-12T23:34:40.006019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}