{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-10T08:33:46.63671Z","iopub.execute_input":"2024-12-10T08:33:46.637035Z","iopub.status.idle":"2024-12-10T08:33:46.969828Z","shell.execute_reply.started":"2024-12-10T08:33:46.637002Z","shell.execute_reply":"2024-12-10T08:33:46.968937Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Importing Necessary Libraries","metadata":{}},{"cell_type":"code","source":"# Import core libraries\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport os\n\n# Import machine learning tools\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import mean_squared_log_error\n\n# Import XGBoost and optimization tools\nimport xgboost as xgb\nimport optuna\n\n# Ignore warnings for a cleaner output\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T08:34:01.265092Z","iopub.execute_input":"2024-12-10T08:34:01.265494Z","iopub.status.idle":"2024-12-10T08:34:02.458992Z","shell.execute_reply.started":"2024-12-10T08:34:01.265452Z","shell.execute_reply":"2024-12-10T08:34:02.458331Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"# Load datasets\ntrain = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\n\n# Drop the 'id' column as it is not required for training\ntrain.drop('id', axis=1, inplace=True)\ntest.drop('id', axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T08:34:02.460005Z","iopub.execute_input":"2024-12-10T08:34:02.46051Z","iopub.status.idle":"2024-12-10T08:34:10.649114Z","shell.execute_reply.started":"2024-12-10T08:34:02.460463Z","shell.execute_reply":"2024-12-10T08:34:10.648428Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Explore and Understand the Data","metadata":{}},{"cell_type":"code","source":"# Display the first few rows of the training dataset\nprint(\"Training Data Overview:\")\ndisplay(train.head())\n\n# Display the first few rows of the test dataset\nprint(\"Test Data Overview:\")\ndisplay(test.head())\n\n# Summary of the training data\nprint(\"Training Data Summary:\")\ntrain.info()\n\n# Check for missing values\nprint(\"Missing Values in Training Data:\")\ntrain.isnull().sum()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T08:34:10.650889Z","iopub.execute_input":"2024-12-10T08:34:10.651502Z","iopub.status.idle":"2024-12-10T08:34:11.783104Z","shell.execute_reply.started":"2024-12-10T08:34:10.65146Z","shell.execute_reply":"2024-12-10T08:34:11.782313Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering - Extract Date Features","metadata":{}},{"cell_type":"code","source":"# Function to extract date-related features\ndef date_features(df):\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['Year'] = df['Policy Start Date'].dt.year\n    df['Month'] = df['Policy Start Date'].dt.month\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Quarter'] = df['Policy Start Date'].dt.quarter\n    df['Day of Week'] = df['Policy Start Date'].dt.dayofweek\n    df.drop('Policy Start Date', axis=1, inplace=True)\n    return df\n\n# Apply the function to train and test datasets\ntrain = date_features(train)\ntest = date_features(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T08:34:27.956693Z","iopub.execute_input":"2024-12-10T08:34:27.95751Z","iopub.status.idle":"2024-12-10T08:34:29.277531Z","shell.execute_reply.started":"2024-12-10T08:34:27.957475Z","shell.execute_reply":"2024-12-10T08:34:29.27661Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering - Create Interaction Features","metadata":{}},{"cell_type":"code","source":"# Create new interaction features\ntrain['Annual_Income_Health_Score_Ratio'] = train['Health Score'] / train['Annual Income']\ntest['Annual_Income_Health_Score_Ratio'] = test['Health Score'] / test['Annual Income']\n\ntrain['Annual_Income_Age_Ratio'] = train['Annual Income'] / train['Age']\ntest['Annual_Income_Age_Ratio'] = test['Annual Income'] / test['Age']\n\ntrain['Credit_Age'] = train['Credit Score'] / train['Age']\ntest['Credit_Age'] = test['Credit Score'] / test['Age']\n\ntrain['Vehicle_Age_Insurance_Duration'] = train['Vehicle Age'] / train['Insurance Duration']\ntest['Vehicle_Age_Insurance_Duration'] = test['Vehicle Age'] / test['Insurance Duration']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T08:34:31.094348Z","iopub.execute_input":"2024-12-10T08:34:31.094716Z","iopub.status.idle":"2024-12-10T08:34:31.13431Z","shell.execute_reply.started":"2024-12-10T08:34:31.094686Z","shell.execute_reply":"2024-12-10T08:34:31.133477Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering - Categorical Transformation","metadata":{}},{"cell_type":"code","source":"# Combine location and property type into a new feature\ntrain['Property Location Type'] = train['Location'] + '_' + train['Property Type']\ntest['Property Location Type'] = test['Location'] + '_' + test['Property Type']\n\n# Drop the original 'Property Type' feature\ntrain.drop('Property Type', axis=1, inplace=True)\ntest.drop('Property Type', axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T08:34:32.516438Z","iopub.execute_input":"2024-12-10T08:34:32.517274Z","iopub.status.idle":"2024-12-10T08:34:33.22001Z","shell.execute_reply.started":"2024-12-10T08:34:32.517241Z","shell.execute_reply":"2024-12-10T08:34:33.219296Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Transformation - Log Transformation of Skewed Features","metadata":{}},{"cell_type":"code","source":"# Apply log transformation to skewed numerical features\nfor col in ['Annual Income', 'Vehicle Age', 'Health Score', 'Credit Score']:\n    train[col] = np.log1p(train[col])\n    test[col] = np.log1p(test[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T08:34:34.95618Z","iopub.execute_input":"2024-12-10T08:34:34.956529Z","iopub.status.idle":"2024-12-10T08:34:35.011652Z","shell.execute_reply.started":"2024-12-10T08:34:34.956497Z","shell.execute_reply":"2024-12-10T08:34:35.010926Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Preprocessing - Memory Optimization","metadata":{}},{"cell_type":"code","source":"# Function to reduce memory usage\ndef reduce_memory_usage(df):\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type == 'float64':\n            df[col] = df[col].astype('float32')\n        elif col_type == 'int64':\n            df[col] = df[col].astype('int32')\n    return df\n\n# Apply the function to both train and test datasets\ntrain = reduce_memory_usage(train)\ntest = reduce_memory_usage(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T08:34:37.116104Z","iopub.execute_input":"2024-12-10T08:34:37.116457Z","iopub.status.idle":"2024-12-10T08:34:37.16612Z","shell.execute_reply.started":"2024-12-10T08:34:37.116426Z","shell.execute_reply":"2024-12-10T08:34:37.165436Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Preprocessing - Define Target and Preprocessing Pipeline","metadata":{}},{"cell_type":"code","source":"# Define the target variable\ntarget = 'Premium Amount'\n\n# Identify numerical and categorical columns\nnumerical_cols = train.select_dtypes(include=['float32', 'int32']).columns.tolist()\nif target in numerical_cols:\n    numerical_cols.remove(target)\n\ncategorical_cols = train.select_dtypes(include=['object']).columns.tolist()\n\n# Define a preprocessing pipeline\npreprocessing = ColumnTransformer([\n    ('num', make_pipeline(SimpleImputer(strategy='mean'), StandardScaler()), numerical_cols),\n    ('cat', make_pipeline(SimpleImputer(strategy='constant', fill_value='unknown'),\n                          OneHotEncoder(handle_unknown='ignore')), categorical_cols)\n], remainder='drop')\n\n# Prepare training and test data\nX_train = train.drop(columns=[target])\ny_train = np.log1p(train[target])  # Log transform target for RMSLE\nX_test = test.copy()\n\nX_train_preprocessed = preprocessing.fit_transform(X_train)\nX_test_preprocessed = preprocessing.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T08:34:39.036188Z","iopub.execute_input":"2024-12-10T08:34:39.036569Z","iopub.status.idle":"2024-12-10T08:34:48.251763Z","shell.execute_reply.started":"2024-12-10T08:34:39.036536Z","shell.execute_reply":"2024-12-10T08:34:48.25103Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Hyperparameter Tuning using Optuna","metadata":{}},{"cell_type":"code","source":"# Define the objective function for Optuna\ndef objective(trial):\n    params = {\n        'objective': 'reg:squarederror',\n        'eval_metric': 'rmse',\n        'tree_method': 'gpu_hist',\n        'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.3),\n        'max_depth': trial.suggest_int('max_depth', 3, 10),\n        'min_child_weight': trial.suggest_int('min_child_weight', 1, 10),\n        'subsample': trial.suggest_uniform('subsample', 0.5, 1.0),\n        'colsample_bytree': trial.suggest_uniform('colsample_bytree', 0.5, 1.0),\n        'lambda': trial.suggest_loguniform('lambda', 1e-3, 10),\n        'alpha': trial.suggest_loguniform('alpha', 1e-3, 10),\n    }\n    dtrain = xgb.DMatrix(X_train_preprocessed, label=y_train)\n    cv_results = xgb.cv(\n        params,\n        dtrain,\n        num_boost_round=1000,\n        nfold=5,\n        early_stopping_rounds=50,\n        metrics=\"rmse\",\n        as_pandas=True,\n        seed=42,\n    )\n    return cv_results['test-rmse-mean'].min()\n\n# Perform hyperparameter tuning\nstudy = optuna.create_study(direction='minimize')\nstudy.optimize(objective, n_trials=15)\nbest_params = study.best_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T08:38:42.35614Z","iopub.execute_input":"2024-12-10T08:38:42.356727Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train the Final Model","metadata":{}},{"cell_type":"code","source":"# Train the final model using the best parameters\nbest_params['eval_metric'] = 'rmse'\ndtrain = xgb.DMatrix(X_train_preprocessed, label=y_train)\nfinal_model = xgb.train(best_params, dtrain, num_boost_round=100)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Make Predictions and Create Submission File","metadata":{}},{"cell_type":"code","source":"# Generate predictions for the test set\ndtest = xgb.DMatrix(X_test_preprocessed)\ny_pred = final_model.predict(dtest)\ny_pred_final = np.expm1(y_pred)  # Reverse log transformation\n\n# Prepare the submission file\nsub = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")\noutput = pd.DataFrame({\"id\": sub.id, \"Premium Amount\": y_pred_final})\noutput.to_csv('submission.csv', index=False)\n\n# Display the first few rows of the submission file\noutput.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}