{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:10:55.07128Z","iopub.execute_input":"2024-12-13T00:10:55.072122Z","iopub.status.idle":"2024-12-13T00:10:55.4661Z","shell.execute_reply.started":"2024-12-13T00:10:55.07206Z","shell.execute_reply":"2024-12-13T00:10:55.465034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport re\nfrom datetime import datetime\n\nfrom sklearn.svm import SVR\nfrom xgboost import XGBRegressor\n\nfrom sklearn.base import BaseEstimator, TransformerMixin \nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline \nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler, OrdinalEncoder, LabelEncoder\nfrom sklearn.model_selection import GridSearchCV, train_test_split, cross_val_score \nfrom sklearn.metrics import accuracy_score, roc_curve, auc, mean_squared_error, r2_score \n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:10:55.576152Z","iopub.execute_input":"2024-12-13T00:10:55.576645Z","iopub.status.idle":"2024-12-13T00:10:56.352981Z","shell.execute_reply.started":"2024-12-13T00:10:55.576612Z","shell.execute_reply":"2024-12-13T00:10:56.351886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv',index_col='id' )\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv',index_col='id' )\ndf_sample_submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:10:56.354823Z","iopub.execute_input":"2024-12-13T00:10:56.355397Z","iopub.status.idle":"2024-12-13T00:11:04.425211Z","shell.execute_reply.started":"2024-12-13T00:10:56.35536Z","shell.execute_reply":"2024-12-13T00:11:04.423908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = df_train['Premium Amount']\nX = df_train.drop(columns=['Premium Amount']) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:04.427051Z","iopub.execute_input":"2024-12-13T00:11:04.427471Z","iopub.status.idle":"2024-12-13T00:11:04.603711Z","shell.execute_reply.started":"2024-12-13T00:11:04.427434Z","shell.execute_reply":"2024-12-13T00:11:04.602507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:04.605093Z","iopub.execute_input":"2024-12-13T00:11:04.605795Z","iopub.status.idle":"2024-12-13T00:11:05.271295Z","shell.execute_reply.started":"2024-12-13T00:11:04.605756Z","shell.execute_reply":"2024-12-13T00:11:05.270229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:05.273467Z","iopub.execute_input":"2024-12-13T00:11:05.273798Z","iopub.status.idle":"2024-12-13T00:11:05.911461Z","shell.execute_reply.started":"2024-12-13T00:11:05.273766Z","shell.execute_reply":"2024-12-13T00:11:05.910206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_columns = X.select_dtypes(include=['number']).columns\ncategorical_columns = X.select_dtypes(include=['object']).columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:05.912957Z","iopub.execute_input":"2024-12-13T00:11:05.913423Z","iopub.status.idle":"2024-12-13T00:11:06.115389Z","shell.execute_reply.started":"2024-12-13T00:11:05.913373Z","shell.execute_reply":"2024-12-13T00:11:06.114041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:06.116739Z","iopub.execute_input":"2024-12-13T00:11:06.117201Z","iopub.status.idle":"2024-12-13T00:11:06.125152Z","shell.execute_reply.started":"2024-12-13T00:11:06.117153Z","shell.execute_reply":"2024-12-13T00:11:06.123986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:06.126523Z","iopub.execute_input":"2024-12-13T00:11:06.126882Z","iopub.status.idle":"2024-12-13T00:11:06.140362Z","shell.execute_reply.started":"2024-12-13T00:11:06.126851Z","shell.execute_reply":"2024-12-13T00:11:06.139201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ncategorical_columns = categorical_columns.drop('Policy Start Date')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:06.142163Z","iopub.execute_input":"2024-12-13T00:11:06.142536Z","iopub.status.idle":"2024-12-13T00:11:06.15236Z","shell.execute_reply.started":"2024-12-13T00:11:06.1425Z","shell.execute_reply":"2024-12-13T00:11:06.151169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['Annual Income'].quantile(0.025)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:06.153681Z","iopub.execute_input":"2024-12-13T00:11:06.154008Z","iopub.status.idle":"2024-12-13T00:11:06.1943Z","shell.execute_reply.started":"2024-12-13T00:11:06.153976Z","shell.execute_reply":"2024-12-13T00:11:06.193078Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"# Calculate missing percentages\ncategorical_missing_percentages = df_train[categorical_columns].isnull().sum() / len(df_train) * 100\ncategorical_missing_percentages = categorical_missing_percentages[categorical_missing_percentages > 0]\n\ncategorical_missing_percentages","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:06.196799Z","iopub.execute_input":"2024-12-13T00:11:06.197183Z","iopub.status.idle":"2024-12-13T00:11:06.909818Z","shell.execute_reply.started":"2024-12-13T00:11:06.197138Z","shell.execute_reply":"2024-12-13T00:11:06.908671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the barplot\nsns.barplot(x=categorical_missing_percentages.index, y=categorical_missing_percentages.values)\n\n# Rotate x-axis labels by 90 degrees\nplt.xticks(rotation=90)\n\n# Label the axes\nplt.xlabel('Features')\nplt.ylabel('Percentage of Missing Values')\n\n# Display the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:06.911725Z","iopub.execute_input":"2024-12-13T00:11:06.912071Z","iopub.status.idle":"2024-12-13T00:11:07.140559Z","shell.execute_reply.started":"2024-12-13T00:11:06.912036Z","shell.execute_reply":"2024-12-13T00:11:07.139447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate missing percentages for numerical columns\nnumerical_missing_percentages = df_train[numerical_columns].isnull().sum() / len(df_train) * 100\nnumerical_missing_percentages = numerical_missing_percentages[numerical_missing_percentages > 0]\n\nnumerical_missing_percentages","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:07.23863Z","iopub.execute_input":"2024-12-13T00:11:07.239037Z","iopub.status.idle":"2024-12-13T00:11:07.289503Z","shell.execute_reply.started":"2024-12-13T00:11:07.239002Z","shell.execute_reply":"2024-12-13T00:11:07.288249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the barplot\nsns.barplot(x=numerical_missing_percentages.index, y=numerical_missing_percentages.values)\n\n# Rotate x-axis labels by 90 degrees\nplt.xticks(rotation=90)\n\n# Label the axes\nplt.xlabel('Features')\nplt.ylabel('Percentage of Missing Values')\n\n# Display the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:07.689399Z","iopub.execute_input":"2024-12-13T00:11:07.689832Z","iopub.status.idle":"2024-12-13T00:11:07.95598Z","shell.execute_reply.started":"2024-12-13T00:11:07.689796Z","shell.execute_reply":"2024-12-13T00:11:07.954791Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Exploratory Data Analytics","metadata":{}},{"cell_type":"markdown","source":"#### Target Variable Analysis","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 6), gridspec_kw={'width_ratios': [1, 2]})\n\n# Box plot on the left\nsns.boxplot(y=df_train['Premium Amount'], ax=axes[0], color='skyblue')\naxes[0].set(title='Box Plot of Premium Amount', xlabel='', ylabel='Premium Amount')\n\n# Histogram with KDE on the right\nsns.histplot(df_train['Premium Amount'], kde=True, bins=50, ax=axes[1], color='green')\naxes[1].set(title='Histogram and KDE of Premium Amount', xlabel='Premium Amount', ylabel='Frequency')\n\n# Adjust layout\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:10.377553Z","iopub.execute_input":"2024-12-13T00:11:10.378136Z","iopub.status.idle":"2024-12-13T00:11:16.61579Z","shell.execute_reply.started":"2024-12-13T00:11:10.378025Z","shell.execute_reply":"2024-12-13T00:11:16.61458Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Categorical Variables Distribution with Premium Amount","metadata":{}},{"cell_type":"code","source":"\nfor column in categorical_columns:\n    print(f'Column: {column}')\n    print(df_train[column].value_counts())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train[categorical_columns].isna().sum()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Using sns.displot for faceted histogram\nsns.displot(\n    data=df_train,\n    x='Premium Amount',\n    col='Marital Status',\n    kde=True,  # Add KDE curve for smoothness\n    col_wrap=3,  # Number of facets per row (adjust as needed)\n    bins=10,  # Number of bins\n    height=4,  # Height of each facet\n    aspect=1.2,  # Aspect ratio of each facet\n    palette = 'Set2'# Optional color\n)\n\n# Adding a title to the entire plot\nplt.subplots_adjust(top=0.9)\nplt.suptitle('Distribution of Premium Amount Faceted by Marital Status', fontsize=16)\n\n# Show the plot\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Using sns.displot for faceted histogram\nsns.displot(\n    data=df_train,\n    x='Premium Amount',\n    col='Occupation',\n    kde=True,  # Add KDE curve for smoothness\n    col_wrap=3,  # Number of facets per row (adjust as needed)\n    bins=10,  # Number of bins\n    height=4,  # Height of each facet\n    aspect=1.2,  # Aspect ratio of each facet\n    palette = 'Set2'# Optional color\n)\n\n# Adding a title to the entire plot\nplt.subplots_adjust(top=0.9)\nplt.suptitle('Distribution of Premium Amount Faceted by Occupation', fontsize=16)\n\n# Show the plot\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Using sns.displot for faceted histogram\nsns.displot(\n    data=df_train,\n    x='Premium Amount',\n    col='Education Level',\n    kde=True,  # Add KDE curve for smoothness\n    col_wrap=3,  # Number of facets per row (adjust as needed)\n    bins=10,  # Number of bins\n    height=4,  # Height of each facet\n    aspect=1.2,  # Aspect ratio of each facet\n    palette = 'Set2'# Optional color\n)\n\n# Adding a title to the entire plot\nplt.subplots_adjust(top=0.9)\nplt.suptitle('Distribution of Premium Amount Faceted by Education Level', fontsize=16)\n\n# Show the plot\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 6), gridspec_kw={'width_ratios': [1, 2]})\n\n# Box plot on the left\nsns.boxplot(y=df_train['Annual Income'], ax=axes[0], color='skyblue')\naxes[0].set(title='Box Plot of Annual Income', xlabel='', ylabel='Annual Income')\n\n# Histogram with KDE on the right\nsns.histplot(df_train['Annual Income'], kde=True, bins=15, ax=axes[1], color='green')\naxes[1].set(title='Histogram and KDE of Annual Income', xlabel='Annual Income', ylabel='Frequency')\n\n# Adjust layout\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Line plot with seaborn\nplt.figure(figsize=(12, 6))\nsns.countplot(\n    data=df_train,\n    x='Number of Dependents', \n    palette='viridis',         # Optional color palette\n    linewidth=2                # Line thickness\n)\n\n# Adding labels and title\nplt.title(' Number of Dependents Distribution', fontsize=16)  \nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Line plot with seaborn\nplt.figure(figsize=(12, 6))\nsns.histplot(\n    x=df_train['Credit Score'], \n    palette='viridis',         # Optional color palette\n    kde=True,\n    bins=10\n)\n\n# Adding labels and title\nplt.title(' Credit Score Distribution', fontsize=16)  \nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Bar plot\nplt.figure(figsize=(10, 6))\nsns.barplot(\n    data=df_train,\n    x='Policy Type',\n    y='Premium Amount',\n    ci=None,  # Remove confidence intervals for cleaner visualization\n    palette='Set2'\n)\n\n# Adding labels and title\nplt.title('Average Premium Amount by Policy Type', fontsize=16)\nplt.xlabel('Policy Type', fontsize=14)\nplt.ylabel('Average Premium Amount', fontsize=14)\nplt.xticks(rotation=45)  # Rotate x-axis labels for better readability\nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"#### Categorical Column Missing Value Impute","metadata":{}},{"cell_type":"code","source":"class HandleOutliers(BaseEstimator, TransformerMixin):\n    def __init__(self, p_min=0.01, p_max=0.99):\n        self.p_min = p_min\n        self.p_max = p_max\n\n    def fit(self, X, y=None):\n        # Fit method is required by the estimator interface but we don't need to do anything here\n        return self\n\n    def transform(self, X):\n        X = pd.DataFrame(X)  # Ensure it's a DataFrame for easy processing\n        for column in X.columns:\n            # Calculate the min and max percentiles\n            p_min_value = X[column].quantile(self.p_min)\n            p_max_value = X[column].quantile(self.p_max)\n\n            # Replace values below min percentile and above max percentile\n            X[column] = X[column].clip(lower=p_min_value, upper=p_max_value)\n\n            # Replace missing values with the median\n            median = X[column].median()\n            X[column] = X[column].fillna(median)\n\n        return X.values  # Return as numpy array for compatibility with sklearn\n\n    def get_feature_names_out(self, input_features=None):\n        # Return the same feature names as input\n        return input_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:21.458219Z","iopub.execute_input":"2024-12-13T00:11:21.458892Z","iopub.status.idle":"2024-12-13T00:11:21.466768Z","shell.execute_reply.started":"2024-12-13T00:11:21.458855Z","shell.execute_reply":"2024-12-13T00:11:21.465501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AgeToRangeTransformer(BaseEstimator, TransformerMixin):\n    def __init__(self, age_column):\n        self.age_column = age_column  # Allow passing the column name dynamically\n\n    def fit(self, X, y=None):\n        return self  # No fitting needed\n\n    def transform(self, X):\n        # Ensure the provided column exists in the data\n        if self.age_column not in X.columns:\n            raise ValueError(f\"Column '{self.age_column}' not found in the input data.\")\n\n        # Handle missing values (adjust as needed)\n        X[self.age_column].fillna(X[self.age_column].median(), inplace=True)\n        \n        # Create the 'age_range' column based on the age column\n        age_range = pd.cut(X[self.age_column], bins=[0, 20, 25, 30, 35, 40, 50, float('inf')],\n                           labels=['Below 20', '20-25', '25-30', '30-35', '35-40', '40-50', '50+'])\n        \n        X['age_range'] = age_range  # Add the new 'age_range' column\n        X = X.drop(columns=[self.age_column])  # Drop the original age column\n        \n        return X\n\n    def get_feature_names_out(self, input_features=None):\n        # Return the feature names after the transformation, which will be 'age_range'\n        return ['age_range']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:22.537723Z","iopub.execute_input":"2024-12-13T00:11:22.538164Z","iopub.status.idle":"2024-12-13T00:11:22.546895Z","shell.execute_reply.started":"2024-12-13T00:11:22.538101Z","shell.execute_reply":"2024-12-13T00:11:22.545788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass NumericalImputer(BaseEstimator, TransformerMixin):\n    \"\"\"\n    Custom transformer to impute numerical columns with specified strategies.\n    \"\"\"\n    def __init__(self, impute_dict=None):\n        \"\"\"\n        Initialize the transformer with a dictionary of imputation strategies.\n\n        Args:\n        - impute_dict (dict): Dictionary mapping column names to imputation strategies.\n          e.g., {'Age': 'median', 'Number of Dependents': 0}\n        \"\"\"\n        self.impute_dict = impute_dict\n\n    def fit(self, X, y=None):\n        \"\"\"\n        Fit the transformer by calculating the required statistics for each column.\n\n        Args:\n        - X (pd.DataFrame): Input data.\n        - y: Ignored.\n        \n        Returns:\n        - self: Fitted transformer.\n        \"\"\"\n        self.statistics_ = {}\n        for col, strategy in self.impute_dict.items():\n            if strategy == \"median\":\n                self.statistics_[col] = X[col].median()\n            elif strategy == \"mean\":\n                self.statistics_[col] = X[col].mean()\n            elif strategy == 0:  # Use 0 for imputation\n                self.statistics_[col] = 0\n            else:\n                raise ValueError(f\"Unsupported strategy: {strategy}\")\n        return self\n\n    def transform(self, X):\n        \"\"\"\n        Transform the data by imputing the specified columns.\n\n        Args:\n        - X (pd.DataFrame): Input data.\n\n        Returns:\n        - pd.DataFrame: Transformed data with imputed values.\n        \"\"\"\n        X = X.copy()\n        for col, value in self.statistics_.items():\n            X[col] = X[col].fillna(value)\n        return X\n\n    def get_feature_names_out(self, input_features=None):\n        \"\"\"\n        Get output feature names for compatibility with ColumnTransformer.\n\n        Args:\n        - input_features (array-like or None): Input feature names. Ignored.\n\n        Returns:\n        - list: Feature names (column names) after transformation.\n        \"\"\"\n        return list(self.impute_dict.keys())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:23.153043Z","iopub.execute_input":"2024-12-13T00:11:23.153459Z","iopub.status.idle":"2024-12-13T00:11:23.162825Z","shell.execute_reply.started":"2024-12-13T00:11:23.153424Z","shell.execute_reply":"2024-12-13T00:11:23.161608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DateTimeConverter(BaseEstimator, TransformerMixin):\n    def __init__(self):\n        self.feature_names_ = None  # Initialize a variable to store feature names\n\n    def fit(self, X, y=None):\n        # Save feature names during the fit step (this is the DataFrame's column names)\n        self.feature_names_ = X.columns\n        return self\n\n    def transform(self, X):\n        # Ensure input is a DataFrame\n        if not isinstance(X, pd.DataFrame):\n            raise ValueError(\"Input must be a pandas DataFrame\")\n\n        # Convert the date columns to datetime format\n        X_transformed = X.apply(pd.to_datetime, errors='coerce')  # Invalid dates will become NaT\n        \n        # Calculate the number of days from today\n        today = datetime.today()\n        days_from_today = X_transformed.apply(lambda col: (today - col).dt.days)\n        \n        return days_from_today\n\n    def get_feature_names_out(self, input_features=None):\n        if self.feature_names_ is None:\n            raise RuntimeError(\"The transformer has not been fit yet.\")\n        return self.feature_names_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:23.721085Z","iopub.execute_input":"2024-12-13T00:11:23.721515Z","iopub.status.idle":"2024-12-13T00:11:23.729491Z","shell.execute_reply.started":"2024-12-13T00:11:23.721471Z","shell.execute_reply":"2024-12-13T00:11:23.728179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class LabelEncoderTransformer(BaseEstimator, TransformerMixin):\n    \"\"\"\n    A custom transformer to apply LabelEncoder to specified columns.\n    \"\"\"\n    def __init__(self, columns=None):\n        \"\"\"\n        Initialize the transformer.\n\n        Args:\n        - columns (list): List of column names to encode.\n        \"\"\"\n        self.columns = columns\n        self.encoders = {}\n\n    def fit(self, X, y=None):\n        \"\"\"\n        Fit LabelEncoder for each specified column.\n\n        Args:\n        - X (pd.DataFrame): Input data.\n        - y: Ignored.\n\n        Returns:\n        - self: Fitted transformer.\n        \"\"\"\n        if self.columns is None:\n            self.columns = X.columns.tolist()\n\n        for col in self.columns:\n            encoder = LabelEncoder()\n            X[col] = X[col].astype(str)  # Ensure all data is string\n            encoder.fit(X[col])\n            self.encoders[col] = encoder\n\n        return self\n\n    def transform(self, X):\n        \"\"\"\n        Transform the data by encoding the specified columns.\n\n        Args:\n        - X (pd.DataFrame): Input data.\n\n        Returns:\n        - pd.DataFrame: Transformed data with encoded values.\n        \"\"\"\n        X = X.copy()\n\n        for col in self.columns:\n            X[col] = X[col].astype(str)  # Ensure all data is string\n            X[col] = self.encoders[col].transform(X[col])\n\n        return X\n\n    def inverse_transform(self, X):\n        \"\"\"\n        Inverse transform the data to decode encoded values back to original categories.\n\n        Args:\n        - X (pd.DataFrame): Input data with encoded values.\n\n        Returns:\n        - pd.DataFrame: Decoded data with original categories.\n        \"\"\"\n        X = X.copy()\n\n        for col in self.columns:\n            X[col] = self.encoders[col].inverse_transform(X[col].astype(int))\n\n        return X\n\n    def get_feature_names_out(self, input_features=None):\n        \"\"\"\n        Get output feature names for compatibility.\n\n        Args:\n        - input_features (array-like or None): Input feature names.\n\n        Returns:\n        - list: Feature names after transformation.\n        \"\"\"\n        return self.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:24.273631Z","iopub.execute_input":"2024-12-13T00:11:24.274005Z","iopub.status.idle":"2024-12-13T00:11:24.283338Z","shell.execute_reply.started":"2024-12-13T00:11:24.273969Z","shell.execute_reply":"2024-12-13T00:11:24.281992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Define the categorical imputation pipeline\ncategorical_impute_pipeline = Pipeline(steps=[\n    ('lab_encode', LabelEncoderTransformer()),  # Encode categorical data\n    ('knn_impute', KNNImputer(n_neighbors=2))  # Apply KNN Imputation\n])\n\nordinal_impute_pipeline = Pipeline(steps=[\n    ('ord_encode', OrdinalEncoder(categories=[['Poor', 'Average', 'Good']],\n                                  handle_unknown='use_encoded_value', unknown_value=-1)),  # Handle NaN\n    ('knn_impute_ord', KNNImputer(n_neighbors=2))  # Apply KNN Imputation\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:24.897267Z","iopub.execute_input":"2024-12-13T00:11:24.897669Z","iopub.status.idle":"2024-12-13T00:11:24.903708Z","shell.execute_reply.started":"2024-12-13T00:11:24.897634Z","shell.execute_reply":"2024-12-13T00:11:24.902515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" df_train['Policy Start Date'].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:25.753301Z","iopub.execute_input":"2024-12-13T00:11:25.753679Z","iopub.status.idle":"2024-12-13T00:11:25.820906Z","shell.execute_reply.started":"2024-12-13T00:11:25.753647Z","shell.execute_reply":"2024-12-13T00:11:25.819843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Create a pipeline for age transformation followed by one-hot encoding\nage_pipeline = Pipeline(steps=[\n    ('age_transformer',  AgeToRangeTransformer(age_column='Age')),  # Apply age transformation\n    ('onehot', OneHotEncoder(handle_unknown='ignore', drop='first'))  # OneHotEncode the age_range column\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:26.888216Z","iopub.execute_input":"2024-12-13T00:11:26.888623Z","iopub.status.idle":"2024-12-13T00:11:26.894168Z","shell.execute_reply.started":"2024-12-13T00:11:26.888588Z","shell.execute_reply":"2024-12-13T00:11:26.892806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Define the imputation strategies\nimpute_strategies = {\n    'Annual Income': 'median',\n    'Number of Dependents': 0,\n    'Health Score': 'mean',\n    'Previous Claims': 0,\n    'Vehicle Age': 'median',\n    'Credit Score': 'median',\n    'Insurance Duration': 'median'\n}\n\n# Create the custom transformer for numerical imputation\nnumerical_imputer = NumericalImputer(impute_dict=impute_strategies)\n\n# Define the numerical pipeline\nnumerical_pipeline = Pipeline(steps=[\n    ('num_processing', numerical_imputer),  # Apply numerical imputation and outlier handling\n    ('outlier_handling', HandleOutliers(p_min=0.0025, p_max=0.9975)),  # Apply outlier handling\n    ('scaling', StandardScaler())  # Apply standard scaling\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:27.889721Z","iopub.execute_input":"2024-12-13T00:11:27.890139Z","iopub.status.idle":"2024-12-13T00:11:27.896584Z","shell.execute_reply.started":"2024-12-13T00:11:27.890084Z","shell.execute_reply":"2024-12-13T00:11:27.895383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"date_columns = ['Policy Start Date']\n\ndate_pipeline = Pipeline(steps = [\n    (\"date_transformation\",DateTimeConverter()),\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:28.785914Z","iopub.execute_input":"2024-12-13T00:11:28.786302Z","iopub.status.idle":"2024-12-13T00:11:28.791348Z","shell.execute_reply.started":"2024-12-13T00:11:28.786268Z","shell.execute_reply":"2024-12-13T00:11:28.790265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nnumerical_columns = numerical_columns.drop(['Age'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:29.535178Z","iopub.execute_input":"2024-12-13T00:11:29.535579Z","iopub.status.idle":"2024-12-13T00:11:29.541326Z","shell.execute_reply.started":"2024-12-13T00:11:29.53553Z","shell.execute_reply":"2024-12-13T00:11:29.540052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['Customer Feedback'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:30.332806Z","iopub.execute_input":"2024-12-13T00:11:30.333228Z","iopub.status.idle":"2024-12-13T00:11:30.431458Z","shell.execute_reply.started":"2024-12-13T00:11:30.33319Z","shell.execute_reply":"2024-12-13T00:11:30.430265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Define the categorical columns to be imputed\ncategorical_impute_columns = ['Marital Status', 'Occupation','Customer Feedback']\n\nremaining_cat_columns = [col for col in categorical_columns if col not in categorical_impute_columns]\n# ColumnTransformer with preprocessor pipeline\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('date_transform',date_pipeline, date_columns ),\n        ('age_transformation',age_pipeline, ['Age']),\n        (\"cat_transformation\", OneHotEncoder(handle_unknown='ignore', drop='first'), remaining_cat_columns),  # OneHotEncode other categorical columns\n        (\"cat_impute_transformation\", categorical_impute_pipeline, ['Marital Status', 'Occupation']),  # Impute categorical columns \n        (\"customer_feedback_transformation\",  ordinal_impute_pipeline,['Customer Feedback']),\n        (\"num_transformation\", numerical_pipeline, numerical_columns),  # Apply numerical transformations\n    ],\n    remainder='passthrough'  # Pass through any columns not explicitly transformed\n)\n \n# Create a pipeline with PCA or any other estimator after preprocessing\npipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor)  # Apply preprocessing for both categorical and numerical columns\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:31.25723Z","iopub.execute_input":"2024-12-13T00:11:31.257621Z","iopub.status.idle":"2024-12-13T00:11:31.264957Z","shell.execute_reply.started":"2024-12-13T00:11:31.257591Z","shell.execute_reply":"2024-12-13T00:11:31.263786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipeline.fit(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:32.337656Z","iopub.execute_input":"2024-12-13T00:11:32.338074Z","iopub.status.idle":"2024-12-13T00:11:39.017612Z","shell.execute_reply.started":"2024-12-13T00:11:32.338035Z","shell.execute_reply":"2024-12-13T00:11:39.016369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transformed_X = pipeline.transform(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:46.404154Z","iopub.execute_input":"2024-12-13T00:11:46.404533Z","iopub.status.idle":"2024-12-13T00:11:52.136379Z","shell.execute_reply.started":"2024-12-13T00:11:46.4045Z","shell.execute_reply":"2024-12-13T00:11:52.13537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transformed_X","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:52.138378Z","iopub.execute_input":"2024-12-13T00:11:52.138733Z","iopub.status.idle":"2024-12-13T00:11:52.146267Z","shell.execute_reply.started":"2024-12-13T00:11:52.1387Z","shell.execute_reply":"2024-12-13T00:11:52.145061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preprocessor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:52.147522Z","iopub.execute_input":"2024-12-13T00:11:52.147845Z","iopub.status.idle":"2024-12-13T00:11:52.224838Z","shell.execute_reply.started":"2024-12-13T00:11:52.147813Z","shell.execute_reply":"2024-12-13T00:11:52.223497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:11:56.641559Z","iopub.execute_input":"2024-12-13T00:11:56.641978Z","iopub.status.idle":"2024-12-13T00:11:56.679372Z","shell.execute_reply.started":"2024-12-13T00:11:56.641942Z","shell.execute_reply":"2024-12-13T00:11:56.67817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # Get the feature names after transformation\ntransformed_feature_names = preprocessor.get_feature_names_out()\ntransformed_feature_names","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:12:07.811829Z","iopub.execute_input":"2024-12-13T00:12:07.812881Z","iopub.status.idle":"2024-12-13T00:12:07.820022Z","shell.execute_reply.started":"2024-12-13T00:12:07.81284Z","shell.execute_reply":"2024-12-13T00:12:07.818969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_transformed_X =  pd.DataFrame(transformed_X, columns=transformed_feature_names)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:12:08.777968Z","iopub.execute_input":"2024-12-13T00:12:08.778424Z","iopub.status.idle":"2024-12-13T00:12:08.784275Z","shell.execute_reply.started":"2024-12-13T00:12:08.778385Z","shell.execute_reply":"2024-12-13T00:12:08.783184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_transformed_X.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:12:09.866131Z","iopub.execute_input":"2024-12-13T00:12:09.866581Z","iopub.status.idle":"2024-12-13T00:12:09.893813Z","shell.execute_reply.started":"2024-12-13T00:12:09.866546Z","shell.execute_reply":"2024-12-13T00:12:09.8926Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Transform Test Data","metadata":{}},{"cell_type":"code","source":"test_transformed = pipeline.transform(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:12:27.668076Z","iopub.execute_input":"2024-12-13T00:12:27.668498Z","iopub.status.idle":"2024-12-13T00:12:31.436777Z","shell.execute_reply.started":"2024-12-13T00:12:27.668461Z","shell.execute_reply":"2024-12-13T00:12:31.435672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test_transformed=  pd.DataFrame(test_transformed, columns=transformed_feature_names)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:12:31.43862Z","iopub.execute_input":"2024-12-13T00:12:31.438969Z","iopub.status.idle":"2024-12-13T00:12:31.444131Z","shell.execute_reply.started":"2024-12-13T00:12:31.438935Z","shell.execute_reply":"2024-12-13T00:12:31.442973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test_transformed.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:12:31.445302Z","iopub.execute_input":"2024-12-13T00:12:31.445687Z","iopub.status.idle":"2024-12-13T00:12:31.530723Z","shell.execute_reply.started":"2024-12-13T00:12:31.445653Z","shell.execute_reply":"2024-12-13T00:12:31.529616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test_transformed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:12:31.532556Z","iopub.execute_input":"2024-12-13T00:12:31.532891Z","iopub.status.idle":"2024-12-13T00:12:31.921809Z","shell.execute_reply.started":"2024-12-13T00:12:31.532858Z","shell.execute_reply":"2024-12-13T00:12:31.920703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Apply Machine Learning Alogrithms","metadata":{}},{"cell_type":"code","source":"X_train,X_test,y_train,y_test = train_test_split(df_transformed_X,y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:12:31.922951Z","iopub.execute_input":"2024-12-13T00:12:31.923291Z","iopub.status.idle":"2024-12-13T00:12:32.13918Z","shell.execute_reply.started":"2024-12-13T00:12:31.923259Z","shell.execute_reply":"2024-12-13T00:12:32.138022Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### XGBoost Regression","metadata":{}},{"cell_type":"code","source":"# Define the XGBoost model for regression\nbest_parameter =  {'random_state': 42,\n                   'grow_policy': 'lossguide',  \n                   'eval_metric': 'rmsle'}\nxgb_model = XGBRegressor(**best_parameter)\n\n# Define hyperparameters for tuning\nparam_grid = {\n    'n_estimators': [   400],\n    'learning_rate': [ 0.1],\n    'max_depth': [ 4, 5],\n    'subsample': [ 1.0],\n    'colsample_bytree': [ 1.0], \n}\n\n# Perform hyperparameter tuning using GridSearchCV\ngrid_search_xgb = GridSearchCV(\n    estimator=xgb_model,\n    param_grid=param_grid,\n    cv=5,\n    scoring='neg_mean_squared_error',  # Use negative MSE as scoring\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:12:33.122084Z","iopub.execute_input":"2024-12-13T00:12:33.123035Z","iopub.status.idle":"2024-12-13T00:12:33.128813Z","shell.execute_reply.started":"2024-12-13T00:12:33.122993Z","shell.execute_reply":"2024-12-13T00:12:33.127689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ngrid_search_xgb.fit(df_transformed_X,y)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:12:35.282563Z","iopub.execute_input":"2024-12-13T00:12:35.282968Z","iopub.status.idle":"2024-12-13T00:15:42.714266Z","shell.execute_reply.started":"2024-12-13T00:12:35.28293Z","shell.execute_reply":"2024-12-13T00:15:42.713064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Print best parameters and best score\nprint(\"Best Parameters:\", grid_search_xgb.best_params_)\nprint(\"Best CV Score (Negative MSE):\", grid_search_xgb.best_score_)\n\n# Get the best model from grid search\nbest_xg_model = grid_search_xgb.best_estimator_\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:15:52.165081Z","iopub.execute_input":"2024-12-13T00:15:52.165482Z","iopub.status.idle":"2024-12-13T00:15:52.171604Z","shell.execute_reply.started":"2024-12-13T00:15:52.165447Z","shell.execute_reply":"2024-12-13T00:15:52.170349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the mean test scores for each parameter combination\nmean_test_scores = grid_search_xgb.cv_results_['mean_test_score']\nstd_test_scores = grid_search_xgb.cv_results_['std_test_score']\n\n# Convert negative MSE to positive MSE for better interpretation\nmean_test_scores = -mean_test_scores  # Convert negative MSE to positive\nstd_test_scores = std_test_scores\n\n# Create a bar chart\nx_labels = [f\"Param Set {i+1}\" for i in range(len(mean_test_scores))]\nx_pos = np.arange(len(mean_test_scores))\n\nplt.figure(figsize=(12, 6))\nplt.bar(x_pos, mean_test_scores, yerr=std_test_scores, capsize=5, color='skyblue', alpha=0.8)\n\n# Add labels and title\nplt.xlabel(\"Parameter Set Index\", fontsize=12)\nplt.ylabel(\"Mean MSE Across CV Folds\", fontsize=12)\nplt.title(\"Mean MSE of Cross-Validation for Each Parameter Set\", fontsize=14)\nplt.xticks(x_pos, x_labels, rotation=90, ha='right', fontsize=10)\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:15:52.819018Z","iopub.execute_input":"2024-12-13T00:15:52.820046Z","iopub.status.idle":"2024-12-13T00:15:53.157273Z","shell.execute_reply.started":"2024-12-13T00:15:52.82Z","shell.execute_reply":"2024-12-13T00:15:53.15618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Create a DataFrame to show the importance of each original feature\nimportance_df = pd.DataFrame({\n    'Original Feature': transformed_feature_names,  # Replace with actual feature names\n    'Importance': best_xg_model.feature_importances_\n}).sort_values(by='Importance', ascending=False)\n\nprint(importance_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:15:53.623605Z","iopub.execute_input":"2024-12-13T00:15:53.624447Z","iopub.status.idle":"2024-12-13T00:15:53.63543Z","shell.execute_reply.started":"2024-12-13T00:15:53.624403Z","shell.execute_reply":"2024-12-13T00:15:53.634257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming the importance_df is already computed as described earlier:\n# Sort the DataFrame by importance in descending order and select top 15 features\ntop_features = importance_df.sort_values(by='Importance', ascending=False).head(15)\n\n# Create a horizontal bar chart\nplt.figure(figsize=(10, 6))\nsns.barplot(x='Importance', y='Original Feature', data=top_features, palette='viridis')\n\n# Add labels and title\nplt.title('Top 15 Important Features', fontsize=16)\nplt.xlabel('Importance', fontsize=14)\nplt.ylabel('Original Feature', fontsize=14)\n\n# Show the plot\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:15:55.260576Z","iopub.execute_input":"2024-12-13T00:15:55.260925Z","iopub.status.idle":"2024-12-13T00:15:55.735597Z","shell.execute_reply.started":"2024-12-13T00:15:55.260894Z","shell.execute_reply":"2024-12-13T00:15:55.734415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Make predictions on the test set\ny_pred_xgb = best_xg_model.predict(df_test_transformed)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:15:55.980812Z","iopub.execute_input":"2024-12-13T00:15:55.98145Z","iopub.status.idle":"2024-12-13T00:15:58.400402Z","shell.execute_reply.started":"2024-12-13T00:15:55.981404Z","shell.execute_reply":"2024-12-13T00:15:58.399554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"output_xgb = pd.DataFrame(y_pred_xgb, index=df_test.index, columns=['Premium Amount']).reset_index()\noutput_xgb['Premium Amount'] = output_xgb['Premium Amount'].round()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:15:58.401963Z","iopub.execute_input":"2024-12-13T00:15:58.402376Z","iopub.status.idle":"2024-12-13T00:15:58.411997Z","shell.execute_reply.started":"2024-12-13T00:15:58.402339Z","shell.execute_reply":"2024-12-13T00:15:58.411193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"output_xgb.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:15:58.413153Z","iopub.execute_input":"2024-12-13T00:15:58.413902Z","iopub.status.idle":"2024-12-13T00:15:58.428577Z","shell.execute_reply.started":"2024-12-13T00:15:58.413863Z","shell.execute_reply":"2024-12-13T00:15:58.427492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = output_xgb.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:16:00.668621Z","iopub.execute_input":"2024-12-13T00:16:00.668992Z","iopub.status.idle":"2024-12-13T00:16:01.712511Z","shell.execute_reply.started":"2024-12-13T00:16:00.66896Z","shell.execute_reply":"2024-12-13T00:16:01.711497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}