{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30839,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.preprocessing import MinMaxScaler, LabelEncoder, StandardScaler\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\nnp.seterr(invalid='ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:20:52.071757Z","iopub.execute_input":"2025-02-04T18:20:52.072112Z","iopub.status.idle":"2025-02-04T18:20:54.040338Z","shell.execute_reply.started":"2025-02-04T18:20:52.072077Z","shell.execute_reply":"2025-02-04T18:20:54.039532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train=pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ndf_test=pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\ndf_sample=pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:21:00.542915Z","iopub.execute_input":"2025-02-04T18:21:00.543423Z","iopub.status.idle":"2025-02-04T18:21:10.3696Z","shell.execute_reply.started":"2025-02-04T18:21:00.543392Z","shell.execute_reply":"2025-02-04T18:21:10.368662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:21:10.370699Z","iopub.execute_input":"2025-02-04T18:21:10.370983Z","iopub.status.idle":"2025-02-04T18:21:10.376394Z","shell.execute_reply.started":"2025-02-04T18:21:10.370958Z","shell.execute_reply":"2025-02-04T18:21:10.375598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:21:10.37816Z","iopub.execute_input":"2025-02-04T18:21:10.378447Z","iopub.status.idle":"2025-02-04T18:21:11.042674Z","shell.execute_reply.started":"2025-02-04T18:21:10.378423Z","shell.execute_reply":"2025-02-04T18:21:11.04165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:21:11.043938Z","iopub.execute_input":"2025-02-04T18:21:11.044232Z","iopub.status.idle":"2025-02-04T18:21:11.049922Z","shell.execute_reply.started":"2025-02-04T18:21:11.044209Z","shell.execute_reply":"2025-02-04T18:21:11.04901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df1=df_train.dropna()\ndf1.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:21:11.050886Z","iopub.execute_input":"2025-02-04T18:21:11.051245Z","iopub.status.idle":"2025-02-04T18:21:11.749692Z","shell.execute_reply.started":"2025-02-04T18:21:11.051211Z","shell.execute_reply":"2025-02-04T18:21:11.748649Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(1200000-384004)/1200000","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:21:13.501906Z","iopub.execute_input":"2025-02-04T18:21:13.502355Z","iopub.status.idle":"2025-02-04T18:21:13.508509Z","shell.execute_reply.started":"2025-02-04T18:21:13.50232Z","shell.execute_reply":"2025-02-04T18:21:13.507574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:21:15.090675Z","iopub.execute_input":"2025-02-04T18:21:15.090985Z","iopub.status.idle":"2025-02-04T18:21:15.097933Z","shell.execute_reply.started":"2025-02-04T18:21:15.090961Z","shell.execute_reply":"2025-02-04T18:21:15.097131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns=[\"_\".join(i.split()).lower() for i in df_train.columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:21:18.056758Z","iopub.execute_input":"2025-02-04T18:21:18.057125Z","iopub.status.idle":"2025-02-04T18:21:18.061975Z","shell.execute_reply.started":"2025-02-04T18:21:18.057093Z","shell.execute_reply":"2025-02-04T18:21:18.060954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cat=df_train.select_dtypes(include=[\"object\"])\ndf_num=df_train.select_dtypes(exclude=[\"object\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:21:22.868279Z","iopub.execute_input":"2025-02-04T18:21:22.86864Z","iopub.status.idle":"2025-02-04T18:21:23.035642Z","shell.execute_reply.started":"2025-02-04T18:21:22.86861Z","shell.execute_reply":"2025-02-04T18:21:23.034661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cat.dtypes.index","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:22:09.262503Z","iopub.execute_input":"2025-02-04T18:22:09.262886Z","iopub.status.idle":"2025-02-04T18:22:09.269135Z","shell.execute_reply.started":"2025-02-04T18:22:09.262854Z","shell.execute_reply":"2025-02-04T18:22:09.268089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cat.isnull().sum()[df_cat.isnull().sum()>0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:21:36.623239Z","iopub.execute_input":"2025-02-04T18:21:36.623538Z","iopub.status.idle":"2025-02-04T18:21:37.837306Z","shell.execute_reply.started":"2025-02-04T18:21:36.623516Z","shell.execute_reply":"2025-02-04T18:21:37.836244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cat[\"marital_status\"].value_counts(dropna=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:21:40.677482Z","iopub.execute_input":"2025-02-04T18:21:40.677783Z","iopub.status.idle":"2025-02-04T18:21:40.73032Z","shell.execute_reply.started":"2025-02-04T18:21:40.67776Z","shell.execute_reply":"2025-02-04T18:21:40.729242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cat[\"occupation\"].value_counts(dropna=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:21:42.839804Z","iopub.execute_input":"2025-02-04T18:21:42.840197Z","iopub.status.idle":"2025-02-04T18:21:42.885088Z","shell.execute_reply.started":"2025-02-04T18:21:42.840168Z","shell.execute_reply":"2025-02-04T18:21:42.884137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# customer_feedback\ndf_cat[\"customer_feedback\"].value_counts(dropna=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-04T18:21:45.283868Z","iopub.execute_input":"2025-02-04T18:21:45.284255Z","iopub.status.idle":"2025-02-04T18:21:45.33487Z","shell.execute_reply.started":"2025-02-04T18:21:45.284223Z","shell.execute_reply":"2025-02-04T18:21:45.333846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cat.loc[:, 'marital_status'] = df_cat['marital_status'].fillna('unknown')\n# Define the probabilities\nprobs = df_cat['occupation'].value_counts(normalize=True)\n# Sample based on probabilities\ndf_cat['occupation'] = df_cat['occupation'].apply(\n    lambda x: np.random.choice(probs.index, p=probs.values) if pd.isnull(x) else x\n)\n\nprobs = df_cat['customer_feedback'].value_counts(normalize=True)\n# Sample based on probabilities\ndf_cat['customer_feedback'] = df_cat['customer_feedback'].apply(\n    lambda x: np.random.choice(probs.index, p=probs.values) if pd.isnull(x) else x\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:44:53.285743Z","iopub.execute_input":"2025-02-03T18:44:53.286062Z","iopub.status.idle":"2025-02-03T18:45:04.484872Z","shell.execute_reply.started":"2025-02-03T18:44:53.286038Z","shell.execute_reply":"2025-02-03T18:45:04.484174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cat[\"policy_start_date\"] = pd.to_datetime(df_cat[\"policy_start_date\"])\n\ndf_cat[\"year\"] = df_cat[\"policy_start_date\"].dt.year\ndf_cat[\"month\"] = df_cat[\"policy_start_date\"].dt.month\ndf_cat[\"day\"] = df_cat[\"policy_start_date\"].dt.day\ndf_cat[\"day_of_week\"] = df_cat[\"policy_start_date\"].dt.dayofweek  # Monday=0, Sunday=6\ndf_cat[\"week_of_year\"] = df_cat[\"policy_start_date\"].dt.isocalendar().week\ndf_cat[\"quarter\"] = df_cat[\"policy_start_date\"].dt.quarter\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:04.486089Z","iopub.execute_input":"2025-02-03T18:45:04.486394Z","iopub.status.idle":"2025-02-03T18:45:05.207972Z","shell.execute_reply.started":"2025-02-03T18:45:04.486365Z","shell.execute_reply":"2025-02-03T18:45:05.207306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cat[\"is_weekend\"] = df_cat[\"policy_start_date\"].dt.weekday >= 5  # 1 for Sat/Sun, 0 for others\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:05.209221Z","iopub.execute_input":"2025-02-03T18:45:05.20949Z","iopub.status.idle":"2025-02-03T18:45:05.271386Z","shell.execute_reply.started":"2025-02-03T18:45:05.20946Z","shell.execute_reply":"2025-02-03T18:45:05.270546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pandas.tseries.holiday import USFederalHolidayCalendar\n\ncal = USFederalHolidayCalendar()\nholidays = cal.holidays(start=df_cat[\"policy_start_date\"].min(), end=df_cat[\"policy_start_date\"].max())\n\ndf_cat[\"is_holiday\"] = df_cat[\"policy_start_date\"].isin(holidays)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:05.272683Z","iopub.execute_input":"2025-02-03T18:45:05.272981Z","iopub.status.idle":"2025-02-03T18:45:05.326939Z","shell.execute_reply.started":"2025-02-03T18:45:05.272951Z","shell.execute_reply":"2025-02-03T18:45:05.326093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cat[\"month_sin\"] = np.sin(2 * np.pi * df_cat[\"month\"] / 12)\ndf_cat[\"month_cos\"] = np.cos(2 * np.pi * df_cat[\"month\"] / 12)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:05.32781Z","iopub.execute_input":"2025-02-03T18:45:05.328123Z","iopub.status.idle":"2025-02-03T18:45:05.360645Z","shell.execute_reply.started":"2025-02-03T18:45:05.328093Z","shell.execute_reply":"2025-02-03T18:45:05.359812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cat.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:05.361483Z","iopub.execute_input":"2025-02-03T18:45:05.361711Z","iopub.status.idle":"2025-02-03T18:45:05.88128Z","shell.execute_reply.started":"2025-02-03T18:45:05.361683Z","shell.execute_reply":"2025-02-03T18:45:05.880546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_num.isnull().sum()[df_num.isnull().sum()>0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:05.882157Z","iopub.execute_input":"2025-02-03T18:45:05.882403Z","iopub.status.idle":"2025-02-03T18:45:05.969042Z","shell.execute_reply.started":"2025-02-03T18:45:05.882383Z","shell.execute_reply":"2025-02-03T18:45:05.968203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_num['age']=df_num['age'].fillna(df_num['age'].median())  # Median is better for skewed distributions\ndf_num['annual_income']=df_num['annual_income'].fillna(df_num['annual_income'].mean())\ndf_num['number_of_dependents']=df_num['number_of_dependents'].fillna(df_num['number_of_dependents'].mode()[0])\ndf_num['health_score']=df_num['health_score'].fillna(df_num['health_score'].median())\ndf_num['previous_claims']=df_num['previous_claims'].fillna(0)\ndf_num['vehicle_age']=df_num['vehicle_age'].fillna(df_num['vehicle_age'].mode()[0])\ndf_num['credit_score']=df_num['credit_score'].fillna(df_num['credit_score'].mean())\ndf_num['insurance_duration']=df_num['insurance_duration'].fillna(df_num['insurance_duration'].mode()[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:05.970744Z","iopub.execute_input":"2025-02-03T18:45:05.970981Z","iopub.status.idle":"2025-02-03T18:45:06.166369Z","shell.execute_reply.started":"2025-02-03T18:45:05.97096Z","shell.execute_reply":"2025-02-03T18:45:06.165422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train0=pd.concat([df_cat, df_num], axis=1)\ndf_train0.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:06.167341Z","iopub.execute_input":"2025-02-03T18:45:06.167568Z","iopub.status.idle":"2025-02-03T18:45:06.746589Z","shell.execute_reply.started":"2025-02-03T18:45:06.167549Z","shell.execute_reply":"2025-02-03T18:45:06.745886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming df is your DataFrame\ncategorical_cols = df_train0.select_dtypes(include=['object', 'category']).columns\nlabel_encoders = {}  # Dictionary to store encoders for future use\n\nfor col in categorical_cols:\n    le = LabelEncoder()\n    df_train0[col] = le.fit_transform(df_train0[col])  # Transform categorical values into numerical labels\n    label_encoders[col] = le  # Store the encoder for inverse transformation if needed later\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:06.747736Z","iopub.execute_input":"2025-02-03T18:45:06.747977Z","iopub.status.idle":"2025-02-03T18:45:08.931442Z","shell.execute_reply.started":"2025-02-03T18:45:06.747956Z","shell.execute_reply":"2025-02-03T18:45:08.930759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train0.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:08.932399Z","iopub.execute_input":"2025-02-03T18:45:08.932608Z","iopub.status.idle":"2025-02-03T18:45:08.938759Z","shell.execute_reply.started":"2025-02-03T18:45:08.93259Z","shell.execute_reply":"2025-02-03T18:45:08.938107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train0.drop(columns=[\"policy_start_date\"], inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:09.770956Z","iopub.execute_input":"2025-02-03T18:45:09.771288Z","iopub.status.idle":"2025-02-03T18:45:09.874499Z","shell.execute_reply.started":"2025-02-03T18:45:09.771264Z","shell.execute_reply":"2025-02-03T18:45:09.873764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def detect_outliers_iqr(df, columns):\n    outlier_indices = {}\n    for col in columns:\n        Q1 = df[col].quantile(0.25)\n        Q3 = df[col].quantile(0.75)\n        IQR = Q3 - Q1\n        lower_bound = Q1 - 1.5 * IQR\n        upper_bound = Q3 + 1.5 * IQR\n\n        outliers = df[(df[col] < lower_bound) | (df[col] > upper_bound)].index\n        outlier_indices[col] = outliers\n\n    return outlier_indices\n\n# Columns to check for outliers (excluding categorical/binary)\nnumerical_cols = [\n    \"age\", \"annual_income\", \"number_of_dependents\", \"health_score\",\n    \"previous_claims\", \"vehicle_age\", \"credit_score\", \"insurance_duration\",\n    \"premium_amount\"\n]\n\noutliers_iqr = detect_outliers_iqr(df_train0, numerical_cols)\noutliers_iqr\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:15.032897Z","iopub.execute_input":"2025-02-03T18:45:15.033226Z","iopub.status.idle":"2025-02-03T18:45:15.495455Z","shell.execute_reply.started":"2025-02-03T18:45:15.033199Z","shell.execute_reply":"2025-02-03T18:45:15.494735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy import stats\n\ndef detect_outliers_zscore(df, columns, threshold=3):\n    outlier_indices = {}\n    for col in columns:\n        z_scores = np.abs(stats.zscore(df[col]))\n        outliers = df[z_scores > threshold].index\n        outlier_indices[col] = outliers\n    return outlier_indices\n\noutliers_zscore = detect_outliers_zscore(df_train0, numerical_cols)\noutliers_zscore\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:26.329294Z","iopub.execute_input":"2025-02-03T18:45:26.32976Z","iopub.status.idle":"2025-02-03T18:45:26.471287Z","shell.execute_reply.started":"2025-02-03T18:45:26.329734Z","shell.execute_reply":"2025-02-03T18:45:26.470272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import TensorDataset, DataLoader\nimport pandas as pd\nimport numpy as np\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\n\n# --------------- Identify Features ---------------\n# Categorical (label-encoded)\ncategorical_cols = [\"year\", \"month\", \"day\", \"day_of_week\", \"week_of_year\", \"quarter\", \"is_weekend\", \"is_holiday\"]\n\n# Continuous numerical columns (to be standardized)\nnumerical_cols = [\"age\", \"annual_income\", \"number_of_dependents\", \"health_score\",\n                  \"previous_claims\", \"vehicle_age\", \"credit_score\", \"insurance_duration\"]\n\nall_numerical_cols=categorical_cols+numerical_cols\n# all_numerical_cols\ntarget_col=\"premium_amount\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:28.855519Z","iopub.execute_input":"2025-02-03T18:45:28.855847Z","iopub.status.idle":"2025-02-03T18:45:28.860911Z","shell.execute_reply.started":"2025-02-03T18:45:28.855809Z","shell.execute_reply":"2025-02-03T18:45:28.859774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🔹 Standardize all numerical features\nscaler = StandardScaler()\ndf_train0[all_numerical_cols] = scaler.fit_transform(df_train0[all_numerical_cols])\n\n# 🔹 Split Data\nX = df_train0[all_numerical_cols].values\ny = df_train0[target_col].values\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:34.30736Z","iopub.execute_input":"2025-02-03T18:45:34.307687Z","iopub.status.idle":"2025-02-03T18:45:36.049783Z","shell.execute_reply.started":"2025-02-03T18:45:34.307662Z","shell.execute_reply":"2025-02-03T18:45:36.049043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:39.244947Z","iopub.execute_input":"2025-02-03T18:45:39.245252Z","iopub.status.idle":"2025-02-03T18:45:39.250774Z","shell.execute_reply.started":"2025-02-03T18:45:39.245229Z","shell.execute_reply":"2025-02-03T18:45:39.249832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check if GPU is available\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f'Using device: {device}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:45:40.761091Z","iopub.execute_input":"2025-02-03T18:45:40.761367Z","iopub.status.idle":"2025-02-03T18:45:40.845876Z","shell.execute_reply.started":"2025-02-03T18:45:40.761346Z","shell.execute_reply":"2025-02-03T18:45:40.845105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:46:16.059256Z","iopub.execute_input":"2025-02-03T18:46:16.059544Z","iopub.status.idle":"2025-02-03T18:46:16.064623Z","shell.execute_reply.started":"2025-02-03T18:46:16.059522Z","shell.execute_reply":"2025-02-03T18:46:16.063846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert to PyTorch tensors\nX_train_tensor = torch.tensor(X_train, dtype=torch.float32)\ny_train_tensor = torch.tensor(y_train, dtype=torch.float32).view(-1, 1)\nX_test_tensor = torch.tensor(X_test, dtype=torch.float32)\ny_test_tensor = torch.tensor(y_test, dtype=torch.float32).view(-1, 1)\n\n# Create DataLoader for batching\ntrain_data = TensorDataset(X_train_tensor, y_train_tensor)\ntest_data = TensorDataset(X_test_tensor, y_test_tensor)\n\ntrain_loader = DataLoader(train_data, batch_size=64, shuffle=True)  # Batching and shuffling for training\ntest_loader = DataLoader(test_data, batch_size=64, shuffle=False)  # No shuffling for evaluation\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Define the neural network model with more layers\nclass ComplexRegressionNN(nn.Module):\n    def __init__(self):\n        super(ComplexRegressionNN, self).__init__()\n        self.layer1 = nn.Linear(16, 128)  # Input to 1st hidden layer\n        self.batch_norm1 = nn.BatchNorm1d(128)  # Batch Normalization\n        self.dropout1 = nn.Dropout(0.3)  # Dropout to prevent overfitting\n        \n        self.layer2 = nn.Linear(128, 256)  # 1st to 2nd hidden layer\n        self.batch_norm2 = nn.BatchNorm1d(256)\n        self.dropout2 = nn.Dropout(0.3)\n        \n        self.layer3 = nn.Linear(256, 512)  # 2nd to 3rd hidden layer\n        self.batch_norm3 = nn.BatchNorm1d(512)\n        self.dropout3 = nn.Dropout(0.3)\n        \n        self.layer4 = nn.Linear(512, 256)  # 3rd to 4th hidden layer\n        self.batch_norm4 = nn.BatchNorm1d(256)\n        self.dropout4 = nn.Dropout(0.3)\n        \n        self.layer5 = nn.Linear(256, 128)  # 4th to 5th hidden layer\n        self.batch_norm5 = nn.BatchNorm1d(128)\n        self.dropout5 = nn.Dropout(0.3)\n        \n        self.output_layer = nn.Linear(128, 1)  # 5th hidden layer to output layer\n\n    def forward(self, x):\n        x = torch.relu(self.batch_norm1(self.layer1(x)))  # ReLU + BatchNorm1\n        x = self.dropout1(x)  # Dropout\n\n        x = torch.relu(self.batch_norm2(self.layer2(x)))  # ReLU + BatchNorm2\n        x = self.dropout2(x)  # Dropout\n        \n        x = torch.relu(self.batch_norm3(self.layer3(x)))  # ReLU + BatchNorm3\n        x = self.dropout3(x)  # Dropout\n        \n        x = torch.relu(self.batch_norm4(self.layer4(x)))  # ReLU + BatchNorm4\n        x = self.dropout4(x)  # Dropout\n        \n        x = torch.relu(self.batch_norm5(self.layer5(x)))  # ReLU + BatchNorm5\n        x = self.dropout5(x)  # Dropout\n        \n        x = self.output_layer(x)  # Output layer (no activation for regression)\n        return x\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Instantiate the model and move it to GPU if available\nmodel = ComplexRegressionNN().to(device)\ncriterion = nn.MSELoss()  # Mean Squared Error Loss for regression\noptimizer = optim.Adam(model.parameters(), lr=0.001)\n\n# Train the model with DataLoader on GPU\nnum_epochs = 50\nfor epoch in range(num_epochs):\n    model.train()\n\n    running_loss = 0.0\n    for inputs, labels in train_loader:\n        # Move inputs and labels to the GPU\n        inputs, labels = inputs.to(device), labels.to(device)\n\n        # Forward pass\n        outputs = model(inputs)\n        loss = criterion(outputs, labels)\n\n        # Backward pass and optimization\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        running_loss += loss.item()\n\n    avg_loss = running_loss / len(train_loader)\n    if (epoch + 1) % 50 == 0:\n        print(f'Epoch [{epoch+1}/{num_epochs}], Loss: {avg_loss:.4f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T18:48:44.784534Z","iopub.execute_input":"2025-02-03T18:48:44.784818Z","iopub.status.idle":"2025-02-03T19:34:33.036277Z","shell.execute_reply.started":"2025-02-03T18:48:44.784795Z","shell.execute_reply":"2025-02-03T19:34:33.035357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.eval()\ntest_loss = 0.0\nwith torch.no_grad():\n    for inputs, labels in test_loader:\n        # Move inputs and labels to the GPU\n        inputs, labels = inputs.to(device), labels.to(device)\n\n        y_pred = model(inputs)\n        loss = criterion(y_pred, labels)\n        test_loss += loss.item()\n\navg_test_loss = test_loss / len(test_loader)\nprint(f'Test Loss: {avg_test_loss:.4f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T19:39:34.485066Z","iopub.execute_input":"2025-02-03T19:39:34.485367Z","iopub.status.idle":"2025-02-03T19:39:39.190826Z","shell.execute_reply.started":"2025-02-03T19:39:34.485346Z","shell.execute_reply":"2025-02-03T19:39:39.190114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport numpy as np\n\n# Assuming the model is already trained and you're using the same device (GPU/CPU)\n# Assuming you've defined and fitted a scaler (e.g., StandardScaler) for scaling the input features\n\n# Step 1: Prepare a single input for prediction (16 features)\n# Replace these values with the actual input values for prediction\nsingle_input = np.array([[25, 50000, 2, 80, 1, 5, 750, 3, 2022, 12, 15, 2, 50, 3, 1, 0]])  # Example\n\n# Step 2: Standardize the input using the same scaler used during training\nsingle_input_scaled = scaler.transform(single_input)  # Apply the same scaling as training data\n\n# Step 3: Convert the input to a PyTorch tensor\nsingle_input_tensor = torch.tensor(single_input_scaled, dtype=torch.float32)\n\n# Step 4: Move the input tensor to the same device (GPU/CPU) as the model\nsingle_input_tensor = single_input_tensor.to(device)\n\n# Step 5: Set the model to evaluation mode (important to disable dropout and batch norm)\nmodel.eval()\n\n# Step 6: Perform prediction (without gradient calculation)\nwith torch.no_grad():\n    prediction = model(single_input_tensor)\n\n# Step 7: Convert the prediction to CPU and then to NumPy for easier interpretation\nprediction = prediction.cpu().numpy()\n\n# Step 8: Print the prediction\nprint(f\"Prediction for input {single_input[0]}: {prediction[0][0]:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T19:44:35.0207Z","iopub.execute_input":"2025-02-03T19:44:35.021008Z","iopub.status.idle":"2025-02-03T19:44:35.05599Z","shell.execute_reply.started":"2025-02-03T19:44:35.020985Z","shell.execute_reply":"2025-02-03T19:44:35.055105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prediction","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T19:45:01.011531Z","iopub.execute_input":"2025-02-03T19:45:01.01184Z","iopub.status.idle":"2025-02-03T19:45:01.01779Z","shell.execute_reply.started":"2025-02-03T19:45:01.011816Z","shell.execute_reply":"2025-02-03T19:45:01.016767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-03T19:45:41.884333Z","iopub.execute_input":"2025-02-03T19:45:41.884662Z","iopub.status.idle":"2025-02-03T19:45:41.908055Z","shell.execute_reply.started":"2025-02-03T19:45:41.884634Z","shell.execute_reply":"2025-02-03T19:45:41.9073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}