{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-16T17:12:08.034486Z","iopub.execute_input":"2024-12-16T17:12:08.034788Z","iopub.status.idle":"2024-12-16T17:12:08.041992Z","shell.execute_reply.started":"2024-12-16T17:12:08.034761Z","shell.execute_reply":"2024-12-16T17:12:08.041127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nimport torch\nfrom transformers import AutoModel, AutoTokenizer, AutoConfig\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch import nn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T17:12:08.756051Z","iopub.execute_input":"2024-12-16T17:12:08.756419Z","iopub.status.idle":"2024-12-16T17:12:08.761557Z","shell.execute_reply.started":"2024-12-16T17:12:08.756387Z","shell.execute_reply":"2024-12-16T17:12:08.760636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.random.seed(42)\ntorch.manual_seed(42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T17:12:09.444851Z","iopub.execute_input":"2024-12-16T17:12:09.445679Z","iopub.status.idle":"2024-12-16T17:12:09.45445Z","shell.execute_reply.started":"2024-12-16T17:12:09.445643Z","shell.execute_reply":"2024-12-16T17:12:09.453536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ninference_data = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ndata.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def fill_missing_values(df, inference_data):\n    \"\"\"\n    Fill missing values in the DataFrame.\n    - For numerical columns: fill with the mean.\n    - For categorical columns: fill with the mode.\n\n    Parameters:\n        df (pd.DataFrame): Input DataFrame with potential missing values.\n\n    Returns:\n        pd.DataFrame: DataFrame with missing values filled.\n    \"\"\"\n    for column in df.columns:\n        if df[column].dtype == 'object':  # Categorical column\n            df[column] = df[column].fillna(df[column].mode()[0])\n            inference_data[column] = inference_data[column].fillna(df[column].mode()[0])\n        elif column != 'Premium Amount':  # Numerical column\n            df[column] = df[column].fillna(df[column].mean())\n            inference_data[column] = inference_data[column].fillna(df[column].mode()[0])\n    return df, inference_data\n\n# Fill missing values in the dataset\ndata, inference_data = fill_missing_values(data, inference_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T17:14:13.042327Z","iopub.execute_input":"2024-12-16T17:14:13.042674Z","iopub.status.idle":"2024-12-16T17:14:14.014971Z","shell.execute_reply.started":"2024-12-16T17:14:13.042644Z","shell.execute_reply":"2024-12-16T17:14:14.014141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.isna().sum()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_features = data.select_dtypes(include=[\"object\", \"category\"]).columns.tolist()\nnumerical_features = data.select_dtypes(include=[\"int64\", \"float64\"]).columns.tolist()\nnumerical_features.remove('Premium Amount')\ntarget_col = 'Premium Amount'","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_encoders = {}\nfor col in categorical_features:\n    le = LabelEncoder()\n    data[col] = le.fit_transform(data[col])\n    label_encoders[col] = le\n\n# # Scale numerical features\n# scaler = StandardScaler()\n# data[numerical_features] = scaler.fit_transform(data[numerical_features])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = data[categorical_features + numerical_features]\ny = data[target_col]\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T16:51:35.706116Z","iopub.execute_input":"2024-12-16T16:51:35.706468Z","iopub.status.idle":"2024-12-16T16:51:36.240556Z","shell.execute_reply.started":"2024-12-16T16:51:35.706441Z","shell.execute_reply":"2024-12-16T16:51:36.239835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RegressionDataset(Dataset):\n    def __init__(self, X, y=None):\n        self.X = torch.tensor(X.values, dtype=torch.float32)\n        self.y = torch.tensor(y.values, dtype=torch.float32) if y is not None else None\n\n    def __len__(self):\n        return len(self.X)\n\n    def __getitem__(self, idx):\n        if self.y is not None:\n            return self.X[idx], self.y[idx]\n        return self.X[idx]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T16:51:38.153312Z","iopub.execute_input":"2024-12-16T16:51:38.1537Z","iopub.status.idle":"2024-12-16T16:51:38.159248Z","shell.execute_reply.started":"2024-12-16T16:51:38.153669Z","shell.execute_reply":"2024-12-16T16:51:38.158363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = RegressionDataset(X_train, y_train)\ntest_dataset = RegressionDataset(X_test, y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T16:51:39.97902Z","iopub.execute_input":"2024-12-16T16:51:39.97986Z","iopub.status.idle":"2024-12-16T16:51:40.117578Z","shell.execute_reply.started":"2024-12-16T16:51:39.979825Z","shell.execute_reply":"2024-12-16T16:51:40.116737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_loader = DataLoader(train_dataset, batch_size=32, shuffle=True)\ntest_loader = DataLoader(test_dataset, batch_size=32, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T16:51:41.616142Z","iopub.execute_input":"2024-12-16T16:51:41.616503Z","iopub.status.idle":"2024-12-16T16:51:41.622259Z","shell.execute_reply.started":"2024-12-16T16:51:41.616472Z","shell.execute_reply":"2024-12-16T16:51:41.621332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ComplexRegressionModel(nn.Module):\n    def __init__(self, input_dim):\n        super(ComplexRegressionModel, self).__init__()\n        self.layer1 = nn.Sequential(\n            nn.Linear(input_dim, 256),\n            nn.ReLU(),\n            nn.BatchNorm1d(256),\n            nn.Dropout(0.4)\n        )\n        self.layer2 = nn.Sequential(\n            nn.Linear(256, 128),\n            nn.ReLU(),\n            nn.BatchNorm1d(128),\n            nn.Dropout(0.4)\n        )\n        self.layer3 = nn.Sequential(\n            nn.Linear(128, 64),\n            nn.ReLU(),\n            nn.BatchNorm1d(64),\n            nn.Dropout(0.4)\n        )\n        self.layer4 = nn.Sequential(\n            nn.Linear(64, 32),\n            nn.ReLU()\n        )\n        self.output = nn.Linear(32, 1)\n\n    def forward(self, x):\n        x = self.layer1(x)\n        # print(f\"After layer1: {x}\")\n        x = self.layer2(x)\n        # print(f\"After layer2: {x}\")\n        x = self.layer3(x)\n        # print(f\"After layer3: {x}\")\n        x = self.layer4(x)\n        # print(f\"After layer4: {x}\")\n        x = self.output(x)\n        # print(f\"Output: {x}\")\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T16:51:44.940129Z","iopub.execute_input":"2024-12-16T16:51:44.940799Z","iopub.status.idle":"2024-12-16T16:51:44.947481Z","shell.execute_reply.started":"2024-12-16T16:51:44.940768Z","shell.execute_reply":"2024-12-16T16:51:44.946531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\ninput_dim = X_train.shape[1]\nmodel = ComplexRegressionModel(input_dim).to(device)\ncriterion = nn.MSELoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-5)\nprint(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T16:51:45.985738Z","iopub.execute_input":"2024-12-16T16:51:45.986093Z","iopub.status.idle":"2024-12-16T16:51:47.416652Z","shell.execute_reply.started":"2024-12-16T16:51:45.986064Z","shell.execute_reply":"2024-12-16T16:51:47.41579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# RMSLE metric\ndef rmsle(y_true, y_pred):\n    y_true = torch.clamp(y_true, min=0)  # Убедимся, что значения >= 0\n    y_pred = torch.clamp(y_pred, min=0)  # Убедимся, что предсказания >= 0\n    return torch.sqrt(torch.mean((torch.log1p(y_pred) - torch.log1p(y_true))**2))\n\n# Training loop\ndef train(model, loader, criterion, optimizer):\n    model.train()\n    total_loss = 0\n    total_rmsle = 0\n    for X_batch, y_batch in loader:\n        X_batch, y_batch = X_batch.to(device), y_batch.to(device)\n\n        optimizer.zero_grad()\n        outputs = model(X_batch).squeeze()\n\n        loss = criterion(outputs, y_batch)\n        loss.backward()\n        optimizer.step()\n\n        total_loss += loss.item()\n        total_rmsle += rmsle(y_batch, outputs).item()\n            \n        # break\n\n    avg_loss = total_loss / len(loader)\n    avg_rmsle = total_rmsle / len(loader)\n    return avg_loss, avg_rmsle\n\n# Evaluation loop\ndef evaluate(model, loader):\n    model.eval()\n    total_rmsle = 0\n    with torch.no_grad():\n        for X_batch, y_batch in loader:\n            # print('uwbiuwiuwdiuwiuwquwuu')\n            # print(X_batch)\n            X_batch, y_batch = X_batch.to(device), y_batch.to(device)\n            outputs = model(X_batch).squeeze()\n            total_rmsle += rmsle(y_batch, outputs).item()\n            # print(outputs)\n            # break\n    return total_rmsle / len(loader)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T16:51:48.901357Z","iopub.execute_input":"2024-12-16T16:51:48.902788Z","iopub.status.idle":"2024-12-16T16:51:48.912845Z","shell.execute_reply.started":"2024-12-16T16:51:48.902742Z","shell.execute_reply":"2024-12-16T16:51:48.912016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training process\nn_epochs = 6\nfor epoch in range(n_epochs):\n    train_loss, train_rmsle = train(model, train_loader, criterion, optimizer)\n    val_rmsle = evaluate(model, test_loader)\n    print(f\"Epoch {epoch + 1}/{n_epochs}, Loss: {train_loss:.4f}, Train RMSLE: {train_rmsle:.4f}, Validation RMSLE: {val_rmsle:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T16:51:50.643211Z","iopub.execute_input":"2024-12-16T16:51:50.643613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}