{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:05:56.672231Z","iopub.execute_input":"2024-12-08T08:05:56.672738Z","iopub.status.idle":"2024-12-08T08:05:56.679267Z","shell.execute_reply.started":"2024-12-08T08:05:56.672706Z","shell.execute_reply":"2024-12-08T08:05:56.678453Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 回归问题","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.impute import SimpleImputer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:55:22.314644Z","iopub.execute_input":"2024-12-08T08:55:22.315316Z","iopub.status.idle":"2024-12-08T08:55:24.565304Z","shell.execute_reply.started":"2024-12-08T08:55:22.315275Z","shell.execute_reply":"2024-12-08T08:55:24.56436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 载入数据\ntrain_data = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:55:24.566964Z","iopub.execute_input":"2024-12-08T08:55:24.56739Z","iopub.status.idle":"2024-12-08T08:55:29.514128Z","shell.execute_reply.started":"2024-12-08T08:55:24.56736Z","shell.execute_reply":"2024-12-08T08:55:29.513122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:55:29.515473Z","iopub.execute_input":"2024-12-08T08:55:29.515872Z","iopub.status.idle":"2024-12-08T08:55:30.09645Z","shell.execute_reply.started":"2024-12-08T08:55:29.515832Z","shell.execute_reply":"2024-12-08T08:55:30.095517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 数据预处理\n# 提取日期中的年份和月份\ntrain_data['Policy Start Date'] = pd.to_datetime(train_data['Policy Start Date'])\ntrain_data['Policy Year'] = train_data['Policy Start Date'].dt.year\ntrain_data['Policy Month'] = train_data['Policy Start Date'].dt.month\ntrain_data = train_data.drop(columns=['Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:55:30.098436Z","iopub.execute_input":"2024-12-08T08:55:30.098707Z","iopub.status.idle":"2024-12-08T08:55:30.715421Z","shell.execute_reply.started":"2024-12-08T08:55:30.098682Z","shell.execute_reply":"2024-12-08T08:55:30.714678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 选择特征和目标变量\nX = train_data.drop(columns=['Premium Amount', 'id'])\ny = train_data['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:55:30.716513Z","iopub.execute_input":"2024-12-08T08:55:30.717182Z","iopub.status.idle":"2024-12-08T08:55:30.858441Z","shell.execute_reply.started":"2024-12-08T08:55:30.71714Z","shell.execute_reply":"2024-12-08T08:55:30.857715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 对数值型特征进行标准化，对分类特征进行独热编码\nnumeric_features = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n                    'Vehicle Age', 'Credit Score', 'Insurance Duration']\ncategorical_features = ['Gender', 'Marital Status', 'Education Level', 'Occupation', \n                        'Location', 'Policy Type', 'Previous Claims', 'Customer Feedback', \n                        'Smoking Status', 'Exercise Frequency', 'Property Type']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:55:30.859311Z","iopub.execute_input":"2024-12-08T08:55:30.859561Z","iopub.status.idle":"2024-12-08T08:55:30.864039Z","shell.execute_reply.started":"2024-12-08T08:55:30.859536Z","shell.execute_reply":"2024-12-08T08:55:30.863203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 对数值型特征进行标准化，对分类特征进行独热编码\nnumeric_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='mean')),\n    ('scaler', StandardScaler())\n])\n\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:55:30.865054Z","iopub.execute_input":"2024-12-08T08:55:30.865263Z","iopub.status.idle":"2024-12-08T08:55:30.877484Z","shell.execute_reply.started":"2024-12-08T08:55:30.865241Z","shell.execute_reply":"2024-12-08T08:55:30.876822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建预处理器\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numeric_transformer, numeric_features),\n        ('cat', categorical_transformer, categorical_features)\n    ])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T08:55:30.878406Z","iopub.execute_input":"2024-12-08T08:55:30.878633Z","iopub.status.idle":"2024-12-08T08:55:30.887045Z","shell.execute_reply.started":"2024-12-08T08:55:30.878611Z","shell.execute_reply":"2024-12-08T08:55:30.886313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 划分训练集和测试集\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:04:00.113858Z","iopub.execute_input":"2024-12-08T09:04:00.114746Z","iopub.status.idle":"2024-12-08T09:04:00.706833Z","shell.execute_reply.started":"2024-12-08T09:04:00.114712Z","shell.execute_reply":"2024-12-08T09:04:00.706102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 应用预处理器\nX_train = preprocessor.fit_transform(X_train)\nX_val = preprocessor.transform(X_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:04:00.708173Z","iopub.execute_input":"2024-12-08T09:04:00.708421Z","iopub.status.idle":"2024-12-08T09:04:06.864981Z","shell.execute_reply.started":"2024-12-08T09:04:00.708398Z","shell.execute_reply":"2024-12-08T09:04:06.86418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 检查是否有 GPU 可用\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:04:06.86603Z","iopub.execute_input":"2024-12-08T09:04:06.86631Z","iopub.status.idle":"2024-12-08T09:04:06.871226Z","shell.execute_reply.started":"2024-12-08T09:04:06.866279Z","shell.execute_reply":"2024-12-08T09:04:06.87039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 转换为PyTorch的Tensor并移动到GPU\nX_train_tensor = torch.tensor(X_train, dtype=torch.float32).to(device)\nX_val_tensor = torch.tensor(X_val, dtype=torch.float32).to(device)\n\n\ny_train_tensor = torch.tensor(y_train.values, dtype=torch.float32).view(-1, 1).to(device)\ny_val_tensor = torch.tensor(y_val.values, dtype=torch.float32).view(-1, 1).to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:04:06.873468Z","iopub.execute_input":"2024-12-08T09:04:06.873799Z","iopub.status.idle":"2024-12-08T09:04:07.034541Z","shell.execute_reply.started":"2024-12-08T09:04:06.873762Z","shell.execute_reply":"2024-12-08T09:04:07.03354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 定义DNN模型\nclass DNNModel(nn.Module):\n    def __init__(self, input_dim):\n        super(DNNModel, self).__init__()\n        self.fc1 = nn.Linear(input_dim, 64)\n        self.fc2 = nn.Linear(64, 32)\n        self.fc3 = nn.Linear(32, 1)\n    \n    def forward(self, x):\n        x = torch.relu(self.fc1(x))\n        x = torch.relu(self.fc2(x))\n        x = self.fc3(x)\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:04:07.035729Z","iopub.execute_input":"2024-12-08T09:04:07.036094Z","iopub.status.idle":"2024-12-08T09:04:07.042565Z","shell.execute_reply.started":"2024-12-08T09:04:07.036055Z","shell.execute_reply":"2024-12-08T09:04:07.041316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建模型并移动到GPU\nmodel = DNNModel(X_train.shape[1]).to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:04:07.043897Z","iopub.execute_input":"2024-12-08T09:04:07.044238Z","iopub.status.idle":"2024-12-08T09:04:07.0546Z","shell.execute_reply.started":"2024-12-08T09:04:07.044201Z","shell.execute_reply":"2024-12-08T09:04:07.053736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 定义损失函数和优化器\ncriterion = nn.MSELoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)\n\n# 使用动态学习率调整策略\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=10, verbose=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:04:07.055849Z","iopub.execute_input":"2024-12-08T09:04:07.056127Z","iopub.status.idle":"2024-12-08T09:04:07.06707Z","shell.execute_reply.started":"2024-12-08T09:04:07.056103Z","shell.execute_reply":"2024-12-08T09:04:07.066155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 早停法参数\npatience = 200\nbest_val_loss = np.inf\npatience_counter = 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:04:07.068112Z","iopub.execute_input":"2024-12-08T09:04:07.068366Z","iopub.status.idle":"2024-12-08T09:04:07.0802Z","shell.execute_reply.started":"2024-12-08T09:04:07.068342Z","shell.execute_reply":"2024-12-08T09:04:07.079285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 训练模型\nepochs = 1000\nfor epoch in range(epochs):\n    model.train()\n    optimizer.zero_grad()\n    \n    # 前向传播\n    outputs = model(X_train_tensor)\n    \n    loss = criterion(outputs, y_train_tensor)\n    \n    # 反向传播\n    loss.backward()\n    optimizer.step()\n    \n    # 在验证集上评估\n    model.eval()\n    with torch.no_grad():\n        val_outputs = model(X_val_tensor)\n        val_loss = criterion(val_outputs, y_val_tensor)\n    \n    # 动态调整学习率\n    scheduler.step(val_loss)\n\n    # 早停法判断\n    if val_loss < best_val_loss:\n        best_val_loss = val_loss\n        patience_counter = 0\n    else:\n        patience_counter += 1\n    \n    # 如果验证集损失没有改进，停止训练\n    if patience_counter >= patience:\n        print(f\"Early stopping at epoch {epoch+1}\")\n        break\n    \n    if (epoch + 1) % 20 == 0:\n        print(f'Epoch [{epoch+1}/{epochs}], Loss: {loss.item():.4f}, Val Loss: {val_loss.item():.4f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:04:07.082116Z","iopub.execute_input":"2024-12-08T09:04:07.082356Z","iopub.status.idle":"2024-12-08T09:04:22.449943Z","shell.execute_reply.started":"2024-12-08T09:04:07.082332Z","shell.execute_reply":"2024-12-08T09:04:22.448993Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# submission","metadata":{}},{"cell_type":"code","source":"test_data = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\n\n# 处理测试数据\ntest_data['Policy Start Date'] = pd.to_datetime(test_data['Policy Start Date'])\ntest_data['Policy Year'] = test_data['Policy Start Date'].dt.year\ntest_data['Policy Month'] = test_data['Policy Start Date'].dt.month\ntest_data = test_data.drop(columns=['Policy Start Date'])\n\nX_test = test_data.drop(columns=['id'])\nX_test = preprocessor.transform(X_test)\n\n# 转换为Tensor并移动到GPU\nX_test_tensor = torch.tensor(X_test, dtype=torch.float32).to(device)\n\n# 预测\nmodel.eval()\nwith torch.no_grad():\n    test_predictions = model(X_test_tensor).cpu().numpy()  # 转回CPU\n\n# 创建提交文件\nsubmission = pd.DataFrame({\n    'id': test_data['id'],\n    'Premium Amount': test_predictions.flatten()\n})\n\n# 保存提交文件\nsubmission.to_csv(\"submission.csv\", index=False)\n\nprint(f'submission:\\n{submission}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:04:23.892609Z","iopub.execute_input":"2024-12-08T09:04:23.892969Z","iopub.status.idle":"2024-12-08T09:04:31.501439Z","shell.execute_reply.started":"2024-12-08T09:04:23.892939Z","shell.execute_reply":"2024-12-08T09:04:31.500469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}