{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport seaborn as sb\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:56:51.068014Z","iopub.execute_input":"2024-12-03T10:56:51.068417Z","iopub.status.idle":"2024-12-03T10:56:52.418372Z","shell.execute_reply.started":"2024-12-03T10:56:51.068365Z","shell.execute_reply":"2024-12-03T10:56:52.41736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nsample_submission = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")\npd.set_option('display.max_columns', 500)\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:56:52.420098Z","iopub.execute_input":"2024-12-03T10:56:52.42057Z","iopub.status.idle":"2024-12-03T10:57:03.883629Z","shell.execute_reply.started":"2024-12-03T10:56:52.420535Z","shell.execute_reply":"2024-12-03T10:57:03.88237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_mean = [\"Age\", \"Annual Income\", \"Number of Dependents\", \"Health Score\", \"Previous Claims\", \"Credit Score\", \"Insurance Duration\"]\ntarget_mode = [\"Marital Status\", \"Occupation\", \"Customer Feedback\", \"Vehicle Age\"]\n\nfor x in target_mean:\n    train[x] = train[x].fillna(train[x].mean())\n    test[x] = test[x].fillna(test[x].mean())\n\nfor x in target_mode:\n    train[x] = train[x].fillna(train[x].mode()[0])\n    test[x] = test[x].fillna(test[x].mode()[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:57:03.884912Z","iopub.execute_input":"2024-12-03T10:57:03.885336Z","iopub.status.idle":"2024-12-03T10:57:05.021153Z","shell.execute_reply.started":"2024-12-03T10:57:03.88529Z","shell.execute_reply":"2024-12-03T10:57:05.020261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_columns = ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location', 'Policy Type', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type']\ntrain = pd.get_dummies(train, columns=target_columns)\ntest = pd.get_dummies(test, columns=target_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:57:05.023215Z","iopub.execute_input":"2024-12-03T10:57:05.023596Z","iopub.status.idle":"2024-12-03T10:57:07.025597Z","shell.execute_reply.started":"2024-12-03T10:57:05.023563Z","shell.execute_reply":"2024-12-03T10:57:07.024437Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 数値データを標準化\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\n\nscaler = StandardScaler()\ntrain[target_mean] = scaler.fit_transform(train[target_mean])\ntest[target_mean] = scaler.fit_transform(test[target_mean])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:57:07.027135Z","iopub.execute_input":"2024-12-03T10:57:07.027493Z","iopub.status.idle":"2024-12-03T10:57:07.326604Z","shell.execute_reply.started":"2024-12-03T10:57:07.027432Z","shell.execute_reply":"2024-12-03T10:57:07.325426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 説明変数と目的変数に分ける & いらなそうなデータの消去\nX = train.drop(columns=[\"id\", \"Policy Start Date\", \"Premium Amount\"])\nY = train['Premium Amount']\nX_test = test.drop(columns=[\"id\", \"Policy Start Date\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:57:07.328228Z","iopub.execute_input":"2024-12-03T10:57:07.329108Z","iopub.status.idle":"2024-12-03T10:57:07.438173Z","shell.execute_reply.started":"2024-12-03T10:57:07.32907Z","shell.execute_reply":"2024-12-03T10:57:07.437235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_features = np.where(X.dtypes != float)[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:57:07.439386Z","iopub.execute_input":"2024-12-03T10:57:07.439748Z","iopub.status.idle":"2024-12-03T10:57:07.445354Z","shell.execute_reply.started":"2024-12-03T10:57:07.439716Z","shell.execute_reply":"2024-12-03T10:57:07.44417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(X, Y, test_size=0.2, random_state=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:57:07.44659Z","iopub.execute_input":"2024-12-03T10:57:07.447Z","iopub.status.idle":"2024-12-03T10:57:07.999882Z","shell.execute_reply.started":"2024-12-03T10:57:07.446968Z","shell.execute_reply":"2024-12-03T10:57:07.998902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Poolクラスの利用\nfrom catboost import Pool\n\ntrain_pool = Pool(data=X_train, label=y_train, cat_features=cat_features)\nval_pool = Pool(data=X_val, label=y_val, cat_features=cat_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:57:08.001128Z","iopub.execute_input":"2024-12-03T10:57:08.001479Z","iopub.status.idle":"2024-12-03T10:58:31.596793Z","shell.execute_reply.started":"2024-12-03T10:57:08.001422Z","shell.execute_reply":"2024-12-03T10:58:31.595699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CatBoostClassifierクラスでモデル構築\nfrom catboost import CatBoostClassifier\nfrom catboost import CatBoostRegressor\n\nmodel = CatBoostRegressor(\n    iterations=500,\n    depth=10,\n    loss_function='MAE',\n    #eval_metric='Accuracy',\n    l2_leaf_reg=20.0,\n    verbose=50,\n    early_stopping_rounds=100,\n    #subsample=0.8,\n    #boosting_type='Ordered',\n    #one_hot_max_size=2\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:58:31.599737Z","iopub.execute_input":"2024-12-03T10:58:31.600065Z","iopub.status.idle":"2024-12-03T10:58:31.60806Z","shell.execute_reply.started":"2024-12-03T10:58:31.600035Z","shell.execute_reply":"2024-12-03T10:58:31.606898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# モデルの学習\nmodel.fit(train_pool, eval_set=val_pool, use_best_model=True, early_stopping_rounds=10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:58:31.609491Z","iopub.execute_input":"2024-12-03T10:58:31.609903Z","iopub.status.idle":"2024-12-03T10:59:26.125618Z","shell.execute_reply.started":"2024-12-03T10:58:31.60986Z","shell.execute_reply":"2024-12-03T10:59:26.124594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:59:26.127058Z","iopub.execute_input":"2024-12-03T10:59:26.127498Z","iopub.status.idle":"2024-12-03T10:59:26.137602Z","shell.execute_reply.started":"2024-12-03T10:59:26.127427Z","shell.execute_reply":"2024-12-03T10:59:26.136483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# テストデータの予測\npredictions = model.predict(X_test)\n\n# データフレーム作成\nsubmission = pd.DataFrame({\n    'id': test['id'],\n    'Premium Amount': predictions\n})\n\n# csvで出力\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T10:59:26.139138Z","iopub.execute_input":"2024-12-03T10:59:26.140129Z","iopub.status.idle":"2024-12-03T11:00:24.320413Z","shell.execute_reply.started":"2024-12-03T10:59:26.140094Z","shell.execute_reply":"2024-12-03T11:00:24.319241Z"}},"outputs":[],"execution_count":null}]}