{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:13.534819Z","iopub.execute_input":"2024-12-29T03:47:13.535471Z","iopub.status.idle":"2024-12-29T03:47:13.546257Z","shell.execute_reply.started":"2024-12-29T03:47:13.535419Z","shell.execute_reply":"2024-12-29T03:47:13.544792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dfSubmission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:13.548367Z","iopub.execute_input":"2024-12-29T03:47:13.54889Z","iopub.status.idle":"2024-12-29T03:47:13.79659Z","shell.execute_reply.started":"2024-12-29T03:47:13.548836Z","shell.execute_reply":"2024-12-29T03:47:13.795503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train=pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntrain.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:13.799112Z","iopub.execute_input":"2024-12-29T03:47:13.799549Z","iopub.status.idle":"2024-12-29T03:47:18.862391Z","shell.execute_reply.started":"2024-12-29T03:47:13.799508Z","shell.execute_reply":"2024-12-29T03:47:18.861194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test=pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ntest.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:18.86395Z","iopub.execute_input":"2024-12-29T03:47:18.864587Z","iopub.status.idle":"2024-12-29T03:47:21.883623Z","shell.execute_reply.started":"2024-12-29T03:47:18.864555Z","shell.execute_reply":"2024-12-29T03:47:21.88251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:21.884589Z","iopub.execute_input":"2024-12-29T03:47:21.884912Z","iopub.status.idle":"2024-12-29T03:47:22.504804Z","shell.execute_reply.started":"2024-12-29T03:47:21.884874Z","shell.execute_reply":"2024-12-29T03:47:22.503849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_x=train.drop(['Premium Amount'], axis=1)\ntrain_y=['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:22.505762Z","iopub.execute_input":"2024-12-29T03:47:22.50605Z","iopub.status.idle":"2024-12-29T03:47:22.735809Z","shell.execute_reply.started":"2024-12-29T03:47:22.506003Z","shell.execute_reply":"2024-12-29T03:47:22.734881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_x=test.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:22.737301Z","iopub.execute_input":"2024-12-29T03:47:22.737695Z","iopub.status.idle":"2024-12-29T03:47:22.889449Z","shell.execute_reply.started":"2024-12-29T03:47:22.737654Z","shell.execute_reply":"2024-12-29T03:47:22.888494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:22.890455Z","iopub.execute_input":"2024-12-29T03:47:22.890865Z","iopub.status.idle":"2024-12-29T03:47:23.592171Z","shell.execute_reply.started":"2024-12-29T03:47:22.890822Z","shell.execute_reply":"2024-12-29T03:47:23.590992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:23.595144Z","iopub.execute_input":"2024-12-29T03:47:23.595466Z","iopub.status.idle":"2024-12-29T03:47:24.26617Z","shell.execute_reply.started":"2024-12-29T03:47:23.595436Z","shell.execute_reply":"2024-12-29T03:47:24.264978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns=list(train_x.columns)\nnumCol = []  # 수치형 컬럼\ncatCol = []  # 문자형 컬럼\n\nfor col in columns:\n    if train[col].dtype in ['int64', 'float64']:  # 수치형 데이터 타입 확인\n        numCol.append(col)\n    elif train[col].dtype == 'object':  # 문자형 데이터 타입 확인\n        catCol.append(col)\n\n# 결과 출력\nprint(\"Numerical Columns:\", numCol)\nprint(\"Categorical Columns:\", catCol)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:24.26827Z","iopub.execute_input":"2024-12-29T03:47:24.268579Z","iopub.status.idle":"2024-12-29T03:47:24.276811Z","shell.execute_reply.started":"2024-12-29T03:47:24.268551Z","shell.execute_reply":"2024-12-29T03:47:24.275769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target=['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:24.277971Z","iopub.execute_input":"2024-12-29T03:47:24.278395Z","iopub.status.idle":"2024-12-29T03:47:24.298972Z","shell.execute_reply.started":"2024-12-29T03:47:24.278356Z","shell.execute_reply":"2024-12-29T03:47:24.297826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in columns :\n    print(col, ':',train[col].nunique()) #nunique ! not unique","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:24.300142Z","iopub.execute_input":"2024-12-29T03:47:24.300569Z","iopub.status.idle":"2024-12-29T03:47:25.457809Z","shell.execute_reply.started":"2024-12-29T03:47:24.300531Z","shell.execute_reply":"2024-12-29T03:47:25.456675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in catCol:\n    print(col ,':',train[col].value_counts()) #value_counts! not count ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:25.458764Z","iopub.execute_input":"2024-12-29T03:47:25.459054Z","iopub.status.idle":"2024-12-29T03:47:26.77145Z","shell.execute_reply.started":"2024-12-29T03:47:25.459001Z","shell.execute_reply":"2024-12-29T03:47:26.770351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:26.772516Z","iopub.execute_input":"2024-12-29T03:47:26.772815Z","iopub.status.idle":"2024-12-29T03:47:26.777454Z","shell.execute_reply.started":"2024-12-29T03:47:26.772789Z","shell.execute_reply":"2024-12-29T03:47:26.776247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def missingvalues (train,title, color='blue'):\n    missing_ratio= train.isnull().sum()/len(train) *100\n    missingtrain=pd.DataFrame({'column': missing_ratio.index , 'missing ratio':missing_ratio.values})\n\n    plt.figure(figsize=(10,6))\n    plt.bar(missingtrain['column'], missingtrain['missing_ratio'], color=color)\n    plt.grid(axis='y', linestyle='--', alpha=0.7)\n    plt.ylabel('missing ration(%)')\n    plt.xlabel('columns')\n\n    for idx, val in enumerate(missingtrain['missing_ratio']):\n        plt.text(idx, val + 0.8, f'{val:.1f}%', ha='center', fontsize=10)\n    plt.tight_layout()\n    plt.show()\n    \n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:26.778612Z","iopub.execute_input":"2024-12-29T03:47:26.778969Z","iopub.status.idle":"2024-12-29T03:47:26.8007Z","shell.execute_reply.started":"2024-12-29T03:47:26.778928Z","shell.execute_reply":"2024-12-29T03:47:26.799704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def missingvalues(train, title, color='blue'):\n    # 결측값 비율 계산\n    missing_ratio = train.isnull().sum() / len(train) * 100\n    missingtrain = pd.DataFrame({'column': missing_ratio.index, 'missing_ratio': missing_ratio.values})\n    \n    # 열 이름 확인 (디버깅용)\n    print(\"Columns in missingtrain:\", missingtrain.columns)\n    \n    # 막대 그래프 생성\n    plt.figure(figsize=(15,6))\n    plt.bar(missingtrain['column'], missingtrain['missing_ratio'], color=color)\n    plt.grid(axis='y', linestyle='--', alpha=0.7)\n    plt.title(title, fontsize=20)\n    plt.ylabel('Missing Ratio (%)',fontsize=14)\n    plt.xlabel('Columns', fontsize=14)\n    \n    # 값 레이블 추가\n    for idx, val in enumerate(missingtrain['missing_ratio']):\n        plt.text(idx, val + 0.8, f'{val:.1f}%', ha='center', fontsize=10)\n    \n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:26.801922Z","iopub.execute_input":"2024-12-29T03:47:26.802304Z","iopub.status.idle":"2024-12-29T03:47:26.819225Z","shell.execute_reply.started":"2024-12-29T03:47:26.802269Z","shell.execute_reply":"2024-12-29T03:47:26.818122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missingvalues(train,'Missing Values Ratio of train Set','blue')\nmissingvalues(test,'Missing Values Ratio of Test Set','yellow')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:26.820472Z","iopub.execute_input":"2024-12-29T03:47:26.820759Z","iopub.status.idle":"2024-12-29T03:47:28.887125Z","shell.execute_reply.started":"2024-12-29T03:47:26.820736Z","shell.execute_reply":"2024-12-29T03:47:28.885997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:28.88845Z","iopub.execute_input":"2024-12-29T03:47:28.888776Z","iopub.status.idle":"2024-12-29T03:47:29.525442Z","shell.execute_reply.started":"2024-12-29T03:47:28.888749Z","shell.execute_reply":"2024-12-29T03:47:29.524383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess(df):\n    num_miss_col=['Age','Annual Income','Number of Dependents','Health Score','Previous Claims','Credit Score','Insurance Duration','Vehicle Age']\n    for column in num_miss_col:\n        if column in df.columns:\n            skewness = df[column].skew()\n            if skewness > 0.5 or skewness < -0.5:\n                modeValue = df[column].mode()[0]\n                df[column].fillna(modeValue,inplace=True)\n            else : \n                meanValue = df[column].mean()\n                df[column].fillna(meanValue,inplace=True) \n    if 'Marital Status' in df.columns:\n        df['Marital Status'].fillna('missing', inplace=True)\n    if 'Occupation' in df.columns:\n        df['Occupation'].fillna('missing', inplace=True)\n    if 'Customer Feedback' in df.columns:\n        df['Customer Feedback'].fillna(df['Customer Feedback'].mode()[0], inplace=True)\n    if 'Policy Start Date' in df.columns:\n        df['Policy Start Year']=pd.to_datetime(df['Policy Start Date']).dt.year\n        catCol.append('Policy Start Year')\n        df.drop(columns=['Policy Start Date'],inplace=True)\n    for col in catCol:\n        if col in df.columns:\n            df[col]=pd.to_numeric(df[column])\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:29.526361Z","iopub.execute_input":"2024-12-29T03:47:29.526611Z","iopub.status.idle":"2024-12-29T03:47:29.534355Z","shell.execute_reply.started":"2024-12-29T03:47:29.52659Z","shell.execute_reply":"2024-12-29T03:47:29.53317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(type(train))\nprint(type(test))\ntrain.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:29.535526Z","iopub.execute_input":"2024-12-29T03:47:29.535794Z","iopub.status.idle":"2024-12-29T03:47:29.581424Z","shell.execute_reply.started":"2024-12-29T03:47:29.535771Z","shell.execute_reply":"2024-12-29T03:47:29.58011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train=preprocess(train)\ndf_test=preprocess(test)\n# df_train.head(5)\ndf_test.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:53:48.222841Z","iopub.execute_input":"2024-12-29T03:53:48.223383Z","iopub.status.idle":"2024-12-29T03:53:48.689704Z","shell.execute_reply.started":"2024-12-29T03:53:48.223313Z","shell.execute_reply":"2024-12-29T03:53:48.688136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:31.725651Z","iopub.execute_input":"2024-12-29T03:47:31.725942Z","iopub.status.idle":"2024-12-29T03:47:31.771775Z","shell.execute_reply.started":"2024-12-29T03:47:31.725902Z","shell.execute_reply":"2024-12-29T03:47:31.770789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:53:57.322049Z","iopub.execute_input":"2024-12-29T03:53:57.322537Z","iopub.status.idle":"2024-12-29T03:53:57.361337Z","shell.execute_reply.started":"2024-12-29T03:53:57.322499Z","shell.execute_reply":"2024-12-29T03:53:57.360144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x = df_train.drop(columns=['Premium Amount'])\ny = df_train['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:31.807692Z","iopub.execute_input":"2024-12-29T03:47:31.808104Z","iopub.status.idle":"2024-12-29T03:47:31.903827Z","shell.execute_reply.started":"2024-12-29T03:47:31.808065Z","shell.execute_reply":"2024-12-29T03:47:31.902909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(x.shape)\nprint(y.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:31.904892Z","iopub.execute_input":"2024-12-29T03:47:31.905292Z","iopub.status.idle":"2024-12-29T03:47:31.910409Z","shell.execute_reply.started":"2024-12-29T03:47:31.905252Z","shell.execute_reply":"2024-12-29T03:47:31.909177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.model_selection import KFold, RandomizedSearchCV, train_test_split, cross_val_score, cross_validate\nfrom random import *\nfrom lightgbm import LGBMRegressor\nfrom lightgbm import log_evaluation, early_stopping\n\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import StackingRegressor , RandomForestRegressor\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.neural_network import MLPClassifier,MLPRegressor\nfrom sklearn.metrics import *","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:31.911429Z","iopub.execute_input":"2024-12-29T03:47:31.911756Z","iopub.status.idle":"2024-12-29T03:47:31.929431Z","shell.execute_reply.started":"2024-12-29T03:47:31.911686Z","shell.execute_reply":"2024-12-29T03:47:31.928278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(x, y, test_size=0.2, random_state=42)\n\n# 예측 함수\ndef hand_draw(model, x_train, y_train, x_val):\n    model.fit(x_train, y_train)\n    y_pred = model.predict(x_val)\n    return y_pred\n\n# RMSLE 계산 함수\ndef rmsl_metric(y_true, y_pred):\n    return np.sqrt(mean_squared_error(np.log1p(y_true), np.log1p(y_pred)))\n\n# # 모델 생성 및 학습\n# L_model = LinearRegression()  # 회귀 모델 사용\n\n\n# mlp_model = MLPRegressor(\n#     hidden_layer_sizes=(20, 10),  # 은닉층 2개 (100개, 50개 뉴런)\n#     activation='relu',            # 활성화 함수\n#     solver='adam',                # 옵티마이저 (adam)\n#     alpha=0.0001,                 # 정규화 파라미터\n#     learning_rate='adaptive',     # 학습률\n#     max_iter=50,                 # 최대 반복 수\n#     random_state=42\n# )\n# y_pred_lin = hand_draw(L_model, x_train, y_train, x_val)\n# y_pred_mlp= hand_draw(mlp_model, x_train, y_train, x_val)\n# # RMSLE 출력\n# rmsle_lin = rmsl_metric(y_val, y_pred_lin)\n# print(f\"Validation RMSLE_Linear: {rmsle_lin:.5f}\")\n# rmsle_mlp= rmsl_metric(y_val, y_pred_mlp)\n# print(f\"Validation RMSLE_MLP: {rmsle_mlp:.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:31.934428Z","iopub.execute_input":"2024-12-29T03:47:31.934764Z","iopub.status.idle":"2024-12-29T03:47:32.356741Z","shell.execute_reply.started":"2024-12-29T03:47:31.934736Z","shell.execute_reply":"2024-12-29T03:47:32.355748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_model=lgb.LGBMRegressor(num_leaves=50,learning_rate=0.05,n_estimators=200,max_depth=7)\ny_pred_lgb = hand_draw(lgb_model, x_train, y_train, x_val)\nDT_model=DecisionTreeRegressor(max_depth=10,min_samples_split=5,min_samples_leaf=4,random_state=42)\ny_pred_dt=hand_draw(DT_model, x_train, y_train, x_val)\nrmsle_lgb = rmsl_metric(y_val, y_pred_lgb)\nrmsle_dt=rmsl_metric(y_val, y_pred_dt)\nprint(f\"Validation RMSLE_LGB: {rmsle_lgb:.5f}\")\nprint(f\"Validation RMSLE_DT: {rmsle_dt:.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:32.358025Z","iopub.execute_input":"2024-12-29T03:47:32.358285Z","iopub.status.idle":"2024-12-29T03:47:55.166599Z","shell.execute_reply.started":"2024-12-29T03:47:32.358262Z","shell.execute_reply":"2024-12-29T03:47:55.165476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import Ridge\n\n# LGBM과 DecisionTree 모델 예측\ny_pred_lgb_train = hand_draw(lgb_model, x_train, y_train, x_train)\ny_pred_dt_train = hand_draw(DT_model, x_train, y_train, x_train)\n\n# Validation set 예측\ny_pred_lgb_val = hand_draw(lgb_model, x_train, y_train, x_val)\ny_pred_dt_val = hand_draw(DT_model, x_train, y_train, x_val)\n\n# Stacking Feature 생성\nstacked_features_train = np.column_stack((y_pred_lgb_train, y_pred_dt_train))\nstacked_features_val = np.column_stack((y_pred_lgb_val, y_pred_dt_val))\n\n# 메타 모델 학습 (예: Ridge Regression)\nmeta_model = Ridge()\nmeta_model.fit(stacked_features_train, y_train)\n\n# 메타 모델을 통한 최종 예측\ny_pred_ensemble_stacking = meta_model.predict(stacked_features_val)\n\n# RMSLE 평가\nrmsle_stacking = rmsl_metric(y_val, y_pred_ensemble_stacking)\nprint(f\"Validation RMSLE_Stacking: {rmsle_stacking:.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:47:55.167756Z","iopub.execute_input":"2024-12-29T03:47:55.168049Z","iopub.status.idle":"2024-12-29T03:48:48.647311Z","shell.execute_reply.started":"2024-12-29T03:47:55.168024Z","shell.execute_reply":"2024-12-29T03:48:48.644209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"weight_lgb = 0.7\nweight_dt = 0.3\n\n# 가중 평균 계산\ny_pred_ensemble_weighted = (weight_lgb * y_pred_lgb) + (weight_dt * y_pred_dt)\n\n# RMSLE 평가\nrmsle_ensemble_weighted = rmsl_metric(y_val, y_pred_ensemble_weighted)\nprint(f\"Validation RMSLE_Weighted Ensemble: {rmsle_ensemble_weighted:.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:49:18.500779Z","iopub.execute_input":"2024-12-29T03:49:18.501282Z","iopub.status.idle":"2024-12-29T03:49:18.521223Z","shell.execute_reply.started":"2024-12-29T03:49:18.501236Z","shell.execute_reply":"2024-12-29T03:49:18.520135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model_rfr = RandomForestRegressor(n_estimators=100, random_state=42)\n# y_pred_rfr=hand_draw(model_rfr, x_train, y_train, x_val)\n# rmsle_rfr = rmsl_metric(y_val, y_pred_rfr)\n# print(f\"Validation RMSLE_rfr: {rmsle_rfr:.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:48:48.687229Z","iopub.execute_input":"2024-12-29T03:48:48.687748Z","iopub.status.idle":"2024-12-29T03:48:48.696939Z","shell.execute_reply.started":"2024-12-29T03:48:48.687703Z","shell.execute_reply":"2024-12-29T03:48:48.692361Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"from sklearn.svm import SVR\nfrom sklearn.preprocessing import StandardScaler\nscaler_x = StandardScaler()\nscaler_y = StandardScaler()\ny_train = y_train.to_numpy().reshape(-1, 1)\nx_train_scaled = scaler_x.fit_transform(x_train)\nx_val_scaled = scaler_x.transform(x_val)\ny_train_scaled = scaler_y.fit_transform(y_train).ravel()\nsvr = SVR(kernel='rbf', C=1.0, epsilon=0.1)\nsvr.fit(x_train_scaled, y_train_scaled) -->","metadata":{}},{"cell_type":"code","source":"print(x_train.shape)\nprint(df_test.shape)\nprint(stacked_features_val.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:54:54.299443Z","iopub.execute_input":"2024-12-29T03:54:54.299783Z","iopub.status.idle":"2024-12-29T03:54:54.31755Z","shell.execute_reply.started":"2024-12-29T03:54:54.299757Z","shell.execute_reply":"2024-12-29T03:54:54.315864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_lgb_test = hand_draw(lgb_model, x_train, y_train, df_test)\ny_pred_dt_test = hand_draw(DT_model, x_train, y_train, df_test)\n\n# Stacking Feature 생성\nstacked_features_test = np.column_stack((y_pred_lgb_test, y_pred_dt_test))\n\n# Ridge 모델로 예측 수행\ny_pred_stacking_test = meta_model.predict(stacked_features_test)\n\n# 결과 저장\nsubmission = pd.DataFrame({\n    'id': dfSubmission['id'], \n    'Premium Amount': y_pred_stacking_test\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T03:58:57.203795Z","iopub.execute_input":"2024-12-29T03:58:57.204312Z","iopub.status.idle":"2024-12-29T03:59:24.525403Z","shell.execute_reply.started":"2024-12-29T03:58:57.204275Z","shell.execute_reply":"2024-12-29T03:59:24.524035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submissionV3.csv', index=False)\nconfirm = pd.read_csv('submissionV3.csv')\nconfirm.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T04:00:16.344671Z","iopub.execute_input":"2024-12-29T04:00:16.345206Z","iopub.status.idle":"2024-12-29T04:00:18.294942Z","shell.execute_reply.started":"2024-12-29T04:00:16.345167Z","shell.execute_reply":"2024-12-29T04:00:18.293932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}