{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30839,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:03.793525Z","iopub.execute_input":"2025-01-21T17:48:03.793847Z","iopub.status.idle":"2025-01-21T17:48:03.800017Z","shell.execute_reply.started":"2025-01-21T17:48:03.79382Z","shell.execute_reply":"2025-01-21T17:48:03.799181Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sklearn\n\nprint(sklearn.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:03.801481Z","iopub.execute_input":"2025-01-21T17:48:03.801739Z","iopub.status.idle":"2025-01-21T17:48:03.815633Z","shell.execute_reply.started":"2025-01-21T17:48:03.801718Z","shell.execute_reply":"2025-01-21T17:48:03.814876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_FILE_PATH = \"/kaggle/input/playground-series-s4e12/train.csv\"\nTEST_FILE_PATH = \"/kaggle/input/playground-series-s4e12/test.csv\"\nSUBMISSION_FILE_PATH = \"/kaggle/input/playground-series-s4e12/sample_submission.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T18:55:20.082866Z","iopub.execute_input":"2025-01-21T18:55:20.083138Z","iopub.status.idle":"2025-01-21T18:55:20.086926Z","shell.execute_reply.started":"2025-01-21T18:55:20.083115Z","shell.execute_reply":"2025-01-21T18:55:20.085909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\ndf = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndisplay(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:03.827817Z","iopub.execute_input":"2025-01-21T17:48:03.828041Z","iopub.status.idle":"2025-01-21T17:48:07.246738Z","shell.execute_reply.started":"2025-01-21T17:48:03.82802Z","shell.execute_reply":"2025-01-21T17:48:07.245807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\npd.set_option('display.max_columns', 50)\n\ndisplay(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:07.247493Z","iopub.execute_input":"2025-01-21T17:48:07.247738Z","iopub.status.idle":"2025-01-21T17:48:07.264378Z","shell.execute_reply.started":"2025-01-21T17:48:07.247718Z","shell.execute_reply":"2025-01-21T17:48:07.263587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if df.index.name != \"id\":\n    df.set_index(\"id\", inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:07.265125Z","iopub.execute_input":"2025-01-21T17:48:07.265328Z","iopub.status.idle":"2025-01-21T17:48:07.275627Z","shell.execute_reply.started":"2025-01-21T17:48:07.26531Z","shell.execute_reply":"2025-01-21T17:48:07.274815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:07.278248Z","iopub.execute_input":"2025-01-21T17:48:07.278498Z","iopub.status.idle":"2025-01-21T17:48:07.855334Z","shell.execute_reply.started":"2025-01-21T17:48:07.278476Z","shell.execute_reply":"2025-01-21T17:48:07.854445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_threshold = 0.8\n\nmissing_values = df.columns[df.isnull().mean() > missing_threshold]\n\nprint(missing_values) # no mising values more than 0.8","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:07.856932Z","iopub.execute_input":"2025-01-21T17:48:07.857158Z","iopub.status.idle":"2025-01-21T17:48:08.426884Z","shell.execute_reply.started":"2025-01-21T17:48:07.857138Z","shell.execute_reply":"2025-01-21T17:48:08.426109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def change_date(df, column_name):\n    if column_name not in df.columns:\n        print(f\"Column '{column_name}' not found in the DataFrame.\")\n        return df\n    day_df = pd.to_datetime(df[column_name]).dt.day\n    month_df= pd.to_datetime(df[column_name]).dt.month\n    year_df = pd.to_datetime(df[column_name]).dt.year\n\n    df[\"day\"] = day_df\n    df[\"month\"] = month_df\n    df[\"year\"] = year_df\n\n    return df\ndf = change_date(df, \"Policy Start Date\")\n\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:08.427731Z","iopub.execute_input":"2025-01-21T17:48:08.428055Z","iopub.status.idle":"2025-01-21T17:48:09.574227Z","shell.execute_reply.started":"2025-01-21T17:48:08.428023Z","shell.execute_reply":"2025-01-21T17:48:09.573413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove unimportant columns\n\ncolumns_to_drop = [\"Policy Start Date\"]\n\ndf = df.drop(columns=columns_to_drop, errors=\"ignore\")\n\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:09.574978Z","iopub.execute_input":"2025-01-21T17:48:09.575192Z","iopub.status.idle":"2025-01-21T17:48:09.727679Z","shell.execute_reply.started":"2025-01-21T17:48:09.575174Z","shell.execute_reply":"2025-01-21T17:48:09.726887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical = df.select_dtypes(include=['object'])\ncategorical_cols = categorical.columns\n\nnumerical = df.select_dtypes(include=['int64', 'float64', 'int32'])\nnumerical_cols = numerical.columns\n\n\nprint(f\"categorical_cols: {categorical_cols}\")\nprint(f\"numerical_cols: {numerical_cols}\")\n\nassert len(categorical_cols) + len(numerical_cols) == len(df.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:09.728326Z","iopub.execute_input":"2025-01-21T17:48:09.728544Z","iopub.status.idle":"2025-01-21T17:48:10.280062Z","shell.execute_reply.started":"2025-01-21T17:48:09.728524Z","shell.execute_reply":"2025-01-21T17:48:10.27928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[categorical_cols] = df[categorical_cols].fillna(df[categorical_cols].mode().iloc[0])\ndf[numerical_cols] = df[numerical_cols].fillna(df[numerical_cols].mean())\n\ndf.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:10.280765Z","iopub.execute_input":"2025-01-21T17:48:10.280983Z","iopub.status.idle":"2025-01-21T17:48:13.303304Z","shell.execute_reply.started":"2025-01-21T17:48:10.280962Z","shell.execute_reply":"2025-01-21T17:48:13.302415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\ncorrelation = numerical.corr()\n\nplt.figure(figsize=(10, 8))\n\nsns.heatmap(\n    correlation,\n    annot=True,\n    # cmap=sns.diverging_palette(230, 20, as_cmap=True),\n    cmap=\"coolwarm\",\n    linewidths=0.5\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:13.304182Z","iopub.execute_input":"2025-01-21T17:48:13.30443Z","iopub.status.idle":"2025-01-21T17:48:14.626973Z","shell.execute_reply.started":"2025-01-21T17:48:13.304409Z","shell.execute_reply":"2025-01-21T17:48:14.626082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.histplot(\n    df['Premium Amount'],\n    bins=50,\n    kde=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:14.62791Z","iopub.execute_input":"2025-01-21T17:48:14.628326Z","iopub.status.idle":"2025-01-21T17:48:19.286554Z","shell.execute_reply.started":"2025-01-21T17:48:14.6283Z","shell.execute_reply":"2025-01-21T17:48:19.285555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Handling Single Value Columns\n\nsingle_value_columns = df.columns[df.nunique() == 1]\n\nprint(single_value_columns) # no single value columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:19.287444Z","iopub.execute_input":"2025-01-21T17:48:19.287679Z","iopub.status.idle":"2025-01-21T17:48:20.013946Z","shell.execute_reply.started":"2025-01-21T17:48:19.287659Z","shell.execute_reply":"2025-01-21T17:48:20.013169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_column = \"Premium Amount\"\n\ndata, label = df.drop(columns=[target_column], axis=1), df[target_column]\n\ndata.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:20.014706Z","iopub.execute_input":"2025-01-21T17:48:20.014959Z","iopub.status.idle":"2025-01-21T17:48:20.179662Z","shell.execute_reply.started":"2025-01-21T17:48:20.014933Z","shell.execute_reply":"2025-01-21T17:48:20.17879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\nsns.histplot(\n    label,\n    bins=50,\n    kde=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:20.1805Z","iopub.execute_input":"2025-01-21T17:48:20.180748Z","iopub.status.idle":"2025-01-21T17:48:24.926217Z","shell.execute_reply.started":"2025-01-21T17:48:20.180726Z","shell.execute_reply":"2025-01-21T17:48:24.925421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nlabel_encoder = LabelEncoder()\n\nfor col in categorical_cols:\n    data[col] = label_encoder.fit_transform(df[col])\n\ndata.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:24.926995Z","iopub.execute_input":"2025-01-21T17:48:24.927237Z","iopub.status.idle":"2025-01-21T17:48:27.054373Z","shell.execute_reply.started":"2025-01-21T17:48:24.927208Z","shell.execute_reply":"2025-01-21T17:48:27.053385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\n\ndata_scaled = scaler.fit_transform(data)\n\ndata = pd.DataFrame(data_scaled, columns=data.columns)\n\ndata.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:27.055282Z","iopub.execute_input":"2025-01-21T17:48:27.055828Z","iopub.status.idle":"2025-01-21T17:48:27.533342Z","shell.execute_reply.started":"2025-01-21T17:48:27.055786Z","shell.execute_reply":"2025-01-21T17:48:27.5324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(data, label, test_size=0.2, random_state=42)\n\nprint(f\"X_train shape: {X_train.shape}\")\nprint(f\"X_test shape: {X_test.shape}\")\n\nprint(f\"y_train shape: {y_train.shape}\")\nprint(f\"y_test shape: {y_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:27.5343Z","iopub.execute_input":"2025-01-21T17:48:27.534584Z","iopub.status.idle":"2025-01-21T17:48:27.837755Z","shell.execute_reply.started":"2025-01-21T17:48:27.534555Z","shell.execute_reply":"2025-01-21T17:48:27.836851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train = np.log(y_train)\n\ny_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:27.838689Z","iopub.execute_input":"2025-01-21T17:48:27.839188Z","iopub.status.idle":"2025-01-21T17:48:27.8467Z","shell.execute_reply.started":"2025-01-21T17:48:27.839154Z","shell.execute_reply":"2025-01-21T17:48:27.845634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.histplot(\n    y_train,\n    bins=50,\n    kde=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:27.849618Z","iopub.execute_input":"2025-01-21T17:48:27.849872Z","iopub.status.idle":"2025-01-21T17:48:31.625914Z","shell.execute_reply.started":"2025-01-21T17:48:27.849851Z","shell.execute_reply":"2025-01-21T17:48:31.624965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:31.626914Z","iopub.execute_input":"2025-01-21T17:48:31.627229Z","iopub.status.idle":"2025-01-21T17:48:31.644311Z","shell.execute_reply.started":"2025-01-21T17:48:31.627207Z","shell.execute_reply":"2025-01-21T17:48:31.643395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import r2_score, mean_absolute_error, root_mean_squared_log_error\n\nlrModel = LinearRegression()\nlrModel.fit(X_train, y_train)\n\ny_pred = lrModel.predict(X_test)\n\ny_pred = np.exp(y_pred)\n\nmae = mean_absolute_error(y_test, y_pred)\nrmsle = root_mean_squared_log_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\nprint(f\"mae: {mae}\")\nprint(f\"rmsle: {rmsle}\")\nprint(f\"r2: {r2}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:48:35.188187Z","iopub.execute_input":"2025-01-21T17:48:35.188537Z","iopub.status.idle":"2025-01-21T17:48:35.791704Z","shell.execute_reply.started":"2025-01-21T17:48:35.188502Z","shell.execute_reply":"2025-01-21T17:48:35.790805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\neff_max_depth = eff_estimator = eff_mae = eff_rmsle = eff_r2 = None\n\neff_mae_features = eff_rmsle_features = eff_r2_features = None\n\nfor n_estimator in [50, 75, 100]:\n    for max_depth in [5, 10, 15]:\n        rfrModel = RandomForestRegressor(n_estimators=n_estimator, max_depth=max_depth, random_state=42)\n        rfrModel.fit(X_train, y_train)\n        \n        y_pred = rfrModel.predict(X_test)\n        y_pred = np.exp(y_pred)\n        mae = mean_absolute_error(y_test, y_pred)\n        rmsle = root_mean_squared_log_error(y_test, y_pred)\n        r2 = r2_score(y_test, y_pred)\n\n        print(f\"n-estimators_{n_estimator}, max_depth_{max_depth} --> mae: {mae}, rmsle: {rmsle}, r2_score: {r2}\")\n\n        if eff_r2 is None or r2 > eff_r2:\n            eff_r2 = r2\n            eff_r2_features = {\"n_estimators\": n_estimator, \"max_depth\": max_depth}\n\n        if eff_mae is None or mae < eff_mae:\n            eff_mae = mae\n            eff_mae_features = {\"n_estimators\": n_estimator, \"max_depth\": max_depth}\n\n        if eff_rmsle is None or rmsle < eff_rmsle:\n            eff_rmsle = rmsle\n            eff_rmsle_features = {\"n_estimators\": n_estimator, \"max_depth\": max_depth}\n\nprint(f\"eff mae1: {eff_mae}, best_mae_features: {eff_mae_features}\")\nprint(f\"eff rmse1: {eff_rmsle}, eff_rmse_features: {eff_rmsle_features}\")\nprint(f\"eff r2_max1: {eff_r2}, r2_max_features: {eff_r2_features}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T17:53:55.620093Z","iopub.execute_input":"2025-01-21T17:53:55.620409Z","iopub.status.idle":"2025-01-21T18:46:26.399653Z","shell.execute_reply.started":"2025-01-21T17:53:55.620384Z","shell.execute_reply":"2025-01-21T18:46:26.39886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rfrModel = RandomForestRegressor(n_estimators=75, max_depth=10, random_state=42)\nrfrModel.fit(X_train, y_train)\n\ny_pred = rfrModel.predict(X_test)\n\ny_pred = np.exp(y_pred)\n\nmae = mean_absolute_error(y_test, y_pred)\nrmsle = root_mean_squared_log_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\nprint(f\"mae: {mae}\")\nprint(f\"rmsle: {rmsle}\")\nprint(f\"r2: {r2}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T18:48:20.637094Z","iopub.execute_input":"2025-01-21T18:48:20.637405Z","iopub.status.idle":"2025-01-21T18:54:28.853678Z","shell.execute_reply.started":"2025-01-21T18:48:20.637382Z","shell.execute_reply":"2025-01-21T18:54:28.852785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_csv(TEST_FILE_PATH)\n\ntest_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T18:57:11.18125Z","iopub.execute_input":"2025-01-21T18:57:11.181531Z","iopub.status.idle":"2025-01-21T18:57:13.380498Z","shell.execute_reply.started":"2025-01-21T18:57:11.181507Z","shell.execute_reply":"2025-01-21T18:57:13.379618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if test_df.index.name != \"id\":\n    test_df.set_index(\"id\", inplace=True)\n\ntest_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T18:57:18.657971Z","iopub.execute_input":"2025-01-21T18:57:18.658252Z","iopub.status.idle":"2025-01-21T18:57:18.675156Z","shell.execute_reply.started":"2025-01-21T18:57:18.65823Z","shell.execute_reply":"2025-01-21T18:57:18.674386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = change_date(test_df, \"Policy Start Date\")\n\ntest_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T18:57:22.886066Z","iopub.execute_input":"2025-01-21T18:57:22.886376Z","iopub.status.idle":"2025-01-21T18:57:23.64166Z","shell.execute_reply.started":"2025-01-21T18:57:22.886347Z","shell.execute_reply":"2025-01-21T18:57:23.640914Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if columns_to_drop:\n    test_df = test_df.drop(columns=columns_to_drop)\n\ntest_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T18:57:26.266347Z","iopub.execute_input":"2025-01-21T18:57:26.266659Z","iopub.status.idle":"2025-01-21T18:57:26.358756Z","shell.execute_reply.started":"2025-01-21T18:57:26.266627Z","shell.execute_reply":"2025-01-21T18:57:26.357962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_categorical = test_df.select_dtypes(include=['object'])\ntest_categorical_cols = test_categorical.columns\n\ntest_numerical = test_df.select_dtypes(include=['int64', 'float64'])\ntest_numerical_cols = test_numerical.columns\n\ntest_df[test_categorical_cols] = test_df[test_categorical_cols].fillna(test_df[test_categorical_cols].mode().iloc[0])\ntest_df[test_numerical_cols] = test_df[test_numerical_cols].fillna(test_df[test_numerical_cols].mean())\n\nprint(f\"test_numerical_cols: {test_numerical_cols}\")\nprint(f\"test_categorical_cols: {test_categorical_cols}\")\n\nassert test_df.isna().sum().any() == False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T18:57:31.956451Z","iopub.execute_input":"2025-01-21T18:57:31.956769Z","iopub.status.idle":"2025-01-21T18:57:34.235467Z","shell.execute_reply.started":"2025-01-21T18:57:31.956743Z","shell.execute_reply":"2025-01-21T18:57:34.234525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in test_categorical_cols:\n    test_df[col] = label_encoder.fit_transform(test_df[col])\n\ntest_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T18:57:34.236701Z","iopub.execute_input":"2025-01-21T18:57:34.237009Z","iopub.status.idle":"2025-01-21T18:57:35.606378Z","shell.execute_reply.started":"2025-01-21T18:57:34.236977Z","shell.execute_reply":"2025-01-21T18:57:35.605562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df_scaled = scaler.fit_transform(test_df)\n\ntest_df = pd.DataFrame(test_df_scaled, columns=test_df.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T18:57:35.607552Z","iopub.execute_input":"2025-01-21T18:57:35.607863Z","iopub.status.idle":"2025-01-21T18:57:35.919711Z","shell.execute_reply.started":"2025-01-21T18:57:35.607838Z","shell.execute_reply":"2025-01-21T18:57:35.919027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def download_submissions_csv_from_predictions(predictions):\n    submission_df = pd.read_csv(SUBMISSION_FILE_PATH)\n    submission_df[target_column] = predictions\n\n    submission_df.to_csv(f\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T18:58:11.709918Z","iopub.execute_input":"2025-01-21T18:58:11.710193Z","iopub.status.idle":"2025-01-21T18:58:11.714001Z","shell.execute_reply.started":"2025-01-21T18:58:11.710171Z","shell.execute_reply":"2025-01-21T18:58:11.713194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predict = rfrModel.predict(test_df)\n\ntest_predict = np.exp(test_predict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T18:57:36.438846Z","iopub.execute_input":"2025-01-21T18:57:36.43908Z","iopub.status.idle":"2025-01-21T18:57:40.489256Z","shell.execute_reply.started":"2025-01-21T18:57:36.43906Z","shell.execute_reply":"2025-01-21T18:57:40.488542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"download_submissions_csv_from_predictions(test_predict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T18:59:33.247293Z","iopub.execute_input":"2025-01-21T18:59:33.24757Z","iopub.status.idle":"2025-01-21T18:59:34.675765Z","shell.execute_reply.started":"2025-01-21T18:59:33.247548Z","shell.execute_reply":"2025-01-21T18:59:34.674812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import the modules we'll need\nfrom IPython.display import HTML\nimport pandas as pd\nimport numpy as np\nimport base64\n\ndef create_download_link(df, title = \"Download CSV file\", filename = \"submission.csv\"): \n    submission_df = pd.read_csv(SUBMISSION_FILE_PATH)\n    submission_df[target_column] = df\n    csv = submission_df.to_csv(index=False)\n    b64 = base64.b64encode(csv.encode())\n    payload = b64.decode()\n    html = '<a download=\"{filename}\" href=\"data:text/csv;base64,{payload}\" target=\"_blank\">{title}</a>'\n    html = html.format(payload=payload,title=title,filename=filename)\n    return HTML(html)\n\n# create a random sample dataframe\n\n# create a link to download the dataframe\ncreate_download_link(test_predict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T19:06:50.670133Z","iopub.execute_input":"2025-01-21T19:06:50.670412Z","iopub.status.idle":"2025-01-21T19:06:52.620418Z","shell.execute_reply.started":"2025-01-21T19:06:50.67039Z","shell.execute_reply":"2025-01-21T19:06:52.619415Z"}},"outputs":[],"execution_count":null}]}