{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30886,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Install tabpfn with pip\npip install tabpfn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T13:31:00.005014Z","iopub.execute_input":"2025-02-19T13:31:00.005465Z","iopub.status.idle":"2025-02-19T13:31:03.329421Z","shell.execute_reply.started":"2025-02-19T13:31:00.005422Z","shell.execute_reply":"2025-02-19T13:31:03.328315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load required packages\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport torch\n#from tabpfn import TabPFNClassifier\nfrom tabpfn import TabPFNRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport gc\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-19T13:31:03.330876Z","iopub.execute_input":"2025-02-19T13:31:03.331137Z","iopub.status.idle":"2025-02-19T13:31:03.336994Z","shell.execute_reply.started":"2025-02-19T13:31:03.33111Z","shell.execute_reply":"2025-02-19T13:31:03.336128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Load train and test datasets\ntrain_path = \"/kaggle/input/playground-series-s4e12/train.csv\"\ntest_path = \"/kaggle/input/playground-series-s4e12/test.csv\"\n\ndf_train = pd.read_csv(train_path)\ndf_test = pd.read_csv(test_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T13:31:03.338347Z","iopub.execute_input":"2025-02-19T13:31:03.338612Z","iopub.status.idle":"2025-02-19T13:31:03.355414Z","shell.execute_reply.started":"2025-02-19T13:31:03.338571Z","shell.execute_reply":"2025-02-19T13:31:03.354587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Concatenate test and train datasets for feature transformation\ndf = pd.concat([df_train, df_test], ignore_index=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T13:35:51.759493Z","iopub.execute_input":"2025-02-19T13:35:51.75987Z","iopub.status.idle":"2025-02-19T13:35:52.056219Z","shell.execute_reply.started":"2025-02-19T13:35:51.759838Z","shell.execute_reply":"2025-02-19T13:35:52.055501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Replace missing values of category variables\nfor col in df.select_dtypes(exclude=[np.number]).columns:\n    df[col] = df[col].fillna(\"Missing\")\n    df[col] = df[col].fillna(\"Missing\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T13:35:54.148535Z","iopub.execute_input":"2025-02-19T13:35:54.148967Z","iopub.status.idle":"2025-02-19T13:35:57.430867Z","shell.execute_reply.started":"2025-02-19T13:35:54.148931Z","shell.execute_reply":"2025-02-19T13:35:57.430148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Transform Policy Start Date to datetime\ndf['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'], errors='coerce')\n\n# Set reference date\ntarget_date = pd.Timestamp('2025-01-01')\n\n# Calculate difference in days\ndf['Policy Start Date'] = (target_date - df['Policy Start Date']).dt.days.astype('int64')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T13:36:00.981224Z","iopub.execute_input":"2025-02-19T13:36:00.981502Z","iopub.status.idle":"2025-02-19T13:36:01.550999Z","shell.execute_reply.started":"2025-02-19T13:36:00.981479Z","shell.execute_reply":"2025-02-19T13:36:01.55029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Set target column\ntarget_column = 'Premium Amount'\n\n#Split back into test and train data\nX_train = df.dropna(subset=[target_column]).drop(columns=[target_column])\ny_train=df[target_column].dropna()\nX_test = df[df['Premium Amount'].isna()].drop(columns=[target_column])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T13:36:47.338804Z","iopub.execute_input":"2025-02-19T13:36:47.339131Z","iopub.status.idle":"2025-02-19T13:36:47.631483Z","shell.execute_reply.started":"2025-02-19T13:36:47.339102Z","shell.execute_reply":"2025-02-19T13:36:47.630522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize model\nmodel = TabPFNRegressor(device='cuda') #activate GPU!\nmodel.fit(X_train[:10000], y_train[:10000]) #tabpfn is limited to 10000 rows and 500 features :(","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T13:37:04.58678Z","iopub.execute_input":"2025-02-19T13:37:04.587075Z","iopub.status.idle":"2025-02-19T13:37:06.218379Z","shell.execute_reply.started":"2025-02-19T13:37:04.587052Z","shell.execute_reply":"2025-02-19T13:37:06.217529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Inference is memory intensive and large workloads must be processed in batches\ndef batch_predict(model, X_test, batch_size=1000):\n    predictions = []\n    for i in range(0, len(X_test), batch_size):\n        torch.cuda.empty_cache()\n        batch = X_test[i:i + batch_size]\n        batch_predictions = model.predict(batch)\n        predictions.extend(batch_predictions)\n        gc.collect()\n        print(i)\n    return np.array(predictions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T13:40:13.28635Z","iopub.execute_input":"2025-02-19T13:40:13.286694Z","iopub.status.idle":"2025-02-19T13:40:13.291364Z","shell.execute_reply.started":"2025-02-19T13:40:13.286662Z","shell.execute_reply":"2025-02-19T13:40:13.290652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions\npredictions = batch_predict(model, X_test, batch_size=25000)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T13:40:17.393641Z","iopub.execute_input":"2025-02-19T13:40:17.393958Z","iopub.status.idle":"2025-02-19T15:22:15.44615Z","shell.execute_reply.started":"2025-02-19T13:40:17.393931Z","shell.execute_reply":"2025-02-19T15:22:15.445177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"output_csv=\"submission.csv\"\n\nresults_df = pd.DataFrame({\"id\": X_test[\"id\"].values, \"Premium Amount\": predictions})\nresults_df.to_csv(output_csv, index=False)\nprint(f\"Submission saved under {output_csv}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T15:22:15.447806Z","iopub.execute_input":"2025-02-19T15:22:15.448127Z","iopub.status.idle":"2025-02-19T15:22:16.356772Z","execution_failed":"2025-02-19T16:17:04.355Z"}},"outputs":[],"execution_count":null}]}