{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-23T20:26:18.408204Z","iopub.execute_input":"2022-09-23T20:26:18.409083Z","iopub.status.idle":"2022-09-23T20:26:18.542705Z","shell.execute_reply.started":"2022-09-23T20:26:18.40903Z","shell.execute_reply":"2022-09-23T20:26:18.54143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission requirments ","metadata":{}},{"cell_type":"code","source":"# Import libraries \nimport pandas as pd\nimport numpy as np\ndf =  pd.read_csv(r'/kaggle/input/mayo-clinic-strip-ai/sample_submission.csv')\ndf","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:20.35745Z","iopub.execute_input":"2022-09-23T20:26:20.358247Z","iopub.status.idle":"2022-09-23T20:26:20.389474Z","shell.execute_reply.started":"2022-09-23T20:26:20.358191Z","shell.execute_reply":"2022-09-23T20:26:20.388389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# We ensure the final solution matches these 3 columns\n# \tpatient_id, CE, LAA","metadata":{}},{"cell_type":"markdown","source":"# Train set ","metadata":{}},{"cell_type":"code","source":"# Create a copy dataset for further work\ndf_copy  = pd.read_csv(r'/kaggle/input/mayo-clinic-strip-ai/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:21.194275Z","iopub.execute_input":"2022-09-23T20:26:21.194717Z","iopub.status.idle":"2022-09-23T20:26:21.205597Z","shell.execute_reply.started":"2022-09-23T20:26:21.194681Z","shell.execute_reply":"2022-09-23T20:26:21.204565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df =  pd.read_csv(r'/kaggle/input/mayo-clinic-strip-ai/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:21.234468Z","iopub.execute_input":"2022-09-23T20:26:21.23511Z","iopub.status.idle":"2022-09-23T20:26:21.250283Z","shell.execute_reply.started":"2022-09-23T20:26:21.235075Z","shell.execute_reply":"2022-09-23T20:26:21.249448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:21.269536Z","iopub.execute_input":"2022-09-23T20:26:21.270189Z","iopub.status.idle":"2022-09-23T20:26:21.282362Z","shell.execute_reply.started":"2022-09-23T20:26:21.270154Z","shell.execute_reply":"2022-09-23T20:26:21.281058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_tab = pd.crosstab(index = df[\"label\"],  # Make a crosstab\n                              columns=\"center_id_x\")      # Name the count column\n\nmy_tab.plot.bar()","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:21.310907Z","iopub.execute_input":"2022-09-23T20:26:21.31153Z","iopub.status.idle":"2022-09-23T20:26:21.587784Z","shell.execute_reply.started":"2022-09-23T20:26:21.311488Z","shell.execute_reply":"2022-09-23T20:26:21.586796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### CE is more common among patients","metadata":{}},{"cell_type":"code","source":"# Create New columns of LAA and CE and transform them \ndf['LAA'] = df[\"label\"]\ndf['CE'] = df[\"label\"]","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:21.589482Z","iopub.execute_input":"2022-09-23T20:26:21.589825Z","iopub.status.idle":"2022-09-23T20:26:21.597224Z","shell.execute_reply.started":"2022-09-23T20:26:21.589793Z","shell.execute_reply":"2022-09-23T20:26:21.595823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Replace values with binary identifiers for LAA and CE","metadata":{}},{"cell_type":"code","source":"df['LAA'].replace(to_replace=\"LAA\",\n           value=1, inplace=True)\ndf['CE'].replace(to_replace=\"LAA\",\n           value=0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:21.59915Z","iopub.execute_input":"2022-09-23T20:26:21.600537Z","iopub.status.idle":"2022-09-23T20:26:21.611985Z","shell.execute_reply.started":"2022-09-23T20:26:21.600405Z","shell.execute_reply":"2022-09-23T20:26:21.6108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['LAA'].replace(to_replace=\"CE\",\n           value=0, inplace=True)\ndf['CE'].replace(to_replace=\"CE\",\n           value=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:21.614362Z","iopub.execute_input":"2022-09-23T20:26:21.615121Z","iopub.status.idle":"2022-09-23T20:26:21.625658Z","shell.execute_reply.started":"2022-09-23T20:26:21.61508Z","shell.execute_reply":"2022-09-23T20:26:21.624686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:21.627236Z","iopub.execute_input":"2022-09-23T20:26:21.627705Z","iopub.status.idle":"2022-09-23T20:26:21.643577Z","shell.execute_reply.started":"2022-09-23T20:26:21.627653Z","shell.execute_reply":"2022-09-23T20:26:21.642221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# We now drop Un-nessecary columns \n","metadata":{}},{"cell_type":"code","source":"df.drop(labels=['label', 'center_id', 'patient_id', 'image_id'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:23.049253Z","iopub.execute_input":"2022-09-23T20:26:23.049662Z","iopub.status.idle":"2022-09-23T20:26:23.060349Z","shell.execute_reply.started":"2022-09-23T20:26:23.04963Z","shell.execute_reply":"2022-09-23T20:26:23.0585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.count()","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:23.55639Z","iopub.execute_input":"2022-09-23T20:26:23.557364Z","iopub.status.idle":"2022-09-23T20:26:23.567512Z","shell.execute_reply.started":"2022-09-23T20:26:23.557322Z","shell.execute_reply":"2022-09-23T20:26:23.566191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n#creating instance of one-hot-encoder\nencoder = OneHotEncoder(handle_unknown='ignore')\n\n#perform one-hot encoding on 'team' column \ndf['image_num'] = encoder.fit_transform(df[['image_num']]).toarray()\ndf['LAA'] = encoder.fit_transform(df[['LAA']]).toarray()\ndf['CE'] = encoder.fit_transform(df[['CE']]).toarray()","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:23.588618Z","iopub.execute_input":"2022-09-23T20:26:23.589041Z","iopub.status.idle":"2022-09-23T20:26:24.109883Z","shell.execute_reply.started":"2022-09-23T20:26:23.588994Z","shell.execute_reply":"2022-09-23T20:26:24.108857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:24.111675Z","iopub.execute_input":"2022-09-23T20:26:24.1122Z","iopub.status.idle":"2022-09-23T20:26:24.120442Z","shell.execute_reply.started":"2022-09-23T20:26:24.112168Z","shell.execute_reply":"2022-09-23T20:26:24.119279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['LAA'].count()","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:24.121835Z","iopub.execute_input":"2022-09-23T20:26:24.122203Z","iopub.status.idle":"2022-09-23T20:26:24.1355Z","shell.execute_reply.started":"2022-09-23T20:26:24.122174Z","shell.execute_reply":"2022-09-23T20:26:24.133706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling for LAA with Random Forest","metadata":{}},{"cell_type":"code","source":"# Splitting Train and test\nfrom sklearn.model_selection import train_test_split\n# split the dataset\n# Getting the target (y) from the splitted DataFrames\ntrain_y = df[\"LAA\"].astype(float).values\neval_y = df[\"LAA\"].astype(float).values\n\n# Getting the features (X) from the splitted DataFrames\ntrain_X = df.drop(['LAA'], axis=1)\neval_X = df.drop(['LAA'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:24.139372Z","iopub.execute_input":"2022-09-23T20:26:24.139886Z","iopub.status.idle":"2022-09-23T20:26:24.16761Z","shell.execute_reply.started":"2022-09-23T20:26:24.139839Z","shell.execute_reply":"2022-09-23T20:26:24.166394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, confusion_matrix, classification_report\nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:24.169417Z","iopub.execute_input":"2022-09-23T20:26:24.169757Z","iopub.status.idle":"2022-09-23T20:26:24.270164Z","shell.execute_reply.started":"2022-09-23T20:26:24.169727Z","shell.execute_reply":"2022-09-23T20:26:24.26913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_RF = RandomForestClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:24.286831Z","iopub.execute_input":"2022-09-23T20:26:24.287257Z","iopub.status.idle":"2022-09-23T20:26:24.292293Z","shell.execute_reply.started":"2022-09-23T20:26:24.287223Z","shell.execute_reply":"2022-09-23T20:26:24.291187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_RF.fit(train_X, train_y)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:24.642762Z","iopub.execute_input":"2022-09-23T20:26:24.643489Z","iopub.status.idle":"2022-09-23T20:26:24.805091Z","shell.execute_reply.started":"2022-09-23T20:26:24.643451Z","shell.execute_reply":"2022-09-23T20:26:24.80387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_RF = model_RF.predict(eval_X)\npredict_RF","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:25.010188Z","iopub.execute_input":"2022-09-23T20:26:25.010611Z","iopub.status.idle":"2022-09-23T20:26:25.045173Z","shell.execute_reply.started":"2022-09-23T20:26:25.010579Z","shell.execute_reply":"2022-09-23T20:26:25.044067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error\nmean_absolute_error(predict_RF, eval_y)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:25.363426Z","iopub.execute_input":"2022-09-23T20:26:25.364121Z","iopub.status.idle":"2022-09-23T20:26:25.37213Z","shell.execute_reply.started":"2022-09-23T20:26:25.364085Z","shell.execute_reply":"2022-09-23T20:26:25.37073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import r2_score\nr2_score(predict_RF, eval_y)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:25.642496Z","iopub.execute_input":"2022-09-23T20:26:25.642922Z","iopub.status.idle":"2022-09-23T20:26:25.652124Z","shell.execute_reply.started":"2022-09-23T20:26:25.642885Z","shell.execute_reply":"2022-09-23T20:26:25.650974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold, cross_val_score, StratifiedKFold\ncv1 = KFold(n_splits=10, random_state=12,shuffle= True)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:25.970354Z","iopub.execute_input":"2022-09-23T20:26:25.971089Z","iopub.status.idle":"2022-09-23T20:26:25.977668Z","shell.execute_reply.started":"2022-09-23T20:26:25.971042Z","shell.execute_reply":"2022-09-23T20:26:25.976541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# evaluate the model with cross validation\nscores = cross_val_score(model_RF, train_X, train_y, scoring='accuracy', cv=cv1, n_jobs=-1)\nscores","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:26.233719Z","iopub.execute_input":"2022-09-23T20:26:26.234154Z","iopub.status.idle":"2022-09-23T20:26:28.694139Z","shell.execute_reply.started":"2022-09-23T20:26:26.234118Z","shell.execute_reply":"2022-09-23T20:26:28.692668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from statistics import mean, stdev\n#report perofmance\nprint('Accuracy: %.3f(%.3f)'% (mean(scores), stdev(scores)))","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:28.696968Z","iopub.execute_input":"2022-09-23T20:26:28.698341Z","iopub.status.idle":"2022-09-23T20:26:28.708449Z","shell.execute_reply.started":"2022-09-23T20:26:28.69828Z","shell.execute_reply":"2022-09-23T20:26:28.707096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(accuracy_score(predict_RF, eval_y)*100)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:28.710188Z","iopub.execute_input":"2022-09-23T20:26:28.711523Z","iopub.status.idle":"2022-09-23T20:26:28.721036Z","shell.execute_reply.started":"2022-09-23T20:26:28.711481Z","shell.execute_reply":"2022-09-23T20:26:28.719565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We have a 100% accurate model","metadata":{}},{"cell_type":"code","source":"print(classification_report(eval_y, predict_RF))","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:28.723099Z","iopub.execute_input":"2022-09-23T20:26:28.723532Z","iopub.status.idle":"2022-09-23T20:26:28.738246Z","shell.execute_reply.started":"2022-09-23T20:26:28.723498Z","shell.execute_reply":"2022-09-23T20:26:28.736938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create new predictive column","metadata":{}},{"cell_type":"code","source":"df_copy['pred_LAA'] = predict_RF","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:30.479483Z","iopub.execute_input":"2022-09-23T20:26:30.47994Z","iopub.status.idle":"2022-09-23T20:26:30.487344Z","shell.execute_reply.started":"2022-09-23T20:26:30.479896Z","shell.execute_reply":"2022-09-23T20:26:30.486263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling for CE","metadata":{}},{"cell_type":"code","source":"# split the dataset\n# Getting the target (y) from the splitted DataFrames\ntrain_y = df[\"LAA\"].astype(float).values\neval_y = df[\"LAA\"].astype(float).values\n\n# Getting the features (X) from the splitted DataFrames\ntrain_X = df.drop(['LAA'], axis=1)\neval_X = df.drop(['LAA'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:33.057029Z","iopub.execute_input":"2022-09-23T20:26:33.057438Z","iopub.status.idle":"2022-09-23T20:26:33.066729Z","shell.execute_reply.started":"2022-09-23T20:26:33.057398Z","shell.execute_reply":"2022-09-23T20:26:33.065624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split the dataset\n# Getting the target (y) from the splitted DataFrames\ntrain_y_c = df[\"CE\"].astype(float).values\neval_y_c = df[\"CE\"].astype(float).values\n\n# Getting the features (X) from the splitted DataFrames\ntrain_X_c = df.drop(['CE'], axis=1)\neval_X_c = df.drop(['CE'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:33.083903Z","iopub.execute_input":"2022-09-23T20:26:33.084849Z","iopub.status.idle":"2022-09-23T20:26:33.092912Z","shell.execute_reply.started":"2022-09-23T20:26:33.084809Z","shell.execute_reply":"2022-09-23T20:26:33.091574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_RF.fit(train_X_c, train_y_c)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:33.126009Z","iopub.execute_input":"2022-09-23T20:26:33.126445Z","iopub.status.idle":"2022-09-23T20:26:33.287042Z","shell.execute_reply.started":"2022-09-23T20:26:33.126408Z","shell.execute_reply":"2022-09-23T20:26:33.285645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_RF_c = model_RF.predict(eval_X_c)\npredict_RF_c","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:33.289685Z","iopub.execute_input":"2022-09-23T20:26:33.290161Z","iopub.status.idle":"2022-09-23T20:26:33.328277Z","shell.execute_reply.started":"2022-09-23T20:26:33.290117Z","shell.execute_reply":"2022-09-23T20:26:33.326935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_copy['pred_CE'] = predict_RF_c","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:33.329919Z","iopub.execute_input":"2022-09-23T20:26:33.331113Z","iopub.status.idle":"2022-09-23T20:26:33.336544Z","shell.execute_reply.started":"2022-09-23T20:26:33.331075Z","shell.execute_reply":"2022-09-23T20:26:33.335396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_copy","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:33.339062Z","iopub.execute_input":"2022-09-23T20:26:33.339768Z","iopub.status.idle":"2022-09-23T20:26:33.367738Z","shell.execute_reply.started":"2022-09-23T20:26:33.339735Z","shell.execute_reply":"2022-09-23T20:26:33.366605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r2_score(predict_RF_c, eval_y_c)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:33.368977Z","iopub.execute_input":"2022-09-23T20:26:33.369324Z","iopub.status.idle":"2022-09-23T20:26:33.381508Z","shell.execute_reply.started":"2022-09-23T20:26:33.369293Z","shell.execute_reply":"2022-09-23T20:26:33.380512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv2 = KFold(n_splits=10, random_state=12,shuffle= True)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:33.383366Z","iopub.execute_input":"2022-09-23T20:26:33.38376Z","iopub.status.idle":"2022-09-23T20:26:33.393518Z","shell.execute_reply.started":"2022-09-23T20:26:33.383683Z","shell.execute_reply":"2022-09-23T20:26:33.392652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = cross_val_score(model_RF, train_X_c, train_y_c, scoring='accuracy', cv=cv2, n_jobs=-1)\nscores","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:33.395178Z","iopub.execute_input":"2022-09-23T20:26:33.395577Z","iopub.status.idle":"2022-09-23T20:26:34.133063Z","shell.execute_reply.started":"2022-09-23T20:26:33.395547Z","shell.execute_reply":"2022-09-23T20:26:34.132165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(accuracy_score(predict_RF_c, eval_y_c)*100)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:35.234373Z","iopub.execute_input":"2022-09-23T20:26:35.234835Z","iopub.status.idle":"2022-09-23T20:26:35.241887Z","shell.execute_reply.started":"2022-09-23T20:26:35.234795Z","shell.execute_reply":"2022-09-23T20:26:35.24065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# A perfect classification model","metadata":{}},{"cell_type":"code","source":"print(classification_report(eval_y_c, predict_RF_c))","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:36.595307Z","iopub.execute_input":"2022-09-23T20:26:36.595722Z","iopub.status.idle":"2022-09-23T20:26:36.610711Z","shell.execute_reply.started":"2022-09-23T20:26:36.595689Z","shell.execute_reply":"2022-09-23T20:26:36.609527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Data","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv(r'/kaggle/input/mayo-clinic-strip-ai/test.csv')\ntest","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:26:38.330693Z","iopub.execute_input":"2022-09-23T20:26:38.331141Z","iopub.status.idle":"2022-09-23T20:26:38.349951Z","shell.execute_reply.started":"2022-09-23T20:26:38.331107Z","shell.execute_reply":"2022-09-23T20:26:38.348576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission To Kaggle competition","metadata":{}},{"cell_type":"code","source":"# create test_X which comes from test_data but includes only the columns you used for prediction.\n# The list of columns is stored in a variable called features\n#test = test_data[features]\n\n# make predictions which we will submit. \n#test_preds = model_R.predict(test_X)\n\n# The lines below shows how to save predictions in format used for competition scoring\noutput = pd.DataFrame({'Id': test.patient_id,\n                      'CE': test.pred_CE,\n                      'LAA': test.pred_LAA})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-23T20:20:15.491523Z","iopub.execute_input":"2022-09-23T20:20:15.491948Z","iopub.status.idle":"2022-09-23T20:20:15.502252Z","shell.execute_reply.started":"2022-09-23T20:20:15.491913Z","shell.execute_reply":"2022-09-23T20:20:15.500863Z"},"trusted":true},"execution_count":null,"outputs":[]}]}