{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-01T19:03:35.816737Z","iopub.execute_input":"2023-10-01T19:03:35.817208Z","iopub.status.idle":"2023-10-01T19:03:35.823091Z","shell.execute_reply.started":"2023-10-01T19:03:35.817174Z","shell.execute_reply":"2023-10-01T19:03:35.821836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<H1>This notebook is inspired from below link . However, I have made modifications based on my understanding of the data. For example loading nii files as npy, calculating accurate slice information based on the position of the slice within the nii file, and including additional features like pixel spacing during training</H1>\n https://www.kaggle.com/code/samuelcortinhas/extracting-vertebrae-c1-c7/notebook. ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:04:08.983333Z","iopub.execute_input":"2023-10-01T19:04:08.983779Z","iopub.status.idle":"2023-10-01T19:04:08.989289Z","shell.execute_reply.started":"2023-10-01T19:04:08.983742Z","shell.execute_reply":"2023-10-01T19:04:08.988239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_data_dir = r'/kaggle/input/rsna-cervical-fracture-detection-training-metadata'\nmeta_data = pd.read_csv(os.path.join(meta_data_dir,'train_meta_data.csv'))\n","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:04:10.138441Z","iopub.execute_input":"2023-10-01T19:04:10.138903Z","iopub.status.idle":"2023-10-01T19:04:12.616069Z","shell.execute_reply.started":"2023-10-01T19:04:10.138867Z","shell.execute_reply":"2023-10-01T19:04:12.614604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:04:14.114538Z","iopub.execute_input":"2023-10-01T19:04:14.115025Z","iopub.status.idle":"2023-10-01T19:04:14.122727Z","shell.execute_reply.started":"2023-10-01T19:04:14.114984Z","shell.execute_reply":"2023-10-01T19:04:14.121644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_data.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:04:16.554573Z","iopub.execute_input":"2023-10-01T19:04:16.554989Z","iopub.status.idle":"2023-10-01T19:04:16.572885Z","shell.execute_reply.started":"2023-10-01T19:04:16.554961Z","shell.execute_reply":"2023-10-01T19:04:16.571696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Store segmentation paths in a dataframe\nfrom glob import glob\nbase_path = \"../input/rsna-cervical-fracture-segmentations-npy\"\nseg_paths = glob(f\"{base_path}/npy_segmentations/*\")\nseg_df = pd.DataFrame({'path': seg_paths})\nseg_df['StudyInstanceUID'] = seg_df['path'].apply(lambda x:x.split('/')[-1][:-4])\nseg_df = seg_df[['StudyInstanceUID','path']]\nprint('seg_df shape:', seg_df.shape)\nseg_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T20:13:57.020184Z","iopub.execute_input":"2023-10-01T20:13:57.020614Z","iopub.status.idle":"2023-10-01T20:13:57.069963Z","shell.execute_reply.started":"2023-10-01T20:13:57.020584Z","shell.execute_reply":"2023-10-01T20:13:57.068962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seg_df.iloc[0:1]['path'].values","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:03:11.180543Z","iopub.execute_input":"2023-10-01T21:03:11.181046Z","iopub.status.idle":"2023-10-01T21:03:11.188534Z","shell.execute_reply.started":"2023-10-01T21:03:11.181011Z","shell.execute_reply":"2023-10-01T21:03:11.187853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_data['SOPInstanceUID'].iloc[0:1].","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:20:59.546972Z","iopub.execute_input":"2023-10-01T19:20:59.547643Z","iopub.status.idle":"2023-10-01T19:20:59.557178Z","shell.execute_reply.started":"2023-10-01T19:20:59.547599Z","shell.execute_reply":"2023-10-01T19:20:59.555944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_data[meta_data['SOPInstanceUID']=='1.2.826.0.1.3680043.6200.1.240']","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:26:46.7643Z","iopub.execute_input":"2023-10-01T19:26:46.76531Z","iopub.status.idle":"2023-10-01T19:26:46.844592Z","shell.execute_reply.started":"2023-10-01T19:26:46.765271Z","shell.execute_reply":"2023-10-01T19:26:46.843305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Add study ID to the meta data\nmeta_data['StudyInstanceUID'] = meta_data['SOPInstanceUID'].apply(lambda x:'.'.join(x.split('.')[:7]))","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:36:31.835096Z","iopub.execute_input":"2023-10-01T19:36:31.835716Z","iopub.status.idle":"2023-10-01T19:36:32.363781Z","shell.execute_reply.started":"2023-10-01T19:36:31.835671Z","shell.execute_reply":"2023-10-01T19:36:32.362974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_data[meta_data['SOPInstanceUID']=='1.2.826.0.1.3680043.6200.1.240']","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:36:44.292018Z","iopub.execute_input":"2023-10-01T19:36:44.292436Z","iopub.status.idle":"2023-10-01T19:36:44.367607Z","shell.execute_reply.started":"2023-10-01T19:36:44.292404Z","shell.execute_reply":"2023-10-01T19:36:44.3663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#select patients with segmentations from meta data\nmeta_seg = meta_data[meta_data['StudyInstanceUID'].isin(seg_df['StudyInstanceUID'])].reset_index(drop=True)\nprint('meta_seg shape:', meta_seg.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:42:39.181456Z","iopub.execute_input":"2023-10-01T19:42:39.181907Z","iopub.status.idle":"2023-10-01T19:42:39.269831Z","shell.execute_reply.started":"2023-10-01T19:42:39.181872Z","shell.execute_reply":"2023-10-01T19:42:39.268511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_seg.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:43:00.125901Z","iopub.execute_input":"2023-10-01T19:43:00.126332Z","iopub.status.idle":"2023-10-01T19:43:00.144756Z","shell.execute_reply.started":"2023-10-01T19:43:00.126302Z","shell.execute_reply":"2023-10-01T19:43:00.143437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#indetifying which vertebrae does the segmentation represent\nsample_seg = np.load('/kaggle/input/rsna-cervical-fracture-segmentations-npy/npy_segmentations/1.2.826.0.1.3680043.10633.npy')\nprint(sample_seg.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:47:55.845048Z","iopub.execute_input":"2023-10-01T19:47:55.846229Z","iopub.status.idle":"2023-10-01T19:47:55.992783Z","shell.execute_reply.started":"2023-10-01T19:47:55.846192Z","shell.execute_reply":"2023-10-01T19:47:55.991637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(sample_seg[199]) # 0 is air, 2 and 3 represents vertbrae C2 and C3","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:48:23.075529Z","iopub.execute_input":"2023-10-01T19:48:23.076018Z","iopub.status.idle":"2023-10-01T19:48:23.088397Z","shell.execute_reply.started":"2023-10-01T19:48:23.075981Z","shell.execute_reply":"2023-10-01T19:48:23.087071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot as plt\nplt.imshow(sample_seg[199])\nplt.title('Segmentation example')\nplt.colorbar()\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:50:48.65048Z","iopub.execute_input":"2023-10-01T19:50:48.650902Z","iopub.status.idle":"2023-10-01T19:50:48.924032Z","shell.execute_reply.started":"2023-10-01T19:50:48.650871Z","shell.execute_reply":"2023-10-01T19:50:48.922882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<H1>Using the above technique lets extract vertebrae information from each slice of segmentation data</H1>","metadata":{}},{"cell_type":"code","source":"#Define target vertebrae for this competition we need C1 to C7 data. Initialise all of them to 0\n\ntargets = ['C1','C2','C3','C4','C5','C6','C7']\nmeta_seg[targets]=0","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:53:05.663571Z","iopub.execute_input":"2023-10-01T19:53:05.664047Z","iopub.status.idle":"2023-10-01T19:53:05.674467Z","shell.execute_reply.started":"2023-10-01T19:53:05.66401Z","shell.execute_reply":"2023-10-01T19:53:05.67345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_seg.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:53:16.572625Z","iopub.execute_input":"2023-10-01T19:53:16.573133Z","iopub.status.idle":"2023-10-01T19:53:16.594137Z","shell.execute_reply.started":"2023-10-01T19:53:16.573092Z","shell.execute_reply":"2023-10-01T19:53:16.593027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_seg.columns","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:53:44.747687Z","iopub.execute_input":"2023-10-01T19:53:44.748115Z","iopub.status.idle":"2023-10-01T19:53:44.755617Z","shell.execute_reply.started":"2023-10-01T19:53:44.748085Z","shell.execute_reply":"2023-10-01T19:53:44.754528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_seg['WindowCenter'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-10-01T20:05:54.685983Z","iopub.execute_input":"2023-10-01T20:05:54.686769Z","iopub.status.idle":"2023-10-01T20:05:54.697763Z","shell.execute_reply.started":"2023-10-01T20:05:54.686725Z","shell.execute_reply":"2023-10-01T20:05:54.697005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_seg['SliceThickness'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-10-01T20:14:47.935121Z","iopub.execute_input":"2023-10-01T20:14:47.935532Z","iopub.status.idle":"2023-10-01T20:14:47.94388Z","shell.execute_reply.started":"2023-10-01T20:14:47.935505Z","shell.execute_reply":"2023-10-01T20:14:47.943167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_seg.iloc[0:1].values","metadata":{"execution":{"iopub.status.busy":"2023-10-01T20:08:27.468865Z","iopub.execute_input":"2023-10-01T20:08:27.469336Z","iopub.status.idle":"2023-10-01T20:08:27.477082Z","shell.execute_reply.started":"2023-10-01T20:08:27.469305Z","shell.execute_reply":"2023-10-01T20:08:27.475943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_seg[meta_seg['StudyInstanceUID']=='1.2.826.0.1.3680043.27016'].sort_values('InstanceNumber')","metadata":{"execution":{"iopub.status.busy":"2023-10-01T20:25:22.714339Z","iopub.execute_input":"2023-10-01T20:25:22.714748Z","iopub.status.idle":"2023-10-01T20:25:22.749059Z","shell.execute_reply.started":"2023-10-01T20:25:22.714713Z","shell.execute_reply":"2023-10-01T20:25:22.747907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#meta_seg = meta_seg.drop(['SliceRank'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T20:56:18.887851Z","iopub.execute_input":"2023-10-01T20:56:18.888247Z","iopub.status.idle":"2023-10-01T20:56:18.898421Z","shell.execute_reply.started":"2023-10-01T20:56:18.88822Z","shell.execute_reply":"2023-10-01T20:56:18.897371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#For the above instance id, the slices start at 2 and end at 286 with a total of 285 slices. However, we don't have,\n#slice information within the 3D segmentation file. Python starts each slice at 0. Hence, there would be \n#few mismatches when we try to join npy slice with slice info available from dcm in meta data. The best way to mitigate it is\n#by assigning ranks to each slice in ascending order starting with 0. This makes it easy to join the segmentation slice as they start with 0. \nmeta_seg[\"Slice\"] = meta_seg.groupby(\"StudyInstanceUID\")[\"InstanceNumber\"].rank(method=\"dense\", ascending=True)-1 #subtract 1 to start with 0","metadata":{"execution":{"iopub.status.busy":"2023-10-01T20:56:22.953982Z","iopub.execute_input":"2023-10-01T20:56:22.954404Z","iopub.status.idle":"2023-10-01T20:56:22.970044Z","shell.execute_reply.started":"2023-10-01T20:56:22.954349Z","shell.execute_reply":"2023-10-01T20:56:22.969058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_seg[meta_seg['StudyInstanceUID']=='1.2.826.0.1.3680043.27016'].sort_values('InstanceNumber')","metadata":{"execution":{"iopub.status.busy":"2023-10-01T20:56:28.059049Z","iopub.execute_input":"2023-10-01T20:56:28.060218Z","iopub.status.idle":"2023-10-01T20:56:28.095775Z","shell.execute_reply.started":"2023-10-01T20:56:28.060179Z","shell.execute_reply":"2023-10-01T20:56:28.094995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Think about sorting slices just like while creating HU units?\nk=0\n# Loop over 87 patients with segmentations\nfor path, UID in zip(seg_df['path'], seg_df['StudyInstanceUID']):\n    # Get segmentations for patient\n    seg = np.load(path)\n    num_slices, _, _ = seg.shape\n    \n    # Loop over slices\n    for i in range(num_slices):\n        mask = seg[i]\n        unique_vals = np.unique(mask)\n        \n        # Loop over unique values (except 0)\n        for j in unique_vals[1:]:\n            \n            # Ignore thoratic spine etc\n            if j <= 7:   \n                meta_seg.loc[(meta_seg['StudyInstanceUID']==UID)&(meta_seg['Slice']==i),f'C{int(j)}'] = 1\n                \n    # Iteration tracker\n    if (k%10)==0:\n        print(f'Iteration:{k}')\n    k+=1","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:03:56.156134Z","iopub.execute_input":"2023-10-01T21:03:56.156574Z","iopub.status.idle":"2023-10-01T21:07:59.400038Z","shell.execute_reply.started":"2023-10-01T21:03:56.156541Z","shell.execute_reply":"2023-10-01T21:07:59.398574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_seg[meta_seg['C1']==1].sort_values('InstanceNumber')","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:08:46.911425Z","iopub.execute_input":"2023-10-01T21:08:46.911966Z","iopub.status.idle":"2023-10-01T21:08:46.955026Z","shell.execute_reply.started":"2023-10-01T21:08:46.911924Z","shell.execute_reply":"2023-10-01T21:08:46.953974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_seg.to_csv(\"meta_segmentation.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:09:52.505579Z","iopub.execute_input":"2023-10-01T21:09:52.506023Z","iopub.status.idle":"2023-10-01T21:09:52.895848Z","shell.execute_reply.started":"2023-10-01T21:09:52.505992Z","shell.execute_reply":"2023-10-01T21:09:52.8945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_seg[['StudyInstanceUID','Slice']+targets].iloc[199:204,:]","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:10:31.754049Z","iopub.execute_input":"2023-10-01T21:10:31.754512Z","iopub.status.idle":"2023-10-01T21:10:31.772699Z","shell.execute_reply.started":"2023-10-01T21:10:31.754475Z","shell.execute_reply":"2023-10-01T21:10:31.771471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"slice_max_seg = meta_seg.groupby('StudyInstanceUID')['Slice'].max().to_dict()\nmeta_seg['SliceRatio'] = 0\nmeta_seg['SliceRatio'] = meta_seg['Slice']/meta_seg['StudyInstanceUID'].map(slice_max_seg)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:12:39.597609Z","iopub.execute_input":"2023-10-01T21:12:39.598676Z","iopub.status.idle":"2023-10-01T21:12:39.617643Z","shell.execute_reply.started":"2023-10-01T21:12:39.598637Z","shell.execute_reply":"2023-10-01T21:12:39.616648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_seg.to_csv(\"meta_segmentation.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:15:48.477123Z","iopub.execute_input":"2023-10-01T21:15:48.477573Z","iopub.status.idle":"2023-10-01T21:15:48.932333Z","shell.execute_reply.started":"2023-10-01T21:15:48.477541Z","shell.execute_reply":"2023-10-01T21:15:48.931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_seg.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:12:51.097383Z","iopub.execute_input":"2023-10-01T21:12:51.097812Z","iopub.status.idle":"2023-10-01T21:12:51.118306Z","shell.execute_reply.started":"2023-10-01T21:12:51.097762Z","shell.execute_reply":"2023-10-01T21:12:51.117067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<H1>\nWith this information train a model to identify target values. Use the model on the train images which doesn't have segmentations and predict their target values. </H1>","metadata":{}},{"cell_type":"code","source":"meta_seg.columns","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:24:08.79752Z","iopub.execute_input":"2023-10-01T21:24:08.798036Z","iopub.status.idle":"2023-10-01T21:24:08.806775Z","shell.execute_reply.started":"2023-10-01T21:24:08.797999Z","shell.execute_reply":"2023-10-01T21:24:08.8055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets start with these features\nfeatures = ['SliceRatio','SliceThickness','ImagePositionPatientX','ImagePositionPatientY','ImagePositionPatientZ','PixelSpacingX','PixelSpacingY','RescaleIntercept','RescaleSlope']","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:24:54.203962Z","iopub.execute_input":"2023-10-01T21:24:54.204411Z","iopub.status.idle":"2023-10-01T21:24:54.210211Z","shell.execute_reply.started":"2023-10-01T21:24:54.204379Z","shell.execute_reply":"2023-10-01T21:24:54.209164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = meta_seg[['StudyInstanceUID']+features]\ny = meta_seg[targets]","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:24:56.30462Z","iopub.execute_input":"2023-10-01T21:24:56.305092Z","iopub.status.idle":"2023-10-01T21:24:56.315317Z","shell.execute_reply.started":"2023-10-01T21:24:56.305058Z","shell.execute_reply":"2023-10-01T21:24:56.314379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X.shape,y.shape) #6 features, 7 classes","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:24:58.681559Z","iopub.execute_input":"2023-10-01T21:24:58.682092Z","iopub.status.idle":"2023-10-01T21:24:58.688322Z","shell.execute_reply.started":"2023-10-01T21:24:58.682049Z","shell.execute_reply":"2023-10-01T21:24:58.687449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split, StratifiedGroupKFold, GroupKFold\nfrom sklearn.metrics import confusion_matrix, accuracy_score\n\n\n# Models\nfrom sklearn.linear_model import LinearRegression, LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom catboost import CatBoostClassifier\nfrom sklearn.naive_bayes import GaussianNB","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:19:28.909471Z","iopub.execute_input":"2023-10-01T21:19:28.909962Z","iopub.status.idle":"2023-10-01T21:19:31.57795Z","shell.execute_reply.started":"2023-10-01T21:19:28.909923Z","shell.execute_reply":"2023-10-01T21:19:31.576869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=5)\n(train_idx, valid_idx) = next(gkf.split(X, y, groups = X['StudyInstanceUID']))\n\n# Train set\nX_train, y_train = X.iloc[train_idx,:], y.iloc[train_idx,:]\n\n# Validation set\nX_valid, y_valid = X.iloc[valid_idx,:], y.iloc[valid_idx,:]\n\n# Drop patient id\nX_train = X_train.drop('StudyInstanceUID', axis=1)\nX_valid = X_valid.drop('StudyInstanceUID', axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:25:03.366349Z","iopub.execute_input":"2023-10-01T21:25:03.367756Z","iopub.status.idle":"2023-10-01T21:25:03.416283Z","shell.execute_reply.started":"2023-10-01T21:25:03.367719Z","shell.execute_reply":"2023-10-01T21:25:03.415141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.shape, X_valid.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:25:06.308355Z","iopub.execute_input":"2023-10-01T21:25:06.308812Z","iopub.status.idle":"2023-10-01T21:25:06.314918Z","shell.execute_reply.started":"2023-10-01T21:25:06.30876Z","shell.execute_reply":"2023-10-01T21:25:06.313755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = RandomForestClassifier()\nclf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:25:08.636245Z","iopub.execute_input":"2023-10-01T21:25:08.636705Z","iopub.status.idle":"2023-10-01T21:25:13.923528Z","shell.execute_reply.started":"2023-10-01T21:25:08.636666Z","shell.execute_reply":"2023-10-01T21:25:13.922784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate model\ny_preds = clf.predict(X_valid)\n\ntotal_acc = 0 \nfor i in range(7):\n    acc = (y_valid[f'C{i+1}']==y_preds[:,i]).sum()/len(y_preds[:,i])\n    total_acc+=acc/7\n    print(f'Accuracy of C{i+1}: {acc} %')\n\nprint('')\nprint(f'Overall accuracy: {total_acc} %')","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:25:18.883137Z","iopub.execute_input":"2023-10-01T21:25:18.88356Z","iopub.status.idle":"2023-10-01T21:25:19.200815Z","shell.execute_reply.started":"2023-10-01T21:25:18.883532Z","shell.execute_reply":"2023-10-01T21:25:19.199699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature importances\npd.DataFrame({'Feature':features, 'Importance':clf.feature_importances_}).sort_values(by='Importance', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:25:27.255634Z","iopub.execute_input":"2023-10-01T21:25:27.256346Z","iopub.status.idle":"2023-10-01T21:25:27.281823Z","shell.execute_reply.started":"2023-10-01T21:25:27.256301Z","shell.execute_reply":"2023-10-01T21:25:27.280981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\npreds = clf.predict(X_valid)\n\n# Confusion matrices\nfig = plt.figure(figsize=(20,10))\nfor i in range(7):\n    cm = confusion_matrix(preds[:,i], y_valid.values[:,i])\n    plt.subplot(2,4,i+1)\n    CBAR=False\n    if (i==3) or (i==6):\n        CBAR=True\n    sns.heatmap(cm, annot=True, fmt='d', cbar=CBAR, cmap='Blues')\n    plt.xlabel('Predicted label')\n    plt.ylabel('True label')\n    plt.title(f'C{i+1}')\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:26:36.830653Z","iopub.execute_input":"2023-10-01T21:26:36.831151Z","iopub.status.idle":"2023-10-01T21:26:38.922765Z","shell.execute_reply.started":"2023-10-01T21:26:36.831115Z","shell.execute_reply":"2023-10-01T21:26:38.921634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_data.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:30:36.570713Z","iopub.execute_input":"2023-10-01T21:30:36.571737Z","iopub.status.idle":"2023-10-01T21:30:36.599623Z","shell.execute_reply.started":"2023-10-01T21:30:36.5717Z","shell.execute_reply":"2023-10-01T21:30:36.598537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read in metadata for entire train set\nmeta_train = meta_data.copy()\n\n#Create slice information\nmeta_train[\"Slice\"] = meta_train.groupby(\"StudyInstanceUID\")[\"InstanceNumber\"].rank(method=\"dense\", ascending=True)-1\n# Calculate slice ratio (to generalise better)\nslice_max_train = meta_train.groupby('StudyInstanceUID')['Slice'].max().to_dict()\nmeta_train['SliceRatio'] = 0\nmeta_train['SliceRatio'] = meta_train['Slice']/meta_train['StudyInstanceUID'].map(slice_max_train)\n\n# Initialise targets\nmeta_train[targets]=0\n\n# Predict targets for entire train set\nmeta_train[targets] = clf.predict(meta_train[features])\n\n# We know images with segmentations have 100% accurate targets so put these back in\nmeta_train.loc[meta_train['StudyInstanceUID'].isin(meta_seg['StudyInstanceUID']),targets] = meta_seg[targets].values\n\n# Save to csv\nmeta_train.to_csv('meta_train_with_vertebrae.csv', index=False)\n\n# Preview\nmeta_train.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:32:17.126359Z","iopub.execute_input":"2023-10-01T21:32:17.126836Z","iopub.status.idle":"2023-10-01T21:33:13.08219Z","shell.execute_reply.started":"2023-10-01T21:32:17.126803Z","shell.execute_reply":"2023-10-01T21:33:13.081434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(20,16))\nfor i, Cx in enumerate(targets):\n    plt.subplot(4,2,i+1)\n    sns.histplot(meta_train.groupby('StudyInstanceUID')[Cx].sum())\n    plt.title(f'Number of slices: {Cx}')\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-10-01T21:33:40.512156Z","iopub.execute_input":"2023-10-01T21:33:40.512589Z","iopub.status.idle":"2023-10-01T21:33:44.466113Z","shell.execute_reply.started":"2023-10-01T21:33:40.512559Z","shell.execute_reply":"2023-10-01T21:33:44.465082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# import os\n# for study in os.listdir(r'/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images'):\n#     if len(study.split('.'))<7:\n    ","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:33:09.322963Z","iopub.execute_input":"2023-10-01T19:33:09.323398Z","iopub.status.idle":"2023-10-01T19:33:09.331674Z","shell.execute_reply.started":"2023-10-01T19:33:09.323365Z","shell.execute_reply":"2023-10-01T19:33:09.33084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# '.'.join('1.2.826.0.1.3680043.6200.1.240'.split('.')[:7])","metadata":{"execution":{"iopub.status.busy":"2023-10-01T19:34:55.099318Z","iopub.execute_input":"2023-10-01T19:34:55.099708Z","iopub.status.idle":"2023-10-01T19:34:55.107474Z","shell.execute_reply.started":"2023-10-01T19:34:55.09968Z","shell.execute_reply":"2023-10-01T19:34:55.106358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}