{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"import glob, pylab, pandas as pd\nimport pydicom, numpy as np\nfrom os import listdir\nfrom os.path import isfile, join\nimport matplotlib.pylab as plt\nimport os\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-11-12T10:40:13.12296Z","iopub.execute_input":"2023-11-12T10:40:13.123336Z","iopub.status.idle":"2023-11-12T10:40:14.597161Z","shell.execute_reply.started":"2023-11-12T10:40:13.123307Z","shell.execute_reply":"2023-11-12T10:40:14.596171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras import layers\nfrom keras.applications import DenseNet121\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-11-12T10:40:16.915425Z","iopub.execute_input":"2023-11-12T10:40:16.915933Z","iopub.status.idle":"2023-11-12T10:40:25.329879Z","shell.execute_reply.started":"2023-11-12T10:40:16.9159Z","shell.execute_reply":"2023-11-12T10:40:25.328795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH=\"../input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection\"\n!ls ../input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection","metadata":{"execution":{"iopub.status.busy":"2023-11-12T10:40:26.713491Z","iopub.execute_input":"2023-11-12T10:40:26.714227Z","iopub.status.idle":"2023-11-12T10:40:27.841399Z","shell.execute_reply.started":"2023-11-12T10:40:26.714192Z","shell.execute_reply":"2023-11-12T10:40:27.840033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(join(PATH,'stage_2_train.csv'))\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T10:40:28.477301Z","iopub.execute_input":"2023-11-12T10:40:28.477703Z","iopub.status.idle":"2023-11-12T10:40:33.500455Z","shell.execute_reply.started":"2023-11-12T10:40:28.477669Z","shell.execute_reply":"2023-11-12T10:40:33.499491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Analysis","metadata":{}},{"cell_type":"code","source":"print(train.Label.value_counts())\nsns.countplot(x='Label', data=train)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T10:40:35.441768Z","iopub.execute_input":"2023-11-12T10:40:35.442186Z","iopub.status.idle":"2023-11-12T10:40:36.210155Z","shell.execute_reply.started":"2023-11-12T10:40:35.442139Z","shell.execute_reply":"2023-11-12T10:40:36.208995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Sub_type'] = train['ID'].str.split(\"_\", n = 3, expand = True)[2]\ntrain['PatientID'] = train['ID'].str.split(\"_\", n = 3, expand = True)[1]\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T10:40:37.00382Z","iopub.execute_input":"2023-11-12T10:40:37.004434Z","iopub.status.idle":"2023-11-12T10:41:09.16461Z","shell.execute_reply.started":"2023-11-12T10:40:37.004382Z","shell.execute_reply":"2023-11-12T10:41:09.163312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig=plt.figure(figsize=(20, 8))\n\nsns.countplot(x=\"Sub_type\", hue=\"Label\", data=train)\n\nplt.title(\"Total Images by Subtype\")\n#The samples labeled with 0s are too much when compared to the samples labeled with 1s.","metadata":{"execution":{"iopub.status.busy":"2023-11-12T10:41:09.166415Z","iopub.execute_input":"2023-11-12T10:41:09.166746Z","iopub.status.idle":"2023-11-12T10:41:17.529396Z","shell.execute_reply.started":"2023-11-12T10:41:09.166719Z","shell.execute_reply":"2023-11-12T10:41:17.528266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=\"Label\", hue=\"Sub_type\", data=train)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T09:37:34.411889Z","iopub.execute_input":"2023-11-12T09:37:34.412341Z","iopub.status.idle":"2023-11-12T09:37:39.38709Z","shell.execute_reply.started":"2023-11-12T09:37:34.41231Z","shell.execute_reply":"2023-11-12T09:37:39.385896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subtype_counts = train.groupby(\"Sub_type\").Label.value_counts().unstack()\nsubtype_counts = subtype_counts.loc[:, 1] / train.groupby(\"Sub_type\").size() * 100\nsubtype_counts\n#We can see here that the data is not balanced, some subtypes have few examples, and that will make it hard to train the model to detectthose subtypes(epidural for example).","metadata":{"execution":{"iopub.status.busy":"2023-11-12T10:41:21.473507Z","iopub.execute_input":"2023-11-12T10:41:21.473902Z","iopub.status.idle":"2023-11-12T10:41:23.306939Z","shell.execute_reply.started":"2023-11-12T10:41:21.473872Z","shell.execute_reply":"2023-11-12T10:41:23.30586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.Label.isnull().sum()\n#As we see here we don't have null values in the csv file, so no need to clean the data from the missing values.","metadata":{"execution":{"iopub.status.busy":"2023-11-12T10:41:24.944174Z","iopub.execute_input":"2023-11-12T10:41:24.945364Z","iopub.status.idle":"2023-11-12T10:41:24.956618Z","shell.execute_reply.started":"2023-11-12T10:41:24.945326Z","shell.execute_reply":"2023-11-12T10:41:24.955689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I load a loaded a preprocessed file of the same file (Prepare dataset (resizing and saving as png)) \n# IN this file the Dicom files are converted to png\n","metadata":{"execution":{"iopub.status.busy":"2023-11-13T11:48:08.576297Z","iopub.execute_input":"2023-11-13T11:48:08.576754Z","iopub.status.idle":"2023-11-13T11:48:08.714481Z","shell.execute_reply.started":"2023-11-13T11:48:08.576721Z","shell.execute_reply":"2023-11-13T11:48:08.712539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}