{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport pydicom\nimport matplotlib.pylab as plt\nfrom matplotlib import rcParams\nimport os\n\nrcParams['figure.figsize'] = 11.7,8.27\n\ntraindir = \"/kaggle/input/rsna-intracranial-hemorrhage-detection/stage_1_train_images\"\ntestdir = \"/kaggle/input/rsna-intracranial-hemorrhage-detection/stage_1_test_images\"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### If you find the kernel helpfull please upvote"},{"metadata":{},"cell_type":"markdown","source":"# Lets look at the list of files given"},{"metadata":{},"cell_type":"markdown","source":"### The file structure looks very simple.Two folders with images and two csv files one with training labels and other is sample submission file"},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls /kaggle/input/rsna-intracranial-hemorrhage-detection/","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Lets dig into the traing labels csv"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv = pd.read_csv(\"/kaggle/input/rsna-intracranial-hemorrhage-detection/stage_1_train.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv.head(5)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### This gives us the insight that this is a multi class classification problem"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv[\"type\"] = train_csv[\"ID\"].apply(lambda a:a.split(\"_\")[2])\ntrain_csv[\"ID\"] = train_csv[\"ID\"].apply(lambda a:\"_\".join(a.split(\"_\")[0:2]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv.head(10)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Lets look at the distibution of Labels"},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(train_csv.Label)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## What is the distribution of labels"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv.groupby(\"type\").sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n\nlabel_distribution = train_csv.groupby(\"type\").sum().reset_index()\nsns.barplot(x=label_distribution.type,y=label_distribution.Label)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Distribution of label count"},{"metadata":{"trusted":true},"cell_type":"code","source":"type_count_distribution = train_csv.groupby(\"ID\").sum().reset_index()\nvc = type_count_distribution.Label.value_counts()\nsns.barplot(x=vc.index,y=vc)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Lets plot some images"},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_images(hem_type):\n    images = train_csv[(train_csv[\"type\"] == hem_type) & (train_csv[\"Label\"] == 1)][\"ID\"].values[:100]\n    width = 5\n    height = 2\n    fig, axs = plt.subplots(height, width, figsize=(15,5))\n\n    for im in range(0, height * width):\n        image = pydicom.read_file(os.path.join(traindir,images[im]+ '.dcm')).pixel_array\n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image, cmap=plt.cm.bone) \n        axs[i,j].axis('off')\n\n    plt.suptitle(hem_type)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Image containing intraparenchymal"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_images(\"intraparenchymal\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Image containing epidural"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_images(\"epidural\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Image containing intraventricular"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_images(\"intraventricular\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Image containing subarachnoid"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_images(\"subarachnoid\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Image containing subdural"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_images(\"subdural\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Lets see if any corelation between occurence of any type of hemorrhage"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv.drop_duplicates(inplace=True)\npivot = train_csv.pivot(index='ID', columns='type', values='Label').reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"corrs = pivot[[\"epidural\",\"intraparenchymal\",\"intraventricular\",\"subarachnoid\",\"subdural\"]].corr()\ncorrs.style.background_gradient(cmap='coolwarm')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}