{"cells":[{"metadata":{},"cell_type":"markdown","source":"**Exploratory Data Analysis and Graph generation**\n\nMany thanks to Laura Fink and her EDA notebooks which can be found [here](https://www.kaggle.com/allunia/rsna-ih-detection-eda)\n\n*Abdullah Hasan*","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pydicom\nfrom PIL import Image\nimport os\nimport cv2\nfrom tqdm import tqdm\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nsns.set()\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"CSV file is read in here and transformed for processing later.\n\nAdditional columns are generated to make later operations easier","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"BASE_PATH = \"../input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/\"\nTRAIN_DIR = \"stage_2_train/\"\n\ntrain_df = pd.read_csv(BASE_PATH + 'stage_2_train.csv')\ntrain_df['filename'] = train_df['ID'].apply(lambda st: \"ID_\" + st.split('_')[1] + \".png\")\ntrain_df['type'] = train_df['ID'].apply(lambda st: st.split('_')[2])\ntrain_df['id'] = train_df['ID'].apply(lambda st: st.split('_')[1])\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Label distribution plot**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(15, 10))\nsns.countplot(train_df.Label,ax=ax)\nax.set_xlabel(\"Label\")\nax.set_ylabel(\"Count\")\nax.set_title(\"Label distribution\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Type distribution plot**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(15, 10))\nplt.rcParams['axes.labelsize'] = 20\nplt.rcParams['axes.titlesize'] = 20\ntype_counts = train_df.groupby(\"type\").Label.value_counts().unstack()\ntrue_cases = type_counts.loc[:,1] / train_df.groupby(\"type\").size() * 100\nsns.barplot(x=true_cases.index,y=true_cases.values,ax=ax)\nplt.yticks(rotation=0,size=15)\nplt.xticks(rotation=0,size=15)\nax.set_xlabel(\"ICH Type\")\nax.set_ylabel(\"%\")\nax.set_title(\"Type distribution\",pad=20)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Label co-occurence plot**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(15, 10))\nmulti_count = train_df.groupby(\"id\").Label.sum()\nsns.countplot(multi_count,ax=ax)\nax.set_title(\"Co-occurences\")\nax.set_xlabel(\"Targets per image\")\nax.set_ylabel(\"Frequency\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Label co-occurence matrix**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df = train_df[['Label', 'filename', 'type']].drop_duplicates().pivot(\n    index='filename', columns='type', values='Label').reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df[df['any']==1]['any'].count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df[df['any']==0]['any'].count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\ncols = ['epidural','intraparenchymal','intraventricular','subarachnoid','subdural']\nlabels = df[cols]\noutput = pd.DataFrame(\n{x:[(df[x] & df[y]).sum() for y in cols] for x in cols},\nindex=cols)\noutput\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Co-occurence matrix plot**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#output\noutput[\"epidural\"][output[\"epidural\"].index != \"epidural\"].sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"normalized_df = output.copy().astype(np.float32)\nfor col in cols:\n    total = output[col][col]\n    total_col = total - output[col][output[col].index != col].sum()\n    normalized_df[col][col] = total_col / total\n    for other in cols:\n        if other == col:\n            continue\n        normalized_df[col][other] = output[col][other] / total\nnormalized_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sum(normalized_df['epidural'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"normalized_df = normalized_df.rename({\n    \"epidural\":\"EDH\",\n    \"intraparenchymal\":\"IPH\",\n    \"intraventricular\":\"IVH\",\n    \"subarachnoid\":\"SAH\",\n    \"subdural\":\"SDH\"\n},axis=\"index\")\nnormalized_df = normalized_df.rename(index=str,columns={\n    \"epidural\":\"EDH\",\n    \"intraparenchymal\":\"IPH\",\n    \"intraventricular\":\"IVH\",\n    \"subarachnoid\":\"SAH\",\n    \"subdural\":\"SDH\"\n})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.rcParams['axes.labelsize'] = 40\nplt.rcParams['axes.titlesize'] = 20\nfig, ax = plt.subplots(figsize=(15, 10))\n\nsns.heatmap(normalized_df,ax=ax,annot=True,annot_kws={\"size\": 20})\nfor i in np.arange(0,5,1.0):\n    ax.axvline(i, color='white', lw=10)\nplt.yticks(rotation=0,size=20)\nplt.xticks(rotation=0,size=20)\nax.set_title(\"Co-occurring haemorrhages matrix\",pad=20)\n#ax.set_title(\"Co-occurence Matrix\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ax.get_xticks()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#for i in range(type_list):\n#    title = cols[i]\n#    dataset = type_list[i]\n#    fig,ax = plt.subplots()\n#    sns.countplot(dataset,ax=ax)\n#    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}