{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<span class=\"label label-default\" style=\"background-color:#8ACFDD; border-radius:12px; font-weight: bold; font-size:26px; color:#FBFAFC; \">RSNA - Breast Cancer Detection EDA</span>\n\n\n- [<span style=\"font-weight: bold; font-size:16px; color:#8ACFDD; \">Libraries and Utilities</span>](#1)\n- [<span style=\"font-weight: bold; font-size:16px; color:#8ACFDD; \">Loading Data</span>](#2)\n- [<span style=\"font-weight: bold; font-size:16px; color:#8ACFDD; \">Understanding Data</span>](#3)\n- [<span style=\"font-weight: bold; font-size:16px; color:#8ACFDD; \">Evaluation</span>](#4)\n- [<span style=\"font-weight: bold; font-size:16px; color:#8ACFDD; \">Descriptive Statistics</span>](#5)\n- [<span style=\"font-weight: bold; font-size:16px; color:#8ACFDD; \">Correlation Coefficients</span>](#6)\n- [<span style=\"font-weight: bold; font-size:16px; color:#8ACFDD; \">Missing Values</span>](#7)\n- [<span style=\"font-weight: bold; font-size:16px; color:#8ACFDD; \">Target Distribution</span>](#8)\n- [<span style=\"font-weight: bold; font-size:16px; color:#8ACFDD; \">Feature Distributions by Target</span>](#9)\n- [<span style=\"font-weight: bold; font-size:16px; color:#8ACFDD; \">Checking Some Images (Not Cancer)</span>](#10)\n- [<span style=\"font-weight: bold; font-size:16px; color:#8ACFDD; \">Checking Some Images (Cancer)</span>](#11)\n- [<span style=\"font-weight: bold; font-size:16px; color:#8ACFDD; \">Number of Images Distribution</span>](#12)","metadata":{}},{"cell_type":"markdown","source":"<a id = \"1\"></a><h2 id=\"Libraries and Utilities\"><span class=\"label label-default\" style=\"background-color:#8ACFDD; border-radius:12px; font-weight: bold; font-size:22px; color:#FBFAFC; \">Libraries and Utilities</span></h2>","metadata":{}},{"cell_type":"code","source":"!pip install nb_black\n!pip install python-gdcm\n!pip install pylibjpeg pylibjpeg-libjpeg pydicom","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-12-04T19:15:01.784228Z","iopub.execute_input":"2022-12-04T19:15:01.784639Z","iopub.status.idle":"2022-12-04T19:15:32.979079Z","shell.execute_reply.started":"2022-12-04T19:15:01.784606Z","shell.execute_reply":"2022-12-04T19:15:32.977821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport glob\nimport pydicom\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom pathlib import Path\nimport matplotlib.pyplot as plt\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\n\ncmap1 = 'Blues'\ncmap2 = 'coolwarm'\n%matplotlib inline\n%load_ext nb_black\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-04T19:15:32.984603Z","iopub.execute_input":"2022-12-04T19:15:32.987274Z","iopub.status.idle":"2022-12-04T19:15:33.034354Z","shell.execute_reply.started":"2022-12-04T19:15:32.987228Z","shell.execute_reply":"2022-12-04T19:15:33.033361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id = \"2\"></a><h2 id=\"Libraries and Utilities\"><span class=\"label label-default\" style=\"background-color:#8ACFDD; border-radius:12px; font-weight: bold; font-size:22px; color:#FBFAFC; \">Loading Data</span></h2>","metadata":{}},{"cell_type":"code","source":"raw_data_path = \"/kaggle/input/rsna-breast-cancer-detection/\"\ntrain_images_path = \"/kaggle/input/rsna-breast-cancer-detection/train_images\"\ntest_images_path = \"/kaggle/input/rsna-breast-cancer-detection/test_images\"\n\ntrain_df = pd.read_csv(os.path.join(raw_data_path, 'train.csv'))\ntest_df = pd.read_csv(os.path.join(raw_data_path, 'test.csv'))\nsub = pd.read_csv(os.path.join(raw_data_path, 'sample_submission.csv'))\n\nprint(f'Train Shape: {train_df.shape}\\nTest Shape: {test_df.shape}')\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-04T19:15:33.038774Z","iopub.execute_input":"2022-12-04T19:15:33.041255Z","iopub.status.idle":"2022-12-04T19:15:33.265723Z","shell.execute_reply.started":"2022-12-04T19:15:33.041215Z","shell.execute_reply":"2022-12-04T19:15:33.264676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-04T19:15:33.271082Z","iopub.execute_input":"2022-12-04T19:15:33.273508Z","iopub.status.idle":"2022-12-04T19:15:33.300588Z","shell.execute_reply.started":"2022-12-04T19:15:33.273468Z","shell.execute_reply":"2022-12-04T19:15:33.299565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id = \"3\"></a><h2 id=\"Libraries and Utilities\"><span class=\"label label-default\" style=\"background-color:#8ACFDD; border-radius:12px; font-weight: bold; font-size:22px; color:#FBFAFC; \">📃 Understanding Data</span></h2>\n\nThe dataset for this challenge contains radiographic breast images of female subjects.\n\n- <code>site_id</code> - ID code for the source hospital.\n- <code>patient_id</code> - ID code for the patient.\n- <code>image_id</code> - ID code for the image.\n- <code>laterality</code> - Whether the image is of the left or right breast.\n- <code>view</code> - The orientation of the image. The default for a screening exam is to capture two views per breast.\n- <code>age</code> - The patient's age in years.\n- <code>implant</code> - Whether or not the patient had breast implants. Site 1 only provides breast implant information at the patient level, not at the breast level.\n- <code>density</code> - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult.\n- <code>machine_id</code> - An ID code for the imaging device.\n- <code>cancer</code> - The target value. Only provided for train.\n- <code>biopsy</code> - Whether or not a follow-up biopsy was performed on the breast. Only provided for train.\n- <code>invasive</code> - If the breast is positive for cancer, whether or not the cancer proved to be invasive. Only provided for train.\n- <code>BIRADS</code> - 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and 2 if the breast was rated as normal. Only provided for train.\n- <code>prediction_id</code> - The ID for the matching submission row. Multiple images will share the same prediction ID. Test only.\n- <code>difficult_negative_case</code> - True if the case was unusually difficult. Only provided for train.\n\n<a id = \"4\"></a><h2 id=\"Libraries and Utilities\"><span class=\"label label-default\" style=\"background-color:#8ACFDD; border-radius:12px; font-weight: bold; font-size:22px; color:#FBFAFC; \">Evaluation</span></h2>\n\nSubmissions are evaluated using the **probabilistic F1 score** (pF1). This extension of the traditional F score accepts probabilities instead of binary classifications. You can find a Python implementation [here](https://www.kaggle.com/code/sohier/probabilistic-f-score).\n\nWith pX as the probabilistic version of X:\n\n$pF_1 = 2\\frac{pPrecision \\cdot pRecall}{pPrecision+pRecall}$\n\nwhere:\n\n$pPrecision = \\frac{pTP}{pTP+pFP}$\n\n$pRecall = \\frac{pTP}{pTP+pFN}$","metadata":{}},{"cell_type":"markdown","source":"<a id = \"5\"></a><h2 id=\"Libraries and Utilities\"><span class=\"label label-default\" style=\"background-color:#8ACFDD; border-radius:12px; font-weight: bold; font-size:22px; color:#FBFAFC; \">Descriptive Statistics</span></h2>","metadata":{}},{"cell_type":"code","source":"def desc_stats(dataframe):\n    desc = dataframe.describe().T\n    f,ax = plt.subplots(figsize = (10,\n                                   desc.shape[0] * 0.75)\n                       )\n    sns.heatmap(\n                desc,\n                annot = True,\n                cmap = cmap1,\n                fmt= '.2f',\n                ax = ax,\n                linecolor = 'white',\n                linewidths = 1.3,\n                cbar = False,\n                annot_kws = {\"size\": 12}\n               )\n    plt.xticks(size = 14)\n    plt.yticks(size = 12,\n               rotation = 0)\n    plt.title(\"Descriptive Statistics\", size = 14)\n    plt.show()\n    \ndesc_stats(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-12-04T18:35:24.407405Z","iopub.execute_input":"2022-12-04T18:35:24.408001Z","iopub.status.idle":"2022-12-04T18:35:24.995307Z","shell.execute_reply.started":"2022-12-04T18:35:24.407966Z","shell.execute_reply":"2022-12-04T18:35:24.99443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id = \"6\"></a><h2 id=\"Libraries and Utilities\"><span class=\"label label-default\" style=\"background-color:#8ACFDD; border-radius:12px; font-weight: bold; font-size:22px; color:#FBFAFC; \">Correlation Coefficients</span></h2>","metadata":{}},{"cell_type":"code","source":"def corr_map(dataframe, method: str = 'pearson', title: str = None):\n    assert method in ['pearson', 'spearman'], 'Invalid Correlation Method'\n    sns.set_style(\"white\")\n    corr = dataframe.corr(method=method)\n    matrix = np.triu(corr)\n    f,ax = plt.subplots(figsize = (\n                                 matrix.shape[0]*0.75,\n                                 matrix.shape[1]*0.75\n                                )\n                     )\n    sns.heatmap(\n                corr,\n                annot= True,\n                fmt = \".2f\",\n                cbar = False,\n                ax=ax,\n                vmin = -1,\n                vmax = 1,\n                mask = matrix,\n                cmap = cmap2,\n                linewidth = 0.4,\n                linecolor = \"white\",\n                annot_kws = {\"size\": 12}\n               )\n    plt.xticks(rotation = 80,size = 14)\n    plt.yticks(rotation = 0,size = 14)\n    if title == None:\n        title = f'{method.title()} Correlation Map'\n    plt.title(title, size = 14)\n    plt.show()\n    \ncorr_map(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-12-04T18:35:24.996955Z","iopub.execute_input":"2022-12-04T18:35:24.997633Z","iopub.status.idle":"2022-12-04T18:35:25.578821Z","shell.execute_reply.started":"2022-12-04T18:35:24.997594Z","shell.execute_reply":"2022-12-04T18:35:25.577825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id = \"7\"></a><h2 id=\"Libraries and Utilities\"><span class=\"label label-default\" style=\"background-color:#8ACFDD; border-radius:12px; font-weight: bold; font-size:22px; color:#FBFAFC; \">Missing Values</span></h2>","metadata":{}},{"cell_type":"code","source":"def check_missing(dataframe):\n    \n    return pd.DataFrame(\n                        {\n                           'feature': dataframe.columns,\n                           'missing': [dataframe[i].isnull().sum() for i in dataframe.columns],\n                           'ratio': [dataframe[i].isnull().sum() / dataframe.shape[0] \n                                     for i in dataframe.columns]\n                        }\n                          ).sort_values('missing', ascending = False\n                              ).reset_index(drop = True)\n    \ncheck_missing(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-12-04T18:35:25.580104Z","iopub.execute_input":"2022-12-04T18:35:25.580499Z","iopub.status.idle":"2022-12-04T18:35:25.63717Z","shell.execute_reply.started":"2022-12-04T18:35:25.580462Z","shell.execute_reply":"2022-12-04T18:35:25.636267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id = \"8\"></a><h2 id=\"Libraries and Utilities\"><span class=\"label label-default\" style=\"background-color:#8ACFDD; border-radius:12px; font-weight: bold; font-size:22px; color:#FBFAFC; \">Target Distribution</span></h2>","metadata":{}},{"cell_type":"code","source":"fig = go.Figure(data = [go.Pie(\n                               labels = train_df['cancer'].value_counts().keys(),\n                               values = train_df['cancer'].value_counts().values,\n                             )\n                       ]\n               )\n\nfig.update_traces(\n                  hoverinfo ='label',\n                  textinfo ='value',\n                  textfont_size = 16,\n                  textposition ='outside',\n                 )\n\nfig.update_layout(title = {\n                           'text': \"Cancer\",\n                           'y':0.95,\n                           'x':0.5,\n                           'xanchor': 'center',\n                           'yanchor': 'top'\n                          },\n                  legend = dict(\n                              x = 0.85,\n                              y = 1.0,\n                              bgcolor = 'rgba(255, 255, 255, 0)',\n                              bordercolor = 'rgba(255, 255, 255, 0)'\n                             ),\n                  width = 700,\n                  height = 500,\n                  template = 'simple_white'\n                 )\n\niplot(fig)","metadata":{"execution":{"iopub.status.busy":"2022-12-04T18:35:25.638498Z","iopub.execute_input":"2022-12-04T18:35:25.639144Z","iopub.status.idle":"2022-12-04T18:35:25.731917Z","shell.execute_reply.started":"2022-12-04T18:35:25.639108Z","shell.execute_reply":"2022-12-04T18:35:25.7308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id = \"9\"></a><h2 id=\"Libraries and Utilities\"><span class=\"label label-default\" style=\"background-color:#8ACFDD; border-radius:12px; font-weight: bold; font-size:22px; color:#FBFAFC; \">Feature Distributions by Target</span></h2>","metadata":{}},{"cell_type":"code","source":"def dist_by_target(data, col: str, target: str):\n    \n    fig = go.Figure()\n    for i in data[target].unique():\n        fig.add_trace(\n                go.Histogram(\n            x = data[data[target] == i][col],\n            name = str(i),\n            marker = dict(opacity = 0.8)\n                            )\n                     )\n    \n    fig.update_layout(\n                      title = dict(text = f'{col.title()} Histogram by {target.title()}',\n                                   y = 0.9,\n                                   x = 0.5,\n                                   xanchor =  'center',\n                                   yanchor =  'top'\n                                  ),\n                      barmode='overlay',\n                      xaxis = dict(title = col.title()),\n                      yaxis =dict(title = 'Frequency'),\n                      width = 700,\n                      height = 500,\n                      template = 'simple_white'   \n                     )\n    iplot(fig)\n    \ndist_by_target(train_df, 'age', 'cancer')","metadata":{"execution":{"iopub.status.busy":"2022-12-04T18:35:25.733602Z","iopub.execute_input":"2022-12-04T18:35:25.733953Z","iopub.status.idle":"2022-12-04T18:35:25.831837Z","shell.execute_reply.started":"2022-12-04T18:35:25.733918Z","shell.execute_reply":"2022-12-04T18:35:25.831016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dist_by_target(train_df, 'invasive', 'cancer')","metadata":{"execution":{"iopub.status.busy":"2022-12-04T18:35:25.83496Z","iopub.execute_input":"2022-12-04T18:35:25.835809Z","iopub.status.idle":"2022-12-04T18:35:25.900122Z","shell.execute_reply.started":"2022-12-04T18:35:25.835773Z","shell.execute_reply":"2022-12-04T18:35:25.899258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dist_by_target(train_df, 'implant', 'cancer')","metadata":{"execution":{"iopub.status.busy":"2022-12-04T18:35:25.901344Z","iopub.execute_input":"2022-12-04T18:35:25.902249Z","iopub.status.idle":"2022-12-04T18:35:25.965287Z","shell.execute_reply.started":"2022-12-04T18:35:25.902213Z","shell.execute_reply":"2022-12-04T18:35:25.964315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id = \"10\"></a><h2 id=\"Libraries and Utilities\"><span class=\"label label-default\" style=\"background-color:#8ACFDD; border-radius:12px; font-weight: bold; font-size:22px; color:#FBFAFC; \">Checking Some Images (Not Cancer)</span></h2>","metadata":{}},{"cell_type":"code","source":"def rescale_img_to_hu(dcm_ds):\n    \"\"\"\n    Rescales the image to Hounsfield unit.\n    \n    \n    \"\"\"\n    return dcm_ds.pixel_array * dcm_ds.RescaleSlope + dcm_ds.RescaleIntercept\n\ndef sample_images(data, sample_size: int = 10, cancer: bool = False):\n    sample_data =  data.loc[data['cancer'] == int(cancer)].sample(sample_size)\n    ncols = 4\n    nrows = int(sample_size / ncols) if sample_size % ncols == 0 else int(sample_size / ncols) + 1\n    counter = 1\n    fig = plt.figure(figsize = (16, nrows*3.5))\n    for index, row in sample_data.iterrows():\n        img_path = os.path.join(train_images_path, str(row['patient_id']), str(row['image_id']) + '.dcm')\n        try:\n            img = pydicom.dcmread(img_path)\n            rescaled_img = rescale_img_to_hu(img)\n            plt.subplot(nrows, ncols, counter)\n            plt.title(f\"Patient ID: {row['patient_id']} - Image: {row['image_id']}\", size = 12)\n            plt.imshow(rescaled_img, cmap = 'gray')\n            plt.axis(\"off\")\n            counter += 1\n        except:\n            continue\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-04T19:55:36.638529Z","iopub.execute_input":"2022-12-04T19:55:36.638887Z","iopub.status.idle":"2022-12-04T19:55:36.69138Z","shell.execute_reply.started":"2022-12-04T19:55:36.638858Z","shell.execute_reply":"2022-12-04T19:55:36.689808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_images(train_df, 20)","metadata":{"execution":{"iopub.status.busy":"2022-12-04T19:55:36.785218Z","iopub.execute_input":"2022-12-04T19:55:36.7856Z","iopub.status.idle":"2022-12-04T19:56:17.217478Z","shell.execute_reply.started":"2022-12-04T19:55:36.785567Z","shell.execute_reply":"2022-12-04T19:56:17.216619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id = \"11\"></a><h2 id=\"Libraries and Utilities\"><span class=\"label label-default\" style=\"background-color:#8ACFDD; border-radius:12px; font-weight: bold; font-size:22px; color:#FBFAFC; \">Checking Some Images (Cancer)</span></h2>","metadata":{}},{"cell_type":"code","source":"sample_images(train_df, 20, cancer = True)","metadata":{"execution":{"iopub.status.busy":"2022-12-04T19:56:35.040265Z","iopub.execute_input":"2022-12-04T19:56:35.040664Z","iopub.status.idle":"2022-12-04T19:57:15.277682Z","shell.execute_reply.started":"2022-12-04T19:56:35.040632Z","shell.execute_reply":"2022-12-04T19:57:15.276856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id = \"12\"></a><h2 id=\"Libraries and Utilities\"><span class=\"label label-default\" style=\"background-color:#8ACFDD; border-radius:12px; font-weight: bold; font-size:22px; color:#FBFAFC; \">Number of Images Distribution</span></h2>","metadata":{"execution":{"iopub.status.busy":"2022-11-30T10:22:52.194739Z","iopub.execute_input":"2022-11-30T10:22:52.195037Z","iopub.status.idle":"2022-11-30T10:22:52.204801Z","shell.execute_reply.started":"2022-11-30T10:22:52.195011Z","shell.execute_reply":"2022-11-30T10:22:52.203896Z"}}},{"cell_type":"code","source":"num_images = list()\nfor patient_id in train_df['patient_id'].unique():\n    patient_dir = os.path.join(train_images_path, str(patient_id))\n    num_images.append(len(glob.glob(f\"{patient_dir}/*\")))\n    \nfig = go.Figure()\nfig.add_trace(\n        go.Histogram(\n    x = num_images,\n    marker = dict(opacity = 0.8)\n                    )\n             )\n\nfig.update_layout(\n                  title = dict(text = f'Number of Images Distribution',\n                               y = 0.9,\n                               x = 0.5,\n                               xanchor =  'center',\n                               yanchor =  'top'\n                              ),\n                  yaxis = dict(title = 'Frequency'),\n                  xaxis = dict(title = 'Number of Images'),\n                  width = 700,\n                  height = 500,\n                  template = 'simple_white'   \n                 )\niplot(fig)","metadata":{"execution":{"iopub.status.busy":"2022-12-04T19:57:34.844359Z","iopub.execute_input":"2022-12-04T19:57:34.844762Z","iopub.status.idle":"2022-12-04T19:57:50.214677Z","shell.execute_reply.started":"2022-12-04T19:57:34.844731Z","shell.execute_reply":"2022-12-04T19:57:50.213506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:#8ACFDD;\n             font-weight: bold;\n             font-size:16px;\">\nIf you liked this notebook, please upvote 😊\n    \n<span style=\"color:#8ACFDD;\n             font-weight: bold;\n             font-size:16px;\">\nIf you have any suggestions or questions, feel free to comment!\n    \n<span style=\"color:#8ACFDD;\n             font-weight: bold;\n             font-size:16px;\">\nBest Wishes!","metadata":{}}]}