{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1><center>RSNA Screening Mammography Breast Cancer Detection. Simple Exploratory Data Analysis.</center></h1>\n\n<center><img src=\"https://www.invitra.com/en/wp-content/uploads/2019/11/mamography.png\"></center>","metadata":{}},{"cell_type":"markdown","source":"<a id=\"0\"></a>\n<h1 style='background:#815391; border:0; color:white'><center>Navigation</center></h1>\n\n* [1. Acknowledgments](#1)\n* [2. Data Overview](#2)\n* [3. Images Overview](#3)\n* [4. DICOM Metadata Overview](#4)","metadata":{}},{"cell_type":"markdown","source":"<a id=\"1\"></a>\n<h1 style='background:#815391; border:0; color:white'><center>Acknowledgments</center></h1>","metadata":{}},{"cell_type":"markdown","source":"This kernel uses such good notebooks:\n\n* [EDA~[RSNA]](https://www.kaggle.com/code/asimple/eda-rsna)\n* [RSNA - EDA and Modeling](https://www.kaggle.com/code/xxxxyyyy80008/rsna-eda-and-modeling/notebook)\n* [RSNA - Breast Cancer EDA](https://www.kaggle.com/code/allunia/rsna-breast-cancer-eda)","metadata":{}},{"cell_type":"markdown","source":"<h1 style='background:#815391; border:0; color:white'><center>Libraries</center></h1>","metadata":{}},{"cell_type":"code","source":"# Installing additional libraries for working with DICOM image format\n!pip install python-gdcm -q\n!pip install pylibjpeg -q","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:09:29.043476Z","iopub.execute_input":"2022-12-09T09:09:29.044369Z","iopub.status.idle":"2022-12-09T09:09:57.895876Z","shell.execute_reply.started":"2022-12-09T09:09:29.044271Z","shell.execute_reply":"2022-12-09T09:09:57.894839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom pathlib import Path\n\nimport gdcm\nimport pydicom\nimport pylibjpeg\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:09:57.897665Z","iopub.execute_input":"2022-12-09T09:09:57.898053Z","iopub.status.idle":"2022-12-09T09:09:58.810179Z","shell.execute_reply.started":"2022-12-09T09:09:57.898013Z","shell.execute_reply":"2022-12-09T09:09:58.809202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set style for plots\nsns.set(style=\"darkgrid\")\nplt.style.use('seaborn-darkgrid')","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:09:58.811568Z","iopub.execute_input":"2022-12-09T09:09:58.8122Z","iopub.status.idle":"2022-12-09T09:09:58.817104Z","shell.execute_reply.started":"2022-12-09T09:09:58.812167Z","shell.execute_reply":"2022-12-09T09:09:58.816255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style='background:#815391; border:0; color:white'><center>Data Loading</center></h1>","metadata":{}},{"cell_type":"code","source":"# Paths to the base dir/files of the dataset\nbase_dir = Path(\"/kaggle/input/rsna-breast-cancer-detection\")\n\ntrain_csv = f\"{base_dir}/train.csv\"\ntest_csv = f\"{base_dir}/test.csv\"\n\ntrain_images_dir = f\"{base_dir}/train_images\"\ntest_images_dir = f\"{base_dir}/test_images\"","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:09:58.819059Z","iopub.execute_input":"2022-12-09T09:09:58.819793Z","iopub.status.idle":"2022-12-09T09:09:58.832985Z","shell.execute_reply.started":"2022-12-09T09:09:58.81976Z","shell.execute_reply":"2022-12-09T09:09:58.831787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read csv files which containing data\ntrain_df = pd.read_csv(train_csv)\ntest_df = pd.read_csv(test_csv)","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:09:58.999175Z","iopub.execute_input":"2022-12-09T09:09:58.999503Z","iopub.status.idle":"2022-12-09T09:09:59.127033Z","shell.execute_reply.started":"2022-12-09T09:09:58.999475Z","shell.execute_reply":"2022-12-09T09:09:59.125895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"2\"></a>\n<h1 style='background:#815391; border:0; color:white'><center>Data Overview</center></h1>","metadata":{}},{"cell_type":"code","source":"print(f\"Training DataFrame Shape: {train_df.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:09:59.128611Z","iopub.execute_input":"2022-12-09T09:09:59.129207Z","iopub.status.idle":"2022-12-09T09:09:59.13567Z","shell.execute_reply.started":"2022-12-09T09:09:59.129171Z","shell.execute_reply":"2022-12-09T09:09:59.134569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:09:59.136954Z","iopub.execute_input":"2022-12-09T09:09:59.137297Z","iopub.status.idle":"2022-12-09T09:09:59.167942Z","shell.execute_reply.started":"2022-12-09T09:09:59.137268Z","shell.execute_reply":"2022-12-09T09:09:59.166761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Test DataFrame Shape: {test_df.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:09:59.1692Z","iopub.execute_input":"2022-12-09T09:09:59.169511Z","iopub.status.idle":"2022-12-09T09:09:59.174963Z","shell.execute_reply.started":"2022-12-09T09:09:59.169483Z","shell.execute_reply":"2022-12-09T09:09:59.173762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:09:59.179182Z","iopub.execute_input":"2022-12-09T09:09:59.179569Z","iopub.status.idle":"2022-12-09T09:09:59.1937Z","shell.execute_reply.started":"2022-12-09T09:09:59.179529Z","shell.execute_reply":"2022-12-09T09:09:59.192634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:09:59.195588Z","iopub.execute_input":"2022-12-09T09:09:59.19607Z","iopub.status.idle":"2022-12-09T09:09:59.23495Z","shell.execute_reply.started":"2022-12-09T09:09:59.196026Z","shell.execute_reply":"2022-12-09T09:09:59.233578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:09:59.236624Z","iopub.execute_input":"2022-12-09T09:09:59.237099Z","iopub.status.idle":"2022-12-09T09:09:59.250486Z","shell.execute_reply.started":"2022-12-09T09:09:59.237054Z","shell.execute_reply":"2022-12-09T09:09:59.249217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Description**\n\n* `site_id` - ID core for the sourse hospital.\n* `patient_id` - ID code for the patient.\n* `image_id` - ID code for the image.\n* `laterality` - Whether the image is of the left or right breast.\n* `view` - The orientation of the image. The default for a screening exam is to capture two views per breast.\n* `age` - The patient's age in years.\n* `implant` - Whether or not the patient had breast implants. Site 1 only provides breast implant information at the patient level, not at the breast level.\n* `density` - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult. Only provided for train.\n* `machine_id` - An ID code for the imaging device.\n* `cancer` - Whether or not the breast was positive for cancer. The target value. Only provided for train.\n* `biopsy` - Whether or not a follow-up biopsy was performed on the breast. Only provided for train.\n* `invasive` - If the breast is positive for cancer, whether or not the cancer proved to be invasive. Only provided for train.\n* `BIRADS` - 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and 2 if the breast was rated as normal. Only provided for train.\n* `prediction_id` - The ID for the matching submission row. Multiple images will share the same prediction ID. Test only.\n* `difficult_negative_case` - True if the case was unusually difficult. Only provided for train.","metadata":{}},{"cell_type":"code","source":"# See what the columns are not present in test df\nprint(f\"Columns are not present in test dataset: {list(set(train_df.columns) - set(test_df.columns))}\")","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:43:30.260775Z","iopub.execute_input":"2022-12-09T09:43:30.261474Z","iopub.status.idle":"2022-12-09T09:43:30.267011Z","shell.execute_reply.started":"2022-12-09T09:43:30.261423Z","shell.execute_reply":"2022-12-09T09:43:30.265948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See missing values in training dataset\nmissing_values = train_df.isnull().sum()\nall_value = train_df.count()\n\nmissing_values_df = pd.concat([missing_values, all_value], axis=1, keys=[\"Missing Values\", \"All Values\"])\nmissing_values_df","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:43:38.108971Z","iopub.execute_input":"2022-12-09T09:43:38.110147Z","iopub.status.idle":"2022-12-09T09:43:38.142297Z","shell.execute_reply.started":"2022-12-09T09:43:38.110097Z","shell.execute_reply":"2022-12-09T09:43:38.141098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See the number of unique patients presented in the training dataset\nprint(f\"Number of unique patients presented in the training dataset: {train_df['patient_id'].nunique()}\")","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:44:26.250807Z","iopub.execute_input":"2022-12-09T09:44:26.251636Z","iopub.status.idle":"2022-12-09T09:44:26.259728Z","shell.execute_reply.started":"2022-12-09T09:44:26.251589Z","shell.execute_reply":"2022-12-09T09:44:26.258489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See the number of cases of cancer in patients\ntrain_df.groupby(\"patient_id\")[\"cancer\"].nunique().value_counts().to_frame(\"Patients Count\")","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:51:15.014305Z","iopub.execute_input":"2022-12-09T09:51:15.014687Z","iopub.status.idle":"2022-12-09T09:51:15.032001Z","shell.execute_reply.started":"2022-12-09T09:51:15.014655Z","shell.execute_reply":"2022-12-09T09:51:15.031009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see from this distribution:\n* 11433 patients are healthy;\n* 480 patients have cancer.","metadata":{}},{"cell_type":"code","source":"# Let's see how many samples contain cancer\nfig, ax = plt.subplots(1, 1, figsize=(20, 8))\n\nsns.countplot(train_df[\"cancer\"], palette=\"Reds_r\")\nax.set_title('Distribution of cancer by cases');","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:53:44.481586Z","iopub.execute_input":"2022-12-09T09:53:44.482094Z","iopub.status.idle":"2022-12-09T09:53:44.703091Z","shell.execute_reply.started":"2022-12-09T09:53:44.482052Z","shell.execute_reply":"2022-12-09T09:53:44.701906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This plot shows that we are dealing with a highly unbalanced dataset.","metadata":{}},{"cell_type":"code","source":"# Plot other dataset features\nfig, ax = plt.subplots(2, 2, figsize=(20, 8))\n\nsns.countplot(train_df[\"laterality\"], ax=ax[0, 0], palette=\"Reds_r\")\nsns.countplot(train_df[\"view\"], ax=ax[0, 1], palette=\"Blues_r\")\nsns.countplot(train_df[\"implant\"], ax=ax[1, 0], palette=\"BrBG_r\")\nsns.countplot(train_df[\"density\"], ax=ax[1, 1], palette=\"pink\");","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:56:24.001553Z","iopub.execute_input":"2022-12-09T09:56:24.002011Z","iopub.status.idle":"2022-12-09T09:56:25.001645Z","shell.execute_reply.started":"2022-12-09T09:56:24.001973Z","shell.execute_reply":"2022-12-09T09:56:25.000595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See age distribution\npatient_ages = train_df[train_df[\"age\"].isnull() == False].groupby(\n    \"patient_id\"\n)[\"age\"].apply(lambda x: np.unique(x)[0])\nplt.figure(figsize=(20,8))\n\n\nsns.histplot(train_df[\"age\"], color=\"purple\", bins=60)\nplt.title(\"Age distribution of patients\");","metadata":{"execution":{"iopub.status.busy":"2022-12-09T09:58:20.423613Z","iopub.execute_input":"2022-12-09T09:58:20.424068Z","iopub.status.idle":"2022-12-09T09:58:21.304582Z","shell.execute_reply.started":"2022-12-09T09:58:20.424029Z","shell.execute_reply":"2022-12-09T09:58:21.30377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"3\"></a>\n<h1 style='background:#815391; border:0; color:white'><center>Images Overview</center></h1>","metadata":{}},{"cell_type":"markdown","source":"Firstly, let's prepare the functions for reading and visualization DICOM images.","metadata":{}},{"cell_type":"code","source":"def load_dicom(path):\n    dicom = pydicom.dcmread(path)\n    \n    image = dicom.pixel_array\n    image = (image - image.min()) / (image.max() - image.min())\n    \n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        image = 1 - image\n        \n    return image","metadata":{"execution":{"iopub.status.busy":"2022-12-09T10:05:52.493523Z","iopub.execute_input":"2022-12-09T10:05:52.493951Z","iopub.status.idle":"2022-12-09T10:05:52.501268Z","shell.execute_reply.started":"2022-12-09T10:05:52.493908Z","shell.execute_reply":"2022-12-09T10:05:52.499813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize_samples(path, df):\n    fig, ax = plt.subplots(1, df.shape[0], figsize=(20, 8), sharey=False, sharex=False)\n    \n    i = 0\n    \n    for _, row in df.iterrows():\n        patient_id = row[\"patient_id\"]\n        image_id = row[\"image_id\"]\n        dicom_path = f\"{path}/{patient_id}/{image_id}.dcm\"\n        data = load_dicom(dicom_path)\n        \n        ax[i].imshow(data, cmap=\"gray\")\n        ax[i].set_title(f\"Image ID: {image_id},\\nSide: {row['laterality']},\\nCancer positive: {row['cancer']},\\nAge:{row['age']}\")\n        \n        i += 1\n        \n    plt.suptitle(f\"Patient ID: {patient_id}\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-09T10:05:53.650066Z","iopub.execute_input":"2022-12-09T10:05:53.650489Z","iopub.status.idle":"2022-12-09T10:05:53.659536Z","shell.execute_reply.started":"2022-12-09T10:05:53.650455Z","shell.execute_reply":"2022-12-09T10:05:53.658167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Randomly choice several unique samples from the training dataset\nsample_idx = np.random.choice(train_df[\"patient_id\"].unique(), 6, replace=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-09T10:07:28.585843Z","iopub.execute_input":"2022-12-09T10:07:28.586297Z","iopub.status.idle":"2022-12-09T10:07:28.593281Z","shell.execute_reply.started":"2022-12-09T10:07:28.58626Z","shell.execute_reply":"2022-12-09T10:07:28.592138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize selected samples\nfor patient_id in sample_idx:\n    sample = train_df[train_df[\"patient_id\"] == patient_id]\n    \n    visualize_samples(train_images_dir, sample)","metadata":{"execution":{"iopub.status.busy":"2022-12-09T10:07:28.876515Z","iopub.execute_input":"2022-12-09T10:07:28.876949Z","iopub.status.idle":"2022-12-09T10:08:28.275189Z","shell.execute_reply.started":"2022-12-09T10:07:28.876912Z","shell.execute_reply":"2022-12-09T10:08:28.273979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Let's look at samples that contain cancer**","metadata":{}},{"cell_type":"code","source":"cancer_positive_samples = train_df[(train_df[\"cancer\"] == 1)].head(4)\n    \nvisualize_samples(train_images_dir, cancer_positive_samples)","metadata":{"execution":{"iopub.status.busy":"2022-12-09T10:08:45.603605Z","iopub.execute_input":"2022-12-09T10:08:45.604117Z","iopub.status.idle":"2022-12-09T10:08:55.121129Z","shell.execute_reply.started":"2022-12-09T10:08:45.604079Z","shell.execute_reply":"2022-12-09T10:08:55.119943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**And also look at healthy samples**","metadata":{}},{"cell_type":"code","source":"cancer_negative_samples = train_df[(train_df[\"cancer\"] == 0)].head(4)\n\nvisualize_samples(train_images_dir, cancer_negative_samples)","metadata":{"execution":{"iopub.status.busy":"2022-12-09T10:09:23.393965Z","iopub.execute_input":"2022-12-09T10:09:23.394361Z","iopub.status.idle":"2022-12-09T10:09:36.546303Z","shell.execute_reply.started":"2022-12-09T10:09:23.394328Z","shell.execute_reply":"2022-12-09T10:09:36.54522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"4\"></a>\n<h1 style='background:#815391; border:0; color:white'><center>DICOM Metadata Overview</center></h1>","metadata":{}},{"cell_type":"code","source":"dicom_example = f\"{train_images_dir}/10086/1479874325.dcm\"\npydicom.dcmread(dicom_example)","metadata":{"execution":{"iopub.status.busy":"2022-12-09T10:18:11.618256Z","iopub.execute_input":"2022-12-09T10:18:11.618909Z","iopub.status.idle":"2022-12-09T10:18:11.633339Z","shell.execute_reply.started":"2022-12-09T10:18:11.618861Z","shell.execute_reply":"2022-12-09T10:18:11.632159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[Back to navigation](#0)","metadata":{}}]}