{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Table of Contents\n\n* [1. Loading libraries](#1)\n* [2. Exploring metadata](#2)\n    * [2.1. Basic tabular EDA](#2.1)\n    * [2.2. Sanity check](#2.2)\n    * [2.3. Missing values](#2.3)\n    * [2.4. Metadata column analysis](#2.4)\n    * [2.5. Correlation table](#2.5)\n* [3. Exploring dicom images](#3)","metadata":{}},{"cell_type":"markdown","source":"# 1. Loading libraries <a id=\"1\"></a>","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport missingno as msno # missing data visualization\n\nimport matplotlib.pyplot as plt  # plotting\n\nimport os    # file manipulation\n\nimport plotly.express as px   # interactive plotting\nimport plotly.graph_objects as go\n\nimport pydicom  # to handle dicom files\nfrom pydicom.pixel_data_handlers import apply_voi_lut\nimport random   # generating (pseudo-)random numbers","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:16:14.374126Z","iopub.execute_input":"2022-12-12T20:16:14.374971Z","iopub.status.idle":"2022-12-12T20:16:15.659081Z","shell.execute_reply.started":"2022-12-12T20:16:14.374835Z","shell.execute_reply":"2022-12-12T20:16:15.658173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -qU pylibjpeg pylibjpeg-openjpeg pylibjpeg-libjpeg python-gdcm","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:16:15.660526Z","iopub.execute_input":"2022-12-12T20:16:15.661108Z","iopub.status.idle":"2022-12-12T20:16:27.186597Z","shell.execute_reply.started":"2022-12-12T20:16:15.661072Z","shell.execute_reply":"2022-12-12T20:16:27.185366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -qU dython\nfrom dython import nominal # for correlation matrix","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:16:27.188714Z","iopub.execute_input":"2022-12-12T20:16:27.189103Z","iopub.status.idle":"2022-12-12T20:16:39.006671Z","shell.execute_reply.started":"2022-12-12T20:16:27.189068Z","shell.execute_reply":"2022-12-12T20:16:39.005129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Exploring metadata <a id=\"2\"></a> \n## 2.1. Basic tabular EDA  <a id=\"2.1\"></a>\nFirst, we declare a global variable for the main working directory and use it to define absolute paths for train/test directories and metadata files.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"BASE_DIR = \"/kaggle/input/rsna-breast-cancer-detection\"\ntrain_dir = os.path.join(BASE_DIR, \"train_images\")\ntest_dir = os.path.join(BASE_DIR, \"test_images\")\n\ntrain_meta_file = os.path.join(BASE_DIR, \"train.csv\")\ntest_meta_file = os.path.join(BASE_DIR, \"test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:16:39.010339Z","iopub.execute_input":"2022-12-12T20:16:39.010712Z","iopub.status.idle":"2022-12-12T20:16:39.018227Z","shell.execute_reply.started":"2022-12-12T20:16:39.010677Z","shell.execute_reply":"2022-12-12T20:16:39.016637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loading train/test metadata into pandas dataframes and inspecting some rows.","metadata":{}},{"cell_type":"code","source":"train_meta_data = pd.read_csv(train_meta_file)\ntest_meta_data = pd.read_csv(test_meta_file)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:16:39.019824Z","iopub.execute_input":"2022-12-12T20:16:39.020179Z","iopub.status.idle":"2022-12-12T20:16:39.116022Z","shell.execute_reply.started":"2022-12-12T20:16:39.020148Z","shell.execute_reply":"2022-12-12T20:16:39.11484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_meta_data.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:16:39.117497Z","iopub.execute_input":"2022-12-12T20:16:39.117867Z","iopub.status.idle":"2022-12-12T20:16:39.143425Z","shell.execute_reply.started":"2022-12-12T20:16:39.117817Z","shell.execute_reply":"2022-12-12T20:16:39.142022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_meta_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:16:39.145659Z","iopub.execute_input":"2022-12-12T20:16:39.146038Z","iopub.status.idle":"2022-12-12T20:16:39.153419Z","shell.execute_reply.started":"2022-12-12T20:16:39.146003Z","shell.execute_reply":"2022-12-12T20:16:39.152147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_meta_data.columns","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:16:39.154819Z","iopub.execute_input":"2022-12-12T20:16:39.155231Z","iopub.status.idle":"2022-12-12T20:16:39.165521Z","shell.execute_reply.started":"2022-12-12T20:16:39.155198Z","shell.execute_reply":"2022-12-12T20:16:39.164629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_meta_data[\"patient_id\"].unique())","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:16:39.168383Z","iopub.execute_input":"2022-12-12T20:16:39.168742Z","iopub.status.idle":"2022-12-12T20:16:39.18122Z","shell.execute_reply.started":"2022-12-12T20:16:39.168711Z","shell.execute_reply":"2022-12-12T20:16:39.179904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train metadata contains 54706 records and spans over 12 columns, as mentioned in the Data Description of the Competition. In total, there are 11913 different patients. Some columns are provided only for train metadata, not for test one (see below).","metadata":{}},{"cell_type":"code","source":"test_meta_data","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:16:39.182981Z","iopub.execute_input":"2022-12-12T20:16:39.183671Z","iopub.status.idle":"2022-12-12T20:16:39.198361Z","shell.execute_reply.started":"2022-12-12T20:16:39.183619Z","shell.execute_reply":"2022-12-12T20:16:39.197191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Test metadata contains only 4 records for 1 patient.","metadata":{}},{"cell_type":"markdown","source":"## 2.2. Sanity check  <a id=\"2.2\"></a>\nHere we check that all records in the train/test metadata tables have a corresponding dicom file.","metadata":{}},{"cell_type":"code","source":"def sanity_check(file_type):\n    \"\"\"\n    Check metadata file\n    \n    This function performs a simple check of whether all records in a metadata file have\n    a corresponding mammogram in dicom format. It prints the total number of records \n    in the metadata file and percentage of metadata records with a .dcm file.\n    \n    Parameters\n    ----------\n    file_type : str\n        Type of the metadata file, either \"train\" or \"test\".\n    \"\"\"\n    \n    if file_type not in [\"train\", \"test\"]:\n        raise ValueError(\"Input parameter file_type should be either 'train' or 'test'!\")\n    \n    meta_file = os.path.join(BASE_DIR, f\"{file_type}.csv\")\n    data_dir = os.path.join(BASE_DIR, f\"{file_type}_images\")\n    meta_data = pd.read_csv(meta_file)\n\n    missing_file_number = 0\n    for i in meta_data.index:\n        patient_id = meta_data.iloc[i][\"patient_id\"]\n        image_id = meta_data.iloc[i][\"image_id\"]\n        file_name = os.path.join(data_dir, str(patient_id), str(image_id)+\".dcm\")\n\n        if not os.path.isfile(file_name):\n            missing_file_number += 1\n            \n    string = f\"Metadata file {file_type}.csv contains {len(meta_data)} records. Percentage of records with a .dcm file: {100*(len(meta_data)-missing_file_number)/len(meta_data)} %\"\n    print(string)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:16:39.199671Z","iopub.execute_input":"2022-12-12T20:16:39.200057Z","iopub.status.idle":"2022-12-12T20:16:39.210533Z","shell.execute_reply.started":"2022-12-12T20:16:39.200025Z","shell.execute_reply":"2022-12-12T20:16:39.209631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sanity_check(\"train\")\nsanity_check(\"test\")","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:16:39.212105Z","iopub.execute_input":"2022-12-12T20:16:39.21262Z","iopub.status.idle":"2022-12-12T20:17:29.440174Z","shell.execute_reply.started":"2022-12-12T20:16:39.212586Z","shell.execute_reply":"2022-12-12T20:17:29.437929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.3. Missing values  <a id=\"2.3\"></a>\nTo explore the data completeness, we are going to use the missingno library.","metadata":{}},{"cell_type":"code","source":"msno.matrix(train_meta_data)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:17:29.444192Z","iopub.execute_input":"2022-12-12T20:17:29.444576Z","iopub.status.idle":"2022-12-12T20:17:30.254058Z","shell.execute_reply.started":"2022-12-12T20:17:29.444539Z","shell.execute_reply":"2022-12-12T20:17:30.252912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As one can see, the train metadata is quite complete, with only three columns containing some missing values: age, BIRADS, and density.","metadata":{}},{"cell_type":"markdown","source":"## 2.4. Metadata column analysis  <a id=\"2.4\"></a>\nHere we are going to interactively investigate the distribution of data in each column. For this, we can use the Plotly library and custom function. It means that one can press buttons, zoom in and out, or even hide data on the figures.","metadata":{}},{"cell_type":"code","source":"def plot_df_distribution(df, columns, dropna=False):\n    \"\"\"\n    Display distribution of data in dataframe columns\n    \n    This function plots the data distribution in given columns of a Pandas dataframe.\n    One can control whether to drop NaN values before plotting.\n    \n    Parameters\n    ----------\n    df : pandas.DataFrame\n        Source tabular data.\n    columns: list\n        List of columns to include in the figure.\n    dropna: bool, default False\n        Parameter to include or drop NaN.\n    \"\"\"\n    \n    fig = go.Figure()\n\n    buttons = []\n    for i, col in enumerate(columns):\n        visible = np.zeros(len(columns)).tolist()\n        visible[i] = 1\n        visible = [True if i==1 else False for i in visible]\n        d = dict(label=col,\n                 method=\"update\",\n                 args=[{\"visible\": visible, \"showlegend\": True}]\n                )\n        buttons.append(d)\n\n        s = df[col]\n        fig.add_trace(\n            go.Pie(\n                labels = s.value_counts(dropna=dropna).index,\n                values = s.value_counts(dropna=dropna).values,\n                name = col,\n                hole=.3\n            ),\n        )\n        \n    fig.update_layout(\n        updatemenus=[\n            go.layout.Updatemenu(\n                active=0,\n                buttons=buttons\n            )\n        ],\n        width=800,\n        height=600,\n        autosize=False,\n        uniformtext_minsize=10,\n        uniformtext_mode=\"hide\"\n    )\n\n    fig.update_traces(textposition=\"inside\", textinfo=\"percent+label\")\n\n    for i, trace in enumerate(fig.data):\n        if i != 0:\n            trace.update(visible=False)\n\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:17:30.255684Z","iopub.execute_input":"2022-12-12T20:17:30.256175Z","iopub.status.idle":"2022-12-12T20:17:30.442701Z","shell.execute_reply.started":"2022-12-12T20:17:30.256141Z","shell.execute_reply":"2022-12-12T20:17:30.441053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = [\n    \"view\",\n    \"site_id\",\n    \"machine_id\",\n    \"laterality\",\n    \"cancer\",\n    \"biopsy\",\n    \"invasive\",\n    \"BIRADS\",\n    \"implant\",\n    \"density\",\n    \"difficult_negative_case\"\n]\n    \nplot_df_distribution(train_meta_data, columns, dropna=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:17:30.444671Z","iopub.execute_input":"2022-12-12T20:17:30.445108Z","iopub.status.idle":"2022-12-12T20:17:30.538933Z","shell.execute_reply.started":"2022-12-12T20:17:30.445072Z","shell.execute_reply":"2022-12-12T20:17:30.53792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here are some insights:\n* The laterality column is fairly balanced\n* The majority of the view column are either CC or MLO\n* There are slightly more images made at site 1 than at site 2","metadata":{}},{"cell_type":"markdown","source":"One more interesting feature is how many images are for each patient in the dataset. As we can see below, the majority of patients have 4 images, 2 per breast. However, some of the patients have more attached images, between 5 and 14.","metadata":{}},{"cell_type":"code","source":"images_per_patient = train_meta_data[\"patient_id\"].value_counts()\n\nfig = go.Figure()\nfig.add_trace(\n    go.Bar(\n        x = images_per_patient.value_counts().index,\n        y = images_per_patient.value_counts()\n    )\n)\nfig.update_layout(\n    width=600,\n    height=400,\n    xaxis_title=\"Images per patient\",\n    yaxis_title=\"Patients\",\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:17:30.540157Z","iopub.execute_input":"2022-12-12T20:17:30.540472Z","iopub.status.idle":"2022-12-12T20:17:30.570278Z","shell.execute_reply.started":"2022-12-12T20:17:30.540443Z","shell.execute_reply":"2022-12-12T20:17:30.568834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The next type of visualization allows seeing how patients with cancer are distributed across other categories. Each bar on the barchart is annotated with the percentage level, so one can see the relative fraction of patients having cancer (if the percentage is not visible, zoom in!).","metadata":{}},{"cell_type":"code","source":"def plot_cancer_distribution(meta_df, columns):\n    \"\"\"\n    Display cancer value distribution accross columns\n    \n    This function plots how many cancer positive/negative values in given columns of \n    a Pandas dataframe. This function is applicable only to the metadata file of RSNA 2022\n    Kaggle competition.\n    \n    Parameters\n    ----------\n    meta_df : pandas.DataFrame\n        Source tabular data with a boolean column 'cancer'.\n    columns: list\n        List of columns to include in the figure.\n    \"\"\"\n\n    fig = go.Figure()\n\n    buttons = []\n    for i, col in enumerate(columns):\n        visible = np.zeros(2*len(columns)).tolist()\n        visible[2*i] = 1\n        visible[2*i+1] = 1\n        visible = [True if i==1 else False for i in visible]\n        axis_type = None\n        if col != \"age\":\n            axis_type = \"category\"\n        d = dict(label=col,\n                method=\"update\",\n                args=[{\"visible\": visible, \"showlegend\": True}, {\"xaxis.range\": None, \"xaxis.type\": axis_type}]\n            )\n        buttons.append(d)\n\n        s1 = meta_df[meta_df[\"cancer\"] == 1][col]\n        s0 = meta_df[meta_df[\"cancer\"] == 0][col]\n\n        if col == \"age\":\n            x1 = s1.value_counts().index\n            x0 = s0.value_counts().index\n        else:\n            x1 = [str(i) for i in s1.value_counts().index]\n            x0 = [str(i) for i in s0.value_counts().index]\n\n        total = meta_df[col].value_counts()\n\n        fig.add_trace(go.Bar(\n            x=x1,\n            y=s1.value_counts(),\n            name=\"Cancer\",\n            marker_color=\"indianred\",\n            customdata = [100*count/total.at[idx] for (idx,count) in s1.value_counts().items()],\n            texttemplate=\"%{customdata:.1f} %\",\n            textposition=\"inside\",\n            textangle=0\n        ))\n\n        fig.add_trace(go.Bar(\n            x=x0,\n            y=s0.value_counts(),\n            name=\"No cancer\",\n            marker_color=\"lightgreen\",\n            customdata = [100*count/total.at[idx] for (idx,count) in s0.value_counts().items()],\n            texttemplate=\"%{customdata:.1f} %\",\n            textposition=\"inside\",\n            textangle=0\n        ))\n\n    fig.update_layout(\n        updatemenus=[\n            go.layout.Updatemenu(\n                active=0,\n                buttons=buttons\n            )\n        ],\n        width=800,\n        height=600,\n        barmode=\"stack\",\n        uniformtext=dict(mode=\"hide\", minsize=8)\n    )\n\n    for i, trace in enumerate(fig.data):\n        if i > 1 :\n            trace.update(visible=False)\n\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:17:30.572403Z","iopub.execute_input":"2022-12-12T20:17:30.573122Z","iopub.status.idle":"2022-12-12T20:17:30.591583Z","shell.execute_reply.started":"2022-12-12T20:17:30.573072Z","shell.execute_reply":"2022-12-12T20:17:30.59047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = [\n    \"age\",\n    \"view\",\n    \"site_id\",\n    \"machine_id\",\n    \"laterality\",\n    \"biopsy\",\n    \"invasive\",\n    \"BIRADS\",\n    \"implant\",\n    \"density\",\n    \"difficult_negative_case\"\n]\n\nplot_cancer_distribution(train_meta_data, columns)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:17:30.593166Z","iopub.execute_input":"2022-12-12T20:17:30.593665Z","iopub.status.idle":"2022-12-12T20:17:30.811221Z","shell.execute_reply.started":"2022-12-12T20:17:30.593621Z","shell.execute_reply":"2022-12-12T20:17:30.809907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see some new patterns in data:\n* The age distribution shows that most patients are between 40 and 75 years old.\n* All patients with cancer had biopsy.","metadata":{}},{"cell_type":"markdown","source":"## 2.5. Correlation table  <a id=\"2.5\"></a>\nHere we can see correlations between various columns summarized in a matrix.","metadata":{}},{"cell_type":"code","source":"nominal.associations(train_meta_data,figsize=(20,10),mark_columns=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:17:30.813318Z","iopub.execute_input":"2022-12-12T20:17:30.81377Z","iopub.status.idle":"2022-12-12T20:17:32.714897Z","shell.execute_reply.started":"2022-12-12T20:17:30.813726Z","shell.execute_reply":"2022-12-12T20:17:32.714013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Exploring dicom images <a id=\"3\"></a>\nFinally, we are going to look into raw images using the pydicom library. First, we define a function to perform scaling and another function to link images with the relevant metadata. The last function would plot all the dicom files for a given patient ID.","metadata":{}},{"cell_type":"code","source":"def rescale_img_to_hu(dcm):\n    \"\"\"Rescales the image to Hounsfield unit\"\"\"\n    data = dcm.pixel_array\n    if dcm.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    return data * dcm.RescaleSlope + dcm.RescaleIntercept","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:17:32.716269Z","iopub.execute_input":"2022-12-12T20:17:32.717116Z","iopub.status.idle":"2022-12-12T20:17:32.724066Z","shell.execute_reply.started":"2022-12-12T20:17:32.717081Z","shell.execute_reply":"2022-12-12T20:17:32.722568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fetch_annotation(patient_id, image_id):\n    \"\"\"\n    Parse annotation for a given image\n    \n    This function parses a row from the metadata table as a multiline string.\n    \n    Parameters\n    ----------\n    patient_id : int\n        patient ID, one of available options in train_meta_data[\"patient_id\"].unique().\n    iamge_id: int\n        image ID, one of available options in train_meta_data[train_meta_data[\"patient_id\"]==patient_id][\"image_id\"].\n    \n    Returns\n    -------\n    str\n       Annotation string. \n    \"\"\"\n    data = train_meta_data[(train_meta_data[\"patient_id\"]==patient_id) & \n                           (train_meta_data[\"image_id\"]==image_id)].squeeze()\n    data = data.replace(np.nan, \"NaN\")\n    annotation_string = \"\"\n    for k,v in data.items():\n        if isinstance(v, float):\n            v = int(v)      \n        annotation_string += f\"{k:<25}{v:>10}<br>\"\n    return annotation_string","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:17:32.725775Z","iopub.execute_input":"2022-12-12T20:17:32.726804Z","iopub.status.idle":"2022-12-12T20:17:32.735268Z","shell.execute_reply.started":"2022-12-12T20:17:32.726764Z","shell.execute_reply":"2022-12-12T20:17:32.734095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_images_for_patient(patient_id):\n    \"\"\"\n    Display image for a given patient ID\n    \n    This function plots all available mammograms together with the associated metadata.\n    \n    Parameters\n    ----------\n    patient_id : int\n        patient ID, one of available options in train_meta_data[\"patient_id\"].unique().\n    \"\"\"\n\n    print(f\"{'#'*80}\\nProcessing data for patient {patient_id}\\n{'#'*80}\")\n    patient_dir = os.path.join(train_dir, str(patient_id))\n    images = os.listdir(patient_dir)\n    print(f\"Number of images for patient: {len(images)}\")\n    \n    fig = go.Figure()\n    \n    buttons = []\n    for i, image in enumerate(images):\n        image_id = int(image.split(\".\")[0])\n        text = fetch_annotation(patient_id, image_id)\n        \n        annotation=[\n            dict(\n                text=text,\n                align=\"left\",\n                showarrow=False,\n                xref=\"paper\",\n                yref=\"paper\",\n                x=2.3,\n                y=0.5,\n                bordercolor=\"black\",\n                borderwidth=1,\n                font=dict(family=\"Courier New\", size=18)\n            )\n        ]\n        if i == 0:\n            annotation0 = annotation\n        \n        visible = np.zeros(len(images)).tolist()\n        visible[i] = 1\n        visible = [True if i==1 else False for i in visible]\n        d = dict(label=image,\n                 method=\"update\",\n                 args=[{\"visible\": visible}, {\"annotations\": annotation}]\n                )\n        buttons.append(d)\n        \n        image_path = os.path.join(patient_dir, image)\n        dcm = pydicom.dcmread(image_path)\n        f = rescale_img_to_hu(dcm)\n        fig.add_trace(px.imshow(f, binary_string=True).data[0])\n\n    fig.update_layout(\n        updatemenus=[\n            go.layout.Updatemenu(\n                active=0,\n                buttons=buttons\n            )\n        ],\n        width=1000,\n        height=600,\n        margin=dict(r=500),\n        xaxis={\"visible\": False, \"showticklabels\": False},\n        yaxis={\"visible\": False, \"showticklabels\": False} ,\n        annotations=annotation0\n    )\n\n    for i, trace in enumerate(fig.data):\n        if i != 0:\n            trace.update(visible=False)\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:17:32.73704Z","iopub.execute_input":"2022-12-12T20:17:32.737505Z","iopub.status.idle":"2022-12-12T20:17:32.756339Z","shell.execute_reply.started":"2022-12-12T20:17:32.737456Z","shell.execute_reply":"2022-12-12T20:17:32.755023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here is a collection of dicom images for a random patient. Note that unlike matplotlib plots these figures are interactive.","metadata":{}},{"cell_type":"code","source":"patient_id = int(random.choice(os.listdir(train_dir)))\nshow_images_for_patient(patient_id)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T20:17:32.758115Z","iopub.execute_input":"2022-12-12T20:17:32.759309Z","iopub.status.idle":"2022-12-12T20:17:38.030389Z","shell.execute_reply.started":"2022-12-12T20:17:32.759271Z","shell.execute_reply":"2022-12-12T20:17:38.028615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"If you found the listed interactive EDA methods useful, please upvote this notebook! Do not hesitate to leave comments/remarks below. And thanks for reading! ","metadata":{}}]}