{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Libraries 📚","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nfrom os import listdir\nimport glob\nfrom pathlib import Path\n\nfrom tqdm.notebook import tqdm\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-10T09:32:32.705488Z","iopub.execute_input":"2022-12-10T09:32:32.705979Z","iopub.status.idle":"2022-12-10T09:32:32.712655Z","shell.execute_reply.started":"2022-12-10T09:32:32.705947Z","shell.execute_reply":"2022-12-10T09:32:32.711739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get Data","metadata":{}},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/rsna-breast-cancer-detection\"\n\nTRAIN_PATH = \"/kaggle/input/rsna-breast-cancer-detection/train_images\"\nTEST_PATH = \"/kaggle/input/rsna-breast-cancer-detection/test_images\"\n\nTRAIN_CSV_PATH = os.path.join(BASE_PATH, \"train.csv\")\nTEST_CSV_PATH = os.path.join(BASE_PATH, \"test.csv\")\nSUB_CSV_PATH = os.path.join(BASE_PATH, \"sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:32.714409Z","iopub.execute_input":"2022-12-10T09:32:32.715228Z","iopub.status.idle":"2022-12-10T09:32:32.728385Z","shell.execute_reply.started":"2022-12-10T09:32:32.715195Z","shell.execute_reply":"2022-12-10T09:32:32.727254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(TRAIN_CSV_PATH)\ntest_df = pd.read_csv(TEST_CSV_PATH)\nsub_df = pd.read_csv(SUB_CSV_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:32.729818Z","iopub.execute_input":"2022-12-10T09:32:32.730156Z","iopub.status.idle":"2022-12-10T09:32:32.824236Z","shell.execute_reply.started":"2022-12-10T09:32:32.730119Z","shell.execute_reply":"2022-12-10T09:32:32.823122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:32.825898Z","iopub.execute_input":"2022-12-10T09:32:32.826623Z","iopub.status.idle":"2022-12-10T09:32:32.834788Z","shell.execute_reply.started":"2022-12-10T09:32:32.826568Z","shell.execute_reply":"2022-12-10T09:32:32.833673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:32.838017Z","iopub.execute_input":"2022-12-10T09:32:32.838349Z","iopub.status.idle":"2022-12-10T09:32:32.85952Z","shell.execute_reply.started":"2022-12-10T09:32:32.838302Z","shell.execute_reply":"2022-12-10T09:32:32.858271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Amount of the patients in train_images: {len(os.listdir(TRAIN_PATH))}\")","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:32.861248Z","iopub.execute_input":"2022-12-10T09:32:32.861828Z","iopub.status.idle":"2022-12-10T09:32:32.87465Z","shell.execute_reply.started":"2022-12-10T09:32:32.861795Z","shell.execute_reply":"2022-12-10T09:32:32.87345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Amount of the patients in test_images: {len(os.listdir(TEST_PATH))}\")","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:32.876047Z","iopub.execute_input":"2022-12-10T09:32:32.877225Z","iopub.status.idle":"2022-12-10T09:32:32.885038Z","shell.execute_reply.started":"2022-12-10T09:32:32.877161Z","shell.execute_reply":"2022-12-10T09:32:32.883831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check out meta data\nexample = \"/kaggle/input/rsna-breast-cancer-detection/train_images/10042/102733848.dcm\"\npydicom.dcmread(example)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:32.8865Z","iopub.execute_input":"2022-12-10T09:32:32.886885Z","iopub.status.idle":"2022-12-10T09:32:32.904178Z","shell.execute_reply.started":"2022-12-10T09:32:32.886852Z","shell.execute_reply":"2022-12-10T09:32:32.903101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ## About Medical Images\n \n Hounsfield Units (HU) are used in CT images it is a measure of radio-density, calibrated to distilled water and free air.\n\n HUs can be calculated from the pixel data with a Dicom Image via two methods.\n \n----------------------------------\n  1. The first (and harder) method involves using the Slope and Intercept value from the Dicom image and applying it to a target pixel. (HU = m * P + b)\n  \n  Normally, these values are stored in the DICOM file itself. The tags are generally called the Rescale Slope and Rescale Intercept, and typically have values of 1 and -1024, respectively.\n  \n  2. Better and easier way is by using ROIMean function.\n  \n----------------------------------\n ## References || Sources\n \n* https://theaisummer.com/medical-image-python/\n* medicalconnections.co.uk/kb/Hounsfield-Units\n* https://gist.github.com/somada141/df9af37e567ba566902e\n* https://www.idlcoyote.com/fileio_tips/hounsfield.html\n* https://vincentblog.xyz/posts/medical-images-in-python-computed-tomography\n* https://vincentblog.xyz/posts/magnetic-resonance-imaging-and-positron-emission-tomography-images","metadata":{}},{"cell_type":"code","source":"# source: https://www.kaggle.com/code/allunia/rsna-csf-cervical-spine-fracture-eda/notebook\ndef rescale_img_to_hu(dcm_ds):\n    \"\"\"Rescales the image to Hounsfield unit.\"\"\"\n    data = dcm_ds.pixel_array\n    if dcm_ds.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    return data * dcm_ds.RescaleSlope + dcm_ds.RescaleIntercept\n\ndef show_images_for_patient(patient_id):\n    patient_dir = os.path.join(TRAIN_PATH, str(patient_id))\n    num_images = len(glob.glob(f\"{patient_dir}/*\"))\n    print(f\"Number of images for patient: {num_images}\")\n    fig, axs = plt.subplots(2, 2, figsize=(24,15))\n    axs = axs.flatten()\n    for i, img_path in enumerate(list(Path(patient_dir).iterdir())):\n        ds = pydicom.dcmread(img_path)\n        axs[i].imshow(rescale_img_to_hu(ds), cmap=\"bone\")","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:32.905428Z","iopub.execute_input":"2022-12-10T09:32:32.905765Z","iopub.status.idle":"2022-12-10T09:32:32.915047Z","shell.execute_reply.started":"2022-12-10T09:32:32.905735Z","shell.execute_reply":"2022-12-10T09:32:32.91392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_images_for_patient(10144)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:32.919208Z","iopub.execute_input":"2022-12-10T09:32:32.91956Z","iopub.status.idle":"2022-12-10T09:32:39.835444Z","shell.execute_reply.started":"2022-12-10T09:32:32.919506Z","shell.execute_reply":"2022-12-10T09:32:39.834377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA & Visualization of csv file","metadata":{}},{"cell_type":"code","source":"# highly imbalanced data\nax1 = sns.countplot(train_df[\"cancer\"])\nfor container in ax1.containers:\n    ax1.bar_label(container)\nplt.title('Distribution of targets');","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:39.836956Z","iopub.execute_input":"2022-12-10T09:32:39.837288Z","iopub.status.idle":"2022-12-10T09:32:40.046337Z","shell.execute_reply.started":"2022-12-10T09:32:39.837257Z","shell.execute_reply":"2022-12-10T09:32:40.045149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(train_df.groupby('patient_id').size())\nplt.title('Number of images per patient')","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:40.047792Z","iopub.execute_input":"2022-12-10T09:32:40.048126Z","iopub.status.idle":"2022-12-10T09:32:40.315212Z","shell.execute_reply.started":"2022-12-10T09:32:40.048095Z","shell.execute_reply":"2022-12-10T09:32:40.314402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2,2,figsize=(20,10))\nsns.countplot(train_df.laterality, ax=ax[0,0], palette='Greens_r')\nsns.countplot(train_df.view, ax=ax[0,1], palette='Reds_r')\nsns.countplot(train_df.implant, ax=ax[1,0], palette='Blues_r')\nsns.countplot(train_df.density, ax=ax[1,1], palette='Purples_r');","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:40.316572Z","iopub.execute_input":"2022-12-10T09:32:40.317162Z","iopub.status.idle":"2022-12-10T09:32:40.964576Z","shell.execute_reply.started":"2022-12-10T09:32:40.317128Z","shell.execute_reply":"2022-12-10T09:32:40.963415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(train_df.machine_id, palette='tab10')","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:40.966193Z","iopub.execute_input":"2022-12-10T09:32:40.967489Z","iopub.status.idle":"2022-12-10T09:32:42.094109Z","shell.execute_reply.started":"2022-12-10T09:32:40.967444Z","shell.execute_reply":"2022-12-10T09:32:42.092573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 'cancer', 'implant', 'biopsy', 'laterality'\nhuman_df = train_df.copy()\n\nhuman_df[\"cancer\"] = human_df[\"cancer\"].replace(0, \"non-cancer\")\nhuman_df[\"cancer\"] = human_df[\"cancer\"].replace(1, \"cancer\")\n\nhuman_df[\"implant\"] = human_df[\"implant\"].replace(0, \"non-implant\")\nhuman_df[\"implant\"] = human_df[\"implant\"].replace(1, \"implant\")\n\nhuman_df[\"biopsy\"] = human_df[\"biopsy\"].replace(0, \"non-biopsy\")\nhuman_df[\"biopsy\"] = human_df[\"biopsy\"].replace(1, \"biopsy\")\n\nhuman_df[\"laterality\"] = human_df[\"laterality\"].replace(\"L\", \"left-breast\")\nhuman_df[\"laterality\"] = human_df[\"laterality\"].replace(\"R\", \"right-breast\")\n\nimport plotly.express as px\n\nfig = px.sunburst(\n    human_df,\n    path=['cancer', 'implant', 'laterality', 'biopsy'],\n)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:42.09585Z","iopub.execute_input":"2022-12-10T09:32:42.096234Z","iopub.status.idle":"2022-12-10T09:32:43.184285Z","shell.execute_reply.started":"2022-12-10T09:32:42.096199Z","shell.execute_reply":"2022-12-10T09:32:43.182888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Observation\n\n---------------------------------------------------------------\n* cancer and non-cancer patients are highly imbalanced.\n* left and right breast data are balanced within both cancer and non-cancer patients.\n---------------------------------------------------------------\n* all cancer patients are non-implant.\n* most of the non-cancer patients are non-implant.\n---------------------------------------------------------------\n* all cancer patients had biopsy.\n* some of the non-cancer patients had biopsy.\n---------------------------------------------------------------","metadata":{}},{"cell_type":"code","source":"from pandas_profiling import ProfileReport\n\nprofile = ProfileReport(train_df, title=\"General Report\")\nprofile.to_file(\"general_report.html\")\n\nprofile","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:32:43.186069Z","iopub.execute_input":"2022-12-10T09:32:43.186529Z","iopub.status.idle":"2022-12-10T09:33:00.128534Z","shell.execute_reply.started":"2022-12-10T09:32:43.186485Z","shell.execute_reply":"2022-12-10T09:33:00.127563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Observation\n---------------------------------------------------------------\n* cancer, invasive and biopsy have high correlation.\n* BIRADS has high correlation with difficult negative case.\n* density and side id have high correlation.\n* BIRADS and density have highly missing values.\n---------------------------------------------------------------","metadata":{}},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:33:00.129827Z","iopub.execute_input":"2022-12-10T09:33:00.130164Z","iopub.status.idle":"2022-12-10T09:33:00.15878Z","shell.execute_reply.started":"2022-12-10T09:33:00.130128Z","shell.execute_reply":"2022-12-10T09:33:00.157267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:33:00.16133Z","iopub.execute_input":"2022-12-10T09:33:00.161933Z","iopub.status.idle":"2022-12-10T09:33:00.225947Z","shell.execute_reply.started":"2022-12-10T09:33:00.161875Z","shell.execute_reply":"2022-12-10T09:33:00.224788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age_cancer_df = train_df[[\"age\", \"cancer\"]].dropna()\nage_cancer_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:33:00.227375Z","iopub.execute_input":"2022-12-10T09:33:00.228282Z","iopub.status.idle":"2022-12-10T09:33:00.244645Z","shell.execute_reply.started":"2022-12-10T09:33:00.228248Z","shell.execute_reply":"2022-12-10T09:33:00.243139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.figure_factory as ff\n\nhist_data =[age_cancer_df[\"age\"].values]\ngroup_labels = [\"cancer\"] \n\nfig = ff.create_distplot(hist_data, group_labels)\nfig.update_layout(title_text=\"Age Distribution plot\")\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:33:00.246003Z","iopub.execute_input":"2022-12-10T09:33:00.246318Z","iopub.status.idle":"2022-12-10T09:33:01.054297Z","shell.execute_reply.started":"2022-12-10T09:33:00.246289Z","shell.execute_reply":"2022-12-10T09:33:01.05339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Observation\n* between 50-65 ages, cancer is common.","metadata":{}},{"cell_type":"code","source":"import plotly.express as px\n\nfig = px.box(age_cancer_df, x='cancer', y='age', points=\"all\")\nfig.update_layout(\n    title_text=\"Gender wise Age Spread - Cancer = 1 Non-cancer =0\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:33:01.055645Z","iopub.execute_input":"2022-12-10T09:33:01.056135Z","iopub.status.idle":"2022-12-10T09:33:01.131315Z","shell.execute_reply.started":"2022-12-10T09:33:01.056103Z","shell.execute_reply":"2022-12-10T09:33:01.129805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# examine correlation\ncmap = sns.diverging_palette(2, 165, s=80, l=55, n=9)\ncorrmat = train_df.corr()\nplt.subplots(figsize=(20,20))\nsns.heatmap(corrmat,cmap= cmap,annot=True, square=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:33:01.133408Z","iopub.execute_input":"2022-12-10T09:33:01.133783Z","iopub.status.idle":"2022-12-10T09:33:02.215136Z","shell.execute_reply.started":"2022-12-10T09:33:01.13375Z","shell.execute_reply":"2022-12-10T09:33:02.214146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA of Train Images","metadata":{}},{"cell_type":"code","source":"def load_scans(path, patient_id):\n    dcm_path = path + \"/\" + str(patient_id)\n    slices = [pydicom.dcmread(dcm_path + '/' + file, force=True) for file in os.listdir(dcm_path)]\n    return slices","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:33:02.216534Z","iopub.execute_input":"2022-12-10T09:33:02.216913Z","iopub.status.idle":"2022-12-10T09:33:02.223678Z","shell.execute_reply.started":"2022-12-10T09:33:02.21688Z","shell.execute_reply":"2022-12-10T09:33:02.222166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_ids = train_df.patient_id.unique()\nscans = load_scans(TRAIN_PATH, patient_ids[0])","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:33:02.229615Z","iopub.execute_input":"2022-12-10T09:33:02.229951Z","iopub.status.idle":"2022-12-10T09:33:02.25509Z","shell.execute_reply.started":"2022-12-10T09:33:02.229922Z","shell.execute_reply":"2022-12-10T09:33:02.253841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2,figsize=(20,5))\nsns.histplot(scans[0].pixel_array.flatten(), ax=ax[0])\nax[1].imshow(scans[0].pixel_array)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:33:02.256727Z","iopub.execute_input":"2022-12-10T09:33:02.257183Z","iopub.status.idle":"2022-12-10T09:33:25.772232Z","shell.execute_reply.started":"2022-12-10T09:33:02.257141Z","shell.execute_reply":"2022-12-10T09:33:25.770701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ALL_IMG_PATHS = []\n\nfor patient in os.listdir(TRAIN_PATH):\n    patient_dir = os.path.join(TRAIN_PATH, patient)\n    for filename in os.listdir(patient_dir):\n        path = os.path.join(patient_dir, filename)\n        ALL_IMG_PATHS.append(path)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:59:10.908742Z","iopub.execute_input":"2022-12-10T09:59:10.909163Z","iopub.status.idle":"2022-12-10T09:59:42.33576Z","shell.execute_reply.started":"2022-12-10T09:59:10.909129Z","shell.execute_reply":"2022-12-10T09:59:42.334593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ALL_IMG_PATHS[:5]","metadata":{"execution":{"iopub.status.busy":"2022-12-10T09:59:42.337812Z","iopub.execute_input":"2022-12-10T09:59:42.338232Z","iopub.status.idle":"2022-12-10T09:59:42.346791Z","shell.execute_reply.started":"2022-12-10T09:59:42.338188Z","shell.execute_reply":"2022-12-10T09:59:42.345345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = pydicom.dcmread('/kaggle/input/rsna-breast-cancer-detection/train_images/10006/1459541791.dcm').pixel_array\nimg = np.array(img, dtype = float) \nimg = (img - img.min()) / (img.max() - img.min()) * 255.0  \nimg = img.astype(np.float16)\nplt.imshow(img, cmap=\"bone\")","metadata":{"execution":{"iopub.status.busy":"2022-12-10T10:00:46.288105Z","iopub.execute_input":"2022-12-10T10:00:46.288591Z","iopub.status.idle":"2022-12-10T10:00:52.929873Z","shell.execute_reply.started":"2022-12-10T10:00:46.288549Z","shell.execute_reply":"2022-12-10T10:00:52.928695Z"},"trusted":true},"execution_count":null,"outputs":[]}]}