{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Problem Description <a class=\"anchor\" id=\"Problem Description\"></a>\n   * **problem** : In this competition our goal is to predict the presence or absence of cancer in mammography images. \n   * **helped Notebooks**: in our journey i will use some hopefull ideas from other notebook and i will mention them in description \n   * **Tasks we will cover in this section** \n       1. Per-Processing images \n          - understand the data \n          - explore data from diffrent view perspective \n          - Technic to processes data to feed into the mode \n               1. read images in each case Patient_ID \n               2. Resize the image and Crop the ROI ( region of intersted \n               3. save the process the image in npy format \n               4. extrat the image label from Train.Csv file \n               5. Virtualize few sample \n   ","metadata":{}},{"cell_type":"code","source":"%%capture\n!pip install python-gdcm\n!pip install pylibjpeg\n# !pip install  pylibjpeg pylibjpeg-libjpeg","metadata":{"execution":{"iopub.status.busy":"2023-03-02T17:18:05.015457Z","iopub.execute_input":"2023-03-02T17:18:05.016208Z","iopub.status.idle":"2023-03-02T17:18:20.623471Z","shell.execute_reply.started":"2023-03-02T17:18:05.01617Z","shell.execute_reply":"2023-03-02T17:18:20.622145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import platform, sys, pydicom\n\nprint(\n    platform.platform(),\n    \"\\nPython\", sys.version,\n    \"\\npydicom\", pydicom.__version__\n)","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:46:33.745296Z","iopub.execute_input":"2023-02-25T04:46:33.745616Z","iopub.status.idle":"2023-02-25T04:46:33.892913Z","shell.execute_reply.started":"2023-02-25T04:46:33.74554Z","shell.execute_reply":"2023-02-25T04:46:33.891785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import display_html\ndef restartkernel() :\n    display_html(\"<script>Jupyter.notebook.kernel.restart()</script>\",raw=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:10:54.612162Z","iopub.execute_input":"2023-02-22T19:10:54.612518Z","iopub.status.idle":"2023-02-22T19:10:54.61959Z","shell.execute_reply.started":"2023-02-22T19:10:54.612489Z","shell.execute_reply":"2023-02-22T19:10:54.618221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### import nesecccery packages \n# %matplotlib inline\nimport matplotlib.pyplot as plt \nimport plotly.express as px\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\nfrom tqdm.notebook import tqdm\n\nimport numpy as np \nimport PIL\nimport math\nfrom collections.abc import Iterable\nimport types\nfrom pydicom.data import get_testdata_file\n\nimport pydicom as dc \nimport SimpleITK as sitk\n\n\nimport pandas as pd \nimport cv2\n\nfrom pathlib import Path\nimport os \nfrom glob import glob\n\nimport torch\nimport torchvision\nfrom torchvision import transforms\nimport torchmetrics\nimport pytorch_lightning as pl\nfrom pytorch_lightning.callbacks import ModelCheckpoint\nfrom pytorch_lightning.loggers import TensorBoardLogger\n\nfrom pydicom.pixel_data_handlers import pillow_handler\npillow_handler.PillowJPEGTransferSyntaxes.append('1.2.840.10008.1.2.4.70')\n\n#from pytorch_lightning.plugins.training_type.ddp import DDPPlugin\nimport warnings \ndef fxn():\n    warnings.warn(\"deprecated\", DeprecationWarning)\n\nwith warnings.catch_warnings():\n    warnings.simplefilter(\"ignore\")\n    fxn()\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-02T18:48:19.966307Z","iopub.execute_input":"2023-03-02T18:48:19.96793Z","iopub.status.idle":"2023-03-02T18:48:19.976374Z","shell.execute_reply.started":"2023-03-02T18:48:19.967892Z","shell.execute_reply":"2023-03-02T18:48:19.975337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Exploer data Csv file** <a class=\"anchor\" id =\"Exploer data Csv file\"></a>\n- we will read the file csv file and check if the data is unbalane corresponding Class we have and let us clean and remove informations thses wil not use in our task \n    1. remvoe unecessery Columns \n    2. check the Class balancing \n    3. read few samples from data \n       - **Notation :** keep in mind here each patient has own series diagnois Slices Image Dicom File followed by [Case_study] **-->** [Slices-Images]\n\n     ","metadata":{}},{"cell_type":"code","source":"Path_data = \"../input/rsna-breast-cancer-detection/train.csv\"\ndata_df = pd.read_csv(Path_data)\n### let us get the number samples we have \nnumber_samples =len(data_df.patient_id)\nlist_col = list(data_df.columns)\nprint(f\"number of total samples {number_samples }\\n Columns dataFrame {list_col}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:47:03.122555Z","iopub.execute_input":"2023-02-25T04:47:03.12389Z","iopub.status.idle":"2023-02-25T04:47:03.26285Z","shell.execute_reply.started":"2023-02-25T04:47:03.12384Z","shell.execute_reply":"2023-02-25T04:47:03.261866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:47:06.395233Z","iopub.execute_input":"2023-02-25T04:47:06.395913Z","iopub.status.idle":"2023-02-25T04:47:06.42014Z","shell.execute_reply.started":"2023-02-25T04:47:06.395876Z","shell.execute_reply":"2023-02-25T04:47:06.418969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nNotaion : we will remove the some columes thses will not use to Pipline models and Per-Processing Step \n-  ['site_id','laterality',\n'view', 'age', 'biopsy',\n'invasive', 'BIRADS',\n'implant','machine_id',\n'difficult_negative_case']\n\"\"\"\ndf_Cols = data_df.drop(['site_id','laterality',\n'view', 'age', 'biopsy',\n'invasive', 'BIRADS',\n'implant','machine_id',\n'difficult_negative_case'],axis=1)\ndata_df_col = df_Cols.fillna(0) ## replace Nan with zeroes\ndata_df_col.head(6)","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:47:12.797262Z","iopub.execute_input":"2023-02-25T04:47:12.797704Z","iopub.status.idle":"2023-02-25T04:47:12.820266Z","shell.execute_reply.started":"2023-02-25T04:47:12.797659Z","shell.execute_reply":"2023-02-25T04:47:12.819376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### check labels Counts between data distrubtion \nTemp_before= data_df_col[\"cancer\"].value_counts()\nsave_count_before = pd.DataFrame({ \"Class\":Temp_before.index,\n                              \"Value\": Temp_before.values})\n## let us figure out solution how to balance the data \n\"\"\"\nsolution 1 : we will try to remove some samples from data that has class 0 \nand make the samples equale class 0 == class 1 \nNotation : this will cause a problem is reducing the number\nsamples Low data repesentation to feed into the model to lean from \n\"\"\"\nsub_stract_classes = save_count_before.Value.iloc[0] - save_count_before.Value.iloc[1]\nprint(sub_stract_classes)\n### now will slice from the results substraction we got \nsave_count_before.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:47:18.530674Z","iopub.execute_input":"2023-02-25T04:47:18.531088Z","iopub.status.idle":"2023-02-25T04:47:18.548975Z","shell.execute_reply.started":"2023-02-25T04:47:18.531057Z","shell.execute_reply":"2023-02-25T04:47:18.547725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alter-","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-warning\" role=\"alert\"> <strong>Important :</strong> here we can see that Class Cancer {0} has a huge amount of data samples greater than Class Cancer {1} which means , insteat we will need to balance the data using Technics such Augmentation that i will not cover in this notebook i wil explore this in another note  <br>\nCancer 97 % <br>\nno Cancer 2.7 %     \n</div>","metadata":{}},{"cell_type":"code","source":"px.pie(save_count_before, values=\"Value\", names=\"Class\", title='Cancer distribution before')","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:47:20.85688Z","iopub.execute_input":"2023-02-25T04:47:20.857256Z","iopub.status.idle":"2023-02-25T04:47:21.829124Z","shell.execute_reply.started":"2023-02-25T04:47:20.857223Z","shell.execute_reply":"2023-02-25T04:47:21.828147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-info\" role=\"alert\"> <strong>Consideration need to keep in mind</strong>:\n    <ul>\n        <li><strong>First:</strong> if we will balancing the data distrubtion between the labele that will lead to decreas the number of samples at end the mode may couldn't to have enough feature</li>\n        <li><strong>Second:</strong> the Images isn't localize to yet we will need to augment and process data to make it better repesentation to feed to Model</li>\n    </ul>\n","metadata":{}},{"cell_type":"markdown","source":"#### ","metadata":{}},{"cell_type":"code","source":"#### now we will need to sorte the data by cancer Column \ndata_sorted = data_df_col.sort_values(by=\"cancer\")\nbalanced_df = data_sorted[52390:]\n","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:47:23.73727Z","iopub.execute_input":"2023-02-25T04:47:23.737732Z","iopub.status.idle":"2023-02-25T04:47:23.749566Z","shell.execute_reply.started":"2023-02-25T04:47:23.737691Z","shell.execute_reply":"2023-02-25T04:47:23.748333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Temp_after= balanced_df[\"cancer\"].value_counts()\nsave_count_after= pd.DataFrame({ \"Class\":Temp_after.index,\n                              \"Value\": Temp_after.values})\nsave_count_after.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:47:26.664886Z","iopub.execute_input":"2023-02-25T04:47:26.665261Z","iopub.status.idle":"2023-02-25T04:47:26.679126Z","shell.execute_reply.started":"2023-02-25T04:47:26.665228Z","shell.execute_reply":"2023-02-25T04:47:26.678247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = make_subplots(rows=1, cols=2)\nLabel = [\"no cancer\", \"cancer\"]\ncolors = ['lightslategray',] * 2\ncolors[1] = 'crimson'\nfig.add_trace(\n    go.Bar(x=Label, y=save_count_after.Value, \n           marker_color=colors,\n           name=\"After balancing Labels\"),\n    row=1, col=1\n)\n\nfig.add_trace(\n    go.Bar(x=Label, y=save_count_before.Value,\n            marker_color=colors,\n            name=\"Before balancing Labels\"),\n    row=1, col=2\n)\n\nfig.update_layout(height=500, width=700, title_text=\"Cancer distribution balancing \", showlegend=True)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:47:31.063251Z","iopub.execute_input":"2023-02-25T04:47:31.063609Z","iopub.status.idle":"2023-02-25T04:47:31.111078Z","shell.execute_reply.started":"2023-02-25T04:47:31.063576Z","shell.execute_reply":"2023-02-25T04:47:31.110033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Total_samples=len(balanced_df.patient_id)\nprint(f\"the total samples we have now : {Total_samples}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:47:34.842498Z","iopub.execute_input":"2023-02-25T04:47:34.842856Z","iopub.status.idle":"2023-02-25T04:47:34.849432Z","shell.execute_reply.started":"2023-02-25T04:47:34.842826Z","shell.execute_reply":"2023-02-25T04:47:34.848102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## set the seed \nnp.random.seed(2000)\n### here will read few samples from dataset \nPath_train = \"../input/rsna-breast-cancer-detection/train_images/\"\nfig , axis = plt.subplots(2,2,figsize=(5,5))\nrand_index = np.random.randint([900,2316],size=(1,)).item()\nassing_indx = rand_index\nfor row in range(2):\n    for col in range(2):\n        ## get the full path Imae\n        sub_path = str(balanced_df.patient_id.iloc[assing_indx]) + \"/\" + str(balanced_df.image_id.iloc[assing_indx])\n        Label = balanced_df.cancer.iloc[assing_indx]\n        Image_path = os.path.join(Path_train,sub_path) + \".dcm\"\n        Image_array = sitk.ReadImage(Image_path)\n        Image_arry = sitk.GetArrayFromImage(Image_array)\n        axis[row][col].imshow(Image_arry[0],cmap=\"bone\")\n        if Label == 0:\n            axis[row][col].set_title(\"Negative\")\n        else:\n            axis[row][col].set_title(\"Positive\")\n    assing_indx += 1     \n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-03-02T17:15:47.332236Z","iopub.execute_input":"2023-03-02T17:15:47.333328Z","iopub.status.idle":"2023-03-02T17:15:47.465806Z","shell.execute_reply.started":"2023-03-02T17:15:47.333219Z","shell.execute_reply":"2023-03-02T17:15:47.464107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Per-Processing Data**   \n   - **Pipline Steps**\n        1. read images in each case Patient_ID\n        2. Resize the image and Crop the ROI ( region of intersted )\n        3. save the process the image in npy format\n        4. extrat the image label from Train.Csv file\n        5. Virtualize few sample \n        \n   <div class=\"alert alert-warning\" role=\"alert\"> <strong>Hopefull Notebook</strong>:\n    <ul>\n        <li><strong>Resize AND Crop methods </strong> Notebook link: <a  class=\"anchor\" href=\"https://www.kaggle.com/code/snnclsr/roi-extraction-using-opencv\"  id=\"Per-Processing\"> Sinan Calisir </a> Please Upvote</li>\n        <li><strong>DICOM images to PNGs:</strong> Notebook link : <a  class=\"anchor\" href=\"https://www.kaggle.com/code/radek1/how-to-process-dicom-images-to-pngs\"  id=\"Per-Processing\"> Radek Osmulski </a>  Please Upvote </li>\n    </ul>\n     \n","metadata":{}},{"cell_type":"markdown","source":"### **how to work with Dicom file and understand it better**   \n   - **in this section we will try to go through how to handle dicom files**\n     1. get the tags dicomConfiguration and use important tags \n     2. Rescaling algorithm <br>\n        **Note** The following is an algorithm I did google and i grabed some Stackover flow posts, some mentions in the DICOM specs.\n        1. Rescale type attribute\n        2. Window Center\n        3. RescaleIntercept \n        \n<div class=\"alert alert-light\" role=\"alert\"> <strong>dicomConfiguration  :</strong><br>\n<strong>note : Photometric Interpretation tages relate to data we have</strong><br>\n    \nMONOCHROME1 : indicates that the greyscale ranges from bright to dark with ascending pixel values,<br>\nMONOCHROME2 : ranges from dark to bright with ascending pixel values.\nImage compression is independent of the Photometric Interpretation (0028,0004) attribute. Compression is instead given by the Transfer Syntax UID (0002,0010) attribute <br> <br>\n\n<div class=\"alert alert-warning\" role=\"alert\"> <strong>Hopefull Notebook</strong>:\n    <ul>\n        <li><strong>Preprocessing images to normalize </strong> Notebook link: <a  class=\"anchor\" href=\"https://www.kaggle.com/code/donkeys/preprocessing-images-to-normalize-colors-and-sizes\"  id=\"Per-Processing\"> averagemn  </a> Please Upvote</li>\n        <li><strong>Article Meduim :</strong> Article link : <a  class=\"anchor\" href=\"https://towardsdatascience.com/understanding-dicoms-835cd2e57d0b\"  id=\"Per-Processing\"> Radek Osmulski </a>  read and enjoy </li>\n    </ul>    \n    </div>\n\n\n<strong>Here are some key points on the tag information above:</strong> <br>\n\n**Pixel Data (7fe0 0010)(last entry)** : This is where the raw pixel data is stored. The order of pixels encoded for each image plane is left to right, top to bottom, i.e., the upper left pixel (labeled 1,1) is encoded first <br>\n**Photometric Interpretation (0028, 0004)** : aka color space. In this case it is MONOCHROME2 where pixel data is represented as a single monochrome image plane where the minimum sample value is intended to be displayed as black info <br>\n**Samples per Pixel (0028, 0002)** : This should be 1 as this image is monochrome. This value would be 3 if the color space was RGB for example <br>\n**Bits Stored (0028 0101)** : Number of bits stored for each pixel sample <br>\n**Pixel Represenation (0028 0103)** : can either be unsigned(0) or signed(1) <br>\n**Lossy Image Compression (0028 2110)** : 00 image has not been subjected to lossy compression. 01 image has been subjected to lossy compression. <br>\n**Lossy Image Compression Method (0028 2114)** : states the type of lossy compression used (in this case JPEG Lossy Compression, as denotated by CS:ISO_10918_1)   <br>        \n\n    ","metadata":{}},{"cell_type":"code","source":"balanced_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:48:01.766927Z","iopub.execute_input":"2023-02-25T04:48:01.767365Z","iopub.status.idle":"2023-02-25T04:48:01.777599Z","shell.execute_reply.started":"2023-02-25T04:48:01.767329Z","shell.execute_reply":"2023-02-25T04:48:01.776652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_image_config_dicom(id_iloc,path_train):\n    data_id_= balanced_df.iloc[id_iloc]\n    patient_path = str(data_id_.patient_id ) +\"/\"+ str(data_id_.image_id)\n    full_path = os.path.join(path_train,patient_path) + \".dcm\"\n    ds = dc.read_file(full_path)\n    config_list = dir(ds)\n### let us get the dicom Configuration Tags\n    for attr in config_list:\n        if attr.startswith(\"_\"):\n            continue\n        if attr == \"PixelData\" or attr == \"pixel_array\":\n            #skip printing the long arrays as they will just spam the output too much with hex code\n            continue\n        var_type = type(getattr(ds,attr))\n        if var_type == types.MethodType:\n            continue\n    return full_path  , getattr(ds,attr)","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:48:03.791227Z","iopub.execute_input":"2023-02-25T04:48:03.791602Z","iopub.status.idle":"2023-02-25T04:48:03.798841Z","shell.execute_reply.started":"2023-02-25T04:48:03.791569Z","shell.execute_reply":"2023-02-25T04:48:03.797638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_img(img_path, colormap = None, extra_brightness=0):\n    ds = dc.read_file(img_path)\n    shape = ds.pixel_array.shape\n    target = 255\n    # Convert to float to avoid overflow or underflow losses.\n    image_2d = ds.pixel_array.astype(float)\n    img_data = image_2d\n    print(f\"data min: {img_data.min()}, max: {img_data.max()}\")\n    print(f\"window center: {ds.WindowCenter}, rescale intercept: {ds.RescaleIntercept}\")\n    multival = isinstance(ds.WindowCenter, Iterable)\n    if multival:\n        scale_center = -ds.WindowCenter[0]\n    else:\n        scale_center = -ds.WindowCenter\n    intercept = scale_center+ds.RescaleIntercept+extra_brightness\n    print(f\"final intercept: {intercept}\")\n    #image_2d += intercept \n    print(f\"after applying intercept, min: {image_2d.min()}, max: {image_2d.max()}\")\n\n    # Rescaling grey scale between 0-255\n    image_2d_scaled = (np.maximum(image_2d,0) / image_2d.max()) * 255.0\n    print(f\"after scaling to 0-255, min: {image_2d_scaled.min()}, max: {image_2d_scaled.max()}\")\n\n    # Convert to uint\n    image_2d_scaled = np.uint8(image_2d_scaled)\n\n    plt.figure(figsize=(4,4))\n    plt.imshow(image_2d_scaled, cmap=colormap)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:48:10.985422Z","iopub.execute_input":"2023-02-25T04:48:10.985858Z","iopub.status.idle":"2023-02-25T04:48:11.001991Z","shell.execute_reply.started":"2023-02-25T04:48:10.985817Z","shell.execute_reply":"2023-02-25T04:48:11.000923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_path , config_dicom_file = get_image_config_dicom(2315,Path_train)\nprint(config_dicom_file)","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:48:16.843001Z","iopub.execute_input":"2023-02-25T04:48:16.84338Z","iopub.status.idle":"2023-02-25T04:48:17.000183Z","shell.execute_reply.started":"2023-02-25T04:48:16.843349Z","shell.execute_reply":"2023-02-25T04:48:16.999051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import display_html\ndef restartkernel() :\n    display_html(\"<script>Jupyter.notebook.kernel.restart()</script>\",raw=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:03.829677Z","iopub.execute_input":"2023-02-22T19:11:03.830066Z","iopub.status.idle":"2023-02-22T19:11:03.842917Z","shell.execute_reply.started":"2023-02-22T19:11:03.829992Z","shell.execute_reply":"2023-02-22T19:11:03.841992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show_img(full_path, colormap=plt.cm.bone) #image 1","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:48:40.536767Z","iopub.execute_input":"2023-02-25T04:48:40.537162Z","iopub.status.idle":"2023-02-25T04:48:40.541703Z","shell.execute_reply.started":"2023-02-25T04:48:40.537127Z","shell.execute_reply":"2023-02-25T04:48:40.540534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**NOTETTION :** after reseclaing the pixels data between 0-255 that will reduce huge computational cost of finding wieghts Paramaters through BackPropagations <br>\n**Next Step :** use that method of scaling data image to Resampling all data images ","metadata":{"execution":{"iopub.status.busy":"2022-12-03T17:24:55.408253Z","iopub.execute_input":"2022-12-03T17:24:55.408608Z","iopub.status.idle":"2022-12-03T17:24:55.415267Z","shell.execute_reply.started":"2022-12-03T17:24:55.408576Z","shell.execute_reply":"2022-12-03T17:24:55.413924Z"}}},{"cell_type":"code","source":"# https://www.kaggle.com/code/snnclsr/roi-extraction-using-opencv\ndef crop_coords(img):\n    \"\"\"\n    Crop ROI from image.\n    \"\"\"\n    # Otsu's thresholding after Gaussian filtering\n    blur = cv2.GaussianBlur(img, (5, 5), 0)\n    _, breast_mask = cv2.threshold(blur,0,255,cv2.THRESH_BINARY+cv2.THRESH_OTSU)\n    \n    cnts, _ = cv2.findContours(breast_mask.astype(np.uint8), cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n    cnt = max(cnts, key = cv2.contourArea)\n    x, y, w, h = cv2.boundingRect(cnt)\n    return (x, y, w, h)\n\n\ndef normalization_(img):\n    \"\"\"\n    WindowCenter and normalize pixels in the breast ROI.\n    return: numpy array of the normalized image\n    \"\"\"\n    # Convert to float to avoid overflow or underflow losses.\n    image_2d = img.astype(float)\n    image_2d_scaled = (np.maximum(image_2d,0) / image_2d.max()) * 255.0\n    # Convert to uint\n    normalized = np.uint8(image_2d_scaled)\n    return normalized","metadata":{"execution":{"iopub.status.busy":"2023-02-25T04:48:43.985421Z","iopub.execute_input":"2023-02-25T04:48:43.985802Z","iopub.status.idle":"2023-02-25T04:48:43.993765Z","shell.execute_reply.started":"2023-02-25T04:48:43.98577Z","shell.execute_reply":"2023-02-25T04:48:43.992758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Processing_img(Path_dir , label_csv , image_id):\n    data_id = label_csv.iloc[image_id]\n    image = str(data_id.patient_id)+ \"/\" + str(data_id.image_id) \n    Image_path = os.path.join(Path_dir,image) + \".dcm\" \n    read_image = dc.read_file(Image_path)\n    Array_pixel = read_image.pixel_array\n#     if read_image.PhotometricInterpretation == \"MONOCHROME1\":\n#         Array_pixel = np.amax(Array_pixel) - Array_pixel\n    (x,y,w,h) = crop_coords(Array_pixel)\n    image_crop = Array_pixel[y:y+h,x:x+w]\n    normalize_img =normalization_(image_crop)\n    # Resize the image to the final shape. \n    img_final = cv2.resize(normalize_img, (120, 120)).astype(np.float32)\n    return img_final , data_id","metadata":{"execution":{"iopub.status.busy":"2023-02-25T05:02:26.788858Z","iopub.execute_input":"2023-02-25T05:02:26.789263Z","iopub.status.idle":"2023-02-25T05:02:26.797492Z","shell.execute_reply.started":"2023-02-25T05:02:26.789228Z","shell.execute_reply":"2023-02-25T05:02:26.796334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we will need to randome selecte the number rows for not keep it sorted \nrandom_label_df = balanced_df.sample(n=2316,random_state =70,replace=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:03.8781Z","iopub.execute_input":"2023-02-22T19:11:03.879157Z","iopub.status.idle":"2023-02-22T19:11:03.888562Z","shell.execute_reply.started":"2023-02-22T19:11:03.87912Z","shell.execute_reply":"2023-02-22T19:11:03.887582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## set the clean dataFrame balanced_df\nsums = 0 \nsums_square = 0 ### to computer Normalize means and STD to use later \nNormalizer = 120*120\nProcessed_data = \"./Processed\"\nfor idx_img , Patient_ID in enumerate(tqdm(balanced_df.image_id)):\n    ### Extract label Patient \n    image_processed , label_id = Processing_img(Path_train,balanced_df,image_id=idx_img)\n    Image_id = str(label_id.image_id)\n    label = label_id.cancer\n    Train_or_Val = \"train\" if idx_img < 1800 else \"val\" ## total samples are 2316 we split it by \n    ## creat save folder \n    Save_path = Path(Processed_data + \"/\"+ Train_or_Val +\"/\"+ str(label))\n    Save_path.mkdir(parents=True,exist_ok=True)\n    #         ### the struct dircotry is \n#         ## Train/\n#         #        lable_0/\n#         #              Image_ID\n#         #        labek_1/\n#         #              Image_ID \n#         #####\n    np.save(os.path.join(Save_path,Image_id),image_processed)\n    if Train_or_Val == \"train\":\n        sums += np.sum(image_processed) / Normalizer \n        sums_square += (image_processed ** 2).sum() / Normalizer ","metadata":{"execution":{"iopub.status.busy":"2023-03-02T17:26:44.969619Z","iopub.execute_input":"2023-03-02T17:26:44.970063Z","iopub.status.idle":"2023-03-02T17:26:44.995599Z","shell.execute_reply.started":"2023-03-02T17:26:44.970028Z","shell.execute_reply":"2023-03-02T17:26:44.993392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pylibjpeg import decode","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:15:27.999499Z","iopub.execute_input":"2023-02-22T19:15:27.999913Z","iopub.status.idle":"2023-02-22T19:15:28.025657Z","shell.execute_reply.started":"2023-02-22T19:15:27.999856Z","shell.execute_reply":"2023-02-22T19:15:28.024173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean = sums / 2316\nstd = np.sqrt(sums_square / 2316 - mean**2)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.342934Z","iopub.status.idle":"2023-02-22T19:11:04.34344Z","shell.execute_reply.started":"2023-02-22T19:11:04.343183Z","shell.execute_reply":"2023-02-22T19:11:04.343207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean , std\n### we can see here the STD and Means not range between [0-1] \n### because of the pixle sixe is between 0-255","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.344735Z","iopub.status.idle":"2023-02-22T19:11:04.34598Z","shell.execute_reply.started":"2023-02-22T19:11:04.345641Z","shell.execute_reply":"2023-02-22T19:11:04.34568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Traning Step using Pytorch Lighting** <a class=\"anchor\" id=\"Section-2\"></a>\n* loading the data has been Processed \n  - function to load and save ans float \n     - augementation \n     - Dataloader \n     - model CNN from Renstnet","metadata":{}},{"cell_type":"code","source":"### Loading the numpy format as .npy\ndef Loader_Data(path):\n    return np.load(path).astype(np.float32)\n### Augmenetation pipline from transform Objet trochvision \n\nTransform_aug_Train = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.RandomAffine(degrees=(-5, 5), translate=(0, 0.05), scale=(0.9, 1.1)),\n    ])\nTransform_aug_Val = transforms.Compose([\n    transforms.ToTensor(),\n    ])\nLoading_DataFolder_Train = torchvision.datasets.DatasetFolder(root=\"./Processed/train\",\n                                                              loader=Loader_Data , \n                                                              extensions=\"npy\",\n                                                              transform=Transform_aug_Train)\nLoading_DataFolder_Val= torchvision.datasets.DatasetFolder(root=\"./Processed/val\",\n                                                           loader=Loader_Data , \n                                                           extensions=\"npy\",\n                                                           transform=Transform_aug_Val)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.347629Z","iopub.status.idle":"2023-02-22T19:11:04.34811Z","shell.execute_reply.started":"2023-02-22T19:11:04.34786Z","shell.execute_reply":"2023-02-22T19:11:04.347882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### Virtualize few samples from Data after augmenration .\nfig , axes = plt.subplots(2,2,figsize=(6,6))\ncounter_Img = 0\nfor i in range(2):\n    for j in range(2):\n        rand_indx = np.random.randint(0,500,size=(1,)).item() ### this will return value betweem [0,24000]\n        Image_ray , label = Loading_DataFolder_Train[rand_indx] # will retrun tuple correspond image X_ray woht label \n        if label == 0 :\n            class_ = \"Negative\"\n        else :\n            class_ = \"Positive\"\n        axes[i][j].imshow(Image_ray[0],cmap=\"gray\")\n        axes[i][j].set_title(f\"label class is : {class_}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.349719Z","iopub.status.idle":"2023-02-22T19:11:04.350324Z","shell.execute_reply.started":"2023-02-22T19:11:04.350064Z","shell.execute_reply":"2023-02-22T19:11:04.35009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### check the data if is imblance or not by seeing the distrubtion \nnp.unique(Loading_DataFolder_Train.targets, return_counts=True)\n### here we can see that the data is imblanc but we can go through this even do \n","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.35195Z","iopub.status.idle":"2023-02-22T19:11:04.352445Z","shell.execute_reply.started":"2023-02-22T19:11:04.352171Z","shell.execute_reply":"2023-02-22T19:11:04.352208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 4\nTraining_Loader = torch.utils.data.DataLoader(Loading_DataFolder_Train,batch_size=batch_size, shuffle=True)\nValidation_Loader = torch.utils.data.DataLoader(Loading_DataFolder_Val,batch_size=batch_size, shuffle=False)\nprint(f\"lenght of training data is :{len(Training_Loader)}\\nlenght of validationi s: {len(Validation_Loader)}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.354044Z","iopub.status.idle":"2023-02-22T19:11:04.354514Z","shell.execute_reply.started":"2023-02-22T19:11:04.35428Z","shell.execute_reply":"2023-02-22T19:11:04.354303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Pipline model** Building model <a class=\"anchor\" id=\"Section3\"></a>\n- by calling this API torchvision \n  * torchvision.models.resnet152()\n    * change the number of channel Conv1 from 3 to 1 \n       * **(conv1): Conv2d(3, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)**\n    * change the output of last layer fully connected layer to 1 \n       * **(fc): Linear(in_features=2048, out_features=1000, bias=True)**\n- **Notation** :  <div class=\"alert alert-info\">since the competition is started few weeks ago here i just tried give a briefy implementation of<strong> Pytorch **Lighiting**</strong></div> <br>\n    - **that's why we will just run the Model on 10 Epochs and you can raise it to wished value if you would like , i will improve this notebook in the next few days**\n\n","metadata":{}},{"cell_type":"code","source":"###### Creating the model \nclass CancerModel(pl.LightningModule):\n    def __init__(self,weight=1):\n        super().__init__()\n        ### the PL of model \n        self.model = torchvision.models.resnet152()\n        self.model.conv1 = torch.nn.Conv2d(1, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)\n        self.model.fc = torch.nn.Linear(in_features=2048, out_features=1)\n        ### setup the Optimizer and loss functions \n        self.Optimizer = torch.optim.Adam(self.model.parameters(),lr = 1e-4)\n        self.Loss_F = torch.nn.BCEWithLogitsLoss(pos_weight=torch.Tensor([weight]))\n         # simple accuracy computation\n        self.train_acc = torchmetrics.Accuracy()\n        self.val_acc = torchmetrics.Accuracy()\n        \n    def forward(self, input_):\n        Predicted_label = self.model(input_)\n        return Predicted_label\n        \n    def training_step(self,batch , batch_idx):\n        Image , label = batch \n        label= label.float()\n        Predicted_label = self(Image)[:,0]\n        loss = self.Loss_F(Predicted_label,label)\n        # Log loss and batch accuracy\n        self.log(\"Train Loss\", loss,sync_dist=True)\n        self.log(\"Step Train Acc\", self.train_acc(torch.sigmoid(Predicted_label), label.int()))\n        return loss\n    \n    def training_epoch_end(self, outs):\n        # After one epoch compute the whole train_data accuracy\n        self.log(\"Train Acc\", self.train_acc.compute(),sync_dist=True)\n        \n    #######\n    ### here we did the same as Traiing PL we changed only the input_data disttro\n    #######\n    def validation_step(self,batch , batch_idx):\n        Image , label = batch \n        label = label.float()\n        Predicted_label = self(Image)[:,0]\n        loss = self.Loss_F(Predicted_label,label)\n        # Log loss and batch accuracy\n        self.log(\"Val Loss\", loss,sync_dist=True)\n        self.log(\"Step Val Acc\", self.train_acc(torch.sigmoid(Predicted_label), label.int()))\n        return loss\n    \n    def validation_epoch_end(self, outs):\n        # After one epoch compute the whole train_data accuracy\n        self.log(\"Val Acc\", self.train_acc.compute(),sync_dist=True)  \n        \n    def configure_optimizers(self):\n        #Caution! You always need to return a list here (just pack your optimizer into one :))\n        return [self.Optimizer]     \n        ","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.356453Z","iopub.status.idle":"2023-02-22T19:11:04.356935Z","shell.execute_reply.started":"2023-02-22T19:11:04.356677Z","shell.execute_reply":"2023-02-22T19:11:04.356701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Model_ = CancerModel() ### instanace the model from the class \nCheck_Point_Callbacks = ModelCheckpoint(\n    monitor=\"Val Acc\", \n    save_top_k=12,\n    mode=\"max\")","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.358593Z","iopub.status.idle":"2023-02-22T19:11:04.359116Z","shell.execute_reply.started":"2023-02-22T19:11:04.358836Z","shell.execute_reply":"2023-02-22T19:11:04.358861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the trainer\n# Change the gpus parameter to the number of available gpus on your system. Use 0 for CPU training\n\ngpus = 2 #TODO\nTrainer = pl.Trainer(accelerator='gpu', devices=2, logger=TensorBoardLogger(save_dir= \"./processed/logs_weights_ex\"), log_every_n_steps=1,\n                     callbacks=Check_Point_Callbacks,                    \n                     max_epochs=50)","metadata":{"execution":{"iopub.status.busy":"2023-03-02T18:48:46.179144Z","iopub.execute_input":"2023-03-02T18:48:46.179607Z","iopub.status.idle":"2023-03-02T18:48:46.595459Z","shell.execute_reply.started":"2023-03-02T18:48:46.179568Z","shell.execute_reply":"2023-03-02T18:48:46.594202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Trainer.fit(Model_,Training_Loader,Validation_Loader)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.363059Z","iopub.status.idle":"2023-02-22T19:11:04.363524Z","shell.execute_reply.started":"2023-02-22T19:11:04.363289Z","shell.execute_reply":"2023-02-22T19:11:04.363311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Evaluation Model** <a class=\"anchor\" id=\"Secteion-Eval\"> </a>\n* here we will load the weights the model \n* preduct few samples from the validation dataset \n* save the reults in tensor data type \n* compute few matrices results ","metadata":{}},{"cell_type":"code","source":"## Loading the model check point weight \ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\npath_model = \"./processed/logs_weights_ex/lightning_logs/version_0/checkpoints/\"\nlast_epoch = glob(path_model+'*.ckpt')[-1]\nmodel_load = CancerModel.load_from_checkpoint(last_epoch)\nmodel_load.to(device)\nmodel_load.eval();","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.365154Z","iopub.status.idle":"2023-02-22T19:11:04.365706Z","shell.execute_reply.started":"2023-02-22T19:11:04.365425Z","shell.execute_reply":"2023-02-22T19:11:04.365452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_ = []\nlabels_ = []\nwith torch.no_grad():\n    for image , label in tqdm(Loading_DataFolder_Val):\n        ## here we will need to load the data into device by hand \n        image = image.to(device).float().unsqueeze(0)\n        label = label\n        predicted= torch.sigmoid(model_load(image)[0]).cpu()\n        predicted_.append(predicted)\n        labels_.append(label)\n    print(image.shape) \n    print(predicted.shape)\n            \nTensor_Pre = torch.tensor(predicted_)        \nTensor_Lab = torch.tensor(labels_).int() ","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.367923Z","iopub.status.idle":"2023-02-22T19:11:04.368506Z","shell.execute_reply.started":"2023-02-22T19:11:04.368257Z","shell.execute_reply":"2023-02-22T19:11:04.368281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Tensor_Lab.shape , Tensor_Pre.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.369801Z","iopub.status.idle":"2023-02-22T19:11:04.370612Z","shell.execute_reply.started":"2023-02-22T19:11:04.370359Z","shell.execute_reply":"2023-02-22T19:11:04.370389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission Files <a class=\"anchor\" id=\"Submission\"></a>\n\n* **here will predict the sampels from vaidations data and save the Patient_ID and predicte value**\n\n","metadata":{}},{"cell_type":"code","source":"acc = torchmetrics.Accuracy()(Tensor_Pre, Tensor_Lab)\nprecision = torchmetrics.Precision()(Tensor_Pre, Tensor_Lab)\nrecall = torchmetrics.Recall()(Tensor_Pre, Tensor_Lab)\ncm = torchmetrics.ConfusionMatrix(num_classes=2)(Tensor_Pre, Tensor_Lab)\ncm_threshed = torchmetrics.ConfusionMatrix(num_classes=2, threshold=0.25)(Tensor_Pre, Tensor_Lab)\n\nprint(f\"Val Accuracy: {acc}\")\nprint(f\"Val Precision: {precision}\")\nprint(f\"Val Recall: {recall}\")\nprint(f\"Confusion Matrix:\\n {cm}\")\nprint(f\"Confusion Matrix 2:\\n {cm_threshed}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.377531Z","iopub.status.idle":"2023-02-22T19:11:04.378248Z","shell.execute_reply.started":"2023-02-22T19:11:04.377986Z","shell.execute_reply":"2023-02-22T19:11:04.37801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#### Virtualiza some images \nDATA_DIR = \"../input/rsna-breast-cancer-detection/\"\n# Directory to save logs and trained model\nROOT_DIR = '/kaggle/working'\n\ntest_dicom_dir = os.path.join(DATA_DIR, 'test_images/10008')\nprint(test_dicom_dir)\nsub_read = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv\")\nsub_read.head(3)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.379512Z","iopub.status.idle":"2023-02-22T19:11:04.380214Z","shell.execute_reply.started":"2023-02-22T19:11:04.379962Z","shell.execute_reply":"2023-02-22T19:11:04.379986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\ndef get_dicom_fps(dicom_dir):\n    dicom_fps = glob.glob(dicom_dir+'/*.dcm')\n    return list(set(dicom_fps))","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.381515Z","iopub.status.idle":"2023-02-22T19:11:04.382251Z","shell.execute_reply.started":"2023-02-22T19:11:04.381963Z","shell.execute_reply":"2023-02-22T19:11:04.381987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get filenames of test dataset DICOM images\ntest_image_fps = get_dicom_fps(test_dicom_dir)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.383639Z","iopub.status.idle":"2023-02-22T19:11:04.384351Z","shell.execute_reply.started":"2023-02-22T19:11:04.384087Z","shell.execute_reply":"2023-02-22T19:11:04.384111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions on test images, write out sample submission\npatient_ID = []\nScore_ID= []\n#my_formatter = \"{0:.2f}\"\ndef predict(image_fps, min_conf=0.90):\n        with torch.no_grad():\n            for image_id in tqdm(image_fps):\n                ds = dc.read_file(image_id)\n                image = ds.pixel_array  / 255 ##normalize \n                image_2d = image.astype(float)\n                image = cv2.resize(image , (120,120)).astype(np.float32)\n                image = np.expand_dims(image,axis=0)\n                patient_id = os.path.splitext(os.path.basename(image_id))[0]\n                patient_ID.append(patient_id)\n                image= torch.from_numpy(image)\n                image = image.to(device).float().unsqueeze(0)\n                results = torch.sigmoid(model_load(image)[0]).cpu()\n                score =  results[0].item()\n                Score = str(score) if float(score) > 0.60 else str(score) \n                Score_ID.append(Score)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.385734Z","iopub.status.idle":"2023-02-22T19:11:04.386532Z","shell.execute_reply.started":"2023-02-22T19:11:04.386269Z","shell.execute_reply":"2023-02-22T19:11:04.386297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict(test_image_fps)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.387819Z","iopub.status.idle":"2023-02-22T19:11:04.388558Z","shell.execute_reply.started":"2023-02-22T19:11:04.388282Z","shell.execute_reply":"2023-02-22T19:11:04.388323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### Virtualiza some images \nDATA_DIR = \"../input/rsna-breast-cancer-detection/\"\n# Directory to save logs and trained model\nROOT_DIR = '/kaggle/working'\n\ntest_dicom_dir = os.path.join(DATA_DIR, 'test_images/10008')\nprint(test_dicom_dir)\nsub_read = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv\")\nsub_read.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.389805Z","iopub.status.idle":"2023-02-22T19:11:04.390521Z","shell.execute_reply.started":"2023-02-22T19:11:04.39027Z","shell.execute_reply":"2023-02-22T19:11:04.390295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DCM_TEST_IMAGES_PATH = '/kaggle/input/rsna-breast-cancer-detection/test_images/10008'\nRSNA_2022_PATH = '/kaggle/input/rsna-breast-cancer-detection'","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.39177Z","iopub.status.idle":"2023-02-22T19:11:04.392528Z","shell.execute_reply.started":"2023-02-22T19:11:04.392249Z","shell.execute_reply":"2023-02-22T19:11:04.392292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndf_test = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\ndf_test[\"img_name\"] = df_test[\"patient_id\"].astype(str) + \"_\" + df_test[\"image_id\"].astype(str) + \".png\"\ndf_test[\"dcm_path\"] = df_test[\"patient_id\"].astype(str) + \"_\" + df_test[\"image_id\"].astype(str) + \".dcm\"\ndf_test.head()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.39386Z","iopub.status.idle":"2023-02-22T19:11:04.394622Z","shell.execute_reply.started":"2023-02-22T19:11:04.394356Z","shell.execute_reply":"2023-02-22T19:11:04.394381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_id = df_test['patient_id'].astype(str) + \"_\" + df_test['laterality']","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.396075Z","iopub.status.idle":"2023-02-22T19:11:04.396873Z","shell.execute_reply.started":"2023-02-22T19:11:04.396599Z","shell.execute_reply":"2023-02-22T19:11:04.396625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = {\n    \"prediction_id\": np.array((prediction_id)),\n    \"cancer\": Score_ID\n}","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.398192Z","iopub.status.idle":"2023-02-22T19:11:04.398931Z","shell.execute_reply.started":"2023-02-22T19:11:04.398674Z","shell.execute_reply":"2023-02-22T19:11:04.398698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.DataFrame(data=data)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.400187Z","iopub.status.idle":"2023-02-22T19:11:04.400905Z","shell.execute_reply.started":"2023-02-22T19:11:04.400637Z","shell.execute_reply":"2023-02-22T19:11:04.400662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subb = sub_df.drop_duplicates(\"prediction_id\")\nsubb\n","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.402198Z","iopub.status.idle":"2023-02-22T19:11:04.402963Z","shell.execute_reply.started":"2023-02-22T19:11:04.402674Z","shell.execute_reply":"2023-02-22T19:11:04.402701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subb.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.404262Z","iopub.status.idle":"2023-02-22T19:11:04.404997Z","shell.execute_reply.started":"2023-02-22T19:11:04.404727Z","shell.execute_reply":"2023-02-22T19:11:04.404752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Activation Map** <a class=\"anchor\" id=\"Section-Map\"></a>\n1. **this section containe activation map extrat feature from Layer Convolution**\n     - this technic is based on save the weights of layer in list Sequencetial model </br>\n       and calculate the weights on Validation test \n     - i did virtualization plot to check heatMap ","metadata":{}},{"cell_type":"code","source":"#### Loader data function \ndef loader(path):\n    return np.load(path).astype(np.float32)\n#### set the transfoms augmentation \ntransform = transforms.Compose([\n    transforms.ToTensor(),\n    ])\nvalidation_set = Path(\"./Processed/val/\")\n\nload_data = torchvision.datasets.DatasetFolder(root=validation_set , loader=loader,extensions=\"npy\",transform=transform)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.406492Z","iopub.status.idle":"2023-02-22T19:11:04.407262Z","shell.execute_reply.started":"2023-02-22T19:11:04.406985Z","shell.execute_reply":"2023-02-22T19:11:04.407011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### building the model \n\nclass CancerModel_val(pl.LightningModule):\n    def __init__(self):\n        super().__init__()\n        self.model = torchvision.models.resnet152()\n        self.model.conv1 = torch.nn.Conv2d(1, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)\n        self.model.fc = torch.nn.Linear(in_features=2048, out_features=1)\n        \n        ### get the features map from the model 2\n        self.feature_map = torch.nn.Sequential(*list(self.model.children())[:-2])\n        \n    def forward(self, input_):\n        feature_map = self.feature_map(input_)\n        pooling_avg = torch.nn.functional.adaptive_avg_pool2d(feature_map , output_size=(1,1))\n        flatten_pool = torch.flatten(pooling_avg)\n        Predict_= self.model.fc(flatten_pool)\n        return feature_map ,Predict_ \n        ","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.408633Z","iopub.status.idle":"2023-02-22T19:11:04.409419Z","shell.execute_reply.started":"2023-02-22T19:11:04.409126Z","shell.execute_reply":"2023-02-22T19:11:04.409153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### intialze the model \nModel_Cancer = CancerModel_val()\nModel_Cancer.load_from_checkpoint(last_epoch,strict=False)\nModel_Cancer.eval();","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.410679Z","iopub.status.idle":"2023-02-22T19:11:04.411403Z","shell.execute_reply.started":"2023-02-22T19:11:04.411141Z","shell.execute_reply":"2023-02-22T19:11:04.411165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### compute the Grad-Cam function \n\ndef Cam(image,model):\n    with torch.no_grad():\n        image = image[0]\n        feature_map , pred = Model_Cancer(image.unsqueeze(0))\n    print(feature_map.shape)\n    feature = feature_map.reshape((2048,4*4))\n    weights = list(model.model.fc.parameters())[0]\n    weights_param = weights[0].detach()\n    cam = torch.matmul(weights_param,feature) \n    print(cam.shape)\n    cam_img = cam.reshape(4,4).cpu()\n    return cam_img , torch.sigmoid(pred)\n        ","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.412737Z","iopub.status.idle":"2023-02-22T19:11:04.413502Z","shell.execute_reply.started":"2023-02-22T19:11:04.41322Z","shell.execute_reply":"2023-02-22T19:11:04.413256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def virtulizer(img,cam ,pred):\n    img = img[0]\n    cam_img =transforms.functional.resize(cam.unsqueeze(0),(120,120))[0]\n    fig ,axis = plt.subplots(1,2 , figsize=(4,4))\n    axis[0].imshow(img,cmap=\"bone\")\n    axis[1].imshow(img,cmap=\"bone\")\n    axis[1].imshow(cam_img,cmap=\"jet\")\n    if pred >0.5 :\n        plt.title(\"positive\")\n    else:\n        plt.title(\"negative\")\n        ","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.414882Z","iopub.status.idle":"2023-02-22T19:11:04.415791Z","shell.execute_reply.started":"2023-02-22T19:11:04.415504Z","shell.execute_reply":"2023-02-22T19:11:04.415532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### testing \n# 1. compute \nimg = load_data[-5][0]\nmap_activation , Predicted = Cam(img.unsqueeze(0),Model_Cancer)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.417169Z","iopub.status.idle":"2023-02-22T19:11:04.41787Z","shell.execute_reply.started":"2023-02-22T19:11:04.417618Z","shell.execute_reply":"2023-02-22T19:11:04.417642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"virtulizer(img,map_activation,Predicted)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T19:11:04.419185Z","iopub.status.idle":"2023-02-22T19:11:04.419902Z","shell.execute_reply.started":"2023-02-22T19:11:04.419634Z","shell.execute_reply":"2023-02-22T19:11:04.419659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-info\" role=\"alert\"> <strong>Next Notebook will be about TransMIL: Transformer based Correlated Multiple Instance Learning for Whole Slide\nImage Classification</strong>\n    </div>\n","metadata":{}},{"cell_type":"markdown","source":"<center>\n<img src=\"https://img.shields.io/badge/Upvote-If%20you%20like%20my%20work-07b3c8?style=for-the-badge&logo=kaggle\">\n</center>","metadata":{}}]}