{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Breast Cancer Detection\n\n### Project Lifecycle:\n\n#### 1) Environment Setup & Data Loading\n#### 2) Data Exploration\n#### 3) Creating a custom dataset\n#### 4) Building the classification model\n#### 5) Defining the loss function\n#### 6) Defining the optimizer\n#### 7) Training and evaluation of the model\n#### 8) Deploying the model\n#### 9) Model inference on test data\n\n","metadata":{}},{"cell_type":"code","source":"!python -m venv /kaggle/working/breast_cancer_detection\n","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:20:43.519992Z","iopub.execute_input":"2023-06-21T09:20:43.520344Z","iopub.status.idle":"2023-06-21T09:20:47.323354Z","shell.execute_reply.started":"2023-06-21T09:20:43.520316Z","shell.execute_reply":"2023-06-21T09:20:47.322232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!source /kaggle/working/breast_cancer_detection/bin/activate\n","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:20:51.169501Z","iopub.execute_input":"2023-06-21T09:20:51.170053Z","iopub.status.idle":"2023-06-21T09:20:51.427613Z","shell.execute_reply.started":"2023-06-21T09:20:51.170025Z","shell.execute_reply":"2023-06-21T09:20:51.426345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pydicom \n!pip install python-gdcm\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:20:55.348003Z","iopub.execute_input":"2023-06-21T09:20:55.348586Z","iopub.status.idle":"2023-06-21T09:21:14.707515Z","shell.execute_reply.started":"2023-06-21T09:20:55.348556Z","shell.execute_reply":"2023-06-21T09:21:14.706285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U pylibjpeg\n!pip install pillow\n!pip install jpeg_ls\n!pip install pylibjpeg pylibjpeg-libjpeg pylibjpeg-openjpeg pylibjpeg-rle\n","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:21:20.239624Z","iopub.execute_input":"2023-06-21T09:21:20.239951Z","iopub.status.idle":"2023-06-21T09:21:48.470628Z","shell.execute_reply.started":"2023-06-21T09:21:20.239926Z","shell.execute_reply":"2023-06-21T09:21:48.469592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom matplotlib import pyplot as plt\nimport pandas as pd\nimport pylibjpeg\nfrom PIL import Image\nimport numpy as np\nimport gdcm\nimport pydicom\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:22:05.296189Z","iopub.execute_input":"2023-06-21T09:22:05.296565Z","iopub.status.idle":"2023-06-21T09:22:05.479437Z","shell.execute_reply.started":"2023-06-21T09:22:05.296529Z","shell.execute_reply":"2023-06-21T09:22:05.478501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport glob\nfrom joblib import Parallel, delayed\nimport time","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:22:13.227335Z","iopub.execute_input":"2023-06-21T09:22:13.227705Z","iopub.status.idle":"2023-06-21T09:22:13.416715Z","shell.execute_reply.started":"2023-06-21T09:22:13.227678Z","shell.execute_reply":"2023-06-21T09:22:13.415679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Read dicom file","metadata":{}},{"cell_type":"code","source":"\n\n\n\n#Before run this code add rsna-breast-cancer dataset\ndata_set_address = '/kaggle/input/rsna-breast-cancer-detection/'\ntrain_images = sorted(glob.glob(f'{data_set_address}train_images/*/*.dcm'))\n#use glob to work easier on files in directories\nnumber_of_train_images = len(train_images)\nprint('Number of train_images is:',number_of_train_images) \n\n# Define a function to read dicom file and return its pixel\ndef read_dicom(dicom_file):\n    gdcm_image = gdcm.ImageReader()\n    gdcm_image.SetFileName(dicom_file)\n    gdcm_image.Read()\n    \n    dcm_data = gdcm_image.GetImage()\n    pydicom_data = pydicom.dcmread(dicom_file)\n    \n    dicom_pixel = pydicom_data.pixel_array\n    dicom_pixel = (dicom_pixel - dicom_pixel.min()) / (dicom_pixel.max() - dicom_pixel.min())\n    \n    if pydicom_data.PhotometricInterpretation == \"MONOCHROME1\":\n        dicom_pixel = 1 - dicom_pixel\n    \n    return dicom_pixel\n","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:22:16.453539Z","iopub.execute_input":"2023-06-21T09:22:16.453866Z","iopub.status.idle":"2023-06-21T09:24:05.117635Z","shell.execute_reply.started":"2023-06-21T09:22:16.453842Z","shell.execute_reply":"2023-06-21T09:24:05.116754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files=train_images[:2]\nfor i in files:\n    dicom_pixel= read_dicom(dicom_file=i)\ndicom_pixel","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:24:05.119477Z","iopub.execute_input":"2023-06-21T09:24:05.119815Z","iopub.status.idle":"2023-06-21T09:24:07.261309Z","shell.execute_reply.started":"2023-06-21T09:24:05.119784Z","shell.execute_reply":"2023-06-21T09:24:07.260106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Crop dicom file","metadata":{}},{"cell_type":"code","source":"def crop_dicom(dicom_file):\n    \n    result= cv2.connectedComponentsWithStats((dicom_file > 0.005).astype(np.uint8)[:, :], 8, cv2.CV_32S)\n    # result has 4 diffrenet outputs. You can see different outputs here.\n    # But for reconizing the black color(0) from others(1), we just need result[2] , called stat matrix\n    stat_matrix = result[2] # include left, top, width, height, area_size columns\n    # the second row shows the boxes of pixels which are different from background.Also, we don't need are_size cloumn. So:\n    second_row = stat_matrix[1:, 4].argmax() + 1\n    x1, y1, w, h = stat_matrix[second_row][:4]\n    x2 = x1 + w\n    y2 = y1 + h\n    cropped_dicom = dicom_file[y1: y2, x1: x2]\n    return cropped_dicom","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:24:25.39271Z","iopub.execute_input":"2023-06-21T09:24:25.393092Z","iopub.status.idle":"2023-06-21T09:24:25.399367Z","shell.execute_reply.started":"2023-06-21T09:24:25.393063Z","shell.execute_reply":"2023-06-21T09:24:25.398447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### original and cropped dicom file\n\n","metadata":{}},{"cell_type":"code","source":"files = train_images[:2]\nfor file in files:\n    patient_id = file.split('/')[-2] # split the patient_id from the whole address\n    print(patient_id)\n    image_id = file.split('/')[-1][:-4] # split the image_id from the whole address\n # Plot Orginial iamge\n    plt.figure(figsize=(8,8))\n    dicom_pixel = read_dicom(dicom_file=file)\n    plt.imshow(dicom_pixel,cmap='gray')\n    plt.title(f\"Original image:{patient_id}_{image_id}\",fontsize=16)\n    plt.show()\n    \n    plt.figure(figsize=(8,8))\n    cropped_iamge = crop_dicom(dicom_file=dicom_pixel)\n    plt.imshow(cropped_iamge,cmap='gray')\n    plt.title(f\"Cropped image:{patient_id}_{image_id}\",fontsize=16)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:24:32.222104Z","iopub.execute_input":"2023-06-21T09:24:32.222498Z","iopub.status.idle":"2023-06-21T09:24:37.816643Z","shell.execute_reply.started":"2023-06-21T09:24:32.222473Z","shell.execute_reply":"2023-06-21T09:24:37.815463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_df_dir = \"/kaggle/input/rsna-breast-cancer-detection/train.csv\"\ntrain_df = pd.read_csv(train_df_dir)\ntrain_df[['patient_id','image_id','site_id','machine_id']] = train_df[['patient_id','image_id','site_id','machine_id']].astype(str)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:24:49.685779Z","iopub.execute_input":"2023-06-21T09:24:49.686105Z","iopub.status.idle":"2023-06-21T09:24:49.87644Z","shell.execute_reply.started":"2023-06-21T09:24:49.68608Z","shell.execute_reply":"2023-06-21T09:24:49.875606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:24:54.033629Z","iopub.execute_input":"2023-06-21T09:24:54.033961Z","iopub.status.idle":"2023-06-21T09:24:54.106942Z","shell.execute_reply.started":"2023-06-21T09:24:54.03393Z","shell.execute_reply":"2023-06-21T09:24:54.106034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nPNG_images = \"/kaggle/working/PNG_train_images/\"\nos.makedirs(PNG_images,exist_ok = True)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:24:57.586095Z","iopub.execute_input":"2023-06-21T09:24:57.586528Z","iopub.status.idle":"2023-06-21T09:24:57.592245Z","shell.execute_reply.started":"2023-06-21T09:24:57.586499Z","shell.execute_reply":"2023-06-21T09:24:57.591086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:25:02.472772Z","iopub.execute_input":"2023-06-21T09:25:02.473184Z","iopub.status.idle":"2023-06-21T09:25:02.560651Z","shell.execute_reply.started":"2023-06-21T09:25:02.473155Z","shell.execute_reply":"2023-06-21T09:25:02.559555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef dcm_to_png(file,size= 512,save_folder= PNG_images, extension = 'png'):\n    patient_id = file.split('/')[-2]\n    image_id = file.split('/')[-1][:-4]\n    dicom_pixel = read_dicom(dicom_file=file) # read dicom image pixels\n    cropped_dicom = crop_dicom(dicom_file=dicom_pixel) # crop the dicom image\n    resized_img = cv2.resize(cropped_dicom, dsize=(size, size)) # resize it to specific size\n    cv2.imwrite(save_folder + f\"{patient_id}_{image_id}.{extension}\", (resized_img*255).astype(np.uint8))\n        \nif __name__ == '__main__':\n    start = time.time()\n    _ = Parallel(n_jobs=4, backend=\"multiprocessing\")(\n        delayed(dcm_to_png)(file=file, size=256)\n        for file in tqdm(train_images)  # you can use train_images[:100] for testing\n    )\n    stop = time.time()\n    print('This process took:', stop - start)\n    print('Now, you can access the cropped files in PNG format.')\n\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:25:05.884398Z","iopub.execute_input":"2023-06-21T09:25:05.884744Z","iopub.status.idle":"2023-06-21T10:04:02.152662Z","shell.execute_reply.started":"2023-06-21T09:25:05.884712Z","shell.execute_reply":"2023-06-21T10:04:02.148186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}