{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# https://www.kaggle.com/code/deannahedges/mammography-challenge-dicom-to-png\n# results: https://www.kaggle.com/datasets/deannahedges/mammography-challenge-pngs\n\n# Sources:\n    # To go from Dicom -> PNG:\n        # https://www.kaggle.com/code/radek1/how-to-process-dicom-images-to-pngs/notebook?scriptVersionId=113529850\n    # To load the data, configure for performance, and build model in keras:\n        # https://www.tensorflow.org/tutorials/load_data/images#:~:text=This%20tutorial%20shows%20how%20to%20load%20and%20preprocess,from%20the%20large%20catalog%20available%20in%20TensorFlow%20Datasets.\n    # To augment the data:\n        # https://www.tensorflow.org/tutorials/images/data_augmentation\n    # To make the submission notebook:\n        # https://www.kaggle.com/code/radek1/fast-ai-starter-pack-train-inference/notebook\n        \n\nimport numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-04-09T13:17:18.389852Z","iopub.execute_input":"2023-04-09T13:17:18.390493Z","iopub.status.idle":"2023-04-09T13:17:18.397525Z","shell.execute_reply.started":"2023-04-09T13:17:18.390429Z","shell.execute_reply":"2023-04-09T13:17:18.396341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploring training csv with labels","metadata":{}},{"cell_type":"code","source":"train_file = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ntrain_file.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T13:17:18.400117Z","iopub.execute_input":"2023-04-09T13:17:18.40099Z","iopub.status.idle":"2023-04-09T13:17:18.480525Z","shell.execute_reply.started":"2023-04-09T13:17:18.400755Z","shell.execute_reply":"2023-04-09T13:17:18.479423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_file['cancer'] = train_file['cancer'].astype('float32')\ntrain_file.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T13:17:18.482322Z","iopub.execute_input":"2023-04-09T13:17:18.482701Z","iopub.status.idle":"2023-04-09T13:17:18.506294Z","shell.execute_reply.started":"2023-04-09T13:17:18.482663Z","shell.execute_reply":"2023-04-09T13:17:18.505109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_file['cancer'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T13:17:18.508802Z","iopub.execute_input":"2023-04-09T13:17:18.509243Z","iopub.status.idle":"2023-04-09T13:17:18.518989Z","shell.execute_reply.started":"2023-04-09T13:17:18.509204Z","shell.execute_reply":"2023-04-09T13:17:18.517804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating function to categorize images as \"positive\" or \"negative\" based on file path","metadata":{}},{"cell_type":"code","source":"def pos_or_neg(img_directory):\n    img_id = str(img_directory).split('/')[-1][:-4]\n    diagnosis = train_file.loc[train_file['image_id']==int(img_id), 'cancer'].values[0]\n    if diagnosis == 0:\n        return \"negative\"\n    else:\n        return \"positive\"","metadata":{"execution":{"iopub.status.busy":"2023-04-09T13:17:18.521147Z","iopub.execute_input":"2023-04-09T13:17:18.521498Z","iopub.status.idle":"2023-04-09T13:17:18.529764Z","shell.execute_reply.started":"2023-04-09T13:17:18.521461Z","shell.execute_reply":"2023-04-09T13:17:18.528361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install dicomsdl","metadata":{"execution":{"iopub.status.busy":"2023-04-09T13:17:18.531548Z","iopub.execute_input":"2023-04-09T13:17:18.532316Z","iopub.status.idle":"2023-04-09T13:17:28.609987Z","shell.execute_reply.started":"2023-04-09T13:17:18.532273Z","shell.execute_reply":"2023-04-09T13:17:28.608771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Transforming images from DICOM format to PNG and sorting them into a \"positive\" and \"negative\" folder","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport cv2\nimport os\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\nfrom pathlib import Path\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport dicomsdl\nimport sys\nimport time\n\nRESIZE_TO = (256, 256)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T13:17:28.611917Z","iopub.execute_input":"2023-04-09T13:17:28.614423Z","iopub.status.idle":"2023-04-09T13:17:28.621867Z","shell.execute_reply.started":"2023-04-09T13:17:28.614375Z","shell.execute_reply":"2023-04-09T13:17:28.620782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n!mkdir -p /kaggle/working/train_images_processed_cv2_dicomsdl_{RESIZE_TO[0]}/positive/\n!mkdir -p /kaggle/working/train_images_processed_cv2_dicomsdl_{RESIZE_TO[0]}/negative/\n\n# https://www.kaggle.com/code/tanlikesmath/brain-tumor-radiogenomic-classification-eda/notebook\ndef dicom_file_to_ary(path):\n    dcm_file = dicomsdl.open(str(path))\n    data = dcm_file.pixelData()\n\n    data = (data - data.min()) / (data.max() - data.min())\n\n    if dcm_file.getPixelDataInfo()['PhotometricInterpretation'] == \"MONOCHROME1\":\n        data = 1 - data\n\n    data = cv2.resize(data, RESIZE_TO)\n    #data = (data * 255).astype(np.uint8)\n    return data\n\nimage_directories = []\nfor patient_dir in Path('/kaggle/input/rsna-breast-cancer-detection/train_images/').iterdir():\n    for pic_dir in patient_dir.iterdir():\n#         if pic_dir.stem not in done_ids:\n        image_directories.append(pic_dir)\nprint(len(image_directories))\n\ndef process_directory(directory_path):\n    parent_directory = pos_or_neg(directory_path)\n    \n    processed_ary = dicom_file_to_ary(directory_path)\n        \n    cv2.imwrite(\n        f'train_images_processed_cv2_dicomsdl_{RESIZE_TO[0]}/{parent_directory}/{directory_path.stem}.png',\n        processed_ary\n    )\npos_dir = Path(\"/kaggle/working/train_images_processed_cv2_dicomsdl_256/positive/\")\n    \nimport multiprocessing as mp\n\nwith mp.Pool(64) as p:\n    p.map(process_directory, image_directories)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Insuring that the final number of images matches the original number","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\ndata_dir = Path(\"/kaggle/working/train_images_processed_cv2_dicomsdl_256/\")\ndone_paths = list(data_dir.glob('*/*.png'))\nimage_count = len(list(data_dir.glob('*/*.png')))\nprint(image_count)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T13:48:06.833053Z","iopub.execute_input":"2023-04-09T13:48:06.833418Z","iopub.status.idle":"2023-04-09T13:48:06.946087Z","shell.execute_reply.started":"2023-04-09T13:48:06.833384Z","shell.execute_reply":"2023-04-09T13:48:06.945044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Classifying the Images with pytorch**","metadata":{}},{"cell_type":"code","source":"def extract_all_image_paths(input_path):\n    image_paths = [image_path for image_path in Path.glob(input_path,pattern = '*.png')]\n    return image_paths\n\npositive_paths = extract_all_image_paths('/kaggle/working/train_images_processed_cv2_dicomsdl_256/positive')\nnegative_paths = extract_all_image_paths('/kaggle/working/train_images_processed_cv2_dicomsdl_256/negative')\n","metadata":{"execution":{"iopub.status.busy":"2023-04-09T14:02:10.410764Z","iopub.execute_input":"2023-04-09T14:02:10.411167Z","iopub.status.idle":"2023-04-09T14:02:10.444699Z","shell.execute_reply.started":"2023-04-09T14:02:10.411107Z","shell.execute_reply":"2023-04-09T14:02:10.442846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print_accuracy","metadata":{"execution":{"iopub.status.busy":"2023-04-09T13:48:59.305023Z","iopub.execute_input":"2023-04-09T13:48:59.305485Z","iopub.status.idle":"2023-04-09T13:48:59.311936Z","shell.execute_reply.started":"2023-04-09T13:48:59.305445Z","shell.execute_reply":"2023-04-09T13:48:59.310659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_paths = extract_all_image_paths()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'We have a total of {len(image_paths)} images.')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's check that we correctly extracted our paths\ndisplay(image_paths[0:3])\ndisplay(image_paths[10000:10003])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracts a dict of lists containing the informaion of each path\ndef extract_metadata(image_paths) -> dict:\n    path_data = {'path':[],'patient_id':[],'x_coord':[] ,'y_coord':[],'target':[]}\n    pattern = '\\/(\\d+)_.+_x(\\d+)_y(\\d+)_.+(\\d)'\n    for image_path in tqdm(image_paths,total = 277524):\n        meta_data = re.search(pattern, str(image_path))\n        path_data['path'].append(image_path)\n        path_data['patient_id'].append(meta_data.group(1))\n        path_data['x_coord'].append(meta_data.group(2))\n        path_data['y_coord'].append(meta_data.group(3))\n        path_data['target'].append(meta_data.group(4))\n    return path_data","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for value in path_data.values():\n    display(value[0:3]) ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert dictionary to pandas dataframe --> convert path_data dict to dataframe\ndf = pd.DataFrame.from_dict(path_data)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prints the first five rows of the dataframe\ndf.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# to get a better understanding of our dataframe we use the .info() method\ndf.info()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.iloc[:,2:5]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loop through the desired columns and change the types\nfor col in df.iloc[:,2:5]:\n    df[col] = df[col].astype('int')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read image from path\nimg = io.imread('../input/breast-histopathology-images/10253/0/10253_idx5_x1001_y1001_class0.png')\n# show the image after being read\nio.imshow(img);","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}