{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Links:\nhttps://hengloose.medium.com/a-comprehensive-starter-guide-to-visualizing-and-analyzing-dicom-images-in-python-7a8430fcb7ed\n\nhttps://towardsdatascience.com/medical-image-pre-processing-with-python-d07694852606\n\nhttps://www.kaggle.com/code/subbuvolvosekar/rsna-breastcancer-data-understanding\n\nhttps://www.kaggle.com/code/javigallego/rsna-complete-eda-external-data","metadata":{}},{"cell_type":"code","source":"!pip install -U pylibjpeg pylibjpeg-openjpeg pylibjpeg-libjpeg pydicom python-gdcm","metadata":{"execution":{"iopub.status.busy":"2023-01-01T07:13:11.823031Z","iopub.execute_input":"2023-01-01T07:13:11.82349Z","iopub.status.idle":"2023-01-01T07:13:23.397313Z","shell.execute_reply.started":"2023-01-01T07:13:11.823402Z","shell.execute_reply":"2023-01-01T07:13:23.395816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Exploring dataset\nimport pandas as pd\n\ndataset_path=\"/kaggle/input/rsna-breast-cancer-detection\"\ntrain_df_path=dataset_path+\"/train.csv\"\ntest_df_path=dataset_path+\"/test.csv\"\n\ntrain_df = pd.read_csv(train_df_path)\ntest_df = pd.read_csv(test_df_path)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T07:13:23.399712Z","iopub.execute_input":"2023-01-01T07:13:23.400198Z","iopub.status.idle":"2023-01-01T07:13:23.490051Z","shell.execute_reply.started":"2023-01-01T07:13:23.400158Z","shell.execute_reply":"2023-01-01T07:13:23.488709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Total train dataset: {len(train_df)}\")\ntrain_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Total test dataset: {len(test_df)}\")\ntest_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\ntrain_images = glob.glob(dataset_path+\"/train_images/*/*\")","metadata":{"execution":{"iopub.status.busy":"2023-01-01T07:13:34.953532Z","iopub.execute_input":"2023-01-01T07:13:34.954353Z","iopub.status.idle":"2023-01-01T07:13:43.977626Z","shell.execute_reply.started":"2023-01-01T07:13:34.954302Z","shell.execute_reply":"2023-01-01T07:13:43.976417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dir_name in [\"train\",\"test\"]:\n    os.makedirs(dir_name, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T07:32:43.862198Z","iopub.execute_input":"2023-01-01T07:32:43.862644Z","iopub.status.idle":"2023-01-01T07:32:43.868322Z","shell.execute_reply.started":"2023-01-01T07:32:43.8626Z","shell.execute_reply":"2023-01-01T07:32:43.867367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2023-01-01T07:32:50.923287Z","iopub.execute_input":"2023-01-01T07:32:50.923683Z","iopub.status.idle":"2023-01-01T07:32:52.024273Z","shell.execute_reply.started":"2023-01-01T07:32:50.923651Z","shell.execute_reply":"2023-01-01T07:32:52.022651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nfrom PIL import Image\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nfrom multiprocessing import Pool,cpu_count\n\ndef process_dcm(path:str):\n    \n    # Open the DICOM file\n    ds = pydicom.dcmread(path)\n    \n    # Get the image data\n    image = ds.pixel_array\n\n    # Convert the image data to a PIL image\n    pil_image = Image.fromarray(image) \n    \n    # plt.imshow(pil_image,cmap=\"gray\")\n    # print(image.shape,path)\n    \n    patient_id = path.split(\"/\")[-2]\n    image_id = path.split(\"/\")[-1].split(\".\")[0]\n    \n    # Resize the image\n    resized_image = pil_image.resize((256, 256))\n\n    # Save the image to a new file\n    resized_image.save(f'train/{patient_id}_{image_id}.png')\n\n\nwith Pool(cpu_count()) as p:\n        p.map(process_dcm, tqdm(train_images))","metadata":{"execution":{"iopub.status.busy":"2023-01-01T07:32:57.925346Z","iopub.execute_input":"2023-01-01T07:32:57.925831Z","iopub.status.idle":"2023-01-01T07:40:56.018675Z","shell.execute_reply.started":"2023-01-01T07:32:57.925788Z","shell.execute_reply":"2023-01-01T07:40:56.015979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot images\nprocessed_train_images = glob.glob(\"train/*\")\n\n# Create a figure with a grid of subplots\nfig, axs = plt.subplots(nrows=3, ncols=5,figsize=(20, 12))\n\n# Plot the images in the subplots\nfor i, ax in enumerate(axs.flat):\n    # Get the image and label for the current subplot\n    image = Image.open(processed_train_images[i+20])\n    \n#     # Convert the image to a numpy array and plot it\n#     img = image.numpy().transpose((1, 2, 0))\n#     # mean = np.array([0.485, 0.456, 0.406])\n#     # std = np.array([0.229, 0.224, 0.225])\n#     # img = std * img + mean\n#     img = np.clip(img, 0, 1)\n\n    ax.imshow(image,cmap=\"gray\")\n    \n#     # Set the title of the subplot to the label\n#     ax.set_title(class_names[label])\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T07:44:41.317737Z","iopub.execute_input":"2023-01-01T07:44:41.318227Z","iopub.status.idle":"2023-01-01T07:44:43.133374Z","shell.execute_reply.started":"2023-01-01T07:44:41.318187Z","shell.execute_reply":"2023-01-01T07:44:43.132064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}