{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":13451,"datasetId":654585,"databundleVersionId":1188070}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport skimage\nfrom skimage.io import imshow, imread, imsave\nimport numpy as np\nfrom glob import glob\nimport os\nimport cv2\nfrom matplotlib import pyplot as plt\nfrom PIL import Image\nimport tensorflow as tf\nfrom tensorflow import keras as K\nimport pandas as pd\nimport numpy as np\nimport pydicom\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2024-09-15T07:12:43.921112Z","iopub.execute_input":"2024-09-15T07:12:43.921625Z","iopub.status.idle":"2024-09-15T07:13:01.433821Z","shell.execute_reply.started":"2024-09-15T07:12:43.921578Z","shell.execute_reply":"2024-09-15T07:13:01.43259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-09-15T07:13:01.435854Z","iopub.execute_input":"2024-09-15T07:13:01.436627Z","iopub.status.idle":"2024-09-15T07:13:06.898102Z","shell.execute_reply.started":"2024-09-15T07:13:01.436578Z","shell.execute_reply":"2024-09-15T07:13:06.896738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path ='/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train/'","metadata":{"execution":{"iopub.status.busy":"2024-09-15T07:26:00.09625Z","iopub.execute_input":"2024-09-15T07:26:00.097713Z","iopub.status.idle":"2024-09-15T07:26:00.102791Z","shell.execute_reply.started":"2024-09-15T07:26:00.097658Z","shell.execute_reply":"2024-09-15T07:26:00.101549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CT_image_file_names = os.listdir(train_path)","metadata":{"execution":{"iopub.status.busy":"2024-09-15T07:26:01.316866Z","iopub.execute_input":"2024-09-15T07:26:01.317342Z","iopub.status.idle":"2024-09-15T07:26:05.939565Z","shell.execute_reply.started":"2024-09-15T07:26:01.317297Z","shell.execute_reply":"2024-09-15T07:26:05.938139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=df[df['Label']==1]","metadata":{"execution":{"iopub.status.busy":"2024-09-15T07:26:09.21772Z","iopub.execute_input":"2024-09-15T07:26:09.21819Z","iopub.status.idle":"2024-09-15T07:26:09.411838Z","shell.execute_reply.started":"2024-09-15T07:26:09.218146Z","shell.execute_reply":"2024-09-15T07:26:09.410573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to split the ID column manually\ndef split_id(id_value):\n    parts = id_value.split('_', 2)\n    return parts[1], parts[2] if len(parts) > 2 else 'N/A'\n\n# Apply the function\ndf[['File', 'Subcategory']] = df['ID'].apply(split_id).apply(pd.Series)\n\n# Drop unnecessary columns\ndf = df.drop(columns=['ID'])\n\n# Rearrange columns\ndf = df[['File', 'Subcategory', 'Label']]\n\n# Rename columns\ndf.columns = ['File Name', 'Subcategory', 'Label']\n\n# Display the DataFrame\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2024-09-15T07:26:10.774596Z","iopub.execute_input":"2024-09-15T07:26:10.775027Z","iopub.status.idle":"2024-09-15T07:26:51.771599Z","shell.execute_reply.started":"2024-09-15T07:26:10.774988Z","shell.execute_reply":"2024-09-15T07:26:51.770267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add 'ID_' prefix to the 'File Name' column\ndf['File Name'] = 'ID_' + df['File Name']\n\n# Display the DataFrame\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2024-09-15T07:37:36.119586Z","iopub.execute_input":"2024-09-15T07:37:36.120063Z","iopub.status.idle":"2024-09-15T07:37:36.171523Z","shell.execute_reply.started":"2024-09-15T07:37:36.120022Z","shell.execute_reply":"2024-09-15T07:37:36.169919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the subcategories\nsubcategories = ['epidural', 'intraparenchymal', 'intraventricular', 'subarachnoid', 'subdural', 'any']\n\n# Create binary columns for each subcategory\nfor subcategory in subcategories:\n    df[subcategory] = df.apply(lambda row: 1 if row['Subcategory'] == subcategory and row['Label'] == 1 else 0, axis=1)\n\n# Drop the 'Subcategory' and 'Label' columns\ndf = df.drop(columns=['Subcategory', 'Label'])\n\n# Aggregate by 'File Name', taking the maximum value for each subcategory column\ndf_aggregated = df.groupby('File Name').max().reset_index()\n\n# Display the DataFrame\nprint(df_aggregated)","metadata":{"execution":{"iopub.status.busy":"2024-09-15T07:37:39.756587Z","iopub.execute_input":"2024-09-15T07:37:39.757037Z","iopub.status.idle":"2024-09-15T07:38:01.184925Z","shell.execute_reply.started":"2024-09-15T07:37:39.756995Z","shell.execute_reply":"2024-09-15T07:38:01.183711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_aggregated.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-15T07:57:12.830937Z","iopub.execute_input":"2024-09-15T07:57:12.831697Z","iopub.status.idle":"2024-09-15T07:57:12.853885Z","shell.execute_reply.started":"2024-09-15T07:57:12.831647Z","shell.execute_reply":"2024-09-15T07:57:12.851967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the 'File Name' values into a list\nfile_names_list = df_aggregated['File Name'].tolist()","metadata":{"execution":{"iopub.status.busy":"2024-09-15T07:57:16.419864Z","iopub.execute_input":"2024-09-15T07:57:16.420348Z","iopub.status.idle":"2024-09-15T07:57:16.435047Z","shell.execute_reply.started":"2024-09-15T07:57:16.420303Z","shell.execute_reply":"2024-09-15T07:57:16.433689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ICH_dicom_image_paths = pd.DataFrame({'File Name':[n for n in file_names_list],\n                       'image_path':[train_path+n+'.dcm' for n in file_names_list]})","metadata":{"execution":{"iopub.status.busy":"2024-09-15T07:57:21.971425Z","iopub.execute_input":"2024-09-15T07:57:21.971885Z","iopub.status.idle":"2024-09-15T07:57:22.074468Z","shell.execute_reply.started":"2024-09-15T07:57:21.971842Z","shell.execute_reply":"2024-09-15T07:57:22.073176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ICH_dicom_image_paths","metadata":{"execution":{"iopub.status.busy":"2024-09-15T07:57:22.840402Z","iopub.execute_input":"2024-09-15T07:57:22.840828Z","iopub.status.idle":"2024-09-15T07:57:22.855821Z","shell.execute_reply.started":"2024-09-15T07:57:22.840791Z","shell.execute_reply":"2024-09-15T07:57:22.854136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df = pd.merge(left=ICH_dicom_image_paths,right=df_aggregated, on='File Name',how='inner')\nmerged_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-15T07:57:27.326053Z","iopub.execute_input":"2024-09-15T07:57:27.326536Z","iopub.status.idle":"2024-09-15T07:57:27.410532Z","shell.execute_reply.started":"2024-09-15T07:57:27.326493Z","shell.execute_reply":"2024-09-15T07:57:27.409309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df.to_csv('/kaggle/working/out.csv', index=False) ","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:18:40.062286Z","iopub.execute_input":"2024-09-15T08:18:40.062776Z","iopub.status.idle":"2024-09-15T08:18:41.160122Z","shell.execute_reply.started":"2024-09-15T08:18:40.062735Z","shell.execute_reply":"2024-09-15T08:18:41.158786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ICH_300_images_df = merged_df[0:300]","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:18:44.585343Z","iopub.execute_input":"2024-09-15T08:18:44.585793Z","iopub.status.idle":"2024-09-15T08:18:44.591761Z","shell.execute_reply.started":"2024-09-15T08:18:44.58575Z","shell.execute_reply":"2024-09-15T08:18:44.590353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ICH_300_images_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:19:18.497343Z","iopub.execute_input":"2024-09-15T08:19:18.497789Z","iopub.status.idle":"2024-09-15T08:19:18.506001Z","shell.execute_reply.started":"2024-09-15T08:19:18.497745Z","shell.execute_reply":"2024-09-15T08:19:18.504712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def convert_dicom_to_jpg(df, output_folder):\n#     \"\"\"\n#     Convert DICOM images listed in the DataFrame to JPG format and save them to the specified folder.\n\n#     Parameters:\n#     df (pd.DataFrame): DataFrame containing 'File Name', 'image_path' and subcategories.\n#     output_folder (str): Directory where JPG images will be saved.\n#     \"\"\"\n#     # Ensure the output folder exists\n#     if not os.path.exists(output_folder):\n#         os.makedirs(output_folder)\n        \n#     patient_id=[]\n\n#     for index, row in merged_df.iterrows():\n#         file_name = row['File Name']\n#         image_path = row['image_path']\n        \n#         try:\n#             # Read the DICOM image\n#             dicom_image = pydicom.dcmread(image_path)\n#             Image_array = dicom_image.pixel_array.astype(float)\n            \n            \n#             #Read Patient\n#             patient_id.append(dicom_image.PatientID)\n            \n#             # Convert to PIL Image\n#             Image_array = (np.maximum(Image_array,0)/Image_array.max())*255\n#             Image_array = np.uint8(Image_array)\n#             pil_image = Image.fromarray(Image_array)\n            \n#             # Ensure image mode is 'L' (grayscale) or 'RGB'\n#             if pil_image.mode != 'RGB':\n#                 pil_image = pil_image.convert('RGB')\n            \n#             # Save as JPG\n#             jpg_path = os.path.join(output_folder, f\"{file_name}.jpg\")\n#             pil_image.save(jpg_path, format='JPEG')\n\n#             #print(f\"Saved: {jpg_path}\")\n\n#         except Exception as e:\n#             print(f\"Failed to process {image_path}: {e}\")","metadata":{"execution":{"iopub.status.busy":"2024-09-14T18:21:09.829964Z","iopub.execute_input":"2024-09-14T18:21:09.831164Z","iopub.status.idle":"2024-09-14T18:21:09.8436Z","shell.execute_reply.started":"2024-09-14T18:21:09.831097Z","shell.execute_reply":"2024-09-14T18:21:09.841578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pydicom\nfrom PIL import Image\nimport numpy as np\nimport pandas as pd\n\ndef convert_dicom_to_tif(df, output_folder):\n    \"\"\"\n    Convert DICOM images listed in the DataFrame to TIFF format and save them to the specified folder.\n\n    Parameters:\n    df (pd.DataFrame): DataFrame containing 'File Name', 'image_path' and subcategories.\n    output_folder (str): Directory where TIFF images will be saved.\n    \"\"\"\n    # Ensure the output folder exists\n    if not os.path.exists(output_folder):\n        os.makedirs(output_folder)\n        \n    patient_id = []\n\n    for index, row in df.iterrows():\n        file_name = row['File Name']\n        image_path = row['image_path']\n        \n        try:\n            # Read the DICOM image\n            dicom_image = pydicom.dcmread(image_path)\n            image_array = dicom_image.pixel_array.astype(float)\n            \n            # Read Patient ID (optional)\n            patient_id.append(dicom_image.PatientID)\n            \n            # Normalize the image array to 0-255\n            image_array = (np.maximum(image_array, 0) / image_array.max()) * 255\n            image_array = np.uint8(image_array)\n            \n            # Convert to PIL Image\n            pil_image = Image.fromarray(image_array)\n            \n            # Ensure image mode is 'L' (grayscale)\n            if pil_image.mode != 'L':\n                pil_image = pil_image.convert('L')\n            \n            # Save as TIFF\n            tif_path = os.path.join(output_folder, f\"{file_name}.tif\")\n            pil_image.save(tif_path, format='TIFF')\n\n            #print(f\"Saved: {tif_path}\")\n\n        except Exception as e:\n            print(f\"Failed to process {image_path}: {e}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:19:23.978338Z","iopub.execute_input":"2024-09-15T08:19:23.979548Z","iopub.status.idle":"2024-09-15T08:19:23.991537Z","shell.execute_reply.started":"2024-09-15T08:19:23.979494Z","shell.execute_reply":"2024-09-15T08:19:23.990248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert DICOM images to JPG and save them\nconvert_dicom_to_tif(ICH_300_images_df, '/kaggle/working/ICH_images')","metadata":{"execution":{"iopub.status.busy":"2024-09-15T08:19:31.991336Z","iopub.execute_input":"2024-09-15T08:19:31.991785Z","iopub.status.idle":"2024-09-15T08:19:37.493866Z","shell.execute_reply.started":"2024-09-15T08:19:31.991732Z","shell.execute_reply":"2024-09-15T08:19:37.492557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_aggregated.to_csv('/kaggle/working/out.csv', index=False) ","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.597546Z","iopub.status.idle":"2024-09-14T16:50:38.598029Z","shell.execute_reply.started":"2024-09-14T16:50:38.597777Z","shell.execute_reply":"2024-09-14T16:50:38.597797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train_image_name_df = pd.DataFrame({'ID':[n.split('.')[0] for n in CT_image_file_names],\n#                        'image_path':[train_path+ n for n in CT_image_file_names]})","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.599214Z","iopub.status.idle":"2024-09-14T16:50:38.599627Z","shell.execute_reply.started":"2024-09-14T16:50:38.599424Z","shell.execute_reply":"2024-09-14T16:50:38.599444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train_image_name_df","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.601271Z","iopub.status.idle":"2024-09-14T16:50:38.601734Z","shell.execute_reply.started":"2024-09-14T16:50:38.601506Z","shell.execute_reply":"2024-09-14T16:50:38.601528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train_image_name_df[Train_image_name_df['ID']=='ID_aec8e68b3']['image_path']","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.604225Z","iopub.status.idle":"2024-09-14T16:50:38.60471Z","shell.execute_reply.started":"2024-09-14T16:50:38.604461Z","shell.execute_reply":"2024-09-14T16:50:38.604484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dicom_image_path","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.60588Z","iopub.status.idle":"2024-09-14T16:50:38.608026Z","shell.execute_reply.started":"2024-09-14T16:50:38.607752Z","shell.execute_reply":"2024-09-14T16:50:38.607777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np\n# import pydicom\n# from PIL import Image","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.609854Z","iopub.status.idle":"2024-09-14T16:50:38.61031Z","shell.execute_reply.started":"2024-09-14T16:50:38.610084Z","shell.execute_reply":"2024-09-14T16:50:38.610106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dicom_im = pydicom.dcmread('/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train/ID_aec8e68b3.dcm')","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.611898Z","iopub.status.idle":"2024-09-14T16:50:38.612293Z","shell.execute_reply.started":"2024-09-14T16:50:38.612097Z","shell.execute_reply":"2024-09-14T16:50:38.612117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dicom_im","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.613885Z","iopub.status.idle":"2024-09-14T16:50:38.614309Z","shell.execute_reply.started":"2024-09-14T16:50:38.614104Z","shell.execute_reply":"2024-09-14T16:50:38.614125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# img = dicom_im.pixel_array.astype(float)","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.615725Z","iopub.status.idle":"2024-09-14T16:50:38.61618Z","shell.execute_reply.started":"2024-09-14T16:50:38.615971Z","shell.execute_reply":"2024-09-14T16:50:38.615992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# img.shape","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.617733Z","iopub.status.idle":"2024-09-14T16:50:38.618174Z","shell.execute_reply.started":"2024-09-14T16:50:38.617964Z","shell.execute_reply":"2024-09-14T16:50:38.617984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.imshow(img,cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.619418Z","iopub.status.idle":"2024-09-14T16:50:38.619874Z","shell.execute_reply.started":"2024-09-14T16:50:38.619637Z","shell.execute_reply":"2024-09-14T16:50:38.619659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dicom_im_2 = pydicom.dcmread('/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train/ID_9b7d000a2.dcm')","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.622087Z","iopub.status.idle":"2024-09-14T16:50:38.622501Z","shell.execute_reply.started":"2024-09-14T16:50:38.622288Z","shell.execute_reply":"2024-09-14T16:50:38.622307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#img_2 = dicom_im_2.pixel_array.astype(float)","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.624354Z","iopub.status.idle":"2024-09-14T16:50:38.62477Z","shell.execute_reply.started":"2024-09-14T16:50:38.624557Z","shell.execute_reply":"2024-09-14T16:50:38.624579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.imshow(img_2,cmap=plt.cm.bone)","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.626065Z","iopub.status.idle":"2024-09-14T16:50:38.626488Z","shell.execute_reply.started":"2024-09-14T16:50:38.626263Z","shell.execute_reply":"2024-09-14T16:50:38.626283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#CT_ID_0002081b6= pydicom.dcmread('/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train/ID_0002081b6.dcm')","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.627874Z","iopub.status.idle":"2024-09-14T16:50:38.628363Z","shell.execute_reply.started":"2024-09-14T16:50:38.628151Z","shell.execute_reply":"2024-09-14T16:50:38.628172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# img_3 = CT_ID_0002081b6.pixel_array.astype(float) \n# rescaled_img_3 = (np.maximum(img_3,0)/img_3.max())*255\n# final_img_3 = np.uint8(rescaled_img_3)\n# #plt.imshow(img_3,cmap='gray')\n# final_img_3=Image.fromarray(final_img_3)\n# #final_img_3.show()\n# plt.imshow(final_img_3,cmap=plt.cm.bone)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.630068Z","iopub.status.idle":"2024-09-14T16:50:38.63047Z","shell.execute_reply.started":"2024-09-14T16:50:38.630262Z","shell.execute_reply":"2024-09-14T16:50:38.630282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.imshow(final_img_3,cmap=plt.cm.gist_gray)","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.631722Z","iopub.status.idle":"2024-09-14T16:50:38.632173Z","shell.execute_reply.started":"2024-09-14T16:50:38.631962Z","shell.execute_reply":"2024-09-14T16:50:38.631984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#CT_ID_0002081b6.PatientID ","metadata":{"execution":{"iopub.status.busy":"2024-09-14T16:50:38.63393Z","iopub.status.idle":"2024-09-14T16:50:38.634369Z","shell.execute_reply.started":"2024-09-14T16:50:38.634152Z","shell.execute_reply":"2024-09-14T16:50:38.634172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}