{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":13451,"datasetId":654585,"databundleVersionId":1188070},{"sourceType":"datasetVersion","sourceId":7115669,"datasetId":4103616,"databundleVersionId":7203628}],"dockerImageVersionId":29840,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"The datasets are too big in many kaggle competitions. For example, the RSNA intracranial hemorrhage detection dataset is 180G. Sometimes I just want to download a very small subset of the original dataset so I can play with it in my local computer. Below is an example of how I achieve this goal. The idea is to create a Kaggle kernel to copy some data files into an output folder, then zip the folder and download the generated .zip file from Kaggle kernel interface.","metadata":{}},{"cell_type":"code","source":"# import os\n# import random\n# from shutil import copy, make_archive\n\n\n# data_root = '/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection'\n\n# k = 10 # randomly select 100 images from both train and test data folder\n\n# os.makedirs('./dataset', exist_ok=True) # create a dataset folder to hold all the files that I wanted to download\n# copy(os.path.join(data_root, 'stage_2_sample_submission.csv'), 'dataset/stage_2_sample_submission.csv')\n# copy(os.path.join(data_root, 'stage_2_train.csv'), 'dataset/stage_2_train.csv')\n# for d in ['stage_2_train', 'stage_2_test']:\n#     # list all images in train/test folder\n#     dir_path = os.path.join(data_root, d)\n#     files = os.listdir(dir_path)\n    \n#     # copy images to target folder\n#     target_dir = os.path.join('dataset', d)\n#     x = 0\n#     os.makedirs(target_dir, exist_ok=True) \n#     for f in random.choices(files, k=k): # randomly select k images and copy them to the target folder\n#         src_file = os.path.join(dir_path, f)\n#         copy(src_file, target_dir)\n#         x+=1\n# #         print(f'generated file {x}')\n        \n# # zip generated files\n# make_archive(base_name='download_dataset_100', format='zip', root_dir='dataset')\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-03T22:26:28.885918Z","iopub.execute_input":"2023-12-03T22:26:28.886488Z","iopub.status.idle":"2023-12-03T22:26:28.891778Z","shell.execute_reply.started":"2023-12-03T22:26:28.886416Z","shell.execute_reply":"2023-12-03T22:26:28.890707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now you can download the `download_dataset.zip` from the kernel interface.\n\n![](https://user-images.githubusercontent.com/1262709/69744216-c2e11780-110d-11ea-82b4-88006cc6d0aa.png)","metadata":{}},{"cell_type":"code","source":"import os\nfrom shutil import copy, make_archive\n\ndata_root = '/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection'\n\nk = 100 # select the first 100 images from both train and test data folder\n\nos.makedirs('./dataset', exist_ok=True) # create a dataset folder to hold all the files that I wanted to download\ncopy(os.path.join(data_root, 'stage_2_sample_submission.csv'), './dataset/stage_2_sample_submission.csv')\ncopy(os.path.join(data_root, 'stage_2_train.csv'), './dataset/stage_2_train.csv')\n\nfor d in ['stage_2_train']:\n    # list all images in train/test folder\n    dir_path = os.path.join(data_root, d)\n    files = sorted(os.listdir(dir_path))\n    print(f\"Total images in {d}: {len(files)}\")\n    print(dir_path)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:26:50.392119Z","iopub.execute_input":"2023-12-03T22:26:50.392491Z","iopub.status.idle":"2023-12-03T22:26:52.529183Z","shell.execute_reply.started":"2023-12-03T22:26:50.392435Z","shell.execute_reply":"2023-12-03T22:26:52.528316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:14:23.485132Z","iopub.execute_input":"2023-12-04T01:14:23.485489Z","iopub.status.idle":"2023-12-04T01:14:23.489428Z","shell.execute_reply.started":"2023-12-04T01:14:23.485444Z","shell.execute_reply":"2023-12-04T01:14:23.488582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#     # copy images to target folder\n#     target_dir = os.path.join('./dataset', d)\n#     os.makedirs(target_dir, exist_ok=True) \n#     for i in range(k): # select the first k images and copy them to the target folder\n#         src_file = os.path.join(dir_path, files[i])\n#         copy(src_file, target_dir)\n#         print(f'Copied file {src_file }_ {i+1}/{k} from {d}')\n\n# # zip generated files\n# # ID_000012eaf\n# # ID_000039fa0\n# # ID_00005679d\n# make_archive(base_name='download_dataset_train_100', format='zip', root_dir='./dataset')","metadata":{"execution":{"iopub.status.busy":"2023-11-29T22:35:09.045223Z","iopub.execute_input":"2023-11-29T22:35:09.045629Z","iopub.status.idle":"2023-11-29T22:35:09.050945Z","shell.execute_reply.started":"2023-11-29T22:35:09.045568Z","shell.execute_reply":"2023-11-29T22:35:09.049618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in files :\n    file = os.path.join(dir_path, i)\n    file = os.path.join(dir_path, 'ID_45785016b.dcm')\n\n    print(file)\n    break","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:28:17.180185Z","iopub.execute_input":"2023-12-03T22:28:17.180542Z","iopub.status.idle":"2023-12-03T22:28:17.187602Z","shell.execute_reply.started":"2023-12-03T22:28:17.180491Z","shell.execute_reply":"2023-12-03T22:28:17.186383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = pydicom.dcmread(file)","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:28:20.73835Z","iopub.execute_input":"2023-12-03T22:28:20.738949Z","iopub.status.idle":"2023-12-03T22:28:20.74619Z","shell.execute_reply.started":"2023-12-03T22:28:20.738896Z","shell.execute_reply":"2023-12-03T22:28:20.745446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:28:26.567347Z","iopub.execute_input":"2023-12-03T22:28:26.567714Z","iopub.status.idle":"2023-12-03T22:28:26.57535Z","shell.execute_reply.started":"2023-12-03T22:28:26.567655Z","shell.execute_reply":"2023-12-03T22:28:26.573892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:27:25.591116Z","iopub.execute_input":"2023-12-03T22:27:25.591458Z","iopub.status.idle":"2023-12-03T22:27:25.595817Z","shell.execute_reply.started":"2023-12-03T22:27:25.591408Z","shell.execute_reply":"2023-12-03T22:27:25.594973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# csv functions \ndef get_train_csv(df1, filenames):\n    # Add the '.jpg' extension to filenames if not already there\n    filenames = [f if f.endswith('.jpg') else f + '.jpg' for f in filenames]\n\n    df1['filename'] = df1['ID'].apply(lambda x: '_'.join(x.split('_')[:-1]) + '.jpg')\n    df1['Illness'] = df1['ID'].apply(lambda x: x.split('_')[-1])\n    \n    # Drop the old 'ID' column as it's no longer needed\n    df1.drop(columns=['ID'], inplace=True)\n    \n    # Aggregate rows that have the same 'filename' and 'Illness' by summing up the 'Label'\n    df1 = df1.groupby(['filename', 'Illness']).agg('sum').reset_index()\n    \n    # Pivot the table to get Illness types as columns\n    df2 = df1.pivot(index='filename', columns='Illness', values='Label').reset_index()\n    \n    # Reorder the columns as per your requirements\n    columns = ['filename', 'any', 'epidural', 'intraparenchymal', 'intraventricular', 'subarachnoid', 'subdural']\n    df2 = df2.reindex(columns=columns)\n    \n    # Fill NaN with zeros since some filenames might not have all types of illnesses\n    df2.fillna(0, inplace=True)\n    \n    # Cast the label values to integers\n    label_columns = ['any', 'epidural', 'intraparenchymal', 'intraventricular', 'subarachnoid', 'subdural']\n    df2[label_columns] = df2[label_columns].astype(int)\n    \n    # Filter the DataFrame to only include filenames provided in the 'filenames' list\n    df2 = df2[df2['filename'].isin(filenames)]\n    \n    return df2\n\n# def get_train_csv_2(df1):\n#     columns = ['filename', 'any', 'epidural', 'intraparenchymal',\n#                'intraventricular', 'subarachnoid', 'subdural']\n#     df2 = pd.DataFrame(columns=columns)\n# #     df1 = df1[0:180]\n#     for id in tqdm(df1['ID']):\n#         ImageID = '_'.join(id.split('_')[:-1])\n#         ImageID = ImageID + '.jpg'\n#         Illness = id.split('_')[-1]\n#         value = df1[df1['ID'] == id]['Label'].values[0]\n#         if not (ImageID in df2['filename'].tolist()):\n#             new_index = len(df2)\n#             df2.loc[new_index] = [ImageID, 0, 0, 0, 0, 0, 0]\n#         df2.loc[df2['filename'] == ImageID, Illness] = value\n    \n#     return df2\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:34:36.854116Z","iopub.execute_input":"2023-12-04T01:34:36.854633Z","iopub.status.idle":"2023-12-04T01:34:36.865765Z","shell.execute_reply.started":"2023-12-04T01:34:36.854587Z","shell.execute_reply":"2023-12-04T01:34:36.864592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_dicom_value(x, cast=int):\n    if type(x) in [pydicom.multival.MultiValue, tuple]:\n        return cast(x[0])\n    else:\n        return cast(x)\n\n\ndef cast(value):\n    if type(value) is pydicom.valuerep.MultiValue:\n        return tuple(value)\n    return value\n\n\ndef get_dicom_raw(dicom):\n    return {attr:cast(getattr(dicom,attr)) for attr in dir(dicom) if attr[0].isupper() and attr not in ['PixelData']}\n\n\ndef rescale_image(image, slope, intercept):\n    return image * slope + intercept\n\ndef apply_window(image, center, width):\n    image = image.copy()\n    min_value = center - width // 2\n    max_value = center + width // 2\n    image[image < min_value] = min_value\n    image[image > max_value] = max_value\n    return image\n\ndef apply_window_policy(image):\n\n    image1 = apply_window(image, 40, 80) # brain\n    image2 = apply_window(image, 80, 200) # subdural\n    image3 = apply_window(image, 40, 380) # bone\n    image1 = (image1 - 0) / 80\n    image2 = (image2 - (-20)) / 200\n    image3 = (image3 - (-150)) / 380\n    image = np.array([\n        image1 - image1.mean(),\n        image2 - image2.mean(),\n        image3 - image3.mean(),\n    ]).transpose(1,2,0)\n\n    return image\n\n\ndef convert_dicom_to_jpg(name):\n#     imgnm = (name.split('/')[-1]).replace('.dcm', '')\n#     dicom = pydicom.dcmread(DicomBytesIO(data))\n    dicom = pydicom.dcmread(name)\n    image = dicom.pixel_array\n    image = rescale_image(image,dicom.RescaleSlope, dicom.RescaleIntercept )\n    image = apply_window_policy(image)\n    image -= image.min((0,1))\n    image = (255*image).astype(np.uint8)\n    return image\n\n\n# plt.imsave('/kaggle/working/dataset/output.jpg',output )\n\n\n\n\n# function loop through the data_root and return \ndef create_dataset(train_data_path , train_labels_csv  ,number_of_imgs = 100 , output_name = 'imgs_jpg_dataset') :\n    os.makedirs(f\"./{output_name}/images\", exist_ok=True) \n    dataset_path = f'/kaggle/working/{output_name}'\n    files = os.listdir(train_data_path)\n    \n    for file in files:\n        number_of_imgs -= 1\n        img_path = os.path.join(train_data_path,file)\n        img = convert_dicom_to_jpg(img_path)\n        plt.imsave(f\"{dataset_path}/images/{file.split('.')[0]}.jpg\", img)\n        if number_of_imgs <= 0 :\n            break\n            \n            \n    filenames = os.listdir(f'{dataset_path}/images')\n    df2 = get_train_csv(pd.read_csv(train_labels_csv) , filenames)\n    df2.to_csv(f\"{dataset_path}/labels.csv\", index=False)\n    make_archive(base_name=f'download_{output_name}', format='zip', root_dir=dataset_path)\n    return(dataset_path)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:43:46.29429Z","iopub.execute_input":"2023-12-04T01:43:46.294649Z","iopub.status.idle":"2023-12-04T01:43:46.316456Z","shell.execute_reply.started":"2023-12-04T01:43:46.294607Z","shell.execute_reply":"2023-12-04T01:43:46.315412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf /kaggle/working/*","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:43:48.878968Z","iopub.execute_input":"2023-12-04T01:43:48.879639Z","iopub.status.idle":"2023-12-04T01:43:49.995743Z","shell.execute_reply.started":"2023-12-04T01:43:48.879584Z","shell.execute_reply":"2023-12-04T01:43:49.994458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_path = '/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train'\ntrain_labels_csv = \"/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train.csv\"\n\ndataset_path = create_dataset(train_data_path , train_labels_csv  ,number_of_imgs = 100 , output_name = 'dataset_new')\n","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:46:28.285038Z","iopub.execute_input":"2023-12-04T01:46:28.285468Z","iopub.status.idle":"2023-12-04T01:46:59.837651Z","shell.execute_reply.started":"2023-12-04T01:46:28.2854Z","shell.execute_reply":"2023-12-04T01:46:59.836866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndf = pd.read_csv('/kaggle/input/github-repo/filenames_grouped_by_patient.csv')\ndf","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:27:50.535936Z","iopub.execute_input":"2023-12-03T22:27:50.536312Z","iopub.status.idle":"2023-12-03T22:27:51.163457Z","shell.execute_reply.started":"2023-12-03T22:27:50.536248Z","shell.execute_reply":"2023-12-03T22:27:51.162315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_of_list_of_filenames = []\nfor i in range(len(df)):\n    if i == 0:\n        list_of_filenames = []\n        list_of_filenames.append(df['filename'][i])\n        patient_id = df['PatientID'][i]\n    else:\n        if df['PatientID'][i] == patient_id:\n            list_of_filenames.append(df['filename'][i])\n        else:\n            list_of_list_of_filenames.append(list_of_filenames)\n            list_of_filenames = []\n            list_of_filenames.append(df['filename'][i])\n            patient_id = df['PatientID'][i]\n    if len(list_of_list_of_filenames) > 1: \n        break","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:27:52.860753Z","iopub.execute_input":"2023-12-03T22:27:52.861067Z","iopub.status.idle":"2023-12-03T22:27:52.901788Z","shell.execute_reply.started":"2023-12-03T22:27:52.861019Z","shell.execute_reply":"2023-12-03T22:27:52.900995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_of_list_of_filenames","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:27:53.880837Z","iopub.execute_input":"2023-12-03T22:27:53.881262Z","iopub.status.idle":"2023-12-03T22:27:53.888447Z","shell.execute_reply.started":"2023-12-03T22:27:53.881183Z","shell.execute_reply":"2023-12-03T22:27:53.887309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res_imgs = []\nfor list_ in list_of_list_of_filenames:\n    for file in list_:\n        path = os.path.join(dir_path, file)\n#         print(path)\n        res_imgs.append(convert_dicom_to_jpg(path))\n    break\nprint(len(res_imgs))","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:32:51.839686Z","iopub.execute_input":"2023-12-03T22:32:51.840043Z","iopub.status.idle":"2023-12-03T22:32:52.484504Z","shell.execute_reply.started":"2023-12-03T22:32:51.839981Z","shell.execute_reply":"2023-12-03T22:32:52.483609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_images_in_grid(array_of_images):\n    fig = plt.figure(figsize=(20,20))\n    columns = 6\n    rows =67\n    for i in range(1, columns*rows +1):\n        img = array_of_images[i-1]\n        fig.add_subplot(rows, columns, i)\n        plt.imshow(img)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:33:00.002695Z","iopub.execute_input":"2023-12-03T22:33:00.003073Z","iopub.status.idle":"2023-12-03T22:33:00.009556Z","shell.execute_reply.started":"2023-12-03T22:33:00.003008Z","shell.execute_reply":"2023-12-03T22:33:00.008306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_images_in_grid(res_imgs)","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:33:00.26565Z","iopub.execute_input":"2023-12-03T22:33:00.266021Z","iopub.status.idle":"2023-12-03T22:33:04.657631Z","shell.execute_reply.started":"2023-12-03T22:33:00.265954Z","shell.execute_reply":"2023-12-03T22:33:04.656765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GEZZO\nimport os\nimport random\nfrom shutil import copy, make_archive, copy2\nimport pydicom\n\ndata_root = '/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection'\n\nos.makedirs('./ordering', exist_ok=True) \n\nlist all images in train/test folder\ndir_path = os.path.join(data_root, 'stage_2_train')\nfiles = os.listdir(dir_path)\n\ncopy images to target folder\ntarget_dir = os.path.join('ordering', 'stage_2_train')\nos.makedirs(target_dir, exist_ok=True)\nfor f in files: \n    src_file = os.path.join(dir_path, f)\n    if ():\n        copy(src_file, target_dir)\n\nSpecify the directory containing DICOM files\ndirectory = target_dir\n\nList all DICOM files in the directory\ndicom_files = [file for file in os.listdir(directory) if file.endswith(\".dcm\")]\n\nsave or use the sorted list as needed\nos.makedirs('./ordering/ORDERED', exist_ok=True)\nordered_dir = os.path.join('ordering', 'ORDERED')\nfor file in dicom_files:\n    destination_path = os.path.join(ordered_dir, NewName)\n    copy2(file_path, destination_path)\n\nmake_archive(base_name='ORDERED', format='zip', root_dir=ordered_dir)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-03T22:22:41.20523Z","iopub.execute_input":"2023-12-03T22:22:41.205604Z","iopub.status.idle":"2023-12-03T22:22:41.209289Z","shell.execute_reply.started":"2023-12-03T22:22:41.205541Z","shell.execute_reply":"2023-12-03T22:22:41.208519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}