{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nhuggingface_token = user_secrets.get_secret(\"huggingface\")\n\n!mkdir -p ~/.huggingface\n!echo -n $huggingface_token > ~/.huggingface/token\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-29T23:47:08.197552Z","iopub.execute_input":"2023-01-29T23:47:08.198334Z","iopub.status.idle":"2023-01-29T23:47:10.530341Z","shell.execute_reply.started":"2023-01-29T23:47:08.198238Z","shell.execute_reply":"2023-01-29T23:47:10.528386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U python-gdcm pylibjpeg[all] timm pydicom","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:47:17.619408Z","iopub.execute_input":"2023-01-29T23:47:17.61992Z","iopub.status.idle":"2023-01-29T23:47:32.197831Z","shell.execute_reply.started":"2023-01-29T23:47:17.619878Z","shell.execute_reply":"2023-01-29T23:47:32.195937Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport os\nimport sys\nimport PIL\nimport time\nimport pydicom\nimport datasets\nimport argparse\nimport numpy as np\nimport pandas as pd\nimport huggingface_hub\nfrom huggingface_hub import HfApi\nfrom huggingface_hub import hf_hub_download\nfrom pydicom.pixel_data_handlers import apply_windowing\n","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:47:33.749835Z","iopub.execute_input":"2023-01-29T23:47:33.750268Z","iopub.status.idle":"2023-01-29T23:47:34.846652Z","shell.execute_reply.started":"2023-01-29T23:47:33.750236Z","shell.execute_reply":"2023-01-29T23:47:34.845596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\n\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:53:16.911822Z","iopub.execute_input":"2023-01-29T23:53:16.912198Z","iopub.status.idle":"2023-01-29T23:53:16.98997Z","shell.execute_reply.started":"2023-01-29T23:53:16.912168Z","shell.execute_reply":"2023-01-29T23:53:16.988504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_dicom_with_windowing(dcm_file):\n    # from: https://www.kaggle.com/code/davidbroberts/mammography-apply-windowing/\n    im = pydicom.dcmread(dcm_file)\n    data = im.pixel_array\n\n    # This line is the only difference in the two functions\n    data = apply_windowing(data, im)\n\n    if im.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    else:\n        data = data - np.min(data)\n\n    if np.max(data) != 0:\n        data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n\n    return data\n\n\ndef save_image(image: np.ndarray, file_path: str):\n    image = PIL.Image.fromarray(image)\n    os.makedirs(os.path.dirname(file_path), exist_ok=True)\n    image.save(file_path, \"PNG\")\n\n\ndef resize_image(image: np.ndarray, size: int) -> np.ndarray:\n    image = PIL.Image.fromarray(image)\n    image = image.resize((size, size))\n    image = np.array(image)\n    return image\n\nprint(len(os.listdir(\"/kaggle/input/rsna-breast-cancer-detection/train_images\")))","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:47:43.551936Z","iopub.execute_input":"2023-01-29T23:47:43.552349Z","iopub.status.idle":"2023-01-29T23:47:43.68889Z","shell.execute_reply.started":"2023-01-29T23:47:43.5523Z","shell.execute_reply":"2023-01-29T23:47:43.687677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_uploaded_patient_list(dataset_id: str) -> list:\n    # get list of uploaded patients\n    api = HfApi()\n    repo_files = api.list_repo_files(\n        repo_id=dataset_id, repo_type=\"dataset\",\n    )\n    repo_files = [file for file in repo_files if file.endswith(\".png\")]\n    all_patients = [file.split(\"/\")[1] for file in repo_files]\n    return list(set(all_patients))\n","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:50:50.501288Z","iopub.execute_input":"2023-01-29T23:50:50.501733Z","iopub.status.idle":"2023-01-29T23:50:50.508402Z","shell.execute_reply.started":"2023-01-29T23:50:50.501699Z","shell.execute_reply":"2023-01-29T23:50:50.507245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_id = \"mm-ai/kaggle-png-unresized\"","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:47:47.992037Z","iopub.execute_input":"2023-01-29T23:47:47.992418Z","iopub.status.idle":"2023-01-29T23:47:47.997925Z","shell.execute_reply.started":"2023-01-29T23:47:47.992388Z","shell.execute_reply":"2023-01-29T23:47:47.996599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"huggingface_hub.create_repo(\n        repo_id=dataset_id,\n        repo_type=\"dataset\",\n        exist_ok=True,\n        private=True,\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:47:57.203159Z","iopub.execute_input":"2023-01-29T23:47:57.203577Z","iopub.status.idle":"2023-01-29T23:47:58.020604Z","shell.execute_reply.started":"2023-01-29T23:47:57.203545Z","shell.execute_reply":"2023-01-29T23:47:58.019722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"uploaded_patients = get_uploaded_patient_list(dataset_id)\nprint(len(uploaded_patients))\nprint(train_df.shape)\ntrain_df[\"patient_id\"] = train_df[\"patient_id\"].astype(str)\ntrain_df = train_df[~train_df[\"patient_id\"].isin(uploaded_patients)]\nprint(train_df.shape)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:53:53.084411Z","iopub.execute_input":"2023-01-29T23:53:53.084838Z","iopub.status.idle":"2023-01-29T23:53:53.256998Z","shell.execute_reply.started":"2023-01-29T23:53:53.084802Z","shell.execute_reply":"2023-01-29T23:53:53.255753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for patient_id, df in train_df.groupby(\"patient_id\"):\n    files = {laterality + view: df[df[\"view\"] == view][\"image_id\"].values[0] for laterality in [\"R\", \"L\"] for view in [\"CC\", \"MLO\"]}\n    for view_name, image_id in files.items():\n        image = read_dicom_with_windowing(f\"/kaggle/input/rsna-breast-cancer-detection/train_images/{patient_id}/{image_id}.dcm\")\n#         image = resize_image(image, 512)\n        save_image(image, f\"/kaggle/working/train/{patient_id}/{view_name}.png\")\n        \n    try:\n        huggingface_hub.upload_folder(\n            folder_path=f\"train/{patient_id}/\",\n            path_in_repo=f\"data/{patient_id}\",\n            repo_id=dataset_id,\n            repo_type=\"dataset\",\n        )\n    except:\n        print(f\"Could not upload {patient_id}\")\n        continue\n        \n    \n","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:53:58.517161Z","iopub.execute_input":"2023-01-29T23:53:58.517536Z","iopub.status.idle":"2023-01-29T23:54:50.346001Z","shell.execute_reply.started":"2023-01-29T23:53:58.517504Z","shell.execute_reply":"2023-01-29T23:54:50.34465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}