{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"try:\n    import pylibjpeg\nexcept:\n#     !pip install /kaggle/input/utils-whl/utils/Pillow-9.2.0-cp37-cp37m-manylinux_2_28_x86_64.whl\n#     !pip install /kaggle/input/utils-whl/utils/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n#     !pip install /kaggle/input/utils-whl/utils/opencv_python-4.6.0.66-cp37-abi3-macosx_11_0_arm64.whl\n#     !pip install /kaggle/input/utils-whl/utils/psutil-5.9.4-cp38-abi3-macosx_11_0_arm64.whl\n#     !pip install /kaggle/input/utils-whl/utils/pydicom-2.3.0-py3-none-any.whl\n#     !pip install /kaggle/input/utils-whl/utils/pylibjpeg-1.4.0-py3-none-any.whl\n    !pip install /kaggle/input/utils-whl/utils/python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n    !pip install /kaggle/input/utils-whl/utils/requests-2.28.1-py3-none-any.whl\n# #     !pip install /kaggle/input/utils-whl/utils/thop-0.1.1.post2209072238-py3-none-any.whl\n#     !pip install /kaggle/input/utils-whl/utils/torch-1.12.1-cp37-cp37m-manylinux1_x86_64.whl\n#     !pip install /kaggle/input/utils-whl/utils/torchvision-0.13.1-cp37-cp37m-manylinux1_x86_64.whl\n#     !pip install /kaggle/input/utils-whl/utils/tqdm-4.64.1-py2.py3-none-any.whl\n        ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-21T02:15:41.437633Z","iopub.execute_input":"2023-01-21T02:15:41.438389Z","iopub.status.idle":"2023-01-21T02:16:45.607574Z","shell.execute_reply.started":"2023-01-21T02:15:41.438351Z","shell.execute_reply":"2023-01-21T02:16:45.606333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!unzip -q /kaggle/input/timm-with-dependencies/'timm with dependencies'/timm_all -d timm-with-dependencies\n!pip install --no-index --find-links timm-with-dependencies timm\n!pip install ../input/discom/discom/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n","metadata":{"execution":{"iopub.status.busy":"2023-01-21T02:16:45.611548Z","iopub.execute_input":"2023-01-21T02:16:45.611877Z","iopub.status.idle":"2023-01-21T02:17:57.281802Z","shell.execute_reply.started":"2023-01-21T02:16:45.611839Z","shell.execute_reply":"2023-01-21T02:17:57.280507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEBUG = False","metadata":{"execution":{"iopub.status.busy":"2023-01-21T02:17:57.283604Z","iopub.execute_input":"2023-01-21T02:17:57.285308Z","iopub.status.idle":"2023-01-21T02:17:57.29087Z","shell.execute_reply.started":"2023-01-21T02:17:57.285263Z","shell.execute_reply":"2023-01-21T02:17:57.289509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys\nimport cv2\nimport glob\nimport gdcm\nimport json\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom tqdm.notebook import tqdm\nfrom joblib import Parallel, delayed","metadata":{"execution":{"iopub.status.busy":"2023-01-21T02:17:57.292812Z","iopub.execute_input":"2023-01-21T02:17:57.293237Z","iopub.status.idle":"2023-01-21T02:17:58.295626Z","shell.execute_reply.started":"2023-01-21T02:17:57.293202Z","shell.execute_reply":"2023-01-21T02:17:58.29459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Preparation\n\n- I use the same strategy as in https://www.kaggle.com/code/theoviel/dicom-resized-png-jpg","metadata":{}},{"cell_type":"code","source":"test_images = glob.glob(\"/kaggle/input/rsna-breast-cancer-detection/test_images/*/*.dcm\")\n\nif DEBUG:\n    test_images = glob.glob(\"/kaggle/input/rsna-breast-cancer-detection/train_images/10042/*.dcm\")\n    \nprint(\"Number of images :\", len(test_images))","metadata":{"execution":{"iopub.status.busy":"2023-01-21T02:17:58.296912Z","iopub.execute_input":"2023-01-21T02:17:58.29766Z","iopub.status.idle":"2023-01-21T02:17:58.322869Z","shell.execute_reply.started":"2023-01-21T02:17:58.297622Z","shell.execute_reply":"2023-01-21T02:17:58.321785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAVE_FOLDER = \"/kaggle/tmp/output/\"\nSIZE = (1024,512)\nEXTENSION = \"png\"\n\nos.makedirs(SAVE_FOLDER, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-21T02:17:58.324277Z","iopub.execute_input":"2023-01-21T02:17:58.32491Z","iopub.status.idle":"2023-01-21T02:17:58.330493Z","shell.execute_reply.started":"2023-01-21T02:17:58.324869Z","shell.execute_reply":"2023-01-21T02:17:58.329338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(f, size=512, save_folder=\"\", extension=\"png\"):\n    patient = f.split('/')[-2]\n    image = f.split('/')[-1][:-4]\n\n    dicom = pydicom.dcmread(f)\n    img = dicom.pixel_array\n\n    img = (img - img.min()) / (img.max() - img.min())\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n\n    img = cv2.resize(img, (size[0], size[1]))\n    \n    path = save_folder + f\"{patient}\"\n    os.makedirs(path, exist_ok=True)\n    cv2.imwrite(path + f\"/{image}.{extension}\", (img * 255).astype(np.uint8))","metadata":{"execution":{"iopub.status.busy":"2023-01-21T02:17:58.331982Z","iopub.execute_input":"2023-01-21T02:17:58.333111Z","iopub.status.idle":"2023-01-21T02:17:58.342992Z","shell.execute_reply.started":"2023-01-21T02:17:58.333071Z","shell.execute_reply":"2023-01-21T02:17:58.342054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = Parallel(n_jobs=4)(\n    delayed(process)(uid, size=SIZE, save_folder=SAVE_FOLDER, extension=EXTENSION)\n    for uid in tqdm(test_images)\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-21T02:17:58.346576Z","iopub.execute_input":"2023-01-21T02:17:58.346963Z","iopub.status.idle":"2023-01-21T02:18:02.33373Z","shell.execute_reply.started":"2023-01-21T02:17:58.34693Z","shell.execute_reply":"2023-01-21T02:18:02.332461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls /kaggle/tmp/output","metadata":{"execution":{"iopub.status.busy":"2023-01-21T02:18:02.335141Z","iopub.execute_input":"2023-01-21T02:18:02.335511Z","iopub.status.idle":"2023-01-21T02:18:03.353904Z","shell.execute_reply.started":"2023-01-21T02:18:02.33547Z","shell.execute_reply":"2023-01-21T02:18:03.352725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mkdir /kaggle/tmp/test","metadata":{"execution":{"iopub.status.busy":"2023-01-21T02:18:03.356041Z","iopub.execute_input":"2023-01-21T02:18:03.356455Z","iopub.status.idle":"2023-01-21T02:18:04.316662Z","shell.execute_reply.started":"2023-01-21T02:18:03.356405Z","shell.execute_reply":"2023-01-21T02:18:04.315252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cp /kaggle/input/fastai/fastai/models -r /kaggle/tmp/models","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd /kaggle/input/fastai/fastai","metadata":{"execution":{"iopub.status.busy":"2023-01-21T02:35:47.367294Z","iopub.execute_input":"2023-01-21T02:35:47.367703Z","iopub.status.idle":"2023-01-21T02:35:47.375885Z","shell.execute_reply.started":"2023-01-21T02:35:47.367666Z","shell.execute_reply":"2023-01-21T02:35:47.374659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python infer.py \\\n    --model /kaggle/tmp/models \\\n    --data /kaggle/tmp/output \\\n    --csv /kaggle/tmp/test/submission.csv \\\n    --threshold 0.25 \\\n    --split 4 \\\n    --stt 0 \\\n    --test_csv /kaggle/input/rsna-breast-cancer-detection/test.csv","metadata":{"execution":{"iopub.status.busy":"2023-01-21T02:37:33.402593Z","iopub.execute_input":"2023-01-21T02:37:33.40306Z","iopub.status.idle":"2023-01-21T02:37:44.739375Z","shell.execute_reply.started":"2023-01-21T02:37:33.403Z","shell.execute_reply":"2023-01-21T02:37:44.738028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd /kaggle/tmp/test/","metadata":{"execution":{"iopub.status.busy":"2023-01-21T02:38:26.02635Z","iopub.execute_input":"2023-01-21T02:38:26.027205Z","iopub.status.idle":"2023-01-21T02:38:26.046605Z","shell.execute_reply.started":"2023-01-21T02:38:26.027154Z","shell.execute_reply":"2023-01-21T02:38:26.045455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\ndf_1 = pd.read_csv('submission.csv')\n# df_2 = pd.read_csv('submission_2.csv')\n\n# cancer_1 = np.array(list(df_1.cancer))\n# cancer_2 = np.array(list(df_2.cancer))\n# cancer = (cancer_1 + cancer_2) / 2\n# arr = cancer_1, cancer_2\n# a = list(zip(*arr[::-1]))\n# cancer = list(map(max, a)) \n# df_2.cancer = cancer\n\nsub = df_1[['prediction_id', 'cancer']].groupby('prediction_id').max().reset_index()\nsub.to_csv('/kaggle/working/submission.csv',index=False)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-21T02:38:30.885566Z","iopub.execute_input":"2023-01-21T02:38:30.886014Z","iopub.status.idle":"2023-01-21T02:38:30.937699Z","shell.execute_reply.started":"2023-01-21T02:38:30.885974Z","shell.execute_reply":"2023-01-21T02:38:30.936749Z"},"trusted":true},"execution_count":null,"outputs":[]}]}