{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip download transformers datasets accelerate evaluate albumentations swifter mapply pylibjpeg pylibjpeg ","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:17:00.859528Z","iopub.execute_input":"2023-01-12T08:17:00.860502Z","iopub.status.idle":"2023-01-12T08:20:19.33179Z","shell.execute_reply.started":"2023-01-12T08:17:00.860376Z","shell.execute_reply":"2023-01-12T08:20:19.329525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install transformers","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:20:19.33603Z","iopub.execute_input":"2023-01-12T08:20:19.337787Z","iopub.status.idle":"2023-01-12T08:20:33.776857Z","shell.execute_reply.started":"2023-01-12T08:20:19.337733Z","shell.execute_reply":"2023-01-12T08:20:33.775426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import (\n    AutoFeatureExtractor,\n    AutoModelForImageClassification,\n)\nmodel_name_or_path = \"google/vit-base-patch16-384\"\nfeature_extractor = AutoFeatureExtractor.from_pretrained(model_name_or_path)\n\nlabels = [\"no_cancer\",\"cancer\"]\n\nmodel = AutoModelForImageClassification.from_pretrained(\n    model_name_or_path,\n    num_labels=len(labels),\n    id2label={str(i): c for i, c in enumerate(labels)},\n    label2id={c: str(i) for i, c in enumerate(labels)},\n    ignore_mismatched_sizes=True\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:20:33.778663Z","iopub.execute_input":"2023-01-12T08:20:33.779172Z","iopub.status.idle":"2023-01-12T08:21:16.426432Z","shell.execute_reply.started":"2023-01-12T08:20:33.779097Z","shell.execute_reply":"2023-01-12T08:21:16.42502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ~/.cache/huggingface/transformers\n!mkdir -p /kaggle/working/models\n!cp -r ~/.cache/huggingface/transformers /kaggle/working/models","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:21:16.43Z","iopub.execute_input":"2023-01-12T08:21:16.430408Z","iopub.status.idle":"2023-01-12T08:21:20.277015Z","shell.execute_reply.started":"2023-01-12T08:21:16.430367Z","shell.execute_reply":"2023-01-12T08:21:20.27543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p /kaggle/working/models","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:21:20.279212Z","iopub.execute_input":"2023-01-12T08:21:20.279593Z","iopub.status.idle":"2023-01-12T08:21:21.396481Z","shell.execute_reply.started":"2023-01-12T08:21:20.279561Z","shell.execute_reply":"2023-01-12T08:21:21.394945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"  !pip install /kaggle/input/rsna-2022-whl/{pydicom-2.3.0-py3-none-any.whl,pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:21:21.398483Z","iopub.execute_input":"2023-01-12T08:21:21.399019Z","iopub.status.idle":"2023-01-12T08:21:37.354199Z","shell.execute_reply.started":"2023-01-12T08:21:21.398973Z","shell.execute_reply":"2023-01-12T08:21:37.352282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport random\nimport shutil\n\nrandom.seed(1337)\n\nimport ssl\n\nimport numpy as np\nimport pandas as pd\nimport cv2\nfrom PIL import Image, ImageDraw, ImageFont\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nssl._create_default_https_context = ssl._create_stdlib_context\n\nimport torch\nimport albumentations as A\nfrom datasets import load_dataset, load_metric\nfrom transformers import (\n    Trainer,\n    TrainingArguments,\n    AutoFeatureExtractor,\n    AutoModelForImageClassification,\n)\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import class_weight\n\nfrom skimage.transform import resize\n\nfrom typing import List, Tuple\nimport csv\n\nfrom pydicom import dcmread\nfrom PIL import Image, UnidentifiedImageError\nimport torchvision.transforms as transforms\n\nfrom tqdm.notebook import tqdm\nfrom torch import nn\nfrom transformers import Trainer\n\ntqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:21:37.356378Z","iopub.execute_input":"2023-01-12T08:21:37.357578Z","iopub.status.idle":"2023-01-12T08:21:45.896283Z","shell.execute_reply.started":"2023-01-12T08:21:37.357529Z","shell.execute_reply":"2023-01-12T08:21:45.894674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:21:45.897818Z","iopub.execute_input":"2023-01-12T08:21:45.900148Z","iopub.status.idle":"2023-01-12T08:21:46.044802Z","shell.execute_reply.started":"2023-01-12T08:21:45.900092Z","shell.execute_reply":"2023-01-12T08:21:46.043572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:21:46.047598Z","iopub.execute_input":"2023-01-12T08:21:46.048039Z","iopub.status.idle":"2023-01-12T08:21:46.080474Z","shell.execute_reply.started":"2023-01-12T08:21:46.047999Z","shell.execute_reply":"2023-01-12T08:21:46.079112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cancer_true = df[df[\"cancer\"]==1]\ncancer_false = df[df[\"cancer\"]==0]\nprint(len(cancer_true),len(cancer_false))\ncancer_false = cancer_false[cancer_false[\"difficult_negative_case\"]==False]\n \nprint(len(cancer_true),len(cancer_false))","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:21:46.082574Z","iopub.execute_input":"2023-01-12T08:21:46.083453Z","iopub.status.idle":"2023-01-12T08:21:46.116908Z","shell.execute_reply.started":"2023-01-12T08:21:46.083397Z","shell.execute_reply":"2023-01-12T08:21:46.115812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_cancer_true = list(cancer_true[\"image_id\"].unique())\nlist_cancer_false = list(cancer_false[\"image_id\"].unique())\nprint(set(list_cancer_true) & set(list_cancer_false))","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:21:46.11841Z","iopub.execute_input":"2023-01-12T08:21:46.11972Z","iopub.status.idle":"2023-01-12T08:21:46.14806Z","shell.execute_reply.started":"2023-01-12T08:21:46.119653Z","shell.execute_reply":"2023-01-12T08:21:46.146541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(list_cancer_true),len(list_cancer_false))","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:21:46.150518Z","iopub.execute_input":"2023-01-12T08:21:46.151225Z","iopub.status.idle":"2023-01-12T08:21:46.158467Z","shell.execute_reply.started":"2023-01-12T08:21:46.151158Z","shell.execute_reply":"2023-01-12T08:21:46.157049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random.shuffle(list_cancer_true)\nrandom.shuffle(list_cancer_false)","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:21:46.162641Z","iopub.execute_input":"2023-01-12T08:21:46.163485Z","iopub.status.idle":"2023-01-12T08:21:46.223109Z","shell.execute_reply.started":"2023-01-12T08:21:46.163441Z","shell.execute_reply":"2023-01-12T08:21:46.221905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_cancer_false = list_cancer_false[:int(len(list_cancer_false)*0.05)]\nprint(len(list_cancer_true),len(list_cancer_false))\nsplit_frac = 0.5\ntrain_ids = list_cancer_true[:int(len(list_cancer_true)*split_frac)] + list_cancer_false[:int(len(list_cancer_false)*split_frac)]\nval_ids = list_cancer_true[int(len(list_cancer_true)*split_frac):] + list_cancer_false[int(len(list_cancer_false)*split_frac):]\nprint(len(train_ids),len(val_ids))","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:21:46.226328Z","iopub.execute_input":"2023-01-12T08:21:46.226911Z","iopub.status.idle":"2023-01-12T08:21:46.239279Z","shell.execute_reply.started":"2023-01-12T08:21:46.226858Z","shell.execute_reply":"2023-01-12T08:21:46.238137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom multiprocessing import Pool\n\ntry:\n    os.makedirs(\"/kaggle/working/data/\")\nexcept:\n    pass\ntry:\n    os.makedirs(\"/kaggle/working/data/train/\")\nexcept:\n    pass\n\ndef load_images(df_item):\n    import os\n    import numpy as np\n    import pydicom\n    from PIL import Image, UnidentifiedImageError\n    import cv2\n    i_row, item = df_item\n    try:\n        new_path = f'{item[\"output_folder_path\"]}{item[\"patient_id\"]}_{item[\"image_id\"]}.jpg'\n        if not os.path.exists(new_path):\n            path = f'{item[\"input_folder_path\"]}{item[\"patient_id\"]}/{item[\"image_id\"]}.dcm'\n            if not os.path.exists(path):\n                return\n            dicom = pydicom.read_file(path)\n            data = dicom.pixel_array\n            if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n                data = np.amax(data) - data\n            data = data - np.min(data)\n            data = data / np.max(data)\n            data = (data*255).astype(np.uint8)\n\n            rgba = Image.fromarray(data).convert(\"RGB\")\n            img = np.array(rgba)\n            og_width, og_height = rgba.size\n\n            width, height = rgba.size\n            border={\n                \"left\":0,\n                \"right\":width,\n                \"top\":0,\n                \"bottom\":height\n            }\n            ratio = 5\n            bg_color = [0,0,0]\n\n            any_break = False\n            for j in range(10,width,ratio):\n                break_true = False\n                for i in range(0, height,ratio):\n                    r,g,b = img[i,j]\n                    if r != bg_color[0] and g != bg_color[1] and b != bg_color[2]:\n                        break_true=True\n                        any_break=True\n                        break\n                if any_break and not break_true:\n                    border[\"right\"]=j+1\n                    break\n                if not any_break:\n                    break\n\n            any_break = False\n            for j in range(width-11,-1,ratio*-1):\n                break_true = False\n                for i in range(0, height,ratio):\n                    r,g,b = img[i,j]\n                    if r != bg_color[0] and g != bg_color[1] and b != bg_color[2]:\n                        break_true=True\n                        any_break=True\n                        break\n                if any_break and not break_true:\n                    border[\"left\"]=j-1\n                    break\n                if not any_break:\n                    break\n                \n            if border[\"right\"]<=border[\"left\"]:\n                border[\"left\"] = 0\n                border[\"right\"] = width\n            clip_img = img[border[\"top\"]:border[\"bottom\"],border[\"left\"]:border[\"right\"]]\n            Image.fromarray(clip_img).resize((384,384)).save(new_path)\n#         print(f\"Done with {i_row}\")\n    except (UnidentifiedImageError, OSError) as e:\n        print(e)","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:21:46.240781Z","iopub.execute_input":"2023-01-12T08:21:46.241454Z","iopub.status.idle":"2023-01-12T08:21:46.265002Z","shell.execute_reply.started":"2023-01-12T08:21:46.241412Z","shell.execute_reply":"2023-01-12T08:21:46.263956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# just_ids = df[(df[\"image_id\"].isin(train_ids)) | (df[\"image_id\"].isin(val_ids))]\ndf[\"input_folder_path\"] = \"/kaggle/input/rsna-breast-cancer-detection/train_images/\"\ndf[\"output_folder_path\"] = \"/kaggle/working/data/train/\"\nwith Pool(os.cpu_count()-1) as p:\n  r = list(tqdm(p.imap(load_images, df.iterrows()), total=len(just_ids)))","metadata":{"execution":{"iopub.status.busy":"2023-01-12T08:21:46.266243Z","iopub.execute_input":"2023-01-12T08:21:46.267002Z","iopub.status.idle":"2023-01-12T09:04:11.89966Z","shell.execute_reply.started":"2023-01-12T08:21:46.266963Z","shell.execute_reply":"2023-01-12T09:04:11.89728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}