{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Introduction\n\nIn The Mythical Man Month, Fred Brooks, the father of the IBM 360, opined \"you should always \"plan to throw one away ... you will anyway\"\n\nThis notebook:\n\n* Is derived from [my initial competion entry](https://www.kaggle.com/code/julianmacnamara/rsna-screening-mammography-breast-cancer-detection)\n* Uses:\n  * fp16\n  * The [RSNA Breast Cancer Detection - 512x512 pngs](https://www.kaggle.com/datasets/theoviel/rsna-breast-cancer-512-pngs) dataset which was created by Theo Viel\n  * A batch size of 32","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport timm\n\nimport cv2\n\nimport shutil\nimport random\nimport warnings\n\nwarnings.filterwarnings(\"ignore\", category=UserWarning) \n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nfrom pathlib import Path\nimport glob\nimport random\n\nfrom fastai.data.all import *\nfrom fastai.vision.all import *\nfrom fastai.callback.fp16 import *","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:15:57.362685Z","iopub.execute_input":"2023-02-12T16:15:57.36346Z","iopub.status.idle":"2023-02-12T16:15:57.376664Z","shell.execute_reply.started":"2023-02-12T16:15:57.363402Z","shell.execute_reply":"2023-02-12T16:15:57.375673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# timm.list_models('convnext*')","metadata":{"execution":{"iopub.status.busy":"2023-02-12T15:32:34.71589Z","iopub.execute_input":"2023-02-12T15:32:34.716587Z","iopub.status.idle":"2023-02-12T15:32:34.727858Z","shell.execute_reply.started":"2023-02-12T15:32:34.716547Z","shell.execute_reply":"2023-02-12T15:32:34.7267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Source: https://www.kaggle.com/datasets/almateya/offline-resnet50\n# Load dataset offline-resnet50\n\n# !mkdir -p /root/.cache/torch/hub/checkpoints\n\n# src = '/kaggle/input/offline-resnet50/resnet50'   #'../input/offline-resnet50/resnet50'\n# dst = '/root/.cache/torch/hub/checkpoints/resnet50-0676ba61.pth'\n\n# shutil.copy(src, dst)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T10:14:33.815757Z","iopub.execute_input":"2023-02-12T10:14:33.816101Z","iopub.status.idle":"2023-02-12T10:14:33.826728Z","shell.execute_reply.started":"2023-02-12T10:14:33.81606Z","shell.execute_reply":"2023-02-12T10:14:33.825598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p /root/.cache/torch/hub/checkpoints\n\nsrc = '/kaggle/input/convnext/convnext_small_22k_224.pth'\ndst = '/root/.cache/torch/hub/checkpoints/convnext_small_22k_224.pth'\n\nshutil.copy(src, dst)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:16:07.230514Z","iopub.execute_input":"2023-02-12T16:16:07.231061Z","iopub.status.idle":"2023-02-12T16:16:09.357287Z","shell.execute_reply.started":"2023-02-12T16:16:07.23101Z","shell.execute_reply":"2023-02-12T16:16:09.355938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# /kaggle/input/convnext-weights/convnext/convnext_tiny_1k_224_ema.pth\n#if not os.path.exists('/root/.cache/torch/hub/checkpoints/'):\n#       os.makedirs('/root/.cache/torch/hub/checkpoints/')\n#!cp '/kaggle/input/convnext-weights/convnext/convnext_tiny_1k_224_ema.pth' '/root/.cache/torch/hub/checkpoints/convnext_tiny_1k_224_ema.pth'","metadata":{"execution":{"iopub.status.busy":"2023-02-12T10:14:35.25462Z","iopub.status.idle":"2023-02-12T10:14:35.255026Z","shell.execute_reply.started":"2023-02-12T10:14:35.254845Z","shell.execute_reply":"2023-02-12T10:14:35.254866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path='/kaggle/input/rsna-difficult-images-20230206/images_difficult_cases/'","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:16:14.080719Z","iopub.execute_input":"2023-02-12T16:16:14.081195Z","iopub.status.idle":"2023-02-12T16:16:14.090203Z","shell.execute_reply.started":"2023-02-12T16:16:14.081146Z","shell.execute_reply":"2023-02-12T16:16:14.088783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking image file\n# /kaggle/input/rsna-sample-images-512-pngs/sample_images_512_pngs/5_640805896.png\n# /kaggle/input/rsna-revised-images-512-pngs/revised_512_pngs/10130_1360338805_cnt.png\n# 53158_1629601463.png\n\nimg = PILImage.create(path+'18283_1177095033.png')\nimg.to_thumb(128)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:16:20.489721Z","iopub.execute_input":"2023-02-12T16:16:20.490179Z","iopub.status.idle":"2023-02-12T16:16:20.516065Z","shell.execute_reply.started":"2023-02-12T16:16:20.490139Z","shell.execute_reply":"2023-02-12T16:16:20.51512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_csv = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ntest_csv","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:16:24.376974Z","iopub.execute_input":"2023-02-12T16:16:24.377425Z","iopub.status.idle":"2023-02-12T16:16:24.40448Z","shell.execute_reply.started":"2023-02-12T16:16:24.377386Z","shell.execute_reply":"2023-02-12T16:16:24.403555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### As noted in [A Brief Intro to Mammography](https://www.kaggle.com/competitions/rsna-breast-cancer-detection/discussion/369262) \n\n* For this task, we are focused on screening mammograms, which again means that only scores of BI-RADS 0, 1, or 2 are possible. One can essentially think of 0 as \"abnormal\" and 1 and 2 as \"normal.\" There can be subjectivity in assigning BI-RADS 1 or 2. For example, if there are stable findings that are almost certainly benign breast cysts, the mammogram may be assigned 1 or 2 depending on the radiologist.\n\n* Our task in the challenge is to predict cancer or no cancer, a binary value. This may be obvious, but not all BI-RADS 0 cases have cancer. The final cancer label will depend on the outcome of the diagnostic mammogram and the biopsy results, if obtained.\n\nSince the test set only includes MLO and CC views only rows that have these views are included\n\n[revised_01_20230206.csv](/kaggle/input/rsna-csv-files/revised_01_20230206.csv) contains:\n\n* 6,946 original and augmented images where cancer was diagnosed regardless of their BIRADS score<br>\nMammograms diagnosed with cancer were augmented using [this notebook](https://www.kaggle.com/code/julianmacnamara/rsna-augmentation). Essentially each image was flipped, inverted, mirrored, autocontrasted with a cutoff of 0.5 and equalised\n\n* 18,021 images where the BIRADS score was either 1 or 2 but cancer had not been diagnosed\n\nbut this scored poorly on submission\n\n[revised_02_20230206](/kaggle/input/rsna-csv-files/revised_01_20230206.csv) contains:\n\n* The same 6,946 original and augmented images where cancer was diagnosed<br>\n\n* 7,697 images where \"difficult_negative_case\" was true\n\n#### 7. February\n\n* Ran with randomized version of revised_02_20230206.csv. Have yet to submit\n* Changed resnet18 to resnet50. Submission score of 0.02 suggests over-fitting\n\n#### 12 February\n\n* Changed architecture to ConvNeXt\n* Created revised_03_20230212.csv which is the same as revised_02_20230206 but with an additional 8,000 records where \"difficult_negative_case\" was false\n\n","metadata":{}},{"cell_type":"code","source":"train_csv = pd.read_csv('/kaggle/input/rsna-csv-files/revised_03_20230212.csv')\ntrain_csv['cancer'] = train_csv['cancer'].astype(str)\n\ntrain_csv.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:16:29.190333Z","iopub.execute_input":"2023-02-12T16:16:29.190836Z","iopub.status.idle":"2023-02-12T16:16:29.497245Z","shell.execute_reply.started":"2023-02-12T16:16:29.190795Z","shell.execute_reply":"2023-02-12T16:16:29.496252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df=train_csv[train_csv['BIRADS'].notna()\n#df=train_csv\ndf=train_csv[[\"img_path\",\"cancer\"]]\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:16:39.759809Z","iopub.execute_input":"2023-02-12T16:16:39.760272Z","iopub.status.idle":"2023-02-12T16:16:39.784937Z","shell.execute_reply.started":"2023-02-12T16:16:39.760233Z","shell.execute_reply":"2023-02-12T16:16:39.783986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,6))\nplt.subplot(1,2,1)\nax1 = sns.countplot(data=df, x='cancer')\nplt.xticks(rotation = -45)\nfor container in ax1.containers:\n    ax1.bar_label(container)\nplt.title('Labels');","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:16:44.439021Z","iopub.execute_input":"2023-02-12T16:16:44.43946Z","iopub.status.idle":"2023-02-12T16:16:44.863352Z","shell.execute_reply.started":"2023-02-12T16:16:44.43942Z","shell.execute_reply":"2023-02-12T16:16:44.862457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating the learner with a batch size of 32 and training","metadata":{}},{"cell_type":"code","source":"def get_x(r): return r['img_path']\ndef get_y(r): return r['cancer']","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:16:50.195404Z","iopub.execute_input":"2023-02-12T16:16:50.195932Z","iopub.status.idle":"2023-02-12T16:16:50.202024Z","shell.execute_reply.started":"2023-02-12T16:16:50.195892Z","shell.execute_reply":"2023-02-12T16:16:50.200851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dblock = DataBlock(blocks=(ImageBlock, CategoryBlock),\n                   get_x = get_x,\n                   get_y = get_y,\n                   splitter=RandomSplitter (valid_pct=0.2),                   \n                   item_tfms=Resize(512))\n\ndsets = dblock.datasets(df)\ndsets.train[0]","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:17:02.924097Z","iopub.execute_input":"2023-02-12T16:17:02.924611Z","iopub.status.idle":"2023-02-12T16:17:07.938283Z","shell.execute_reply.started":"2023-02-12T16:17:02.924568Z","shell.execute_reply":"2023-02-12T16:17:07.937284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = dblock.dataloaders(df, bs=32)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:17:12.776887Z","iopub.execute_input":"2023-02-12T16:17:12.777338Z","iopub.status.idle":"2023-02-12T16:17:12.941021Z","shell.execute_reply.started":"2023-02-12T16:17:12.777298Z","shell.execute_reply":"2023-02-12T16:17:12.939844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#/kaggle/input/resnet18/resnet18.pth\nif not os.path.exists('/root/.cache/torch/hub/checkpoints/'):\n        os.makedirs('/root/.cache/torch/hub/checkpoints/')\n!cp '/kaggle/input/resnet18/resnet18.pth' '/root/.cache/torch/hub/checkpoints/resnet18-f37072fd.pth'","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:17:18.542051Z","iopub.execute_input":"2023-02-12T16:17:18.542614Z","iopub.status.idle":"2023-02-12T16:17:20.281982Z","shell.execute_reply.started":"2023-02-12T16:17:18.542568Z","shell.execute_reply":"2023-02-12T16:17:20.280453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#learn = vision_learner(dls, models.resnet50, metrics=accuracy, pretrained=True).to_fp16()\n#learn = vision_learner(dls, models.resnet18, metrics=accuracy, pretrained=True).to_fp16()\n# /kaggle/input/convnext-weights/convnext/convnext_tiny_1k_224_ema.pth\nlearn = vision_learner(dls, 'convnext_small_in22k', metrics=accuracy, pretrained=True).to_fp16()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:17:36.208108Z","iopub.execute_input":"2023-02-12T16:17:36.208681Z","iopub.status.idle":"2023-02-12T16:17:38.444577Z","shell.execute_reply.started":"2023-02-12T16:17:36.208626Z","shell.execute_reply":"2023-02-12T16:17:38.443431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lrs = learn.lr_find()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:17:43.519706Z","iopub.execute_input":"2023-02-12T16:17:43.520209Z","iopub.status.idle":"2023-02-12T16:19:35.661096Z","shell.execute_reply.started":"2023-02-12T16:17:43.520166Z","shell.execute_reply":"2023-02-12T16:19:35.659889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fit_one_cycle(10, 0.01)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T16:20:11.45394Z","iopub.execute_input":"2023-02-12T16:20:11.454411Z","iopub.status.idle":"2023-02-12T17:23:19.082074Z","shell.execute_reply.started":"2023-02-12T16:20:11.454365Z","shell.execute_reply":"2023-02-12T17:23:19.080899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interp = ClassificationInterpretation.from_learner(learn)\ninterp.plot_confusion_matrix()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T17:23:32.372725Z","iopub.execute_input":"2023-02-12T17:23:32.373214Z","iopub.status.idle":"2023-02-12T17:26:23.695858Z","shell.execute_reply.started":"2023-02-12T17:23:32.373164Z","shell.execute_reply":"2023-02-12T17:26:23.694186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction and submission","metadata":{}},{"cell_type":"code","source":"path='/kaggle/input/rsna-test-images-as-pngs/'","metadata":{"execution":{"iopub.status.busy":"2023-02-12T17:28:32.868332Z","iopub.execute_input":"2023-02-12T17:28:32.868979Z","iopub.status.idle":"2023-02-12T17:28:32.878319Z","shell.execute_reply.started":"2023-02-12T17:28:32.868919Z","shell.execute_reply":"2023-02-12T17:28:32.876716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Source: https://www.kaggle.com/code/beezus666/dicom-to-png-to-predict\n\n# Get images\ntest_images = sorted([os.path.join(path, file) for file in os.listdir(path)])\nimages = [cv2.imread(file) for file in test_images]\n\n# pass images to fast.ai learner and get predictions\ntest_dl = learn.dls.test_dl(images)\npreds_batch, _ = learn.get_preds(dl=test_dl)\npredsdec, _, decoded = learn.get_preds(dl=test_dl, with_decoded=True)\npredsdec[:10]","metadata":{"execution":{"iopub.status.busy":"2023-02-12T17:28:40.854583Z","iopub.execute_input":"2023-02-12T17:28:40.855159Z","iopub.status.idle":"2023-02-12T17:28:41.864011Z","shell.execute_reply.started":"2023-02-12T17:28:40.855118Z","shell.execute_reply":"2023-02-12T17:28:41.862748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extract the prediction, which is the 2nd value in the tensor above\n\narray = preds_batch.numpy()\nlist_of_lists = array.tolist()\nsecond_values = [lst[1] for lst in list_of_lists]\n\nsorted_files = sorted(os.listdir(path))\nno_extensions = [os.path.splitext(name)[0] for name in sorted_files]\n\npreds_df = pd.DataFrame(data = {'concat_id':no_extensions, 'cancer':second_values})\n\ninfo_df=pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ninfo_df['concat_id'] = info_df['image_id'].astype(str)\ninfo_df=info_df[[\"patient_id\",\"image_id\",\"laterality\",\"concat_id\"]]\n\nmerged_df = pd.merge(right=info_df, left=preds_df, on='concat_id')\nmerged_df['prediction_id'] = merged_df['patient_id'].astype(str)+'_'+merged_df['laterality'].astype(str)\n\n# Select only the 'prediction_id' and 'cancer' columns\nresulting_df = merged_df[['prediction_id', 'cancer']]\nresulting_df = resulting_df.groupby('prediction_id', as_index=False).mean()\nresulting_df = resulting_df.sort_index()\n\ni=resulting_df['cancer'][0]\nj=resulting_df['cancer'][1]\n\nfinal_df=pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')\nfinal_df.loc[0:,\"cancer\"]=i\nfinal_df.loc[1:,\"cancer\"]=j\nfinal_df","metadata":{"execution":{"iopub.status.busy":"2023-02-12T17:28:48.745621Z","iopub.execute_input":"2023-02-12T17:28:48.746115Z","iopub.status.idle":"2023-02-12T17:28:48.814521Z","shell.execute_reply.started":"2023-02-12T17:28:48.746064Z","shell.execute_reply":"2023-02-12T17:28:48.813567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shutil.rmtree(\"/kaggle/working/models\")","metadata":{"execution":{"iopub.status.busy":"2023-02-12T17:28:57.700268Z","iopub.execute_input":"2023-02-12T17:28:57.700771Z","iopub.status.idle":"2023-02-12T17:28:57.731543Z","shell.execute_reply.started":"2023-02-12T17:28:57.700727Z","shell.execute_reply":"2023-02-12T17:28:57.730574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T17:29:08.030525Z","iopub.execute_input":"2023-02-12T17:29:08.031019Z","iopub.status.idle":"2023-02-12T17:29:08.043231Z","shell.execute_reply.started":"2023-02-12T17:29:08.030979Z","shell.execute_reply":"2023-02-12T17:29:08.041788Z"},"trusted":true},"execution_count":null,"outputs":[]}]}