{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"\n!pip install pydicom\n!pip install kornia\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!pip install torch torchvision","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!nvidia-smi","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.cuda.is_available()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install torch torchvision feather-format kornia pyarrow --upgrade   \n!pip install git+https://github.com/fastai/fastai                    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"","metadata":{}},{"cell_type":"code","source":"from fastai.basics           import *\nfrom fastai.vision.all       import *\nfrom fastai.medical.imaging  import *\nfrom fastai.callback.tracker import *\nfrom fastai.callback.all     import *\n\n\n\nnp.set_printoptions(linewidth=120)\nmatplotlib.rcParams['image.cmap'] = 'bone'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"First we read in the metadata files (linked in the introduction).","metadata":{}},{"cell_type":"code","source":"\"\"\"path = Path('../input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/')\npath_trn_img = path/'stage_2_train'\npath_tst_img = path/'stage_2_test'\n\"\"\"\npath_inp = Path('../input')\npath_xtra = path_inp/'rsna-hemorrhage-jpg'\npath_meta = path_xtra/'meta'/'meta'\n\npath_img_jpg = path_xtra/'train_jpg'/'train_jpg'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_comb = pd.read_feather(path_meta/'comb.fth').set_index('SOPInstanceUID')\n\ndf_tst  = pd.read_feather(path_meta/'df_tst.fth').set_index('SOPInstanceUID')\ndf_samp = pd.read_feather(path_meta/'wgt_sample.fth').set_index('SOPInstanceUID')\nwith open (path_meta/'bins.pkl','rb') as f:\n  bins = pickle.load(f)\npd.set_option('max_columns',None)\n#df_comb.head()\n\n        ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train vs valid","metadata":{}},{"cell_type":"markdown","source":"To get better validation measures, we should split on patients, not just on studies, since that's how the test set is created.\n\nHere's a list of random patients:","metadata":{}},{"cell_type":"code","source":"set_seed(42)\npatients = df_comb.PatientID.unique()\npat_mask = np.random.random(len(patients))<0.8\npat_trn = patients[pat_mask]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#len(df_samp)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can use that to take just the patients in a dataframe that match that mask:","metadata":{}},{"cell_type":"code","source":"def split_data(df):\n    idx = L.range(df)\n    mask = df.PatientID.isin(pat_trn)\n    return idx[mask],idx[~mask]\n\nsplits = split_data(df_samp)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"splits[1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's double-check that for a patient in the training set that their images are all in the first split.","metadata":{}},{"cell_type":"code","source":"df_trn = df_samp.iloc[splits[0]]\np1 = L.range(df_samp)[df_samp.PatientID==df_trn.PatientID[0]]\nassert len(p1) == len(set(p1) & set(splits[0]))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_validate=df_samp.iloc[splits[1]]\ndf_validate.head()\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare sample DataBunch","metadata":{}},{"cell_type":"markdown","source":"We will grab our sample filenames for the initial pretraining.","metadata":{}},{"cell_type":"code","source":"def filename(o): return os.path.splitext(os.path.basename(o))[0]\n\nfns = L(list(df_samp.fname)).map(filename)\n#fn = fns[0]\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We need to create a `DataBunch` that contains our sample data, so we need a function to convert a filename (pointing at a DICOM file) into a path to our sample JPEG files:","metadata":{}},{"cell_type":"code","source":"path_test=Path('../input/test-sample')\n\ndef fn2test_im(fn):\n    return PILCTScan.create((path_test/fn).with_suffix('.jpg'))\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fn2image(fn): return PILCTScan.create((path_img_jpg/fn).with_suffix('.jpg'))\n#fn2image(fn).show();","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We also need to be able to grab the labels from this, which we can do by simply indexing into our sample `DataFrame`.","metadata":{}},{"cell_type":"code","source":"htypes = ['any','epidural','intraparenchymal','intraventricular','subarachnoid','subdural']\ndef fn2label(fn): return df_comb.loc[fn][htypes].values.astype(np.float32)\n#fn2label(fn)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"choose and change batchsize and number-of-workers here:\n","metadata":{}},{"cell_type":"code","source":"bs,nw = 64,4","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import torch\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#torch.cuda.device_count()\n#torch.cuda.is_available()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We're going to use fastai's new [Transform Pipeline API](http://dev.fast.ai/pets.tutorial.html) to create the DataBunch, since this is extremely flexible. We create two transform pipelines, one to open the image file, and one to look up the label and create a tensor of categories.","metadata":{}},{"cell_type":"code","source":"tfms = [[fn2image], [fn2label,EncodedMultiCategorize(htypes)]]\ndsrc = Datasets(fns, tfms,splits=splits)\nnrm = Normalize(tensor([0.6]),tensor([0.25]))\naug = aug_transforms(p_lighting=0.)\nbatch_tfms = [IntToFloatTensor(), nrm, *aug]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To support progressive resizing  we create a function that returns a dataset resized to a requested size:","metadata":{}},{"cell_type":"code","source":"def get_data(bs, sz):return dsrc.dataloaders(bs=bs, num_workers=nw,after_item=[ToTensor],after_batch=batch_tfms+[AffineCoordTfm(size=sz)])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's try it out!","metadata":{}},{"cell_type":"code","source":"dbch = get_data(64, 224)\n\"\"\"xb,yb = to_cpu(dbch.one_batch())\ndbch.show_batch(max_n=4, figsize=(9,6))\nxb.shape\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(yb)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's track the accuracy of the *any* label as our main metric, since it's easy to interpret.","metadata":{}},{"cell_type":"code","source":"def accuracy_any(inp, targ, thresh=0.5, sigmoid=True):\n    inp,targ = flatten_check(inp[:,0],targ[:,0])\n    if sigmoid: inp = inp.sigmoid()\n    return ((inp>thresh)==targ.bool()).float().mean()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n","metadata":{}},{"cell_type":"code","source":"def get_loss(scale=1.0):\n    loss_weights = tensor(2.0, 1, 1, 1, 1, 1).cuda()*scale\n    return BaseLoss(nn.BCEWithLogitsLoss, pos_weight=loss_weights, floatify=True, flatten=False, \n        is_2d=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We'll scale the loss initially to account for our sampling (since the original data had 14% rows with a positive label, and we resampled it to 50/50).","metadata":{}},{"cell_type":"code","source":"loss_func = get_loss(0.14*2)\nopt_func = partial(Adam, wd=0.01, eps=1e-3)\nF1Score_Multi=F1ScoreMulti()\nPrecision_Multi=PrecisionMulti()\nRecall_Multi=RecallMulti()\nmetrics=[accuracy_multi,accuracy_any,F1Score_Multi,Precision_Multi,Recall_Multi]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we're ready to create our learner. We can use mixed precision (fp16) by simply adding a call to `to_fp16()`!","metadata":{}},{"cell_type":"code","source":"def get_learner():\n    dbch = get_data(64,224)xresnet50, \n    learn = cnn_learner(dbch, loss_func=loss_func, opt_func=opt_func, metrics=metrics)\n    return learn","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = get_learner()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Leslie Smith's famous LR finder will give us a reasonable learning rate suggestion.","metadata":{}},{"cell_type":"code","source":"learn.lr_find()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pretrain on sample","metadata":{}},{"cell_type":"markdown","source":"Here's our main routine for changing the size of the images in our DataBunch, doing one fine-tuning of the final layers, and then training the whole model for a few epochs.","metadata":{}},{"cell_type":"code","source":"def do_fit(bs,sz,epochs,lr, freeze=True):\n    learn.dbunch = get_data(bs, sz)\n    \"\"\"if freeze:\n        if learn.opt is not None: learn.opt.clear_state()\n        learn.freeze()\n        learn.fit_one_cycle(1, slice(lr))  \"\"\"\n    learn.unfreeze()\n    learn.fit_one_cycle(epochs, slice(lr))  ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we can pre-train at different sizes.","metadata":{}},{"cell_type":"code","source":"do_fit(64, 224,16, 1e-3)","metadata":{"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#choosing lr One order of magnitude less than where the minimum loss was achieved (i.e., the minimum divided by 10\ndo_fit(64,224,16,1e-3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_loss=learn.recorder.plot_loss()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot_metrics=learn.recorder.plot_metrics","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.save('resnet_final_16_epoc')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import pickle","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trained_model_path=Path('../input/trained-network')\nmodel=load_learner(trained_model_path/'resnet_final_16_epoc.pkl')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1=model.model.eval()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1[1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"filename='fuck_10_epoc_model.pkl'\npickle.dump(learn,open(filename,'wb'))\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.export('resnet_final_16_epoc.pkl')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#newpath=Path('./')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#path2=Path('../input/pickle-data')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model_test=load_learner(path2/'training.pkl')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"l,id,pre=model_test.predict(test_im)\nO=pre.sigmoid()\nO\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#A=O.detach().cpu().numpy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model=load_learner(newpath/'training.pkl')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"interp = ClassificationInterpretation.from_learner(learn)\n\ninterp.plot_confusion_matrix(figsize=(12,12), dpi=60)\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#interp.plot_confusion_matrix(figsize=(12,12), dpi=60)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#img=dbch.train_ds[10000][0]\n","metadata":{"trusted":true}},{"cell_type":"markdown","source":"> <a href=\"./\"> Download File </a>","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"<a href=\"./models\"> Download File </a>","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}