{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import glob, pylab, pandas as pd\nimport pydicom, numpy as np\nfrom os import listdir\nfrom os.path import isfile, join\nimport matplotlib.pylab as plt\n\nimport seaborn as sns\n\nfrom tqdm import tqdm_notebook as tqdm\nfrom fastai.vision import *","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dir_csv = '../input/rsna-intracranial-hemorrhage-detection'\ndir_train_img = '../input/rsna-train-stage-1-images-png-224x/stage_1_train_png_224x'\ndir_test_img = '../input/rsna-test-stage-1-images-png-224x/stage_1_test_png_224x'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv(dir_csv+'/stage_1_train.csv')\ndf[['ID', 'Image', 'Diagnosis']] = df['ID'].str.split('_', expand=True)\ndf = df[['Image', 'Diagnosis', 'Label']]\ndf.drop_duplicates(inplace=True)\ndf = df.pivot(index='Image', columns='Diagnosis', values='Label').reset_index()\ndf['Image'] = 'ID_' + df['Image']\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['fn'] = df['Image'].apply(lambda x: str(x)+\".png\")\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Some files didn't contain legitimate images, so we need to remove them\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"path_train = Path(dir_train_img)\npath_test = Path(dir_test_img)\n\nimg = open_image(path_train/df['fn'][5])\nimg.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"png = glob.glob(os.path.join(dir_train_img, '*.png'))\npng = [os.path.basename(png)[:-4] for png in png]\npng = np.array(png)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = df[df['Image'].isin(png)]\ndf.to_csv('train.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.drop(\"Image\", inplace=True, axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = df[[\"fn\", \"any\", \"epidural\", \"intraparenchymal\", \"intraventricular\", \"subarachnoid\", \"subdural\"]]\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"mask = df[\"any\"] == 0\ndf[\"None\"] = \"\"\ndf.loc[mask, \"None\"] = \"None\"\n\nlabels = ['any', 'epidural', 'intraparenchymal', 'intraventricular','subarachnoid', 'subdural', 'None']\n\nfor col in labels:\n    print(col,end=\", \")\n    df[col] = df[col].replace({0:\"\", 1:col})\n\ndf[\"mul_label\"] = df[labels].apply(lambda x: \" \".join((' '.join(x)).split()), axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Test data"},{"metadata":{"trusted":true},"cell_type":"code","source":"test = pd.read_csv(dir_csv+\"/stage_1_sample_submission.csv\")\ntest[\"fn\"] = test.ID.apply(lambda x: \"_\".join(x.split(\"_\")[:2])+\".png\")\ntest[\"label\"] = test.ID.apply(lambda x: x.split(\"_\")[-1])\ntest.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pivot_test = test.pivot(index=\"fn\", columns=\"label\", values=\"Label\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pivot_test.reset_index(inplace=True)\npivot_test[\"Multilabel\"]= \" \"\npivot_test.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Image Data Bunch"},{"metadata":{"trusted":true},"cell_type":"code","source":"tfms = get_transforms(do_flip=True)\n\nnp.random.seed(42)\ndata_train = (ImageList\n             .from_df(path=path_train, df=df[[\"fn\", \"mul_label\"]])\n             .split_by_rand_pct(0.1)\n             .label_from_df(cols=1, label_delim=\" \")\n             .transform(size=(128, 128))\n             .databunch()\n             .normalize(imagenet_stats))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_train.show_batch(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_test = (ImageList\n             .from_df(path=path_test, df=pivot_test[[\"fn\", \"Multilabel\"]])\n             .split_by_rand_pct(valid_pct=0)\n             .label_from_df(cols=1, label_delim=\" \")\n             .transform(size=(128, 128))\n             .databunch()\n             .normalize(imagenet_stats))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_test.show_batch(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"acc_02 = partial(accuracy_thresh, thresh=0.2)\nf_score = partial(fbeta, thresh=0.2)\n\n\nmodels_path = Path(\"/kaggle/working/models\")\nif not models_path.exists(): models_path.mkdir()\n\nmodel_34 = cnn_learner(data_train, models.resnet34, metrics=[acc_02, f_score], model_dir = models_path).to_fp16()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_34.lr_find()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_34.recorder.plot()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lr = 5e-2\nmodel_34.freeze()\nmodel_34.fit_one_cycle(2, slice(lr))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_34.save(\"95%_acc\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Predictions"},{"metadata":{},"cell_type":"markdown","source":"## Submissions"},{"metadata":{"trusted":true},"cell_type":"code","source":"model_34.data = data_test\nmodel_34.model.float()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = model_34.get_preds(ds_type=DatasetType.Test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.read_csv(dir_csv + /\"stage_1_sample_submission.csv\")\n\n\npreds = np.array(preds[0])\nany_probs = 1 - np.prod(1 - preds, axis=1)\nsubmission.Label = np.hstack([preds, np.expand_dims(any_probs, -1)]).reshape(-1)\nsubmission.head()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}