{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom keras.applications import resnet50\nfrom PIL import Image\nfrom numpy import asarray\n\nimport random\nimport os\nimport pandas as pd\nimport numpy as np\nfrom skimage.io import imread_collection\nimport skimage.io\nimport skimage.color\nimport skimage.transform\nfrom platform import python_version\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom keras.applications.resnet50 import ResNet50\nfrom keras.preprocessing import image\nfrom keras.applications.resnet50 import preprocess_input, decode_predictions\n\nprint(tf.__version__)\nprint(python_version())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# extract filenames from the folder of images\nfilenames = []\nfor root, dirs, files in os.walk('../input/rsna-hemorrhage-jpg/train_jpg/train_jpg'):\n    for file in files:\n        if file.endswith('.jpg'):\n            filenames.append(file)\n\nprint(\"Number test images hemorrhage positive: \"+\"{}\".format(len(filenames)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = pd.read_feather('../input/rsna-hemorrhage-jpg/meta/meta/labels.fth')\n\n#manipulate the filenames list, stripping the .jpg at the end\nidstosearch = [item.rstrip(\".jpg\") for item in filenames]\n\n#now search the \"ID\" column for ids that correspond to our filenames\nlabels = labels[labels['ID'].isin(idstosearch)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"random.seed(10)\nnew_hem = labels[labels[['epidural', 'intraparenchymal', 'intraventricular', 'subarachnoid', 'subdural']].sum(1) >= 3].copy()\nnew_normal = labels[labels['any'] == 0].copy()\nnew_normal = new_normal.sample(n = 20000)\nframes = [new_normal, new_hem]\nnew = pd.concat(frames)\nnew = new.sort_values('ID')\nprint(\"Number of images with hemorrhage: \"+\"{}\".format(len(new_hem)))\nprint(\"Number of healthy images: \"+\"{}\".format(len(new_normal)))\nprint(\"Percent of dataset with 3+ hemorrhage types: \"+\"{:.2%}\".format(len(new_hem)/len(new)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new, test = train_test_split(new, test_size = 0.1)\ntestlist = test['ID']\ntestids = [\"../input/rsna-hemorrhage-jpg/train_jpg/train_jpg/\"+ x + \".jpg\" for x in testlist]\nnewlist = new['ID']\nnewids = [\"../input/rsna-hemorrhage-jpg/train_jpg/train_jpg/\"+ x + \".jpg\" for x in newlist]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"first_image = image.load_img(testids[0], target_size=(224, 224, 3))\n\ntestimages_arr = np.empty(len(testids), dtype = type(first_image))\nfor i in range(len(testids)):\n    testimages_arr[i] = asarray(image.load_img(testids[i], target_size=(224, 224, 3)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"newimages_arr = np.empty(len(newids), dtype = type(first_image))\nfor i in range(len(newids)):\n    newimages_arr[i] = asarray(image.load_img(newids[i], target_size=(224, 224, 3)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"newimages_arr","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels_new = new.iloc[:, 1]\nprint(\"Percent train images hemorrhage positive: \"+\"{:.2%}\".format(labels_new.sum()/len(new)))\nlabels_test = test.iloc[:, 1]\nprint(\"Percent test images hemorrhage positive: \"+\"{:.2%}\".format(labels_test.sum()/len(test)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels_new = pd.Series.to_numpy(labels_new)\nlabels_test = pd.Series.to_numpy(labels_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.applications import resnet50\n\nmodel = resnet50.ResNet50(weights=\"imagenet\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Compile the model\nmodel.compile(optimizer='adam',\n              loss='sparse_categorical_crossentropy',\n              metrics=['accuracy'])\n#model.compile(optimizer=keras.optimizers.Adadelta(),\n#              loss='binary_crossentropy',\n#              metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Train model\nmodel.fit(newimages_arr, labels_new, epochs = 5, validation_split = 0.1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#### PREVIOUSLY\n\n# images = imread_collection(col_dir)\n# images_arr = skimage.io.collection.concatenate_images(images)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Validate model\n#test_loss, test_acc = model.evaluate(images_val, labels_val, verbose=2)\n\n#print('\\nTest accuracy:', test_acc)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.7.4"}},"nbformat":4,"nbformat_minor":1}