{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\n\nimport os\nimport pandas as pd\nimport numpy as np\nfrom skimage.io import imread_collection\nimport skimage.io\nimport skimage.color\nimport skimage.transform\nfrom platform import python_version\nimport matplotlib.pyplot as plt\n\nprint(tf.__version__)\nprint(python_version())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# extract filenames from the folder of images\nfilenames = []\nfor root, dirs, files in os.walk('../input/rsna-hemorrhage-jpg/train_jpg/train_jpg'):\n    for file in files:\n        if file.endswith('.jpg'):\n            filenames.append(file)\n            \n# should be the same as the images imported\nlen(filenames)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"col_dir = '../input/rsna-hemorrhage-jpg/train_jpg/train_jpg/*.jpg'\n\n# Create a collection with the available images\nimages = imread_collection(col_dir)\n#we could also try what is below,\n#this should load the images in the order that we expect, \n#but if automatically alphabetical this isn't necessary:\n#images = imread_collection(col_dir, load_pattern = filenames)\n\n#make sure this is equivalent with the number of filenames\nlen(images)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot the first image\nplt.figure()\nplt.imshow(images[0])\nplt.colorbar()\nplt.grid(False)\nplt.show()\n\nprint(images[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check shape\nprint(images[0].shape)\nprint(images[1].shape)\nprint(images[2].shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Select only the first 5000 images\nimages_trn = images[:20000]\nprint(len(images_trn))\nimages_val = images[20000:25000]\nprint(len(images_val))\nimages_tst = images[25000:30000]\nprint(len(images_tst))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images_arr_trn = skimage.io.collection.concatenate_images(images_trn)\nimages_arr_val = skimage.io.collection.concatenate_images(images_val)\nimages_arr_tst = skimage.io.collection.concatenate_images(images_tst)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Import labels and selct only first 5000 labels without any additional columns\n#labels = pd.read_feather('../input/rsna-hemorrhage-jpg/meta/meta/labels.fth')\n#labels = labels.iloc[:5000, 1]\n#print(labels)\n#print(type(labels))\n#print(labels.sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = pd.read_feather('../input/rsna-hemorrhage-jpg/meta/meta/labels.fth')\n\n#manipulate the filenames list, stripping the .jpg at the end\nidstosearch = [item.rstrip(\".jpg\") for item in filenames]\n\n#now search the \"ID\" column for ids that correspond to our filenames\n#made the reduced dataframe \"labels2\" for now\nlabels2 = labels[labels['ID'].isin(idstosearch)]\nlabels2.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = labels2.iloc[:, 1]\nprint(labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels_trn = labels[:20000]\nprint(len(labels_trn))\nlabels_val = labels[20000:25000]\nprint(len(labels_val))\nlabels_tst = labels[25000:30000]\nprint(len(labels_tst))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(type(labels_trn))\nprint(labels_trn.sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Transform labels into array\nlabels_trn = pd.Series.to_numpy(labels_trn)\nlen(labels_trn)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels_val = pd.Series.to_numpy(labels_val)\nlen(labels_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels_tst = pd.Series.to_numpy(labels_tst)\nlen(labels_tst)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Build the model\nmodel = keras.Sequential([\n    keras.layers.Flatten(input_shape=(256, 256, 3)),\n    keras.layers.Dense(128, activation='relu'),\n    keras.layers.Dense(23, activation='softmax')\n])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Compile the model\nmodel.compile(optimizer='adam',\n              loss='sparse_categorical_crossentropy',\n              metrics=['accuracy'])\n\n# data = train_images.reshape(2000,75,100,1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Train model\nmodel.fit(images_arr_trn, labels_trn, epochs=10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Validate model\ntest_loss, test_acc = model.evaluate(images_arr_val, labels_val, verbose=2)\n\nprint('\\nTest accuracy:', test_acc)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# ToDos:\n# 1. Assign labels correctly\n# 2. Train / Validation / Test split\n# 3. Increase data size\n# 4. Use pretrained model to compare","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#df_comb = pd.read_feather('../input/rsna-hemorrhage-jpg/meta/meta/comb.fth').set_index('SOPInstanceUID')   \n#print(df_comb.shape)\n\n#df_tst = pd.read_feather('../input/rsna-hemorrhage-jpg/meta/meta/df_tst.fth').set_index('SOPInstanceUID')\n#print(df_tst.shape)\n\n#df_samp = pd.read_feather('../input/rsna-hemorrhage-jpg/meta/meta/wgt_sample.fth').set_index('SOPInstanceUID')\n#print(df_samp.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#from PIL import Image\n#import glob\n\n#image_list = []\n\n#for filename in glob.glob('../input/rsna-hemorrhage-jpg/train_jpg/train_jpg/*.jpg'):\n#    im=Image.open(filename)\n#    image_list.append(im)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.7.4"}},"nbformat":4,"nbformat_minor":1}