{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-21T17:17:15.134776Z","iopub.execute_input":"2022-08-21T17:17:15.135253Z","iopub.status.idle":"2022-08-21T17:17:15.264201Z","shell.execute_reply.started":"2022-08-21T17:17:15.135163Z","shell.execute_reply":"2022-08-21T17:17:15.262853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport cv2\nimport tifffile\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tqdm.notebook import tqdm\nfrom tensorflow import reshape\nfrom tensorflow.keras import Model,backend\nimport tensorflow_addons as tfa\nfrom keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Flatten, Dropout, Embedding, LSTM, Activation,ZeroPadding1D,Conv2D","metadata":{"execution":{"iopub.status.busy":"2022-08-21T17:17:26.7755Z","iopub.execute_input":"2022-08-21T17:17:26.776004Z","iopub.status.idle":"2022-08-21T17:17:33.668603Z","shell.execute_reply.started":"2022-08-21T17:17:26.775958Z","shell.execute_reply":"2022-08-21T17:17:33.667438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 20","metadata":{"execution":{"iopub.status.busy":"2022-08-21T17:17:35.025571Z","iopub.execute_input":"2022-08-21T17:17:35.026665Z","iopub.status.idle":"2022-08-21T17:17:35.034053Z","shell.execute_reply.started":"2022-08-21T17:17:35.02662Z","shell.execute_reply":"2022-08-21T17:17:35.032979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#debug = False\n#train_df = pd.read_csv(\"../input/mayo-clinic-strip-ai/train.csv\").head(10 if debug else 1000)\ntrain_df = pd.read_csv(\"../input/mayo-clinic-strip-ai/train.csv\")\n#max_count = max(train_df.label.value_counts())\n#for label in train_df.label.unique():\n#    df = train_df.loc[train_df.label == label]\n#    while(train_df.label.value_counts()[label] < max_count):\n#        train_df = pd.concat([train_df, df.head(max_count - train_df.label.value_counts()[label])], axis = 0)\n\ntrain_x = train_df.iloc[:,:]\nvalid_x = train_df.sample(2)\n\ntrain_img = []\nvalid_img = []\n\n\nif(True):\n    #os.mkdir('./train')\n    for i in tqdm(range(train_x.shape[0])):\n        img_id = train_x.iloc[i].image_id\n        img = cv2.resize(tifffile.imread('../input/mayo-clinic-strip-ai/train/' + img_id + \".tif\"), (224,224))\n        #cv2.imwrite(f\"./train/{img_id}.jpg\", img)\n        train_img.append(img)\n        del img\n        gc.collect()\n        \nif(True):\n    #os.mkdir('./valid')\n    for i in tqdm(range(valid_x.shape[0])):\n        img_id = valid_x.iloc[i].image_id\n        img = cv2.resize(tifffile.imread('../input/mayo-clinic-strip-ai/train/' + img_id + \".tif\"), (224,224))\n        #cv2.imwrite(f\"./valid/{img_id}.jpg\", img)\n        valid_img.append(img)\n        del img\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-21T17:17:37.535264Z","iopub.execute_input":"2022-08-21T17:17:37.535796Z","iopub.status.idle":"2022-08-21T20:50:38.345947Z","shell.execute_reply.started":"2022-08-21T17:17:37.535751Z","shell.execute_reply":"2022-08-21T20:50:38.344359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\nohe = OneHotEncoder()\ntrain_y = ohe.fit_transform(train_x[['label']])\nvalid_y = ohe.transform(valid_x[['label']])","metadata":{"execution":{"iopub.status.busy":"2022-08-21T20:59:27.500823Z","iopub.execute_input":"2022-08-21T20:59:27.502202Z","iopub.status.idle":"2022-08-21T20:59:28.259541Z","shell.execute_reply.started":"2022-08-21T20:59:27.502108Z","shell.execute_reply":"2022-08-21T20:59:28.258171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_y.toarray())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p_min = 0.0005\np_max = 0.9995\n\ndef logloss(y_true, y_pred):\n    y_pred = tf.clip_by_value(y_pred,p_min,p_max)\n    return -backend.mean(y_true*backend.log(y_pred) + (1-y_true)*backend.log(1-y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-21T20:59:56.338098Z","iopub.execute_input":"2022-08-21T20:59:56.339103Z","iopub.status.idle":"2022-08-21T20:59:56.347651Z","shell.execute_reply.started":"2022-08-21T20:59:56.339049Z","shell.execute_reply":"2022-08-21T20:59:56.346265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model():\n    input_pri = tf.keras.layers.Input(shape = (224, 224, 3), name = 'Input1')\n    \n    mouth_1 = Sequential([\n        tf.keras.layers.Conv2D(64, (3,3), padding='valid'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.2),\n        tf.keras.layers.Activation('relu'),\n        tf.keras.layers.MaxPool2D(pool_size=(2, 2), padding='valid'),\n        tf.keras.layers.Conv2D(32,(4,4), padding='valid'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.5),\n        tf.keras.layers.Activation('relu'),\n        tf.keras.layers.MaxPool2D(pool_size=(2, 2), padding='valid'),\n        tf.keras.layers.Conv2D(16,(5,5), padding='valid'),\n        tf.keras.layers.Activation('relu'),\n        tf.keras.layers.MaxPool2D(pool_size=(2, 2), padding='valid'),\n        tf.keras.layers.Flatten()\n        ],name='Head')\n    \n    input_mouth = mouth_1(input_pri)\n    \n    num_columns = 10000\n    \n    Stomach_1 = Sequential([    \n        tf.keras.layers.Input(num_columns),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(1024, activation=\"relu\")),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(512, activation=\"relu\")),\n        tf.keras.layers.BatchNormalization(), \n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(256, activation=\"relu\")),\n        tf.keras.layers.BatchNormalization(), \n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(128, activation=\"relu\")),        \n    ],name='stomach_1')\n    \n    \n    Stomach_2 = Sequential([    \n        tf.keras.layers.Input(num_columns),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(1024, activation=\"relu\")),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(256, activation=\"relu\")),\n        tf.keras.layers.BatchNormalization(), \n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(256, activation=\"relu\")),\n        tf.keras.layers.BatchNormalization(), \n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(1024, activation=\"relu\")),        \n    ],name='stomach_2')\n    \n    Stomach_3 = Sequential([    \n        tf.keras.layers.Input(num_columns),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(256, activation=\"relu\")),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(1024, activation=\"relu\")),\n        tf.keras.layers.BatchNormalization(), \n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(1024, activation=\"relu\")),\n        tf.keras.layers.BatchNormalization(), \n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(256, activation=\"relu\")),        \n    ],name='stomach_3')\n    \n    \n    Stomach_4 = Sequential([    \n        tf.keras.layers.Input(num_columns),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(128, activation=\"relu\")),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(256, activation=\"relu\")),\n        tf.keras.layers.BatchNormalization(), \n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(512, activation=\"relu\")),\n        tf.keras.layers.BatchNormalization(), \n        tf.keras.layers.Dropout(0.5),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(1024, activation=\"relu\")),        \n    ],name='stomach_4')\n    \n    input_1 = Stomach_1(input_mouth)\n    input_2 = Stomach_2(input_mouth)\n    input_3 = Stomach_3(input_mouth)\n    input_4 = Stomach_4(input_mouth)\n    \n    input_final = tf.keras.layers.Concatenate()([input_1,input_2,input_3,input_4])\n    \n    num_columns_1 = 2432\n    \n    tail = tf.keras.Sequential([\n        tf.keras.layers.Input(num_columns_1),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(128, activation='relu')),\n        tf.keras.layers.BatchNormalization(),\n        tfa.layers.WeightNormalization(tf.keras.layers.Dense(2, activation=\"sigmoid\"))\n        ],name='tail')\n    \n    output = tail(input_final)\n    \n    \n    model = Model(inputs = input_pri, outputs = output)\n    model.compile(optimizer=tf.optimizers.Adam(),\n                  loss = tf.keras.losses.BinaryCrossentropy(),metrics=logloss)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-08-21T21:00:01.165734Z","iopub.execute_input":"2022-08-21T21:00:01.166128Z","iopub.status.idle":"2022-08-21T21:00:01.197839Z","shell.execute_reply.started":"2022-08-21T21:00:01.166096Z","shell.execute_reply":"2022-08-21T21:00:01.196818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='loss', factor=0.1, verbose=1,mode='min',patience=3, min_lr=1E-6)\nearly_st = tf.keras.callbacks.EarlyStopping(monitor='loss', min_delta=1E-5, patience=5, verbose=1, mode='min',baseline=None, restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-21T21:00:06.207551Z","iopub.execute_input":"2022-08-21T21:00:06.207938Z","iopub.status.idle":"2022-08-21T21:00:06.214407Z","shell.execute_reply.started":"2022-08-21T21:00:06.207896Z","shell.execute_reply":"2022-08-21T21:00:06.213289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model()\nhistory = model.fit(np.array(train_img),train_y.toarray(), batch_size= batch_size, epochs=1000,verbose=2,validation_data = (np.array(valid_img),valid_y.toarray()),callbacks =[reduce_lr, early_st])","metadata":{"execution":{"iopub.status.busy":"2022-08-21T21:00:23.885649Z","iopub.execute_input":"2022-08-21T21:00:23.886031Z","iopub.status.idle":"2022-08-21T21:35:02.243451Z","shell.execute_reply.started":"2022-08-21T21:00:23.886001Z","shell.execute_reply":"2022-08-21T21:35:02.242114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model.save('first_model.h5')","metadata":{"execution":{"iopub.status.busy":"2022-08-21T21:37:23.493381Z","iopub.execute_input":"2022-08-21T21:37:23.493898Z","iopub.status.idle":"2022-08-21T21:37:25.199196Z","shell.execute_reply.started":"2022-08-21T21:37:23.49386Z","shell.execute_reply":"2022-08-21T21:37:25.198364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(\"../input/mayo-clinic-strip-ai/test.csv\")\ntest_x = []\nif(True):\n    #os.mkdir('./train')\n    for i in tqdm(range(test_df.shape[0])):\n        img_id = test_df.iloc[i].image_id\n        img = cv2.resize(tifffile.imread('../input/mayo-clinic-strip-ai/test/' + img_id + \".tif\"), (224,224))\n        #cv2.imwrite(f\"./train/{img_id}.jpg\", img)\n        test_x.append(img)\n        del img\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-21T21:47:09.956152Z","iopub.execute_input":"2022-08-21T21:47:09.956628Z","iopub.status.idle":"2022-08-21T21:49:00.416036Z","shell.execute_reply.started":"2022-08-21T21:47:09.956578Z","shell.execute_reply":"2022-08-21T21:49:00.414687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_y = model.predict(np.array(test_x))","metadata":{"execution":{"iopub.status.busy":"2022-08-21T21:50:29.521764Z","iopub.execute_input":"2022-08-21T21:50:29.522927Z","iopub.status.idle":"2022-08-21T21:50:33.303211Z","shell.execute_reply.started":"2022-08-21T21:50:29.522882Z","shell.execute_reply":"2022-08-21T21:50:33.302088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_y","metadata":{"execution":{"iopub.status.busy":"2022-08-21T21:50:40.890325Z","iopub.execute_input":"2022-08-21T21:50:40.890717Z","iopub.status.idle":"2022-08-21T21:50:40.900137Z","shell.execute_reply.started":"2022-08-21T21:50:40.890686Z","shell.execute_reply":"2022-08-21T21:50:40.899346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.savetxt('results.csv',test_y,delimiter=\",\")","metadata":{"execution":{"iopub.status.busy":"2022-08-21T21:54:39.869858Z","iopub.execute_input":"2022-08-21T21:54:39.870286Z","iopub.status.idle":"2022-08-21T21:54:39.877463Z","shell.execute_reply.started":"2022-08-21T21:54:39.870247Z","shell.execute_reply":"2022-08-21T21:54:39.876031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}