{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":45867,"databundleVersionId":6688004,"sourceType":"competition"}],"dockerImageVersionId":30559,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Lib\nimport tensorflow as tf\nimport zipfile,os\nimport random\nimport shutil\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom matplotlib.image import imread\nfrom matplotlib.pyplot import imshow\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.layers import Dense, Flatten, Conv2D, MaxPooling2D, Dropout, BatchNormalization\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:08:52.409421Z","iopub.execute_input":"2023-10-16T03:08:52.409777Z","iopub.status.idle":"2023-10-16T03:09:02.098194Z","shell.execute_reply.started":"2023-10-16T03:08:52.409754Z","shell.execute_reply":"2023-10-16T03:09:02.097302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traincsv = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntraincsv.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.100246Z","iopub.execute_input":"2023-10-16T03:09:02.101166Z","iopub.status.idle":"2023-10-16T03:09:02.133883Z","shell.execute_reply.started":"2023-10-16T03:09:02.101128Z","shell.execute_reply":"2023-10-16T03:09:02.13287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traincsv.describe()","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.135053Z","iopub.execute_input":"2023-10-16T03:09:02.135969Z","iopub.status.idle":"2023-10-16T03:09:02.156036Z","shell.execute_reply.started":"2023-10-16T03:09:02.135907Z","shell.execute_reply":"2023-10-16T03:09:02.155105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=traincsv, x=\"label\")\nplt.title(\"Ovarian Cancer Types Distributions\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.158484Z","iopub.execute_input":"2023-10-16T03:09:02.159188Z","iopub.status.idle":"2023-10-16T03:09:02.409221Z","shell.execute_reply.started":"2023-10-16T03:09:02.15916Z","shell.execute_reply":"2023-10-16T03:09:02.408373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainimg = \"/kaggle/input/UBC-OCEAN/train_images\"\ntestimg = \"/kaggle/input/UBC-OCEAN/test_images\"\ntrain_folder = os.listdir(trainimg)\ntest_folder = os.listdir(testimg)\n\nprint(len(train_folder))\nprint(len(test_folder))","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.410509Z","iopub.execute_input":"2023-10-16T03:09:02.411048Z","iopub.status.idle":"2023-10-16T03:09:02.553027Z","shell.execute_reply.started":"2023-10-16T03:09:02.411017Z","shell.execute_reply":"2023-10-16T03:09:02.552057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traincsv['is_tma'].value_counts().plot(kind='pie',autopct=\"%.1f%%\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.554274Z","iopub.execute_input":"2023-10-16T03:09:02.554768Z","iopub.status.idle":"2023-10-16T03:09:02.701003Z","shell.execute_reply.started":"2023-10-16T03:09:02.554739Z","shell.execute_reply":"2023-10-16T03:09:02.7002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_folder[:5]","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.702325Z","iopub.execute_input":"2023-10-16T03:09:02.702825Z","iopub.status.idle":"2023-10-16T03:09:02.710237Z","shell.execute_reply.started":"2023-10-16T03:09:02.702795Z","shell.execute_reply":"2023-10-16T03:09:02.709305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_folder","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.711545Z","iopub.execute_input":"2023-10-16T03:09:02.712061Z","iopub.status.idle":"2023-10-16T03:09:02.721603Z","shell.execute_reply.started":"2023-10-16T03:09:02.712032Z","shell.execute_reply":"2023-10-16T03:09:02.720672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Images in small size\npath_train_copy = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\npath_test_copy = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\ntrain_folder_copy = os.listdir(path_train_copy)\ntest_folder_copy = os.listdir(path_test_copy)\n\nprint(len(train_folder_copy))\nprint(len(test_folder_copy))","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.722838Z","iopub.execute_input":"2023-10-16T03:09:02.723357Z","iopub.status.idle":"2023-10-16T03:09:02.816789Z","shell.execute_reply.started":"2023-10-16T03:09:02.723328Z","shell.execute_reply":"2023-10-16T03:09:02.815858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traincsv","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.819997Z","iopub.execute_input":"2023-10-16T03:09:02.820409Z","iopub.status.idle":"2023-10-16T03:09:02.834148Z","shell.execute_reply.started":"2023-10-16T03:09:02.820388Z","shell.execute_reply":"2023-10-16T03:09:02.832866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Tma True\ntraintmatrue = traincsv[traincsv['is_tma']==True]\ntraintmatrue","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.835619Z","iopub.execute_input":"2023-10-16T03:09:02.836129Z","iopub.status.idle":"2023-10-16T03:09:02.848942Z","shell.execute_reply.started":"2023-10-16T03:09:02.836101Z","shell.execute_reply":"2023-10-16T03:09:02.847948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Tma false\ntraintmafalse = traincsv[traincsv['is_tma']==False]\ntraintmafalse","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.850091Z","iopub.execute_input":"2023-10-16T03:09:02.850883Z","iopub.status.idle":"2023-10-16T03:09:02.864275Z","shell.execute_reply.started":"2023-10-16T03:09:02.850855Z","shell.execute_reply":"2023-10-16T03:09:02.863172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traincsv.loc[:, 'image_id_path'] = traincsv['image_id'].apply(lambda i: f\"{i}_thumbnail.png\")\ntraincsv","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.865609Z","iopub.execute_input":"2023-10-16T03:09:02.866263Z","iopub.status.idle":"2023-10-16T03:09:02.882135Z","shell.execute_reply.started":"2023-10-16T03:09:02.866231Z","shell.execute_reply":"2023-10-16T03:09:02.881167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traintmatrue.loc[:, 'image_id_path'] = traintmatrue['image_id'].apply(lambda i: f\"{i}.png\")\ntraintmatrue\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.883175Z","iopub.execute_input":"2023-10-16T03:09:02.883826Z","iopub.status.idle":"2023-10-16T03:09:02.898381Z","shell.execute_reply.started":"2023-10-16T03:09:02.883805Z","shell.execute_reply":"2023-10-16T03:09:02.897536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traintmafalse.loc[:, 'image_id_path'] = traintmafalse['image_id'].apply(lambda i: f\"{i}_thumbnail.png\")\ntraintmafalse","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.899634Z","iopub.execute_input":"2023-10-16T03:09:02.9002Z","iopub.status.idle":"2023-10-16T03:09:02.915716Z","shell.execute_reply.started":"2023-10-16T03:09:02.90017Z","shell.execute_reply":"2023-10-16T03:09:02.914822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import imageio\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\nfrom keras.regularizers import l2\nfrom sklearn.utils import resample\n\nimage_data = []\nimage_label = []\n\npath_thumbnails = \"/kaggle/input/UBC-OCEAN/train_thumbnails/\"\npath_images = \"/kaggle/input/UBC-OCEAN/train_images/\"\n\nfor img, label in zip(traintmafalse['image_id_path'], traintmafalse['label']):\n    image_path = os.path.join(path_thumbnails, img)\n    image = Image.fromarray(imageio.imread(image_path))\n    image = image.resize((224, 224))\n    image = np.array(image)\n    image_data.append(image)\n    image_label.append(label)\n\nfor img, label in zip(traintmatrue['image_id_path'], traintmatrue['label']):\n    image_path = os.path.join(path_images, img)\n    image = Image.fromarray(imageio.imread(image_path))\n    image = image.resize((224, 224))\n    image = np.array(image)\n    image_data.append(image)\n    image_label.append(label)\n\n# Konversi label-label ke angka\nlabel_mapping = {\"CC\": 0, \"EC\": 1, \"HGSC\": 2, \"LGSC\": 3, \"MC\": 4}\nimage_label_1 = [label_mapping[label] for label in image_label]\n\n# Pisahkan data menjadi data pelatihan dan pengujian\nx_train, x_test, y_train, y_test = train_test_split(image_data, image_label_1, test_size=0.15, shuffle=True)\nx_train_undersampled, y_train_undersampled = resample(x_train, y_train, n_samples=len(np.unique(y_train)), random_state=42)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:09:02.917011Z","iopub.execute_input":"2023-10-16T03:09:02.917557Z","iopub.status.idle":"2023-10-16T03:11:54.815724Z","shell.execute_reply.started":"2023-10-16T03:09:02.917526Z","shell.execute_reply":"2023-10-16T03:11:54.814738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train_tf = tf.convert_to_tensor(x_train_undersampled, dtype=tf.float32)\ny_train_tf = tf.convert_to_tensor(y_train_undersampled, dtype=tf.float32)\nx_test_tf = tf.convert_to_tensor(x_test, dtype=tf.float32)\ny_test_tf = tf.convert_to_tensor(y_test, dtype=tf.float32)\nprint ('setup compler')","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:11:54.817132Z","iopub.execute_input":"2023-10-16T03:11:54.817449Z","iopub.status.idle":"2023-10-16T03:12:05.041055Z","shell.execute_reply.started":"2023-10-16T03:11:54.817409Z","shell.execute_reply":"2023-10-16T03:12:05.04012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.applications import ResNet50\n\nbase_model = ResNet50(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\nmodel = Sequential()\nmodel.add(Conv2D(32, (3, 3), activation='relu', input_shape=(224, 224, 3)))\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(Conv2D(64, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(Conv2D(128, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(Flatten())\n\nmodel.add(Dense(512, activation='relu'))\nmodel.add(Dense(5, activation='softmax'))\n\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:12:05.042241Z","iopub.execute_input":"2023-10-16T03:12:05.042544Z","iopub.status.idle":"2023-10-16T03:12:07.5726Z","shell.execute_reply.started":"2023-10-16T03:12:05.042514Z","shell.execute_reply":"2023-10-16T03:12:07.57188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(x_train_tf, y_train_tf, epochs=10, batch_size=32, validation_data=(x_test_tf, y_test_tf))","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:12:07.573599Z","iopub.execute_input":"2023-10-16T03:12:07.573973Z","iopub.status.idle":"2023-10-16T03:12:28.914138Z","shell.execute_reply.started":"2023-10-16T03:12:07.573941Z","shell.execute_reply":"2023-10-16T03:12:28.912985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss, acc = model.evaluate(x_test_tf,y_test_tf)\nprint(\"Accuracy on Test Data:\",acc)\nprint()\nloss, acc = model.evaluate(x_train_tf,y_train_tf)\nprint(\"Accuracy on Train Data:\",acc)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:12:28.915536Z","iopub.execute_input":"2023-10-16T03:12:28.91647Z","iopub.status.idle":"2023-10-16T03:12:29.176204Z","shell.execute_reply.started":"2023-10-16T03:12:28.916423Z","shell.execute_reply":"2023-10-16T03:12:29.175291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(x_test_tf)\ny_pred_label = [np.argmax(i) for i in y_pred]","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:12:29.177505Z","iopub.execute_input":"2023-10-16T03:12:29.177802Z","iopub.status.idle":"2023-10-16T03:12:29.445441Z","shell.execute_reply.started":"2023-10-16T03:12:29.177775Z","shell.execute_reply":"2023-10-16T03:12:29.444413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit=pd.read_csv('/kaggle/input/UBC-OCEAN/sample_submission.csv')\nsubmit[]=y_pred\ndisplay(submit)\nsubmit.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T03:12:29.447043Z","iopub.execute_input":"2023-10-16T03:12:29.447374Z","iopub.status.idle":"2023-10-16T03:12:30.361713Z","shell.execute_reply.started":"2023-10-16T03:12:29.447343Z","shell.execute_reply":"2023-10-16T03:12:30.36036Z"},"trusted":true},"execution_count":null,"outputs":[]}]}