{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\ndf = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\nselected_columns = ['image_id', 'label']\ndf_id_lable = df[selected_columns]\nprint(df_id_lable)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:35:14.256376Z","iopub.execute_input":"2023-11-09T04:35:14.256733Z","iopub.status.idle":"2023-11-09T04:35:14.630496Z","shell.execute_reply.started":"2023-11-09T04:35:14.256704Z","shell.execute_reply":"2023-11-09T04:35:14.62943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:35:14.632125Z","iopub.execute_input":"2023-11-09T04:35:14.632407Z","iopub.status.idle":"2023-11-09T04:35:14.640682Z","shell.execute_reply.started":"2023-11-09T04:35:14.632382Z","shell.execute_reply":"2023-11-09T04:35:14.639843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# สร้าง DataFrames สองอันแยกกันสำหรับค่า 'is_tma' 'True' และ 'False'\n# เนื่องจากค่า  is_tama = True มีอยู่ใน train_images อย่างเดียว ไม่มีใน train_thumbnails\ntma_true_df = df[df['is_tma'] == True].copy()\ntma_false_df = df[df['is_tma'] == False].copy()\n\nbase_path_tma_true = '/kaggle/input/UBC-OCEAN/train_images/'\nbase_path_tma_false = '/kaggle/input/UBC-OCEAN/train_thumbnails/'\n\n#Function เพื่อสร้างเส้นทางไฟล์แบบเต็มตาม 'image_id'\ndef get_image_path(row):\n    if row['is_tma'] == True :\n        base_path = base_path_tma_true\n    elif row['is_tma'] == False :\n        base_path = base_path_tma_false  \n    # รวม '.png' ต่อท้ายเส้นทางสำหรับ 'is_tma' True\n    if row['is_tma'] == True:\n        return os.path.join(base_path, str(row['image_id']) + '.png')\n    # รวม '_thumbnail.png' ต่อท้ายเส้นทางสำหรับ 'is_tma' False\n    elif row['is_tma'] == False:\n        return os.path.join(base_path, str(row['image_id']) + '_thumbnail.png')\n\n\n# ใช้ฟังก์ชันเพื่อสร้างคอลัมน์ใหม่พร้อมเส้นทางไฟล์แบบเต็ม\ntma_true_df['image_path'] = tma_true_df.apply(get_image_path, axis=1)\ntma_false_df['image_path'] = tma_false_df.apply(get_image_path, axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:35:14.641706Z","iopub.execute_input":"2023-11-09T04:35:14.641965Z","iopub.status.idle":"2023-11-09T04:35:14.674124Z","shell.execute_reply.started":"2023-11-09T04:35:14.641943Z","shell.execute_reply":"2023-11-09T04:35:14.673237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tma_true_df['image_path'])","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:35:14.675183Z","iopub.execute_input":"2023-11-09T04:35:14.675459Z","iopub.status.idle":"2023-11-09T04:35:14.680798Z","shell.execute_reply.started":"2023-11-09T04:35:14.675435Z","shell.execute_reply":"2023-11-09T04:35:14.679788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tma_false_df['image_path'])","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:35:14.683466Z","iopub.execute_input":"2023-11-09T04:35:14.683729Z","iopub.status.idle":"2023-11-09T04:35:14.693283Z","shell.execute_reply.started":"2023-11-09T04:35:14.6837Z","shell.execute_reply":"2023-11-09T04:35:14.692445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport os\nimport numpy as np\n\ndef process_image_is_tma(image_path):\n    image = cv2.imread(image_path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = cv2.resize(image, (64, 64))\n    return image\n\nnormalized_images = []\n\nfor i, image_path in enumerate(tma_true_df['image_path']):\n    img_np = process_image_is_tma(image_path)\n    img_np_normalized = img_np / 255.0\n    # Append the tiles to the list\n    normalized_images.append(img_np_normalized)\n    print(f\"Image Index: {i}, File Path: {image_path}\")\n\n# Convert the normalized image tiles into a NumPy array\nis_tma_image = np.array(normalized_images)\n\n# Check the shape of the array\nprint(is_tma_image.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:35:14.694301Z","iopub.execute_input":"2023-11-09T04:35:14.694574Z","iopub.status.idle":"2023-11-09T04:35:27.189778Z","shell.execute_reply.started":"2023-11-09T04:35:14.694544Z","shell.execute_reply":"2023-11-09T04:35:27.188806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(is_tma_image.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:35:27.190975Z","iopub.execute_input":"2023-11-09T04:35:27.191273Z","iopub.status.idle":"2023-11-09T04:35:27.196092Z","shell.execute_reply.started":"2023-11-09T04:35:27.191247Z","shell.execute_reply":"2023-11-09T04:35:27.195235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#เปลี่ยนภาพ is_tma_image == False ทั้งหมด ให้เป็น numpy array\ndef process_image_not_tma(image_path):\n    image = cv2.imread(image_path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    # is_tma_image == False ทั้งหมด มา resize 64x64\n    image = cv2.resize(image, (64, 64))\n    return image\n\nnormalized_images = []\n\nfor i, image_path in enumerate(tma_false_df['image_path']):\n    if os.path.isfile(image_path):\n        img_np = process_image_not_tma(image_path)\n        img_np_normalized = img_np / 255.0\n        normalized_images.append(img_np_normalized)\n        print(f\"Image Index: {i}, File Path: {image_path}\")\n    else:\n        print(f\"File not found: {image_path}\")\n\n# แปลงไทล์รูปภาพมาตรฐานให้เป็นอาร์เรย์ NumPy\nnot_tma_image = np.array(normalized_images)\n\n# Check the shape of the array\nprint(not_tma_image.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:35:27.197232Z","iopub.execute_input":"2023-11-09T04:35:27.197493Z","iopub.status.idle":"2023-11-09T04:37:07.956632Z","shell.execute_reply.started":"2023-11-09T04:35:27.19747Z","shell.execute_reply":"2023-11-09T04:37:07.955661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\n#รวมตัว train_images และ train_thumbnails ที่เป็น numpy array แล้ว เข้าด้วยกัน\ntrain_data = np.vstack((not_tma_image, is_tma_image))\n\n# Check the shape of the training data\nprint(train_data.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:37:07.957847Z","iopub.execute_input":"2023-11-09T04:37:07.958153Z","iopub.status.idle":"2023-11-09T04:37:07.983336Z","shell.execute_reply.started":"2023-11-09T04:37:07.958127Z","shell.execute_reply":"2023-11-09T04:37:07.98234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#สร้าง label จากข้อมูลทั้งหมด\ntrain_label = pd.concat([tma_false_df['label'],tma_true_df['label']], ignore_index=True)\nprint(train_label.shape)\nprint(train_label)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:37:07.98453Z","iopub.execute_input":"2023-11-09T04:37:07.984834Z","iopub.status.idle":"2023-11-09T04:37:07.991902Z","shell.execute_reply.started":"2023-11-09T04:37:07.984809Z","shell.execute_reply":"2023-11-09T04:37:07.991018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom tensorflow.keras.utils import to_categorical\n#เข้ารหัส label เปลี่ยนเป็น 0 หรือ 1\nlabel_encoder = LabelEncoder()\nencoded_labels = label_encoder.fit_transform(train_label)\ntrain_label = to_categorical(encoded_labels, num_classes=5)\n\nprint(train_label)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:37:07.993091Z","iopub.execute_input":"2023-11-09T04:37:07.993428Z","iopub.status.idle":"2023-11-09T04:37:16.21547Z","shell.execute_reply.started":"2023-11-09T04:37:07.993394Z","shell.execute_reply":"2023-11-09T04:37:16.214442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_label.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:37:16.216654Z","iopub.execute_input":"2023-11-09T04:37:16.217246Z","iopub.status.idle":"2023-11-09T04:37:16.222036Z","shell.execute_reply.started":"2023-11-09T04:37:16.217215Z","shell.execute_reply":"2023-11-09T04:37:16.221075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data.shape)\n# reshape train_data เป็น 1 มิติ\ntrain_data = train_data.reshape((538, 64*64*3))\nprint(train_data.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:37:16.223418Z","iopub.execute_input":"2023-11-09T04:37:16.223676Z","iopub.status.idle":"2023-11-09T04:37:16.235827Z","shell.execute_reply.started":"2023-11-09T04:37:16.223653Z","shell.execute_reply":"2023-11-09T04:37:16.235085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.optimizers import SGD\nimport matplotlib.pyplot as plt\nMLP = Sequential()\nMLP.add(Dense(128, input_shape=(64*64*3, ), activation='sigmoid'))\nMLP.add(Dense(5, activation='sigmoid'))\n\nnew_SGD = SGD(learning_rate = 0.03)\nMLP.compile(loss='binary_crossentropy', optimizer=new_SGD)\nMLP.summary()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:37:16.238886Z","iopub.execute_input":"2023-11-09T04:37:16.239471Z","iopub.status.idle":"2023-11-09T04:37:19.405666Z","shell.execute_reply.started":"2023-11-09T04:37:16.239447Z","shell.execute_reply":"2023-11-09T04:37:19.404718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hist = MLP.fit(train_data, train_label, epochs=200, verbose=0)\n\nplt.plot(hist.history['loss'])\nplt.ylabel('Loss')\nplt.xlabel('Epochs')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:37:19.406819Z","iopub.execute_input":"2023-11-09T04:37:19.407131Z","iopub.status.idle":"2023-11-09T04:37:32.382003Z","shell.execute_reply.started":"2023-11-09T04:37:19.407105Z","shell.execute_reply":"2023-11-09T04:37:32.381026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\npredic = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\nselected_columns = ['image_id']\npredic = predic[selected_columns]\nprint(predic)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:37:32.383038Z","iopub.execute_input":"2023-11-09T04:37:32.383331Z","iopub.status.idle":"2023-11-09T04:37:32.398805Z","shell.execute_reply.started":"2023-11-09T04:37:32.383301Z","shell.execute_reply":"2023-11-09T04:37:32.397773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport cv2\nimport numpy as np\n\npredic = df\n# สร้าง DataFrames สองอันแยกกันสำหรับค่า 'is_tma' 'True' และ 'False'\n# เนื่องจากค่า  is_tama = True มีอยู่ใน train_images อย่างเดียว ไม่มีใน train_thumbnails\ntma_true_df = df[df['is_tma'] == True].copy()\ntma_false_df = df[df['is_tma'] == False].copy()\n\nbase_path_tma_true = '/kaggle/input/UBC-OCEAN/train_images/'\nbase_path_tma_false = '/kaggle/input/UBC-OCEAN/train_thumbnails/'\n\n#Function เพื่อสร้างเส้นทางไฟล์แบบเต็มตาม 'image_id'\ndef get_image_path(row):\n    if row['is_tma'] == True :\n        predic_path = base_path_tma_true\n    elif row['is_tma'] == False :\n        predic_path = base_path_tma_false  \n    # รวม '.png' ต่อท้ายเส้นทางสำหรับ 'is_tma' True\n    if row['is_tma'] == True:\n        return os.path.join(predic_path, str(row['image_id']) + '.png')\n    # รวม '_thumbnail.png' ต่อท้ายเส้นทางสำหรับ 'is_tma' False\n    elif row['is_tma'] == False:\n        return os.path.join(predic_path, str(row['image_id']) + '_thumbnail.png')\n# Threshold for classifying predictions\nthreshold = 0.5\npredictions_df = pd.DataFrame(columns=['image_id', 'label'])\n\n# Iterate over image IDs in the DataFrame 'predic'\nfor image_id in predic['image_id']:\n    # Get the image path\n    image_path = get_image_path(predic[predic['image_id'] == image_id].squeeze())\n    \n    if predic[predic['image_id'] == image_id]['is_tma'].values[0]:\n        # Process image when is_tma is True\n        image = process_image_is_tma(image_path)\n        image = image.reshape(64, 64, 3)\n        image = image / 255.0 \n    else :\n        # Process image when is_tma is False\n        image = process_image_not_tma(image_path)\n        image = image / 255.0\n        \n    image = image.reshape((64*64*3))\n    # Make predictions using the MLP model\n    predictions = MLP.predict(np.array([image]))\n\n# Calculate the label based on predictions and threshold\n    if (predictions.max() < threshold):\n        label = 'Other'\n    else:\n        class_labels = ['CC', 'EC', 'HGSC', 'LGSC', 'MC']\n        predicted_class = class_labels[np.argmax(predictions)]\n        label = predicted_class\n\n# Print the image_id and the corresponding predicted class label\n    print(f\"Image ID: {image_id}, Predicted Label: {label}\")\n\n# Add the image_id and label to the DataFrame\n\n    predictions_df = pd.concat([predictions_df, pd.DataFrame({'image_id': [image_id], 'label': [label]})], ignore_index=True)\n\n# Save the DataFrame to a CSV file\npredictions_df.to_csv('submission.csv', index=False)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:37:32.400531Z","iopub.execute_input":"2023-11-09T04:37:32.401Z","iopub.status.idle":"2023-11-09T04:39:38.751428Z","shell.execute_reply.started":"2023-11-09T04:37:32.400963Z","shell.execute_reply":"2023-11-09T04:39:38.750446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndf = pd.read_csv('/kaggle/working/submission.csv')\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T04:39:38.752643Z","iopub.execute_input":"2023-11-09T04:39:38.752942Z","iopub.status.idle":"2023-11-09T04:39:38.76286Z","shell.execute_reply.started":"2023-11-09T04:39:38.752917Z","shell.execute_reply":"2023-11-09T04:39:38.761868Z"},"trusted":true},"execution_count":null,"outputs":[]}]}