{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":7073992,"sourceType":"datasetVersion","datasetId":4074261},{"sourceId":7082883,"sourceType":"datasetVersion","datasetId":4080514}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.callbacks import EarlyStopping\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\"\"\"\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\"\"\"\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n\n#先resize  test image\nfrom PIL import Image\nImage.MAX_IMAGE_PIXELS = 1024*1024*1024*5\ndef resize_images(input_folder, output_folder, scale_factor):\n    # 確保輸出資料夾存在\n    if not os.path.exists(output_folder):\n        os.makedirs(output_folder)\n\n    # 取得輸入資料夾中的所有檔案\n    image_files = [f for f in os.listdir(input_folder) if f.endswith(('.jpg', '.png', '.jpeg'))]\n\n    for image_file in image_files:\n        print(image_file)\n        input_path = os.path.join(input_folder, image_file)\n        output_path = os.path.join(output_folder, image_file)\n\n        # 開啟圖片\n        with Image.open(input_path) as img:\n            # 取得原始寬高\n            width, height = img.size\n            # 計算等比例縮小後的寬高\n            new_width = int(width * scale_factor)\n            new_height = int(height * scale_factor)\n            if new_width > 300 :\n                img.thumbnail((256, 256))\n                img.save(output_path)\n            else :\n                # 縮小圖片\n                resized_img = img.resize((new_width, new_height), Image.Resampling.LANCZOS)\n                # 儲存縮小後的圖片\n                resized_img.save(output_path)\n\nif __name__ == \"__main__\":\n    # 輸入資料夾的路徑\n    input_folder_path = '/kaggle/input/UBC-OCEAN/test_images/'\n    # 輸出資料夾的路徑\n    output_folder_path = '/kaggle/working/UBC-OCEAN/test/'\n    # 設定縮小比例，這裡設定為原尺寸的一半\n    scale_factor = 0.5\n    # 呼叫函式進行縮小相片\n    resize_images(input_folder_path, output_folder_path, scale_factor)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-29T19:11:13.166502Z","iopub.execute_input":"2023-11-29T19:11:13.166898Z","iopub.status.idle":"2023-11-29T19:11:31.018874Z","shell.execute_reply.started":"2023-11-29T19:11:13.166859Z","shell.execute_reply":"2023-11-29T19:11:31.017685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 讀取訓練集和測試集的 CSV 檔案\ntrain_df = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntest_df = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\n\n# 圖片資料夾路徑\ntrain_image_folder = '/kaggle/input/train-resizee/train/'\ntest_image_folder = '/kaggle/working/UBC-OCEAN/test/'\n\n# build訓練集和測試集的路徑和標籤\ntrain_image_paths = [os.path.join(train_image_folder, f'{image_id}.png') for image_id in train_df['image_id']]\ntrain_labels = train_df['label'].tolist()\n\n# 將標籤字串mapping到整數\nlabel_mapping = {'CC': 0, 'EC': 1, 'HGSC': 2, 'LGSC': 3, 'MC': 4, 'Other': 5}\n\n# 創建反向mapping字典\nreverse_label_mapping = {v: k for k, v in label_mapping.items()}\n\n# 將標籤列替換為整數標籤\ntrain_df['label'] = train_df['label'].map(label_mapping)\n\n# 切分訓練集和驗證集\nfrom sklearn.model_selection import train_test_split\ntrain_image_paths, val_image_paths, train_labels, val_labels = train_test_split(\n    train_image_paths, train_labels, test_size=0.2, random_state=42\n)\n\ntest_image_paths = [os.path.join(test_image_folder, f'{image_id}.png') for image_id in test_df['image_id']]\n\n# 定義 EarlyStopping callback\nearly_stopping = EarlyStopping(monitor='val_loss', patience=6, restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-29T19:11:31.021648Z","iopub.execute_input":"2023-11-29T19:11:31.022114Z","iopub.status.idle":"2023-11-29T19:11:31.04294Z","shell.execute_reply.started":"2023-11-29T19:11:31.022071Z","shell.execute_reply":"2023-11-29T19:11:31.041903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 建立模型\nmodel = models.Sequential([\n    layers.Conv2D(32, (3, 3), activation='relu', input_shape=(256, 256, 3)),\n    layers.MaxPooling2D((2, 2)),\n    layers.Dropout(0.25),\n    layers.Conv2D(64, (3, 3), activation='relu'),\n    layers.MaxPooling2D((2, 2)),\n    layers.Dropout(0.25),\n    layers.Conv2D(128, (3, 3), activation='relu'),\n    layers.MaxPooling2D((2, 2)),\n    layers.Flatten(),\n    layers.Dropout(0.5),\n    layers.Dense(64, activation='relu'),\n    layers.Dense(6, activation='softmax')  # 6 类别的输出层\n])\n\n# 編譯模型\nmodel.compile(optimizer='adam',\n              loss='sparse_categorical_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-11-29T19:11:31.04441Z","iopub.execute_input":"2023-11-29T19:11:31.044751Z","iopub.status.idle":"2023-11-29T19:11:31.199291Z","shell.execute_reply.started":"2023-11-29T19:11:31.044724Z","shell.execute_reply":"2023-11-29T19:11:31.198097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 預處理圖像\ndef preprocess_image(image_path, label):\n    img = load_img(image_path, target_size=(256, 256))\n    label = label_mapping[label]\n    img_array = img_to_array(img)\n    img_array = img_array / 255.0  # 归一化\n    return img_array, label","metadata":{"execution":{"iopub.status.busy":"2023-11-29T19:11:31.200652Z","iopub.execute_input":"2023-11-29T19:11:31.200989Z","iopub.status.idle":"2023-11-29T19:11:31.207087Z","shell.execute_reply.started":"2023-11-29T19:11:31.200959Z","shell.execute_reply":"2023-11-29T19:11:31.205837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 創建訓練集\ntrain_dataset = tf.data.Dataset.from_generator(\n    lambda: (preprocess_image(image_path, label) for image_path, label in zip(train_image_paths, train_labels)),\n    output_signature=(tf.TensorSpec(shape=(256, 256, 3), dtype=tf.float32), tf.TensorSpec(shape=(), dtype=tf.int64))\n)\ntrain_dataset = (\n    train_dataset\n    .shuffle(buffer_size=1000)\n    .batch(32)\n    .prefetch(tf.data.experimental.AUTOTUNE)\n)\n\nval_dataset = tf.data.Dataset.from_generator(\n    lambda: (preprocess_image(image_path, label) for image_path, label in zip(val_image_paths, val_labels)),\n    output_signature=(tf.TensorSpec(shape=(256, 256, 3), dtype=tf.float32), tf.TensorSpec(shape=(), dtype=tf.int64))\n)\nval_dataset = (\n    val_dataset\n    .batch(32)\n    .prefetch(tf.data.experimental.AUTOTUNE)\n)","metadata":{"execution":{"iopub.status.busy":"2023-11-29T19:11:31.209997Z","iopub.execute_input":"2023-11-29T19:11:31.21049Z","iopub.status.idle":"2023-11-29T19:11:31.282449Z","shell.execute_reply.started":"2023-11-29T19:11:31.210436Z","shell.execute_reply":"2023-11-29T19:11:31.281318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 訓練模型\nmodel.fit(train_dataset, epochs=99999, validation_data=val_dataset, callbacks=[early_stopping])","metadata":{"execution":{"iopub.status.busy":"2023-11-29T19:11:31.283885Z","iopub.execute_input":"2023-11-29T19:11:31.284221Z","iopub.status.idle":"2023-11-29T19:18:29.375338Z","shell.execute_reply.started":"2023-11-29T19:11:31.284191Z","shell.execute_reply":"2023-11-29T19:18:29.374205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 預測測試集\ntest_predictions = []\nfor test_image_path in test_image_paths:\n    img = load_img(test_image_path, target_size=(256, 256))\n    img_array = img_to_array(img)\n    img_array = img_array / 255.0  # 归一化\n    img_array = np.expand_dims(img_array, 0)  # 扩展维度以符合模型输入\n    prediction = model.predict(img_array)\n    predicted_label = np.argmax(prediction, axis=1)[0]\n    test_predictions.append(predicted_label)","metadata":{"execution":{"iopub.status.busy":"2023-11-29T19:18:29.376568Z","iopub.execute_input":"2023-11-29T19:18:29.376952Z","iopub.status.idle":"2023-11-29T19:18:29.600797Z","shell.execute_reply.started":"2023-11-29T19:18:29.376925Z","shell.execute_reply":"2023-11-29T19:18:29.599835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 把數字mapping回string label\ndef custom_formatting_function(value):\n    return reverse_label_mapping[value]","metadata":{"execution":{"iopub.status.busy":"2023-11-29T19:18:29.604551Z","iopub.execute_input":"2023-11-29T19:18:29.605208Z","iopub.status.idle":"2023-11-29T19:18:29.610163Z","shell.execute_reply.started":"2023-11-29T19:18:29.605175Z","shell.execute_reply":"2023-11-29T19:18:29.609067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 保存預測結果為 submission.csv\nsubmission_df = pd.DataFrame({'image_id': test_df['image_id'], 'label': test_predictions})\nsubmission_df['label'] = submission_df['label'].apply(custom_formatting_function)\nprint(submission_df)\nsubmission_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-29T19:18:29.612083Z","iopub.execute_input":"2023-11-29T19:18:29.613251Z","iopub.status.idle":"2023-11-29T19:18:29.634473Z","shell.execute_reply.started":"2023-11-29T19:18:29.613177Z","shell.execute_reply":"2023-11-29T19:18:29.632956Z"},"trusted":true},"execution_count":null,"outputs":[]}]}