{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":10038,"sourceType":"datasetVersion","datasetId":6978},{"sourceId":7261427,"sourceType":"datasetVersion","datasetId":4208369}],"dockerImageVersionId":30626,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from fastai.vision.all import *\nset_seed(400)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-22T14:26:00.319334Z","iopub.execute_input":"2023-12-22T14:26:00.319731Z","iopub.status.idle":"2023-12-22T14:26:00.327056Z","shell.execute_reply.started":"2023-12-22T14:26:00.319702Z","shell.execute_reply":"2023-12-22T14:26:00.325808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_path = \"/kaggle/input/UBC-OCEAN/train_images\"","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:00.32912Z","iopub.execute_input":"2023-12-22T14:26:00.329455Z","iopub.status.idle":"2023-12-22T14:26:00.338756Z","shell.execute_reply.started":"2023-12-22T14:26:00.329426Z","shell.execute_reply":"2023-12-22T14:26:00.33757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from fastai.vision.all import *\n\n# Path to your dataset\ndata_path = Path('/kaggle/input/UBC-OCEAN/train_images')\n\n# Define data augmentation and transforms\n# Modify these according to your needs\nitem_tfms = Resize(224)\nbatch_tfms = aug_transforms(mult=1.0, do_flip=True, flip_vert=True, max_rotate=20.0, max_zoom=1.1, max_lighting=0.2,\n                            max_warp=0.2, p_affine=0.75, p_lighting=0.75)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:00.340769Z","iopub.execute_input":"2023-12-22T14:26:00.342368Z","iopub.status.idle":"2023-12-22T14:26:00.354378Z","shell.execute_reply.started":"2023-12-22T14:26:00.342309Z","shell.execute_reply":"2023-12-22T14:26:00.353508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from fastai.vision.all import *\n\n# Load the CSV file\ndf = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\n\n# Update path to your actual directory where thumbnails are stored\ndata_path = Path('/kaggle/input/UBC-OCEAN/train_thumbnails')\n\n# Function to check if the thumbnail file exists\ndef thumbnail_exists(row):\n    image_path = data_path / f\"{row['image_id']}_thumbnail.png\"\n    return image_path.exists()\n\n# Keep rows where the thumbnail exists\nexisting_thumbnails_mask = df.apply(thumbnail_exists, axis=1)\ndf = df[existing_thumbnails_mask]\n\n# Custom function to get image paths\ndef get_image_path(row):\n    return data_path / f\"{row['image_id']}_thumbnail.png\"\n\n# Define transforms if needed\nitem_tfms = [Resize(10)]\nbatch_tfms = aug_transforms()\n\n# Create DataBlock and DataLoaders\ndblock = DataBlock(blocks=(ImageBlock, CategoryBlock),\n                   get_x=get_image_path,\n                   get_y=ColReader('label'),  # Assuming a column 'label' contains the labels\n                   item_tfms=item_tfms,\n                   batch_tfms=batch_tfms)\n\ndls = dblock.dataloaders(df)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:00.35608Z","iopub.execute_input":"2023-12-22T14:26:00.356881Z","iopub.status.idle":"2023-12-22T14:26:01.258212Z","shell.execute_reply.started":"2023-12-22T14:26:00.356838Z","shell.execute_reply":"2023-12-22T14:26:01.256646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"dls.show_batch()\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:01.261896Z","iopub.execute_input":"2023-12-22T14:26:01.262279Z","iopub.status.idle":"2023-12-22T14:26:17.055336Z","shell.execute_reply.started":"2023-12-22T14:26:01.262243Z","shell.execute_reply":"2023-12-22T14:26:17.05395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"try:\n    learn = cnn_learner(dls, resnet34, metrics=accuracy)\nexcept Exception as e:\n    pass\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:17.056724Z","iopub.execute_input":"2023-12-22T14:26:17.057286Z","iopub.status.idle":"2023-12-22T14:26:17.06516Z","shell.execute_reply.started":"2023-12-22T14:26:17.057224Z","shell.execute_reply":"2023-12-22T14:26:17.063794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"try:\n    learn.lr_find()\nexcept Exception as e:\n    pass\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:17.066492Z","iopub.execute_input":"2023-12-22T14:26:17.066806Z","iopub.status.idle":"2023-12-22T14:26:17.080909Z","shell.execute_reply.started":"2023-12-22T14:26:17.066772Z","shell.execute_reply":"2023-12-22T14:26:17.07955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"try:\n    learn.fine_tune(epochs=5, base_lr=1e-3)\nexcept Exception as e:\n    pass\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:17.082484Z","iopub.execute_input":"2023-12-22T14:26:17.083579Z","iopub.status.idle":"2023-12-22T14:26:17.092563Z","shell.execute_reply.started":"2023-12-22T14:26:17.083529Z","shell.execute_reply":"2023-12-22T14:26:17.091592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"try:\n    learn.export('model.pkl')\nexcept Exception as e:\n    pass\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:17.09422Z","iopub.execute_input":"2023-12-22T14:26:17.094532Z","iopub.status.idle":"2023-12-22T14:26:17.106409Z","shell.execute_reply.started":"2023-12-22T14:26:17.094505Z","shell.execute_reply":"2023-12-22T14:26:17.105323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"try:\n    test_dls = dls.test_dl(get_image_files(path/'test_thumbnails'))\n    preds, _ = learn.get_preds(dl=test_dls)\nexcept Exception as e:\n    pass\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:17.10821Z","iopub.execute_input":"2023-12-22T14:26:17.108577Z","iopub.status.idle":"2023-12-22T14:26:17.120319Z","shell.execute_reply.started":"2023-12-22T14:26:17.108548Z","shell.execute_reply":"2023-12-22T14:26:17.119181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = load_learner('/kaggle/input/model-imm/test_model.pkl')","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:17.12165Z","iopub.execute_input":"2023-12-22T14:26:17.121989Z","iopub.status.idle":"2023-12-22T14:26:17.187693Z","shell.execute_reply.started":"2023-12-22T14:26:17.121959Z","shell.execute_reply":"2023-12-22T14:26:17.186572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dls = dls.test_dl(get_image_files('/kaggle/input/UBC-OCEAN/test_thumbnails'))\nprobs,_,idxs = learn.get_preds(dl=test_dls, with_decoded=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:17.192472Z","iopub.execute_input":"2023-12-22T14:26:17.193616Z","iopub.status.idle":"2023-12-22T14:26:17.655889Z","shell.execute_reply.started":"2023-12-22T14:26:17.193561Z","shell.execute_reply":"2023-12-22T14:26:17.654683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_names = [str(f) for f in test_dls.dataset.items]","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:17.657792Z","iopub.execute_input":"2023-12-22T14:26:17.658183Z","iopub.status.idle":"2023-12-22T14:26:17.663866Z","shell.execute_reply.started":"2023-12-22T14:26:17.658151Z","shell.execute_reply":"2023-12-22T14:26:17.662409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_classes = [dls.vocab[i] for i in idxs]\n\n# Create a DataFrame to store file names and predictions\ndata = {'image_id': file_names, 'Prediction': predicted_classes}\n\n# Convert the data to a pandas DataFrame\ndf = pd.DataFrame(data)","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:17.665719Z","iopub.execute_input":"2023-12-22T14:26:17.666264Z","iopub.status.idle":"2023-12-22T14:26:17.677229Z","shell.execute_reply.started":"2023-12-22T14:26:17.666222Z","shell.execute_reply":"2023-12-22T14:26:17.675862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_number(filename):\n    parts = filename.split('/')\n    image_name = parts[-1]\n    number = image_name.split('_')[0]\n    return number\n\n# Applying the function to the 'image_id' column to extract the numbers\ndf['image_id'] = df['image_id'].apply(extract_number)\n# Renaming 'Prediction' column to 'label'\ndf = df.rename(columns={'Prediction': 'label'})\n\ndf","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:17.678984Z","iopub.execute_input":"2023-12-22T14:26:17.679922Z","iopub.status.idle":"2023-12-22T14:26:17.702513Z","shell.execute_reply.started":"2023-12-22T14:26:17.67987Z","shell.execute_reply":"2023-12-22T14:26:17.70115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[['image_id', 'label']].to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-22T14:26:17.704066Z","iopub.execute_input":"2023-12-22T14:26:17.7045Z","iopub.status.idle":"2023-12-22T14:26:17.71545Z","shell.execute_reply.started":"2023-12-22T14:26:17.704467Z","shell.execute_reply":"2023-12-22T14:26:17.714326Z"},"trusted":true},"execution_count":null,"outputs":[]}]}