{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 0. Imports and Setup\n\nThe code below imports required modules, loads the dataframes, and sets up some needed variables.","metadata":{}},{"cell_type":"code","source":"from fastai.imports import *\nfrom fastai.vision.all import *\nfrom fastai.metrics import BalancedAccuracy\nimport shutil\nimport random\nimport os\n\n# Normal max size causes decompression bomb warnings\nImage.MAX_IMAGE_PIXELS = 10000000000","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-29T16:18:53.715341Z","iopub.execute_input":"2023-10-29T16:18:53.715796Z","iopub.status.idle":"2023-10-29T16:18:58.919814Z","shell.execute_reply.started":"2023-10-29T16:18:53.71576Z","shell.execute_reply":"2023-10-29T16:18:58.91883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = Path('../input/UBC-OCEAN')","metadata":{"execution":{"iopub.status.busy":"2023-10-29T16:18:58.923789Z","iopub.execute_input":"2023-10-29T16:18:58.924094Z","iopub.status.idle":"2023-10-29T16:18:58.92891Z","shell.execute_reply.started":"2023-10-29T16:18:58.924069Z","shell.execute_reply":"2023-10-29T16:18:58.927706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(path/\"train.csv\")\ntest_df = pd.read_csv(path/\"test.csv\")\nsubmission = pd.read_csv(path/\"sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-10-29T16:18:58.930142Z","iopub.execute_input":"2023-10-29T16:18:58.930512Z","iopub.status.idle":"2023-10-29T16:18:58.987918Z","shell.execute_reply.started":"2023-10-29T16:18:58.930464Z","shell.execute_reply":"2023-10-29T16:18:58.9869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. First Pass","metadata":{}},{"cell_type":"markdown","source":"## 1a. Nice Functions\n\nThis section consists of functions that do the bulk of necessary jobs for prediction. These include:\n* return_default_value_if_fails - Returns a default failure should a decorated function throw an exception\n* return_label - Fetches the label of a training image located at a certain file path using the train dataframe | Required by the learner's dataloaders, even though we don't use it at inference time\n* get_file_path - Fetches the path of an image(thumbnail if it exists, regular if not) using its image_id(chooses correct folder depending on whether test=True), and throws an error if the image doesn't exist\n* get_cancer_images - Creates a list of all the train images or all the test images | Required by the learner's dataloaders, not used during inference","metadata":{}},{"cell_type":"code","source":"fails = []\n\ndef return_default_value_if_fails(default_value):\n    def decorator(func):\n        def inner(*args, **kwargs):\n            try:\n                return func(*args, **kwargs)\n            except Exception as e:\n                fails.append((func, (args, kwargs), e))\n                return default_value\n        return inner\n    return decorator","metadata":{"execution":{"iopub.status.busy":"2023-10-29T16:18:58.990663Z","iopub.execute_input":"2023-10-29T16:18:58.991005Z","iopub.status.idle":"2023-10-29T16:18:58.997588Z","shell.execute_reply.started":"2023-10-29T16:18:58.990975Z","shell.execute_reply":"2023-10-29T16:18:58.996554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def return_label(o):\n    return df.loc[df['image_id'] == int(\"\".join(filter(str.isdigit, Path(o).name)))][\"label\"].iloc[0]\n\ndef get_file_path(o, test=False):\n    base_path = f\"{path}/{'test' if test else 'train'}\" \n    if os.path.exists(f\"{base_path}_thumbnails/{o}_thumbnail.png\"):\n        return f\"{base_path}_thumbnails/{o}_thumbnail.png\"\n    elif os.path.exists(f\"{base_path}_images/{o}.png\"):\n        return f\"{base_path}_images/{o}.png\"\n    else:\n        raise OSError(o, \"Image does not exist in data\")\n        \ndef get_cancer_images(_, test=False):\n    files = []\n    if test:\n        for i in test_df[\"image_id\"]:\n            files.append(get_file_path(i, test=True))\n    else:\n        for i in df[\"image_id\"]:\n            files.append(get_file_path(i))\n    return files","metadata":{"execution":{"iopub.status.busy":"2023-10-29T16:18:58.999208Z","iopub.execute_input":"2023-10-29T16:18:58.999947Z","iopub.status.idle":"2023-10-29T16:18:59.009796Z","shell.execute_reply.started":"2023-10-29T16:18:58.999914Z","shell.execute_reply":"2023-10-29T16:18:59.008917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1b. Inference\n\nThis section is all the code that does the prediction and submission using the test data. Instead of training a model here, I'm instead loading a pretrained model that is stored in a custom dataset. The original training notebook is [here](https://www.kaggle.com/japancolorado/fast-ai-first-pass/).\n\n","metadata":{}},{"cell_type":"code","source":"balanced_accuracy = BalancedAccuracy()\n\nlearner = load_learner(\"/kaggle/input/first-pass-model/export.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-10-29T16:18:59.011009Z","iopub.execute_input":"2023-10-29T16:18:59.011472Z","iopub.status.idle":"2023-10-29T16:19:00.611549Z","shell.execute_reply.started":"2023-10-29T16:18:59.011438Z","shell.execute_reply":"2023-10-29T16:19:00.610631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#@return_default_value_if_fails(default_value=\"LGSC\")\ndef predict_from_id(image_id):\n    file = get_file_path(image_id, True)\n    label, _, probs = learner.predict(file)\n    return label","metadata":{"execution":{"iopub.status.busy":"2023-10-29T16:19:00.612811Z","iopub.execute_input":"2023-10-29T16:19:00.613162Z","iopub.status.idle":"2023-10-29T16:19:00.618675Z","shell.execute_reply.started":"2023-10-29T16:19:00.613128Z","shell.execute_reply":"2023-10-29T16:19:00.617777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[\"label\"] = submission[\"image_id\"].apply(predict_from_id)","metadata":{"execution":{"iopub.status.busy":"2023-10-29T16:19:00.619882Z","iopub.execute_input":"2023-10-29T16:19:00.621001Z","iopub.status.idle":"2023-10-29T16:19:01.376579Z","shell.execute_reply.started":"2023-10-29T16:19:00.620974Z","shell.execute_reply":"2023-10-29T16:19:01.375852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)\n!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-10-29T16:19:01.377769Z","iopub.execute_input":"2023-10-29T16:19:01.37846Z","iopub.status.idle":"2023-10-29T16:19:02.324438Z","shell.execute_reply.started":"2023-10-29T16:19:01.37843Z","shell.execute_reply":"2023-10-29T16:19:02.323397Z"},"trusted":true},"execution_count":null,"outputs":[]}]}