{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Cancer🔬 Classif: Random preds with the same ⚖️ distrib.\n\nWe will make random label predition just with honoring the same label distribution as the trainign dataset...","metadata":{}},{"cell_type":"code","source":"import os, glob\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nDATASET_FOLDER = \"/kaggle/input/UBC-OCEAN/\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-07T16:27:44.131721Z","iopub.execute_input":"2023-10-07T16:27:44.132135Z","iopub.status.idle":"2023-10-07T16:27:44.539359Z","shell.execute_reply.started":"2023-10-07T16:27:44.132104Z","shell.execute_reply":"2023-10-07T16:27:44.538168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## train & test data","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(os.path.join(DATASET_FOLDER, \"train.csv\"))\n# labels = list(df_train[\"label\"].unique())\nprint(f\"Dataset/train size: {len(df_train)}\")\ndisplay(df_train.head())","metadata":{"execution":{"iopub.status.busy":"2023-10-07T16:27:44.54093Z","iopub.execute_input":"2023-10-07T16:27:44.541329Z","iopub.status.idle":"2023-10-07T16:27:44.582186Z","shell.execute_reply.started":"2023-10-07T16:27:44.541303Z","shell.execute_reply":"2023-10-07T16:27:44.581123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(os.path.join(DATASET_FOLDER, \"test.csv\"))\n# labels = list(df_train[\"label\"].unique())\nprint(f\"Dataset/test size: {len(df_test)}\")\ndisplay(df_test.head())","metadata":{"execution":{"iopub.status.busy":"2023-10-07T16:27:44.583709Z","iopub.execute_input":"2023-10-07T16:27:44.584656Z","iopub.status.idle":"2023-10-07T16:27:44.600085Z","shell.execute_reply.started":"2023-10-07T16:27:44.584592Z","shell.execute_reply":"2023-10-07T16:27:44.599147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sample submission","metadata":{}},{"cell_type":"code","source":"df_sub = pd.read_csv(os.path.join(DATASET_FOLDER, \"sample_submission.csv\"))\nprint(f\"Submission size: {len(df_sub)}\")\ndisplay(df_sub.head())","metadata":{"execution":{"iopub.status.busy":"2023-10-07T16:27:44.601382Z","iopub.execute_input":"2023-10-07T16:27:44.601776Z","iopub.status.idle":"2023-10-07T16:27:44.619061Z","shell.execute_reply.started":"2023-10-07T16:27:44.601747Z","shell.execute_reply":"2023-10-07T16:27:44.618218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate dummy submission\n\nRandomly select labels from train dataset so they shall have the same distribution","metadata":{}},{"cell_type":"code","source":"df_test[\"label\"] = np.random.choice(df_train[\"label\"].to_list(), len(df_test))\ndisplay(df_test.head())","metadata":{"execution":{"iopub.status.busy":"2023-10-07T16:27:44.620923Z","iopub.execute_input":"2023-10-07T16:27:44.621836Z","iopub.status.idle":"2023-10-07T16:27:44.635034Z","shell.execute_reply.started":"2023-10-07T16:27:44.621805Z","shell.execute_reply":"2023-10-07T16:27:44.634221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(ncols=2, figsize=(8, 5))\n_= df_train[[\"label\"]].value_counts().plot.pie(ax=axes[0], autopct='%1.1f%%', ylabel=\"label\", title=\"Training\")\n_= df_test[[\"label\"]].value_counts().plot.pie(ax=axes[1], autopct='%1.1f%%', ylabel=\"label\", title=\"Testing\")","metadata":{"execution":{"iopub.status.busy":"2023-10-07T16:27:44.636244Z","iopub.execute_input":"2023-10-07T16:27:44.636556Z","iopub.status.idle":"2023-10-07T16:27:44.944819Z","shell.execute_reply.started":"2023-10-07T16:27:44.636529Z","shell.execute_reply":"2023-10-07T16:27:44.943387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save submision 💾","metadata":{}},{"cell_type":"code","source":"df_test[[\"image_id\", \"label\"]].to_csv(\"submission.csv\", index=False)\n\n! head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-10-07T16:27:44.946536Z","iopub.execute_input":"2023-10-07T16:27:44.947881Z","iopub.status.idle":"2023-10-07T16:27:46.072256Z","shell.execute_reply.started":"2023-10-07T16:27:44.947834Z","shell.execute_reply":"2023-10-07T16:27:46.070749Z"},"trusted":true},"execution_count":null,"outputs":[]}]}