{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom pathlib import Path\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    #for filename in filenames:\n        #print(os.path.join(dirname, filename))\n        pass\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-27T14:06:32.889979Z","iopub.execute_input":"2022-12-27T14:06:32.890412Z","iopub.status.idle":"2022-12-27T14:07:35.930015Z","shell.execute_reply.started":"2022-12-27T14:06:32.890373Z","shell.execute_reply":"2022-12-27T14:07:35.929028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -Uq fastai","metadata":{"execution":{"iopub.status.busy":"2022-12-26T17:36:05.033367Z","iopub.execute_input":"2022-12-26T17:36:05.033878Z","iopub.status.idle":"2022-12-26T17:36:24.717156Z","shell.execute_reply.started":"2022-12-26T17:36:05.033836Z","shell.execute_reply":"2022-12-26T17:36:24.715825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install fastkaggle","metadata":{"execution":{"iopub.status.busy":"2022-12-26T17:36:24.721194Z","iopub.execute_input":"2022-12-26T17:36:24.725285Z","iopub.status.idle":"2022-12-26T17:36:41.952206Z","shell.execute_reply.started":"2022-12-26T17:36:24.725243Z","shell.execute_reply":"2022-12-26T17:36:41.950889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -Uq timm","metadata":{"execution":{"iopub.status.busy":"2022-12-27T12:13:09.624442Z","iopub.execute_input":"2022-12-27T12:13:09.624896Z","iopub.status.idle":"2022-12-27T12:13:27.389629Z","shell.execute_reply.started":"2022-12-27T12:13:09.624856Z","shell.execute_reply":"2022-12-27T12:13:27.388327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\n\nimport os\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedShuffleSplit\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import StratifiedKFold\nfrom dataclasses import dataclass, field\nfrom collections import Counter\nimport shutil\nfrom typing import List, Tuple","metadata":{"execution":{"iopub.status.busy":"2022-12-27T12:21:37.58032Z","iopub.execute_input":"2022-12-27T12:21:37.580794Z","iopub.status.idle":"2022-12-27T12:21:37.588781Z","shell.execute_reply.started":"2022-12-27T12:21:37.580753Z","shell.execute_reply":"2022-12-27T12:21:37.587843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"comp = 'rsna-breast-cancer-detection'\n#path = setup_comp(comp, install='fastai \"timm>=0.6.2.dev0\"')","metadata":{"execution":{"iopub.status.busy":"2022-12-27T12:21:39.15386Z","iopub.execute_input":"2022-12-27T12:21:39.154291Z","iopub.status.idle":"2022-12-27T12:21:39.159222Z","shell.execute_reply.started":"2022-12-27T12:21:39.154254Z","shell.execute_reply":"2022-12-27T12:21:39.158247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import timm\nfrom fastai.imports import *\nfrom fastai.vision.all import *\n\nrnd_seed = 42\nn_s = 5\nset_seed(rnd_seed)\nset_seed(42)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = Path(r'/kaggle/input/')\nsymlink_path = Path(r'/kaggle/input/rsna-small-dataset-for-faster-experimentation')\nsymlink_path.ls()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T14:04:02.102071Z","iopub.execute_input":"2022-12-27T14:04:02.102966Z","iopub.status.idle":"2022-12-27T14:04:02.111603Z","shell.execute_reply.started":"2022-12-27T14:04:02.102918Z","shell.execute_reply":"2022-12-27T14:04:02.110181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"- The problem is that, with all the images it takes almost 1 hour for each epoch\n- Therefore need a subset of data for faster experimentation.\n- Possitive image number is very less. So I tried to create the same distribution of in the newer smaller dataset","metadata":{}},{"cell_type":"markdown","source":"- Split the data stratified way and take desired amount into training data\n- Then use those training data\n- We can create symbolic link to those subset of the data.","metadata":{}},{"cell_type":"code","source":"#| export\n@dataclass\nclass StratifyYSplitData:\n\n    X:str # column name for X\n    Y:str # column name for Y which needs to be stratified\n    df_t:pd.DataFrame=field(repr=False) # dataframe with all images and context data\n    train_path:Path=field(repr=False) # path to train images\n    shuffle_d:bool=True # shuffle data or not\n    random_state_:int=42 # random state\n    test_size:float=0.2 # test size\n\n    def __post_init__(self):\n        self.x_trn,  self.x_tst, self.y_trn, self.y_tst = train_test_split(\n                                                                         self.df_t[self.X].values,\n                                                                         self.df_t[self.Y].values,\n                                                                         test_size=self.test_size,\n                                                                         stratify=self.df_t[self.Y].values\n                                                                         )\n    def get_data(\n                    self,\n                    )->Tuple[List[str], List[int], List[str], List[int]]:\n        trn_img = [self.train_path/str(i) for i in self.x_trn]\n        trn_lbl = self.y_trn\n        val_img = [self.train_path/str(i) for i in self.x_tst]\n        val_lbl = self.y_tst\n\n        return (trn_img, trn_lbl), (val_img, val_lbl)\n\n    def __getitem__(self, idx:int)->Tuple[List[str], List[int]]:\n        if idx == 0:\n            (trn_img, trn_lbl), (_, _) = self.get_data()\n            return trn_img, trn_lbl\n        else:\n            (_, _), (val_img, val_lbl) = self.get_data()\n            return val_img, val_lbl\n    \n    def get_label_dict(\n                       self,\n                       data_list:str,\n                       image_list:List[Path]\n                       )->Dict[int, str]:\n        \n        if data_list == 'train':\n            x, y = self[0]\n            #print(x)\n            return {i:j for i,j in zip(x,y)}\n\n        elif data_list == 'valid':\n            x, y = self[1]\n            return {i:j for i,j in zip(x, y)}\n\n        else:\n            root_path = image_list[0].parent\n            im_name_list = [Path(i).name for i in image_list]\n\n            x,y = self[0]\n            actual_dic =  {i.name:j for i,j in zip(x, y) if Path(i).name in im_name_list}\n            return {k:v for k,v in actual_dic.items()}","metadata":{"execution":{"iopub.status.busy":"2022-12-27T12:21:45.037424Z","iopub.execute_input":"2022-12-27T12:21:45.037915Z","iopub.status.idle":"2022-12-27T12:21:45.059039Z","shell.execute_reply.started":"2022-12-27T12:21:45.037871Z","shell.execute_reply":"2022-12-27T12:21:45.058029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#| export \ndef process_trn_df(df_train):\n    cols = ['image_name', 'cancer', 'laterality']\n    return (\n        df_train\n        .assign(image_name=lambda df1: df1['patient_id'].astype(str) + \"_\"+ df1['image_id'].astype(str) +\".png\")\n        .loc[:,cols]\n    )","metadata":{"execution":{"iopub.status.busy":"2022-12-27T14:13:19.879348Z","iopub.execute_input":"2022-12-27T14:13:19.879829Z","iopub.status.idle":"2022-12-27T14:13:19.888584Z","shell.execute_reply.started":"2022-12-27T14:13:19.87979Z","shell.execute_reply":"2022-12-27T14:13:19.887363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train_path = Path(r'/kaggle/input/rsna-breast-cancer-detection-poi-images/bc_768_roi/bc_768_roi/train')\ndata_symlink_path = Path(r'/kaggle/input/rsna_small_dataset_for_faster_experimentation')","metadata":{"execution":{"iopub.status.busy":"2022-12-27T14:08:31.859804Z","iopub.execute_input":"2022-12-27T14:08:31.860178Z","iopub.status.idle":"2022-12-27T14:08:31.86533Z","shell.execute_reply.started":"2022-12-27T14:08:31.860147Z","shell.execute_reply":"2022-12-27T14:08:31.864008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_root = Path(r'/kaggle/input/rsna-breast-cancer-detection')\npath_root.ls()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T14:12:04.895128Z","iopub.execute_input":"2022-12-27T14:12:04.895511Z","iopub.status.idle":"2022-12-27T14:12:04.9083Z","shell.execute_reply.started":"2022-12-27T14:12:04.895465Z","shell.execute_reply":"2022-12-27T14:12:04.907173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(path_root/'train.csv')\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T14:12:52.919339Z","iopub.execute_input":"2022-12-27T14:12:52.919718Z","iopub.status.idle":"2022-12-27T14:12:52.997233Z","shell.execute_reply.started":"2022-12-27T14:12:52.919684Z","shell.execute_reply":"2022-12-27T14:12:52.996362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_t = process_trn_df(df_train)\ndf_t.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T14:13:50.988608Z","iopub.execute_input":"2022-12-27T14:13:50.989808Z","iopub.status.idle":"2022-12-27T14:13:51.085556Z","shell.execute_reply.started":"2022-12-27T14:13:50.989761Z","shell.execute_reply":"2022-12-27T14:13:51.084596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train_path.ls()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T14:08:49.587598Z","iopub.execute_input":"2022-12-27T14:08:49.588303Z","iopub.status.idle":"2022-12-27T14:08:49.885394Z","shell.execute_reply.started":"2022-12-27T14:08:49.588267Z","shell.execute_reply":"2022-12-27T14:08:49.884388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strY = StratifyYSplitData(\n                        X='image_name', \n                        Y='cancer',\n                        df_t=df_t,\n                        train_path=data_train_path,\n                        test_size=0.9)\n\n(xtrn, ytrn), (xval, yval) = strY.get_data()\n\nprint(f' Training size = {len(xtrn)},  and validation size = {len(xval)}')\n","metadata":{"execution":{"iopub.status.busy":"2022-12-27T14:14:18.733348Z","iopub.execute_input":"2022-12-27T14:14:18.733729Z","iopub.status.idle":"2022-12-27T14:14:19.024171Z","shell.execute_reply.started":"2022-12-27T14:14:18.733696Z","shell.execute_reply":"2022-12-27T14:14:19.023133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_small_dataset(\n                    destination_=data_symlink_path,\n                    image_list=xtrn,\n                    symlink_=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`create_small_dataset` function should create a dataset which is same distribution as original dataset.\n- I have tried in windows subsystem in linux, so in linux it should work. In actual windows, I guess one need to activate deveoper mode and also need root access.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}}]}