{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, glob\nimport datetime\nimport gc\nimport torch\nfrom torch.optim import Adam, lr_scheduler\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision.transforms import PILToTensor, ToPILImage\nfrom torch.utils.tensorboard import SummaryWriter\n# from iterstrat.ml_stratifiers import MultilabelStratifiedKFold\nfrom sklearn.metrics import roc_auc_score, roc_curve, RocCurveDisplay\nfrom sklearn.metrics import ConfusionMatrixDisplay, PrecisionRecallDisplay\nfrom sklearn.metrics import classification_report\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm, trange\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nImage.MAX_IMAGE_PIXELS = None\n%load_ext tensorboard","metadata":{"executionInfo":{"elapsed":68,"status":"ok","timestamp":1661773516438,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"KxMTtJ6v0tp5","outputId":"f752848a-eb7d-434a-a81e-02bf8769608c","execution":{"iopub.status.busy":"2022-10-13T21:17:53.551849Z","iopub.execute_input":"2022-10-13T21:17:53.552156Z","iopub.status.idle":"2022-10-13T21:17:57.298425Z","shell.execute_reply.started":"2022-10-13T21:17:53.552079Z","shell.execute_reply":"2022-10-13T21:17:57.297242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !chmod 600 ../input/tabular-data/kaggle.json\nos.environ['KAGGLE_CONFIG_DIR'] = '../input/tabular-data'","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:17:57.307471Z","iopub.execute_input":"2022-10-13T21:17:57.312099Z","iopub.status.idle":"2022-10-13T21:17:57.325051Z","shell.execute_reply.started":"2022-10-13T21:17:57.312061Z","shell.execute_reply":"2022-10-13T21:17:57.324056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if torch.cuda.is_available():\n  torch.set_default_tensor_type(torch.cuda.FloatTensor)\n  print('using cuda:', torch.cuda.get_device_name(0))\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"executionInfo":{"elapsed":62,"status":"ok","timestamp":1661773516439,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"m2sbEo2wdAgg","outputId":"42a0710d-1841-4c2b-e155-51aa6ab4c504","execution":{"iopub.status.busy":"2022-10-13T21:17:57.326581Z","iopub.execute_input":"2022-10-13T21:17:57.327312Z","iopub.status.idle":"2022-10-13T21:17:57.414229Z","shell.execute_reply.started":"2022-10-13T21:17:57.327258Z","shell.execute_reply":"2022-10-13T21:17:57.412847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' Provided that the sklearn cross validators with stratification don't offer \nthe ability to stratify multilabel data, this notebook includes the multilabel \ncross validator MultilabelStratifiedKFold copied from the iterative-stratification \nproject implementation of the Iterative Stratification algorithm described in \nthe following paper (as cited by the authors of the aforementioned project):\nSechidis K., Tsoumakas G., Vlahavas I. (2011) On the Stratification of Multi-\nLabel Data. In: Gunopulos D., Hofmann T., Malerba D., Vazirgiannis M. (eds)\nMachine Learning and Knowledge Discovery in Databases. ECML PKDD 2011. Lecture\nNotes in Computer Science, vol 6913. Springer, Berlin, Heidelberg.\n\nAttribution to authors of iterative-stratification/iterstrat/ml_stratifiers.py \nunder BSD 3 clause:\n\nCopyright (c) 2018, Trent J. Bradberry\nAll rights reserved.\n\nRedistribution and use in source and binary forms, with or without\nmodification, are permitted provided that the following conditions are met:\n\n* Redistributions of source code must retain the above copyright notice, this\n  list of conditions and the following disclaimer.\n\n* Redistributions in binary form must reproduce the above copyright notice,\n  this list of conditions and the following disclaimer in the documentation\n  and/or other materials provided with the distribution.\n\n* Neither the name of the copyright holder nor the names of its\n  contributors may be used to endorse or promote products derived from\n  this software without specific prior written permission.\n\nTHIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS \"AS IS\"\nAND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE\nIMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE\nDISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE\nFOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL\nDAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR\nSERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER\nCAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,\nOR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE\nOF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.\n'''\n\nfrom sklearn.utils import check_random_state\nfrom sklearn.utils.validation import check_array\nfrom sklearn.utils.multiclass import type_of_target\nfrom sklearn.model_selection._split import _BaseKFold\n\ndef IterativeStratification(labels, r, random_state):\n    \"\"\"This function implements the Iterative Stratification algorithm described\n    in the following paper:\n    Sechidis K., Tsoumakas G., Vlahavas I. (2011) On the Stratification of\n    Multi-Label Data. In: Gunopulos D., Hofmann T., Malerba D., Vazirgiannis M.\n    (eds) Machine Learning and Knowledge Discovery in Databases. ECML PKDD\n    2011. Lecture Notes in Computer Science, vol 6913. Springer, Berlin,\n    Heidelberg.\n    \"\"\"\n\n    n_samples = labels.shape[0]\n    test_folds = np.zeros(n_samples, dtype=int)\n\n    # Calculate the desired number of examples at each subset\n    c_folds = r * n_samples\n\n    # Calculate the desired number of examples of each label at each subset\n    c_folds_labels = np.outer(r, labels.sum(axis=0))\n\n    labels_not_processed_mask = np.ones(n_samples, dtype=bool)\n\n    while np.any(labels_not_processed_mask):\n        # Find the label with the fewest (but at least one) remaining examples,\n        # breaking ties randomly\n        num_labels = labels[labels_not_processed_mask].sum(axis=0)\n\n        # Handle case where only all-zero labels are left by distributing\n        # across all folds as evenly as possible (not in original algorithm but\n        # mentioned in the text). (By handling this case separately, some\n        # code redundancy is introduced; however, this approach allows for\n        # decreased execution time when there are a relatively large number\n        # of all-zero labels.)\n        if num_labels.sum() == 0:\n            sample_idxs = np.where(labels_not_processed_mask)[0]\n\n            for sample_idx in sample_idxs:\n                fold_idx = np.where(c_folds == c_folds.max())[0]\n\n                if fold_idx.shape[0] > 1:\n                    fold_idx = fold_idx[random_state.choice(fold_idx.shape[0])]\n\n                test_folds[sample_idx] = fold_idx\n                c_folds[fold_idx] -= 1\n\n            break\n\n        label_idx = np.where(num_labels == num_labels[np.nonzero(num_labels)].min())[0]\n        if label_idx.shape[0] > 1:\n            label_idx = label_idx[random_state.choice(label_idx.shape[0])]\n\n        sample_idxs = np.where(np.logical_and(labels[:, label_idx].flatten(), labels_not_processed_mask))[0]\n\n        for sample_idx in sample_idxs:\n            # Find the subset(s) with the largest number of desired examples\n            # for this label, breaking ties by considering the largest number\n            # of desired examples, breaking further ties randomly\n            label_folds = c_folds_labels[:, label_idx]\n            fold_idx = np.where(label_folds == label_folds.max())[0]\n\n            if fold_idx.shape[0] > 1:\n                temp_fold_idx = np.where(c_folds[fold_idx] ==\n                                         c_folds[fold_idx].max())[0]\n                fold_idx = fold_idx[temp_fold_idx]\n\n                if temp_fold_idx.shape[0] > 1:\n                    fold_idx = fold_idx[random_state.choice(temp_fold_idx.shape[0])]\n\n            test_folds[sample_idx] = fold_idx\n            labels_not_processed_mask[sample_idx] = False\n\n            # Update desired number of examples\n            c_folds_labels[fold_idx, labels[sample_idx]] -= 1\n            c_folds[fold_idx] -= 1\n\n    return test_folds\n\n\nclass MultilabelStratifiedKFold(_BaseKFold):\n    \"\"\"Multilabel stratified K-Folds cross-validator\n    Provides train/test indices to split multilabel data into train/test sets.\n    This cross-validation object is a variation of KFold that returns\n    stratified folds for multilabel data. The folds are made by preserving\n    the percentage of samples for each label.\n    Parameters\n    ----------\n    n_splits : int, default=3\n        Number of folds. Must be at least 2.\n    shuffle : boolean, optional\n        Whether to shuffle each stratification of the data before splitting\n        into batches.\n    random_state : int, RandomState instance or None, optional, default=None\n        If int, random_state is the seed used by the random number generator;\n        If RandomState instance, random_state is the random number generator;\n        If None, the random number generator is the RandomState instance used\n        by `np.random`. Unlike StratifiedKFold that only uses random_state\n        when ``shuffle`` == True, this multilabel implementation\n        always uses the random_state since the iterative stratification\n        algorithm breaks ties randomly.\n    Examples\n    --------\n    >>> from iterstrat.ml_stratifiers import MultilabelStratifiedKFold\n    >>> import numpy as np\n    >>> X = np.array([[1,2], [3,4], [1,2], [3,4], [1,2], [3,4], [1,2], [3,4]])\n    >>> y = np.array([[0,0], [0,0], [0,1], [0,1], [1,1], [1,1], [1,0], [1,0]])\n    >>> mskf = MultilabelStratifiedKFold(n_splits=2, random_state=0)\n    >>> mskf.get_n_splits(X, y)\n    2\n    >>> print(mskf)  # doctest: +NORMALIZE_WHITESPACE\n    MultilabelStratifiedKFold(n_splits=2, random_state=0, shuffle=False)\n    >>> for train_index, test_index in mskf.split(X, y):\n    ...    print(\"TRAIN:\", train_index, \"TEST:\", test_index)\n    ...    X_train, X_test = X[train_index], X[test_index]\n    ...    y_train, y_test = y[train_index], y[test_index]\n    TRAIN: [0 3 4 6] TEST: [1 2 5 7]\n    TRAIN: [1 2 5 7] TEST: [0 3 4 6]\n    Notes\n    -----\n    Train and test sizes may be slightly different in each fold.\n    See also\n    --------\n    RepeatedMultilabelStratifiedKFold: Repeats Multilabel Stratified K-Fold\n    n times.\n    \"\"\"\n\n    def __init__(self, n_splits=3, *, shuffle=False, random_state=None):\n        super(MultilabelStratifiedKFold, self).__init__(n_splits=n_splits, shuffle=shuffle, random_state=random_state)\n\n    def _make_test_folds(self, X, y):\n        y = np.asarray(y, dtype=bool)\n        type_of_target_y = type_of_target(y)\n\n        if type_of_target_y != 'multilabel-indicator':\n            raise ValueError(\n                'Supported target type is: multilabel-indicator. Got {!r} instead.'.format(type_of_target_y))\n\n        num_samples = y.shape[0]\n\n        rng = check_random_state(self.random_state)\n        indices = np.arange(num_samples)\n\n        if self.shuffle:\n            rng.shuffle(indices)\n            y = y[indices]\n\n        r = np.asarray([1 / self.n_splits] * self.n_splits)\n\n        test_folds = IterativeStratification(labels=y, r=r, random_state=rng)\n\n        return test_folds[np.argsort(indices)]\n\n    def _iter_test_masks(self, X=None, y=None, groups=None):\n        test_folds = self._make_test_folds(X, y)\n        for i in range(self.n_splits):\n            yield test_folds == i\n\n    def split(self, X, y, groups=None):\n        \"\"\"Generate indices to split data into training and test set.\n        Parameters\n        ----------\n        X : array-like, shape (n_samples, n_features)\n            Training data, where n_samples is the number of samples\n            and n_features is the number of features.\n            Note that providing ``y`` is sufficient to generate the splits and\n            hence ``np.zeros(n_samples)`` may be used as a placeholder for\n            ``X`` instead of actual training data.\n        y : array-like, shape (n_samples, n_labels)\n            The target variable for supervised learning problems.\n            Multilabel stratification is done based on the y labels.\n        groups : object\n            Always ignored, exists for compatibility.\n        Returns\n        -------\n        train : ndarray\n            The training set indices for that split.\n        test : ndarray\n            The testing set indices for that split.\n        Notes\n        -----\n        Randomized CV splitters may return different results for each call of\n        split. You can make the results identical by setting ``random_state``\n        to an integer.\n        \"\"\"\n        y = check_array(y, ensure_2d=False, dtype=None)\n        return super(MultilabelStratifiedKFold, self).split(X, y, groups)","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:17:57.418084Z","iopub.execute_input":"2022-10-13T21:17:57.422334Z","iopub.status.idle":"2022-10-13T21:17:57.468647Z","shell.execute_reply.started":"2022-10-13T21:17:57.422296Z","shell.execute_reply":"2022-10-13T21:17:57.46765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Helper functions\n\ndef mskf(df, n_classes):\n  ''' Returns a multilevel stratified k-fold from a given dataframe'''\n  \n  mskf = MultilabelStratifiedKFold(n_splits=10, shuffle=True, random_state=0)\n  X, y = df.iloc[:, :-n_classes], df.iloc[:, -n_classes:]\n  train_index, test_index = next(mskf.split(X, y))\n  df_train = df.loc[X.iloc[train_index].index]\n  df_test = df.loc[X.iloc[test_index].index]\n\n  return df_train, df_test\n\n\ndef sanitize_split(df_train, df_test, n_classes):\n  ''' Returns new train and test sets without any patient in common '''\n\n  df_test_common = df_test[df_test['patient_id'].isin(df_train['patient_id'])]\n\n  # Computes how many categories we need to import from the train examples\n  labels = df_test_common.iloc[:, -n_classes:].sum()\n\n  # Looks for examples in the train set for every label and stores their indices\n  train2test_idx = pd.Index([])\n  for label, num_examples in zip(labels.index, labels.values.astype(int)):\n    no_common = ~df_train['patient_id'].isin(df_test_common['patient_id'])\n    same_label = df_train['label'] == label\n    unrepeated = ~df_train['patient_id'].duplicated(keep=False)\n    new_index = df_train[no_common & same_label & unrepeated] \\\n                .sample(num_examples, random_state=0).index\n    train2test_idx = train2test_idx.union(new_index)\n\n  # Adds to every set the examples coming from the other\n  df_test = pd.concat([df_test, df_train.loc[train2test_idx]])\n  df_train = pd.concat([df_train, df_test_common])\n\n  # Drops from every set the samples transferred to the other\n  df_test = df_test.drop(index=df_test_common.index)\n  df_train = df_train.drop(index=train2test_idx)\n\n  # Sorts the index of every set\n  df_test = df_test.sort_index()\n  df_train = df_train.sort_index()\n\n  # Checks the sets doesn't share any common patient\n  assert df_test['patient_id'].isin(df_train['patient_id']).sum() == 0\n\n  return df_train, df_test\n\n\ndef show_distribution(df_train, df_test, n_classes):\n  ''' Shows the distribution of the classes in the train and test sets '''\n\n  train_distr = df_train.iloc[:, -n_classes:].sum()\n  test_distr = df_test.iloc[:, -n_classes:].sum()\n  distr = pd.concat([train_distr, test_distr], axis=1)\n  distr.columns = ['train', 'test']\n  print('Final distribution in the train and test sets in %:\\n')\n  print((distr / distr.sum() * 100).round(1))\n  print('\\n\\nPercentage of examples in the new test set: %.1f' \\\n        % (len(df_test) / len(df_train) * 100))\n  \n\ndef img2tensor(img_path):\n  return PILToTensor()(Image.open(img_path))\n\n\ndef tensor2img(img_pt):\n  return ToPILImage()(img_pt)\n\n\ndef class_weights(df_train, n_classes):\n  ''' Returns the class weights to feed in the loss function '''\n  \n  labels = df_train.iloc[:, -n_classes:]\n  weight = 1.0 / torch.tensor(labels.sum())\n  weight /= weight.sum()\n  return weight\n\n\ndef roc_auc(model, df_test):\n  ''' Returns the targets, the preds and the ROC-AUC score on the test set '''\n\n  preds = model.predict_proba(df_test)  \n  labels = df_test.iloc[:, -model.n_classes:]\n  support = labels.sum(axis=0)           \n  weight_dict = {k: v for k, v in zip(labels.columns, support)}\n  weights = df_test['label'].map(weight_dict).values\n  ras = roc_auc_score(labels, preds, average='weighted', sample_weight=weights, \n                      multi_class='ovr')\n  return ras\n\n\ndef clf_report(model, df_test, labels, preds):\n  ''' Displays the sklearn classification report '''\n\n  y_true = labels.values.argmax(axis=1)\n  y_pred = preds.argmax(axis=1)\n  target_names = df_test.columns[-model.n_classes:]\n  print(classification_report(y_true, y_pred, target_names=target_names))\n\n\ndef conf_matrix(model, df_test, labels, preds):\n  ''' Plots the confusion matrix '''\n\n  np.set_printoptions(precision=2)\n  y_true = labels.values.argmax(axis=1)\n  y_pred = preds.argmax(axis=1)\n  target_names=df_test.columns[-model.n_classes:]\n\n  titles_options = [(\"Confusion matrix without normalization\", None),\n                    (\"Normalized confusion matrix\", 'true')]\n\n  for title, normalize in titles_options:\n    disp = ConfusionMatrixDisplay.from_predictions(y_true, y_pred,\n                                                   display_labels=target_names,\n                                                   cmap=plt.cm.Blues,\n                                                   normalize=normalize)\n    disp.ax_.set_title(title)\n  plt.show()\n\n\ndef transform_image(src, dst, max_dim=512):\n  ''' Resizes and pads an image and saves the result '''\n  with Image.open(src) as im:\n    size = np.array(im.size) \n    size = (size / size.max() * max_dim).astype(int)   \n    if size[0] > size[1]:\n      padding_right = 0\n      padding_bottom = max_dim - size[1]\n    else:\n      padding_right = max_dim - size[0]\n      padding_bottom = 0    \n    im.resize(size).save(dst)\n  del im\n  gc.collect()\n  im_pt = img2tensor(dst)\n  im_pt = nn.ZeroPad2d((0, padding_right, 0, padding_bottom))(im_pt)\n  tensor2img(im_pt).save(dst)\n  return\n\n\ndef transform_images(src, dst, max_input_size= 2., max_dim=512):\n  '''\n  Transforms the selected images in src and saves the transformed ones in dst.\n  The selection criterion is the image size to be less than max_input_size Gb.\n  '''\n  src = glob.glob(os.path.join(src, '*.tif'))\n  bytes_Gb = pow(1024, 3)\n  for image in tqdm(src):\n    if os.path.getsize(image) < max_input_size * bytes_Gb:\n      gc.collect()\n      dest = os.path.join(dst, os.path.splitext(image)[0][-8:] + '.png')\n      transform_image(image, dest, max_dim=max_dim)\n      gc.collect()\n  return","metadata":{"executionInfo":{"elapsed":54,"status":"ok","timestamp":1661773516440,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"sGQ9Lo1y-gNN","execution":{"iopub.status.busy":"2022-10-13T21:17:57.473262Z","iopub.execute_input":"2022-10-13T21:17:57.476084Z","iopub.status.idle":"2022-10-13T21:17:57.518109Z","shell.execute_reply.started":"2022-10-13T21:17:57.476044Z","shell.execute_reply":"2022-10-13T21:17:57.516819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(42)\ntorch.manual_seed(123456)\ntorch.cuda.manual_seed(123456)\ntorch.backends.cudnn.deterministic = True\ntorch.backends.cudnn.benchmark = False","metadata":{"executionInfo":{"elapsed":21,"status":"ok","timestamp":1661773517234,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"1WOZoqJl8Gjn","execution":{"iopub.status.busy":"2022-10-13T21:17:57.523098Z","iopub.execute_input":"2022-10-13T21:17:57.524237Z","iopub.status.idle":"2022-10-13T21:17:57.534668Z","shell.execute_reply.started":"2022-10-13T21:17:57.524196Z","shell.execute_reply":"2022-10-13T21:17:57.533544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Input directories\nimg_dir = '../input/images512/content/gdrive/MyDrive/Kaggle/mayo-clinic-strip-ai/images512'\ntest_dir = '../input/mayo-clinic-strip-ai/test'\nmodel_dir = '../input/mayo-clinic-models'\n\n# Output directories\nruns = './runs'\nif not os.path.exists(runs): os.mkdir(runs)","metadata":{"executionInfo":{"elapsed":53,"status":"ok","timestamp":1661773516441,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"3Tz2r6yV79Hu","execution":{"iopub.status.busy":"2022-10-13T21:17:57.539332Z","iopub.execute_input":"2022-10-13T21:17:57.539862Z","iopub.status.idle":"2022-10-13T21:17:57.549528Z","shell.execute_reply.started":"2022-10-13T21:17:57.539825Z","shell.execute_reply":"2022-10-13T21:17:57.548401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In order to make it simple, I've gathered in only one DataFrame all the tabular information, including:\n- an identifier of the original csv file ('train' or 'other') -column 'dataset' in the next table\n- the width and height of every image\n- a one-hot-encoded version of the label for 4 classes","metadata":{}},{"cell_type":"code","source":"df_data = pd.read_csv('../input/tabular-data/kaggle_data.csv')\ndf_data.head()","metadata":{"executionInfo":{"elapsed":51,"status":"ok","timestamp":1661773516441,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"CvBvUn1cOWtV","outputId":"0d0e83ad-4187-4e30-9c35-0f8aa15eccd2","execution":{"iopub.status.busy":"2022-10-13T21:17:57.554637Z","iopub.execute_input":"2022-10-13T21:17:57.557181Z","iopub.status.idle":"2022-10-13T21:17:57.602374Z","shell.execute_reply.started":"2022-10-13T21:17:57.557145Z","shell.execute_reply":"2022-10-13T21:17:57.601462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To make the problem tractable, I've downsized the images limiting the maximum dimension (either width or height) to 512 pixels, keeping the aspect ratio unchanged.\n\nThe images with width > height become 512 pixels wide after this downside. I've transposed them to obtain a set of equally heighted images (all of them having height = 512 and whatever width lower than 512). This is the set of images to feed the model. I've stored the set of transformed images in img_dir (see above).\n\nAs the images need to be of the same dimensions, their width will be zero-padded at the time to feed the model. The column 'padding' keeps track of this in the next DataFrame.","metadata":{}},{"cell_type":"code","source":"df_resized = pd.read_csv('../input/tabular-data/resized.csv')\ndf_resized.head()","metadata":{"executionInfo":{"elapsed":49,"status":"ok","timestamp":1661773516442,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"jAKidrx0Ead-","outputId":"93355bb4-ad3e-47e9-878a-7545f91ffdbe","execution":{"iopub.status.busy":"2022-10-13T21:17:57.606299Z","iopub.execute_input":"2022-10-13T21:17:57.608553Z","iopub.status.idle":"2022-10-13T21:17:57.635226Z","shell.execute_reply.started":"2022-10-13T21:17:57.608518Z","shell.execute_reply":"2022-10-13T21:17:57.634308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.info()","metadata":{"executionInfo":{"elapsed":47,"status":"ok","timestamp":1661773516443,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"pOFxqH2c6ioZ","outputId":"109392b9-9b02-446f-afc7-765f4a74da1c","execution":{"iopub.status.busy":"2022-10-13T21:17:57.642188Z","iopub.execute_input":"2022-10-13T21:17:57.644395Z","iopub.status.idle":"2022-10-13T21:17:57.674425Z","shell.execute_reply.started":"2022-10-13T21:17:57.644361Z","shell.execute_reply":"2022-10-13T21:17:57.673559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's merge and keep all the info we need in only one dataframe\ndf_data = pd.concat([df_data.iloc[:, :-4], df_resized['padding'], \n                     df_data.iloc[:, -4:]], axis=1)","metadata":{"executionInfo":{"elapsed":42,"status":"ok","timestamp":1661773516445,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"Qj5qmP-04pPm","execution":{"iopub.status.busy":"2022-10-13T21:17:57.678909Z","iopub.execute_input":"2022-10-13T21:17:57.681208Z","iopub.status.idle":"2022-10-13T21:17:57.69299Z","shell.execute_reply.started":"2022-10-13T21:17:57.681172Z","shell.execute_reply":"2022-10-13T21:17:57.691632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.head()","metadata":{"executionInfo":{"elapsed":43,"status":"ok","timestamp":1661773516447,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"U-yZHjHmVniT","outputId":"4a481c74-2f48-400f-9986-bafa2b3066c4","execution":{"iopub.status.busy":"2022-10-13T21:17:57.697658Z","iopub.execute_input":"2022-10-13T21:17:57.699914Z","iopub.status.idle":"2022-10-13T21:17:57.724568Z","shell.execute_reply.started":"2022-10-13T21:17:57.699879Z","shell.execute_reply":"2022-10-13T21:17:57.723834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.info()","metadata":{"id":"P_J9A8BW4Dq5","executionInfo":{"status":"ok","timestamp":1661773516451,"user_tz":-120,"elapsed":45,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"}},"outputId":"1074f765-3908-41da-88f1-991add8e0f2e","execution":{"iopub.status.busy":"2022-10-13T21:17:57.726014Z","iopub.execute_input":"2022-10-13T21:17:57.726943Z","iopub.status.idle":"2022-10-13T21:17:57.741789Z","shell.execute_reply.started":"2022-10-13T21:17:57.726906Z","shell.execute_reply":"2022-10-13T21:17:57.740717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Gets the original test set\ndf_test = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')","metadata":{"executionInfo":{"elapsed":41,"status":"ok","timestamp":1661773516453,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"v3-0L8pq3vyc","execution":{"iopub.status.busy":"2022-10-13T21:17:57.743195Z","iopub.execute_input":"2022-10-13T21:17:57.743813Z","iopub.status.idle":"2022-10-13T21:17:57.754515Z","shell.execute_reply.started":"2022-10-13T21:17:57.743741Z","shell.execute_reply":"2022-10-13T21:17:57.753057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drops from the train dataset the examples already living in the test csv file\ndf_data_new = df_data[4:].copy()\n\n# Creates a stratified k-fold from the new train set\nn_classes = 4\ndf_train, df_test = mskf(df_data_new, n_classes)\n\n# Adds the examples in the test csv file to the new test set\ndf_test = pd.concat([df_data[:4], df_test])","metadata":{"executionInfo":{"elapsed":41,"status":"ok","timestamp":1661773516454,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"kx0UD0yrCqH2","execution":{"iopub.status.busy":"2022-10-13T21:17:57.757516Z","iopub.execute_input":"2022-10-13T21:17:57.760445Z","iopub.status.idle":"2022-10-13T21:17:57.838611Z","shell.execute_reply.started":"2022-10-13T21:17:57.760403Z","shell.execute_reply":"2022-10-13T21:17:57.837561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Is there any intersection between the splitted sets?\ndf_test_common = df_test[df_test['patient_id'].isin(df_train['patient_id'])]\nprint('%d out of %d patients in the test set were found in the train set. '\n      'Percentage: %.1f' % (len(df_test_common), len(df_test),\n                            len(df_test_common) / len(df_test) * 100))","metadata":{"executionInfo":{"elapsed":43,"status":"ok","timestamp":1661773516456,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"2XlMjtDv_awy","outputId":"48848ce4-ccf6-4941-9830-90506d63ca39","execution":{"iopub.status.busy":"2022-10-13T21:17:57.842928Z","iopub.execute_input":"2022-10-13T21:17:57.843191Z","iopub.status.idle":"2022-10-13T21:17:57.853643Z","shell.execute_reply.started":"2022-10-13T21:17:57.843167Z","shell.execute_reply":"2022-10-13T21:17:57.852175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, df_test = sanitize_split(df_train, df_test, n_classes=4)","metadata":{"executionInfo":{"elapsed":39,"status":"ok","timestamp":1661773516457,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"oYALuR9KDl5i","execution":{"iopub.status.busy":"2022-10-13T21:17:57.855693Z","iopub.execute_input":"2022-10-13T21:17:57.857072Z","iopub.status.idle":"2022-10-13T21:17:57.892571Z","shell.execute_reply.started":"2022-10-13T21:17:57.857035Z","shell.execute_reply":"2022-10-13T21:17:57.891573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_distribution(df_train, df_test, n_classes)","metadata":{"executionInfo":{"elapsed":815,"status":"ok","timestamp":1661773517233,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"gdBU3gnVcFP8","outputId":"350a685c-6968-49fc-9c45-80096dc5b643","execution":{"iopub.status.busy":"2022-10-13T21:17:57.894131Z","iopub.execute_input":"2022-10-13T21:17:57.894473Z","iopub.status.idle":"2022-10-13T21:17:57.913351Z","shell.execute_reply.started":"2022-10-13T21:17:57.894439Z","shell.execute_reply":"2022-10-13T21:17:57.912443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MayoClinicDataset(Dataset):\n\n  def __init__(self, dir_path, df_data=df_data, n_classes=4, random=True):\n    self.data_df = df_data\n    self.img_dir = dir_path\n    self.n_classes = n_classes\n    self.random = random\n    \n  def __len__(self):\n    return len(self.data_df)\n\n  def __getitem__(self, index):\n    # image target (label)\n    if self.n_classes == 4:\n      target = torch.tensor(self.data_df.iloc[index, -4:] \\\n                            .values.astype('float32'))\n    elif self.n_classes == 2:\n      target = torch.tensor(self.data_df.iloc[index, -4:-2] \\\n                            .values.astype('float32'))\n    else:\n      print('Wrong number of classes')\n      return\n    \n    # image data zero-padded, normalized from 0-255 to 0-1 (and randomly \n    # transposed with 50% probability in training mode)\n    im_id = self.data_df.iloc[index, 0]\n    padding = self.data_df.iloc[index, 7]\n    im_path = os.path.join(self.img_dir, im_id + '.png')\n    im_pt = img2tensor(im_path)\n    im_pt = torch.cuda.FloatTensor(self.__transform__(im_pt, padding) / 255.0)\n    return im_pt, target\n\n  def __transform__(self, im_pt, padding):\n    ''' Pipeline of transformations: padding and transposition '''\n    im_transformed_pt = self.__pad__(im_pt, padding)\n    if self.random:\n      if bool(np.random.randint(0, 2)):\n        im_transformed_pt = torch.transpose(im_transformed_pt, 1, 2)\n    return im_transformed_pt\n\n  def __pad__(self, im_pt, padding):\n    ''' Returns a zero-padded tensor matching the standard width '''\n    if self.random:\n      if padding != 0:\n        padding_left = np.random.randint(padding)\n      else:\n        padding_left = 0\n    else:\n      padding_left = 0\n    padding_right = padding - padding_left\n    m = nn.ZeroPad2d((padding_left, padding_right, 0, 0))\n    img_pt_padded = m(im_pt)\n    return img_pt_padded\n\n\nclass SubsetMayoClinicDataset(MayoClinicDataset):\n\n  def __init__(self, dir_path, df_data, index, n_classes, random):\n    super().__init__(dir_path, df_data, n_classes, random)\n    self.data_df = self.data_df.iloc[index, :]","metadata":{"executionInfo":{"elapsed":20,"status":"ok","timestamp":1661773517235,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"HOKYbFKCcos8","execution":{"iopub.status.busy":"2022-10-13T21:17:57.917154Z","iopub.execute_input":"2022-10-13T21:17:57.919394Z","iopub.status.idle":"2022-10-13T21:17:57.934424Z","shell.execute_reply.started":"2022-10-13T21:17:57.919355Z","shell.execute_reply":"2022-10-13T21:17:57.933535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checks the dataset\nall_data = MayoClinicDataset(img_dir, df_data)\nprint('Dataset size: ', len(all_data))\nplt.imshow(tensor2img(all_data[1][0]));","metadata":{"executionInfo":{"elapsed":17,"status":"ok","timestamp":1661773517235,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"4yNshHHU1Sgl","execution":{"iopub.status.busy":"2022-10-13T21:17:57.935734Z","iopub.execute_input":"2022-10-13T21:17:57.936329Z","iopub.status.idle":"2022-10-13T21:18:01.735024Z","shell.execute_reply.started":"2022-10-13T21:17:57.936294Z","shell.execute_reply":"2022-10-13T21:18:01.733786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Model(nn.Module):\n\n  def __init__(self, n_classes, finetune=False):\n    super().__init__()\n\n    # number of classes in the train set\n    self.n_classes = n_classes\n\n    # activation function\n    if finetune:\n      self.activation = nn.GELU()\n    else:\n      self.activation = nn.GELU()\n\n    # defines neural network layers\n    self.model = nn.Sequential(\n      self.__cbl__(in_channels=3, out_channels=128, kernel_size=8, stride=8),\n      self.__cbl__(in_channels=128, out_channels=256, kernel_size=8, stride=2),\n      self.__cbl__(in_channels=256, out_channels=128, kernel_size=4, stride=1),\n      nn.Flatten(),\n      nn.Linear(26*26*128, 1000),\n      self.activation,\n      nn.Linear(1000, self.n_classes)\n    )\n\n    # creates loss function\n    weight = class_weights(df_train, self.n_classes)\n    self.loss_function = nn.CrossEntropyLoss(weight=weight, reduction='sum')\n\n    # creates optimizer\n    self.optimizer = Adam(self.parameters())\n\n    # accumulators to keep track of losses\n    self.train_losses = []\n    self.valid_losses = []\n\n    # initial values for the early_stopping callback use\n    self.patience = None\n    self.min_valid_loss = np.inf\n    self.best_epoch = 0\n    self.best_model = None\n\n\n  def __cbl__(self, *args,**kwargs):\n    ''' Builds the basic block of the net '''\n    return nn.Sequential(nn.Conv2d(*args,**kwargs),\n                         nn.BatchNorm2d(kwargs['out_channels']),\n                         self.activation,)\n                         # nn.Dropout(p=0.4))\n    \n\n  def forward(self, inputs):\n    return self.model(inputs)\n\n\n  def train_model(self, epochs, patience=None, bestmodel=True):\n\n    self.patience = patience\n    np.random.seed(42)\n    writer = SummaryWriter()\n\n    for epoch in range(epochs):\n      # Creates 2 lists to keep track of the train and validation losses\n      train_losses_cv = []\n      valid_losses_cv = []\n      \n      # Random iterative stratification\n      random_state = np.random.randint(0, 10000)\n      mskf = MultilabelStratifiedKFold(n_splits=5, shuffle=True, \n                                       random_state=random_state)\n      X = df_train.iloc[:, :-self.n_classes]\n      y = df_train.iloc[:, -self.n_classes:]\n\n      # Cross-validation loop\n      for train_index, valid_index in mskf.split(X, y):\n\n        train_subset = SubsetMayoClinicDataset(img_dir, df_data, train_index,\n                                               self.n_classes, random=True)\n        valid_subset = SubsetMayoClinicDataset(img_dir, df_data, valid_index,\n                                               self.n_classes, random=True)\n        trainloader = DataLoader(train_subset, \n                                 batch_size=batch_size,\n                                 shuffle=True,\n                                 generator=torch.Generator(device='cuda'))\n        validloader = DataLoader(valid_subset, \n                                 batch_size=batch_size,\n                                 shuffle=True,\n                                 generator=torch.Generator(device='cuda'))\n        scheduler = lr_scheduler.OneCycleLR(self.optimizer, max_lr=0.01,\n                                            steps_per_epoch=len(trainloader), \n                                            epochs=epochs, base_momentum=0.9)\n        train_size_cv = len(train_index)\n        valid_size_cv = len(valid_index)\n        train_loss = 0.0\n        valid_loss = 0.0\n\n        self.model.train()\n\n        # Training loop\n        for inputs, targets in trainloader:\n\n          # calculates the output of the network\n          outputs = self.forward(inputs)\n\n          # calculates and accumulates the loss\n          loss = self.loss_function(outputs, targets)\n          train_loss += loss.item()\n\n          # Zeroes the gradients and performs a backward pass \n          self.optimizer.zero_grad()\n          loss.backward()\n\n          # Updates the weights and the learning rate\n          self.optimizer.step()\n          scheduler.step()\n        \n        self.model.eval()\n\n        # Validation loop\n        for inputs, targets in validloader:\n          outputs = self.forward(inputs)\n          loss = self.loss_function(outputs, targets)\n          valid_loss += loss.item()\n        \n        # Stores the train and validation losses in the folder\n        train_losses_cv.append(train_loss/train_size_cv)\n        valid_losses_cv.append(valid_loss/valid_size_cv)           \n      \n      # Averages the train and validation losses from the k-folders\n      train_loss = np.array(train_losses_cv).mean()\n      valid_loss = np.array(valid_losses_cv).mean()\n\n      # Stores the average train and validation losses\n      self.train_losses.append(train_loss)\n      self.valid_losses.append(valid_loss)\n\n      print('Epoch %03d/%d:  Train loss: %.6f --- Val loss: %.6f' % \\\n            (epoch + 1, epochs, train_loss, valid_loss))\n      writer.add_scalars('Loss', \n                         {'Train': self.train_losses[-1], \n                          'Valid': self.valid_losses[-1]},\n                         epoch + 1)\n      \n      # Keeps track of the best model\n      self.__update_best_params__(epoch)\n      \n      # Early stopping callback\n      if self.__early_stop__(epoch): break         \n\n    writer.close()\n    if bestmodel: self.load_state_dict(self.best_model, strict=True)\n    self.__print_best_epoch__()\n    self.__save_model__(runs, 'model')\n\n\n  def predict_proba(self, X):\n    ''' Returns the predicted probabilities for a set of examples '''\n\n    self.model.eval()\n    subset = SubsetMayoClinicDataset(img_dir, df_data, X.index, self.n_classes,  \n                                     random=False)\n    loader = DataLoader(subset, batch_size=len(X.index), shuffle=False)\n    inputs, _ = next(iter(loader))\n    outputs = self.model(inputs)\n    sm = nn.Softmax(1)\n    preds = sm(outputs).detach().cpu().numpy()\n    return preds\n\n\n  def __update_best_params__(self, epoch):\n    ''' Updates some attributes of the model to keep the best one's '''\n\n    valid_loss = self.valid_losses[-1]\n    if valid_loss < self.min_valid_loss:\n      self.min_valid_loss = valid_loss\n      self.best_epoch = epoch\n      self.best_model = self.state_dict()\n    return\n\n\n  def __early_stop__(self, epoch):\n    ''' Activates the early stopping functionality if the conditions are met '''\n\n    if self.patience != None:\n      if self.best_epoch == epoch:\n        return False\n      elif epoch < self.best_epoch + self.patience:\n        return False\n      else:\n        return True\n    else:\n      return False\n\n\n  def __print_best_epoch__(self):\n    ''' Prints the minimal validation loss and the epoch when reached '''\n    print('\\nMinimal validation loss: {:.6f} --> Epoch {:d}' \\\n          .format(self.min_valid_loss, self.best_epoch + 1))\n    \n\n  def __save_model__(self, dir_path, name):\n    ''' Saves the model in a given directory including datetime in its name '''\n    dt = datetime.datetime.now()\n    model_name= os.path.join(name+'_'+dt.strftime('%Y%m%d-%H%M%S')+'.pt')\n    torch.save(self.state_dict(), os.path.join(dir_path, model_name))\n    print('\\nRecorded at %d/%d/%d  %dh %d\\'' % \\\n          (dt.day, dt.month, dt.year, dt.hour, dt.minute))","metadata":{"executionInfo":{"elapsed":16,"status":"ok","timestamp":1661773517236,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"LMXOERgev91e","execution":{"iopub.status.busy":"2022-10-13T21:18:01.738041Z","iopub.execute_input":"2022-10-13T21:18:01.738755Z","iopub.status.idle":"2022-10-13T21:18:01.769883Z","shell.execute_reply.started":"2022-10-13T21:18:01.73871Z","shell.execute_reply":"2022-10-13T21:18:01.768702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Instantiates the base model\nn_classes = 4\nbasemodel = Model(n_classes=n_classes).to(device)","metadata":{"executionInfo":{"elapsed":14,"status":"ok","timestamp":1661773517236,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"l0Q7C6pCy_1U","execution":{"iopub.status.busy":"2022-10-13T21:18:01.771386Z","iopub.execute_input":"2022-10-13T21:18:01.772219Z","iopub.status.idle":"2022-10-13T21:18:01.804773Z","shell.execute_reply.started":"2022-10-13T21:18:01.772173Z","shell.execute_reply":"2022-10-13T21:18:01.803786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Trains the base model\nbatch_size = 32\nepochs = 400\npatience = 100\n# basemodel.train_model(epochs, patience)\nmodel_path = os.path.join(model_dir, 'model_20221002-135227.pt')\nbasemodel.load_state_dict(torch.load(model_path))","metadata":{"id":"5KBmUd1CyozR","executionInfo":{"status":"ok","timestamp":1661785892349,"user_tz":-120,"elapsed":12375125,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"}},"outputId":"2e185cc8-93b9-4cc0-8278-4cfd49f0dc30","execution":{"iopub.status.busy":"2022-10-13T21:18:01.80641Z","iopub.execute_input":"2022-10-13T21:18:01.806771Z","iopub.status.idle":"2022-10-13T21:19:09.472806Z","shell.execute_reply.started":"2022-10-13T21:18:01.806736Z","shell.execute_reply":"2022-10-13T21:19:09.471619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shows the ROC-AUC score on the test set\nras = roc_auc(basemodel, df_test)\nprint('ROC-AUC score: %.4f' % ras)","metadata":{"executionInfo":{"elapsed":1914,"status":"ok","timestamp":1661785895552,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"X_6lKbwQu268","outputId":"0cc296a4-6f78-4a83-81fa-a5594a0b2e81","execution":{"iopub.status.busy":"2022-10-13T21:19:09.474603Z","iopub.execute_input":"2022-10-13T21:19:09.474998Z","iopub.status.idle":"2022-10-13T21:19:10.129901Z","shell.execute_reply.started":"2022-10-13T21:19:09.47496Z","shell.execute_reply":"2022-10-13T21:19:10.128934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = df_test.iloc[:, -basemodel.n_classes:]\npreds = basemodel.predict_proba(df_test)  ","metadata":{"id":"nbdHw1_WLCEv","executionInfo":{"status":"ok","timestamp":1661785897643,"user_tz":-120,"elapsed":2102,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"}},"execution":{"iopub.status.busy":"2022-10-13T21:19:10.131131Z","iopub.execute_input":"2022-10-13T21:19:10.131795Z","iopub.status.idle":"2022-10-13T21:19:10.71307Z","shell.execute_reply.started":"2022-10-13T21:19:10.131755Z","shell.execute_reply":"2022-10-13T21:19:10.712117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_report(basemodel, df_test, labels, preds)","metadata":{"executionInfo":{"elapsed":15,"status":"ok","timestamp":1661785897647,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"eAn-0s_l7L3Q","outputId":"d4212bf3-d751-4c03-cef0-b9aee65d4105","execution":{"iopub.status.busy":"2022-10-13T21:19:10.714529Z","iopub.execute_input":"2022-10-13T21:19:10.714908Z","iopub.status.idle":"2022-10-13T21:19:10.729497Z","shell.execute_reply.started":"2022-10-13T21:19:10.71487Z","shell.execute_reply":"2022-10-13T21:19:10.728439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conf_matrix(basemodel, df_test, labels, preds)","metadata":{"executionInfo":{"elapsed":642,"status":"ok","timestamp":1661785898277,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"Bm_i7jtW8sga","outputId":"b61cb895-0f1e-4993-8d69-3a070dbcb6df","execution":{"iopub.status.busy":"2022-10-13T21:19:10.730841Z","iopub.execute_input":"2022-10-13T21:19:10.731818Z","iopub.status.idle":"2022-10-13T21:19:11.204911Z","shell.execute_reply.started":"2022-10-13T21:19:10.731766Z","shell.execute_reply":"2022-10-13T21:19:11.20402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fine-tuning the base model","metadata":{"id":"KRZ3jB6I03nT"}},{"cell_type":"code","source":"# Selects the examples to the new train and test sets\ndf_train = df_train[df_train['dataset']=='train'].iloc[:, :-2]\ndf_test = df_test[df_test['dataset']=='train'].iloc[:, :-2]","metadata":{"executionInfo":{"elapsed":36,"status":"ok","timestamp":1661785898278,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"oZg8JuJ67U8R","execution":{"iopub.status.busy":"2022-10-13T21:19:11.210363Z","iopub.execute_input":"2022-10-13T21:19:11.210638Z","iopub.status.idle":"2022-10-13T21:19:11.218875Z","shell.execute_reply.started":"2022-10-13T21:19:11.210612Z","shell.execute_reply":"2022-10-13T21:19:11.217695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_distribution(df_train, df_test, n_classes=2)","metadata":{"executionInfo":{"elapsed":34,"status":"ok","timestamp":1661785898278,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"QqSNDrl-fcox","outputId":"32fec7a0-6b59-4881-9521-d9b78155f1f0","execution":{"iopub.status.busy":"2022-10-13T21:19:11.220302Z","iopub.execute_input":"2022-10-13T21:19:11.220763Z","iopub.status.idle":"2022-10-13T21:19:11.235474Z","shell.execute_reply.started":"2022-10-13T21:19:11.220726Z","shell.execute_reply":"2022-10-13T21:19:11.234456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's get the convolutional part of the model, freeze its weights and train \n# the classifier on the top to finetune the model.","metadata":{"id":"nD5-eLSaqu83","executionInfo":{"status":"ok","timestamp":1661785898279,"user_tz":-120,"elapsed":28,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"}},"execution":{"iopub.status.busy":"2022-10-13T21:19:11.237051Z","iopub.execute_input":"2022-10-13T21:19:11.237402Z","iopub.status.idle":"2022-10-13T21:19:11.241809Z","shell.execute_reply.started":"2022-10-13T21:19:11.237369Z","shell.execute_reply":"2022-10-13T21:19:11.240702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Instantiates the fine-tuned model for 2 classes only\nn_classes = 2\nfinetuned_model = Model(n_classes=n_classes, finetune=True).to(device)\n\n# Builds a new model reusing the parameters from the pretrained base model\nfinetuned_model.model = nn.Sequential(basemodel.model[0:4],\n                                      finetuned_model.model[4:])\n\n# Freezes the weights from the pretrained model\nfinetuned_model.model.eval()\nfor param in finetuned_model.model[0].parameters():\n  param.requires_grad = False\n\n# Checks the weights are frozen\nfor i, w in enumerate(finetuned_model.parameters()):\n  print(i, w.shape, w.requires_grad)","metadata":{"executionInfo":{"elapsed":28,"status":"ok","timestamp":1661785898280,"user":{"displayName":"José Miguel Máiz","userId":"07508041429164906142"},"user_tz":-120},"id":"oZr92mjoVjf1","outputId":"4ab00389-f6e4-43e6-f976-7064ecfae3cc","execution":{"iopub.status.busy":"2022-10-13T21:19:11.243419Z","iopub.execute_input":"2022-10-13T21:19:11.244234Z","iopub.status.idle":"2022-10-13T21:19:11.258521Z","shell.execute_reply.started":"2022-10-13T21:19:11.244195Z","shell.execute_reply":"2022-10-13T21:19:11.257314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Trains the fine-tuned model\nbatch_size = 64\nepochs = 200\npatience = 80\n# finetuned_model.train_model(epochs, patience)\nmodel_path = os.path.join(model_dir, 'model_20221002-161408.pt')\nfinetuned_model.load_state_dict(torch.load(model_path))","metadata":{"id":"gMZXvF01nVKV","outputId":"963f6860-aed6-44f0-cb60-c17edb028677","execution":{"iopub.status.busy":"2022-10-13T21:19:11.259887Z","iopub.execute_input":"2022-10-13T21:19:11.260295Z","iopub.status.idle":"2022-10-13T21:19:30.421554Z","shell.execute_reply.started":"2022-10-13T21:19:11.26026Z","shell.execute_reply":"2022-10-13T21:19:30.420453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(torch.cuda.memory_summary(device=None, abbreviated=True))","metadata":{"id":"VaPff7iXCFt7","execution":{"iopub.status.busy":"2022-10-13T21:19:30.422759Z","iopub.execute_input":"2022-10-13T21:19:30.423621Z","iopub.status.idle":"2022-10-13T21:19:30.432059Z","shell.execute_reply.started":"2022-10-13T21:19:30.423588Z","shell.execute_reply":"2022-10-13T21:19:30.430964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %tensorboard --logdir runs","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:19:30.433802Z","iopub.execute_input":"2022-10-13T21:19:30.434894Z","iopub.status.idle":"2022-10-13T21:19:30.441543Z","shell.execute_reply.started":"2022-10-13T21:19:30.434856Z","shell.execute_reply":"2022-10-13T21:19:30.44051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shows the ROC-AUC score on the test set\nras = roc_auc(finetuned_model, df_test)\nprint('ROC-AUC score: %.4f' % ras)","metadata":{"id":"IJJWS0mc-Cyc","execution":{"iopub.status.busy":"2022-10-13T21:19:30.443011Z","iopub.execute_input":"2022-10-13T21:19:30.443453Z","iopub.status.idle":"2022-10-13T21:19:30.857706Z","shell.execute_reply.started":"2022-10-13T21:19:30.443416Z","shell.execute_reply":"2022-10-13T21:19:30.856631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = df_test.iloc[:, -finetuned_model.n_classes:]\npreds = finetuned_model.predict_proba(df_test)","metadata":{"id":"UlRrZTJ1LZ5E","execution":{"iopub.status.busy":"2022-10-13T21:19:30.858981Z","iopub.execute_input":"2022-10-13T21:19:30.860015Z","iopub.status.idle":"2022-10-13T21:19:31.271187Z","shell.execute_reply.started":"2022-10-13T21:19:30.859973Z","shell.execute_reply":"2022-10-13T21:19:31.270227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RocCurveDisplay.from_predictions(labels['LAA'], preds[:, 1])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:19:31.272535Z","iopub.execute_input":"2022-10-13T21:19:31.273036Z","iopub.status.idle":"2022-10-13T21:19:31.469624Z","shell.execute_reply.started":"2022-10-13T21:19:31.272997Z","shell.execute_reply":"2022-10-13T21:19:31.468717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Precision-recall curve\ny_true = labels.iloc[:, 1]\ny_pred = preds[:, 1]\ndisplay = PrecisionRecallDisplay.from_predictions(y_true, y_pred, name='fine-tuned model')\n_ = display.ax_.set_title('2-class Precision-Recall curve')","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:19:31.470868Z","iopub.execute_input":"2022-10-13T21:19:31.471218Z","iopub.status.idle":"2022-10-13T21:19:31.694986Z","shell.execute_reply.started":"2022-10-13T21:19:31.471181Z","shell.execute_reply":"2022-10-13T21:19:31.694115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_report(finetuned_model, df_test, labels, preds)","metadata":{"id":"d3d9D7P0-Kao","execution":{"iopub.status.busy":"2022-10-13T21:19:31.696407Z","iopub.execute_input":"2022-10-13T21:19:31.696762Z","iopub.status.idle":"2022-10-13T21:19:31.709644Z","shell.execute_reply.started":"2022-10-13T21:19:31.696727Z","shell.execute_reply":"2022-10-13T21:19:31.708438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conf_matrix(finetuned_model, df_test, labels, preds)","metadata":{"id":"dBoH7zGj-Kao","execution":{"iopub.status.busy":"2022-10-13T21:19:31.711265Z","iopub.execute_input":"2022-10-13T21:19:31.711611Z","iopub.status.idle":"2022-10-13T21:19:32.112868Z","shell.execute_reply.started":"2022-10-13T21:19:31.711577Z","shell.execute_reply":"2022-10-13T21:19:32.111985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saves the transformed images in the outfolder\noutfolder = './test512'\nif not os.path.exists(outfolder): os.mkdir(outfolder)\ntransform_images(test_dir, outfolder, max_input_size= 0.8)\n\n# Submission inputs\nim_list = glob.glob(os.path.join(outfolder, '*.png'))\ninputs = torch.cat([img2tensor(im_path).unsqueeze(0) \\\n                    for im_path in im_list]).to(device) / 255.\n\n# Submission output\noutputs = finetuned_model.model(inputs)\nsm = nn.Softmax(1)\npreds = pd.DataFrame({'image_id': im_list})\npreds['image_id'] = preds['image_id'].map(lambda x: os.path.splitext(x)[0][-8:])\npreds = pd.concat([preds, pd.DataFrame(sm(outputs).detach().cpu().numpy(), columns=['CE', 'LAA'])],\n                  axis=1)\n\n# Submission\ndf = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\ndf['CE'] = df['image_id'].map(preds[['image_id', 'CE']].set_index('image_id').to_dict()['CE'])\ndf['LAA'] = df['image_id'].map(preds[['image_id', 'LAA']].set_index('image_id').to_dict()['LAA'])\ndf.fillna(0.5, inplace=True)\ndf_mean = df[['patient_id', 'CE', 'LAA']].groupby('patient_id').mean()\ndf_sub = pd.read_csv('../input/mayo-clinic-strip-ai/sample_submission.csv')\ndf_sub['patient_id'] = df_mean.index.to_list()\ndf_sub['CE'] = df_mean['CE'].to_list()\ndf_sub['LAA'] = df_mean['LAA'].to_list()\ndf_sub.to_csv('./submission.csv', index=False)\nprint('Your submission was successfully saved!')\ndf_sub.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-13T21:19:32.11436Z","iopub.execute_input":"2022-10-13T21:19:32.114726Z","iopub.status.idle":"2022-10-13T21:21:01.734343Z","shell.execute_reply.started":"2022-10-13T21:19:32.11469Z","shell.execute_reply":"2022-10-13T21:21:01.733306Z"},"trusted":true},"execution_count":null,"outputs":[]}]}