{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport glob\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom ipywidgets import interact\nimport matplotlib.animation as animation\nfrom IPython.display import HTML\n\nimport pydicom\nimport nibabel as nib\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport torch\nimport torchvision\nimport torchvision.transforms as transforms\n\nDATA_PATH = '/kaggle/input/rsna-2023-abdominal-trauma-detection/'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-29T02:02:05.767544Z","iopub.execute_input":"2023-08-29T02:02:05.767815Z","iopub.status.idle":"2023-08-29T02:02:11.994725Z","shell.execute_reply.started":"2023-08-29T02:02:05.76779Z","shell.execute_reply":"2023-08-29T02:02:11.993696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv(f'{DATA_PATH}/train.csv')\ntrain_csv = train_csv.set_index('patient_id')\n\ntrain_series_meta = pd.read_csv(f'{DATA_PATH}/train_series_meta.csv')\ntest_series_meta = pd.read_csv(f'{DATA_PATH}/test_series_meta.csv')\n\nimage_level_labels = pd.read_csv(f'{DATA_PATH}/image_level_labels.csv')\n\ntrain_dicom_tags = pd.read_parquet(f'{DATA_PATH}/train_dicom_tags.parquet')\ntest_dicom_tags = pd.read_parquet(f'{DATA_PATH}/test_dicom_tags.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-08-29T02:02:11.996569Z","iopub.execute_input":"2023-08-29T02:02:11.997487Z","iopub.status.idle":"2023-08-29T02:02:18.804813Z","shell.execute_reply.started":"2023-08-29T02:02:11.997452Z","shell.execute_reply":"2023-08-29T02:02:18.803535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-29T02:02:18.806294Z","iopub.execute_input":"2023-08-29T02:02:18.806952Z","iopub.status.idle":"2023-08-29T02:02:18.831854Z","shell.execute_reply.started":"2023-08-29T02:02:18.806905Z","shell.execute_reply":"2023-08-29T02:02:18.830829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_dcms = glob.glob(f'{DATA_PATH}/train_images/*/*/*.dcm')\n# train_dcms.sort()\n\n# test_dcms = glob.glob(f'{DATA_PATH}/test_images/*/*/*.dcm')\n# test_dcms.sort()\n\n# segmentations = glob.glob(f'{DATA_PATH}/segmentations/*.nii')\n# segmentations.sort()","metadata":{"execution":{"iopub.status.busy":"2023-08-29T02:02:18.834642Z","iopub.execute_input":"2023-08-29T02:02:18.834979Z","iopub.status.idle":"2023-08-29T02:02:18.839307Z","shell.execute_reply.started":"2023-08-29T02:02:18.834949Z","shell.execute_reply":"2023-08-29T02:02:18.838273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_256_jpgs = glob.glob('/kaggle/input/rsna-2023-atd-reduced-256-5mm/reduced_256_tickness_5/*/*/*.jpeg')\nnp.random.shuffle(train_256_jpgs)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T02:02:18.840956Z","iopub.execute_input":"2023-08-29T02:02:18.841687Z","iopub.status.idle":"2023-08-29T02:04:31.961809Z","shell.execute_reply.started":"2023-08-29T02:02:18.841652Z","shell.execute_reply":"2023-08-29T02:04:31.960783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\ntorch.manual_seed(0)\ntorch.backends.cudnn.deterministic = False\ntorch.backends.cudnn.benchmark = True\n\nfrom PIL import Image\nimport torchvision.models as models\nimport torchvision.transforms as transforms\nimport torchvision.datasets as datasets\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.autograd import Variable\nfrom torch.utils.data.dataset import Dataset\n\n# Define the transform\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),  # Resize the image to 224x224 pixels\n    transforms.RandomVerticalFlip(0.5),  # Resize the image to 224x224 pixels\n    transforms.RandomHorizontalFlip(0.5),  # Resize the image to 224x224 pixels\n    transforms.ToTensor()  # Convert the image to a PyTorch tensor\n])","metadata":{"execution":{"iopub.status.busy":"2023-08-29T02:04:31.963078Z","iopub.execute_input":"2023-08-29T02:04:31.963415Z","iopub.status.idle":"2023-08-29T02:04:31.976221Z","shell.execute_reply.started":"2023-08-29T02:04:31.963384Z","shell.execute_reply":"2023-08-29T02:04:31.975332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class RSNADataset(Dataset):\n    def __init__(self, img_path, transform=None):\n        self.img_path = img_path\n        if transform is not None:\n            self.transform = transform\n        else:\n            self.transform = None\n    \n    def __getitem__(self, index):\n        pid = self.img_path[index].split('/')[-3]\n        pid = int(pid)\n        \n        img = Image.open(self.img_path[index])\n        if self.transform is not None:\n            img = self.transform(img)\n        \n        return img, torch.from_numpy(train_csv.loc[pid].values[:]).float()\n    \n    def __len__(self):\n        return len(self.img_path)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T02:04:31.979546Z","iopub.execute_input":"2023-08-29T02:04:31.979836Z","iopub.status.idle":"2023-08-29T02:04:31.988518Z","shell.execute_reply.started":"2023-08-29T02:04:31.979813Z","shell.execute_reply":"2023-08-29T02:04:31.987496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(train_loader, model, criterion, optimizer):\n    model.train()\n    train_loss = 0.0\n    for i, (input, target) in enumerate(train_loader):\n        input = input.cuda()\n        target = target.cuda()\n\n        # compute output\n        output = model(input)\n        loss = criterion(output, target)\n\n        # compute gradient and do SGD step\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n            \n        train_loss += loss.item()\n    \n    return train_loss/len(train_loader)\n\ndef validate(val_loader, model, criterion):\n    model.eval()\n    val_loss = 0.0\n    \n    val_pred = []\n    val_label = []\n    with torch.no_grad():\n        for i, (input, target) in enumerate(val_loader):\n            input = input.cuda()\n            target = target.cuda()\n            output = model(input)\n            val_loss += criterion(output, target)\n            \n            output = torch.sigmoid(output)\n            val_pred.append(output.data.cpu().numpy())\n            val_label.append(target.data.cpu().numpy())\n    return val_loss / len(val_loader.dataset), val_pred, val_label","metadata":{"execution":{"iopub.status.busy":"2023-08-29T02:04:31.989917Z","iopub.execute_input":"2023-08-29T02:04:31.990516Z","iopub.status.idle":"2023-08-29T02:04:32.015341Z","shell.execute_reply.started":"2023-08-29T02:04:31.990479Z","shell.execute_reply":"2023-08-29T02:04:32.014162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pids = list(train_csv.index)\n\ntrain_loader = torch.utils.data.DataLoader(\n    RSNADataset([x for x in train_256_jpgs if int(x.split('/')[-3]) in pids[:-200]], transform), \n    batch_size=6, shuffle=True, num_workers=4, pin_memory=False\n)\n\nval_loader = torch.utils.data.DataLoader(\n    RSNADataset([x for x in train_256_jpgs if int(x.split('/')[-3]) in pids[-200:]], transform), \n    batch_size=6, shuffle=True, num_workers=4, pin_memory=False\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T02:12:57.136396Z","iopub.execute_input":"2023-08-29T02:12:57.136816Z","iopub.status.idle":"2023-08-29T02:13:03.859596Z","shell.execute_reply.started":"2023-08-29T02:12:57.136781Z","shell.execute_reply":"2023-08-29T02:13:03.858633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pandas.api.types\nimport sklearn.metrics\n\ndef normalize_probabilities_to_one(df: pd.DataFrame, group_columns: list) -> pd.DataFrame:\n    # Normalize the sum of each row's probabilities to 100%.\n    # 0.75, 0.75 => 0.5, 0.5\n    # 0.1, 0.1 => 0.5, 0.5\n    row_totals = df[group_columns].sum(axis=1)\n    if row_totals.min() == 0:\n        raise ParticipantVisibleError('All rows must contain at least one non-zero prediction')\n    for col in group_columns:\n        df[col] /= row_totals\n    return df\n\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str) -> float:\n    '''\n    Pseudocode:\n    1. For every label group (liver, bowel, etc):\n        - Normalize the sum of each row's probabilities to 100%.\n        - Calculate the sample weighted log loss.\n    2. Derive a new any_injury label by taking the max of 1 - p(healthy) for each label group\n    3. Calculate the sample weighted log loss for the new label group\n    4. Return the average of all of the label group log losses as the final score.\n    '''\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n\n    # Run basic QC checks on the inputs\n    if not pandas.api.types.is_numeric_dtype(submission.values):\n        raise ParticipantVisibleError('All submission values must be numeric')\n\n    if not np.isfinite(submission.values).all():\n        raise ParticipantVisibleError('All submission values must be finite')\n\n    if solution.min().min() < 0:\n        raise ParticipantVisibleError('All labels must be at least zero')\n    if submission.min().min() < 0:\n        raise ParticipantVisibleError('All predictions must be at least zero')\n\n    # Calculate the label group log losses\n    binary_targets = ['bowel', 'extravasation']\n    triple_level_targets = ['kidney', 'liver', 'spleen']\n    all_target_categories = binary_targets + triple_level_targets\n\n    label_group_losses = []\n    for category in all_target_categories:\n        if category in binary_targets:\n            col_group = [f'{category}_healthy', f'{category}_injury']\n        else:\n            col_group = [f'{category}_healthy', f'{category}_low', f'{category}_high']\n\n        solution = normalize_probabilities_to_one(solution, col_group)\n\n        for col in col_group:\n            if col not in submission.columns:\n                raise ParticipantVisibleError(f'Missing submission column {col}')\n        submission = normalize_probabilities_to_one(submission, col_group)\n        label_group_losses.append(\n            sklearn.metrics.log_loss(\n                y_true=solution[col_group].values,\n                y_pred=submission[col_group].values,\n                sample_weight=solution[f'{category}_weight'].values\n            )\n        )\n\n    # Derive a new any_injury label by taking the max of 1 - p(healthy) for each label group\n    healthy_cols = [x + '_healthy' for x in all_target_categories]\n    any_injury_labels = (1 - solution[healthy_cols]).max(axis=1)\n    any_injury_predictions = (1 - submission[healthy_cols]).max(axis=1)\n    any_injury_loss = sklearn.metrics.log_loss(\n        y_true=any_injury_labels.values,\n        y_pred=any_injury_predictions.values,\n        sample_weight=solution['any_injury_weight'].values\n    )\n\n    label_group_losses.append(any_injury_loss)\n    return np.mean(label_group_losses)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T02:13:03.86222Z","iopub.execute_input":"2023-08-29T02:13:03.862612Z","iopub.status.idle":"2023-08-29T02:13:03.88161Z","shell.execute_reply.started":"2023-08-29T02:13:03.862577Z","shell.execute_reply":"2023-08-29T02:13:03.880534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = torchvision.models.resnet34(False)\nmodel.load_state_dict(torch.load(\"/kaggle/input/pretrained-model-weights-pytorch/resnet34-333f7ec4.pth\"))\nmodel.fc = torch.nn.Linear(512, 14)\nmodel.conv1 = nn.Conv2d(1, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)\n\nmodel = model.cuda()\npos_weight = torch.Tensor([1,2,1,6,1,2,4,1,2,4,1,2,4,6]).cuda()\ncriterion = nn.BCEWithLogitsLoss(pos_weight=pos_weight)\noptimizer = torch.optim.SGD(model.parameters(), 0.0005)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T02:13:03.883144Z","iopub.execute_input":"2023-08-29T02:13:03.883785Z","iopub.status.idle":"2023-08-29T02:13:04.482686Z","shell.execute_reply.started":"2023-08-29T02:13:03.883752Z","shell.execute_reply":"2023-08-29T02:13:04.481673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_loss = 100\nfor epoch in range(2):\n    train_loss = train(train_loader, model, criterion, optimizer)\n    val_loss, val_pred, val_label = validate(val_loader, model, criterion)\n    \n    # metrics\n    val_pred = np.vstack(val_pred)\n    val_pred = val_pred[:, :-1]\n    val_label = np.vstack(val_label)\n\n    val_pred = pd.DataFrame(val_pred)\n    val_pred.columns = ['bowel_healthy', 'bowel_injury', 'extravasation_healthy',\n           'extravasation_injury', 'kidney_healthy', 'kidney_low', 'kidney_high',\n           'liver_healthy', 'liver_low', 'liver_high', 'spleen_healthy',\n           'spleen_low', 'spleen_high']\n    val_pred['patient_id'] = range(len(val_pred))\n\n    \n    val_label = pd.DataFrame(val_label)\n    val_label.columns = ['bowel_healthy', 'bowel_injury', 'extravasation_healthy',\n           'extravasation_injury', 'kidney_healthy', 'kidney_low', 'kidney_high',\n           'liver_healthy', 'liver_low', 'liver_high', 'spleen_healthy',\n           'spleen_low', 'spleen_high', 'any_injury']\n    val_label['patient_id'] = range(len(val_pred))\n    \n    # metrics weight\n    val_label['bowel_weight'] = val_label['bowel_healthy'].map({1:1}).fillna(2)\n    val_label['extravasation_weight'] = val_label['extravasation_healthy'].map({1:1}).fillna(6)\n    val_label['extravasation_weight'] = val_label['extravasation_healthy'].map({1:1}).fillna(6)\n\n    kidney_label = val_label[['kidney_healthy', 'kidney_low','kidney_high']].values.argmax(1)\n    kidney_label = pd.Series(kidney_label)\n    val_label['kidney_weight'] = kidney_label.map({0: 1, 1:2, 2:4})\n\n    liver_label = val_label[['liver_healthy', 'liver_low','liver_high']].values.argmax(1)\n    liver_label = pd.Series(liver_label)\n    val_label['liver_weight'] = liver_label.map({0: 1, 1:2, 2:4})\n\n    spleen_label = val_label[['spleen_healthy', 'spleen_low','spleen_high']].values.argmax(1)\n    spleen_label = pd.Series(spleen_label)\n    val_label['spleen_weight'] = spleen_label.map({0: 1, 1:2, 2:4})\n    val_label['any_injury_weight'] = val_label['any_injury'].map({0:1, 1:6})\n    \n    val_loss = score(val_label, val_pred, 'patient_id')\n    \n    print(train_loss, val_loss)\n    torch.save(model.state_dict(), f'model_{epoch}.pth')","metadata":{"execution":{"iopub.status.busy":"2023-08-29T02:13:04.485162Z","iopub.execute_input":"2023-08-29T02:13:04.485547Z","iopub.status.idle":"2023-08-29T02:31:52.816198Z","shell.execute_reply.started":"2023-08-29T02:13:04.485514Z","shell.execute_reply":"2023-08-29T02:31:52.815049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}