{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":36363,"databundleVersionId":4050810,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# sample_submission.csv'\ndf = pd.read_csv('/kaggle/input/rsna-2022-cervical-spine-fracture-detection/sample_submission.csv')\nprint(df.head())  ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 'train.csv'\ntrain = pd.read_csv('/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train.csv')\nprint(train.head()) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 'train_bounding_boxes.csv'\ntest = pd.read_csv('/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_bounding_boxes.csv')\nprint(test.head()) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install torch torchvision","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install pydicom","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install --upgrade pydicom","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install pylibjpeg pylibjpeg-libjpeg pydicom","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Resnet50 fo baseline","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nfrom PIL import Image\nimport torch\nimport numpy as np\n\nclass SpineFractureDataset(Dataset):\n    def __init__(self, csv_file, root_dir, transform=None):\n        self.data_frame = pd.read_csv(csv_file)\n        self.root_dir = root_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.data_frame)\n\n    def __getitem__(self, idx):\n\n        study_instance_uid = self.data_frame.iloc[idx]['StudyInstanceUID']\n        img_dir = os.path.join(self.root_dir, study_instance_uid)\n        images = [os.path.join(img_dir, f) for f in os.listdir(img_dir) if f.endswith('.dcm')]\n        \n        img_path = images[0]\n        dicom = pydicom.dcmread(img_path)\n        image = apply_voi_lut(dicom.pixel_array, dicom)  # Convert to proper format\n        image = image.astype(np.float32)\n        image = (image - np.min(image)) / (np.max(image) - np.min(image))  # Normalize\n        image = Image.fromarray(np.uint8(image*255)).convert(\"RGB\")  # Convert to RGB\n\n        if self.transform:\n            image = self.transform(image)\n\n        # Extract labels\n        labels = self.data_frame.iloc[idx][2:].values.astype(np.float32)  # Assumes labels start at column 3\n\n        return image, torch.from_numpy(labels)\n\n# Define transformations\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\n# Create dataset and dataloader\ndataset = SpineFractureDataset(\n    csv_file='/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train.csv',\n    root_dir='/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images',\n    transform=transform\n)\ndataloader = DataLoader(dataset, batch_size=50, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T15:42:28.266778Z","iopub.execute_input":"2024-04-14T15:42:28.267277Z","iopub.status.idle":"2024-04-14T15:42:28.286212Z","shell.execute_reply.started":"2024-04-14T15:42:28.267242Z","shell.execute_reply":"2024-04-14T15:42:28.285327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchvision.models as models\nimport torch.nn as nn\n\nmodel = models.resnet50(pretrained=True)\nnum_ftrs = model.fc.in_features\nmodel.fc = nn.Linear(num_ftrs, 7) \n\nmodel = model.to('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2024-04-14T15:42:36.042756Z","iopub.execute_input":"2024-04-14T15:42:36.043677Z","iopub.status.idle":"2024-04-14T15:42:36.959781Z","shell.execute_reply.started":"2024-04-14T15:42:36.04364Z","shell.execute_reply":"2024-04-14T15:42:36.958938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.optim as optim\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)\n\nnum_epochs = 10\nmodel.train()\nfor epoch in range(num_epochs):\n    for images, labels in dataloader:\n        images = images.to('cuda' if torch.cuda.is_available() else 'cpu')\n        labels = labels.to('cuda' if torch.cuda.is_available() else 'cpu')\n        \n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n    print(f'Epoch {epoch+1}, Loss: {loss.item()}')","metadata":{"execution":{"iopub.status.busy":"2024-04-14T15:42:49.206661Z","iopub.execute_input":"2024-04-14T15:42:49.207289Z","iopub.status.idle":"2024-04-14T15:47:42.631065Z","shell.execute_reply.started":"2024-04-14T15:42:49.207254Z","shell.execute_reply":"2024-04-14T15:47:42.630082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Simple CNN","metadata":{}},{"cell_type":"code","source":"class DICOMDataset(Dataset):\n    def __init__(self, csv_file, root_dir, transform=None):\n        self.labels_df = pd.read_csv(csv_file)\n        self.root_dir = root_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.labels_df)\n\n    def __getitem__(self, idx):\n        study_instance_uid = self.labels_df.iloc[idx]['StudyInstanceUID']\n        img_dir = os.path.join(self.root_dir, study_instance_uid)\n        images = [os.path.join(img_dir, f) for f in os.listdir(img_dir) if f.endswith('.dcm')]\n        img_path = images[0]  \n        \n        # Load DICOM image\n        dicom = pydicom.dcmread(img_path)\n        image = apply_voi_lut(dicom.pixel_array, dicom)\n        image = image.astype(float)\n        image = (image - image.min()) / (image.max() - image.min())\n        image = Image.fromarray((image * 255).astype(np.uint8)).convert(\"RGB\")\n\n        if self.transform:\n            image = self.transform(image)\n\n        # Extract labels\n        labels = self.labels_df.iloc[idx][2:].values.astype(float)  # Adjust index depending on label columns\n        return image, torch.tensor(labels)\n\n# Transformations\ntransform = transforms.Compose([\n    transforms.Resize((256, 256)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\n# Creating the dataset and dataloader\ndataset = DICOMDataset(csv_file='/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train.csv', root_dir='/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images', transform=transform)\ndataloader = DataLoader(dataset, batch_size=10, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T15:50:53.881229Z","iopub.execute_input":"2024-04-14T15:50:53.882087Z","iopub.status.idle":"2024-04-14T15:50:53.900491Z","shell.execute_reply.started":"2024-04-14T15:50:53.882057Z","shell.execute_reply":"2024-04-14T15:50:53.899565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn as nn\nimport torch.nn.functional as F\n\nclass SimpleCNN(nn.Module):\n    def __init__(self):\n        super(SimpleCNN, self).__init__()\n        self.conv1 = nn.Conv2d(3, 16, kernel_size=3, padding=1)\n        self.conv2 = nn.Conv2d(16, 32, kernel_size=3, padding=1)\n        self.conv3 = nn.Conv2d(32, 64, kernel_size=3, padding=1)\n        self.pool = nn.MaxPool2d(2, 2)\n        self.fc1 = nn.Linear(64 * 32 * 32, 512)\n        self.fc2 = nn.Linear(512, 7)  # 7 outputs for C1 to C7 labels\n\n    def forward(self, x):\n        x = self.pool(F.relu(self.conv1(x)))\n        x = self.pool(F.relu(self.conv2(x)))\n        x = self.pool(F.relu(self.conv3(x)))\n        x = torch.flatten(x, 1)  # flatten all dimensions except the batch dimension\n        x = F.relu(self.fc1(x))\n        x = self.fc2(x)\n        return x\n\nmodel = SimpleCNN()\nmodel = model.to('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2024-04-14T15:50:57.892785Z","iopub.execute_input":"2024-04-14T15:50:57.893584Z","iopub.status.idle":"2024-04-14T15:50:58.261652Z","shell.execute_reply.started":"2024-04-14T15:50:57.89355Z","shell.execute_reply":"2024-04-14T15:50:58.260587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.optim as optim\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)\n\nnum_epochs = 10\nfor epoch in range(num_epochs):\n    for images, labels in dataloader:\n        images = images.to('cuda' if torch.cuda.is_available() else 'cpu')\n        labels = labels.to('cuda' if torch.cuda.is_available() else 'cpu')\n        \n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n    print(f'Epoch {epoch+1}, Loss: {loss.item()}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Evaluation**","metadata":{}},{"cell_type":"code","source":"test_dataset = DICOMDataset(csv_file='/kaggle/input/rsna-2022-cervical-spine-fracture-detection/test.csv', root_dir='/kaggle/input/rsna-2022-cervical-spine-fracture-detection/test_images', transform=transform)\ntest_loader = DataLoader(test_dataset, batch_size=10, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T15:54:04.674339Z","iopub.execute_input":"2024-04-14T15:54:04.674747Z","iopub.status.idle":"2024-04-14T15:54:04.683172Z","shell.execute_reply.started":"2024-04-14T15:54:04.674717Z","shell.execute_reply":"2024-04-14T15:54:04.682227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nimport numpy as np\n\n# Make sure to import necessary libraries\nimport torch\n\ndef calculate_metrics(outputs, labels):\n    with torch.no_grad():\n        predictions = torch.sigmoid(outputs).data > 0.5  # Convert logits to binary predictions\n        predictions = predictions.cpu().numpy()\n        labels = labels.cpu().numpy()\n\n        accuracy = accuracy_score(labels, predictions)\n        class_report = classification_report(labels, predictions)\n        conf_matrix = confusion_matrix(labels.argmax(axis=1), predictions.argmax(axis=1))\n\n        return accuracy, class_report, conf_matrix","metadata":{"execution":{"iopub.status.busy":"2024-04-14T15:54:08.186342Z","iopub.execute_input":"2024-04-14T15:54:08.18725Z","iopub.status.idle":"2024-04-14T15:54:08.193712Z","shell.execute_reply.started":"2024-04-14T15:54:08.187215Z","shell.execute_reply":"2024-04-14T15:54:08.192721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.eval()  # Set the model to evaluation mode\n\nall_labels = []\nall_predictions = []\n\n# Evaluate the model\nfor images, labels in test_loader:\n    images = images.to('cuda' if torch.cuda.is_available() else 'cpu')\n    labels = labels.to('cuda' if torch.cuda.is_available() else 'cpu')\n\n    outputs = model(images)\n    predictions = torch.sigmoid(outputs).data > 0.5  # Sigmoid and thresholding\n    all_labels.append(labels.cpu().numpy())\n    all_predictions.append(predictions.cpu().numpy())\n\n# Convert lists to numpy arrays\nall_labels = np.vstack(all_labels)\nall_predictions = np.vstack(all_predictions)\n\n# Calculate overall accuracy and other metrics\naccuracy = accuracy_score(all_labels, all_predictions)\nprint(\"Accuracy:\", accuracy)\nprint(\"Classification Report:\\n\", classification_report(all_labels, all_predictions))\nconf_matrix = confusion_matrix(all_labels.argmax(axis=1), all_predictions.argmax(axis=1))\nprint(\"Confusion Matrix:\\n\", conf_matrix)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T15:54:11.967705Z","iopub.execute_input":"2024-04-14T15:54:11.968359Z","iopub.status.idle":"2024-04-14T15:54:12.071396Z","shell.execute_reply.started":"2024-04-14T15:54:11.968313Z","shell.execute_reply":"2024-04-14T15:54:12.070173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}