{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pydicom\nimport pandas as pd\nimport numpy as np\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nfrom torchvision.transforms import Resize\n\n# Paths\nBASE_DIR = \"/kaggle/input/rsna-2023-abdominal-trauma-detection\"\nTRAIN_CSV_PATH = os.path.join(BASE_DIR, \"train.csv\")\nTRAIN_IMAGES_PATH = os.path.join(BASE_DIR, \"train_images\")\nTEST_IMAGES_PATH = os.path.join(BASE_DIR, \"test_images\")\nSAMPLE_SUBMISSION_PATH = os.path.join(BASE_DIR, \"sample_submission.csv\")\n\n# Load CSV files\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\nsample_submission = pd.read_csv(SAMPLE_SUBMISSION_PATH)\n# Create the test dataframe\ntest_df = sample_submission[['patient_id']]\n\n# Function to get DICOM paths\ndef get_dicom_paths(patient_id, is_train=True):\n    folder = TRAIN_IMAGES_PATH if is_train else TEST_IMAGES_PATH\n    paths = []\n    for dirpath, _, filenames in os.walk(os.path.join(folder, patient_id)):\n        for file in filenames:\n            paths.append(os.path.join(dirpath, file))\n    return paths\n\n# Transformations added\ntransform = transforms.Compose([\n    transforms.ToTensor(),\n])","metadata":{"execution":{"iopub.status.busy":"2023-10-05T11:51:29.532181Z","iopub.execute_input":"2023-10-05T11:51:29.532652Z","iopub.status.idle":"2023-10-05T11:51:33.049408Z","shell.execute_reply.started":"2023-10-05T11:51:29.532615Z","shell.execute_reply":"2023-10-05T11:51:33.048409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torchvision.transforms import Resize, ToPILImage, ToTensor\n\n\n\n# Define CTDataset\nclass CTDataset(Dataset):\n    def __init__(self, dataframe, root_dir, transform=None, is_train=True):\n        self.dataframe = dataframe\n        self.root_dir = root_dir\n        self.transform = transform\n        self.is_train = is_train\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        patient_id = str(self.dataframe.iloc[idx, 0])\n\n        # Get all DICOM paths for the patient and use the first one\n        dicom_paths = get_dicom_paths(patient_id, self.is_train)\n        if not dicom_paths:\n            raise FileNotFoundError(f\"No DICOM images found for patient {patient_id}\")\n\n        # Load the first DICOM image for simplicity\n        image = pydicom.dcmread(dicom_paths[0]).pixel_array\n\n        # Convert to float32 and normalize to [0, 1]\n        image = image.astype(np.float32) / 65535.0  # 65535 is the maximum value for uint16\n\n        # Convert single channel to three-channel by repeating\n        image = np.stack((image,) * 3, axis=-1)\n\n        # Multiply by 255 and convert to uint8 for PIL conversion\n        image *= 255\n        image = image.astype(np.uint8)\n\n        # Convert to PIL Image and then resize\n        to_pil = ToPILImage()\n        image = to_pil(image)\n        resize_transform = Resize((512, 512))\n        image = resize_transform(image)\n\n        # Convert back to tensor\n        to_tensor = ToTensor()\n        image = to_tensor(image)\n        \n        \n        # Get labels\n        labels = self.dataframe.iloc[idx, 1:].values\n        labels = torch.tensor(labels, dtype=torch.float32)\n\n        # Apply other transformations to the image \n        if self.transform:\n            image = self.transform(image)\n\n        return image, labels\n\nbatch_size = 8\n\n# Check data loaders\ntrain_dataset = CTDataset(dataframe=train_df, root_dir=TRAIN_IMAGES_PATH, is_train=True)\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\n\n# Create the DataLoader for test set\ntest_dataset = CTDataset(dataframe=test_df, root_dir=TEST_IMAGES_PATH, is_train=False)\ntest_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False)\n\ntest_loader\n\ntrain_dataset, train_loader\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-05T11:51:33.051408Z","iopub.execute_input":"2023-10-05T11:51:33.052007Z","iopub.status.idle":"2023-10-05T11:51:33.065592Z","shell.execute_reply.started":"2023-10-05T11:51:33.051973Z","shell.execute_reply":"2023-10-05T11:51:33.064614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from timm import create_model\nimport torch\nimport os\nimport torch.nn as nn\n\nmodel = create_model('efficientnet_b4', pretrained=False)\ntorch.save(model.state_dict(), 'efficientnet_b4_weights.pth')\n","metadata":{"execution":{"iopub.status.busy":"2023-10-05T11:51:33.067111Z","iopub.execute_input":"2023-10-05T11:51:33.067741Z","iopub.status.idle":"2023-10-05T11:51:34.590829Z","shell.execute_reply.started":"2023-10-05T11:51:33.067712Z","shell.execute_reply":"2023-10-05T11:51:34.589889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model Building\n\n# Path for saved weights\nweights_path = 'efficientnet_b4_weights.pth'\n\n# Function to build the model\ndef build_model(num_classes):\n    # Create the model WITHOUT pre-trained weights (since there's no internet connection)\n    model = create_model('efficientnet_b4', pretrained=False)\n    \n    # If the weights file exists, load it\n    if os.path.exists(weights_path):\n        model.load_state_dict(torch.load(weights_path, map_location='cpu'), strict=False)  \n    else:\n        print(f'Warning: Weights file not found in path {weights_path}, training from scratch.')\n    \n    # Modify the classifier layer to have the desired number of output classes\n    model.classifier = nn.Linear(model.classifier.in_features, num_classes)\n    \n    return model\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-05T11:51:34.593026Z","iopub.execute_input":"2023-10-05T11:51:34.593375Z","iopub.status.idle":"2023-10-05T11:51:34.598803Z","shell.execute_reply.started":"2023-10-05T11:51:34.593346Z","shell.execute_reply":"2023-10-05T11:51:34.597865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model Training\n\nimport torch.optim as optim\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = build_model(len(train_df.columns) - 1).to(device)\n\ncriterion = torch.nn.BCEWithLogitsLoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)\n\nnum_epochs = 4\nfor epoch in range(num_epochs):\n    model.train()\n    running_loss = 0.0\n    for i, (inputs, labels) in enumerate(train_loader):\n        inputs, labels = inputs.to(device), labels.to(device)\n        optimizer.zero_grad()\n        outputs = model(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        running_loss += loss.item()\n\n    print(f\"Epoch {epoch+1}, Loss: {running_loss/len(train_loader)}\")\n\nprint(\"Finished Training\")\n","metadata":{"execution":{"iopub.status.busy":"2023-10-05T11:51:34.600035Z","iopub.execute_input":"2023-10-05T11:51:34.600745Z","iopub.status.idle":"2023-10-05T12:09:44.510829Z","shell.execute_reply.started":"2023-10-05T11:51:34.600716Z","shell.execute_reply":"2023-10-05T12:09:44.50989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluation & Prediction\n\nmodel.eval()\npredictions = []\n\nwith torch.no_grad():\n    for images, _ in test_loader:  # unpacking two values\n        images = images.to(device)\n        outputs = torch.sigmoid(model(images))\n        predictions.append(outputs.cpu().numpy())\n\npredictions = np.vstack(predictions)\nprint(predictions.shape)\nprint(sample_submission.columns[1:])\n\npredictions = predictions[:, :-1]\n\nsubmission_df = pd.DataFrame(sample_submission['patient_id'], columns=['patient_id'])\npredictions_df = pd.DataFrame(predictions, columns=sample_submission.columns[1:])\nsubmission_df = pd.concat([submission_df, predictions_df], axis=1)\n\n# Save to CSV.\nsubmission_filename = \"submission.csv\"\nsubmission_df.to_csv(submission_filename, index=False)\n\nprint(submission_df.head(2))","metadata":{"execution":{"iopub.status.busy":"2023-10-05T12:09:44.512293Z","iopub.execute_input":"2023-10-05T12:09:44.512865Z","iopub.status.idle":"2023-10-05T12:09:44.638163Z","shell.execute_reply.started":"2023-10-05T12:09:44.512834Z","shell.execute_reply":"2023-10-05T12:09:44.63724Z"},"trusted":true},"execution_count":null,"outputs":[]}]}