{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":36363,"databundleVersionId":4050810,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom glob import glob\n\nimport pydicom\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision.transforms import v2\n\nfrom transformers import BertModel\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pylab as plt\nfrom pathlib import Path\nimport random\n\nfrom tqdm import tqdm\nimport timm\nimport SimpleITK as sitk","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-03T13:56:04.75663Z","iopub.execute_input":"2025-01-03T13:56:04.757305Z","iopub.status.idle":"2025-01-03T13:56:23.4214Z","shell.execute_reply.started":"2025-01-03T13:56:04.757272Z","shell.execute_reply":"2025-01-03T13:56:23.420645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device('cuda') if torch.cuda.is_available() else torch.device('cpu')\nimage_encoder_model = timm.create_model('hf_hub:timm/mobilenetv3_small_050.lamb_in1k', pretrained=True)\nif torch.cuda.device_count() > 1:\n    image_encoder_model = nn.DataParallel(image_encoder_model)\nimage_encoder_model = image_encoder_model.to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T13:56:23.422913Z","iopub.execute_input":"2025-01-03T13:56:23.423176Z","iopub.status.idle":"2025-01-03T13:56:24.261535Z","shell.execute_reply.started":"2025-01-03T13:56:23.423151Z","shell.execute_reply":"2025-01-03T13:56:24.260728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\n\ndicom_path = \"/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images/1.2.826.0.1.3680043.10001/1.dcm\"\ndicom_data = pydicom.dcmread(dicom_path)\n\n# Step 2: Extract the pixel data (assuming it's in the pixel_data field)\nimage_data = dicom_data.pixel_array\n\n# Step 3: Convert to tensor\ntensor_data = torch.tensor(image_data)\n\n# Step 4 (Optional): Normalize the tensor data (if needed)\ntensor_data = tensor_data.float()  # Convert to float if necessary\ntensor_data /= tensor_data.max()  # Normalize to [0, 1] range\n\ntensor_data = tensor_data.unsqueeze(0).repeat(3,1,1)\n# Print the tensor shape to verify\nprint(tensor_data.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T13:56:24.262738Z","iopub.execute_input":"2025-01-03T13:56:24.263109Z","iopub.status.idle":"2025-01-03T13:56:24.298015Z","shell.execute_reply.started":"2025-01-03T13:56:24.263064Z","shell.execute_reply":"2025-01-03T13:56:24.297094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def img_process(path, transform):\n    image = sitk.ReadImage(path)\n    \n    # Convert the SimpleITK image to a NumPy array\n    image_data = sitk.GetArrayFromImage(image)\n    \n    # Convert the NumPy array to a tensor\n    tensor_data = torch.tensor(image_data)\n    tensor_data = tensor_data.repeat(3,1,1)\n    # print(tensor_data.shape)\n    return tensor_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T13:56:24.299703Z","iopub.execute_input":"2025-01-03T13:56:24.299986Z","iopub.status.idle":"2025-01-03T13:56:24.368706Z","shell.execute_reply.started":"2025-01-03T13:56:24.299957Z","shell.execute_reply":"2025-01-03T13:56:24.367865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"root_dir = \"/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images/\"\ntrain_df = pd.read_csv(\"/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train.csv\")\ntrain_df[\"index\"]=train_df.index\nscan_encoders_train = train_df[['StudyInstanceUID', 'patient_overall']]\ntransform = v2.Compose([\n            v2.ToImage(),\n            v2.Resize((224, 224)),\n            v2.ToDtype(torch.float32, scale=True)\n        ])\nscan_encoders_train['embeddings'] = None\nembedding_dir = '/kaggle/working/'\nfor i in tqdm(scan_encoders_train['StudyInstanceUID']):\n    slit_vectors = []\n    imgs = []\n    patient_dir = os.path.join(root_dir, str(i))\n    if os.path.isdir(patient_dir):\n        # Get the list of files in the folder\n        files = os.listdir(patient_dir)\n        \n        # Count the number of files\n        file_count = len(files)\n        first_img = img_process(os.path.join(patient_dir, f\"1.dcm\"), transform)\n        img = transform(first_img)\n        img_cuda = img.unsqueeze(0).to(device)\n    for slit in range(2, file_count+1):\n        img = img_process(os.path.join(patient_dir, f\"{slit}.dcm\"), transform)\n        img = transform(img)\n        img = img.unsqueeze(0).to(device)\n        img_cuda = torch.cat([img_cuda, img], dim=0)\n        # img_cuda = img.to(device)\n        # imgs.append(img_cuda)\n    # imgs = torch.stack(imgs, dim=0)\n    # print(img_cuda.shape)\n    with torch.no_grad():\n        vector = image_encoder_model(img_cuda)\n        slit_vectors.append(vector.cpu().numpy())\n    embedding_file = os.path.join(embedding_dir, f\"{i}_embeddings.npy\")\n    np.save(embedding_file, np.array(slit_vectors))\n    scan_encoders_train.loc[scan_encoders_train['StudyInstanceUID'] == i, 'embeddings'] = embedding_file","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T13:59:39.474548Z","iopub.execute_input":"2025-01-03T13:59:39.474889Z","iopub.status.idle":"2025-01-03T14:06:52.25556Z","shell.execute_reply.started":"2025-01-03T13:59:39.474858Z","shell.execute_reply":"2025-01-03T14:06:52.253995Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scan_encoders_train.to_csv('/kaggle/working/filtered_data.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T14:06:52.256647Z","iopub.status.idle":"2025-01-03T14:06:52.257089Z","shell.execute_reply.started":"2025-01-03T14:06:52.256863Z","shell.execute_reply":"2025-01-03T14:06:52.256886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}