{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":22307,"databundleVersionId":1502524,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ====================================================\n# IMPORT LIBRARIES\n# ====================================================\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pydicom\nfrom pydicom.pixel_data_handlers import pylibjpeg_handler  # Ensure pylibjpeg is available\nimport glob\nimport os\nimport warnings\nfrom scipy.ndimage import gaussian_filter\nfrom skimage import exposure  # For histogram equalization\nfrom skimage.transform import resize  # For resizing images\n\n# Suppress warnings for clean output\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\nprint(\"Libraries imported successfully.\")\n\n# ====================================================\n# SET DIRECTORY PATHS\n# ====================================================\n\nDATA_DIR = \"../input/rsna-str-pulmonary-embolism-detection/\"\nTRAIN_CSV = os.path.join(DATA_DIR, \"train.csv\")\nTEST_CSV = os.path.join(DATA_DIR, \"test.csv\")\nTRAIN_PATH = os.path.join(DATA_DIR, \"train/\")\nTEST_PATH = os.path.join(DATA_DIR, \"test/\")\nprint(\"Directory paths set.\")\n\n\n# ====================================================\n# LOAD DATASETS\n# ====================================================\ntrain_df = pd.read_csv(TRAIN_CSV)\ntest_df = pd.read_csv(TEST_CSV)\nprint(f\"Train DataFrame shape: {train_df.shape}\")\nprint(f\"Test DataFrame shape : {test_df.shape}\") ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-05T04:40:59.880738Z","iopub.execute_input":"2024-10-05T04:40:59.881204Z","iopub.status.idle":"2024-10-05T04:41:04.156959Z","shell.execute_reply.started":"2024-10-05T04:40:59.881154Z","shell.execute_reply":"2024-10-05T04:41:04.155916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n\n\n# Import dicom images","metadata":{}},{"cell_type":"code","source":"# ==================================================== \n# LOAD DICOM IMAGE PATHS\n# ====================================================\n\ntrain_image_file_paths = glob.glob(os.path.join(TRAIN_PATH, '*/*/*.dcm'))\ntest_image_file_paths = glob.glob(os.path.join(TEST_PATH, '*/*/*.dcm'))\n\nprint(f\"Number of training images: {len(train_image_file_paths)}\")\nprint(f\"Number of testing images : {len(test_image_file_paths)}\") ","metadata":{"execution":{"iopub.status.busy":"2024-10-05T04:41:06.632233Z","iopub.execute_input":"2024-10-05T04:41:06.632657Z","iopub.status.idle":"2024-10-05T04:43:52.543385Z","shell.execute_reply.started":"2024-10-05T04:41:06.632617Z","shell.execute_reply":"2024-10-05T04:43:52.54235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nimport matplotlib.pyplot as plt\n\n# ====================================================\n# DISPLAY FIRST 5 CTPA IMAGES\n# ====================================================\n# Create a figure for plotting\nplt.figure(figsize=(20, 8))  # Increased figure size for larger images\n\n# Loop through the first 5 image file paths\nfor i in range(5):\n    # Load the DICOM file\n    dicom_image = pydicom.dcmread(train_image_file_paths[i]).pixel_array\n    \n    # Plot the image\n    plt.subplot(1, 5, i + 1)  # 1 row, 5 columns\n    plt.imshow(dicom_image, cmap='gray')\n    plt.axis('off')  # Hide axes\n    plt.title(f\"Image {i + 1}\", fontsize=14)  # Increased font size for readability\n\n# Adjust the layout\nplt.tight_layout()\nplt.suptitle(\"Original DICOM Images\", fontsize=16, y=1.02)  # Add an overall title\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-05T05:23:44.156564Z","iopub.execute_input":"2024-10-05T05:23:44.156964Z","iopub.status.idle":"2024-10-05T05:23:45.01514Z","shell.execute_reply.started":"2024-10-05T05:23:44.156926Z","shell.execute_reply":"2024-10-05T05:23:45.014209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# preprocessing func","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport numpy as np\nfrom skimage.transform import resize\nfrom skimage import exposure\nfrom skimage.filters import gaussian, unsharp_mask\nfrom scipy.ndimage import median_filter\n\ndef preprocess_dicom_image(file_path, target_size=(512, 512)):\n    try:\n        # Load the DICOM file\n        ds = pydicom.dcmread(file_path)\n        if 'PixelData' not in ds:\n            print(f\"No pixel data found in {file_path}.\")\n            return None\n            \n        image = ds.pixel_array.astype(float)\n        \n        # Hounsfield Unit (HU) scaling\n        intercept = ds.RescaleIntercept if 'RescaleIntercept' in ds else 0.0\n        slope = ds.RescaleSlope if 'RescaleSlope' in ds else 1.0\n        image = slope * image + intercept\n        \n        # Pulmonary vessel-specific window\n        pe_window = np.array([-700.0, 300.0])  # Wider window to capture more vessel detail\n        image = np.clip(image, pe_window[0], pe_window[1])\n        \n        # Normalize to [0, 1]\n        image -= pe_window[0]\n        image /= (pe_window[1] - pe_window[0])\n        \n        # Resize first to reduce noise impact\n        image_resized = resize(image, target_size, anti_aliasing=True)\n        \n        # Structure-preserving noise reduction\n        # First pass: larger median filter for background noise\n        image_median = median_filter(image_resized, size=3)\n        \n        # Second pass: small gaussian for fine detail preservation\n        image_smoothed = gaussian(image_median, sigma=0.7)\n        \n        # Edge enhancement optimized for vessels with reduced sharpening\n        # Reduced amount from 1.2 to 0.8 and increased radius from 2 to 3\n        image_sharp = unsharp_mask(image_smoothed, radius=3, amount=0.8)\n        \n        # Adaptive histogram equalization with vessel-optimized parameters\n        image_equalized = exposure.equalize_adapthist(image_sharp, \n                                                    kernel_size=8,\n                                                    clip_limit=0.012,\n                                                    nbins=256)\n        \n        # Final contrast enhancement focusing on mid-range values where vessels are\n        p20, p80 = np.percentile(image_equalized, (20, 80))\n        image_rescaled = exposure.rescale_intensity(image_equalized, in_range=(p20, p80))\n        \n        return image_rescaled\n    except pydicom.errors.InvalidDicomError:\n        print(f\"Invalid DICOM file: {file_path}\")\n        return None\n    except Exception as e:\n        print(f\"Error processing {file_path}: {e}\")\n        return None","metadata":{"execution":{"iopub.status.busy":"2024-10-05T05:18:38.691815Z","iopub.execute_input":"2024-10-05T05:18:38.692251Z","iopub.status.idle":"2024-10-05T05:18:38.705459Z","shell.execute_reply.started":"2024-10-05T05:18:38.692211Z","shell.execute_reply":"2024-10-05T05:18:38.704454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Sample preprocessing results","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# ====================================================\n# EXAMPLE OF PREPROCESSING IMAGES\n# ====================================================\nsample_images = [preprocess_dicom_image(file) for file in train_image_file_paths[:5]]\n\n# Create a figure for plotting\nplt.figure(figsize=(20, 8))  # Width of 20 inches, height of 8 inches\n\n# Display the processed images\nfor i, img in enumerate(sample_images):\n    plt.subplot(1, 5, i + 1)  # 1 row, 5 columns\n    plt.imshow(img, cmap='gray')\n    plt.axis('off')  # Hide axes\n    plt.title(f'Image {i + 1}', fontsize=14)\n\n# Adjust the layout\nplt.tight_layout()\nplt.suptitle(\"Preprocessed DICOM Images\", fontsize=16, y=1.02)  # Add an overall title\n\n# Show the plot\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2024-10-05T05:24:42.663093Z","iopub.execute_input":"2024-10-05T05:24:42.663531Z","iopub.status.idle":"2024-10-05T05:24:45.089317Z","shell.execute_reply.started":"2024-10-05T05:24:42.663467Z","shell.execute_reply":"2024-10-05T05:24:45.08833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# apply preprocessing with BATCHING","metadata":{}},{"cell_type":"code","source":"\n# ====================================================\n# BATCH PROCESSING OF IMAGES\n# ====================================================\ndef process_in_batches(file_paths, batch_size=100):\n    \"\"\"Process images in batches to manage memory usage\"\"\"\n    all_preprocessed_images = []\n    \n    # Calculate total number of batches\n    num_batches = len(file_paths) // batch_size + (1 if len(file_paths) % batch_size != 0 else 0)\n    \n    for batch_idx in range(num_batches):\n        start_idx = batch_idx * batch_size\n        end_idx = min((batch_idx + 1) * batch_size, len(file_paths))\n        \n        print(f\"Processing batch {batch_idx + 1}/{num_batches}\")\n        \n        # Process current batch\n        batch_files = file_paths[start_idx:end_idx]\n        batch_processed = []\n        \n        for file_path in batch_files:\n            processed_image = preprocess_dicom_image(file_path)\n            if processed_image is not None:\n                batch_processed.append(processed_image)\n        \n        # Extend the main list with processed batch\n        all_preprocessed_images.extend(batch_processed)\n        \n        # Display progress\n        print(f\"Processed {len(all_preprocessed_images)} images so far\")\n        \n        # EVERY 10 BATCHES, DISPLAY FIRST 5 CTPA FROM EACH BATCH\n        if batch_processed and batch_idx % 10 == 0:  \n            plt.figure(figsize=(12, 6))\n            for i, img in enumerate(batch_processed[:5]):\n                plt.subplot(1, 5, i + 1)\n                plt.imshow(img, cmap='gray')\n                plt.axis('off')\n            plt.show()\n    \n    return all_preprocessed_images\n\n# Process training images in batches\nprint(\"Starting batch processing of training images...\")\npreprocessed_train_images = process_in_batches(train_image_file_paths, batch_size=100)\nprint(f\"Preprocessed {len(preprocessed_train_images)} training images successfully.\") ","metadata":{"execution":{"iopub.status.busy":"2024-10-05T05:25:29.691988Z","iopub.execute_input":"2024-10-05T05:25:29.69241Z","iopub.status.idle":"2024-10-05T05:26:10.323628Z","shell.execute_reply.started":"2024-10-05T05:25:29.692372Z","shell.execute_reply":"2024-10-05T05:26:10.322103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n\n\n# Apply NORMAL preprocessing on ALL TRAINING images","metadata":{}},{"cell_type":"code","source":"\n# List to hold preprocessed images\npreprocessed_train_images = []\n\n# Assuming train_image_file_paths is defined\n# Apply preprocessing to all training images\nfor file_path in tqdm(train_image_file_paths, desc=\"Preprocessing images\"):\n    processed_image = preprocess_dicom_image(file_path)\n    if processed_image is not None:\n        preprocessed_train_images.append(processed_image)\n\nprint(f\"Preprocessed {len(preprocessed_train_images)} training images successfully.\")\n\n# Display some of the preprocessed images\nvalid_images = [img for img in preprocessed_train_images if img is not None]\n\nif valid_images:\n    sample_images = valid_images[:5]  # Display the first 5 valid images\n    plt.figure(figsize=(12, 6))\n    for i, img in enumerate(sample_images):\n        plt.subplot(1, 5, i + 1)\n        plt.imshow(img, cmap='gray')\n        plt.axis('off')\n    plt.show()\nelse:\n    print(\"No valid images to display.\") ","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}