{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99552,"databundleVersionId":13851420,"sourceType":"competition"},{"sourceId":13913205,"sourceType":"datasetVersion","datasetId":8834618}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Libraries from Demo\nimport os\nimport shutil\nfrom collections import defaultdict\n\nimport pandas as pd\nimport polars as pl\nimport pydicom as dicom\n\n\n#Libraries from attempt\nimport glob\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport random\nimport scipy.ndimage as ndi\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nimport os, numpy as np, torch\n\nfrom collections import Counter\nfrom scipy import ndimage\n\nfrom scipy.ndimage import zoom as ndi_zoom\nfrom sklearn.model_selection import train_test_split, StratifiedShuffleSplit\nfrom torch.utils.data import Dataset, DataLoader, Subset\nfrom tqdm import tqdm\nfrom typing import Tuple, List\n\nfrom sklearn.preprocessing import StandardScaler","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:16:09.753172Z","iopub.execute_input":"2025-11-30T06:16:09.75412Z","iopub.status.idle":"2025-11-30T06:16:16.794323Z","shell.execute_reply.started":"2025-11-30T06:16:09.754076Z","shell.execute_reply":"2025-11-30T06:16:16.793456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_everything(seed=42):\n    \"\"\"\n    Set random seeds for reproducibility in deep learning projects.\n    \n    Args:\n        seed (int): Random seed value (default: 42)\n    \"\"\"\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)  # if using multi-GPU\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    os.environ['PYTHONHASHSEED'] = str(seed)\n\nseed_everything()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:16:22.851664Z","iopub.execute_input":"2025-11-30T06:16:22.852212Z","iopub.status.idle":"2025-11-30T06:16:22.864937Z","shell.execute_reply.started":"2025-11-30T06:16:22.852183Z","shell.execute_reply":"2025-11-30T06:16:22.863887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nTRAIN_CSV = \"/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv\"\ntest_frac = 0.2\nval_frac = 0.1\nval_frac_within_trainval = val_frac / (1 - test_frac)\ngenerated_mask_dir = '/kaggle/working/generated_masks'\nseed = 42","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:16:25.681164Z","iopub.execute_input":"2025-11-30T06:16:25.681486Z","iopub.status.idle":"2025-11-30T06:16:25.686845Z","shell.execute_reply.started":"2025-11-30T06:16:25.681462Z","shell.execute_reply":"2025-11-30T06:16:25.685937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(TRAIN_CSV)\nprint(f\"On the original Dataset, the percentage of aneurysms is: {100 * sum(train['Aneurysm Present'])/len(train)}%\")\nprint(f\"The original dataset has {len(train)} samples.\")\nprint(f\"The original dataset has {sum(train['Aneurysm Present']==1)} positive samples.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:16:27.621497Z","iopub.execute_input":"2025-11-30T06:16:27.622223Z","iopub.status.idle":"2025-11-30T06:16:27.676545Z","shell.execute_reply.started":"2025-11-30T06:16:27.622188Z","shell.execute_reply":"2025-11-30T06:16:27.675335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:16:30.418501Z","iopub.execute_input":"2025-11-30T06:16:30.418933Z","iopub.status.idle":"2025-11-30T06:16:30.456659Z","shell.execute_reply.started":"2025-11-30T06:16:30.4189Z","shell.execute_reply":"2025-11-30T06:16:30.455473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Aneurysm Present']==1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:16:33.898526Z","iopub.execute_input":"2025-11-30T06:16:33.898852Z","iopub.status.idle":"2025-11-30T06:16:33.908387Z","shell.execute_reply.started":"2025-11-30T06:16:33.898826Z","shell.execute_reply":"2025-11-30T06:16:33.907501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PROCESSED_DATA_DIRS_MASKS = [\n    \"/kaggle/input/binary-masks-dataset/masks_quart1_2_5\",\n    \"/kaggle/input/binary-masks-dataset/masks_quart2_2_5\",\n    \"/kaggle/input/binary-masks-dataset/masks_quart3_2_5\",\n    \"/kaggle/input/binary-masks-dataset/masks_quart4_2_5\",\n]\nmask_sizes = []\nall_files = []\nsids_empty_not_lost = []\nfor mdir in PROCESSED_DATA_DIRS_MASKS:\n    for mfile in os.listdir(mdir):\n        all_files.append(os.path.join(mdir, mfile))\n        sids_empty_not_lost.append(mfile[:-4])\n\nprint(f'After removing masks fully lost in the 160^3 volume we got from {len(train)} to {len(sids_empty_not_lost)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:16:40.189556Z","iopub.execute_input":"2025-11-30T06:16:40.189805Z","iopub.status.idle":"2025-11-30T06:16:40.205694Z","shell.execute_reply.started":"2025-11-30T06:16:40.189786Z","shell.execute_reply":"2025-11-30T06:16:40.20469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for fpath in tqdm(all_files, desc=\"Processing masks\"):\n    mask = np.load(fpath)[\"vol\"].astype(np.float32)\n    mask_sizes.append(float(mask.sum()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:16:45.668125Z","iopub.execute_input":"2025-11-30T06:16:45.668435Z","iopub.status.idle":"2025-11-30T06:20:47.682204Z","shell.execute_reply.started":"2025-11-30T06:16:45.668414Z","shell.execute_reply":"2025-11-30T06:20:47.68121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"max_size = max(mask_sizes)\nprint(\"Max mask size using a 1.5mm isotropic resampling with a 50 mm cube size:\", max_size)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:20:47.683871Z","iopub.execute_input":"2025-11-30T06:20:47.684211Z","iopub.status.idle":"2025-11-30T06:20:47.690178Z","shell.execute_reply.started":"2025-11-30T06:20:47.684186Z","shell.execute_reply":"2025-11-30T06:20:47.688827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import Counter\n\nsize_counts = Counter(mask_sizes)\nsize_counts_sorted = dict(sorted(size_counts.items()))\nprint(size_counts_sorted)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:20:47.691051Z","iopub.execute_input":"2025-11-30T06:20:47.691361Z","iopub.status.idle":"2025-11-30T06:20:47.70747Z","shell.execute_reply.started":"2025-11-30T06:20:47.691334Z","shell.execute_reply":"2025-11-30T06:20:47.706432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"original_positive_samples = sum(train['Aneurysm Present']==1)\nprint(f'Originally we had a total number of postivie samples: {original_positive_samples} out of {len(train)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:20:47.709673Z","iopub.execute_input":"2025-11-30T06:20:47.710215Z","iopub.status.idle":"2025-11-30T06:20:47.725703Z","shell.execute_reply.started":"2025-11-30T06:20:47.710192Z","shell.execute_reply":"2025-11-30T06:20:47.724681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"total = sum(size_counts_sorted.values())\n\nprint(f'After normalizing input size to 160 cubic sizes {original_positive_samples- total+2485} were lost, having now {total-2485}')\nprint(f'That leaves us with {100*(total-2485)/original_positive_samples}%')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:20:47.726538Z","iopub.execute_input":"2025-11-30T06:20:47.726821Z","iopub.status.idle":"2025-11-30T06:20:47.748483Z","shell.execute_reply.started":"2025-11-30T06:20:47.726793Z","shell.execute_reply":"2025-11-30T06:20:47.747248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"samples_above_80 = 0\nsamples_above_75 = 0\nsamples_above_60 = 0\nfor size, count in size_counts_sorted.items():\n    if size > 110592.0 * 0.8:\n        samples_above_80 += count\n    if size > 110592.0 * 0.75:\n        samples_above_75 += count\n    if size > 110592.0 * 0.60:\n        samples_above_60 += count\n\nprint(f'If we want to keep masks that kept 80% of its original size after standardizing we get {samples_above_80} samples.')\nprint(f'This leaves us with the {100*samples_above_80/original_positive_samples}% of data')\nprint(f'If we want to keep masks that kept 75% of its original size after standardizing we get {samples_above_75} samples.')\nprint(f'This leaves us with the {100*samples_above_75/original_positive_samples}% of data')\nprint(f'If we want to keep masks that kept 60% of its original size after standardizing we get {samples_above_60} samples.')\nprint(f'This leaves us with the {100*samples_above_60/original_positive_samples}% of data')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:20:47.749525Z","iopub.execute_input":"2025-11-30T06:20:47.749982Z","iopub.status.idle":"2025-11-30T06:20:47.767997Z","shell.execute_reply.started":"2025-11-30T06:20:47.749927Z","shell.execute_reply":"2025-11-30T06:20:47.767027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sids_usefull = []\nfor fpath in tqdm(all_files, desc=\"Processing masks\"):\n    mask = np.load(fpath)[\"vol\"].astype(np.float32)\n    size = float(mask.sum())\n    if size > 0 and size >= 110592.0 * 0.8:\n        sids_usefull.append(fpath[len(mdir)+1:-4])\n    if size == 0: \n        sids_usefull.append(fpath[len(mdir)+1:-4])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:25:43.502686Z","iopub.execute_input":"2025-11-30T06:25:43.503532Z","iopub.status.idle":"2025-11-30T06:29:34.719547Z","shell.execute_reply.started":"2025-11-30T06:25:43.503505Z","shell.execute_reply":"2025-11-30T06:29:34.71854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(len(sids_not_usefull)), \n# np.savez_compressed('/kaggle/working/not_usefull.npz', lst=sids_empty_not_lost)\n# #242","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(sids_usefull)), \nnp.savez_compressed('/kaggle/working/usefull.npz', lst=sids_usefull)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:35:31.441866Z","iopub.execute_input":"2025-11-30T06:35:31.442842Z","iopub.status.idle":"2025-11-30T06:35:31.508796Z","shell.execute_reply.started":"2025-11-30T06:35:31.442811Z","shell.execute_reply":"2025-11-30T06:35:31.507662Z"}},"outputs":[],"execution_count":null}]}