{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99552,"databundleVersionId":13190393,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport zipfile\nfrom tqdm import tqdm\n\nk = 6\nchk = 250","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-16T12:02:38.51263Z","iopub.execute_input":"2025-08-16T12:02:38.512887Z","iopub.status.idle":"2025-08-16T12:02:38.886743Z","shell.execute_reply.started":"2025-08-16T12:02:38.512861Z","shell.execute_reply":"2025-08-16T12:02:38.885787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(os.listdir('/kaggle/input/rsna-intracranial-aneurysm-detection/series'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T12:02:38.888206Z","iopub.execute_input":"2025-08-16T12:02:38.888637Z","iopub.status.idle":"2025-08-16T12:02:39.053712Z","shell.execute_reply.started":"2025-08-16T12:02:38.88861Z","shell.execute_reply":"2025-08-16T12:02:39.052783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv')\ntrain.tail()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T12:02:39.054685Z","iopub.execute_input":"2025-08-16T12:02:39.055037Z","iopub.status.idle":"2025-08-16T12:02:39.120478Z","shell.execute_reply.started":"2025-08-16T12:02:39.055006Z","shell.execute_reply":"2025-08-16T12:02:39.119524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_positives = train[train['Aneurysm Present'] == 1].SeriesInstanceUID.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T12:02:39.12246Z","iopub.execute_input":"2025-08-16T12:02:39.123377Z","iopub.status.idle":"2025-08-16T12:02:39.138946Z","shell.execute_reply.started":"2025-08-16T12:02:39.123344Z","shell.execute_reply":"2025-08-16T12:02:39.137683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"segmented_cases = os.listdir('/kaggle/input/rsna-intracranial-aneurysm-detection/segmentations')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T12:02:39.14017Z","iopub.execute_input":"2025-08-16T12:02:39.140492Z","iopub.status.idle":"2025-08-16T12:02:39.178926Z","shell.execute_reply.started":"2025-08-16T12:02:39.140459Z","shell.execute_reply":"2025-08-16T12:02:39.17797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cases = [c for c in all_positives if c not in segmented_cases]\ncases.sort()\nprint(len(cases))\ncases[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T12:02:39.180441Z","iopub.execute_input":"2025-08-16T12:02:39.181186Z","iopub.status.idle":"2025-08-16T12:02:39.199552Z","shell.execute_reply.started":"2025-08-16T12:02:39.181152Z","shell.execute_reply":"2025-08-16T12:02:39.198553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the zip file name for this chunk\nzip_path = f'/kaggle/working/cases_chunk_{k}.zip'\n\n# Create a new zip file\nwith zipfile.ZipFile(zip_path, 'w', zipfile.ZIP_DEFLATED) as zipf:\n    for c in tqdm(cases[k*chk:(k+1)*chk], desc=f'Zipping chunk {k}'):\n        src_path = f'/kaggle/input/rsna-intracranial-aneurysm-detection/series/{c}'\n        \n        # Walk through the source directory and add files to the zip\n        for root, _, files in os.walk(src_path):\n            for file in files:\n                file_path = os.path.join(root, file)\n                \n                # Maintain relative path structure in the zip\n                rel_path = os.path.relpath(file_path, start=src_path)\n                zipf.write(\n                    filename=file_path,\n                    arcname=os.path.join(c, rel_path)  # Store in zip under case directory\n                )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-16T12:02:39.200447Z","iopub.execute_input":"2025-08-16T12:02:39.200747Z","iopub.status.idle":"2025-08-16T12:30:42.9099Z","shell.execute_reply.started":"2025-08-16T12:02:39.200719Z","shell.execute_reply":"2025-08-16T12:30:42.908652Z"}},"outputs":[],"execution_count":null}]}