{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":18647,"databundleVersionId":1126921,"sourceType":"competition"},{"sourceId":1113957,"sourceType":"datasetVersion","datasetId":624783}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, shutil\nimport pandas as pd\nimport numpy as np\n\n# Config\nN_PATIENTS = 1062      # Number of patients you want\nN_TILES = 12           # Number of tiles per patient\nTILE_DIR = \"/kaggle/input/panda-16x128x128-tiles-data/train\"\nLABEL_CSV = \"/kaggle/input/prostate-cancer-grade-assessment/train.csv\"\nOUT_DIR = \"/kaggle/working/mini_dataset\"\n\n# Make output directory\nos.makedirs(f\"{OUT_DIR}/train\", exist_ok=True)\n\n# Load original labels\ndf = pd.read_csv(LABEL_CSV)\n\n# Identify patients that have _0.png through _11.png\ntile_files = set(os.listdir(TILE_DIR))\ncandidate_ids = []\n\nfor image_id in df['image_id']:\n    has_all_tiles = all(f\"{image_id}_{i}.png\" in tile_files for i in range(N_TILES))\n    if has_all_tiles:\n        candidate_ids.append(image_id)\n\nprint(f\"Found {len(candidate_ids)} patients with exactly {N_TILES} tiles.\")\n\n# Check if enough patients\nif len(candidate_ids) < N_PATIENTS:\n    print(f\"Only {len(candidate_ids)} patients available. Reducing N_PATIENTS.\")\n    N_PATIENTS = len(candidate_ids)\n\n# Sample the patients\nsampled_ids = np.random.choice(candidate_ids, N_PATIENTS, replace=False)\ndf_small = df[df['image_id'].isin(sampled_ids)].reset_index(drop=True)\n\n# Save CSV\ndf_small.to_csv(f\"{OUT_DIR}/train.csv\", index=False)\n\n# Copy exactly 12 tiles for each patient\ntiles_copied = 0\nfor pid in sampled_ids:\n    for i in range(N_TILES):\n        src = os.path.join(TILE_DIR, f\"{pid}_{i}.png\")\n        dst = os.path.join(OUT_DIR, \"train\", f\"{pid}_{i}.png\")\n        shutil.copy(src, dst)\n        tiles_copied += 1\n\nprint(f\"Copied {tiles_copied} tiles for {N_PATIENTS} patients (each with {N_TILES} tiles).\")\nprint(\"Mini dataset is ready at:\", OUT_DIR)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-30T09:44:34.463069Z","iopub.execute_input":"2025-04-30T09:44:34.463828Z","iopub.status.idle":"2025-04-30T09:45:00.283473Z","shell.execute_reply.started":"2025-04-30T09:44:34.463799Z","shell.execute_reply":"2025-04-30T09:45:00.282508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!head /kaggle/working/mini_dataset/train.csv\n!ls /kaggle/working/mini_dataset/train | head\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T09:47:41.871669Z","iopub.execute_input":"2025-04-30T09:47:41.871999Z","iopub.status.idle":"2025-04-30T09:47:42.147356Z","shell.execute_reply.started":"2025-04-30T09:47:41.871974Z","shell.execute_reply":"2025-04-30T09:47:42.146218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tile_counts = pd.Series([f[:32] for f in os.listdir(TILE_DIR)]).value_counts()\nprint(tile_counts.describe())\nprint(tile_counts.value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T08:18:10.097183Z","iopub.execute_input":"2025-04-30T08:18:10.097476Z","iopub.status.idle":"2025-04-30T08:18:11.638971Z","shell.execute_reply.started":"2025-04-30T08:18:10.097454Z","shell.execute_reply":"2025-04-30T08:18:11.637779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndf = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/train.csv')\nprint(len(df))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:09:26.951571Z","iopub.execute_input":"2025-04-30T10:09:26.952337Z","iopub.status.idle":"2025-04-30T10:09:26.977738Z","shell.execute_reply.started":"2025-04-30T10:09:26.9523Z","shell.execute_reply":"2025-04-30T10:09:26.976988Z"}},"outputs":[],"execution_count":null}]}