{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":18647,"databundleVersionId":1126921,"sourceType":"competition"}],"dockerImageVersionId":30559,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom tqdm.notebook import tqdm\nimport zipfile\nimport tensorflow.keras as K\nfrom tensorflow.keras.layers import Input\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import EarlyStopping\nimport numpy as np\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import roc_auc_score, roc_curve\nimport matplotlib.pyplot as plt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nimport skimage.io\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport PIL.Image\nfrom sklearn.model_selection import StratifiedKFold\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import cohen_kappa_score\nfrom tqdm import tqdm_notebook as tqdm","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:23:17.317034Z","iopub.execute_input":"2024-08-26T01:23:17.317554Z","iopub.status.idle":"2024-08-26T01:23:17.851795Z","shell.execute_reply.started":"2024-08-26T01:23:17.317528Z","shell.execute_reply":"2024-08-26T01:23:17.851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '../input/prostate-cancer-grade-assessment'\ndf_train = pd.read_csv(os.path.join(data_dir, 'train.csv'))\n","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:23:17.853116Z","iopub.execute_input":"2024-08-26T01:23:17.853387Z","iopub.status.idle":"2024-08-26T01:23:17.885623Z","shell.execute_reply.started":"2024-08-26T01:23:17.853363Z","shell.execute_reply":"2024-08-26T01:23:17.884919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gleason_6 = df_train[df_train['isup_grade'].isin([0, 1, 2, 3, 4, 5])]\n\n# Select 100 rows from the filtered DataFrame\nselected_data = gleason_6.head(100)\n\n# Print the selected data\nprint(selected_data)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:23:17.887511Z","iopub.execute_input":"2024-08-26T01:23:17.887791Z","iopub.status.idle":"2024-08-26T01:23:17.908171Z","shell.execute_reply.started":"2024-08-26T01:23:17.887766Z","shell.execute_reply":"2024-08-26T01:23:17.907301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image ids in NAME variable with their respective classes in CLASS variable\nNAME= selected_data['image_id'].tolist()\nCLASS=selected_data['isup_grade'].tolist()\nNAME","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:23:17.909401Z","iopub.execute_input":"2024-08-26T01:23:17.909722Z","iopub.status.idle":"2024-08-26T01:23:17.919079Z","shell.execute_reply.started":"2024-08-26T01:23:17.909689Z","shell.execute_reply":"2024-08-26T01:23:17.918341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=[]\ncount=0\nfor i in NAME:\n df.append(df_train[df_train['image_id']==i])\n count+=1\ndf_train=pd.DataFrame(np.array(df).reshape(100,4),columns=['image_id','data_provider','isup_grade','gleason_score'])\ndf_train","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:23:17.920218Z","iopub.execute_input":"2024-08-26T01:23:17.920547Z","iopub.status.idle":"2024-08-26T01:23:18.157681Z","shell.execute_reply.started":"2024-08-26T01:23:17.920518Z","shell.execute_reply":"2024-08-26T01:23:18.156674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN = '../input/prostate-cancer-grade-assessment/train_images/'\nMASKS = '../input/prostate-cancer-grade-assessment/train_label_masks/'\nOUT_TRAIN = 'train.zip'\nOUT_MASKS = 'masks.zip'\nsz = 256 # Size of each tile\nN = 36 # Total no of tiles","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:23:18.158973Z","iopub.execute_input":"2024-08-26T01:23:18.159341Z","iopub.status.idle":"2024-08-26T01:23:18.164541Z","shell.execute_reply.started":"2024-08-26T01:23:18.1593Z","shell.execute_reply":"2024-08-26T01:23:18.163632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tile(img, mask, sz=256, N=16, cutoff_threshold=0.9):\n    result = []\n    shape = img.shape\n    pad0, pad1 = (sz - shape[0] % sz) % sz, (sz - shape[1] % sz) % sz\n    img = np.pad(img, [[pad0 // 2, pad0 - pad0 // 2], [pad1 // 2, pad1 - pad1 // 2], [0, 0]],\n                 constant_values=255)\n    mask = np.pad(mask, [[pad0 // 2, pad0 - pad0 // 2], [pad1 // 2, pad1 - pad1 // 2], [0, 0]],\n                  constant_values=0)\n    img = img.reshape(img.shape[0] // sz, sz, img.shape[1] // sz, sz, 3)\n    img = img.transpose(0, 2, 1, 3, 4).reshape(-1, sz, sz, 3)\n    mask = mask.reshape(mask.shape[0] // sz, sz, mask.shape[1] // sz, sz, 3)\n    mask = mask.transpose(0, 2, 1, 3, 4).reshape(-1, sz, sz, 3)\n\n    # Filter tiles using tile_cutoff\n    valid_tiles = tile_cutoff(img, cutoff_threshold)\n    img = img[valid_tiles]\n    mask = mask[valid_tiles]\n\n    if len(img) < N:\n        mask = np.pad(mask, [[0, N - len(img)], [0, 0], [0, 0], [0, 0]], constant_values=0)\n        img = np.pad(img, [[0, N - len(img)], [0, 0], [0, 0], [0, 0]], constant_values=255)\n    idxs = np.argsort(img.reshape(img.shape[0], -1).sum(-1))[:N]\n    img = img[idxs]\n    mask = mask[idxs]\n    for i in range(len(img)):\n        augmented_img, augmented_mask = augment_tile(img[i], mask[i])\n        result.append({'img': augmented_img, 'mask': augmented_mask, 'idx': i})\n    return result\n\ndef tile_cutoff(img_tiles, threshold):\n    \"\"\"Filter out tiles that are predominantly white or gray.\"\"\"\n    valid_tiles = []\n    for i, tile in enumerate(img_tiles):\n        if np.mean(tile) / 255.0 < threshold:\n            valid_tiles.append(i)\n    return np.array(valid_tiles)\n\ndef augment_tile(img, mask):\n    \"\"\"Apply augmentations to a tile.\"\"\"\n    # Random horizontal flip\n    if np.random.rand() > 0.5:\n        img = np.fliplr(img)\n        mask = np.fliplr(mask)\n    # Random vertical flip\n    if np.random.rand() > 0.5:\n        img = np.flipud(img)\n        mask = np.flipud(mask)\n    # Random rotation\n    if np.random.rand() > 0.5:\n        k = np.random.randint(0, 4)\n        img = np.rot90(img, k)\n        mask = np.rot90(mask, k)\n    # Additional augmentations can be added here\n    return img, mask\n","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:23:18.16615Z","iopub.execute_input":"2024-08-26T01:23:18.166608Z","iopub.status.idle":"2024-08-26T01:23:18.184947Z","shell.execute_reply.started":"2024-08-26T01:23:18.166574Z","shell.execute_reply":"2024-08-26T01:23:18.184102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_tot, x2_tot = [], []\nnames = [name[:-10] for name in os.listdir(MASKS)][:100]  # Limit to 100 images\nwith zipfile.ZipFile(OUT_TRAIN, 'w') as img_out, zipfile.ZipFile(OUT_MASKS, 'w') as mask_out:\n    for name in tqdm(names):\n        img = skimage.io.MultiImage(os.path.join(TRAIN, name + '.tiff'))[-1]\n        mask = skimage.io.MultiImage(os.path.join(MASKS, name + '_mask.tiff'))[-1]\n        tiles = tile(img, mask)\n        for t in tiles:\n            img, mask, idx = t['img'], t['mask'], t['idx']\n            x_tot.append((img / 255.0).reshape(-1, 3).mean(0))\n            x2_tot.append(((img / 255.0) ** 2).reshape(-1, 3).mean(0))\n            # if read with PIL RGB turns into BGR\n            img = cv2.imencode('.png', cv2.cvtColor(img, cv2.COLOR_RGB2BGR))[1]\n            img_out.writestr(f'{name}_{idx}.png', img)\n            mask = cv2.imencode('.png', mask[:, :, 0])[1]\n            mask_out.writestr(f'{name}_{idx}.png', mask)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:23:18.186071Z","iopub.execute_input":"2024-08-26T01:23:18.186403Z","iopub.status.idle":"2024-08-26T01:23:57.460909Z","shell.execute_reply.started":"2024-08-26T01:23:18.18637Z","shell.execute_reply":"2024-08-26T01:23:57.459401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!unzip /kaggle/working/train.zip","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:23:57.462059Z","iopub.status.idle":"2024-08-26T01:23:57.462562Z","shell.execute_reply.started":"2024-08-26T01:23:57.462301Z","shell.execute_reply":"2024-08-26T01:23:57.462325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport os\n\nout_dir = '/kaggle/working/'\nimg_train = np.empty((100, 4 * 256, 4 * 256, 3))  # Adjust dimensions for 4x4 grid\ncount = -1\n\nfor i in NAME:  # NAME should be a list of base names of the images\n    img_var = []\n    count += 1\n    valid_tiles = True\n    for j in range(0, 16):  # Iterate through each tile index (0 to 15, 16 tiles)\n        img_path = os.path.join(out_dir, f'{i}_{j}.png')\n        if os.path.exists(img_path):\n            img_tile = cv2.imread(img_path)\n            if img_tile is None:\n                print(f\"Warning: Tile image {img_path} could not be read (NoneType).\")\n                valid_tiles = False\n                break\n            img_var.append(img_tile)\n        else:\n            print(f\"Warning: Tile image {img_path} does not exist.\")\n            valid_tiles = False\n            break\n    \n    if valid_tiles:\n        img_var = np.array(img_var)\n        if img_var.shape[0] == 16:  # Ensure there are 16 tiles\n            try:\n                img_var = img_var.reshape(4 * 256, 4 * 256, 3)  # Reshape to 4x4 grid of tiles\n                img_train[count] = img_var\n            except ValueError as e:\n                print(f\"Error reshaping the image tiles for {i}: {e}\")\n        else:\n            print(f\"Warning: Unexpected number of tiles for {i}. Expected 16, got {img_var.shape[0]}\")\n    else:\n        print(f\"Warning: Skipping image {i} due to missing or unreadable tiles.\")\n\nprint(\"Finished processing images.\")\n\n# Ensure the rest of the script uses `img_train` as required\n","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:24:03.826558Z","iopub.execute_input":"2024-08-26T01:24:03.826924Z","iopub.status.idle":"2024-08-26T01:24:03.841018Z","shell.execute_reply.started":"2024-08-26T01:24:03.826894Z","shell.execute_reply":"2024-08-26T01:24:03.839781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_train=img_train.reshape(100,-1) # for smote function to work, it requires only 2 dimensional array","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:24:09.33986Z","iopub.execute_input":"2024-08-26T01:24:09.340208Z","iopub.status.idle":"2024-08-26T01:24:09.344617Z","shell.execute_reply.started":"2024-08-26T01:24:09.340181Z","shell.execute_reply":"2024-08-26T01:24:09.343665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 12 to 24 images using smote analysis\nfrom collections import Counter\nfrom imblearn.over_sampling import SMOTE\ncounter = Counter(CLASS)\nprint('Before',counter)\n# oversampling the train dataset using SMOTE\nsmt = SMOTE(sampling_strategy={0:31,1:31,2:31,3:31,4:31,5:31},k_neighbors=7)\n#X_train, y_train = smt.fit_resample(X_train, y_train)\nX_train_sm, y_train_sm = smt.fit_resample(img_train, CLASS)\n\ncounter = Counter(y_train_sm)\nprint('After',counter)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:24:14.478531Z","iopub.execute_input":"2024-08-26T01:24:14.478891Z","iopub.status.idle":"2024-08-26T01:24:23.919375Z","shell.execute_reply.started":"2024-08-26T01:24:14.47886Z","shell.execute_reply":"2024-08-26T01:24:23.918444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train_sm.shape)\nprint(X_train_sm.size)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:24:23.921283Z","iopub.execute_input":"2024-08-26T01:24:23.921977Z","iopub.status.idle":"2024-08-26T01:24:23.926732Z","shell.execute_reply.started":"2024-08-26T01:24:23.921924Z","shell.execute_reply":"2024-08-26T01:24:23.925882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\n\n# Define the desired dimensions for each image\nnew_width = 256\nnew_height = 256\n\n# List to store resized images\nresized_images = []\n\n# Loop through each image in X_train_sm and resize\nfor img_data in X_train_sm:\n    # Reshape the 1D array back into image dimensions\n    img = img_data.reshape(4*256, 4*256, 3)\n    \n    # Convert numpy array to PIL Image\n    pil_img = Image.fromarray(img.astype('uint8'))\n    \n    # Resize the image\n    resized_img = pil_img.resize((new_width, new_height))\n    \n    # Convert back to numpy array\n    resized_img_data = np.array(resized_img)\n    \n    # Append resized image to the list\n    resized_images.append(resized_img_data)\n\n# Convert the list of resized images back to numpy array\nX_train_resized = np.array(resized_images)\n\n# Check the shape of the resized array\nprint(X_train_resized.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:24:23.927912Z","iopub.execute_input":"2024-08-26T01:24:23.928246Z","iopub.status.idle":"2024-08-26T01:24:26.866969Z","shell.execute_reply.started":"2024-08-26T01:24:23.928214Z","shell.execute_reply":"2024-08-26T01:24:26.86567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_train=pd.get_dummies(y_train_sm).values # onehotencoding","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:24:26.874917Z","iopub.execute_input":"2024-08-26T01:24:26.878042Z","iopub.status.idle":"2024-08-26T01:24:26.891118Z","shell.execute_reply.started":"2024-08-26T01:24:26.877995Z","shell.execute_reply":"2024-08-26T01:24:26.890007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n!pip install git+https://github.com/qubvel/classification_models.git","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:24:26.892309Z","iopub.execute_input":"2024-08-26T01:24:26.892713Z","iopub.status.idle":"2024-08-26T01:24:43.245115Z","shell.execute_reply.started":"2024-08-26T01:24:26.892674Z","shell.execute_reply":"2024-08-26T01:24:43.24378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from classification_models.tfkeras import Classifiers\n\n# Get the model and preprocessing function\nResNeXt50, preprocess_input = Classifiers.get('resnext50')\n\n# Create the model with desired input shape and weights (optional)\nresmodel = ResNeXt50(\n    include_top=False,  # Set to True for classification (optional)\n    input_shape=(256, 256, 3),\n    weights='imagenet'  # Load pre-trained weights (optional)\n)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:24:43.247063Z","iopub.execute_input":"2024-08-26T01:24:43.247438Z","iopub.status.idle":"2024-08-26T01:24:53.321782Z","shell.execute_reply.started":"2024-08-26T01:24:43.247402Z","shell.execute_reply":"2024-08-26T01:24:53.320966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining the model\nimport tensorflow.keras as K\nfrom tensorflow.keras.layers import Input\n\n# Define the input shape based on the resized image dimensions\ninput_t = Input(shape=(256, 256, 3))\n\n# Load ResNet50 model with the updated input shape\n# res_model = K.applications.ResNeXt50(include_top=False, weights=\"imagenet\", input_tensor=input_t)\n\n# Build the rest of the model\nmodel = K.models.Sequential()\nmodel.add(resmodel)\nmodel.add(K.layers.Flatten())\nmodel.add(K.layers.Dense(6, activation='softmax'))\nmodel.compile(loss='categorical_crossentropy',optimizer=K.optimizers.RMSprop(lr=0.001),metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:24:53.323081Z","iopub.execute_input":"2024-08-26T01:24:53.323358Z","iopub.status.idle":"2024-08-26T01:24:56.151047Z","shell.execute_reply.started":"2024-08-26T01:24:53.323322Z","shell.execute_reply":"2024-08-26T01:24:56.150243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nk= 5  # Number of folds\nkf = KFold(n_splits=k, shuffle=True, random_state=2)\nfold_no = 1\nscores = []\n\nfor train_index, test_index in kf.split(X_train_resized):\n    print(f'Fold {fold_no}')\n\n    X_train, X_test = X_train_resized[train_index], X_train_resized[test_index]\n    y_train, y_test = Y_train[train_index], Y_train[test_index]\n\n    model.fit(X_train, y_train, epochs=10)\n\n    score = model.evaluate(X_test, y_test)\n    scores.append(score)\n    print(f'Score for fold {fold_no}: {model.metrics_names[1]} of {score[1]*100}%')\n    fold_no += 1\n        # Predict the probability scores for the test set\n    y_pred_prob = model.predict(X_test)\n    for i in range(y_test.shape[1]):\n        fpr, tpr, _ = roc_curve(y_test[:, i], y_pred_prob[:, i])\n        roc_auc = roc_auc_score(y_test[:, i], y_pred_prob[:, i])\n        plt.plot(fpr, tpr, label=f'Class {i} (area = {roc_auc:.2f})')\n        print(f\"{fold_no}: \\nFPR:{fpr}\\nTPR:{tpr}\")\n    plt.title(f'ROC Curve for Fold {fold_no}')\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.legend(loc=\"lower right\")\n    plt.show()\n\nprint(f'Average accuracy: {np.mean([s[1] for s in scores])*100}%')","metadata":{"execution":{"iopub.status.busy":"2024-08-26T01:24:56.152121Z","iopub.execute_input":"2024-08-26T01:24:56.152404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining the model\nimport tensorflow.keras as K\nfrom tensorflow.keras.layers import Input\n\n# Define the input shape based on the resized image dimensions\ninput_t = Input(shape=(256, 256, 3))\n\n# Load ResNet50 model with the updated input shape\nres_model = K.applications.ResNet50(include_top=False, weights=\"imagenet\", input_tensor=input_t)\n\n# Build the rest of the model\nmodel = K.models.Sequential()\nmodel.add(res_model)\nmodel.add(K.layers.Flatten())\nmodel.add(K.layers.Dense(6, activation='softmax'))\nmodel.compile(loss='categorical_crossentropy',optimizer=K.optimizers.RMSprop(lr=0.01),metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:29:33.086481Z","iopub.execute_input":"2024-08-25T05:29:33.08704Z","iopub.status.idle":"2024-08-25T05:29:35.917818Z","shell.execute_reply.started":"2024-08-25T05:29:33.087007Z","shell.execute_reply":"2024-08-25T05:29:35.916874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nk= 5  # Number of folds\nkf = KFold(n_splits=k, shuffle=True, random_state=3)\nfold_no = 1\nscores = []\n\nfor train_index, test_index in kf.split(X_train_resized):\n    print(f'Fold {fold_no}')\n\n    X_train, X_test = X_train_resized[train_index], X_train_resized[test_index]\n    y_train, y_test = Y_train[train_index], Y_train[test_index]\n    model.fit(X_train, y_train, epochs=10)\n\n    score = model.evaluate(X_test, y_test)\n    scores.append(score)\n    print(f'Score for fold {fold_no}: {model.metrics_names[1]} of {score[1]*100}%')\n    fold_no += 1\n        # Predict the probability scores for the test set\n    y_pred_prob = model.predict(X_test)\n    for i in range(y_test.shape[1]):\n        fpr, tpr, _ = roc_curve(y_test[:, i], y_pred_prob[:, i])\n        roc_auc = roc_auc_score(y_test[:, i], y_pred_prob[:, i])\n        plt.plot(fpr, tpr, label=f'Class {i} (area = {roc_auc:.2f})')\n\n    plt.title(f'ROC Curve for Fold {fold_no}')\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.legend(loc=\"lower right\")\n    plt.show()\n    print(f\"{fold_no-1}: \\nFPR:{fpr}\\nTPR:{tpr}\")\n\nprint(f'Average accuracy: {np.mean([s[1] for s in scores])*100}%')\n\n","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:30:24.056236Z","iopub.execute_input":"2024-08-25T05:30:24.056609Z","iopub.status.idle":"2024-08-25T05:32:00.754439Z","shell.execute_reply.started":"2024-08-25T05:30:24.05658Z","shell.execute_reply":"2024-08-25T05:32:00.753404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"try","metadata":{}},{"cell_type":"code","source":"import tensorflow.keras as K\nfrom tensorflow.keras.layers import Input\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import EarlyStopping\nimport numpy as np\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import roc_auc_score, roc_curve\nimport matplotlib.pyplot as plt\n\n\n# Define the input shape based on the resized image dimensions\ninput_t = Input(shape=(256, 256, 3))\n\n# Load ResNet50 model with the updated input shape\nres_model = K.applications.ResNet50(include_top=False, weights=\"imagenet\", input_tensor=input_t)\n\n# Freeze earlier layers for fine-tuning (optional)\nfor layer in res_model.layers[:100]:  # Freeze the first 100 layers\n    layer.trainable = False\n\n# Build the rest of the model\nmodel = K.models.Sequential()\nmodel.add(res_model)\nmodel.add(K.layers.Flatten())\nmodel.add(K.layers.Dense(6, activation='softmax'))\n\n# Hyperparameter Tuning\noptimizer = K.optimizers.RMSprop(learning_rate=0.001)  # Lower learning rate\n\n# Regularization (L2 regularization)\nmodel.compile(loss='categorical_crossentropy', optimizer=optimizer, metrics=['accuracy'])\n\n# Assuming X_train_resized and Y_train are already defined\nk = 5  # Number of folds\nkf = KFold(n_splits=k, shuffle=True, random_state=42)\nfold_no = 1\nscores = []\n\n# Early Stopping (monitor validation loss)\nearly_stopping = EarlyStopping(monitor='val_loss', patience=5)  # Stop training after 5 epochs of no improvement in validation loss\n\nfor train_index, test_index in kf.split(X_train_resized):\n    print(f'Fold {fold_no}')\n\n    X_train, X_test = X_train_resized[train_index], X_train_resized[test_index]\n    y_train, y_test = Y_train[train_index], Y_train[test_index]\n\n    # Decode one-hot encoded labels to categorical labels\n    y_train_cat = np.argmax(y_train, axis=1)\n    y_test_cat = np.argmax(y_test, axis=1)\n\n    # Ensure `classes` contains unique class labels\n    classes = list(set(y_train_cat))\n\n    # Data Augmentation\n    datagen = ImageDataGenerator(\n        rotation_range=20,\n        width_shift_range=0.2,\n        height_shift_range=0.2,\n        shear_range=0.2,\n        zoom_range=0.2,\n        horizontal_flip=True,\n        fill_mode='nearest'\n    )\n\n    train_generator = datagen.flow(\n        X_train, y_train,\n        batch_size=32\n    )\n\n    validation_generator = datagen.flow(\n        X_test, y_test,\n        batch_size=32,\n        shuffle=False\n    )\n\n    # Fit the model with augmented training data and early stopping\n    model.fit(train_generator, steps_per_epoch=len(train_generator), epochs=100, validation_data=validation_generator, callbacks=[early_stopping])\n    score = model.evaluate(X_test, y_test)\n    scores.append(score)\n    print(f'Score for fold {fold_no}: {model.metrics_names[1]} of {score[1]*100}%')\n\n    # Predict the probability scores for the test set\n    y_pred_prob = model.predict(X_test)\n    for i in range(y_test.shape[1]):\n        fpr, tpr, _ = roc_curve(y_test[:, i], y_pred_prob[:, i])\n        roc_auc = roc_auc_score(y_test[:, i], y_pred_prob[:, i])\n        plt.plot(fpr, tpr, label=f'Class {i} (area = {roc_auc:.2f})')\n\n    plt.title(f'ROC Curve for Fold {fold_no}')\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.legend(loc=\"lower right\")\n    plt.show()\n    print(f\"{fold_no}: \\nFPR:{fpr}\\nTPR:{tpr}\")\n\n    fold_no += 1\n\nprint(f'Average accuracy: {np.mean([s[1] for s in scores])*100}%')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom sklearn.model_selection import StratifiedKFold\nfrom tensorflow.keras.applications import VGG16\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Flatten, Dropout, GlobalAveragePooling2D\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n# # Limit GPU memory growth\n# physical_devices = tf.config.list_physical_devices('GPU')\n# if physical_devices:\n#     tf.config.experimental.set_memory_growth(physical_devices[0], True)\n\n# Data preparation\nX = X_train_resized[:149]  # Ensure X and y lengths match\ny = y_train\n\n# Convert one-hot encoded labels to single labels\ny = np.argmax(y, axis=1)\n\n# Check the shapes and lengths of X and y\nprint(\"Adjusted shape of X_train_resized:\", X.shape)\nprint(\"Length of y_train:\", len(y))\n\n# Create a StratifiedKFold object\nkfold = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Data augmentation\ndatagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    horizontal_flip=True\n)\n\n# Callbacks for early stopping and learning rate reduction\nearly_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n\n# Loop through each fold\nfor train_index, val_index in kfold.split(X, y):\n    X_train_fold, X_val = X[train_index], X[val_index]\n    y_train_fold, y_val = y[train_index], y[val_index]\n\n    # Load VGG16 model pre-trained on ImageNet\n    base_model = VGG16(include_top=False, input_shape=(256, 256, 3))\n    \n    # Unfreeze some of the deeper layers\n    for layer in base_model.layers[-4:]:\n        layer.trainable = True\n\n    # Add custom layers on top of the base model\n    x = base_model.output\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(256, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    output = Dense(1, activation='sigmoid')(x)\n\n    # Create the model\n    model = Model(inputs=base_model.input, outputs=output)\n\n    # Compile the model with a smaller learning rate\n    model.compile(optimizer=Adam(learning_rate=0.00001), loss='binary_crossentropy', metrics=['accuracy'])\n\n    # Create data generators\n    train_generator = datagen.flow(X_train_fold, y_train_fold, batch_size=16)\n    \n    # Fit the model\n    history = model.fit(train_generator, \n                        epochs=100, \n                        validation_data=(X_val, y_val), \n                        steps_per_epoch=len(X_train_fold) // 16,\n                        callbacks=[early_stopping],\n                        verbose=1)\n\n    # Evaluate the model\n    scores = model.evaluate(X_val, y_val, verbose=0)\n    print(f'Fold Accuracy: {scores[1] * 100:.2f}%')\n        # Predict the probability scores for the test set\n    y_pred_prob = model.predict(X_test)\n    for i in range(y_test.shape[1]):\n        fpr, tpr, _ = roc_curve(y_test[:, i], y_pred_prob[:, i])\n        roc_auc = roc_auc_score(y_test[:, i], y_pred_prob[:, i])\n        plt.plot(fpr, tpr, label=f'Class {i} (area = {roc_auc:.2f})')\n\n    plt.title(f'ROC Curve for Fold {fold_no}')\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.legend(loc=\"lower right\")\n    plt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom sklearn.model_selection import StratifiedKFold\nfrom tensorflow.keras.applications import VGG16\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Flatten, Dropout\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import Adam\n\n# # Limit GPU memory growth\n# physical_devices = tf.config.list_physical_devices('GPU')\n# if physical_devices:\n#     tf.config.experimental.set_memory_growth(physical_devices[0], True)\n\n# Data preparation\nX = X_train_resized[:149]  # Trimming to match the length of y_train\ny = y_train\n\n# Convert one-hot encoded labels to single labels\ny = np.argmax(y, axis=1)\n\n# Check the shapes and lengths of X and y\nprint(\"Adjusted shape of X_train_resized:\", X.shape)\nprint(\"Length of y_train:\", len(y))\n\n# Create a StratifiedKFold object\nkfold = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Data augmentation\ndatagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    horizontal_flip=True\n)\n\n# Loop through each fold\nfor train_index, val_index in kfold.split(X, y):\n    X_train_fold, X_val = X[train_index], X[val_index]\n    y_train_fold, y_val = y[train_index], y[val_index]\n\n    # Load VGG16 model pre-trained on ImageNet\n    base_model = VGG16(include_top=False, input_shape=(256, 256, 3))\n    \n    # Freeze the layers in the base model\n    for layer in base_model.layers:\n        layer.trainable = False\n\n    # Add custom layers on top of the base model\n    x = Flatten()(base_model.output)\n    x = Dense(256, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    x = Dense(128, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    output = Dense(1, activation='sigmoid')(x)\n\n    # Create the model\n    model = Model(inputs=base_model.input, outputs=output)\n\n    # Compile the model\n    model.compile(optimizer=Adam(learning_rate=0.0001), loss='binary_crossentropy', metrics=['accuracy'])\n\n    # Create data generators\n    train_generator = datagen.flow(X_train_fold, y_train_fold, batch_size=16)\n    \n    # Fit the model\n    history = model.fit(train_generator, \n                        epochs=100, \n                        validation_data=(X_val, y_val), \n                        steps_per_epoch=len(X_train_fold) // 16,\n                        verbose=1)\n\n    # Evaluate the model\n    scores = model.evaluate(X_val, y_val, verbose=0)\n    print(f'Fold Accuracy: {scores[1] * 100:.2f}%')\n        # Predict the probability scores for the test set\n    y_pred_prob = model.predict(X_test)\n    for i in range(y_test.shape[1]):\n        fpr, tpr, _ = roc_curve(y_test[:, i], y_pred_prob[:, i])\n        roc_auc = roc_auc_score(y_test[:, i], y_pred_prob[:, i])\n        plt.plot(fpr, tpr, label=f'Class {i} (area = {roc_auc:.2f})')\n\n    plt.title(f'ROC Curve for Fold {fold_no}')\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.legend(loc=\"lower right\")\n    plt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import StratifiedKFold\n\n# Define image dimensions\nimg_width, img_height = X_train_resized.shape[1], X_train_resized.shape[2]\n\n# Initialize a StratifiedKFold object\nkfold = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Loop over each fold\nfold_no = 1\nfor train_index, test_index in kfold.split(X_train_resized, y_train):\n    # Split data into train and validation sets for this fold\n    X_train_fold, X_val = X_train_resized[train_index], X_train_resized[test_index]\n    y_train_fold, y_val = y_train[train_index], y_train[test_index]\n\n    # Define the model architecture\n    model = Sequential([\n        Conv2D(32, (3, 3), activation='relu', input_shape=(img_width, img_height, 3)),\n        MaxPooling2D((2, 2)),\n        Conv2D(64, (3, 3), activation='relu'),\n        MaxPooling2D((2, 2)),\n        Conv2D(128, (3, 3), activation='relu'),\n        MaxPooling2D((2, 2)),\n        Conv2D(256, (3, 3), activation='relu'),\n        MaxPooling2D((2, 2)),\n        Flatten(),\n        Dense(512, activation='relu'),\n        Dropout(0.5),\n        Dense(1, activation='sigmoid')\n    ])\n\n    # Compile the model\n    model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n    # Data augmentation\n    datagen = ImageDataGenerator(rescale=1./255, rotation_range=20, width_shift_range=0.2,\n                                 height_shift_range=0.2, shear_range=0.2, zoom_range=0.2, horizontal_flip=True)\n\n    # Initialize the ImageDataGenerator for training data\n    train_generator = datagen.flow(X_train_fold, y_train_fold, batch_size=32)\n\n    # Fit the model\n    history = model.fit(train_generator, \n                        epochs=100, \n                        validation_data=(X_val, y_val), \n                        steps_per_epoch=len(X_train_fold) // 32,\n                        verbose=1)\n\n    # Evaluate the model\n    scores = model.evaluate(X_val, y_val, verbose=0)\n    print(f\"Score for fold {fold_no}: accuracy of {scores[1]*100}%\")\n\n    # Increment fold number\n    fold_no += 1\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Defining the model\n# import tensorflow.keras as K\n# from tensorflow.keras.layers import Input\n\n# # Define the input shape based on the resized image dimensions\n# input_t = Input(shape=(1536, 1536, 3))\n\n# # Load ResNet50 model with the updated input shape\n# res_model = K.applications.ResNet50(include_top=False, weights=\"imagenet\", input_tensor=input_t)\n\n# # Build the rest of the model\n# model = K.models.Sequential()\n# model.add(res_model)\n# model.add(K.layers.Flatten())\n# model.add(K.layers.Dense(6, activation='softmax'))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import train_test_split\n\n# train_x, val_x, train_y, val_y = train_test_split(\n#     X_train_resized, Y_train, train_size=0.6, random_state=42)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import train_test_split\n\n# test_x, val_x, test_y, val_y = train_test_split(\n#     val_x, val_y, train_size=0.5, random_state=42)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.compile(loss='categorical_crossentropy',optimizer=K.optimizers.RMSprop(lr=2e-5),metrics=['accuracy'])\n# history = model.fit(\n#     train_x,train_y,\n#     epochs=10,\n#     validation_data=(val_x, val_y)\n# )\n# mode.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import cross_val_score, KFold\n# kf = KFold(n_splits=5, shuffle=True, random_state=42)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cross_val_results = cross_val_score(model, X_train_resized, Y_train, cv=kf)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}