{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":6917177,"sourceType":"datasetVersion","datasetId":3889865},{"sourceId":150735049,"sourceType":"kernelVersion"}],"dockerImageVersionId":30580,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%matplotlib inline\n\nimport os\n# import pyvips\nimport numpy as np\nfrom PIL import Image\nfrom sklearn.preprocessing import LabelEncoder\nimport matplotlib.pyplot as plt\nimport pandas as pd\n\nimport os, glob\n# import pyvips\nimport numpy as np\nimport random\nfrom PIL import Image\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\n\nimport torch\n\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nimport tensorflow_addons as tfa\nimport cv2\nfrom collections import Counter\n\n\n\ntrain_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/test.csv\")\n\n\nBASE_DIR = [\"/kaggle/input/UBC-OCEAN/train_thumbnails/\", \"/kaggle/input/UBC-OCEAN/test_thumbnails/\"]\nTRAIN_DIR= \"/kaggle/input/UBC-OCEAN/train_images\"\nTEST_DIR= \"/kaggle/input/UBC-OCEAN/test_images\"","metadata":{"execution":{"iopub.status.busy":"2023-11-15T14:23:30.631914Z","iopub.execute_input":"2023-11-15T14:23:30.632744Z","iopub.status.idle":"2023-11-15T14:23:36.641409Z","shell.execute_reply.started":"2023-11-15T14:23:30.63271Z","shell.execute_reply":"2023-11-15T14:23:36.640636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.cluster import MiniBatchKMeans\nfrom tensorflow.keras.applications.resnet50 import ResNet50, preprocess_input\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Model\n\nimport numpy as np\n\n# Load the ResNet50 model pre-trained on ImageNet data, without the top classification layer\nmodel = ResNet50(weights='imagenet', include_top=False, pooling='avg', input_shape=(512, 512, 3))\n\n# Set up your data generator - adjust based on your data\ndatagen = ImageDataGenerator(preprocessing_function=preprocess_input)\ngenerator = datagen.flow_from_directory(\n    '/kaggle/input/tiles-of-cancer-2048px-scale-0-25/',\n    target_size=(512, 512),\n    batch_size=64,\n    class_mode=None,  # This ensures the generator does not return labels\n    shuffle=False)\n\n# Predict using the generator to get feature vectors\n# The 'steps' parameter is set so the model predicts the whole dataset\nfeatures = model.predict(generator, steps=len(generator))\n\nnp.save(\"features\", features)\n\n# Assuming 'train_df' is a DataFrame with 'image_id' and 'label' columns\n# Create a dictionary for quick lookup\nid_label_dict = train_df.set_index('image_id')['label'].to_dict()\n\n# Use list comprehensions for efficiency\nimage_ids = [int(os.path.dirname(file).split('/')[4]) for file in generator.filepaths]\nclasses = [id_label_dict[image_id] for image_id in image_ids]\n\n# Save to a numpy file\nnp.save(\"classes\", np.array(classes))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-15T14:47:01.505845Z","iopub.execute_input":"2023-11-15T14:47:01.506615Z","iopub.status.idle":"2023-11-15T14:50:08.220267Z","shell.execute_reply.started":"2023-11-15T14:47:01.506582Z","shell.execute_reply":"2023-11-15T14:50:08.219176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.cluster import MiniBatchKMeans\nimport pickle\n\n# Assuming X is your data, a NumPy array or Pandas DataFrame\n# Set the number of clusters and batch size\nn_clusters = 4\nbatch_size = 100\n\n# Initialize MiniBatchKMeans\nmbkmeans = MiniBatchKMeans(n_clusters=n_clusters, batch_size=batch_size)\n\n# Fit the model to the data\nmbkmeans.fit(features)\n\n# Get the cluster centers and labels\ncenters = mbkmeans.cluster_centers_\nlabels = mbkmeans.labels_\n\n# Saving the model to a file\nwith open('model-kmeans.pkl', 'wb') as file:\n    pickle.dump(mbkmeans, file)\n\n# Later on, loading the model from the file\nwith open('model-kmeans.pkl', 'rb') as file:\n    loaded_model = pickle.load(file)\n\ncolumns = [f'pixel{i}' for i in range(1, 2049)]\ndf = pd.DataFrame(features, columns=columns)\ndf['cluster']=labels\ndf['class']=classes\ndf['image_id']=image_ids\n\ndf.to_csv('feature_vectors.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-15T14:54:27.598277Z","iopub.execute_input":"2023-11-15T14:54:27.598664Z","iopub.status.idle":"2023-11-15T14:54:44.28899Z","shell.execute_reply.started":"2023-11-15T14:54:27.598634Z","shell.execute_reply":"2023-11-15T14:54:44.288075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import PIL\n\ndef generate_plot(index):\n    \n    selected_images=[]\n    for idx in index:\n        selected_images.append(np.array(PIL.Image.open(generator.filepaths[idx])))\n\n    # Set up the figure and axes for a 2x5 grid to display the images\n    fig, axes = plt.subplots(nrows=5, ncols=10, figsize=(10, 4))\n\n    # Flatten the array of axes for easy iteration\n    axes = axes.flatten()\n\n    # Iterate through the selected images and plot them\n    for i, ax in enumerate(axes):\n        # Squeeze the single channel for display\n        image = np.squeeze(selected_images[i])\n\n        # Display the image\n        ax.imshow(image, cmap='gray')  # Use grayscale color map\n        ax.axis('off')  # Hide the axes ticks\n\n    # Adjust the layout\n    plt.tight_layout()\n    plt.show()\n    \ndef create_index_dict(lst):\n    index_dict = {}\n    \n    for i, value in enumerate(lst):\n        if value not in index_dict:\n            index_dict[value] = []\n        index_dict[value].append(i)\n\n    return index_dict\n\nmy_list = labels\nindex_dict = create_index_dict(my_list)\n\nfor label in index_dict.keys():\n    print(\"----------------Label:\", label, \"----------------\")\n    generate_plot(random.sample(index_dict[label], 50))","metadata":{"execution":{"iopub.status.busy":"2023-11-15T14:54:44.290545Z","iopub.execute_input":"2023-11-15T14:54:44.290853Z","iopub.status.idle":"2023-11-15T14:54:59.030227Z","shell.execute_reply.started":"2023-11-15T14:54:44.290827Z","shell.execute_reply":"2023-11-15T14:54:59.029265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}