{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2 as cv\nimport matplotlib.pyplot as plt\nfrom matplotlib import gridspec\nimport matplotlib.image as img\nfrom keras.datasets import mnist\nfrom keras.utils import to_categorical\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D\nfrom keras.layers import MaxPooling2D\nfrom keras.layers import Dense\nfrom keras.layers import Flatten\nfrom keras.preprocessing.image import ImageDataGenerator\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-27T01:45:37.504959Z","iopub.execute_input":"2024-01-27T01:45:37.505435Z","iopub.status.idle":"2024-01-27T01:45:47.048947Z","shell.execute_reply.started":"2024-01-27T01:45:37.505397Z","shell.execute_reply":"2024-01-27T01:45:47.047826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_data = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\nprint(my_data.head())\nprint(my_data.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-27T01:45:47.051155Z","iopub.execute_input":"2024-01-27T01:45:47.052185Z","iopub.status.idle":"2024-01-27T01:45:47.067314Z","shell.execute_reply.started":"2024-01-27T01:45:47.05214Z","shell.execute_reply":"2024-01-27T01:45:47.066503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\nprint(train.head())\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-27T01:45:47.068852Z","iopub.execute_input":"2024-01-27T01:45:47.069507Z","iopub.status.idle":"2024-01-27T01:45:47.084158Z","shell.execute_reply.started":"2024-01-27T01:45:47.069467Z","shell.execute_reply":"2024-01-27T01:45:47.08322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info","metadata":{"execution":{"iopub.status.busy":"2024-01-27T01:45:47.086221Z","iopub.execute_input":"2024-01-27T01:45:47.086532Z","iopub.status.idle":"2024-01-27T01:45:47.097847Z","shell.execute_reply.started":"2024-01-27T01:45:47.086505Z","shell.execute_reply":"2024-01-27T01:45:47.097085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\nprint(sample_submission.head())\nsample_submission.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-27T01:45:47.099045Z","iopub.execute_input":"2024-01-27T01:45:47.099904Z","iopub.status.idle":"2024-01-27T01:45:47.121341Z","shell.execute_reply.started":"2024-01-27T01:45:47.099872Z","shell.execute_reply":"2024-01-27T01:45:47.120319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = img.imread(\"/kaggle/input/UBC-OCEAN/train_images/13387.png\")\nplt.imshow(image)\nimage.size","metadata":{"execution":{"iopub.status.busy":"2024-01-27T01:45:47.122507Z","iopub.execute_input":"2024-01-27T01:45:47.122818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image2 = img.imread(\"/kaggle/input/UBC-OCEAN/train_thumbnails/13387_thumbnail.png\")\nplt.imshow(image2)\nimage2.size","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels=train['label'].unique()\nprint(labels)\nid_train=train['image_id'].value_counts()\nprint(id_train)\ntrain_is_tma=train[train['is_tma']==False].value_counts()\nprint(train_is_tma)\ntrain_is_tma1=train[train['is_tma']==True].value_counts()\nprint(train_is_tma1)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_is_tma_false = train[train['is_tma'] == False]\nunique_image_ids = train_is_tma_false['image_id'].unique()\nprint(unique_image_ids)\n\nunique_labels = train_is_tma_false['label'].unique()\n\nprint(unique_labels)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os , sys\nfrom PIL import Image\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\nfnames = os.listdir(root)\nlen(fnames)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root3 = \"/kaggle/input/UBC-OCEAN/train_images\"\nfnames3 = os.listdir(root3)\nlen(fnames3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root2 = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\nfnames2 = os.listdir(root2)\nlen(fnames2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root4 = \"/kaggle/input/UBC-OCEAN/test_images\"\nfnames4 = os.listdir(root4)\nlen(fnames4)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=2, ncols=5, figsize=(10, 8))\naxs = axs.flatten()\nfor i in range(10):\n    filepath = os.path.join(root,fnames[i])\n    img = Image.open(filepath)\n    #plt.subplots_adjust(wspace=0.5, hspace=0.5)\n    #axs[i].set_title(fnames[i], fontsize=8, y=1.02)  # Adjust the y value as needed\n    axs[i].imshow(img)\n    axs[i].axis(\"off\")\n    axs[i].set_title(fnames[i])\n       \nplt.tight_layout()    \nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_thumbnails =  (\"/kaggle/input/UBC-OCEAN/train_thumbnails\")\n\ndef resize():\n    for item in os.listdir(path_thumbnails):\n        if os.path.isfile(item):\n            im = Image.open(item)\n            f, e = os.path.splitext(item)\n            imResize = im.resize((200,200), Image.ANTIALIAS)\n            imResize.save(f + ' resized.jpg', 'JPEG', quality=90)\n\nresize()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_data=[]\nimage_label=[]\nfor img , label in zip(train_is_tma_false['image_id'],train_is_tma_false['label']):\n    image_path=os.path.join('/kaggle/input/UBC-OCEAN/train_thumbnails',str(img))\n    image = Image.open(filepath)\n    image = image.resize((512,512))\n    image = image.convert(\"RGB\")\n    image = np.array(image)\n    image_data.append(image)\n    image_label.append(label)\n\nprint(len(image_data))\nprint(len( image_label))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_mapping = {\"CC\": 0, \"EC\": 1, \"HGSC\": 2, \"LGSC\": 3, \"MC\": 4}\n\nimage_label_1 = [label_mapping[i] for i in image_label]\n\nprint(image_label_1)\nprint(len(image_label_1))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x=np.array(image_data)\ny=np.array(image_label_1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''image_label_1 = []\nfor i in image_label:\n    if i==\"CC\":\n        image_label_1.append(0)\n    elif i==\"EC\":\n        image_label_1.append(1)\n    elif i==\"HGSC\":\n        image_label_1.append(2)\n    elif i==\"LGSC\":\n        image_label_1.append(3)\n    elif i==\"MC\":\n        image_label_1.append(4)\nprint(image_label_1 )\nprint(len(image_label_1 ))'''\n","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = cv.imread('/kaggle/input/UBC-OCEAN/train_thumbnails/13387_thumbnail.png')\n#img= cv.cvtColor(img, cv.COLOR_BGR2GRAY)\nplt.figure(figsize=(15, 5))#(15inch, 5inch)= (15*80 pixels, 5*80 pixels)\nplt.title('input image')\nplt.imshow(img)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Normalize the image\nimg_normalized = cv.normalize(img, None, 0, 1.0,cv.NORM_MINMAX, dtype=cv.CV_32F)\n#img_normalized = cv.normalize(img, None, 0, 255,cv.NORM_MINMAX, dtype=cv.CV_8U)\nplt.figure(figsize=(15, 5))\nplt.title('Normalized Image')\nplt.imshow(img_normalized)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom sklearn.model_selection import train_test_split\n\n# Specify the directory containing your image data\ndata_dir = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\n\n# List all image files in the directory\nimage_files = [f for f in os.listdir(data_dir) if f.endswith('.png')]  # Adjust the file extension accordingly\n\n# Split the image files into training, testing, and validation sets\ntrain_files, temp_files = train_test_split(image_files, test_size=0.3, random_state=42)\ntest_files, val_files = train_test_split(temp_files, test_size=0.5, random_state=42)\n\n# Print the number of images in each set\nprint(\"Training set size:\",len(train_files))\nprint(\"Testing set size:\",len(test_files))\nprint(\"Validation set size:\",len(val_files))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\n# Define a custom function to normalize an image\ndef normalize_image(img):\n    # Normalize the image to the range 0 to 1\n    norm_img = cv2.normalize(img, None, 0, 1, cv2.NORM_MINMAX, dtype=cv2.CV_32F)\n    return norm_img\n\n# Create a list of images\nimages = [cv2.imread('/kaggle/input/UBC-OCEAN/train_thumbnails/13987_thumbnail.png'), \n          cv2.imread('/kaggle/input/UBC-OCEAN/train_thumbnails/13526_thumbnail.png'), \n          cv2.imread('/kaggle/input/UBC-OCEAN/train_thumbnails/13387_thumbnail.png')]\n\n# Iterate over the list and normalize each image\nnormalized_images = []\nfor img in images:\n    normalized_images.append(normalize_image(img))\n\n# Display the first normalized image\ncv2.imshow('Normalized Image', normalized_images[0])\ncv2.waitKey(0)\ncv2.destroyAllWindows","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nfrom PIL import Image\n\n# Define the folder path and the pattern for the images\nfolder_path = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\nimage_pattern = \"*.png\"\n\n# Create an empty list to store the images\nimage_list = []\n\n# Loop through all the files that match the pattern\nfor img in glob.glob(folder_path + image_pattern):\n    # Open the image file\n    filepath = os.path.join(root,fnames[i])\n    img = Image.open(filepath)\n    # Append the image to the list\n    image_list.append(img)\n\n# Print the number of images in the list\nprint(len(image_list)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import required library\nimport cv2 as cv\n\n# read the input image in grayscale\nimg = cv.imread(\"/kaggle/input/UBC-OCEAN/train_thumbnails/13387_thumbnail.png\",0)\nprint(\"Image data before Normalize:\\n\", img)\n\n# Normalize the image\nimg_normalized = cv.normalize(img, None, 0, 1,\ncv.NORM_MINMAX, dtype=cv.CV_32F)\n\n# visualize the normalized image\n#cv.imshow('Normalized Image' , img_normalized)\n#cv.waitKey(0)\n#cv.destroyAllWindows()\nprint(\"Image data after Normalize:\\n\", img_normalized)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_normalized = cv.normalize(img, None, 0, 1.0, cv.NORM_MINMAX, dtype=cv.CV_32F)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train, x_test, y_train, y_test = train_test_split(root.id_code, root.diagnosis, test_size=0.2,\n                                                    random_state=SEED, stratify=df_train.diagnosis)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(trainX, trainY), (testX, testY) = mnist.load_data()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}