{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-07T07:34:18.993303Z","iopub.execute_input":"2023-10-07T07:34:18.993673Z","iopub.status.idle":"2023-10-07T07:34:18.999912Z","shell.execute_reply.started":"2023-10-07T07:34:18.993643Z","shell.execute_reply":"2023-10-07T07:34:18.998643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Exploring the top 20 rows of data","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nfrom skimage import io\nimport os\nimport seaborn as sns\ntrain_df = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntrain_df.head(20)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-07T06:12:54.324041Z","iopub.execute_input":"2023-10-07T06:12:54.324429Z","iopub.status.idle":"2023-10-07T06:12:54.341269Z","shell.execute_reply.started":"2023-10-07T06:12:54.324398Z","shell.execute_reply":"2023-10-07T06:12:54.340214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"See the distribution of data\n","metadata":{}},{"cell_type":"code","source":"train_df['label'].value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-07T06:14:01.465242Z","iopub.execute_input":"2023-10-07T06:14:01.46565Z","iopub.status.idle":"2023-10-07T06:14:01.480501Z","shell.execute_reply.started":"2023-10-07T06:14:01.465619Z","shell.execute_reply":"2023-10-07T06:14:01.479335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's view an image","metadata":{}},{"cell_type":"code","source":"io.imshow(\"/kaggle/input/UBC-OCEAN/train_thumbnails/1080_thumbnail.png\")\n","metadata":{"execution":{"iopub.status.busy":"2023-10-07T06:35:33.656593Z","iopub.execute_input":"2023-10-07T06:35:33.656943Z","iopub.status.idle":"2023-10-07T06:35:34.99999Z","shell.execute_reply.started":"2023-10-07T06:35:33.656917Z","shell.execute_reply":"2023-10-07T06:35:34.999054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Count non-black pixels in image","metadata":{}},{"cell_type":"code","source":"import cv2\nimage = cv2.imread(\"/kaggle/input/UBC-OCEAN/train_thumbnails/1080_thumbnail.png\", 0)\ncount = cv2.countNonZero(image)\nprint(count)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-07T06:40:54.716769Z","iopub.execute_input":"2023-10-07T06:40:54.71771Z","iopub.status.idle":"2023-10-07T06:40:54.92291Z","shell.execute_reply.started":"2023-10-07T06:40:54.717673Z","shell.execute_reply":"2023-10-07T06:40:54.921589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's count the number of objects in an image.","metadata":{}},{"cell_type":"code","source":"image = cv2.imread(\"/kaggle/input/UBC-OCEAN/train_thumbnails/1080_thumbnail.png\") \ngray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY) \n# plt.imshow(gray,cmap='gray')\nblur = cv2.GaussianBlur(gray, (11, 11), 0) \n# plt.imshow(blur,cmap='gray')\n\ncanny = cv2.Canny(blur, 30, 150, 3) \n# plt.imshow(canny,cmap='gray')\n\ndilated = cv2.dilate(canny, (1, 1), iterations=0) \nplt.imshow(dilated,cmap='gray')\n\n(cnt, hierarchy) = cv2.findContours( \n    dilated.copy(), cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_NONE) \nrgb = cv2.cvtColor(image, cv2.COLOR_BGR2RGB) \ncv2.drawContours(rgb, cnt, -1, (0, 255, 0), 2) \nprint(len(cnt))  \nplt.imshow(rgb) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Hello, i am looking into the features that can contribute towards the successfull detection of type of carcinoma in this dataset. Cancer is usually hard cells of the body so they usually show up in multiple colors. We can find pixels of specific color and connected regions that can help us to get a good accuracy. Support my work and findings uptil now and do comment if you find the direction should be in some other way. Thanks**","metadata":{}}]}