{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ***Please, upvote if you find the code useful :D***","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:43:50.767754Z","iopub.execute_input":"2022-08-23T14:43:50.768202Z","iopub.status.idle":"2022-08-23T14:43:52.100156Z","shell.execute_reply.started":"2022-08-23T14:43:50.768105Z","shell.execute_reply":"2022-08-23T14:43:52.099078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:43:52.102873Z","iopub.execute_input":"2022-08-23T14:43:52.103403Z","iopub.status.idle":"2022-08-23T14:43:52.120333Z","shell.execute_reply.started":"2022-08-23T14:43:52.10336Z","shell.execute_reply":"2022-08-23T14:43:52.119496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:43:52.121804Z","iopub.execute_input":"2022-08-23T14:43:52.122658Z","iopub.status.idle":"2022-08-23T14:43:52.146304Z","shell.execute_reply.started":"2022-08-23T14:43:52.122599Z","shell.execute_reply":"2022-08-23T14:43:52.14526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import skimage.io as skio\nim1 = skio.imread('../input/mayo-clinic-strip-ai/train/006388_0.tif')\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:43:52.14956Z","iopub.execute_input":"2022-08-23T14:43:52.150227Z","iopub.status.idle":"2022-08-23T14:44:32.824843Z","shell.execute_reply.started":"2022-08-23T14:43:52.150193Z","shell.execute_reply":"2022-08-23T14:44:32.823916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(im1)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:44:32.826175Z","iopub.execute_input":"2022-08-23T14:44:32.826746Z","iopub.status.idle":"2022-08-23T14:44:32.833808Z","shell.execute_reply.started":"2022-08-23T14:44:32.826711Z","shell.execute_reply":"2022-08-23T14:44:32.832675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(im1[0])","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:44:32.835583Z","iopub.execute_input":"2022-08-23T14:44:32.836447Z","iopub.status.idle":"2022-08-23T14:44:32.846627Z","shell.execute_reply.started":"2022-08-23T14:44:32.8364Z","shell.execute_reply":"2022-08-23T14:44:32.845603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(im1)*len(im1[0])*len(im1[0][0])","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:44:32.84811Z","iopub.execute_input":"2022-08-23T14:44:32.848592Z","iopub.status.idle":"2022-08-23T14:44:32.857136Z","shell.execute_reply.started":"2022-08-23T14:44:32.84856Z","shell.execute_reply":"2022-08-23T14:44:32.856229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"im1.size","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:44:32.858844Z","iopub.execute_input":"2022-08-23T14:44:32.859154Z","iopub.status.idle":"2022-08-23T14:44:32.870476Z","shell.execute_reply.started":"2022-08-23T14:44:32.859127Z","shell.execute_reply":"2022-08-23T14:44:32.869436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 6))\n\nsns.countplot(df_train['label'])\nplt.title(\"Distribution of Ratings\", fontsize=20)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:44:32.871926Z","iopub.execute_input":"2022-08-23T14:44:32.872294Z","iopub.status.idle":"2022-08-23T14:44:33.061039Z","shell.execute_reply.started":"2022-08-23T14:44:32.872262Z","shell.execute_reply":"2022-08-23T14:44:33.060178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 6))\n\nsns.countplot(df_train['center_id'])\nplt.title(\"Distribution of Ratings\", fontsize=20)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:44:33.062451Z","iopub.execute_input":"2022-08-23T14:44:33.063079Z","iopub.status.idle":"2022-08-23T14:44:33.313584Z","shell.execute_reply.started":"2022-08-23T14:44:33.063045Z","shell.execute_reply":"2022-08-23T14:44:33.312772Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 6))\n\nsns.countplot(df_train['image_num'])\nplt.title(\"Distribution of Ratings\", fontsize=20)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:44:33.317167Z","iopub.execute_input":"2022-08-23T14:44:33.317783Z","iopub.status.idle":"2022-08-23T14:44:33.537338Z","shell.execute_reply.started":"2022-08-23T14:44:33.317747Z","shell.execute_reply":"2022-08-23T14:44:33.536325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"director = '../input/mayo-clinic-strip-ai/train'\ncase = os.listdir(director)\ncase\nfile_name = []\nimage_matrix = []\n","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:44:33.540758Z","iopub.execute_input":"2022-08-23T14:44:33.541092Z","iopub.status.idle":"2022-08-23T14:44:33.780725Z","shell.execute_reply.started":"2022-08-23T14:44:33.541062Z","shell.execute_reply":"2022-08-23T14:44:33.779758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(case)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:44:33.781931Z","iopub.execute_input":"2022-08-23T14:44:33.782251Z","iopub.status.idle":"2022-08-23T14:44:33.789045Z","shell.execute_reply.started":"2022-08-23T14:44:33.782222Z","shell.execute_reply":"2022-08-23T14:44:33.787907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"k=0\nlabel = []\nfor i in case[0:20]:\n    im1 = []\n    directory_ = '../input/mayo-clinic-strip-ai/train/'\n    imagenes = os.listdir(directory_)\n    label.append(df_train['label'][k])\n    k = k+1\n    im1 = skio.imread('../input/mayo-clinic-strip-ai/train/'+ i)\n    res = cv2.resize(im1, dsize=(1700, 3040), interpolation=cv2.INTER_CUBIC)\n    cv2.imwrite(str(k)+'.jpg', res)\n    #image_matrix.append(res)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:44:33.790673Z","iopub.execute_input":"2022-08-23T14:44:33.791012Z","iopub.status.idle":"2022-08-23T14:50:07.064509Z","shell.execute_reply.started":"2022-08-23T14:44:33.790981Z","shell.execute_reply":"2022-08-23T14:50:07.063504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train['label'].index)\nlabels = []\nfor i in range(len(df_train['label'])):\n    labels.append(df_train['label'][i])\nprint(labels)\nlabels_code = []\nfor i in labels:\n    labels_code.append(labels.index(i))\nlabels_code\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-23T15:03:00.758525Z","iopub.execute_input":"2022-08-23T15:03:00.759376Z","iopub.status.idle":"2022-08-23T15:03:00.788432Z","shell.execute_reply.started":"2022-08-23T15:03:00.759335Z","shell.execute_reply":"2022-08-23T15:03:00.787305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport os\nfrom skimage.transform import resize\nfrom skimage.io import imread\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport imageio.v3 as iio\nimport imghdr\nCategories=['CE','LAA']\nflat_data_arr=[] #input array\ntarget_arr=[] #output array\ndatadir='./' \n#path which contains all the categories of images\npath=datadir\nfor img in os.listdir(path):\n    if imghdr.what(img)=='jpeg':\n        print(img)\n    #img_array=skio.imread(path + img)\n    #img_array=iio.imread(img)\n    #img_resized=resize(img_array,(150,150,3))\n    #flat_data_arr.append(img_resized.flatten())","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:50:07.065746Z","iopub.execute_input":"2022-08-23T14:50:07.066041Z","iopub.status.idle":"2022-08-23T14:50:07.389696Z","shell.execute_reply.started":"2022-08-23T14:50:07.066014Z","shell.execute_reply":"2022-08-23T14:50:07.387801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can reduce the size of the images with cv2\nk=0\nlabel = []\nfor i in case[0:200]:\n    im1 = []\n    directory_ = '../input/mayo-clinic-strip-ai/train/'\n    imagenes = os.listdir(directory_)\n    label.append(df_train['label'][k])\n    k = k+1\n    im1 = skio.imread('../input/mayo-clinic-strip-ai/train/'+ i)\n    res = cv2.resize(im1, dsize=(1700, 3040), interpolation=cv2.INTER_CUBIC)\n    cv2.imwrite(str(k)+'.jpg', res)\n    #image_matrix.append(res)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:50:07.391233Z","iopub.status.idle":"2022-08-23T14:50:07.392436Z","shell.execute_reply.started":"2022-08-23T14:50:07.392115Z","shell.execute_reply":"2022-08-23T14:50:07.392147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can reduce the size of the images with cv2\nlabel = []\nfor i in case[200:400]:\n    im1 = []\n    directory_ = '../input/mayo-clinic-strip-ai/train/'\n    imagenes = os.listdir(directory_)\n    label.append(df_train['label'][k])\n    k = k+1\n    im1 = skio.imread('../input/mayo-clinic-strip-ai/train/'+ i)\n    res = cv2.resize(im1, dsize=(1700, 3040), interpolation=cv2.INTER_CUBIC)\n    cv2.imwrite(str(k)+'.jpg', res)\n    #image_matrix.append(res)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:50:07.39447Z","iopub.status.idle":"2022-08-23T14:50:07.394945Z","shell.execute_reply.started":"2022-08-23T14:50:07.394737Z","shell.execute_reply":"2022-08-23T14:50:07.394758Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can reduce the size of the images with cv2\nlabel = []\nfor i in case[400:600]:\n    im1 = []\n    directory_ = '../input/mayo-clinic-strip-ai/train/'\n    imagenes = os.listdir(directory_)\n    label.append(df_train['label'][k])\n    k = k+1\n    im1 = skio.imread('../input/mayo-clinic-strip-ai/train/'+ i)\n    res = cv2.resize(im1, dsize=(1700, 3040), interpolation=cv2.INTER_CUBIC)\n    cv2.imwrite(str(k)+'.jpg', res)\n    #image_matrix.append(res)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:50:07.396972Z","iopub.status.idle":"2022-08-23T14:50:07.397421Z","shell.execute_reply.started":"2022-08-23T14:50:07.397232Z","shell.execute_reply":"2022-08-23T14:50:07.39725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can reduce the size of the images with cv2\nlabel = []\nfor i in case[600:754]:\n    im1 = []\n    directory_ = '../input/mayo-clinic-strip-ai/train/'\n    imagenes = os.listdir(directory_)\n    label.append(df_train['label'][k])\n    k = k+1\n    im1 = skio.imread('../input/mayo-clinic-strip-ai/train/'+ i)\n    res = cv2.resize(im1, dsize=(1700, 3040), interpolation=cv2.INTER_CUBIC)\n    cv2.imwrite(str(k)+'.jpeg', res)\n    #image_matrix.append(res)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:50:07.398901Z","iopub.status.idle":"2022-08-23T14:50:07.399491Z","shell.execute_reply.started":"2022-08-23T14:50:07.3993Z","shell.execute_reply":"2022-08-23T14:50:07.399319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport os\nfrom skimage.transform import resize\nfrom skimage.io import imread\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport imageio.v3 as iio\nCategories=['CE','LAA']\nflat_data_arr=[] #input array\ntarget_arr=[] #output array\ndatadir='./' \n#path which contains all the categories of images\npath=datadir\nfor img in os.listdir(path):\n    print(img)\n    if imghdr.what(img)=='jpeg':\n        print(img)\n        img_array=skio.imread(path + img)\n        img_array=iio.imread(img)\n        img_resized=resize(img_array,(150,150,3))\n        flat_data_arr.append(img_resized.flatten())#path which contains all the categories \nflat_data=np.array(flat_data_arr)\ntarget=np.array(labels_code)\ndf=pd.DataFrame(flat_data) #dataframe\ndf.head()\ndf['Target']=target\n        #x=df.iloc[:,:-1] #input data \n       # y=df.iloc[:,-1] #output data\nx=df.iloc[:,:-1] #input data \ny=df.iloc[:,-1] #output data","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:50:07.40097Z","iopub.status.idle":"2022-08-23T14:50:07.401371Z","shell.execute_reply.started":"2022-08-23T14:50:07.401184Z","shell.execute_reply":"2022-08-23T14:50:07.401202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.imshow(res,interpolation='nearest', aspect='auto')","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:50:07.402311Z","iopub.status.idle":"2022-08-23T14:50:07.402699Z","shell.execute_reply.started":"2022-08-23T14:50:07.402488Z","shell.execute_reply":"2022-08-23T14:50:07.402504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import svm\nfrom sklearn.model_selection import GridSearchCV\nparam_grid={'C':[0.1,1,10,100],'gamma':[0.0001,0.001,0.1,1],'kernel':['rbf','poly']}\nsvc=svm.SVC(probability=True)\nmodel=GridSearchCV(svc,param_grid)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train,x_test,y_train,y_test=train_test_split(x,y,test_size=0.20,random_state=77,stratify=y)\nprint('Splitted Successfully')\nmodel.fit(x_train,y_train)\nprint('The Model is trained well with the given images')\n# model.best_params_ contains the best parameters obtained from GridSearchCV","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\ny_pred=model.predict(x_test)\nprint(\"The predicted Data is :\")\nprint(y_pred)\nprint(\"The actual data is:\")\nprint(np.array(y_test))\nprint(f\"The model is {accuracy_score(y_pred,y_test)*100}% accurate\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dill\ndill.dump_session('notebook_env.db')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#len(label)\n#len(imagenes)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:50:07.405453Z","iopub.status.idle":"2022-08-23T14:50:07.406305Z","shell.execute_reply.started":"2022-08-23T14:50:07.406014Z","shell.execute_reply":"2022-08-23T14:50:07.406041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_other = df_train.copy()\ndf_other = pd.read_csv('../input/mayo-clinic-strip-ai/other.csv')\ndf_other.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:50:07.408026Z","iopub.status.idle":"2022-08-23T14:50:07.408556Z","shell.execute_reply.started":"2022-08-23T14:50:07.408283Z","shell.execute_reply":"2022-08-23T14:50:07.408308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 6))\n\nsns.countplot(df_other['image_num'])\nplt.title(\"Distribution of Ratings\", fontsize=20)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:50:07.410697Z","iopub.status.idle":"2022-08-23T14:50:07.411059Z","shell.execute_reply.started":"2022-08-23T14:50:07.410881Z","shell.execute_reply":"2022-08-23T14:50:07.410898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 6))\n\nsns.countplot(df_other['label'])\nplt.title(\"Distribution of Ratings\", fontsize=20)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:50:07.411863Z","iopub.status.idle":"2022-08-23T14:50:07.41225Z","shell.execute_reply.started":"2022-08-23T14:50:07.412054Z","shell.execute_reply":"2022-08-23T14:50:07.412072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 6))\n\nsns.countplot(df_other['other_specified'])\nplt.title(\"Distribution of Ratings\", fontsize=20)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-23T14:50:07.413839Z","iopub.status.idle":"2022-08-23T14:50:07.414204Z","shell.execute_reply.started":"2022-08-23T14:50:07.414016Z","shell.execute_reply":"2022-08-23T14:50:07.414032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ***Working in progress :D***","metadata":{}}]}