{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport os\nimport PIL\nfrom tqdm.notebook import tqdm\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-09T13:19:47.353843Z","iopub.execute_input":"2023-02-09T13:19:47.354356Z","iopub.status.idle":"2023-02-09T13:19:48.417167Z","shell.execute_reply.started":"2023-02-09T13:19:47.354233Z","shell.execute_reply":"2023-02-09T13:19:48.415925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analizing training set","metadata":{}},{"cell_type":"code","source":"csvpathtrain = '/kaggle/input/rsna-breast-cancer-detection/train.csv'\n\ndftrain = pd.read_csv(csvpathtrain)\ndftrain.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:19:55.093789Z","iopub.execute_input":"2023-02-09T13:19:55.094753Z","iopub.status.idle":"2023-02-09T13:19:55.263224Z","shell.execute_reply.started":"2023-02-09T13:19:55.094684Z","shell.execute_reply":"2023-02-09T13:19:55.261888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dftrain.tail()","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:19:57.175482Z","iopub.execute_input":"2023-02-09T13:19:57.175975Z","iopub.status.idle":"2023-02-09T13:19:57.199203Z","shell.execute_reply.started":"2023-02-09T13:19:57.175932Z","shell.execute_reply":"2023-02-09T13:19:57.197441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Seaborn setup\nsns.set_theme(style=\"whitegrid\")\nsns.set(rc={'figure.figsize':(15,5)})\n","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:19:58.487687Z","iopub.execute_input":"2023-02-09T13:19:58.488133Z","iopub.status.idle":"2023-02-09T13:19:58.495119Z","shell.execute_reply.started":"2023-02-09T13:19:58.4881Z","shell.execute_reply":"2023-02-09T13:19:58.49373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ages = dftrain.groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\nages_cancer = dftrain[dftrain['cancer'] == 1].groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\nages_no_cancer = dftrain[dftrain['cancer'] == 0].groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\n\nfig, axes = plt.subplots(3, 1, figsize=(15, 15))\nxticks = range(int(dftrain['age'].min()), int(dftrain['age'].max()), 10)\n\nsplot = sns.histplot(ax = axes[0], x = ages, bins = ages.nunique(), kde = True, palette = \"crest\")\naxes[0].xaxis.labelpad = -5\nsplot.set_title(\"Total distribution\", fontsize = 15, y = 0.98)\nsplot.set_xticks(xticks)\n\nsplot = sns.histplot(ax = axes[1], x = ages_cancer, bins = ages_cancer.nunique(), kde = True)\naxes[1].xaxis.labelpad = -5\nsplot.set_title(\"Cancer age distribution\", fontsize = 15, y = 0.98)\nsplot.set_xticks(xticks) \n\nsplot = sns.histplot(ax = axes[2], x = ages_no_cancer, bins = ages_no_cancer.nunique(), kde = True)\naxes[2].xaxis.labelpad = -5\nsplot.set_title(\"No cancer age distribution\", fontsize = 15, y = 0.98)\nsplot.set_xticks(xticks) \n\nplt.show()\n# for some reason, in set_xticks, 0 is referenced as starting tick (in this case 26)","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:19:59.60961Z","iopub.execute_input":"2023-02-09T13:19:59.610969Z","iopub.status.idle":"2023-02-09T13:20:02.07223Z","shell.execute_reply.started":"2023-02-09T13:19:59.610911Z","shell.execute_reply":"2023-02-09T13:20:02.071038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Age distribution is normal (Gaussian) distribution, which is exepected.\n\nAge distribution is also much different for examples with and without cancer. There is more older people with cancer, out of which most of them are older than 40.","metadata":{}},{"cell_type":"code","source":"\n# plotting view\nsplot = sns.countplot(x = dftrain['view'])\nsplot.set_title(\"View distribution\", fontsize = 10)","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:01:09.29022Z","iopub.execute_input":"2023-02-09T13:01:09.291013Z","iopub.status.idle":"2023-02-09T13:01:09.586457Z","shell.execute_reply.started":"2023-02-09T13:01:09.290972Z","shell.execute_reply":"2023-02-09T13:01:09.585376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It would be interesting to see if there are any view class examples that are not CC and MLO.","metadata":{}},{"cell_type":"code","source":"dftrain['view'][(dftrain['view'] != \"CC\") & (dftrain['view'] != \"MLO\")]","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:01:11.137332Z","iopub.execute_input":"2023-02-09T13:01:11.137963Z","iopub.status.idle":"2023-02-09T13:01:11.154802Z","shell.execute_reply.started":"2023-02-09T13:01:11.137928Z","shell.execute_reply":"2023-02-09T13:01:11.154013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since there are couple of examples, and from previous histogram can't be seen how many, it will be shown on seperate histogram.","metadata":{}},{"cell_type":"code","source":"splot = sns.countplot(x = dftrain['view'][(dftrain['view'] != \"CC\") & (dftrain['view'] != \"MLO\")])\nsns.set(rc={'figure.figsize':(9,5)})\nsplot.set_title(\"View distribution\", fontsize = 20)\nsplot.set_yticks(range(0, 20, 2)) ","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:01:13.073472Z","iopub.execute_input":"2023-02-09T13:01:13.074268Z","iopub.status.idle":"2023-02-09T13:01:13.320469Z","shell.execute_reply.started":"2023-02-09T13:01:13.074223Z","shell.execute_reply":"2023-02-09T13:01:13.319558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Very big imbalance here! Just one example of LMO, and ~25000 of CC and MLO. Alsto ML, LM and AT are very underrepresented.","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 4, figsize=(20, 10))\n\n########## PLOTING CANCER ################\nsplot = sns.countplot(ax = axes[0, 0], x = dftrain['cancer'])\nsplot.set_title(\"Cancer distribution\", fontsize = 15)\n\ns = dftrain['cancer'].value_counts()\naxes[1, 0].pie(s, autopct=\"%.1f%%\", labels = s.keys())\n\n########## PLOTING BIOPSY ################\nsplot = sns.countplot(ax = axes[0, 1], x = dftrain['biopsy'])\nsplot.set_title(\"Biopsy distribution\", fontsize = 15)\n\ns = dftrain['biopsy'].value_counts()\naxes[1, 1].pie(s, autopct=\"%.1f%%\", labels = s.keys())\n\n########## PLOTING INVASIVE ################\nsplot = sns.countplot(ax = axes[0, 2], x = dftrain['invasive'])\nsplot.set_title(\"Invasive distribution\", fontsize = 15)\n\ns = dftrain['invasive'].value_counts()\naxes[1, 2].pie(s, autopct=\"%.1f%%\", labels = s.keys())\n\n########## PLOTING IMPLANT ################\nsplot = sns.countplot(ax = axes[0, 3], x = dftrain['implant'])\nsplot.set_title(\"Implant distribution\", fontsize = 15)\n\ns = dftrain['implant'].value_counts()\naxes[1, 3].pie(s, autopct=\"%.1f%%\", labels = s.keys())\n","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:01:14.647738Z","iopub.execute_input":"2023-02-09T13:01:14.648164Z","iopub.status.idle":"2023-02-09T13:01:15.451089Z","shell.execute_reply.started":"2023-02-09T13:01:14.648128Z","shell.execute_reply":"2023-02-09T13:01:15.449976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analyze this!","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 2, figsize=(20, 10))\n\n########## PLOTING BIRADS ################\nsplot = sns.countplot(ax = axes[0, 0], x = dftrain['BIRADS'].dropna())\nsplot.set_title(\"BIRADS distribution\", fontsize = 15)\n\ns = dftrain['BIRADS'].dropna().value_counts()\naxes[1, 0].pie(s, autopct=\"%.1f%%\", labels = s.keys())\n\n########## PLOTING DENSITY ################\nsplot = sns.countplot(ax = axes[0, 1], x = dftrain['density'].dropna())\nsplot.set_title(\"Density distribution\", fontsize = 15)\n\ns = dftrain['density'].dropna().value_counts()\naxes[1, 1].pie(s, autopct=\"%.1f%%\", labels = s.keys())","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:01:16.695643Z","iopub.execute_input":"2023-02-09T13:01:16.696773Z","iopub.status.idle":"2023-02-09T13:01:17.231911Z","shell.execute_reply.started":"2023-02-09T13:01:16.696683Z","shell.execute_reply":"2023-02-09T13:01:17.230792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"splot = sns.countplot(x = dftrain['machine_id'])\nsns.set(rc={'figure.figsize':(9,5)})\nsplot.set_title(\"Machines distribution\", fontsize = 20)","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:01:17.593484Z","iopub.execute_input":"2023-02-09T13:01:17.594336Z","iopub.status.idle":"2023-02-09T13:01:17.95705Z","shell.execute_reply.started":"2023-02-09T13:01:17.594281Z","shell.execute_reply":"2023-02-09T13:01:17.955793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = dftrain['difficult_negative_case'].dropna().value_counts()\nplt.pie(s, autopct=\"%.1f%%\", labels = s.keys())\nplt.title(\"Difficult negative case distribution\", fontsize=20);\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:01:19.194888Z","iopub.execute_input":"2023-02-09T13:01:19.195272Z","iopub.status.idle":"2023-02-09T13:01:19.295523Z","shell.execute_reply.started":"2023-02-09T13:01:19.19524Z","shell.execute_reply":"2023-02-09T13:01:19.294679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dftrain.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:01:22.391159Z","iopub.execute_input":"2023-02-09T13:01:22.391556Z","iopub.status.idle":"2023-02-09T13:01:22.41137Z","shell.execute_reply.started":"2023-02-09T13:01:22.391524Z","shell.execute_reply":"2023-02-09T13:01:22.410013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For age, 37/54705 data are missing, so it's not much of a problem, But for atrubutes BIRADS and density, missing values are approximatly 1/2 of a dataset, so some predictor should be used to fill these missing values. For density prediction, all listed atributes should be used (except image id) + image for data sample.","metadata":{}},{"cell_type":"markdown","source":"density - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult. Only provided for train.\n<br/>\nBIRADS atribute is 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and 2 if the breast was rated as normal. Only provided for train.","metadata":{}},{"cell_type":"code","source":"dftrain['patient_id'].nunique","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:01:33.961034Z","iopub.execute_input":"2023-02-09T13:01:33.962402Z","iopub.status.idle":"2023-02-09T13:01:33.97395Z","shell.execute_reply.started":"2023-02-09T13:01:33.96234Z","shell.execute_reply":"2023-02-09T13:01:33.972099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are totally of 10006 unique patients, and for them there is totally 54701 data.","metadata":{}},{"cell_type":"markdown","source":"# Test data analysis","metadata":{}},{"cell_type":"code","source":"csvpathtest = '/kaggle/input/rsna-breast-cancer-detection/test.csv'\n\ndftest = pd.read_csv(csvpathtest)\ndftest.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:01:37.047446Z","iopub.execute_input":"2023-02-09T13:01:37.047879Z","iopub.status.idle":"2023-02-09T13:01:37.069791Z","shell.execute_reply.started":"2023-02-09T13:01:37.047842Z","shell.execute_reply":"2023-02-09T13:01:37.068784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is example of test data. The image will be provided, as well as the following attributes:\n* laterality\n* view\n* age\n* implant\n* machine_id\n<br/>\nprediction_id just is just patient_id and laterality merged, so I don't see any use of this atribute.\n<br/>\nAtributes that just provided for training data are:\n* density\n* cancer (THIS IS TARGET)\n* biopsy\n* invasive\n* BIRADS\n* difficult_negative_case\n","metadata":{}},{"cell_type":"markdown","source":"First idea is to train clasificators for each atribute that is provided just for training data, to predict it from image. \nAfter that, create clasificator that predicts out of all this atributes, weather person has a cancer or not.","metadata":{}},{"cell_type":"markdown","source":"# Image analysis","metadata":{}},{"cell_type":"code","source":"# load first 1000 images \ndirname = '/kaggle/input/rsna-breast-cancer-256-pngs'\nimgs = []\nimg_ids = []\npatient_ids = []\n\ni = 0\nfor dirname, _, filenames in os.walk(dirname):\n    for filename in tqdm(filenames):\n        if filename.endswith(\".png\"):\n            # new resize because of memory issue when converting from uint8 to float\n            img = PIL.Image.open(dirname + \"/\" + filename)\n            \n            #img = np.array(img, dtype = 'float16') / 255.0 # resacling and converting to float\n            img = np.array(img) # resacling and converting to float\n\n            imgs.append(img)\n            \n            img_id = filename.split(sep = \".\")[0].split(sep = \"_\")[1]\n            img_ids.append(int(img_id))\n            patient_id = filename.split(sep=\"_\")[0]\n            patient_ids.append(patient_id)\n            \n        #if i > 1000:\n        #    break\n        #i += 1\n        \n#imgs = np.array(imgs)","metadata":{"execution":{"iopub.status.busy":"2023-02-09T13:55:08.407233Z","iopub.execute_input":"2023-02-09T13:55:08.407667Z","iopub.status.idle":"2023-02-09T13:57:08.570277Z","shell.execute_reply.started":"2023-02-09T13:55:08.407629Z","shell.execute_reply":"2023-02-09T13:57:08.568914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = np.array([dftrain.loc[dftrain['image_id'] == img_id]['implant'].values for img_id in img_ids])[:, 0]","metadata":{"execution":{"iopub.status.busy":"2023-02-09T14:05:09.017344Z","iopub.execute_input":"2023-02-09T14:05:09.017807Z","iopub.status.idle":"2023-02-09T14:05:40.500837Z","shell.execute_reply.started":"2023-02-09T14:05:09.017767Z","shell.execute_reply":"2023-02-09T14:05:40.499761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_implants_inds = np.squeeze(np.argwhere(labels == 1))\n\nf = plt.figure(figsize=(18,7))\nfor i in range(10):\n    implant1 = img_implants_inds[i]\n    imp_example = imgs[implant1]\n\n    plt.subplot(2, 5, i + 1)\n    plt.imshow(imp_example, cmap = 'gray')\nf.suptitle(\"With implant\", fontsize = 20)","metadata":{"execution":{"iopub.status.busy":"2023-02-09T14:05:40.5026Z","iopub.execute_input":"2023-02-09T14:05:40.503345Z","iopub.status.idle":"2023-02-09T14:05:42.284212Z","shell.execute_reply.started":"2023-02-09T14:05:40.503303Z","shell.execute_reply":"2023-02-09T14:05:42.282992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = np.array([dftrain.loc[dftrain['image_id'] == img_id]['cancer'].values for img_id in img_ids])[:, 0]","metadata":{"execution":{"iopub.status.busy":"2023-02-09T14:05:42.286133Z","iopub.execute_input":"2023-02-09T14:05:42.286964Z","iopub.status.idle":"2023-02-09T14:06:14.094574Z","shell.execute_reply.started":"2023-02-09T14:05:42.286926Z","shell.execute_reply":"2023-02-09T14:06:14.093454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_cancer = np.squeeze(np.argwhere(labels == 1))\n\nf = plt.figure(figsize=(20,7))\nfor i in range(10):\n    implant1 = img_cancer[i]\n    imp_example = imgs[implant1]\n\n    plt.subplot(2, 5, i + 1)\n    plt.imshow(imp_example, cmap = 'gray')\nf.suptitle(\"With cancer\", fontsize = 20)","metadata":{"execution":{"iopub.status.busy":"2023-02-09T14:06:14.09711Z","iopub.execute_input":"2023-02-09T14:06:14.097466Z","iopub.status.idle":"2023-02-09T14:06:15.898305Z","shell.execute_reply.started":"2023-02-09T14:06:14.097434Z","shell.execute_reply":"2023-02-09T14:06:15.897276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_not_cancer = np.squeeze(np.argwhere(labels == 0))\n\nf = plt.figure(figsize=(20,7))\nfor i in range(10):\n    implant1 = img_not_cancer[i]\n    imp_example = imgs[implant1]\n\n    plt.subplot(2, 5, i + 1)\n    plt.imshow(imp_example, cmap = 'gray')\nf.suptitle(\"Without cancer\", fontsize = 20)","metadata":{"execution":{"iopub.status.busy":"2023-02-09T14:06:15.899655Z","iopub.execute_input":"2023-02-09T14:06:15.900758Z","iopub.status.idle":"2023-02-09T14:06:17.815905Z","shell.execute_reply.started":"2023-02-09T14:06:15.900684Z","shell.execute_reply":"2023-02-09T14:06:17.814507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Plot the pixel histograms for first 10 images","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 5, figsize=(20,7))\n\nfor i in range(5):\n    img_example = imgs[i]\n    plt.subplot(2, 5, i + 1)\n    plt.hist(img_example.ravel(), bins=256, range=(0, 255), fc='k', ec='k') \n    plt.subplot(2, 5, i + 6)\n    plt.imshow(img_example, cmap = 'gray')","metadata":{"execution":{"iopub.status.busy":"2023-02-09T14:06:34.6158Z","iopub.execute_input":"2023-02-09T14:06:34.616339Z","iopub.status.idle":"2023-02-09T14:06:39.06777Z","shell.execute_reply.started":"2023-02-09T14:06:34.61629Z","shell.execute_reply":"2023-02-09T14:06:39.066852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since the most pixels are black, and can't be seen the distribution of other pixels, those black pixels whose value is 0, will be left out","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 5, figsize=(20,7))\n\nfor i in range(5):\n    img_example = imgs[i]\n    plt.subplot(2, 5, i + 1)\n    plt.hist(img_example[img_example > 0.0].ravel(), bins=256, range=(0, 255), fc='k', ec='k') \n    plt.subplot(2, 5, i + 6)\n    plt.imshow(img_example, cmap = 'gray')","metadata":{"execution":{"iopub.status.busy":"2023-02-09T14:06:50.965428Z","iopub.execute_input":"2023-02-09T14:06:50.965874Z","iopub.status.idle":"2023-02-09T14:06:55.413877Z","shell.execute_reply.started":"2023-02-09T14:06:50.965835Z","shell.execute_reply":"2023-02-09T14:06:55.412702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Distributions vary very much, which can be from the fact that images are taken with different machines!<br>Now will be analyzed how machine id will infulence pixel distribution. <br> \nFrom previous analysis, it is seen that most of the images are taken with machines 28, 29, 48 and 49, so the pixel distribution for them will be analysed. ","metadata":{}},{"cell_type":"markdown","source":"#### Check for possible outliers or image with noise\n<br>\nIf image has less than 25 shades of gray, something is wrong.","metadata":{}},{"cell_type":"code","source":"for i, img in enumerate(imgs):\n    if len(np.unique(img.flatten())) < 25:\n        print(f\"Image with id {img_ids[i]} is suspicious\")\n        plt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2023-02-09T14:25:48.979333Z","iopub.execute_input":"2023-02-09T14:25:48.979832Z","iopub.status.idle":"2023-02-09T14:27:18.594403Z","shell.execute_reply.started":"2023-02-09T14:25:48.979788Z","shell.execute_reply":"2023-02-09T14:27:18.593477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Image with id 1942326353 is totally white!\n<br>\nAlso, images of patient 27770 looks suspicious","metadata":{}},{"cell_type":"code","source":"plt.subplots(2, 3, figsize = (15, 15))\nj = 0\nfor i, patient_id in enumerate(patient_ids):\n    if patient_id == '27770':\n        j += 1\n        print(f\"Image with id {img_ids[i]} is suspicious\")\n        img_example = imgs[i]\n        plt.subplot(2, 3, j)\n        plt.imshow(img_example, cmap = 'gray')\n        #plt.hist(img_example[img_example > 0].ravel(), bins=255, range=(0, 255), fc='k', ec='k') \n        \n    ","metadata":{"execution":{"iopub.status.busy":"2023-02-09T14:22:16.971883Z","iopub.execute_input":"2023-02-09T14:22:16.97277Z","iopub.status.idle":"2023-02-09T14:22:18.289404Z","shell.execute_reply.started":"2023-02-09T14:22:16.972672Z","shell.execute_reply":"2023-02-09T14:22:18.288096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Concusion: those images (and maybe more) should be removed from the dataset","metadata":{}},{"cell_type":"code","source":"labels = np.array([dftrain.loc[dftrain['image_id'] == img_id]['machine_id'].values for img_id in img_ids])[:, 0]\n\nunique_machines = np.unique(labels)\nprint(f\"There is {len(unique_machines)} different mahine ids.\")","metadata":{"execution":{"iopub.status.busy":"2023-01-26T07:16:31.817951Z","iopub.execute_input":"2023-01-26T07:16:31.818735Z","iopub.status.idle":"2023-01-26T07:16:32.389052Z","shell.execute_reply.started":"2023-01-26T07:16:31.818693Z","shell.execute_reply":"2023-01-26T07:16:32.387768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(4, 5, figsize=(20,14))\n\nfor i in range(len(unique_machines)):\n    img_for_machine = imgs[np.squeeze(np.argwhere(labels == unique_machines[i])[0])] # taking first image for machine\n    plt.subplot(4, 5, i + 1 if i < 5 else i + 6)\n    plt.hist(img_for_machine[img_for_machine > 0.0].ravel(), bins=256, range=(0.0, 1.0), fc='k', ec='k') \n    plt.title(f\"Machine {unique_machines[i]}\")\n    plt.subplot(4, 5, i + 6 if i < 5 else i + 11)\n    plt.imshow(img_for_machine, cmap = 'gray')\n    plt.title(f\"Machine {unique_machines[i]}\")\n\n","metadata":{"execution":{"iopub.status.busy":"2023-01-26T07:16:32.390701Z","iopub.execute_input":"2023-01-26T07:16:32.391605Z","iopub.status.idle":"2023-01-26T07:16:41.046867Z","shell.execute_reply.started":"2023-01-26T07:16:32.391562Z","shell.execute_reply":"2023-01-26T07:16:41.045464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nThese images that are very bright when you exclude background (machines 93, 190, 197 and 216) seems that doesn't capture tissue density very well, and that could be problematic for cancer detection. <br>\nOn the other hand, other machines shots looks good. The general conclusion is that the more histogram is shifted to left, the better (because white pixels will represent regions of interest, and they will be more easily noticed since there will be more dark pixels).\n<br>\nLuckily, most images are taken with first 4 machines, which have fairly good quality (machine 29 is a bit problematic).\n<br>\nAlso, normalisation can help to shift the histograms.\n","metadata":{}},{"cell_type":"markdown","source":"#### Check for potential irregular images","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = dftrain.corr(method='kendall')","metadata":{"execution":{"iopub.status.busy":"2023-01-26T07:22:13.200045Z","iopub.execute_input":"2023-01-26T07:22:13.201538Z","iopub.status.idle":"2023-01-26T07:22:13.601611Z","shell.execute_reply.started":"2023-01-26T07:22:13.201421Z","shell.execute_reply":"2023-01-26T07:22:13.600427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (15, 10))\nsns.heatmap(corr, cmap = 'Blues', annot = True)","metadata":{"execution":{"iopub.status.busy":"2023-01-26T07:23:51.015735Z","iopub.execute_input":"2023-01-26T07:23:51.016173Z","iopub.status.idle":"2023-01-26T07:23:52.028667Z","shell.execute_reply.started":"2023-01-26T07:23:51.016125Z","shell.execute_reply":"2023-01-26T07:23:52.027562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"abs(corr['cancer']) > 0.01","metadata":{"execution":{"iopub.status.busy":"2023-01-26T07:26:52.616042Z","iopub.execute_input":"2023-01-26T07:26:52.616489Z","iopub.status.idle":"2023-01-26T07:26:52.626538Z","shell.execute_reply.started":"2023-01-26T07:26:52.616453Z","shell.execute_reply":"2023-01-26T07:26:52.625244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From this, we can se that there is no correlation between having cancer and site id, patient id, image id and machine id. <br>\nThe other, personal features, are more correlated with cancer. Biopsy and invasive are strongly correlated, which makes sense from descriptions of these features. Age has a positive correlation, which also makes sense, since the older persons have higher probability of having a cancer. On the other hand, implant has small negative correlation, which would mean that having an implant kind of prevents cancer better than not having an implant. That absolutly makes no sense to me, and this is probably some kind of coincidence.\n\nSo, to sumarize, from features given in test.csv there is only sense to use age information, and maybe view and laterality, which are categorical variables and as such are not represented in correlation matrix.","metadata":{}}]}