{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Introduction/Context","metadata":{}},{"cell_type":"code","source":"#imports\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:26.240048Z","iopub.execute_input":"2023-01-28T12:12:26.240484Z","iopub.status.idle":"2023-01-28T12:12:26.245539Z","shell.execute_reply.started":"2023-01-28T12:12:26.240448Z","shell.execute_reply":"2023-01-28T12:12:26.2446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:26.713764Z","iopub.execute_input":"2023-01-28T12:12:26.714661Z","iopub.status.idle":"2023-01-28T12:12:26.822262Z","shell.execute_reply.started":"2023-01-28T12:12:26.714601Z","shell.execute_reply":"2023-01-28T12:12:26.821238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['patient_id'] == 3313]","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:27.245335Z","iopub.execute_input":"2023-01-28T12:12:27.245677Z","iopub.status.idle":"2023-01-28T12:12:27.277184Z","shell.execute_reply.started":"2023-01-28T12:12:27.245647Z","shell.execute_reply":"2023-01-28T12:12:27.276146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-01-28T10:57:19.76068Z","iopub.execute_input":"2023-01-28T10:57:19.76112Z","iopub.status.idle":"2023-01-28T10:57:19.771093Z","shell.execute_reply.started":"2023-01-28T10:57:19.761075Z","shell.execute_reply":"2023-01-28T10:57:19.769659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore the Data","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-01-21T10:06:22.66616Z","iopub.execute_input":"2023-01-21T10:06:22.666523Z","iopub.status.idle":"2023-01-21T10:06:22.852917Z","shell.execute_reply.started":"2023-01-21T10:06:22.666495Z","shell.execute_reply":"2023-01-21T10:06:22.851842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Singular Analysis","metadata":{}},{"cell_type":"markdown","source":"### Site ID","metadata":{}},{"cell_type":"markdown","source":"#### site_id - ID code for the source hospital\n\nQuestions to answer through exploring the data:\n* How many sites are there in the training data?\n* Should we discount any sites due to biases?","metadata":{}},{"cell_type":"code","source":"df['site_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:24.4975Z","iopub.execute_input":"2023-01-14T12:52:24.498114Z","iopub.status.idle":"2023-01-14T12:52:24.511228Z","shell.execute_reply.started":"2023-01-14T12:52:24.498084Z","shell.execute_reply":"2023-01-14T12:52:24.510504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='site_id', data=df)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:26.134081Z","iopub.execute_input":"2023-01-14T12:52:26.134406Z","iopub.status.idle":"2023-01-14T12:52:26.283442Z","shell.execute_reply.started":"2023-01-14T12:52:26.134383Z","shell.execute_reply":"2023-01-14T12:52:26.282571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('NaN values: ' + str(len(df)-sum(df['site_id'].value_counts())))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:27.872887Z","iopub.execute_input":"2023-01-14T12:52:27.873292Z","iopub.status.idle":"2023-01-14T12:52:27.880646Z","shell.execute_reply.started":"2023-01-14T12:52:27.873259Z","shell.execute_reply":"2023-01-14T12:52:27.879897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Laterality","metadata":{}},{"cell_type":"markdown","source":"#### laterality - Whether the image is of the left or right breast\n\nQuestions to answer through exploring the data:\n* Is there an inbalance of the laterailities?\n* Does breast cancer exist more on either side, and could potentially get used as an input variable?","metadata":{}},{"cell_type":"code","source":"df['laterality'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:29.493196Z","iopub.execute_input":"2023-01-14T12:52:29.493558Z","iopub.status.idle":"2023-01-14T12:52:29.502636Z","shell.execute_reply.started":"2023-01-14T12:52:29.493532Z","shell.execute_reply":"2023-01-14T12:52:29.501847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='laterality', data=df)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:30.384512Z","iopub.execute_input":"2023-01-14T12:52:30.384875Z","iopub.status.idle":"2023-01-14T12:52:30.519636Z","shell.execute_reply.started":"2023-01-14T12:52:30.384849Z","shell.execute_reply":"2023-01-14T12:52:30.518799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('NaN values: ' + str(len(df)-sum(df['laterality'].value_counts())))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:31.142683Z","iopub.execute_input":"2023-01-14T12:52:31.143704Z","iopub.status.idle":"2023-01-14T12:52:31.151968Z","shell.execute_reply.started":"2023-01-14T12:52:31.14367Z","shell.execute_reply":"2023-01-14T12:52:31.150892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### View","metadata":{}},{"cell_type":"markdown","source":"#### view - The orientation of the image. The default for a screening exam is to capture two views per breast\n\nQuestions to answer through exploring the data:\n* Is there an inbalance of the views?\n* Is cancer more detectable in certain views? How should we treat that in modelling our data?","metadata":{}},{"cell_type":"code","source":"df['view'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:32.509432Z","iopub.execute_input":"2023-01-14T12:52:32.509809Z","iopub.status.idle":"2023-01-14T12:52:32.519727Z","shell.execute_reply.started":"2023-01-14T12:52:32.509782Z","shell.execute_reply":"2023-01-14T12:52:32.518777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='view', data=df)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:33.127532Z","iopub.execute_input":"2023-01-14T12:52:33.127849Z","iopub.status.idle":"2023-01-14T12:52:33.282891Z","shell.execute_reply.started":"2023-01-14T12:52:33.127825Z","shell.execute_reply":"2023-01-14T12:52:33.282025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('NaN values: ' + str(len(df)-sum(df['view'].value_counts())))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:33.8133Z","iopub.execute_input":"2023-01-14T12:52:33.813797Z","iopub.status.idle":"2023-01-14T12:52:33.821434Z","shell.execute_reply.started":"2023-01-14T12:52:33.81377Z","shell.execute_reply":"2023-01-14T12:52:33.820184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Age","metadata":{}},{"cell_type":"markdown","source":"#### age - The patient's age in years\n\nQuestions to answer through exploring the data:\n* Is there an inbalance of ages?\n* How does cancer increase as ages increase? Should we include age as an input variable for the model to train off?","metadata":{}},{"cell_type":"code","source":"df['age'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:35.279619Z","iopub.execute_input":"2023-01-14T12:52:35.279975Z","iopub.status.idle":"2023-01-14T12:52:35.290646Z","shell.execute_reply.started":"2023-01-14T12:52:35.279947Z","shell.execute_reply":"2023-01-14T12:52:35.289471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='age', data=df)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:35.883929Z","iopub.execute_input":"2023-01-14T12:52:35.884243Z","iopub.status.idle":"2023-01-14T12:52:36.529728Z","shell.execute_reply.started":"2023-01-14T12:52:35.884219Z","shell.execute_reply":"2023-01-14T12:52:36.528907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(x='age', data=df, bins=10)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:36.531101Z","iopub.execute_input":"2023-01-14T12:52:36.531356Z","iopub.status.idle":"2023-01-14T12:52:36.713144Z","shell.execute_reply.started":"2023-01-14T12:52:36.531332Z","shell.execute_reply":"2023-01-14T12:52:36.712241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('NaN values: ' + str(len(df)-sum(df['age'].value_counts())))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:36.829081Z","iopub.execute_input":"2023-01-14T12:52:36.829947Z","iopub.status.idle":"2023-01-14T12:52:36.836582Z","shell.execute_reply.started":"2023-01-14T12:52:36.8299Z","shell.execute_reply":"2023-01-14T12:52:36.835724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cancer","metadata":{}},{"cell_type":"markdown","source":"#### cancer - Whether or not the breast was positive for malignant cancer. The target value. Only provided for train\n\nQuestions to answer through exploring the data:\n* How much is the data inbalanced under this variable?\n* How do we improve the model by measuring the correct metrics, and prevent a model that simply predicts the under-represented outcome?","metadata":{}},{"cell_type":"code","source":"df['cancer'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:38.159977Z","iopub.execute_input":"2023-01-14T12:52:38.160339Z","iopub.status.idle":"2023-01-14T12:52:38.168065Z","shell.execute_reply.started":"2023-01-14T12:52:38.160311Z","shell.execute_reply":"2023-01-14T12:52:38.16735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='cancer', data=df)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:38.706857Z","iopub.execute_input":"2023-01-14T12:52:38.707624Z","iopub.status.idle":"2023-01-14T12:52:38.814132Z","shell.execute_reply.started":"2023-01-14T12:52:38.707597Z","shell.execute_reply":"2023-01-14T12:52:38.813098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('NaN values: ' + str(len(df)-sum(df['cancer'].value_counts())))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:39.328628Z","iopub.execute_input":"2023-01-14T12:52:39.328971Z","iopub.status.idle":"2023-01-14T12:52:39.335757Z","shell.execute_reply.started":"2023-01-14T12:52:39.328944Z","shell.execute_reply":"2023-01-14T12:52:39.334705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Biopsy","metadata":{}},{"cell_type":"markdown","source":"#### biopsy - Whether or not a follow-up biopsy was performed on the breast. Only provided for train\n\nQuestions to answer through exploring the data:\n* Is biopsy a useful variable to factor when considering the goal of this model?","metadata":{}},{"cell_type":"code","source":"df['biopsy'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:40.310161Z","iopub.execute_input":"2023-01-14T12:52:40.311396Z","iopub.status.idle":"2023-01-14T12:52:40.318998Z","shell.execute_reply.started":"2023-01-14T12:52:40.311334Z","shell.execute_reply":"2023-01-14T12:52:40.317986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='biopsy', data=df)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:41.213924Z","iopub.execute_input":"2023-01-14T12:52:41.215244Z","iopub.status.idle":"2023-01-14T12:52:41.324054Z","shell.execute_reply.started":"2023-01-14T12:52:41.215182Z","shell.execute_reply":"2023-01-14T12:52:41.323063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('NaN values: ' + str(len(df)-sum(df['biopsy'].value_counts())))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:41.934079Z","iopub.execute_input":"2023-01-14T12:52:41.934432Z","iopub.status.idle":"2023-01-14T12:52:41.940147Z","shell.execute_reply.started":"2023-01-14T12:52:41.934403Z","shell.execute_reply":"2023-01-14T12:52:41.939304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Invasive","metadata":{}},{"cell_type":"markdown","source":"#### invasive - If the breast is positive for cancer, whether or not the cancer proved to be invasive. Only provided for train\n\nQuestions to answer through exploring the data:\n* Is invasive a useful variable to factor when considering the goal of this model?","metadata":{}},{"cell_type":"code","source":"df['invasive'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:43.837534Z","iopub.execute_input":"2023-01-14T12:52:43.838031Z","iopub.status.idle":"2023-01-14T12:52:43.8451Z","shell.execute_reply.started":"2023-01-14T12:52:43.837998Z","shell.execute_reply":"2023-01-14T12:52:43.844491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='invasive', data=df)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:44.272414Z","iopub.execute_input":"2023-01-14T12:52:44.27276Z","iopub.status.idle":"2023-01-14T12:52:44.385254Z","shell.execute_reply.started":"2023-01-14T12:52:44.272735Z","shell.execute_reply":"2023-01-14T12:52:44.384501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('NaN values: ' + str(len(df)-sum(df['invasive'].value_counts())))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:44.879155Z","iopub.execute_input":"2023-01-14T12:52:44.879509Z","iopub.status.idle":"2023-01-14T12:52:44.886309Z","shell.execute_reply.started":"2023-01-14T12:52:44.879481Z","shell.execute_reply":"2023-01-14T12:52:44.885389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### BIRADS","metadata":{}},{"cell_type":"markdown","source":"#### BIRADS - 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and 2 if the breast was rated as normal. Only provided for train\n\nQuestions to answer through exploring the data:\n* Is BIRADS a useful variable to factor when considering the goal of this model?","metadata":{}},{"cell_type":"code","source":"df['BIRADS'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:45.862841Z","iopub.execute_input":"2023-01-14T12:52:45.863172Z","iopub.status.idle":"2023-01-14T12:52:45.870511Z","shell.execute_reply.started":"2023-01-14T12:52:45.863146Z","shell.execute_reply":"2023-01-14T12:52:45.869857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='BIRADS', data=df)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:46.245276Z","iopub.execute_input":"2023-01-14T12:52:46.245772Z","iopub.status.idle":"2023-01-14T12:52:46.38213Z","shell.execute_reply.started":"2023-01-14T12:52:46.245746Z","shell.execute_reply":"2023-01-14T12:52:46.38137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('NaN values: ' + str(len(df)-sum(df['BIRADS'].value_counts())))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:46.652551Z","iopub.execute_input":"2023-01-14T12:52:46.653076Z","iopub.status.idle":"2023-01-14T12:52:46.659228Z","shell.execute_reply.started":"2023-01-14T12:52:46.653039Z","shell.execute_reply":"2023-01-14T12:52:46.658159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Implant","metadata":{}},{"cell_type":"markdown","source":"#### implant - Whether or not the patient had breast implants. Site 1 only provides breast implant information at the patient level, not at the breast level\n\nQuestions to answer through exploring the data:\n* How inbalanced is the variable?\n* How does an implant impact the target variable, cancer and should it be used as an input variable to train our model?","metadata":{}},{"cell_type":"code","source":"df['implant'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:47.569172Z","iopub.execute_input":"2023-01-14T12:52:47.570177Z","iopub.status.idle":"2023-01-14T12:52:47.576947Z","shell.execute_reply.started":"2023-01-14T12:52:47.570129Z","shell.execute_reply":"2023-01-14T12:52:47.576331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='implant', data=df)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:47.95789Z","iopub.execute_input":"2023-01-14T12:52:47.958382Z","iopub.status.idle":"2023-01-14T12:52:48.065325Z","shell.execute_reply.started":"2023-01-14T12:52:47.958355Z","shell.execute_reply":"2023-01-14T12:52:48.064278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('NaN values: ' + str(len(df)-sum(df['implant'].value_counts())))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:48.360389Z","iopub.execute_input":"2023-01-14T12:52:48.360752Z","iopub.status.idle":"2023-01-14T12:52:48.367345Z","shell.execute_reply.started":"2023-01-14T12:52:48.360725Z","shell.execute_reply":"2023-01-14T12:52:48.36623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Density","metadata":{}},{"cell_type":"markdown","source":"#### density - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult. Only provided for train\n\nQuestions to answer through exploring the data:\n* How inbalanced is the variable?\n* How does an implant impact the target variable, cancer and should it be used as an input variable to train our model?\n* If it's harder to detect cancer in highly dense tissue, should we account for this bias somehow in training our model?","metadata":{}},{"cell_type":"code","source":"df['density'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:49.073746Z","iopub.execute_input":"2023-01-14T12:52:49.074114Z","iopub.status.idle":"2023-01-14T12:52:49.084263Z","shell.execute_reply.started":"2023-01-14T12:52:49.074084Z","shell.execute_reply":"2023-01-14T12:52:49.083153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='density', data=df)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:49.382013Z","iopub.execute_input":"2023-01-14T12:52:49.382363Z","iopub.status.idle":"2023-01-14T12:52:49.530103Z","shell.execute_reply.started":"2023-01-14T12:52:49.382335Z","shell.execute_reply":"2023-01-14T12:52:49.529378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('NaN values: ' + str(len(df)-sum(df['density'].value_counts())))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:49.807975Z","iopub.execute_input":"2023-01-14T12:52:49.808975Z","iopub.status.idle":"2023-01-14T12:52:49.816527Z","shell.execute_reply.started":"2023-01-14T12:52:49.808928Z","shell.execute_reply":"2023-01-14T12:52:49.815614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Machine ID","metadata":{}},{"cell_type":"markdown","source":"#### machine_id - An ID code for the imaging device\n\nQuestions to answer through exploring the data:\n* How many machines are there in the training data?\n* Should we discount any machines due to biases?","metadata":{}},{"cell_type":"code","source":"df['machine_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:50.585978Z","iopub.execute_input":"2023-01-14T12:52:50.586318Z","iopub.status.idle":"2023-01-14T12:52:50.59454Z","shell.execute_reply.started":"2023-01-14T12:52:50.586291Z","shell.execute_reply":"2023-01-14T12:52:50.593497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='machine_id', data=df)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:50.963177Z","iopub.execute_input":"2023-01-14T12:52:50.963546Z","iopub.status.idle":"2023-01-14T12:52:51.121274Z","shell.execute_reply.started":"2023-01-14T12:52:50.963518Z","shell.execute_reply":"2023-01-14T12:52:51.120459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('NaN values: ' + str(len(df)-sum(df['machine_id'].value_counts())))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:51.278205Z","iopub.execute_input":"2023-01-14T12:52:51.278606Z","iopub.status.idle":"2023-01-14T12:52:51.284559Z","shell.execute_reply.started":"2023-01-14T12:52:51.278574Z","shell.execute_reply":"2023-01-14T12:52:51.283731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Difficult Negative Case","metadata":{}},{"cell_type":"markdown","source":"#### difficult_negative_case - True if the case was unusually difficult. Only provided for train\n\nQuestions to answer through exploring the data:\n* Could a separate model be trained to detect difficult to detect negative cases to better remove false positives that might arise from the original model?","metadata":{}},{"cell_type":"code","source":"df['difficult_negative_case'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:52.050791Z","iopub.execute_input":"2023-01-14T12:52:52.051222Z","iopub.status.idle":"2023-01-14T12:52:52.05954Z","shell.execute_reply.started":"2023-01-14T12:52:52.051187Z","shell.execute_reply":"2023-01-14T12:52:52.05878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='difficult_negative_case', data=df)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:52.386868Z","iopub.execute_input":"2023-01-14T12:52:52.387647Z","iopub.status.idle":"2023-01-14T12:52:52.49649Z","shell.execute_reply.started":"2023-01-14T12:52:52.38761Z","shell.execute_reply":"2023-01-14T12:52:52.495411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('NaN values: ' + str(len(df)-sum(df['difficult_negative_case'].value_counts())))","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:52.748971Z","iopub.execute_input":"2023-01-14T12:52:52.750059Z","iopub.status.idle":"2023-01-14T12:52:52.757105Z","shell.execute_reply.started":"2023-01-14T12:52:52.750015Z","shell.execute_reply":"2023-01-14T12:52:52.755693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Double Comparisons","metadata":{}},{"cell_type":"markdown","source":"### Age vs Cancer","metadata":{}},{"cell_type":"code","source":"sns.lineplot(data=df,x='age',y='cancer')","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:53.802242Z","iopub.execute_input":"2023-01-14T12:52:53.802782Z","iopub.status.idle":"2023-01-14T12:52:55.539683Z","shell.execute_reply.started":"2023-01-14T12:52:53.802756Z","shell.execute_reply":"2023-01-14T12:52:55.538827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2 = df[['cancer', 'age']]","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:55.541241Z","iopub.execute_input":"2023-01-14T12:52:55.541534Z","iopub.status.idle":"2023-01-14T12:52:55.547003Z","shell.execute_reply.started":"2023-01-14T12:52:55.541509Z","shell.execute_reply":"2023-01-14T12:52:55.545872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins = [25, 35, 45, 55, 65, 75, 85, 95]\n\ndf2.groupby([pd.cut(df2.age, bins)]).mean()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:55.548149Z","iopub.execute_input":"2023-01-14T12:52:55.54842Z","iopub.status.idle":"2023-01-14T12:52:55.576171Z","shell.execute_reply.started":"2023-01-14T12:52:55.548397Z","shell.execute_reply":"2023-01-14T12:52:55.575169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Implant vs Cancer","metadata":{}},{"cell_type":"code","source":"sns.lineplot(data=df,x='implant',y='cancer')","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:55.811422Z","iopub.execute_input":"2023-01-14T12:52:55.811734Z","iopub.status.idle":"2023-01-14T12:52:56.587674Z","shell.execute_reply.started":"2023-01-14T12:52:55.811711Z","shell.execute_reply":"2023-01-14T12:52:56.586492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df3 = df[['cancer', 'implant']]","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:56.590427Z","iopub.execute_input":"2023-01-14T12:52:56.590711Z","iopub.status.idle":"2023-01-14T12:52:56.595913Z","shell.execute_reply.started":"2023-01-14T12:52:56.590688Z","shell.execute_reply":"2023-01-14T12:52:56.595009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df3.groupby('implant').mean()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T12:52:56.745934Z","iopub.execute_input":"2023-01-14T12:52:56.746958Z","iopub.status.idle":"2023-01-14T12:52:56.757559Z","shell.execute_reply.started":"2023-01-14T12:52:56.746924Z","shell.execute_reply":"2023-01-14T12:52:56.756442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Density vs Cancer","metadata":{}},{"cell_type":"code","source":"sns.lineplot(data=df.sort_values('density'),x='density',y='cancer', sort=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T13:01:10.866111Z","iopub.execute_input":"2023-01-14T13:01:10.866415Z","iopub.status.idle":"2023-01-14T13:01:11.424148Z","shell.execute_reply.started":"2023-01-14T13:01:10.866391Z","shell.execute_reply":"2023-01-14T13:01:11.423418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## View vs Cancer","metadata":{}},{"cell_type":"code","source":"sns.lineplot(data=df,x='view',y='cancer', sort=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T13:04:56.138193Z","iopub.execute_input":"2023-01-14T13:04:56.13855Z","iopub.status.idle":"2023-01-14T13:04:56.963741Z","shell.execute_reply.started":"2023-01-14T13:04:56.138524Z","shell.execute_reply":"2023-01-14T13:04:56.962855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Laterality vs Cancer","metadata":{}},{"cell_type":"code","source":"sns.lineplot(data=df,x='laterality',y='cancer', sort=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T13:07:11.646922Z","iopub.execute_input":"2023-01-14T13:07:11.647284Z","iopub.status.idle":"2023-01-14T13:07:12.413207Z","shell.execute_reply.started":"2023-01-14T13:07:11.647256Z","shell.execute_reply":"2023-01-14T13:07:12.412506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Machine ID vs Cancer","metadata":{}},{"cell_type":"code","source":"df4 = df[['cancer', 'machine_id']]","metadata":{"execution":{"iopub.status.busy":"2023-01-14T13:08:17.038887Z","iopub.execute_input":"2023-01-14T13:08:17.039214Z","iopub.status.idle":"2023-01-14T13:08:17.044346Z","shell.execute_reply.started":"2023-01-14T13:08:17.039189Z","shell.execute_reply":"2023-01-14T13:08:17.043601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df4.groupby('machine_id').mean()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T13:14:38.677829Z","iopub.execute_input":"2023-01-14T13:14:38.678159Z","iopub.status.idle":"2023-01-14T13:14:38.690313Z","shell.execute_reply.started":"2023-01-14T13:14:38.678133Z","shell.execute_reply":"2023-01-14T13:14:38.689487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Site ID vs Cancer","metadata":{}},{"cell_type":"code","source":"df5 = df[['cancer', 'site_id']]","metadata":{"execution":{"iopub.status.busy":"2023-01-14T13:14:04.382568Z","iopub.execute_input":"2023-01-14T13:14:04.383659Z","iopub.status.idle":"2023-01-14T13:14:04.389608Z","shell.execute_reply.started":"2023-01-14T13:14:04.383606Z","shell.execute_reply":"2023-01-14T13:14:04.388753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df5.groupby('site_id').mean()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T13:14:17.265143Z","iopub.execute_input":"2023-01-14T13:14:17.265501Z","iopub.status.idle":"2023-01-14T13:14:17.278225Z","shell.execute_reply.started":"2023-01-14T13:14:17.265474Z","shell.execute_reply":"2023-01-14T13:14:17.276928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Biopsy vs Cancer","metadata":{}},{"cell_type":"code","source":"df6 = df[['cancer', 'biopsy']]","metadata":{"execution":{"iopub.status.busy":"2023-01-14T16:16:49.251675Z","iopub.execute_input":"2023-01-14T16:16:49.252797Z","iopub.status.idle":"2023-01-14T16:16:49.263467Z","shell.execute_reply.started":"2023-01-14T16:16:49.252729Z","shell.execute_reply":"2023-01-14T16:16:49.262305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df6.groupby('cancer').sum()","metadata":{"execution":{"iopub.status.busy":"2023-01-14T16:17:50.422667Z","iopub.execute_input":"2023-01-14T16:17:50.423502Z","iopub.status.idle":"2023-01-14T16:17:50.442209Z","shell.execute_reply.started":"2023-01-14T16:17:50.423461Z","shell.execute_reply":"2023-01-14T16:17:50.440687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Multiple Comparisons","metadata":{}},{"cell_type":"markdown","source":"### Multi-collinearity","metadata":{}},{"cell_type":"code","source":"df.dtypes == object","metadata":{"execution":{"iopub.status.busy":"2023-01-14T15:24:09.079295Z","iopub.execute_input":"2023-01-14T15:24:09.079773Z","iopub.status.idle":"2023-01-14T15:24:09.090023Z","shell.execute_reply.started":"2023-01-14T15:24:09.079719Z","shell.execute_reply":"2023-01-14T15:24:09.088532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"laterality_ohe = pd.get_dummies(df.laterality, prefix='Laterality')\nview_ohe = pd.get_dummies(df.view, prefix='View')\ndensity_ohe = pd.get_dummies(df.density, prefix='Density')","metadata":{"execution":{"iopub.status.busy":"2023-01-14T15:22:03.67783Z","iopub.execute_input":"2023-01-14T15:22:03.678265Z","iopub.status.idle":"2023-01-14T15:22:03.705445Z","shell.execute_reply.started":"2023-01-14T15:22:03.678232Z","shell.execute_reply":"2023-01-14T15:22:03.704116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.concat([df.drop(['laterality','view','density'],axis=1),laterality_ohe,view_ohe,density_ohe],axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-01-14T15:27:37.267882Z","iopub.execute_input":"2023-01-14T15:27:37.268323Z","iopub.status.idle":"2023-01-14T15:27:37.282095Z","shell.execute_reply.started":"2023-01-14T15:27:37.268287Z","shell.execute_reply":"2023-01-14T15:27:37.280309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-01-14T15:27:39.262891Z","iopub.execute_input":"2023-01-14T15:27:39.263364Z","iopub.status.idle":"2023-01-14T15:27:39.298671Z","shell.execute_reply.started":"2023-01-14T15:27:39.263328Z","shell.execute_reply":"2023-01-14T15:27:39.297112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(50, 6))\nsns.heatmap(df.corr(), annot=True)\nplt.savefig(\"Heatmap.png\")","metadata":{"execution":{"iopub.status.busy":"2023-01-14T15:29:18.085602Z","iopub.execute_input":"2023-01-14T15:29:18.086054Z","iopub.status.idle":"2023-01-14T15:29:22.108323Z","shell.execute_reply.started":"2023-01-14T15:29:18.086014Z","shell.execute_reply":"2023-01-14T15:29:22.107078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Outcome","metadata":{}},{"cell_type":"markdown","source":"Notes on site_id:\n* We would prefer to have more than 2 sites to base our training data on. However, cancer should look the same across all train/test sites so we shouldn't have any problems generalizing.\n* There doesn't seem to be any problems of certain sites incorrectly labelling the training data that we should distrust the data provided with both sites showing similar cancer rates.\n* There doesn't seem to hold any predictive power in the sites e.g. breast cancer being more prevelant in one site due to the similar cancer rates. We don't want to use it as a predictor for a second reason being that the model's target is to generalize to other hospitals.","metadata":{}},{"cell_type":"markdown","source":"Notes on laterality:\n* The data is pretty evenly split between lateralities and it doesn't seem to hold any predictive power nor any imbalances that we need to account for.","metadata":{}},{"cell_type":"markdown","source":"Notes on view:\n* The view should not have any effect on the outcome won't be used in our inputs so we don't need to account for any biases even if there were some (which there aren't). MLO and CC views will be used mostly in the test data and these both show similar cancer rates.","metadata":{}},{"cell_type":"markdown","source":"Notes on age:\n* Age does seem to show a small predictive power. Cancer is still completely dependent on the image actually showing cancer **but the inclusion of age may give a better prediction of whether a certain pattern in the image should be predicted as cancer-positive.**\n* We should also be aware of the imbalance that this might cause. Old people have higher rates of cancer and the model may incorrectly learn to simply identify an 'old looking' brain. The imbalance is not massive in this case but it's something to keep an eye on for future experimenting.","metadata":{}},{"cell_type":"markdown","source":"Notes on cancer:\n* Cancer is pretty imbalanced between false and true (thankfully!). The model may incorrectly learn to simply predict false and get a high accuracy. W**e will have to ensure we are measuring the correct metrics as we're interested in true positives probably more so that true negatives. Using F2 will account for this and we'll use data augmentation to grow the synthetically grow the true set.**","metadata":{}},{"cell_type":"markdown","source":"Notes on biopsy:\n* Biopsy is not a useful variable to consider. The goal of this competition is to identify cases of breast cancer. What happens after is not hugely important. We might ask if the cancer prediction is accurate unless a follow up has been performed and reduce the training data to the 2000 cases where there was a biopsy. However, that will have more of a negative effect reducing the training data size than catching a few negative falses in the training data that weren't caught in the initial screening.","metadata":{}},{"cell_type":"markdown","source":"Notes on invasive:\n* Invasive is not a useful variable to consider. The goal of this competition is to identify cases of breast cancer, not invasiveness of the cancer.","metadata":{}},{"cell_type":"markdown","source":"Notes on BIRADS:\n* BIRADS is not a useful variable to consider once again as it doesn't matter for the goal of this model.","metadata":{}},{"cell_type":"markdown","source":"Notes on implants:\n* It has been shown the women with implants tend to have a lower rate of breast cancer. However, research has concluded that this is most likely due to younger women getting implants and so it's more correlation rather than cause and therefore, shouldn't be considered as a predictive variable.","metadata":{}},{"cell_type":"markdown","source":"Notes on density:\n* Density seems to be unbalanced with the high density and low density mammograms underrepresented. The high density mammograms are also harder to identify cancer in. **We will measure the correct cancer detections in each density category and balance the data using data augmentation if necessary.**","metadata":{}},{"cell_type":"markdown","source":"Notes on machine_id:\n* There are no 'red flags' with any machine indicating that we might need to discount any of the training data due to a faulty machine. Some machines have 0 or close to 0 cancer detection rates but the number of these machines in the training data is too low to make any assumptions about faultiness.","metadata":{}},{"cell_type":"markdown","source":"Notes on difficult_negative_case:\n* There are fairly low difficult_negative_case=True values in the training data causing an imbalance. The model may learn the easy cases but not the hard ones and **so we should measure the accuracy of correctly identifying the difficult ones during training rather than just the overall accuracy and ensure these tricky cases are getting the right prediction. If not, we may need to include data augmentation to balance the data.**","metadata":{}},{"cell_type":"markdown","source":"**Conclusion: In general, this project works like a classic CNN problem. We might gain some insights by considering variables such as age and implant, by balancing the data so true cancers are more represented and the training data is represented across the secondary variables and by measuring the correct metrics to ensure the model isn't being oversimplified.**","metadata":{}},{"cell_type":"markdown","source":"# Pre-processing","metadata":{}},{"cell_type":"markdown","source":"## Quick Check","metadata":{}},{"cell_type":"markdown","source":"First, we will do a quick check to make sure that the training csv matches the images available to us in the 'train_images' folder:","metadata":{}},{"cell_type":"code","source":"import glob\n\nimg_list = glob.glob('/kaggle/input/rsna-breast-cancer-detection/train_images/*/*')","metadata":{"execution":{"iopub.status.busy":"2023-01-15T11:59:01.001433Z","iopub.execute_input":"2023-01-15T11:59:01.001859Z","iopub.status.idle":"2023-01-15T11:59:31.377937Z","shell.execute_reply.started":"2023-01-15T11:59:01.001826Z","shell.execute_reply":"2023-01-15T11:59:31.376766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(img_list)","metadata":{"execution":{"iopub.status.busy":"2023-01-15T12:02:17.972627Z","iopub.execute_input":"2023-01-15T12:02:17.973059Z","iopub.status.idle":"2023-01-15T12:02:17.98347Z","shell.execute_reply.started":"2023-01-15T12:02:17.973021Z","shell.execute_reply":"2023-01-15T12:02:17.982187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_list = []\n\nfor x in df.index:\n    training_list.append('/kaggle/input/rsna-breast-cancer-detection/train_images/' + str(df['patient_id'][x]) + '/' + str(df['image_id'][x]) + '.dcm')\n\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-01-15T12:14:39.469059Z","iopub.execute_input":"2023-01-15T12:14:39.469496Z","iopub.status.idle":"2023-01-15T12:14:40.336075Z","shell.execute_reply.started":"2023-01-15T12:14:39.469463Z","shell.execute_reply":"2023-01-15T12:14:40.334856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(training_list)","metadata":{"execution":{"iopub.status.busy":"2023-01-15T12:15:08.231504Z","iopub.execute_input":"2023-01-15T12:15:08.231902Z","iopub.status.idle":"2023-01-15T12:15:08.242371Z","shell.execute_reply.started":"2023-01-15T12:15:08.231871Z","shell.execute_reply":"2023-01-15T12:15:08.240999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(training_list == img_list)","metadata":{"execution":{"iopub.status.busy":"2023-01-15T12:18:43.142395Z","iopub.execute_input":"2023-01-15T12:18:43.143644Z","iopub.status.idle":"2023-01-15T12:18:43.153677Z","shell.execute_reply.started":"2023-01-15T12:18:43.143595Z","shell.execute_reply":"2023-01-15T12:18:43.152481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We know we can trust the training csv to load the full and complete images available to us in the folder.","metadata":{}},{"cell_type":"markdown","source":"## Extra Training Data","metadata":{}},{"cell_type":"markdown","source":"We should provide as much targetting help as possible to our model. We should group our inputs by patient's breasts (so distinguishing between L and R). This should make intuitive sense sinse just because one breast has cancer doesn't mean the other will. Let's do a simple check to see if there are any cases where the ccancer outcome contradicts within a single patient's breast.","metadata":{}},{"cell_type":"markdown","source":"*If this is the case, grouping by patient_id and laterality with mean as the processor will return none 0 or 1 means of cancer. E.g. if one patient had 4 L images and 3 returned True for cancer and 1 returned False then the mean would be 0.75. In the case below the only options in this grouped dataframe are 1's and 0's:*","metadata":{}},{"cell_type":"code","source":"df8=df.groupby(['patient_id','laterality']).mean()\nlen(df8) == len(df8[df8['cancer'] == 1]) + len(df8[df8['cancer'] == 0])","metadata":{"execution":{"iopub.status.busy":"2023-01-15T14:33:25.544805Z","iopub.execute_input":"2023-01-15T14:33:25.545224Z","iopub.status.idle":"2023-01-15T14:33:25.585422Z","shell.execute_reply.started":"2023-01-15T14:33:25.54519Z","shell.execute_reply":"2023-01-15T14:33:25.584079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The number of patients in the training data is: ' + str(len(df.groupby(['patient_id']).mean())))\nprint('The number of inputs in the training data is: ' + str(len(df.groupby(['patient_id','laterality']).mean())) +'. This is two times the number of patients so all patients have a L and R breast assessed.') ","metadata":{"execution":{"iopub.status.busy":"2023-01-15T14:17:47.200576Z","iopub.execute_input":"2023-01-15T14:17:47.201187Z","iopub.status.idle":"2023-01-15T14:17:47.244057Z","shell.execute_reply.started":"2023-01-15T14:17:47.201154Z","shell.execute_reply":"2023-01-15T14:17:47.242805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"24,000 is a somewhat suitable training data size. We will add to this using data augmentation. \n\nI found a semi-useful data set for potentially adding some further training data: https://www.kaggle.com/datasets/awsaf49/cbis-ddsm-breast-cancer-image-dataset\n\nRunning some queries on the data, I find it holds the following values:\n* Patients: 1,566\n* Inputs: 6775 (617 positive cases)\n\n**The extra training might be significant. I will train without it to begin with and assess the accuracy to avoid the added complexity of exporting from multiple sources and matching the metadata.**","metadata":{}},{"cell_type":"markdown","source":"## Import DCM into Tensors","metadata":{}},{"cell_type":"markdown","source":"The preprocessing of the images from the DCM files will include multiple steps:\n* Importing the DCM image into a computational array\n* Converting this array into Grayscale RGB\n* Using CLAHE to equalize the images. The images have a lot of un-needed space which creates contrast between the white background and the breast tissue. Using normal Histogram Equalisation won't improve visibility of the image as you can see in the appendix. We could use segmentation using the LIBRA package and then apply normal Histogram Equalisation, but an easier approach is to apply CLAHE which stops the over-amplification of the contrast.\n* Resizing the images\n* Standardisation of pixels (will happen at the loading stage)","metadata":{}},{"cell_type":"code","source":"import cv2\nimport pydicom as dicom\nfrom skimage.transform import resize ","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:33.633816Z","iopub.execute_input":"2023-01-28T12:12:33.634172Z","iopub.status.idle":"2023-01-28T12:12:34.386813Z","shell.execute_reply.started":"2023-01-28T12:12:33.634142Z","shell.execute_reply":"2023-01-28T12:12:34.385842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dcm2rgbgray(dcm):\n    rgbgray = (dcm.astype(np.float64)/np.max(dcm))*255\n    rgbgray = rgbgray.astype(np.uint8)\n    \n    return rgbgray\n\n\n# plt.imshow(img, cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:34.388682Z","iopub.execute_input":"2023-01-28T12:12:34.389043Z","iopub.status.idle":"2023-01-28T12:12:34.394594Z","shell.execute_reply.started":"2023-01-28T12:12:34.389011Z","shell.execute_reply":"2023-01-28T12:12:34.393568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import SimpleITK as sitk\nimport numpy as np\n\nclass Mammogram_Processed_Image:\n  def __init__(self, path):\n    self.path = path\n    \n    try:\n        image_read = dicom.dcmread(self.path)\n        image_read.PhotometricInterpretation = 'YBR RCT'\n        data=image_read.pixel_array\n    except:\n        image_read = sitk.ReadImage(self.path)\n        image_read.PhotometricInterpretation = 'YBR RCT'\n        data=sitk.GetArrayFromImage(image_read)\n    \n    image_grayscale = dcm2rgbgray(data)\n    \n    if len(image_grayscale.shape) == 3:\n        image_grayscale = image_grayscale.squeeze(0)\n    \n    clahe = cv2.createCLAHE(clipLimit = 20)\n    image_clahe = clahe.apply(image_grayscale) + 30\n    \n    image_resize = resize(image_clahe,(512,512))\n    \n    image_normalised = image_resize/np.max(image_resize)\n    self.image = image_normalised\n    \n    \n  def getimg(self):\n    return self.image","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:35.003884Z","iopub.execute_input":"2023-01-28T12:12:35.004832Z","iopub.status.idle":"2023-01-28T12:12:35.361354Z","shell.execute_reply.started":"2023-01-28T12:12:35.004789Z","shell.execute_reply":"2023-01-28T12:12:35.360423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchvision.transforms as T\nfrom PIL import Image\n\nclass Mammogram_Processed_Augmented_Image:\n    def __init__(self,path,augmentation_dict=0):\n        self.path = path\n        self.augmentation_dict = augmentation_dict\n        \n        self.image_array = Mammogram_Processed_Image(self.path).getimg()\n        self.image = Image.fromarray(self.image_array)\n        \n        if 'hflip' in self.augmentation_dict:\n            hflipper = T.RandomHorizontalFlip(p=0.5)\n            transformed_image = hflipper(self.image)\n            \n        if 'vflip' in self.augmentation_dict:\n            vflipper = T.RandomVerticalFlip(p=0.5)\n            transformed_image = vflipper(transformed_image)\n            \n        if 'noise' in self.augmentation_dict:\n            blurrer = T.GaussianBlur(kernel_size=(1, 479), sigma=(0.1, 5))\n            transformed_image = blurrer(transformed_image)\n        \n        if 'rotate' in self.augmentation_dict:\n            rotater = T.RandomRotation(degrees=(0, 180))\n            transformed_image = rotater(transformed_image)\n            \n        self.output = np.array(transformed_image)\n    \n    def getimg(self):\n        return self.output\n            ","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:35.967555Z","iopub.execute_input":"2023-01-28T12:12:35.969856Z","iopub.status.idle":"2023-01-28T12:12:37.934867Z","shell.execute_reply.started":"2023-01-28T12:12:35.969821Z","shell.execute_reply":"2023-01-28T12:12:37.933969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Grouping by Laterality and Patient","metadata":{}},{"cell_type":"markdown","source":"Cancer is detected per breast and so we should group the data respectively. Some of these groupings return up to 8 images but at a minimum there's 2, an MLO and a CC view so we'll train accordingly as a baseline, and then explore ways we could introduce the extra 1-6 images in future modelling.","metadata":{}},{"cell_type":"code","source":"from collections import namedtuple\n\ndef getCancerInfoList() :\n    df16=df.groupby(['patient_id','laterality','cancer','density','difficult_negative_case'],dropna=False)['image_id'].apply(list).reset_index(name='image_ids')\n    \n    CancerInfoTuple = namedtuple('CancerInfo','patient_id,laterality,cancer,image_ids,density,difficult_negative_case')\n    \n    CancerInfo_list = []\n\n    for x in df16.index:\n        CancerInfo_list.append(CancerInfoTuple(df16['patient_id'][x],df16['laterality'][x],df16['cancer'][x],df16['image_ids'][x],df16['density'][x],df16['difficult_negative_case'][x]))\n        \n    return CancerInfo_list","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:37.936567Z","iopub.execute_input":"2023-01-28T12:12:37.937253Z","iopub.status.idle":"2023-01-28T12:12:37.951141Z","shell.execute_reply.started":"2023-01-28T12:12:37.937218Z","shell.execute_reply":"2023-01-28T12:12:37.950085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df16=df.groupby(['patient_id','laterality','cancer','density','difficult_negative_case'],dropna=False)['image_id'].apply(list).reset_index(name='image_ids')\ndf16","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:38.234556Z","iopub.execute_input":"2023-01-28T12:12:38.234843Z","iopub.status.idle":"2023-01-28T12:12:38.77707Z","shell.execute_reply.started":"2023-01-28T12:12:38.234817Z","shell.execute_reply":"2023-01-28T12:12:38.775879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"code","source":"CancerInfo_list = getCancerInfoList()","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:40.156918Z","iopub.execute_input":"2023-01-28T12:12:40.157306Z","iopub.status.idle":"2023-01-28T12:12:41.585256Z","shell.execute_reply.started":"2023-01-28T12:12:40.157273Z","shell.execute_reply":"2023-01-28T12:12:41.584198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CancerInfo_list[0]\n\nCancerInfo_tup = CancerInfo_list[0]","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:41.587221Z","iopub.execute_input":"2023-01-28T12:12:41.587604Z","iopub.status.idle":"2023-01-28T12:12:41.593084Z","shell.execute_reply.started":"2023-01-28T12:12:41.587569Z","shell.execute_reply":"2023-01-28T12:12:41.591969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CancerInfo_tup.difficult_negative_case","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:41.755697Z","iopub.execute_input":"2023-01-28T12:12:41.756572Z","iopub.status.idle":"2023-01-28T12:12:41.763208Z","shell.execute_reply.started":"2023-01-28T12:12:41.756533Z","shell.execute_reply":"2023-01-28T12:12:41.762275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import copy\nimport random\nimport torch\n\nfrom torch.utils.data import Dataset\n\n\nclass CancerDataset(Dataset):\n    \n    def __init__(self,\n                 val_stride=0,\n                 isValSet_bool=None,\n                 sortby_str='random',\n                 ratio_int=0,\n                 augmentation_dict=None,\n                 CancerInfo_list=None, \n                 transform = None):\n        self.ratio_int      = ratio_int\n        self.augmentation_dict = augmentation_dict\n        self.transform = transform\n        \n        # Take's the input data, otherwise the full dataset as defined above\n        if CancerInfo_list:\n            self.CancerInfo_list = copy.copy(CancerInfo_list)\n        else:\n            self.CancerInfo_list = copy.copy(getCancerInfoList())\n            \n        # Setting up validation to measure accuracy throughout training    \n        if isValSet_bool:\n            assert val_stride > 0, val_stride\n            self.CancerInfo_list = self.CancerInfo_list[::val_stride]\n            assert self.CancerInfo_list\n        elif val_stride > 0:\n            del self.CancerInfo_list[::val_stride]\n            assert self.CancerInfo_list\n            \n        # Shuffle the data\n        if sortby_str == 'random':\n            random.shuffle(self.CancerInfo_list)\n            \n        # Split the data into positive and negative\n        self.negative_list = [\n            nt for nt in self.CancerInfo_list if not nt.cancer\n        ]\n        self.positive_list = [\n            nt for nt in self.CancerInfo_list if nt.cancer\n        ]\n        \n    # Build a function to shuffle respective lists if we set a ratio between the posituve and negative lists\n    def shuffleSamples(self):\n        if self.ratio_int:\n            random.shuffle(self.negative_list)\n            random.shuffle(self.positive_list)\n            \n    # Define the length of the Dataset as a minimum with __getitem__    \n    def __len__(self):\n        if self.ratio_int:\n            return 20000\n        else:\n            return len(self.CancerInfo_list)\n    \n    def __getitem__(self, ndx):\n        if self.ratio_int:\n            pos_ndx = ndx // (self.ratio_int + 1)\n\n            if ndx % (self.ratio_int + 1):\n                neg_ndx = ndx - 1 - pos_ndx\n                neg_ndx %= len(self.negative_list)\n                CancerInfo_tup = self.negative_list[neg_ndx]\n            else:\n                pos_ndx %= len(self.positive_list)\n                CancerInfo_tup = self.positive_list[pos_ndx]\n        else:\n            CancerInfo_tup = self.CancerInfo_list[ndx]\n            \n        #Only include 2 views to provide a standard input to the model\n        image_list = CancerInfo_tup[3]\n        image1 = image_list[0]\n        for i in range(1,len(image_list)):\n            if (df[df['image_id']== image1]['view'] == 'MLO').bool() == (df[df['image_id']== image_list[i]]['view'] =='MLO').bool():\n                pass\n            else:\n                image2 = image_list[i]\n                break\n        \n        path1 = '/kaggle/input/rsna-breast-cancer-detection/train_images/' + str(CancerInfo_tup[0]) + '/' + str(image1) +'.dcm'\n        path2 = '/kaggle/input/rsna-breast-cancer-detection/train_images/' + str(CancerInfo_tup[0]) + '/' + str(image2) +'.dcm'\n        if self.augmentation_dict:\n        #Augmentation\n            output1 = Mammogram_Processed_Augmented_Image(path1,self.augmentation_dict).getimg()\n            output2 = Mammogram_Processed_Augmented_Image(path2,self.augmentation_dict).getimg()\n        else:\n        #Load Image Class + Concatenate\n            output1 = Mammogram_Processed_Image(path1).getimg()\n            output2 = Mammogram_Processed_Image(path2).getimg()\n        final_output = np.concatenate((np.expand_dims(output1,axis=0),np.expand_dims(output2,axis=0)),axis=0)\n        \n            \n        # augmentations\n        if self.transform is not None:\n            final_output = self.transform(image = final_output)['image']\n            \n        candidate_t = torch.from_numpy(final_output).to(torch.float32)\n        \n        pos_t = torch.tensor([\n                not CancerInfo_tup.cancer,\n                CancerInfo_tup.cancer\n            ],\n            dtype=torch.long,\n        )\n        \n        density_t = 0\n        if CancerInfo_tup.density == 'A':\n            density_t = 1\n        if CancerInfo_tup.density == 'B':\n            density_t = 2\n        if CancerInfo_tup.density == 'C':\n            density_t = 3\n        if CancerInfo_tup.density == 'D':\n            density_t = 4\n        density_t = torch.tensor(density_t)\n        \n        return candidate_t, pos_t, density_t, CancerInfo_tup.difficult_negative_case","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:43.395125Z","iopub.execute_input":"2023-01-28T12:12:43.395717Z","iopub.status.idle":"2023-01-28T12:12:43.417355Z","shell.execute_reply.started":"2023-01-28T12:12:43.395683Z","shell.execute_reply":"2023-01-28T12:12:43.416255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check that our dataset is working as expected\n\ntest_dataset = CancerDataset()\ntest_result = test_dataset.__getitem__(5)","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:44.218786Z","iopub.execute_input":"2023-01-28T12:12:44.219592Z","iopub.status.idle":"2023-01-28T12:12:48.204531Z","shell.execute_reply.started":"2023-01-28T12:12:44.219559Z","shell.execute_reply":"2023-01-28T12:12:48.203504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"import math\n\nfrom torch import nn as nn\n\n\nclass CancerModel(nn.Module):\n    def __init__(self, in_channels=2, conv_channels=8):\n        super().__init__()\n\n        self.tail_batchnorm = nn.BatchNorm2d(2)\n\n        # Input is (B,2,512,512) Output is (B,8,256,256)\n        self.block1 = CancerBlock(in_channels, conv_channels)\n        # Input is (B,8,256,256) Output is (B,16,128,128)\n        self.block2 = CancerBlock(conv_channels, conv_channels * 2)\n        # Input is (B,16,128,128) Output is (B,32,64,64)\n        self.block3 = CancerBlock(conv_channels * 2, conv_channels * 4)\n        # Input is (B,32,64,64) Output is (B,64,32,32)\n        self.block4 = CancerBlock(conv_channels * 4, conv_channels * 8)\n        # Input is (B,64,32,32) Output is (B,128,16,16)\n        self.block5 = CancerBlock(conv_channels * 8, conv_channels * 16)\n\n        #Flatten layer will output (batch_size, 32768)\n\n        #Extra layers might be required\n        self.head_linear = nn.Linear(32768, 2)\n        self.head_softmax = nn.Softmax(dim=1)\n\n        self._init_weights()\n    \n    def _init_weights(self):\n        for m in self.modules():\n            if type(m) in {\n                nn.Linear,\n                nn.Conv3d,\n                nn.Conv2d,\n                nn.ConvTranspose2d,\n                nn.ConvTranspose3d,\n            }:\n                nn.init.kaiming_normal_(\n                    m.weight.data, a=0, mode='fan_out', nonlinearity='relu',\n                )\n                if m.bias is not None:\n                    fan_in, fan_out = \\\n                        nn.init._calculate_fan_in_and_fan_out(m.weight.data)\n                    bound = 1 / math.sqrt(fan_out)\n                    nn.init.normal_(m.bias, -bound, bound)\n\n  \n\n\n    def forward(self, input_batch):\n        bn_output1 = self.tail_batchnorm(input_batch)\n\n        block_out1 = self.block1(bn_output1)\n        block_out2 = self.block2(block_out1)\n        block_out3 = self.block3(block_out2)\n        block_out4 = self.block4(block_out3)\n        block_out5 = self.block5(block_out4)\n\n        # To be changed when optimizing storage\n        conv_flat = block_out5.reshape(\n            block_out5.size(0),\n            -1,\n        )\n        linear_output = self.head_linear(conv_flat)\n\n        return linear_output, self.head_softmax(linear_output)","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:48.206738Z","iopub.execute_input":"2023-01-28T12:12:48.207113Z","iopub.status.idle":"2023-01-28T12:12:48.221163Z","shell.execute_reply.started":"2023-01-28T12:12:48.207077Z","shell.execute_reply":"2023-01-28T12:12:48.219739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CancerBlock(nn.Module):\n    def __init__(self, in_channels, conv_channels):\n        super().__init__()\n\n        self.conv1 = nn.Conv2d(\n            in_channels, conv_channels, kernel_size=3, padding=1, bias=True,\n        )\n        self.relu1 = nn.ReLU(inplace=True)\n        self.conv2 = nn.Conv2d(\n            conv_channels, conv_channels, kernel_size=3, padding=1, bias=True,\n        )\n        self.relu2 = nn.ReLU(inplace=True)\n\n        self.maxpool = nn.MaxPool2d(2, 2)\n\n    def forward(self, input_batch):\n        block_out = self.conv1(input_batch)\n        block_out = self.relu1(block_out)\n        block_out = self.conv2(block_out)\n        block_out = self.relu2(block_out)\n\n        return self.maxpool(block_out)","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:48.222952Z","iopub.execute_input":"2023-01-28T12:12:48.22335Z","iopub.status.idle":"2023-01-28T12:12:48.234721Z","shell.execute_reply.started":"2023-01-28T12:12:48.223315Z","shell.execute_reply":"2023-01-28T12:12:48.233675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Test to make sure the model is working as expected\n\ntestmodel = CancerModel()\ninput = torch.unsqueeze(test_result[0],0)\ntestmodel(input)","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:12:48.238017Z","iopub.execute_input":"2023-01-28T12:12:48.238309Z","iopub.status.idle":"2023-01-28T12:12:48.500793Z","shell.execute_reply.started":"2023-01-28T12:12:48.238268Z","shell.execute_reply":"2023-01-28T12:12:48.499413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\nimport torch\nimport torch.nn as nn\nfrom torch.optim import Adam\nfrom torch.utils.data import DataLoader\n\nimport datetime\nimport os\nimport logging\nfrom torch.utils.tensorboard import SummaryWriter\n\nlog = logging.getLogger(__name__)\n# log.setLevel(logging.WARN)\nlog.setLevel(logging.INFO)\n# log.setLevel(logging.DEBUG)\n\n# Used for computeBatchLoss and logMetrics to index into metrics_t/metrics_a\nMETRICS_LABEL_NDX=0\nMETRICS_PRED_NDX=1\nMETRICS_LOSS_NDX=2\nMETRICS_DENSITY=3\nMETRICS_DIFFICULTY=4\nMETRICS_SIZE = 5\n\nepochs = 25\n\nclass CancerTrainingApp:\n    def __init__(self):\n        self.time_str = datetime.datetime.now().strftime('%Y-%m-%d_%H.%M.%S')\n\n        self.trn_writer = None\n        self.val_writer = None\n        \n        self.totalTrainingSamples_count = 0\n\n        self.use_cuda = torch.cuda.is_available()\n        self.device = torch.device(\"cuda\" if self.use_cuda else \"cpu\")\n\n        self.model = self.initModel()\n        self.optimizer = self.initOptimizer()\n        \n        self.batch_size = 20\n        \n        self.augmentation_dict = {\n            'vflip':True,\n            'hflip': True,\n            'noise': True,\n            'rotate': True\n        }\n\n\n    def initModel(self):\n        model = CancerModel()\n        if self.use_cuda:\n            log.info(\"Using CUDA; {} devices.\".format(torch.cuda.device_count()))\n            if torch.cuda.device_count() > 1:\n                model = nn.DataParallel(model)\n            model = model.to(self.device)\n        return model\n\n    def initOptimizer(self):\n        return Adam(self.model.parameters())\n\n    def initTrainDl(self):\n        train_ds = CancerDataset(\n            val_stride=10,\n            isValSet_bool=False,\n            ratio_int=3,\n            augmentation_dict=self.augmentation_dict,\n        )\n\n        batch_size = self.batch_size\n        if self.use_cuda:\n            batch_size *= torch.cuda.device_count()\n\n        train_dl = DataLoader(\n            train_ds,\n            batch_size=batch_size,\n            num_workers=2,\n            pin_memory=self.use_cuda,\n        )\n\n        return train_dl\n\n    def initValDl(self):\n        val_ds = CancerDataset(\n            val_stride=10,\n            isValSet_bool=True,\n        )\n\n        batch_size = self.batch_size\n        if self.use_cuda:\n            batch_size *= torch.cuda.device_count()\n\n        val_dl = DataLoader(\n            val_ds,\n            batch_size=batch_size,\n            num_workers=2,\n            pin_memory=self.use_cuda,\n        )\n\n        return val_dl\n\n    def initTensorboardWriters(self):\n        if self.trn_writer is None:\n            log_dir = os.path.join('runs', baseline, self.time_str)\n\n            self.trn_writer = SummaryWriter(\n                log_dir=log_dir + '-trn_cls-' + 'dlwpt')\n            self.val_writer = SummaryWriter(\n                log_dir=log_dir + '-val_cls-' + 'dlwpt')\n    \n    \n    def main(self):\n        log.info(\"Starting {}\".format(type(self).__name__))\n        train_dl = self.initTrainDl()\n        val_dl = self.initValDl()\n\n        for epoch_ndx in range(1, epochs + 1):\n            print(\"Epoch {} of {}, {}/{} batches of size {}*{}\".format(\n                epoch_ndx,\n                epochs,\n                len(train_dl),\n                len(val_dl),\n                self.batch_size,\n                (torch.cuda.device_count() if self.use_cuda else 1),\n            ))\n            \n            trnMetrics_t = self.doTraining(epoch_ndx, train_dl)\n            self.logMetrics(epoch_ndx, 'trn', trnMetrics_t)\n\n            valMetrics_t = self.doValidation(epoch_ndx, val_dl)\n            self.logMetrics(epoch_ndx, 'val', valMetrics_t)\n            \n        if hasattr(self, 'trn_writer'):\n            self.trn_writer.close()\n            self.val_writer.close()    \n\n\n    def doTraining(self, epoch_ndx, train_dl):\n        self.model.train()\n        train_dl.dataset.shuffleSamples()\n        trnMetrics_g = torch.zeros(\n            METRICS_SIZE,\n            len(train_dl.dataset),\n            device=self.device,\n        )\n        \n        batch_iter = enumerate(\n            train_dl\n        )\n        \n        for batch_ndx, batch_tup in batch_iter:\n            self.optimizer.zero_grad()\n\n            loss_var = self.computeBatchLoss(\n                batch_ndx,\n                batch_tup,\n                train_dl.batch_size,\n                trnMetrics_g\n            )\n\n            loss_var.backward()\n            self.optimizer.step()\n\n        self.totalTrainingSamples_count += len(train_dl.dataset)\n\n        return trnMetrics_g.to('cpu')\n\n\n    def doValidation(self, epoch_ndx, val_dl):\n        with torch.no_grad():\n            self.model.eval()\n            valMetrics_g = torch.zeros(\n                METRICS_SIZE,\n                len(val_dl.dataset),\n                device=self.device,\n            )\n\n            batch_iter = enumerate(\n                val_dl\n            )\n            \n            for batch_ndx, batch_tup in batch_iter:\n                loss_val = self.computeBatchLoss(\n                    batch_ndx,\n                    batch_tup,\n                    val_dl.batch_size,\n                    valMetrics_g\n                )\n\n        return valMetrics_g.to('cpu')\n\n\n\n    def computeBatchLoss(self, batch_ndx, batch_tup, batch_size,metrics_g):\n        input_t, label_t, density_t, difficult_t = batch_tup\n\n        input_g = input_t.to(self.device, non_blocking=True)\n        label_g = label_t.to(self.device, non_blocking=True)\n        density_g = density_t\n        difficult_g = difficult_t.to(self.device, non_blocking=True)\n\n        logits_g, probability_g = self.model(input_g)\n\n        loss_func = nn.CrossEntropyLoss(reduction='none')\n        loss_g = loss_func(\n            logits_g,\n            label_g[:,1],\n        )\n        \n        start_ndx = batch_ndx * batch_size\n        end_ndx = start_ndx + label_t.size(0)\n        \n        metrics_g[METRICS_LABEL_NDX, start_ndx:end_ndx] = label_g[:,1]\n        metrics_g[METRICS_PRED_NDX, start_ndx:end_ndx] = probability_g[:,1]\n        metrics_g[METRICS_LOSS_NDX, start_ndx:end_ndx] = loss_g\n        metrics_g[METRICS_DENSITY, start_ndx:end_ndx] = density_t\n        metrics_g[METRICS_DIFFICULTY, start_ndx:end_ndx] = difficult_t\n        \n        \n        print(\"The loss of batch {} of this epoch is {}\".format(batch_ndx,loss_g.mean()))\n\n        return loss_g.mean()\n    \n    def logMetrics(\n            self,\n            epoch_ndx,\n            mode_str,\n            metrics_t,\n            classificationThreshold=0.5,\n    ):\n        self.initTensorboardWriters()\n        log.info(\"E{} {}\".format(\n            epoch_ndx,\n            type(self).__name__,\n        ))\n        \n        #Whole Dataset\n        negLabel_mask = metrics_t[METRICS_LABEL_NDX] <= classificationThreshold\n        negPred_mask = metrics_t[METRICS_PRED_NDX] <= classificationThreshold\n\n        posLabel_mask = ~negLabel_mask\n        posPred_mask = ~negPred_mask\n\n        neg_count = int(negLabel_mask.sum())\n        pos_count = int(posLabel_mask.sum())\n\n        trueNeg_count = neg_correct = int((negLabel_mask & negPred_mask).sum())\n        truePos_count = pos_correct = int((posLabel_mask & posPred_mask).sum())\n\n        falsePos_count = neg_count - neg_correct\n        falseNeg_count = pos_count - pos_correct\n\n        metrics_dict = {}\n        metrics_dict['loss/all'] = metrics_t[METRICS_LOSS_NDX].mean()\n        metrics_dict['loss/neg'] = metrics_t[METRICS_LOSS_NDX, negLabel_mask].mean()\n        metrics_dict['loss/pos'] = metrics_t[METRICS_LOSS_NDX, posLabel_mask].mean()\n\n        metrics_dict['correct/all'] = (pos_correct + neg_correct) / metrics_t.shape[1] * 100\n        metrics_dict['correct/neg'] = (neg_correct) / neg_count * 100\n        metrics_dict['correct/pos'] = (pos_correct) / pos_count * 100\n\n        precision = metrics_dict['pr/precision'] = \\\n            truePos_count / np.float32(truePos_count + falsePos_count)\n        recall    = metrics_dict['pr/recall'] = \\\n            truePos_count / np.float32(truePos_count + falseNeg_count)\n\n        metrics_dict['pr/f1_score'] = \\\n            2 * (precision * recall) / (precision + recall)\n\n        log.info(\n            (\"E{} {:8} {loss/all:.4f} loss, \"\n                 + \"{correct/all:-5.1f}% correct, \"\n                 + \"{pr/precision:.4f} precision, \"\n                 + \"{pr/recall:.4f} recall, \"\n                 + \"{pr/f1_score:.4f} f1 score\"\n            ).format(\n                epoch_ndx,\n                mode_str,\n                **metrics_dict,\n            )\n        )\n        log.info(\n            (\"E{} {:8} {loss/neg:.4f} loss, \"\n                 + \"{correct/neg:-5.1f}% correct ({neg_correct:} of {neg_count:})\"\n            ).format(\n                epoch_ndx,\n                mode_str + '_neg',\n                neg_correct=neg_correct,\n                neg_count=neg_count,\n                **metrics_dict,\n            )\n        )\n        log.info(\n            (\"E{} {:8} {loss/pos:.4f} loss, \"\n                 + \"{correct/pos:-5.1f}% correct ({pos_correct:} of {pos_count:})\"\n            ).format(\n                epoch_ndx,\n                mode_str + '_pos',\n                pos_correct=pos_correct,\n                pos_count=pos_count,\n                **metrics_dict,\n            )\n        )\n        writer = getattr(self, mode_str + '_writer')\n\n        for key, value in metrics_dict.items():\n            writer.add_scalar(key, value, self.totalTrainingSamples_count)\n\n        writer.add_pr_curve(\n            'pr',\n            metrics_t[METRICS_LABEL_NDX],\n            metrics_t[METRICS_PRED_NDX],\n            self.totalTrainingSamples_count,\n        )\n\n        bins = [x/50.0 for x in range(51)]\n\n        negHist_mask = negLabel_mask & (metrics_t[METRICS_PRED_NDX] > 0.01)\n        posHist_mask = posLabel_mask & (metrics_t[METRICS_PRED_NDX] < 0.99)\n\n        if negHist_mask.any():\n            writer.add_histogram(\n                'is_neg',\n                metrics_t[METRICS_PRED_NDX, negHist_mask],\n                self.totalTrainingSamples_count,\n                bins=bins,\n            )\n        if posHist_mask.any():\n            writer.add_histogram(\n                'is_pos',\n                metrics_t[METRICS_PRED_NDX, posHist_mask],\n                self.totalTrainingSamples_count,\n                bins=bins,\n            )\n        \n        #MUST REDO\n        #Loop to cover same metrics for density specific to check that the model performs the same over all densities\n        d ={}\n        for x in ['A','B','C','D']:\n            negLabel_mask = metrics_t[METRICS_LABEL_NDX] <= classificationThreshold\n            negPred_mask = metrics_t[METRICS_PRED_NDX] <= classificationThreshold\n\n            posLabel_mask = ~negLabel_mask\n            posPred_mask = ~negPred_mask\n\n            neg_count_density = int((negLabel_mask_density & metrics_t[METRICS_DENSITY] == x).sum())\n            pos_count_density = int((posLabel_mask_density & metrics_t[METRICS_DENSITY] == x).sum())\n\n            trueNeg_count_density = neg_correct_density = int((negLabel_mask & negPred_mask & metrics_t[METRICS_DENSITY] == x).sum())\n            truePos_count_density = pos_correct_density = int((posLabel_mask & posPred_mask & metrics_t[METRICS_DENSITY] == x).sum())\n\n            falsePos_count_density = neg_count_density - neg_correct_density\n            falseNeg_count_density = pos_count_density - pos_correct_density\n\n            metrics_dict_density = {}\n            d[\"{}\".format(x)] = metrics_dict_density\n                \n            d[\"{}\".format(x)]['loss/all'] = (metrics_t[METRICS_LOSS_NDX] & metrics_t[METRICS_DENSITY] == x).mean()\n            d[\"{}\".format(x)]['loss/neg'] = (metrics_t[METRICS_LOSS_NDX, negLabel_mask] & metrics_t[METRICS_DENSITY] == x).mean()\n            d[\"{}\".format(x)]['loss/pos'] = (metrics_t[METRICS_LOSS_NDX, posLabel_mask] & metrics_t[METRICS_DENSITY] == x).mean()\n\n            d[\"{}\".format(x)]['correct/all'] = (pos_correct_density + neg_correct_density) / (metrics_t[METRICS_DENSITY] == x).count() * 100\n            d[\"{}\".format(x)]['correct/neg'] = (neg_correct_density) / neg_count_density * 100\n            d[\"{}\".format(x)]['correct/pos'] = (pos_correct_density) / pos_count_density * 100\n\n            precision_density = d[\"{}\".format(x)]['pr/precision'] = \\\n                truePos_count_density / np.float32(truePos_count_density + falsePos_count_density)\n            recall_density    = d[\"{}\".format(x)]['pr/recall'] = \\\n                truePos_count_density / np.float32(truePos_count_density + falseNeg_count_density)\n\n            d[\"{}\".format(x)]['pr/f1_score'] = \\\n                2 * (precision_density * recall_density) / (precision_density + recall_density)\n\n            log.info(\n                (\"E{} {:8} {loss/all:.4f} loss, \"\n                    + \"{correct/all:-5.1f}% correct, \"\n                    + \"{pr/precision:.4f} precision, \"\n                    + \"{pr/recall:.4f} recall, \"\n                    + \"{pr/f1_score:.4f} f1 score\"\n                ).format(\n                    epoch_ndx,\n                    mode_str,\n                    **d[\"{}\".format(x)],\n                    )\n                )\n            log.info(\n                (\"E{} {:8} {loss/neg:.4f} loss, \"\n                    + \"{correct/neg:-5.1f}% correct ({neg_correct_density:} of {neg_count_density:})\"\n                ).format(\n                    epoch_ndx,\n                    mode_str + '_neg',\n                    neg_correct_density=neg_correct_density,\n                    neg_count_density=neg_count_density,\n                    **d[\"{}\".format(x)],\n                )\n            )\n            log.info(\n                (\"E{} {:8} {loss/pos:.4f} loss, \"\n                    + \"{correct/pos:-5.1f}% correct ({pos_correct_density:} of {pos_count_density:})\"\n                ).format(\n                    epoch_ndx,\n                    mode_str + '_pos',\n                    pos_correct_density=pos_correct_density,\n                    pos_count_density=pos_count_density,\n                    **d[\"{}\".format(x)],\n                )\n            )\n            writer = getattr(self, mode_str + '_writer')\n\n            for key, value in d[\"{}\".format(x)].items():\n                writer.add_scalar(key, value, self.totalTrainingSamples_count)\n\n            bins = [x/50.0 for x in range(51)]\n\n            negHist_mask_density = negLabel_mask_density & (metrics_t[METRICS_PRED_NDX] > 0.01) & metrics_t[METRICS_DENSITY] == x\n            posHist_mask_density = posLabel_mask_density & (metrics_t[METRICS_PRED_NDX] < 0.99) & metrics_t[METRICS_DENSITY] == x\n\n            if negHist_mask_density.any():\n                writer.add_histogram(\n                    'is_neg {}'.format(x),\n                    metrics_t[METRICS_PRED_NDX, negHist_mask_density],\n                    self.totalTrainingSamples_count,\n                    bins=bins,\n                )\n            if posHist_mask_density.any():\n                writer.add_histogram(\n                    'is_pos {}'.format(x),\n                    metrics_t[METRICS_PRED_NDX, posHist_mask_density],\n                    self.totalTrainingSamples_count,\n                    bins=bins,\n                )\n        \n        #MUST DO\n        #Loop to cover same metrics for difficult specific to check that the model performs the same over all difficulties\n        e ={}\n        for x in [True, False]:\n            negLabel_mask = metrics_t[METRICS_LABEL_NDX] <= classificationThreshold\n            negPred_mask = metrics_t[METRICS_PRED_NDX] <= classificationThreshold\n\n            posLabel_mask = ~negLabel_mask\n            posPred_mask = ~negPred_mask\n\n            neg_count_difficulty = int((negLabel_mask_difficulty & metrics_t[METRICS_DIFFICULTY] == x).sum())\n            pos_count_difficulty = int((posLabel_mask_difficulty & metrics_t[METRICS_DIFFICULTY] == x).sum())\n\n            trueNeg_count_difficulty = neg_correct_difficulty = int((negLabel_mask & negPred_mask & metrics_t[METRICS_DIFFICULTY] == x).sum())\n            truePos_count_difficulty = pos_correct_difficulty = int((posLabel_mask & posPred_mask & metrics_t[METRICS_DIFFICULTY] == x).sum())\n\n            falsePos_count_difficulty = neg_count_difficulty - neg_correct_difficulty\n            falseNeg_count_difficulty = pos_count_difficulty - pos_correct_difficulty\n\n            metrics_dict_difficulty = {}\n            e[\"{}\".format(x)] = metrics_dict_difficulty\n                \n            e[\"{}\".format(x)]['loss/all'] = (metrics_t[METRICS_LOSS_NDX] & metrics_t[METRICS_DIFFICULTY] == x).mean()\n            e[\"{}\".format(x)]['loss/neg'] = (metrics_t[METRICS_LOSS_NDX, negLabel_mask] & metrics_t[METRICS_DIFFICULTY] == x).mean()\n            e[\"{}\".format(x)]['loss/pos'] = (metrics_t[METRICS_LOSS_NDX, posLabel_mask] & metrics_t[METRICS_DIFFICULTY] == x).mean()\n\n            e[\"{}\".format(x)]['correct/all'] = (pos_correct_difficulty + neg_correct_difficulty) / (metrics_t[METRICS_DIFFICULTY] == x).count() * 100\n            e[\"{}\".format(x)]['correct/neg'] = (neg_correct_difficulty) / neg_count_difficulty * 100\n            e[\"{}\".format(x)]['correct/pos'] = (pos_correct_difficulty) / pos_count_difficulty * 100\n\n            precision_difficulty = e[\"{}\".format(x)]['pr/precision'] = \\\n                truePos_count_difficulty / np.float32(truePos_count_difficulty + falsePos_count_difficulty)\n            recall_difficulty = e[\"{}\".format(x)]['pr/recall'] = \\\n                truePos_count_difficulty / np.float32(truePos_count_difficulty + falseNeg_count_difficulty)\n\n            e[\"{}\".format(x)]['pr/f1_score'] = \\\n                2 * (precision_difficulty * recall_difficulty) / (precision_difficulty + recall_difficulty)\n\n            log.info(\n                (\"E{} {:8} {loss/all:.4f} loss, \"\n                    + \"{correct/all:-5.1f}% correct, \"\n                    + \"{pr/precision:.4f} precision, \"\n                    + \"{pr/recall:.4f} recall, \"\n                    + \"{pr/f1_score:.4f} f1 score\"\n                ).format(\n                    epoch_ndx,\n                    mode_str,\n                    **e[\"{}\".format(x)],\n                    )\n                )\n            log.info(\n                (\"E{} {:8} {loss/neg:.4f} loss, \"\n                    + \"{correct/neg:-5.1f}% correct ({neg_correct_density:} of {neg_count_density:})\"\n                ).format(\n                    epoch_ndx,\n                    mode_str + '_neg',\n                    neg_correct_difficulty=neg_correct_difficulty,\n                    neg_count_difficulty=neg_count_difficulty,\n                    **e[\"{}\".format(x)],\n                )\n            )\n            log.info(\n                (\"E{} {:8} {loss/pos:.4f} loss, \"\n                    + \"{correct/pos:-5.1f}% correct ({pos_correct_density:} of {pos_count_density:})\"\n                ).format(\n                    epoch_ndx,\n                    mode_str + '_pos',\n                    pos_correct_difficulty=pos_correct_difficulty,\n                    pos_count_difficulty=pos_count_difficulty,\n                    **e[\"{}\".format(x)],\n                )\n            )\n            writer = getattr(self, mode_str + '_writer')\n\n            for key, value in e[\"{}\".format(x)].items():\n                writer.add_scalar(key, value, self.totalTrainingSamples_count)\n\n            bins = [x/50.0 for x in range(51)]\n\n            negHist_mask_difficulty = negLabel_mask_difficulty & (metrics_t[METRICS_PRED_NDX] > 0.01) & metrics_t[METRICS_DIFFICULTY] == x\n            posHist_mask_difficulty = posLabel_mask_difficulty & (metrics_t[METRICS_PRED_NDX] < 0.99) & metrics_t[METRICS_DIFFICULTY] == x\n\n            if negHist_mask_difficulty.any():\n                writer.add_histogram(\n                    'is_neg {}'.format(x),\n                    metrics_t[METRICS_PRED_NDX, negHist_mask_difficulty],\n                    self.totalTrainingSamples_count,\n                    bins=bins,\n                )\n            if posHist_mask_difficulty.any():\n                writer.add_histogram(\n                    'is_pos {}'.format(x),\n                    metrics_t[METRICS_PRED_NDX, posHist_mask_difficulty],\n                    self.totalTrainingSamples_count,\n                    bins=bins,\n                )\n        \n","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:13:34.78259Z","iopub.execute_input":"2023-01-28T12:13:34.784984Z","iopub.status.idle":"2023-01-28T12:13:35.061082Z","shell.execute_reply.started":"2023-01-28T12:13:34.784937Z","shell.execute_reply":"2023-01-28T12:13:35.058195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training = CancerTrainingApp()","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:13:36.363753Z","iopub.execute_input":"2023-01-28T12:13:36.364529Z","iopub.status.idle":"2023-01-28T12:13:36.386252Z","shell.execute_reply.started":"2023-01-28T12:13:36.364492Z","shell.execute_reply":"2023-01-28T12:13:36.384683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training.main()\n\n# Add print functions to figure out where it's going wrong\n# Re-run with correct divisables to see what needs to be changed","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:13:37.948442Z","iopub.execute_input":"2023-01-28T12:13:37.948913Z","iopub.status.idle":"2023-01-28T13:05:12.6318Z","shell.execute_reply.started":"2023-01-28T12:13:37.948875Z","shell.execute_reply":"2023-01-28T13:05:12.629946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utility","metadata":{}},{"cell_type":"code","source":"import logging\nimport logging.handlers\n\nroot_logger = logging.getLogger()\nroot_logger.setLevel(logging.INFO)\n\n# Some libraries attempt to add their own root logger handlers. This is\n# annoying and so we get rid of them.\nfor handler in list(root_logger.handlers):\n    root_logger.removeHandler(handler)\n\nlogfmt_str = \"%(asctime)s %(levelname)-8s pid:%(process)d %(name)s:%(lineno)03d:%(funcName)s %(message)s\"\nformatter = logging.Formatter(logfmt_str)\n\nstreamHandler = logging.StreamHandler()\nstreamHandler.setFormatter(formatter)\nstreamHandler.setLevel(logging.DEBUG)\n\nroot_logger.addHandler(streamHandler)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T15:49:20.114831Z","iopub.execute_input":"2023-01-22T15:49:20.11604Z","iopub.status.idle":"2023-01-22T15:49:20.123546Z","shell.execute_reply.started":"2023-01-22T15:49:20.115993Z","shell.execute_reply":"2023-01-22T15:49:20.122476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Appendix -- Mostly research or random pockets of code that helps with direction","metadata":{}}]}