{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Importing packages\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pickle\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import cross_validate, cross_val_score, GridSearchCV, RandomizedSearchCV\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score, f1_score, roc_curve, auc\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom bayes_opt import BayesianOptimization\nfrom sklearn.impute import SimpleImputer\nimport lightgbm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-09T11:59:28.818075Z","iopub.execute_input":"2023-04-09T11:59:28.818419Z","iopub.status.idle":"2023-04-09T11:59:32.789179Z","shell.execute_reply.started":"2023-04-09T11:59:28.818392Z","shell.execute_reply":"2023-04-09T11:59:32.787809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Loading the files Xtrain_scaled_pca, Xtest_scaled_pca, Ytrain, Ytest from the \n#preprocessing notebook.\n\nfolder_path = '/kaggle/input/rsna-breast-cancer-detection-preprocessing-2/'\n\nwith open(folder_path + 'Xtest_scaled_pca.pickle', 'rb') as file:\n    Xtest_scaled_pca = pickle.load(file)\nwith open(folder_path + 'Xtrain_scaled_pca.pickle', 'rb') as file:\n    Xtrain_scaled_pca = pickle.load(file)\nwith open(folder_path + 'Ytest.pickle', 'rb') as file:\n    Ytest = pickle.load(file)\nwith open(folder_path + 'Ytrain.pickle', 'rb') as file:\n    Ytrain = pickle.load(file)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:59:32.791074Z","iopub.execute_input":"2023-04-09T11:59:32.791441Z","iopub.status.idle":"2023-04-09T11:59:32.975049Z","shell.execute_reply.started":"2023-04-09T11:59:32.791414Z","shell.execute_reply":"2023-04-09T11:59:32.974316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Logistic Regression","metadata":{}},{"cell_type":"markdown","source":"Let's first use logistic regression. We will use Bayesian optimization to do the hyperparameter search. We will use the F1-score as the evaluation metric, because we want to minimize the number of false positives.","metadata":{}},{"cell_type":"code","source":"#def logreg_eval(C):\n    #logreg = LogisticRegression(solver = 'liblinear', max_iter = 500, C=C)\n    #cv_results = cross_val_score(logreg, Xtrain_scaled_pca, Ytrain, scoring=\"f1\",cv=5)\n    #return cv_results.mean()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:59:32.97635Z","iopub.execute_input":"2023-04-09T11:59:32.977101Z","iopub.status.idle":"2023-04-09T11:59:32.98305Z","shell.execute_reply.started":"2023-04-09T11:59:32.977066Z","shell.execute_reply":"2023-04-09T11:59:32.981624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#logregBO = BayesianOptimization(logreg_eval, {'C': (0.01,200)})\n#logregBO.maximize(init_points=5,n_iter=10,acq='ucb')","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:59:32.986522Z","iopub.execute_input":"2023-04-09T11:59:32.986955Z","iopub.status.idle":"2023-04-09T11:59:57.331211Z","shell.execute_reply.started":"2023-04-09T11:59:32.986917Z","shell.execute_reply":"2023-04-09T11:59:57.329242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We see that the F1-score doesn't vary much over the whole range of C, and stays at around 0.95. Let's now choose the value C=38.05 and make predictions on the test set.","metadata":{}},{"cell_type":"code","source":"logreg = LogisticRegression(solver = 'liblinear', max_iter = 500, C=38.05)\nlogreg.fit(Xtrain_scaled_pca, Ytrain)\nprint(f'F1-score on training data: {f1_score(logreg.predict(Xtrain_scaled_pca), Ytrain):.2f}')\nprint(f'F1-score on test data: {f1_score(logreg.predict(Xtest_scaled_pca), Ytest):.2f}')","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:59:57.332468Z","iopub.execute_input":"2023-04-09T11:59:57.332812Z","iopub.status.idle":"2023-04-09T11:59:57.638097Z","shell.execute_reply.started":"2023-04-09T11:59:57.332781Z","shell.execute_reply":"2023-04-09T11:59:57.637186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Forest","metadata":{}},{"cell_type":"markdown","source":"Next, we try random forest. We go back to using GridSearchCV because we want to optimize over discrete (integer) values of the parameter min_samples_split, and it's hard to deal with discrete hyperparameter values with BayesianOptimization.","metadata":{}},{"cell_type":"code","source":"rf = RandomForestClassifier(random_state=47, n_estimators=20, n_jobs=-1)\nparameters = {\"min_samples_split\": [3,4],\n             \"max_depth\": [2,3],\n             \"criterion\": [\"gini\",\"entropy\"],\n             \"max_features\": [\"sqrt\", \"log2\", None],\n             \"bootstrap\": [True, False]}\n\ncv_rf = RandomizedSearchCV(rf, param_distributions=parameters, scoring='f1', cv=5, \n                           random_state=13, n_jobs=-1)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:59:57.639515Z","iopub.execute_input":"2023-04-09T11:59:57.640116Z","iopub.status.idle":"2023-04-09T11:59:57.647154Z","shell.execute_reply.started":"2023-04-09T11:59:57.640081Z","shell.execute_reply":"2023-04-09T11:59:57.646127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Printing the best parameter value and the best F1-score of the train set:","metadata":{}},{"cell_type":"code","source":"cv_rf.fit(Xtrain_scaled_pca, Ytrain)\nprint(cv_rf.best_score_)\nprint(cv_rf.best_params_)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:59:57.648385Z","iopub.execute_input":"2023-04-09T11:59:57.648699Z","iopub.status.idle":"2023-04-09T12:00:05.068251Z","shell.execute_reply.started":"2023-04-09T11:59:57.648668Z","shell.execute_reply":"2023-04-09T12:00:05.067095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Printing the F1-score on the test set:","metadata":{}},{"cell_type":"code","source":"Ypred = cv_rf.predict(Xtest_scaled_pca)\nprint(\"F1-score of the test set:\", f1_score(Ytest, Ypred))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T12:00:05.069519Z","iopub.execute_input":"2023-04-09T12:00:05.069751Z","iopub.status.idle":"2023-04-09T12:00:05.180042Z","shell.execute_reply.started":"2023-04-09T12:00:05.069725Z","shell.execute_reply":"2023-04-09T12:00:05.178675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Unfortunately, the F1-score of the test set is much lower than the one of the train set.","metadata":{}},{"cell_type":"markdown","source":"# Gradient boosting","metadata":{}},{"cell_type":"code","source":"gb = GradientBoostingClassifier(random_state=47, n_estimators=20)\nparameters = {\"learning_rate\": [0.2,0.5],\n             \"min_samples_split\": [3,4],\n             \"max_depth\": [2,3],\n             \"criterion\": [\"friedman_mse\", \"squared_error\"],\n             \"max_features\": [\"auto\", \"sqrt\", \"log2\", None]}\n\ncv_gb = RandomizedSearchCV(gb, param_distributions=parameters, scoring='f1', cv=5, \n                           random_state=13, n_jobs=-1)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T12:00:05.181307Z","iopub.execute_input":"2023-04-09T12:00:05.181851Z","iopub.status.idle":"2023-04-09T12:00:05.188829Z","shell.execute_reply.started":"2023-04-09T12:00:05.181815Z","shell.execute_reply":"2023-04-09T12:00:05.187227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Printing the best parameter value and the best F1-score of the train set:","metadata":{}},{"cell_type":"code","source":"cv_gb.fit(Xtrain_scaled_pca, Ytrain)\nprint(cv_gb.best_score_)\nprint(cv_gb.best_params_)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T12:00:05.192957Z","iopub.execute_input":"2023-04-09T12:00:05.193388Z","iopub.status.idle":"2023-04-09T12:00:13.486163Z","shell.execute_reply.started":"2023-04-09T12:00:05.193345Z","shell.execute_reply":"2023-04-09T12:00:13.484874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Printing the F1-score on the test set:","metadata":{}},{"cell_type":"code","source":"Ypred = cv_gb.predict(Xtest_scaled_pca)\nprint(\"F1-score of the test set:\", f1_score(Ytest, Ypred))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T12:00:13.487309Z","iopub.execute_input":"2023-04-09T12:00:13.48759Z","iopub.status.idle":"2023-04-09T12:00:13.501327Z","shell.execute_reply.started":"2023-04-09T12:00:13.487565Z","shell.execute_reply":"2023-04-09T12:00:13.49926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Again, the F1-score of the test set is considerably lower than the one of the train set.","metadata":{}},{"cell_type":"markdown","source":"Overall, it looks like logistic regression is the best algorithm.","metadata":{}},{"cell_type":"markdown","source":"# ROC-AUC for Logistic Regression","metadata":{}},{"cell_type":"markdown","source":"Even though our metric has been chosen to be the F1-score, it's useful to produce the ROC curve and compute the AUC metric for our best algorithm - logistic regression.","metadata":{}},{"cell_type":"code","source":"Ypred_probs = logreg.predict_proba(Xtest_scaled_pca)[:,1]\nfpr, tpr, thresholds = roc_curve(Ytest, Ypred_probs)\nplt.plot([0, 1], [0, 1], 'k--')\nplt.plot(fpr, tpr)\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve for Logistic Regression')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T12:00:13.50282Z","iopub.execute_input":"2023-04-09T12:00:13.503175Z","iopub.status.idle":"2023-04-09T12:00:13.676595Z","shell.execute_reply.started":"2023-04-09T12:00:13.503147Z","shell.execute_reply":"2023-04-09T12:00:13.675336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Print out the area under the curve.\nprint(auc(fpr, tpr))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T12:00:13.677933Z","iopub.execute_input":"2023-04-09T12:00:13.678522Z","iopub.status.idle":"2023-04-09T12:00:13.685211Z","shell.execute_reply.started":"2023-04-09T12:00:13.678494Z","shell.execute_reply":"2023-04-09T12:00:13.683287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's also plot the false positive rate as a function of the threshold.","metadata":{}},{"cell_type":"code","source":"plt.plot(fpr, thresholds)\nplt.xlabel('Thresholds')\nplt.ylabel('False Positive Rate')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T12:00:13.687101Z","iopub.execute_input":"2023-04-09T12:00:13.687499Z","iopub.status.idle":"2023-04-09T12:00:13.811987Z","shell.execute_reply.started":"2023-04-09T12:00:13.687472Z","shell.execute_reply":"2023-04-09T12:00:13.811287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We see that the usual threshold of 0.5 corresponds to an excellent FPR, which is much smaller than the largest acceptable FPR of 25 percent (which we demand at the beginning of the project, in the Project Proposal).","metadata":{}},{"cell_type":"markdown","source":"# Saving models","metadata":{}},{"cell_type":"code","source":"fileObj = open('breast_cancer_detection_model.pickle', 'wb')\npickle.dump(logreg,fileObj)\nfileObj.close()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T12:00:13.813324Z","iopub.execute_input":"2023-04-09T12:00:13.813847Z","iopub.status.idle":"2023-04-09T12:00:13.819014Z","shell.execute_reply.started":"2023-04-09T12:00:13.813812Z","shell.execute_reply":"2023-04-09T12:00:13.818043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fileObj = open('breast_cancer_detection_model2.pickle', 'wb')\npickle.dump(cv_rf,fileObj)\nfileObj.close()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fileObj = open('breast_cancer_detection_model3.pickle', 'wb')\npickle.dump(cv_gb,fileObj)\nfileObj.close()","metadata":{},"execution_count":null,"outputs":[]}]}