{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow_decision_forests as tfdf\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-17T15:38:02.673324Z","iopub.execute_input":"2023-02-17T15:38:02.673732Z","iopub.status.idle":"2023-02-17T15:38:02.680687Z","shell.execute_reply.started":"2023-02-17T15:38:02.6737Z","shell.execute_reply":"2023-02-17T15:38:02.679346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load train and test dataset in a Pandas dataframe.\ntrain_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:02.686025Z","iopub.execute_input":"2023-02-17T15:38:02.686432Z","iopub.status.idle":"2023-02-17T15:38:02.774935Z","shell.execute_reply.started":"2023-02-17T15:38:02.686398Z","shell.execute_reply":"2023-02-17T15:38:02.773513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:02.777386Z","iopub.execute_input":"2023-02-17T15:38:02.777745Z","iopub.status.idle":"2023-02-17T15:38:02.804135Z","shell.execute_reply.started":"2023-02-17T15:38:02.777712Z","shell.execute_reply":"2023-02-17T15:38:02.803247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:02.805229Z","iopub.execute_input":"2023-02-17T15:38:02.806112Z","iopub.status.idle":"2023-02-17T15:38:02.824373Z","shell.execute_reply.started":"2023-02-17T15:38:02.806047Z","shell.execute_reply":"2023-02-17T15:38:02.822956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nimport random\nratio=0.80\npatient_ids = train_df['patient_id'].unique().tolist()\nrandom.shuffle(patient_ids)\nindices = math.ceil(len(patient_ids) * ratio)","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:02.827645Z","iopub.execute_input":"2023-02-17T15:38:02.828175Z","iopub.status.idle":"2023-02-17T15:38:02.850855Z","shell.execute_reply.started":"2023-02-17T15:38:02.828129Z","shell.execute_reply":"2023-02-17T15:38:02.849634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_df= train_df[train_df['patient_id'].isin(patient_ids[indices:])]\ndisplay(val_df)","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:02.854121Z","iopub.execute_input":"2023-02-17T15:38:02.85484Z","iopub.status.idle":"2023-02-17T15:38:02.884714Z","shell.execute_reply.started":"2023-02-17T15:38:02.854801Z","shell.execute_reply":"2023-02-17T15:38:02.883411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_s = train_df[train_df['patient_id'].isin(patient_ids[:indices])]\ndisplay(train_df_s)","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:02.886008Z","iopub.execute_input":"2023-02-17T15:38:02.88645Z","iopub.status.idle":"2023-02-17T15:38:02.920235Z","shell.execute_reply.started":"2023-02-17T15:38:02.886419Z","shell.execute_reply":"2023-02-17T15:38:02.919332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"{} for training, {} for validation.\".format(len(train_df_s), len(val_df)))\nprint(val_df['patient_id'].unique())\nprint(train_df_s['patient_id'].unique())","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:02.921155Z","iopub.execute_input":"2023-02-17T15:38:02.921491Z","iopub.status.idle":"2023-02-17T15:38:02.931816Z","shell.execute_reply.started":"2023-02-17T15:38:02.921462Z","shell.execute_reply":"2023-02-17T15:38:02.930528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert the dataset into a TensorFlow dataset.\ntrain_ds = tfdf.keras.pd_dataframe_to_tf_dataset(train_df_s.loc[:, ['laterality', 'view', 'age','implant','cancer']], label=\"cancer\")\nval_ds = tfdf.keras.pd_dataframe_to_tf_dataset(val_df.loc[:, ['laterality', 'view', 'age','implant','cancer']], label=\"cancer\")\ntest_ds = tfdf.keras.pd_dataframe_to_tf_dataset(test_df.loc[:, ['laterality', 'view', 'age','implant']])","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:02.933402Z","iopub.execute_input":"2023-02-17T15:38:02.933746Z","iopub.status.idle":"2023-02-17T15:38:02.987177Z","shell.execute_reply.started":"2023-02-17T15:38:02.933715Z","shell.execute_reply":"2023-02-17T15:38:02.984935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train a Random Forest model.\nmodel = tfdf.keras.GradientBoostedTreesModel(verbose=10)\nmodel.fit(train_ds)","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:02.988556Z","iopub.execute_input":"2023-02-17T15:38:02.989106Z","iopub.status.idle":"2023-02-17T15:38:05.380585Z","shell.execute_reply.started":"2023-02-17T15:38:02.989072Z","shell.execute_reply":"2023-02-17T15:38:05.379512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Summary of the model structure.\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:05.383635Z","iopub.execute_input":"2023-02-17T15:38:05.383992Z","iopub.status.idle":"2023-02-17T15:38:05.399489Z","shell.execute_reply.started":"2023-02-17T15:38:05.383961Z","shell.execute_reply":"2023-02-17T15:38:05.398591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the model\ntfdf.model_plotter.plot_model_in_colab(model, tree_idx=0, max_depth=5)","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:05.400642Z","iopub.execute_input":"2023-02-17T15:38:05.40095Z","iopub.status.idle":"2023-02-17T15:38:05.41544Z","shell.execute_reply.started":"2023-02-17T15:38:05.40092Z","shell.execute_reply":"2023-02-17T15:38:05.414265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model.\nmodel.evaluate(val_ds)\n\n# Export the model to a SavedModel.\n# model.save(\"project/model\")","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:05.416652Z","iopub.execute_input":"2023-02-17T15:38:05.416966Z","iopub.status.idle":"2023-02-17T15:38:05.543216Z","shell.execute_reply.started":"2023-02-17T15:38:05.416937Z","shell.execute_reply":"2023-02-17T15:38:05.541958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perform the prediction.\npredictions = model.predict(test_ds)\npredictions","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:05.544855Z","iopub.execute_input":"2023-02-17T15:38:05.545328Z","iopub.status.idle":"2023-02-17T15:38:05.654803Z","shell.execute_reply.started":"2023-02-17T15:38:05.545286Z","shell.execute_reply":"2023-02-17T15:38:05.653579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reminder for the submissiom format\npd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:05.656167Z","iopub.execute_input":"2023-02-17T15:38:05.656579Z","iopub.status.idle":"2023-02-17T15:38:05.673276Z","shell.execute_reply.started":"2023-02-17T15:38:05.656546Z","shell.execute_reply":"2023-02-17T15:38:05.672032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['cancer'] = predictions\ntest_df[['prediction_id','cancer']].groupby(['prediction_id']).mean()","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:05.676482Z","iopub.execute_input":"2023-02-17T15:38:05.677551Z","iopub.status.idle":"2023-02-17T15:38:05.691358Z","shell.execute_reply.started":"2023-02-17T15:38:05.677512Z","shell.execute_reply":"2023-02-17T15:38:05.69008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_df = test_df[['prediction_id','cancer']].groupby(['prediction_id']).mean()\nprediction_df","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:05.692466Z","iopub.execute_input":"2023-02-17T15:38:05.692954Z","iopub.status.idle":"2023-02-17T15:38:05.710901Z","shell.execute_reply.started":"2023-02-17T15:38:05.692907Z","shell.execute_reply":"2023-02-17T15:38:05.709355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_df.to_csv('submission.csv',index=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:05.712642Z","iopub.execute_input":"2023-02-17T15:38:05.713108Z","iopub.status.idle":"2023-02-17T15:38:05.720119Z","shell.execute_reply.started":"2023-02-17T15:38:05.713045Z","shell.execute_reply":"2023-02-17T15:38:05.718914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final check of the submissiom csv file\npd.read_csv('/kaggle/working/submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-02-17T15:38:05.721264Z","iopub.execute_input":"2023-02-17T15:38:05.721616Z","iopub.status.idle":"2023-02-17T15:38:05.740318Z","shell.execute_reply.started":"2023-02-17T15:38:05.721585Z","shell.execute_reply":"2023-02-17T15:38:05.739385Z"},"trusted":true},"execution_count":null,"outputs":[]}]}