{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"/kaggle/input/UBC-OCEAN/train.csv\n\n/kaggle/input/UBC-OCEAN/test.csv","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import GridSearchCV\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import balanced_accuracy_score\n# Load the CSV data\ndata = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')  # Replace 'your_data.csv' with your actual data file\n\n# Data Preprocessing\n# Encode the 'label' column\nlabel_encoder = LabelEncoder()\ndata['label_encoded'] = label_encoder.fit_transform(data['label'])\n\n# Split the data into features and target\nX = data[['image_width', 'image_height', 'is_tma']]\ny = data['label_encoded']\n\n# Split the data into training and validation sets\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-19T16:58:37.343392Z","iopub.execute_input":"2023-10-19T16:58:37.34451Z","iopub.status.idle":"2023-10-19T16:58:37.358248Z","shell.execute_reply.started":"2023-10-19T16:58:37.34448Z","shell.execute_reply":"2023-10-19T16:58:37.357221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Define the Random Forest model\nrf_model = RandomForestClassifier(random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T16:58:37.360279Z","iopub.execute_input":"2023-10-19T16:58:37.36067Z","iopub.status.idle":"2023-10-19T16:58:37.369018Z","shell.execute_reply.started":"2023-10-19T16:58:37.360644Z","shell.execute_reply":"2023-10-19T16:58:37.367952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fine-tune the Random Forest model using GridSearchCV\nparam_grid = {\n    'n_estimators': [100, 200, 300],  # You can adjust the number of estimators\n    'max_depth': [None, 10, 20, 30],  # You can adjust the maximum depth\n    # Add more hyperparameters and values as needed\n}","metadata":{"execution":{"iopub.status.busy":"2023-10-19T16:58:37.370549Z","iopub.execute_input":"2023-10-19T16:58:37.370987Z","iopub.status.idle":"2023-10-19T16:58:37.382588Z","shell.execute_reply.started":"2023-10-19T16:58:37.370906Z","shell.execute_reply":"2023-10-19T16:58:37.38157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_search = GridSearchCV(rf_model, param_grid, scoring='balanced_accuracy', cv=5)\ngrid_search.fit(X_train, y_train)\n\nbest_rf_model = grid_search.best_estimator_\nprint(f'Best Random Forest Model: {best_rf_model}')","metadata":{"execution":{"iopub.status.busy":"2023-10-19T16:58:58.871469Z","iopub.execute_input":"2023-10-19T16:58:58.871857Z","iopub.status.idle":"2023-10-19T16:59:21.178884Z","shell.execute_reply.started":"2023-10-19T16:58:58.871826Z","shell.execute_reply":"2023-10-19T16:59:21.177823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import balanced_accuracy_score\n\n# Assuming you have the Random Forest model and validation data as 'best_rf_model' and 'X_valid', 'y_valid'\n\n# Make predictions on the validation set\ny_pred = best_rf_model.predict(X_valid)\n\n# Calculate balanced accuracy\nbalanced_acc = balanced_accuracy_score(y_valid, y_pred)\n\nprint(f'Balanced Accuracy on Validation Set: {balanced_acc:.4f}')","metadata":{"execution":{"iopub.status.busy":"2023-10-19T17:08:28.750072Z","iopub.execute_input":"2023-10-19T17:08:28.750456Z","iopub.status.idle":"2023-10-19T17:08:28.779512Z","shell.execute_reply.started":"2023-10-19T17:08:28.750428Z","shell.execute_reply":"2023-10-19T17:08:28.778293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-10-19T17:09:29.05474Z","iopub.execute_input":"2023-10-19T17:09:29.055084Z","iopub.status.idle":"2023-10-19T17:09:29.082193Z","shell.execute_reply.started":"2023-10-19T17:09:29.055058Z","shell.execute_reply":"2023-10-19T17:09:29.081053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}