{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Description\n### This is an annotated explanation of the probabilistic **FBeta** score evaluation metric that is being used in this competition. I also used a dummy dataset to illustrate the working of this metric. The code for the metric is modified from the implementation present [here](https://www.kaggle.com/code/sohier/probabilistic-f-score).","metadata":{}},{"cell_type":"code","source":"# Imports\nimport numpy as np\nfrom sklearn.datasets import make_classification\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import fbeta_score\nfrom sklearn.model_selection import train_test_split","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-29T10:52:04.334559Z","iopub.execute_input":"2022-11-29T10:52:04.334999Z","iopub.status.idle":"2022-11-29T10:52:04.341283Z","shell.execute_reply.started":"2022-11-29T10:52:04.334959Z","shell.execute_reply":"2022-11-29T10:52:04.339739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pfbeta(labels, predictions, beta):\n    '''\n    This function returns the probablistic fbeta-score.\n    Parameters\n        labels: \n            - The ground truth labels.\n        predictions: \n            - The probability of a certain datapoint belonging to a specific class.\n        beta:\n            - A parameter that determines the weight of recall in the combined score.\n    '''\n    y_true_count = 0\n    \n    # probabilistic true positives\n    probabilistic_tp = 0\n    \n    # probabilistic false positives\n    probabilistic_fp = 0\n    \n    # Loop over every ground truth label.\n    for idx in range(len(labels)):\n        prediction = min(max(predictions[idx], 0), 1) # This line makes sure that the prediction probability is always between 0 and 1.\n        if (labels[idx]):\n            y_true_count += 1\n            probabilistic_tp += prediction\n            probabilistic_fp += 1 - prediction\n        else:\n            probabilistic_fp += prediction\n\n    beta_squared = beta * beta\n    \n    # Probabilistic precision\n    c_precision = probabilistic_tp / (probabilistic_tp + probabilistic_fp)\n    \n    # Probabilistic recall\n    c_recall = probabilistic_tp / y_true_count\n\n    # This part calculates the probabilistic Fbeta-Score\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return result\n    else:\n        return 0","metadata":{"execution":{"iopub.status.busy":"2022-11-29T10:41:12.723488Z","iopub.execute_input":"2022-11-29T10:41:12.723885Z","iopub.status.idle":"2022-11-29T10:41:12.733808Z","shell.execute_reply.started":"2022-11-29T10:41:12.723852Z","shell.execute_reply":"2022-11-29T10:41:12.73246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fitting a logreg model to a dummy dataset\nX, y = make_classification(random_state = 42)\nX_train, X_test, y_train, y_test = train_test_split(X, y, train_size = 0.95, random_state = 42)\nmodel = LogisticRegression(random_state = 42)\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-11-29T10:47:47.773499Z","iopub.execute_input":"2022-11-29T10:47:47.77397Z","iopub.status.idle":"2022-11-29T10:47:47.783541Z","shell.execute_reply.started":"2022-11-29T10:47:47.773934Z","shell.execute_reply":"2022-11-29T10:47:47.782106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The predictions would be passed in as the maximum of probability of both classes.\npredictions = model.predict_proba(X_test).max(axis = 1)\npbeta_score = pfbeta(y_test, predictions, 0.2)\nprint(f\"The probablistic Fbeta score is {pbeta_score}\")","metadata":{"execution":{"iopub.status.busy":"2022-11-29T10:53:37.276915Z","iopub.execute_input":"2022-11-29T10:53:37.277347Z","iopub.status.idle":"2022-11-29T10:53:37.283736Z","shell.execute_reply.started":"2022-11-29T10:53:37.277308Z","shell.execute_reply":"2022-11-29T10:53:37.28269Z"},"trusted":true},"execution_count":null,"outputs":[]}]}