{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"What is the run time for the leaders was the simple question I asked myself.  \n5 hours later this is as close as I got.\n\n\nAssumption - the benchmark score is 0.500.    \nThe only justification is that in past competitions kaggle has often submitted the sample_submission as a simple mean or median. \nSince auc is our metric, guessing it would be median. \nWill do a submission later just for giggles using median of the gold, \nbut I thinking kaggle might have cheated and did the full private test.\n\nMy run time looks about right, but I don't watch it that close.\n\nRun for the 8/28 revision to the list.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\ndef calculate_exact_runtimes(\n    local_filepath=\"/home/james/Documents/full_leaderboard.txt\",\n    kaggle_url=\"/kaggle/input/notebooks/ryanholbrook/rsna-knee-abnormalities-efficiency-lb/full_leaderboard.csv\"\n):\n    # Determine if the script is running inside a Kaggle environment\n    # Kaggle always sets specific environment variables like KAGGLE_KERNEL_RUN_TYPE\n    is_kaggle = os.environ.get('KAGGLE_KERNEL_RUN_TYPE') is not None\n    \n    # Set up the logic to load the file depending on the environment\n    if is_kaggle:\n        print(\"Kaggle environment detected.\")\n        # When running on Kaggle, it is best to attach the efficiency notebook output as a dataset.\n        # This is the standard path structure when you add another notebook's output to your current notebook.\n        kaggle_input_path = \"../input/rsna-knee-abnormalities-efficiency-lb/full_leaderboard.txt\"\n        \n        # Check if the file is attached locally in the Kaggle environment\n        if os.path.exists(kaggle_input_path):\n            print(f\"Reading file from Kaggle input path: {kaggle_input_path}\")\n            df = pd.read_csv(kaggle_input_path)\n        else:\n            # Fallback to the direct URL if the dataset is not attached\n            # Note: Pandas read_csv can sometimes fail on Kaggle URLs if they render as HTML instead of raw text\n            print(f\"Reading file directly from Kaggle URL: {kaggle_url}\")\n            df = pd.read_csv(kaggle_url)\n            \n        # Define where the output file should be saved in Kaggle (must be in /kaggle/working/)\n        output_path = \"/kaggle/working/leaderboard_realistic_runtimes_2.csv\"\n        \n    else:\n        print(\"Local Linux environment detected.\")\n        # Check if the file exists at the specified absolute local path\n        if os.path.exists(local_filepath):\n            print(f\"Reading file from local path: {local_filepath}\")\n            df = pd.read_csv(local_filepath)\n        # Fallback to the current working directory if the absolute path fails\n        elif os.path.exists(\"full_leaderboard.txt\"):\n            print(\"Reading file from current working directory.\")\n            df = pd.read_csv(\"full_leaderboard.txt\")\n        else:\n            raise FileNotFoundError(f\"Could not find the leaderboard file at {local_filepath}\")\n            \n        # Define where the output file should be saved locally\n        output_path = \"/home/james/Documents/leaderboard_realistic_runtimes_2.csv\"\n\n    # Strip any leading or trailing spaces from column names to avoid key errors\n    df.columns = df.columns.str.strip()\n    \n    # Extract the PublicScore column as floating point numbers\n    auc = df[\"PublicScore\"].astype(float).values\n    \n    # Extract the expected ranks as integers\n    expected_ranks = df[\"EfficiencyRank\"].astype(int).values\n    \n    # Get the total number of teams in the dataset\n    n = len(auc)\n    \n    # Dynamically calculate the maximum AUC directly from the current file\n    max_auc = auc.max()\n    \n    # Set the baseline benchmark score for random guessing\n    benchmark = 0.500\n    \n    # Calculate the absolute difference for the denominator\n    denom_abs = max_auc - benchmark \n    \n    # Calculate the actual formula denominator (negative value)\n    denom_formula = benchmark - max_auc \n    \n    # Initialize an array to hold the calculated runtimes (starts with all zeros)\n    runtimes = np.zeros(n)\n    \n    # Define a tiny value to break ties strictly when ranks are different\n    epsilon = 1e-4\n    \n    # Loop through all teams starting from the second one (index 1)\n    for i in range(1, n):\n        # Calculate the mathematical time limit between the current and previous team\n        limit = 32400 * (auc[i] - auc[i-1]) / denom_abs\n        \n        # If the current team has the same rank as the previous team (a tie)\n        if expected_ranks[i] == expected_ranks[i-1]:\n            # Set their runtime to exactly match the limit boundary to keep the score equal\n            runtimes[i] = runtimes[i-1] + limit\n        else:\n            # If they have a different rank, strictly enforce a higher time penalty\n            runtimes[i] = max(0, runtimes[i-1] + limit + epsilon)\n            \n    # Calculate how much we can shift the runtimes without exceeding the 9-hour limit\n    max_shift = 32400 - runtimes.max()\n    \n    # Shift runtimes up by a maximum of 1000 seconds to avoid 0.0 second runtimes\n    shift_amount = min(1000, max_shift - 1)\n    runtimes += shift_amount\n    \n    # Add the used benchmark and max AUC to the dataframe for tracking\n    df[\"Benchmark_Used\"] = benchmark\n    df[\"Max_AUC_Used\"] = max_auc\n    \n    # Round the final estimated runtimes to 1 decimal place\n    df[\"Estimated_Runtime_Sec\"] = np.round(runtimes, 1)\n    \n    # Define a helper function to format seconds into HH:MM:SS.s\n    def format_time(seconds):\n        # Calculate total hours\n        h = int(seconds // 3600)\n        # Calculate remaining minutes\n        m = int((seconds % 3600) // 60)\n        # Calculate remaining seconds\n        s = seconds % 60\n        # Return formatted string\n        return f\"{h:02d}:{m:02d}:{s:04.1f}\"\n        \n    # Apply the formatting function to the runtime column\n    df[\"Estimated_Time\"] = df[\"Estimated_Runtime_Sec\"].apply(format_time)\n    \n    # Recalculate the exact efficiency score using the official formula\n    E_exact = (auc / denom_formula) + (runtimes / 32400.0)\n    \n    # Round to 9 decimal places to prevent float precision errors from breaking ties\n    E_exact_rounded = np.round(E_exact, 9)\n    df[\"Calculated_EfficiencyScore\"] = E_exact_rounded\n    \n    # Rank the recalculated scores, ensuring ties get the same rank (method='min')\n    df[\"Calculated_EfficiencyRank\"] = df[\"Calculated_EfficiencyScore\"].rank(ascending=True, method=\"min\").astype(int)\n    \n    # Store the original ranks for easy comparison\n    df[\"Initial_EfficiencyRank\"] = expected_ranks\n    \n    # Count how many calculated ranks match the original expected ranks\n    match_count = np.sum(df[\"Calculated_EfficiencyRank\"].values == expected_ranks)\n    print(f\"Match Verification: {match_count} / {n} ranks perfectly matched.\")\n    \n    # Define the exact columns we want in our final output file\n    output_columns = [\n        \"Initial_EfficiencyRank\",\n        \"Calculated_EfficiencyRank\",\n        \"TeamName\",\n        \"PublicScore\",\n        \"Calculated_EfficiencyScore\",\n        \"Estimated_Runtime_Sec\",\n        \"Estimated_Time\",\n        \"Benchmark_Used\",\n        \"Max_AUC_Used\"\n    ]\n    \n    # Filter the dataframe to only include the specified output columns\n    df_output = df[[col for col in output_columns if col in df.columns]]\n    \n    # Save the dataframe to the designated environment path without the index column\n    df_output.to_csv(output_path, index=False)\n    print(f\"Results saved to {output_path}\")\n    \n    # Return the final dataframe\n    return df_output\n\nif __name__ == \"__main__\":\n    # Execute the main function\n    results = calculate_exact_runtimes()\n    \n    # Print a preview of the top 15 teams\n    print(\"\\nTop 15 Teams Runtimes:\")\n    preview_cols = [\"Initial_EfficiencyRank\", \"Calculated_EfficiencyRank\", \"TeamName\", \"PublicScore\", \"Estimated_Time\", \"Max_AUC_Used\"]\n    print(results[preview_cols].head(15))","metadata":{},"outputs":[],"execution_count":null}]}