{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaL4","dataSources":[{"sourceId":84795,"databundleVersionId":11281725,"sourceType":"competition"},{"sourceId":221096520,"sourceType":"kernelVersion"},{"sourceId":265863,"sourceType":"modelInstanceVersion","modelInstanceId":227466,"modelId":224053},{"sourceId":265872,"sourceType":"modelInstanceVersion","modelInstanceId":227475,"modelId":224053},{"sourceId":276793,"sourceType":"modelInstanceVersion","modelInstanceId":237027,"modelId":258700}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"References\n\n- https://www.kaggle.com/code/mpware/vllm-0-7 for the current installation script\n- https://www.kaggle.com/code/richolson/ai-math-olympiad-qwen2-5-72b for showing how to submit\n- https://www.kaggle.com/code/abdullahmeda/load-72b-awq-model-using-vllm-on-l4-x4","metadata":{"_uuid":"0861311c-b7fb-488b-9f0a-835cc4d8a6d6","_cell_guid":"bb79fd83-7bbb-46b6-bb3e-d9eb5cca051f","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-06T16:16:02.914987Z","iopub.execute_input":"2025-03-06T16:16:02.9154Z","iopub.status.idle":"2025-03-06T16:16:02.922962Z","shell.execute_reply.started":"2025-03-06T16:16:02.915364Z","shell.execute_reply":"2025-03-06T16:16:02.921271Z"},"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"!cp /kaggle/usr/lib/pip_install_kprize/triton/backends/nvidia/bin/* /usr/local/cuda/bin/","metadata":{"_uuid":"b7c0bb02-b3c7-4f30-a5f0-474b32530294","_cell_guid":"1fe3fc61-2f7d-4816-940f-4cd9715eba6b","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T21:52:49.295841Z","iopub.execute_input":"2025-03-08T21:52:49.296143Z","iopub.status.idle":"2025-03-08T21:52:50.930451Z","shell.execute_reply.started":"2025-03-08T21:52:49.29611Z","shell.execute_reply":"2025-03-08T21:52:50.929046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport time\nimport warnings\n\nimport pandas as pd\nimport polars as pl\nimport numpy as np\n\nimport torch\nimport kaggle_evaluation.konwinski_prize_inference_server\n\nimport io\nfrom typing import List, Tuple, Dict, Optional\n\npd.set_option('display.max_colwidth', None)\nstart_time = time.time()\nallowed_time = [start_time + 60 * 60]  # 60 minute starting grace period, six minute allowance increments","metadata":{"_uuid":"652661b2-a070-446e-92a8-a9e408caa3a1","_cell_guid":"d6ffd952-63f6-47e7-b6c8-207af8b8b4d9","trusted":true,"collapsed":false,"_kg_hide-output":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T21:52:50.931424Z","iopub.execute_input":"2025-03-08T21:52:50.931677Z","iopub.status.idle":"2025-03-08T21:53:18.9537Z","shell.execute_reply.started":"2025-03-08T21:52:50.931655Z","shell.execute_reply":"2025-03-08T21:53:18.952848Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Configs","metadata":{"_uuid":"9ea9fb5a-9762-45d3-9e98-ac3b17197ff7","_cell_guid":"740e8df8-5859-48df-934c-77b5b30c94d4","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Checklist for GPU commits - LLM_SERVER_URL, REQUIRED_SCORE, MAXIMUM_SUBMISSIONS, EVALUATION_COUNT, %%writefile, INTERNET, ACCELERATOR\n\n# MODEL_PATH = '/kaggle/input/deepseek-r1/transformers/deepseek-r1-distill-qwen-14b-awq-casperhansen/1'\n# MODEL_NAME = 'casperhansen/deepseek-r1-distill-qwen-14b-awq'\n# LLM_SERVER_URL = \"https://tonghuikang--example-vllm-openai-compatible-salt1337-14b-serve.modal.run/v1\"\n\nMODEL_PATH = '/kaggle/input/deepseek-r1/transformers/deepseek-r1-distill-qwen-32b-awq-casperhansen/1'\nMODEL_NAME = 'casperhansen/deepseek-r1-distill-qwen-32b-awq'\nLLM_SERVER_URL = \"https://tonghuikang--example-vllm-openai-compatible-salt1337-32b-serve.modal.run/v1\"\n\n# MODEL_PATH = '/kaggle/input/qwq-32b/transformers/qwq-32b-awq/1'\n# MODEL_NAME = 'qwen/qwq-32b-awq'\n# LLM_SERVER_URL = \"https://tonghuikang--example-vllm-openai-compatible-salt1337-qwq-442ea8.modal.run/v1\"\n\nLLM_SERVER_URL = \"http://0.0.0.0:8000/v1\"\n\nN_GPU = 4\nMAX_MODEL_LEN = 8192 * 4\nMAX_NUM_SEQS = 7\n\nBATCH_SIZE: int = 8\nMAX_TOKENS: int = 2048\n\nUSE_REMOTE_LLM: int = bool(LLM_SERVER_URL == \"http://0.0.0.0:8000/v1\")\nREPO_PATH: int = \"repo\"\nBASE_RESULT_SCORE: int = 1000  # so that file names have the same length\nBASE_ACCEPTED_SCORE: int = 5000\nREQUIRED_SCORE: int = 5004  # 5001 or 5004\nPERFECT_SCORE: int = 5004\n\nQUERY_GENERATION_ITERATION_COUNT: int = 4\nPATCH_GENERATION_ITERATION_COUNT: int = 8\nSECONDS_CUTOFF_SHORT_CIRCUIT_ITERATION: int = 6 * 60\n\nMAXIMUM_SUBMISSIONS: int = 1  # overridden to a large number in submissions, ignored during evaluation\nEVALUATION_COUNT: int = 1  # skipped in submissions\nEVALUATION_PROBLEM_INDEX: int = 0\n\nsettings = dict(\n    MODEL_PATH = MODEL_PATH,\n    MODEL_NAME = MODEL_NAME,\n    N_GPU = N_GPU,\n    MAX_MODEL_LEN = MAX_MODEL_LEN,\n    MAX_NUM_SEQS = MAX_NUM_SEQS,\n    LLM_SERVER_URL = LLM_SERVER_URL,\n    USE_REMOTE_LLM = USE_REMOTE_LLM,\n    \n    BATCH_SIZE = BATCH_SIZE,\n    MAX_TOKENS = MAX_TOKENS,\n\n    REPO_PATH = REPO_PATH,\n    BASE_RESULT_SCORE = BASE_RESULT_SCORE,\n    BASE_ACCEPTED_SCORE = BASE_ACCEPTED_SCORE,\n    REQUIRED_SCORE = REQUIRED_SCORE,\n\n    QUERY_GENERATION_ITERATION_COUNT = QUERY_GENERATION_ITERATION_COUNT,\n    PATCH_GENERATION_ITERATION_COUNT = PATCH_GENERATION_ITERATION_COUNT,\n    SECONDS_CUTOFF_SHORT_CIRCUIT_ITERATION = SECONDS_CUTOFF_SHORT_CIRCUIT_ITERATION,\n)\n\nassert LLM_SERVER_URL.endswith(\"/v1\")  # so that I can delete and append /tokenize\n\nimport json\nfrom typing import Union\n\nwith open(\"settings.json\", \"w\") as json_file:\n    json.dump(settings, json_file, indent=4)\n\ndef read_settings() -> dict[str, Union[str, int, bool]]:\n    with open(\"settings.json\", \"r\") as json_file:\n        settings = json.load(json_file)\n    return settings","metadata":{"_uuid":"e04ca1da-05f1-4edc-813c-c06ea451a190","_cell_guid":"b4237a9c-048c-4de2-b2a5-79e9611965f6","trusted":true,"collapsed":false,"papermill":{"duration":0.011949,"end_time":"2024-12-11T03:22:08.838279","exception":false,"start_time":"2024-12-11T03:22:08.82633","status":"completed"},"tags":[],"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T21:53:18.954376Z","iopub.execute_input":"2025-03-08T21:53:18.955049Z","iopub.status.idle":"2025-03-08T21:53:18.962442Z","shell.execute_reply.started":"2025-03-08T21:53:18.955028Z","shell.execute_reply":"2025-03-08T21:53:18.961211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import requests\n\ndef count_tokens(prompt: str, add_special_tokens=True) -> int:\n    payload = {\n        \"model\": MODEL_NAME,\n        \"prompt\": prompt,\n        \"add_special_tokens\": add_special_tokens\n    }\n\n    headers = {\"Content-Type\": \"application/json\"}\n\n    response = requests.post(LLM_SERVER_URL[:-3] + \"/tokenize\", json=payload, headers=headers)\n    \n    if response.status_code == 200:\n        result = response.json()\n        return len(result.get('tokens', []))\n    else:\n        raise ValueError","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T21:53:18.963345Z","iopub.execute_input":"2025-03-08T21:53:18.963629Z","iopub.status.idle":"2025-03-08T21:53:18.99036Z","shell.execute_reply.started":"2025-03-08T21:53:18.963604Z","shell.execute_reply":"2025-03-08T21:53:18.98916Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Communications","metadata":{"_uuid":"df31286c-a7de-483f-a4bb-a9d3ff478796","_cell_guid":"13e618fa-e3a2-4638-beb0-967d9ccdb065","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"question_number = 0\n\ndef write_question(question: str):\n    global question_number\n    question_number += 1\n    with open(\"question_number.txt\", \"w\") as f:\n        f.write(str(question_number))\n    with open(\"question.txt\", \"w\") as f:\n        f.write(question)\n\n\ndef read_answers(suffix: str = \"\") -> list[str]:\n    import glob\n    answers = []\n    for answer_filename in glob.glob(f\"answer_{question_number}_{suffix}*.txt\"):\n        with open(answer_filename) as f:\n            answer = f.read()\n            answers.append(answer)\n    \n    return answers\n\n\ndef delete_answers():\n    import glob\n    import os\n\n    with open(\"question.txt\", \"w\") as f:\n        f.write(\"\")\n    with open(\"question_number.txt\", \"w\") as f:\n        f.write(\"9999\")\n    for file_path in glob.glob(\"answer_*.txt\"):\n        os.remove(file_path)","metadata":{"_uuid":"4817a0b3-8a6e-4e78-9a22-2231a92f4be9","_cell_guid":"55fa0f2a-e9e8-48b6-a609-32581c79aca3","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T21:53:18.991173Z","iopub.execute_input":"2025-03-08T21:53:18.99141Z","iopub.status.idle":"2025-03-08T21:53:19.022274Z","shell.execute_reply.started":"2025-03-08T21:53:18.991386Z","shell.execute_reply":"2025-03-08T21:53:19.020946Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Environment","metadata":{"_uuid":"f38310cd-ab3b-487e-8b31-ef5487e8177c","_cell_guid":"41555dc4-e005-48fc-8380-152976a56841","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import os\n\n# Possible environments\n# - local\n# - Kaggle interactive\n# - Kaggle commit\n# - Kaggle competition (public and private)\n\n\ndef is_on_kaggle() -> bool:\n    return bool(os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\")) or bool(\n        os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\")\n    )\n\n\ndef is_on_kaggle_interactive() -> bool:\n    return os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\") == \"Interactive\" and not bool(os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"))\n\n\ndef is_on_kaggle_commit() -> bool:\n    return (\n        os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\")\n        and not os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\") == \"Interactive\" \n        and not bool(os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"))\n    )\n\n\ndef is_on_kaggle_submission() -> bool:\n    return bool(os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"))","metadata":{"_uuid":"4f076f54-85ca-4b07-a96c-ec9621e01316","_cell_guid":"e099a4d2-57e4-4cd2-8440-cbd8eaead0c3","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T21:53:19.024768Z","iopub.execute_input":"2025-03-08T21:53:19.025038Z","iopub.status.idle":"2025-03-08T21:53:19.050473Z","shell.execute_reply.started":"2025-03-08T21:53:19.025013Z","shell.execute_reply":"2025-03-08T21:53:19.049137Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# vLLM Serving","metadata":{"_uuid":"c8c9c56e-c097-4189-a29a-b5a62caf5895","_cell_guid":"e495fd80-4218-4415-b7ba-968d765bdac7","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"%%writefile vllm_serve.py\n# ----- Configs (shared) -----\n\n\nimport json\nfrom typing import Union\n\n\ndef read_settings() -> dict[str, Union[str, int, bool]]:\n    with open(\"settings.json\", \"r\") as json_file:\n        settings = json.load(json_file)\n    return settings\n\n\nsettings = read_settings()\nMODEL_PATH: str = settings[\"MODEL_PATH\"]  # type: ignore\nMODEL_NAME: str = settings[\"MODEL_NAME\"]  # type: ignore\nN_GPU: int = settings[\"N_GPU\"]  # type: ignore\nMAX_MODEL_LEN: int = settings[\"MAX_MODEL_LEN\"]  # type: ignore\nMAX_NUM_SEQS: int = settings[\"MAX_NUM_SEQS\"]  # type: ignore\nLLM_SERVER_URL: str = settings[\"LLM_SERVER_URL\"]  # type: ignore\nUSE_REMOTE_LLM: bool = settings[\"USE_REMOTE_LLM\"]  # type: ignore\n\nBATCH_SIZE: int = settings[\"BATCH_SIZE\"]  # type: ignore\nMAX_TOKENS: int = settings[\"MAX_TOKENS\"]  # type: ignore\n\nREPO_PATH: str = settings[\"REPO_PATH\"]\nBASE_RESULT_SCORE: int = settings[\"BASE_RESULT_SCORE\"]\nBASE_ACCEPTED_SCORE: int = settings[\"BASE_ACCEPTED_SCORE\"]\n\n\n# ----- Environment (shared) -----\n\n\nimport os\n\n# Possible environments\n# - local\n# - Kaggle interactive\n# - Kaggle commit\n# - Kaggle competition (public and private)\n\n\ndef is_on_kaggle() -> bool:\n    return bool(os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\")) or bool(\n        os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\")\n    )\n\n\ndef is_on_kaggle_interactive() -> bool:\n    return os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\") == \"Interactive\" and not bool(os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"))\n\n\ndef is_on_kaggle_commit() -> bool:\n    return (\n        bool(os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\"))\n        and not os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\") == \"Interactive\" \n        and not bool(os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"))\n    )\n\n\ndef is_on_kaggle_submission() -> bool:\n    return bool(os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"))\n\n\n# ----- Serving -----\n\n\nif __name__ == \"__main__\" and USE_REMOTE_LLM:\n\n    from vllm.entrypoints.openai.cli_args import make_arg_parser\n    from vllm.utils import FlexibleArgumentParser\n\n    serve_parser = FlexibleArgumentParser()\n    serve_parser.add_argument(\"model_tag\", type=str, help=\"The model tag to serve\")\n    serve_parser = make_arg_parser(serve_parser)\n\n    args = serve_parser.parse_args([MODEL_PATH])\n    args.model = MODEL_PATH\n    args.served_model_name = [MODEL_NAME]\n    args.max_model_len = MAX_MODEL_LEN\n    args.max_num_seqs = MAX_NUM_SEQS\n    args.max_seq_len_to_capture = MAX_MODEL_LEN\n    args.tensor_parallel_size = N_GPU\n    args.enable_prefix_caching = True\n    # args.enforce_eager = False\n\n    if is_on_kaggle_submission():\n        args.disable_log_requests = True\n        args.disable_log_stats = True\n\n    import os\n\n    # https://www.kaggle.com/competitions/ai-mathematical-olympiad-progress-prize-2/discussion/560682#3113134\n    os.environ[\"TRITON_PTXAS_PATH\"] = \"/usr/local/cuda/bin/ptxas\"\n    os.environ[\"CUDA_VISIBLE_DEVICES\"] = \"0,1,2,3\"\n    os.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"\n    os.environ[\"VLLM_WORKER_MULTIPROC_METHOD\"] = \"spawn\"\n\n    import uvloop\n    from vllm.entrypoints.openai.api_server import run_server\n\n    uvloop.run(run_server(args))\n\nelse:\n    print(f\"{USE_REMOTE_LLM=}\")","metadata":{"_uuid":"c38b4e3f-87e3-4cfb-a77a-e0c9f67e4984","_cell_guid":"010b48da-22c6-411c-bec6-3d1648a451a6","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T21:53:19.051888Z","iopub.execute_input":"2025-03-08T21:53:19.05218Z","iopub.status.idle":"2025-03-08T21:53:19.07292Z","shell.execute_reply.started":"2025-03-08T21:53:19.052152Z","shell.execute_reply":"2025-03-08T21:53:19.07201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%bash\nnohup python vllm_serve.py > vllm_serve.log 2>&1 &","metadata":{"_uuid":"25647370-0930-41de-9952-b33dc4ec018c","_cell_guid":"ab725a90-a0c9-45fd-ae37-1ec2139369fe","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T21:53:19.073782Z","iopub.execute_input":"2025-03-08T21:53:19.074055Z","iopub.status.idle":"2025-03-08T21:53:19.104717Z","shell.execute_reply.started":"2025-03-08T21:53:19.074028Z","shell.execute_reply":"2025-03-08T21:53:19.103619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"\n\nfrom transformers import AutoTokenizer\n\n# Note: this takes around 25 seconds\ntokenizer = AutoTokenizer.from_pretrained(\n    MODEL_PATH,\n    trust_remote_code=True,\n)\n\n\ndef count_tokens(text: str) -> int:\n    return len(tokenizer.encode(text))","metadata":{"_uuid":"542d846e-8a23-45a4-9097-b4afd1dad8d7","_cell_guid":"b99d1af5-3b11-4c86-be02-b6938c2ac1b1","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T21:53:19.105451Z","iopub.execute_input":"2025-03-08T21:53:19.105714Z","iopub.status.idle":"2025-03-08T21:53:40.390752Z","shell.execute_reply.started":"2025-03-08T21:53:19.105689Z","shell.execute_reply":"2025-03-08T21:53:40.389977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from openai import OpenAI, APIConnectionError\nclient = OpenAI(base_url=LLM_SERVER_URL, api_key=\"aimo\")","metadata":{"_uuid":"4c2804d2-3de0-489a-acc0-5d9930b5d7f8","_cell_guid":"1302198a-c6f4-42f3-8b66-eee4e8e97bcb","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T21:53:40.391428Z","iopub.execute_input":"2025-03-08T21:53:40.391966Z","iopub.status.idle":"2025-03-08T21:53:46.219748Z","shell.execute_reply.started":"2025-03-08T21:53:40.39194Z","shell.execute_reply":"2025-03-08T21:53:46.218759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import time\nprint(\"polling\")\n\npoll_count = 0\nwhile True:\n    poll_count += 1\n    if not is_on_kaggle_submission():\n        if poll_count > 10 * 60:\n            raise\n    try:\n        print(client.models.list())\n        break\n    except APIConnectionError as e:\n        time.sleep(1)\n","metadata":{"_uuid":"d771e9d8-ae02-47b5-83bf-bdbde5c4ff19","_cell_guid":"55c3e1ef-2700-4fd8-9f42-8713a998efa7","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T21:53:46.220441Z","iopub.execute_input":"2025-03-08T21:53:46.220655Z","iopub.status.idle":"2025-03-08T21:53:47.052472Z","shell.execute_reply.started":"2025-03-08T21:53:46.220636Z","shell.execute_reply":"2025-03-08T21:53:47.051037Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Worker","metadata":{"_uuid":"6eb8600c-592d-4a51-a56d-b761f8347b3e","_cell_guid":"99a4da40-24a5-4f96-8a86-953179d55eb6","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"%%writefile worker.py\n# ----- Configs (shared) -----\n\n\nimport json\nfrom typing import Union\n\n\ndef read_settings() -> dict[str, Union[str, int, bool]]:\n    with open(\"settings.json\", \"r\") as json_file:\n        settings = json.load(json_file)\n    return settings\n\n\nsettings = read_settings()\nMODEL_PATH: str = settings[\"MODEL_PATH\"]  # type: ignore\nMODEL_NAME: str = settings[\"MODEL_NAME\"]  # type: ignore\nN_GPU: int = settings[\"N_GPU\"]  # type: ignore\nMAX_MODEL_LEN: int = settings[\"MAX_MODEL_LEN\"]  # type: ignore\nMAX_NUM_SEQS: int = settings[\"MAX_NUM_SEQS\"]  # type: ignore\nLLM_SERVER_URL: str = settings[\"LLM_SERVER_URL\"]  # type: ignore\nUSE_REMOTE_LLM: bool = settings[\"USE_REMOTE_LLM\"]  # type: ignore\n\nBATCH_SIZE: int = settings[\"BATCH_SIZE\"]  # type: ignore\nMAX_TOKENS: int = settings[\"MAX_TOKENS\"]  # type: ignore\n\nREPO_PATH: str = settings[\"REPO_PATH\"]  # type: ignore\nBASE_RESULT_SCORE: int = settings[\"BASE_RESULT_SCORE\"]  # type: ignore\nBASE_ACCEPTED_SCORE: int = settings[\"BASE_ACCEPTED_SCORE\"]  # type: ignore\n\nQUERY_GENERATION_ITERATION_COUNT: int = settings[\"QUERY_GENERATION_ITERATION_COUNT\"]  # type: ignore\nPATCH_GENERATION_ITERATION_COUNT: int = settings[\"PATCH_GENERATION_ITERATION_COUNT\"]  # type: ignore\nSECONDS_CUTOFF_SHORT_CIRCUIT_ITERATION: int = settings[\"SECONDS_CUTOFF_SHORT_CIRCUIT_ITERATION\"]  # type: ignore\n\n# ----- Environment (shared) -----\n\n\nimport os\n\n# Possible environments\n# - local\n# - Kaggle interactive\n# - Kaggle commit\n# - Kaggle competition (public and private)\n\n\ndef is_on_kaggle() -> bool:\n    return bool(os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\")) or bool(\n        os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\")\n    )\n\n\ndef is_on_kaggle_interactive() -> bool:\n    return os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\") == \"Interactive\" and not bool(\n        os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\")\n    )\n\n\ndef is_on_kaggle_commit() -> bool:\n    return (\n        bool(os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\"))\n        and not os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\") == \"Interactive\"\n        and not bool(os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"))\n    )\n\n\ndef is_on_kaggle_submission() -> bool:\n    return bool(os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"))\n\n\n# ----- Communication (shared) -----\n\n\ndef read_question() -> tuple[str, int]:\n    with open(\"question.txt\") as f:\n        question = f.read()\n    with open(\"question_number.txt\") as f:\n        question_number = int(f.read())\n    return question, question_number\n\n\nquestion, question_number = read_question()\n\n\ndef is_still_on_question() -> bool:\n    _, current_question_number = read_question()\n    return current_question_number == question_number\n\n\nimport random\nimport pandas as pd\nimport os\n\nresponse_idx = os.getpid()\n\n\ndef write_answer(answer: str, suffix: str = \"\") -> None:\n    with open(f\"answer_{question_number}_{suffix}_{response_idx}.txt\", \"w\") as f:\n        f.write(answer)\n\n\ndef save_csv(df: pd.DataFrame, suffix: str = \"\") -> None:\n    df.to_csv(f\"logs_{question_number}_{suffix}_{response_idx}.csv\", index=False)\n\n\n# ----- Section divider -----\n\n\nimport requests\n\n\ndef count_tokens(prompt: str, add_special_tokens=True) -> int:\n    payload = {\n        \"model\": MODEL_NAME,\n        \"prompt\": prompt,\n        \"add_special_tokens\": add_special_tokens,\n    }\n\n    headers = {\"Content-Type\": \"application/json\"}\n\n    response = requests.post(\n        LLM_SERVER_URL[:-3] + \"/tokenize\", json=payload, headers=headers\n    )\n\n    if response.status_code == 200:\n        result = response.json()\n        return len(result.get(\"tokens\", []))\n    else:\n        raise ValueError\n\n\n# ----- Section divider -----\n\n\nfrom openai import OpenAI, APIConnectionError\n\nclient = OpenAI(base_url=LLM_SERVER_URL, api_key=\"aimo\")\n\n\n# ----- Section divider -----\n\n\nimport os\nfrom typing import List\n\n\ndef stringify_directory(directory: str) -> str:\n    full_paths: List[str] = []\n    banned_strings = [\".venv\"]\n\n    for root, dirs, files in os.walk(directory):\n        for file in files:\n            if not file.endswith(\".py\"):\n                continue\n            for banned_string in banned_strings:\n                if banned_string in root or banned_string in file:\n                    break\n            else:\n                full_path: str = os.path.join(root, file)\n                full_paths.append(full_path)\n    return \"\\n\".join(full_paths)\n\n\n# ----- Section divider -----\n\n\nimport re\nfrom typing import Dict, List, Optional\n\n\ndef extract_file_query(xml_content: str) -> Dict[str, List[str]]:\n    import xml.etree.ElementTree as ET\n\n    # Prepare a data structure to collect results\n    parsed_data: Dict[str, List[str]] = {}\n    pattern: str = r\"<root>(.*?)</root>\"\n    matches: List[str] = re.findall(pattern, xml_content, re.DOTALL)\n\n    for match in matches[::-1]:\n        try:\n            # Parse the XML\n            root = ET.fromstring(\"<root>\" + match + \"</root>\")\n\n            # Find all <entry> elements\n            for entry in root.findall(\"entry\"):\n                # Extract the <glob_pattern> text\n                glob_pattern = entry.find(\"glob_pattern\")\n                filepath_text: Optional[str] = (\n                    glob_pattern.text.strip()\n                    if glob_pattern is not None and glob_pattern.text is not None\n                    else None\n                )\n\n                if filepath_text is not None:\n                    # Locate <strings_to_search> container\n                    strings_container = entry.find(\"strings_to_search\")\n\n                    # Gather each <string_to_search> text\n                    search_strings: List[str] = []\n                    if strings_container is not None:\n                        for s in strings_container.findall(\"string_to_search\"):\n                            if s.text is not None:\n                                search_strings.append(s.text.strip())\n\n                    # Store in a dictionary: { filepath: [search_strings...] }\n                    parsed_data[filepath_text] = search_strings\n        except:\n            return {}\n        break  # check the last match only\n\n    return parsed_data\n\n\n# ----- Section divider -----\n\n\n# NOTE: <｜begin▁of▁sentence｜> is intentionally omitted - https://github.com/vllm-project/vllm/issues/12985\nselection_prompt: str = (\n    \"\"\"\n<｜User｜>\nYou are a senior software developer tasked with locating all relevant code needed to diagnose and fix a bug. Your goal is to create a comprehensive search strategy that finds all files and code related to the issue.\n\nPROBLEM STATEMENT:\n{problem_statement}\n\nREPOSITORY STRUCTURE:\n<directory>\n{directory_string}\n</directory>\n\nTASK:\nCreate a structured search query to identify ALL code that might be related to this issue. Your search should be:\n1. COMPREHENSIVE - Include all potentially relevant files and code areas\n2. PRECISE - Use specific search terms to minimize irrelevant results\n3. SYSTEMATIC - Cover the full execution path of the affected functionality\n\nREQUIRED OUTPUT FORMAT:\nReturn a structured XML search query following this exact format:\n\n```xml\n<root>\n    <entry>\n        <glob_pattern>GLOB_PATTERN</glob_pattern>  \n        <strings_to_search>\n            <string_to_search>SEARCH_STRING_1</string_to_search>\n            <string_to_search>SEARCH_STRING_2</string_to_search>\n            <!-- More search strings as needed -->\n        </strings_to_search>\n    </entry>\n    <!-- Additional entries as needed -->\n</root>\n```\n\nSEARCH STRATEGY GUIDELINES:\n- MAIN CODE: Search for implementation files related to the affected functionality\n- INTERFACES: Look for related interfaces, base classes, and APIs\n- DEPENDENCIES: Include dependencies that might affect the issue\n- TESTS: Include test files to understand expected behavior\n- ERROR HANDLING: Search for error messages or error handling code\n\nTECHNICAL TIPS:\n- Use precise glob patterns (e.g., \"repo/module/specific_file.py\" or \"repo/module/**/*.py\")\n- For function searches, include parenthesis (e.g., \"function_name(\" instead of just \"function_name\")\n- For class/method searches, include declaration syntax (e.g., \"def method_name\" or \"class ClassName\")\n- For variables, include assignment or usage patterns (e.g., \"variable =\" or \"if variable\")\n- Prioritize unique identifiers over common terms\n- Include error messages, log statements, and unique string literals\n<｜Assistant｜><think>\n\nLet me analyze this problem thoroughly to determine what code is relevant.\n\"\"\".strip()\n)\n\n\nfrom typing import Tuple\n\n\ndef get_selection_query(\n    directory_string: str,\n    problem_statement: str,\n) -> Tuple[str, Dict[str, List[str]], str]:\n\n    prompt_text: str = selection_prompt.format(\n        problem_statement=problem_statement[:20_000],  # estimated 5_000 tokens\n        directory_string=directory_string[:80_000],  # estimated 20_000 tokens\n    )\n\n    print(\n        response_idx,\n        \"get_selection_query input 1\",\n        count_tokens(prompt_text),\n        flush=True,\n    )\n    time_input = time.time()\n    completion = client.completions.create(\n        model=MODEL_NAME,\n        prompt=prompt_text,\n        max_tokens=min(MAX_TOKENS, MAX_MODEL_LEN - count_tokens(prompt_text)),\n        temperature=0.6,\n        stop=\"</think>\",\n    )\n    response_text: str = completion.choices[0].text\n    print(\n        response_idx,\n        \"get_selection_query output 1\",\n        count_tokens(response_text),\n        int(time.time() - time_input),\n        flush=True,\n    )\n\n    prompt_text += response_text + \"</think>\"\n\n    if not is_still_on_question():\n        return prompt_text + response_text, {}, \"\"\n\n    print(\n        response_idx,\n        \"get_selection_query input 1\",\n        count_tokens(prompt_text),\n        flush=True,\n    )\n    time_input = time.time()\n    completion = client.completions.create(\n        model=MODEL_NAME,\n        prompt=prompt_text,\n        max_tokens=min(MAX_TOKENS, MAX_MODEL_LEN - count_tokens(prompt_text)),\n        temperature=0.6,\n        stop=\"<root>\",\n    )\n    response_text: str = completion.choices[0].text\n    print(\n        response_idx,\n        \"get_selection_query output 1\",\n        count_tokens(response_text),\n        int(time.time() - time_input),\n        flush=True,\n    )\n\n    prompt_text += response_text + \"<root>\"\n\n    if not is_still_on_question():\n        return prompt_text + response_text, {}, \"\"\n\n    for iteration in range(QUERY_GENERATION_ITERATION_COUNT):\n        print(\n            response_idx,\n            f\"get_selection_query input 2 {iteration}\",\n            count_tokens(prompt_text),\n            flush=True,\n        )\n        time_input = time.time()\n        completion = client.completions.create(\n            model=MODEL_NAME,\n            prompt=prompt_text,\n            max_tokens=min(MAX_TOKENS, MAX_MODEL_LEN - count_tokens(prompt_text)),\n            temperature=0.6,\n            stop=\"</root>\",\n        )\n        response_text: str = completion.choices[0].text\n        print(\n            response_idx,\n            f\"get_selection_query output 2 {iteration}\",\n            count_tokens(response_text),\n            int(time.time() - time_input),\n            flush=True,\n        )\n\n        response_text += \"</root>\"\n\n        completion_text = prompt_text + response_text\n        select_query: Dict[str, List[str]] = extract_file_query(completion_text)\n        select_content_string, _ = fetch_file_contents(select_query)\n\n        if not is_still_on_question():\n            break\n        if time.time() - start_time > SECONDS_CUTOFF_SHORT_CIRCUIT_ITERATION:\n            break\n        if select_content_string:\n            break\n\n    return completion_text, select_query, select_content_string\n\n\n# ----- Section divider -----\n\n\ndef fetch_file_contents(\n    files_to_search: Dict[str, List[str]], context_lines: int = 12, max_gap: int = 0\n) -> tuple[str, bool]:\n    from io import StringIO\n    from typing import Tuple\n    import glob\n    import math\n\n    def find_matches_in_file(\n        file_path: str, terms: List[str], context_lines: int, add_prefix: bool = False\n    ) -> List[List[Tuple[int, str]]]:\n        \"\"\"Get matching snippets from a single file\"\"\"\n        if not os.path.isfile(file_path):\n            return []\n\n        with open(file_path, \"r\", encoding=\"utf-8\", errors=\"replace\") as f:\n            lines = f.readlines()\n\n        file_snippets = []\n        num_lines = len(lines)\n\n        # Find matches and add context\n        for i, line in enumerate(lines, start=1):\n            if any(term in line for term in terms):\n                start_idx = max(1, i - context_lines)\n                end_idx = min(num_lines, i + context_lines)\n\n                snippet = []\n                for snippet_no in range(start_idx, end_idx + 1):\n                    text = lines[snippet_no - 1].rstrip(\"\\n\")\n                    if add_prefix:\n                        text = f\"[{file_path}] {text}\"\n                    snippet.append((snippet_no, text))\n\n                file_snippets.append(snippet)\n\n        return file_snippets\n\n    def merge_snippets(\n        snippets: List[List[Tuple[int, str]]], gap: int = 0\n    ) -> List[List[Tuple[int, str]]]:\n        \"\"\"Merge overlapping snippets\"\"\"\n        if not snippets:\n            return []\n\n        # Convert to intervals with start, end, and content\n        intervals = [(s[0][0], s[-1][0], s) for s in snippets if s]\n        intervals.sort(key=lambda x: x[0])  # Sort by start line\n\n        merged = []\n        for start, end, snippet in intervals:\n            # First interval or non-overlapping\n            if not merged or start > merged[-1][1] + gap:\n                merged.append((start, end, snippet))\n                continue\n\n            # Merge with previous interval\n            prev_start, prev_end, prev_snippet = merged[-1]\n            merged_end = max(end, prev_end)\n\n            # Combine lines, prioritizing later snippets for duplicates\n            combined = {ln: txt for ln, txt in prev_snippet}\n            for ln, txt in snippet:\n                combined[ln] = txt\n\n            # Create merged snippet and update last entry\n            merged_snippet = [(ln, combined[ln]) for ln in sorted(combined)]\n            merged[-1] = (prev_start, merged_end, merged_snippet)\n\n        # Return just the snippets\n        return [interval[2] for interval in merged]\n\n    # Sort queries by precision (exact file paths first, then wildcards)\n    # and by pattern specificity (more specific patterns first)\n    def query_precision_score(item):\n        pattern, terms = item\n        is_wildcard = \"*\" in pattern or \"?\" in pattern\n        \n        # Calculate search term specificity score (lower is better)\n        # More search terms and longer search terms = more specific query\n        term_count = len(terms)\n        avg_term_length = sum(len(term) for term in terms) / max(1, term_count)\n        term_specificity = -((term_count * 3) + avg_term_length)  # Negative to prioritize higher values\n        \n        # Exact files have highest priority (lowest score is better)\n        if not is_wildcard:\n            return (0, term_specificity, 0)\n            \n        # For wildcards, prioritize by specificity\n        wildcard_count = pattern.count('*') + pattern.count('?')\n        subdirectory_depth = pattern.count('/')\n        \n        # More specific patterns have fewer wildcards and more subdirectories\n        pattern_specificity = wildcard_count - subdirectory_depth\n        \n        return (1, term_specificity, pattern_specificity)\n    \n    # Sort the queries by precision\n    sorted_queries = sorted(files_to_search.items(), key=query_precision_score)\n    \n    # Process all files and collect results\n    all_results = []\n\n    # First, collect all matching snippets for sorted queries\n    for glob_pattern, terms in sorted_queries:\n        is_wildcard = \"*\" in glob_pattern or \"?\" in glob_pattern\n\n        if is_wildcard:\n            # Handle wildcard patterns\n            matching_files = glob.glob(glob_pattern, recursive=True)\n            file_snippets = []\n\n            for file_path in matching_files:\n                # Add file prefix for wildcard matches\n                file_snippets.extend(\n                    find_matches_in_file(\n                        file_path, terms, context_lines, add_prefix=True\n                    )\n                )\n\n            all_results.append(\n                (\n                    glob_pattern,\n                    terms,\n                    is_wildcard,\n                    merge_snippets(file_snippets, max_gap),\n                )\n            )\n        else:\n            # Handle direct file paths\n            snippets = find_matches_in_file(glob_pattern, terms, context_lines)\n            all_results.append(\n                (glob_pattern, terms, is_wildcard, merge_snippets(snippets, max_gap))\n            )\n\n    # Calculate specificity for each result based on number of appearances\n    result_specificity = []\n    for i, (glob_pattern, terms, is_wildcard, snippet_list) in enumerate(all_results):\n        # Count the total number of matches/appearances\n        num_appearances = sum(len(snippet) for snippet in snippet_list)\n        \n        # Calculate term specificity - more specific terms are better\n        term_count = len(terms)\n        avg_term_length = sum(len(term) for term in terms) / max(1, term_count)\n        term_specificity = (term_count * 2) + avg_term_length\n        \n        # Calculate match density - fewer matches is better (more specific)\n        # We use a logarithmic scale to prevent extreme values from dominating\n        if num_appearances > 0:\n            # Lower is better - results with fewer matches are more specific\n            # Use logarithm to dampen the effect of very high match counts\n            match_density = 100 - min(95, 20 * math.log10(num_appearances + 1))\n        else:\n            match_density = 0  # No matches at all\n        \n        # Final specificity score (higher is better)\n        # Prioritize queries with actual matches but fewer occurrences\n        specificity = (\n            100 if num_appearances > 0 else 0,  # First priority: has any matches at all\n            match_density,                      # Second priority: fewer matches is better (more specific)\n            term_specificity                    # Third priority: more search terms is better\n        )\n        \n        result_specificity.append((specificity, i))\n    \n    # Sort results by specificity (highest first)\n    sorted_result_indices = [i for _, i in sorted(result_specificity, reverse=True)]\n    \n    # Format the output\n    output = StringIO()\n    output.write(\"Search Results:\\n\\n\")\n\n    # Show all unique search terms at the top\n    all_terms = []\n    for _, terms, _, _ in all_results:\n        all_terms.extend(terms)\n\n    # Only include terms that resulted in actual matches\n    matching_terms = []\n    for _, terms, _, snippet_list in all_results:\n        if snippet_list:  # If this query found some matches\n            matching_terms.extend(terms)\n    \n    output.write(\"[terms searched]:\\n\")\n    output.write(\"\\n\".join(sorted(set(matching_terms))) + \"\\n\\n\")\n\n    has_any_matches = False\n    current_output_length = output.tell()\n    max_output_length = 80_000\n\n    # Process results in order of specificity\n    for idx in sorted_result_indices:\n        # Stop if we've reached the maximum output length\n        if current_output_length >= max_output_length:\n            output.write(\"\\n[Output truncated due to size limit]\\n\")\n            break\n            \n        glob_pattern, terms, is_wildcard, snippet_list = all_results[idx]\n        \n        # Display file pattern/name header\n        if is_wildcard:\n            header = f\"[file pattern]: {glob_pattern[len(REPO_PATH) + 1:]}\\n\"\n        else:\n            header = f\"[file name]: {glob_pattern[len(REPO_PATH) + 1:]}\\n\"\n            \n        output.write(header)\n        output.write(\"[file content begin]\\n\")\n        current_output_length += len(header) + 19  # +19 for \"[file content begin]\\n\"\n\n        # Handle case with no matches\n        if not snippet_list:\n            if is_wildcard:\n                msg = f\"  No files found matching pattern '{glob_pattern}'.\\n\"\n            elif not os.path.isfile(glob_pattern):\n                msg = f\"  File not found '{glob_pattern}'.\\n\"\n            else:\n                msg = f\"  No matches found in '{glob_pattern}'.\\n\"\n            output.write(msg)\n            current_output_length += len(msg)\n        else:\n            has_any_matches = True\n            # Track seen snippets to avoid duplicates\n            seen_keys = set()\n            for snippet_idx, snippet in enumerate(snippet_list, start=1):\n                # Check if we're about to exceed the limit\n                if current_output_length >= max_output_length:\n                    output.write(\"\\n[Additional matches truncated due to size limit]\\n\")\n                    break\n                    \n                snippet_start: int = snippet[0][0]\n                snippet_end: int = snippet[-1][0]\n\n                # Generate a key to identify this snippet\n                snippet_key = f\"{glob_pattern}_{snippet_start}_{snippet_end}\"\n                if snippet_key in seen_keys:\n                    # Skip this snippet as we've already shown it\n                    continue\n                seen_keys.add(snippet_key)\n\n                # Find context information\n                context_line = None\n                first_match_line = None\n\n                # Check if this is a wildcard result with file path in the text\n                first_line_text = snippet[0][1]\n                if is_wildcard and first_line_text.startswith(\"[\"):\n                    # Extract the file path from the first line if it's a wildcard match\n                    file_indicator = first_line_text.split(\"] \")[0] + \"]\"\n\n                    # Find the first occurrence of any search term\n                    for line_no, text in snippet:\n                        clean_text = text.split(\"] \", 1)[1] if \"] \" in text else text\n                        if any(term in clean_text for term in terms):\n                            first_match_line = line_no\n                            break\n\n                    if first_match_line:\n                        # Extract the match file path from the indicator\n                        match_file = first_line_text.split(\"]\")[0][1:]  # [path] -> path\n\n                        # Try to find context by scanning the original file\n                        try:\n                            if os.path.isfile(match_file):\n                                with open(\n                                    match_file, \"r\", encoding=\"utf-8\", errors=\"replace\"\n                                ) as f:\n                                    file_lines = f.readlines()\n\n                                # Look for class/function definitions before match line\n                                for i in range(first_match_line - 1, -1, -1):\n                                    if i >= len(file_lines):\n                                        continue\n\n                                    line = file_lines[i].strip()\n                                    if line.startswith(\"class \"):\n                                        context_line = line[:50]\n                                        break\n                                    elif (\n                                        line.startswith(\"def \")\n                                        or line.startswith(\"async def \")\n                                    ) and not context_line:\n                                        context_line = line[:50]\n                        except Exception:\n                            pass\n\n                    # If file scanning failed, try to find context within the snippet\n                    if not context_line and first_match_line:\n                        # Find match line indentation\n                        match_indent = None\n                        for line_no, text in snippet:\n                            if line_no == first_match_line:\n                                clean_text = (\n                                    text.split(\"] \", 1)[1] if \"] \" in text else text\n                                )\n                                match_indent = len(clean_text) - len(\n                                    clean_text.lstrip()\n                                )\n                                break\n\n                        if match_indent is not None:\n                            # Find a definition with less indentation\n                            for line_no, text in snippet:\n                                if line_no >= first_match_line:\n                                    continue\n\n                                clean_text = (\n                                    text.split(\"] \", 1)[1] if \"] \" in text else text\n                                )\n                                indent = len(clean_text) - len(clean_text.lstrip())\n                                stripped = clean_text.strip()\n\n                                if indent < match_indent and (\n                                    stripped.startswith(\"class \")\n                                    or stripped.startswith(\"def \")\n                                    or stripped.startswith(\"async def \")\n                                ):\n                                    context_line = stripped[:50]\n                                    break\n\n                        # If still no context, use the match line itself\n                        if not context_line:\n                            for line_no, text in snippet:\n                                if line_no == first_match_line:\n                                    clean_text = (\n                                        text.split(\"] \", 1)[1] if \"] \" in text else text\n                                    )\n                                    context_line = clean_text.strip()[:50]\n                                    break\n\n                    header = f\"\\nMatch #{snippet_idx} in {file_indicator}, lines {snippet_start} to {snippet_end}\"\n                    if context_line:\n                        header += f\": {context_line}\"\n                    output.write(f\"{header}\\n\")\n                    current_output_length += len(header) + 1\n\n                    # Remove the file path prefix from each line\n                    for line_no, text in snippet:\n                        if \"] \" in text:\n                            clean_text = text.split(\"] \", 1)[1]\n                            line_output = f\"  {line_no:3d} | {clean_text}\\n\"\n                        else:\n                            line_output = f\"  {line_no:3d} | {text}\\n\"\n                        output.write(line_output)\n                        current_output_length += len(line_output)\n                        \n                        # Check if we've exceeded the limit during line output\n                        if current_output_length >= max_output_length:\n                            output.write(\"\\n[Additional matches truncated due to size limit]\\n\")\n                            break\n                else:\n                    # Regular single file output\n                    # Find the first occurrence of any search term\n                    for line_no, text in snippet:\n                        if any(term in text for term in terms):\n                            first_match_line = line_no\n                            break\n\n                    if first_match_line:\n                        # Try to find context by scanning the original file\n                        try:\n                            if os.path.isfile(glob_pattern):\n                                with open(\n                                    glob_pattern,\n                                    \"r\",\n                                    encoding=\"utf-8\",\n                                    errors=\"replace\",\n                                ) as f:\n                                    file_lines = f.readlines()\n\n                                # Look for class/function definitions before match line\n                                for i in range(first_match_line - 1, -1, -1):\n                                    if i >= len(file_lines):\n                                        continue\n\n                                    line = file_lines[i].strip()\n                                    if line.startswith(\"class \"):\n                                        context_line = line[:50]\n                                        break\n                                    elif (\n                                        line.startswith(\"def \")\n                                        or line.startswith(\"async def \")\n                                    ) and not context_line:\n                                        context_line = line[:50]\n                        except Exception:\n                            pass\n\n                    # If file scanning failed, try to find context within the snippet\n                    if not context_line and first_match_line:\n                        # Find match line indentation\n                        match_indent = None\n                        for line_no, text in snippet:\n                            if line_no == first_match_line:\n                                match_indent = len(text) - len(text.lstrip())\n                                break\n\n                        if match_indent is not None:\n                            # Find a definition with less indentation\n                            for line_no, text in snippet:\n                                if line_no >= first_match_line:\n                                    continue\n\n                                indent = len(text) - len(text.lstrip())\n                                stripped = text.strip()\n\n                                if indent < match_indent and (\n                                    stripped.startswith(\"class \")\n                                    or stripped.startswith(\"def \")\n                                    or stripped.startswith(\"async def \")\n                                ):\n                                    context_line = stripped[:50]\n                                    break\n\n                        # If still no context, use the match line itself\n                        if not context_line:\n                            for line_no, text in snippet:\n                                if line_no == first_match_line:\n                                    context_line = text.strip()[:50]\n                                    break\n\n                    header = f\"\\nMatch #{snippet_idx}, lines {snippet_start} to {snippet_end}\"\n                    if context_line:\n                        header += f\": {context_line}\"\n                    output.write(f\"{header}\\n\")\n                    current_output_length += len(header) + 1\n\n                    for line_no, text in snippet:\n                        line_output = f\"  {line_no:3d} | {text}\\n\"\n                        output.write(line_output)\n                        current_output_length += len(line_output)\n                        \n                        # Check if we've exceeded the limit during line output\n                        if current_output_length >= max_output_length:\n                            break\n\n                output.write(\"\\n\")\n                current_output_length += 1\n                \n                # Check again if we've exceeded the limit\n                if current_output_length >= max_output_length:\n                    break\n                \n        output.write(\"[file content end]\\n\\n\")\n        current_output_length += 19  # +19 for \"[file content end]\\n\\n\"\n        \n        # Final check for size limit\n        if current_output_length >= max_output_length:\n            break\n\n    # Return the final string and always False for retry flag\n    return output.getvalue() if has_any_matches else \"\", False\n\n\n# ----- Section divider -----\n\n\nimport re\nfrom typing import Optional\n\n\ndef extract_patch_string(text: str) -> Optional[str]:\n    pattern: str = r\"\\n```diff\\n(.*?)\\n```\"\n    matches: List[str] = re.findall(pattern, text, re.DOTALL)\n    if not matches:\n        return None\n    return matches[-1] + \"\\n\"\n\n\n# ----- Section divider -----\n\n\n# NOTE: <｜begin▁of▁sentence｜> is intentionally omitted - https://github.com/vllm-project/vllm/issues/12985\npatching_prompt: str = (\n    \"\"\"\n<｜User｜>\nYou will be writing a patch string in the unified diff format to solve the following problem with the code repository.\n\nThis is how a patch string in the unified diff format looks like. Pay special attention to the line numbers.\n\n```diff\n--- a/first.txt\n+++ b/first.txt\n@@ -1,3 +1,3 @@\n start\n-first change\n+new first change\n middle\n@@ -14,3 +14,3 @@\n some content\n-second change\n+new second change\n more content\n--- a/deep/second.txt\n+++ b/deep/second.txt\n@@ -5,4 +5,7 @@\n line 5 in second.txt\n line 6 in second.txt\n+added line 1 in second.txt\n+added line 2 in second.txt\n+added line 3 in second.txt\n line 7 in second.txt\n line 8 in second.txt\n--- a/deep/directory/third.txt\n+++ b/deep/directory/third.txt\n@@ -4,7 +4,4 @@\n line 4 in third.txt\n line 5 in third.txt\n-deleted line 6 in third.txt\n-deleted line 7 in third.txt\n-deleted line 8 in third.txt\n line 9 in third.txt\n line 10 in third.txt\n--- a/deep/directory/fifth.txt\n+++ b/deep/directory/fifth.txt\n@@ -4,6 +4,8 @@\n line 4 in fifth.txt\n line 5 in fifth.txt\n-removed line one in fifth.txt\n-removed line two in fifth.txt\n+added line one in fifth.txt\n+added line two in fifth.txt\n+added line three in fifth.txt\n+added line four in fifth.txt\n line 6 in fifth.txt\n line 7 in fifth.txt\n@@ -16,7 +18,4 @@\n line 16 in fifth.txt\n line 17 in fifth.txt\n-deleted additional line 18 in fifth.txt\n-deleted additional line 19 in fifth.txt\n-deleted additional line 20 in fifth.txt\n line 21 in fifth.txt\n line 22 in fifth.txt\n```\n\nThis is the problem statement.\n\n<problem_statement>\n\n{problem_statement}\n\n</problem_statement>\n\nThese are the files that are thought to be relevant\n\n<search_results>\n\n{select_content_string}\n\n</search_results>\n\nWrite a git diff within ```diff and ``` that fully fixes the problem.\nOnly edit files that are in the search results.\nDo not edit the test files.\n\n\nReminder\n- Reason with code, citing their line numbers\n- Cite the code you want to change and what it should be changed into.\n- Only change code that are in the search results.\n- Return ONE unified diff with ALL the changes you intend to make.\n- Make sure that the patch string is a valid patch string with the CORRECT context and line numbers.\n- Do not edit the test code.\n<｜Assistant｜><think>\n\"\"\".strip()\n)\n\nimport re\n\n\ndef get_patch_string(\n    problem_statement: str,\n    select_content_string: str,\n) -> Tuple[str, Optional[str]]:\n\n    if select_content_string == \"\":\n        return \"\", None\n\n    prompt_text: str = patching_prompt.format(\n        problem_statement=problem_statement[:20_000],  # estimated 5000 tokens\n        select_content_string=select_content_string[:80_000],  # estimated 20000 tokens\n    )\n\n    print(\n        response_idx,\n        \"get_patch_string input 0\",\n        count_tokens(prompt_text),\n        flush=True,\n    )\n    time_input = time.time()\n    completion = client.completions.create(\n        model=MODEL_NAME,\n        prompt=prompt_text,\n        max_tokens=min(MAX_TOKENS, MAX_MODEL_LEN - count_tokens(prompt_text)),\n        temperature=0.6,\n        stop=\"</think>\",\n    )\n    response_text: str = completion.choices[0].text\n    print(\n        response_idx,\n        \"get_patch_string output 0\",\n        count_tokens(response_text),\n        int(time.time() - time_input),\n        flush=True,\n    )\n\n    prompt_text += response_text + \"</think>\"\n\n    if not is_still_on_question():\n        return prompt_text, None\n\n    print(\n        response_idx,\n        \"get_patch_string input 1\",\n        count_tokens(prompt_text),\n        flush=True,\n    )\n    time_input = time.time()\n    completion = client.completions.create(\n        model=MODEL_NAME,\n        prompt=prompt_text,\n        max_tokens=min(MAX_TOKENS, MAX_MODEL_LEN - count_tokens(prompt_text)),\n        temperature=0.6,\n        stop=\"```diff\",\n    )\n    response_text: str = completion.choices[0].text\n    print(\n        response_idx,\n        \"get_patch_string output 1\",\n        count_tokens(response_text),\n        int(time.time() - time_input),\n        flush=True,\n    )\n\n    prompt_text += response_text + \"```diff\"\n\n    if not is_still_on_question():\n        return prompt_text, None\n\n    for iteration in range(PATCH_GENERATION_ITERATION_COUNT):\n        print(\n            response_idx,\n            f\"get_patch_string input 2 {iteration}\",\n            count_tokens(prompt_text),\n            flush=True,\n        )\n        time_input = time.time()\n        completion = client.completions.create(\n            model=MODEL_NAME,\n            prompt=prompt_text,\n            max_tokens=min(MAX_TOKENS, MAX_MODEL_LEN - count_tokens(prompt_text)),\n            temperature=0.6,\n            stop=\"\\n```\",\n        )\n        response_text: str = completion.choices[0].text\n        print(\n            response_idx,\n            f\"get_patch_string output 2 {iteration}\",\n            count_tokens(response_text),\n            int(time.time() - time_input),\n            flush=True,\n        )\n\n        response_text += \"\\n```\"\n\n        completion_text = prompt_text + response_text\n        patch_string: Optional[str] = extract_patch_string(completion_text)\n\n        if not is_still_on_question():\n            break\n        if time.time() - start_time > SECONDS_CUTOFF_SHORT_CIRCUIT_ITERATION:\n            break\n        if patch_string is None:\n            continue\n        if pass_static_checks(patch_string):\n            break\n\n    return completion_text, patch_string\n\n\n# ----- Section divider -----\n\n\nfrom pathlib import Path\n\n# NOTE: <｜begin▁of▁sentence｜> is intentionally omitted - https://github.com/vllm-project/vllm/issues/12985\nverifying_prompt: str = (\n    \"\"\"\n<｜User｜>\nYou are a CRITICAL CODE REVIEWER with the highest standards of software quality. Your team relies on you to catch ALL issues in patches, no matter how subtle. Your reputation is built on NEVER letting problematic code pass review.\n\nPROBLEM STATEMENT:\n{problem_statement}\n\nRELEVANT CODE CONTEXT:\n{research_content_string}\n\nPROPOSED PATCH:\n{patch_string}\n\nYOUR MISSION:\nConduct an EXTREMELY RIGOROUS review of this patch, actively searching for ANY possible problems. Assume the patch is flawed until proven otherwise. You must identify issues that others would miss.\n\nCOMPREHENSIVE REVIEW CHECKLIST - Scrutinize each of these aspects with extreme skepticism:\n\n1. ROOT CAUSE ANALYSIS:\n   - Does the patch address the TRUE underlying issue or merely mask symptoms?\n   - Is the fix complete or does it handle only certain manifestations of the problem?\n   - Does the patch make incorrect assumptions about the problem's cause?\n\n2. COMPLETENESS CHECK:\n   - Are there unmodified files or code paths that also need changes?\n   - Are there similar patterns elsewhere in the code that remain unfixed?\n   - Does the patch handle ALL variations of the issue?\n\n3. EDGE CASE DETECTION:\n   - Does the patch fail under ANY unusual inputs or conditions?\n   - Are there race conditions, timing issues, or thread-safety concerns?\n   - What happens with null values, empty collections, or boundary conditions?\n   - Are there exception paths or error conditions not adequately handled?\n\n4. REGRESSION RISK:\n   - Could this patch break existing functionality in ANY way?\n   - Does it change behavior that other code might depend on?\n   - Does it modify interfaces, return values, or side effects?\n\n5. CODE CORRECTNESS:\n   - Are there ANY logical errors in the implementation?\n   - Could the patch introduce off-by-one errors, boundary errors, or indexing issues?\n   - Are there potential null pointer or reference exceptions?\n   - Does the patch handle all error conditions?\n\n6. ROBUSTNESS CONCERNS:\n   - Does the patch fail to handle invalid input?\n   - Is there missing validation or sanitization?\n   - Could the patch lead to resource leaks, memory issues, or performance bottlenecks?\n\nREQUIRED FORMAT:\nDocument ALL issues, concerns, and potential improvements, no matter how small.\nBE EXTREMELY CRITICAL - assume the patch contains flaws and actively search for them.\nYour final verdict MUST be ONE of these exact sentences as the last line:\n- Yes, I found a mistake in the patch.\n- No, I did not find a mistake in the patch.\n\nEVALUATION CRITERIA - The patch fails review if ANY of these apply:\n✘ Does not completely fix the issue\n✘ Leaves similar issues unfixed elsewhere\n✘ Fails in any edge cases\n✘ Could cause regressions or side effects\n✘ Contains any logical errors\n✘ Has inadequate error handling\n✘ Violates coding standards\n✘ Has performance or security implications\n✘ Does not follow best practices\n✘ Is in any way incomplete or suboptimal\n\nYour default position should be to REJECT the patch unless you are 100% confident it is flawless. The cost of letting a bad patch through is much higher than rejecting a good one.\n<｜Assistant｜><think>\n\"\"\".strip()\n)\n\n\n\nfrom functools import cache\nimport unidiff\nfrom unidiff.errors import UnidiffParseError\n\n\n@cache\ndef is_valid_patch_format(patch_string: str) -> tuple[bool, str]:\n    \"\"\"\n    A quick check to confirm if a patch could be valid.\n    \"\"\"\n    if not isinstance(patch_string, str):\n        raise\n    try:\n        patch_set = unidiff.PatchSet(patch_string)\n        if len(patch_set) == 0:\n            return False, \"Patch is empty\"\n    except UnidiffParseError as e:\n        return False, repr(e)\n    except ImportError:\n        raise\n    except Exception as e:\n        print(\"uncaught error in is_valid_patch_format\", repr(e))\n        return False, repr(e)\n    return True, \"\"\n\n\nfrom functools import cache\nfrom unidiff import PatchSet\n\n\n@cache\ndef is_patch_noop(patch_string: str) -> bool:\n    patch = PatchSet(patch_string)\n\n    for patched_file in patch:\n        for hunk in patched_file:\n            removed_lines = [line.value for line in hunk if line.is_removed]\n            added_lines = [line.value for line in hunk if line.is_added]\n\n            # If the number of removed and added lines differ, it's definitely a change\n            if len(removed_lines) != len(added_lines):\n                return False\n\n            # Check line-by-line if added lines exactly match removed lines\n            for removed, added in zip(removed_lines, added_lines):\n                if removed != added:\n                    return False\n\n    # If we reach here, no actual changes were introduced\n    return True\n\n\nimport subprocess\nimport tempfile\nfrom functools import cache\n\n\n@cache\ndef patch_dry_run_succeeds(\n    patch_string: str, repo_path: str = REPO_PATH, timeout: int = 60\n) -> tuple[bool, str]:\n    \"\"\"\n    A robust check if the patch will proceed without any errors.\n    Should be run after `is_valid_patch_format()`: the patch\n    command can hang if the inputs are sufficiently invalid.\n\n    Args:\n        patch_path: Path to a file containing the patch.\n        repo_path: Path to the directory to be patched.\n        timeout: Number of seconds before the dry run will be cancelled.\n    \"\"\"\n    with tempfile.TemporaryDirectory() as tmpdir:\n        patch_path = f\"{tmpdir}/patch.txt\"\n        with open(patch_path, \"w\") as f:\n            f.write(patch_string)\n\n        cmd = f\"patch --verbose --dry-run -p1 -i {patch_path} -d {repo_path} 2>&1\"\n        try:\n            output = subprocess.run(\n                cmd, capture_output=True, shell=True, check=True, timeout=timeout\n            )\n            return True, str(output.stdout.decode(\"utf-8\"))\n        except subprocess.CalledProcessError as e:\n            return False, str(e.stdout.decode(\"utf-8\"))\n        except subprocess.TimeoutExpired as e:\n            return False, \"Timed out\"\n\n\nimport subprocess\nimport tempfile\nimport shutil\nimport os\nimport ast\nfrom typing import Tuple, Dict, Any\n\n\n@cache\ndef pass_static_analysis(\n    patch_string: str, repo_dir: str = REPO_PATH\n) -> Tuple[bool, Dict[str, Any]]:\n    \"\"\"\n    Apply a patch represented as a string to a temporary copy of the Python project\n    and check if the patched files contain syntax errors.\n\n    Parameters:\n        patch_string (str): The patch content in unified diff format.\n        repo_dir (str): Path to the repository directory containing Python files.\n\n    Returns:\n        Tuple[bool, Dict[str, Any]]:\n            - False if syntax errors occurred, True otherwise.\n            - Dict containing detailed syntax error information if errors occurred.\n    \"\"\"\n    syntax_errors: Dict[str, Any] = {}\n\n    # Get modified files using PatchSet\n    patch_set = PatchSet(patch_string)\n    modified_files = [\n        os.path.join(patched_file.target_file[2:]) for patched_file in patch_set\n    ]\n    modified_python_files = [f for f in modified_files if f.endswith(\".py\")]\n\n    with tempfile.TemporaryDirectory() as temp_dir:\n        temp_repo_dir = os.path.join(temp_dir, \"repo_copy\")\n        shutil.copytree(repo_dir, temp_repo_dir)\n\n        # Apply patch from string via subprocess and stdin\n        try:\n            process = subprocess.run(\n                [\"patch\", \"-p1\"],\n                input=patch_string,\n                cwd=temp_repo_dir,\n                check=True,\n                stdout=subprocess.PIPE,\n                stderr=subprocess.PIPE,\n                text=True,\n            )\n        except subprocess.CalledProcessError as e:\n            error_message = f\"Failed to apply patch:\\n{e.stderr}\"\n            return False, {\"patch_error\": error_message}\n\n        # Check syntax of only modified Python files\n        for py_file in modified_python_files:\n            file_path = os.path.join(temp_repo_dir, py_file)\n            relative_path = os.path.relpath(file_path, temp_repo_dir)\n\n            if os.path.exists(file_path):\n                with open(file_path, \"r\", encoding=\"utf-8\") as f:\n                    source = f.read()\n                try:\n                    ast.parse(source, filename=relative_path)\n                except SyntaxError as se:\n                    syntax_errors[relative_path] = {\n                        \"error\": str(se),\n                        \"lineno\": se.lineno,\n                        \"offset\": se.offset,\n                        \"text\": se.text.strip() if se.text else \"\",\n                    }\n\n    if syntax_errors:\n        return False, syntax_errors\n\n    return True, {}\n\n\n@cache\ndef pass_static_checks(patch_string: str, repo_dir: str = REPO_PATH) -> bool:\n    valid_patch_format, _ = is_valid_patch_format(patch_string)\n    if valid_patch_format is False:\n        return False\n    if is_patch_noop(patch_string):\n        return False\n    patch_success, _ = patch_dry_run_succeeds(patch_string, repo_dir)\n    if not patch_success:\n        return False\n    static_success, _ = pass_static_analysis(patch_string, repo_dir)\n    if not static_success:\n        return False\n    return True\n\n\n# ----- Section divider -----\n\n\n# NOTE: <｜begin▁of▁sentence｜> is intentionally omitted - https://github.com/vllm-project/vllm/issues/12985\nresearch_prompt: str = (\n    \"\"\"\n<｜User｜>\nYou are a FORENSIC CODE AUDITOR specializing in finding subtle bugs and potential regressions. You have been called in to exhaustively validate a critical patch. Your mission is to gather ALL necessary code context to perform the most thorough review possible.\n\nPROBLEM STATEMENT:\n{problem_statement}\n\nPROPOSED PATCH:\n{patch_string}\n\nREPOSITORY STRUCTURE:\n<directory>\n{directory_string}\n</directory>\n\nCRITICAL MISSION:\nDevelop an ULTRA-COMPREHENSIVE search strategy to locate EVERY piece of related code needed to verify that this patch:\n1. COMPLETELY addresses the root cause from ALL angles\n2. Will NOT introduce ANY regressions or side effects\n3. Handles EVERY possible edge case and exception path\n4. Fixes ALL affected code paths, not just the obvious ones\n5. Doesn't overlook SIMILAR issues elsewhere in the codebase\n\nCOMPREHENSIVE SEARCH REQUIREMENTS:\nYour search MUST include ALL of the following categories of code:\n\nCORE ISSUE VERIFICATION:\n- EXACT locations of modified code plus 30+ lines of context\n- ALL implementations of functions/methods being modified\n- EVERY call site where modified code is invoked\n- ALL code that depends on the modified behavior\n\nREGRESSION PREVENTION:\n- EVERY file with similar patterns or functionality\n- ALL possible execution paths through modified code\n- EVERY test case that exercises related functionality\n- ANY code that might be affected by changed behavior\n\nEDGE CASE DETECTION:\n- FULL error handling and exception paths\n- ALL boundary condition checks\n- EVERY validation routine for inputs/outputs\n- ANY special cases or conditional logic\n\nARCHITECTURAL UNDERSTANDING:\n- COMPLETE class/interface hierarchies\n- ALL dependencies and dependent modules\n- EVERY configuration that affects behavior\n- FULL initialization and setup code\n\nREQUIRED OUTPUT FORMAT:\nReturn a structured XML search query following this EXACT format:\n\n```xml\n<root>\n    <entry>\n        <glob_pattern>PRECISE_FILE_PATH</glob_pattern>  \n        <strings_to_search>\n            <string_to_search>SPECIFIC_STRING_1</string_to_search>\n            <string_to_search>SPECIFIC_STRING_2</string_to_search>\n            <!-- More strings as needed -->\n        </strings_to_search>\n    </entry>\n    <!-- More entries -->\n</root>\n```\n\nSEARCH STRATEGY GUIDANCE:\n- BE EXHAUSTIVE - Your search must be maximally comprehensive\n- BE SPECIFIC - Use precise patterns to find exactly what's needed\n- BE SYSTEMATIC - Cover all angles methodically and thoroughly\n- ASSUME INCOMPLETENESS - Assume the patch might miss important cases\n- PRIORITIZE COMPLETENESS OVER BREVITY - Better to return too much than too little\n\nTECHNICAL SEARCH TIPS:\n- For functions/methods: Include declaration patterns and call sites (\" function_name(\" or \"def function_name\")\n- For classes: Search for full hierarchy (\"class ClassName\", \"inherit\", \"super()\")\n- For variables: Include declaration, assignment, and usage patterns\n- For interfaces: Find all implementations and call points\n- For error handling: Search for exception types, try/except blocks, and error messages\n- For tests: Find all test files and test methods related to the functionality\n\nThe success of the code review ENTIRELY depends on your ability to find ALL relevant code. Your search strategy should be BOTH exhaustive and precise.\n<｜Assistant｜><think>\n\nI must create the most comprehensive search strategy possible to validate this patch thoroughly. Let me analyze the problem and patch in detail\n\"\"\".strip()\n)\n\n\nfrom typing import Tuple\n\n\ndef get_research_query(\n    directory_string: str,\n    problem_statement: str,\n    patch_string: str,\n) -> Tuple[str, Dict[str, List[str]], str]:\n\n    prompt_text: str = research_prompt.format(\n        problem_statement=problem_statement[:20_000],  # estimated 5_000 tokens\n        directory_string=directory_string[:60_000],  # estimated 15_000 tokens\n        patch_string=patch_string,\n    )\n\n    print(\n        response_idx,\n        \"get_research_query input 1\",\n        count_tokens(prompt_text),\n        flush=True,\n    )\n    time_input = time.time()\n    completion = client.completions.create(\n        model=MODEL_NAME,\n        prompt=prompt_text,\n        max_tokens=min(MAX_TOKENS, MAX_MODEL_LEN - count_tokens(prompt_text)),\n        temperature=0.6,\n        stop=\"</think>\",\n    )\n    response_text: str = completion.choices[0].text\n    print(\n        response_idx,\n        \"get_research_query output 1\",\n        count_tokens(response_text),\n        int(time.time() - time_input),\n        flush=True,\n    )\n\n    prompt_text += response_text + \"</think>\"\n\n    if not is_still_on_question():\n        return prompt_text + response_text, {}, \"\"\n\n    print(\n        response_idx,\n        \"get_research_query input 1\",\n        count_tokens(prompt_text),\n        flush=True,\n    )\n    time_input = time.time()\n    completion = client.completions.create(\n        model=MODEL_NAME,\n        prompt=prompt_text,\n        max_tokens=min(MAX_TOKENS, MAX_MODEL_LEN - count_tokens(prompt_text)),\n        temperature=0.6,\n        stop=\"<root>\",\n    )\n    response_text: str = completion.choices[0].text\n    print(\n        response_idx,\n        \"get_research_query output 1\",\n        count_tokens(response_text),\n        int(time.time() - time_input),\n        flush=True,\n    )\n\n    prompt_text += response_text + \"<root>\"\n\n    if not is_still_on_question():\n        return prompt_text + response_text, {}, \"\"\n\n    for iteration in range(QUERY_GENERATION_ITERATION_COUNT):\n        print(\n            response_idx,\n            f\"get_research_query input 2 {iteration}\",\n            count_tokens(prompt_text),\n            flush=True,\n        )\n        time_input = time.time()\n        completion = client.completions.create(\n            model=MODEL_NAME,\n            prompt=prompt_text,\n            max_tokens=min(MAX_TOKENS, MAX_MODEL_LEN - count_tokens(prompt_text)),\n            temperature=0.6,\n            stop=\"</root>\",\n        )\n        response_text: str = completion.choices[0].text\n        print(\n            response_idx,\n            f\"get_research_query output 2 {iteration}\",\n            count_tokens(response_text),\n            int(time.time() - time_input),\n            flush=True,\n        )\n\n        response_text += \"</root>\"\n\n        completion_text = prompt_text + response_text\n        research_query: Dict[str, List[str]] = extract_file_query(completion_text)\n        research_content_string, _ = fetch_file_contents(research_query)\n\n        if not is_still_on_question():\n            break\n        if time.time() - start_time > SECONDS_CUTOFF_SHORT_CIRCUIT_ITERATION:\n            break\n        if research_content_string:\n            break\n\n    return completion_text, research_query, research_content_string\n\n\n# ----- Section divider -----\n\n\ndef get_verification(\n    problem_statement: str,\n    research_content_string: str,\n    patch_string: Optional[str],\n) -> Tuple[str, bool]:\n\n    prompt_text = verifying_prompt.format(\n        problem_statement=problem_statement[:20_000],  # estimated 5_000 tokens\n        research_content_string=research_content_string[:80_000],  # estimated 20_000 tokens\n        patch_string=patch_string,\n    )\n\n    print(\n        response_idx,\n        \"get_verification input\",\n        count_tokens(prompt_text),\n        flush=True,\n    )\n    time_input = time.time()\n    completion = client.completions.create(\n        model=MODEL_NAME,\n        prompt=prompt_text,\n        max_tokens=min(MAX_TOKENS, MAX_MODEL_LEN - count_tokens(prompt_text)),\n        temperature=0.6,\n    )\n    response_text = completion.choices[0].text\n    print(\n        response_idx,\n        \"get_verification output\",\n        count_tokens(response_text),\n        int(time.time() - time_input),\n        flush=True,\n    )\n\n    completion_text = prompt_text + response_text\n    \n    # Check the last occurrence of yes/no to determine if mistakes were found\n    yes_index = (\"yes\" + response_text).lower().rindex(\"yes\")\n    no_index = (\"no\" + response_text).lower().rindex(\"no\")\n    judgment = no_index > yes_index  # True if \"no, I did not find a mistake\" appears later\n\n    return completion_text, judgment\n\n\n# ----- Section divider -----\n\n\nimport time\nimport json\n\nstart_time = time.time()\n\nproblem_statement, question_number = read_question()\n\ndirectory: str = REPO_PATH\n\nis_valid_patch_format.cache_clear()\nis_patch_noop.cache_clear()\npatch_dry_run_succeeds.cache_clear()\npass_static_analysis.cache_clear()\npass_static_checks.cache_clear()\n\ndata = {}\npatch_string = None\n\ndirectory_string = stringify_directory(directory)\ndata[\"problem_statement\"] = problem_statement\ndata[\"question_number\"] = question_number\ndata[\"score\"] = BASE_RESULT_SCORE\n\nfor _ in range(1):\n    if not __name__ == \"__main__\":\n        break\n\n    # SELECT\n\n    data[\"score\"] = BASE_RESULT_SCORE + 1\n    selection_completion_text, select_query, select_content_string = (\n        get_selection_query(\n            directory_string,\n            problem_statement,\n        )\n    )\n    data[f\"selection_completion_text\"] = selection_completion_text\n    data[f\"selection_completion_length\"] = count_tokens(selection_completion_text)\n    data[f\"select_query\"] = json.dumps(\n        select_query, indent=4, default=str, sort_keys=True\n    )\n    print(\n        response_idx,\n        \"select_query\",\n        select_query,\n        flush=True,\n    )\n\n    if not is_still_on_question():\n        break\n    if not select_query:\n        break\n\n    data[\"score\"] = BASE_RESULT_SCORE + 2\n    data[\"select_content_string\"] = select_content_string\n\n    if not is_still_on_question():\n        break\n    if not select_content_string:\n        break\n\n    # PATCH\n\n    data[\"score\"] = BASE_RESULT_SCORE + 3\n    patch_completion_text, patch_string = get_patch_string(\n        problem_statement, select_content_string\n    )\n    data[f\"patch_completion_text\"] = patch_completion_text\n    data[f\"patch_completion_length\"] = count_tokens(patch_completion_text)\n    data[f\"patch_string\"] = patch_string\n\n    if not is_still_on_question():\n        break\n    if patch_string is None:\n        break\n\n    data[\"score\"] = BASE_RESULT_SCORE + 4\n    valid_patch_format, patch_error = is_valid_patch_format(patch_string)\n    if valid_patch_format is False:\n        data[f\"patch_error\"] = patch_error\n        print(\n            response_idx,\n            \"patch_string fail is_valid_patch_format\",\n            flush=True,\n        )\n        break\n\n    data[\"score\"] = BASE_RESULT_SCORE + 5\n    if is_patch_noop(patch_string):\n        data[f\"patch_error\"] = \"patch_noop\"\n        print(\n            response_idx,\n            \"patch_string fail is_patch_noop\",\n            flush=True,\n        )\n        break\n\n    data[\"score\"] = BASE_RESULT_SCORE + 6\n    patch_success, patch_error = patch_dry_run_succeeds(patch_string, directory)\n    if not patch_success:\n        data[f\"patch_error\"] = patch_error\n        print(\n            response_idx,\n            \"patch_string fail patch_dry_run_succeeds\",\n            flush=True,\n        )\n        break\n\n    data[\"score\"] = BASE_RESULT_SCORE + 7\n    static_success, patch_error = pass_static_analysis(patch_string, directory)\n    if not static_success:\n        data[f\"patch_error\"] = str(patch_error)\n        print(\n            response_idx,\n            \"patch_string fail patch_dry_run_succeeds\",\n            flush=True,\n        )\n        break\n\n    print(\n        response_idx,\n        \"patch_string pass statics\",\n        flush=True,\n    )\n\n    # VERIFY\n\n    data[\"score\"] = BASE_ACCEPTED_SCORE\n    verification_completion_text, judgment = get_verification(\n        problem_statement, select_content_string, patch_string\n    )\n\n    data[\"verification_completion_text\"] = verification_completion_text\n    data[\"verification_completion_length\"] = count_tokens(verification_completion_text)\n    data[\"judgment\"] = judgment\n    print(\n        response_idx,\n        \"judgment\",\n        judgment,\n        flush=True,\n    )\n\n    if not is_still_on_question():\n        break\n    if judgment is False:\n        break\n\n    # RESEARCH\n\n    data[\"score\"] = BASE_ACCEPTED_SCORE + 1\n    # select_content_string is intended not be an input\n    research_completion_text, research_query, research_content_string = (\n        get_research_query(directory_string, problem_statement, patch_string)\n    )\n    data[f\"research_completion_text\"] = research_completion_text\n    data[f\"research_completion_length\"] = count_tokens(research_completion_text)\n    data[f\"research_query\"] = json.dumps(\n        research_query, indent=4, default=str, sort_keys=True\n    )\n    print(\n        response_idx,\n        \"research_query\",\n        research_query,\n        flush=True,\n    )\n\n    if not is_still_on_question():\n        break\n    if not research_query:\n        break\n\n    data[\"score\"] = BASE_ACCEPTED_SCORE + 2\n    data[\"research_content_string\"] = research_content_string\n\n    if not is_still_on_question():\n        break\n    if not research_content_string:\n        break\n\n    # REVERIFY\n\n    data[\"score\"] = BASE_ACCEPTED_SCORE + 3\n    reverification_completion_text, rejudgment = get_verification(\n        problem_statement, research_content_string, patch_string\n    )\n\n    data[\"reverification_completion_text\"] = reverification_completion_text\n    data[\"reverification_completion_length\"] = count_tokens(\n        reverification_completion_text\n    )\n    data[\"rejudgment\"] = rejudgment\n    print(\n        response_idx,\n        \"rejudgment\",\n        rejudgment,\n        flush=True,\n    )\n\n    if rejudgment is False:\n        break\n\n    data[\"score\"] = BASE_ACCEPTED_SCORE + 4\n\n\ndata[\"duration\"] = time.time() - start_time\n\nif not is_on_kaggle_submission():\n    save_csv(pd.DataFrame([data]), suffix=str(data[\"score\"]))\n\nif patch_string is None:\n    patch_string = \"\"\n\nif is_still_on_question():\n    write_answer(patch_string, suffix=str(data[\"score\"]))\n","metadata":{"_uuid":"90d9ae39-8a8b-4b70-944c-c82d8e683355","_cell_guid":"63af6a74-12a2-4a70-8dc4-9be56415118c","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-09T01:50:16.06083Z","iopub.execute_input":"2025-03-09T01:50:16.061391Z","iopub.status.idle":"2025-03-09T01:50:16.082637Z","shell.execute_reply.started":"2025-03-09T01:50:16.061342Z","shell.execute_reply":"2025-03-09T01:50:16.081171Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Setup","metadata":{}},{"cell_type":"code","source":"import sys\nimport subprocess\n\ndef start_execution() -> None:\n    subprocess.Popen([sys.executable, \"worker.py\"], bufsize=1)\n\n\nimport io\nimport shutil\nimport subprocess\n\n\ndef setup(\n    problem_statement: str,\n    repo_archive: io.BytesIO,\n    pip_packages_archive: io.BytesIO,\n    env_setup_cmds_templates: list[str],\n    repo_path: str = REPO_PATH,\n) -> None:\n    delete_answers()\n    write_question(problem_statement)\n\n    if not os.path.exists(repo_path):\n        os.makedirs(repo_path)\n    \n    \"\"\"Replace this function with your inference code.\n    Args:\n        problem_statement: The text of the git issue.\n        repo_path: A BytesIO buffer path with a .tar containing the codebase that must be patched. The gateway will make this directory available immediately before this function runs.\n        pip_packages_archive: A BytesIO buffer path with a .tar containing the wheel files necessary for running unit tests.\n        env_setup_cmds_templates: Commands necessary for installing the pip_packages_archive.\n    \"\"\"\n\n    # Unpack the codebase to be patched into a directory that won't be exported when\n    # the notebook is saved.\n    archive_path = \"/tmp/repo_archive.tar\"\n    with open(archive_path, \"wb\") as f:\n        f.write(repo_archive.read())\n    if os.path.exists(repo_path):\n        shutil.rmtree(repo_path)\n    shutil.unpack_archive(archive_path, extract_dir=repo_path)\n    os.remove(archive_path)\n\n    \"\"\"\n    Unpack pip_packages if you want to run unit tests on your patch.\n    Note that editing unit tests with your patch -- even to add valid tests -- can cause your submission to be flagged as a failure.\n    Most of the relevant repos use pytest for running tests. You will almost certainly need to run only a subset of the unit tests to avoid running out of inference time.\n    \"\"\"\n    pip_archive_dir = \"/tmp/pip_packages_archive.tar\"\n    with open(pip_archive_dir, \"wb\") as f:\n        f.write(pip_packages_archive.read())\n    pip_packages_path = \"/tmp/path/to/pip_packages\"\n    if os.path.exists(pip_packages_path):\n        shutil.rmtree(pip_packages_path)\n    shutil.unpack_archive(pip_archive_dir, extract_dir=pip_packages_path)\n    os.remove(pip_archive_dir)\n\n    # Get env setup cmds by setting the pip_packages_path\n    env_setup_cmds = [\n        cmd.format(pip_packages_path=pip_packages_path)\n        for cmd in env_setup_cmds_templates\n    ]\n\n    # # Run env setup for the repo\n    # subprocess.run(\n    #     \"\\n\".join(env_setup_cmds),\n    #     shell=True,\n    #     executable=\"/bin/bash\",\n    #     cwd=repo_path,\n    # )\n\n\ndef teardown(repo_path: str = REPO_PATH) -> None:\n    if os.path.exists(repo_path):\n        shutil.rmtree(repo_path)\n    delete_answers()","metadata":{"_uuid":"0ac628c6-5e5d-422d-bf21-2d9aa2d7e297","_cell_guid":"e04340e3-e9aa-483a-8868-6e66a410d6fe","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T22:32:19.498979Z","iopub.execute_input":"2025-03-08T22:32:19.499243Z","iopub.status.idle":"2025-03-08T22:32:19.506691Z","shell.execute_reply.started":"2025-03-08T22:32:19.499226Z","shell.execute_reply":"2025-03-08T22:32:19.505712Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Predict function","metadata":{}},{"cell_type":"code","source":"import io\nimport time\n\n\nsubmissions_left = MAXIMUM_SUBMISSIONS\nif is_on_kaggle_submission():\n    submissions_left = 1000\n\n\n\ndef predict(\n    problem_statement: str,\n    repo_archive: io.BytesIO,\n    pip_packages_archive: io.BytesIO,\n    env_setup_cmds_templates: List[str],\n) -> Optional[str]:\n    \"\"\"Replace this function with your inference code.\n    Args:\n        problem_statement: The text of the git issue.\n        repo_archive: A BytesIO buffer path with a .tar containing the codebase that must be patched. The gateway will make this directory available immediately before this function runs.\n    \"\"\"\n    allowed_time[-1] += 6 * 60\n    if time.time() > allowed_time[-1]:\n        return None\n\n    global submissions_left\n    if submissions_left == 0:\n        return None\n\n    setup(problem_statement, repo_archive, pip_packages_archive, env_setup_cmds_templates)\n\n    ##### CALL BEGINS #####\n    prediction_start_time = time.time()\n    print(\"starting prediction\")\n\n    for _ in range(BATCH_SIZE):\n        start_execution()\n\n    num_minutes = 12\n    seconds_interval = 20\n    for interval_count in range(60 // seconds_interval * num_minutes + 1):\n        answers = read_answers(suffix=str(PERFECT_SCORE))\n        if len(answers) >= 1:\n            print(f\"found patch deemed perfect\")\n            break\n        answers = read_answers()\n        if len(answers) >= BATCH_SIZE:\n            print(f\"found {len(answers)} answers\")\n            break\n        if len(answers) >= BATCH_SIZE - 1 and interval_count * seconds_interval >= 9 * 60:\n            # avoid waiting for the one last very slow thread\n            # but this might mean we miss out on the judgment / rejudgment\n            print(f\"found {len(answers)} answers, skipping last\")\n            break\n        # NOTE: not checking time.time() > allowed_time[-1] here\n        # We depend on the num_minutes limit instead\n        # We rather attempt one question in full and skip the other question\n        # than to half-answer two questions\n        print(f\"current answers length {len(answers)}\")\n        time.sleep(seconds_interval)\n    else:\n        print(f\"timebox exceeded\")\n\n    patch_string = None\n    for score_to_search in range(REQUIRED_SCORE, PERFECT_SCORE + 1):\n        answers = read_answers(suffix=str(score_to_search))\n        for answer in answers:\n            if answer:\n                patch_string = answer\n\n    print(\"ended prediction\", int(time.time() - prediction_start_time))\n    ###### CALL ENDS #####\n\n    teardown()\n\n    if is_on_kaggle_submission():\n        if patch_string is None:\n            pass\n        else:\n            submissions_left -= 1\n    else:\n        submissions_left -= 1\n    \n    print(\"submitted patch_string\")\n    print(patch_string)\n\n    if patch_string is None:\n        return None\n\n    return patch_string","metadata":{"_uuid":"bc379814-b967-4008-859a-018a17e1d72c","_cell_guid":"f16b9227-8945-4e91-9732-22d4085138df","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T22:32:20.927231Z","iopub.execute_input":"2025-03-08T22:32:20.927622Z","iopub.status.idle":"2025-03-08T22:32:20.936095Z","shell.execute_reply.started":"2025-03-08T22:32:20.927589Z","shell.execute_reply":"2025-03-08T22:32:20.934865Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Get predict data without server","metadata":{"_uuid":"876d9a65-7de7-452a-84d8-9971aab63bd6","_cell_guid":"8ea4d9f5-fdfc-418e-9b72-c4b519b8fd11","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import os\nimport zipfile\n\nif not is_on_kaggle_submission():\n    # !mkdir -p /kaggle/tmp/konwinski-prize-alt\n    os.makedirs(\"/tmp/kaggle/tmp/konwinski-prize-alt\", exist_ok=True)\n    \n    # !unzip -q -o /kaggle/input/konwinski-prize/data.a_zip -d /kaggle/tmp/konwinski-prize-alt/ 2>/dev/null || true\n    try:\n        with zipfile.ZipFile(\"/kaggle/input/konwinski-prize/data.a_zip\", \"r\") as zip_ref:\n            zip_ref.extractall(\"/tmp/kaggle/tmp/konwinski-prize-alt/\")\n    except:\n        pass","metadata":{"_uuid":"227756ef-2fca-4e27-b99d-fe373f0cf88a","_cell_guid":"621f1d4c-1d67-43ee-be33-a2f2d694f805","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T22:32:22.407982Z","iopub.execute_input":"2025-03-08T22:32:22.408259Z","iopub.status.idle":"2025-03-08T22:32:24.896085Z","shell.execute_reply.started":"2025-03-08T22:32:22.408238Z","shell.execute_reply":"2025-03-08T22:32:24.894949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport io\n\ntemp_data_dir = \"/tmp/kaggle/tmp/konwinski-prize-alt/data/\"\nmetadata_path = os.path.join(temp_data_dir, \"data.parquet\")\npip_packages_dir = os.path.join(temp_data_dir, \"pip_packages\")\nrepo_config_dir = os.path.join(temp_data_dir, \"repo_configs\")\nrepo_dir = os.path.join(temp_data_dir, \"repos\")\n\nfrom kprize_setup.kprize.evaluation.kprize_env_handler import KprizeEnvHandler\n\n\ndef get_problem(problem_index: int) -> tuple[str, io.BytesIO, io.BytesIO, list[str]]:\n    df = pd.read_parquet(\"/tmp/kaggle/tmp/konwinski-prize-alt/data/data.parquet\")\n    problem_statement: str = df[\"problem_statement\"][problem_index]\n\n    repo_path = os.path.join(repo_dir, f\"repo__{df['instance_id'][problem_index]}\")\n    pip_packages_path = os.path.join(pip_packages_dir, df[\"instance_id\"][problem_index])\n\n    import shutil\n    import tempfile\n\n    with tempfile.TemporaryDirectory() as tmpdir:\n        # instance repo\n        shutil.make_archive(os.path.join(tmpdir, \"a_repo\"), \"tar\", repo_path)\n        with open(os.path.join(tmpdir, \"a_repo.tar\"), \"rb\") as f:\n            repo_buffer = io.BytesIO(f.read())\n        # instance pip packages\n        shutil.make_archive(\n            os.path.join(tmpdir, \"a_pip_packages_dir\"), \"tar\", pip_packages_path\n        )\n        with open(os.path.join(tmpdir, \"a_pip_packages_dir.tar\"), \"rb\") as f:\n            pip_packages_buffer = io.BytesIO(f.read())\n\n    repo_config_path = os.path.join(\n        repo_config_dir, df[\"instance_id\"][problem_index].rsplit(\"-\", maxsplit=1)[0]\n    )\n    env_setup_cmd_templates = KprizeEnvHandler.get_env_setup_cmds_templates(\n        repo_config_path\n    )\n    return problem_statement, repo_buffer, pip_packages_buffer, env_setup_cmd_templates","metadata":{"_uuid":"38c49d86-7b5d-4aa9-a2b8-ebe9fb055748","_cell_guid":"6af6932e-ecdd-4610-878b-2f147736919b","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T22:32:24.89736Z","iopub.execute_input":"2025-03-08T22:32:24.897657Z","iopub.status.idle":"2025-03-08T22:32:24.908346Z","shell.execute_reply.started":"2025-03-08T22:32:24.897627Z","shell.execute_reply":"2025-03-08T22:32:24.906841Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Sample execution","metadata":{"_uuid":"ce58df36-5ae5-43e9-8714-8c237f3b0fa5","_cell_guid":"4643ed41-d2ee-4f78-b1d7-e39af894d321","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"if not is_on_kaggle_submission():\n    if is_on_kaggle_interactive():\n        problem_statement, repo_buffer, pip_packages_buffer, env_setup_cmd_templates = (\n            get_problem(problem_index=EVALUATION_PROBLEM_INDEX)\n        )\n    \n        print(problem_statement)\n        print(len(list(repo_buffer)))\n        print(len(list(repo_buffer)))\n        print(len(list(pip_packages_buffer)))\n        print(len(list(pip_packages_buffer)))\n        print(env_setup_cmd_templates)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T22:32:24.909563Z","iopub.execute_input":"2025-03-08T22:32:24.909865Z","iopub.status.idle":"2025-03-08T22:32:26.144561Z","shell.execute_reply.started":"2025-03-08T22:32:24.909838Z","shell.execute_reply":"2025-03-08T22:32:26.143216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if not is_on_kaggle_submission():\n    if is_on_kaggle_interactive() or is_on_kaggle_commit():\n        problem_statement, repo_buffer, pip_packages_buffer, env_setup_cmd_templates = (\n            get_problem(problem_index=EVALUATION_PROBLEM_INDEX)\n        )\n        setup(problem_statement, repo_buffer, pip_packages_buffer, env_setup_cmd_templates)","metadata":{"_uuid":"abb4dd8d-0eba-41c7-a151-818a0638b56c","_cell_guid":"2db23573-5b35-42b0-954a-e535ea7a55ab","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T22:32:34.77385Z","iopub.execute_input":"2025-03-08T22:32:34.77415Z","iopub.status.idle":"2025-03-08T22:32:34.99022Z","shell.execute_reply.started":"2025-03-08T22:32:34.774126Z","shell.execute_reply":"2025-03-08T22:32:34.989174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"! ([ \"$KAGGLE_KERNEL_RUN_TYPE\" = \"Interactive\" ] || [ \"$KAGGLE_KERNEL_RUN_TYPE\" = \"Batch\" ]) && [ -z \"$KAGGLE_IS_COMPETITION_RERUN\" ] && python -u worker.py || true","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T22:32:34.991684Z","iopub.execute_input":"2025-03-08T22:32:34.992068Z","iopub.status.idle":"2025-03-08T22:37:19.7551Z","shell.execute_reply.started":"2025-03-08T22:32:34.992043Z","shell.execute_reply":"2025-03-08T22:37:19.753605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if not is_on_kaggle_submission():\n    if is_on_kaggle_interactive() or is_on_kaggle_commit():\n        teardown()","metadata":{"_uuid":"dbad72c3-70ba-4d8b-a459-2f295b215f7f","_cell_guid":"62eac63e-850a-4fdf-9ec1-ddd9ed312e7e","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T22:11:08.607097Z","iopub.execute_input":"2025-03-08T22:11:08.607511Z","iopub.status.idle":"2025-03-08T22:11:08.629251Z","shell.execute_reply.started":"2025-03-08T22:11:08.607472Z","shell.execute_reply":"2025-03-08T22:11:08.627686Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"When your notebook is run on the hidden test set, inference_server.serve must be called within 15 minutes of the notebook starting or the gateway will throw an error. If you need more than 15 minutes to load your model you can do so during the very first predict call, which does not have the usual 30 minute response deadline.","metadata":{"_uuid":"727dda6e-39c3-4064-9f44-1ab12c9e7cc2","_cell_guid":"2f1f980d-954a-4d46-8874-215be176e348","trusted":true,"collapsed":false,"papermill":{"duration":0.001889,"end_time":"2024-12-11T03:22:08.856283","exception":false,"start_time":"2024-12-11T03:22:08.854394","status":"completed"},"tags":[],"jupyter":{"outputs_hidden":false}}},{"cell_type":"markdown","source":"# With inference server","metadata":{"_uuid":"096e622b-e27f-4a56-a851-642c1a652262","_cell_guid":"8a73c90b-e2ec-4da3-b9e5-b65ac9ed05dd","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"instance_count: Optional[int] = None\n\n\ndef get_number_of_instances(num_instances: int) -> None:\n    \"\"\"The very first message from the gateway will be the total number of instances to be served.\n    You don't need to edit this function.\n    \"\"\"\n    global instance_count\n    instance_count = num_instances\n\n\ninference_server = (\n    kaggle_evaluation.konwinski_prize_inference_server.KPrizeInferenceServer(\n        get_number_of_instances, predict\n    )\n)\n\nif os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            \"/kaggle/input/konwinski-prize/\",  # Path to the entire competition dataset\n            \"/kaggle/tmp/konwinski-prize/\",  # Path to a scratch directory for unpacking data.a_zip.\n        )  # type: ignore\n    )","metadata":{"_uuid":"75ecb8a4-9db6-4106-97fd-46ea6ed9979d","_cell_guid":"170a50a8-5543-45d7-907d-2622483e70c9","trusted":true,"collapsed":false,"_kg_hide-output":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-08T22:11:08.631217Z","iopub.execute_input":"2025-03-08T22:11:08.631526Z","iopub.status.idle":"2025-03-08T22:11:15.517549Z","shell.execute_reply.started":"2025-03-08T22:11:08.631507Z","shell.execute_reply":"2025-03-08T22:11:15.515993Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluation","metadata":{"_uuid":"c8364054-e811-4c00-9aa7-6cbe2b4188a1","_cell_guid":"21564afb-2ae7-4bd4-81ee-99ac3b0c0ab9","collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"if not is_on_kaggle_submission():\n    import kaggle_evaluation.konwinski_prize_gateway\n    k_prize_gateway = kaggle_evaluation.konwinski_prize_gateway.KPrizeGateway()\n    k_prize_gateway.unpack_data_paths()\n\n    import polars as pl\n    df = pl.read_parquet('/tmp/kaggle/tmp/konwinski-prize-alt/data/data.parquet')\n\n    \"\"\"\n    # test usage\n    problem_index = 5\n    results = test_patch(df.row(problem_index, named=True)[\"patch\"], problem_index)\n    \"\"\"\n    \n    def test_patch(patch: str, problem_index: int) -> None:\n    \n        from pathlib import Path\n        print(\"testing\")\n        print(patch)\n        \n        # because the utility script messes with the library versions\n        original_pythonpath = os.environ['PYTHONPATH']\n        os.environ['PYTHONPATH'] = \"/kaggle/lib/kagglegym:/kaggle/lib:/kaggle/usr/lib:/kaggle/input/konwinski-prize\"\n        \n        results = k_prize_gateway._evaluate_instance(\n            instance = df.row(problem_index, named=True),\n            patch = patch,\n        )\n        \n        os.environ['PYTHONPATH'] = original_pythonpath\n    \n        with open(\"patch.txt\", \"w\") as f:\n            f.write(patch)\n    \n        from collections import Counter\n        print(\n            problem_index,\n            kaggle_evaluation.konwinski_prize_gateway.is_valid_patch_format(patch),\n            kaggle_evaluation.konwinski_prize_gateway.patch_dry_run_succeeds(\n                Path(\"/kaggle/working/patch.txt\"),\n                Path(f'/tmp/kaggle/tmp/konwinski-prize-alt/data/repos/repo__{df[\"instance_id\"][problem_index]}'),\n            ),\n            Counter(result.unit_test_outcome for result in results[1:])\n        )\n    \n        from kaggle_evaluation.konwinski_prize_gateway import UnitTestOutcome\n    \n        for result in results[1:]:\n            if result.unit_test_outcome != UnitTestOutcome.PASSED:\n                print(result.test_name)\n                print(result.fail_description)\n\n        return results","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T22:11:15.518667Z","iopub.execute_input":"2025-03-08T22:11:15.518965Z","iopub.status.idle":"2025-03-08T22:11:17.79131Z","shell.execute_reply.started":"2025-03-08T22:11:15.518942Z","shell.execute_reply":"2025-03-08T22:11:17.789942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nif not is_on_kaggle_submission():\n\n    import kaggle_evaluation.konwinski_prize_gateway\n\n    for idx in range(EVALUATION_COUNT):\n\n        submissions_left = 1  # ignore MAXIMUM_SUBMISSIONS\n        problem_statement, repo_buffer, pip_packages_buffer, env_setup_cmd_templates = (\n            get_problem(problem_index=EVALUATION_PROBLEM_INDEX)\n        )\n        patch_string = predict(problem_statement, repo_buffer, pip_packages_buffer, env_setup_cmd_templates)\n\n        if patch_string is not None:\n            results = test_patch(patch_string, problem_index=EVALUATION_PROBLEM_INDEX)\n            pd.DataFrame(results).to_csv(f\"results_{idx:03d}.csv\", index=False)\n\n\n    import pandas as pd\n    import glob\n    import os\n    \n    file_list = sorted(glob.glob('logs_*.csv'))\n    \n    df_list = [\n        pd.read_csv(file).assign(source_file=os.path.basename(file))\n        for file in file_list\n    ]\n\n    if df_list:\n        combined_df = pd.concat(df_list, ignore_index=True) if df_list else pd.DataFrame()    \n        combined_df.to_csv('combined_logs.csv', index=False)\n\n        combined_df[combined_df[\"score\"] >= BASE_ACCEPTED_SCORE].to_csv('combined_logs_valid.csv', index=False)\n\n        for file in file_list:\n            os.remove(file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-08T22:11:17.79226Z","iopub.execute_input":"2025-03-08T22:11:17.792502Z","iopub.status.idle":"2025-03-08T22:19:28.108277Z","shell.execute_reply.started":"2025-03-08T22:11:17.792483Z","shell.execute_reply":"2025-03-08T22:19:28.106108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}