{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaL4","dataSources":[{"sourceId":84795,"databundleVersionId":11281725,"sourceType":"competition"},{"sourceId":10998331,"sourceType":"datasetVersion","datasetId":6846525},{"sourceId":10998679,"sourceType":"datasetVersion","datasetId":6846747},{"sourceId":221096520,"sourceType":"kernelVersion"},{"sourceId":265863,"sourceType":"modelInstanceVersion","modelInstanceId":227466,"modelId":224053},{"sourceId":265872,"sourceType":"modelInstanceVersion","modelInstanceId":227475,"modelId":224053},{"sourceId":276458,"sourceType":"modelInstanceVersion","modelInstanceId":236741,"modelId":224053}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### This notebook was copied and edited from kaggle user named Tong Hui Kang whose original profile link and notebook link i will be adding.\n#### The changes we made here is we tried using AST's to feed the model with context and are willing to see how well the models perform in patch generation which is different from the approach used in the original notebook of using the keywords generated by the llm.\n#### user profile : https://www.kaggle.com/huikang\n#### original notebook : https://www.kaggle.com/code/huikang/starter-notebook-select-patch-verify","metadata":{"_uuid":"086f9e64-7a54-43f8-a39c-b9d66916a121","_cell_guid":"3036cc2a-e7d0-4a3d-bd06-eda7b2e84c62","collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import os\nimport time\n# https://www.kaggle.com/competitions/ai-mathematical-olympiad-progress-prize-2/discussion/560682#3113134\nos.environ[\"TRITON_PTXAS_PATH\"] = \"/usr/local/cuda/bin/ptxas\"\n\nglobal_start_time = time.time() ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip install /kaggle/input/codemod-wheel-dataset/codemod-1.0.0-py3-none-any.whl\n!pip install --no-deps --ignore-installed --force-reinstall /kaggle/input/codemod-wheel-dataset/codemod-1.0.0-py3-none-any.whl\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip install /kaggle/input/pyanalyze-k-prize/pyanalyze/*.whl\n!pip install --no-deps --ignore-installed --force-reinstall /kaggle/input/pyanalyze-k-prize/pyanalyze/*.whl || true","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pyanalyze","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import io\nimport time\nimport shutil\nimport re\nimport xml.etree.ElementTree as ET\nimport pandas as pd\nimport polars as pl\nimport os\nfrom collections import defaultdict\nimport kaggle_evaluation.konwinski_prize_inference_server\nfrom typing import List, Tuple, Dict, Optional\n\nstart_time = time.time()\nallowed_time = [start_time + 60 * 60]","metadata":{"_uuid":"481ef31d-36ba-45a3-9a98-138a459f6d85","_cell_guid":"eb862ac3-4642-4ecd-a1a0-b2701c00d800","trusted":true,"collapsed":false,"_kg_hide-output":true,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The evaluation API requires that you set up a server which will respond to inference requests. We have already defined the server; you just need write the predict function. When we evaluate your submission on the hidden test set the client defined in `konwinski_prize_gateway` will run in a different container with direct access to the hidden test set and hand off the data.\n#\nYour code will always have access to the published copies of the files.","metadata":{"_uuid":"d6516b72-e2ca-4261-a058-e4bdf9eae190","_cell_guid":"5b24338e-0844-44fc-9288-8c064601e578","trusted":true,"collapsed":false,"papermill":{"duration":0.002032,"end_time":"2024-12-11T03:22:08.823897","exception":false,"start_time":"2024-12-11T03:22:08.821865","status":"completed"},"tags":[],"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"instance_count: Optional[int] = None\n\n\ndef get_number_of_instances(num_instances: int) -> None:\n    \"\"\"The very first message from the gateway will be the total number of instances to be served.\n    You don't need to edit this function.\n    \"\"\"\n    global instance_count\n    instance_count = num_instances","metadata":{"_uuid":"3464f257-aeab-4b4c-bf8b-73591ca7fce9","_cell_guid":"af345919-bd32-4541-9e26-86b1d1370f87","trusted":true,"collapsed":false,"papermill":{"duration":0.011949,"end_time":"2024-12-11T03:22:08.838279","exception":false,"start_time":"2024-12-11T03:22:08.82633","status":"completed"},"tags":[],"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Initialize LLM","metadata":{"_uuid":"9601f137-9946-4cd0-9ef2-712d8517bd93","_cell_guid":"1a312bd3-27ae-4242-bee3-b632f1bf65fe","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# from vllm import LLM, SamplingParams, RequestOutput\n# import warnings\n\n# warnings.simplefilter(\"ignore\")\n\n# os.environ[\"CUDA_VISIBLE_DEVICES\"] = \"0,1,2,3\"\n# os.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"\n\n# if os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\") or os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"):\n#     llm_model_pth: str = (\n#         \"/kaggle/input/deepseek-r1/transformers/qwen-qwq-32b-awq/1\"\n#     )\n# else:\n#     llm_model_pth: str = \"/root/volume/KirillR/QwQ-32B-Preview-AWQ\"\n\n# BATCH_SIZE: int = 6\n# VALIDATION_COPY_COUNT: int = 1\n# MAX_TOKENS: int = 8192\n\n# MAX_NUM_SEQS: int = 6\n# MAX_MODEL_LEN: int = 32_768\n\n# llm: LLM = LLM(\n#     llm_model_pth,\n#     max_num_seqs=MAX_NUM_SEQS,  # Maximum number of sequences per iteration. Default is 256\n#     max_model_len=MAX_MODEL_LEN,  # Model context length\n#     trust_remote_code=True,  # Trust remote code (e.g., from HuggingFace) when downloading the model and tokenizer\n#     tensor_parallel_size=4,  # The number of GPUs to use for distributed execution with tensor parallelism\n#     enable_prefix_caching=True,\n#     gpu_memory_utilization=0.95,  # The ratio (between 0 and 1) of GPU memory to reserve for the model\n#     seed=2024,\n# )","metadata":{"_uuid":"6b90e41a-106f-47a5-a8c2-6b79f727bdb5","_cell_guid":"e99a7258-4aee-49d8-bd39-3c10ca40ddf0","trusted":true,"collapsed":false,"_kg_hide-output":true,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# tokenizer = llm.get_tokenizer()\n\n\ndef count_tokens(text: str) -> int:\n    return len(tokenizer.encode(text))\n\nelapsed_time = time.time() - global_start_time  # Calculate elapsed time\nprint(f\"Total elapsed time since start: {elapsed_time:.2f} seconds\")\n","metadata":{"_uuid":"08b952e7-cd51-4cff-9d84-5f02675ed1dc","_cell_guid":"0406e19f-9d43-49c6-bc58-0a2a5e507595","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Helper functions","metadata":{"_uuid":"0bf5c8d3-7e74-4a4b-a31d-da48f5b361c4","_cell_guid":"42420a5a-5116-41f0-a68e-1186d97f5f85","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"REPO_PATH = \"repo\"\n\n\ndef setup(\n    repo_archive: io.BytesIO,\n    pip_packages_archive: io.BytesIO,\n    env_setup_cmds_templates: list[str],\n    repo_path: str,\n) -> None:\n    \"\"\"Replace this function with your inference code.\n    Args:\n        problem_statement: The text of the git issue.\n        repo_path: A BytesIO buffer path with a .tar containing the codebase that must be patched. The gateway will make this directory available immediately before this function runs.\n        pip_packages_archive: A BytesIO buffer path with a .tar containing the wheel files necessary for running unit tests.\n        env_setup_cmds_templates: Commands necessary for installing the pip_packages_archive.\n    \"\"\"\n\n    # Unpack the codebase to be patched into a directory that won't be exported when\n    # the notebook is saved.\n    archive_path = \"/tmp/repo_archive.tar\"\n    with open(archive_path, \"wb\") as f:\n        f.write(repo_archive.read())\n    if os.path.exists(repo_path):\n        shutil.rmtree(repo_path)\n    shutil.unpack_archive(archive_path, extract_dir=repo_path)\n    os.remove(archive_path)\n\n    \"\"\"\n    Unpack pip_packages if you want to run unit tests on your patch.\n    Note that editing unit tests with your patch -- even to add valid tests -- can cause your submission to be flagged as a failure.\n    Most of the relevant repos use pytest for running tests. You will almost certainly need to run only a subset of the unit tests to avoid running out of inference time.\n    \"\"\"\n    pip_archive_dir = \"/tmp/pip_packages_archive.tar\"\n    with open(pip_archive_dir, \"wb\") as f:\n        f.write(pip_packages_archive.read())\n    pip_packages_path = \"/path/to/pip_packages\"\n    if os.path.exists(pip_packages_path):\n        shutil.rmtree(pip_packages_path)\n    shutil.unpack_archive(pip_archive_dir, extract_dir=pip_packages_path)\n    os.remove(pip_archive_dir)\n\n    # Get env setup cmds by setting the pip_packages_path\n    env_setup_cmds = [\n        cmd.format(pip_packages_path=pip_packages_path)\n        for cmd in env_setup_cmds_templates\n    ]\n\n    # Run env setup for the repo\n    subprocess.run(\n        \"\\n\".join(env_setup_cmds),\n        shell=True,\n        executable=\"/bin/bash\",\n        cwd=repo_path,\n    )","metadata":{"_uuid":"a08a265b-729f-4b45-9eda-e2b5917d1b97","_cell_guid":"a97c9025-75f9-449b-93be-698746b23c7c","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from vllm import LLM, SamplingParams, RequestOutput\nimport warnings\nimport os\n\nwarnings.simplefilter(\"ignore\")\n\n# Global LLM variable (Initially None)\nllm = None  \ntokenizer = None\ndef initialize_llm():\n    \"\"\"Initializes the LLM model and makes it globally accessible.\"\"\"\n    global llm  # Ensure llm is accessible throughout the script\n\n    # Set environment variables\n    os.environ[\"CUDA_VISIBLE_DEVICES\"] = \"0,1,2,3\"\n    os.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"\n\n    # Choose model path based on execution environment\n    if os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\") or os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"):\n        llm_model_pth: str = \"/kaggle/input/deepseek-r1/transformers/qwen-qwq-32b-awq/1\"\n    else:\n        llm_model_pth: str = \"/root/volume/KirillR/QwQ-32B-Preview-AWQ\"\n\n    # Model Parameters\n    BATCH_SIZE: int = 6\n    MAX_TOKENS: int = 8192\n    MAX_NUM_SEQS: int = 6\n    MAX_MODEL_LEN: int = 32_768\n\n    # Initialize LLM globally\n    llm = LLM(\n        llm_model_pth,\n        max_num_seqs=MAX_NUM_SEQS,\n        max_model_len=MAX_MODEL_LEN,\n        trust_remote_code=True,\n        tensor_parallel_size=4,  # Adjust based on available GPUs\n        enable_prefix_caching=True,\n        gpu_memory_utilization=0.95,  # Utilize 95% of GPU memory\n        seed=2024,\n    )\n    tokenizer = llm.get_tokenizer()\n    print(\"LLM has been initialized successfully!\")\n\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n\n# def stringify_directory(directory: str) -> str:\n#     full_paths: List[str] = []\n#     banned_strings = [\".venv\", \".pyc\", \".txt\", \".pytest_cache\", \".github\", \"/doc/\"]\n\n#     for root, dirs, files in os.walk(directory):\n#         for file in files:\n#             for banned_string in banned_strings:\n#                 if banned_string in root or banned_string in file:\n#                     break\n#             else:\n#                 full_path: str = os.path.join(root, file)\n#                 full_paths.append(full_path)\n#     return \"\\n\".join(full_paths)","metadata":{"_uuid":"957ea2f5-0402-41f1-86d2-dfee2fb9fdd2","_cell_guid":"df138201-3898-4cd1-9a43-89bdb7c6393d","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import re\n\n\n# def extract_file_query(xml_content: str) -> Dict[str, List[str]]:\n#     import xml.etree.ElementTree as ET\n\n#     # Prepare a data structure to collect results\n#     parsed_data: Dict[str, List[str]] = {}\n#     pattern: str = r\"<root>(.*?)</root>\"\n#     matches: List[str] = re.findall(pattern, xml_content, re.DOTALL)\n\n#     for match in matches:\n#         try:\n#             # Parse the XML\n#             root = ET.fromstring(\"<root>\" + match + \"</root>\")\n\n#             # Find all <entry> elements\n#             for entry in root.findall(\"entry\"):\n#                 # Extract the <filepath> text\n#                 filepath = entry.find(\"filepath\")\n#                 filepath_text: Optional[str] = (\n#                     filepath.text.strip()\n#                     if filepath is not None and filepath.text is not None\n#                     else None\n#                 )\n\n#                 # Locate <strings_to_search> container\n#                 strings_container = entry.find(\"strings_to_search\")\n\n#                 # Gather each <string_to_search> text\n#                 search_strings: List[str] = []\n#                 if strings_container is not None:\n#                     for s in strings_container.findall(\"string_to_search\"):\n#                         if s.text is not None:\n#                             search_strings.append(s.text.strip())\n\n#                 # Store in a dictionary: { filepath: [search_strings...] }\n#                 parsed_data[filepath_text] = search_strings  # type: ignore\n#         except:\n#             print(\"Error parsing output\")\n#             print(xml_content)\n#             return {}\n\n#     return parsed_data","metadata":{"_uuid":"b32f50c7-4a25-49e3-8516-3bc95c1c5900","_cell_guid":"768b2b94-2a83-40a1-b9d6-678b3d1ca1ad","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# reading_prompt: str = (\n#     \"\"\"\n# You will be implementing a git diff patch to solve an issue with the code repository.\n# You will first need to select files in the file directory.\n\n# This is the problem statement.\n\n# {problem_statement}\n\n# This is the file directory\n\n# <directory>\n# {directory_string}\n# </directory>\n\n# Which files should be inspected so that we can solve the problem?\n# When we inspect each file, what strings should be searched?\n\n# Return the strings to search in this format\n\n# (explanation)\n\n# <root>\n#     <entry>\n#         <filepath>filepath</filepath>  \n#         <strings_to_search>\n#             <string_to_search>string_to_search</string_to_search>\n#             ...\n#             <string_to_search>string_to_search</string_to_search>\n#         </strings_to_search>\n#     </entry>\n#     <entry>\n#         <filepath>filepath</filepath>\n#         <strings_to_search>\n#             <string_to_search>string_to_search</string_to_search>\n#             ...\n#             <string_to_search>string_to_search</string_to_search>\n#         </strings_to_search>\n#     </entry>\n#     ...\n# </root>\n# ...\n\n# Notes:\n# - Make sure to encode each entry between <root> and </root>\n# - Return the FULL filepath - exactly as specified in <directory> and </directory>\n#     - Example: <filepath>repo/path/to/directory/file.py</filepath>\n# - If you are searching for a word instead of a substring, maybe add spaces or brackets before and after the string\n#     - For example, if you are searching for uses of the function `calculate`, use ` calculate(` as the search string instead of `calculate`\n# - Prefer searching longer strings\n#     - Avoid searching for strings that might appear in many parts of the codebase\n# - Search the test files as well to understand the feature behavior\n#     - Also search for the relevant function calls in the test files\n# \"\"\".strip()\n# )\n\n\n# def get_selection_query(\n#     directory_string: str, problem_statement: str\n# ) -> Tuple[List[str], List[Dict[str, List[str]]]]:\n#     sampling_params: SamplingParams = SamplingParams(\n#         temperature=0.6,  # randomness of the sampling\n#         min_p=0.01,\n#         skip_special_tokens=True,  # Whether to skip special tokens in the output\n#         max_tokens=MAX_TOKENS,\n#     )\n\n#     list_of_messages: List[List[Dict[str, str]]] = [\n#         [\n#             {\n#                 \"role\": \"user\",\n#                 \"content\": reading_prompt.format(\n#                     problem_statement=problem_statement[:20_000],\n#                     directory_string=directory_string[:30_000],\n#                 ),\n#             },\n#         ]\n#         for _ in range(BATCH_SIZE)\n#     ]\n\n#     prompt_texts: List[str] = [\n#         (\n#             tokenizer.apply_chat_template(\n#                 conversation=messages, tokenize=False, add_generation_prompt=True\n#             )  # type: ignore\n#         )\n#         + \"<think>\\n\"\n#         for messages in list_of_messages\n#     ]\n#     # print(prompt_texts)\n\n#     print(\"get_selection_query\", [count_tokens(text) for text in prompt_texts])\n#     request_outputs: list[RequestOutput] = llm.generate(\n#         prompt_texts, sampling_params=sampling_params\n#     )\n#     if not request_outputs:\n#         return [], []\n#     response_texts: List[str] = [\n#         request_output.outputs[0].text for request_output in request_outputs\n#     ]\n#     print(\"get_selection_query\", [count_tokens(text) for text in response_texts])\n\n#     completion_texts = [\n#         prompt_text + response_text\n#         for prompt_text, response_text in zip(prompt_texts, response_texts)\n#     ]\n#     file_queries: List[Dict[str, List[str]]] = [\n#         extract_file_query(response_text) for response_text in response_texts\n#     ]\n#     return completion_texts, file_queries","metadata":{"_uuid":"e83192b1-1843-4c6a-a886-363810d8501b","_cell_guid":"3385fe9b-8e27-476e-a395-33072a3fe894","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_deciding: str = (\n    \"\"\"\nTask:\nyou will be given a list of files like below. your task is to identify the most probable one's where the said error/problem might be occuring\n\n{list_of_files}\n\nexample list of files:\n['path/to/file1',\n.\n.\n'path/to/filen-1'\n'path/to/filen']\n\nproblem statement:\n{problem_statement}\n\ncontext/additional information: you must give me the most probable file where the said error might be occuring considering the problem statement.\naccuracy or absolute clarity is not important as the information is not enough to decide the probable file.\njust give the most probable file where the problem might be occuring considering teh general ways/patterns or situations where these kind of errors might occur\nor has the most chances of the error being present in that file. It's ok if the judgement is vague as it doesn't have enough information. \n\nanswer format: you will give me the output in this format \n<root>\n    <entry>\n        <file_path>filepath</file_path>\n    </entry>\n</root>\n\nexample answer format path:\n<root>\n    <entry>\n        <file_path>C:/Users/Dell/Documents/projects/swe-bench/konwinski-prize/data/data/repos/repo__pylint-dev__astroid-2496/astroid/nodes/const</file_path>\n    </entry>\n</root>\n\n\"\"\".strip()\n)\ndef deciding_the_files(files_list_list:List[List[str]] , problem_statement: str)-> List[str]:\n\n    import re\n    import xml.etree.ElementTree as ET\n    from typing import List\n    \n    def extract_xml_from_text(text: str) -> str:\n        \"\"\"Extracts the XML portion from a text string using regex.\"\"\"\n        xml_match = re.search(r\"<root>.*?</root>\", text, re.DOTALL)\n        return xml_match.group(0) if xml_match else None\n    \n    def extract_file_path_from_text(text: str) -> str:\n        \"\"\"Extracts the file path from the XML inside the text.\"\"\"\n        xml_data = extract_xml_from_text(text)\n        if not xml_data:\n            return None  # No XML found, return None\n    \n        # Parse the XML and extract the file path\n        root = ET.fromstring(xml_data)\n        file_path_element = root.find(\".//file_path\")\n    \n        return file_path_element.text if file_path_element is not None else None\n    \n    def extract_file_paths(response_texts: List[str]) -> List[str]:\n        \"\"\"Extracts file paths from a list of response texts.\"\"\"\n        file_paths = [\n            extract_file_path_from_text(text) for text in response_texts\n        ]\n        \n        # Remove None values in case some texts don't contain valid XML\n        return [fp for fp in file_paths if fp is not None]\n    \n  \n    def convert_file_list_to_string(file_list):\n    return \"[\\n    '\" + \"',\\n    '\".join(file_list) + \"'\\n]\"\n    \n    sampling_params: SamplingParams = SamplingParams(\n        temperature=0.6,  # randomness of the sampling\n        min_p=0.01,\n        skip_special_tokens=True,  # Whether to skip special tokens in the output\n        max_tokens=MAX_TOKENS,\n    )\n    file_list_str = [convert_file_list_to_string(ele) for ele in file_list_list]\n    \n    list_of_messages: List[List[Dict[str, str]]] = [\n        [\n            {\n                \"role\": \"user\",\n                \"content\": file_deciding.format(\n                    problem_statement=problem_statement[:20_000],\n                    list_of_files=ele[:30_000],\n                ),\n            },\n        ]\n        for ele in file_list_str\n    ]\n\n    prompt_texts: List[str] = [\n        (\n            tokenizer.apply_chat_template(\n                conversation=messages, tokenize=False, add_generation_prompt=True\n            )  # type: ignore\n        )\n        + \"<think>\\n\"\n        for messages in list_of_messages\n    ]\n\n    request_outputs: list[RequestOutput] = llm.generate(\n        prompt_texts, sampling_params=sampling_params\n    )\n    if not request_outputs:\n        return [], []\n    response_texts: List[str] = [\n        request_output.outputs[0].text for request_output in request_outputs\n    ]\n\n    final_file_paths = extract_file_paths(response_texts)\n    return final_file_paths","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"deciding_lines: str = (\n    \"\"\"\nTask:you will be given a problem statement for a unit test case..you will be giving me the lines at which the error may have occured.\nyou should also give a line that occurs before the line where error occured.\n\nThis is the problem statement.\n\n{problem_statement}\n\nyou will be giving me the otput in this format:\n<root>\n    <entry>\n        <actual_error_occuring_code_line>actual_error_occuring_code_line</actual_error_occuring_code_line>\n    </entry>\n    <entry>\n        <code_line_before_error_occurence>code_line_before_error_occurence</code_line_before_error_occurence>\n    </entry>\n</root>\n\nExample:\n\n```python\nfrom astropy.table import QTable, Column\nimport numpy as np\ndata = np.ones((5,2))\ntt = QTable(data=data, names=[\"a\",\"weight\"],units={\"weight\":u.dimensionless_unscaled})\n```\nResults in \n```\n---------------------------------------------------------------------------\nZeroDivisionError                         Traceback (most recent call last)\nCell In[15], line 2\n      1 data = np.ones((5,2))\n----> 2 tt = QTable(data=data, names=[\"a\",\"weight\"],units={\"weight\":u.dimensionless_unscaled})\n      3 tt\n\nFile ~/miniconda3/envs/datapipe-testbench/lib/python3.11/site-packages/astropy/table/table.py:887, in Table.__init__(self, data, masked, names, dtype, meta, copy, rows, copy_indices, units, descriptions, **kwargs)\n    884 if self.masked not in (None, True, False):\n    885     raise TypeError(\"masked property must be None, True or False\")\n--> 887 self._set_column_attribute(\"unit\", units)\n    888 self._set_column_attribute(\"description\", descriptions)\n\nExample Answer: \n<root>\n    <entry>\n        <actual_error_occuring_code_line>tt = QTable(data=data, names=[\"a\",\"weight\"],units={\"weight\":u.dimensionless_unscaled})</actual_error_occuring_code_line>\n    </entry>\n    <entry>\n        <code_line_before_error_occurence>data = np.ones((5,2))</code_line_before_error_occurence>\n    </entry>\n</root>\n\nNotes: \n1. note that for any error the tracing might lead to the actual error being mentined encountered in the dependencies/library code.\n2. In that case you should be giving me the line where the error occured in actual code and not the dependencies like mentioned in the example here.\n3. In cases where the actual line of error cannot be found just give the code line before that as actual_error_occuring_code_line and the previous line to that as the code_line_before_error_occurence.\n4. let us get only two lines of code as requested.\n\"\"\".strip()\n)\ndef getting_lines(problem_statement : str)->List[List[str]]:\n    sampling_params: SamplingParams = SamplingParams(\n        temperature=0.6,  # randomness of the sampling\n        min_p=0.01,\n        skip_special_tokens=True,  # Whether to skip special tokens in the output\n        max_tokens=MAX_TOKENS,\n    )\n    import re\n    import xml.etree.ElementTree as ET\n    from typing import List\n    def extract_xml_from_text(text: str) -> str:\n        \"\"\"Extracts the XML portion from a text string using regex.\"\"\"\n        xml_match = re.search(r\"<root>.*?</root>\", text, re.DOTALL)\n        return xml_match.group(0) if xml_match else None\n    \n    def extract_code_lines_from_text(text: str) -> List[str]:\n        \"\"\"Extracts code lines from the XML inside the text.\"\"\"\n        xml_data = extract_xml_from_text(text)\n        if not xml_data:\n            return []  # Return empty list if no XML found\n    \n        # Parse the XML and extract code lines\n        root = ET.fromstring(xml_data)\n    \n        # Extract all code-related elements\n        code_lines = []\n        for tag in [\"actual_error_occuring_code_line\", \"code_line_before_error_occurence\"]:\n            for element in root.findall(f\".//{tag}\"):\n                if element.text:\n                    code_lines.append(element.text.strip())\n    \n        return code_lines\n        \n    list_of_messages: List[List[Dict[str, str]]] = [\n        [\n            {\n                \"role\": \"user\",\n                \"content\": deciding_lines.format(\n                    problem_statement=problem_statement[:20_000],\n                ),\n            },\n        ]\n        for _ in range(BATCH_SIZE)\n    ]\n\n    prompt_texts: List[str] = [\n        (\n            tokenizer.apply_chat_template(\n                conversation=messages, tokenize=False, add_generation_prompt=True\n            )  # type: ignore\n        )\n        + \"<think>\\n\"\n        for messages in list_of_messages\n    ]\n    \n    request_outputs: list[RequestOutput] = llm.generate(\n        prompt_texts, sampling_params=sampling_params\n    )\n    if not request_outputs:\n        return [], []\n    response_texts: List[str] = [\n        request_output.outputs[0].text for request_output in request_outputs\n    ]\n    code_lines_list = [\n        extract_code_lines_from_text(response_text) for response_text in response_texts\n    ]\n    return code_lines_list\n    ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def fetch_file_contents(\n#     files_to_search: Dict[str, List[str]], context_lines: int = 12, max_gap: int = 0\n# ) -> str:\n#     from io import StringIO\n#     from typing import Tuple\n\n#     def find_lines_in_files_with_context(\n#         search_map: Dict[str, List[str]], context_lines: int = context_lines\n#     ) -> List[List[List[Tuple[int, str]]]]:\n#         \"\"\"\n#         Given a dictionary mapping file paths to a list of search terms,\n#         open each file and gather *snippets* of lines that contain any\n#         of those search terms, including 'context_lines' before and after.\n\n#         Returns a list of lists:\n#         [\n#           [  # For file1\n#              [ (line_number, text), (line_number, text), ... ],\n#              [ ... ],\n#           ],\n#           [  # For file2\n#              ...\n#           ],\n#           ...\n#         ]\n#         \"\"\"\n#         all_matches_per_file: List[List[List[Tuple[int, str]]]] = []\n\n#         for path, terms in search_map.items():\n#             if not os.path.isfile(path):\n#                 # If the file is not found, record an empty list\n#                 all_matches_per_file.append([])\n#                 continue\n\n#             with open(path, \"r\", encoding=\"utf-8\", errors=\"replace\") as f:\n#                 lines = f.readlines()\n\n#             file_snippets: List[List[Tuple[int, str]]] = []\n#             num_lines: int = len(lines)\n\n#             for i, line in enumerate(lines, start=1):\n#                 if any(t in line for t in terms):\n#                     start_idx: int = max(1, i - context_lines)\n#                     end_idx: int = min(num_lines, i + context_lines)\n#                     snippet: List[Tuple[int, str]] = []\n#                     for snippet_no in range(start_idx, end_idx + 1):\n#                         text_content: str = lines[snippet_no - 1].rstrip(\"\\n\")\n#                         snippet.append((snippet_no, text_content))\n#                     file_snippets.append(snippet)\n\n#             all_matches_per_file.append(file_snippets)\n\n#         return all_matches_per_file\n\n#     # ---------------------------------------------------------\n#     # 3. MERGE OVERLAPPING/ADJACENT SNIPPETS\n#     # ---------------------------------------------------------\n\n#     def merge_file_snippets(\n#         file_snippets: List[List[Tuple[int, str]]], gap: int = 0\n#     ) -> List[List[Tuple[int, str]]]:\n#         \"\"\"\n#         Merge overlapping or nearly adjacent snippets in a single file’s snippet list.\n#         \"\"\"\n#         intervals: List[Tuple[int, int, List[Tuple[int, str]]]] = []\n#         for snippet in file_snippets:\n#             if snippet:\n#                 start_line: int = snippet[0][0]\n#                 end_line: int = snippet[-1][0]\n#                 intervals.append((start_line, end_line, snippet))\n\n#         intervals.sort(key=lambda x: x[0])  # sort by start line\n\n#         merged: List[Tuple[int, int, List[Tuple[int, str]]]] = []\n#         for start, end, snippet in intervals:\n#             if not merged:\n#                 merged.append((start, end, snippet))\n#                 continue\n\n#             prev_start, prev_end, prev_snippet = merged[-1]\n#             if start <= prev_end + gap:\n#                 new_end: int = max(end, prev_end)\n#                 combined_dict: Dict[int, str] = {}\n#                 for ln, txt in prev_snippet:\n#                     combined_dict[ln] = txt\n#                 for ln, txt in snippet:\n#                     combined_dict[ln] = txt\n#                 merged_snippet: List[Tuple[int, str]] = [\n#                     (ln, combined_dict[ln]) for ln in sorted(combined_dict)\n#                 ]\n#                 merged[-1] = (prev_start, new_end, merged_snippet)\n#             else:\n#                 merged.append((start, end, snippet))\n\n#         # Extract just the merged snippet portion\n#         return [x[2] for x in merged]\n\n#     def merge_all_snippets(\n#         all_files_snips: List[List[List[Tuple[int, str]]]], gap: int = 0\n#     ) -> List[List[List[Tuple[int, str]]]]:\n#         \"\"\"\n#         Merge snippet blocks within each file.\n#         all_files_snips is a list-of-lists:\n#           [\n#             [ snippetA, snippetB, ... ],  # file 1\n#             [ snippetC, snippetD, ... ],  # file 2\n#           ]\n#         \"\"\"\n#         merged: List[List[List[Tuple[int, str]]]] = []\n#         for snips in all_files_snips:\n#             merged.append(merge_file_snippets(snips, gap=gap))\n#         return merged\n\n#     # ---------------------------------------------------------\n#     # 4. RUN LOGIC: generate files, search, merge, and BUILD A STRING\n#     # ---------------------------------------------------------\n\n#     has_any_matches: bool = False\n\n#     # 1) Gather snippets around each match\n#     context_snippets: List[List[List[Tuple[int, str]]]] = (\n#         find_lines_in_files_with_context(files_to_search, context_lines=context_lines)\n#     )\n\n#     # 2) Merge overlapping snippets\n#     merged_snips: List[List[List[Tuple[int, str]]]] = merge_all_snippets(\n#         context_snippets, gap=max_gap\n#     )\n\n#     # 3) Build a string (instead of printing)\n#     output = StringIO()\n\n#     # Header\n#     output.write(\"Sample files created successfully.\\n\\n\")\n#     output.write(\"Search Results (by file, merging any overlapping context):\\n\\n\")\n\n#     # For each file\n#     for (filepath, terms), snippet_list in zip(files_to_search.items(), merged_snips):\n#         output.write(f\"[file name]: {filepath[len(REPO_PATH) + 1:]}\\n\")\n#         terms_searched_as_str = \"\\n\".join(terms)\n#         output.write(f\"[terms searched]:\\n{terms_searched_as_str}\\n\")\n#         output.write(\"[file content begin]\\n\")\n#         if not snippet_list:\n#             output.write(\"  No matches found.\\n\")\n#         else:\n#             has_any_matches = True\n#             for snippet_idx, snippet in enumerate(snippet_list, start=1):\n#                 snippet_start: int = snippet[0][0]\n#                 snippet_end: int = snippet[-1][0]\n#                 output.write(\n#                     f\"\\nMatch #{snippet_idx}, lines {snippet_start} to {snippet_end}:\\n\"\n#                 )\n#                 for line_no, text in snippet:\n#                     output.write(f\"  {line_no:3d} | {text}\\n\")\n#                 output.write(\"\\n\")\n#         output.write(\"[file content end]\\n\\n\")\n\n#     file_content_string: str = output.getvalue()\n\n#     if has_any_matches:\n#         return file_content_string\n#     return \"\"","metadata":{"_uuid":"28a297dd-a6f2-4b9e-8c6d-3553187930ae","_cell_guid":"97cbebf3-16b9-4971-a1b2-c5645c0fd3ef","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\n\n\ndef extract_patch_string(text: str) -> Optional[str]:\n    pattern: str = r\"\\n```diff\\n(.*?)\\n```\"\n    matches: List[str] = re.findall(pattern, text, re.DOTALL)\n    if not matches:\n        return None\n    return matches[-1] + \"\\n\"","metadata":{"_uuid":"7f4d31bd-6729-40d7-b648-0c5d36742788","_cell_guid":"99c7c237-6b7a-4160-b576-550fde9de659","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"patching_prompt: str = (\n    \"\"\"\nYou will be implementing a git diff patch to solve an issue with the code repository.\nThis is the problem statement.\n\n{problem_statement}\n\nThese are the files that is thought to be relevant\n\n{file_content_string}\n\nWrite a git diff within ```diff and ``` that fully fixes the problem.\nThe git diff should not cause other tests to fail.\nDo not edit the test files.\n\nExample:\n\n```diff\n--- a/first.txt\n+++ b/first.txt\n@@ -1,3 +1,3 @@\n start\n-first change\n+new first change\n middle\n@@ -7,4 +7,4 @@\n some content\n-second change\n+new second change\n more content\n--- a/second.txt\n+++ b/second.txt\n@@ -1,3 +1,3 @@\n beginning\n-old line\n+new line\n end\n```\n\nReminder\n- Put your diff within ```diff and ``` and make sure the diff is valid.\n- Only the last diff printed will be considered.\n- Do not edit the test files.\n\"\"\".strip()\n)\n\nimport re\n\n\ndef get_patch_string(\n    problem_statement: str, file_content_strings: List[str]\n) -> Tuple[List[str], List[Optional[str]]]:\n    sampling_params: SamplingParams = SamplingParams(\n        temperature=0.6,  # randomness of the sampling\n        min_p=0.01,\n        skip_special_tokens=True,  # Whether to skip special tokens in the output\n        max_tokens=MAX_TOKENS,\n    )\n\n    inference_idx_to_input_idx: list[int] = [\n        input_idx\n        for input_idx, file_content_string in enumerate(file_content_strings)\n        if file_content_string != \"\"\n    ]\n\n    list_of_messages: List[List[Dict[str, str]]] = [\n        [\n            {\n                \"role\": \"user\",\n                \"content\": patching_prompt.format(\n                    problem_statement=problem_statement[:20_000],\n                    file_content_string=file_content_strings[input_idx][:30_000],\n                ),\n            },\n        ]\n        for input_idx in inference_idx_to_input_idx\n    ]\n\n    prompt_texts: List[str] = [\n        (\n            tokenizer.apply_chat_template(\n                conversation=messages, tokenize=False, add_generation_prompt=True\n            )  # type: ignore\n        )\n        + \"<think>\\n\"\n        for messages in list_of_messages\n    ]\n    # print(prompt_texts)\n\n    print(\"get_patch_string\", [count_tokens(text) for text in prompt_texts])\n    request_outputs: list[RequestOutput] = llm.generate(\n        prompt_texts, sampling_params=sampling_params\n    )\n    response_texts_from_inference: List[str] = [\n        request_output.outputs[0].text for request_output in request_outputs\n    ]\n    print(\n        \"get_patch_string\",\n        [count_tokens(text) for text in response_texts_from_inference],\n    )\n    completion_texts_from_inference = [\n        prompt_text + response_text\n        for prompt_text, response_text in zip(\n            prompt_texts, response_texts_from_inference\n        )\n    ]\n    patch_strings_from_inference: List[Optional[str]] = [\n        extract_patch_string(response_text)\n        for response_text in response_texts_from_inference\n    ]\n\n    completion_texts: list[str] = [\"\" for _ in file_content_strings]\n    patch_strings: List[Optional[str]] = [None for _ in file_content_strings]\n    for inference_idx, (completion_text, patch_string) in enumerate(\n        zip(completion_texts_from_inference, patch_strings_from_inference)\n    ):\n        input_idx = inference_idx_to_input_idx[inference_idx]\n        completion_texts[input_idx] = completion_text\n        patch_strings[input_idx] = patch_string\n\n    return completion_texts, patch_strings","metadata":{"_uuid":"69511c48-00b6-4d38-bd78-e1fe14891450","_cell_guid":"d487fae6-778a-474f-a9cb-a97403c0033a","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nverifying_prompt: str = (\n    \"\"\"\nThis is the problem statement.\n\n{problem_statement}\n\nThese are the files that is thought to be relevant, which may not be complete.\n\n{file_content_string}\n\nThis is the proposed patch to fix the problem.\n\n{patch_string}\n\nEvaluate whether the patch works\n- The patch fully fixes the problem described in the problem statement.\n- The patch does not cause side effects and make any other tests fail.\n\nEnd your response with exactly either of\n- <label>Yes</label>, this fixes the problem.\n- <label>No</label>, this does not fix the problem.\n\nReminder\n- Only evaluate, do not provide suggestion on how to fix.\n- Remember to write exactly either of <label>Yes</label> or <label>No</label> in the last line\n\"\"\".strip()\n)\n\n\nfrom functools import cache\n\n\n@cache\ndef is_valid_patch_format(patch_string: str) -> bool:\n    \"\"\"\n    A quick check to confirm if a patch could be valid.\n    \"\"\"\n    if not isinstance(patch_string, str):\n        return False\n    try:\n        patch_set = unidiff.PatchSet(patch_string)\n        if len(patch_set) == 0:\n            return False\n    except Exception:\n        return False\n    return True\n\n\n@cache\ndef patch_dry_run_succeeds(\n    patch_string: str, repo_path: str = REPO_PATH, timeout: int = 60\n) -> bool:\n    \"\"\"\n    A robust check if the patch will proceed without any errors.\n    Should be run after `is_valid_patch_format()`: the patch\n    command can hang if the inputs are sufficiently invalid.\n\n    Args:\n        patch_path: Path to a file containing the patch.\n        repo_path: Path to the directory to be patched.\n        timeout: Number of seconds before the dry run will be cancelled.\n    \"\"\"\n    with open(\"patch.txt\", \"w\") as f:\n        f.write(patch_string)\n    patch_path = \"/kaggle/working/patch.txt\"\n\n    cmd = f\"patch --quiet --dry-run -p1 -i {patch_path} -d {repo_path}\"\n    try:\n        subprocess.run(cmd, shell=True, check=True, timeout=timeout)\n        return True\n    except subprocess.CalledProcessError:\n        return False\n\n\ndef get_verification(\n    problem_statement: str,\n    file_content_strings: List[str],\n    patch_strings: List[Optional[str]],\n    repo_path: str,\n) -> Tuple[List[List[str]], List[List[bool]]]:\n    assert len(file_content_strings) == len(patch_strings)\n    sampling_params: SamplingParams = SamplingParams(\n        temperature=0.6,  # randomness of the sampling\n        min_p=0.01,\n        skip_special_tokens=True,  # Whether to skip special tokens in the output\n        max_tokens=MAX_TOKENS,\n    )\n\n    inference_idx_to_input_idx: list[int] = [\n        input_idx\n        for _ in range(VALIDATION_COPY_COUNT)\n        for input_idx, patch_string in enumerate(patch_strings)\n        if patch_string is not None\n        and is_valid_patch_format(patch_string)\n        and patch_dry_run_succeeds(patch_string, repo_path)\n    ]\n    print(inference_idx_to_input_idx)\n\n    list_of_messages: List[List[Dict[str, str]]] = [\n        [\n            {\n                \"role\": \"user\",\n                \"content\": verifying_prompt.format(\n                    problem_statement=problem_statement[:20_000],\n                    file_content_string=file_content_strings[input_idx][:30_000],\n                    patch_string=patch_strings[input_idx],\n                ),\n            },\n        ]\n        for input_idx in inference_idx_to_input_idx\n    ]\n\n    prompt_texts: List[str] = [\n        (\n            tokenizer.apply_chat_template(\n                conversation=messages, tokenize=False, add_generation_prompt=True\n            )  # type: ignore\n        )\n        + \"<think>\\n\"\n        for messages in list_of_messages\n    ]\n    # print(prompt_texts)\n\n    print(\"get_verification\", [count_tokens(text) for text in prompt_texts])\n    request_outputs: list[RequestOutput] = llm.generate(\n        prompt_texts, sampling_params=sampling_params\n    )\n    response_texts: List[str] = [\n        request_output.outputs[0].text for request_output in request_outputs\n    ]\n    print(\"get_verification\", [count_tokens(text) for text in response_texts])\n\n    completion_texts = [\n        prompt_text + response_text\n        for prompt_text, response_text in zip(prompt_texts, response_texts)\n    ]\n    judgments_flattened: List[bool] = [\n        \"<label>Yes</label>\" in response_text for response_text in response_texts\n    ]\n    print(judgments_flattened)\n\n    judgments_aggregated: List[List[bool]] = [[] for _ in file_content_strings]\n    completion_text_aggregated: List[List[str]] = [[] for _ in patch_strings]\n    for inference_idx, (completion_text, judgement) in enumerate(\n        zip(completion_texts, judgments_flattened)\n    ):\n        input_idx = inference_idx_to_input_idx[inference_idx]\n        completion_text_aggregated[input_idx].append(completion_text)\n        judgments_aggregated[input_idx].append(judgement)\n    print(judgments_aggregated)\n\n    return completion_text_aggregated, judgments_aggregated","metadata":{"_uuid":"6e92f7ca-1e80-45d8-8fbb-9d17debe2ab6","_cell_guid":"c7239b62-1e48-45ca-8ef8-c701ebec687e","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import unidiff\nimport subprocess\n\n\ndef choose_patch_string(\n    patch_strings: list[Optional[str]],\n    judgments_aggregated: List[List[bool]],\n    repo_path: str,\n) -> tuple[list[int], Optional[str]]:\n    best_score = -1\n    best_patch_string = None\n\n    scores = []\n    for judgments, patch_string in zip(judgments_aggregated, patch_strings):\n\n        if patch_string is None:\n            score = -103\n            scores.append(score)\n            continue\n\n        if not is_valid_patch_format(patch_string):\n            score = -102\n            scores.append(score)\n            continue\n\n        if not patch_dry_run_succeeds(patch_string, repo_path):\n            score = -101\n            scores.append(score)\n            continue\n\n        score = judgments.count(True)\n        scores.append(score)\n\n        if score > best_score:\n            best_score = score\n            best_patch_string = patch_string\n\n    return scores, best_patch_string","metadata":{"_uuid":"16cae032-ec07-427c-9f97-14ee6032f9b7","_cell_guid":"be966b8c-14b0-4c59-a77d-15566f315988","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport re\nfrom collections import defaultdict\n\n# Helper function to get all Python files in the repo\ndef get_python_files(repo_path):\n    python_files = []\n    for root, _, files in os.walk(repo_path):\n        for file in files:\n            if file.endswith(\".py\"):\n                python_files.append(os.path.join(root, file))\n    return python_files\n\n# Function to find all patterns of a substring across the entire file content\ndef find_code_patterns_in_repo(code_pattern, repo_path):\n    python_files = get_python_files(repo_path)\n    found_files = []  # List to store paths of files where the pattern is found\n    found = False  # Flag to check if any match is found\n    matched_lines = defaultdict(list)\n    # Compile regex pattern for faster matching\n    pattern = re.compile(re.escape(code_pattern.strip()))\n\n    # Debug: Print the regex pattern we are searching for\n    print(f\"Searching for pattern: '{code_pattern.strip()}'\\n\")\n\n    for file in python_files:\n        with open(file, 'r', encoding='utf-8') as f:\n            # Read the entire file as a single string\n            file_content = f.read()\n\n        # Find all occurrences of the pattern in the file content\n        matches = list(pattern.finditer(file_content))\n\n        if matches:\n            found = True\n            found_files.append(file)  # Add the file path to the result list\n            print(f\"\\nFound in file: {file}\")\n            # Split the file content into lines for line number mapping\n            lines = file_content.splitlines()\n\n            # Print each match with its line number and context\n            for match in matches:\n                # Get the start position of the match\n                start_pos = match.start()\n                # Find the line number by counting newlines up to the match position\n                line_number = file_content.count('\\n', 0, start_pos) + 1\n                matched_line = lines[line_number - 1].strip()\n                # Print the matched line with line number\n                print(f\"Line {line_number}: {matched_line}\")\n                # Store the matched line and its line number\n                matched_lines[file].append((line_number, matched_line))\n\n    if not found:\n        print(\"The pattern was not found in any file.\")\n    else:\n        print(\"\\nSearch complete.\")\n    \n    return found_files,matched_lines  # Return list of files where the pattern was found\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to find matching lines based on the condition\ndef find_adjacent_matches(matched_lines1, matched_lines2):\n    # Dictionary to store filtered results\n    filtered_matches = {}\n\n    # Iterate over files that exist in both matched_lines1 and matched_lines2\n    for file in matched_lines1:\n        if file in matched_lines2:\n            # Extract lines and line numbers for both sets\n            lines1 = matched_lines1[file]\n            lines2 = matched_lines2[file]\n\n            # Initialize list to store matched results for current file\n            filtered_matches[file] = []\n\n            # Iterate over lines in matched_lines1\n            for line_num1, line1 in lines1:\n                # Check if any line in matched_lines2 is 1 line before or after the current line in matched_lines1\n                for line_num2, line2 in lines2:\n                    if line_num2 == line_num1 - 1 or line_num2 == line_num1 + 1:\n                        # Store the result if condition is met\n                        filtered_matches[file].append((line_num2, line2, line_num1, line1))\n\n    # Print the results\n    if filtered_matches:\n        print(\"\\nFiltered Matches (Where matched_lines2 is 1 line before or after matched_lines1):\")\n        for file, matches in filtered_matches.items():\n            print(f\"\\nIn file: {file}\")\n            for line_num2, line2, line_num1, line1 in matches:\n                print(f\"Line {line_num2}: {line2}\")\n                print(f\"Line {line_num1}: {line1}\")\n    else:\n        print(\"No adjacent matches found.\")\n    \n    return filtered_matches  # Return the filtered matches if needed","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# from collections import defaultdict\n\n# def extract_code_context(traced_lines, file_path, total_lines, line_number):\n\n#     # Read the entire source code from the file\n#     with open(file_path, \"r\", encoding=\"utf-8\") as f:\n#         file_lines = f.readlines()\n\n#     file_lines = [line.rstrip(\"\\n\") for line in file_lines]  # Remove newlines\n\n#     # Extract the **Main Window** (4 lines above, code line, 4 lines below)\n#     start_window = max(0, line_number - 5)  # 4 lines above\n#     end_window = min(len(file_lines), line_number + 4)  # 4 lines below\n\n#     main_window_lines = [(i + 1, file_lines[i]) for i in range(start_window, end_window)]\n    \n#     # Lines to be added in the final context\n#     extracted_lines = set(range(start_window, end_window))  # Track used line numbers\n#     formatted_context = []\n\n#     # Add File Path at the Start\n#     formatted_context.append(f\"### File Path: {file_path} ###\\n\")\n\n#     # Format the main window first\n#     formatted_context.append(\"### Main Window Context ###\\n\")\n#     for lineno, code in main_window_lines:\n#         formatted_context.append(f\"{lineno}: {code}\")\n\n#     # Remaining lines needed after main window\n#     remaining_lines_needed = total_lines - len(main_window_lines)\n\n#     # Fetch **Tracing Lines** (Before the start of main window)\n#     tracing_windows = []\n#     if file_path in traced_lines:\n#         # Get tracing points before start_window (i.e., before line_number - 4)\n#         for lineno, code in traced_lines[file_path]:\n#             if lineno < start_window:\n#                 # Ensure we take 1 line before and 1 line after if possible\n#                 tracing_start = max(0, lineno - 1)\n#                 tracing_end = min(len(file_lines), lineno + 2)\n\n#                 # Ensure we don't add already included lines (to avoid duplication)\n#                 tracing_range = set(range(tracing_start, tracing_end))\n#                 if not tracing_range.intersection(extracted_lines):  # No overlap\n#                     extracted_lines.update(tracing_range)\n#                     tracing_windows.append([(i + 1, file_lines[i]) for i in range(tracing_start, tracing_end)])\n\n#                 # Stop if we reach the required total lines\n#                 if len(extracted_lines) >= total_lines:\n#                     break\n\n#     # Sort tracing windows in **descending order** to print nearest tracing first\n#     tracing_windows.sort(reverse=True, key=lambda x: x[0][0])  # Sort by first line number\n\n#     # Add tracing windows to formatted output\n#     formatted_context.append(\"\\n### Backward Tracing Context ###\\n\")\n#     for tracing_block in tracing_windows:\n#         formatted_context.append(\"---\")\n#         for lineno, code in tracing_block:\n#             formatted_context.append(f\"{lineno}: {code}\")\n#         formatted_context.append(\"---\")\n\n#     # Ensure we have exactly `total_lines`, extend further if needed\n#     if len(extracted_lines) < total_lines:\n#         additional_needed = total_lines - len(extracted_lines)\n#         extra_start = max(0, start_window - additional_needed)\n#         extra_lines = [(i + 1, file_lines[i]) for i in range(extra_start, start_window) if i not in extracted_lines]\n\n#         if extra_lines:\n#             formatted_context.append(\"\\n### Extra Lines for Context ###\\n\")\n#             for lineno, code in extra_lines:\n#                 formatted_context.append(f\"{lineno}: {code}\")\n\n#     return \"\\n\".join(formatted_context)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom collections import defaultdict\n\ndef extract_code_context(traced_lines, file_path, total_lines, code_line):\n    \"\"\"\n    Extracts a well-formatted code context for an LLM.\n    \n    - First extracts a 9-line window around the `code_line` in `file_path`.\n    - Then fetches backward tracing from `traced_lines`, ensuring total_lines is reached.\n    - Formats everything correctly with indentation and line numbers, printing nearest tracing first.\n    \n    Parameters:\n        traced_lines (dict): Traced lines across the repository.\n        file_path (str): Path of the file to extract context from.\n        total_lines (int): Total number of code lines required in context.\n        code_line (str): The exact code line whose tracings were found.\n\n    Returns:\n        str: Formatted context with indentation and line numbers.\n    \"\"\"\n\n    # Read the entire source code from the file\n    with open(file_path, \"r\", encoding=\"utf-8\") as f:\n        file_lines = f.readlines()\n\n    file_lines = [line.rstrip(\"\\n\") for line in file_lines]  # Remove newlines\n\n    # Find the line number of the given `code_line`\n    try:\n        line_number = file_lines.index(code_line) + 1  # Convert 0-based index to 1-based\n    except ValueError:\n        return f\"Error: The given code line was not found in {file_path}\"\n\n    # Extract the **Main Window** (4 lines above, code line, 4 lines below)\n    start_window = max(0, line_number - 5)  # 4 lines above\n    end_window = min(len(file_lines), line_number + 4)  # 4 lines below\n\n    main_window_lines = [(i + 1, file_lines[i]) for i in range(start_window, end_window)]\n    \n    # Lines to be added in the final context\n    extracted_lines = set(range(start_window, end_window))  # Track used line numbers\n    formatted_context = []\n\n    # Add File Path at the Start\n    formatted_context.append(f\"### File Path: {file_path} ###\\n\")\n\n    # Format the main window first\n    formatted_context.append(\"### Main Window Context ###\\n\")\n    for lineno, code in main_window_lines:\n        formatted_context.append(f\"{lineno}: {code}\")\n\n    # Remaining lines needed after main window\n    remaining_lines_needed = total_lines - len(main_window_lines)\n\n    # Fetch **Tracing Lines** (Before the start of main window)\n    tracing_windows = []\n    if file_path in traced_lines:\n        # Get tracing points before start_window (i.e., before line_number - 4)\n        for lineno, code in traced_lines[file_path]:\n            if lineno < start_window:\n                # Ensure we take 1 line before and 1 line after if possible\n                tracing_start = max(0, lineno - 1)\n                tracing_end = min(len(file_lines), lineno + 2)\n\n                # Ensure we don't add already included lines (to avoid duplication)\n                tracing_range = set(range(tracing_start, tracing_end))\n                if not tracing_range.intersection(extracted_lines):  # No overlap\n                    extracted_lines.update(tracing_range)\n                    tracing_windows.append([(i + 1, file_lines[i]) for i in range(tracing_start, tracing_end)])\n\n                # Stop if we reach the required total lines\n                if len(extracted_lines) >= total_lines:\n                    break\n\n    # Sort tracing windows in **descending order** to print nearest tracing first\n    tracing_windows.sort(reverse=True, key=lambda x: x[0][0])  # Sort by first line number\n\n    # Add tracing windows to formatted output\n    formatted_context.append(\"\\n### Backward Tracing Context ###\\n\")\n    for tracing_block in tracing_windows:\n        formatted_context.append(\"---\")\n        for lineno, code in tracing_block:\n            formatted_context.append(f\"{lineno}: {code}\")\n        formatted_context.append(\"---\")\n\n    # Ensure we have exactly `total_lines`, extend further if needed\n    if len(extracted_lines) < total_lines:\n        additional_needed = total_lines - len(extracted_lines)\n        extra_start = max(0, start_window - additional_needed)\n        extra_lines = [(i + 1, file_lines[i]) for i in range(extra_start, start_window) if i not in extracted_lines]\n\n        if extra_lines:\n            formatted_context.append(\"\\n### Extra Lines for Context ###\\n\")\n            for lineno, code in extra_lines:\n                formatted_context.append(f\"{lineno}: {code}\")\n\n    return \"\\n\".join(formatted_context)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Predict function","metadata":{"_uuid":"0c7746fe-a75d-4f77-8cb3-e9d9dde86dbe","_cell_guid":"d117733c-784a-43c3-96f4-014299fa39c6","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def predict_inner(problem_statement: str, directory: str) -> Optional[str]:\n    is_valid_patch_format.cache_clear()\n    patch_dry_run_succeeds.cache_clear()\n\n    # directory_string = stringify_directory(directory)\n\n    # selection_completion_texts, file_queries = get_selection_query(\n    #     directory_string, problem_statement\n    # )\n\n    # file_content_strings: List[str] = [\n    #     fetch_file_contents(file_query) for file_query in file_queries\n    # ]\n    lines_list = getting_lines(problem_statement)\n    res_code1 = []\n    res_code2 = []\n    results = []\n    for ele in lines_list:\n        if ele:\n            code1 = ele[0]\n            code2 = ele[1]\n            files1, result1 = find_code_patterns_in_repo(code1,directory)\n            files2, result2 = find_code_patterns_in_repo(code2,directory)\n            res_code1.append(result1)\n            res_code2.append(result2)\n    for matched_lines1, matched_lines2 in zip(res_code1,res_code2):\n        result = find_adjacent_matches(matched_lines1,matched_lines2)\n        results.append(result)\n    file_names_list = [list(result.keys()) for result in results]\n    choosen_files=deciding_the_files(file_names_list,problem_statement)\n    node_results = []\n    for code in lines_list:\n        code_line1=code[0]\n        node_results.append(extract_nodes_from_line(code_line1))\n    tracing_results = []\n    for nodes in node_results:\n        tracing = trace_back_nodes(directory, nodes)\n        tracing_results.append(tracing)\n    file_content_strings = []\n    for i in range(len(tracing_results)):\n        traced_lines = tracing_results[i]\n        file_path = choosen_files[i]\n        code_line = lines_list[i][0]\n        file_content_strings.append(extract_code_context(traced_lines, file_path, total_lines = 15 , code_line))\n        \n    # deciding_the_files(files_list:List[str] , problem_statement: str)-> List[str]\n    patch_completion_texts, patch_strings = get_patch_string(\n        problem_statement, file_content_strings\n    )\n\n    verification_completion_texts_aggregated, judgments_aggregated = get_verification(\n        problem_statement, file_content_strings, patch_strings, directory\n    )\n\n    scores, patch_string = choose_patch_string(\n        patch_strings, judgments_aggregated, directory\n    )\n\n    if not os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"):\n        data = {\n            \"problem_statement\": [problem_statement] * len(file_queries),\n            \"selection_completion_text\": selection_completion_texts,\n            \"selection_completion_length\": [\n                count_tokens(completion_text)\n                for completion_text in selection_completion_texts\n            ],\n            \"file_query\": file_queries,\n            \"file_content_string\": file_content_strings,\n            \"patch_completion_text\": patch_completion_texts,\n            \"patch_completion_length\": [\n                count_tokens(completion_text)\n                for completion_text in patch_completion_texts\n            ],\n            \"patch_string\": patch_strings,\n        }\n\n        for copy_idx in range(VALIDATION_COPY_COUNT):\n            data[f\"verification_completion_text_{copy_idx}\"] = [\n                completion_texts[copy_idx] if completion_texts else None\n                for completion_texts in verification_completion_texts_aggregated\n            ]\n            data[f\"verification_completion_length_{copy_idx}\"] = [\n                count_tokens(completion_texts[copy_idx]) if completion_texts else None\n                for completion_texts in verification_completion_texts_aggregated\n            ]\n            data[f\"judgment_{copy_idx}\"] = [\n                judgments[copy_idx] if judgments else None\n                for judgments in judgments_aggregated\n            ]\n\n        data[\"judgment_count_true\"] = [\n            judgments.count(True) for judgments in judgments_aggregated\n        ]\n        data[\"score\"] = scores\n\n        pd.DataFrame(data).to_csv(\n            f\"{str(int(time.time() - start_time)).zfill(5)}.csv\", index=False\n        )\n\n    return patch_string","metadata":{"_uuid":"bfc7f6a8-4465-4729-ad94-15dd100bd3ae","_cell_guid":"4cd441a0-1ab8-4dbf-b55a-a1dd42bed643","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import io\nfrom typing import Optional, List\n\ninitial_predictions_left = 1000\npredictions_left = initial_predictions_left\nfirst_pred = True\n\ndef predict(\n    problem_statement: str,\n    repo_archive: io.BytesIO,\n    pip_packages_archive: io.BytesIO,\n    env_setup_cmds_templates: List[str],\n) -> Optional[str]:\n    \"\"\"Replace this function with your inference code.\n    Args:\n        problem_statement: The text of the git issue.\n        repo_archive: A BytesIO buffer path with a .tar containing the codebase that must be patched. The gateway will make this directory available immediately before this function runs.\n    \"\"\"\n    if first_pred:\n        initialize_llm()\n    first_pred = False\n    allowed_time[-1] += 6 * 60\n    if time.time() > allowed_time[-1]:\n        return None\n\n    global predictions_left\n    if predictions_left == 0:\n        return None\n\n    repo_path: str = REPO_PATH\n    if not os.path.exists(repo_path):\n        os.makedirs(repo_path)\n\n    setup(repo_archive, pip_packages_archive, env_setup_cmds_templates, repo_path)\n\n    patch_string = predict_inner(\n        problem_statement=problem_statement,\n        directory=repo_path,\n    )\n\n    if os.path.exists(repo_path):\n        shutil.rmtree(repo_path)\n\n    if not os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\") and not os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\") == \"Interactive\":\n        predictions_left = 0\n    \n    print(\"submitted patch_string\")\n    print(patch_string)\n\n    if patch_string is None:\n        return None\n\n    if os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"):\n        predictions_left -= 1\n\n    return patch_string","metadata":{"_uuid":"c43def43-2e77-4cca-bdf7-6d10c08815fc","_cell_guid":"7e0d9952-8792-4310-8d0c-43d823bd06b9","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Get predict data without server","metadata":{"_uuid":"d0174590-1085-45ff-b098-5122486f42ec","_cell_guid":"594d02e7-7ef2-4aa2-8c2b-01bfdc4f66eb","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import os\nimport zipfile\n\n# !mkdir -p /kaggle/tmp/konwinski-prize-alt\nos.makedirs(\"/kaggle/tmp/konwinski-prize-alt\", exist_ok=True)\n\n# !unzip -q -o /kaggle/input/konwinski-prize/data.a_zip -d /kaggle/tmp/konwinski-prize-alt/ 2>/dev/null || true\ntry:\n    with zipfile.ZipFile(\"/kaggle/input/konwinski-prize/data.a_zip\", \"r\") as zip_ref:\n        zip_ref.extractall(\"/kaggle/tmp/konwinski-prize-alt/\")\nexcept:\n    pass","metadata":{"_uuid":"59c8d09c-c2c1-4889-9ace-05c3a038d52b","_cell_guid":"668b6305-e306-43bb-8156-5c49f9784ad3","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ntemp_data_dir = \"/kaggle/tmp/konwinski-prize-alt/data/\"\nmetadata_path = os.path.join(temp_data_dir, \"data.parquet\")\npip_packages_dir = os.path.join(temp_data_dir, \"pip_packages\")\nrepo_config_dir = os.path.join(temp_data_dir, \"repo_configs\")\nrepo_dir = os.path.join(temp_data_dir, \"repos\")\n\nfrom kprize_setup.kprize.evaluation.kprize_env_handler import KprizeEnvHandler\n\n\ndef get_problem(problem_index: int) -> tuple[str, io.BytesIO, io.BytesIO, list[str]]:\n    df = pd.read_parquet(\"/kaggle/tmp/konwinski-prize-alt/data/data.parquet\")\n    problem_statement: str = df[\"problem_statement\"][problem_index]\n\n    repo_path = os.path.join(repo_dir, f\"repo__{df['instance_id'][problem_index]}\")\n    pip_packages_path = os.path.join(pip_packages_dir, df[\"instance_id\"][problem_index])\n\n    import shutil\n    import tempfile\n\n    with tempfile.TemporaryDirectory() as tmpdir:\n        # instance repo\n        shutil.make_archive(os.path.join(tmpdir, \"a_repo\"), \"tar\", repo_path)\n        with open(os.path.join(tmpdir, \"a_repo.tar\"), \"rb\") as f:\n            repo_buffer = io.BytesIO(f.read())\n        # instance pip packages\n        shutil.make_archive(\n            os.path.join(tmpdir, \"a_pip_packages_dir\"), \"tar\", pip_packages_path\n        )\n        with open(os.path.join(tmpdir, \"a_pip_packages_dir.tar\"), \"rb\") as f:\n            pip_packages_buffer = io.BytesIO(f.read())\n\n    repo_config_path = os.path.join(\n        repo_config_dir, df[\"instance_id\"][problem_index].rsplit(\"-\", maxsplit=1)[0]\n    )\n    env_setup_cmd_templates = KprizeEnvHandler.get_env_setup_cmds_templates(\n        repo_config_path\n    )\n    return problem_statement, repo_buffer, pip_packages_buffer, env_setup_cmd_templates","metadata":{"_uuid":"358eb3ee-1500-4784-b5f7-44489a3f5e5a","_cell_guid":"a2da7861-bb26-4b3a-86ed-5d2b7c0ec33b","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"demo_problem_index: int = 0\n\nif os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\") == \"Interactive\" and not os.getenv(\n    \"KAGGLE_IS_COMPETITION_RERUN\"\n):\n    problem_statement, repo_buffer, pip_packages_buffer, env_setup_cmd_templates = (\n        get_problem(problem_index=demo_problem_index)\n    )\n\n    print(problem_statement)\n    print(len(list(repo_buffer)))\n    print(len(list(repo_buffer)))\n    print(len(list(pip_packages_buffer)))\n    print(len(list(pip_packages_buffer)))\n    print(env_setup_cmd_templates)","metadata":{"_uuid":"6c027058-d972-420d-bd4f-6fb7b4f263b8","_cell_guid":"a088661b-d627-47c4-89ba-f275bb96d5d0","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\") == \"Interactive\" and not os.getenv(\n    \"KAGGLE_IS_COMPETITION_RERUN\"\n):\n    predictions_left = 1\n    problem_statement, repo_buffer, pip_packages_buffer, env_setup_cmd_templates = (\n        get_problem(problem_index=demo_problem_index)\n    )\n    patch_string = predict(\n        problem_statement, repo_buffer, pip_packages_buffer, env_setup_cmd_templates\n    )","metadata":{"_uuid":"e6d8f807-9248-4953-acf2-96610df3b98e","_cell_guid":"775d0efc-23a9-4202-9f53-94f6ac42a415","trusted":true,"collapsed":false,"_kg_hide-output":true,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if (\n    os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\") == \"Interactive\"\n    and not os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\")\n    and patch_string is not None\n):\n    import polars as pl\n\n    df = pl.read_parquet(\"/kaggle/tmp/konwinski-prize-alt/data/data.parquet\")\n\n    import kaggle_evaluation.konwinski_prize_gateway\n\n    k_prize_gateway = kaggle_evaluation.konwinski_prize_gateway.KPrizeGateway()\n    k_prize_gateway.unpack_data_paths()\n\n    results = k_prize_gateway._evaluate_instance(\n        instance=df.row(demo_problem_index, named=True),\n        patch=patch_string,\n    )\n\n    from collections import Counter\n\n    print(\n        demo_problem_index, Counter(result.unit_test_outcome for result in results[1:])\n    )","metadata":{"_uuid":"5292db7c-2f74-4e87-b9c5-d68c061cc0e9","_cell_guid":"65bc0028-8bd7-48e2-8f3a-e319889bdc08","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if (\n    os.getenv(\"KAGGLE_KERNEL_RUN_TYPE\") == \"Interactive\"\n    and not os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\")\n    and patch_string is not None\n):\n    from kaggle_evaluation.konwinski_prize_gateway import UnitTestOutcome\n\n    for result in results[1:]:\n        if result.unit_test_outcome != UnitTestOutcome.PASSED:\n            print(result.test_name)\n            print(result.fail_description)","metadata":{"_uuid":"08300287-f22c-477a-83b9-e092ee467778","_cell_guid":"1dcd6015-a9f3-4560-8021-a18c11660cee","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"When your notebook is run on the hidden test set, inference_server.serve must be called within 15 minutes of the notebook starting or the gateway will throw an error. If you need more than 15 minutes to load your model you can do so during the very first predict call, which does not have the usual 30 minute response deadline.","metadata":{"_uuid":"5e6e67cc-b3b3-4658-ae73-72d6214aeb56","_cell_guid":"15d7d77d-480c-464b-8cc8-33adf9c405ed","trusted":true,"collapsed":false,"papermill":{"duration":0.001889,"end_time":"2024-12-11T03:22:08.856283","exception":false,"start_time":"2024-12-11T03:22:08.854394","status":"completed"},"tags":[],"jupyter":{"outputs_hidden":false}}},{"cell_type":"markdown","source":"# Evaluation with inference server","metadata":{"_uuid":"7d254b31-d130-45b6-8173-558beb10c31d","_cell_guid":"153c9dfe-9a51-4428-b594-d591b38e274c","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"predictions_left = initial_predictions_left\n\nelapsed_time = time.time() - global_start_time  # Calculate elapsed time\nprint(f\"Total elapsed time since start: {elapsed_time:.2f} seconds\")","metadata":{"_uuid":"e81dc05e-cf68-4e6e-bb2d-2afd6d2a1968","_cell_guid":"8108d3d6-025a-4d4b-81bb-71befac78363","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = (\n    kaggle_evaluation.konwinski_prize_inference_server.KPrizeInferenceServer(\n        get_number_of_instances, predict\n    )\n)\n\nif os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            \"/kaggle/input/konwinski-prize/\",  # Path to the entire competition dataset\n            \"/kaggle/tmp/konwinski-prize/\",  # Path to a scratch directory for unpacking data.a_zip.\n        )  # type: ignore\n    )","metadata":{"_uuid":"71fea716-cca0-478b-8d29-383a59254aaf","_cell_guid":"e2ec01ec-e95a-40be-8074-99e382e1a9a4","trusted":true,"collapsed":false,"_kg_hide-output":true,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"_uuid":"16e8618e-fbfa-43ef-a0fc-5ab227edea0b","_cell_guid":"0ce34cf1-7275-4373-a521-5cec87057e15","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null}]}