{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaL4","dataSources":[{"sourceId":84795,"databundleVersionId":10934030,"sourceType":"competition"},{"sourceId":10503391,"sourceType":"datasetVersion","datasetId":6502489},{"sourceId":10580610,"sourceType":"datasetVersion","datasetId":6548001},{"sourceId":10580755,"sourceType":"datasetVersion","datasetId":6548046},{"sourceId":10588937,"sourceType":"datasetVersion","datasetId":6553359},{"sourceId":10599100,"sourceType":"datasetVersion","datasetId":6560473}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-01T03:17:55.599065Z","iopub.execute_input":"2025-02-01T03:17:55.599399Z","iopub.status.idle":"2025-02-01T03:17:55.646221Z","shell.execute_reply.started":"2025-02-01T03:17:55.599373Z","shell.execute_reply":"2025-02-01T03:17:55.645614Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Path to the competition data directory\ncompetition_data_path = \"/kaggle/input/konwinski-prize\"\nprint(\"Files in the competition data folder:\")\nprint(os.listdir(competition_data_path))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T03:18:00.710634Z","iopub.execute_input":"2025-02-01T03:18:00.71091Z","iopub.status.idle":"2025-02-01T03:18:00.714823Z","shell.execute_reply.started":"2025-02-01T03:18:00.710889Z","shell.execute_reply":"2025-02-01T03:18:00.714199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\nimport os\nimport pandas as pd\n\n# Paths\nzip_file_path = \"/kaggle/input/konwinski-prize/data.a_zip\"\nextract_to_path = \"/kaggle/working/extracted_data\"\n\n# Recreate the working folder and extract\nif not os.path.exists(extract_to_path):\n    os.makedirs(extract_to_path)\n\nprint(\"Extracting data...\")\nwith zipfile.ZipFile(zip_file_path, 'r') as zip_ref:\n    zip_ref.extractall(extract_to_path)\n\n# Check contents of the extracted folder\nprint(\"Contents of extracted folder:\", os.listdir(extract_to_path))\n\n# Check contents of 'data' subfolder\ndata_folder = os.path.join(extract_to_path, \"data\")\nprint(\"Contents of 'data' subfolder:\", os.listdir(data_folder))\n\n# Read the parquet file\nparquet_file_path = os.path.join(data_folder, \"data.parquet\")  # Adjust if necessary\ndf = pd.read_parquet(parquet_file_path)\n\nprint(f\"Total rows in dataset: {len(df)}\")\nprint(df.head(3))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T03:18:02.587664Z","iopub.execute_input":"2025-02-01T03:18:02.587944Z","iopub.status.idle":"2025-02-01T03:18:05.785663Z","shell.execute_reply.started":"2025-02-01T03:18:02.587923Z","shell.execute_reply":"2025-02-01T03:18:05.785012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print the column names and data types\nprint(\"\\nColumn Names and Data Types:\")\nprint(df.dtypes)\n\n# Print the total number of rows\nprint(f\"\\nTotal number of rows: {len(df)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-01T03:18:07.512562Z","iopub.execute_input":"2025-02-01T03:18:07.512805Z","iopub.status.idle":"2025-02-01T03:18:07.517236Z","shell.execute_reply.started":"2025-02-01T03:18:07.512785Z","shell.execute_reply":"2025-02-01T03:18:07.516643Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"!pip uninstall -y bitsandbytes\n!pip install -U bitsandbytes\n!pip install accelerate\n!pip install transformers\n!pip install bitsandbytes-cuda11x\n","metadata":{"execution":{"iopub.status.busy":"2025-01-16T03:14:01.227766Z","iopub.execute_input":"2025-01-16T03:14:01.228103Z","iopub.status.idle":"2025-01-16T03:14:15.119982Z","shell.execute_reply.started":"2025-01-16T03:14:01.22807Z","shell.execute_reply":"2025-01-16T03:14:15.118976Z"}}},{"cell_type":"markdown","source":"!pip install peft","metadata":{"execution":{"iopub.status.busy":"2025-01-16T03:15:33.696084Z","iopub.execute_input":"2025-01-16T03:15:33.696423Z","iopub.status.idle":"2025-01-16T03:15:37.102696Z","shell.execute_reply.started":"2025-01-16T03:15:33.696392Z","shell.execute_reply":"2025-01-16T03:15:37.101515Z"}}},{"cell_type":"markdown","source":"!pip install /kaggle/input/konwinski-prize/kprize_setup/kprize-1.0.0-py3-none-any.whl --find-links /kaggle/input/konwinski-prize/kprize_setup/pip_packages\n","metadata":{"execution":{"iopub.status.busy":"2025-01-17T00:43:00.293896Z","iopub.execute_input":"2025-01-17T00:43:00.294213Z","iopub.status.idle":"2025-01-17T00:43:06.216342Z","shell.execute_reply.started":"2025-01-17T00:43:00.29419Z","shell.execute_reply":"2025-01-17T00:43:06.215119Z"}}},{"cell_type":"markdown","source":"import os\nfrom kaggle_evaluation.konwinski_prize_inference_server import KPrizeInferenceServer\nimport io\nimport shutil\n\n# Define a lightweight, dummy prediction function\ndef generate_dummy_diff(problem_statement: str) -> str:\n    \"\"\"\n    Generate a dummy Git diff for the given problem statement.\n    This is a placeholder function to reduce computational load.\n    \"\"\"\n    # Example of a trivial dummy diff\n    return \"diff --git a/dummy_file.py b/dummy_file.py\\nindex 0000000..1111111 100644\\n--- a/dummy_file.py\\n+++ b/dummy_file.py\\n@@ -1 +1 @@\\n- print('Hello, World!')\\n+ print('Hello, Kaggle!')\"\n\n# Define the `get_number_of_instances` function\ninstance_count = None\n\ndef get_number_of_instances(num_instances: int):\n    \"\"\"\n    Callback function to set the number of instances to process.\n    \"\"\"\n    global instance_count\n    instance_count = min(num_instances, 1)  # Limit to processing only one instance\n\n# Define the `predict` function with the correct signature\ndef predict(problem_statement: str, repo_archive: io.BytesIO, env_setup_cmd_templates=None, repo_config_path=None) -> str:\n    \"\"\"\n    Predict the Git diff for a single instance.\n    \n    Args:\n        problem_statement: The text of the GitHub issue.\n        repo_archive: A BytesIO object containing the repository archive (.tar).\n        env_setup_cmd_templates: Additional commands for setting up the environment (unused in dummy).\n        repo_config_path: Path to the repository configuration (unused in dummy).\n        \n    Returns:\n        str: The generated Git diff.\n    \"\"\"\n    # Unpack the repository archive\n    with open(\"repo_archive.tar\", \"wb\") as f:\n        f.write(repo_archive.read())\n    repo_path = \"repo\"\n    if os.path.exists(repo_path):\n        shutil.rmtree(repo_path)\n    shutil.unpack_archive(\"repo_archive.tar\", extract_dir=repo_path)\n    os.remove(\"repo_archive.tar\")\n\n    # Return a dummy diff\n    diff = generate_dummy_diff(problem_statement)\n    return diff\n\n# Set up the inference server\ninference_server = KPrizeInferenceServer(\n    get_number_of_instances,  # Register the get_number_of_instances function\n    predict                   # Register the predict function\n)\n\n# Handle submission or local testing\nis_submission = os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\")\n\nif is_submission:\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            \"/kaggle/input/konwinski-prize/\",  # Path to the competition dataset\n            \"/kaggle/tmp/konwinski-prize/\",   # Path for temporary files\n        )\n    )\n","metadata":{"execution":{"iopub.status.busy":"2025-01-18T01:14:17.647499Z","iopub.execute_input":"2025-01-18T01:14:17.647871Z","iopub.status.idle":"2025-01-18T01:14:29.995842Z","shell.execute_reply.started":"2025-01-18T01:14:17.647843Z","shell.execute_reply":"2025-01-18T01:14:29.994335Z"}}},{"cell_type":"markdown","source":"Step 3: Prepare Your Input Data\nSWE-Llama expects inputs in the following format:\n\nIssue text: The GitHub issue description.\nRelevant file contexts: Codebase snippets retrieved using BM25 retrieval.\nExample Patch: (Optional) You can include an example patch as part of the input prompt.\nHere’s how to adapt your Kaggle data for SWE-Llama:","metadata":{}},{"cell_type":"markdown","source":"CORRECT BM25 working","metadata":{}},{"cell_type":"code","source":"#!pip install rank_bm25\n!pip install /kaggle/input/rank-bm25/rank_bm25-0.2.2-py3-none-any.whl\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-02T05:34:43.754773Z","iopub.execute_input":"2025-02-02T05:34:43.755087Z","iopub.status.idle":"2025-02-02T05:34:47.361549Z","shell.execute_reply.started":"2025-02-02T05:34:43.755063Z","shell.execute_reply":"2025-02-02T05:34:47.360702Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"BM25 top files + bm25-llama7b to predict correct files.","metadata":{}},{"cell_type":"markdown","source":"import os\nimport gc\nimport pandas as pd\nimport torch\nfrom transformers import AutoTokenizer, AutoModelForCausalLM\nimport zipfile\nfrom glob import glob\nfrom rank_bm25 import BM25Okapi\nimport re\n\n# **Paths**\nMODEL_DIR_7B = \"/kaggle/input/bm25-llama-7b\"\nZIP_FILE_PATH = \"/kaggle/input/konwinski-prize/data.a_zip\"\nEXTRACTED_DATA_DIR = \"/kaggle/working/extracted_data\"\nPARQUET_FILE = os.path.join(EXTRACTED_DATA_DIR, \"data\", \"data.parquet\")\nCOMP_REPOS_DIR = os.path.join(EXTRACTED_DATA_DIR, \"data\", \"repos\")\n\n# **Set PyTorch CUDA Memory Config**\nos.environ[\"PYTORCH_CUDA_ALLOC_CONF\"] = \"expandable_segments:True\"\n\n# **Extract ZIP file**\nif not os.path.exists(EXTRACTED_DATA_DIR):\n    with zipfile.ZipFile(ZIP_FILE_PATH, 'r') as zip_ref:\n        zip_ref.extractall(EXTRACTED_DATA_DIR)\n\n# **Load dataset**\nkprize_df = pd.read_parquet(PARQUET_FILE)\n\n# **Clear GPU memory**\ndef clear_gpu_memory():\n    gc.collect()\n    torch.cuda.empty_cache()\n\n# **BM25 Setup**\ndef prepare_bm25_documents(repo_path, min_tokens=1, max_tokens=100000, exclude_patterns=None):\n    \"\"\"\n    Prepare documents for BM25 by reading repository files.\n    Excludes specific files and directories based on provided patterns.\n    \"\"\"\n    if exclude_patterns is None:\n        exclude_patterns = [\n            r'\\.git',       # Exclude .git directories\n            r'\\.pylint',    # Exclude .pylint files\n            r'\\.lock$',     # Exclude .lock files\n            r'\\.log$',      # Exclude .log files\n            r'__pycache__', # Exclude __pycache__ directories\n            r'\\.DS_Store',  # Exclude macOS system files\n            r'\\.png$',      # Exclude image files\n            r'\\.jpg$',      # Exclude image files\n            r'\\.jpeg$',     # Exclude image files\n            r'\\.zip$',      # Exclude compressed files\n            r'\\.tar$',      # Exclude compressed files\n            r'\\.egg$',      # Exclude Python egg files\n        ]\n\n    # Compile regex patterns\n    exclude_regex = [re.compile(pattern) for pattern in exclude_patterns]\n\n    documents = []\n    file_paths = []\n    for root, _, files in os.walk(repo_path):\n        for file in files:\n            file_path = os.path.join(root, file)\n\n            # Check if the file or directory should be excluded\n            if any(regex.search(file_path) for regex in exclude_regex):\n                continue\n\n            try:\n                with open(file_path, \"r\", encoding=\"utf-8\") as f:\n                    content = f.read().strip()  # Remove leading/trailing spaces\n                    token_count = len(content.split())  # Count words\n                    if token_count < min_tokens or token_count > max_tokens:\n                        continue  # Skip very short or extremely large files\n                    augmented_content = f\"<file: {file_path}>\\n{content}\"\n                    documents.append(augmented_content)\n                    file_paths.append(file_path)\n            except (UnicodeDecodeError, Exception):\n                pass  # Suppress warnings\n\n    if not documents:\n        raise ValueError(\"[ERROR] No valid documents found for BM25 retrieval!\")\n    return documents, file_paths\n\n\ndef bm25_retrieve_limited(query, documents, file_paths, max_tokens=100000, min_files=10):\n    \"\"\"\n    Retrieve the most relevant files using BM25.\n    \"\"\"\n    tokenized_docs = [doc.split() for doc in documents]\n    bm25 = BM25Okapi(tokenized_docs)\n    scores = bm25.get_scores(query.split())\n\n    ranked_indices = sorted(range(len(scores)), key=lambda i: scores[i], reverse=True)\n    selected_docs = []\n    total_tokens = 0\n\n    for i in ranked_indices:\n        doc_tokens = len(documents[i].split())\n        if total_tokens + doc_tokens > max_tokens:\n            break\n        selected_docs.append((file_paths[i], documents[i]))\n        total_tokens += doc_tokens\n        if len(selected_docs) >= min_files:\n            break\n\n    return selected_docs\n\n\n\ndef process_query_extract_first_lines(query, num_lines=3):\n    \"\"\"\n    Extract the first few lines from the query.\n    \"\"\"\n    lines = query.split(\"\\n\")[:num_lines]\n    return \" \".join(line.strip() for line in lines if line.strip())\n\ndef process_query_extract_keywords(query):\n    \"\"\"\n    Extract technical terms from the query using regex.\n    \"\"\"\n    # Match technical terms such as file paths, functions, classes, or keywords\n    keywords = re.findall(r'\\b(def|class|import|return|Error|Exception|traceback)\\b', query)\n    \n    # Add file-like patterns (e.g., `something.py`)\n    file_paths = re.findall(r'[\\w\\-/]+\\.py', query)\n\n    return \" \".join(set(keywords + file_paths))\n    \ndef process_query_extract_errors(query):\n    \"\"\"\n    Extract lines containing error messages from the query.\n    \"\"\"\n    error_lines = [line.strip() for line in query.split(\"\\n\") if \"error\" in line.lower() or \"exception\" in line.lower()]\n    return \" \".join(error_lines)\n\ndef process_query_with_code_snippets(query):\n    \"\"\"\n    Extract and prioritize code snippets from the query.\n    \"\"\"\n    # Extract code snippets enclosed in triple backticks\n    code_snippets = re.findall(r\"```(.*?)```\", query, re.DOTALL)\n    \n    # Add heuristic for standalone code-like lines\n    code_lines = [line.strip() for line in query.split(\"\\n\") if any(kw in line for kw in [\"def \", \"class \", \"import \"])]\n    \n    # Combine snippets and standalone lines\n    code_content = \" \".join(code_snippets + code_lines)\n    \n    # Combine code snippets with the original query for context\n    return f\"{query.strip()} {code_content.strip()}\"\n\ndef process_query_with_context(query):\n    \"\"\"\n    Extract context-specific terms from the query.\n    \"\"\"\n    # Common context keywords\n    action_keywords = [\"fix\", \"add\", \"update\", \"remove\"]\n    issue_keywords = [\"bug\", \"error\", \"crash\", \"slow\", \"unexpected\"]\n    component_keywords = [\"UI\", \"database\", \"API\", \"server\", \"backend\"]\n\n    # Extract relevant words from query\n    extracted = [word for word in query.lower().split() if word in action_keywords + issue_keywords + component_keywords]\n    \n    return \" \".join(extracted)\n\n\ndef process_query_combined(query):\n    \"\"\"\n    Combine multiple strategies for query processing.\n    \"\"\"\n    # Extract parts using different approaches\n    first_lines = process_query_extract_first_lines(query, num_lines=2)\n    keywords = process_query_extract_keywords(query)\n    errors = process_query_extract_errors(query)\n    code_snippets = process_query_with_code_snippets(query)\n\n    # Combine them into a single query\n    return f\"{first_lines} {keywords} {errors} {code_snippets}\".strip()\n\n\n# **Random Issue**\nrandom_issue = kprize_df.iloc[1]\nrepo = random_issue[\"repo\"]\nproblem_statement = random_issue[\"problem_statement\"]\nreal_diff = random_issue[\"patch\"]\n\n# **Extract Repository Path**\nnamespace, repo_name = repo.split(\"/\")\nrepo_paths = glob(f\"{COMP_REPOS_DIR}/repo__{namespace}__{repo_name}-*\")\nif not repo_paths:\n    raise ValueError(f\"[ERROR] Repository not found for: {repo}\")\nrepo_path = repo_paths[0]\n\n# **Prepare BM25 Documents**\ndocuments, file_paths = prepare_bm25_documents(repo_path)\n\n# **Retrieve Top 10 Most Relevant Files**\nquery = process_query_combined(problem_statement)\ntop_files = bm25_retrieve_limited(query, documents, file_paths)\n\n# **Extract Non-Overlapping Code Chunks from the Top 10 Files**\ndef rank_chunks_in_top_files(top_files, query, top_n=30, chunk_size=10):\n    \"\"\"\n    Rank overlapping chunks of size `chunk_size` lines from the top 10 files.\n    Select the top non-overlapping ones.\n    \"\"\"\n    ranked_chunks = []\n    bm25_corpus = []\n    chunk_file_map = []\n\n    for file_path, content in top_files[:10]:  # Process only top 10 files\n        lines = content.split(\"\\n\")\n\n        # Generate overlapping chunks (10-line chunks)\n        for idx in range(len(lines) - (chunk_size - 1)):  \n            chunk = \"\\n\".join(lines[idx : idx + chunk_size])  \n            bm25_corpus.append(chunk)\n            chunk_file_map.append((file_path, idx))  \n\n    # Apply BM25 on 10-line chunks\n    bm25 = BM25Okapi([chunk.split() for chunk in bm25_corpus])\n    scores = bm25.get_scores(query.split())\n\n    # Select top-ranked non-overlapping chunks\n    top_chunk_indices = sorted(range(len(scores)), key=lambda i: scores[i], reverse=True)\n\n    selected_chunks = []\n    selected_ranges = []\n\n    for idx in top_chunk_indices:\n        file_path, start_line = chunk_file_map[idx]\n        end_line = start_line + (chunk_size - 1)\n\n        # Check for overlap with previously selected chunks\n        if any(s <= end_line and e >= start_line for _, s, e in selected_ranges):\n            continue  # Skip if overlapping\n\n        selected_ranges.append((file_path, start_line, end_line))\n        code_chunk = bm25_corpus[idx]\n        selected_chunks.append((file_path, code_chunk))\n\n        if len(selected_chunks) >= top_n:\n            break  # Stop once we have enough non-overlapping chunks\n\n    return selected_chunks\n\n# **Retrieve Top 30 Non-Overlapping Chunks**\nranked_code_chunks = rank_chunks_in_top_files(top_files, problem_statement)\n\n# **Format File Paths Correctly**\ndef format_file_path(long_path):\n    \"\"\"\n    Convert long file paths into the required format: `/repo_name/module/file.py`\n    \"\"\"\n    parts = long_path.split(\"/\")\n    if \"repos\" in parts:\n        idx = parts.index(\"repos\") + 1  # Find \"repos\" and get repo name\n        return \"/\" + \"/\".join(parts[idx+1:])  # Return formatted path\n    return long_path  # Default case (should not happen)\n\n# **Merge Code Chunks Per File**\nfrom collections import defaultdict\n\nmerged_code_chunks = defaultdict(lambda: {\"start\": float(\"inf\"), \"end\": float(\"-inf\"), \"lines\": []})\n\nfor file, chunk in ranked_code_chunks:\n    file_path = format_file_path(file)\n    lines = chunk.split(\"\\n\")\n    \n    # Find the smallest and largest line number across all selected chunks\n    min_line = merged_code_chunks[file_path][\"start\"]\n    max_line = merged_code_chunks[file_path][\"end\"]\n\n    if min_line > 0:  # Update the start line\n        merged_code_chunks[file_path][\"start\"] = min(min_line, 0)\n    merged_code_chunks[file_path][\"end\"] = max(max_line, len(lines) - 1)\n    \n    merged_code_chunks[file_path][\"lines\"].extend(lines)\n\n# **Format the Context with Correct File Paths and Merged Chunks**\ncode_chunks_str = \"\\n\\n\".join(\n    [f\"<file: {file}>\\n\" + \"\\n\".join(set(data[\"lines\"])) for file, data in merged_code_chunks.items()]\n)\n\n# **Prepare Input Prompt for BM25-LLaMA 7B**\ninput_prompt = f\"\"\"\nThe following is a description of a software issue and the most relevant code base context extracted from the repository.\nYour task is to analyze the issue and the provided files' content to predict the specific file paths that are most likely to be modified to resolve the issue.\n\n### Task:\n1. Carefully read the issue description provided under `<issue>`.\n2. Review the most relevant code snippets under `<repository_context>`. These are the most relevant files retrieved based on their similarity to the issue.\n3. Identify the file paths that are most likely to require modifications in order to fix the issue.\n4. Your response should only include the file paths.\n\n<issue>\n{problem_statement}\n</issue>\n\n<repository_context>\n{code_chunks_str}\n</repository_context>\n\n\nProvide only the file paths that require modifications to fix the issue,\n\n<files>\n\"\"\"\n\n\n\n# **Load LLaMA 7B Model**\ntokenizer = AutoTokenizer.from_pretrained(MODEL_DIR_7B)\n\nclear_gpu_memory()  # Free memory\n\nmodel = AutoModelForCausalLM.from_pretrained(\n    MODEL_DIR_7B,\n    device_map=\"auto\",\n    torch_dtype=torch.float16\n)\n\nclear_gpu_memory()  # **Free memory again**\n\n# **Tokenize and Generate File Path Predictions**\ninputs = tokenizer(input_prompt, return_tensors=\"pt\").to(\"cuda\")\noutputs = model.generate(inputs.input_ids, max_new_tokens=300)\npredicted_paths = tokenizer.decode(outputs[0], skip_special_tokens=True)\n\n# **Output Results**\nprint(\"\\n[INFO] Problem Statement:\\n\", problem_statement)\nprint(\"\\n[INFO] Top 30 Code Chunks:\\n\")\nfor file, chunk in ranked_code_chunks:\n    print(f\"<file: {format_file_path(file)}>\\n{chunk}\\n\")\nprint(\"\\n[INFO] Predicted Relevant Paths:\\n\", predicted_paths)","metadata":{}},{"cell_type":"markdown","source":"bm25 file ranking","metadata":{}},{"cell_type":"markdown","source":"# **BM25 Setup (Fixed)**\ndef prepare_bm25_documents(repo_path, min_tokens=10, max_tokens=1200):\n    \"\"\"Prepare augmented documents for BM25.\"\"\"\n    documents = []\n    file_paths = []\n    \n    for root, _, files in os.walk(repo_path):\n        for file in files:\n            file_path = os.path.join(root, file)\n            try:\n                with open(file_path, \"r\", encoding=\"utf-8\") as f:\n                    content = f.read().strip()  # Remove leading/trailing spaces\n                    token_count = len(content.split())  # Count words\n\n                    if token_count < min_tokens or token_count > max_tokens:\n                        continue  # **Skip very short or extremely large files**\n                    \n                    augmented_content = f\"<file: {file_path}>\\n{content}\"\n                    documents.append(augmented_content)\n                    file_paths.append(file_path)\n\n            except Exception as e:\n                print(f\"[WARNING] Could not read file {file_path}: {e}\")\n    \n    if not documents:\n        raise ValueError(\"[ERROR] No valid documents found for BM25 retrieval!\")\n\n    print(f\"[INFO] Loaded {len(documents)} documents for BM25.\")\n    return documents, file_paths\n\ndef bm25_retrieve_limited(query, documents, file_paths, max_tokens=3000, min_files=3):\n    \"\"\"Retrieve relevant files using BM25, ensuring total tokens stay within `max_tokens`.\"\"\"\n    tokenized_docs = [doc.split() for doc in documents]\n    bm25 = BM25Okapi(tokenized_docs)\n    query_tokens = query.split()\n    scores = bm25.get_scores(query_tokens)\n\n    ranked_indices = sorted(range(len(scores)), key=lambda i: scores[i], reverse=True)\n    selected_docs = []\n    total_tokens = 0\n\n    for i in ranked_indices:\n        doc_tokens = len(documents[i].split())\n\n        # **Skip files that are too big**\n        if doc_tokens > max_tokens * 0.5:  # If a single file is > 50% of max_tokens, skip\n            continue\n\n        if total_tokens + doc_tokens > max_tokens:\n            break  # Stop if adding another file exceeds token limit\n\n        selected_docs.append((file_paths[i], documents[i]))\n        total_tokens += doc_tokens\n\n        # **Ensure at least `min_files` are added**\n        if len(selected_docs) < min_files:\n            continue\n\n    if not selected_docs:\n        raise ValueError(\"[ERROR] BM25 did not retrieve enough relevant files!\")\n\n    print(f\"[INFO] BM25 retrieved {len(selected_docs)} relevant files.\")\n    return selected_docs\n\n# **Random issue**\nrandom_issue = kprize_df.sample(n=1).iloc[0]\nrepo = random_issue[\"repo\"]\nproblem_statement = random_issue[\"problem_statement\"]\nreal_diff = random_issue[\"patch\"]\n\n# **Extract repository path**\nnamespace, repo_name = repo.split(\"/\")\nrepo_paths = glob(f\"{COMP_REPOS_DIR}/repo__{namespace}__{repo_name}-*\")\n\nif not repo_paths:\n    raise ValueError(f\"[ERROR] Repository not found for: {repo}\")\n\nrepo_path = repo_paths[0]  # Pick first matched path\nprint(f\"[INFO] Extracted Repository Path: {repo_path}\")\n\n# **Prepare BM25 documents**\ndocuments, file_paths = prepare_bm25_documents(repo_path)\n\n# **Retrieve relevant files (limit total token size)**\ntop_files = bm25_retrieve_limited(problem_statement, documents, file_paths, max_tokens=3000, min_files=3)\n\n# **Debug: Show retrieved file paths**\nprint(\"\\n[DEBUG] BM25 Retrieved Files:\")\nfor file, _ in top_files:\n    print(f\"- {file}\")\n\n# **Prepare context for model input**\nrepo_context = \"\\n\".join([doc for _, doc in top_files])\n\nif not repo_context.strip():\n    raise ValueError(\"[ERROR] Repository context is empty! BM25 didn't return valid results.\")\n\nprint(\"\\n[INFO] Repository Context Length:\", len(repo_context.split()), \"tokens\")\n","metadata":{"execution":{"iopub.status.busy":"2025-01-28T19:15:36.847669Z","iopub.execute_input":"2025-01-28T19:15:36.848009Z","iopub.status.idle":"2025-01-28T19:15:36.923533Z","shell.execute_reply.started":"2025-01-28T19:15:36.847982Z","shell.execute_reply":"2025-01-28T19:15:36.922866Z"}}},{"cell_type":"markdown","source":"BM25 ranking files + ranking code chunks (top3) + diff prediction","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport pandas as pd\nimport torch\nimport zipfile\nimport shutil\nfrom glob import glob\nfrom rank_bm25 import BM25Okapi\nimport re\nfrom transformers import AutoTokenizer, AutoModelForCausalLM\nimport json\nimport io\nfrom kaggle_evaluation.konwinski_prize_inference_server import KPrizeInferenceServer\n\n\n\n# **Clear GPU memory**\ndef clear_gpu_memory():\n    gc.collect()\n    torch.cuda.empty_cache()\n\n# **BM25 Setup**\ndef prepare_bm25_documents(repo_path, min_tokens=1, max_tokens=100000):\n    documents = []\n    file_paths = []\n    for root, _, files in os.walk(repo_path):\n        for file in files:\n            file_path = os.path.join(root, file)\n            try:\n                with open(file_path, \"r\", encoding=\"utf-8\") as f:\n                    content = f.read().strip()\n                    token_count = len(content.split())\n                    if min_tokens <= token_count <= max_tokens:\n                        documents.append(f\"<file: {file_path}>\\n{content}\")\n                        file_paths.append(file_path)\n            except UnicodeDecodeError:\n                # Suppress UnicodeDecodeError warnings (non-text files)\n                pass\n            except Exception:\n                # Suppress all other file reading errors\n                pass\n    return documents, file_paths\n\n\ndef bm25_retrieve_limited(query, documents, file_paths, max_tokens=100000, min_files=10):\n    tokenized_docs = [doc.split() for doc in documents]\n    bm25 = BM25Okapi(tokenized_docs)\n    scores = bm25.get_scores(query.split())\n\n    ranked_indices = sorted(range(len(scores)), key=lambda i: scores[i], reverse=True)\n    selected_docs = []\n    total_tokens = 0\n\n    for i in ranked_indices:\n        doc_tokens = len(documents[i].split())\n        if total_tokens + doc_tokens > max_tokens:\n            break\n        selected_docs.append((file_paths[i], documents[i]))\n        total_tokens += doc_tokens\n        if len(selected_docs) >= min_files:\n            break\n\n    return selected_docs\n\n\ndef process_query_extract_first_lines(query, num_lines=3):\n    \"\"\"\n    Extract the first few lines from the query.\n    \"\"\"\n    lines = query.split(\"\\n\")[:num_lines]\n    return \" \".join(line.strip() for line in lines if line.strip())\n\ndef process_query_extract_keywords(query):\n    \"\"\"\n    Extract technical terms from the query using regex.\n    \"\"\"\n    # Match technical terms such as file paths, functions, classes, or keywords\n    keywords = re.findall(r'\\b(def|class|import|return|Error|Exception|traceback)\\b', query)\n    \n    # Add file-like patterns (e.g., `something.py`)\n    file_paths = re.findall(r'[\\w\\-/]+\\.py', query)\n\n    return \" \".join(set(keywords + file_paths))\n    \ndef process_query_extract_errors(query):\n    \"\"\"\n    Extract lines containing error messages from the query.\n    \"\"\"\n    error_lines = [line.strip() for line in query.split(\"\\n\") if \"error\" in line.lower() or \"exception\" in line.lower()]\n    return \" \".join(error_lines)\n\ndef process_query_with_code_snippets(query):\n    \"\"\"\n    Extract and prioritize code snippets from the query.\n    \"\"\"\n    # Extract code snippets enclosed in triple backticks\n    code_snippets = re.findall(r\"```(.*?)```\", query, re.DOTALL)\n    \n    # Add heuristic for standalone code-like lines\n    code_lines = [line.strip() for line in query.split(\"\\n\") if any(kw in line for kw in [\"def \", \"class \", \"import \"])]\n    \n    # Combine snippets and standalone lines\n    code_content = \" \".join(code_snippets + code_lines)\n    \n    # Combine code snippets with the original query for context\n    return f\"{query.strip()} {code_content.strip()}\"\n\ndef process_query_with_context(query):\n    \"\"\"\n    Extract context-specific terms from the query.\n    \"\"\"\n    # Common context keywords\n    action_keywords = [\"fix\", \"add\", \"update\", \"remove\"]\n    issue_keywords = [\"bug\", \"error\", \"crash\", \"slow\", \"unexpected\"]\n    component_keywords = [\"UI\", \"database\", \"API\", \"server\", \"backend\"]\n\n    # Extract relevant words from query\n    extracted = [word for word in query.lower().split() if word in action_keywords + issue_keywords + component_keywords]\n    \n    return \" \".join(extracted)\n\n\ndef process_query_combined(query):\n    \"\"\"\n    Combine multiple strategies for query processing.\n    \"\"\"\n    # Extract parts using different approaches\n    first_lines = process_query_extract_first_lines(query, num_lines=2)\n    keywords = process_query_extract_keywords(query)\n    errors = process_query_extract_errors(query)\n    code_snippets = process_query_with_code_snippets(query)\n\n    # Combine them into a single query\n    return f\"{first_lines} {keywords} {errors} {code_snippets}\".strip()\n\n\n\ndef combine_chunks_from_same_file(ranked_chunks):\n    \"\"\"\n    Combine chunks from the same file into larger chunks, preserving context.\n    \"\"\"\n    file_chunks = {}\n    \n    for file_path, chunk in ranked_chunks:\n        if file_path not in file_chunks:\n            file_chunks[file_path] = []\n        file_chunks[file_path].append(chunk)\n    \n    combined_chunks = {}\n    \n    for file_path, chunks in file_chunks.items():\n        # Sort chunks by their starting line number\n        sorted_chunks = sorted(chunks, key=lambda x: int(re.search(r\"\\[start of .+\\](\\n.*?)?\\n\", x).group(1).split(\"\\n\")[0].split(\":\")[0]))\n        \n        combined_chunk = []\n        previous_end_line = -1\n        \n        for chunk in sorted_chunks:\n            lines = chunk.split(\"\\n\")\n            start_line = int(re.search(r\"\\[start of .+\\](\\n.*?)?\\n\", chunk).group(1).split(\"\\n\")[0].split(\":\")[0])\n            \n            if previous_end_line != -1 and start_line > previous_end_line:\n                # Add the lines between the previous chunk and the current chunk\n                with open(file_path, \"r\") as f:\n                    all_lines = f.readlines()\n                    combined_chunk.extend(all_lines[previous_end_line:start_line])\n            \n            combined_chunk.extend(lines)\n            previous_end_line = start_line + len(lines) - 1\n        \n        combined_chunks[file_path] = \"\\n\".join(combined_chunk)\n    \n    return combined_chunks\n\n\ndef format_file_path(long_path):\n    \"\"\"\n    Convert long file paths into the required format: `/module/file.py`\n    \"\"\"\n    parts = long_path.split(\"/\")\n    if \"repos\" in parts:\n        idx = parts.index(\"repos\") + 1  # Find \"repos\" and get repo name\n        return \"/\" + \"/\".join(parts[idx+1:])  # Return formatted path\n    return long_path  # Default case (should not happen)\n\n\n\ndef format_code_chunks(merged_chunks):\n    \"\"\"\n    Convert merged chunks into the final prompt format.\n    \"\"\"\n    return \"\\n\\n\".join(\n        [f\"<file: {format_file_path(file)}>\\n{chunk}\" for file, (_, _, chunk) in merged_chunks.items()]\n    )\n\n\n\n\n\ndef merge_chunks_within_files(top_files, ranked_chunks):\n    \"\"\"\n    Merge chunks within the same file and add +50 lines up/down for the two largest files.\n    \"\"\"\n    merged_chunks = {}\n\n    for file_path, ranges in ranked_chunks.items():\n        # Sort chunks by starting line number\n        sorted_chunks = sorted(ranges, key=lambda x: x[0])\n\n        # Find the min and max lines across all chunks for this file\n        min_line = sorted_chunks[0][0]\n        max_line = sorted_chunks[-1][1]\n\n        # Extract actual content\n        file_content = next((content for path, content in top_files if path == file_path), None)\n        if not file_content:\n            continue  \n\n        lines = file_content.split(\"\\n\")\n        merged_chunk = \"\\n\".join(lines[min_line : max_line + 1])\n\n        merged_chunks[file_path] = (min_line, max_line, merged_chunk)\n\n    # Find top 2 files with the most merged lines\n    sorted_files = sorted(merged_chunks.items(), key=lambda x: x[1][1] - x[1][0], reverse=True)[:2]\n\n    # Expand context by +50 lines for top 2 files\n    for file_path, (min_line, max_line, chunk) in sorted_files:\n        file_content = next((content for path, content in top_files if path == file_path), None)\n        if not file_content:\n            continue  \n\n        lines = file_content.split(\"\\n\")\n        min_line = max(0, min_line - 50)\n        max_line = min(len(lines) - 1, max_line + 50)\n\n        merged_chunks[file_path] = (min_line, max_line, \"\\n\".join(lines[min_line:max_line + 1]))\n\n    return merged_chunks\n\n\n\ndef rank_chunks_in_top_files(top_files, query, top_n=5, context_window=10):\n    \"\"\"\n    Rank 10-line overlapping chunks inside the top 10 files and extract the highest-ranked non-overlapping ones.\n    Merge chunks from the same file to create a continuous context.\n    \"\"\"\n    bm25_corpus = []\n    chunk_file_map = []\n\n    for file_path, content in top_files[:10]:  # Process the top 10 files\n        lines = content.split(\"\\n\")\n\n        # Generate 10-line overlapping chunks\n        for idx in range(len(lines) - 9):  \n            chunk = \"\\n\".join(lines[idx : idx + 10])\n            bm25_corpus.append(chunk)\n            chunk_file_map.append((file_path, idx))  # Track which file and start index\n\n    # Apply BM25 on 10-line chunks\n    bm25 = BM25Okapi([chunk.split() for chunk in bm25_corpus])\n    scores = bm25.get_scores(query.split())\n\n    # Select top-ranked non-overlapping chunks\n    top_chunk_indices = sorted(range(len(scores)), key=lambda i: scores[i], reverse=True)\n\n    selected_chunks = {}\n    selected_ranges = {}\n\n    for idx in top_chunk_indices:\n        file_path, start_line = chunk_file_map[idx]\n        end_line = start_line + 9  \n\n        # Ensure non-overlapping chunks\n        if file_path in selected_ranges:\n            existing_ranges = selected_ranges[file_path]\n            if any(s <= end_line and e >= start_line for s, e in existing_ranges):\n                continue  \n\n        # Store chunk with its file\n        if file_path not in selected_chunks:\n            selected_chunks[file_path] = []\n            selected_ranges[file_path] = []\n\n        selected_chunks[file_path].append((start_line, end_line, bm25_corpus[idx]))\n        selected_ranges[file_path].append((start_line, end_line))\n\n        # Stop if we have enough\n        if sum(len(r) for r in selected_ranges.values()) >= top_n:\n            break  \n\n    return selected_chunks\n\n\n\n\nimport re\n\ndef extract_diff_patch(output_text):\n    \"\"\"\n    Extracts the diff patch from the model's output and cleans up extra <patch> tags.\n    \"\"\"\n    matches = re.findall(r\"<patch>\\s*(diff --git .*?)\\s*</patch>\", output_text, re.DOTALL)\n\n    if len(matches) < 1:\n        print(\"[WARNING] No valid <patch> found.\")\n        return None\n\n    extracted_patch = matches[0] if len(matches) == 1 else matches[1]\n\n    # Remove redundant <patch> tags line by line\n    cleaned_patch = \"\\n\".join(\n        [line for line in extracted_patch.split(\"\\n\") if \"<patch>\" not in line and \"</patch>\" not in line]\n    ).strip()\n\n    # Sanity check: Ensure valid diff format\n    if len(cleaned_patch.split(\"\\n\")) < 3 or \"diff --git\" not in cleaned_patch:\n        print(\"[WARNING] Invalid or malformed patch detected:\\n\", cleaned_patch)\n        return None\n\n    return cleaned_patch\n\n# **Prepare Submission Entry**\ndef prepare_submission(instance_id, model_patch):\n    \"\"\"\n    Creates a properly formatted JSON submission entry.\n    \"\"\"\n    return {\n        \"instance_id\": instance_id,\n        \"model_patch\": model_patch,\n        \"model_name_or_path\": \"SWE-Llama-13b\"\n    }\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-02T05:34:55.016933Z","iopub.execute_input":"2025-02-02T05:34:55.017235Z","iopub.status.idle":"2025-02-02T05:34:55.048581Z","shell.execute_reply.started":"2025-02-02T05:34:55.017211Z","shell.execute_reply":"2025-02-02T05:34:55.047853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport io\nimport shutil\nimport zipfile\nimport subprocess\nimport pandas as pd\nimport torch\nfrom glob import glob\nfrom transformers import AutoTokenizer, AutoModelForCausalLM\nimport kaggle_evaluation.konwinski_prize_inference_server as kp_server\n\n# **Paths**\nMODEL_DIR_13B = \"/kaggle/input/swe-llama-model/swe-llama-model\"\nZIP_FILE_PATH = \"/kaggle/input/konwinski-prize/data.a_zip\"\nEXTRACTED_DATA_DIR = \"/kaggle/working/extracted_data\"\nPARQUET_FILE = os.path.join(EXTRACTED_DATA_DIR, \"data\", \"data.parquet\")\nCOMP_REPOS_DIR = os.path.join(EXTRACTED_DATA_DIR, \"data\", \"repos\")\n\n# **Extract ZIP file if necessary**\nif not os.path.exists(EXTRACTED_DATA_DIR):\n    with zipfile.ZipFile(ZIP_FILE_PATH, 'r') as zip_ref:\n        zip_ref.extractall(EXTRACTED_DATA_DIR)\n\n# **Load dataset**\nkprize_df = pd.read_parquet(PARQUET_FILE)\n\n# **Kaggle Evaluation Server**\ninstance_count = None\n\ndef get_number_of_instances(num_instances: int):\n    global instance_count\n    instance_count = num_instances\n\ndef predict(problem_statement: str, repo_archive: io.BytesIO, pip_packages_archive: io.BytesIO, env_setup_cmds_templates: list[str]) -> str:\n    \"\"\" Handles prediction dynamically for Kaggle inference server. \"\"\"\n    \n    # **Extract Repository**\n    repo_path = '/kaggle/working/repo'\n    if os.path.exists(repo_path):\n        shutil.rmtree(repo_path)\n    with open('/kaggle/working/repo_archive.tar', 'wb') as f:\n        f.write(repo_archive.read())\n    shutil.unpack_archive('/kaggle/working/repo_archive.tar', extract_dir=repo_path)\n    os.remove('/kaggle/working/repo_archive.tar')\n    \n    # **Setup Environment**\n    pip_packages_path = '/kaggle/working/pip_packages'\n    if os.path.exists(pip_packages_path):\n        shutil.rmtree(pip_packages_path)\n    with open('/kaggle/working/pip_packages_archive.tar', 'wb') as f:\n        f.write(pip_packages_archive.read())\n    shutil.unpack_archive('/kaggle/working/pip_packages_archive.tar', extract_dir=pip_packages_path)\n    os.remove('/kaggle/working/pip_packages_archive.tar')\n    \n    env_setup_cmds = [cmd.format(pip_packages_path=pip_packages_path) for cmd in env_setup_cmds_templates]\n    subprocess.run(\"\\n\".join(env_setup_cmds), shell=True, executable=\"/bin/bash\", cwd=repo_path)\n    \n    # **Find Matching Issue**\n    instance_id = None\n    for idx, issue in kprize_df.iterrows():\n        if issue[\"problem_statement\"] == problem_statement:\n            instance_id = f\"instance_{idx}\"\n            break\n    \n    if instance_id is None:\n        print(\"[WARNING] Issue not found in dataset.\")\n        return None\n    \n    # **Prepare BM25 Documents**\n    documents, file_paths = prepare_bm25_documents(repo_path)\n    \n    # **Retrieve Top 10 Files**\n    query = process_query_combined(problem_statement)\n    top_files = bm25_retrieve_limited(query, documents, file_paths)\n    \n    # **Rank and Merge Chunks**\n    ranked_chunks = rank_chunks_in_top_files(top_files, problem_statement)\n    merged_chunks = merge_chunks_within_files(top_files, ranked_chunks)\n    code_chunks_str = format_code_chunks(merged_chunks)\n    \n    # **Generate Input Prompt**\n    input_prompt = f\"\"\"\n    You will be provided with a partial code base and an issue statement explaining a problem to resolve.\n    \n    <issue>\n    {problem_statement}\n    </issue>\n    \n    <code>\n    {code_chunks_str}\n    </code>\n    \n    Here is an example of a patch file. It consists of changes to the code base. It specifies the file names, \n    the line numbers of each change, and the removed and added lines. A single patch file can contain changes \n    to multiple files.\n    \n    <patch>\n    --- a/file.py\n    +++ b/file.py\n    @@ -1,27 +1,35 @@\n    def euclidean(a, b):\n    - while b:\n    -     a, b = b, a % b\n    - return a\n    + if b == 0:\n    +     return a\n    + return euclidean(b, a % b)\n    \n    def bresenham(x0, y0, x1, y1):\n        points = []\n        dx = abs(x1 - x0)\n        dy = abs(y1 - y0)\n    -   sx = 1 if x0 < x1 else -1\n    -   sy = 1 if y0 < y1 else -1\n    -   err = dx - dy\n    +   x, y = x0, y0\n    +   sx = -1 if x0 > x1 else 1\n    +   sy = -1 if y0 > y1 else 1\n    \n    -   while True:\n    -       points.append((x0, y0))\n    -       if x0 == x1 and y0 == y1:\n    -           break\n    -       e2 = 2 * err\n    -       if e2 > -dy:\n    +   if dx > dy:\n    +       err = dx / 2.0\n    +       while x != x1:\n    +           points.append((x, y))\n                err -= dy\n    -           x0 += sx\n    -       if e2 < dx:\n    -           err += dx\n    -           y0 += sy\n    +           if err < 0:\n    +               y += sy\n    +               err += dx\n    +           x += sx\n    +   else:\n    +       err = dy / 2.0\n    +       while y != y1:\n    +           points.append((x, y))\n                err -= dx\n    +           if err < 0:\n    +               x += sx\n    +               err += dy\n    +           y += sy\n    +   points.append((x, y))\n        return points\n    </patch>\n    \n    I need you to solve the provided issue by generating a single patch file that I can apply directly to this repository using `git apply`. \n    Please respond with a single patch file in the format shown above.\n    \n    <patch>\n    \"\"\"\n    \n    # **Load LLaMA 13B Model**\n    tokenizer = AutoTokenizer.from_pretrained(MODEL_DIR_13B)\n    model = AutoModelForCausalLM.from_pretrained(\n        MODEL_DIR_13B, device_map=\"auto\", torch_dtype=torch.float16, offload_folder=\"/kaggle/tmp/offload\"\n    )\n    inputs = tokenizer(input_prompt, return_tensors=\"pt\").to(\"cuda\")\n    outputs = model.generate(inputs.input_ids, max_new_tokens=1000)\n    predicted_output = tokenizer.decode(outputs[0], skip_special_tokens=True)\n    \n    # **Extract Patch**\n    extracted_patch = extract_diff_patch(predicted_output)\n    \n    if extracted_patch is None:\n        print(f\"[WARNING] Skipping instance {instance_id} due to invalid patch.\")\n        return None\n    \n    return extracted_patch\n\n# **Start Kaggle Inference Server**\ninference_server = kp_server.KPrizeInferenceServer(get_number_of_instances, predict)\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            '/kaggle/input/konwinski-prize/',\n            '/kaggle/tmp/konwinski-prize/',\n        ),\n        use_concurrency=True,\n    )\n","metadata":{"execution":{"iopub.status.busy":"2025-02-02T05:35:10.557217Z","iopub.execute_input":"2025-02-02T05:35:10.557567Z","iopub.status.idle":"2025-02-02T05:51:33.529111Z","shell.execute_reply.started":"2025-02-02T05:35:10.55754Z","shell.execute_reply":"2025-02-02T05:51:33.528352Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n\n\n# **Paths**\nMODEL_DIR_7B = \"/kaggle/input/bm25-llama-7b\"\nMODEL_DIR_13B = \"/kaggle/input/swe-llama-model/swe-llama-model\"\nZIP_FILE_PATH = \"/kaggle/input/konwinski-prize/data.a_zip\"\nEXTRACTED_DATA_DIR = \"/kaggle/working/extracted_data\"\nPARQUET_FILE = os.path.join(EXTRACTED_DATA_DIR, \"data\", \"data.parquet\")\nCOMP_REPOS_DIR = os.path.join(EXTRACTED_DATA_DIR, \"data\", \"repos\")\n\n# **Set PyTorch CUDA Memory Config**\n#os.environ[\"PYTORCH_CUDA_ALLOC_CONF\"] = \"expandable_segments:True\"\n\n# **Extract ZIP file**\nif not os.path.exists(EXTRACTED_DATA_DIR):\n    with zipfile.ZipFile(ZIP_FILE_PATH, 'r') as zip_ref:\n        zip_ref.extractall(EXTRACTED_DATA_DIR)\n\n# **Load dataset**\nkprize_df = pd.read_parquet(PARQUET_FILE)\n\n\n\n\n# **Inference Logic**\npredictions = []\nfor idx in range(len(kprize_df)):  \n    random_issue = kprize_df.iloc[idx]\n    repo = random_issue[\"repo\"]\n    problem_statement = random_issue[\"problem_statement\"]\n    instance_id = f\"instance_{idx}\"\n\n    # **Extract Repository Path**\n    namespace, repo_name = repo.split(\"/\")\n    repo_paths = glob(f\"{COMP_REPOS_DIR}/repo__{namespace}__{repo_name}-*\")\n    if not repo_paths:\n        print(f\"[WARNING] Skipping instance {instance_id}: Repository not found.\")\n        continue  # Skip this instance\n\n    repo_path = repo_paths[0]\n\n    # **Prepare BM25 Documents**\n    documents, file_paths = prepare_bm25_documents(repo_path)\n\n    # **Retrieve Top 10 Files**\n    query = process_query_combined(problem_statement)\n    top_files = bm25_retrieve_limited(query, documents, file_paths)\n\n    # **Rank and Merge Chunks**\n    ranked_chunks = rank_chunks_in_top_files(top_files, problem_statement)\n    merged_chunks = merge_chunks_within_files(top_files, ranked_chunks)\n    code_chunks_str = format_code_chunks(merged_chunks)\n\n    # **Generate Input Prompt**\n    input_prompt = f\"\"\"\n    You will be provided with a partial code base and an issue statement explaining a problem to resolve.\n    \n    <issue>\n    {problem_statement}\n    </issue>\n    \n    <code>\n    {code_chunks_str}\n    </code>\n    \n    Here is an example of a patch file. It consists of changes to the code base. It specifies the file names, \n    the line numbers of each change, and the removed and added lines. A single patch file can contain changes \n    to multiple files.\n    \n    <patch>\n    --- a/file.py\n    +++ b/file.py\n    @@ -1,27 +1,35 @@\n    def euclidean(a, b):\n    - while b:\n    -     a, b = b, a % b\n    - return a\n    + if b == 0:\n    +     return a\n    + return euclidean(b, a % b)\n    \n    def bresenham(x0, y0, x1, y1):\n        points = []\n        dx = abs(x1 - x0)\n        dy = abs(y1 - y0)\n    -   sx = 1 if x0 < x1 else -1\n    -   sy = 1 if y0 < y1 else -1\n    -   err = dx - dy\n    +   x, y = x0, y0\n    +   sx = -1 if x0 > x1 else 1\n    +   sy = -1 if y0 > y1 else 1\n    \n    -   while True:\n    -       points.append((x0, y0))\n    -       if x0 == x1 and y0 == y1:\n    -           break\n    -       e2 = 2 * err\n    -       if e2 > -dy:\n    +   if dx > dy:\n    +       err = dx / 2.0\n    +       while x != x1:\n    +           points.append((x, y))\n                err -= dy\n    -           x0 += sx\n    -       if e2 < dx:\n    -           err += dx\n    -           y0 += sy\n    +           if err < 0:\n    +               y += sy\n    +               err += dx\n    +           x += sx\n    +   else:\n    +       err = dy / 2.0\n    +       while y != y1:\n    +           points.append((x, y))\n                err -= dx\n    +           if err < 0:\n    +               x += sx\n    +               err += dy\n    +           y += sy\n    +   points.append((x, y))\n        return points\n    </patch>\n    \n    I need you to solve the provided issue by generating a single patch file that I can apply directly to this repository using `git apply`. \n    Please respond with a single patch file in the format shown above.\n    \n    <patch>\n    \"\"\"\n\n    # **Load LLaMA 13B Model**\n    tokenizer = AutoTokenizer.from_pretrained(MODEL_DIR_13B)\n    \n    clear_gpu_memory()  # Free memory\n    \n    model = AutoModelForCausalLM.from_pretrained(\n        MODEL_DIR_13B,\n        device_map=\"auto\",\n        torch_dtype=torch.float16,\n        offload_folder=\"/kaggle/tmp/offload\"\n        \n    )\n\n\n    #model = AutoModelForCausalLM.from_pretrained(\n    #    MODEL_DIR_13B,\n    #    device_map=\"cpu\",\n    #    torch_dtype=torch.float32,\n    #    offload_folder=\"/kaggle/tmp/offload\"\n    #).quantize(8)  # Uses INT8 quantization (reduces RAM usage)\n\n    clear_gpu_memory()  # **Free memory again**\n\n    # **Generate Prediction**\n    inputs = tokenizer(input_prompt, return_tensors=\"pt\").to(\"cuda\")\n    #inputs = tokenizer(input_prompt, return_tensors=\"pt\").to(\"cpu\")\n    outputs = model.generate(inputs.input_ids, max_new_tokens=1000)\n    predicted_output = tokenizer.decode(outputs[0], skip_special_tokens=True)\n\n    # **Extract Patch**\n    extracted_patch = extract_diff_patch(predicted_output)\n\n    # **Skip Invalid Predictions**\n    if extracted_patch is None:\n        print(f\"[WARNING] Skipping instance {instance_id} due to invalid patch.\")\n        continue\n\n    # **Prepare and Store Prediction**\n    predictions.append(prepare_submission(instance_id, extracted_patch))\n\n# **Save Submission File**\nsubmission_path = \"/kaggle/working/predictions.json\"\nwith open(submission_path, \"w\") as f:\n    json.dump(predictions, f, indent=4)\n\nprint(f\"[INFO] Predictions saved to {submission_path}\")\n\n# **Kaggle Inference Server**\ninstance_count = None\n\ndef get_number_of_instances(num_instances: int):\n    \"\"\"Receive the total number of test instances.\"\"\"\n    global instance_count\n    instance_count = num_instances\n\ndef predict(problem_statement: str, repo_archive: io.BytesIO, pip_packages_archive: io.BytesIO, env_setup_cmds_templates: list[str]) -> str:\n    \"\"\"Handles prediction dynamically for Kaggle inference server.\"\"\"\n    \n    # Extract Patch from Model Output\n    extracted_patch = extract_diff_patch(predicted_output)\n\n    if extracted_patch is None:\n        print(\"[WARNING] Skipping instance due to invalid patch.\")\n        return None  # Skip invalid patch\n\n    return extracted_patch  # Return the valid patch\n\n# **Setup and Run Server**\ninference_server = KPrizeInferenceServer(get_number_of_instances, predict)\n\nif os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\"):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            \"/kaggle/input/konwinski-prize/\",\n            \"/kaggle/tmp/konwinski-prize/\",\n        )\n    )\n\n\n\n\n\n'''\n# **Random Issue**\nrandom_issue = kprize_df.iloc[5]\nrepo = random_issue[\"repo\"]\nproblem_statement = random_issue[\"problem_statement\"]\nreal_diff = random_issue[\"patch\"]\n\n# **Extract Repository Path**\nnamespace, repo_name = repo.split(\"/\")\nrepo_paths = glob(f\"{COMP_REPOS_DIR}/repo__{namespace}__{repo_name}-*\")\nif not repo_paths:\n    raise ValueError(f\"[ERROR] Repository not found for: {repo}\")\nrepo_path = repo_paths[0]\n\n# **Prepare BM25 Documents**\ndocuments, file_paths = prepare_bm25_documents(repo_path)\n\n\n\n# **Retrieve Top 10 Files**\nquery = process_query_combined(problem_statement)\ntop_files = bm25_retrieve_limited(query, documents, file_paths)\n\n# **Step 1: Rank Chunks Within Top Files**\nranked_chunks = rank_chunks_in_top_files(top_files, problem_statement)\n\n# **Step 2: Merge Chunks Per File**\nmerged_chunks = merge_chunks_within_files(top_files, ranked_chunks)\n\n# **Step 3: Format for Prompt**\ncode_chunks_str = format_code_chunks(merged_chunks)\n\n\n\n\n\n\ninput_prompt = f\"\"\"\nYou will be provided with a partial code base and an issue statement explaining a problem to resolve.\n\n<issue>\n{problem_statement}\n</issue>\n\n<code>\n{code_chunks_str}\n</code>\n\nHere is an example of a patch file. It consists of changes to the code base. It specifies the file names, \nthe line numbers of each change, and the removed and added lines. A single patch file can contain changes \nto multiple files.\n\n<patch>\n--- a/file.py\n+++ b/file.py\n@@ -1,27 +1,35 @@\ndef euclidean(a, b):\n- while b:\n-     a, b = b, a % b\n- return a\n+ if b == 0:\n+     return a\n+ return euclidean(b, a % b)\n\ndef bresenham(x0, y0, x1, y1):\n    points = []\n    dx = abs(x1 - x0)\n    dy = abs(y1 - y0)\n-   sx = 1 if x0 < x1 else -1\n-   sy = 1 if y0 < y1 else -1\n-   err = dx - dy\n+   x, y = x0, y0\n+   sx = -1 if x0 > x1 else 1\n+   sy = -1 if y0 > y1 else 1\n\n-   while True:\n-       points.append((x0, y0))\n-       if x0 == x1 and y0 == y1:\n-           break\n-       e2 = 2 * err\n-       if e2 > -dy:\n+   if dx > dy:\n+       err = dx / 2.0\n+       while x != x1:\n+           points.append((x, y))\n            err -= dy\n-           x0 += sx\n-       if e2 < dx:\n-           err += dx\n-           y0 += sy\n+           if err < 0:\n+               y += sy\n+               err += dx\n+           x += sx\n+   else:\n+       err = dy / 2.0\n+       while y != y1:\n+           points.append((x, y))\n            err -= dx\n+           if err < 0:\n+               x += sx\n+               err += dy\n+           y += sy\n+   points.append((x, y))\n    return points\n</patch>\n\nI need you to solve the provided issue by generating a single patch file that I can apply directly to this repository using `git apply`. \nPlease respond with a single patch file in the format shown above.\n\n<patch>\n\"\"\"\n\n\n# **Load LLaMA 13B Model**\ntokenizer = AutoTokenizer.from_pretrained(MODEL_DIR_13B)\n\nclear_gpu_memory()  # Free memory\n\nmodel = AutoModelForCausalLM.from_pretrained(\n    MODEL_DIR_13B,\n    device_map=\"auto\",\n    torch_dtype=torch.float16\n)\n\nclear_gpu_memory()  # **Free memory again**\n\n# **Tokenize and Generate Diff Prediction**\ninputs = tokenizer(input_prompt, return_tensors=\"pt\").to(\"cuda\")\noutputs = model.generate(inputs.input_ids, max_new_tokens=1000)\npredicted_output = tokenizer.decode(outputs[0], skip_special_tokens=True)\n\n# **Extract the diff patch**\nextracted_patch = extract_diff_patch(predicted_output)\n\n# **Print Results**\nprint(\"\\n[INFO] Problem Statement:\\n\", problem_statement)\nprint(\"\\n[INFO] Top 3 Code Chunks:\\n\")\n#for file, chunk in ranked_code_chunks:\n    #print(f\"<file: {file}>\\n{chunk}\\n\")\nprint(\"\\n[INFO] Predicted Diff:\\n\", predicted_output)\nprint(\"\\n[INFO] Extracted Diff Patch:\\n\", extracted_patch if extracted_patch else \"[INVALID PATCH - SKIPPED]\")\nprint(\"\\n[INFO] Actual Diff:\\n\", real_diff)\n'''\n\n\nStep 6: Save Predictions for Evaluation\nSWE-bench expects predictions to be saved in a .json file:","metadata":{"execution":{"iopub.status.busy":"2025-02-01T03:18:50.745536Z","iopub.execute_input":"2025-02-01T03:18:50.745845Z","iopub.status.idle":"2025-02-01T03:32:41.966144Z","shell.execute_reply.started":"2025-02-01T03:18:50.745821Z","shell.execute_reply":"2025-02-01T03:32:41.96546Z"}},"attachments":{}},{"cell_type":"markdown","source":"# Format prediction\nprediction = {\n    \"instance_id\": \"example_instance_1\",\n    \"model_patch\": predicted_patch,\n    \"model_name_or_path\": \"SWE-Llama-13b\",\n}\n\n# Save to a JSON file\nwith open(\"/kaggle/working/predictions.json\", \"w\") as f:\n    json.dump([prediction], f, indent=4)\n\nprint(\"Predictions saved to /kaggle/working/predictions.json\")\n","metadata":{}},{"cell_type":"markdown","source":"import os\nimport torch\nfrom transformers import AutoTokenizer, AutoModelForCausalLM\nfrom peft import LoraConfig, get_peft_model\nfrom kaggle_evaluation.konwinski_prize_inference_server import KPrizeInferenceServer\nimport io\nimport shutil\n\n# Load the base model and tokenizer\nmodel_path = \"/kaggle/input/codellama-7b-dataset/local_codellama_7b\"  # Updated path to the subfolder\n\ntokenizer = AutoTokenizer.from_pretrained(model_path)\nmodel = AutoModelForCausalLM.from_pretrained(model_path, device_map=\"auto\", torch_dtype=torch.float16)\n\n# Configure LoRA\nlora_config = LoraConfig(\n    r=8,                      # Rank of LoRA\n    lora_alpha=16,            # LoRA scaling factor\n    target_modules=[\"q_proj\", \"v_proj\"],  # Parts of the model to apply LoRA (attention heads)\n    lora_dropout=0.1,         # Dropout rate for LoRA\n    bias=\"none\"               # No bias adjustment\n)\n\n# Apply LoRA to the model\nmodel = get_peft_model(model, lora_config)\nmodel.eval()  # Set to evaluation mode\n\n# Global counter to keep track of processed instances\nprocessed_instance_count = 0\nMAX_VALID_PREDICTIONS = 25  # Set the number of valid predictions to generate\n\ndef generate_diff(problem_statement: str) -> str:\n    \"\"\"\n    Generate a Git diff for the given problem statement.\n    \"\"\"\n    prompt = (\n        \"You are a coding assistant trained to write Git diffs. \"\n        \"Write only the required Git diff lines without including the issue description. \"\n        \"Avoid unnecessary lines and ensure the diff applies to the specified file and method.\\n\\n\"\n        f\"Issue: {problem_statement}\\n\\n\"\n        \"Git Diff (start with 'diff --git'):\\n\"\n    )\n\n    # Tokenize the input\n    inputs = tokenizer(prompt, return_tensors=\"pt\").to(\"cuda\")\n\n    # Generate a prediction (maximum 400 tokens)\n    outputs = model.generate(**inputs, max_new_tokens=400)\n\n    # Decode the output to text\n    diff_output = tokenizer.decode(outputs[0], skip_special_tokens=True)\n\n    # Post-process to extract only the diff content\n    diff_start_index = diff_output.find(\"diff --git\")\n    if diff_start_index != -1:\n        diff_output = diff_output[diff_start_index:]\n\n    return diff_output\n\n# Define the `get_number_of_instances` function\ninstance_count = None\n\ndef get_number_of_instances(num_instances: int):\n    \"\"\"\n    Callback function to set the number of instances to process.\n    \"\"\"\n    global instance_count\n    instance_count = num_instances\n\ndef predict(problem_statement: str, repo_archive: io.BytesIO, instance_id: int, metadata: dict) -> str:\n    \"\"\"\n    Predict the Git diff for a single instance.\n    \n    Args:\n        problem_statement: The text of the GitHub issue.\n        repo_archive: A BytesIO object containing the repository archive (.tar).\n        instance_id: An identifier for the instance being processed.\n        metadata: Additional metadata about the instance.\n        \n    Returns:\n        str: The generated Git diff or None for skipped instances.\n    \"\"\"\n    global processed_instance_count\n\n    # Skip prediction after generating `MAX_VALID_PREDICTIONS` real ones\n    if processed_instance_count >= MAX_VALID_PREDICTIONS:\n        print(f\"Instance {instance_id} skipped.\")\n        return None  # Skip this instance\n\n    # Unpack the repository archive\n    with open(\"repo_archive.tar\", \"wb\") as f:\n        f.write(repo_archive.read())\n    repo_path = \"repo\"\n    if os.path.exists(repo_path):\n        shutil.rmtree(repo_path)\n    shutil.unpack_archive(\"repo_archive.tar\", extract_dir=repo_path)\n    os.remove(\"repo_archive.tar\")\n\n    # Increment processed instance count\n    processed_instance_count += 1\n\n    # Generate the diff\n    diff = generate_diff(problem_statement)\n    print(f\"Instance {instance_id}: Generated Prediction:\\n{diff}\\n\")\n    return diff\n\n\n# Set up the inference server\ninference_server = KPrizeInferenceServer(\n    get_number_of_instances,  # Register the get_number_of_instances function\n    predict                   # Register the predict function\n)\n\n# Handle submission or local testing\nis_submission = os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\")\n\nif is_submission:\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            \"/kaggle/input/konwinski-prize/\",  # Path to the competition dataset\n            \"/kaggle/tmp/konwinski-prize/\",   # Path for temporary files\n        )\n    )\n\n","metadata":{"execution":{"iopub.status.busy":"2025-01-19T03:07:43.025753Z","iopub.execute_input":"2025-01-19T03:07:43.02623Z","iopub.status.idle":"2025-01-19T03:12:26.689143Z","shell.execute_reply.started":"2025-01-19T03:07:43.026153Z","shell.execute_reply":"2025-01-19T03:12:26.688262Z"}}},{"cell_type":"markdown","source":"pip install transformers huggingface-hub\n","metadata":{"execution":{"iopub.status.busy":"2025-01-18T01:31:42.275244Z","iopub.execute_input":"2025-01-18T01:31:42.275751Z","iopub.status.idle":"2025-01-18T01:31:46.632086Z","shell.execute_reply.started":"2025-01-18T01:31:42.275718Z","shell.execute_reply":"2025-01-18T01:31:46.630719Z"}}},{"cell_type":"markdown","source":"!huggingface-cli login","metadata":{}},{"cell_type":"markdown","source":"import os\nimport torch\nfrom transformers import AutoTokenizer, AutoModelForCausalLM\nfrom peft import LoraConfig, get_peft_model\nfrom kaggle_evaluation.konwinski_prize_inference_server import KPrizeInferenceServer\nimport io\nimport shutil\n\n# Load the base model and tokenizer\nmodel_name = \"codellama/CodeLlama-7b-hf\"\ntokenizer = AutoTokenizer.from_pretrained(model_name)\nmodel = AutoModelForCausalLM.from_pretrained(model_name, device_map=\"auto\")\n\n# Configure LoRA\nlora_config = LoraConfig(\n    r=8,                      # Rank of LoRA\n    lora_alpha=16,            # LoRA scaling factor\n    target_modules=[\"q_proj\", \"v_proj\"],  # Parts of the model to apply LoRA (attention heads)\n    lora_dropout=0.1,         # Dropout rate for LoRA\n    bias=\"none\"               # No bias adjustment\n)\n\n# Apply LoRA to the model\nmodel = get_peft_model(model, lora_config)\nmodel.eval()  # Set to evaluation mode\n\ndef generate_diff(problem_statement: str) -> str:\n    \"\"\"\n    Generate a Git diff for the given problem statement.\n    \"\"\"\n    prompt = (\n        \"You are a coding assistant trained to write Git diffs. \"\n        \"Write only the required Git diff lines without including the issue description. \"\n        \"Avoid unnecessary lines and ensure the diff applies to the specified file and method.\\n\\n\"\n        f\"Issue: {problem_statement}\\n\\n\"\n        \"Git Diff (start with 'diff --git'):\\n\"\n    )\n\n    # Tokenize the input\n    inputs = tokenizer(prompt, return_tensors=\"pt\").to(\"cuda\")\n\n    # Generate a prediction (maximum 400 tokens)\n    outputs = model.generate(**inputs, max_new_tokens=400)\n\n    # Decode the output to text\n    diff_output = tokenizer.decode(outputs[0], skip_special_tokens=True)\n\n    # Post-process to extract only the diff content\n    diff_start_index = diff_output.find(\"diff --git\")\n    if diff_start_index != -1:\n        diff_output = diff_output[diff_start_index:]\n\n    return diff_output\n\n# Define the `get_number_of_instances` function\ninstance_count = None\n\ndef get_number_of_instances(num_instances: int):\n    \"\"\"\n    Callback function to set the number of instances to process.\n    \"\"\"\n    global instance_count\n    instance_count = num_instances\n\n# Define the `predict` function\ndef predict(problem_statement: str, repo_archive: io.BytesIO) -> str:\n    \"\"\"\n    Predict the Git diff for a single instance.\n    \n    Args:\n        problem_statement: The text of the GitHub issue.\n        repo_archive: A BytesIO object containing the repository archive (.tar).\n        \n    Returns:\n        str: The generated Git diff.\n    \"\"\"\n    # Unpack the repository archive\n    with open(\"repo_archive.tar\", \"wb\") as f:\n        f.write(repo_archive.read())\n    repo_path = \"repo\"\n    if os.path.exists(repo_path):\n        shutil.rmtree(repo_path)\n    shutil.unpack_archive(\"repo_archive.tar\", extract_dir=repo_path)\n    os.remove(\"repo_archive.tar\")\n\n    # Generate the diff\n    diff = generate_diff(problem_statement)\n    return diff\n\n# Set up the inference server\ninference_server = KPrizeInferenceServer(\n    get_number_of_instances,  # Register the get_number_of_instances function\n    predict                   # Register the predict function\n)\n\n# Handle submission or local testing\nis_submission = os.getenv(\"KAGGLE_IS_COMPETITION_RERUN\")\n\nif is_submission:\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            \"/kaggle/input/konwinski-prize/\",  # Path to the competition dataset\n            \"/kaggle/tmp/konwinski-prize/\",   # Path for temporary files\n        )\n    )\n","metadata":{"execution":{"iopub.status.busy":"2025-01-17T01:23:22.291856Z","iopub.execute_input":"2025-01-17T01:23:22.292241Z"}}},{"cell_type":"markdown","source":"from kaggle_evaluation.konwinski_prize_inference_server import KPrizeInferenceServer\nhelp(KPrizeInferenceServer)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-17T00:59:54.412954Z","iopub.execute_input":"2025-01-17T00:59:54.41332Z","iopub.status.idle":"2025-01-17T00:59:54.428866Z","shell.execute_reply.started":"2025-01-17T00:59:54.413291Z","shell.execute_reply":"2025-01-17T00:59:54.428054Z"}}},{"cell_type":"markdown","source":"from sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.metrics.pairwise import cosine_similarity\n\n# Function to calculate similarity between actual and generated diff\ndef calculate_similarity(actual_diff, predicted_diff):\n    vectorizer = TfidfVectorizer()\n    vectors = vectorizer.fit_transform([actual_diff, predicted_diff])\n    similarity_score = cosine_similarity(vectors[0], vectors[1])[0][0]\n    return similarity_score\n\nsimilarity_score = calculate_similarity(actual_patch, predicted_diff)\nprint(f\"Cosine Similarity Score: {similarity_score:.2f}\")","metadata":{"execution":{"iopub.status.busy":"2025-01-16T03:19:11.647012Z","iopub.execute_input":"2025-01-16T03:19:11.647449Z","iopub.status.idle":"2025-01-16T03:19:11.687152Z","shell.execute_reply.started":"2025-01-16T03:19:11.647411Z","shell.execute_reply":"2025-01-16T03:19:11.686301Z"}}}]}