{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84795,"databundleVersionId":10821240,"sourceType":"competition"},{"sourceId":9212312,"sourceType":"datasetVersion","datasetId":5570435},{"sourceId":10517184,"sourceType":"datasetVersion","datasetId":6509921}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":22.341371,"end_time":"2024-12-11T03:22:13.479076","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-12-11T03:21:51.137705","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ############################################################################################### #\n# !pip install -q \\\n#     /kaggle/input/konwinski-prize/kprize_setup/kprize-1.1.0-py3-none-any.whl \\\n#     --no-index \\\n#     --find-links /kaggle/input/konwinski-prize/kprize_setup/kprize_setup/pip_packages/kprize\n# ############################################################################################### #\n# Do this instead of installing because for some reason we can't see the `bundling` part.\n# Since we have internet I can always install the dependencies as needed.\nimport sys; sys.path.insert(0, \"/kaggle/input/konwinski-prize/kprize_setup\")\n# ############################################################################################### #\n\nimport io\nimport os\nimport shutil\nimport subprocess\nfrom pathlib import Path\nfrom datasets import load_dataset\nfrom datasets.dataset_dict import DatasetDict\n\nimport pandas as pd; pd.options.mode.chained_assignment = None; pd.set_option('display.max_columns', None)\nimport polars as pl; print(f\"\\t\\t– POLARS VERSION: {pl.__version__}\")\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\")\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\")\n\n# Built-In Imports (mostly don't worry about these)\nfrom typing import Iterable, Any, Literal, Callable, Generator\nfrom kaggle_datasets import KaggleDatasets\nfrom dataclasses import dataclass\nfrom collections import Counter\nfrom datetime import datetime\nfrom zipfile import ZipFile\nfrom io import StringIO\nfrom glob import glob\nimport subprocess\nimport tempfile\nimport warnings\nimport requests\nimport textwrap\nimport hashlib\nimport imageio\nimport IPython\nimport urllib\nimport zipfile\nimport tarfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport copy\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport gc\nimport re\nimport os\n\n# Rich\nfrom rich import pretty; pretty.install()\nfrom rich.markdown import Markdown\nfrom rich import print as rprint\nfrom rich.console import Console\nfrom rich.style import Style\nfrom rich.live import Live\nfrom rich.text import Text\nfrom rich import inspect\nimport rich\n\n# --------------------------------------------------------- #\nimport kaggle_evaluation.konwinski_prize_inference_server\n# --------------------------------------------------------- #","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":14.873526,"end_time":"2024-12-11T03:22:08.818755","exception":false,"start_time":"2024-12-11T03:21:53.945229","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T16:12:53.044588Z","iopub.execute_input":"2025-01-19T16:12:53.045181Z","iopub.status.idle":"2025-01-19T16:13:06.092223Z","shell.execute_reply.started":"2025-01-19T16:12:53.045142Z","shell.execute_reply":"2025-01-19T16:13:06.090952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Rudimentary Paths\nBASE_DIR = \"/kaggle\"\nTMP_DIR = os.path.join(BASE_DIR, \"tmp\")\nWORKING_DIR = os.path.join(BASE_DIR, \"working\")\nINPUT_DIR = os.path.join(BASE_DIR, \"input\")\nTEMP_DIR = os.path.join(\"/tmp\")\n\n# Basic Competition Paths\nCOMP_DIR = os.path.join(INPUT_DIR, \"konwinski-prize\")\nCOMP_KAGGLE_EVALUATION_DIR = os.path.join(COMP_DIR, \"kaggle_evaluation\")\nCOMP_KPRIZE_SETUP_DIR = os.path.join(COMP_DIR, \"kprize_setup\")\n\n# Dataset Competition Paths\nCOMP_DATA_ZIP_PATH = os.path.join(COMP_DIR, \"data.a_zip\")\nCOMP_TMP_DIR = os.path.join(TMP_DIR, \"konwinski-prize-alt\")\nCOMP_TMP_DATA_DIR = os.path.join(COMP_TMP_DIR, \"data\")\nCOMP_DATA_PARQUET_PATH = os.path.join(COMP_TMP_DATA_DIR, \"data.parquet\")\nCOMP_CONDA_PACKAGES_DIR = os.path.join(COMP_TMP_DATA_DIR, \"conda_packages\")\nCOMP_PIP_PACKAGES_DIR = os.path.join(COMP_TMP_DATA_DIR, \"pip_packages\")\nCOMP_REPO_CONFIGS_DIR = os.path.join(COMP_TMP_DATA_DIR, \"repo_configs\")\nCOMP_REPOS_DIR = os.path.join(COMP_TMP_DATA_DIR, \"repos\")\n\n# SWE Dataset Paths ... https://huggingface.co/datasets/...\nHF_SWE_BENCH_PROVIDER = \"princeton-nlp\"\nHF_SWE_BENCH_PATH = os.path.join(HF_SWE_BENCH_PROVIDER, \"SWE-bench\")\nHF_SWE_BENCH_LITE_PATH = os.path.join(HF_SWE_BENCH_PROVIDER, \"SWE-bench_Lite\")\nHF_SWE_BENCH_VERIFIED_PATH = os.path.join(HF_SWE_BENCH_PROVIDER, \"SWE-bench_Verified\")\n\ndef load_kprize_df(add_local_paths: bool = True) -> pd.DataFrame:\n    \"\"\"Loader function\"\"\"\n    if not os.path.isfile(COMP_DATA_PARQUET_PATH):    \n        # Make the directory to unzip to\n        os.makedirs(COMP_TMP_DIR, exist_ok=True)\n        \n        # Open and extract the zip file\n        with ZipFile(COMP_DATA_ZIP_PATH, 'r') as zip_ref:\n            zip_ref.extractall(COMP_TMP_DIR)\n    _df = pd.read_parquet(COMP_DATA_PARQUET_PATH)\n\n    try:\n        if add_local_paths:\n            _df.insert(1, \"local_pip_packages_path\", _df.instance_id.apply(lambda x: os.path.join(COMP_PIP_PACKAGES_DIR , x)))\n            _df.insert(1, \"local_repo_path\", _df.instance_id.apply(lambda x: os.path.join(COMP_REPOS_DIR, f\"repo__{x}\")))\n    except:\n        print(f\"Could not add local path using {COMP_REPOS_DIR} as root competition directory path.\")\n    return _df\n    \n\n# Competition dataset for comparison\nkprize_df = load_kprize_df(add_local_paths=True)\n\n# Load the huggingface datasets (put in order so the biggest one is last)\n#   - Test with just the small one\nhf_datasets = {\n    \"swe_bench_lite\": load_dataset(HF_SWE_BENCH_LITE_PATH),          # N_EX = 300 + 23 = 323\n    #\"swe_bench_verified\": load_dataset(HF_SWE_BENCH_VERIFIED_PATH),  # N_EX = 500\n    #\"swe_bench\": load_dataset(HF_SWE_BENCH_PATH),                    # N_EX = 19008 + 2294 + 225 = 21527\n}\n\n# Let's see 'em\nrich.print(\"\\n\\nKPRIZE DATASET:\\n\")\ndisplay(kprize_df)\n\nrich.print(\"\\n\\n\\n\\nSWE BENCH DATASETS:\\n\")\nfor ds_name, ds in hf_datasets.items(): \n    rich.print(f\"\\n\\n\\n\\n[bold]{ds_name}[/bold]\")\n    display(ds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T16:13:06.093597Z","iopub.execute_input":"2025-01-19T16:13:06.094225Z","iopub.status.idle":"2025-01-19T16:13:14.728455Z","shell.execute_reply.started":"2025-01-19T16:13:06.094191Z","shell.execute_reply":"2025-01-19T16:13:14.727338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_json_file(\n    file_path: str | Path,\n    encoding: str = 'utf-8',\n    force_jsonl: bool = False\n) -> dict | list | list[dict]:\n    \"\"\"\n    Read JSON or JSONL files with custom encoding and format options.\n    \n    Args:\n        file_path (str | Path): \n            Path to the JSON/JSONL file\n        encoding (str, optional): \n            File encoding (default: utf-8)\n        force_jsonl (bool, optional): \n            Force reading as JSONL format\n        \n    Returns:\n        Contents of the file as a dict, list, or list of dicts\n    \"\"\"\n    # The file must exist\n    file_path = Path(file_path)\n    if not file_path.exists():\n        raise FileNotFoundError(f\"File not found: {file_path}\")\n\n    with open(file_path, 'r', encoding=encoding) as f:\n\n        # If doing JSONL we perform json.loads on each line and return\n        if force_jsonl:\n            return [json.loads(line.strip()) for line in f if line.strip()]\n            \n        # Otherwise we load the entire thing as JSON and only fallback to JSONL if that fails.\n        try:\n            return json.load(f)\n        except json.JSONDecodeError:\n            # If regular JSON fails, try JSONL\n            f.seek(0)\n            lines = f.readlines()\n            \n            try:\n                return [json.loads(line.strip()) for line in lines if line.strip()]\n            except json.JSONDecodeError as e:\n                raise json.JSONDecodeError(\n                    f\"File is neither valid JSON nor JSONL: {str(e)}\", \n                    e.doc, \n                    e.pos\n                )\n\n\ndef create_instances_dir_data(\n    hf_datasets: dict[str, DatasetDict],\n    instances_path: str = \"task_instances\",\n    dataset_specifier_str: str = \"{ds_name}-{split_name}\",\n    jsonl_filename: str = \"tasks.jsonl\"\n) -> None:\n    \"\"\"Convert Hugging Face datasets into JSONL files organized in directories.\n    \n    This function takes a dictionary of Hugging Face datasets and saves each split\n    as a JSONL file in a dedicated directory structure. The directory names are\n    formatted using the dataset name and split name.\n   \n    Args:\n        hf_datasets (str, datasets.dataset_dict.DatasetDict): \n            Dictionary mapping dataset names to DatasetDict objects\n        instances_path (str, optional): \n            Base directory to store the JSONL files\n        dataset_specifier_str (str, optional): \n            Template string for directory names.\n            Must use {ds_name} and {split_name} as placeholders.\n        jsonl_filename (str, optional): \n            Name of the output JSONL file in each directory\n       \n    Example:\n        >>> datasets = {'squad': squad_dataset, 'conll': conll_dataset}\n        >>> create_instances_dir_data(datasets)\n        # Creates structure like:\n        # task_instances/\n        #   squad-train/\n        #     tasks.jsonl\n        #   squad-validation/\n        #     tasks.jsonl\n        #   conll-train/\n        #     tasks.jsonl\n        #   ...\n    \"\"\"\n    # Iterate through each dataset in the dictionary\n    for ds_name, ds_dict in hf_datasets.items():\n        # Process each split (train/validation/test) in the dataset\n        for split_name, split_hf_dataset in ds_dict.items():\n            # Create the output directory path using the template\n            jsonl_output_dir = os.path.join(\n                instances_path,\n                dataset_specifier_str.format(ds_name=ds_name, split_name=split_name),\n            )\n            \n            # Create the output directory if it doesn't exist\n            if not os.path.isdir(jsonl_output_dir):\n                os.makedirs(jsonl_output_dir, exist_ok=True)\n            \n            # Save the dataset split as a JSONL file\n            split_hf_dataset.to_json(os.path.join(jsonl_output_dir, jsonl_filename))\n\n            # Print an update message\n            print(f\"Saved {ds_name} to {os.path.join(jsonl_output_dir, jsonl_filename)} ...\")\n\n\ncreate_instances_dir_data(hf_datasets)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T16:13:14.730872Z","iopub.execute_input":"2025-01-19T16:13:14.731304Z","iopub.status.idle":"2025-01-19T16:13:14.855178Z","shell.execute_reply.started":"2025-01-19T16:13:14.731262Z","shell.execute_reply":"2025-01-19T16:13:14.853725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from kprize.bundling.kprize_bundler import KPrizeBundler\n\n\ndef setup_bundler(\n    instances_path: str | Path = \"task_instances/\",\n    output_path: str | Path = \"dependencies/\",\n    compress_bundles: bool = False,\n    selected_splits: list[str] = None,\n) -> KPrizeBundler:\n    \"\"\"Initialize KPrize bundler with specified configuration.\n    \n    Args:\n        instances_path (str | Path): \n            Directory containing task instances\n        output_path (str | Path, optional): \n            Directory for output files\n        compress_bundles (bool, optional): \n            Whether to compress dataset bundles\n        selected_splits (list[str], optional): \n            List of specific splits to process (e.g., 'dev', 'test', etc.)\n    \"\"\"\n    # Create the bundler\n    bundler = KPrizeBundler(\n        instances_dir=Path(instances_path).absolute(),\n        output_dir=Path(output_path).absolute(), \n        bundle_compress=compress_bundles,\n        split_filter=selected_splits\n    )   \n    return bundler\n\n# Initialize bundler with default settings\npip_bundler = setup_bundler()\ninspect(pip_bundler, help=True, methods=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T16:13:14.85686Z","iopub.execute_input":"2025-01-19T16:13:14.857319Z","iopub.status.idle":"2025-01-19T16:13:14.933271Z","shell.execute_reply.started":"2025-01-19T16:13:14.85728Z","shell.execute_reply":"2025-01-19T16:13:14.932017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Would work if we had docker... but we don't...\n# def collect_dependencies(\n#     bundler: KPrizeBundler,\n#     dataset_id: str = \"dschettler8845/swe-kprize-cv-assets\",\n#     upload_kaggle: bool = True,\n#     skip_docker: bool = True,\n#     create_dataset: bool = True,\n#     force_docker_copy: bool = False,\n#     clear_instances: list[str] = None,\n# ) -> None:\n#     \"\"\"Run dependency collection process with specified settings.\n    \n#     Args:\n#         bundler (KPrizeBundler): \n#             Initialized KPrizeBundler instance\n#         dataset_id (str, optional): \n#             The kaggle dataset identifier\n#         upload_kaggle (bool, optional): \n#             Whether to upload results to Kaggle\n#         skip_docker (bool, optional): \n#             Skip running in Docker container\n#         create_dataset (bool, optional): \n#             Whether to create a new dataset\n#         force_docker_copy (bool, optional): \n#             Force copy assets to Docker\n#         clear_instances (list[str], optional): \n#             List of instance IDs to clear\n#     \"\"\"\n#     bundler.run_dependency_collection(\n#         upload_to_kaggle=upload_kaggle,\n#         skip_run_in_docker=skip_docker,\n#         skip_dataset_creation=not create_dataset,\n#         clear_instance_ids=clear_instances,\n#         dataset_id=dataset_id,\n#         force_copy_assets_to_docker=force_docker_copy,\n#     )\n\n# # Run collection with default settings\n# collect_dependencies(pip_bundler)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T16:13:14.934342Z","iopub.execute_input":"2025-01-19T16:13:14.934655Z","iopub.status.idle":"2025-01-19T16:13:14.941187Z","shell.execute_reply.started":"2025-01-19T16:13:14.934626Z","shell.execute_reply":"2025-01-19T16:13:14.940133Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**NOTE:**\n\n**`repos`** is just the list of unique values found under the \"repo\" field in our loaded SWE bench datasets.","metadata":{}},{"cell_type":"code","source":"def merge_hf_datasets_to_dataframe(\n    hf_datasets: dict[str, DatasetDict],\n    instances_path: str = \"task_instances\",\n    dataset_specifier_str: str = \"{ds_name}-{split_name}\",\n    jsonl_filename: str = \"tasks.jsonl\"\n) -> pd.DataFrame:\n    \"\"\"Merge multiple Hugging Face datasets into a single pandas DataFrame.\n    \n    This function takes a dictionary of Hugging Face DatasetDicts, extracts all\n    records, and combines them into a single DataFrame. It also adds:\n      - A 'dataset' column in the format '{dataset_name}/{split_name}'.\n      - A 'jsonl_path' column with the expected location of the corresponding JSONL file.\n\n    Args:\n        hf_datasets (dict[str, datasets.DatasetDict]): \n            Dictionary mapping dataset names to DatasetDict objects.\n        instances_path (str, optional): \n            Base directory where JSONL files are expected to be stored.\n        dataset_specifier_str (str, optional): \n            Template string for directory names. Must include {ds_name} and {split_name}.\n        jsonl_filename (str, optional): \n            Name of the JSONL file expected in each directory.\n\n    Returns:\n        pd.DataFrame: A DataFrame containing all dataset records, with additional columns.\n    \n    Example:\n        >>> df = merge_hf_datasets_to_dataframe(hf_datasets)\n        >>> df.head()\n    \n    Example Output:\n        | repo         | instance_id | base_commit | ... | dataset               | jsonl_path                          |\n        |-------------|------------|-------------|-----|----------------------|-----------------------------------|\n        | my_repo_1   | 1234       | abcde123    | ... | swe_bench_lite/dev   | task_instances/swe_bench_lite-dev/tasks.jsonl  |\n        | my_repo_2   | 5678       | fghij456    | ... | swe_bench/test       | task_instances/swe_bench-test/tasks.jsonl  |\n    \"\"\"\n    all_records = []\n\n    # Iterate through datasets and their splits\n    for ds_name, ds_dict in hf_datasets.items():\n        for split_name, split_dataset in ds_dict.items():\n            # Convert Dataset to pandas DataFrame\n            df = split_dataset.to_pandas()\n\n            # Add metadata columns\n            df[\"dataset\"] = f\"{ds_name}/{split_name}\"\n            df[\"jsonl_path\"] = os.path.join(\n                instances_path,\n                dataset_specifier_str.format(ds_name=ds_name, split_name=split_name),\n                jsonl_filename\n            )\n\n            all_records.append(df)\n\n    # Concatenate all DataFrames into one\n    merged_df = pd.concat(all_records, ignore_index=True)\n\n    return merged_df\n\n# All merged together\ndf = merge_hf_datasets_to_dataframe(hf_datasets)\n\nall_unique_repos = sorted(df.repo.unique())\nrich.print(all_unique_repos)\n\nall_instances = []\nfor x in glob(os.path.join(\"task_instances\", \"**\", \"*.jsonl\"), recursive=True):\n    all_instances.extend(read_json_file(x))\n\n# Sql fluff is stupid\nall_instances_without_sqlfluff = [x for x in all_instances if 'sqlfluff' not in x[\"repo\"]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T16:13:14.94242Z","iopub.execute_input":"2025-01-19T16:13:14.942975Z","iopub.status.idle":"2025-01-19T16:13:15.019275Z","shell.execute_reply.started":"2025-01-19T16:13:14.942898Z","shell.execute_reply":"2025-01-19T16:13:15.018153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from kprize.bundling.repo_collector import RepoCollector\n\n_output_dir = Path(os.path.join(TEMP_DIR, \"dependencies\"))\n_collected_deps_dir = Path(os.path.join(_output_dir, \"collected\"))\n_collected_repos_dir = Path(os.path.join(_collected_deps_dir, \"repos\"))\n_collected_instance_repos_dir = Path(os.path.join(_collected_deps_dir, \"instance_repos\"))\n_collected_pip_packages_dir = Path(os.path.join(_collected_deps_dir, \"pip_packages\"))\n_collected_python_packages_dir = Path(os.path.join(_collected_deps_dir, \"python3.11\"))\n_collected_uv_packages_dir = Path(os.path.join(_collected_deps_dir, \"uv\"))\n\nrepo_collector = RepoCollector(\n    collected_repos_dir=_collected_repos_dir,\n    collected_instance_repos_dir=_collected_instance_repos_dir,\n)\n\n# Collect the repos we need\nrepo_collector.run_repo_collection(all_unique_repos)\nrepo_collector.run_instance_repo_collection(all_instances_without_sqlfluff)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T16:29:34.116972Z","iopub.execute_input":"2025-01-19T16:29:34.117323Z","iopub.status.idle":"2025-01-19T16:29:41.958729Z","shell.execute_reply.started":"2025-01-19T16:29:34.117297Z","shell.execute_reply":"2025-01-19T16:29:41.957488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# UV\nfrom kprize.bundling.uv_downloader import UvDownloader\nos.makedirs(_collected_uv_packages_dir, exist_ok=True)\nUvDownloader.download(_collected_uv_packages_dir)\n\n# Python Debs\nshutil.copytree(\"/kaggle/input/python-3-11-debs/python3.11/\", _collected_python_packages_dir, dirs_exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T16:32:16.509537Z","iopub.execute_input":"2025-01-19T16:32:16.509882Z","iopub.status.idle":"2025-01-19T16:32:17.747868Z","shell.execute_reply.started":"2025-01-19T16:32:16.509852Z","shell.execute_reply":"2025-01-19T16:32:17.746623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_docker_pip_packages_dir = Path(\"/tmp/docker/pip_packages\")\ninstances_without_pip_packages = []\ninstances_with_pip_packages = []\nfor instance in all_instances_without_sqlfluff:\n    instance_pip_dir = _docker_pip_packages_dir / instance[\"instance_id\"]\n    if instance_pip_dir.exists():\n        instances_with_pip_packages.append(instance)\n    else:\n        instances_without_pip_packages.append(instance)\nprint(f\"Instances without pip packages: {len(instances_without_pip_packages)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T16:40:05.880663Z","iopub.execute_input":"2025-01-19T16:40:05.881042Z","iopub.status.idle":"2025-01-19T16:40:05.894765Z","shell.execute_reply.started":"2025-01-19T16:40:05.881013Z","shell.execute_reply":"2025-01-19T16:40:05.89324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from kprize.collection.configs.make_configs import make_configs\nmake_configs(True, \"repo_configs\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T18:17:54.714325Z","iopub.execute_input":"2025-01-19T18:17:54.714733Z","iopub.status.idle":"2025-01-19T18:17:54.757805Z","shell.execute_reply.started":"2025-01-19T18:17:54.714703Z","shell.execute_reply":"2025-01-19T18:17:54.756647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from kprize.collection.configs.repo_config import RepoConfig\n\n# def create_repo_config(repo_path: str, specs_dict: dict) -> RepoConfig:\n#     repo_name = repo_path.split(\"/\")[-1]\n\n#     return RepoConfig.from_dict(\n#         {\n#             \"repo_name\": repo_name,\n#             \"repo_path\": repo_path,\n#             \"github_url\": f\"https://github.com/{repo_path}\",\n#             \"log_parser\": \"parse_log_pytest\",\n#             \"specs\": specs_dict,\n#         }\n#     )\n\n# SPECS_HUMANEVAL = {\n#     k: {\"python\": \"3.9\", \"test_cmd\": \"python\"} \n#     for k in [\"1.0\"]\n# }\n# SPECS_DBT_CORE = {\n#     k: {\"python\": \"3.9\", \"packages\": \"requirements.txt\", \"install\": \"python -m pip install -e .\", \"test_cmd\": \"pytest -rA\"}\n#     for k in [\"0.13\", \"0.14\", \"0.15\", \"0.16\", \"0.17\", \"0.18\", \"0.19\", \"0.20\", \"0.21\", \"1.0\", \"1.1\", \"1.2\", \"1.3\", \"1.4\", \"1.5\", \"1.6\", \"1.7\",]\n# }\n# _MAP_REPO_VERSION_TO_SPECS_PY = {\"humaneval\": SPECS_HUMANEVAL, \"dbt-core\": SPECS_DBT_CORE}\n\n# _output_dir = Path(\"repo_configs\")\n# for repo_path, specs_dict in _MAP_REPO_VERSION_TO_SPECS_PY.items():\n#     repo_name = repo_path.split(\"/\")[-1]\n#     output_file = _output_dir / f\"{repo_name}.json\"\n#     config = create_repo_config(repo_path, specs_dict)\n#     config.to_json(output_file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T17:18:00.293482Z","iopub.execute_input":"2025-01-19T17:18:00.293953Z","iopub.status.idle":"2025-01-19T17:18:00.305245Z","shell.execute_reply.started":"2025-01-19T17:18:00.293899Z","shell.execute_reply":"2025-01-19T17:18:00.303701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from kprize.collection.configs.repo_config import RepoConfig\n# # !pip install anthropic\n# import kprize.collection.validation.validator\n# import kprize.collection.collector\n\n# collector = kprize.collection.collector.Collector(\n#     Path(\"dependencies/docker/input/kprize-assets/repos\"),\n    \n# )\n\n\ndef _update_repo_config(instances: list[dict[str, Any]]) -> None:\n    if len(instances) == 0:\n        return\n    for repo in all_unique_repos:\n        repo = instances[0][\"repo\"]\n        repo_stem = RepoConfig.map_repo_path_to_repo_stem(repo).rsplit(\"__\", 1)[-1]\n        repo_config_path = Path(\"/kaggle/working/repo_configs\") / f\"{repo_stem}.json\"\n        if repo_config_path.exists():\n            repo_config=RepoConfig.from_json(repo_config_path)\n            print(\".\", end=\"\")\n        else:\n            print(repo_stem,)\n            raise ValueError\n            # repo_config=RepoConfig(\n            #     repo_name=repo.split(\"/\")[-1],\n            #     repo_path=repo,\n            #     github_url=f\"https://github.com/{repo}\",\n            #     log_parser=self.default_config.log_parser,\n            #     specs={},\n            # )\n    \n        if len(repo_config.specs) > 0:\n            print(\".\", end=\"\")\n            default_version, default_specs = max(repo_config.specs.items(), key=lambda x: x[0])\n        else:\n            print(repo_stem,)\n            raise ValueError\n    \n        is_updated = False\n    \n        for instance in instances:\n            any_fail_to_pass = len(instance.get(\"FAIL_TO_PASS\", [])) > 0\n    \n            if any_fail_to_pass and \"version\" in instance and instance[\"version\"] not in repo_config.specs:\n                repo_config.specs[instance[\"version\"]] = default_specs\n                is_updated = True\n            elif any_fail_to_pass and default_version not in repo_config.specs and len(repo_config.specs) == 0:\n                repo_config.specs[default_version] = default_specs\n                is_updated = True\n    \n        if is_updated:\n            repo_config.to_json(repo_config_path)\n\n\n# _update_repo_config(all_instances_without_sqlfluff)\n\n# Update with default\nfor f_path in glob(\"repo_configs/*.json\"):\n    config = read_json_file(f_path)\n    config[\"specs\"][\"default\"] = config[\"specs\"][next(iter(config[\"specs\"]))]\n    with open(f_path, 'w') as f:\n        json.dump(config, f, indent=4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T18:18:21.197837Z","iopub.execute_input":"2025-01-19T18:18:21.198228Z","iopub.status.idle":"2025-01-19T18:18:21.221284Z","shell.execute_reply.started":"2025-01-19T18:18:21.1982Z","shell.execute_reply":"2025-01-19T18:18:21.219977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import argparse\nimport tomli as tomllib\n\n\ndef current_ms():\n    return round(time.time_ns() / 1000000)\n\ndef seconds_since(time_ms):\n    return (current_ms() - time_ms) / 1000\n\ndef create_dir_path(path: Path|str) -> Path:\n    if isinstance(path, str):\n        path = Path(path)\n    path.mkdir(parents=True, exist_ok=True)\n    return path\n\ndef run_commands(cmds: list[str], log_file=None, console_log=True, log_commands=False):\n    if log_file:\n        print(f\"Writing command logs to {log_file.name}\")\n    commands = '\\n'.join(cmds)\n    if log_commands:\n        print(\"Running commands:\")\n        print(commands)\n        print(\"\\n\")\n    process = subprocess.Popen(\n        '/bin/bash',\n        stdin=subprocess.PIPE,\n        stdout=subprocess.PIPE,\n        stderr=subprocess.PIPE,\n        text=True)\n    out, err = process.communicate(commands)\n    if console_log:\n        print(\"Command stderr:\")\n        print(err)\n        print(\"Command stdout:\")\n        print(out)\n    if log_file:\n        log_file.write(err)\n        log_file.write(out)\n\ndef get_repo_config_name_from_repo_path(instance_id: str) -> str:\n    return instance_id.rsplit(\"-\", maxsplit=1)[0]\n\ndef convert_pip_install_to_pip_download(cmd_pip_install: str, download_dir_path: Path) -> str:\n    \"\"\"\n    Convert a `pip install` command to a `pip download` command\n    \"\"\"\n    cmd_pip_download = cmd_pip_install.replace(\"install\", f\"download -d {download_dir_path}\")\n    cmd_pip_download = cmd_pip_download.replace(\"-e\", \"\").replace(\"--verbose\", \"\")\n    return cmd_pip_download\n\ndef get_build_system_requires(toml_path: Path) -> list[str]:\n    requirements = []\n    toml_data = tomllib.loads(toml_path.read_text())\n    # get build-system requires\n    build_system = toml_data.get(\"build-system\", None)\n    if build_system:\n        requirements.extend(build_system.get(\"requires\", []))\n        # get build-system build-backend\n        build_backend = build_system.get(\"build-backend\", None)\n        if build_backend:\n            if build_backend == \"setuptools.build_meta\":\n                requirements.append(\"wheel\")\n            elif build_backend == \"hatchling.build\":\n                requirements.append(\"editables\")\n    # wrap in double quotes for special cases\n    # e.g. \"setuptools >= 65.5.1\", \"setuptools_scm[toml]\", \"cython>=3.0.0, <4\"\n    return list(map(lambda r: f'\"{r}\"', requirements))\n\ndef download_pip_packages(\n    repos_path: str = \"dependencies/docker/input/kprize-assets/repos\",\n    repo_configs_path: str = \"dependencies/docker/input/kprize-assets/repo_configs\",\n    output_path: str = \"dependencies/collected/downloaded_pip_packages\",\n    instances_json_file: str = \"task_instances/test/q3-task-instances-all-new-log-parser-test.jsonl\",\n    limit: int = None,\n    python_exec: str = \"python3.11\"\n) -> tuple[list, list, list, list]:\n    \"\"\"\n    Run `pip download` for each repo in a given directory of git repos\n    \n    Args:\n        repos_path: Path to repos directory\n        repo_configs_path: Path to repo configs directory\n        output_path: Path to output downloaded pip packages\n        instances_json_file: Path to task instances json file\n        limit: Limit number of instances to process\n        python_exec: Python executable to use\n    \n    Returns:\n        tuple containing lists of:\n        - skipped_repos: repos that were already processed\n        - collected_repos: repos successfully processed\n        - missing_install_cmd_repos: repos missing install commands\n        - missing_download_cmd_repos: repos where download command creation failed\n    \"\"\"\n    repos_path = Path(repos_path)\n    repo_configs_path = Path(repo_configs_path)\n    output_path = Path(output_path)\n    instances_json_file = Path(instances_json_file)\n\n    print(\" > repos_path:\", repos_path)\n    print(\" > repo_configs_path:\", repo_configs_path)\n    print(\" > output_path:\", output_path)\n    print(\" > instances_json_file:\", instances_json_file)\n    print(\" > limit:\", limit)\n\n    if not repos_path.exists():\n        raise ValueError(f\"Error: repos path does not exist: {repos_path}\")\n    if not repo_configs_path.exists():\n        raise ValueError(f\"Error: repo configs path does not exist: {repo_configs_path}\")\n    if not instances_json_file.exists():\n        raise ValueError(f\"Error: instances json file does not exist: {instances_json_file}\")\n    if not output_path.exists():\n        create_dir_path(output_path)\n\n    instance_ids = []\n    with open(instances_json_file, \"r\") as f:\n        for line in f:\n            instance = json.loads(line)\n            instance_ids.append(instance[\"instance_id\"])\n    print(\" > instance_ids:\", instance_ids)\n\n    start_ms = current_ms()\n    skipped_repos = []\n    missing_install_cmd_repos = []\n    missing_download_cmd_repos = []\n    collected_repos = []\n\n    if (limit is not None) and (limit > 0):\n        print(f\"\\nLimiting to {limit} instances\")\n        instance_ids = instance_ids[:limit]\n\n    for instance_id in instance_ids:\n        repo_start_ms = current_ms()\n        repo_name = f\"repo__{instance_id}\"\n        repo_path = repos_path / repo_name\n\n        repo_pip_packages_path = output_path / instance_id\n        if repo_pip_packages_path.exists():\n            # skip if already downloaded pip packages for this repo\n            skipped_repos.append(repo_name)\n            continue\n\n        # get repo config\n        repo_config_name = get_repo_config_name_from_repo_path(instance_id).rsplit(\"__\", 1)[-1]\n        repo_config_path = repo_configs_path / f\"{repo_config_name}.json\"\n        if not repo_config_path.exists():\n            # raise ValueError(f\"Error: repo config does not exist: {repo_config_path}\")\n            print(f'ValueError(f\"Error: repo config does not exist: {repo_config_path}\") ... SKIPPING')\n            continue\n            \n        repo_config = json.loads(Path(repo_config_path).read_text())\n\n        specs = repo_config[\"specs\"]\n        if not specs.get(\"default\", None):\n            # error: no default install command in repo config\n            missing_install_cmd_repos.append(repo_name)\n            continue\n\n        cmd_install = specs[\"default\"][\"install\"]\n\n        # add extra pip packages\n        extra_pip_packages = specs[\"default\"].get(\"pip_packages\", None)\n        if extra_pip_packages:\n            cmd_install = f\"{cmd_install} && pip install {' '.join(extra_pip_packages)}\"\n\n        # get build-system requirements\n        toml_path = repo_path / \"pyproject.toml\"\n        if toml_path.exists():\n            build_requirements = get_build_system_requires(toml_path)\n            print(f\" > Found build-system requirements in {toml_path.absolute()}:\\n > {build_requirements}\")\n            if build_requirements and len(build_requirements) > 0:\n                cmd_install = f\"{cmd_install} && pip install {' '.join(build_requirements)}\"\n\n        # set env vars\n        cmd_env_vars = specs[\"default\"].get(\"env_vars\", None)\n        if cmd_env_vars:\n            env_vars = ' '.join(list(cmd_env_vars))\n            cmd_install = ' && '.join(map(lambda c: f\"{env_vars} {c}\", cmd_install.split(' && ')))\n\n        # convert `pip install` to `pip download`\n        cmd_download = convert_pip_install_to_pip_download(cmd_install, repo_pip_packages_path.absolute())\n        if \"download\" not in cmd_download:\n            missing_download_cmd_repos.append(repo_name)\n            print(f\"Error: no download command in repo config: {repo_config_path}\")\n            continue\n\n        cmds = [\n            f\"cd {repo_path}\",\n            \"rm -rf venv\",\n            f\"{python_exec} -m venv venv\",\n            \"source venv/bin/activate\",\n            cmd_download,\n            \"deactivate\"\n        ]\n        # run `pip download`\n        run_commands(\n            cmds,\n            log_commands=True\n        )\n        collected_repos.append(repo_name)\n        print(f\" > Finished `pip download` for repo: {repo_name} in {seconds_since(repo_start_ms)}s\")\n    \n    print(f\"\\nFinished `pip download` for {len(instance_ids)} repos in {seconds_since(start_ms)}s\")\n    print(f\"Already collected repos (skipped): {skipped_repos}\")\n    print(f\"Collected repos: {collected_repos}\")\n    print(f\"Missing install command repos: {missing_install_cmd_repos}\")\n    \n    return skipped_repos, collected_repos, missing_install_cmd_repos, missing_download_cmd_repos","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T18:18:23.236869Z","iopub.execute_input":"2025-01-19T18:18:23.237315Z","iopub.status.idle":"2025-01-19T18:18:23.262574Z","shell.execute_reply.started":"2025-01-19T18:18:23.237282Z","shell.execute_reply":"2025-01-19T18:18:23.26087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mkdir -p /tmp/dependencies/collected/download_pip_packages\n\n# Dev SWE Bench Lite\ndownload_pip_packages(\n    repos_path = \"/tmp/dependencies/collected/repos\",\n    repo_configs_path = \"/kaggle/working/repo_configs\",\n    output_path = \"/tmp/dependencies/collected/download_pip_packages\",\n    instances_json_file = \"/kaggle/working/task_instances/swe_bench_lite-dev/tasks.jsonl\",\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T18:18:23.673191Z","iopub.execute_input":"2025-01-19T18:18:23.673536Z","iopub.status.idle":"2025-01-19T18:18:47.086244Z","shell.execute_reply.started":"2025-01-19T18:18:23.673509Z","shell.execute_reply":"2025-01-19T18:18:47.084819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"read_json_file(\"/kaggle/working/repo_configs/astropy.json\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T18:17:10.255485Z","iopub.execute_input":"2025-01-19T18:17:10.255896Z","iopub.status.idle":"2025-01-19T18:17:10.520519Z","shell.execute_reply.started":"2025-01-19T18:17:10.255861Z","shell.execute_reply":"2025-01-19T18:17:10.519191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}