{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84795,"databundleVersionId":10821240,"sourceType":"competition"},{"sourceId":9212312,"sourceType":"datasetVersion","datasetId":5570435},{"sourceId":10517184,"sourceType":"datasetVersion","datasetId":6509921}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":22.341371,"end_time":"2024-12-11T03:22:13.479076","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-12-11T03:21:51.137705","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ############################################################################################### #\n!pip install -q /kaggle/input/konwinski-prize/kprize_setup/kprize-1.1.0-py3-none-any.whl\n# ############################################################################################### #\n# Do this instead of installing because for some reason we can't see the `bundling` part.\n# Since we have internet I can always install the dependencies as needed.\nimport sys; sys.path.insert(0, \"/kaggle/input/konwinski-prize/kprize_setup\")\n# ############################################################################################### #\n\nimport io\nimport os\nimport shlex\nimport logging\nimport shutil\nimport subprocess\nfrom enum import Enum\nfrom pathlib import Path\nfrom datasets import load_dataset\nfrom datasets.dataset_dict import DatasetDict\n\nimport pandas as pd; pd.options.mode.chained_assignment = None; pd.set_option('display.max_columns', None)\nimport polars as pl; print(f\"\\t\\t– POLARS VERSION: {pl.__version__}\")\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\")\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\")\n\n# Built-In Imports (mostly don't worry about these)\nfrom typing import Iterable, Any, Literal, Callable, Generator\nfrom kaggle_datasets import KaggleDatasets\nfrom dataclasses import dataclass\nfrom collections import Counter\nfrom datetime import datetime\nfrom zipfile import ZipFile\nfrom io import StringIO\nfrom glob import glob\nimport subprocess\nimport tempfile\nimport warnings\nimport requests\nimport textwrap\nimport hashlib\nimport imageio\nimport IPython\nimport urllib\nimport zipfile\nimport tarfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport uuid\nimport copy\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport gc\nimport re\nimport os\n\n# Rich\nfrom rich import pretty; pretty.install()\nfrom rich.markdown import Markdown\nfrom rich import print as rprint\nfrom rich.console import Console\nfrom rich.style import Style\nfrom rich.live import Live\nfrom rich.text import Text\nfrom rich import inspect\nimport rich\n\n# --------------------------------------------------------- #\nimport kaggle_evaluation.konwinski_prize_inference_server\n# --------------------------------------------------------- #","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":14.873526,"end_time":"2024-12-11T03:22:08.818755","exception":false,"start_time":"2024-12-11T03:21:53.945229","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:40:12.153437Z","iopub.execute_input":"2025-01-26T19:40:12.153779Z","iopub.status.idle":"2025-01-26T19:40:26.750794Z","shell.execute_reply.started":"2025-01-26T19:40:12.153751Z","shell.execute_reply":"2025-01-26T19:40:26.74955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Rudimentary Paths\nBASE_DIR = \"/kaggle\"\nTMP_DIR = os.path.join(BASE_DIR, \"tmp\")\nWORKING_DIR = os.path.join(BASE_DIR, \"working\")\nINPUT_DIR = os.path.join(BASE_DIR, \"input\")\nTEMP_DIR = os.path.join(\"/tmp\")\n\n# Basic Competition Paths\nCOMP_DIR = os.path.join(INPUT_DIR, \"konwinski-prize\")\nCOMP_KAGGLE_EVALUATION_DIR = os.path.join(COMP_DIR, \"kaggle_evaluation\")\nCOMP_KPRIZE_SETUP_DIR = os.path.join(COMP_DIR, \"kprize_setup\")\n\n# Dataset Competition Paths\nCOMP_DATA_ZIP_PATH = os.path.join(COMP_DIR, \"data.a_zip\")\nCOMP_TMP_DIR = os.path.join(TMP_DIR, \"konwinski-prize-alt\")\nCOMP_TMP_DATA_DIR = os.path.join(COMP_TMP_DIR, \"data\")\nCOMP_DATA_PARQUET_PATH = os.path.join(COMP_TMP_DATA_DIR, \"data.parquet\")\nCOMP_CONDA_PACKAGES_DIR = os.path.join(COMP_TMP_DATA_DIR, \"conda_packages\")\nCOMP_PIP_PACKAGES_DIR = os.path.join(COMP_TMP_DATA_DIR, \"pip_packages\")\nCOMP_REPO_CONFIGS_DIR = os.path.join(COMP_TMP_DATA_DIR, \"repo_configs\")\nCOMP_REPOS_DIR = os.path.join(COMP_TMP_DATA_DIR, \"repos\")\n\n# SWE Dataset Paths ... https://huggingface.co/datasets/...\nHF_SWE_BENCH_PROVIDER = \"princeton-nlp\"\nHF_SWE_BENCH_PATH = os.path.join(HF_SWE_BENCH_PROVIDER, \"SWE-bench\")\nHF_SWE_BENCH_LITE_PATH = os.path.join(HF_SWE_BENCH_PROVIDER, \"SWE-bench_Lite\")\nHF_SWE_BENCH_VERIFIED_PATH = os.path.join(HF_SWE_BENCH_PROVIDER, \"SWE-bench_Verified\")\n\ndef load_kprize_df(add_local_paths: bool = True) -> pd.DataFrame:\n    \"\"\"Loader function\"\"\"\n    if not os.path.isfile(COMP_DATA_PARQUET_PATH):    \n        # Make the directory to unzip to\n        os.makedirs(COMP_TMP_DIR, exist_ok=True)\n        \n        # Open and extract the zip file\n        with ZipFile(COMP_DATA_ZIP_PATH, 'r') as zip_ref:\n            zip_ref.extractall(COMP_TMP_DIR)\n    _df = pd.read_parquet(COMP_DATA_PARQUET_PATH)\n\n    try:\n        if add_local_paths:\n            _df.insert(1, \"local_pip_packages_path\", _df.instance_id.apply(lambda x: os.path.join(COMP_PIP_PACKAGES_DIR , x)))\n            _df.insert(1, \"local_repo_path\", _df.instance_id.apply(lambda x: os.path.join(COMP_REPOS_DIR, f\"repo__{x}\")))\n    except:\n        print(f\"Could not add local path using {COMP_REPOS_DIR} as root competition directory path.\")\n    return _df\n    \n\n# Competition dataset for comparison\nkprize_df = load_kprize_df(add_local_paths=True)\n\n# Load the huggingface datasets (put in order so the biggest one is last)\n#   - Test with just the small one\nhf_datasets = {\n    \"swe_bench_lite\": load_dataset(HF_SWE_BENCH_LITE_PATH),          # N_EX = 300 + 23 = 323\n    \"swe_bench_verified\": load_dataset(HF_SWE_BENCH_VERIFIED_PATH),  # N_EX = 500\n    \"swe_bench\": load_dataset(HF_SWE_BENCH_PATH),                    # N_EX = 19008 + 2294 + 225 = 21527\n}\n\nhf_dfs = {}\nfor k,v in hf_datasets.items():\n    hf_dfs[k] = {}\n    for _k, _v in v.items():\n        _df = pd.DataFrame(_v)\n        for col in [\"PASS_TO_PASS\", \"FAIL_TO_PASS\"]:\n            _df[col] = _df[col].apply(lambda x: ast.literal_eval(x))\n        hf_dfs[k][_k] = _df.copy()\n\n# Let's see 'em\nrich.print(\"\\n\\nKPRIZE DATASET:\\n\")\ndisplay(kprize_df)\n\nrich.print(\"\\n\\n\\n\\nSWE BENCH DATASETS:\\n\")\nfor ds_name, ds in hf_datasets.items(): \n    rich.print(f\"\\n\\n\\n\\n[bold]{ds_name}[/bold]\")\n    display(ds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:40:26.752238Z","iopub.execute_input":"2025-01-26T19:40:26.752911Z","iopub.status.idle":"2025-01-26T19:40:49.903552Z","shell.execute_reply.started":"2025-01-26T19:40:26.752878Z","shell.execute_reply":"2025-01-26T19:40:49.902272Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Explore the complexities of managing environments... within a notebook ... within a docker","metadata":{}},{"cell_type":"code","source":"import tomli\nfrom packaging.requirements import Requirement\nfrom packaging.specifiers import SpecifierSet\nfrom packaging.version import Version, parse\nfrom contextlib import contextmanager\n\n@contextmanager\ndef suppress_logging_below(level):\n    logger = logging.getLogger()\n    old_level = logger.level\n    logger.setLevel(level)\n    try:\n        yield\n    finally:\n        logger.setLevel(old_level)\n\nclass PythonImplementation(Enum):\n    \"\"\"Enumeration of Python implementations.\"\"\"\n    CPYTHON = \"cpython\"\n    PYPY = \"pypy\"\n\n\n@dataclass\nclass PythonVersion:\n    \"\"\"Represents a Python version with its implementation and availability.\n    \n    Attributes:\n        implementation (PythonImplementation): The Python implementation (e.g., CPython, PyPy)\n        version (str): The full version string (e.g., '3.10.12')\n        is_installed (bool): Whether the version is installed on the system\n        path (str | Path | None): The path to the Python executable\n        is_freethreaded (bool): Whether the Python version is free-threaded\n            Free-threaded Python allows multiple threads to run simultaneously.\n    \"\"\"\n    implementation: PythonImplementation\n    version: str\n    is_installed: bool\n    path: str | Path | None = None\n    is_freethreaded: bool = False\n\n    @property\n    def base_version(self) -> str:\n        \"\"\"Returns the base version (e.g., '3.10' from '3.10.12').\"\"\"\n        return '.'.join(self.version.split('.')[:2])\n\n    @property\n    def parsed_version(self) -> Version:\n        \"\"\"Returns a packaging.version.Version object for comparison.\"\"\"\n        return parse(self.version)\n\n@dataclass\nclass VersionConstraints:\n    \"\"\"Represents Python version constraints with metadata.\n    \n    Attributes:\n        min_version (str | None): Minimum version constraint\n        max_version (str | None): Maximum version constraint\n        specifier_set (SpecifierSet): Set of version specifiers\n        source_file (str): The file where the constraints were extracted from\n        confidence (float): Confidence level of the constraints (0.0 to 1.0)\n    \"\"\"\n    min_version: str | None\n    max_version: str | None\n    specifier_set: SpecifierSet\n    source_file: str\n    confidence: float  # 0.0 to 1.0\n\n    @classmethod\n    def from_specifier_string(cls, spec_str: str, source_file: str, confidence: float = 1.0) -> 'VersionConstraints':\n        \"\"\"Create VersionConstraints from a version specifier string.\n        \n        Args:\n            spec_str (str): The version specifier string\n            source_file (str): The file where the constraints were extracted from\n            confidence (float): Confidence level of the constraints (0.0 to 1.0)\n            \n        Returns:\n            VersionConstraints: The parsed version constraints\n        \"\"\"\n        # (1) Parse the version specifier string\n        spec_set = SpecifierSet(spec_str)\n\n        # (2) Extract min and max versions from specifiers\n        min_ver = None\n        max_ver = None\n        # (2a) Iterate over the specifiers\n        for spec in spec_set:\n            # (2b) Skip non-version specifiers\n            ver_str = spec.version\n            # (2c) Check for minimum and maximum versions\n            if spec.operator in ('>=', '>'):\n                # (2d) Update min_version if needed\n                if min_ver is None or parse(ver_str) > parse(min_ver):\n                    min_ver = ver_str\n            # (2e) Check for maximum version\n            elif spec.operator in ('<=', '<'):\n                # (2f) Update max_version if needed\n                if max_ver is None or parse(ver_str) < parse(max_ver):\n                    max_ver = ver_str\n        # (3) Create and return the VersionConstraints object\n        return cls(\n            min_version=min_ver,\n            max_version=max_ver,\n            specifier_set=spec_set,\n            source_file=source_file,\n            confidence=confidence\n        )\n\n@dataclass\nclass CommandResult:\n    \"\"\"Holds information about a subprocess command result.\n\n    Args:\n        command (str): The command that was executed.\n        returncode (int): The return code of the command.\n        stdout (str): The standard output of the command.\n        stderr (str): The standard error of the command.\n    \"\"\"\n    command: str\n    returncode: int\n    stdout: str\n    stderr: str\n\n    @property\n    def success(self) -> bool:\n        \"\"\"Indicates if returncode == 0.\"\"\"\n        return self.returncode == 0\n\n    def __str__(self) -> str:\n        \"\"\"Informal string representation, used for user-facing display.\"\"\"\n        if self.success:\n            return \"CommandResult(command={!r}, returncode={!r}, success={!r}, stdout={!r})\".format(\n                self.command, self.returncode, self.success, self.stdout\n            )\n        else:\n            return \"CommandResult(command={!r}, returncode={!r}, success={!r}, stdout={!r}, stderr={!r})\".format(\n                self.command, self.returncode, self.success, self.stdout, self.stderr\n            )\n\n    def __repr__(self) -> str:\n        \"\"\"Official string representation, used for debugging.\"\"\"\n        return str(self)\n\n    def raise_for_status(self) -> None:\n        \"\"\"Raise a subprocess.CalledProcessError if the command failed.\n\n        Raises:\n            subprocess.CalledProcessError: If the command's return code is non-zero.\n        \"\"\"\n        if not self.success:\n            raise subprocess.CalledProcessError(\n                returncode=self.returncode,\n                cmd=self.command,\n                output=self.stdout,\n                stderr=self.stderr\n            )\n\n\n@dataclass\nclass EnvironmentConfig:\n    \"\"\"Configuration for UV environment setup.\n    \n    Args:\n        python_version (str, optional): \n            Python version to use (e.g. \"3.10\")\n        base_dir (str, optional): \n            Base directory for environments to be stored.\n            NOTE: We default to the /kaggle/tmp directory as it\n                  has access to a larger storage volume.\n        pytest_options (str, optional): \n            Additional pytest options to pass.\n    \"\"\"\n    python_version: str = \"3.10\"\n    base_dir: str = \"/kaggle/tmp\"\n    pytest_options: str = \"\"\n\n\n@dataclass\nclass TestResult:\n    \"\"\"Results from running tests in the environment.\n    \n    Args:\n        success (bool): Whether tests passed\n        output (str): Test output\n        error (str): Error output if any\n        duration (float): Test duration in seconds\n    \"\"\"\n    success: bool\n    output: str\n    error: str\n    duration: float\n\n\n@dataclass\nclass SWEBenchInstance:\n    \"\"\"Represents a single instance from the SWE-Bench dataset.\n    \n    Attributes:\n        repo (str): Repository URL or identifier (e.g. \"owner/repo\")\n        instance_id (str): Unique identifier for this instance\n        base_commit (str): The commit hash where the bug exists\n        patch (str): The code changes that fix or modify the bug\n        test_patch (str): The test changes associated with the fix\n        problem_statement (str): Description of the bug/issue\n        hints_text (str, optional): Additional hints or context about the bug\n        created_at (datetime): When the instance was created\n        version (str): Version identifier for this instance\n        fail_to_pass (list[str]): Whether this instance should go from failing to passing\n        pass_to_pass (list[str]): Whether this instance should maintain passing status\n        environment_setup_commit (str): Commit hash used for environment setup\n    \"\"\"\n    repo: str\n    instance_id: str\n    base_commit: str\n    patch: str\n    test_patch: str\n    problem_statement: str\n    hints_text: str | None\n    created_at: datetime\n    version: str\n    fail_to_pass: list[str]\n    pass_to_pass: list[str]\n    environment_setup_commit: str\n\n    @classmethod\n    def from_df_row(cls, row: Any) -> \"SWEBenchInstance\":\n        \"\"\"Create an instance from a pandas DataFrame row.\n        \n        Args:\n            row: A single row (pd.Series or dict) from the SWE-Bench-Lite dataset.\n        \n        Returns:\n            A SWEBenchInstance object populated with row data.\n        \"\"\"\n        return cls(\n            repo=row[\"repo\"],\n            instance_id=str(row[\"instance_id\"]),\n            base_commit=row[\"base_commit\"],\n            patch=row[\"patch\"],\n            test_patch=row[\"test_patch\"],\n            problem_statement=row[\"problem_statement\"],\n            hints_text=row[\"hints_text\"],\n            created_at=datetime.fromisoformat(row[\"created_at\"].replace('Z', '+00:00')),\n            version=row[\"version\"],\n            fail_to_pass=row[\"FAIL_TO_PASS\"],\n            pass_to_pass=row[\"PASS_TO_PASS\"],\n            environment_setup_commit=row[\"environment_setup_commit\"]\n        )\n\n    @property\n    def github_repo_url(self) -> str:\n        \"\"\"Constructs the full GitHub URL for the repo.\n\n        Returns:\n            str: The GitHub repo URL.\n        \"\"\"\n        return os.path.join(f\"https://github.com\", self.repo)\n        \n    @property\n    def github_pull_url(self) -> str:\n        \"\"\"Constructs the full GitHub URL for the PR that fixes the issue.\n\n        Returns:\n            str: The GitHub URL for the PR.\n        \"\"\"\n        pull_number = self.instance_id.rsplit(\"-\", 1)[-1]\n        return os.path.join(self.github_repo_url, \"pull\", pull_number)\n\n    @property\n    def repo_at_base_commit_url(self) -> str:\n        \"\"\"Constructs the full GitHub URL for the base commit (where the bug exists).\n\n        Returns:\n            str: The GitHub URL for the base commit.\n        \"\"\"\n        return os.path.join(self.github_repo_url, \"tree\", self.base_commit)\n\n    @property\n    def repo_at_environment_setup_commit_url(self) -> str:\n        \"\"\"Constructs the full GitHub URL for the environment setup commit.\n\n        Returns:\n            str: The GitHub URL for the environment setup commit.\n        \"\"\"\n        return os.path.join(self.github_repo_url, \"tree\", self.environment_setup_commit)\n\ndemo_df = hf_dfs[\"swe_bench_verified\"][\"test\"]\ndemo_row = demo_df.iloc[-1]\ndemo_instance = SWEBenchInstance.from_df_row(demo_row)\n\nrich.print(\"[bold]SWE-BENCH-VERIFIED DEMO INSTANCE:[/bold]\")\ndisplay(demo_instance)\ndisplay(demo_instance.github_pull_url)\ndisplay(demo_instance.repo_at_base_commit_url)\ndisplay(demo_instance.repo_at_environment_setup_commit_url)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:40:49.905502Z","iopub.execute_input":"2025-01-26T19:40:49.905826Z","iopub.status.idle":"2025-01-26T19:40:49.971518Z","shell.execute_reply.started":"2025-01-26T19:40:49.905799Z","shell.execute_reply":"2025-01-26T19:40:49.970373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class UVManager:\n    \"\"\"Manages a UV virtual environment with persistent shell session support.\n    \n    This class provides functionality to create, manage, and interact with UV virtual\n    environments. It supports both context manager and direct usage patterns, maintains\n    a persistent shell session, and provides methods for package installation and\n    command execution. This class was designed to be used within a Kaggle notebook\n    to allow the creation of small environments for recreating swe-bench \n    github issue environments.\n    \n    Attributes:\n        venv_path (Path): Absolute path to the virtual environment\n        python_version (str): Python version being used (e.g. \"3.10\")\n        env_ready (bool): Whether the environment is ready for use\n        \n    Example:\n        >>> # Using as context manager\n        >>> with UVManager(\"./my_venv\", python_version=\"3.10\") as uv:\n        ...     result = uv.pip_install(\"requests\")\n        ...     assert result.success\n        \n        >>> # Direct usage\n        >>> uv = UVManager(\"./other_venv\")\n        >>> uv.initialize()\n        >>> result = uv.send(\"python --version\")\n        >>> print(result.stdout)\n        Python 3.10.x\n        >>> uv.cleanup()\n    \"\"\"\n    \n    def __init__(\n        self, \n        venv_path: str | Path, \n        python_version: str = \"3.10\",\n        env_vars: dict[str, str] | None = None\n    ):\n        \"\"\"Initialize the UV environment manager.\n        \n        Args:\n            venv_path (str | Path): Path where virtual environment should be created\n            python_version (str): Python version to use (e.g. \"3.10\")\n            env_vars (dict[str, str] | None): Additional environment variables to set in the shell\n            \n        Note:\n            This doesn't create the environment immediately. Call initialize()\n            or use as context manager to create and activate the environment.\n        \"\"\"\n        self.venv_path = Path(venv_path).absolute()\n        self.python_version = python_version\n        self._shell = None\n        self._logger = logging.getLogger(self.__class__.__name__)\n        self.env_ready = False\n        \n        # Base environment variables\n        self._env_vars = {\n            'UV_LINK_MODE': 'copy',  # Prevent hardlink warnings\n            'VIRTUAL_ENV': str(self.venv_path),\n            'PATH': f\"{self.venv_path}/bin:{os.environ.get('PATH', '')}\"\n        }\n        if env_vars:\n            self._env_vars.update(env_vars)\n            \n    def __enter__(self) -> 'UVManager':\n        \"\"\"Initialize environment when used as context manager.\"\"\"\n        self.initialize()\n        return self\n        \n    def __exit__(self, exc_type: Any, exc_val: Any, exc_tb: Any) -> None:\n        \"\"\"Cleanup when exiting context manager.\"\"\"\n        self.cleanup()\n\n    def _verify_environment(self) -> bool:\n        \"\"\"Verify that the UV environment is properly set up and functional.\n        \n        Returns:\n            bool: True if environment is ready, False otherwise\n        \"\"\"\n        try:\n            # Check that $VIRTUAL_ENV matches self.venv_path\n            venv_check = self.send(\"echo $VIRTUAL_ENV\", bypass_env_check=True)\n            if not venv_check.stdout.strip() == str(self.venv_path):\n                self._logger.warning(\n                    \"VIRTUAL_ENV does not match the expected path. \"\n                    f\"Expected: {self.venv_path}, Got: {venv_check.stdout}\"\n                )\n                return False\n\n            # Verify Python is accessible and correct version substring\n            py_check = self.send(\"python --version\", bypass_env_check=True)\n            if not py_check.success:\n                self._logger.warning(\"Running 'python --version' failed.\")\n                return False\n            # Check if the declared python_version (e.g. '3.10') is in the output\n            if self.python_version not in py_check.stdout:\n                self._logger.warning(\n                    f\"Python version mismatch. Expected string '{self.python_version}' \"\n                    f\"in '{py_check.stdout.strip()}'\"\n                )\n                return False\n\n            # Try importing a basic module and check sys.prefix\n            import_check = self.send('python -c \"import sys; print(sys.prefix)\"', bypass_env_check=True)\n            if not import_check.success:\n                self._logger.warning(\"Failed to import and print sys.prefix.\")\n                return False\n            if str(self.venv_path) not in import_check.stdout:\n                self._logger.warning(\n                    f\"sys.prefix does not match the expected path: {import_check.stdout}\"\n                )\n                return False\n\n            return True\n\n        except Exception as e:\n            self._logger.warning(f\"Environment verification failed: {e}\")\n            return False\n\n    def _initialize_shell(self) -> None:\n        \"\"\"Initialize a persistent shell session with UV environment setup.\n        \n        Raises:\n            RuntimeError: If shell initialization fails\n        \"\"\"\n        if self._shell is not None:\n            return\n\n        self._shell = subprocess.Popen(\n            ['bash'],\n            stdin=subprocess.PIPE,\n            stdout=subprocess.PIPE,\n            stderr=subprocess.PIPE,\n            text=True,\n            env={**os.environ, **self._env_vars}\n        )\n        \n        # Setup environment variables\n        for name, value in self._env_vars.items():\n            self._run_in_shell(f'export {name}=\"{value}\"')\n        \n        # Activate using UV\n        stdout, stderr = self._run_in_shell(f'eval \"$(uv venv {self.venv_path})\"')\n        if stderr:\n            self._logger.warning(f\"UV environment activation warning: {stderr}\")\n\n    def _run_in_shell(self, command: str) -> tuple[str, str]:\n        \"\"\"Execute a command in the persistent shell session.\n        \n        Args:\n            command (str): Shell command to execute\n            \n        Returns:\n            Tuple of (stdout, stderr) from command execution\n        \"\"\"\n        if not self._shell:\n            self._initialize_shell()\n\n        terminator = f\"__CMD_DONE_{id(command)}__\"\n        full_command = f\"{command}; echo {terminator}; echo {terminator} >&2\"\n        \n        self._shell.stdin.write(full_command + '\\n')\n        self._shell.stdin.flush()\n\n        def read_until_terminator(pipe) -> str:\n            output = []\n            while True:\n                line = pipe.readline()\n                if not line or terminator in line:\n                    break\n                output.append(line)\n            return ''.join(output).rstrip()\n\n        stdout = read_until_terminator(self._shell.stdout)\n        stderr = read_until_terminator(self._shell.stderr)\n        \n        return stdout, stderr\n\n    def initialize(self) -> None:\n        \"\"\"Create and initialize the UV virtual environment.\n        \n        This method:\n            (1) Creates the virtual environment directory\n            (2) Sets up the UV environment\n            (3) Initializes the shell session\n            (4) Verifies the environment is working\n        \n        Raises:\n            RuntimeError: If environment creation or verification fails\n        \"\"\"\n        self._logger.info(f\"Creating UV environment at {self.venv_path}\")\n        \n        try:\n            # Create the virtual environment directory\n            self.venv_path.mkdir(parents=True, exist_ok=True)\n            \n            # Create the venv using uv\n            result = subprocess.run(\n                [\"uv\", \"venv\", \"--python\", self.python_version, str(self.venv_path)],\n                capture_output=True,\n                text=True,\n                check=True\n            )\n            \n            # Initialize shell and verify\n            self._initialize_shell()\n            if not self._verify_environment():\n                raise RuntimeError(\"Environment verification failed.\")\n            self.env_ready = True\n                \n            if not self.env_ready:\n                raise RuntimeError(\"Environment verification failed after waiting\")\n                \n        except subprocess.CalledProcessError as e:\n            raise RuntimeError(f\"Failed to create UV environment: {e.stderr}\")\n        except Exception as e:\n            raise RuntimeError(f\"Error setting up UV environment: {str(e)}\")\n\n    def send(self, command: str, cwd: Path | str | None = None, bypass_env_check: bool = False) -> CommandResult:\n        \"\"\"Sends any arbitrary command to be executed in the UV environment.\n        \n        Args:\n            command (str): The command to execute\n            cwd (Path | str | None): Working directory for the command\n            bypass_env_check (bool, optional): Whether to bypass the env setup check\n                Only really used to allow initial setup commands to pass into the env.\n            \n        Returns:\n            CommandResult containing command output and status\n            \n        Example:\n            >>> uv = UVManager(\"./my_venv\")\n            >>> uv.initialize()\n            >>> result = uv.send(\"python --version\")\n            >>> assert result.success\n            >>> print(result.stdout)\n            Python 3.10.x\n        \"\"\"\n        # (1) We cannot run commands until the env is setup\n        if not self.env_ready and not bypass_env_check:\n            raise RuntimeError(\"Environment not ready. Call initialize() first.\")\n\n        # (2) Add a change of directory to the command if cwd is passed\n        if cwd:\n            command = f\"cd {cwd} && {command}\"\n\n        # Run the command and get the outputs, errors and return code\n        stdout, stderr = self._run_in_shell(command)\n        retcode_out, _ = self._run_in_shell(\"echo $?\")\n        \n        try:\n            returncode = int(retcode_out.strip())\n        except ValueError:\n            returncode = -1\n            \n        return CommandResult(        \n            command=command,\n            returncode=returncode,\n            stdout=stdout,\n            stderr=stderr\n        )\n\n    def run(self, script: str | Path, args: list[str] | None = None) -> CommandResult:\n        \"\"\"Alias to uv_run\n        \n        Args:\n            script (str | Path): Path to the Python script to run\n            args (list[str] | None): List of arguments to pass to the script\n            \n        Returns:\n            CommandResult containing script output and status\n\n        Example:\n            >>> uv = UVManager(\"./my_venv\")\n            >>> uv.initialize()\n            >>> result = uv.run(\"script.py\", [\"--arg1\", \"value1\"])\n        \"\"\"\n        return self.uv_run(script, args)  # Forwarding arguments properly\n    \n    def uv_run(self, script: str | Path, args: list[str] | None = None) -> CommandResult:\n        \"\"\"Run a Python script using UV's run command.\n        \n        Args:\n            script (str | Path): Path to the Python script to run\n            args (list[str] | None): List of arguments to pass to the script\n            \n        Returns:\n            CommandResult containing script output and status\n            \n        Example:\n            >>> uv = UVManager(\"./my_venv\")\n            >>> uv.initialize()\n            >>> result = uv.uv_run(\"script.py\", [\"--arg1\", \"value1\"])\n        \"\"\"\n        cmd = [\"uv\", \"run\"]\n        if args:\n            cmd.extend(args)\n        cmd.append(str(script))\n        \n        return self.send(\" \".join(cmd))\n\n    def pip_install(\n        self, \n        package: str, \n        editable: bool = False, \n        cwd: Path | str | None = None,\n        verbose: bool = True\n    ) -> CommandResult:\n        \"\"\"Install a package using UV's pip interface.\n        \n        Args:\n            package (str): Package specification (name, path, or requirements file)\n            editable (bool): If True, install in editable mode (-e flag)\n            verbose (bool): If True, print installation progress\n            \n        Returns:\n            CommandResult containing installation output and status\n            \n        Example:\n            >>> uv = UVManager(\"./my_venv\")\n            >>> uv.initialize()\n            >>> result = uv.pip_install(\"requests\")\n            >>> assert result.success\n        \"\"\"\n        if not self.env_ready:\n            raise RuntimeError(\"Environment not ready. Call initialize() first.\")\n\n        # If package is a string, split it safely, or handle it as a list\n        if isinstance(package, str):\n            package_list = shlex.split(package)\n        else:\n            package_list = [package,]\n    \n        cmd = [\"uv\", \"pip\", \"install\"]\n        if editable:\n            cmd.append(\"-e\")\n        cmd.extend(package_list)\n        \n        result = self.send(\" \".join(cmd), cwd=cwd)\n        \n        if verbose:\n            if result.stdout:\n                self._logger.info(result.stdout)\n            if result.stderr:\n                if \"Installed\" in result.stderr or \"Resolved\" in result.stderr or \"Using Python\" in result.stderr:\n                    self._logger.info(result.stderr)\n                else:\n                    self._logger.error(result.stderr)\n                \n        return result\n\n    def cleanup(self) -> None:\n        \"\"\"Clean up resources and terminate the shell session.\n        \n        This should be called when done using the environment if not using\n        the context manager.\n        \"\"\"\n        if self._shell:\n            self._shell.terminate()\n            self._shell = None\n        self.env_ready = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:40:49.973286Z","iopub.execute_input":"2025-01-26T19:40:49.973648Z","iopub.status.idle":"2025-01-26T19:40:50.001041Z","shell.execute_reply.started":"2025-01-26T19:40:49.973619Z","shell.execute_reply":"2025-01-26T19:40:49.9998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import logging\nfrom rich.logging import RichHandler\n\n# Remove existing handlers (important in Jupyter environments)\nfor handler in logging.root.handlers[:]:\n    logging.root.removeHandler(handler)\n\n# Setup logging with RichHandler\nlogging.basicConfig(\n    level=logging.DEBUG,  # Ensure debug messages are shown\n    format=\"%(message)s\",\n    datefmt=\"[%X]\",\n    handlers=[RichHandler()]\n)\n\nlogger = logging.getLogger(__name__)\n\n# Test logs\nlogger.info(\"This is an info message.\")\nlogger.warning(\"This is a warning.\")\nlogger.error(\"This is an error message.\")\nlogger.debug(\"This is a debug message (should show if level is DEBUG).\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:40:50.002274Z","iopub.execute_input":"2025-01-26T19:40:50.002681Z","iopub.status.idle":"2025-01-26T19:40:50.055212Z","shell.execute_reply.started":"2025-01-26T19:40:50.00264Z","shell.execute_reply":"2025-01-26T19:40:50.053911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_ENV_NAME = \"testbed_venv\"\n\nif os.path.isdir(_ENV_NAME):\n    shutil.rmtree(_ENV_NAME)\n\n# Context manager\nwith UVManager(f\"./{_ENV_NAME}\") as uv:\n    uv.pip_install(\"requests\")\n    print(uv.send(\"uv pip list\").stdout)\n    print(uv.send(\"pwd\").stdout)\n\n# print(\"\\n\\n\")\n\n# # Direct usage\n# uv = UVManager(f\"./{_ENV_NAME}\")\n# uv.initialize()\n# uv.pip_install(\"requests\")\n# print(uv.send(\"uv pip list\").stdout)\n# print(uv.send(\"pwd\").stdout)\n# uv.cleanup()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:40:50.056489Z","iopub.execute_input":"2025-01-26T19:40:50.056828Z","iopub.status.idle":"2025-01-26T19:40:51.274956Z","shell.execute_reply.started":"2025-01-26T19:40:50.056783Z","shell.execute_reply":"2025-01-26T19:40:51.273588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# First create a basic requirements.txt\nrequirements = \"\"\"\nrequests==2.31.0\npandas>=2.0.0\nnumpy>=1.24.0\nmatplotlib>=3.7.0\npython-dotenv==1.0.0\npyyaml>=6.0.0\n\"\"\"\n\n# Write the requirements file\nwith open(\"requirements.txt\", \"w\") as f:\n    f.write(requirements.strip())\n\n# Now use UV to install from the requirements file\nwith UVManager(f\"./{_ENV_NAME}\") as uv:\n    # First verify Python is working\n    py_ver = uv.send(\"python --version\")\n    print(f\"Using Python: {py_ver.stdout}\")\n    \n    # Install from requirements file\n    print(\"\\nInstalling from requirements.txt...\")\n    result = uv.pip_install(\"-r requirements.txt\")\n    \n    if result.returncode != 0:\n        print(f\"Installation failed: {result.stderr}\")\n    else:\n        # Verify installations\n        print(\"\\nVerifying installations...\")\n        verify = uv.send(\"uv pip list\")\n        print(\"Installed packages:\")\n        print(verify.stdout)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:40:51.275951Z","iopub.execute_input":"2025-01-26T19:40:51.276348Z","iopub.status.idle":"2025-01-26T19:40:53.352463Z","shell.execute_reply.started":"2025-01-26T19:40:51.276284Z","shell.execute_reply":"2025-01-26T19:40:53.351508Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<br>\n\n## Explore how to create minimal UV environments for testing\n\n---\n\n1. Clone the repo at instance.environment_setup_commit\n2. Apply patches (if any)\n3. Create a UV environment\n4. Install dependencies\n5. Run pytest\n6. Clean up automatically","metadata":{}},{"cell_type":"markdown","source":"### Repo Manager","metadata":{}},{"cell_type":"code","source":"class GitHubRepo:\n    \"\"\"Handles operations with a GitHub repository (cloning, checkout, patching).\n\n    Attributes:\n        instance_id (str): A unique string identifier for the instance (aka GitHub issue).\n        repo_name (str): The name of Github repository\n        org_name (str): The name of owner of the Github repository\n        repo_url (str): The full GitHub repository URL (e.g., \"https://github.com/user/repo.git\")\n        env_setup_commit_hash (str): The commit hash used for environment setup.\n        base_commit_hash (str | None): The commit hash for the relevant issue.\n        root_dir (Path): Path to the root directory where all temporary directories will be.\n        root_repo_path (Path): Path to this repo's specific temporary directory\n        _logger (logging.Logger): Internal logger instance\n    \"\"\"\n    instance_id: str\n    repo_name: str\n    org_name: str\n    repo_url: str\n    env_setup_commit_hash: str\n    base_commit_hash: str\n    root_dir: Path\n    root_repo_path: Path\n    _logger: logging.Logger\n\n    def __init__(\n            self,\n            instance_id: str,\n            repo_name: str,\n            org_name: str,\n            repo_url: str,\n            env_setup_commit_hash: str,\n            base_commit_hash: str | None = None,\n            root_dir: str | Path = \"/kaggle/tmp\"\n    ) -> None:\n        \"\"\"Initializes the GitHubRepo object.\n\n        Args:\n            instance_id (str):\n                 A unique string identifier for the instance (aka GitHub issue).\n            repo_name (str):\n                The name of Github repo (repo)\n            org_name (str):\n                The name of owner of the Github repo (repo)\n            repo_url (str):\n                The full GitHub repository URL (e.g., \"https://github.com/user/repo.git\")\n            env_setup_commit_hash (str):\n                The commit hash used for environment setup.\n            base_commit_hash (str, optional):\n                The commit hash for the relevant issue.\n            root_dir (str | Path, optional):\n                The root directory where all temporary directories will be stored.\n        \"\"\"\n        # (1) Set attributes related to commit hashes\n        self.env_setup_commit_hash = env_setup_commit_hash\n        self.base_commit_hash = base_commit_hash or env_setup_commit_hash\n\n        # (2) Set attributes related to the repository name and URL\n        self.instance_id = instance_id\n        self.repo_name = repo_name\n        self.org_name = org_name\n        self.repo_url = repo_url\n\n        # (3) Set attributes related to the root directory and paths\n        self.root_dir = Path(root_dir)\n        self.root_repo_path = self.find_free_directory()\n\n        # (4) Initialize the logger\n        self._logger = logging.getLogger(self.__class__.__name__)\n\n    def find_free_directory(self, base_dir: str | Path | None = None) -> Path:\n        \"\"\"Finds a free directory in the base directory, creating one if necessary.\n        \n        Args:\n            base_dir (str | Path, optional): \n                Base directory to search within.\n        \n        Returns:\n            Path:\n                The path object pointing to the newly created directory.\n        \n        Raises:\n            RuntimeError: If no free directory could be found after multiple attempts.\n        \"\"\"\n        base_path = Path(base_dir or self.root_dir)\n        for i in range(100):\n            dir_name = f\"temp_env_{i}\"\n            temp_dir = base_path / dir_name\n            if not temp_dir.exists():\n                temp_dir.mkdir(parents=True, exist_ok=True)\n                return temp_dir\n        raise RuntimeError(\"Could not find a free temporary directory\")\n    \n    def clone_and_checkout(self, checkout_commit_hash: str | None = None) -> Path:\n        \"\"\"Clones the repository and checks out the specified commit.\n\n        Args:\n            checkout_commit_hash (str, optional):\n                The commit hash that we will checkout.\n                If not provided we will use the commit hash for the environment setup.\n\n        Returns:\n            The local filesystem path to the cloned repo.\n\n        Raises:\n            RuntimeError: If clone or checkout fails.\n        \"\"\"\n        try:\n            # (1) Clone the repository\n            self.clone_repo()\n\n            # (2) Checkout the specified commit\n            self.checkout_commit(checkout_commit_hash or self.env_setup_commit_hash)\n\n        # (3) Handle any errors\n        except subprocess.CalledProcessError as e:\n            self._logger.error(f\"Git clone/checkout error: {e.stderr}\")\n            raise RuntimeError(f\"Git clone/checkout error: {e.stderr}\") from e\n\n        # (4) Return the path to the cloned repository\n        return self.root_repo_path\n\n    def clone_repo(self, force_reclone: bool = True) -> \"GitHubRepo\":\n        \"\"\"Clones the repository into the specified path.\n\n        Args:\n            force_reclone (bool, optional):\n                Whether to force re-cloning the repository if it already exists.\n\n        Returns:\n            GitHubRepo: The instance itself (allows method chaining).\n        \"\"\"\n        # (1) If repo (validated by .git) already exists, do nothing\n        if (self.root_repo_path / \".git\").exists():\n            # (1a) If force_reclone is False, log a warning and return\n            if not force_reclone:\n                self._logger.warning(f\"Repository path {self.root_repo_path} already exists. Skipping clone.\")\n                return self\n            # (1b) If force_reclone is True, remove the existing directory and proceed.\n            self._logger.warning(f\"Repository path {self.root_repo_path} already exists. Recloning...\")\n            shutil.rmtree(self.root_repo_path)\n            self.root_repo_path.mkdir(parents=True, exist_ok=True)\n\n        # (2) Clone a single branch of the repository\n        self._logger.info(f\"Cloning {self.repo_url} to {self.root_repo_path}...\")\n        try:\n            # Clone with --no-single-branch to get all branches\n            subprocess.run(\n                [\"git\", \"clone\", \"--no-single-branch\", self.repo_url, str(self.root_repo_path)],\n                capture_output=True, text=True, check=True\n            )\n            \n            # Fetch all tags and remote branches\n            subprocess.run(\n                [\"git\", \"fetch\", \"--all\", \"--tags\", \"--prune\"],\n                cwd=self.root_repo_path,\n                capture_output=True, text=True, check=True\n            )\n\n        # (3) Handle any errors\n        except subprocess.CalledProcessError as e:\n            self._logger.error(f\"Git clone error: {e.stderr}\")\n            raise RuntimeError(f\"Git clone error: {e.stderr}\") from e\n\n        # (4) Enable method chaining\n        return self\n\n    def checkout_commit(self, commit_hash: str) -> \"GitHubRepo\":\n        \"\"\"Checks out the specified commit hash in the cloned repository.\n\n        Args:\n            commit_hash (str): The commit hash to checkout.\n\n        Returns:\n            GitHubRepo: The instance itself (allows method chaining).\n        \"\"\"\n        # (1) Check if the repository path exists\n        if not (self.root_repo_path / \".git\").exists():\n            raise RuntimeError(f\"Repository path {self.root_repo_path} does not exist. Clone the repo first.\")\n\n        # (2) Checkout the specified commit\n        try:\n            subprocess.run(\n                [\"git\", \"checkout\", commit_hash],\n                cwd=self.root_repo_path, capture_output=True, text=True, check=True\n            )\n            self._logger.info(f\"Checked out commit {commit_hash}.\")\n\n        # (3) Handle any errors\n        except subprocess.CalledProcessError as e:\n            self._logger.error(f\"Git checkout error: {e.stderr}\")\n            raise RuntimeError(f\"Git checkout error: {e.stderr}\") from e\n\n        # (4) Enable method chaining\n        return self\n\n    def apply_patch(self, patch_content: str) -> \"GitHubRepo\":\n        \"\"\"Applies a patch to the cloned repository.\n\n        Args:\n            patch_content (str):\n                The diff/patch content as a string.\n\n        Raises:\n            RuntimeError: If patch application fails.\n        \"\"\"\n        # (1) Write the patch content to a file in the temporary directory\n        patch_file = self.root_repo_path / \"local_changes.patch\"\n        patch_file.write_text(patch_content)\n\n        # (2) Apply the patch\n        self._logger.info(f\"Applying patch at {patch_file}...\")\n        try:\n            subprocess.run(\n                [\"git\", \"apply\", str(patch_file)],\n                cwd=self.root_repo_path, capture_output=True, text=True, check=True\n            )\n            self._logger.info(\"Patch applied successfully.\")\n        except subprocess.CalledProcessError as e:\n            self._logger.error(f\"Patch application error: {e.stderr}\")\n            raise RuntimeError(f\"Patch application error: {e.stderr}\") from e\n\n        # Enable method chaining\n        return self\n        \n    @classmethod\n    def from_swebench_instance(\n            cls,\n            instance: SWEBenchInstance,\n            root_dir: str | Path = \"/kaggle/tmp\"\n    ) -> \"GitHubRepo\":\n        \"\"\"Creates a GitHubRepo instance from a SWEBenchInstance object.\n\n        Args:\n            instance (SWEBenchInstance):\n                The SWE-Bench instance to create the GitHubRepo object from.\n\n        Returns:\n            GitHubRepo:\n                A GitHubRepo object initialized with the instance's repo URL and commit hash.\n        \"\"\"\n        org_name, repo_name = instance.repo.split(\"/\")\n        return GitHubRepo(\n            instance_id=instance.instance_id,\n            repo_name=repo_name.strip(),\n            org_name=org_name.strip(),\n            repo_url=instance.github_repo_url,\n            base_commit_hash=instance.base_commit,\n            env_setup_commit_hash=instance.environment_setup_commit,\n            root_dir=root_dir\n        )\n\n    def cleanup(self) -> None:\n        \"\"\"Cleans up the cloned repository directory.\"\"\"\n        self._logger.info(f\"Cleaning up temporary repository directory {self.root_repo_path}...\")\n        shutil.rmtree(self.root_repo_path, ignore_errors=True)\n        self._logger.info(\"Temporary repository directory cleaned up.\")\n\n    def __repr__(self) -> str:\n        \"\"\"String representation of the GitHubRepo object.\"\"\"\n        return (\n            f\"GitHubRepo(\"\n            f\"rood_dir={self.root_dir}, \"\n            f\"root_repo_path={self.root_repo_path}, \"\n            f\"repo_name={self.repo_name}, \"\n            f\"org_name={self.org_name}, \"\n            f\"repo_url={self.repo_url}, \"\n            f\"base_commit_hash={self.base_commit_hash}\"\n            f\"env_setup_commit_hash={self.env_setup_commit_hash})\"\n        )\n\nrich.print(\"[bold green]Setup our Github Repo object (clones to a new temporary folder) [/bold green]\")\ndemo_github_repo = GitHubRepo.from_swebench_instance(demo_instance, root_dir=\"/kaggle/working\")\ndemo_repo_path = demo_github_repo.clone_and_checkout()\n# demo_github_repo.cleanup()\n\n# rich.print(\"[bold red]Show that we can apply the patches and they don't error out.[/bold red]\")\n# demo_github_repo.apply_patch(demo_instance.patch)\n# demo_github_repo.apply_patch(demo_instance.test_patch)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:40:53.354834Z","iopub.execute_input":"2025-01-26T19:40:53.355169Z","iopub.status.idle":"2025-01-26T19:41:16.595787Z","shell.execute_reply.started":"2025-01-26T19:40:53.355138Z","shell.execute_reply":"2025-01-26T19:41:16.594657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class PythonVersionDetector:\n    \"\"\"Detects appropriate Python versions for a project using modern parsing approaches.\n\n    Attributes:\n        python_versions (list[PythonVersion]):\n            Available Python versions from the Universal Python Installer\n        _logger (logging.Logger): Internal logger instance\n    \"\"\"\n\n    def __init__(self) -> None:\n        self._logger = logging.getLogger(self.__class__.__name__)\n        self.python_versions = self._get_available_python_versions()\n\n    def _parse_uv_python_list(self, output: str) -> list[PythonVersion]:\n        \"\"\"Parse the output of 'uv python list' command.\n\n        Args:\n            output (str): The output string from 'uv python list'.\n\n        Returns:\n            list[PythonVersion]:\n                A list of PythonVersion objects.\n        \"\"\"\n        # (1) Initialize the list of versions\n        versions = []\n\n        # (2) Parse the output line by line\n        for line in output.splitlines():\n\n            # (3a) Skip empty lines\n            if not line.strip():\n                continue\n\n            # (3b) Split the line into parts\n            parts = line.split(maxsplit=1)\n\n            # (3c) Skip if no parts\n            if not parts:\n                continue\n\n            # (3d) Extract version info and availability\n            version_info = parts[0]\n            availability = parts[1] if len(parts) > 1 else \"<download available>\"\n\n            # (3e) Match the version info\n            match = re.match(\n                r'(cpython|pypy)-(\\d+\\.\\d+\\.\\d+(?:a\\d+)?)'\n                r'(\\+freethreaded)?-linux-x86_64-gnu',\n                version_info\n            )\n\n            # (3f) Skip if no match\n            if not match:\n                continue\n\n            # (3g) Extract the implementation, version, and freethreaded info\n            impl, version, freethreaded = match.groups()\n            implementation = (\n                PythonImplementation.CPYTHON if impl == \"cpython\"\n                else PythonImplementation.PYPY\n            )\n\n            # (3h) Check if the version is installed\n            is_installed = not availability.startswith(\"<download\")\n            path = availability if is_installed else None\n\n            # (3i) Append the version to the list if it's installed and available\n            if path and \" -> \" in path:\n                path = path.split(\" -> \")[0].strip()\n\n            # (3j) Append the version to the list\n            versions.append(PythonVersion(\n                implementation=implementation,\n                version=version,\n                is_installed=is_installed,\n                path=path,\n                is_freethreaded=bool(freethreaded)\n            ))\n\n        # (4) Return the list of versions\n        return versions\n\n    def _get_available_python_versions(self) -> list[PythonVersion]:\n        \"\"\"Get list of available Python versions from UV.\n\n        Returns:\n            list[PythonVersion]:\n                A list of PythonVersion objects.\n        \"\"\"\n        # (1) Run the 'uv python list' command\n        try:\n            result = subprocess.run(\n                [\"uv\", \"python\", \"list\"],\n                capture_output=True,\n                text=True,\n                check=True\n            )\n            # (2) Parse the output\n            return self._parse_uv_python_list(result.stdout)\n        # (3) Handle any errors\n        except subprocess.CalledProcessError as e:\n            self._logger.warning(f\"Failed to list Python versions: {e.stderr}\")\n            return []\n\n    def parse_pyproject_toml(self, path: Path) -> VersionConstraints | None:\n        \"\"\"Parse Python version constraints from pyproject.toml using proper TOML parser.\n\n        Args:\n            path (Path): The path to the pyproject.toml file.\n\n        Returns:\n            VersionConstraints | None:\n                The parsed version constraints or None if not found.\n        \"\"\"\n        # (1) Attempt to parse the pyproject.toml file\n        try:\n            # (2a) Load the TOML data\n            data = tomli.loads(path.read_text())\n\n            # (2b) Check project.requires-python (PEP 621)\n            if \"project\" in data and \"requires-python\" in data[\"project\"]:\n                return VersionConstraints.from_specifier_string(\n                    data[\"project\"][\"requires-python\"],\n                    \"pyproject.toml\",\n                    confidence=1.0\n                )\n\n            # (2c) Check poetry dependencies (looks for \"tool.poetry.dependencies.python\")\n            if \"tool\" in data and \"poetry\" in data:\n                if \"dependencies\" in data[\"tool\"][\"poetry\"]:\n                    if \"python\" in data[\"tool\"][\"poetry\"][\"dependencies\"]:\n                        return VersionConstraints.from_specifier_string(\n                            data[\"tool\"][\"poetry\"][\"dependencies\"][\"python\"],\n                            \"pyproject.toml (poetry)\",\n                            confidence=0.9\n                        )\n\n            # (2d) Check PDM dependencies (looks for \"tool.pdm.python\")\n            if \"tool\" in data and \"pdm\" in data:\n                if \"python\" in data[\"tool\"][\"pdm\"]:\n                    return VersionConstraints.from_specifier_string(\n                        data[\"tool\"][\"pdm\"][\"python\"],\n                        \"pyproject.toml (pdm)\",\n                        confidence=0.9\n                    )\n        # (3) Handle any errors\n        except Exception as e:\n            self._logger.warning(f\"Error parsing pyproject.toml: {e}\")\n\n        # (4) Return None if no constraints found\n        return None\n\n    def parse_setup_py(self, path: Path) -> VersionConstraints | None:\n        \"\"\"Parse Python version constraints from setup.py file.\n\n        Uses multiple strategies to detect version constraints:\n            1. Analyzes Python version classifiers\n            2. Checks for explicit python_requires in setup arguments\n            3. Infers version range from classifier information\n\n        Args:\n            path (Path): Path to the setup.py file to analyze\n\n        Returns:\n            VersionConstraints | None: Parsed version constraints if found, None otherwise\n        \"\"\"\n        # (1) Read and parse the setup.py file\n        try:\n            content = path.read_text()\n            \n            # (2a) Look for Python version classifiers with single quotes\n            classifiers = re.findall(\n                r\"'Programming Language :: Python :: (\\d+\\.\\d+)'\",\n                content\n            )\n            \n            # (2b) If none found, try double quotes\n            if not classifiers:\n                classifiers = re.findall(\n                    r'\"Programming Language :: Python :: (\\d+\\.\\d+)\"',\n                    content\n                )\n            \n            # (3) Process classifier versions if found\n            if classifiers:\n                # (3a) Filter to only Python 3 versions\n                versions = [v for v in classifiers if not v.startswith('2')]\n                \n                if versions:\n                    # (3b) Get min and max supported versions\n                    min_ver = min(versions, key=lambda x: parse(x))\n                    max_ver = max(versions, key=lambda x: parse(x))\n                    \n                    # (3c) Create constraint up to next major version\n                    major, minor = map(int, max_ver.split('.'))\n                    next_minor = f\"{major}.{minor + 1}\"\n                    spec = f\">={min_ver},<{next_minor}\"\n                    \n                    return VersionConstraints.from_specifier_string(\n                        spec,\n                        \"setup.py (classifiers)\",\n                        confidence=0.7\n                    )\n            \n            # (4) Check for explicit python_requires\n            requires_match = re.search(\n                r'python_requires\\s*=\\s*[\\'\"]([^\\'\"]+)[\\'\"]',\n                content\n            )\n            if requires_match:\n                return VersionConstraints.from_specifier_string(\n                    requires_match.group(1),\n                    \"setup.py (python_requires)\",\n                    confidence=0.9\n                )\n                \n        # (5) Handle any parsing errors\n        except Exception as e:\n            self._logger.warning(f\"Error parsing setup.py: {e}\")\n            \n        return None\n    \n    def parse_setup_cfg(self, path: Path) -> VersionConstraints | None:\n        \"\"\"Parse Python version constraints from setup.cfg.\n\n        Args:\n            path (Path): The path to the setup.cfg file.\n\n        Returns:\n            VersionConstraints | None:\n                The parsed version constraints or None if not found.\n        \"\"\"\n        # (1) Attempt to parse the setup.cfg file\n        try:\n            # (2a) Load the setup.cfg data\n            from configparser import ConfigParser\n            config = ConfigParser()\n            config.read(path)\n\n            # (2b) Check for \"options.python_requires\" in the \"metadata\" section (PEP 518)\n            if \"options\" in config:\n                if \"python_requires\" in config[\"options\"]:\n                    return VersionConstraints.from_specifier_string(\n                        config[\"options\"][\"python_requires\"],\n                        \"setup.cfg\",\n                        confidence=0.8\n                    )\n        # (3) Handle any errors\n        except Exception as e:\n            self._logger.warning(f\"Error parsing setup.cfg: {e}\")\n\n        # (4) Return None if no constraints found\n        return None\n\n    def parse_requirements_txt(self, path: Path) -> VersionConstraints | None:\n        \"\"\"Parse Python version constraints from requirements.txt using packaging.requirements.\n\n        Args:\n            path (Path): The path to the requirements.txt file.\n\n        Returns:\n            VersionConstraints | None:\n                The parsed version constraints or None if not found.\n        \"\"\"\n        # (1) Attempt to parse the requirements.txt file\n        try:\n            # (2a) Read the requirements.txt file\n            content = path.read_text()\n            constraints = []\n\n            # (2b) Parse each line in the file (ignoring comments and empty lines)\n            for line in content.splitlines():\n                line = line.strip()\n                if not line or line.startswith('#'):\n                    continue\n\n                # (2c) Attempt to parse the requirement (ignoring any errors and looking for python_version)\n                try:\n                    req = Requirement(line)\n                    if \"python_version\" in str(req.marker):\n                        # Extract the version constraint from the marker\n                        marker_str = str(req.marker)\n                        version_part = re.search(r'python_version([^\"\\']+)[\"\\']([^\"\\']+)[\"\\']', marker_str)\n                        if version_part:\n                            op, ver = version_part.groups()\n                            constraints.append(f\"{op}{ver}\")\n                except Exception:\n                    continue\n\n            # (2d) Return the constraints if found\n            if constraints:\n                return VersionConstraints.from_specifier_string(\n                    \",\".join(constraints),\n                    \"requirements.txt\",\n                    confidence=0.7\n                )\n\n        # (3) Handle any errors\n        except Exception as e:\n            self._logger.warning(f\"Error parsing requirements.txt: {e}\")\n\n        # (4) Return None if no constraints found\n        return None\n\n    def get_project_constraints(self, repo_path: Path | str) -> VersionConstraints | None:\n        \"\"\"Get Python version constraints from all project configuration files.\n\n        Checks multiple config files in order of reliability:\n            1. pyproject.toml (PEP 621)\n            2. setup.cfg \n            3. setup.py\n            4. requirements.txt\n\n        Args:\n            repo_path (Path | str): Path to the repository root\n\n        Returns:\n            VersionConstraints | None: Version constraints if found, None otherwise\n        \"\"\"\n        # (1) Ensure path is a Path object\n        repo_path = Path(repo_path)\n\n        # (2) Define config files to check in priority order\n        config_files = [\n            (repo_path / \"pyproject.toml\", self.parse_pyproject_toml),\n            (repo_path / \"setup.cfg\", self.parse_setup_cfg),\n            (repo_path / \"setup.py\", self.parse_setup_py),\n            (repo_path / \"requirements.txt\", self.parse_requirements_txt)\n        ]\n\n        # (3) Try each config file in order\n        for file_path, parser in config_files:\n            if file_path.exists():\n                if constraints := parser(file_path):\n                    self._logger.info(\n                        f\"Found constraints in {file_path.name}: \"\n                        f\"{constraints.specifier_set} \"\n                        f\"(confidence: {constraints.confidence})\"\n                    )\n                    return constraints\n\n        # (4) No constraints found in any file\n        return None\n\n    def select_python_version(\n        self,\n        repo_path: Path | str,\n        python_fallback_version: str = \"3.10\",\n        newest_allowed_is_fallback: bool = True,\n    ) -> PythonVersion:\n        \"\"\"Select the most appropriate Python version for a project.\n\n        Uses the following selection process:\n            1. Gets project version constraints from config files\n            2. Filters available CPython versions to stable releases\n            3. Filters versions to those matching constraints\n            4. Selects newest compatible version\n            5. Falls back to specified version if no match\n            6. If specified version is newer than fallback optionally set to fallback\n\n        Args:\n            repo_path (Path | str): Path to the repository root\n            python_fallback_version (str): Version to use if no compatible version found\n\n        Returns:\n            PythonVersion: Selected Python version object\n        \"\"\"\n        # (1) Get project constraints\n        constraints = self.get_project_constraints(repo_path)\n        \n        # (2) Filter to stable CPython versions\n        cpython_versions = [\n            v for v in self.python_versions\n            if v.implementation == PythonImplementation.CPYTHON\n            and not re.search(r'a|b|rc', v.version)\n        ]\n        \n        # (3) Apply version constraints if found\n        if constraints:\n            try:\n                # (3a) Filter to versions matching constraints\n                compatible_versions = []\n                for version in cpython_versions:\n                    try:\n                        if version.parsed_version in constraints.specifier_set:\n                            compatible_versions.append(version)\n                            self._logger.debug(\n                                f\"Version {version.version} matches constraints\"\n                            )\n                        else:\n                            self._logger.debug(\n                                f\"Version {version.version} excluded by constraints\"\n                            )\n                    except Exception as e:\n                        self._logger.warning(\n                            f\"Error checking version {version.version}: {e}\"\n                        )\n            except Exception as e:\n                self._logger.warning(f\"Error applying constraints: {e}\")\n                compatible_versions = cpython_versions\n        else:\n            compatible_versions = cpython_versions\n            \n        # (4) Sort compatible versions newest first\n        sorted_versions = sorted(\n            compatible_versions,\n            key=lambda v: v.parsed_version,\n            reverse=True\n        )\n        \n        # (5) Return newest compatible version or fallback\n        if not sorted_versions:\n            self._logger.warning(\n                f\"No compatible versions found, using fallback {python_fallback_version}\"\n            )\n            return PythonVersion(\n                implementation=PythonImplementation.CPYTHON,\n                version=f\"{python_fallback_version}.0\",\n                is_installed=False,\n                path=None\n            )\n\n        # (6) Return the selected newest version that meets requirements\n        selected = sorted_versions[0]\n        if selected.base_version>python_fallback_version and newest_allowed_is_fallback:\n            self._logger.info(\n                f\"Selected version {python_fallback_version}.0 \"\n                f\"because actual selected version is newer.\"\n            )\n            return PythonVersion(\n                implementation=PythonImplementation.CPYTHON,\n                version=f\"{python_fallback_version}.0\",\n                is_installed=False,\n                path=None\n            )\n        self._logger.info(\n            f\"Selected version {selected.version} \"\n            f\"from {len(sorted_versions)} compatible versions\"\n        )\n        return selected\n\n### ######################## ###\n### Let's see how to use it. ###\n### ######################## ###\n# Initialize the detector\ndetector = PythonVersionDetector()\n\n# Get project constraints\nconstraints = detector.get_project_constraints(demo_repo_path)\n\nif constraints:\n    rich.print(f\"Min version: {constraints.min_version}\")\n    rich.print(f\"Max version: {constraints.max_version}\")\n    rich.print(f\"Source: {constraints.source_file}\")\n    rich.print(f\"Confidence: {constraints.confidence}\")\n\n# Select the best Python version\nselected_version = detector.select_python_version(demo_repo_path)\nrich.print(f\"\\nSelected Python version: {selected_version.version}\")\n### ######################## ###","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:41:16.597241Z","iopub.execute_input":"2025-01-26T19:41:16.59755Z","iopub.status.idle":"2025-01-26T19:41:17.390182Z","shell.execute_reply.started":"2025-01-26T19:41:16.597526Z","shell.execute_reply":"2025-01-26T19:41:17.389173Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### OLD VERSION OF PYTHON DETECTION...","metadata":{}},{"cell_type":"code","source":"# class PythonVersionDetectionMixin:\n#     \"\"\"A mixin that provides methods to:\n      \n#       (1) Parse installed versions from `uv python list`\n#       (2) Extract Python version constraints from project files\n#       (3) Pick the newest installed CPython version that satisfies constraints\n#     \"\"\"\n#     def __init__(self):\n#         self._logger = logging.getLogger(self.__class__.__name__)\n#         self.python_versions = self._get_available_python_versions()\n        \n#     def _parse_uv_python_list(self, output: str) -> list[PythonVersion]:\n#         \"\"\"Parse the output of 'uv python list' command.\n\n#         Args:\n#             output (str): Raw output from 'uv python list'\n\n#         Returns:\n#             list[PythonVersion]: List of parsed Python versions\n#         \"\"\"\n#         versions = []\n#         for line in output.splitlines():\n#             if not line.strip():\n#                 continue\n\n#             # Split into version info and path/availability\n#             parts = line.split(maxsplit=1)\n#             if not parts:\n#                 continue\n\n#             version_info = parts[0]\n#             availability = parts[1] if len(parts) > 1 else \"<download available>\"\n\n#             # Parse the version string\n#             # Example: cpython-3.10.12-linux-x86_64-gnu\n#             match = re.match(\n#                 r'(cpython|pypy)-(\\d+\\.\\d+\\.\\d+(?:a\\d+)?)'\n#                 r'(\\+freethreaded)?-linux-x86_64-gnu',\n#                 version_info\n#             )\n#             if not match:\n#                 continue\n\n#             impl, version, freethreaded = match.groups()\n#             implementation = (\n#                 PythonImplementation.CPYTHON if impl == \"cpython\"\n#                 else PythonImplementation.PYPY\n#             )\n\n#             is_installed = not availability.startswith(\"<download\")\n#             path = availability if is_installed else None\n#             if path and \" -> \" in path:  # Handle symlinks\n#                 path = path.split(\" -> \")[0].strip()\n\n#             versions.append(PythonVersion(\n#                 implementation=implementation,\n#                 version=version,\n#                 is_installed=is_installed,\n#                 path=path,\n#                 is_freethreaded=bool(freethreaded)\n#             ))\n\n#         return versions\n\n#     def _get_available_python_versions(self) -> list[PythonVersion]:\n#         \"\"\"Get list of available Python versions from UV.\n\n#         Returns:\n#             list[PythonVersion]: List of available Python versions\n#         \"\"\"\n#         try:\n#             result = subprocess.run(\n#                 [\"uv\", \"python\", \"list\"],\n#                 capture_output=True,\n#                 text=True,\n#                 check=True\n#             )\n#             return self._parse_uv_python_list(result.stdout)\n#         except subprocess.CalledProcessError as e:\n#             self._logger.warning(f\"Failed to list Python versions: {e.stderr}\")\n#             return []\n\n#     def _get_pyproject_constraints(self, pyproject_path: Path) -> str | None:\n#         \"\"\"Extract Python version constraints from pyproject.toml file.\n\n#         Args:\n#             pyproject_path (Path | str):\n#                 The path to the pyproject.toml file.\n\n#         Returns:\n#             The version constraints for Python (or None if not found)\n#         \"\"\"\n#         try:\n#             content = pyproject_path.read_text()\n\n#             # Common patterns in pyproject.toml\n#             patterns = [\n#                 r'requires-python\\s*=\\s*[\\'\"]([^\\'\"]*)[\\'\"]',  # PEP 621\n#                 r'python\\s*=\\s*[\\'\"]([^\\'\"]*)[\\'\"]',  # poetry\n#                 r'python_version\\s*=\\s*[\\'\"]([^\\'\"]*)[\\'\"]'  # other tools\n#             ]\n\n#             for pattern in patterns:\n#                 if match := re.search(pattern, content):\n#                     return match.group(1)\n\n#             # If no direct match, look for version patterns near 'python'\n#             python_lines = [line for line in content.splitlines()\n#                             if 'python' in line.lower()]\n#             version_pattern = r'(?:^|\\D)(\\d+\\.\\d+(?:\\.\\d+)?)'\n\n#             versions = []\n#             for line in python_lines:\n#                 if matches := re.finditer(version_pattern, line):\n#                     versions.extend(match.group(1) for match in matches)\n\n#             if versions:\n#                 # Convert to constraint format\n#                 base_versions = sorted(\n#                     set('.'.join(v.split('.')[:2]) for v in versions),\n#                     key=lambda v: tuple(map(int, v.split('.')))\n#                 )\n#                 min_ver = base_versions[0]\n#                 max_ver = base_versions[-1]\n#                 major, minor = map(int, max_ver.split('.'))\n#                 return f\">={min_ver},<{major}.{minor + 1}\"\n\n#             return None\n#         except Exception as e:\n#             self._logger.warning(f\"Error parsing pyproject.toml: {e}\")\n#             return None\n\n#     def _get_requirements_constraints(self, requirements_path: Path | str) -> str | None:\n#         \"\"\"Extract Python version constraints from requirements.txt file.\n\n#         Args:\n#             requirements_path (Path | str):\n#                 The path to the requirements.txt file.\n\n#         Returns:\n#             The version constraints for Python (or None if not found)\n#         \"\"\"\n#         try:\n#             content = Path(requirements_path).read_text()\n\n#             # Look for python_version markers\n#             markers = [\n#                 r'python_version\\s*([><=!~]+)\\s*[\\'\"]([^\\'\"]+)[\\'\"]',\n#                 r'python_version\\s*in\\s*[\\'\"]([^\\'\"]+)[\\'\"]'\n#             ]\n\n#             constraints = []\n#             for line in content.splitlines():\n#                 if 'python_version' in line:\n#                     for pattern in markers:\n#                         if match := re.search(pattern, line):\n#                             if len(match.groups()) == 2:  # operator and version\n#                                 op, ver = match.groups()\n#                                 constraints.append(f\"{op}{ver}\")\n#                             else:  # in operator with version list\n#                                 versions = match.group(1).split(',')\n#                                 clean_versions = [v.strip(' \\'\\\"') for v in versions]\n#                                 return f\">={min(clean_versions)},<{max(clean_versions)}\"\n\n#             return ','.join(constraints) if constraints else None\n\n#         except Exception as e:\n#             self._logger.warning(f\"Error parsing requirements.txt: {e}\")\n#             return None\n\n#     def _get_setup_py_python_constraints(self, setup_path: Path | str) -> str | None:\n#         \"\"\"Extract Python version constraints from setup.py file.\n\n#         Args:\n#             setup_path (Path | str):\n#                 The path to the setup.py file.\n\n#         Returns:\n#             The version constraints for Python (or None if not found)\n#         \"\"\"\n#         try:\n#             content = Path(setup_path).read_text()\n\n#             # Find all lines containing 'python' case-insensitive\n#             python_lines = [line for line in content.splitlines()\n#                             if 'python' in line.lower()]\n\n#             version_pattern = r'(?:^|\\D)(\\d+\\.\\d+(?:\\.\\d+)?)'\n#             versions = set()\n\n#             for line in python_lines:\n#                 # Look for version patterns on python lines\n#                 if matches := re.finditer(version_pattern, line):\n#                     versions.update(match.group(1) for match in matches)\n\n#             if versions:\n#                 # Convert to major.minor only and sort\n#                 base_versions = sorted(\n#                     set('.'.join(v.split('.')[:2]) for v in versions),\n#                     key=lambda v: tuple(map(int, v.split('.')))\n#                 )\n\n#                 # Get min/max versions that start with '3'\n#                 py3_versions = [v for v in base_versions if v.startswith('3')]\n#                 if py3_versions:\n#                     min_version = py3_versions[0]\n#                     max_version = py3_versions[-1]\n\n#                     # Convert to constraint (e.g., \">=3.5,<3.9\")\n#                     major, minor = map(int, max_version.split('.'))\n#                     next_minor = f\"{major}.{minor + 1}\"\n#                     return f\">={min_version},<{next_minor}\"\n\n#             return None\n\n#         except Exception as e:\n#             self._logger.warning(f\"Error parsing setup.py: {e}\")\n#             return None\n\n#     def _get_project_python_constraints(self, repo_path: str | Path) -> str | None:\n#         \"\"\"Get Python version constraints from project configuration files.\n\n#         Checks pyproject.toml, setup.py, and requirements.txt in order of preference.\n\n#         Args:\n#             repo_path (str | Path):\n#                 Path to the repo we are finding the best python version for.\n\n#         Returns:\n#             str | None: Version constraint string if found, None otherwise\n#         \"\"\"\n#         # Check pyproject.toml first (PEP 621)\n#         pyproject_path = Path(repo_path) / \"pyproject.toml\"\n#         if pyproject_path.exists():\n#             if constraints := self._get_pyproject_constraints(pyproject_path):\n#                 return constraints\n\n#         # Then check setup.py\n#         setup_path = Path(repo_path) / \"setup.py\"\n#         if setup_path.exists():\n#             if constraints := self._get_setup_py_python_constraints(setup_path):\n#                 return constraints\n\n#         # Finally check requirements.txt\n#         requirements_path = Path(repo_path) / \"requirements.txt\"\n#         if requirements_path.exists():\n#             if constraints := self._get_requirements_constraints(requirements_path):\n#                 return constraints\n\n#         return None\n\n#     def _version_satisfies_constraint(self, version: PythonVersion, constraint: str) -> bool:\n#         \"\"\"Check if version satisfies the given constraint.\n\n#         Args:\n#             version (PythonVersion): Version to check\n#             constraint (str): Version constraint (e.g., \">=3.8,<3.11\")\n\n#         Returns:\n#             bool: True if version satisfies constraint, False otherwise\n#         \"\"\"\n#         try:\n#             from packaging.specifiers import SpecifierSet\n#             from packaging.version import Version\n\n#             spec = SpecifierSet(constraint)\n#             # Use base version for comparison (e.g., 3.10 instead of 3.10.12)\n#             return Version(version.base_version) in spec\n#         except ImportError:\n#             self._logger.warning(\"packaging module not available, falling back to simple version comparison\")\n#             return version.base_version >= constraint.replace('>=', '').replace('>', '').strip()\n#         except Exception as e:\n#             self._logger.warning(f\"Error checking version constraint: {e}\")\n#             return False\n\n#     def _select_python_version(self, repo_path: str | Path) -> PythonVersion:\n#         \"\"\"Select the most appropriate Python version based on constraints and availability.\n\n#         Args:\n#             repo_path (str | Path):\n#                 Path to the repo we are finding the best python version for.\n        \n#         Returns:\n#             PythonVersion: Selected Python version object\n#         \"\"\"\n#         constraint = self._get_project_python_constraints(repo_path)\n\n#         # Filter to only CPython versions (excluding alpha/beta)\n#         cpython_versions = [\n#             v for v in self.python_versions\n#             if v.implementation == PythonImplementation.CPYTHON\n#                and not re.search(r'a|b|rc', v.version)\n#         ]\n\n#         if constraint:\n#             # Filter versions that satisfy the constraint\n#             compatible_versions = [\n#                 v for v in cpython_versions\n#                 if self._version_satisfies_constraint(v, constraint)\n#             ]\n#         else:\n#             compatible_versions = cpython_versions\n\n#         # Sort by version number (newest first)\n#         sorted_versions = sorted(\n#             compatible_versions,\n#             key=lambda v: tuple(map(int, v.version.split('.'))),\n#             reverse=True\n#         )\n\n#         if not sorted_versions:\n#             # Create a fallback version object\n#             self._logger.warning(f\"No compatible versions found, using fallback {self.fallback_python_version}\")\n#             return PythonVersion(\n#                 implementation=PythonImplementation.CPYTHON,\n#                 version=f\"{self.fallback_python_version}.0\",\n#                 is_installed=False,\n#                 path=None\n#             )\n\n#         return sorted_versions[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:41:17.391626Z","iopub.execute_input":"2025-01-26T19:41:17.39202Z","iopub.status.idle":"2025-01-26T19:41:17.401102Z","shell.execute_reply.started":"2025-01-26T19:41:17.391988Z","shell.execute_reply":"2025-01-26T19:41:17.39971Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RepoUVManager(UVManager):\n    \"\"\"A specialized UVManager for tackling SWE-problems.\n    \n    This class can:\n        - clone a GitHub repo\n        - check out commits\n        - detect Python versions\n        - apply patches\n        - install dependencies from the local cloned directory\n\n    Inherits:\n        UVManager: The base environment manager with persistent shell session.\n\n    Attributes:\n        github_repo (GitHubRepo): A reference to the GitHubRepo object managing our cloned repo.\n        python_version_detector (PythonVersionDetector): The object responsible for identifying the python version to use.\n        repo_path (Path | None): The local filesystem path where the repo is cloned.\n        venv_name (str): The name of the virtual environment (usually unique and generated).\n        venv_path (Path): The full path to the virtual environment including the name.\n\n    Extended Features:\n        - Automatic detection of Python version availability (via `uv python list`).\n        - Support for installing dependencies from requirements.txt, pyproject.toml, or setup.py.\n        - Ability to run pytest easily via `run_pytest`.\n        - Optional patch application through the GitHubRepo reference.\n    \"\"\"\n    \n    def __init__(\n        self,\n        venv_dir_path: str | Path,\n        github_repo: \"GitHubRepo\",\n        auto_detect_python: bool = True,\n        fallback_python_version: str = \"3.10\",\n        venv_name: str | None = None,\n        env_vars: dict[str, str] | None = None,\n        do_clone_and_checkout: bool = True\n    ):\n        \"\"\"Initialize the RepoUVManager.\n\n        Args:\n            venv_dir_path (str | Path):\n                Path where the UV virtual environment should be created.\n            github_repo (GitHubRepo):\n                Manages the cloned repository (commit checkout, patching, etc.).\n            auto_detect_python (bool, optional):\n                Whether to auto-detect a suitable Python version from project constraints.\n            fallback_python_version (str, optional):\n                Used if no constraints are found or no matching version is available in 'uv python list'.\n            venv_name (str, optional):\n                The name of the virtual environment to create in venv_dir_path.\n                If None, generate a default with generate_venv_name().\n            env_vars (dict[str, str] | None, optional):\n                Additional environment variables to set in the shell.\n            do_clone_and_checkout (bool, optional):\n                Whether to clone and checkout the env setup commit during init.\n        \"\"\"\n        self._logger = logging.getLogger(self.__class__.__name__)\n\n        # Initialize the github repo (for python versions and other stuff)\n        self.github_repo = github_repo\n        self.repo_path = github_repo.root_repo_path\n        self.clone_and_checkout_repo()\n        \n        self.venv_name = venv_name or self.generate_venv_name()\n        self.venv_path = Path(venv_dir_path) / self.venv_name\n        \n        # Initialize python detector and get the selected python version\n        self.python_version_detector = PythonVersionDetector()\n        self.selected_python = self.python_version_detector.select_python_version(\n            repo_path=self.repo_path,  \n            python_fallback_version=fallback_python_version\n        )\n\n        super().__init__(venv_path=self.venv_path, python_version=self.selected_python.version, env_vars=env_vars)\n\n    def generate_venv_name(self, instance_id: str | None = None, commit_hash: str | None = None) -> str:\n        \"\"\"Generates a unique environment name based on the repository, instance ID, and commit hash.\n\n        Any arguments not provided are inferred from self.github_repo, where:\n            - instance_id = self.github_repo.instance_id (if defined in GitHubRepo)\n            - commit_hash = self.github_repo.env_setup_commit_hash\n\n        Args:\n            instance_id (str, optional): \n                The repository identifier string, e.g. 'repo__user-repo-issue'.\n            commit_hash (str, optional): \n                The commit hash to reference.\n\n        Returns:\n            str: A unique environment name with a random UUID suffix.\n        \"\"\"\n        instance_id = instance_id or getattr(self.github_repo, \"instance_id\", \"repo\")\n        commit_hash = commit_hash or getattr(self.github_repo, \"env_setup_commit_hash\", \"HEAD\")\n\n        # In case the repo includes slashes\n        sanitized_instance = instance_id.replace(\"/\", \"_\")\n        return f\"{sanitized_instance}-{commit_hash}-{uuid.uuid4().hex[:8]}\"\n    \n    def clone_and_checkout_repo(self, commit_hash: str | None = None) -> Path:\n        \"\"\"Clone the GitHub repo (if not already) and check out the given commit hash.\n\n        Args:\n            commit_hash (str | None):\n                The commit hash to check out. If None, uses the repo's default environment hash.\n\n        Returns:\n            Path: The local path to the cloned repository.\n        \"\"\"\n        repo_path = self.github_repo.clone_and_checkout(checkout_commit_hash=commit_hash)\n        self._logger.info(f\"Repo cloned at {repo_path}\")\n        self.repo_path = repo_path  # This should already be set but we update just in case...\n        return repo_path\n\n    def install_repo_dependencies(self) -> CommandResult | None:\n        \"\"\"Install dependencies from the cloned repo by checking:\n            (1) pyproject.toml build-system requirements\n            (2) requirements.txt\n            (3) pyproject.toml [project] table (install local project)\n            (4) setup.py (install local project)\n            (5) otherwise, do nothing\n\n        We install in editable mode by default, so local code changes are reflected.\n\n        Returns:\n            CommandResult | None:\n                The result of the pip install command, or None if no install was done.\n        \"\"\"\n        if not self.env_ready:\n            raise RuntimeError(\"Environment not ready. Call initialize() first.\")\n\n        if not self.repo_path:\n            self._logger.warning(\"No repo is attached or cloned yet. Skipping dependency install.\")\n            return None\n\n        result: CommandResult | None = None\n\n        # (0) Upgrade pip and setup tools and whatnot\n        self.pip_install(\"--upgrade pip\", cwd=self.repo_path, verbose=True)\n        self.pip_install(\"--upgrade setuptools wheel\", cwd=self.repo_path, verbose=True)\n        \n        # (1) Build-System Requires: install them if present in pyproject.toml\n        pyproj = self.repo_path / \"pyproject.toml\"\n        if pyproj.exists():\n            build_system_requires = self._get_build_system_requires(pyproj)\n            if build_system_requires:\n                self._logger.info(f\"Installing build-system requirements: {build_system_requires}\")\n                build_cmd = \" \".join(build_system_requires)\n                build_result = self.pip_install(build_cmd, editable=True, cwd=self.repo_path, verbose=True)\n                if not build_result.success:\n                    self._logger.error(\"Failed to install build-system requirements!\")\n                    return build_result  # Early return if needed\n            else:\n                self._logger.debug(\"No build-system.requires found in pyproject.toml\")\n\n        # (2) requirements.txt and requirements-dev.txt\n        req_file = self.repo_path / \"requirements.txt\"\n        req_dev_file = self.repo_path / \"requirements-dev.txt\"\n        if req_file.exists():\n            self._logger.info(\"Installing dependencies from requirements.txt...\")\n            result = self.pip_install([\"-r\", str(req_file)], editable=True, cwd=self.repo_path)\n            if req_dev_file.exists():\n                self._logger.info(\"Installing dev dependencies from requirements-dev.txt...\")\n                dev_result = sself.pip_install([\"-r\", str(req_dev_file)], editable=True, cwd=self.repo_path)\n                if not dev_result.success:\n                    self._logger.error(\"Failed to install dev dependencies!\")\n                    return dev_result\n            return result\n\n        # (3) pyproject.toml [project] \n        if pyproj.exists():\n            content = pyproj.read_text()\n            if \"[project]\" in content:\n                self._logger.info(\"Detected [project] table in pyproject.toml; installing with uv pip install . [editable]\")\n                result = self.pip_install(\".\", editable=True, cwd=self.repo_path)\n                return result\n            else:\n                # Rename pyproject.toml -> pyproject.toml.bak (effectively 'hiding' it)\n                pyproj_backup = pyproj.with_name(\"pyproject.toml.bak\")\n                pyproj.rename(pyproj_backup)\n                self._logger.info(\"pyproject.toml found but no [project] table (hiding via rename). Will check setup.py next.\")\n\n        # (4) Fallback to setup.py\n        setup_py = self.repo_path / \"setup.py\"\n        if setup_py.exists():\n            self._logger.info(\"Installing local code via setup.py with uv pip install . (editable)\")\n            result = self.pip_install(\".\", editable=True, cwd=self.repo_path)\n            return result\n\n        # (5) No recognized files\n        self._logger.info(\n            \"No recognized dependency file found (requirements.txt, pyproject.toml, or setup.py). Skipping installation.\"\n        )\n        return result\n\n    def _get_build_system_requires(self, pyproject_path: Path) -> list[str]:\n        \"\"\"Extract build-system dependencies from pyproject.toml.\"\"\"\n        try:\n            import tomllib  # Python 3.11+; use 'tomli' for older versions\n        except ImportError:\n            import tomli as tomllib\n\n        requirements = []\n        toml_data = tomllib.loads(pyproject_path.read_text())\n\n        build_system = toml_data.get(\"build-system\", {})\n        if not build_system:\n            return requirements  # No build-system table found\n\n        requires_list = build_system.get(\"requires\", [])\n        for item in requires_list:\n            requirements.append(f'\"{item}\"')  # Quote items to avoid shell parsing issues\n\n        # Optionally, infer extras based on build-backend\n        build_backend = build_system.get(\"build-backend\")\n        if build_backend == \"setuptools.build_meta\" and not any(\"wheel\" in r for r in requirements):\n            requirements.append('\"wheel\"')\n        elif build_backend == \"hatchling.build\" and not any(\"editables\" in r for r in requirements):\n            requirements.append('\"editables\"')\n\n        return requirements\n\n    def run_pytest(self, test_path: str | None = None, extra_args: list[str] | None = None) -> \"CommandResult\":\n        \"\"\"Convenience method to run pytest in the environment.\n\n        Args:\n            test_path (str | None):\n                Specific test path or module to run (e.g. \"tests/test_file.py::test_func\").\n                If None, runs all tests in the current repo_path.\n            extra_args (list[str] | None):\n                Additional command-line arguments (e.g. [\"-v\", \"--pdb\"]).\n\n        Returns:\n            CommandResult: Contains stdout, stderr, and return code.\n        \"\"\"\n        if not self.env_ready:\n            raise RuntimeError(\"Environment not ready. Call initialize() first.\")\n        if not self.repo_path:\n            raise RuntimeError(\"No repo_path available; cannot run tests in an un-cloned repo.\")\n\n        cmd_tokens = [\"pytest\"]\n        if test_path:\n            cmd_tokens.append(test_path)\n        if extra_args:\n            cmd_tokens.extend(extra_args)\n\n        # Build the final command string\n        cmd_str = \" \".join(cmd_tokens)\n        self._logger.info(f\"Running pytest: {cmd_str} (cwd={self.repo_path})\")\n        return self.send(cmd_str, cwd=self.repo_path)\n\n    def apply_patch(self, patch_content: str) -> None:\n        \"\"\"Applies a patch to the cloned repository by delegating to GitHubRepo.\n\n        Args:\n            patch_content (str): The diff/patch content as a string.\n\n        Raises:\n            RuntimeError: If no GitHubRepo is set or patch application fails.\n        \"\"\"\n        self._logger.info(\"Applying patch content via GitHubRepo...\")\n        self.github_repo.apply_patch(patch_content)\n\n    def get_git_patch(self) -> str | None:\n        \"\"\"Generate a git patch from the current changes in the repo.\"\"\"\n        try:\n            result = self.send(\"git diff\", cwd=self.repo_path)\n            if not result.success or not result.stdout.strip():\n                return None\n            \n            # Check if there's at least one line starting with '+'\n            lines = result.stdout.splitlines()\n            if not any(line.startswith('+') for line in lines):\n                return None\n            \n            # Otherwise return the stdout\n            return result.stdout\n\n        # Catch any errors and log and return None\n        except Exception as e:\n            self._logger.error(f\"Failed to generate git patch: {e}\")\n            return None\n    \n    def remove_github_actions(self) -> None:\n        \"\"\"Remove the .github directory in the cloned repo, if it exists.\n        \n        This is done to prevent any unwanted GitHub Actions from interfering locally.\n        \"\"\"\n        if not self.repo_path:\n            self._logger.warning(\"No repo path is set, cannot remove .github folder.\")\n            return\n\n        github_dir = self.repo_path / \".github\"\n        if github_dir.exists():\n            import shutil\n            self._logger.info(f\"Removing GitHub Actions files at {github_dir}\")\n            shutil.rmtree(github_dir)\n        else:\n            self._logger.info(\"No .github directory found. Skipping removal.\")\n\n    def cleanup_repo(self) -> None:\n        \"\"\"Cleans up the cloned repository directory from disk (if desired).\n        \n        This is separate from environment cleanup, which ends the shell session.\n        \"\"\"\n        if not self.repo_path:\n            self._logger.warning(\"No repo path is set. Nothing to remove.\")\n            return\n\n        import shutil\n        if self.repo_path.exists():\n            self._logger.info(f\"Removing cloned repository at {self.repo_path}...\")\n            shutil.rmtree(self.repo_path, ignore_errors=True)\n            self._logger.info(\"Repository folder removed.\")\n\n    def cleanup(self, remove_venv: bool = True) -> None:\n        \"\"\"Override cleanup() so we can remove the environment directory too, plus the repo.\n\n        This method:\n            - Terminates the persistent shell session from UVManager\n            - Resets env_ready to False\n            - Removes the cloned repo from disk\n            - (Optionally, remove self.venv_path if you want the venv folder gone too)\n\n        Args:\n            remove_venv (bool): Whether to remove the environment directory\n        \"\"\"\n        super().cleanup()  # Kills the shell, sets env_ready=False\n        self.cleanup_repo()\n        \n        if remove_venv:\n            if self.venv_path.exists():\n                self._logger.info(f\"Removing environment directory at {self.venv_path}...\")\n                shutil.rmtree(self.venv_path, ignore_errors=True)\n                self._logger.info(\"Environment directory removed.\")\n            else:\n                self._logger.info(f\"No environment directory found at {self.venv_path}...\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:45:59.666573Z","iopub.execute_input":"2025-01-26T19:45:59.667166Z","iopub.status.idle":"2025-01-26T19:45:59.738857Z","shell.execute_reply.started":"2025-01-26T19:45:59.6671Z","shell.execute_reply":"2025-01-26T19:45:59.736041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def setup_demo_environment(\n    instance: \"SWEBenchInstance\", \n    root_dir: str | Path = \"/kaggle/tmp\",\n    fallback_python_version: str = \"3.10\"\n) -> RepoUVManager:\n    \"\"\"\n    Sets up a demo environment for a given SWE-Bench instance following the steps:\n        (1) Generate a name for the environment (instance ID + commit hash).\n        (2) Initialize the repository handler and clone the repo at environment_setup_commit.\n        (3) Initialize the RepoUVManager with fallback_python_version (or a more advanced detection).\n        (4) Create the virtual environment.\n        (5) Install dependencies (requirements.txt, pyproject.toml, or setup.py).\n        (6) Checkout the base_commit after installation.\n\n    Args:\n        instance (SWEBenchInstance):\n            Contains information about the environment (repo, commits, etc.).\n        root_dir (str | Path):\n            The root directory where the repository should be cloned.\n        fallback_python_version (str):\n            The Python version to use if no advanced detection is done.\n\n    Returns:\n        RepoUVManager:\n            - The specialized UV environment manager.\n    \"\"\"\n    \n    # ----------------------------------------------------------\n    # (1) Create the GitHub repo object\n    # ----------------------------------------------------------\n    github_repo = GitHubRepo.from_swebench_instance(\n        instance,\n        root_dir=root_dir\n    )\n\n    # ----------------------------------------------------------\n    # (2) Initialize a RepoUVManager with the new environment name\n    #     and link to the cloned GitHubRepo\n    # ----------------------------------------------------------\n    # We store the environment in root_dir/env_name (or any path you like)\n    repo_uv = RepoUVManager(\n        venv_dir_path=root_dir,\n        github_repo=github_repo,\n        fallback_python_version=fallback_python_version\n    )\n\n    # ----------------------------------------------------------\n    # (3) Clone the repo and checkout the correct commit for \n    #     env setup. Done internally now within __init__\n    # ----------------------------------------------------------\n    # repo_uv.clone_and_checkout_repo()\n\n    # ----------------------------------------------------------\n    # (4) Remove github actions\n    # ----------------------------------------------------------\n    repo_uv.remove_github_actions()\n    \n    # ----------------------------------------------------------\n    # (5) Create the UV virtual environment\n    # ----------------------------------------------------------\n    repo_uv.initialize()\n\n    # ----------------------------------------------------------\n    # (6) Install dependencies, if any\n    #     - This might look in requirements.txt or do 'uv pip install .'\n    # ----------------------------------------------------------\n    repo_uv.install_repo_dependencies()\n\n    # ----------------------------------------------------------\n    # (7) Checkout the base_commit if different from environment_setup_commit\n    # ----------------------------------------------------------\n    if instance.base_commit != instance.environment_setup_commit:\n        repo_uv.github_repo.checkout_commit(instance.base_commit)\n\n    # Return both objects so user can interact further\n    return repo_uv\n\ndemo_repo_uv = setup_demo_environment(demo_instance, \"/kaggle/working\")\ndemo_repo_uv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:46:00.043985Z","iopub.execute_input":"2025-01-26T19:46:00.044459Z","iopub.status.idle":"2025-01-26T19:46:30.23223Z","shell.execute_reply.started":"2025-01-26T19:46:00.044418Z","shell.execute_reply":"2025-01-26T19:46:30.230448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# https://github.com/sympy/sympy/blob/a36caf5c74fe654cedc488e8a8a05fad388f8406/sympy/release.py\ndemo_repo_uv.send(\"uv run python -c 'import sympy; print(sympy.__version__)'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:46:30.23466Z","iopub.execute_input":"2025-01-26T19:46:30.235141Z","iopub.status.idle":"2025-01-26T19:46:32.677712Z","shell.execute_reply.started":"2025-01-26T19:46:30.23511Z","shell.execute_reply":"2025-01-26T19:46:32.676496Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rich.print(demo_instance.problem_statement)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:46:32.679458Z","iopub.execute_input":"2025-01-26T19:46:32.679858Z","iopub.status.idle":"2025-01-26T19:46:32.689392Z","shell.execute_reply.started":"2025-01-26T19:46:32.679826Z","shell.execute_reply":"2025-01-26T19:46:32.688003Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<br>\n\n### **Side Note on PyTest Node ID Format**\n\n---\n\nWe see something like this in `PASS_TO_PASS` and `FAIL_TO_PASS` columns in our dataframes: **`'path/to/test_file_name.py::x__y__z'`**. This pattern follows **pytest's node ID format**, which is used for specifying individual test cases. \n\nThis format consists of:\n- **Path to the test file**: e.g., **`test/cli/commands_test.py`**\n- **Double colons (`::`)**: Used to separate the file path from the test function name.\n- **Test function name**: e.g., **`x__y__z`**\n  - When a test function name contains double underscores (`__`), it often indicates parameterized tests or specific test cases in dynamically generated tests.\n\nWhen running pytest, you can target individual test functions using this format. \n\nFor example...\n\n```bash\npytest test/cli/commands_test.py::test_some_function\n```\n\n... will execute only the **`test_some_function`** test inside **`test/cli/commands_test.py`**.\n\n---\n\n**Ideally we want to understand from our datasets which tests are updated by the test patch so we can run these updated tests and demonstrate the code failure (prior to us patching it)**","metadata":{}},{"cell_type":"code","source":"def extract_updated_tests(diff_text: str) -> list[str]:\n    \"\"\"Extracts pytest-compatible test identifiers from a git diff.\n    \n    This function identifies modified test files and their respective test functions\n    using regex, returning them in the pytest node ID format: `path/to/test_file.py::test_function_name`.\n    If a test file is modified but no specific test functions are added/modified,\n    returns just the file path to run all tests in that file.\n    \n    Args:\n        diff_text (str): The git diff output containing file changes and modifications.\n    \n    Returns:\n        list[str]: A list of pytest test identifiers in the format `file_path::test_function_name`\n                   or just `file_path` for modified test files.\n    \"\"\"\n    # Patterns for identifying test files and functions\n    test_file_pattern = re.compile(r'^[+]{3} b/(.+?test.*?\\.py)', re.MULTILINE)\n    test_func_pattern = re.compile(r'^\\+\\s*def (test_[a-zA-Z0-9_]+)', re.MULTILINE)\n    \n    # Pattern for parameterized tests (they might have different signature)\n    param_test_pattern = re.compile(\n        r'^\\+.*?@pytest\\.mark\\.parametrize.*?\\n.*?\\n*?^\\+\\s*def (test_[a-zA-Z0-9_]+)',\n        re.MULTILINE | re.DOTALL\n    )\n    \n    test_identifiers = set()\n    \n    # Split the diff into file chunks\n    diff_files = diff_text.split('diff --git ')\n    \n    for diff_chunk in diff_files[1:]:  # Skip the first empty chunk\n        # Extract the test file path\n        file_matches = test_file_pattern.findall(diff_chunk)\n        if not file_matches:\n            continue\n            \n        file_path = file_matches[0]\n        \n        # Skip renamed files without content changes\n        if 'similarity index 100%' in diff_chunk:\n            continue\n            \n        # Extract all test functions (both regular and parameterized)\n        test_funcs = set(test_func_pattern.findall(diff_chunk))\n        param_funcs = set(param_test_pattern.findall(diff_chunk))\n        \n        # Combine all found test functions\n        all_funcs = test_funcs.union(param_funcs)\n        \n        if all_funcs:\n            # If we found specific test functions, add them with the file path\n            for func_name in all_funcs:\n                test_identifiers.add(f\"{file_path}::{func_name}\")\n        else:\n            # If the test file was modified but no specific test functions were found,\n            # add the file path to run all tests in that file\n            test_identifiers.add(file_path)\n    \n    return sorted(list(test_identifiers))\n\n# hf_dfs[\"swe_bench_lite\"][\"dev\"][\"test_patch\"][10]\ntest_identifiers = extract_updated_tests(demo_instance.test_patch)\nall_updated_tests = extract_updated_tests(demo_instance.test_patch)\nrich.print(f\"Running the following updated test(s): {all_updated_tests}\\n\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:49:02.207525Z","iopub.execute_input":"2025-01-26T19:49:02.207958Z","iopub.status.idle":"2025-01-26T19:49:02.222799Z","shell.execute_reply.started":"2025-01-26T19:49:02.207923Z","shell.execute_reply":"2025-01-26T19:49:02.221572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"demo_repo_uv.apply_patch(demo_instance.test_patch)\nwith suppress_logging_below(logging.WARNING):\n    for failing_test in all_updated_tests:\n        rich.print(\"\\n\\n\", Markdown(\"---\"), \"\\n\\n\")\n        rich.print(demo_repo_uv.run_pytest(failing_test).stdout)\n        rich.print(\"\\n\\n\", Markdown(\"---\"), \"\\n\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:49:17.350781Z","iopub.execute_input":"2025-01-26T19:49:17.351144Z","iopub.status.idle":"2025-01-26T19:49:19.664848Z","shell.execute_reply.started":"2025-01-26T19:49:17.351118Z","shell.execute_reply":"2025-01-26T19:49:19.663465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"demo_repo_uv.apply_patch(demo_instance.patch)\nwith suppress_logging_below(logging.WARNING):\n    for test in all_updated_tests:\n        rich.print(\"\\n\\n\", Markdown(\"---\"), \"\\n\\n\")\n        rich.print(demo_repo_uv.run_pytest(test).stdout)\n        rich.print(\"\\n\\n\", Markdown(\"---\"), \"\\n\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-26T19:50:00.604962Z","iopub.execute_input":"2025-01-26T19:50:00.605335Z","iopub.status.idle":"2025-01-26T19:50:02.75281Z","shell.execute_reply.started":"2025-01-26T19:50:00.605285Z","shell.execute_reply":"2025-01-26T19:50:02.751747Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <s>UV Environment Manager</s> **OLD**","metadata":{"_kg_hide-output":true,"_kg_hide-input":true}},{"cell_type":"code","source":"# class UVEnvironmentManager:\n#     \"\"\"Manages creation and usage of a UV environment, optionally as a context manager.\n\n#     Attributes:\n#         env_name (str): Unique name for the environment.\n#         python_version (str): Python version to use (e.g., \"3.10\").\n#         repo_path (Path): Path to the cloned repository on disk.\n#         env_path (Path): Path to the UV environment directory.\n#         _logger (logging.Logger): Internal logger instance.\n#         python_versions (list[PythonVersion]): Python versions we have at our disposal\n#         selected_python (PythonVersion): The version that will be used within this env.\n#     \"\"\"\n\n#     env_name: str\n#     repo_path: Path\n#     env_path: Path\n#     _logger: logging.Logger\n#     python_versions: list[PythonVersion]\n#     selected_python = PythonVersion\n\n#     def __init__(self, env_name: str, repo_path: Path, fallback_python_version: str = \"3.10\") -> None:\n#         \"\"\"Initialize the UVEnvironmentManager.\n\n#         Args:\n#             env_name (str): Unique name for the environment.\n#             repo_path (str | Path): Path to the cloned repository on disk.\n#             fallback_python_version (str): Fallback Python version if no compatible version is found.\n#         \"\"\"\n#         self.env_name = env_name\n#         self.repo_path = Path(repo_path)\n#         self.env_path = repo_path / \".venv\"\n#         self._logger = logging.getLogger(self.__class__.__name__)\n#         self.fallback_python_version = fallback_python_version\n\n        \n\n#     def __enter__(self) -> \"UVEnvironmentManager\":\n#         \"\"\"Context manager entry: create environment.\n\n#         Returns:\n#             The current instance of UVEnvironmentManager.\n#         \"\"\"\n#         self.create_environment()\n#         return self\n\n#     def __exit__(self, exc_type: Any, exc_val: Any, exc_tb: Any) -> None:\n#         \"\"\"Context manager exit: cleanup environment.\n\n#         Args:\n#             exc_type (Any): Exception type (if any).\n#             exc_val (Any): Exception value (if any).\n#             exc_tb (Any): Exception traceback (if any).\n\n#         Returns:\n#             None; any exceptions are re-raised.\n#         \"\"\"\n#         self.cleanup()\n\n#     def __repr__(self) -> str:\n#         \"\"\"Official string representation, used for debugging.\"\"\"\n#         return (f\"{self.__class__.__name__}(env_name={self.env_name!r}, \"\n#                 f\"selected_python={self.selected_python!r}, \"\n#                 f\"repo_path={str(self.repo_path)!r}, \"\n#                 f\"env_path={str(self.env_path)!r})\")\n\n#     def __str__(self) -> str:\n#         \"\"\"Informal string representation, used for user-facing display.\"\"\"\n#         return str(self.__repr__())\n\n    \n\n#     def create_environment(self) -> None:\n#         \"\"\"Creates a UV virtual environment.\"\"\"\n#         self._logger.info(\n#             f\"Creating UV environment '{self.env_name}' \"\n#             f\"with Python {self.selected_python.base_version}...\"\n#         )\n\n#         try:\n#             subprocess.run(\n#                 [\"uv\", \"venv\", \"--python\", self.selected_python.base_version, str(self.env_path)],\n#                 capture_output=True, text=True, check=True\n#             )\n#             self._logger.info(f\"UV environment created successfully at {self.env_path}.\")\n#         except subprocess.CalledProcessError as e:\n#             raise RuntimeError(f\"Failed to create UV environment: {e.stderr}\") from e\n\n#         # (2) Remove Github Actions if Found\n#         self.remove_github_actions()\n\n#     def _uv_pip_install(\n#         self,\n#         package: str = \".\",\n#         try_editable_install: bool = True,\n#         from_requirements: bool = False\n#     ) -> None:\n#         \"\"\"Installs a package using pip within the UV environment.\n\n#         Args:\n#             package (str, optional):\n#                 The path of the package to install (e.g., \"numpy\" or \"requests\").\n#             try_editable_install (bool, optional):\n#                 Whether to attempt an editable install (e.g., \"pip install -e .\").\n#             from_requirements (bool, optional):\n#                 Whether to install from a requirements file.\n#         \"\"\"\n#         # (1) Run the command in the UV environment\n#         self._logger.info(f\"Running pip install within the UV env '{self.env_name}' for package: {package}\")\n\n#         # (2) Initialize install commands\n#         _install_commands = [\"uv\", \"pip\", \"install\"]\n\n#         # (3) Add the install from requirements file if needed\n#         if from_requirements:\n#             _install_commands += [\"-r\", \"requirements.txt\"]\n\n#         # (4) Add the package to install (editable optional\n#         _install_commands += [\"-e\", package] if try_editable_install else [package]\n\n#         return self._execute_subprocess_run(_install_commands, cwd=self.repo_path)\n\n#     def pip_install(\n#             self,\n#             package: str = \".\",\n#             try_editable_install: bool = True,\n#             from_requirements: bool = False\n#     ) -> CommandResult:\n#         \"\"\"Installs a package using pip within the UV environment.\n\n#         Args:\n#             package (str, optional):\n#                 The path of the package to install (e.g., \"numpy\" or \"requests\").\n#             try_editable_install (bool, optional):\n#                 Whether to attempt an editable install (e.g., \"uv pip install -e .\").\n#             from_requirements (bool, optional):\n#                 Whether to install from a requirements file.\n\n#         Returns:\n#             None; raises on failure.\n#         \"\"\"\n#         return self._uv_pip_install(package, try_editable_install, from_requirements)\n\n#     def remove_github_actions(self) -> None:\n#         \"\"\"Removes GitHub Actions files from the repository.\n\n#         This is useful when running tests locally to avoid conflicts with the CI/CD workflow.\n#         \"\"\"\n#         self._logger.info(\"Removing GitHub Actions files...\")\n#         github_dir = self.repo_path / \".github\"\n#         if github_dir.exists():\n#             shutil.rmtree(github_dir)\n#             self._logger.info(\"GitHub Actions files removed.\")\n#         else:\n#             self._logger.info(\"No GitHub Actions files found; skipping removal.\")\n\n#     def force_uv_to_install_from_setup_py(self, pyproject_file: Path):\n#         \"\"\"Forces UV to install dependencies from setup.py instead of pyproject.toml.\n\n#         This is required if the pyproject.toml is invalid in someway as recognized by UV,\n#         and we need to fallback to the legacy setup.py method.\n\n#         Args:\n#             pyproject_file (str): Path to the pyproject.toml file.\n#         \"\"\"\n\n#         # (1) Backup the original pyproject.toml file\n#         backup_path = pyproject_file.with_name(\"pyproject.toml.bak\")\n\n#         # (2) Rename pyproject.toml -> pyproject.toml.bak (effectively 'hiding' it)\n#         pyproject_file.rename(backup_path)\n\n#         # (3) Install dependencies with legacy setup.py\n#         self.pip_install()\n\n#         ## (4) Rename pyproject.toml.bak back to pyproject.toml ('unhiding' it)\n#         # backup_path.rename(pyproject_file)\n\n#     def install_dependencies(self) -> None:\n#         \"\"\"Installs project dependencies using UV within the environment.\"\"\"\n#         self._logger.info(\"Attempting to install dependencies...\")\n#         # (1) Define paths to possible install files\n#         requirements_file = self.repo_path / \"requirements.txt\"\n#         setup_file = self.repo_path / \"setup.py\"\n#         pyproject_file = self.repo_path / \"pyproject.toml\"\n\n#         try:\n#             # (2a) If there's a requirements.txt, install that first.\n#             if requirements_file.exists():\n#                 self._logger.info(\"Installing from requirements.txt...\")\n#                 self.pip_install(from_requirements=True)\n\n#             # (2b) If there's a pyproject.toml with a [project] table, do a normal 'pip install .'\n#             elif pyproject_file.exists():\n#                 content = pyproject_file.read_text()\n\n#                 # (2bi) If all is well install from pyproject.toml\n#                 if \"[project]\" in content:\n#                     self._logger.info(\"Detected [project] table in pyproject.toml; installing with uv pip install .\")\n#                     self.pip_install()\n#                 # (2bii) If all is NNOT well fallback to setup.py (with special hiding logic)\n#                 elif setup_file.exists():\n#                     self._logger.info(\"No [project] table found; using legacy setup.py install.\")\n#                     self.force_uv_to_install_from_setup_py(pyproject_file)\n#                 # (2biii) If neither exists (nor a requirements) we giveup\n#                 else:\n#                     self._logger.info(\"No [project] table or setup.py found; skipping dependency install.\")\n\n#             # (2c) Otherwise, fallback to setup.py if it exists\n#             elif setup_file.exists():\n#                 self._logger.info(\"Installing via setup.py...\")\n#                 self.pip_install()\n\n#             else:\n#                 self._logger.info(\"No recognized dependency file found; skipping installation.\")\n\n#         except subprocess.CalledProcessError as e:\n#             raise RuntimeError(f\"Failed to install dependencies: {e.stderr}\") from e\n\n#     def _execute_subprocess_run(self, command: list[str], cwd: Path) -> CommandResult:\n#         \"\"\"Executes a command using subprocess.run.\n\n#         Args:\n#             command (list[str]): The command to execute as a list of strings.\n#             cwd (str): The working directory to run the command from.\n\n#         Returns:\n#             CommandResult: The result of the command execution\n#         \"\"\"\n#         # (1) Run the subprocess command\n#         result = subprocess.run(\n#             command,\n#             cwd=cwd,\n#             capture_output=True,\n#             text=True,\n#         )\n\n#         # (2) We can choose to raise on non-zero or simply return the result\n#         if result.returncode != 0:\n#             self._logger.error(f\"Command failed (exit code {result.returncode}):\\n{result.stderr}\")\n\n#         # (3) Return the result\n#         return CommandResult(command, result.returncode, result.stdout, result.stderr)\n\n#     def execute_run_command(self, command: str, cwd: Path | str | None = None) -> CommandResult:\n#         \"\"\"Executes a command inside the UV environment.\n\n#         Args:\n#             command (str):\n#                 The shell command to execute (e.g., \"pytest\" or \"python script.py\").\n#             cwd (Path, optional):\n#                 Optional directory to run the command from. Defaults to the repo path.\n\n#         Returns:\n#             A tuple of (returncode, stdout, stderr).\n#                 - returncode (int): The exit code of the command.\n#                 - stdout (str): The standard output of the command.\n#                 - stderr (str): The standard error of the command.\n\n#         Raises:\n#             subprocess.CalledProcessError: If the command fails (non-zero return code).\n#         \"\"\"\n#         # (1) Set the working directory if provided or default to the repo path\n#         cwd = Path(cwd or self.repo_path)\n\n#         # (2) Run the command in the UV environment\n#         self._logger.info(f\"Running command in UV env '{self.env_name}' (cwd={cwd}): {command}\")\n#         return self._execute_subprocess_run([\"uv\", \"run\", \"sh\", \"-c\", command], cwd)\n\n\n#     def run_pytest(self, test_path: str | Path, extra_args: list[str] | None = None) -> CommandResult:\n#         \"\"\"Runs a specific test (or tests) using pytest inside the UV environment.\n\n#         Args:\n#             test_path (str | Path):\n#                 The relative path (or module::test_func) to run.\n#                     - e.g. \"tests/test_something.py::test_something\"\n#             extra_args (list[str], optional):\n#                 Optional extra flags or arguments for pytest\n#                     - e.g. [\"-v\", \"--pdb\"]\n\n#         Returns:\n#             A CommandResult with stdout, stderr, and returncode.\n#         \"\"\"\n#         # (0) Parse extra-args if provided otherwise default to empty list.\n#         if not extra_args:\n#             extra_args = []\n\n#         # (1) Get test path\n#         if self.repo_path not in Path(test_path).parents:\n#             test_path = str(self.repo_path / test_path)\n#         else:\n#             test_path = str(test_path)\n\n#         # (2) Build the command, e.g. \"pytest test/cli/commands_test.py::test__cli__command_directed -v\"\n#         cmd_tokens = [\"pytest\", test_path, *extra_args]\n\n#         # (3) Join into a single shell command\n#         cmd_str = \" \".join(cmd_tokens)\n\n#         # (4) Run the command and return\n#         return self.execute_run_command(cmd_str)\n\n#     def cleanup(self) -> None:\n#         \"\"\"Deletes the UV environment directory from the filesystem.\"\"\"\n#         # (1) Remove the environment directory and its contents\n#         self._logger.info(f\"Cleaning up environment at {self.env_path}...\")\n#         if self.env_path.exists():\n#             shutil.rmtree(self.env_path)\n#             self._logger.info(f\"Environment directory removed: {self.env_path}\")","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"_kg_hide-output":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def setup_demo_environment(\n    instance: SWEBenchInstance, \n    root_dir: str = \"/kaggle/tmp\",\n    fallback_python_version: str = \"3.10\"\n) -> None:\n    \"\"\"Sets up a demo environment for a given SweBench instance.\n    \n    This setup involves:\n        - creating a virtual environment\n        - cloning the repository\n        - installing dependencies\n        - checkout out the correct issue respective commit \n    \n    Args:\n        instance (SweBenchInstance): The instance containing information about the environment and repository.\n        root_dir (str, optional): The root directory where the repository should be cloned.\n        fallback_python_version (str, optional): \n            The Python version to use as a fallback for the virtual environment.\n            Only applies if the python version cannot be automatically detected and used.\n    \n    Returns:\n        None\n    \"\"\"\n    # Step 1: Generate a name for the environment based on instance ID and commit hash.\n    _env_name = generate_env_name(\n        instance_id=instance.instance_id, \n        commit_hash=instance.environment_setup_commit\n    )\n    \n    # Step 2: Initialize the repository handler and clone the repository.\n    _repo_handler = GitHubRepo.from_swebench_instance(demo_instance, root_dir=root_dir)\n    _repo_path = _repo_handler.clone_and_checkout()\n    \n    # Step 3: Initialize the environment manager with the specified Python version.\n    _env_manager = UVEnvironmentManager(_env_name, _repo_path, fallback_python_version=fallback_python_version)\n    \n    # Step 4: Create the virtual environment.\n    _env_manager.create_environment()\n    \n    # Step 5: Install dependencies if a requirements file, setup.py, or pyproject.toml is found.\n    _env_manager.install_dependencies()\n    \n    # Step 6: Checkout the repository at the correct commit after installation.\n    _repo_handler.checkout_commit(_repo_handler.base_commit_hash)\n\n    return _env_manager, _repo_handler\n\ndemo_env_manager, demo_repo_handler = setup_demo_environment(instance=demo_instance, root_dir=\"/kaggle/working\")","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"_kg_hide-output":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    import pydicom\nexcept ImportError as e:\n    rich.print(f\"\\n[bold red]EXCEPTION FROM KAGGLE: {e}[/bold red]\\n\\n[bold]... We cannot import pydicom in the kaggle environment but we SHOULD be able to within the UV environment ... [/bold]\\n\\n\")\n    rich.print(demo_env_manager.execute_command(\"python -c 'import pydicom; print(\\\"\\\\n\\\\nThe version of pydicom within the UV environment is:\\\", pydicom.__version__)'\").stdout)\n    rich.print(\"\\n\\n[bold cyan]For confirmation here is the link to the version file for this commit on Github:[/bold cyan]\")\n    display(\"https://github.com/pydicom/pydicom/blob/7241f5d9db0de589b230bb84212fbb643a7c86c3/pydicom/_version.py#L4\")","metadata":{"trusted":true,"_kg_hide-output":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"demo_env_manager.run_pytest(str(all_updated_tests[0]), extra_args=[\"-vv\",])","metadata":{"trusted":true,"_kg_hide-output":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rich.print(\"\\n\\n[bold red]... THIS WILL FAIL BECAUSE WE HAVE NOT PATCHED THE TESTS YET ...[/bold red]\")\nfor test_to_show_fail in all_updated_tests:    \n    rich.print(demo_env_manager.run_pytest(test_to_show_fail))\ndisplay(Markdown(\"<br><br>---\"))\nrich.print(\"\\n\\n\\n\\n[bold green]... THESE SHOULD WORK BECAUSE WE HAVE NOW APPLIED THE TEST PATCHES ...[/bold green]\")\ndemo_repo_handler.apply_patch(demo_instance.test_patch)\nfor test_to_show_fail in all_updated_tests:    \n    rich.print(demo_env_manager.run_pytest(test_to_show_fail))","metadata":{"trusted":true,"_kg_hide-output":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rich.print(demo_env_manager.pip_install(\"setuptools\", False))\ndemo_env_manager.execute_command(\"python setup.py install\")","metadata":{"trusted":true,"_kg_hide-output":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\n# Create the test script\ntest_script = \"\"\"\nimport sqlfluff\n\ndef test_sql(sql: str, dialect: str = \"tsql\") -> None:\n    print(f\"\\\\nTesting SQL:\\\\n{sql}\\\\n\")\n    linted = sqlfluff.lint(sql, dialect=dialect)\n    if linted:\n        print(\"Violations found:\")\n        for violation in linted:\n            print(f\"Line {violation.line_no}: {violation.code} - {violation.description}\")\n            print(f\"  Context: {violation.line_pos}\")\n    else:\n        print(\"No violations found.\")\n\n# Test cases\nno_alias_sql = '''\nSELECT [hello]\nFROM\n    mytable\n'''\n\nwith_alias_sql = '''\nSELECT a.[hello]\nFROM\n    mytable AS a\n'''\n\nprint(\"=== Testing SQL without alias ===\")\ntest_sql(no_alias_sql)\n\nprint(\"\\\\n=== Testing SQL with alias ===\")\ntest_sql(with_alias_sql)\n\"\"\"\n\n# Write the script to a file\nscript_path = demo_env_manager.repo_path / \"test_sqlfluff.py\"\nscript_path.write_text(test_script)\n\n# Now we can return the path to be executed\nprint(str(script_path))\ndemo_env_manager.execute_command(f\"python {str(script_path)}\")","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"_kg_hide-output":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <s>E2E</s> **OLD**","metadata":{"_kg_hide-output":true,"_kg_hide-input":true}},{"cell_type":"code","source":"def run_swe_bench_test(\n        instance: SWEBenchInstance,\n        python_version: str = \"3.10\",\n        remove_repo: bool = True\n) -> None:\n    \"\"\"Orchestrates the full flow: clone repo, create UV env, apply patches, install deps, run tests.\n\n    Args:\n        instance (SWEBenchInstance):\n            A SWEBenchInstance containing all relevant data.\n        python_version (str):\n            The Python version to use for UV (e.g. '3.10').\n        remove_repo (bool):\n            Whether to remove the cloned repository folder after the test.\n    \"\"\"\n    # (1) Construct GitHub URL - adapt if your 'repo' field is already a full URL\n    repo_url = instance.repo\n    if \"github.com\" not in repo_url:\n        repo_url = f\"https://github.com/{instance.repo}.git\"\n\n    # (2) Determine environment name\n    env_name = generate_env_name(instance.instance_id, instance.environment_setup_commit)\n\n    # (3) Clone the repository at environment_setup_commit (or base_commit if desired)\n    repo_handler = GitHubRepo(repo_url, commit_hash=instance.base_commit)\n    repo_path = repo_handler.clone_and_checkout()\n\n    # (4) Apply patches if present\n    if instance.test_patch:\n        repo_handler.apply_patch(instance.test_patch)\n    if instance.patch:\n        repo_handler.apply_patch(instance.patch)\n\n    # (5) Create UV environment and install dependencies\n    with UVEnvironmentManager(env_name, repo_path, python_version=python_version) as uv_env:\n        uv_env.install_dependencies()\n\n        # (6) Run tests (example: pytest)\n        try:\n            returncode, stdout, stderr = uv_env.execute_command(\"pytest\")\n            logger.info(\n                \"Tests completed successfully. \"\n                f\"Return code: {returncode}\\nOutput:\\n{stdout}\\nErrors:\\n{stderr}\"\n            )\n        except subprocess.CalledProcessError as e:\n            logger.error(f\"Test command failed: {e.stderr}\")\n\n    # (7)(Optional)\n    #   - Additional cleanup beyond context manager if needed\n    #   - For example:\n    #       --> if you want to remove the entire cloned repository folder:\n    if remove_repo and repo_path.exists():\n        shutil.rmtree(repo_path)\n        logger.info(f\"Repository folder removed: {repo_path}\")\n\n\nrun_swe_bench_test(demo_instance)","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"_kg_hide-output":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}