{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaL4","dataSources":[{"sourceId":84795,"databundleVersionId":11281725,"sourceType":"competition"},{"sourceId":10759592,"sourceType":"datasetVersion","datasetId":6673751},{"sourceId":10795003,"sourceType":"datasetVersion","datasetId":6699450},{"sourceId":10822479,"sourceType":"datasetVersion","datasetId":6719681},{"sourceId":10911480,"sourceType":"datasetVersion","datasetId":6763110},{"sourceId":10922064,"sourceType":"datasetVersion","datasetId":6782867},{"sourceId":10956683,"sourceType":"datasetVersion","datasetId":6816215},{"sourceId":10982514,"sourceType":"datasetVersion","datasetId":6834137},{"sourceId":10984818,"sourceType":"datasetVersion","datasetId":6819022},{"sourceId":10985213,"sourceType":"datasetVersion","datasetId":6837011},{"sourceId":10985273,"sourceType":"datasetVersion","datasetId":6828548},{"sourceId":226831111,"sourceType":"kernelVersion"},{"sourceId":162952,"sourceType":"modelInstanceVersion","modelInstanceId":138579,"modelId":161088}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":22.341371,"end_time":"2024-12-11T03:22:13.479076","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-12-11T03:21:51.137705","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import time\nNOTEBOOK_START_TIME = time.time()\nNOTEBOOK_MAX_ALLOWED_TIME = 60 * 60 * 22 # 22 hours\nMAX_ALLOWED_TIME_PER_SAMPLE = 60 * 10 # 10 minutes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:08:15.899828Z","iopub.execute_input":"2025-03-10T22:08:15.900104Z","iopub.status.idle":"2025-03-10T22:08:15.903356Z","shell.execute_reply.started":"2025-03-10T22:08:15.900082Z","shell.execute_reply":"2025-03-10T22:08:15.902769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!cp -r /kaggle/usr/lib/install_llama_cpp_l4/ .\n!chmod +x install_llama_cpp_l4/llama.cpp/build/bin/llama-server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:08:16.045018Z","iopub.execute_input":"2025-03-10T22:08:16.045232Z","iopub.status.idle":"2025-03-10T22:08:32.062772Z","shell.execute_reply.started":"2025-03-10T22:08:16.045214Z","shell.execute_reply":"2025-03-10T22:08:32.061931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import subprocess\nimport time\nimport os\nimport signal\n\n# Function to start the server\ndef start_llama_server(gpu, port, ctx, seed, is_reasoner):\n    # Start the server process\n    if is_reasoner:\n        cmd = f'CUDA_VISIBLE_DEVICES={gpu} LD_LIBRARY_PATH=$LD_LIBRARY_PATH:./install_llama_cpp_l4/llama.cpp/build/bin/ ./install_llama_cpp_l4/llama.cpp/build/bin/llama-server --model /kaggle/input/qwq-unsloth/QwQ-32B-Q4_K_M.gguf --port {port} --threads 12 --ctx-size {ctx} --n-gpu-layers 99 --seed {seed} --no-mmap --prio 2 --temp 0.6 --repeat-penalty 1.0 --dry-multiplier 0.5 --min-p 0.0 --top-k 40 --top-p 0.95 --samplers \"top_k;top_p;min_p;temperature;dry;typ_p;xtc\" -fa'\n    else:\n        cmd = f\"CUDA_VISIBLE_DEVICES={gpu} LD_LIBRARY_PATH=$LD_LIBRARY_PATH:./install_llama_cpp_l4/llama.cpp/build/bin/ ./install_llama_cpp_l4/llama.cpp/build/bin/llama-server --model /kaggle/input/barto-qwen-coder/Barto-Qwen2.5-Coder-32B-Instruct-Q4_K_M.gguf --port {port} --threads 12 --ctx-size {ctx} --n-gpu-layers 99 --seed {seed} --no-mmap --prio 2 --temp 0.8 --min-p 0.01 --top-k 40 --top-p 0.9 -fa\"\n    # Start process and don't wait for it to complete\n    print(cmd)\n    process = subprocess.Popen(\n        cmd,\n        shell=True,\n        stdout=subprocess.PIPE,\n        stderr=subprocess.PIPE,\n        text=True,\n        preexec_fn=os.setsid  # This allows killing the process group later\n    )\n    \n    print(f\"Started llama-server with PID: {process.pid}\")\n    return process","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:08:32.063906Z","iopub.execute_input":"2025-03-10T22:08:32.064145Z","iopub.status.idle":"2025-03-10T22:08:32.068641Z","shell.execute_reply.started":"2025-03-10T22:08:32.06412Z","shell.execute_reply":"2025-03-10T22:08:32.068066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Start the server\nserver_process0 = start_llama_server(gpu='0,1', port=8091, ctx=32_000, seed=123, is_reasoner=False)\nserver_process1 = start_llama_server(gpu='2', port=8092, ctx=14_000, seed=666, is_reasoner=True)\nserver_process2 = start_llama_server(gpu='3', port=8093, ctx=14_000, seed=777, is_reasoner=True)\n# server_process1 = start_llama_server(1, 8092)\n\n# To stop the server later, call:\n# stop_llama_server(server_process)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:08:32.069779Z","iopub.execute_input":"2025-03-10T22:08:32.069982Z","iopub.status.idle":"2025-03-10T22:08:32.099519Z","shell.execute_reply.started":"2025-03-10T22:08:32.069964Z","shell.execute_reply":"2025-03-10T22:08:32.098812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to stop the server when needed\ndef stop_llama_server(process):\n    try:\n        if process:\n            os.killpg(os.getpgid(process.pid), signal.SIGTERM)\n            print(\"Server stopped\")\n    except:\n        print(\"Failed to stop server\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:08:32.100579Z","iopub.execute_input":"2025-03-10T22:08:32.10079Z","iopub.status.idle":"2025-03-10T22:08:32.104101Z","shell.execute_reply.started":"2025-03-10T22:08:32.100768Z","shell.execute_reply":"2025-03-10T22:08:32.10354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.environ[\"OPENAI_API_KEY\"] = \"hf_...\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:08:32.104758Z","iopub.execute_input":"2025-03-10T22:08:32.10495Z","iopub.status.idle":"2025-03-10T22:08:32.115941Z","shell.execute_reply.started":"2025-03-10T22:08:32.104933Z","shell.execute_reply":"2025-03-10T22:08:32.115361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!cp -r /kaggle/input/custom-agentless-cot/* /kaggle/working","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:08:32.116509Z","iopub.execute_input":"2025-03-10T22:08:32.116702Z","iopub.status.idle":"2025-03-10T22:08:32.424845Z","shell.execute_reply.started":"2025-03-10T22:08:32.116685Z","shell.execute_reply":"2025-03-10T22:08:32.424001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!echo $PYTHONPATH","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:08:32.425702Z","iopub.execute_input":"2025-03-10T22:08:32.425933Z","iopub.status.idle":"2025-03-10T22:08:32.538742Z","shell.execute_reply.started":"2025-03-10T22:08:32.425908Z","shell.execute_reply":"2025-03-10T22:08:32.537933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q -t /kaggle/working/site-packages /kaggle/input/eduardo-wheels/wheels/*.whl\n# !pip download -r requirements.txt -d ./wheels\n# !zip -r wheels.py wheels\n# !pip install -t /kaggle/working/site-packages -r requirements.txt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:08:32.54065Z","iopub.execute_input":"2025-03-10T22:08:32.540885Z","iopub.status.idle":"2025-03-10T22:10:48.886078Z","shell.execute_reply.started":"2025-03-10T22:08:32.54086Z","shell.execute_reply":"2025-03-10T22:10:48.885043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import subprocess\nfrom pathlib import Path\n\n\nclass PythonManager:\n    @staticmethod\n    def install_cmds() -> list[str]:\n        return [\"apt install -y python3.11 python3.11-dev python3.11-venv\"]\n\n    @staticmethod\n    def get_ubuntu_version() -> str:\n        result = subprocess.run(\n            \"lsb_release -rs\",\n            shell=True,\n            executable=\"/bin/bash\",\n            capture_output=True,\n            text=True,\n        )\n        return result.stdout.strip()\n\n    @staticmethod\n    def install_offline_cmds(\n        python_debs_dir: Path,\n        ubuntu_version: str = \"22.04\",\n    ) -> list[str]:\n        python_debs_dir = Path(python_debs_dir)\n        return [\n            f\"dpkg -i {python_debs_dir}/ubuntu_{ubuntu_version}/*.deb\"\n        ]\n\ndef install_python311():\n    cmd = PythonManager.install_offline_cmds(\"/kaggle/input/konwinski-prize/kprize_setup/python3.11\", PythonManager.get_ubuntu_version())[0]\n    result = subprocess.run(\n                cmd,\n                shell=True,\n                executable=\"/bin/bash\",\n                capture_output=True,\n                text=True,\n            )\n    print(result.stdout)\n    print(result.stderr)\n\ninstall_python311()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:10:48.887401Z","iopub.execute_input":"2025-03-10T22:10:48.887713Z","iopub.status.idle":"2025-03-10T22:10:55.338779Z","shell.execute_reply.started":"2025-03-10T22:10:48.887683Z","shell.execute_reply":"2025-03-10T22:10:55.33798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip install -q --no-index --target=/kaggle/working /kaggle/input/vllm-wheels/vllm_wheels/*.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:10:55.339765Z","iopub.execute_input":"2025-03-10T22:10:55.339999Z","iopub.status.idle":"2025-03-10T22:10:55.343192Z","shell.execute_reply.started":"2025-03-10T22:10:55.339978Z","shell.execute_reply":"2025-03-10T22:10:55.342417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/working/site-packages')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:10:55.343952Z","iopub.execute_input":"2025-03-10T22:10:55.344161Z","iopub.status.idle":"2025-03-10T22:10:55.353835Z","shell.execute_reply.started":"2025-03-10T22:10:55.344142Z","shell.execute_reply":"2025-03-10T22:10:55.35307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mkdir -p /kaggle/working/site-packages/llama_index/core/_static/nltk_cache/tokenizers/punkt/PY3_tab","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:10:55.354598Z","iopub.execute_input":"2025-03-10T22:10:55.354814Z","iopub.status.idle":"2025-03-10T22:10:55.474869Z","shell.execute_reply.started":"2025-03-10T22:10:55.354795Z","shell.execute_reply":"2025-03-10T22:10:55.473743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n# https://www.kaggle.com/competitions/ai-mathematical-olympiad-progress-prize-2/discussion/560682#3113134\nos.environ[\"TRITON_PTXAS_PATH\"] = \"/usr/local/cuda/bin/ptxas\"\nos.environ[\"MY_LOG_LEVEL\"] = \"40\" # error\n\nos.environ[\"CUDA_VISIBLE_DEVICES\"] = \"0,1,2,3\"\nos.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"\n\nprint(os.getenv('KAGGLE_IS_COMPETITION_RERUN'))\nprint(len(os.sched_getaffinity(0)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:10:55.475927Z","iopub.execute_input":"2025-03-10T22:10:55.476196Z","iopub.status.idle":"2025-03-10T22:10:55.48126Z","shell.execute_reply.started":"2025-03-10T22:10:55.476168Z","shell.execute_reply":"2025-03-10T22:10:55.480475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip download openai==1.61.1 -d openai_whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:10:55.482019Z","iopub.execute_input":"2025-03-10T22:10:55.482234Z","iopub.status.idle":"2025-03-10T22:11:01.532335Z","shell.execute_reply.started":"2025-03-10T22:10:55.482214Z","shell.execute_reply":"2025-03-10T22:11:01.531424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install --no-index  /kaggle/input/openai-whl/openai_whl/*.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:11:01.533251Z","iopub.execute_input":"2025-03-10T22:11:01.533529Z","iopub.status.idle":"2025-03-10T22:11:07.181217Z","shell.execute_reply.started":"2025-03-10T22:11:01.533484Z","shell.execute_reply":"2025-03-10T22:11:07.180203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nimport time\nimport pandas as pd\nimport polars as pl\nimport io\n\nimport argparse\nfrom pathlib import Path\nfrom queue import Queue\nimport threading\nimport time\nimport json\nfrom git import Repo\nfrom pathlib import Path\nimport subprocess\nimport xml.etree.ElementTree as ET\n\nimport main\nfrom agentless.fl.localize import localize\nfrom agentless.test.select_pass_to_pass_tests import get_test_files\nfrom agentless.util.preprocess_data import get_repo_structure\nfrom agentless.util.utils import setup_logger\n\nimport kaggle_evaluation.konwinski_prize_inference_server","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":14.873526,"end_time":"2024-12-11T03:22:08.818755","exception":false,"start_time":"2024-12-11T03:21:53.945229","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:11:07.182193Z","iopub.execute_input":"2025-03-10T22:11:07.182458Z","iopub.status.idle":"2025-03-10T22:11:22.788421Z","shell.execute_reply.started":"2025-03-10T22:11:07.182431Z","shell.execute_reply":"2025-03-10T22:11:22.787647Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The evaluation API requires that you set up a server which will respond to inference requests. We have already defined the server; you just need write the predict function. When we evaluate your submission on the hidden test set the client defined in `konwinski_prize_gateway` will run in a different container with direct access to the hidden test set and hand off the data.\n\nYour code will always have access to the published copies of the files.","metadata":{"papermill":{"duration":0.002032,"end_time":"2024-12-11T03:22:08.823897","exception":false,"start_time":"2024-12-11T03:22:08.821865","status":"completed"},"tags":[]}},{"cell_type":"code","source":"instance_count = None\n\ndef get_number_of_instances(num_instances: int) -> None:\n    \"\"\" The very first message from the gateway will be the total number of instances to be served.\n    You don't need to edit this function.\n    \"\"\"\n    global instance_count\n    instance_count = num_instances","metadata":{"papermill":{"duration":0.011949,"end_time":"2024-12-11T03:22:08.838279","exception":false,"start_time":"2024-12-11T03:22:08.82633","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:13:21.770383Z","iopub.execute_input":"2025-03-10T22:13:21.770917Z","iopub.status.idle":"2025-03-10T22:13:21.774521Z","shell.execute_reply.started":"2025-03-10T22:13:21.770889Z","shell.execute_reply":"2025-03-10T22:13:21.773877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"REPO_PATH = 'repo'\nREPO_CLONE_PATH = 'repo_clone'\nRESULT_PATH = 'result'\nPARQUET_PATH = 'data.parquet'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:13:23.341655Z","iopub.execute_input":"2025-03-10T22:13:23.341923Z","iopub.status.idle":"2025-03-10T22:13:23.34496Z","shell.execute_reply.started":"2025-03-10T22:13:23.341902Z","shell.execute_reply":"2025-03-10T22:13:23.344307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def clean_dirs():\n    paths_to_clean = [\n        \"./pip_packages\",\n        REPO_PATH,\n        REPO_CLONE_PATH,\n        RESULT_PATH,\n        PARQUET_PATH,\n        \"./playground\",\n        \"./embedding\",\n        \"./git.patch\",\n    ]\n    \n    for path_str in paths_to_clean:\n        path = Path(path_str)\n        try:\n            if path.exists():\n                if path.is_file():\n                    path.unlink()\n                else:\n                    shutil.rmtree(path, ignore_errors=True)\n        except Exception as e:\n            print(f\"Error while removing {path}: {e}\")\n            continue","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:13:23.549126Z","iopub.execute_input":"2025-03-10T22:13:23.549333Z","iopub.status.idle":"2025-03-10T22:13:23.553189Z","shell.execute_reply.started":"2025-03-10T22:13:23.549314Z","shell.execute_reply":"2025-03-10T22:13:23.552624Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Init LLM","metadata":{}},{"cell_type":"code","source":"TEMPERATURE = 0.8\nMIN_P = 0.01\nTOP_P = 0.9\nTOP_K = 40\n\nUSE_MEMORY = False\nFINE_GRAIN_NB_SAMPLES=2\nREPAIR_NB_SAMPLES=3\nTEST_NB_SAMPLES=3\n# TEST_SAMPLE_BATCH_SIZE=5\n\nINSTALL_TIMEOUT = 60 * 3\nTESTS_TIMEOUT = 60 * 2\nUNIT_TEST_TIMEOUT = 60 * 2\n\nSEED = 6335","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:13:41.977611Z","iopub.execute_input":"2025-03-10T22:13:41.977902Z","iopub.status.idle":"2025-03-10T22:13:41.98163Z","shell.execute_reply.started":"2025-03-10T22:13:41.977881Z","shell.execute_reply":"2025-03-10T22:13:41.980931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from agentless.util.model import LlamaCppLLM\nimport warnings\nimport os\nfrom transformers import AutoTokenizer\n\nwarnings.simplefilter('ignore')\n\nos.environ[\"CUDA_VISIBLE_DEVICES\"] = \"0,1,2,3\"\nos.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"\n\ndef get_tokenizer(model_path):\n    # kwargs = {}\n    # kwargs[\"gguf_file\"] = Path(model_path).name\n    # model_path = Path(model_path).parent\n\n    tokenizer = AutoTokenizer.from_pretrained(\n                    \"/kaggle/input/qwen2.5-coder/transformers/32b-instruct-awq/1\",\n                    # **kwargs\n                )    \n    return tokenizer\n\nllm_model_pth = '/kaggle/input/barto-qwen-coder/Qwen2.5-Coder-32B-Instruct-Q8_0.gguf'\nllm = LlamaCppLLM(llm_model_pth, get_tokenizer(llm_model_pth), port=8091, is_reasoner=False, temperature=TEMPERATURE, top_p=TOP_P, min_p=MIN_P, top_k=TOP_K)\nllm_qwq1 = LlamaCppLLM(llm_model_pth, get_tokenizer(llm_model_pth), port=8092, is_reasoner=True, temperature=TEMPERATURE, top_p=TOP_P, min_p=MIN_P, top_k=TOP_K)\nllm_qwq2 = LlamaCppLLM(llm_model_pth, get_tokenizer(llm_model_pth), port=8093, is_reasoner=True, temperature=TEMPERATURE, top_p=TOP_P, min_p=MIN_P, top_k=TOP_K)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:13:43.15054Z","iopub.execute_input":"2025-03-10T22:13:43.150838Z","iopub.status.idle":"2025-03-10T22:13:52.445122Z","shell.execute_reply.started":"2025-03-10T22:13:43.150816Z","shell.execute_reply":"2025-03-10T22:13:52.444355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def is_model_ready(llm):\n    # make a curl to the port and check if the model is ready\n    import requests\n\n    url = f\"http://localhost:{llm.port}/health\"\n    try:\n        response = requests.get(url).json()\n        if response[\"status\"] == \"ok\":\n            return True\n        else:\n            return False\n    except Exception as e:\n        return False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:13:55.353127Z","iopub.execute_input":"2025-03-10T22:13:55.35373Z","iopub.status.idle":"2025-03-10T22:13:55.35758Z","shell.execute_reply.started":"2025-03-10T22:13:55.3537Z","shell.execute_reply":"2025-03-10T22:13:55.356894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"embd_model = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:13:55.661313Z","iopub.execute_input":"2025-03-10T22:13:55.661566Z","iopub.status.idle":"2025-03-10T22:13:55.664327Z","shell.execute_reply.started":"2025-03-10T22:13:55.661545Z","shell.execute_reply":"2025-03-10T22:13:55.66365Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Helper Functions","metadata":{}},{"cell_type":"code","source":"def init_working_repo():\n    if os.path.exists(REPO_CLONE_PATH):\n        shutil.rmtree(REPO_CLONE_PATH, ignore_errors=True)\n    shutil.copytree(REPO_PATH, REPO_CLONE_PATH)\n    \n    # Remove any existing git folder\n    git_path = Path(REPO_CLONE_PATH) / \".git\"\n    if git_path.exists():\n        shutil.rmtree(git_path, ignore_errors=True)\n    \n    # Init new repo and commit everything\n    repo = Repo.init(REPO_CLONE_PATH)\n    repo.index.add('*')  # Add all files\n    repo.index.commit(\"Initial commit\")\n    return repo\n\ndef load_jsonl(sample_path):\n    # Load the JSONL file\n    samples = []\n    with open(sample_path, 'r') as f:\n        for line in f:\n            samples.append(json.loads(line))\n    return samples\n\ndef load_repair_samples(\n        result_folder, num_samples=FINE_GRAIN_NB_SAMPLES, \n        num_outputs_per_sample=REPAIR_NB_SAMPLES,\n        base_path=\"repair_sample_{sample_idx}/output_{output_idx}_processed.jsonl\"):\n    samples = []\n    for sample_idx in range(1, num_samples + 1):\n        for output_idx in range(num_outputs_per_sample):\n            file_path = base_path.format(sample_idx=sample_idx, output_idx=output_idx)\n            if os.path.exists(os.path.join(result_folder, file_path)):\n                samples.extend(load_jsonl(os.path.join(result_folder, file_path)))\n    return samples\n\ndef fix_repo_paths_in_patches(patch):\n    return patch.replace(\"repo_clone/\", \"\")\n\ndef _apply_patch(patch_path: Path, repo_path: Path, timeout: int=15, is_dry_run: bool=True) -> bool:\n    patch_path = Path(patch_path)\n    repo_path = Path(repo_path)\n    dry_run_cmd = f\"--dry-run\" if is_dry_run else \"\"\n    cmd = f\"patch --quiet {dry_run_cmd} -p1 -i {patch_path.resolve()} -d {repo_path.resolve()}\"\n    try:\n        result = subprocess.run(cmd, shell=True, check=False, text=True, capture_output=True, timeout=timeout)\n        # print(f\"stdout: {result.stdout}\\nstderr: {result.stderr}\")\n        if result.returncode == 0:\n            return True\n        else:\n            return False\n    except subprocess.CalledProcessError:\n        return False\n\ndef apply_patch(patch_str, repo_path, is_dry_run: bool=True, patch_path: str=\"./git.patch\"):\n    patch_path = Path(patch_path).resolve()\n    if patch_path.exists():\n        patch_path.unlink(missing_ok=True)\n\n    patch_path.write_text(patch_str)\n    result = _apply_patch(patch_path, repo_path, is_dry_run=is_dry_run)\n    patch_path.unlink(missing_ok=True)\n    return result\n\ndef get_valid_patches():\n    patches = list(set([x[\"model_patch\"] for x in load_repair_samples(RESULT_PATH)]))\n    patches = [fix_repo_paths_in_patches(p) for p in patches if p]\n    patches = [patch for patch in patches if apply_patch(patch, REPO_CLONE_PATH, is_dry_run=True)]\n    return patches","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:13:58.179052Z","iopub.execute_input":"2025-03-10T22:13:58.179278Z","iopub.status.idle":"2025-03-10T22:13:58.188952Z","shell.execute_reply.started":"2025-03-10T22:13:58.179259Z","shell.execute_reply":"2025-03-10T22:13:58.188245Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Regression tests functions","metadata":{}},{"cell_type":"code","source":"def exec_run_with_timeout(cmd, timeout: int|None=60, logger=None):\n    # Local variables to store the result of executing the command\n    exec_result = ''\n    exec_id = None\n    exception = None\n    timed_out = False\n    process = None\n\n    # Wrapper function to run the command\n    def run_command():\n        current_env = os.environ.copy()\n        jupyter_vars = ['MPLBACKEND', 'JPY_PARENT_PID', 'JUPYTER_CONFIG_DIR', 'JUPYTER_PATH', 'JUPYTER_DATA_DIR']\n        clean_env = {k: v for k, v in current_env.items() if k not in jupyter_vars}\n                \n        nonlocal exec_result, exec_id, exception, process\n        try:\n            process = subprocess.run(cmd, shell=True, executable=\"/bin/bash\", check=False, text=True, capture_output=True, timeout=timeout, env=clean_env)\n            exec_result = process.stdout\n            logger.info(f\"stdout: {process.stdout}\\nstderr: {process.stderr}\")\n        except Exception as e:\n            logger.info(f\"Error executing command: {e}\")\n            try:\n                logger.info(f\"stdout: {process.stdout}\\nstderr: {process.stderr}\")\n            except:\n                pass\n            exception = e\n\n    # Start the command in a separate thread\n    thread = threading.Thread(target=run_command)\n    start_time = time.time()\n    thread.start()\n    thread.join(timeout)\n\n    if exception:\n        print(exception)\n        return None, True, time.time() - start_time\n\n    # If the thread is still alive, the command timed out\n    if thread.is_alive():\n        # what do I do here for subprocess.run ?\n        # I need to kill the process\n        # if process:\n        #     process.kill()\n        timed_out = True\n    end_time = time.time()\n    return exec_result, timed_out, end_time - start_time\n\ndef load_regression_samples(\n        result_folder,\n        num_test_samples=TEST_NB_SAMPLES,\n        base_path=\"reproduction_test_samples/output_{output_idx}_processed_reproduction_test.jsonl\"):\n    samples = []\n    for output_idx in range(num_test_samples):\n        file_path = base_path.format(output_idx=output_idx)\n        if os.path.exists(os.path.join(result_folder, file_path)):\n            samples.extend(load_jsonl(os.path.join(result_folder, file_path)))\n    return samples\n\ndef get_valid_regression_tests():\n    patches = list(set([x[\"test_patch\"] for x in load_regression_samples(RESULT_PATH, base_path=\"reproduction_test_samples/output_{output_idx}_processed_reproduction_test.jsonl\")] + [x[\"test_patch\"] for x in load_regression_samples(RESULT_PATH, base_path=\"reproduction_test_samples_qwq1/output_{output_idx}_processed_reproduction_test.jsonl\")] + [x[\"test_patch\"] for x in load_regression_samples(RESULT_PATH, base_path=\"reproduction_test_samples_qwq2/output_{output_idx}_processed_reproduction_test.jsonl\")]))\n    patches = [fix_repo_paths_in_patches(p) for p in patches if p]\n    patches = [patch for patch in patches if apply_patch(patch, REPO_CLONE_PATH, is_dry_run=True)]\n    return patches\n    \ndef filter_regression_tests(regression_tests, repo_dir, expected_output, unexpected_outputs, logger):\n    start_time = time.time()\n    outs = []\n    for i, test in enumerate(regression_tests):\n        elapsed_time = time.time() - start_time\n        timeout = max(0, TESTS_TIMEOUT - elapsed_time)\n        if timeout <= 0:\n            break\n        \n        bug_path = os.path.join(repo_dir, 'reproduce_bug.py')\n\n        # Remove the reproduce_bug.py file\n        if os.path.exists(bug_path):\n            os.remove(bug_path)\n            \n        res = apply_patch(test, repo_dir, is_dry_run=False)\n        if not res:\n            continue\n        cmd = f'cd {repo_dir} && . .venv/bin/activate && python reproduce_bug.py'\n        logger.info(f\"Testing regression test {i}:\")\n        result, timed_out, duration = exec_run_with_timeout(cmd, timeout=timeout, logger=logger)\n        \n        # Remove the reproduce_bug.py file\n        if os.path.exists(bug_path):\n            os.remove(bug_path)\n            \n        # print(i, result, timed_out, duration)\n        failed = False\n        if not timed_out:\n            for unexpected_output in unexpected_outputs:\n                if unexpected_output.lower() in result.lower():\n                    failed = True\n                    break\n            if not failed and expected_output.lower() in result.lower():\n                outs.append(test)\n    return outs\n\ndef get_regression_tests_that_reproduced_issue(logger):\n    tests = get_valid_regression_tests()\n    if not tests:\n        return []\n    logger.info(f\"Filtering regression tests that reproduced issue\")\n    tests = filter_regression_tests(tests, REPO_CLONE_PATH, expected_output=\"Issue reproduced\", unexpected_outputs=[\"Issue resolved\", \"Other issues\"], logger=logger)\n    return tests\n\ndef select_patch_with_most_regression_tests_fixed(repo, patch_candidates, regression_tests, src_dir_with_installed_packages=REPO_CLONE_PATH, logger=None):\n    fail_to_pass_patches = []\n\n    for i, patch in enumerate(patch_candidates):\n        try:\n            # More thorough cleanup before testing each patch\n            repo.head.reset(index=True, working_tree=True)\n            repo.index.checkout(force=True)\n            # repo.git.clean('-fd')  # Remove untracked files and directories\n\n            # Apply repair patch\n            res = apply_patch(patch, src_dir_with_installed_packages, is_dry_run=False)\n            if not res:\n                print(f\"Failed to apply patch {i}\")\n                continue\n            \n            # Run regression tests\n            logger.info(f\"Testing F2P for patch {i}\")\n            tests = filter_regression_tests(\n                regression_tests, \n                src_dir_with_installed_packages, \n                expected_output=\"Issue resolved\", \n                unexpected_outputs=[\"Issue reproduced\", \"Other issues\"],\n                logger=logger\n            )\n\n            if len(tests) > 0:\n                print(f\"Patch {i} fixed the issue for {len(tests)} regression tests!!\")\n                fail_to_pass_patches.append((patch, len(tests)))\n\n        except Exception as e:\n            print(f\"Error processing patch {i}: {str(e)}\")\n            continue\n            \n        finally:\n            # Always ensure cleanup, even if an error occurred\n            try:\n                repo.head.reset(index=True, working_tree=True)\n                repo.index.checkout(force=True)\n                # repo.git.clean('-fd')\n            except Exception as e:\n                print(f\"Error during cleanup after patch {i}: {str(e)}\")\n\n    # Select the patches sorted by number of fixed tests\n    if fail_to_pass_patches:\n        fail_to_pass_patches = sorted(fail_to_pass_patches, key=lambda x: x[1], reverse=True)\n        return [x[0] for x in fail_to_pass_patches]\n    return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:38:59.461044Z","iopub.execute_input":"2025-03-10T22:38:59.461331Z","iopub.status.idle":"2025-03-10T22:38:59.478434Z","shell.execute_reply.started":"2025-03-10T22:38:59.461308Z","shell.execute_reply":"2025-03-10T22:38:59.477784Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Installer Manager","metadata":{}},{"cell_type":"code","source":"class Installer:\n    def __init__(self, pip_packages_archive, env_setup_cmds_templates, timeout,\n               repo_path=REPO_CLONE_PATH, tmp_pip_packages_path='pip_packages',\n               pip_archive_dir= '/tmp/pip_packages_archive.tar', logger=None):\n        self.pip_packages_archive = pip_packages_archive\n        self.env_setup_cmds_templates = env_setup_cmds_templates\n        self.timeout = timeout\n        self.repo_path = repo_path\n        self.tmp_pip_packages_path = tmp_pip_packages_path\n        self.pip_archive_dir = pip_archive_dir\n        self.thread = threading.Thread(target=self.run)\n        self.start_time = None\n        self.logger = logger\n\n    def start(self):\n        self.start_time = time.time()\n        self.thread.start()\n\n    def join(self):\n        elapsed_time = time.time() - self.start_time\n        self.thread.join(max(0, self.timeout - elapsed_time))\n        if self.thread.is_alive():\n            return False\n        return True\n    \n    @staticmethod\n    def sanitize_env_commands(env_setup_cmds):\n        def add_options(cmd, opt):\n            if opt not in cmd:\n                return f\" {opt} \"\n            return \"\"\n\n        out = []\n        for cmd in env_setup_cmds:\n            out_inner = []\n            for c in cmd.split(\"&&\"):\n                new_c = c\n                if \"uv venv\" in c:\n                    to_replace = \"uv venv\"\n                    to_replace = to_replace + add_options(c, \"--no-python-downloads\")\n                    to_replace = to_replace + add_options(c, \"--no-index\")\n                    new_c = new_c.replace(\"uv venv\", to_replace)\n                elif \"uv pip install\" in c:\n                    to_replace = \"uv pip install\"\n                    to_replace = to_replace + add_options(c, \"--no-index\")\n                    to_replace = to_replace + add_options(c, \"--find-links={pip_packages_path}\")\n                    new_c = new_c.replace(\"uv pip install\", to_replace)\n                out_inner.append(new_c)\n            out.append(\" && \".join(out_inner))\n\n        uv_install_cmd = \"uv pip install uv --no-index --find-links=/kaggle/input/uv-whl\"\n        uv_install_dev = \"uv pip install pylint pytest black flake8 isort --no-index --find-links=/kaggle/input/dev-wheels/wheels\"\n        out = out[:-1] + [uv_install_cmd, uv_install_dev] + out[-1:]\n        return out    \n    \n    def run(self):\n        repo_path = str(Path(self.repo_path).resolve())\n        tmp_pip_packages_path = str(Path(self.tmp_pip_packages_path).resolve())\n\n        # Extract the pip packages\n        if self.pip_packages_archive is not None:\n            if os.path.exists(self.pip_archive_dir):\n                os.remove(self.pip_archive_dir)\n                \n            with open(self.pip_archive_dir, 'wb') as f:\n                f.write(self.pip_packages_archive.read())\n        \n            if os.path.exists(tmp_pip_packages_path):\n                shutil.rmtree(tmp_pip_packages_path, ignore_errors=True)\n            shutil.unpack_archive(self.pip_archive_dir, extract_dir=tmp_pip_packages_path)\n            os.remove(self.pip_archive_dir)\n\n        # Get env setup cmds by setting the pip_packages_path\n        env_setup_cmds_templates = Installer.sanitize_env_commands(self.env_setup_cmds_templates)\n        prefix_cmd = [] # [\"export PYTHONNOUSERSITE=1\", 'export PYTHONPATH=\"\"'] # avoid conflict by looking to other PYTHONPATHs\n        env_setup_cmds = [cmd.format(pip_packages_path=tmp_pip_packages_path) for cmd in prefix_cmd + env_setup_cmds_templates]\n        print(env_setup_cmds)\n        # Run env setup for the repo\n        try:\n            start_time = time.time()\n            result = subprocess.run(\n                \"\\n\".join(env_setup_cmds),\n                shell=True,\n                executable=\"/bin/bash\",\n                cwd=repo_path,\n                capture_output=True,\n                text=True,\n                timeout=self.timeout,\n                check=True\n            )\n            print(f\"Finshed installing in {(time.time() - start_time) / 60}min\")\n            self.logger.info(f\"Finshed installing in {(time.time() - start_time) / 60}min\")\n            self.logger.info(result.stdout)\n            self.logger.info(result.stderr)\n\n        except Exception as e:\n            try:\n                self.logger.info(f\"Error installing packages: {e}\")\n                self.logger.info(result.stdout)\n                self.logger.info(result.stderr)            \n            except:\n                pass\n            print(e)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:14:14.004139Z","iopub.execute_input":"2025-03-10T22:14:14.004548Z","iopub.status.idle":"2025-03-10T22:14:14.017548Z","shell.execute_reply.started":"2025-03-10T22:14:14.004509Z","shell.execute_reply":"2025-03-10T22:14:14.016806Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Unit tests","metadata":{}},{"cell_type":"code","source":"class UnitTestRunner:\n    def __init__(self, repo_path, file_paths, patch=None, timeout=60 * 2, verbose=2, logger=None):\n        self.repo_path = Path(repo_path).resolve()\n        self.file_paths = file_paths\n        self.timeout = timeout\n        self.verbose = verbose\n        self.patch = patch\n        self.queue = Queue()\n        self.thread = threading.Thread(target=self.run)\n        self.start_time = None\n        self.logger = logger\n    def start(self):\n        self.thread.start()\n        self.start_time = time.time()\n\n    def join(self):\n        elapsed_time = time.time() - self.start_time\n        self.thread.join(max(0, self.timeout - elapsed_time))\n        if self.thread.is_alive():\n            return None\n        return self.queue.get()\n\n    def run(self):\n        # Apply the patch\n        if self.patch:\n            apply_patch(self.patch, str(self.repo_path), is_dry_run=False)\n\n        all_total, all_failures, all_errors, all_skipped, all_passed = 0, 0, 0, 0, 0\n        for filepath in self.file_paths:\n            filepath = Path(filepath).resolve()\n            self.logger.info(f\"Running unit tests for {filepath}\")\n            print(f\"Running unit tests for {filepath}\")\n\n            # Create a temporary file for the junit XML output\n            junit_path = (self.repo_path / 'junit.xml').resolve()\n            if junit_path.exists():\n                junit_path.unlink()\n    \n            current_env = os.environ.copy()\n            jupyter_vars = ['MPLBACKEND', 'JPY_PARENT_PID', 'JUPYTER_CONFIG_DIR', 'JUPYTER_PATH', 'JUPYTER_DATA_DIR']\n            clean_env = {k: v for k, v in current_env.items() if k not in jupyter_vars}\n            \n            cmd = f\". .venv/bin/activate && timeout -k 5 {self.timeout}s pytest -rA --tb=no --junitxml={str(junit_path)} {' '.join(self.file_paths)}\"\n    \n            result = subprocess.run(\n                cmd,\n                shell=True,\n                executable=\"/bin/bash\",\n                cwd=str(self.repo_path),\n                capture_output=True,\n                text=True,\n                timeout=self.timeout,\n                env=clean_env\n            )\n            if self.verbose > 1:\n                print(result.stdout)\n                print(result.stderr)\n    \n            # Parse the junit XML to get test counts\n            if junit_path.exists():\n                tree = ET.parse(junit_path)\n                root = tree.getroot()\n                testsuite = root.find('testsuite')\n                if testsuite is not None:\n                    total = int(testsuite.get('tests', 0))\n                    failures = int(testsuite.get('failures', 0))\n                    errors = int(testsuite.get('errors', 0))\n                    skipped = int(testsuite.get('skipped', 0))\n                    passed = total - failures - errors - skipped\n                    \n                    self.logger.info(f\"Test Results:\")\n                    self.logger.info(f\"Passed: {passed}\")\n                    self.logger.info(f\"Failed: {failures}\")\n                    self.logger.info(f\"Errors: {errors}\")\n                    self.logger.info(f\"Skipped: {skipped}\")\n                    self.logger.info(f\"Total: {total}\")\n                    if self.verbose:\n                        print(f\"Test Results:\")\n                        print(f\"Passed: {passed}\")\n                        print(f\"Failed: {failures}\")\n                        print(f\"Errors: {errors}\")\n                        print(f\"Skipped: {skipped}\")\n                        print(f\"Total: {total}\")\n                        \n                    all_total = all_total + total\n                    all_failures = all_failures + failures\n                    all_errors = all_errors + errors\n                    all_skipped = all_skipped + skipped\n                    all_passed = all_passed + passed    \n                # Clean up the temporary file\n                junit_path.unlink()\n        if self.verbose:\n            print(\"All unit tests ran, accumulated results:\")\n            print(f\"Passed: {all_passed}\")\n            print(f\"Failed: {all_failures}\")\n            print(f\"Errors: {all_errors}\")\n            print(f\"Skipped: {all_skipped}\")\n            print(f\"Total: {all_total}\")            \n        self.queue.put({\"total\": all_total, \"failures\": all_failures, \"errors\": all_errors, \"skipped\": all_skipped, \"passed\": all_passed})\n\ndef filter_patches_with_p2p_tests(repo, patches, p2p_tests, ref_test_results, logger):\n    \"\"\"\n    Validate using unit tests before and after patching. Return True if the patch is valid.\n    \"\"\"\n    for i, patch in enumerate(patches):\n        # Reset the repo to the original state\n        repo.head.reset(index=True, working_tree=True)\n        repo.index.checkout(force=True)\n        \n        try:\n            logger.info(f\"Testing unit for patch {i}\")\n            tester = UnitTestRunner(repo_path=REPO_CLONE_PATH, file_paths=p2p_tests, patch=patch, timeout=UNIT_TEST_TIMEOUT, verbose=0, logger=logger)\n            tester.start()\n            result = tester.join()\n        except Exception as e:\n            print(f\"Error in UnitTestRunner: {e}\")\n            continue\n        finally:\n            # Reset the repo to the original state\n            repo.head.reset(index=True, working_tree=True)\n            repo.index.checkout(force=True)\n\n        if result[\"total\"] == 0:\n            continue\n            \n        if result[\"failures\"] > ref_test_results[\"failures\"] or result[\"errors\"] > ref_test_results[\"errors\"] or result[\"passed\"] < ref_test_results[\"passed\"]:\n            continue\n        return patch\n    return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:14:16.874695Z","iopub.execute_input":"2025-03-10T22:14:16.87501Z","iopub.status.idle":"2025-03-10T22:14:16.88928Z","shell.execute_reply.started":"2025-03-10T22:14:16.874985Z","shell.execute_reply":"2025-03-10T22:14:16.888458Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Prediction functions","metadata":{}},{"cell_type":"code","source":"def init_logger(instance_i):\n    # Create subfolders\n    os.makedirs(\"result\", exist_ok=True)\n    os.makedirs(\"result/installer/\", exist_ok=True)\n    os.makedirs(\"result/test/\", exist_ok=True)\n    os.makedirs(\"result/unit/\", exist_ok=True)\n    os.makedirs(\"result/p2p/\", exist_ok=True)\n        \n    installer_logger = setup_logger(f\"result/installer/{instance_i - 1}.log\")\n    test_logger = setup_logger(f\"result/test/{instance_i - 1}.log\")\n    unit_logger = setup_logger(f\"result/unit/{instance_i - 1}.log\")\n    return installer_logger, test_logger, unit_logger\n\n\ndef localize_for_tests(llm, structure, p2p_tests):\n    test_related_dir = os.path.join(RESULT_PATH, \"related_tests\")\n    os.makedirs(test_related_dir, exist_ok=True)\n    os.makedirs(os.path.join(test_related_dir, \"localization_logs\"), exist_ok=True)\n    rel_args = argparse.Namespace(\n        related_level=True,\n        output_folder=test_related_dir,\n        top_n=3,\n        compress_assign=True,\n        compress=True,\n        start_file=None,\n        num_threads=1,\n        skip_existing=True,\n        backend=\"local\",\n        model=llm_model_pth,\n        is_test=True,\n        # defaults\n        output_file=os.path.join(test_related_dir, \"loc_outputs.jsonl\"),\n        file_level=False,\n        fine_grain_line_level=False,\n        make_summary=False,\n        temperature=0.0,\n        min_p=MIN_P,\n        top_p=TOP_P,\n        num_samples=1,\n        compress_assign_total_lines=30,\n        compress_assign_prefix_lines=10,\n        compress_assign_suffix_lines=10,\n        merge=False,\n        add_space=False,\n        no_line_number=False,\n        sticky_scroll=False,\n        related_level_separate_file=False,\n        context_window=10,\n        keep_old_order=False,\n        irrelevant=False,\n        direct_edit_loc=False,\n        target_id=None,\n        mock=False,\n        dataset=\"princeton-nlp/SWE-bench_Lite\",\n    )\n    return localize(rel_args, llm=llm, structure=structure, start_file_locs=[p2p_tests])[0]\n\ndef check_sample_time_limit(sample_start_time):\n    if time.time() - sample_start_time > MAX_ALLOWED_TIME_PER_SAMPLE:\n        print(\"Sample time limit exceeded, skipping...\")\n        return True\n    return False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:14:19.024758Z","iopub.execute_input":"2025-03-10T22:14:19.025007Z","iopub.status.idle":"2025-03-10T22:14:19.031759Z","shell.execute_reply.started":"2025-03-10T22:14:19.024987Z","shell.execute_reply":"2025-03-10T22:14:19.03103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TestCreatorRunner:\n    def __init__(self, llm_model, nb_samples, structure, related_content, result_dir_name, skip_greedy, timeout=3 * 60):\n        self.llm_model = llm_model\n        self.nb_samples = nb_samples\n        self.structure = structure\n        self.related_content = related_content\n        self.result_dir_name = result_dir_name\n        self.skip_greedy = skip_greedy\n        \n        self.thread = threading.Thread(target=self.run)\n        self.start_time = None\n        self.timeout = timeout\n        \n    def start(self):\n        self.thread.start()\n        self.start_time = time.time()\n\n    def join(self):\n        elapsed_time = time.time() - self.start_time\n        self.thread.join(max(0, self.timeout - elapsed_time))\n        if self.thread.is_alive():\n            return None\n\n    def run(self):\n        args = argparse.Namespace(model_path=llm_model_pth, output_base=RESULT_PATH, temperature=TEMPERATURE, min_p=MIN_P, top_p=TOP_P)\n        main.generate_tests(args, self.llm_model, self.structure, self.related_content, test_nb_samples=self.nb_samples, result_dir_name=self.result_dir_name, skip_greedy=self.skip_greedy)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:18:52.652405Z","iopub.execute_input":"2025-03-10T22:18:52.652767Z","iopub.status.idle":"2025-03-10T22:18:52.658895Z","shell.execute_reply.started":"2025-03-10T22:18:52.652738Z","shell.execute_reply":"2025-03-10T22:18:52.658235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"skip_prediction = False\ninstance_i = 0\n\ndef predict_wrapped(problem_statement: str, repo_archive: io.BytesIO, pip_packages_archive: io.BytesIO, env_setup_cmds_templates: list[str]) -> str:\n    \"\"\" Replace this function with your inference code.\n    Args:\n        problem_statement: The text of the git issue.\n        repo_archive: A BytesIO buffer path with a .tar containing the codebase that must be patched. The gateway will make this directory available immediately before this function runs.\n    \"\"\"\n    print(\"predict: \", problem_statement[:100])\n    sample_start_time = time.time()\n    global instance_i\n    instance_i += 1\n\n    if instance_i == 1:\n        while not is_model_ready(llm):\n            print(\"Waiting for model to be ready...\")\n            time.sleep(10)\n        while not is_model_ready(llm_qwq1):\n            print(\"Waiting for model to be ready...\")\n            time.sleep(10)\n        while not is_model_ready(llm_qwq2):\n            print(\"Waiting for model to be ready...\")\n            time.sleep(10)            \n            \n    # if \"add ModelAdmin.get_inlines() hook to allow\" not in problem_statement:\n    # if \"QTable cannot take\" not in problem_statement:\n    # if \"Migration optimizer does not reduce multiple AlterField\" not in problem_statement:\n    # if \"BUG:\" not in problem_statement:\n        # return None\n    \n    if skip_prediction:\n        return None\n\n    if time.time() - NOTEBOOK_START_TIME > NOTEBOOK_MAX_ALLOWED_TIME:\n        print(\"Global time limit exceeded, skipping...\")\n        return None\n        \n    clean_dirs()\n\n    installer_logger, test_logger, unit_logger = init_logger(instance_i)\n\n    with open('repo_archive.tar', 'wb') as f:\n        f.write(repo_archive.read())\n    if os.path.exists(REPO_PATH):\n        shutil.rmtree(REPO_PATH, ignore_errors=True)\n    shutil.unpack_archive('repo_archive.tar', extract_dir=REPO_PATH)\n    os.remove('repo_archive.tar')\n\n    ########## Clone the repo\n    repo = init_working_repo()\n\n    ######### Create the dummy dataset entry (required by `process_sample` and `generate_tests`)\n    data = pd.DataFrame({\"instance_id\": [instance_i - 1], \"repo\": [REPO_CLONE_PATH], \"problem_statement\": [problem_statement], \"base_commit\": [\"dummy\"]})\n    data.to_parquet(PARQUET_PATH)   \n    \n    ######### Install the packages\n    installer = Installer(pip_packages_archive, env_setup_cmds_templates, timeout=INSTALL_TIMEOUT, repo_path=REPO_CLONE_PATH, logger=installer_logger)\n    installer.start()\n    print(\"Started installing packages\")\n        \n    # Get reference unit tests\n    structure = get_repo_structure([instance_i - 1], REPO_CLONE_PATH, \"dummy_commit\", \"structure_playground\")\n\n    # Start QwQ before related_content\n    qwq1_thread = TestCreatorRunner(llm_qwq1, 2, structure, related_content=None, result_dir_name=\"reproduction_test_samples_qwq1\", skip_greedy=True)\n    qwq1_thread.start()\n    qwq2_thread = TestCreatorRunner(llm_qwq2, 2, structure, related_content=None, result_dir_name=\"reproduction_test_samples_qwq2\", skip_greedy=True)\n    qwq2_thread.start()\n    \n    p2p_tests = get_test_files(instance_id=instance_i - 1, llm=llm, problem_statement=problem_statement, structure=structure, logger_file=f\"result/p2p/{instance_i - 1}.log\")\n    if not p2p_tests:\n        print(\"No pass to pass tests found, skipping...\")\n        return None\n    \n    # Get related files for tests\n    related_content = localize_for_tests(llm, structure, p2p_tests)\n    if not related_content or not related_content.get(\"found_related_locs\", []):\n        print(\"No related content found for tests\")\n        related_content = None\n\n    ######### Create F2P tests\n    # args = argparse.Namespace(model_path=llm_model_pth, output_base=RESULT_PATH, temperature=TEMPERATURE, min_p=MIN_P, top_p=TOP_P)\n    # main.generate_tests(args, llm, structure, related_content, test_nb_samples=TEST_SAMPLE_BATCH_SIZE)\n    coder_thread = TestCreatorRunner(llm, TEST_NB_SAMPLES, structure, related_content=related_content, result_dir_name=\"reproduction_test_samples\", skip_greedy=False)\n    coder_thread.start()\n\n    installer.join()\n    qwq1_thread.join()\n    qwq2_thread.join()\n    coder_thread.join()\n\n    regression_tests = get_regression_tests_that_reproduced_issue(logger=test_logger)\n    if not regression_tests:\n        print(\"No valid regression tests found, skipping...\")\n        return None\n         \n    # if not regression_tests:\n    #     print(\"No valid regression tests found, retrying...\")\n    #     nb_retries = (TEST_NB_SAMPLES // TEST_SAMPLE_BATCH_SIZE) - 1\n    #     should_return_none = True\n    #     for retry_i in range(nb_retries):\n    #         retry_temp = TEMPERATURE + (1 + retry_i) * 0.1\n    #         print(f\"Generating tests for the {retry_i + 1} try with temperature {retry_temp}\")\n    #         args = argparse.Namespace(model_path=llm_model_pth, output_base=RESULT_PATH, temperature=retry_temp, min_p=MIN_P, top_p=TOP_P)\n    #         main.generate_tests(args, llm, structure, related_content, test_nb_samples=TEST_SAMPLE_BATCH_SIZE + 1, skip_greedy=True)\n    #         regression_tests = get_regression_tests_that_reproduced_issue(logger=test_logger)\n    #         if regression_tests:\n    #             should_return_none = False\n    #             break\n    #     if should_return_none:\n    #         print(\"No valid regression tests found, skipping...\")\n    #         return None\n\n    if check_sample_time_limit(sample_start_time):\n        return None\n        \n    ######### Get pass to pass tests and run\n    unit_logger.info(\"Getting reference unit tests\")\n    tester = UnitTestRunner(repo_path=REPO_CLONE_PATH, file_paths=[x.replace(\"repo_clone/\", \"\") for x in p2p_tests[\"found_files\"]], timeout=UNIT_TEST_TIMEOUT, verbose=0, logger=unit_logger)\n    tester.start()\n    \n    ######### Process the sample to generate repair patches\n    args = argparse.Namespace(model_path=llm_model_pth, output_base=RESULT_PATH, temperature=TEMPERATURE, min_p=MIN_P, top_p=TOP_P)\n    main.process_sample(args, llm, embd_model, fine_grain_nb_samples=FINE_GRAIN_NB_SAMPLES, repair_nb_samples=REPAIR_NB_SAMPLES, structure=structure, use_memory=USE_MEMORY)\n    patches = get_valid_patches()\n    ref_test_results = tester.join()\n\n    if not patches:\n        print(\"No valid patches found\")\n        return None\n    print(f\"Found {len(patches)} valid patches, testing now...\")\n\n    ######### Select the patch with the most regression tests fixed\n    patches = select_patch_with_most_regression_tests_fixed(repo, patches, regression_tests, logger=test_logger)\n    if not patches:\n        print(\"Patch rejected in F2P tests, skipping...\")\n        return None\n    \n    if ref_test_results[\"passed\"] > 0:\n        patch = filter_patches_with_p2p_tests(repo, patches, [x.replace(\"repo_clone/\", \"\") for x in p2p_tests[\"found_files\"]], ref_test_results, logger=unit_logger)\n        if not patch:\n            print(\"Patch rejected in P2P tests, skipping...\")\n            return None\n    else:\n        patch = patches[0]\n\n    print(\"Finished processing everything, returning patches...\")\n    print(patch)\n    return patch","metadata":{"papermill":{"duration":0.011382,"end_time":"2024-12-11T03:22:08.852112","exception":false,"start_time":"2024-12-11T03:22:08.84073","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:18:54.751175Z","iopub.execute_input":"2025-03-10T22:18:54.75141Z","iopub.status.idle":"2025-03-10T22:18:54.765339Z","shell.execute_reply.started":"2025-03-10T22:18:54.75139Z","shell.execute_reply":"2025-03-10T22:18:54.764634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(problem_statement: str, repo_archive: io.BytesIO, pip_packages_archive: io.BytesIO, env_setup_cmds_templates: list[str]) -> str:\n    global skip_prediction\n    patch_string = None\n    start_time = time.time()\n    try:\n        patch_string = predict_wrapped(problem_statement, repo_archive, pip_packages_archive, env_setup_cmds_templates)\n    except Exception as err:\n        print(f\"Error in predict_wrapped: {err}\")\n    finally:\n        # Always attempt to clean up, even if prediction fails\n        print(f\"Ran sample in {round((time.time() - start_time) / 60, 2)} minutes\")\n        clean_dirs()\n        if not os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n            skip_prediction = True\n    return patch_string","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:18:57.619964Z","iopub.execute_input":"2025-03-10T22:18:57.620252Z","iopub.status.idle":"2025-03-10T22:18:57.624691Z","shell.execute_reply.started":"2025-03-10T22:18:57.620229Z","shell.execute_reply":"2025-03-10T22:18:57.624Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"When your notebook is run on the hidden test set, inference_server.serve must be called within 15 minutes of the notebook starting or the gateway will throw an error. If you need more than 15 minutes to load your model you can do so during the very first predict call, which does not have the usual 30 minute response deadline.","metadata":{"papermill":{"duration":0.001889,"end_time":"2024-12-11T03:22:08.856283","exception":false,"start_time":"2024-12-11T03:22:08.854394","status":"completed"},"tags":[]}},{"cell_type":"code","source":"skip_prediction = False\n\ninference_server = kaggle_evaluation.konwinski_prize_inference_server.KPrizeInferenceServer(\n    get_number_of_instances,   \n    predict\n)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            '/kaggle/input/konwinski-prize/',  # Path to the entire competition dataset\n            '/kaggle/tmp/konwinski-prize/',   # Path to a scratch directory for unpacking data.a_zip.\n        ),\n        use_concurrency=True,  # This can safely be disabled for purposes of local testing if necessary.\n    )","metadata":{"papermill":{"duration":3.790202,"end_time":"2024-12-11T03:22:12.648591","exception":false,"start_time":"2024-12-11T03:22:08.858389","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:18:59.251406Z","iopub.execute_input":"2025-03-10T22:18:59.251712Z","iopub.status.idle":"2025-03-10T22:28:51.169674Z","shell.execute_reply.started":"2025-03-10T22:18:59.25169Z","shell.execute_reply":"2025-03-10T22:28:51.167933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !zip result_kaggle.zip -r result","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T21:49:59.393632Z","iopub.execute_input":"2025-03-10T21:49:59.393858Z","iopub.status.idle":"2025-03-10T21:49:59.396494Z","shell.execute_reply.started":"2025-03-10T21:49:59.393838Z","shell.execute_reply":"2025-03-10T21:49:59.395833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stop_llama_server(server_process0)\nstop_llama_server(server_process1)\nstop_llama_server(server_process2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:39:26.733827Z","iopub.execute_input":"2025-03-10T22:39:26.734104Z","iopub.status.idle":"2025-03-10T22:39:26.738174Z","shell.execute_reply.started":"2025-03-10T22:39:26.734081Z","shell.execute_reply":"2025-03-10T22:39:26.737526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clean everything except submission.csv\nfiles = os.listdir(\"/kaggle/working/\")\nfor path_str in files:\n    if path_str.endswith(\"submission.csv\"):\n        continue\n    path = Path(\"/kaggle/working/\") / path_str\n    try:\n        if path.exists():\n            if path.is_file():\n                path.unlink()\n            else:\n                shutil.rmtree(path, ignore_errors=True)\n    except Exception as e:\n        print(f\"Error while removing {path}: {e}\")\n        continue","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T21:24:41.738661Z","iopub.execute_input":"2025-03-10T21:24:41.738852Z","iopub.status.idle":"2025-03-10T21:24:43.899203Z","shell.execute_reply.started":"2025-03-10T21:24:41.738834Z","shell.execute_reply":"2025-03-10T21:24:43.898516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}