{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaL4","dataSources":[{"sourceId":84795,"databundleVersionId":11281725,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":10912201,"sourceType":"datasetVersion","datasetId":6783329},{"sourceId":10912204,"sourceType":"datasetVersion","datasetId":6783331},{"sourceId":10912207,"sourceType":"datasetVersion","datasetId":6783333},{"sourceId":10912208,"sourceType":"datasetVersion","datasetId":6783334},{"sourceId":221548458,"sourceType":"kernelVersion"},{"sourceId":225599010,"sourceType":"kernelVersion"},{"sourceId":162952,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":138579,"modelId":161088}],"dockerImageVersionId":30919,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import io\nimport os\nimport shutil\nimport subprocess\n\nimport pandas as pd\nimport polars as pl\n\nimport kaggle_evaluation.konwinski_prize_inference_server","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-04T02:09:16.763303Z","iopub.execute_input":"2025-03-04T02:09:16.763581Z","iopub.status.idle":"2025-03-04T02:09:30.754694Z","shell.execute_reply.started":"2025-03-04T02:09:16.763559Z","shell.execute_reply":"2025-03-04T02:09:30.754015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import Counter\nimport os\nimport copy\nimport time\nfrom typing import List\nfrom agent2.agent.agent import Agent\nfrom agent2.agent.tool import Tool\nfrom agent2.file import File\nfrom agent2.tools_common.element_tools.element_viewing import view_element, search_elements, view_file\nfrom agent2.tools_common.element_tools.element_editing import replace_element, replace_element_with, open_element\nfrom agent2.utils.utils import load_project_files, get_completion, get_rating_keys\nfrom agent2.utils.agent_utils import load_agent_from_json\nimport ast","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T02:09:30.755598Z","iopub.execute_input":"2025-03-04T02:09:30.756201Z","iopub.status.idle":"2025-03-04T02:10:30.508718Z","shell.execute_reply.started":"2025-03-04T02:09:30.756178Z","shell.execute_reply":"2025-03-04T02:10:30.508008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"instance_count = None\n\ndef get_number_of_instances(num_instances: int) -> None:\n    \"\"\" The very first message from the gateway will be the total number of instances to be served.\n    You don't need to edit this function.\n    \"\"\"\n    global instance_count\n    instance_count = num_instances","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T02:10:30.509988Z","iopub.execute_input":"2025-03-04T02:10:30.510542Z","iopub.status.idle":"2025-03-04T02:10:30.513482Z","shell.execute_reply.started":"2025-03-04T02:10:30.510519Z","shell.execute_reply":"2025-03-04T02:10:30.512923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lmdeploy import pipeline, GenerationConfig, TurbomindEngineConfig\nfrom lmdeploy.cli.utils import get_chat_template\n\nprint(\"Begin loading...\")\nbackend_config = TurbomindEngineConfig(tp=4, enable_prefix_caching=True, cache_max_entry_count = 0.6)\nllm = pipeline('/kaggle/input/qwen2.5-coder/transformers/32b-instruct-awq/1',\n                backend_config=backend_config)\nprint(\"Finished loading!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T02:10:30.514409Z","iopub.execute_input":"2025-03-04T02:10:30.514605Z","iopub.status.idle":"2025-03-04T02:13:18.137089Z","shell.execute_reply.started":"2025-03-04T02:10:30.514587Z","shell.execute_reply":"2025-03-04T02:13:18.13633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gen_config_coder = GenerationConfig(do_sample=True,\n                              min_p=0.1,\n                              temperature=0.8,\n                              max_new_tokens=3000)\ncoding_start_temp = 0.8\ncoding_temp_drop = 0.2\ncoding_end_temp = 0.2     # Temperature decreases over time, this means different runs diverge from one another but still make use of a low temperature","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T02:13:18.137734Z","iopub.execute_input":"2025-03-04T02:13:18.137958Z","iopub.status.idle":"2025-03-04T02:13:18.141144Z","shell.execute_reply.started":"2025-03-04T02:13:18.137939Z","shell.execute_reply":"2025-03-04T02:13:18.140548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gen_config_rater = GenerationConfig(do_sample=True,\n                              min_p=0.15,\n                              temperature=0.8,\n                              max_new_tokens=3000)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T02:13:18.14174Z","iopub.execute_input":"2025-03-04T02:13:18.141938Z","iopub.status.idle":"2025-03-04T02:13:18.154327Z","shell.execute_reply.started":"2025-03-04T02:13:18.141921Z","shell.execute_reply":"2025-03-04T02:13:18.153758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_responses(pipeline, gen_config, oai_inputs):\n    responses = pipeline(oai_inputs, gen_config=gen_config)\n    return [response.text for response in responses]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T02:13:18.154862Z","iopub.execute_input":"2025-03-04T02:13:18.155061Z","iopub.status.idle":"2025-03-04T02:13:18.165216Z","shell.execute_reply.started":"2025-03-04T02:13:18.155027Z","shell.execute_reply":"2025-03-04T02:13:18.16462Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_n_most_common(lists, n):\n    counter = Counter()\n    for lst in lists:\n        # Since each list has no duplicates, we can safely update counts\n        for item in lst:\n            counter[item] += 1\n    # Get the n most common items, which are already unique\n    return [item for item, _ in counter.most_common(n)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T02:13:18.165875Z","iopub.execute_input":"2025-03-04T02:13:18.166084Z","iopub.status.idle":"2025-03-04T02:13:18.17766Z","shell.execute_reply.started":"2025-03-04T02:13:18.166067Z","shell.execute_reply":"2025-03-04T02:13:18.177096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"i = 0\nmax_questions = 30\nturn_limit = 10\nagent_count = 4\nrater_iterations = 4\ndef predict(problem_statement: str, repo_archive: io.BytesIO, pip_packages_archive: io.BytesIO, env_setup_cmds_templates: list[str]) -> str:\n    \"\"\" Replace this function with your inference code.\n    Args:\n        problem_statement: The text of the git issue.\n        repo_path: A BytesIO buffer path with a .tar containing the codebase that must be patched. The gateway will make this directory available immediately before this function runs.\n        pip_packages_archive: A BytesIO buffer path with a .tar containing the wheel files necessary for running unit tests.\n        env_setup_cmds_templates: Commands necessary for installing the pip_packages_archive.\n    \"\"\"\n    \n    # Unpack the codebase to be patched into a directory that won't be exported when\n    # the notebook is saved.\n    archive_path = '/tmp/repo_archive.tar'\n    with open(archive_path, 'wb') as f:\n        f.write(repo_archive.read())\n    repo_path = 'repo'\n    if os.path.exists(repo_path):\n        shutil.rmtree(repo_path)\n    shutil.unpack_archive(archive_path, extract_dir=repo_path)\n    os.remove(archive_path)\n\n    \"\"\"\n    Unpack pip_packages if you want to run unit tests on your patch.\n    Note that editing unit tests with your patch -- even to add valid tests -- can cause your submission to be flagged as a failure.\n    Most of the relevant repos use pytest for running tests. You will almost certainly need to run only a subset of the unit tests to avoid running out of inference time.\n    \"\"\"\n    pip_archive_dir = '/tmp/pip_packages_archive.tar'\n    with open(pip_archive_dir, 'wb') as f:\n        f.write(pip_packages_archive.read())\n    pip_packages_path = '/path/to/pip_packages'\n    if os.path.exists(pip_packages_path):\n        shutil.rmtree(pip_packages_path)\n    shutil.unpack_archive(pip_archive_dir, extract_dir=pip_packages_path)\n    os.remove(pip_archive_dir)\n\n    # Get env setup cmds by setting the pip_packages_path\n    env_setup_cmds = [cmd.format(pip_packages_path=pip_packages_path) for cmd in env_setup_cmds_templates]\n\n    # Run env setup for the repo\n    subprocess.run(\n        \"\\n\".join(env_setup_cmds),\n        shell=True,\n        executable=\"/bin/bash\",\n        cwd=repo_path,\n    )\n\n    #### ACTUAL BEHAVIOR\n    global i\n    global max_questions\n    i += 1\n    if i > max_questions:\n        return None\n    \n    # Initialize tools\n    tools = [\n        Tool(view_element),\n        Tool(replace_element),\n        Tool(replace_element_with),\n        Tool(search_elements),\n        Tool(view_file),\n        Tool(open_element)\n    ]\n\n    start_time = time.time()\n    print(f\"==== STARTING ISSUE {i}/{max_questions} at time {start_time} ====\")\n    print(problem_statement)\n\n    print(\"Loading files...\")\n    original_project_files = load_project_files(\"repo\")\n\n    global agent_count\n    global rater_iterations\n    total_agents = []\n    agent_chats = []\n    print(\"Loading agents...\")\n    # The tool tokens I used with mistral aren't working with qwen, likely because they are special tokens, so I change them out here\n    rater_agent = load_agent_from_json(\"/kaggle/input/rater-agent/rater_agent.json\", tools)\n    for j in range(0, agent_count):\n        new_agent = load_agent_from_json(\"/kaggle/input/codeact-agent/codeact_agent.json\", tools)\n        agent_chats += [new_agent.start(task=problem_statement, files=copy.deepcopy(original_project_files)).openai_completion]\n        total_agents += [new_agent]\n    for j in range(0, agent_count):\n        new_agent = load_agent_from_json(\"/kaggle/input/md-agent/md_agent.json\", tools)\n        agent_chats += [new_agent.start(task=problem_statement, files=copy.deepcopy(original_project_files)).openai_completion]\n        total_agents += [new_agent]\n    for j in range(0, agent_count):\n        new_agent = load_agent_from_json(\"/kaggle/input/xml-agent/xml_agent.json\", tools)\n        agent_chats += [new_agent.start(task=problem_statement, files=copy.deepcopy(original_project_files)).openai_completion]\n        total_agents += [new_agent]\n    \n    print(f\"==== DEPLOYING {len(total_agents)} AGENTS ====\")\n    global turn_limit\n    turn_counter = 0\n    current_temperature = coding_start_temp\n    while len(agent_chats) > 0 and turn_counter < turn_limit:\n        print(f\"==== TURN {turn_counter} ====\")\n        gen_config_coder.temperature = current_temperature\n        current_temperature -= coding_temp_drop\n        if current_temperature < coding_end_temp:\n            current_temperature = coding_end_temp\n        responses = get_responses(llm, gen_config_coder, agent_chats)\n        agent_counter = 0\n        new_chats = []\n        for x in total_agents:\n            if x.frozen == False:                \n                resp = x.step(responses[agent_counter])\n                if resp.done == None:\n                    print(\"Continue...\")\n                    new_chats += [resp.openai_completion]\n                else:\n                    print(\"Frozen agent...\")\n                    x.frozen = True\n                agent_counter += 1\n        if (time.time() - start_time)/60 > 20:\n            print(\"!!!!!! EMERGENCY ERROR; RAN OUT OF TIME !!!!!!\")\n            return None\n        agent_chats = new_chats\n        turn_counter += 1\n    print(f\"Time taken to generate solutions: {(time.time() - start_time)/60} minutes\")\n    print(f\"==== COLLECTING SOLUTIONS! ====\")\n    solutions = []\n    aggregate_good_references = []\n    for a in total_agents:\n        diffs = []\n        failed = False\n        for f in a.cached_state.workspace:\n            if f.original_content != f.updated_content:\n                diffs += [f.diff(None)]\n                try:\n                    ast.parse(f.updated_content)\n                except Exception:\n                    failed = True\n        if len(diffs) == 0 or failed:\n            print(\"Discarded broken solution\")\n            continue\n        else:\n            print(\"Working solution got\")\n            solutions += [\"\\n\".join(diffs)]\n            aggregate_good_references += [a.cached_state.saved_elements]\n    if len(solutions) == 0:\n        print(\"All failures...\")\n        return None\n\n    print(f\"==== RATING SOLUTIONS! ====\")\n    print(aggregate_good_references)\n    aggregated_files = get_n_most_common(aggregate_good_references, 10)\n    print(aggregated_files)\n    all_rating_chats = []\n    scores = {}\n    init_message_cache = rater_agent.init_message\n    for sol in solutions:\n        scores[sol] = 0\n        rater_agent.init_message = (init_message_cache.replace(\"{{diffs}}\", sol))\n        all_rating_chats += [rater_agent.start(task=problem_statement, files=original_project_files, copy_saved_elements=aggregated_files).openai_completion] * rater_iterations\n    \n    responses = get_responses(llm, gen_config_rater, all_rating_chats)\n\n    # Track the current position in the responses list\n    current_index = 0\n    for sol in solutions:\n        # Get the chunk of responses for this solution\n        solution_responses = responses[current_index : current_index + rater_iterations]\n        # Sum the scores for this solution\n        scores[sol] = sum(get_rating_keys(response) for response in solution_responses)\n        # Move to the next chunk\n        current_index += rater_iterations\n        for solresp in solution_responses:\n            print(solresp)\n        print(\"SCORE:\", scores[sol])\n        print(\"SOLUTION:\", sol)\n\n    highest_rated_solution = max(scores.items(), key=lambda x: x[1])[0]\n    print(f\"Time taken to finish: {(time.time() - start_time)/60} minutes\")\n    if scores[highest_rated_solution] > 0:\n        print(f\"Returning highest rated solution with score of {scores[highest_rated_solution]}...\")\n        print(highest_rated_solution)\n        return highest_rated_solution\n    else:\n        print(f\"Highest solution only got score of {scores[highest_rated_solution]}, returning nothing...\")\n        return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T02:13:18.17899Z","iopub.execute_input":"2025-03-04T02:13:18.179203Z","iopub.status.idle":"2025-03-04T02:13:18.195464Z","shell.execute_reply.started":"2025-03-04T02:13:18.179185Z","shell.execute_reply":"2025-03-04T02:13:18.194901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.konwinski_prize_inference_server.KPrizeInferenceServer(\n    get_number_of_instances,   \n    predict\n)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            '/kaggle/input/konwinski-prize/',  # Path to the entire competition dataset\n            '/kaggle/tmp/konwinski-prize/',   # Path to a scratch directory for unpacking data.a_zip.\n        ),\n        use_concurrency=True,  # This can safely be disabled for purposes of local testing if necessary.\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T02:13:18.196105Z","iopub.execute_input":"2025-03-04T02:13:18.196289Z","execution_failed":"2025-03-04T02:23:13.18Z"}},"outputs":[],"execution_count":null}]}