{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84795,"databundleVersionId":10462807,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":22.341371,"end_time":"2024-12-11T03:22:13.479076","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-12-11T03:21:51.137705","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from IPython.core.display import HTML\n\n\ndef apply_css_styles(\n    font_name: str = \"Lato\",\n    fallback_font: str = \"Verdana\",\n    css_content: str = None,\n    verbose: bool = False\n) -> HTML:\n    \"\"\"Applies custom CSS styles within a Jupyter notebook cell.\n\n    Args:\n        font_name (str, optional): \n            The primary font to use in the styles.\n        fallback_font (str, optional): \n            The fallback font to use if the primary font is unavailable.\n        css_content (str, optional): \n            Custom CSS content to use. \n            If None, default styles are used.\n        verbose (bool, optional): \n            Whether to print the generated CSS for debugging.\n\n    Returns:\n        IPython.core.display.HTML: \n            HTML object with the injected styles.\n    \"\"\"\n    try:\n        # Default CSS content if none is provided\n        default_css = '''\np, li, a, b, h1, h2, h3, h4, h5, h6, title, ul, strong, sup, sub, em, i, blockquote, label {\n    font-family: Verdana !important;\n}\n\nb, h1 {\n    font-weight: 900 !important;\n}\n\nh2, h3, h4 ul {\n    font-weight: 700 !important;\n}\n\n.fa, .far, .fas {\n    font-family: \"Font Awesome 5 Free\" !important;\n}\n'''\n\n        # Generate font import string dynamically based on the provided font name\n        font_import = (\n            f\"\\n@import url('https://fonts.googleapis.com/css2?family={font_name.replace(' ', '+')}:ital,wght@0,100;0,300;0,400;0,700;0,900;1,100;1,300;1,400;1,700;1,900&display=swap');\\n\"\n        )\n\n        # Use provided CSS content or fallback to default\n        css_to_use = css_content or default_css\n\n        # Replace fallback font in the CSS content\n        css_to_use = css_to_use.replace(\"Verdana\", font_name)\n\n        # Combine the font import and the CSS content into a single HTML style block\n        combined_styles = f\"<style>{font_import}{css_to_use}</style>\"\n\n        if verbose:\n            print(combined_styles)  # Print the CSS for debugging if verbose is True\n\n        return HTML(combined_styles)  # Return the generated styles as an HTML object\n\n    except Exception as e:\n        raise RuntimeError(f\"An error occurred while applying styles: {str(e)}\")\n\n# Apply styles (example usage)\napply_css_styles(verbose=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T14:59:36.024156Z","iopub.execute_input":"2025-01-08T14:59:36.024553Z","iopub.status.idle":"2025-01-08T14:59:36.066319Z","shell.execute_reply.started":"2025-01-08T14:59:36.024517Z","shell.execute_reply":"2025-01-08T14:59:36.065103Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"margin: 0;  border-radius: 1em; background-color: #48B0F7; text-align: center; color: white; padding: 50px 20px; position: relative;\">\n  <h1 style=\"font-size: 36px; letter-spacing: 0.05em; margin: 0 20px; font-weight: 900;\">KPRIZE vs. SWE-BENCH</h1>\n  <p style=\"font-size: 20px; margin: 25px 0 0; padding-left: 10px !important;\">Can Language Models Resolve Real-World GitHub Issues?</p>\n  <img src=\"https://www.swebench.com/img/swellama.png\" alt=\"Swell Llama\" style=\"position: absolute; right: 5%; top: 50%; transform: translateY(-50%); width: 135px;\">\n    <img src=\"https://www.kaggle.com/competitions/84795/images/header\" style=\"position: absolute; right: 0%; top: 0%; transform: translateY(-50%); width: 100px; border-radius: 0.25em; opacity: .9;\">\n</div>\n\n<br style=\"margin: 15px;\">\n\n<h2 style=\"text-align: center; font-size: 30px; font-style: normal; font-weight: 800; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">\n    <span style=\"text-decoration: underline;\">\n        <font color=#48B0F7><b>L</b></font>ET'S \n        <font color=#48B0F7><b>L</b></font>EARN \n        <font color=#48B0F7><b>T</b></font>OGETHER !\n    </span><br><br style=\"margin: 15px;\">\n<span style=\"font-size: 22px; letter-spacing: 1px;\">\n    <font color=#48B0F7><b>U</b></font>NDERSTANDING    \n    <font color=#48B0F7><b>T</b></font>HROUGH\n    <font color=#48B0F7><b>E</b></font>XPLORATION</span>\n<br style=\"margin: 15px;\"></h2>\n\n<p style=\"text-align: center; font-size: 15px; font-style: normal; font-weight: bold; text-decoration: None; text-transform: none; letter-spacing: 1px; color: black; background-color: #ffffff;\">CREATED BY: DARIEN SCHETTLER</p>\n\n<hr>\n\n<center><div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em;\">\n    <b style=\"font-size: 18px;\">🛑 &nbsp; WARNING:</b><br><br><b><s>THIS IS A WORK IN PROGRESS</s> <span style=\"font-size: 24px;\"> ---- SUPER–DUPER–WIP ---- </span></b><br>\n</div></center>\n\n<center><div class=\"alert alert-block alert-warning\" style=\"margin: 2em; line-height: 1.7em;\">\n    <b style=\"font-size: 16px;\">👏 &nbsp; IF YOU FORK THIS OR FIND THIS HELPFUL &nbsp; 👏</b><br><br><b style=\"font-size: 22px; color: darkorange\">PLEASE UPVOTE!</b><br><br>This was a lot of work for me, and it may seem silly, but the 🔼 makes me feel appreciated. 😅\n</div></center>\n\n<hr>","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #203354; background-color: #ffffff;\">\n    CHANGELOG\n</h1>\n\n<ul>\n    <li>\n        <b>Version 1-3</b>\n        <ul>\n            <li>Initial Versions</li>\n            <li>Just getting the notebook setup</li>\n        </ul>\n    </li>\n    <li>\n        <b>Version 4</b>\n        <ul>\n            <li>Added Secret Management</li>\n            <li>Reorganize a Bit</li>\n            <li>Begin the EDA</li>\n        </ul>\n    </li>\n    <li>\n        <b>Version 5-7</b>\n        <ul>\n            <li>Make a semi-manual Gemini based agent to test approximate workflow</li>\n        </ul>\n    </li>\n</ul>","metadata":{}},{"cell_type":"markdown","source":"<p id=\"toc\"></p>\n\n<h1 style=\"font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #203354; background-color: #ffffff;\">\n    TABLE OF CONTENTS\n</h1>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#introduction\" style=\"text-decoration: none; color: #48B0F7;\">1&nbsp;&nbsp;&nbsp;&nbsp;INTRODUCTION & JUSTIFICATION</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#background_information\" style=\"text-decoration: none; color: #48B0F7;\">2&nbsp;&nbsp;&nbsp;&nbsp;BACKGROUND INFORMATION</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#imports\" style=\"text-decoration: none; color: #48B0F7;\">3&nbsp;&nbsp;&nbsp;&nbsp;IMPORTS</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#setup\" style=\"text-decoration: none; color: #48B0F7;\">4&nbsp;&nbsp;&nbsp;&nbsp;SETUP AND HELPER FUNCTIONS</a></h3>\n\n<hr>\n\n<h3 style=\"text-indent: 10vw; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; background-color: #ffffff;\"><a href=\"#eda\" style=\"text-decoration: none; color: #48B0F7;\">5&nbsp;&nbsp;&nbsp;&nbsp;EXPLORATORY DATA ANALYSIS</a></h3>\n\n<hr>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"introduction\"></a>\n\n<h1 style=\"font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #48B0F7;\" id=\"introduction\">1&nbsp;&nbsp;INTRODUCTION & JUSTIFICATION&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #203354;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>\n","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">1.1 <b>WHAT</b> IS THIS?</h3>\n<hr>\n\n<ul>\n    <li>This notebook will follow the authors learning path and highlight relevant terms, information, and useful content about the competition.</li>\n    <li>This notebook will conduct an <b>E</b>xploratory <b>D</b>ata <b>A</b>nalysis for the competition.</li>\n    <li>This notebook <i>may</i> propose an open-source baseline solution.</li>\n</ul>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">1.2 <b>WHY</b> IS THIS?</h3>\n<hr>\n\n<ul>\n    <li>Writing and sharing my learning path and the resulting exploratory data analysis can help improve my own understanding of the competition and the data.</li>\n    <li>Sharing my work may help others who are interested in the competition (or the data). This help may take the form of:\n        <ul>\n            <li>Better understanding the problem and potential common solutions (incl. my baseline).</li>\n            <li>Better understanding of the provided dataset.</li>\n            <li>Better understanding of the background information and research.</li>\n            <li>Better ability to hypothesize new solutions.</li>\n        </ul>\n    </li>\n    <li>Exploratory data analysis is a critical step in any data science project. Sharing my EDA might help others in the competition.</li>\n    <li>Writing and sharing my work is often a fun and rewarding experience! It not only allows me to explore and try different techniques, ideas, and visualizations but also encourages and supports other learners and participants.</li>\n</ul>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">1.3 <b>WHO</b> IS THIS FOR?</h3>\n<hr>\n\n\n<ul>\n    <li>The primary purpose of this notebook is to educate <b>MYSELF</b>, however, my review/learning might be beneficial to others:\n        <ul>\n            <li>Other Kagglers (aka. current and future competition participants).</li>\n            <li>Anyone interested in learning more about using artificial intelligence to tackle software engineering tasks directly.</li>\n        </ul>\n    </li>\n</ul>\n","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">1.4 <b>HOW</b> WILL THIS WORK?</h3>\n<hr>\n\n\n<p>I'm going to assemble some markdown cells (like this one) at the beginning of the notebook to go over some concepts/details/etc.</p>\n\n<p>Following this, I will likely go through a few examples from the SWE-Lite dataset to better understand the problems and how they are solved</p>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"background_information\"></a>\n\n<h1 style=\"font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #48B0F7;\" id=\"background_information\">2&nbsp;&nbsp;BACKGROUND INFORMATION&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #203354;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>\n\n<b>Your challenge is to develop an 'agent' that can resolve real-world GitHub issues.</b> \n\nThis competition builds on <a rel=\"noreferrer nofollow\" aria-label=\"the SWE-bench benchmark (opens in a new tab)\" target=\"_blank\" href=\"https://www.swebench.com/\">the SWE-bench benchmark</a> by using Kaggle's forecasting format to ensure that all of the git issues used for the private test set cannot did not exist when the submitted model was trained.\n\nEssentially, this competition emulates <b><a href=\"https://www.swebench.com/\">SWE-BENCH</a></b> with some minor – BUT IMPORTANT – modifications. Let's briefly discuss those differences beefore we dive into more generic background information about <b><a href=\"https://www.swebench.com/\">SWE-BENCH</a></b> and other related background information.\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">KEY DIFFERENCES BETWEEN KPRIZE AND SWE-BENCH</b>\n<br>\n\nBesides the obvious key differences of where/how the hosting is done (containerization, raw setup, constraints, etc.), the following are the main differences I wanted to call out.\n\n1. **EVALUATION METRICS**\n    * **SWE-Bench:**\n        * Primarily measures the percentage of issues resolved by the AI agent, focusing on the agent's ability to generate correct solutions without explicitly penalizing incorrect ones.\n    * **KPrize:**\n        * The **scoring metric** for the competition is designed to incentivize skipping an issue (if the solution is uncertain) over submitting a bad patch. Here's how the formula works:\n        * $$\\text{score} = \\frac{a - b}{a + b + c}$$\n        * Where:\n            - $a$ is the number of correctly resolved issues.\n            - $b$ is the number of failing issues (incorrect solutions).\n            - $c$ is the number of skipped issues.\n2. **DATA CONTAMINATION AND PHASING**\n    * **SWE-Bench**:\n        * The test set is publicly available, which poses a risk of data contamination.\n        * Models might inadvertently train on test data, leading to inflated performance metrics.\n    * **KPrize**:\n        * To mitigate the contamination issue, KPrize utilizes Kaggle's forecasting format to ensure that all of the git issues used for the private test set cannot did not exist when the submitted model was trained.\n        * This results in a phased competition:\n            > **------** PHASE 1 **------**\n            > \n            > A model training phase with a leaderboard using only the public test set of historical data. This test set has about 100 instances and will require approximately one hour of your notebook's nine hour runtime limit to execute unit tests. We expect to improve this performance in the future. The public test set labels can be scraped from github. Accordingly, we may make updates mid-competition in order to keep the public leaderboard reasonably useful. With that said, it is infeasible to keep the public leaderboard results entirely leak-free.\n            >\n            > \n            > **------** PHASE 2 **------**\n            > \n            > A forecasting phase with a leaderboard using only data collected after the submission deadline. You should expect this test set to contain approximately 150-200 instances. The exact count will be provided by the evaluation API. The public test set instances will not be served by the evaluation API during the forecasting phase.\n            \n\n3. **THE SWE-BENCH PYTHON LIBRARY**\n    * **SWE-Bench**\n        * The [SWE-Bench library](https://github.com/swe-bench/SWE-bench) is designed to evaluate the ability of large language models (LLMs) to generate patches that resolve real-world software issues. It provides tasks where models receive an issue description and a codebase, and must generate fixes to address the issue. Below is a high-level overview of its structure:\n        * **Root Directory (SWE-bench)**:\n            * Contains essential files for the benchmark, such as setup scripts, configuration files, and Docker files for creating reproducible environments.\n            * **Subdirectory (`swebench`)**: The core module of SWE-Bench, organized into several submodules:\n                * **`collect`**: Handles the collection of evaluation tasks from GitHub repositories, including mirroring repositories and extracting task instances.\n                * **`harness`**: Provides the evaluation framework to assess model-generated patches against the benchmark.\n                * **`inference`**: Manages running inference for models, facilitating the generation of fixes for issues in repositories.\n                * **`versioning`**: Maintains version control and manages installation configurations for repositories in the benchmark.\n                * **`utils`**: Includes utility functions shared across the framework to support core functionalities.\n    * **KPrize**\n        * `kprize_setup/kprize` is a package in the competition dataset that contains the files used for installing this competition's adaptation of <a rel=\"noreferrer nofollow\" aria-label=\"the swebench library (opens in a new tab)\" target=\"_blank\" href=\"https://github.com/princeton-nlp/SWE-bench\">the SWE-Bench library</a>. Note that this won't currently work on Windows. It has quite a few differences, but at a high level the structure is (ignoring the root directory):\n        * **Package Directory (`kprize`)**:\n            * Contains many utility files and other essential files (far more than are found in the original implementation). I think some of the utility files here may have been found previously in the **`utils`** submodule?\n            * Contains various submodules that mostly overlap with the original implementation but differ in some regards:\n                * **`collection`**: Same as **`collect`**, but I haven't investigated.\n                * **`harness`**: This is likely the same as the original, but I haven't investigated.\n                * **`bundling`**: This is new. TBD\n                * **`evaluation`**: This is new. TBD\n                * **`scripts`**: This is new. TBD\n","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">2.1 <b>UNDERSTANDING</b> THE <b>K</b>PRIZE</h3>\n<hr>\n\n<center><img src=\"https://andykonwinski.com/assets/img/kprize-tweet.png\" width=80%></center>\n\nThe K Prize competition, announced by Andy Konwinski at NeurIPS in December 2024, inspired by SWE-Bench but born from a passion for open-source innovation and competitive programming, this prize challenges teams to build AI models that can effectively solve real-world GitHub issues.\n\nThe competition builds upon SWE-bench, a benchmark that tests AI models against real-world software engineering problems from GitHub repositories. However, it introduces a crucial innovation: a contamination-free evaluation framework that prevents models from being inadvertently trained on test data.\n\nAs Konwinski notes, \"Automating this task will let human software engineers spend lots more time designing new features, reforming abstractions, interfacing with users, and other tasks that are more inherently human (and, for many of us, more fun).\" \n\nBy providing compute resources through Kaggle and implementing a substantial prize structure, the competition aims to catalyze research progress in the same way the Netflix Prize inspired the creation of Apache Spark.\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">2.2 <b>COMPETITION OVERVIEW</b></h3>\n<hr>\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">PRIMARY TASK DESCRIPTION</b>\n<br>\n<br>\nDevelop open-source large language models capable of achieving a 90% score on a contamination-free version of SWE-bench, significantly surpassing the current highest-performing model's 55% achievement. The competition specifically targets real-world software engineering problems, with test data collected after the submission deadline to ensure genuine evaluation.\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">COMPETITION STRUCTURE</b>\n<br>\n<br>\nThe competition proceeds in two distinct phases:\n\n**Model Training Phase:**\n* Public test set of ~100 historical instances\n* Approximately one hour runtime per evaluation\n* Public leaderboard based on historical data\n* Labels can be scraped from GitHub\n\n**Forecasting Phase:**\n* 150-200 new instances collected post-submission\n* Completely new, unseen test data\n* Final evaluation on contamination-free dataset\n* Exact instance count provided by evaluation API\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">EVALUATION INNOVATIONS AND THE METRIC</b>\n<br>\n<br>\nThe K Prize introduces three key innovations:\n\n**Contamination-Free Testing:**\n* Test set created after model submissions\n* Prevents training on test data\n* Ensures genuine evaluation of AI capabilities\n\n**Focus on Open Source:**\n* Only open-source code and open-weight models eligible\n* Encourages community collaboration\n* Builds on collective innovation\n\n**Support for Independent Developers:**\n* Kaggle provides computing resources\n* Levels playing field for smaller teams\n* Promotes \"small AI\" innovation\n\n<br>\n\nSubmissions are scored using a metric that rewards quality over quantity:\n\n$$\\text{score} = \\frac{a - b}{a + b + c}$$\n\nWhere:\n- $a$ is the number of correctly resolved issues.\n- $b$ is the number of failing issues (incorrect solutions).\n- $c$ is the number of skipped issues.\n\nThis formula **HEAVILY incentivizes** skipping difficult issues rather than guessing or submitting incorrect patches.\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">PRIZE STRUCTURE</b>\n<br>\n<br>\nTotal Prize Fund: **\\$1,225,000**\n\n* **Leaderboard Prizes:**\n    * 1st Place: **\\$50,000**\n    * 2nd Place: **\\$20,000**\n    * 3rd-5th Place: **\\$10,000 each**\n\n* **Threshold Bonuses:**\n    * Additional **\\$50,000** pool for each threshold (30%, 40%, 50%, 60%, 70%, 80%, 90%)\n    * Distributed proportionally among qualifying top-5 teams\n\n* **Grand Prize:**\n    * **\\$775,000** additional for first place if reaching 90%\n    * Brings total potential first place prize to **\\$1 million**\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">TECHNICAL FRAMEWORK REQUIREMENTS</b>\n<br>\n<br>\n**Environment:**\n* Python 3.11-based evaluation environments\n* Custom adaptation of SWE-bench library\n* Kaggle-provided Docker container recommended for testing\n\n**Data Structure:**\n* Issue metadata including repo, problem statement, and test information\n* Complete repository copies for context\n* Specialized evaluation API for submission handling\n\n**Submission Requirements:**\n* Must use provided Python evaluation API\n* Open source code and open weight models only\n* Full documentation and reproducibility required\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">COMPETITION HOST(S)/CONTRIBUTOR(S)</b>\n<br>\n<br>\n<b><u><a href=\"https://andykonwinski.com/about/\">Andy Konwinski</a></u></b> is a co-founder of Databricks, Perplexity and Laude Ventures. The prize is personally funded by Konwinski, leveraging his success from Databricks (valued at $62 billion) and Perplexity (approaching double-digit billion-dollar valuation). \n\nThe competition is hosted on Kaggle, working in collaboration with SWE-bench to develop and maintain the evaluation framework.\n\nThe initiative was <b><a href=\"https://andykonwinski.com/2024/12/12/konwinski-prize.html\">announced at the Neural Information Processing Systems (NeurIPS) conference</a></b> in December 2024.\n\nRead more about Andy and bask in his beardliness here\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">KEY DATES</b>\n<br>\n<br>\n**Start Date:** December 11, 2024\n\n**Entry Deadline:** March 5, 2025\n\n**Team Merger Deadline:** March 5, 2025\n\n**Final Submission Deadline:** March 12, 2025\n\n**Competition End Date:** June 11, 2025\n","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">2.3 AN <b>EXAMPLE</b> TO ILLUSTRATE <a href=\"https://github.com/astropy/astropy/issues/16825\">(ASTROPY ISSUE 16825)</a></h3>\n<hr>\n\nRetrieved with:\n\n```python\n# (0) Necessary imports\nimport pandas as pd\n\n# (1) Unzip so we have access to the KPRIZE dataset\n%%capture\n!mkdir -p /kaggle/tmp/konwinski-prize-alt\n!unzip /kaggle/input/konwinski-prize/data.a_zip -d /kaggle/tmp/konwinski-prize-alt/\n\n# (2) Load into a dataframe\ndf = pd.read_parquet('/kaggle/tmp/konwinski-prize-alt/data/data.parquet')\n\n# (3) Get a pd.Series for a demo example\n#       --> This is actually the row with the least text in `patch` and `test_patch` combined\nDEMO_EX = df.iloc[4]\n```\n\nAnd then I manually formatted and took photos and hosted the relevant images...\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">`instance_id`</b>\n<br>\n<br>\n\n> astropy__astropy-16830\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">`repo`</b>\n<br>\n<br>\n\n> astropy/astropy\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">`problem_statement`</b>\n<br>\n<br>\n\n> KeyError: \\'version_1_3_or_later\\' when parsing certain VOTables\\n### Description\\n\\nWhen parsing VOTables (for instance) VOTables with empty integer literals in some places (e.g., MIN value=\"\" in a VALUES element), astropy crashes with\\r\\n\\r\\n\\`\\`\\`\\r\\nKeyError: \\'version_1_3_or_later\\'\\r\\n\\`\\`\\`\\n\\n### Expected behavior\\n\\nWell, you *could* argue that at least some such VOTables should be rejected (a NULL value in a MIN, indeed, does not make sense) with a sensible error message; with the patch proposed in the accompanying PR, astropy emits a warning.  I notice in passing that the reproducer table passes stilts votlint.  I also note in passing that against the warning, a value=\"\" is accepted in a PARAM (as it should, as this is the way to express NULLs).\\r\\n\\r\\nMy fix in the accompanying PR just stops the crashing, leading to a workable table.\\n\\n### How to Reproduce\\n\\nTry\\r\\n\\r\\n\\`\\`\\`python\\r\\nfrom astropy import table\\r\\ntable.Table.read(\"with-empty-min.vot\", format=\"votable\")\\r\\n\\`\\`\\`\\r\\n\\r\\nwith the table from https://docs.g-vo.org/with-empty-min.vot\\n\\n### Versions\\n\\nBoth current HEAD and what\\'s in Debian bookworm.\\r\\n\\nFix votable 1 3 check\\n<!-- These comments are hidden when you submit the pull request,\\r\\nso you do not need to remove them! -->\\r\\n\\r\\n\\r\\n### Description\\r\\n<!-- Provide a general description of what your pull request does.\\r\\nComplete the following sentence and add relevant details as you see fit. -->\\r\\n\\r\\nWhen parsing VOTables (for instance) VOTables with empty integer literals in some places (e.g., MIN value=\"\" in a VALUES element), astropy crashes with\\r\\n\\r\\n\\`\\`\\`\\r\\nKeyError: \\'version_1_3_or_later\\'\\r\\n\\`\\`\\`\\r\\n\\r\\nThis is a simple fix for the problem at hand, doing the key check in analogy to the other key checks of this sort in the few places where an index rather than the get method was used on the config object.\\r\\n\\r\\nOne *might* want to dig deeper, though; I am not exactly sure why the version_1_3_or_later key is missing on the example table from the bug report.  But I suspect (well: hope:-) that is not critical for the bug fix.\\r\\n\\r\\n<!-- In addition please ensure that the pull request title is descriptive\\r\\nand allows maintainers to infer the applicable subpackage(s). -->\\r\\n\\r\\n<!-- READ THIS FOR MANUAL BACKPORT FROM A MAINTAINER:\\r\\nApply \"skip-basebranch-check\" label **before** you open the PR! -->\\r\\n\\r\\n\\r\\n<!-- If the pull request closes any open issues you can add this.\\r\\nIf you replace <Issue Number> with a number, GitHub will automatically link it.\\r\\nIf this pull request is unrelated to any issues, please remove\\r\\nthe following line. -->\\r\\n\\r\\nFixes #16825.\\n\n\n<img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/astropy_issue_16825.png?raw=true\" width=90%>\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">`patch`</b>\n<br>\n<br>\n\n> diff --git a/astropy/io/votable/tree.py b/astropy/io/votable/tree.py\\nindex 036ca8af732..ed85aeffcfd 100644\\n--- a/astropy/io/votable/tree.py\\n+++ b/astropy/io/votable/tree.py\\n@@ -1053,7 +1053,7 @@ def min(self):\\n     @min.setter\\n     def min(self, min):\\n         if hasattr(self._field, \"converter\") and min is not None:\\n-            self._min = self._field.converter.parse(min)[0]\\n+            self._min = self._field.converter.parse(min, config=self._config)[0]\\n         else:\\n             self._min = min\\n \\n@@ -1089,7 +1089,7 @@ def max(self):\\n     @max.setter\\n     def max(self, max):\\n         if hasattr(self._field, \"converter\") and max is not None:\\n-            self._max = self._field.converter.parse(max)[0]\\n+            self._max = self._field.converter.parse(max, config=self._config)[0]\\n         else:\\n             self._max = max\\n \\ndiff --git a/docs/changes/io.votable/16830.bugfix.rst b/docs/changes/io.votable/16830.bugfix.rst\\nnew file mode 100644\\nindex 00000000000..d30a2a9ff96\\n--- /dev/null\\n+++ b/docs/changes/io.votable/16830.bugfix.rst\\n@@ -0,0 +1,1 @@\\n+Fix KeyError when parsing certain VOTables.\\n\n\n<img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/astropy_pr_16830_patch.png?raw=true\" width=80%>\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">`test_patch`</b>\n<br>\n<br>\n\n> diff --git a/astropy/io/votable/tests/test_tree.py b/astropy/io/votable/tests/test_tree.py\\nindex 7fd771cef77..224d189a33f 100644\\n--- a/astropy/io/votable/tests/test_tree.py\\n+++ b/astropy/io/votable/tests/test_tree.py\\n@@ -90,6 +90,31 @@ def test_namespace_warning():\\n     parse(io.BytesIO(good_namespace_13), verify=\"exception\")\\n \\n \\n+def test_votable_values_empty_min_max():\\n+    \"\"\"Regression test for https://github.com/astropy/astropy/issues/16825\"\"\"\\n+    with_empty_minmax = b\"\"\"<VOTABLE xmlns=\"http://www.ivoa.net/xml/VOTable/v1.3\" version=\"1.4\">\\n+        <RESOURCE type=\"results\">\\n+          <TABLE name=\"main\">\\n+            <PARAM name=\"break\" datatype=\"int\" value=\"\"/>\\n+          <FIELD ID=\"hd\" datatype=\"int\" name=\"hd\" ucd=\"meta.id;meta.main\">\\n+            <DESCRIPTION>HD number for this object</DESCRIPTION>\\n+            <VALUES null=\"-2147483648\">\\n+              <MIN value=\"\"/>\\n+              <MAX value=\"\"/>\\n+            </VALUES>\\n+          </FIELD>\\n+          <DATA>\\n+            <BINARY>\\n+              <STREAM encoding=\"base64\">AAMNIg==</STREAM>\\n+            </BINARY>\\n+          </DATA>\\n+        </TABLE>\\n+      </RESOURCE>\\n+    </VOTABLE>\\n+    \"\"\"\\n+    parse(io.BytesIO(with_empty_minmax), verify=\"exception\")\\n+\\n+\\n def test_version():\\n     \"\"\"\\n     VOTableFile.__init__ allows versions of \\'1.1\\', \\'1.2\\', \\'1.3\\' and \\'1.4\\'.\\n\n\n<img src=\"https://github.com/darien-schettler/asset-hosting/blob/main/astropy_pr_16830_test_patch.png?raw=true\" width=80%>\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">`pull_number`</b>\n<br>\n<br>\n\n> 16830\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">`base_commit`</b>\n<br>\n<br>\n\n> e39f486fec48d87aa3677326167954370d7a7bf9\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">`PASS_TO_PASS`</b>\n<br>\n<br>\n\n> ['astropy/io/votable/tests/test_tree.py::test_check_astroyear_fail'\n 'astropy/io/votable/tests/test_tree.py::test_string_fail'\n 'astropy/io/votable/tests/test_tree.py::test_make_Fields'\n 'astropy/io/votable/tests/test_tree.py::test_unit_format'\n 'astropy/io/votable/tests/test_tree.py::test_namespace_warning'\n 'astropy/io/votable/tests/test_tree.py::test_version'\n 'astropy/io/votable/tests/test_tree.py::test_votable_tag'\n 'astropy/io/votable/tests/test_tree.py::test_mivot_constructor'\n 'astropy/io/votable/tests/test_tree.py::test_mivot_readout'\n 'astropy/io/votable/tests/test_tree.py::test_mivot_write'\n 'astropy/io/votable/tests/test_tree.py::test_mivot_write_after_table'\n 'astropy/io/votable/tests/test_tree.py::test_write_no_mivot'\n 'astropy/io/votable/tests/test_tree.py::test_mivot_write_after_resource'\n 'astropy/io/votable/tests/test_tree.py::test_mivot_forbidden_write'\n 'astropy/io/votable/tests/test_tree.py::test_mivot_order']\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">`FAIL_TO_PASS`</b>\n<br>\n<br>\n\n> ['astropy/io/votable/tests/test_tree.py::test_votable_values_empty_min_max']\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">`issue_numbers`</b>\n<br>\n<br>\n\n> [<a href=\"https://github.com/astropy/astropy/issues/16825\">16825</a>, <a href=\"https://github.com/astropy/astropy/issues/16826\">16826</a>]","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">2.4 EVALUATION INFORMATION</h3>\n<hr>\n\n<br>\n\nTo determine what 'correct' means we should review the description in the original SWE-Bench paper.\n> Evaluation is performed by unit test verification using post-PR behavior as the reference solution.\n\nKnowing this we can reexamine the metric provided by the hosts below:\n\n$$\\text{score} = \\frac{a - b}{a + b + c}$$\n\nWhere:\n- $a$ is the number of correctly resolved issues.\n- $b$ is the number of failing issues (incorrect solutions).\n- $c$ is the number of skipped issues.\n\nRemember, this formula **HEAVILY incentivizes** skipping difficult issues rather than guessing or submitting incorrect patches.\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">HOST PROVIDED IMPLEMENTATION</b>\n<br>\n<br>\n\n```python\n# TBD\ndef host_implementation(...):\n    ...\n```\n\nAs a placeholder here is my overly verbose implementation that would simply take the number of correct, incorrect and skipped examples:\n\n```python\ndef calculate_score(correct: int, incorrect: int, skipped: int) -> float:\n    \"\"\"Calculate the SWE-Bench score based on correct, incorrect, and skipped solutions.\n    \n    Args:\n        correct (int): Number of correctly resolved issues (a)\n        incorrect (int): Number of failing issues (b)\n        skipped (int): Number of skipped issues (c)\n    \n    Returns:\n        float: Score calculated using (a-b)/(a+b+c) formula\n    \"\"\"\n    # All values must be positive or 0 and all values cannot be 0\n    correct, incorrect, skipped = max(0, correct), max(0, incorrect), max(0, skipped)\n    if not any([correct, incorrect, skipped])\n        return 0.0\n        \n    # Calculate score using provided formula (no 0 check required due to above)\n    return (correct - incorrect) / (correct + incorrect + skipped)\n```","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">2.5 COMPETITION DATA OVERVIEW</h3>\n<hr>\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">DATA FILE DESCRIPTIONS</b>\n\n**`data.a_zip` description:**\n* A renamed `.zip` archive created to work around filename and nested archive constraints\n* Unzipping this file creates the main data folder containing all dataset components\n\n**`data/data.parquet` columns:**\n* **`instance_id`** (string)\n    * Unique string identifier for each instance (GitHub issue)\n* **`repo`** (string)\n    * The GitHub repository relevant to the issue\n    * Also accessible through the evaluation API\n* **`problem_statement`** (string)\n    * Textual description of the issue\n    * Also accessible through the evaluation API\n* **`patch`** (string)\n    * The patch that resolves the issue\n    * *Only provided in the train set*\n* **`test_patch`** (string)\n    * The patch that resolves the issue\n    * *Only provided in the train set*\n* **`pull_number`** (int)\n    * The pull request number that resolved the issue\n* **`base_commit`** (string)\n    * The commit used as the foundation for the provided repository copy\n* **`issue_numbers`** (int)\n    * The original ID number of the GitHub issue\n* **`[PASS_TO_PASS/FAIL_TO_PASS]`** (list)\n    * Lists containing unit tests to be executed for this issue\n\n**Directory Structure Details:**\n* **`data/*/`**\n    * All other subdirectories are utilized by the evaluation API\n    * Used to configure evaluation environments\n    * All evaluation environments run Python 3.11\n\n**`kprize_setup/` description:**\n* Contains files for installing the competition's adapted <b><a href=\"https://github.com/princeton-nlp/SWE-bench\">swebench library</a></b>\n* *Note: Currently not compatible with Windows systems*\n\n**`kaggle_evaluation/` description:**\n* Contains files implementing the evaluation API\n* Implementation details may be useful for offline testing\n* Recommended to start with the <b><a href=\"https://www.kaggle.com/code/sohier/konwinski-prize-demo-submission\">demo submission notebook</a></b>\n* Strong recommendation to run API in <b><a href=\"https://github.com/Kaggle/docker-python\">Docker container based on Kaggle's image</a></b> for local execution\n    * Helps avoid conflicts with existing Python environments\n* The API will install if needed:\n    * <b><a href=\"https://mamba.readthedocs.io/en/latest/user_guide/micromamba.html\">Micromamba</a></b>\n    * Several libraries (listed in kprize_setup/pip_packages)\n    * Creates new Python environments as required\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">ADDITIONAL NOTES:</b>\n\n* Further updates are planned for the evaluation API and kprize library\n    * Aimed at improving runtime\n    * Will provide additional useful tooling\n    * Updates not expected to impact core submission workflow\n* Refer to <b><a href=\"https://www.kaggle.com/competitions/konwinski-prize/discussion/552449\">this forum post</a></b> for additional details\n* Users are encouraged to source additional codebases for model training\n* Most metadata is only available in the train set\n\n<br>\n\n<b style=\"text-decoration: underline; font-size: 15px; text-transform: uppercase; letter-spacing: 2px; font-weight: 900;\">POST-EDA DATA OBSERVATIONS</b>\n\nTBD\n\n<br>\n","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">2.6 GLOSSARY AND BACKGROUND RESEARCH</h3>\n<hr>\n\n<br>\n\nI'm going to blow this out more in the future, most of my glossary/research is done by hand in a notebook right now, so this is just some preliminary definitions to get us started...\n\n<br>\n\n| Term | Definition | Example | Example Explanation |\n|------|------------|---------|-------------------|\n| Diff | A textual representation showing the differences between two versions of a file or set of files, highlighting added, removed, and modified lines. | `@@ -10,7 +10,7 @@\\n def hello():\\n-    print 'hello'\\n+    print('hello')\\n` | Shows a change from Python 2.x to 3.x print syntax. The `-` line shows removed code, the `+` line shows added code. |\n| Patch | A file containing a set of changes (diff) that can be applied to a codebase. | `--- a/hello.py\\n+++ b/hello.py\\n@@ -1,3 +1,3 @@\\n def greet(name):\\n-    print \"Hi, \" + name\\n+    print(f\"Hi, {name}\")` | A patch file showing the location (`hello.py`) and changes to update string formatting. |\n| Test Patch | A modification specifically made to test files in a codebase. | `--- a/test_calc.py\\n+++ b/test_calc.py\\n@@ -1,2 +1,6 @@\\n def test_add():\\n-    assert calc.add(2,2) == 4\\n+    assert calc.add(2,2) == 4\\n+    assert calc.add(-1,1) == 0\\n+    assert calc.add(0,0) == 0` | Adds new test cases to verify edge cases for an addition function. |\n| Unix Patch Apply | A command-line operation that applies a patch file to a codebase. | `$ patch -p1 < bugfix.patch` | Applies changes from `bugfix.patch` to the codebase, where `-p1` strips one directory level from paths. |\n| GitHub Issue | A tracked item in GitHub's issue system. | `Title: \"TypeError when using pandas.read_csv() with custom delimiter\"\\nDescription: \"When using a tab delimiter...\"` | A bug report describing unexpected behavior with specific steps to reproduce. |\n| Pull Request (PR) | A GitHub feature that proposes changes from one branch to another. | `PR #1234: \"Fix TypeError in CSV parser with custom delimiters\"` | A proposed solution to the issue, containing code changes and tests. |\n| Base Commit | The specific commit that serves as the starting point for changes. | `git checkout a1b2c3d` | References commit `a1b2c3d` as the state before any changes are made. |\n| Pass to Pass Tests (P2P) | Tests that passed before and should pass after changes. | `def test_existing():\\n    assert len([1,2,3]) == 3` | A test verifying basic list functionality that shouldn't be affected by changes. |\n| Fail to Pass Tests (F2P) | Tests that failed before and should pass after changes. | `def test_fix():\\n    assert parser.read_tab('file.txt')` | A test specifically checking if the bug fix works. |\n| SWE-bench Verified | 500 manually verified solvable problems. | `\"Fix pandas DataFrame.fillna() with datetime values\"` | A verified issue with clear reproduction steps and known solution. |\n| SWE-bench Lite | 300 self-contained issues focused on functional bugs. | `\"Fix off-by-one error in list slicing\"` | A focused bug fix that doesn't require extensive codebase knowledge. |\n| Oracle Retrieval | Providing only the files edited in the reference solution. | `{\"files\": [\"pandas/core/frame.py\", \"pandas/tests/frame/test_datetime.py\"]}` | Only retrieves the specific files that need modification. |\n| BM25 Retrieval | Selecting relevant files based on issue description. | `query: \"DataFrame fillna datetime\"\\nreturned: [\"pandas/core/frame.py\", ...]` | Uses text similarity to find potentially relevant files. |\n| Breaking Resolved | Fixes the target issue but breaks existing functionality. | `Fixed: test_new_feature()\\nBroken: test_old_feature()` | The change fixed the bug but introduced a regression. |\n| Docker Evaluation | Containerized testing environment. | `docker run swe-bench eval --task pandas-1234` | Runs evaluation in an isolated, reproducible environment. |\n| Gold Patch | Reference solution from the original PR. | `diff --git a/src/main.py\\n--- a/src/main.py\\n+++ b/src/main.py\\n...` | The accepted solution that properly fixed the issue. |\n| Cross-Context Edit | Changes requiring modifications across multiple files. | `Changed: src/parser.py, src/utils.py, tests/test_parser.py` | A fix that requires coordinated changes in multiple components. |\n","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"imports\"></a>\n\n<h1 style=\"font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #48B0F7;\" id=\"imports\">3&nbsp;&nbsp;IMPORTS&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #203354;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>\n","metadata":{}},{"cell_type":"code","source":"print(\"\\n... PIP INSTALLS STARTING ...\\n\")\n!pip install -q google-genai  # For free Gemini access\n!pip install -q /kaggle/input/konwinski-prize/kprize_setup/kprize-1.0.0-py3-none-any.whl --no-index --find-links /kaggle/input/konwinski-prize/kprize_setup/pip_packages\nprint(\"\\n... PIP INSTALLS COMPLETE ...\\n\")\n\nprint(\"\\n... IMPORTS STARTING ...\\n\")\nprint(\"\\n\\tVERSION INFORMATION\")\n\n# Competition specific imports\nimport kaggle_evaluation.konwinski_prize_inference_server\nfrom datasets import load_dataset\nfrom google.genai import types\nfrom google import genai\nimport unidiff\n\nimport pandas as pd; pd.options.mode.chained_assignment = None; pd.set_option('display.max_columns', None)\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\")\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\")\nimport polars as pl; print(f\"\\t\\t– POLARS VERSION: {pl.__version__}\")\n\n# Built-In Imports (mostly don't worry about these)\nfrom typing import Iterable, Any, Literal, Callable, Generator\nfrom kaggle_datasets import KaggleDatasets\nfrom dataclasses import dataclass\nfrom collections import Counter\nfrom datetime import datetime\nfrom zipfile import ZipFile\nfrom io import StringIO\nfrom glob import glob\nimport subprocess\nimport tempfile\nimport warnings\nimport requests\nimport textwrap\nimport hashlib\nimport imageio\nimport IPython\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport copy\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport gc\nimport re\nimport os\n\n# Rich\nfrom rich import pretty; pretty.install()\nfrom rich.markdown import Markdown\nfrom rich import print as rprint\nfrom rich.console import Console\nfrom rich.style import Style\nfrom rich.live import Live\nfrom rich.text import Text\nfrom rich import inspect\n\n# Visualization Imports (overkill)\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nfrom tqdm.notebook import tqdm; tqdm.pandas();\nfrom IPython.core.display import HTML\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport seaborn as sns\nimport plotly\n\ndef seed_it_all(seed: int = 7, fix_tf_seed: bool = False):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n\n    if fix_tf_seed:\n        raise NotImplementedError(\"Not importing TF yet...\")\n        # tf.random.set_seed(seed)\n    \nseed_it_all()\n\nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T14:59:36.068669Z","iopub.execute_input":"2025-01-08T14:59:36.069027Z","iopub.status.idle":"2025-01-08T15:00:06.660268Z","shell.execute_reply.started":"2025-01-08T14:59:36.068994Z","shell.execute_reply":"2025-01-08T15:00:06.658795Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<br>\n\n<a id=\"setup\"></a>\n\n<h1 style=\"font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #48B0F7;\" id=\"setup\">4&nbsp;&nbsp;SETUP AND HELPER FUNCTIONS&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #203354;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>\n","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">4.0 HELPER FUNCTIONS</h3>\n<hr>\n\n<br>\n","metadata":{}},{"cell_type":"code","source":"\"\"\"Rich visualization utility.\"\"\"\n\nimport pathlib\nfrom rich.filesize import decimal\nfrom rich.markup import escape\nfrom rich.text import Text\nfrom rich.tree import Tree\n\nDEFAULT_IGNORE_LIST: Iterable[str] = \".ipynb_checkpoints\", \".DS_Store\", \".git\", \".idea\", \".coverage\", \".pytest_cache\"\n\ndef get_directory_tree(\n        directory: pathlib.Path,\n        tree: Tree | None = None,\n        show_hidden: bool = False,\n        inplace: bool = False,\n        ignore_list_extras: Iterable[str] = (),\n) -> Tree | None:\n    \"\"\"Recursively build a Tree with directory contents.\n\n    Args:\n        directory (pathlib.Path): The directory to walk.\n        tree (Tree, optional):\n            The Tree object to build.\n            If not provided, a new Tree is created.\n        show_hidden (bool, optional):\n            Whether to show hidden files.\n        inplace (bool, optional):\n            Whether to print the tree in place.\n            If False, the tree is returned.\n        ignore_list_extras (Iterable[str], optional):\n            Additional file extensions to ignore.\n\n    Returns:\n        If inplace is False, the Tree object with the directory contents.\n        Else, None.\n    \"\"\"\n    # Get the ignore list that includes the default and any extras\n    ignore_list = sorted(set(DEFAULT_IGNORE_LIST) | set(ignore_list_extras))\n\n    # Create a new Tree if one is not provided\n    tree = tree or Tree(label=f\"[bold]{directory!s}[/bold] File Tree\")  # type: ignore\n\n    # Sort dirs first then by filename\n    paths = sorted(\n        pathlib.Path(directory).iterdir(),\n        key=lambda path: (path.is_file(), path.name.lower()),\n    )\n\n    # Sort dirs first then by filename\n    for path in paths:\n\n        # Remove hidden files if show_hidden is False\n        if path.name.startswith(\".\") and not show_hidden:\n            continue\n\n        # Skip files in the ignore list by suffix or name\n        if path.is_file() and (path.suffix in ignore_list or path.name in ignore_list):\n            continue\n\n        # Skip directories only by name (not suffix)\n        if path.is_dir() and path.name in ignore_list:\n            continue\n\n        # Add the directory to the tree\n        if path.is_dir():\n\n            # Style directories starting with \"__\" differently\n            style = \"dim\" if path.name.startswith(\"__\") else \"\"\n\n            # Add the directory to the tree\n            branch = tree.add(\n                f\"[bold magenta]:open_file_folder: [link file://{path}]{escape(path.name)}\",\n                style=style,\n                guide_style=style,\n            )\n            get_directory_tree(path, branch)\n\n        # Add the file to the tree\n        else:\n            main_style = \"dim green\" if path.name.startswith(\"_\") else \"green\"\n            ext_style = \"dim red\" if path.name.startswith(\"_\") else \"bold red\"\n            file_size_style = \"dim blue\" if path.name.startswith(\"_\") else \"blue\"\n            text_filename = Text(path.name, main_style)\n            text_filename.highlight_regex(r\"\\..*$\", ext_style)\n            text_filename.stylize(f\"link file://{path}\")\n            text_filename.append(f\" ({decimal(path.stat().st_size)})\", file_size_style)\n            if path.suffix == \".py\":\n                icon = \"🐍 \"\n            elif path.suffix == \".ipynb\":\n                icon = \"🐍📓 \"\n            elif path.suffix == \".sh\":\n                icon = \"🔧 \"\n            elif \".env\" in path.name.lower():\n                icon = \"🔑 \"\n            elif path.suffix == \".csv\":\n                icon = \"📊 \"\n            elif path.suffix in [\".yaml\", \".yml\", \".json\"]:\n                icon = \"📜 \"\n            elif path.suffix in [\".txt\", \".md\"]:\n                icon = \"📝 \"\n            elif path.suffix in [\".png\", \".jpg\", \".jpeg\", \".gif\", \".svg\"]:\n                icon = \"🖼️ \"\n            elif path.suffix in [\".zip\", \".tar\", \".gz\", \".7z\"]:\n                icon = \"📦 \"\n            elif path.suffix in [\".pdf\"]:\n                icon = \"📰 \"\n            elif path.suffix in [\".mp4\", \".avi\", \".mov\", \".mkv\"]:\n                icon = \"🎥 \"\n            elif path.suffix in [\".mp3\", \".wav\", \".flac\"]:\n                icon = \"🎵 \"\n            elif path.suffix in [\".html\", \".css\", \".js\"]:\n                icon = \"🌐 \"\n            elif path.suffix in [\".exe\", \".msi\"]:\n                icon = \"🛠️ \"\n            elif path.suffix in [\".docx\", \".pptx\", \".xlsx\"]:\n                icon = \"📄 \"\n            elif path.suffix in [\".parquet\", \".feather\"]:\n                icon = \"🧼 \"\n            elif path.suffix in [\".db\", \".sqlite\", \".sql\", \".jsonl\"]:\n                icon = \"🗄️ \"\n            else:\n                icon = \"📄 \"\n\n            # Prefix hidden files with a \"🤫\" emoji\n            if path.name.startswith(\".\"):\n                icon = \"🤫\"+icon\n\n            # Add the file to the tree (with icon prefix)\n            tree.add(Text(icon) + text_filename)\n\n    # If inplace is False, return the tree... otherwise the Tree object is updated in place\n    if not inplace:\n        return tree\n    return None\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T15:00:06.663192Z","iopub.execute_input":"2025-01-08T15:00:06.664127Z","iopub.status.idle":"2025-01-08T15:00:06.689222Z","shell.execute_reply.started":"2025-01-08T15:00:06.664069Z","shell.execute_reply":"2025-01-08T15:00:06.688117Z"},"_kg_hide-output":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def stream_with_styling(chunk_generator: Any, return_completion: bool = True) -> None:\n    \"\"\"Streams text content with real-time Markdown styling using Rich library.\n    \n    Processes chunks of text from a generator and displays them with live Markdown \n    formatting, continuously updating the display as new chunks arrive.\n    \n    Args:\n        chunk_generator (Any): \n            A generator-like object that yields objects containing text attributes.\n            Each chunk object must have a 'text' attribute containing the text to display.\n        return_completion (bool, optional):\n            Whether we return the consolidated text.\n    \n    Returns:\n        str; The text that was returned by the model \n        --OR--\n        None;\n    \"\"\"\n    console = Console()\n    with Live(console=console, auto_refresh=False) as live:\n        current_text = \"\"\n        for chunk in chunk_generator:\n            current_text += chunk.text\n            live.update(Markdown(current_text))\n            live.refresh()\n    \n        # Final render\n        live.update(Markdown(current_text))\n        live.refresh()\n        time.sleep(0.01)\n    \n    return current_text if return_completion else None\n\nclass MyUserSecretsClient:\n    \"\"\"A wrapper around UserSecretsClient enhancing secret management functionality.\n    \n    This class extends the base UserSecretsClient by adding error handling and \n    environment variable integration capabilities.\n    \"\"\"\n    \n    def __init__(self) -> None:\n        \"\"\"Initialize the MyUserSecretsClient with a base UserSecretsClient instance.\n        \n        Attributes:\n            _user_secrets_client (UserSecretsClient):\n                The original secret management client object we wrap.\n        \"\"\"\n        self._user_secrets_client = UserSecretsClient()\n\n    def get_secret(self, label: str, default: Any = None) -> Any:\n        \"\"\"Retrieve a secret by label with fallback to default value.\n        \n        Attempts to retrieve a secret value, handling potential errors gracefully\n        by returning a default value if the secret cannot be accessed.\n        \n        Args:\n            label (str): \n                The identifier of the secret to retrieve.\n            default (Any, optional): \n                Value to return if the secret cannot be retrieved.\n        \n        Returns:\n            The secret value if successfully retrieved (str), \n            Otherwise, the default value specified (by default this is None).\n        \n        Example:\n            >>> client = MyUserSecretsClient()\n            >>> api_key = client.get_secret(\"API_KEY\", default=\"fallback_key\")\n        \"\"\"\n        try:\n            return self._user_secrets_client.get_secret(label)\n        except BackendError:\n            # Expected error when secret doesn't exist\n            return default\n        except Exception as e:\n            # Log unexpected errors while still maintaining graceful fallback\n            print(\n                f\"Unexpected error while retrieving secret '{label}': {str(e)}. \"\n                \"This may indicate a problem beyond the secret not existing.\"\n            )\n            return default\n\n    def secret_to_env(\n        self, \n        label: str, \n        env_var_label: str | None = None, \n        default: Any = None\n    ) -> None:\n        \"\"\"Set an environment variable using a secret value.\n        \n        Retrieves a secret and sets it as an environment variable. If the secret\n        cannot be retrieved, uses the provided default value instead.\n\n        Note the provided value will be cast as a string when set as an environment variable.\n        \n        Args:\n            label (str): \n                The identifier of the secret to retrieve.\n            env_var_label (str, optional): \n                The name of the environment variable to set.\n                If None, uses the secret's label.\n            default (Any, optional): \n                Value to use if the secret cannot be retrieved.\n        \n        Returns:\n            None; \n                An environment variable will be updated/created with the \n                appropriate value.\n                \n        \n        Example:\n            >>> client = MyUserSecretsClient()\n            >>> client.secret_to_env(\"DB_PASSWORD\", \"DATABASE_PASSWORD\", \"default_pass\")\n        \"\"\"\n        env_var_label = env_var_label if env_var_label is not None else label\n        os.environ[env_var_label] = str(self.get_secret(label, default))\n\n\ndef is_valid_patch_format(patch: str | StringIO) -> bool:\n    \"\"\"Validates if the input string represents a valid unified diff patch format.\n\n    This function checks if the provided input can be parsed as a valid unified diff\n    patch using the unidiff library. It verifies both the syntax and presence of\n    actual patch content.\n\n    Based on: https://www.kaggle.com/code/sohier/patch-validation-snippet\n\n    Args:\n        patch (str | StringIO): A string or StringIO object containing the potential patch content.\n            The patch should be in unified diff format\n                - created by diff -u or similar tools\n                - or generated by an LLM as in our case.\n\n    Returns:\n        bool: True if the input is a valid non-empty patch, False otherwise.\n\n    Raises:\n        unidiff.UnidiffParseError: Parsing exception\n        Exception: Any other problems with the string\n    \n    Examples:\n        >>> is_valid_patch_format(\"--- a/file.txt\\n+++ b/file.txt\\n@@ -1,1 +1,1 @@\\n-old\\n+new\")\n        True\n        >>> is_valid_patch_format(\"invalid content\")\n        False\n    \"\"\"\n    # All patches must be of type StringIO\n    if not isinstance(patch, (str, StringIO)):\n        return False\n\n    try:\n        # Convert string to StringIO if needed for consistent handling\n        patch_content = StringIO(patch) if isinstance(patch, str) else patch\n        \n        # Attempt to parse the patch (may trigger an error on fail which results in False)\n        patch_set = unidiff.PatchSet(patch_content)\n        \n        # Verify the patch contains actual changes\n        if len(patch_set) == 0:\n            print(\"Patch is either not a valid patch or contains no actual changes!\\n\")\n            return False\n            \n        # Additional validation: check if there are any actual changes\n        #   - Check if file has any hunks, if so we return True.\n        #   - Otherwise return False\n        return any(len(patched_file) > 0 for patched_file in patch_set)\n\n    # Log expected unidiff parsing errors and return False\n    except unidiff.UnidiffParseError as e:\n        print(f\"Unidiff parsing error (returning False)\\n\\tunidiff.UnidiffParseError: {str(e)}\\n\")\n        return False\n\n    # Log unexpected errors while still returning False\n    except Exception as e:\n        print(f\"Unexpected error validating patch (returning False)\\n\\tGeneral Exception: {str(e)}\\n\")\n        return False\n\n\ndef calculate_kprize_score(correct: int, incorrect: int, skipped: int) -> float:\n    \"\"\"Calculate the SWE-Bench score based on correct, incorrect, and skipped solutions.\n    \n    Args:\n        correct (int): Number of correctly resolved issues (a)\n        incorrect (int): Number of failing issues (b)\n        skipped (int): Number of skipped issues (c)\n    \n    Returns:\n        float: Score calculated using (a-b)/(a+b+c) formula\n    \"\"\"\n    # All values must be positive or 0 and all values cannot be 0\n    correct, incorrect, skipped = max(0, correct), max(0, incorrect), max(0, skipped)\n    if not any([correct, incorrect, skipped]):\n        return 0.0\n        \n    # Calculate score using provided formula (no 0 check required due to above)\n    return (correct - incorrect) / (correct + incorrect + skipped)\n\ndef calculate_swebench_score(correct: int, incorrect: int) -> float:\n    \"\"\"Calculate the SWE-Bench score based on correct and incorrect solutions.\n    \n    Args:\n        correct (int): Number of correctly resolved issues (a)\n        incorrect (int): Number of failing issues (b)\n    \n    Returns:\n        float: Score calculated using a/b formula\n    \"\"\"\n    # All values must be positive or 0 and all values cannot be 0\n    correct, incorrect = max(0, correct), max(0, incorrect)\n    if not any([correct, incorrect]):\n        return 0.0\n        \n    # Calculate score using provided formula (no 0 check required due to above)\n    return correct / incorrect\n\n    \ndef flatten(items: Iterable[Any], as_list: bool = True) -> Generator[Any, None, None] | list[Any]:\n   \"\"\"Flattens an iterable of items or nested iterables into a single level sequence.\n\n    Strings are treated as atomic elements and will not be flattened.\n    Dictionaries are treated as lists where the keys are the elements.\n\n   Args:\n       items (Iterable[Any]): \n           An iterable containing either individual elements or nested iterables.\n       as_list (bool, optional): \n           If True, returns a list. [DEFAULT BEHAVIOUR]\n           If False, returns a generator.\n\n   Returns:\n       If `as_list` is True, returns a flattened list.\n       If `as_list` is False, returns a generator yielding flattened elements.\n\n   Examples:\n       Basic usage returning a list:\n       >>> nested = [1, [2, 3], [4, [5, 6]]]\n       >>> flatten(nested)\n       [1, 2, 3, 4, 5, 6]\n\n       Using generator output:\n       >>> nested = ['a', ['b', 2], 'c']\n       >>> list(flatten(nested, as_list=False))\n       ['a', 'b', 2, 'c']\n\n       Handles non-nested iterables:\n       >>> simple = [1, 2, 3]\n       >>> flatten(simple)\n       [1, 2, 3]\n\n   Raises:\n       TypeError: If input is not an iterable.\n   \"\"\"\n   def _flatten_generator(items: Iterable[Any]) -> Generator[Any, None, None]:\n       # Iterate through each item in the input iterable\n       for item in items:\n           # Check if item is an iterable but not a string/bytes\n           # Strings/bytes are treated as atomic elements\n           if isinstance(item, Iterable) and not isinstance(item, (str, bytes)):\n               # Recursively flatten nested iterables\n               yield from _flatten_generator(item)\n           else:\n               # Yield non-iterable items directly\n               yield item\n\n   # Return either a generator or list based on as_list parameter\n   generator = _flatten_generator(items)\n   return list(generator) if as_list else generator\n        \n### TEST IT OUT IF YOU WANT ###\n# rprint(Markdown(\"---\"))\n# # Valid patch\n# if is_valid_patch_format(dev_df.patch[0]):\n#     rprint(\"[bold green]SUCCESS[/bold green]\")\n# else:\n#     rprint(\"[bold red]FAILURE[/bold red]\")\n# rprint(Markdown(\"---\"))\n\n# # Example showing unidiff parsing error\n# if is_valid_patch_format(dev_df.patch[0].replace(\".py\", \"0.py\", 3)):\n#     rprint(\"[bold green]SUCCESS[/bold green]\")\n# else:\n#     rprint(\"[bold red]FAILURE[/bold red]\")\n# rprint(Markdown(\"---\"))\n\n# # Example showing complete failure (weird dash)\n# if is_valid_patch_format(dev_df.patch[0].replace(\"-\", \"–\")):\n#     rprint(\"[bold green]SUCCESS[/bold green]\")\n# else:\n#     rprint(\"[bold red]FAILURE[/bold red]\")\n# rprint(Markdown(\"---\"))\n\n# # Example showing complete failure (random stringf)\n# if is_valid_patch_format(\"invalid content\"):\n#     rprint(\"[bold green]SUCCESS[/bold green]\")\n# else:\n#     rprint(\"[bold red]FAILURE[/bold red]\")\n# rprint(Markdown(\"---\"))\n\n### TRY IT OUT ###\n# flatten(((1,[2,{3: \"hi\"}]), [\"a b c d\".split(), \"abcd\", \"b\", \"5\", 5, [\"hi\", \"there\"]]), as_list=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T15:00:06.69095Z","iopub.execute_input":"2025-01-08T15:00:06.69138Z","iopub.status.idle":"2025-01-08T15:00:06.717771Z","shell.execute_reply.started":"2025-01-08T15:00:06.69133Z","shell.execute_reply":"2025-01-08T15:00:06.716472Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">4.1 PATH DEFINITIONS</h3>\n<hr>\n\n<br>\n","metadata":{}},{"cell_type":"code","source":"# Rudimentary Paths\nBASE_DIR = \"/kaggle\"\nTMP_DIR = os.path.join(BASE_DIR, \"tmp\")\nWORKING_DIR = os.path.join(BASE_DIR, \"working\")\nINPUT_DIR = os.path.join(BASE_DIR, \"input\")\n\n# Basic Competition Paths\nCOMP_DIR = os.path.join(INPUT_DIR, \"konwinski-prize\")\nCOMP_KAGGLE_EVALUATION_DIR = os.path.join(COMP_DIR, \"kaggle_evaluation\")\nCOMP_KPRIZE_SETUP_DIR = os.path.join(COMP_DIR, \"kprize_setup\")\n\n# Dataset Competition Paths\nCOMP_DATA_ZIP_PATH = os.path.join(COMP_DIR, \"data.a_zip\")\nCOMP_TMP_DIR = os.path.join(TMP_DIR, \"konwinski-prize-alt\")\nCOMP_TMP_DATA_DIR = os.path.join(COMP_TMP_DIR, \"data\")\nCOMP_DATA_PARQUET_PATH = os.path.join(COMP_TMP_DATA_DIR, \"data.parquet\")\nCOMP_CONDA_PACKAGES_DIR = os.path.join(COMP_TMP_DATA_DIR, \"conda_packages\")\nCOMP_PIP_PACKAGES_DIR = os.path.join(COMP_TMP_DATA_DIR, \"pip_packages\")\nCOMP_REPO_CONFIGS_DIR = os.path.join(COMP_TMP_DATA_DIR, \"repo_configs\")\nCOMP_REPOS_DIR = os.path.join(COMP_TMP_DATA_DIR, \"repos\")\n\n# SWE Dataset Paths ... https://huggingface.co/datasets/...\nHF_SWE_BENCH_PROVIDER = \"princeton-nlp\"\nHF_SWE_BENCH_PATH = os.path.join(HF_SWE_BENCH_PROVIDER, \"SWE-bench\")\nHF_SWE_BENCH_LITE_PATH = os.path.join(HF_SWE_BENCH_PROVIDER, \"SWE-bench_Lite\")\nHF_SWE_BENCH_VERIFIED_PATH = os.path.join(HF_SWE_BENCH_PROVIDER, \"SWE-bench_Verified\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T15:00:06.720445Z","iopub.execute_input":"2025-01-08T15:00:06.720861Z","iopub.status.idle":"2025-01-08T15:00:06.742599Z","shell.execute_reply.started":"2025-01-08T15:00:06.720824Z","shell.execute_reply":"2025-01-08T15:00:06.741309Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">4.2 SECRET/ENVIRONMENT LOADING/SETUP</h3>\n<hr>\n\n<br>","metadata":{}},{"cell_type":"code","source":"USE_KAGGLE_SECRETS = True\nSECRETS_TO_USE = [\"GEMINI_API_KEY\", \"HUGGINGFACE_TOKEN\", \"OPENAI_API_KEY\"]\n\nif USE_KAGGLE_SECRETS:\n    from kaggle_secrets import UserSecretsClient, BackendError\n    user_secrets = MyUserSecretsClient()\n    for secret in SECRETS_TO_USE: \n        user_secrets.secret_to_env(secret)\nelse:\n    from dotenv import load_dotenv, find_dotenv\n    _ = load_dotenv(find_dotenv())\n\n# Check if env is loaded\nfor secret in SECRETS_TO_USE:\n    if os.getenv(secret):\n        print(f\"✅  The {repr(secret)} value HAS been set.\")\n    else:\n        print(f\"❌  The {repr(secret)} value HAS NOT set.\")\n\nclient = genai.Client(api_key=os.getenv(\"GEMINI_API_KEY\"))\n\nrprint(\"\\n\\n[bold cyan]TESTING OUR GEMINI AUTHENTICATION WITH A MESSAGE TO GEMINI FLASH![/bold cyan]\\n\")\nDEMO_MESSAGE = \"Tell me about Andy Konwinski the Computer scientist\"\n\nstream_with_styling(client.models.generate_content_stream(model=\"gemini-2.0-flash-exp\", contents=DEMO_MESSAGE))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T15:00:06.744138Z","iopub.execute_input":"2025-01-08T15:00:06.744497Z","iopub.status.idle":"2025-01-08T15:00:14.314487Z","shell.execute_reply.started":"2025-01-08T15:00:06.744464Z","shell.execute_reply":"2025-01-08T15:00:14.313448Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">4.3 LOAD THE SWE-LITE DATASET</h3>\n<hr>\n\n<br>\n\nThis dataset is a slim version of the full SWE-Bench dataset and can be found on <b><a href=\"https://huggingface.co/datasets/princeton-nlp/SWE-bench_Lite\">HuggingFace</a></b>\n\nThis dataset contains two splits:\n* **DEV**\n    * 27 Examples\n* **TEST**\n    * 300 Examples","metadata":{}},{"cell_type":"code","source":"# Load the huggingface dataset\nswe_bench_lite_ds = load_dataset(HF_SWE_BENCH_LITE_PATH)\n\n# Create the sub-dataframes\ndev_df = swe_bench_lite_ds[\"dev\"].to_pandas()\ntest_df = swe_bench_lite_ds[\"test\"].to_pandas()\n\n# Display The Two Dataframes\nrprint(\"\\n[bold blue]DEV SWE BENCH LITE DATAFRAME[/bold blue]\")\ndisplay(dev_df.head(3))\n\nrprint(\"\\n[bold green]TEST SWE BENCH LITE DATAFRAME[/bold green]\")\ndisplay(test_df.head(3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T15:00:14.316014Z","iopub.execute_input":"2025-01-08T15:00:14.316491Z","iopub.status.idle":"2025-01-08T15:00:20.841925Z","shell.execute_reply.started":"2025-01-08T15:00:14.316441Z","shell.execute_reply":"2025-01-08T15:00:20.840839Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">4.4 LOAD THE KPRIZE DATASET</h3>\n<hr>\n\n<br>\n\nThis is the provided 'training' data. Really these are just a few examples to show us the format and structure that we are to expect. It's on us to gather more data probably.","metadata":{}},{"cell_type":"code","source":"# (1) Handle unzipping the compressed dataset (or skip if already done)\n# Check if the zip file has already been uncompressed\nif not os.path.isfile(COMP_DATA_PARQUET_PATH):    \n    # Make the directory to unzip to\n    os.makedirs(COMP_TMP_DIR, exist_ok=True)\n    \n    # Open and extract the zip file\n    with ZipFile(COMP_DATA_ZIP_PATH, 'r') as zip_ref:\n        zip_ref.extractall(COMP_TMP_DIR)\n\n# (2) Load into a dataframe\nkprize_df = pd.read_parquet(COMP_DATA_PARQUET_PATH)\ndisplay(kprize_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T15:00:20.843208Z","iopub.execute_input":"2025-01-08T15:00:20.843556Z","iopub.status.idle":"2025-01-08T15:00:24.790155Z","shell.execute_reply.started":"2025-01-08T15:00:20.843523Z","shell.execute_reply":"2025-01-08T15:00:24.78899Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<br>\n\n<a id=\"eda\"></a>\n\n<h1 style=\"font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #48B0F7;\" id=\"eda\">5&nbsp;&nbsp;EXPLORATORY DATA ANALYSIS&nbsp;&nbsp;&nbsp;&nbsp;<a style=\"text-decoration: none; color: #203354;\" href=\"#toc\">&#10514;</a></h1>\n\n<br>\n","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">5.1 KPRIZE DATA</h3>\n<hr>\n\n<br>\n\nFrom the competition readme and our earlier investigation we know that the dataframe contains the following:\n\n* **`instance_id`** (string)\n    * Unique string identifier for each instance (GitHub issue)\n* **`repo`** (string)\n    * The GitHub repository relevant to the issue\n    * Also accessible through the evaluation API\n* **`problem_statement`** (string)\n    * Textual description of the issue\n    * Also accessible through the evaluation API\n* **`patch`** (string)\n    * The patch that resolves the issue\n    * *Only provided in the train set*\n* **`test_patch`** (string)\n    * The patch that resolves the issue\n    * *Only provided in the train set*\n* **`pull_number`** (int)\n    * The pull request number that resolved the issue\n* **`base_commit`** (string)\n    * The commit used as the foundation for the provided repository copy\n* **`issue_numbers`** (int)\n    * The original ID number of the GitHub issue\n* **`[PASS_TO_PASS/FAIL_TO_PASS]`** (list)\n    * Lists containing unit tests to be executed for this issue\n\n<center><div class=\"alert alert-block alert-warning\" style=\"margin: 2em; line-height: 1.7em;\">\n    <b style=\"font-size: 16px; font-weight: 900;\">🔔 &nbsp; NOTE &nbsp; 🔔</b><br><br><span>Since we only have <b>5 examples</b>, we won’t get extensive statistical insights, but we can still carry out some basic exploratory data analysis (EDA) to understand each feature, spot any quirks in the data, and think through potential next steps.</span>\n</div></center>\n\n\n\n","metadata":{}},{"cell_type":"code","source":"rprint(f\"{kprize_df.shape=}\\n\")\n\nrprint(\"\\nAny Missing Values?\\n\")\nrprint(kprize_df.isnull().sum())\n\nrprint(\"\\nDatatypes?\\n\")\nkprize_df.info(show_counts=True)\n\nrprint(\"\\nRepo Distribution?\\n\")\nrprint(kprize_df['repo'].value_counts())\n# display(kprize_df['repo'].value_counts().plot(kind='bar'))\n\n# Fixes can reference more than one GitHub issue.\nrprint(\"\\nNumber of Issues per PR\\n\")\nkprize_df[\"issue_numbers\"].apply(len).value_counts()\n\nkprize_df['problem_statement_length'] = kprize_df['problem_statement'].apply(lambda x: len(x.split()))\nrprint(\"\\nProblem Statement Lengths\\n\")\ndisplay(kprize_df['problem_statement_length'].describe())\n\nrprint(\"\\nPatch Lengths\\n\")\nkprize_df['patch_length'] = kprize_df['patch'].apply(lambda x: len(x))\nkprize_df['test_patch_length'] = kprize_df['test_patch'].apply(lambda x: len(x))\ndisplay(kprize_df[['patch_length', 'test_patch_length']].describe())\n\nrprint(\"\\nTest Counts\\n\")\nkprize_df['PASS_TO_PASS_count'] = kprize_df['PASS_TO_PASS'].apply(len)\nkprize_df['PASS_TO_PASS_count'] = kprize_df['FAIL_TO_PASS'].apply(len)\ndisplay(kprize_df[['PASS_TO_PASS', 'FAIL_TO_PASS', 'PASS_TO_PASS_count', 'PASS_TO_PASS_count']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T15:00:24.791527Z","iopub.execute_input":"2025-01-08T15:00:24.79188Z","iopub.status.idle":"2025-01-08T15:00:24.923819Z","shell.execute_reply.started":"2025-01-08T15:00:24.791846Z","shell.execute_reply":"2025-01-08T15:00:24.9227Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">5.2 SWE-BENCH LITE DATA</h3>\n<hr>\n\n<br>\n\n\nSWE-bench was designed to provide a diverse set of codebase problems that were verifiable using in-repo unit tests. The full SWE-bench test split comprises 2,294 issue-commit pairs across 12 python repositories.\n<br>\n<br>\nSince its release, we've found that for most systems evaluating on SWE-bench, running each instance can take a lot of time and compute. We've also found that SWE-bench can be a particularly difficult benchmark, which is useful for evaluating LMs in the long term, but discouraging for systems trying to make progress in the short term.\n<br>\n<br>\nTo remedy these issues, we've released a canonical subset of SWE-bench called SWE-bench Lite. SWE-bench Lite comprises 300 instances from SWE-bench that have been sampled to be more self-contained, with a focus on evaluating functional bug fixes. SWE-bench Lite covers 11 of the original 12 repositories in SWE-bench, with a similar diversity and distribution of repositories as the original. We perform similar filtering on the SWE-bench dev set to provide 23 development instances that can be useful for active development on the SWE-bench task. We recommend future systems evaluating on SWE-bench to report numbers on SWE-bench Lite in lieu of the full SWE-bench set if necessary. You can find the source code for how SWE-bench Lite was created in <a href=\"https://github.com/princeton-nlp/SWE-bench/tree/main/swebench/collect/make_lite\">SWE-bench/swebench/collect/make_lite</a>.\n<br>\n<br>\nHere's a list of the general criteria we used to select SWE-bench Lite instances:\n\n<ul>\n    <li> We remove instances with images, external hyperlinks, references to specific commit shas and references to other pull requests or issues. </li>\n    <li> We remove instances that have fewer than 40 words in the problem statement. </li>\n    <li> We remove instances that edit more than 1 file. </li>\n    <li> We remove instances where the gold patch has more than 3 edit hunks (see patch). </li>\n    <li> We remove instances that create or remove files. </li>\n    <li> We remove instances that contain tests with error message checks. </li>\n    <li> Finally, we sample 300 test instances and 23 development instances from the remaining instances. </li>\n</ul>\n<br>\n          \nYou can download SWE-bench Lite and its baselines from Hugging Face Datasets:\n\n<ul>\n    <li><a style=\"width: 100%\" href=\"https://huggingface.co/datasets/princeton-nlp/SWE-bench_Lite\">🤗 SWE-bench Lite</a></li>\n    <li><a style=\"width: 100%\" href=\"https://huggingface.co/datasets/princeton-nlp/SWE-bench_Lite_oracle\">🤗 \"Oracle\" Retrieval Lite</a></li>\n    <li><a style=\"width: 100%\" href=\"https://huggingface.co/datasets/princeton-nlp/SWE-bench_Lite_bm25_13K\">🤗 BM25 Retrieval 13K Lite</a></li>\n    <li><a style=\"width: 100%\" href=\"https://huggingface.co/datasets/princeton-nlp/SWE-bench_Lite_bm25_27K\">🤗 BM25 Retrieval 27K Lite</a></li>\n<ul>\n<br>\n\n<div style=\"display: flex; justify-content: center; align-items: flex-start;\">  \n  <!-- First image + text -->\n  <div style=\"max-width: 400px; text-align: center; margin: 2%;\">\n    <img \n      src=\"https://www.swebench.com/img/swebench-lite-pie.png\"\n      style=\"width: 100%; max-width: 400px;\"\n      alt=\"Pie chart for SWE-bench Lite distribution\"\n    >\n    <p>\n      SWE-bench Lite distribution across repositories. Compare to the full SWE-bench \n      in Figure 3 of the \n      <a href=\"https://arxiv.org/abs/2310.06770\">SWE-bench paper</a>.\n    </p>\n  </div>\n  <!-- Second image + text -->\n  <div style=\"max-width: 400px; text-align: center; margin: 2%;\">\n    <img \n      src=\"https://www.swebench.com/img/swe-bench_lite_results.png\"\n      style=\"width: 100%; max-width: 400px;\"\n      alt=\"Bar chart for SWE-bench Lite performance\"\n    >\n    <p>\n      SWE-bench Lite performance for our baselines. Compare to the full SWE-bench baseline \n      performance in Table 5 of the \n      <a href=\"https://arxiv.org/abs/2310.06770\">SWE-bench paper</a>.\n    </p>\n  </div>\n</div>\n","metadata":{}},{"cell_type":"markdown","source":"| Field Name                | Type   | Description                                                                                      |\n|---------------------------|--------|--------------------------------------------------------------------------------------------------|\n| `instance_id`             | str    | A formatted instance identifier, usually as `repo_owner__repo_name-PR-number`.                 |\n| `patch`                   | str    | The gold patch, the patch generated by the PR (minus test-related code), that resolved the issue.|\n| `repo`                    | str    | The repository owner/name identifier from GitHub.                                               |\n| `base_commit`             | str    | The commit hash of the repository representing the HEAD of the repository before the solution PR is applied. |\n| `hints_text`              | str    | Comments made on the issue prior to the creation of the solution PR’s first commit creation date.|\n| `created_at`              | str    | The creation date of the pull request.                                                          |\n| `test_patch`              | str    | A test-file patch that was contributed by the solution PR.                                      |\n| `problem_statement`       | str    | The issue title and body.                                                                       |\n| `version`                 | str    | Installation version to use for running evaluation.                                             |\n| `environment_setup_commit`| str    | Commit hash to use for environment setup and installation.                                      |\n| `FAIL_TO_PASS`            | str    | A JSON list of strings that represent the set of tests resolved by the PR and tied to the issue resolution. |\n| `PASS_TO_PASS`            | str    | A JSON list of strings that represent tests that should pass before and after the PR application.|\n","metadata":{}},{"cell_type":"code","source":"rprint(Markdown(\"---\"), \"\\n\\n[bold magenta] PROBLEM STATEMENT: [/bold magenta]\\n\", Markdown(\"---\"))\nrprint(test_df[test_df.repo==\"scikit-learn/scikit-learn\"].problem_statement.values[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T15:00:24.925449Z","iopub.execute_input":"2025-01-08T15:00:24.925964Z","iopub.status.idle":"2025-01-08T15:00:24.949087Z","shell.execute_reply.started":"2025-01-08T15:00:24.925913Z","shell.execute_reply":"2025-01-08T15:00:24.948043Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3 style=\"font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #203354; background-color: #ffffff;\">5.3 STEP BY STEP SOLVE WITH GEMINI</h3>\n<hr>\n\n<br>\n\n**First we make some helper functions to play with Gemini**\n\nBasically, you realize almost immediately that we are building an agent framework. The framework will use the best Open Weights model we can leverage (or a custom trained one) and anything else we need. For now I'm just going to manually explore. ","metadata":{}},{"cell_type":"code","source":"def format_google_message(\n    role: Literal[\"user\", \"model\"],\n    content: str\n) -> types.Content:\n    \"\"\"Format a single message for the Gemini API.\n    \n    Args:\n        role (str): \n            The role of the message sender ('user' or 'model')\n        content (str): \n            The content of the message\n            \n    Returns:\n        types.Content: Formatted message ready for the Gemini API\n        \n    Raises:\n        ValueError: If role is not 'user' or 'model'\n    \"\"\"\n    if role not in [\"user\", \"model\"]:\n        raise ValueError(\"Role must be either 'user' or 'model'\")\n        \n    return types.Content(\n        role=role,\n        parts=[types.Part.from_text(content)]\n    )\n\ndef process_google_message_history(\n    message_history: list[dict[str, str]],\n    model_config: dict[str, Any]\n) -> list[types.Content]:\n    \"\"\"Process a list of historical messages and extract any system instructions.\n    \n    Args:\n        message_history (list[dict[str, str]]): \n            A list of previous messages, where each message is a dictionary with:\n                - 'role' or 'type': \n                    Origin of message ('user', 'model', or 'system')\n                - 'parts' or 'content': \n                    The message content\n        model_config (dict[str, Any]): \n            Configuration dictionary that may be modified if system instructions are found\n            \n    Returns:\n        list[types.Content]: \n            Processed message history ready for the Gemini API\n        \n    Example:\n        >>> history = [\n        ...     {\"role\": \"user\", \"content\": \"Hello\"},\n        ...     {\"role\": \"model\", \"content\": \"Hi there!\"},\n        ...     {\"role\": \"system\", \"content\": \"Be concise\"}\n        ... ]\n        >>> config = {}\n        >>> processed = process_message_history(history, config)\n        >>> print(len(processed))  # 2 (system message removed)\n        >>> print(config)          # {'system_instruction': 'Be concise'}\n    \"\"\"\n    # Initialize\n    formatted_messages = []\n    \n    # Iterate over and extract role and content with support for fallback key values.\n    for msg in message_history:\n        role = msg.get(\"role\", msg.get(\"type\", \"user\"))\n        content = msg.get(\"parts\", msg.get(\"content\", \"\"))\n        \n        # Handle system instructions\n        if \"system\" in role.lower() and content:\n            rprint(\"[bold red]Detected system instruction message, overriding any provided system instruction[/bold red]\")\n            model_config[\"system_instruction\"] = content\n            continue\n            \n        # Add regular messages to history\n        try:\n            formatted_messages.append(format_google_message(role, content))\n        except ValueError as e:\n            raise ValueError(f\"Invalid message in history: {e}\")    \n    return formatted_messages\n\ndef to_gemini(\n    message: str,\n    message_history: list[dict[str, str]] | None = None,\n    model_name: str = \"gemini-2.0-flash-exp\",\n    stream: bool = True,\n    system_instruction: str | None = None,\n    **model_config: Any\n) -> str:\n    \"\"\"Send a message to Google's Gemini model and get the response.\n\n    Args:\n        message (str): \n            The message to send to Gemini\n        message_history (list[dict[str, str]], optional): \n            Previous messages in the conversation. Each message should be a dictionary with:\n                - 'role' or 'type': Origin of message ('user', 'model', or 'system')\n                - 'parts' or 'content': The message content\n        model_name (str, optional): \n            Name of the Gemini model to use\n            Defaults to gemini-2.0-flash-exp (available in the free tier model)\n        stream (bool, optional): \n            Whether to stream the response with live formatting\n        system_instruction (str, optional):\n            Instructions for how the model should behave\n            Can also be provided via a system message in message_history\n        **model_config (Any, optional): \n            Additional configuration parameters for the model\n\n    Returns:\n        str: The model's response text\n        \n    Example:\n        >>> # Simple usage\n        >>> response = to_gemini(\"What's the weather like?\")\n        \n        >>> # With message history and system instruction\n        >>> history = [\n        ...     {\"role\": \"user\", \"content\": \"Hi, I'm Darien, the weather is beautiful here.\"},\n        ...     {\"role\": \"model\", \"content\": \"Hello Darien! Thanks for letting me know!\"}\n        ... ]\n        >>> response = to_gemini(\n        ...     message=\"What's something that rhymes with my name?\",\n        ...     message_history=history,\n        ...     system_instruction=\"Be concise and write all names in bold\",\n        ...     temperature=0.7\n        ... )\n        \n        >>> # Without streaming, using a different model\n        >>> response = to_gemini(\n        ...     \"Explain quantum computing\",\n        ...     stream=False,\n        ...     model_name=\"gemini-1.5-pro\"\n        ... )\n    \"\"\"\n    # (0) Get Fresh Client\n    client = genai.Client(api_key=os.getenv(\"GEMINI_API_KEY\"))\n\n    # (1) Initialize config and process system instruction\n    if system_instruction:\n        model_config[\"system_instruction\"] = system_instruction\n        \n    # (2) Process message history if provided\n    contents = []\n    if message_history:\n        contents.extend(process_google_message_history(message_history, model_config))\n        \n    # (3) Add the current message\n    contents.append(format_google_message(\"user\", message))\n\n    # (4) Create generation config\n    config = types.GenerateContentConfig(**model_config) if model_config else None\n\n    # (5) Generate response\n    if stream:\n        response = client.models.generate_content_stream(\n            model=model_name,\n            contents=contents,\n            config=config\n        )\n        return stream_with_styling(response)\n    \n    # else\n    reponse = client.models.generate_content(\n        model=model_name,\n        contents=contents,\n        config=config\n    )\n    return response.text\n\n# Showcase the tool\nresponse = json.loads(to_gemini(\n    message=\"Write a haiku about what we've discussed so far?\",\n    message_history=[{\"role\": \"user\", \"content\": \"Hi, I'm Darien, the weather is beautiful here.\"}, {\"role\": \"model\", \"content\": \"Hello Darien! Thanks for letting me know!\"}],\n    system_instruction=\"Be concise and respond in JSON format\",\n    temperature=0.7, \n    response_mime_type=\"application/json\",\n))\nresponse","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T15:03:29.68761Z","iopub.execute_input":"2025-01-08T15:03:29.688103Z","iopub.status.idle":"2025-01-08T15:03:31.38521Z","shell.execute_reply.started":"2025-01-08T15:03:29.688057Z","shell.execute_reply":"2025-01-08T15:03:31.383944Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Next we look at an example problem from the kprize data** ","metadata":{}},{"cell_type":"code","source":"DEMO_IDX = 3\nDEMO_ROW = kprize_df.iloc[DEMO_IDX]\nDEMO_PROBLEM_STATEMENT = DEMO_ROW[\"problem_statement\"]\nDEMO_PASS_TO_PASS_TESTS = DEMO_ROW[\"PASS_TO_PASS\"]\nDEMO_FAIL_TO_PASS_TESTS = DEMO_ROW[\"FAIL_TO_PASS\"]\nDEMO_REPO_PATH = os.path.join(COMP_REPOS_DIR, f'repo__{DEMO_ROW[\"instance_id\"]}')\n\n# tree = get_directory_tree(DEMO_REPO_PATH)\n# rprint(tree)\n\nrprint(DEMO_PROBLEM_STATEMENT)\n\ndisplay(pd.DataFrame(DEMO_ROW).T)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T15:23:35.276646Z","iopub.execute_input":"2025-01-08T15:23:35.277154Z","iopub.status.idle":"2025-01-08T15:23:35.321862Z","shell.execute_reply.started":"2025-01-08T15:23:35.277118Z","shell.execute_reply":"2025-01-08T15:23:35.320462Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_lines_from_file(file_path: str, line_range: str | None = None, as_list: bool = False) -> str:\n    \"\"\"Returns the lines of code from a specified file and line range.\n\n    Args:\n        file_path (str):\n            The path to the file from which to retrieve lines.\n        line_range (str, optional):\n            A string representing the range of lines in the format \"start-end\".\n            For example, \"10-20\" means lines 10 through 20, inclusive (1-based indexing).\n            When not provided all lines are returned.\n        as_list (bool, optional):\n            Keep the lines separate.\n\n    Returns:\n        str:\n            A concatenated string of lines from the specified range. If the file\n            is not found or if the line range is invalid, returns an error message string.\n    \"\"\"\n    if not os.path.isfile(file_path):\n        return f\"[Error] File not found: {file_path}\"\n\n    with open(file_path, 'r', encoding='utf-8') as f:\n        snippets = f.readlines()\n\n    if line_range:\n        try:\n            if \"-\" not in line_range:\n                line_range=f\"{line_range.strip()}-{line_range.strip()}\"\n            start_line, end_line = [int(x.strip()) for x in line_range.split(\"-\")]\n            snippets = snippets[start_line-1 : end_line]\n        except ValueError:\n            return f\"[Error] Invalid line range: {line_range}. Must be either a single integer or two integers delimited by a single dash.\"\n    return ''.join(snippets) if not as_list else snippets\n    \n\ndef _normalize_imports(lines: list[str]) -> list[str]:\n    \"\"\"Processes a list of lines to collect and normalize all import statements.\n\n    Args:\n        lines (list[str]):\n            The lines of code from which to extract import statements.\n\n    Returns:\n        list[str]:\n            A sorted list of unique import statements, each as a single line.\n    \"\"\"\n    imports = set()\n    current_import = []\n    inside_multiline = False\n\n    import_pattern = re.compile(r'^\\s*(from\\s+\\S+\\s+import\\s+|import\\s+)')\n\n    for line in lines:\n        stripped = line.strip()\n\n        if inside_multiline:\n            # Check if we are inside a multi-line import and process continuation\n            if stripped.endswith(')'):\n                current_import.append(stripped[:-1])\n                imports.add(' '.join(' '.join(current_import).split()))\n                current_import = []\n                inside_multiline = False\n            else:\n                current_import.append(stripped)\n            continue\n\n        match = import_pattern.match(line)\n        if match:\n            if stripped.endswith('('):\n                # Begin multi-line import\n                current_import.append(stripped[:-1])\n                inside_multiline = True\n            elif '(' in stripped and ')' in stripped:\n                # Handle single-line imports with parentheses\n                base_import, items = stripped.split('(', 1)\n                items = items.rstrip(')').split(',')\n                for item in items:\n                    imports.add(f\"{base_import.strip()} {item.strip()}\")\n            else:\n                # Single-line import\n                imports.add(stripped)\n\n    return sorted(imports)\n\n\ndef _collect_imports_from_lines(lines: list[str]) -> list[str]:\n    \"\"\"Collects import statements from a list of lines.\n\n    Args:\n        lines (list[str]):\n            The lines of code in which to search for import statements.\n\n    Returns:\n        list[str]:\n            A list of import statements, each stripped of trailing newlines.\n    \"\"\"\n    imports = []\n    import_pattern = re.compile(r'^\\s*(?:import|from)\\s+')\n    for line in lines:\n        if import_pattern.match(line):\n            imports.append(line.rstrip('\\n'))\n    return _normalize_imports(imports)\n\n\ndef search_code(\n    root_directory: str,\n    search_string: str,\n    n_lines_before: int = 0,\n    n_lines_after: int = 0,\n    return_imports: bool = False\n) -> list[dict[str, str | int | list[str]]]:\n    \"\"\"Searches for a given string in all .py files under root_directory.\n    \n    Optionally returning surrounding lines (context) and import statements from matching files.\n\n    Args:\n        root_directory (str):\n            The path to the root directory of the codebase to search.\n        search_string (str):\n            The string to search for in .py files.\n        n_lines_before (int, optional):\n            Number of lines of context to include before each match.\n            Defaults to 0.\n        n_lines_after (int, optional):\n            Number of lines of context to include after each match.\n            Defaults to 0.\n        return_imports (bool, optional):\n            Whether to collect and return all import statements found in each file\n            that contains at least one match. Defaults to False.\n\n    Returns:\n        list[dict[str, str | int | list[str]]]:\n            A list of dictionaries, each containing:\n              - 'file': The path to the file containing the match\n              - 'line': The line number of the match in that file (1-based)\n              - 'content': The exact line content for that match\n              - 'context_before': A list of lines preceding the match\n              - 'context_after': A list of lines following the match\n              - 'imports': A list of import statements (only if return_imports=True)\n    \"\"\"\n    matches: list[dict[str, str | int | list[str]]] = []\n    for dirpath, _, filenames in os.walk(root_directory):\n        for filename in filenames:\n            if filename.endswith('.py'):\n                full_path = os.path.join(dirpath, filename)\n                with open(full_path, 'r', encoding='utf-8') as f:\n                    lines = f.readlines()\n\n                file_imports = _collect_imports_from_lines(lines) if return_imports else []\n\n                for i, line in enumerate(lines, start=1):\n                    if search_string in line:\n                        start_idx = max(0, i - 1 - n_lines_before)\n                        end_idx = min(len(lines), i - 1 + n_lines_after)\n                        context_before = [l.rstrip('\\n') for l in lines[start_idx:i - 1]]\n                        context_after = [l.rstrip('\\n') for l in lines[i:end_idx]]\n\n                        match_entry = {\n                            'file': full_path,\n                            'line': i,\n                            'content': line.rstrip('\\n'),\n                            'context_before': context_before,\n                            'context_after': context_after\n                        }\n\n                        if return_imports:\n                            match_entry['imports'] = file_imports\n                        matches.append(match_entry)\n    return matches\n\n\ndef _extract_entire_definition(\n    lines: list[str],\n    start_index: int\n) -> list[str]:\n    \"\"\"Extracts the entire definition body (function or class) starting at a given line.\n\n    Args:\n        lines (list[str]):\n            The full list of lines from the file (unmodified).\n        start_index (int):\n            The index (0-based) of the line where the definition ('def' or 'class')\n            was found.\n\n    Returns:\n        list[str]:\n            A list of lines comprising the entire definition block.\n    \"\"\"\n    definition_lines: list[str] = []\n    base_indent = len(lines[start_index]) - len(lines[start_index].lstrip())\n    top_def_pattern = re.compile(r'^\\s*(def|class)\\s+')\n\n    current_index = start_index\n    while current_index < len(lines):\n        line = lines[current_index]\n        current_indent = len(line) - len(line.lstrip())\n\n        if (current_index > start_index and top_def_pattern.match(line) and current_indent <= base_indent):\n            break\n\n        definition_lines.append(line)\n        current_index += 1\n\n    return definition_lines\n\n\ndef _extract_entire_definition(lines: list[str], start_index: int) -> list[str]:\n    \"\"\"Extracts all lines in the definition block (function or class). \n    \n    This is starting at start_index and continuing until we reach a line with \n    less-or-equal indentation that indicates the next top-level definition, or the end of file.\n\n    Args:\n        lines (list[str]):\n            The lines containing the entirety of the class definition.\n        start_index (int):\n            Where we will start checking from looking for the relevant information.\n\n    Returns:\n        A list of strings representing the lines for a given definition block (function/class/method)\n    \"\"\"\n    definition_block = []\n    initial_indent = _get_indent_level(lines[start_index])\n    definition_block.append(lines[start_index])\n    # Gather everything that's part of this definition’s indentation\n    for idx in range(start_index + 1, len(lines)):\n        line = lines[idx]\n        if line.strip() == '':\n            # Blank lines inside the definition are included\n            definition_block.append(line)\n            continue\n        if _get_indent_level(line) <= initial_indent and re.match(r'^\\s*(def|class)\\s+', line):\n            # Found a new top-level definition\n            break\n        definition_block.append(line)\n    return definition_block\n\n\ndef _get_indent_level(line: str) -> int:\n    \"\"\"Utility to count the number of leading spaces in a line.\n\n    leading_spaces = (line-length minus (line-length minus non-prefixing-spaces))\n    \n    Args:\n        line (str): The line of code\n\n    Returns:\n        The number of leading spaces \n    \"\"\"\n    return len(line) - len(line.lstrip(' '))\n\n    \ndef _find_method_block_in_lines(\n    block_lines: list[str], \n    method_name: str\n) -> tuple[int, int] | None:\n    \"\"\"Within a block of lines (e.g. a class block), find the start and end line indices (inclusive).\n    \n    This is used to be able to allow for effective retrieval of method code from a file.\n    For example, the definition for 'def method_name(...)' may exist within a class.\n\n    Args:\n        block_lines (list[str]):\n            The line by line strings making up the class definition\n        method_name (str):\n            The name of the method to be extracted.\n\n    Returns:\n        The start and end line indices (if found and inclusive) for the method.\n    \"\"\"\n    pattern = re.compile(rf'^\\s*def\\s+{re.escape(method_name)}\\s*\\(')\n    for i, line in enumerate(block_lines):\n        if pattern.search(line):\n            # Found the start. Now find where it ends by indentation.\n            start_idx = i\n            init_indent = _get_indent_level(line)\n            # Move forward to find where this method ends.\n            for j in range(i + 1, len(block_lines)):\n                if block_lines[j].strip() == '':\n                    continue\n                if _get_indent_level(block_lines[j]) <= init_indent and re.match(r'^\\s*(def|class)\\s+', block_lines[j]):\n                    # Reached the next method/class -> end of this method’s block\n                    return (start_idx, j - 1)\n            return (start_idx, len(block_lines) - 1)  # Goes until end of block\n    return None\n\ndef _extract_class_up_to_init_or_method(\n    lines: list[str],\n    class_index: int,\n    method_name: str\n) -> list[str]:\n    \"\"\"Grab the first part of a class definition up to the point at which initialization has completed.\n\n    (1) Extract the entire class definition at class_index (using _extract_entire_definition).\n    (2) Within that class block, find the __init__ block (if any) and the block for method_name (if any).\n    (3) Return lines from the start of the class up through the furthest end of either __init__ or the method.\n\n    Args:\n        lines (list[str]):\n            The lines of code containing the class definition.\n        class_index (int):\n            The starting point of the class (indexable) for the definition within the lines.\n        method_name (str):\n            The method we want to retrieve (in addition to the initialization code)\n    \n    Returns:\n        list[str]:\n            The relevant lines as a list of strings.\n    \"\"\"\n    class_block = _extract_entire_definition(lines, class_index)\n    # Look for __init__ and the target method\n    init_block_bounds = _find_method_block_in_lines(class_block, '__init__')\n    method_block_bounds = _find_method_block_in_lines(class_block, method_name)\n\n    # If neither __init__ nor method is found, we just return the whole class\n    if not init_block_bounds and not method_block_bounds:\n        return class_block\n\n    furthest_line = 0\n    if init_block_bounds:\n        furthest_line = max(furthest_line, init_block_bounds[1])\n    if method_block_bounds:\n        furthest_line = max(furthest_line, method_block_bounds[1])\n\n    # Slice from start of the class block up to furthest_line\n    return class_block[:furthest_line + 1]\n\n\ndef _parse_class_and_method(object_name: str) -> tuple[str | None, str]:\n    \"\"\"Get the class and method names separately from an object if applicable.\n\n    For example, for the Cat class with method _meow:\n        - If object_name = \"Cat._meow\", returns (\"Cat\", \"_meow\").\n        - Otherwise (object_name=\"Cat\"), returns (None, object_name) if there's no dot.\n        \n    Args:\n        object_name (str):\n            The string containing the object name, one of:\n                - Class Name: 'Cat'\n                - Method Name: '_meow'\n                - Function Name: make_cat_meow\n                - Method With Class Prefix: Cat._meow\n\n    Returns:\n        tuple[str | None, str]:\n            - The method name (or None if no dot found) \n            - followed by the class name within a tuple\n    \"\"\"\n    # You could make this more robust, e.g., handle multiple dots\n    # or disallow multiple dots. Adjust as you see fit.\n    if '.' in object_name:\n        parts = object_name.split('.', 1)  # split on first dot\n        if len(parts) == 2:\n            return parts[0], parts[1]  # class_name, method_name\n    return None, object_name  # No dot -> treat entire string as the object\n\ndef get_object_definition(\n    root_directory: str,\n    object_name: str,\n    return_imports: bool = False\n) -> dict[str, str | int | list[str]] | None:\n    \"\"\"Searches the codebase for the first definition of a function, class, or method matching object_name.\n\n    If object_name is a method referenced with dot notation (e.g. \"Cat._meow\"),\n    then we find class 'Cat', extract the relevant portion of its definition block,\n    and include the method definition plus any __init__.\n\n    Args:\n        root_directory (str):\n            The path to the root directory of the codebase.\n        object_name (str):\n            The name of the function or class to find (e.g., \"my_function\", \"MyClass\", or \"Cat._meow\").\n        return_imports (bool, optional):\n            Whether to collect import statements found in the file.\n\n    Returns:\n        dict[str, str | int | list[str]] | None: \n            A dictionary describing the object definition, or None if not found.\n                - file (str): Path to the file containing the definition.\n                - line (int): The 1-based line number where the definition appears.\n                - content (str): The exact line that matched (the def/class line).\n                - definition_block (list[str]): The extracted lines of the definition.\n                - imports (list[str], optional): The file’s import statements, if return_imports=True.\n    \"\"\"\n    class_name, method_name = _parse_class_and_method(object_name)\n\n    # If we have a separate class_name, we'll do a 2-phase search:\n    #   - Phase A: find the class definition for class_name\n    #   - Phase B: from that block, locate method_name\n    if class_name:\n        # We only search for 'class class_name'\n        class_pattern = re.compile(rf'^\\s*class\\s+{re.escape(class_name)}\\b')\n\n        for dirpath, _, filenames in os.walk(root_directory):\n            for filename in filenames:\n                if filename.endswith('.py'):\n                    full_path = os.path.join(dirpath, filename)\n                    with open(full_path, 'r', encoding='utf-8') as f:\n                        lines = f.readlines()\n                    file_imports = _collect_imports_from_lines(lines) if return_imports else []\n\n                    for i, line in enumerate(lines, start=1):\n                        if class_pattern.search(line.strip()):\n                            # Found the class\n                            class_definition_block = _extract_entire_definition(lines, i - 1)\n                            # Now see if we can find the method inside\n                            # We need to see if method_name is actually a method\n                            # If method_name is '__init__', same logic applies\n                            bounds = _find_method_block_in_lines(class_definition_block, method_name)\n\n                            if bounds is None:\n                                # Possibly there's no such method, but we did find the class\n                                # If you want to return None in this scenario, do so:\n                                # return None\n                                # Otherwise, maybe you still want to return the entire class?\n                                # For now, let's just return None to signal we didn't find a method\n                                continue\n\n                            # We do have a method -> let's figure out the extended range.\n                            init_bounds = _find_method_block_in_lines(class_definition_block, '__init__')\n                            furthest_line = max(bounds[1], init_bounds[1] if init_bounds else 0)\n                            final_block = class_definition_block[: furthest_line + 1]\n\n                            # Return the combined class + method snippet\n                            result = {\n                                'file': full_path,\n                                'line': i,\n                                'content': line.rstrip('\\n'),  # The class definition line\n                                'definition_block': [l.rstrip('\\n') for l in final_block],\n                            }\n                            if return_imports:\n                                result['imports'] = file_imports\n                            return result\n        # If we exit all loops without finding anything, return None\n        return None\n\n    else:\n        # class_name is None -> (handle \"def object_name\" or \"class object_name\")\n        pattern = re.compile(rf'^\\s*(?:def|class)\\s+{re.escape(method_name)}\\b')\n\n        for dirpath, _, filenames in os.walk(root_directory):\n            for filename in filenames:\n                if filename.endswith('.py'):\n                    full_path = os.path.join(dirpath, filename)\n\n                    with open(full_path, 'r', encoding='utf-8') as f:\n                        lines = f.readlines()\n\n                    file_imports = _collect_imports_from_lines(lines) if return_imports else []\n\n                    for i, line in enumerate(lines, start=1):\n                        if pattern.search(line.strip()):\n                            stripped = line.strip()\n                            if stripped.startswith(f'class {method_name}'):\n                                # It's a class definition -> old approach\n                                definition_block = _extract_entire_definition(lines, i - 1)\n                            else:\n                                # It's a def -> could be a top-level function or a method\n                                def_indent = _get_indent_level(line)\n                                class_line_idx = None\n                                for rev_idx in range(i - 2, -1, -1):\n                                    if lines[rev_idx].lstrip().startswith('class '):\n                                        class_indent = _get_indent_level(lines[rev_idx])\n                                        if class_indent < def_indent:\n                                            class_line_idx = rev_idx\n                                            break\n\n                                if class_line_idx is None:\n                                    # Top-level function\n                                    definition_block = _extract_entire_definition(lines, i - 1)\n                                else:\n                                    # It's a method. Extract the class portion up to end of __init__ or the method\n                                    definition_block = _extract_class_up_to_init_or_method(\n                                        lines, class_line_idx, method_name\n                                    )\n\n                            # Build the final result\n                            result = {\n                                'file': full_path,\n                                'line': i,\n                                'content': line.rstrip('\\n'),\n                                'definition_block': [l.rstrip('\\n') for l in definition_block],\n                            }\n                            if return_imports:\n                                result['imports'] = file_imports\n\n                            return result\n\n        return None\n\n\ndef process_instructions(\n    json_instructions: str | dict,\n    root_directory: str,\n    search_kwargs: Any | None = None,\n    lookup_kwargs: Any | None = None,\n) -> list[dict[str, Any]]:\n    \"\"\"Parses JSON instructions and processes each step to retrieve code snippets or definitions.\n\n    Args:\n        json_instructions (str | dict):\n            The instructions, either as a JSON string or a Python dictionary.\n        root_directory (str):\n            The path to the root directory of the codebase.\n\n    Returns:\n        list[dict[str, Any]]: A list of dictionaries containing the results of each step.\n    \"\"\"\n    instructions = json_instructions\n    if isinstance(instructions, str):\n        instructions = json.loads(instructions)\n        \n    # Initialize\n    search_kwargs = search_kwargs if search_kwargs else {}\n    lookup_kwargs = lookup_kwargs if lookup_kwargs else {}\n    next_steps = instructions.get('clear_next_steps', [])\n    results = []\n\n    for step in next_steps:\n\n        # Initialize the step result\n        result = {}\n\n        if 'search' in step:\n            search_string = step['search']\n            search_results = search_code(\n                root_directory,\n                search_string,\n                n_lines_before=search_kwargs.get('n_lines_before', 0),\n                n_lines_after=search_kwargs.get('n_lines_after', 0),\n                return_imports=search_kwargs.get('return_imports', False)\n            )\n            result['search'] = search_string\n            result['results'] = search_results\n\n        elif 'file' in step and 'lines' in step:\n            file_path = step['file'] if os.path.isfile(step['file']) else os.path.join(root_directory, step['file'])\n            line_range = step['lines']\n            snippet = get_lines_from_file(file_path, line_range)\n            result['file'] = file_path\n            result['lines'] = line_range\n            result['snippet'] = snippet\n\n        elif 'object' in step:\n            object_name = step['object']\n            definition = get_object_definition(\n                root_directory, \n                object_name,\n                return_imports=lookup_kwargs.get('return_imports', False)\n            )\n            result['object'] = object_name\n            result['definition'] = definition\n\n        results.append(result)\n\n    return results\n\n# rprint(\"[bold cyan]Demo Code Search[/bold cyan]\")\n# search_code_results = search_code(DEMO_REPO_PATH, \"yield from nodes[0]._infer(context, **kwargs)\", return_imports=True, n_lines_before=10, n_lines_after=10)\n# rprint(search_code_results)\n\n# rprint(\"[bold cyan]Demo Object Search[/bold cyan]\")\n# object_def_results = get_object_definition(DEMO_REPO_PATH, \"Node\", return_imports=True)\n# rprint(object_def_results)\n\n# rprint(\"[bold cyan]Get Lines from File[/bold cyan]\")\n# line_results = get_lines_from_file(os.path.join(DEMO_REPO_PATH, \"astroid/nodes/node_classes.py\"), \"150-175\")\n# rprint(line_results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T15:46:27.28766Z","iopub.execute_input":"2025-01-08T15:46:27.288084Z","iopub.status.idle":"2025-01-08T15:46:27.342579Z","shell.execute_reply.started":"2025-01-08T15:46:27.288048Z","shell.execute_reply":"2025-01-08T15:46:27.341414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"STEP_1_PROMPT = \"\"\"You are a brilliant software engineer tasked with solving github issues in a reproducible and logical step-by-step way.\n\nContext: \n  - We have a large code repository and a specific GitHub issue (pasted below).\n  - The codebase is too big to share in full, so you must work incrementally. \n  - I can provide you with specific code snippets, files, functions, or lines of code on demand if you tell me which files/lines/keywords you want.\n  - I cannot provide you with access to the internet or previous Github commits/issues/PRs.\n  - I will provide the things you ask for in the section titled Previously Requested Information.\n\nYour Task:\n  - Read the issue text (below) carefully.\n  - Summarize the problem in your own words making sure to understand how the requested information helps you and reframes the issue.\n  - Outline a plan to investigate and solve the issue. This plan does not have to be complete, as at any time we can review and plan anew.\n  - Describe the next steps in a consistently formatted way (JSON LIST of ACTIONS) that describes what searches to perform or which files/functions or lines of code you might want to see first.\n      - You can only ask for very specific things (for each step you can specify these things in narrowing order --> 'file' --> 'object' --> 'lines'):\n          - Specific file(s)\n          - Specific object(s) (will be attempted if no file is provided and will return the first found instance of the function/method/object)\n          - Specific line(s) of code (requires a specified file)\n          - Search for code (will search the codebase for the specified code string and will return the File, Function, and Line Numbers)\n      - For example:\n          - {{'file': 'util_in_here.py', 'lines': '123-130'}}\n              - This would return --> *the 8 lines from the specified file. whatever they are*\n          - {{'object': 'UtilConfig'}}\n              - We would than take this and do the search and on the next step provide you with {{'file': ..., 'object': ..., 'lines': ...}}\n\nYour Deliverables:\n  - issue_restatement: An illuminating and logical restatement of the problem in your own words taking into consideration the previous steps requested information (if provided).\n  - methodical_plan: A plan for identifying what code or information to request from me next (or in the first place).\n  - clear_next_steps: The structured and properly formatted next steps in order that will allow us to solve the problem together.\n\nFinal Deliverable (Output Only When You Have Solved Everything With Absolute Confidence):\n  - final_code_diff: This is only to be output when you are confident you have a solution to the problem statement. You should output a code_diff with the appropriate format so that it will be able to be applied as a unix patch.\n  \nImportant Notes:\n  - Do not make assumptions about the codebase. \n  - Be thorough, ask for more rather than less. This includes when you ask for specific lines of code, in that case ask for maybe 10 before and 10 after as well (or whatever you think is appropriate).\n  - Do not ask me to reproduce the issue. The issue exists as described by the problem statement below.\n  - If you don’t know where a relevant piece of code might be, propose a strategy to search for it (e.g., searching by function name, references to certain classes, or by keywords).\n  - If the section 'Previously Requested Information' is empty, then this is the first step in the process. \n  - Do not return anything other than the deliverables as a JSON object with the deliverables as keys ('issue_restatement', 'methodical_plan', 'clear_next_steps')\n  - If you return a 'file' string in the clear_next_steps, please ensure it only includes the path up to the package name (i.e. 'openai/openai-python/blob/main/src/openai/_client.py')\n  - You must output your answer in JSON. The clear_next_steps should be formatted as JSON list of dicts mapping strings to strings.\n  - If you think you can solve it, you should.\n\nRelevant GitHub Issue Text (Problem Statement):\n```markdown\n{problem_statement}\n```\n\nPreviously Requested Information:\n```\n{requested_info}\n```\n\"\"\"\nCONVERSATION_HISTORY = []\n\n\ndef process_steps(\n    problem_statement: str, \n    initial_prompt: str, \n    repo_path: str, \n    accumulate_requested_info: bool = True, \n    temperature: float = 0.1, \n    max_steps: int = 10\n) -> dict[str, Any]:\n    \"\"\"A generator function to process steps sequentially.\n    \n    Args:\n        problem_statement (str): The initial problem statement.\n        initial_prompt (str): The prompt template for formatting each step's input.\n        repo_path (str): The path to the repository for processing instructions.\n        accumulate_requested_info (bool, optional): Whether to append previously accumulated request info.\n        temperature (float, optional): The temperature to use\n    \n    Yields:\n        dict[str, Any]: All the information for a given step\n    \"\"\"\n    requested_info = ''\n    for i in range(max_steps):\n        # Prepare the input for this step\n        step_input = initial_prompt.format(problem_statement=problem_statement, requested_info=requested_info)\n        # rprint(f\"Step {i + 1} Input: {step_input}\")\n        \n        # Simulate sending input to Gemini\n        step_output = json.loads(to_gemini(\n            step_input,\n            temperature=temperature,\n            response_mime_type=\"application/json\",\n        ))\n        rprint(f\"Step {i + 1} Output: {step_output}\")\n        \n        # Process and accumulate the instructions\n        processed_instructions = process_instructions(json_instructions=step_output, root_directory=repo_path)\n        _requested_info = f\"\\nSTEP {i+1} REQUESTED INFO\\n{processed_instructions}\"\n        if accumulate_requested_info:\n            requested_info += _requested_info\n        else:\n            requested_info = _requested_info\n        \n        # Yield the requested info for this step\n        yield {\n            \"step_output\": step_output,\n            \"requested_info\": processed_instructions,\n        }\n\nsteps = []\nstep_generator = process_steps(DEMO_PROBLEM_STATEMENT, STEP_1_PROMPT, DEMO_REPO_PATH)\nstep_1_output = next(step_generator)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T15:48:42.18708Z","iopub.execute_input":"2025-01-08T15:48:42.187641Z","iopub.status.idle":"2025-01-08T15:48:47.309514Z","shell.execute_reply.started":"2025-01-08T15:48:42.187569Z","shell.execute_reply":"2025-01-08T15:48:47.308416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DEMO_PROBLEM_STATEMENT","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T16:13:12.461188Z","iopub.execute_input":"2025-01-08T16:13:12.461603Z","iopub.status.idle":"2025-01-08T16:13:12.474727Z","shell.execute_reply.started":"2025-01-08T16:13:12.461568Z","shell.execute_reply":"2025-01-08T16:13:12.473048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"STEP_2_PROMPT = \"\"\"You are a meticulous software engineer tasked with generating a precise diff to fix a GitHub issue. You have already investigated the issue and have all the necessary context to create a solution.\n\nContext:\n  - You have been provided with the relevant code snippets and context from the investigation phase\n  - You must generate a diff that can be applied using the Unix patch command\n  - The diff must follow strict formatting requirements to be valid\n  - You have access to the original problem statement and all previously requested information\n\nYour Task:\n  - Review all the information gathered during the investigation phase\n  - Generate a precise diff that solves the issue\n  - Ensure the diff follows proper Unix patch format\n  - Include only the minimum necessary changes to fix the issue\n  - Validate that the diff format is correct before submitting\n\nRequired Diff Format:\n  - The diff must start with the file path relative to the repository root\n  - Use unified diff format (indicated by '---' and '+++' lines)\n  - Include the @@ notation to indicate line numbers\n  - Use - for removed lines and + for added lines\n  - Example format:\n    ```diff\n    --- a/path/to/file.py\n    +++ b/path/to/file.py\n    @@ -1,3 +1,3 @@\n     unchanged line\n    -removed line\n    +added line\n     unchanged line\n    ```\n\nStructure Of Output:\n  reasoning:\n  ...\n  \n  validation:\n  ...\n  \n  final_diff:\n  ```diff\n  ...\n  ```\n\nYour Deliverables:\n  - reasoning: A clear explanation of how your changes fix the issue\n  - validation: A step-by-step verification that your diff is correctly formatted\n  - final_diff: The complete diff in proper Unix patch format\n\nValidation Checklist:\n  1. File paths are correct and relative to repository root\n  2. Unified diff format is used (---, +++, @@)\n  3. Line numbers in @@ notation are accurate\n  4. Only necessary changes are included\n  5. No trailing whitespace in changed lines\n  6. Proper indentation maintained\n  7. Diff can be applied with Unix patch command\n\nImportant Notes:\n  - Do not make assumptions about code you haven't seen\n  - Include only changes you are confident about based on the investigation\n  - The diff must be applicable using the standard Unix patch command\n  - Verify all file paths match the repository structure\n  - The final_diff must be a complete, properly formatted patch\n  - Do not include any explanatory text within the diff itself\n\nOriginal Problem Statement:\n```markdown\n{problem_statement}\n```\n\nInvestigation Results:\n```\n{investigation_results}\n```\n\n\n\"\"\"\n\nother_relevant_info = get_object_definition(DEMO_REPO_PATH, \"QTable\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T16:08:26.425184Z","iopub.execute_input":"2025-01-08T16:08:26.426123Z","iopub.status.idle":"2025-01-08T16:08:26.567733Z","shell.execute_reply.started":"2025-01-08T16:08:26.426072Z","shell.execute_reply":"2025-01-08T16:08:26.566779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"STEP_2_PROMPT.format(\n        problem_statement=DEMO_PROBLEM_STATEMENT, \n        investigation_results=step_1_output[\"requested_info\"]+[other_relevant_info]\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T16:10:02.371046Z","iopub.execute_input":"2025-01-08T16:10:02.371443Z","iopub.status.idle":"2025-01-08T16:10:02.388826Z","shell.execute_reply.started":"2025-01-08T16:10:02.371412Z","shell.execute_reply":"2025-01-08T16:10:02.387367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"raw_output = to_gemini(\n    message=STEP_2_PROMPT.format(\n        problem_statement=DEMO_PROBLEM_STATEMENT, \n        investigation_results=step_1_output[\"requested_info\"]+[other_relevant_info]\n    ),\n    temperature=0.1,\n    # response_mime_type=\"application/json\",\n)\n# output = json.loads(raw_output)\n# rprint(output[\"final_diff\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T16:08:34.60386Z","iopub.execute_input":"2025-01-08T16:08:34.604255Z","iopub.status.idle":"2025-01-08T16:08:38.988828Z","shell.execute_reply.started":"2025-01-08T16:08:34.604222Z","shell.execute_reply":"2025-01-08T16:08:38.987525Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**I TESTED THIS IN MY LOCAL AND IT ACTUALLY DOES FIX THE ISSUE**\n\nI haven't run the tests yet... but I will","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}