{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84795,"databundleVersionId":10462807,"sourceType":"competition"},{"sourceId":10354138,"sourceType":"datasetVersion","datasetId":6411949},{"sourceId":203811899,"sourceType":"kernelVersion"},{"sourceId":216101634,"sourceType":"kernelVersion"},{"sourceId":120005,"sourceType":"modelInstanceVersion","modelInstanceId":100936,"modelId":121027},{"sourceId":166245,"sourceType":"modelInstanceVersion","modelInstanceId":141458,"modelId":164048},{"sourceId":166247,"sourceType":"modelInstanceVersion","modelInstanceId":141460,"modelId":164048},{"sourceId":166258,"sourceType":"modelInstanceVersion","modelInstanceId":141469,"modelId":164048},{"sourceId":166264,"sourceType":"modelInstanceVersion","modelInstanceId":141475,"modelId":164048},{"sourceId":170579,"sourceType":"modelInstanceVersion","modelInstanceId":145133,"modelId":164048}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":22.341371,"end_time":"2024-12-11T03:22:13.479076","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-12-11T03:21:51.137705","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🏆 Konwinski Prize - AI GitHub Issue Resolver\n\n## Overview\nThis competition challenges us to build an AI system that can resolve real GitHub issues, evaluated on a contamination-free test set collected after submission freeze. The goal is to achieve >90% accuracy on the SWE-bench benchmark.\n\n### Competition Goals\n- Build an AI that can resolve GitHub issues automatically\n- Achieve high accuracy on a new test set collected post-submission\n- Use only open-source code and models\n\n### Evaluation Metric\n```\nscore = (a - b) / (a + b + c)\nwhere:\na = correctly resolved issues\nb = failing issues\nc = skipped issues\n```\n\n### Prizes\n- 1st Place: $50,000 (+ $775,000 if score > 90%)\n- 2nd Place: $20,000\n- 3rd-5th Place: $10,000 each\n- Additional threshold prizes at 30%, 40%, ..., 90%","metadata":{}},{"cell_type":"code","source":"# Install the required Python packages from the provided wheel files\n# The --target option specifies the directory where the packages will be installed\n# The --no-deps option ensures that dependencies are not installed automatically\n# The --no-index option prevents pip from checking the Python Package Index (PyPI)\n\n# Install all wheel files from the specified directory\n!pip install --target=/kaggle/working /kaggle/input/konwinski-prize/kprize_setup/pip_packages/*.whl -q \\\n    --no-deps \\\n    --no-index\n\n# Install the specific kprize wheel file\n# The --no-index option is used again to prevent checking the Python Package Index (PyPI)\n# The --no-deps option is used to avoid installing dependencies automatically\n!pip install --target=/kaggle/working /kaggle/input/konwinski-prize/kaggle_evaluation/../kprize_setup/kprize-1.0.0-py3-none-any.whl -q \\\n    --no-index \\\n    --no-deps\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:24.439431Z","iopub.execute_input":"2025-01-06T14:43:24.439722Z","iopub.status.idle":"2025-01-06T14:43:27.830003Z","shell.execute_reply.started":"2025-01-06T14:43:24.439697Z","shell.execute_reply":"2025-01-06T14:43:27.829144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import necessary libraries\nimport io\nimport os\nimport sys\nimport shutil\nimport pandas as pd\nimport polars as pl\n\n# Add the kprize_setup directory to the system path\n# This allows Python to find and import the kaggle_evaluation module\nsys.path.insert(0, \"/kaggle/input/konwinski-prize/kprize_setup\")\n\n# Import the konwinski_prize_inference_server module from the kaggle_evaluation package\nimport kaggle_evaluation.konwinski_prize_inference_server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:27.831085Z","iopub.execute_input":"2025-01-06T14:43:27.831317Z","iopub.status.idle":"2025-01-06T14:43:34.86506Z","shell.execute_reply.started":"2025-01-06T14:43:27.831296Z","shell.execute_reply":"2025-01-06T14:43:34.8644Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Launch Vllm Serve\n\nThis script serves the LLM model using `vllm` in background mode. The script launches a subprocess that executes the `vllm.scripts.serve` module with specified parameters such as tensor parallelism, RoPE scaling, and auto tool selection. The output and errors are logged into a file for monitoring. \n\nKey features include:\n- Tensor parallel size set to 4 GPUs(For L4)\n- GPU memory utilization set to 99%\n- [RoPE scaling to support a maximum context length of 131,072 tokens](https://huggingface.co/Qwen/Qwen2.5-14B-Instruct#processing-long-texts)\n- Prefix caching and eager execution for optimized model performance\n  - `enable_prefix_caching` will save slice inference time but don't know about accuray loss yet.\n- Auto tool choice\n  - [Vllm Tool Calling](https://docs.vllm.ai/en/latest/usage/tool_calling.html)\n  - [Qwen Function Calling](https://qwen.readthedocs.io/en/latest/framework/function_call.html)\n\n**Make sure you are using Settings > Accelerator > GPU L4 x 4**","metadata":{}},{"cell_type":"code","source":"# Import the subprocess module to manage external processes\nimport subprocess\n\n# Open a log file in write mode to capture the output of the background process\nlog_file = open(\"vllm_output.log\", \"w\")\n\n# Define the path to the model\nmodel_path = \"/kaggle/input/qwen2.5/transformers/14b-instruct/1\"\n\n# Define the command to run the background task\ncommand = [\n    \"python\",  # Specify the Python interpreter\n    \"-m\",  # Run a module as a script\n    \"vllm.scripts\",  # The module to run\n    \"serve\",  # The specific script to execute within the module\n    model_path,  # Path to the model\n    \"--tensor_parallel_size\", \"4\",  # Tensor parallel size\n    \"--gpu_memory_utilization\", \"0.99\",  # GPU memory utilization\n    \"--enforce_eager\",  # Enforce eager execution\n    \"--enable-auto-tool-choice\",  # Enable automatic tool choice\n    \"--tool-call-parser\", \"hermes\",  # Tool call parser\n    \"--enable_prefix_caching\",  # Enable prefix caching\n    \"--rope-scaling\", '{\"factor\": 4.0, \"original_max_position_embeddings\": 32768, \"type\": \"yarn\"}'  # ROPE scaling configuration\n    ## 131,072 context length\n]\n\n# Start the background process\nprocess = subprocess.Popen(command, stdout=log_file, stderr=log_file)\n\n# Print the PID of the background process\nprint(f\"Background process started with PID: {process.pid}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:34.866222Z","iopub.execute_input":"2025-01-06T14:43:34.866614Z","iopub.status.idle":"2025-01-06T14:43:34.872483Z","shell.execute_reply.started":"2025-01-06T14:43:34.866592Z","shell.execute_reply":"2025-01-06T14:43:34.871881Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Launch Elasticsearch\n\nThis script sets up and launches Elasticsearch 8.17.0 in a background process within a Kaggle environment. It ensures the necessary configurations and directories are in place before running the Elasticsearch instance.\n\nKey steps:\n1. **Copy Elasticsearch directory:**  \n   Copies the Elasticsearch 8.17.0 directory from the input location to the working directory.\n2. **Create necessary directories:**  \n   Creates `logs` and `config` directories within the Elasticsearch folder to store logs and configuration files.\n3. **Write `elasticsearch.yml`:**  \n   Uses `%%writefile` to create the `elasticsearch.yml` configuration file in the `config` directory.\n4. **Run Elasticsearch in the background:**  \n   Launches the Elasticsearch executable with the `jupyter` user using a subprocess. Both standard output and error output are redirected to `elasticsearch_output.log`.\n5. **Print PID:**  \n   After the background process starts, the process ID (PID) is printed to confirm successful execution.\n\nReference:\n- https://www.kaggle.com/code/linshokaku/4th-elasticsearch-retrieval-example#launch-elasticsearch","metadata":{}},{"cell_type":"code","source":"# Copy the entire directory from the input path to the working directory\n# The -r option ensures that the copy is recursive, meaning all subdirectories and files are copied\n!cp -r /kaggle/input/elasticsearch-8-17-0 /kaggle/working/\n\n# Create the logs directory within the elasticsearch-8.17.0 directory\n# The -p option ensures that any necessary parent directories are created as well\n!mkdir -p /kaggle/working/elasticsearch-8-17-0/elasticsearch-8.17.0/logs\n\n# Create the config directory within the elasticsearch-8.17.0 directory\n# The -p option ensures that any necessary parent directories are created as well\n!mkdir -p /kaggle/working/elasticsearch-8-17-0/elasticsearch-8.17.0/config","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:34.87341Z","iopub.execute_input":"2025-01-06T14:43:34.873616Z","iopub.status.idle":"2025-01-06T14:43:48.772633Z","shell.execute_reply.started":"2025-01-06T14:43:34.873598Z","shell.execute_reply":"2025-01-06T14:43:48.77166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile /kaggle/working/elasticsearch-8-17-0/elasticsearch-8.17.0/config/elasticsearch.yml\n# ======================== Elasticsearch Configuration =========================\n#\n# NOTE: Elasticsearch comes with reasonable defaults for most settings.\n#       Before you set out to tweak and tune the configuration, make sure you\n#       understand what are you trying to accomplish and the consequences.\n#\n# The primary way of configuring a node is via this file. This template lists\n# the most important settings you may want to configure for a production cluster.\n#\n# Please consult the documentation for further information on configuration options:\n# https://www.elastic.co/guide/en/elasticsearch/reference/index.html\n#\n# ---------------------------------- Cluster -----------------------------------\n#\n# Use a descriptive name for your cluster:\n#\ncluster.name: single-node-cluster\n#\n# ------------------------------------ Node ------------------------------------\n#\n# Use a descriptive name for the node:\n#\nnode.name: single-node\n#\n# Add custom attributes to the node:\n#\n#node.attr.rack: r1\n#\n# ----------------------------------- Paths ------------------------------------\n#\n# Path to directory where to store the data (separate multiple locations by comma):\n#\n#path.data: /path/to/data\n#\n# Path to log files:\n#\n#path.logs: /path/to/logs\n#\n# ----------------------------------- Memory -----------------------------------\n#\n# Lock the memory on startup:\n#\n#bootstrap.memory_lock: true\n#\n# Make sure that the heap size is set to about half the memory available\n# on the system and that the owner of the process is allowed to use this\n# limit.\n#\n# Elasticsearch performs poorly when the system is swapping the memory.\n#\n# ---------------------------------- Network -----------------------------------\n#\n# By default Elasticsearch is only accessible on localhost. Set a different\n# address here to expose this node on the network:\n#\n# network.host: 192.168.0.1\n#\n# By default Elasticsearch listens for HTTP traffic on the first free port it\n# finds starting at 9200. Set a specific HTTP port here:\n#\nhttp.port: 9200\n#\n# For more information, consult the network module documentation.\n#\n# --------------------------------- Discovery ----------------------------------\n#\n# Pass an initial list of hosts to perform discovery when this node is started:\n# The default list of hosts is [\"127.0.0.1\", \"[::1]\"]\n#\ndiscovery.type: single-node\n#\n# Bootstrap the cluster using an initial set of master-eligible nodes:\n#\n#cluster.initial_master_nodes: [\"node-1\", \"node-2\"]\n#\n# For more information, consult the discovery and cluster formation module documentation.\n#\n# ---------------------------------- Various -----------------------------------\n#\n# Allow wildcard deletion of indices:\n#\n#action.destructive_requires_name: false\n\n#----------------------- BEGIN SECURITY AUTO CONFIGURATION -----------------------\n#\n# The following settings, TLS certificates, and keys have been automatically\n# generated to configure Elasticsearch security features on 02-01-2025 08:26:22\n#\n# --------------------------------------------------------------------------------\n\n# Enable security features\nxpack.security.enabled: false\n\nxpack.security.enrollment.enabled: false\n\n# Enable encryption for HTTP API client connections, such as Kibana, Logstash, and Agents\nxpack.security.http.ssl:\n  enabled: false\n  # keystore.path: certs/http.p12\n\n# Enable encryption and mutual authentication between cluster nodes\nxpack.security.transport.ssl:\n  enabled: false\n  # verification_mode: certificate\n  # keystore.path: certs/transport.p12\n  # truststore.path: certs/transport.p12\n# Create a new cluster with the current node only\n# Additional nodes can still join the cluster later\n# cluster.initial_master_nodes: [\"a847b5071daa\"]\n\n# Allow HTTP API connections from anywhere\n# Connections are encrypted and require user authentication\nhttp.host: 0.0.0.0\n\n# Allow other nodes to join the cluster from anywhere\n# Connections are encrypted and mutually authenticated\n#transport.host: 0.0.0.0\n\n#----------------------- END SECURITY AUTO CONFIGURATION -------------------------","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:48.77352Z","iopub.execute_input":"2025-01-06T14:43:48.77377Z","iopub.status.idle":"2025-01-06T14:43:48.787448Z","shell.execute_reply.started":"2025-01-06T14:43:48.773747Z","shell.execute_reply":"2025-01-06T14:43:48.786794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%bash\n# Create a new user named 'jupyter' with a home directory\nuseradd -m jupyter\n\n# Change the ownership of the Elasticsearch directory to the 'jupyter' user and group recursively\nchown jupyter:jupyter -R /kaggle/working/elasticsearch-8-17-0/elasticsearch-8.17.0\n\n# Change the permissions of the Elasticsearch directory to make all files executable recursively\nchmod -R +x /kaggle/working/elasticsearch-8-17-0/elasticsearch-8.17.0\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:48.788075Z","iopub.execute_input":"2025-01-06T14:43:48.788276Z","iopub.status.idle":"2025-01-06T14:43:48.918258Z","shell.execute_reply.started":"2025-01-06T14:43:48.788258Z","shell.execute_reply":"2025-01-06T14:43:48.917588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Open a log file in write mode to capture the output of the background process\nlog_file = open(\"elasticsearch_output.log\", \"w\")\n\n# Define the command to run Elasticsearch as the 'jupyter' user\ncommand = [\n    \"su\",  # Switch user command\n    \"-\",  # Indicates that the command should be run as the specified user\n    \"jupyter\",  # The user to switch to\n    \"-c\",  # Execute the following command as the specified user\n    \"/kaggle/working/elasticsearch-8-17-0/elasticsearch-8.17.0/bin/elasticsearch\"  # The command to start Elasticsearch\n]\n\n# Start the background process\nprocess = subprocess.Popen(command, stdout=log_file, stderr=log_file)\n\n# Print the PID of the background process\nprint(f\"Background process started with PID: {process.pid}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:48.919032Z","iopub.execute_input":"2025-01-06T14:43:48.919249Z","iopub.status.idle":"2025-01-06T14:43:48.924795Z","shell.execute_reply.started":"2025-01-06T14:43:48.919231Z","shell.execute_reply":"2025-01-06T14:43:48.924206Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Codebase Syntax Tree Parsing and Flattening for LLM Tool and Elasticsearch Database\n\nThis script processes a directory of Python files by parsing their source code using the `Tree-sitter` library and converting the syntax trees into a flattened structure for analysis. The output is designed for use by an LLM as a tool to understand the structure of the codebase and for storing in Elasticsearch to support advanced search and query capabilities for code-related tasks.\n\n### Key Functions:\n1. **`traverse_tree(node, source_code)`**  \n   - Recursively traverses the syntax tree, converting each node to a dictionary with details such as type, start and end points, and text content.\n   - Captures child nodes for hierarchical representation.\n\n2. **`parse_python_code(source_code)`**  \n   - Parses the source code using the `Tree-sitter` parser and returns a nested structure representing the syntax tree.\n\n3. **`find_python_files(base_dir)`**  \n   - Scans a given directory recursively to find all Python files (`.py`) and returns their file paths.\n\n4. **`process_codebase(base_dir)`**  \n   - Reads and parses all Python files in the directory and generates syntax trees for each file.\n   - Returns a dictionary where the keys are file paths and the values are parsed syntax trees.\n\n5. **`point_to_dict(point)`**  \n   - Converts a `Point` object (representing line and column positions) into a dictionary format for easier serialization.\n\n6. **`flatten_module_data(data)`**  \n   - Flattens the hierarchical syntax tree into a list of dictionaries for database insertion.\n   - Filters the nodes to include only `function_definition` and `class_definition` types.\n   - Assigns unique IDs (`UUID`) to each node and tracks parent-child relationships to maintain structure in a flattened format.\n   - Processes the children recursively to ensure nested definitions (like inner classes or methods) are included.\n\n## Purpose:\n- **LLM Tool Integration:**  \n  Provides structured information to LLMs for understanding Python codebases, aiding in:\n  - Code documentation generation.\n  - Refactoring assistance by analyzing relationships between functions and classes.\n  - Identifying unused code or undocumented functions.\n  \n- **Elasticsearch Database Usage:**  \n  The flattened data structure is designed to be indexed in an Elasticsearch database to enable:\n  - **Full-text search**: Search across Python functions, classes, and modules using keywords or phrases.\n  - **Hierarchical queries**: Retrieve specific definitions based on relationships (e.g., find all functions within a class).\n  - **Filtering and aggregation**: Perform queries to filter by file path, type (function or class), or code snippet content.\n\n## Elasticsearch Data Model:\nEach flattened data entry is structured as a document for Elasticsearch with the following fields:\n- `id`: Unique identifier (UUID).\n- `parent_id`: Reference to the parent node ID (if applicable).\n- `file_path`: The file path of the source code.\n- `type`: The type of the code block (`function_definition`, `class_definition`).\n- `text`: The source code snippet for the function or class definition.\n- `start_point`: Start position (row, column) of the code block.\n- `end_point`: End position (row, column) of the code block.\n\nReference\n- https://ai.globant.com/wp-content/uploads/2024/11/Whitepaper-Globant-Code-Fixer-Agent.pdf","metadata":{}},{"cell_type":"code","source":"from langchain_core.tools import tool\nimport json\nimport uuid\nimport tree_sitter_python as tspython\nfrom tree_sitter import Language, Parser\nPY_LANGUAGE = Language(tspython.language())\nparser = Parser(PY_LANGUAGE)\n\n# Function to traverse the tree\ndef traverse_tree(node, source_code):\n    \"\"\"Recursively traverse the syntax tree.\"\"\"\n    result = {\n        \"type\": node.type,\n        \"start_point\": node.start_point,\n        \"end_point\": node.end_point,\n        \"text\": source_code[node.start_byte:node.end_byte].decode(\"utf-8\"),\n        \"children\": []\n    }\n    if node.child_count > 0:\n        for child in node.children:\n            result[\"children\"].append(traverse_tree(child, source_code))\n    return result\n\n# Parse Python source code\ndef parse_python_code(source_code):\n    tree = parser.parse(source_code)\n    root_node = tree.root_node\n    return traverse_tree(root_node, source_code)\n\n# Find all Python files in a directory\ndef find_python_files(base_dir):\n    python_files = []\n    for root, _, files in os.walk(base_dir):\n        for file in files:\n            if file.endswith(\".py\"):\n                python_files.append(os.path.join(root, file))\n    return python_files\n\n# Process a directory of Python files\ndef process_codebase(base_dir):\n    results = {}\n    python_files = find_python_files(base_dir)\n    for file_path in python_files:\n        with open(file_path, 'rb') as file:\n            source_code = file.read()\n            results[file_path] = parse_python_code(source_code)\n    return results\n\ndef point_to_dict(point):\n    \"\"\"\n    Convert a Point object to a dictionary\n    \"\"\"\n    return {\"row\": point.row, \"column\": point.column}\n    \n\ndef flatten_module_data(data):\n    \"\"\"\n    Convert hierarchical data into a flattened structure, including the top-level file path.\n    :param data: Hierarchical data (dict)\n    :return: List of flattened data\n    \"\"\"\n    flattened = []\n\n    for module_path, module_data in data.items():\n        # Convert top-level module data to a flat structure\n        module_id = str(uuid.uuid4())\n        \n        # Filter types: only process \"function_definition\" or \"class_definition\"\n        if module_data.get(\"type\") not in [\"module\", \"function_definition\", \"class_definition\"]:\n            continue\n\n        flattened.append({\n            \"id\": module_id,\n            \"parent_id\": None,  # Top-level has no parent\n            \"file_path\": module_path,\n            \"type\": module_data.get(\"type\"),\n            \"text\": module_data.get(\"text\"),\n            \"start_point\": point_to_dict(module_data.get(\"start_point\")),\n            \"end_point\": point_to_dict(module_data.get(\"end_point\"))\n        })\n\n        # Process the children data\n        def process_children(children, current_parent_id):\n            for child in children:\n                # Filter types: only process \"function_definition\" or \"class_definition\"\n                if child.get(\"type\") not in [\"module\", \"function_definition\", \"class_definition\"]:\n                    continue\n\n                child_id = str(uuid.uuid4())\n                flattened.append({\n                    \"id\": child_id,\n                    \"parent_id\": current_parent_id,  # Reference to the immediate parent ID\n                    \"file_path\": module_path,  # Retain the top-level module path\n                    \"type\": child.get(\"type\"),\n                    \"text\": child.get(\"text\"),\n                    \"start_point\": point_to_dict(child.get(\"start_point\")),\n                    \"end_point\": point_to_dict(child.get(\"end_point\"))\n                })\n                process_children(child.get(\"children\", []), child_id)  # Pass the current ID as the next parent ID\n        \n        process_children(module_data.get(\"children\", []), module_id)\n\n    return flattened","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:44:09.672014Z","iopub.execute_input":"2025-01-06T14:44:09.672323Z","iopub.status.idle":"2025-01-06T14:44:09.684255Z","shell.execute_reply.started":"2025-01-06T14:44:09.672299Z","shell.execute_reply":"2025-01-06T14:44:09.683576Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Agent Tool Definition: open_file, edit_file, list_folder, search_code_elements\n\nThis script defines agent tools for interacting with the filesystem and performing codebase searches. It integrates functionalities for reading, editing, and listing file contents and provides Elasticsearch-based search capabilities for analyzing Python code elements. Each tool is implemented using `pydantic` for request validation and error handling.\n\n## Tool Definitions:\n\n### 1. `open_file` Tool\n**Purpose:**  \nOpens a file and returns the content within a specified line range for viewing.  \n\n**Key Features:**\n- Supports reading from the beginning or a specified line number.\n- Handles errors such as file not found, permission issues, and invalid line ranges.\n  \n**Input Fields:**\n- `file_path`: Path to the file to open.\n- `start_line` (optional): Line number to start reading from.\n- `end_line` (optional): Line number to stop reading.\n\n**Output Fields:**\n- `content`: The read content from the file.\n- `error`: Error message, if any.\n\n---\n\n### 2. `edit_file` Tool\n**Purpose:**  \nEdits the content of a file by replacing or inserting lines within a specified range.\n\n**Key Features:**\n- Supports both line replacement and insertion.\n- Validates syntax for Python files after editing to avoid introducing errors.\n\n**Input Fields:**\n- `file_path`: Path to the file to edit.\n- `text`: The new text to be added or replace existing lines.\n- `start_line`: Line to start editing (inclusive).\n- `end_line` (optional): Line to end editing (inclusive).\n\n**Output Fields:**\n- `old_text`: The original content that was replaced.\n- `updated_text`: The newly updated content.\n- `error`: Error message, if any.\n\n---\n\n### 3. `list_folder` Tool\n**Purpose:**  \nLists all files and folders within a specified directory.\n\n**Key Features:**\n- Recursively lists folder contents, excluding `.git` directories.\n- Provides separate lists for files and folders.\n\n**Input Fields:**\n- `folder_path`: Path to the folder to list.\n\n**Output Fields:**\n- `files`: List of file paths.\n- `folders`: List of folder paths.\n- `error`: Error message, if any.\n\n---\n\n### 4. `search_code_elements` Tool\n**Purpose:**  \nPerforms a search for specific code elements (e.g., functions, classes) in an Elasticsearch index using BM25-based relevance scoring.\n\n**Key Features:**\n- Supports searching by `element_type` (e.g., function, class).\n- Enables keyword-based searches on the `text` content.\n- Designed to work with Python code structure indexed in Elasticsearch.\n\n**Input Fields:**\n- `index_name`: Elasticsearch index to search.\n- `element_type` (optional): Type of the code element (`function_definition`, `class_definition`, etc.).\n- `keyword` (optional): Keyword for text-based search.\n\n**Output Fields:**\n- List of search results containing:\n  - `id`: Unique ID of the document.\n  - `file_path`: Path to the file containing the element.\n  - `type`: Type of the code element.\n  - `text`: Content snippet of the code element.\n  - `start_point`, `end_point`: Start and end positions of the code element.\n  - `score`: BM25 score for relevance.\n- Returns a maximum of the top 4 results.\n\n---\n\n### Utility Functions:\n- **`index_data(flattened_data, index_name)`**:  \n  Indexes flattened Python code data into Elasticsearch for searchability.","metadata":{}},{"cell_type":"code","source":"from typing import Dict, Optional, List\nfrom pydantic import BaseModel, Field\nimport os  # Import os module for directory operations\n\n# Define the request model for editing a file\nclass EditFileRequest(BaseModel):\n    \"\"\"Request to edit a file.\"\"\"\n    file_path: Optional[str] = Field(\n        default=None,\n        description=(\n            \"The path to the file that will be edited. If not provided, \"\n            \"THE CURRENTLY OPEN FILE will be edited. If provided, the \"\n            \"file at the provided path will be OPENED and edited, changing \"\n            \"the opened file.\"\n        ),\n    )\n    text: str = Field(\n        ...,\n        description=\"The text that will replace the specified line range in the file.\",\n    )\n    start_line: int = Field(\n        ...,\n        description=(\n            \"The line number at which the file edit will start (REQUIRED). \"\n            \"Inclusive - the start line will be included in the edit. \"\n            \"If you just want to add code and not replace any line, \"\n            \"don't provide end_line field.\"\n        ),\n    )\n    end_line: Optional[int] = Field(\n        default=None,\n        description=(\n            \"The line number at which the file edit will end (REQUIRED). \"\n            \"Inclusive - the end line will be included in the edit. \"\n            \"If you just want to add code and not replace any line, \"\n            \"don't provide this field.\"\n        ),\n    )\n\n# Define the response model for editing a file\nclass EditFileResponse(BaseModel):\n    \"\"\"Response to edit a file.\"\"\"\n    old_text: Optional[str] = Field(\n        default=None,\n        description=(\n            \"The updated changes. If the file was not edited, the original file \"\n            \"will be returned.\"\n        ),\n    )\n    error: Optional[str] = Field(\n        default=None,\n        description=\"Error message if any\",\n    )\n    updated_text: Optional[str] = Field(\n        default=None,\n        description=\"The updated text. If the file was not edited, this will be empty.\",\n    )\n\n# Define the request model for opening a file\nclass OpenFileRequest(BaseModel):\n    \"\"\"Request to open a file.\"\"\"\n    file_path: str = Field(\n        ...,\n        description=\"The path to the file that will be opened.\",\n    )\n    start_line: int = Field(\n        default=None,\n        description=(\n            \"The line number at which the file content will start to be read. \"\n            \"If not provided, the file will be read from the beginning.\"\n        ),\n    )\n    end_line: Optional[int] = Field(\n        default=None,\n        description=(\n            \"The line number at which the file content will stop being read. \"\n            \"If not provided, the file will be read until the end.\"\n        ),\n    )\n\n# Define the response model for opening a file\nclass OpenFileResponse(BaseModel):\n    \"\"\"Response for opening a file.\"\"\"\n    content: Optional[str] = Field(\n        default=None,\n        description=\"The content of the opened file.\",\n    )\n    error: Optional[str] = Field(\n        default=None,\n        description=\"Error message if any.\",\n    )\n\n# Define the tool for opening a file\n@tool(args_schema=OpenFileRequest)\ndef open_file(**kwargs) -> OpenFileResponse:\n    \"\"\"\n    Opens a file in the editor based on the provided file path,\n    If start_line or end_line are provided, the window will be moved after that line. (i.e. 100 lines after the line number will be displayed)\n\n    Can result in:\n    - ValueError: If file_path is not a string or if the file does not exist.\n    - FileNotFoundError: If the file does not exist.\n    - IOError: If there's an issue reading the file.\n    - PermissionError: If the user doesn't have permission to read the file.\n    - IsADirectoryError: If the provided path is a directory.\n    \"\"\"\n    request = OpenFileRequest(**kwargs)\n\n    try:\n        with open(request.file_path, \"r\") as file:\n            lines = file.readlines()\n\n        start_line = request.start_line if request.start_line else 1\n        end_line = request.end_line if request.end_line else len(lines)\n\n        if start_line < 1 or end_line > len(lines):\n            return OpenFileResponse(error=\"Invalid line range.\")\n\n        content = \"\".join(lines[start_line - 1:end_line])\n        return OpenFileResponse(content=content)\n\n    except FileNotFoundError:\n        return OpenFileResponse(error=\"File not found.\")\n    except PermissionError:\n        return OpenFileResponse(error=\"Permission denied.\")\n    except OSError as e:\n        return OpenFileResponse(error=f\"OS error occurred: {str(e)}\")\n\n# Define the tool for editing a file\n@tool(args_schema=EditFileRequest)\ndef edit_file(**kwargs) -> EditFileResponse:\n    \"\"\"\n    Use this tool to edit a file on specific line numbers.\n\n    Please note that THE EDIT COMMAND REQUIRES PROPER INDENTATION.\n\n    Python files will be checked for syntax errors after the edit.\n    If you'd like to add the line '        print(x)' you must fully write\n    that out, with all those spaces before the code!\n\n    If a syntax error is detected, the edit won't be executed. Review the error\n    message and modify your edit command accordingly.\n\n    When start and end lines are the same, the new text is inserted at that line,\n    preserving the original line's content.\n\n    Ex A: Start=End=1, Text: \"print(x)\"\n    Result: Adds \"print(x)\" as first line, rest unchanged.\n\n    Ex B: Start=1, End=3, Text: \"print(x)\"\n    Result: Replaces lines 1,2 and 3 with \"print(x)\", rest unchanged.\n\n    This action edits a specific part of the file, if you want to rewrite the\n    complete file, use `write` tool instead.\"\"\"\n    request = EditFileRequest(**kwargs)\n\n    try:\n        if request.file_path is None:\n            return EditFileResponse(error=\"No file path provided.\")\n\n        with open(request.file_path, \"r\") as file:\n            lines = file.readlines()\n\n        # Adjust end_line if not provided\n        end_line = request.end_line if request.end_line is not None else len(lines)\n\n        if request.start_line < 1 or end_line > len(lines):\n            return EditFileResponse(error=\"Invalid line range.\")\n\n        # Capture the old text\n        old_text = \"\".join(lines[request.start_line - 1:end_line])\n\n        # Replace the specified lines\n        new_lines = lines[:request.start_line - 1] + [request.text + \"\\n\"] + lines[end_line:]\n\n        # Write back to the file\n        with open(request.file_path, \"w\") as file:\n            file.writelines(new_lines)\n\n        return EditFileResponse(old_text=old_text, updated_text=request.text)\n\n    except FileNotFoundError:\n        return EditFileResponse(error=\"File not found.\")\n    except PermissionError:\n        return EditFileResponse(error=\"Permission denied.\")\n    except OSError as e:\n        return EditFileResponse(error=f\"OS error occurred: {str(e)}\")\n\n# Define the request model for listing the contents of a folder\nclass ListFolderRequest(BaseModel):\n    \"\"\"Request to list the contents of a folder.\"\"\"\n    folder_path: str = Field(\n        ...,\n        description=\"The path to the folder whose contents will be listed.\"\n    )\n\n# Define the response model for listing the contents of a folder\nclass ListFolderResponse(BaseModel):\n    \"\"\"Response for listing the contents of a folder.\"\"\"\n    files: Optional[List[str]] = Field(\n        default=None,\n        description=\"List of file names in the folder.\"\n    )\n    folders: Optional[List[str]] = Field(\n        default=None,\n        description=\"List of folder names in the folder.\"\n    )\n    error: Optional[str] = Field(\n        default=None,\n        description=\"Error message if any.\"\n    )\n\n# Define the tool for listing the contents of a folder\n@tool(args_schema=ListFolderRequest)\ndef list_folder(**kwargs) -> ListFolderResponse:\n    \"\"\"\n    Recursively lists the contents of a folder at the provided path, excluding .git folders.\n\n    Can result in:\n    - ValueError: If folder_path is not a string.\n    - FileNotFoundError: If the folder does not exist.\n    - NotADirectoryError: If the path is not a directory.\n    - PermissionError: If the user doesn't have permission to access the folder.\n    \"\"\"\n    request = ListFolderRequest(**kwargs)\n\n    try:\n        if not os.path.isdir(request.folder_path):\n            return ListFolderResponse(error=\"The provided path is not a directory.\")\n\n        # Recursively list all files and folders, excluding .git directories\n        all_files = []\n        all_folders = []\n        for root, dirs, files in os.walk(request.folder_path):\n            # Exclude .git folder\n            dirs[:] = [d for d in dirs if d != '.git']\n            for file in files:\n                all_files.append(os.path.join(root, file))\n            for folder in dirs:\n                all_folders.append(os.path.join(root, folder))\n\n        return ListFolderResponse(files=all_files, folders=all_folders)\n\n    except FileNotFoundError:\n        return ListFolderResponse(error=\"Folder not found.\")\n    except PermissionError:\n        return ListFolderResponse(error=\"Permission denied.\")\n    except OSError as e:\n        return ListFolderResponse(error=f\"OS error occurred: {str(e)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:44:15.289126Z","iopub.execute_input":"2025-01-06T14:44:15.289447Z","iopub.status.idle":"2025-01-06T14:44:15.314052Z","shell.execute_reply.started":"2025-01-06T14:44:15.289422Z","shell.execute_reply":"2025-01-06T14:44:15.31342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pydantic import BaseModel, Field, model_validator\nfrom typing import Optional, Dict\nfrom elasticsearch import Elasticsearch\nfrom elasticsearch.helpers import bulk\n\nclass SearchCodeElementsParams(BaseModel):\n    \"\"\"\n    Model for validating input parameters for the search_code_elements function.\n    \"\"\"\n    element_type: Optional[str] = Field(\n        None, description=\"Type of the code element to search for (e.g., 'function_definition', 'class_definition', 'decorated_definition').\"\n    )\n    keyword: Optional[str] = Field(\n        None, description=\"Keyword to search for in the text field. Uses BM25-based matching.\"\n    )\n    # file_path: Optional[str] = Field(\n    #     None, description=\"Specific module path to filter the search results.\"\n    # )\n    # parent_id: Optional[str] = Field(\n    #     None, description=\"Parent ID to filter related data.\"\n    # )\n    # start_point: Optional[Dict[str, int]] = Field(\n    #     None, description=\"Start position filter in the format {'row': int, 'column': int}.\"\n    # )\n    # end_point: Optional[Dict[str, int]] = Field(\n    #     None, description=\"End position filter in the format {'row': int, 'column': int}.\"\n    # )\n    index_name: str = Field(\n        default=None, description=\"Elasticsearch index name to query from.\"\n    )\n\n\n    @model_validator(mode=\"before\")\n    def validate_element_type(cls, values):\n        \"\"\"\n        Validates that element_type is one of the allowed Tree-sitter types.\n        \"\"\"\n        valid_types = {\n            \"function_definition\", \"class_definition\", \"decorated_definition\"}\n        \n        element_type = values.get(\"element_type\")\n\n        if element_type and element_type not in valid_types:\n            raise ValueError(f\"element_type must be one of {valid_types}. Provided: {element_type}\")\n\n        return values    \n\n@tool(args_schema=SearchCodeElementsParams)\ndef search_code_elements(**kwargs):\n    \"\"\"\n    Searches for code elements (e.g., class, function, parameter) in an Elasticsearch index based on BM25.\n\n    Filters can be applied for start_point and end_point ranges, along with other parameters.\n\n    :param kwargs: Dictionary of search parameters. Must match the fields in SearchCodeElementsParams.\n    :return: List of search results.\n    \"\"\"\n    # Validate and parse parameters using Pydantic\n    params = SearchCodeElementsParams(**kwargs)\n\n    # Base search query\n    query = {\n        \"bool\": {\n            \"must\": [],  # BM25-based search\n            \"filter\": []  # Exact match filters\n        }\n    }\n\n    # Add filters for element type\n    if params.element_type:\n        query[\"bool\"][\"filter\"].append({\n            \"term\": {\n                \"type\": params.element_type\n            }\n        })\n\n    # Add BM25-based keyword search\n    if params.keyword:\n        query[\"bool\"][\"must\"].append({\n            \"match\": {\n                \"text\": params.keyword\n            }\n        })\n\n    # Add filter for module path\n    # if params.file_path:\n    #     query[\"bool\"][\"filter\"].append({\n    #         \"term\": {\n    #             \"file_path\": params.file_path\n    #         }\n    #     })\n\n    # Add filter for parent ID\n    # if params.parent_id:\n    #     query[\"bool\"][\"filter\"].append({\n    #         \"term\": {\n    #             \"parent_id\": params.parent_id\n    #         }\n    #     })\n\n    # Add range filter for start_point\n    # if params.start_point:\n    #     query[\"bool\"][\"filter\"].append({\n    #         \"range\": {\n    #             \"start_point.row\": {\n    #                 \"gte\": params.start_point.get(\"row\", 0),\n    #                 \"lte\": float(\"inf\")\n    #             }\n    #         }\n    #     })\n    #     query[\"bool\"][\"filter\"].append({\n    #         \"range\": {\n    #             \"start_point.column\": {\n    #                 \"gte\": params.start_point.get(\"column\", 0),\n    #                 \"lte\": float(\"inf\")\n    #             }\n    #         }\n    #     })\n\n    # # Add range filter for end_point\n    # if params.end_point:\n    #     query[\"bool\"][\"filter\"].append({\n    #         \"range\": {\n    #             \"end_point.row\": {\n    #                 \"gte\": params.end_point.get(\"row\", 0),\n    #                 \"lte\": float(\"inf\")\n    #             }\n    #         }\n    #     })\n    #     query[\"bool\"][\"filter\"].append({\n    #         \"range\": {\n    #             \"end_point.column\": {\n    #                 \"gte\": params.end_point.get(\"column\", 0),\n    #                 \"lte\": float(\"inf\")\n    #             }\n    #         }\n    #     })\n\n    # Perform Elasticsearch search\n    response = es.search(index=params.index_name, body={\"query\": query})\n\n    # Extract and format search results\n    results = [\n        {\n            \"id\": hit[\"_id\"],\n            \"file_path\": hit[\"_source\"][\"file_path\"],\n            \"type\": hit[\"_source\"][\"type\"],\n            \"text\": hit[\"_source\"][\"text\"],\n            \"start_point\": hit[\"_source\"][\"start_point\"],\n            \"end_point\": hit[\"_source\"][\"end_point\"],\n            \"parent_id\": hit[\"_source\"][\"parent_id\"],\n            \"score\": hit[\"_score\"]  # BM25 score\n        }\n        for hit in response[\"hits\"][\"hits\"][:4] # top-2\n    ]\n    return results\n\ndef index_data(flattened_data, index_name):\n    actions = [\n        {\n            \"_index\": index_name,\n            \"_id\": doc[\"id\"],\n            \"_source\": doc\n        }\n        for doc in flattened_data\n    ]\n    bulk(es, actions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:54.568186Z","iopub.status.idle":"2025-01-06T14:43:54.568477Z","shell.execute_reply":"2025-01-06T14:43:54.568343Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Define Agent\n\nThis script defines an agent using a workflow that integrates LLM-based reasoning and tool usage. The agent is configured to utilize a set of predefined tools for interacting with codebases and making decisions based on user queries and responses. The workflow utilizes `StateGraph` to manage the flow between the agent and the tools, allowing dynamic handling of tool invocations during interactions.\n\n### Key Components:\n\n#### 1. **`get_model()` Function**\n**Purpose:**  \nInitializes and returns an application workflow (`app`) that uses a language model capable of invoking tools during conversation.\n\n**Steps:**\n1. **Environment Setup:**\n   - Sets the `OPENAI_API_KEY` for authentication.\n   - Lists the available tools (`search_code_elements`, `open_file`, `edit_file`, `list_folder`).\n\n2. **Tool Node Creation:**\n   - Creates a `ToolNode` with the specified tools.\n   - Binds the tools to the language model (`ChatOpenAI`), allowing the model to make calls to these tools.\n\n3. **Model Response Logic:**\n   - **`should_continue(state: MessagesState)`**: Determines if the agent should continue tool execution or finish the conversation.\n     - If the last message contains tool calls, transitions to `\"tools\"`.\n     - Otherwise, ends the conversation (`END`).\n   \n   - **`call_model(state: MessagesState)`**: Sends the message state to the model and returns the response message from the model.\n\n4. **Workflow Definition:**\n   - Creates a state graph (`StateGraph`) to manage the agent's conversation flow.\n   - Adds nodes:\n     - `\"agent\"`: Handles LLM responses.\n     - `\"tools\"`: Handles tool executions.\n   - Defines transitions:\n     - Starts at `\"agent\"`.\n     - Cycles between `\"agent\"` and `\"tools\"` based on the response.\n     - Ends when no further tool calls are needed.\n\n5. **Application Compilation:**\n   - The workflow is compiled into an executable `app` that can handle incoming requests and manage interactions.\n\n---\n\n## Tools Integrated:\n- **`search_code_elements`:** Searches Python code in Elasticsearch for specific elements (e.g., classes, functions).\n- **`open_file`:** Opens a file and reads content within a line range.\n- **`edit_file`:** Edits a specific range of lines in a file.\n- **`list_folder`:** Lists the contents of a folder (files and subfolders).\n\n---\n\n## Workflow Flow:\n1. **Start:**  \n   The agent receives the user query.\n2. **Agent Decision:**  \n   The agent decides if tool usage is required.\n3. **Tool Invocation:**  \n   If a tool is needed, the workflow invokes the corresponding tool.\n4. **Response:**  \n   The tool response is processed, and the agent generates a reply.\n5. **End:**  \n   The conversation ends if no further tool usage is necessary.\n\n## Reference\n- https://langchain-ai.github.io/langgraph/how-tos/tool-calling-errors/#using-the-prebuilt-toolnode","metadata":{}},{"cell_type":"code","source":"import os\nfrom typing import Literal\n\nfrom langgraph.graph import StateGraph, MessagesState, START, END\nfrom langgraph.prebuilt import ToolNode\nfrom langchain_openai import ChatOpenAI\n\ndef get_model():\n    os.environ[\"OPENAI_API_KEY\"] = \"api_key\"\n    tools = [search_code_elements, open_file, edit_file, list_folder]\n    tool_node = ToolNode(tools)\n    model_with_tools = ChatOpenAI(base_url=\"http://localhost:8000/v1\",model=model_path).bind_tools(tools, temperature=0)\n       \n    def should_continue(state: MessagesState):\n        messages = state[\"messages\"]\n        last_message = messages[-1]\n        if last_message.tool_calls:\n            print(last_message.tool_calls, \"*\"*100)\n            return \"tools\"\n        return END\n    \n    \n    def call_model(state: MessagesState):\n        messages = state[\"messages\"]\n        response = model_with_tools.invoke(messages)\n        return {\"messages\": [response]}\n    \n    \n    workflow = StateGraph(MessagesState)\n    \n    # Define the two nodes we will cycle between\n    workflow.add_node(\"agent\", call_model)\n    workflow.add_node(\"tools\", tool_node)\n    \n    workflow.add_edge(START, \"agent\")\n    workflow.add_conditional_edges(\"agent\", should_continue, [\"tools\", END])\n    workflow.add_edge(\"tools\", \"agent\")\n    \n    app = workflow.compile()\n    return app","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:54.569201Z","iopub.status.idle":"2025-01-06T14:43:54.56948Z","shell.execute_reply":"2025-01-06T14:43:54.569365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"GITHUB_ISSUE_SOLVER_PROMPT = \"\"\"\nYou are an autonomous software engineer tasked with solving coding issues efficiently and concisely. Your primary role is to coordinate between code analysis and editing tasks. Follow these streamlined guidelines:\n\nYou have access to the following tools:\n- `search_code_elements`: Get information about a specific class or function, including start and end lines.\n- `list_folder`: View the repository structure.\n- `open_file`: Open and view file contents, ideally focusing on a specific range based on `search_code_elements` results.\n- `edit_file`: Make changes to the code.\n\nThe task involves working within the provided **Codebase** and **Codebase Folder List** to modify relevant modules and solve the given issue. Focus only on the necessary steps to avoid unnecessary complexity.\n\n### Instructions:\n1. **Identify the Relevant Module**:\n   - Quickly scan the **Codebase Folder List** and determine the likely location of the issue.\n   - Form a hypothesis about the parts of the code that require investigation based on folder names.\n\n2. **Understand the Issue**:\n   - Review the given issue or bug report concisely.\n   - Formulate a hypothesis about the root cause and a potential solution.\n   - Minimize analysis noise and focus directly on solving the issue.\n\n3. **Explore the Codebase**:\n   - Use `list_folder` to confirm the current folder structure.\n   - Use `search_code_elements` to locate relevant code elements (e.g., functions, classes) and determine their start and end lines.\n   - Use `open_file` to view specific sections of the code, focusing on the range provided by `search_code_elements`.\n\n4. **Code Analysis and Editing**:\n   - Use `open_file` to locate the target code section based on the results from `search_code_elements` and review its contents.\n   - Use `edit_file` to make precise, minimal changes that address the issue.\n   - Ensure that your edits preserve existing functionality and syntax.\n   - Pay close attention to line numbers, indentation, and syntax.\n\n5. **Problem-Solving Approach**:\n   - Break down complex problems into smaller tasks as needed but avoid verbose explanations.\n   - Continuously monitor progress and adapt only if required.\n\n6. **Completion**:\n   - When the issue has been fixed, respond with \"PATCH COMPLETED\".\n   - Only respond with \"PATCH COMPLETED\" if you are confident the issue is resolved.\n\n7. **Example**:\n**Problem Statmenet**:\nTypeError: unsupported format string passed to NoneType.__format__\nRegression in #2459\n\n### Steps to reproduce\na.py:\npy\nclass A:\n    def __init__(self):\n        self._magnitude = None\n\n    def name(self) -> str | None:\n        if self._magnitude:\n            return f\"M {self._magnitude:.1f}\"\n\npylint a.py\n\n### Current behavior\nFile \"/Users/jwalls/release/lib/python3.12/site-packages/astroid/nodes/node_classes.py\", line 4778, in _infer_from_values\n    yield from nodes[0]._infer(context, **kwargs)\n  File \"/Users/jwalls/release/lib/python3.12/site-packages/astroid/nodes/node_classes.py\", line 4695, in _infer\n    formatted = format(value.value, format_spec.value)\n                ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^\nTypeError: unsupported format string passed to NoneType.__format__\n\n**Codebase Folder List**\n['repo/tests',\n 'repo/doc',\n 'repo/script',\n 'repo/.github',\n 'repo/astroid',\n 'repo/tests/testdata',\n 'repo/tests/brain', ...]\n\n**Approach Steps**:\na. Understanding I need to solve the problem about `astroid` project.\nb. Use `search_code_elements` to search for `_infer_from_values` and `_infer`, obtaining their start and end lines.\nc. Use `open_file` to open and analyze `_infer_from_values` or `_infer`, focusing on the range identified by `search_code_elements`.\nd. Use `edit_file` to make the necessary changes to `_infer_from_values` or `_infer`.\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:54.570132Z","iopub.status.idle":"2025-01-06T14:43:54.570412Z","shell.execute_reply":"2025-01-06T14:43:54.570285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# instance_count = None\n\ndef get_number_of_instances(num_instances: int) -> None:\n    \"\"\" The very first message from the gateway will be the total number of instances to be served.\n    You don't need to edit this function.\n    \"\"\"\n    global instance_count\n    instance_count = num_instances","metadata":{"execution":{"iopub.status.busy":"2025-01-06T14:43:54.571012Z","iopub.status.idle":"2025-01-06T14:43:54.571262Z","shell.execute_reply":"2025-01-06T14:43:54.571161Z"},"papermill":{"duration":0.011949,"end_time":"2024-12-11T03:22:08.838279","exception":false,"start_time":"2024-12-11T03:22:08.82633","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import unidiff\n\n\ndef is_valid_patch_format(patch: str) -> bool:\n    try:\n        patch_set = unidiff.PatchSet(patch)\n        if len(patch_set) == 0:\n            return False\n    except Exception:\n        return False\n    \n    return True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:54.571817Z","iopub.status.idle":"2025-01-06T14:43:54.572066Z","shell.execute_reply":"2025-01-06T14:43:54.571962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"host = \"http://localhost:9200\"  # Elasticsearch\n\n# Elasticsearch Client\nes = Elasticsearch(\n    hosts=[host],\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:54.572649Z","iopub.status.idle":"2025-01-06T14:43:54.572892Z","shell.execute_reply":"2025-01-06T14:43:54.572793Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Define Agent: GitHub Issue Solver with Elasticsearch and Codebase Management\n\nThis script defines an agent that processes GitHub issues by analyzing the provided codebase, indexing its structure in Elasticsearch, and using LLM-assisted tools to suggest and apply changes. The agent performs health checks, manages the codebase as a Git repository, and generates diffs for code patches.\n\n### Key Components:\n\n#### 1. **`predict(problem_statement: str, repo_archive: io.BytesIO) -> str`**\n**Purpose:**  \nProcesses the given problem statement and codebase archive, performs analysis, and generates a suggested code patch.\n\n---\n\n#### **Main Steps:**\n\n1. **Health Checks:**\n   - Waits for the external service (LLM endpoint) and Elasticsearch to be available. it will called only first_prediction is `true`\n\n2. **Codebase Preparation:**\n   - Unpacks the `.tar` archive containing the codebase.\n   - Initializes the directory as a Git repository for version control.\n   - Makes an initial commit with all files.\n\n3. **Elasticsearch Indexing:**\n   - Creates a unique Elasticsearch index to store the flattened codebase structure.\n   - Mappings include fields like `id`, `type`, `file_path`, and `start_point` for structured search.\n   - Indexes data using `process_codebase` and `flatten_module_data`.\n\n4. **LLM Tool Integration:**\n   - Calls `get_model()` to create an LLM application with integrated tools (`search_code_elements`, `open_file`, `edit_file`, `list_folder`).\n   - Sends a message to the LLM app with:\n     - **Problem Statement:** Description of the issue to be solved.\n     - **Codebase Folder List:** List of directories in the unpacked codebase.\n     - **Elasticsearch Index:** The index used for querying the codebase.\n\n5. **Generating the Response:**\n   - The LLM app processes the request and returns tool-generated responses.\n   - Logs the response content for review.\n\n6. **Committing Changes:**\n   - Uses `git status` to detect any changes made by the LLM's response.\n   - If changes exist, commits them and generates the `git diff` output.\n\n7. **Validation:**\n   - The function validates the generated patch format before returning the diff.","metadata":{}},{"cell_type":"code","source":"import time\nimport uuid\nimport requests\nfirst_prediction = True\n\ndef predict(problem_statement: str, repo_archive: io.BytesIO) -> str:\n    \"\"\" Replace this function with your inference code.\n    Args:\n        problem_statement: The text of the git issue.\n        repo_path: A BytesIO buffer path with a .tar containing the codebase that must be patched. The gateway will make this directory available immediately before this function runs.\n    \"\"\"\n    try:        \n        global first_prediction\n        global is_submission\n        global is_debug\n\n        if not is_submission and not is_debug:\n            return None\n        \n        if first_prediction:\n            # Wait for external service health\n            while True:\n                try:\n                    health_response = requests.get(\"http://localhost:8000/health\")\n                    if health_response.status_code == 200:\n                        print(\"Health check for http://localhost:8000/health passed.\")\n                        break\n                    print(\"Waiting for http://localhost:8000/health...\")\n                except requests.exceptions.RequestException as e:\n                    print(f\"Health check request failed: {e}. Retrying...\")\n                time.sleep(5)  # Retry every 5 seconds\n    \n            # Wait for Elasticsearch connection\n            es = Elasticsearch(hosts=[\"http://localhost:9200\"])\n            while True:\n                try:\n                    if es.ping():\n                        print(\"Elasticsearch is available.\")\n                        break\n                    print(\"Waiting for Elasticsearch to become available...\")\n                except Exception as e:\n                    print(f\"Elasticsearch connection error: {e}. Retrying...\")\n                time.sleep(5)  # Retry every 5 seconds\n    \n            print(\"Health checks passed.\")\n            first_prediction = False\n            \n        # Unpack\n        with open('repo_archive.tar', 'wb') as f:\n            f.write(repo_archive.read())\n        repo_path = 'repo'\n        if os.path.exists(repo_path):\n            shutil.rmtree(repo_path)\n        shutil.unpack_archive('repo_archive.tar', extract_dir=repo_path)    \n        \n        os.remove('repo_archive.tar')\n    \n        # Initialize a Git repository\n        # Ensure Git user identity is set\n        subprocess.run([\"git\", \"config\", \"--global\", \"user.email\", \"example@example.com\"], check=True)\n        subprocess.run([\"git\", \"config\", \"--global\", \"user.name\", \"Example User\"], check=True)    \n        subprocess.run([\"git\", \"init\"], cwd=repo_path, check=True)\n        subprocess.run([\"git\", \"add\", \"-A\"], cwd=repo_path, check=True)\n        subprocess.run([\"git\", \"commit\", \"-m\", \"Initial commit\"], cwd=repo_path, check=True)\n        \n        index_name = str(uuid.uuid4())\n        ## elasticsearch settings    \n        base_directory = repo_path  # Replace with your codebase path\n        codebase_structure = process_codebase(base_directory)\n    \n        host = \"http://localhost:9200\"  \n        \n        # Elasticsearch Client\n        es = Elasticsearch(\n            hosts=[host],\n        )\n        \n        es.indices.create(index=index_name, body={\n            \"mappings\": {\n                \"properties\": {\n                    \"id\": { \"type\": \"keyword\" },\n                    \"parent_id\": { \"type\": \"keyword\" },\n                    \"file_path\": { \"type\": \"keyword\" },\n                    \"type\": { \"type\": \"keyword\" },\n                    \"text\": { \"type\": \"text\" },\n                    \"start_point\": {\n                        \"type\": \"object\",\n                        \"properties\": {\n                            \"row\": { \"type\": \"integer\" },\n                            \"column\": { \"type\": \"integer\" }\n                        }\n                    },\n                    \"end_point\": {\n                        \"type\": \"object\",\n                        \"properties\": {\n                            \"row\": { \"type\": \"integer\" },\n                            \"column\": { \"type\": \"integer\" }\n                        }\n                    }\n                }\n            }\n        })\n        flattened = flatten_module_data(codebase_structure)\n        index_data(flattened, index_name)\n    \n        ##langgraph setting\n        app = get_model()\n        folder_list = list_folder.invoke(input={\"folder_path\": repo_path}).folders\n        messages = [\n        {\"role\": \"system\",\n         \"content\": GITHUB_ISSUE_SOLVER_PROMPT},\n         {\"role\":\n        \"user\",\n        \"content\":\n        f\"##Problem Statement: {problem_statement}\\n\\nCodebase Path: {repo_path}\\n\\nCodebase Folder List: {folder_list}\\n\\nElasticSearch Index: {index_name}\"}]\n    \n        response = app.invoke(\n        {\"messages\": messages}, {\"recursion_limit\": 10})\n      \n        \n        for message in response[\"messages\"]:\n            string_representation = f\"{message.type.upper()}: {message.content}\\n\"\n            print(string_representation)\n        es.indices.delete(index=index_name)\n        # Instead of a valid diff, let's just submit a generic string. This will definitely fail.    \n        first_prediction = False\n        # Check for changes before adding and committing\n        status_result = subprocess.run(\n            [\"git\", \"status\", \"--porcelain\"],\n            cwd=repo_path,\n            check=True,\n            stdout=subprocess.PIPE,\n            text=True\n        )\n    \n        if not status_result.stdout.strip():\n            print(\"No changes to commit.\")\n            git_diff = None\n        else:\n            subprocess.run([\"git\", \"add\", \"-A\"], cwd=repo_path, check=True)\n            subprocess.run([\"git\", \"commit\", \"-m\", \"Apply changes\"], cwd=repo_path, check=True)\n    \n            # Get git diff output\n            git_diff = subprocess.check_output([\"git\", \"diff\", \"HEAD~1\"], cwd=repo_path, text=True)\n            print(git_diff)\n        del app\n\n        if is_valid_patch_format(git_diff):        \n            return git_diff\n        else:\n            return None\n    except Exception as e:\n        print(f\"Exception occurred: {e}\")\n        return None","metadata":{"execution":{"iopub.status.busy":"2025-01-06T14:43:54.573485Z","iopub.status.idle":"2025-01-06T14:43:54.573745Z","shell.execute_reply":"2025-01-06T14:43:54.573636Z"},"papermill":{"duration":0.011382,"end_time":"2024-12-11T03:22:08.852112","exception":false,"start_time":"2024-12-11T03:22:08.84073","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**When your notebook is run on the hidden test set, inference_server.serve must be called within 15 minutes of the notebook starting or the gateway will throw an error. If you need more than 15 minutes to load your model you can do so during the very first predict call, which does not have the usual 30 minute response deadline.**","metadata":{"papermill":{"duration":0.001889,"end_time":"2024-12-11T03:22:08.856283","exception":false,"start_time":"2024-12-11T03:22:08.854394","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## **Inference Server Integration**\nThe script integrates with `KPrizeInferenceServer` to handle competition mode and local testing.\n\n- **`is_debug = True:`**  \n  Predicts all training data to simulate the inference process.\n- **`is_debug = False:`**  \n  Returns `None` for all training data to reduce commit time.\n\n\n**Ensure that for testing purposes, I set recursion_limit=10. It's take around 2hours for inference.**\n---","metadata":{}},{"cell_type":"code","source":"inference_server = kaggle_evaluation.konwinski_prize_inference_server.KPrizeInferenceServer(\n    get_number_of_instances,   \n    predict\n)\nis_debug = True\nis_submission = os.getenv('KAGGLE_IS_COMPETITION_RERUN')\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            '/kaggle/input/konwinski-prize/',  # Path to the entire competition dataset\n            '/kaggle/tmp/konwinski-prize/',   # Path to a scratch directory for unpacking data.a_zip.\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2025-01-06T14:43:54.574364Z","iopub.status.idle":"2025-01-06T14:43:54.574619Z","shell.execute_reply":"2025-01-06T14:43:54.574516Z"},"papermill":{"duration":3.790202,"end_time":"2024-12-11T03:22:12.648591","exception":false,"start_time":"2024-12-11T03:22:08.858389","status":"completed"},"tags":[],"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!cat vllm_output.log","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:54.575051Z","iopub.status.idle":"2025-01-06T14:43:54.575291Z","shell.execute_reply":"2025-01-06T14:43:54.575194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!cat elasticsearch_output.log","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:54.575806Z","iopub.status.idle":"2025-01-06T14:43:54.576047Z","shell.execute_reply":"2025-01-06T14:43:54.575948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\n\ndef delete_subdirectories(path):\n    \"\"\"\n    Function to delete only the subdirectories in a given path.\n\n    Parameters:\n        path (str): Path containing subdirectories to be deleted.\n    \"\"\"\n    if not os.path.exists(path):\n        print(f\"Path does not exist: {path}\")\n        return\n\n    # Iterate over all items in the given path\n    for item in os.listdir(path):\n        full_path = os.path.join(path, item)\n\n        # Delete if the item is a directory\n        if os.path.isdir(full_path):\n            shutil.rmtree(full_path)\n            print(f\"Directory deleted: {full_path}\")\n        else:\n            print(f\"Skipped (not a directory): {full_path}\")\n\n# Test path\ntarget_path = \"/kaggle/working\"\ndelete_subdirectories(target_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T14:43:54.576656Z","iopub.status.idle":"2025-01-06T14:43:54.576901Z","shell.execute_reply":"2025-01-06T14:43:54.576802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.read_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2025-01-06T14:43:54.577493Z","iopub.status.idle":"2025-01-06T14:43:54.577738Z","shell.execute_reply":"2025-01-06T14:43:54.577637Z"},"trusted":true},"outputs":[],"execution_count":null}]}