{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84795,"databundleVersionId":10462807,"sourceType":"competition"},{"sourceId":124328,"sourceType":"modelInstanceVersion","modelInstanceId":104636,"modelId":128845}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"The [Cursor editor](https://www.cursor.com/) is one of the best in the industry solving the same problem as the goal of this competition. As per the [Cursor editor documentations](https://docs.cursor.com/context/codebase-indexing), they are using file embeddings as a means to index the codebase.\n\nThe goal of this notebook is to use embedding based similarity search to see if we can narrow down the number of files that can be sent to LLM as context for asking about the required code change.\n\nI would really appreciate if folks seeing this notebook can post comment on what could be done better with the implementation of this approach to get better results.","metadata":{}},{"cell_type":"code","source":"!pip install faiss-cpu\n!pip install -U sentence-transformers\n!pip install langchain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:56:57.747347Z","iopub.execute_input":"2024-12-30T17:56:57.747712Z","iopub.status.idle":"2024-12-30T17:57:15.131902Z","shell.execute_reply.started":"2024-12-30T17:56:57.747669Z","shell.execute_reply":"2024-12-30T17:57:15.131057Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\nimport os\nimport faiss\nfrom sentence_transformers import SentenceTransformer\nimport torch.nn.functional as F\nimport torch\nimport numpy as np\nfrom langchain_text_splitters.character import Language, RecursiveCharacterTextSplitter","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:57:22.527369Z","iopub.execute_input":"2024-12-30T17:57:22.527703Z","iopub.status.idle":"2024-12-30T17:57:37.23703Z","shell.execute_reply.started":"2024-12-30T17:57:22.527675Z","shell.execute_reply":"2024-12-30T17:57:37.236316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\npd.set_option('display.width', 150)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:57:37.23806Z","iopub.execute_input":"2024-12-30T17:57:37.238667Z","iopub.status.idle":"2024-12-30T17:57:37.242754Z","shell.execute_reply.started":"2024-12-30T17:57:37.238631Z","shell.execute_reply":"2024-12-30T17:57:37.241811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    base_pth = Path('/kaggle/input/konwinski-prize')\n    datazip_pth = Path('/kaggle/input/konwinski-prize/data')\n    working_pth = Path('/kaggle/working')\n    repo_pth = Path('/kaggle/working')\n    data_pth = Path('/kaggle/working/data')\n    repo_pth = Path('/kaggle/working/data/repos/repo__pylint-dev__astroid-2468')\n    model_pth = Path('/kaggle/input/baai/transformers/bge-base-en-v1.5/1')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:57:37.244982Z","iopub.execute_input":"2024-12-30T17:57:37.245305Z","iopub.status.idle":"2024-12-30T17:57:37.669582Z","shell.execute_reply.started":"2024-12-30T17:57:37.245273Z","shell.execute_reply":"2024-12-30T17:57:37.668702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!unzip -q /kaggle/input/konwinski-prize/data.a_zip","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:57:37.670853Z","iopub.execute_input":"2024-12-30T17:57:37.671129Z","iopub.status.idle":"2024-12-30T17:57:41.14343Z","shell.execute_reply.started":"2024-12-30T17:57:37.671108Z","shell.execute_reply":"2024-12-30T17:57:41.142445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"python_splitter = RecursiveCharacterTextSplitter.from_language(\n    language=Language.PYTHON, chunk_size=500, chunk_overlap=0\n)\n\nchunks = []\n\nfor pth, dirs, files in os.walk(Config.repo_pth):\n    for file in files:\n        if file.endswith('.py'):\n            file_pth = f'{pth}/{file}'\n            with open(file_pth) as f:\n                content = f.read()\n                python_docs = python_splitter.create_documents([content])\n                for doc in python_docs:\n                    chunk_detail = {\n                        'file_path': file_pth,\n                        'content': doc.page_content\n                    }\n                    chunks.append(chunk_detail)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:57:41.144752Z","iopub.execute_input":"2024-12-30T17:57:41.14513Z","iopub.status.idle":"2024-12-30T17:57:41.260945Z","shell.execute_reply.started":"2024-12-30T17:57:41.145095Z","shell.execute_reply":"2024-12-30T17:57:41.260328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"chunk_txts = [chunk['content'] for chunk in chunks]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:57:41.261813Z","iopub.execute_input":"2024-12-30T17:57:41.262119Z","iopub.status.idle":"2024-12-30T17:57:41.267011Z","shell.execute_reply.started":"2024-12-30T17:57:41.262085Z","shell.execute_reply":"2024-12-30T17:57:41.265936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"embed_model = SentenceTransformer('/kaggle/input/baai/transformers/bge-base-en-v1.5/1')\npool = embed_model.start_multi_process_pool()\nembeddings = embed_model.encode_multi_process(chunk_txts, pool)\nembeddings.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:57:41.267864Z","iopub.execute_input":"2024-12-30T17:57:41.26824Z","iopub.status.idle":"2024-12-30T17:58:22.14068Z","shell.execute_reply.started":"2024-12-30T17:57:41.268212Z","shell.execute_reply":"2024-12-30T17:58:22.139992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vector_dimension = embeddings.shape[1]\nindex = faiss.IndexFlatL2(vector_dimension)\nfaiss.normalize_L2(embeddings)\nindex.add(embeddings)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:58:22.14239Z","iopub.execute_input":"2024-12-30T17:58:22.142629Z","iopub.status.idle":"2024-12-30T17:58:22.165963Z","shell.execute_reply.started":"2024-12-30T17:58:22.14261Z","shell.execute_reply":"2024-12-30T17:58:22.165261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_parquet(Config.working_pth / \"data/data.parquet\")\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:58:22.166937Z","iopub.execute_input":"2024-12-30T17:58:22.167148Z","iopub.status.idle":"2024-12-30T17:58:22.241929Z","shell.execute_reply.started":"2024-12-30T17:58:22.167129Z","shell.execute_reply":"2024-12-30T17:58:22.241211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"astroid_df = df[df['repo'] == 'pylint-dev/astroid']\nastroid_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:58:22.24259Z","iopub.execute_input":"2024-12-30T17:58:22.242841Z","iopub.status.idle":"2024-12-30T17:58:22.260307Z","shell.execute_reply.started":"2024-12-30T17:58:22.24282Z","shell.execute_reply":"2024-12-30T17:58:22.259453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = {\n    'statement': [],\n    'distance': [],\n    'matched_file': [],\n    'matched_codesnippet': []\n}\n\nfor statement in astroid_df['problem_statement']:\n    search_vector = embed_model.encode(statement)\n    _vector = np.array([search_vector])\n    faiss.normalize_L2(_vector)\n    k = index.ntotal\n    distances, ann = index.search(_vector, k=k)\n\n    # top 10 results\n    for i in range(10):\n        matched_chunk = chunks[ann[0][i]]\n        results['statement'].append(statement)\n        results['distance'].append(distances[0][i])\n        results['matched_file'].append('/'.join(matched_chunk['file_path'].split('/')[-2:]))\n        results['matched_codesnippet'].append(matched_chunk['content'])\n\nresult_df = pd.DataFrame(results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T18:01:05.963681Z","iopub.execute_input":"2024-12-30T18:01:05.964063Z","iopub.status.idle":"2024-12-30T18:01:06.639996Z","shell.execute_reply.started":"2024-12-30T18:01:05.964033Z","shell.execute_reply":"2024-12-30T18:01:06.639166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T18:01:09.532801Z","iopub.execute_input":"2024-12-30T18:01:09.533143Z","iopub.status.idle":"2024-12-30T18:01:09.544692Z","shell.execute_reply.started":"2024-12-30T18:01:09.533115Z","shell.execute_reply":"2024-12-30T18:01:09.543708Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"For the first problem statement with `TypeError` the suggested patch in training data makes changes in `node_classes.py` file which is the 3rd result of this similarity search.\n\nFor the second problem statement, the suggested patch also required changes in `node_classes.py` file but that did not make it to the top 10 results in the similarity search.","metadata":{}}]}