{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84795,"databundleVersionId":10462807,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\nimport re\nimport matplotlib.pyplot as plt\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-28T19:10:01.852882Z","iopub.execute_input":"2024-12-28T19:10:01.853127Z","iopub.status.idle":"2024-12-28T19:10:02.227576Z","shell.execute_reply.started":"2024-12-28T19:10:01.853104Z","shell.execute_reply":"2024-12-28T19:10:02.226528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    base_pth = Path('/kaggle/input/konwinski-prize')\n    datazip_pth = Path('/kaggle/input/konwinski-prize/data')\n    working_pth = Path('/kaggle/working')\n    repo_pth = Path('/kaggle/working')\n    data_pth = Path('/kaggle/working/data')\n    repo_pth = Path('/kaggle/working/data/repos')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T19:10:02.228537Z","iopub.execute_input":"2024-12-28T19:10:02.229007Z","iopub.status.idle":"2024-12-28T19:10:02.234086Z","shell.execute_reply.started":"2024-12-28T19:10:02.228968Z","shell.execute_reply":"2024-12-28T19:10:02.233075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!unzip -q /kaggle/input/konwinski-prize/data.a_zip","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T19:10:02.234994Z","iopub.execute_input":"2024-12-28T19:10:02.235351Z","iopub.status.idle":"2024-12-28T19:10:05.912464Z","shell.execute_reply.started":"2024-12-28T19:10:02.235324Z","shell.execute_reply":"2024-12-28T19:10:05.911326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_parquet(Config.working_pth / \"data/data.parquet\")\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T19:10:05.91373Z","iopub.execute_input":"2024-12-28T19:10:05.914141Z","iopub.status.idle":"2024-12-28T19:10:06.106916Z","shell.execute_reply.started":"2024-12-28T19:10:05.914113Z","shell.execute_reply":"2024-12-28T19:10:06.105928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cntr = {}\n\nfor i in df['instance_id']:\n    x = re.search(\"(.*)-\\d+\", i)\n    if x:\n        repo_name = x.group(1)\n        if repo_name not in cntr:\n            cntr[repo_name] = 1\n        else:\n            cntr[repo_name] += 1\n\nvals = list(cntr.values())\nkeys = list(cntr.keys())\np, tx, autotexts = plt.pie([float(v) for v in vals], labels=[str(k) for k in keys], autopct='%1.1f%%', shadow=True)\n\nfor i, a in enumerate(autotexts):\n    a.set_text(\"{}\".format(vals[i]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T19:10:06.107975Z","iopub.execute_input":"2024-12-28T19:10:06.108335Z","iopub.status.idle":"2024-12-28T19:10:06.287859Z","shell.execute_reply.started":"2024-12-28T19:10:06.108301Z","shell.execute_reply":"2024-12-28T19:10:06.286761Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The above pie chart shows the distinct repos that are present in the input data along with the count of states for the repo.\n\nWhat is the meaning of state?<br/>\nGit is a version control software. A repository managed by Git will have iterations of edits stacked on top of each other. Each iteration is called a commit in Git world. State refers to the state of all the files at the time when the patch was applied on them. Applying the patch mutates the state of the repo.","metadata":{}},{"cell_type":"markdown","source":"# Repository sizes","metadata":{}},{"cell_type":"markdown","source":"A git repository in the most normie sense is directory containing a bunch of files. Ideally one would want the LLM to know each and every line of code in every file of the repo. This is a challenge as repositories can have \"any\" number of files. The goal of this section is to put a number on \"any\".","metadata":{}},{"cell_type":"code","source":"file_cnt = {\"Repo state\": [], \"File count\": [], \"Line count\": []}\n\nfor repo_state in os.listdir(Config.repo_pth):\n    fc = 0\n    lc = 0\n    for r, d, files in os.walk(Config.repo_pth / repo_state): \n        fc += len(files)\n        for file in files:\n            try:\n                lc += sum(1 for _ in open(f'{r}/{file}'))\n            except:\n                pass\n    file_cnt[\"Repo state\"].append(repo_state)\n    file_cnt[\"File count\"].append(fc)\n    file_cnt[\"Line count\"].append(lc)\n\nfile_cnt_df = pd.DataFrame(file_cnt)\nfile_cnt_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T19:10:06.290114Z","iopub.execute_input":"2024-12-28T19:10:06.290491Z","iopub.status.idle":"2024-12-28T19:10:07.040968Z","shell.execute_reply.started":"2024-12-28T19:10:06.29046Z","shell.execute_reply":"2024-12-28T19:10:07.040035Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Gemini pro 1.5 is the current leader in the number of context tokens that it can take in standing at 1M. ([ref](https://token-calculator.net/))\n\n1M context token roughly translates to 50,000 lines of code ([ref](https://ai.google.dev/gemini-api/docs/long-context)) which is less than the least that we are seeing in the sample input.","metadata":{}}]}