{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84795,"databundleVersionId":11281725,"sourceType":"competition"},{"sourceId":165313789,"sourceType":"kernelVersion"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T09:46:18.539708Z","iopub.execute_input":"2025-03-04T09:46:18.539923Z","iopub.status.idle":"2025-03-04T09:46:23.693006Z","shell.execute_reply.started":"2025-03-04T09:46:18.539901Z","shell.execute_reply":"2025-03-04T09:46:23.692173Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Motivation**:\n\nThe organizer, Andy Konwinski, was inspired by SWE-bench and its approach to evaluating AI models on real-world GitHub issues.\nHe believes that a private test set would provide a more accurate and challenging evaluation, so a new test set will be collected after the submission deadline.\n\nOpen Source Requirement:\nCash prizes will only be awarded to submissions that use open-source code and open-weight models. This emphasizes the importance of transparency and community collaboration.\nImpact:\nAutomating the resolution of GitHub issues will allow human software engineers to focus on more creative and strategic tasks, such as designing new features, reforming abstractions, and interfacing with users.\nThe goal is to reduce the time spent on bug fixing and increase the time spent on building new and innovative solutions.\n\n**Evaluation**\n\nThe performance of the AI models is evaluated using a specific scoring metric designed to incentivize the submission of high-quality solutions and penalize incorrect or skipped issues. The scoring formula is:\n\n$$ \\text{score} = \\frac{a - b - \\frac{c}{10,000}}{a + b + c} $$\n\nWhere:\n\n( a ) is the number of correctly resolved issues.\n\n( b ) is the number of failing issues.\n\n( c ) is the number of skipped issues.\n\nThis can be implemented in Python as follows:","metadata":{}},{"cell_type":"code","source":"# install\n!pip install accelerate\n!pip install -U bitsandbytes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T09:46:23.694448Z","iopub.execute_input":"2025-03-04T09:46:23.694794Z","iopub.status.idle":"2025-03-04T09:46:33.690484Z","shell.execute_reply.started":"2025-03-04T09:46:23.694773Z","shell.execute_reply":"2025-03-04T09:46:33.689283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!pip uninstall -y wandb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T09:46:33.692518Z","iopub.execute_input":"2025-03-04T09:46:33.692874Z","iopub.status.idle":"2025-03-04T09:46:33.697439Z","shell.execute_reply.started":"2025-03-04T09:46:33.692841Z","shell.execute_reply":"2025-03-04T09:46:33.696583Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Declare libraries and configuration path","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\nimport re\nimport matplotlib.pyplot as plt\nimport os\nimport kaggle_evaluation.konwinski_prize_inference_server\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T09:46:33.698456Z","iopub.execute_input":"2025-03-04T09:46:33.698766Z","iopub.status.idle":"2025-03-04T09:46:34.708649Z","shell.execute_reply.started":"2025-03-04T09:46:33.698733Z","shell.execute_reply":"2025-03-04T09:46:34.707724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    base_pth = Path('/kaggle/input/konwinski-prize')\n    datazip_pth = Path('//kaggle/input/konwinski-prize/data')\n    working_pth = Path('/kaggle/working')\n    repo_pth = Path('/kaggle/working')\n    data_pth = Path('/kaggle/working/data')\n    repo_pth = Path('/kaggle/working/data/repos')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T09:46:34.709665Z","iopub.execute_input":"2025-03-04T09:46:34.710179Z","iopub.status.idle":"2025-03-04T09:46:34.714644Z","shell.execute_reply.started":"2025-03-04T09:46:34.710154Z","shell.execute_reply":"2025-03-04T09:46:34.713776Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Understanding the Problem and Data\nFirst, load the Parquet dataset and explore it:","metadata":{}},{"cell_type":"code","source":"!unzip -n ../input/konwinski-prize/data.a_zip","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T09:46:34.71567Z","iopub.execute_input":"2025-03-04T09:46:34.71601Z","iopub.status.idle":"2025-03-04T09:46:38.670613Z","shell.execute_reply.started":"2025-03-04T09:46:34.715986Z","shell.execute_reply":"2025-03-04T09:46:38.66946Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualize","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport pyarrow.parquet as pq\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom datasets import Dataset\n\n# Load the dataset\nissues_df = pd.read_parquet(Config.working_pth / \"data/data.parquet\")\nissues_df = issues_df.head(10)\n# Basic exploration\nprint(issues_df.head())\nprint(issues_df.info())\nprint(issues_df.describe())\n\n# Visualize the target variable\nsns.countplot(x='instance_id', data=issues_df)\nplt.show()\n\n\n\ndf_filtered = issues_df[['instance_id', 'repo', 'problem_statement', 'patch', 'test_patch', 'pull_number', 'base_commit', 'issue_numbers']]\n\n# Convert to Hugging Face Dataset\ndataset = Dataset.from_pandas(df_filtered)\n\n# Advanced dataset visualization\nfig, axes = plt.subplots(2, 2, figsize=(14, 10))\n\n# Problem statement length distribution\nsns.histplot(df_filtered['problem_statement'].apply(len), bins=50, kde=True, log_scale=True, ax=axes[0, 0])\naxes[0, 0].set_xlabel(\"Problem Statement Length\")\naxes[0, 0].set_ylabel(\"Frequency\")\naxes[0, 0].set_title(\"Distribution of Problem Statement Lengths\")\n\n# Patch vs. Test Patch length comparison\npatch_lengths = df_filtered[['patch', 'test_patch']].dropna()\npatch_lengths['patch_length'] = patch_lengths['patch'].apply(lambda x: len(str(x)))\npatch_lengths['test_patch_length'] = patch_lengths['test_patch'].apply(lambda x: len(str(x)))\nsns.scatterplot(x=patch_lengths['patch_length'], y=patch_lengths['test_patch_length'], alpha=0.5, ax=axes[0, 1])\naxes[0, 1].set_xlabel(\"Patch Length\")\naxes[0, 1].set_ylabel(\"Test Patch Length\")\naxes[0, 1].set_title(\"Patch vs. Test Patch Length Comparison\")\n\n# Repository distribution\nsns.countplot(y=df_filtered['repo'], order=df_filtered['repo'].value_counts().index[:10], ax=axes[1, 0])\naxes[1, 0].set_xlabel(\"Count\")\naxes[1, 0].set_ylabel(\"Repository\")\naxes[1, 0].set_title(\"Top 10 Repositories by Issue Count\")\n\n# Pull request number distribution\nsns.histplot(df_filtered['pull_number'].dropna(), bins=50, kde=True, ax=axes[1, 1])\naxes[1, 1].set_xlabel(\"Pull Request Number\")\naxes[1, 1].set_ylabel(\"Frequency\")\naxes[1, 1].set_title(\"Distribution of Pull Request Numbers\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T09:46:38.671816Z","iopub.execute_input":"2025-03-04T09:46:38.672082Z","iopub.status.idle":"2025-03-04T09:46:40.836203Z","shell.execute_reply.started":"2025-03-04T09:46:38.672058Z","shell.execute_reply":"2025-03-04T09:46:40.835292Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"GitHub Issue Resolution AI","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Distribution of issues per repository\nplt.figure(figsize=(12, 5))\nsns.countplot(y=issues_df[\"repo\"], order=issues_df[\"repo\"].value_counts().index, palette=\"viridis\")\nplt.title(\"Distribution of GitHub Issues by Repository\")\nplt.xlabel(\"Count\")\nplt.ylabel(\"Repository\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T09:46:40.838185Z","iopub.execute_input":"2025-03-04T09:46:40.838679Z","iopub.status.idle":"2025-03-04T09:46:40.985682Z","shell.execute_reply.started":"2025-03-04T09:46:40.838653Z","shell.execute_reply":"2025-03-04T09:46:40.984753Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Patch Length Analysis","metadata":{}},{"cell_type":"code","source":"issues_df[\"patch_length\"] = issues_df[\"patch\"].apply(lambda x: len(str(x)))\nplt.figure(figsize=(10, 5))\nsns.histplot(issues_df[\"patch_length\"], bins=30, kde=True)\nplt.title(\"Distribution of Patch Lengths\")\nplt.xlabel(\"Patch Length (# characters)\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T09:46:40.986666Z","iopub.execute_input":"2025-03-04T09:46:40.986898Z","iopub.status.idle":"2025-03-04T09:46:41.23155Z","shell.execute_reply.started":"2025-03-04T09:46:40.986878Z","shell.execute_reply":"2025-03-04T09:46:41.230638Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Word Cloud of GitHub Issues","metadata":{}},{"cell_type":"code","source":"from wordcloud import WordCloud\n\ntext = \" \".join(issues_df[\"problem_statement\"])\nwordcloud = WordCloud(width=800, height=400, background_color=\"white\").generate(text)\n\nplt.figure(figsize=(10, 5))\nplt.imshow(wordcloud, interpolation=\"bilinear\")\nplt.axis(\"off\")\nplt.title(\"Most Common Words in GitHub Issues\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T09:46:41.232633Z","iopub.execute_input":"2025-03-04T09:46:41.232916Z","iopub.status.idle":"2025-03-04T09:46:42.15162Z","shell.execute_reply.started":"2025-03-04T09:46:41.232892Z","shell.execute_reply":"2025-03-04T09:46:42.15055Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## Train Model, Prediction and Submission","metadata":{}},{"cell_type":"code","source":"import torch\nfrom transformers import AutoModelForCausalLM, AutoTokenizer, TrainingArguments, Trainer, BitsAndBytesConfig\nfrom peft import LoraConfig, get_peft_model\nfrom datasets import Dataset\n# from kaggle_evaluation.konwinski_prize_inference_server import KonwinskiPrizeInference\nimport kaggle_evaluation.konwinski_prize_inference_server as kp\n\n\n# Ensure GPU is available\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(f\"Using device: {device}\")\n\n# Select AI model (GPT-2 or GPT-J-6B)\n# Select AI model with 4-bit quantization\nMODEL_NAME = \"EleutherAI/gpt-neo-1.3B\"\nprint(f\"Loading tokenizer for {MODEL_NAME}...\")\n\n# Load tokenizer\ntokenizer = AutoTokenizer.from_pretrained(MODEL_NAME, trust_remote_code=True)\ntokenizer.add_special_tokens({\"pad_token\": \"[PAD]\"})\n\n# Tokenization function\ndef tokenize_function(examples):\n    return tokenizer.batch_encode_plus(\n        examples[\"problem_statement\"], padding=\"longest\", truncation=True, max_length=128, return_tensors=\"pt\"\n    )\n\ndataset = dataset.map(tokenize_function, batched=True, batch_size=8)\n\n# Load Model with 4-bit Quantization\n#quant_config = BitsAndBytesConfig(load_in_4bit=True)  # Use 4-bit to reduce memory\n#print(f\"Loading model {MODEL_NAME}...\")\n\n# Load Model with Optimized 4-bit Quantization\nquant_config = BitsAndBytesConfig(\n    load_in_4bit=True, \n    bnb_4bit_compute_dtype=torch.float16,  # Use FP16 for reduced memory\n    bnb_4bit_quant_type=\"nf4\",  # Use Normalized Float 4-bit for better accuracy\n    llm_int8_enable_fp32_cpu_offload=True  # Offload computation to CPU if needed\n)\n\nprint(f\"Loading model {MODEL_NAME} with optimized quantization...\")\n\nmodel = AutoModelForCausalLM.from_pretrained(\n    MODEL_NAME,\n    quantization_config=quant_config, \n    device_map=\"auto\"\n)\n\n# Attach LoRA adapters for fine-tuning\nlora_config = LoraConfig(\n    r=8, lora_alpha=16, lora_dropout=0.1, \n    task_type=\"CAUSAL_LM\"\n)\nmodel = get_peft_model(model, lora_config)\nmodel.print_trainable_parameters()\n\n# Ensure correct tokenization\ndef tokenize_function(examples):\n    return tokenizer(\n        examples[\"problem_statement\"], padding=\"max_length\", truncation=True, max_length=64\n    )\n\ndataset = dataset.map(tokenize_function, batched=True, batch_size=8)\ndataset.set_format(type=\"torch\", columns=[\"input_ids\", \"attention_mask\"])  # Ensure format for Trainer\n\n\n# Training setup\ntraining_args = TrainingArguments(\n    output_dir=\"models/optimized-model\",\n    per_device_train_batch_size=1,  # Keep batch size small\n    gradient_accumulation_steps=8,  # Accumulate gradients to simulate larger batch\n    num_train_epochs=2,\n    logging_steps=50,\n    fp16=True,  # Enable mixed precision\n    save_total_limit=1\n)\n\ntrainer = Trainer(\n    model=model, \n    args=training_args, \n    train_dataset=dataset\n)\ntrainer.train()\n\n# Save model\nmodel.save_pretrained(\"fine_tuned_optimized_model\")\ntokenizer.save_pretrained(\"fine_tuned_optimized_model\")\n\n\ndf_filtered.to_parquet(\"submission.parquet\", index=False)\n\n# Kaggle Evaluation API\ninference_api = kp()\ndef generate_patch(instance):\n    input_text = instance[\"problem_statement\"]\n    inputs = tokenizer(input_text, return_tensors=\"pt\", padding=True, truncation=True, max_length=80).to(device)\n    with torch.no_grad():\n        outputs = model.generate(**inputs, max_length=80)\n    return tokenizer.decode(outputs[0], skip_special_tokens=True)\n\nfor instance in inference_api:\n    suggested_patch = generate_patch(instance)\n    inference_api.submit(instance[\"instance_id\"], suggested_patch)\n\ninference_api.complete()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T09:46:42.152652Z","iopub.execute_input":"2025-03-04T09:46:42.153223Z"}},"outputs":[],"execution_count":null}]}