{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84795,"databundleVersionId":10934030,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:33:35.90464Z","iopub.execute_input":"2025-03-01T05:33:35.905388Z","iopub.status.idle":"2025-03-01T05:33:36.682203Z","shell.execute_reply.started":"2025-03-01T05:33:35.905353Z","shell.execute_reply":"2025-03-01T05:33:36.68097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\nimport os\n\nzip_path = \"/kaggle/input/konwinski-prize/data.a_zip\"\n\nextract_path = \"/kaggle/working/konwinski_data\"\n\nos.makedirs(extract_path, exist_ok=True)\n\nwith zipfile.ZipFile(zip_path, \"r\") as zip_ref:\n    zip_ref.extractall(extract_path)\nprint(\"Extraction complete!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:34:13.695904Z","iopub.execute_input":"2025-03-01T05:34:13.696387Z","iopub.status.idle":"2025-03-01T05:34:18.122881Z","shell.execute_reply.started":"2025-03-01T05:34:13.69635Z","shell.execute_reply":"2025-03-01T05:34:18.121824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"extracted_files = os.listdir(extract_path)\nprint('Extracted files:', extracted_files)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:34:25.007096Z","iopub.execute_input":"2025-03-01T05:34:25.007519Z","iopub.status.idle":"2025-03-01T05:34:25.013881Z","shell.execute_reply.started":"2025-03-01T05:34:25.007486Z","shell.execute_reply":"2025-03-01T05:34:25.012656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:34:37.213992Z","iopub.execute_input":"2025-03-01T05:34:37.214443Z","iopub.status.idle":"2025-03-01T05:34:38.087743Z","shell.execute_reply.started":"2025-03-01T05:34:37.214406Z","shell.execute_reply":"2025-03-01T05:34:38.08668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\n\nzip_path = \"/kaggle/input/konwinski-prize/data.a_zip\"  # Update with actual path\nextract_path = \"/kaggle/working/konwinski_data/\"\n\nwith zipfile.ZipFile(zip_path, \"r\") as zip_ref:\n    zip_ref.extractall(extract_path)\n\nprint(\"Files after extraction:\", os.listdir(extract_path))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:34:48.705777Z","iopub.execute_input":"2025-03-01T05:34:48.706413Z","iopub.status.idle":"2025-03-01T05:34:50.892991Z","shell.execute_reply.started":"2025-03-01T05:34:48.706376Z","shell.execute_reply":"2025-03-01T05:34:50.892075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"extract_path = \"/kaggle/working/konwinski_data/data\"\nparquet_path = os.path.join(extract_path, \"data.parquet\")\n\ndf = pd.read_parquet(parquet_path)\n\nprint(\"Dataset Loaded Successfully!\")\nprint(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:35:02.075042Z","iopub.execute_input":"2025-03-01T05:35:02.075465Z","iopub.status.idle":"2025-03-01T05:35:02.343091Z","shell.execute_reply.started":"2025-03-01T05:35:02.075434Z","shell.execute_reply":"2025-03-01T05:35:02.342138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:35:17.33073Z","iopub.execute_input":"2025-03-01T05:35:17.331046Z","iopub.status.idle":"2025-03-01T05:35:17.35779Z","shell.execute_reply.started":"2025-03-01T05:35:17.331023Z","shell.execute_reply":"2025-03-01T05:35:17.356535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:37:01.151733Z","iopub.execute_input":"2025-03-01T05:37:01.152168Z","iopub.status.idle":"2025-03-01T05:37:01.161144Z","shell.execute_reply.started":"2025-03-01T05:37:01.152133Z","shell.execute_reply":"2025-03-01T05:37:01.160003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe(include=\"all\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:37:13.082877Z","iopub.execute_input":"2025-03-01T05:37:13.083239Z","iopub.status.idle":"2025-03-01T05:37:13.129026Z","shell.execute_reply.started":"2025-03-01T05:37:13.083214Z","shell.execute_reply":"2025-03-01T05:37:13.128031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install nltk transformers datasets\n\nimport nltk\nfrom nltk.tokenize import word_tokenize\nfrom nltk.corpus import stopwords\nimport string\nimport re\n\nnltk.download('punkt')\nnltk.download('stopwords')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:37:28.861855Z","iopub.execute_input":"2025-03-01T05:37:28.862181Z","iopub.status.idle":"2025-03-01T05:37:35.340991Z","shell.execute_reply.started":"2025-03-01T05:37:28.862156Z","shell.execute_reply":"2025-03-01T05:37:35.340098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def clean_text(text):\n    text = text.lower()\n    text = re.sub(r'\\d+', '', text)\n    text = text.translate(str.maketrans(\"\", \"\", string.punctuation))\n    tokens = word_tokenize(text)\n    tokens = [word for word in tokens if word not in stopwords.words('english')]\n    return \" \".join(tokens)\n\ndf['cleaned_problem_statement'] = df['problem_statement'].apply(clean_text)\n\nprint(\"Sample cleaned text:\")\nprint(df['cleaned_problem_statement'].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:37:48.06862Z","iopub.execute_input":"2025-03-01T05:37:48.069029Z","iopub.status.idle":"2025-03-01T05:37:48.269846Z","shell.execute_reply.started":"2025-03-01T05:37:48.068993Z","shell.execute_reply":"2025-03-01T05:37:48.268794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import AutoTokenizer, AutoModel\nimport torch\n\nmodel_name = \"bert-base-uncased\"\ntokenizer = AutoTokenizer.from_pretrained(model_name)\nmodel = AutoModel.from_pretrained(model_name)\n\n\ndef get_embedding(text):\n    inputs = tokenizer(text, return_tensors=\"pt\", truncation=True, padding=True)\n    with torch.no_grad():\n        outputs = model(**inputs)\n    return outputs.last_hidden_state.mean(dim=1).squeeze().numpy()\n\ndf['problem_embedding'] = df['cleaned_problem_statement'].apply(get_embedding)\nprint(\"Embedding generated successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:38:06.718856Z","iopub.execute_input":"2025-03-01T05:38:06.719194Z","iopub.status.idle":"2025-03-01T05:38:37.403456Z","shell.execute_reply.started":"2025-03-01T05:38:06.719168Z","shell.execute_reply":"2025-03-01T05:38:37.40235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_data, test_data = train_test_split(df, test_size=0.2, random_state=42)\n\n\n\nprint(f\"Training Set: {len(train_data)} samples\")\nprint(f'Testing Set: {len(test_data)} samples')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:39:11.576741Z","iopub.execute_input":"2025-03-01T05:39:11.57752Z","iopub.status.idle":"2025-03-01T05:39:11.587156Z","shell.execute_reply.started":"2025-03-01T05:39:11.577473Z","shell.execute_reply":"2025-03-01T05:39:11.586225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import T5Tokenizer, T5ForConditionalGeneration\n\nt5_model_name = \"t5-small\"\nt5_tokenizer = T5Tokenizer.from_pretrained(t5_model_name)\nt5_model = T5ForConditionalGeneration.from_pretrained(t5_model_name)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:41:14.111408Z","iopub.execute_input":"2025-03-01T05:41:14.111823Z","iopub.status.idle":"2025-03-01T05:41:17.932894Z","shell.execute_reply.started":"2025-03-01T05:41:14.111797Z","shell.execute_reply":"2025-03-01T05:41:17.931795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.utils.data import DataLoader, Dataset\n\nclass PatchDataset(Dataset):\n    def __init__(self, df):\n        self.texts = df[\"cleaned_problem_statement\"].tolist()\n        self.labels = df[\"patch\"].tolist()\n\n    def __len__(self):\n        return len(self.texts)\n\n    def __getitem__(self, idx):\n        return self.texts[idx], self.labels[idx]\n\ntrain_dataset = PatchDataset(train_data)\ntest_dataset = PatchDataset(test_data)\n\ntrain_loader = DataLoader(train_dataset, batch_size=8, shuffle=True)\ntest_loader = DataLoader(test_dataset, batch_size=8, shuffle=False)\n\n\noptimizer = torch.optim.Adam(t5_model.parameters(), lr=1e-4)\nt5_model.train()\n\nfor epoch in range(5):\n    for problem, patch in train_loader:\n        inputs = t5_tokenizer(problem, return_tensors=\"pt\", padding=True, truncation=True)\n        labels = t5_tokenizer(patch, return_tensors=\"pt\", padding=True, truncation=True).input_ids\n        optimizer.zero_grad()\n        outputs = t5_model(**inputs, labels=labels)\n        loss = outputs.loss\n        loss.backward()\n        optimizer.step()\n    print(f'Epoch {epoch+1}: Loss = {loss.item()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:41:56.864724Z","iopub.execute_input":"2025-03-01T05:41:56.865075Z","iopub.status.idle":"2025-03-01T05:42:48.953851Z","shell.execute_reply.started":"2025-03-01T05:41:56.865034Z","shell.execute_reply":"2025-03-01T05:42:48.952411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"t5_model.eval()\n\ndef generate_patch(problem_text):\n    inputs = t5_tokenizer(problem_text, return_tensors=\"pt\", padding=True, truncation=True)\n    outputs = t5_model.generate(**inputs)\n    return t5_tokenizer.decode(outputs[0], skip_special_tokens=True)\n\n\nsample_problem = test_data.iloc[0]['cleaned_problem_statement']\ngenerated_patch = generate_patch(sample_problem)\n\nprint(\"Porblem Statement:\", sample_problem)\nprint(\"Generated Patch:\", generated_patch)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:42:48.955388Z","iopub.execute_input":"2025-03-01T05:42:48.955765Z","iopub.status.idle":"2025-03-01T05:42:49.38343Z","shell.execute_reply.started":"2025-03-01T05:42:48.955731Z","shell.execute_reply":"2025-03-01T05:42:49.382373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\n# Define the save path\nsave_path = \"/kaggle/working/t5_model\"\n\n# Save the trained model and tokenizer\nt5_model.save_pretrained(save_path)\nt5_tokenizer.save_pretrained(save_path)\n\nprint(\"Model saved successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:42:55.414874Z","iopub.execute_input":"2025-03-01T05:42:55.415267Z","iopub.status.idle":"2025-03-01T05:42:55.647875Z","shell.execute_reply.started":"2025-03-01T05:42:55.41521Z","shell.execute_reply":"2025-03-01T05:42:55.646786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom kaggle_evaluation.konwinski_prize_inference_server import KPrizeInferenceServer\n\ndef get_number_of_instances():\n    # Define a function to return the number of instances for inference\n    return 10  # Example value, modify as needed\n\ndef predict():\n    # Define a function to handle predictions\n    return \"Prediction logic here\"  # Replace with actual implementation\n\n# Initialize the inference server\ninference_server = KPrizeInferenceServer(get_number_of_instances, predict)\n\n# Check if running in Kaggle competition rerun mode\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            '/kaggle/input/konwinski-prize/',  # Path to the entire competition dataset\n            '/kaggle/tmp/konwinski-prize/',   # Path to a scratch directory for unpacking data\n        ),\n        use_concurrency=True,  # Enable concurrency for better performance\n    )\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:45:16.056603Z","iopub.execute_input":"2025-03-01T05:45:16.057041Z","iopub.status.idle":"2025-03-01T05:45:33.231882Z","shell.execute_reply.started":"2025-03-01T05:45:16.057014Z","shell.execute_reply":"2025-03-01T05:45:33.230361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = []\nfor problem in test_data[\"cleaned_problem_statement\"]:\n    generated_patch = generate_patch(problem)\n    predictions.append(generated_patch)\n\nsubmission_df = pd.DataFrame({\n    \"id\": test_data.index,  \n    \"patch\": predictions\n})\n\n\nsubmission_path = \"/kaggle/working/submission.csv\"\nsubmission_df.to_csv(submission_path, index=False)\n\nprint(\"Submission file saved successfully:\", submission_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:45:46.597615Z","iopub.execute_input":"2025-03-01T05:45:46.597985Z","iopub.status.idle":"2025-03-01T05:45:47.419484Z","shell.execute_reply.started":"2025-03-01T05:45:46.597951Z","shell.execute_reply":"2025-03-01T05:45:47.418501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}