{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84795,"databundleVersionId":10934030,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:08:46.102215Z","iopub.execute_input":"2025-03-01T05:08:46.102593Z","iopub.status.idle":"2025-03-01T05:08:47.424476Z","shell.execute_reply.started":"2025-03-01T05:08:46.102554Z","shell.execute_reply":"2025-03-01T05:08:47.423239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\nimport os\n\nzip_path = \"/kaggle/input/konwinski-prize/data.a_zip\"\n\nextract_path = \"/kaggle/working/konwinski_data\"\n\nos.makedirs(extract_path, exist_ok=True)\n\nwith zipfile.ZipFile(zip_path, \"r\") as zip_ref:\n    zip_ref.extractall(extract_path)\nprint(\"Extraction complete!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:08:47.426117Z","iopub.execute_input":"2025-03-01T05:08:47.426705Z","iopub.status.idle":"2025-03-01T05:08:53.124026Z","shell.execute_reply.started":"2025-03-01T05:08:47.42666Z","shell.execute_reply":"2025-03-01T05:08:53.122861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"extracted_files = os.listdir(extract_path)\nprint('Extracted files:', extracted_files)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:08:53.126502Z","iopub.execute_input":"2025-03-01T05:08:53.126967Z","iopub.status.idle":"2025-03-01T05:08:53.132985Z","shell.execute_reply.started":"2025-03-01T05:08:53.126909Z","shell.execute_reply":"2025-03-01T05:08:53.131893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:08:53.134622Z","iopub.execute_input":"2025-03-01T05:08:53.135057Z","iopub.status.idle":"2025-03-01T05:08:54.086706Z","shell.execute_reply.started":"2025-03-01T05:08:53.135023Z","shell.execute_reply":"2025-03-01T05:08:54.085709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\n\nzip_path = \"/kaggle/input/konwinski-prize/data.a_zip\"  # Update with actual path\nextract_path = \"/kaggle/working/konwinski_data/\"\n\nwith zipfile.ZipFile(zip_path, \"r\") as zip_ref:\n    zip_ref.extractall(extract_path)\n\nprint(\"Files after extraction:\", os.listdir(extract_path))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:08:54.087743Z","iopub.execute_input":"2025-03-01T05:08:54.088269Z","iopub.status.idle":"2025-03-01T05:08:56.274494Z","shell.execute_reply.started":"2025-03-01T05:08:54.088238Z","shell.execute_reply":"2025-03-01T05:08:56.273612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"extract_path = \"/kaggle/working/konwinski_data/data\"\nparquet_path = os.path.join(extract_path, \"data.parquet\")\n\ndf = pd.read_parquet(parquet_path)\n\nprint(\"Dataset Loaded Successfully!\")\nprint(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:08:56.275383Z","iopub.execute_input":"2025-03-01T05:08:56.275672Z","iopub.status.idle":"2025-03-01T05:08:56.554139Z","shell.execute_reply.started":"2025-03-01T05:08:56.275647Z","shell.execute_reply":"2025-03-01T05:08:56.553046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:08:56.555209Z","iopub.execute_input":"2025-03-01T05:08:56.555588Z","iopub.status.idle":"2025-03-01T05:08:56.585214Z","shell.execute_reply.started":"2025-03-01T05:08:56.55555Z","shell.execute_reply":"2025-03-01T05:08:56.58406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:08:56.586156Z","iopub.execute_input":"2025-03-01T05:08:56.586462Z","iopub.status.idle":"2025-03-01T05:08:56.609197Z","shell.execute_reply.started":"2025-03-01T05:08:56.586433Z","shell.execute_reply":"2025-03-01T05:08:56.608066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe(include=\"all\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:08:56.61213Z","iopub.execute_input":"2025-03-01T05:08:56.612455Z","iopub.status.idle":"2025-03-01T05:08:56.668323Z","shell.execute_reply.started":"2025-03-01T05:08:56.612419Z","shell.execute_reply":"2025-03-01T05:08:56.667234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install nltk transformers datasets\n\nimport nltk\nfrom nltk.tokenize import word_tokenize\nfrom nltk.corpus import stopwords\nimport string\nimport re\n\nnltk.download('punkt')\nnltk.download('stopwords')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T05:08:56.669787Z","iopub.execute_input":"2025-03-01T05:08:56.670116Z","iopub.status.idle":"2025-03-01T05:10:07.455643Z","shell.execute_reply.started":"2025-03-01T05:08:56.670083Z","shell.execute_reply":"2025-03-01T05:10:07.454598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def clean_text(text):\n    text = text.lower()\n    text = re.sub(r'\\d+', '', text)\n    text = text.translate(str.maketrans(\"\", \"\", string.punctuation))\n    tokens = word_tokenize(text)\n    tokens = [word for word in tokens if word not in stopwords.words('english')]\n    return \" \".join(tokens)\n\ndf['cleaned_problem_statement'] = df['problem_statement'].apply(clean_text)\n\nprint(\"Sample cleaned text:\")\nprint(df['cleaned_problem_statement'].head())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import AutoTokenizer, AutoModel\nimport torch\n\nmodel_name = \"bert-base-uncased\"\ntokenizer = AutoTokenizer.from_pretrained(model_name)\nmodel = AutoModel.from_pretrained(model_name)\n\n\ndef get_embedding(text):\n    inputs = tokenizer(text, return_tensors=\"pt\", truncation=True, padding=True)\n    with torch.no_grad():\n        outputs = model(**inputs)\n    return outputs.last_hidden_state.mean(dim=1).squeeze().numpy()\n\ndf['problem_embedding'] = df['cleaned_problem_statement'].apply(get_embedding)\nprint(\"Embedding generated successfully!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_data, test_data = train_test_split(df, test_size=0.2, random_state=42)\n\n\n\nprint(f\"Training Set: {len(train_data)} samples\")\nprint(f'Testing Set: {len(test_data)} samples')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import T5Tokenizer, T5ForConditionalGeneration\n\nt5_model_name = \"t5-small\"\nt5_tokenizer = T5Tokenizer.from_pretrained(t5_model_name)\nt5_model = T5ForConditionalGeneration.from_pretrained(t5_model_name)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.utils.data import DataLoader, Dataset\n\nclass PatchDataset(Dataset):\n    def __init__(self, df):\n        self.texts = df[\"cleaned_problem_statement\"].tolist()\n        self.labels = df[\"patch\"].tolist()\n\n    def __len__(self):\n        return len(self.texts)\n\n    def __getitem__(self, idx):\n        return self.texts[idx], self.labels[idx]\n\ntrain_dataset = PatchDataset(train_data)\ntest_dataset = PatchDataset(test_data)\n\ntrain_loader = DataLoader(train_dataset, batch_size=8, shuffle=True)\ntest_loader = DataLoader(test_dataset, batch_size=8, shuffle=False)\n\n\noptimizer = torch.optim.Adam(t5_model.parameters(), lr=1e-4)\nt5_model.train()\n\nfor epoch in range(5):\n    for problem, patch in train_loader:\n        inputs = t5_tokenizer(problem, return_tensors=\"pt\", padding=True, truncation=True)\n        labels = t5_tokenizer(patch, return_tensors=\"pt\", padding=True, truncation=True).input_ids\n        optimizer.zero_grad()\n        outputs = t5_model(**inputs, labels=labels)\n        loss = outputs.loss\n        loss.backward()\n        optimizer.step()\n    print(f'Epoch {epoch+1}: Loss = {loss.item()}')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"t5_model.eval()\n\ndef generate_patch(problem_text):\n    inputs = t5_tokenizer(problem_text, return_tensors=\"pt\", padding=True, truncation=True)\n    outputs = t5_model.generate(**inputs)\n    return t5_tokenizer.decode(outputs[0], skip_special_tokens=True)\n\n\nsample_problem = test_data.iloc[0]['cleaned_problem_statement']\ngenerated_patch = generate_patch(sample_problem)\n\nprint(\"Porblem Statement:\", sample_problem)\nprint(\"Generated Patch:\", generated_patch)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\n# Define the save path\nsave_path = \"/kaggle/working/t5_model\"\n\n# Save the trained model and tokenizer\nt5_model.save_pretrained(save_path)\nt5_tokenizer.save_pretrained(save_path)\n\nprint(\"Model saved successfully!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = []\nfor problem in test_data[\"cleaned_problem_statement\"]:\n    generated_patch = generate_patch(problem)\n    predictions.append(generated_patch)\n\nsubmission_df = pd.DataFrame({\n    \"id\": test_data.index,  \n    \"patch\": predictions\n})\n\n\nsubmission_path = \"/kaggle/working/submission.csv\"\nsubmission_df.to_csv(submission_path, index=False)\n\nprint(\"Submission file saved successfully:\", submission_path)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}