{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84795,"databundleVersionId":10462807,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport io\nimport os\n\nimport shutil\n\nimport pandas as pd\nimport polars as pl\n\nimport kaggle_evaluation.konwinski_prize_inference_server\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:31:55.851737Z","iopub.execute_input":"2024-12-20T09:31:55.852217Z","iopub.status.idle":"2024-12-20T09:32:09.296286Z","shell.execute_reply.started":"2024-12-20T09:31:55.852184Z","shell.execute_reply":"2024-12-20T09:32:09.294813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport zipfile\n\n# Path to the ZIP file\nzip_path = '/kaggle/input/konwinski-prize/data.a_zip'\n\n# Extract the ZIP file\nextract_dir = '/kaggle/working/unzipped_data/'\nos.makedirs(extract_dir, exist_ok=True)\n\nwith zipfile.ZipFile(zip_path, 'r') as zip_ref:\n    zip_ref.extractall(extract_dir)\n\n# List extracted files and directories\nprint(\"Extracted Files:\", os.listdir(extract_dir))\n\n# Check inside the 'data' directory\ndata_dir = os.path.join(extract_dir, 'data')\nprint(\"Contents of 'data' directory:\", os.listdir(data_dir))\n\n# Locate a CSV or readable file inside the 'data' directory\nfor file in os.listdir(data_dir):\n    file_path = os.path.join(data_dir, file)\n    print(\"Found file:\", file_path)\n    \n    # Attempt to read the file as a CSV\n    if file.endswith('.csv'):\n        df = pd.read_csv(file_path)\n        print(\"File successfully read as CSV.\")\n        print(df.head())\n        break\nelse:\n    print(\"No CSV file found. Check the file formats.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:32:16.80466Z","iopub.execute_input":"2024-12-20T09:32:16.805535Z","iopub.status.idle":"2024-12-20T09:32:20.422387Z","shell.execute_reply.started":"2024-12-20T09:32:16.805495Z","shell.execute_reply":"2024-12-20T09:32:20.421367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install pyarrow","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T03:49:41.97748Z","iopub.execute_input":"2024-12-17T03:49:41.977891Z","iopub.status.idle":"2024-12-17T03:49:52.423049Z","shell.execute_reply.started":"2024-12-17T03:49:41.977851Z","shell.execute_reply":"2024-12-17T03:49:52.421944Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Path to the Parquet file\nparquet_file = '/kaggle/working/unzipped_data/data/data.parquet'\n\n# Read the Parquet file\ndf = pd.read_parquet(parquet_file)\n\n# Display the first few rows of the DataFrame\nprint(df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:32:25.155839Z","iopub.execute_input":"2024-12-20T09:32:25.156236Z","iopub.status.idle":"2024-12-20T09:32:25.283542Z","shell.execute_reply.started":"2024-12-20T09:32:25.156197Z","shell.execute_reply":"2024-12-20T09:32:25.282503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_error\n\n# Load Data\nparquet_file = '/kaggle/working/unzipped_data/data/data.parquet'\ndf = pd.read_parquet(parquet_file)\n\n# Step 1: Prepare Target Variable\n# Convert 'issue_numbers' to the count of issues\ndf['issue_count'] = df['issue_numbers'].apply(lambda x: len(x) if isinstance(x, list) else 0)\n\n# Step 2: Text Features Preparation\ntext_columns = ['problem_statement', 'patch', 'test_patch']\ndf['combined_text'] = df[text_columns].apply(lambda row: ' '.join(row.values.astype(str)), axis=1)\n\n# Step 3: Vectorize Text Data (TF-IDF)\nvectorizer = TfidfVectorizer(max_features=5000)\nX_text = vectorizer.fit_transform(df['combined_text'])\n\n# Step 4: Prepare Input Features and Target\nX = X_text  # TF-IDF vectors as features\ny = df['issue_count']  # Target: issue count\n\n# Split the data into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Step 5: Train XGBoost Model\nmodel = XGBRegressor(objective='reg:squarederror', n_estimators=100, learning_rate=0.1)\nmodel.fit(X_train, y_train)\n\n# Step 6: Evaluate the Model\ny_pred = model.predict(X_test)\nrmse = np.sqrt(mean_squared_error(y_test, y_pred))\n\nprint(\"Root Mean Squared Error (RMSE):\", rmse)\ndef instance():\n    return y  # Replace 'y' with the actual value or logic you want to return\n\ndef predict():\n    return y_pred  # Replace 'y_pred' with the actual prediction logic or value\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:32:33.555741Z","iopub.execute_input":"2024-12-20T09:32:33.556136Z","iopub.status.idle":"2024-12-20T09:32:33.86535Z","shell.execute_reply.started":"2024-12-20T09:32:33.556104Z","shell.execute_reply":"2024-12-20T09:32:33.863307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score\n\n# Evaluate the Model on Test Set\ny_pred = model.predict(X_test)\n\n# Calculate RMSE\nrmse = np.sqrt(mean_squared_error(y_test, y_pred))\n\n# Calculate MAE\nmae = mean_absolute_error(y_test, y_pred)\n\n# Calculate R² Score\nr2 = r2_score(y_test, y_pred)\n\nprint(\"Model Performance Metrics:\")\nprint(f\"Root Mean Squared Error (RMSE): {rmse:.4f}\")\nprint(f\"Mean Absolute Error (MAE): {mae:.4f}\")\nprint(f\"R² Score: {r2:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:32:39.9173Z","iopub.execute_input":"2024-12-20T09:32:39.917738Z","iopub.status.idle":"2024-12-20T09:32:39.929638Z","shell.execute_reply.started":"2024-12-20T09:32:39.917703Z","shell.execute_reply":"2024-12-20T09:32:39.928069Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"instance_count = None\n\ndef get_number_of_instances(num_instances: int) -> None:\n    \"\"\" The very first message from the gateway will be the total number of instances to be served.\n    You don't need to edit this function.\n    \"\"\"\n    global instance_count\n    instance_count = num_instances","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:32:46.494895Z","iopub.execute_input":"2024-12-20T09:32:46.495269Z","iopub.status.idle":"2024-12-20T09:32:46.501038Z","shell.execute_reply.started":"2024-12-20T09:32:46.495239Z","shell.execute_reply":"2024-12-20T09:32:46.499708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport shutil\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\nimport io\n\n# Global variable for first prediction check\nfirst_prediction = True\n\n\ndef predict(problem_statement: str, repo_archive: io.BytesIO) -> str:\n    global first_prediction\n\n    # Load Data\n    parquet_file = '/kaggle/working/unzipped_data/data/data.parquet'\n    df = pd.read_parquet(parquet_file)\n\n    # Step 1: Prepare Target Variable\n    # Convert 'issue_numbers' to the count of issues\n    df['issue_count'] = df['issue_numbers'].apply(lambda x: len(x) if isinstance(x, list) else 0)\n\n    # Step 2: Text Features Preparation\n    text_columns = ['problem_statement', 'patch', 'test_patch']\n    df['combined_text'] = df[text_columns].apply(lambda row: ' '.join(row.values.astype(str)), axis=1)\n\n    # Step 3: Vectorize Text Data (TF-IDF)\n    vectorizer = TfidfVectorizer(max_features=5000)\n    X_text = vectorizer.fit_transform(df['combined_text'])\n\n    # Step 4: Prepare Input Features and Target\n    X = X_text  # TF-IDF vectors as features\n    y = df['issue_count']  # Target: issue count\n\n    # Split the data into train and test sets\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n    # Step 5: Train Random Forest Model\n    model = RandomForestRegressor(n_estimators=100, random_state=42)\n    model.fit(X_train, y_train)\n\n    # Step 6: Evaluate the Model\n    y_pred = model.predict(X_test)\n    rmse = np.sqrt(mean_squared_error(y_test, y_pred))\n\n    print(\"Root Mean Squared Error (RMSE):\", rmse)\n\n    # Skip if not first prediction\n    if not first_prediction:\n        return None  # Skip issue.\n\n    # Unpack the repo archive\n    with open('repo_archive.tar', 'wb') as f:\n        f.write(repo_archive.read())\n    repo_path = 'repo'\n    if os.path.exists(repo_path):\n        shutil.rmtree(repo_path)\n    shutil.unpack_archive('repo_archive.tar', extract_dir=repo_path)\n    os.remove('repo_archive.tar')\n\n    # Mark as not the first prediction\n    first_prediction = False\n\n    # Placeholder response\n    return \"Hello World\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:32:51.199836Z","iopub.execute_input":"2024-12-20T09:32:51.200254Z","iopub.status.idle":"2024-12-20T09:32:51.211285Z","shell.execute_reply.started":"2024-12-20T09:32:51.200219Z","shell.execute_reply":"2024-12-20T09:32:51.210053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T04:07:40.741413Z","iopub.execute_input":"2024-12-19T04:07:40.741814Z","iopub.status.idle":"2024-12-19T04:07:40.753538Z","shell.execute_reply.started":"2024-12-19T04:07:40.74178Z","shell.execute_reply":"2024-12-19T04:07:40.752223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure instance_id refers to a valid column in the dataframe\ninstance_id = ''  # Replace 'id' with the actual column name\n\n# Initialize the inference server\ninference_server = kaggle_evaluation.konwinski_prize_inference_server.KPrizeInferenceServer(\n    get_number_of_instances,\n    predict\n)\n\n# Serve the model or run it locally based on the environment\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            '/kaggle/input/konwinski-prize/',  # Path to the competition dataset\n            '/kaggle/tmp/konwinski-prize/',   # Scratch directory for unpacking\n        )\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T09:32:56.4122Z","iopub.execute_input":"2024-12-20T09:32:56.412621Z","iopub.status.idle":"2024-12-20T09:33:22.248Z","shell.execute_reply.started":"2024-12-20T09:32:56.412584Z","shell.execute_reply":"2024-12-20T09:33:22.246826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}