{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":84896,"databundleVersionId":10305135}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1 style=\"text-align: center; margin: 0; padding: 20px; background-color: #2a3128; color: white;\"> Regression with an Insurance Dataset</h1>","metadata":{}},{"cell_type":"markdown","source":"#### The Core Problem: Risk vs. Reward\nInsurance companies survive by correctly guessing how much \"risk\" a customer represents.\n\n+ Underpricing Risk: If the company sets a price that is too low for a high-risk person (someone who has many Previous Claims or a poor Health Score), the company will lose money when they have to pay for that person's accidents.\n\n+ Overpricing Caution: If the company sets a price that is too high for a cautious person (someone with a high Credit Score or zero accidents), that customer will simply leave and go to a competitor who offers a cheaper rate.\n\n**The goal is to find the \"Sweet Spot\"—a price high enough to cover potential risks but low enough to keep the customer happy.**","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"margin: 0; padding: 20px; background-color: #DC143C; color: white; text-align: left;\">1. import libraries and load data</h1>","metadata":{}},{"cell_type":"code","source":"import os\nimport json\nimport subprocess\nimport zipfile\nimport re\nimport textwrap\nimport urllib.request\nfrom pathlib import Path\nimport google.generativeai as genai\nfrom kaggle_secrets import UserSecretsClient","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-19T14:14:44.611075Z","iopub.execute_input":"2026-02-19T14:14:44.611603Z","iopub.status.idle":"2026-02-19T14:14:44.615732Z","shell.execute_reply.started":"2026-02-19T14:14:44.611575Z","shell.execute_reply":"2026-02-19T14:14:44.615069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"user_secrets = UserSecretsClient()\n                           \nMODEL           = \"gemini-3-flash-preview\" \nANTHROPIC_KEY   = os.environ.get(             \n    \"ANTHROPIC_API_KEY\",\n    user_secrets.get_secret(\"GEMINI_API_KEY\")\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T14:14:44.633402Z","iopub.execute_input":"2026-02-19T14:14:44.633617Z","iopub.status.idle":"2026-02-19T14:14:44.703183Z","shell.execute_reply.started":"2026-02-19T14:14:44.633599Z","shell.execute_reply":"2026-02-19T14:14:44.70261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"COMPETITION_ID  = \"playground-series-s4e12\"   \nTOP_N_NOTEBOOKS = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T14:14:44.704166Z","iopub.execute_input":"2026-02-19T14:14:44.704393Z","iopub.status.idle":"2026-02-19T14:14:44.707709Z","shell.execute_reply.started":"2026-02-19T14:14:44.704375Z","shell.execute_reply":"2026-02-19T14:14:44.707019Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"margin: 0; padding: 20px; background-color: #DC143C; color: white; text-align: left;\">2. Fetch Public Notebooks via Kaggle API</h1>","metadata":{}},{"cell_type":"code","source":"user_secrets = UserSecretsClient()\nkaggle_username = user_secrets.get_secret(\"KAGGLE_USERNAME\")\nkaggle_key = user_secrets.get_secret(\"KAGGLE_KEY\")\n\nos.makedirs('/root/.config/kaggle', exist_ok=True)\n\nwith open('/root/.config/kaggle/kaggle.json', 'w') as f:\n    json.dump({\"username\": kaggle_username, \"key\": kaggle_key}, f)\n\nos.chmod('/root/.config/kaggle/kaggle.json', 0o600)\n\nfrom kaggle.api.kaggle_api_extended import KaggleApi\napi = KaggleApi()\napi.authenticate()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T14:14:44.708494Z","iopub.execute_input":"2026-02-19T14:14:44.708819Z","iopub.status.idle":"2026-02-19T14:14:44.861474Z","shell.execute_reply.started":"2026-02-19T14:14:44.708799Z","shell.execute_reply":"2026-02-19T14:14:44.86061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def fetch_top_notebooks(competition: str, top_n: int) -> list[dict]:\n    \"\"\"Use Kaggle Python API to list and download top scored notebooks.\"\"\"\n \n    api = KaggleApi()\n    api.authenticate()\n\n    print(f\"\\n{'='*60}\")\n    print(f\"📥 Fetching top {top_n} notebooks for: {competition}\")\n    print(f\"{'='*60}\")\n\n    try:\n        kernels = api.kernels_list(\n            competition=competition,\n            sort_by='scoreDescending',\n            page_size=top_n\n        )\n    except Exception as e:\n        raise RuntimeError(f\"Kaggle API error during list: {e}\")\n\n    print(f\"✅ Found {len(kernels)} kernels\")\n\n    notebooks = []\n    tmp_dir = Path(\"/kaggle/working/pulled_notebooks\")\n    tmp_dir.mkdir(exist_ok=True, parents=True)\n\n    for i, kernel in enumerate(kernels):\n        ref = kernel.ref  \n \n        title = getattr(kernel, 'title', ref)\n        \n        print(f\"\\n  [{i+1}/{top_n}] Pulling: {title} \")\n\n        pull_dir = tmp_dir / f\"kernel_{i}\"\n        pull_dir.mkdir(exist_ok=True)\n\n        try:\n      \n            api.kernels_pull(ref, path=str(pull_dir))\n            \n            notebook_content = _read_notebook(pull_dir)\n            \n            if notebook_content:\n                notebooks.append({\n                    \"ref\": ref,\n                    \"title\": title,\n                    \"content\": notebook_content\n                })\n                print(f\"    ✅ Loaded ({len(notebook_content)} chars)\")\n            else:\n                print(f\"    ⚠️  No readable content found in {ref}\")\n                \n        except Exception as e:\n            print(f\"    ⚠️  Skipping {ref} — {str(e)}\")\n            continue\n\n    print(f\"\\n✅ Successfully loaded {len(notebooks)} notebooks\")\n    return notebooks","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T14:14:44.863348Z","iopub.execute_input":"2026-02-19T14:14:44.863665Z","iopub.status.idle":"2026-02-19T14:14:44.871141Z","shell.execute_reply.started":"2026-02-19T14:14:44.863636Z","shell.execute_reply":"2026-02-19T14:14:44.87052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def fetch_specific_notebooks(notebook_refs: list[str]) -> list[dict]:\n    \"\"\"Fetch and extract content from a specific list of notebook references.\"\"\"\n    \n    api = KaggleApi()\n    api.authenticate()\n\n    notebooks = []\n    tmp_dir = Path(\"/kaggle/working/targeted_notebooks\")\n    tmp_dir.mkdir(exist_ok=True, parents=True)\n\n    print(f\"\\n{'='*60}\")\n    print(f\"🎯 Fetching {len(notebook_refs)} targeted notebooks\")\n    print(f\"{'='*60}\")\n\n    for i, ref in enumerate(notebook_refs):\n        print(f\"\\n  [{i+1}/{len(notebook_refs)}] Pulling: {ref}\")\n        \n        slug = ref.split('/')[-1]\n        pull_dir = tmp_dir / slug\n        pull_dir.mkdir(exist_ok=True)\n\n        try:\n            api.kernels_pull(ref, path=str(pull_dir))\n            \n            notebook_content = _read_notebook(pull_dir)\n            \n            if notebook_content:\n                notebooks.append({\n                    \"ref\": ref,\n                    \"title\": ref.split('/')[-1],\n                    \"content\": notebook_content\n                })\n                print(f\"    ✅ Successfully extracted.\")\n            else:\n                print(f\"    ⚠️  Empty or unreadable.\")\n                \n        except Exception as e:\n            print(f\"    ❌ Error pulling {ref}: {e}\")\n            continue\n\n    return notebooks","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T14:14:44.87188Z","iopub.execute_input":"2026-02-19T14:14:44.872133Z","iopub.status.idle":"2026-02-19T14:14:44.882929Z","shell.execute_reply.started":"2026-02-19T14:14:44.872101Z","shell.execute_reply":"2026-02-19T14:14:44.882306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def _read_notebook(folder: Path) -> str:\n    \"\"\"Extract code cells from .ipynb or read .py file.\"\"\"\n  \n    for nb_file in folder.glob(\"*.ipynb\"):\n        try:\n            with open(nb_file) as f:\n                nb = json.load(f)\n            cells = []\n            for cell in nb.get(\"cells\", []):\n                source = \"\".join(cell.get(\"source\", []))\n                if cell[\"cell_type\"] == \"code\" and source.strip():\n                    cells.append(f\"# CODE CELL\\n{source}\")\n                elif cell[\"cell_type\"] == \"markdown\" and source.strip():\n                    cells.append(f\"# MARKDOWN: {source[:200]}\")\n            return \"\\n\\n\".join(cells)\n        except Exception:\n            continue\n\n    for py_file in folder.glob(\"*.py\"):\n        try:\n            return py_file.read_text(errors=\"ignore\")\n        except Exception:\n            continue\n\n    return \"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T14:14:44.883814Z","iopub.execute_input":"2026-02-19T14:14:44.88409Z","iopub.status.idle":"2026-02-19T14:14:44.895525Z","shell.execute_reply.started":"2026-02-19T14:14:44.884066Z","shell.execute_reply":"2026-02-19T14:14:44.894928Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"margin: 0; padding: 20px; background-color: #DC143C; color: white; text-align: left;\">3. Build LLM Prompt</h1>","metadata":{}},{"cell_type":"code","source":"def build_prompt(competition: str, notebooks: list[dict]) -> str:\n    \"\"\"Construct the prompt that instructs the LLM to synthesize best code.\"\"\"\n\n    nb_sections = []\n    for i, nb in enumerate(notebooks):\n     \n        content = nb[\"content\"]\n        if len(content) > 8000:\n            content = content[:4000] + \"\\n\\n... [TRUNCATED] ...\\n\\n\" + content[-4000:]\n\n        nb_sections.append(\n            f\"{'─'*50}\\n\"\n            f\"NOTEBOOK #{i+1}: {nb['title']}\\n\"\n            f\"Reference: {nb['ref']} \\n\"\n            f\"{'─'*50}\\n\"\n            f\"{content}\"\n        )\n\n    all_notebooks = \"\\n\\n\".join(nb_sections)\n\n    prompt = textwrap.dedent(f\"\"\"\n    You are an expert Kaggle data scientist. I will show you the top {len(notebooks)} \n    public notebooks from the Kaggle competition: **{competition}**\n\n    Your job is to:\n    1. STUDY all notebooks carefully\n    2. IDENTIFY the best patterns: feature engineering, model choices, CV strategy, \n       preprocessing, ensembling, and any clever tricks\n    3. SYNTHESIZE a single, clean, optimized Python script that combines ALL the \n       best ideas from these notebooks into one superior solution\n\n    STRICT REQUIREMENTS for the generated code:\n    - Must be complete, runnable Python code (no placeholders, no \"...\" gaps)\n    - Use standard Kaggle paths: /kaggle/input/{competition}/train.csv etc.\n    - Include all imports at the top\n    - Add clear section comments (# ── SECTION NAME ──)\n    - Handle missing values robustly\n    - Use the same metric as the competition (detect it from the notebooks)\n    - Save submission to: /kaggle/working/submission.csv\n    - Code must work when passed to Python's exec() function\n    - Do NOT include any markdown formatting, backticks, or explanation outside the code\n    - The ENTIRE response must be valid Python code only\n\n    Here are the top notebooks to study:\n\n    {all_notebooks}\n\n    Output ONLY pure Python code. No markdown, no backticks. \n    CRITICAL: Do NOT use float16 for memory reduction or data types.\n    when EXECUTING GENERATED CODE let me know in which phase we are\n    Now write the single best Python solution that beats all of them:\n    \"\"\").strip()\n\n    return prompt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T14:14:44.896461Z","iopub.execute_input":"2026-02-19T14:14:44.896722Z","iopub.status.idle":"2026-02-19T14:14:44.9103Z","shell.execute_reply.started":"2026-02-19T14:14:44.896694Z","shell.execute_reply":"2026-02-19T14:14:44.909684Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"margin: 0; padding: 20px; background-color: #DC143C; color: white; text-align: left;\">4. Call the LLM </h1>","metadata":{}},{"cell_type":"code","source":"def call_llm(prompt: str, api_key: str, model: str) -> str:\n    \"\"\"Send prompt to Gemini and get generated code back.\"\"\"\n    print(f\"\\n{'='*60}\")\n    print(f\"🤖 Calling LLM (Gemini): {model}\")\n    print(f\"    Prompt size: {len(prompt):,} characters\")\n    print(f\"{'='*60}\")\n\n    payload = json.dumps({\n        \"contents\": [{\n            \"parts\": [{\n                \"text\": f\"System: You are an elite Kaggle grandmaster. Output ONLY pure Python code. No markdown, no backticks.\\n\\nUser: {prompt}\"\n            }]\n        }],\n        \"generationConfig\": {\n            \"maxOutputTokens\": 8192,\n            \"temperature\": 0.2\n        }\n    }).encode(\"utf-8\")\n\n    url = f\"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent?key={api_key}\"\n\n    req = urllib.request.Request(\n        url,\n        data=payload,\n        headers={\"Content-Type\": \"application/json\"},\n        method=\"POST\"\n    )\n\n    try:\n        with urllib.request.urlopen(req, timeout=300) as resp:\n            response = json.loads(resp.read().decode(\"utf-8\"))\n        \n        generated_code = response[\"candidates\"][0][\"content\"][\"parts\"][0][\"text\"]\n        print(f\"✅ LLM returned {len(generated_code):,} characters\")\n        return generated_code\n        \n    except Exception as e:\n        print(f\"💥 Failed to call Gemini: {e}\")\n\n        if hasattr(e, 'read'):\n            print(f\"Error details: {e.read().decode()}\")\n        return \"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T14:14:44.911145Z","iopub.execute_input":"2026-02-19T14:14:44.911769Z","iopub.status.idle":"2026-02-19T14:14:44.925287Z","shell.execute_reply.started":"2026-02-19T14:14:44.911741Z","shell.execute_reply":"2026-02-19T14:14:44.92472Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"margin: 0; padding: 20px; background-color: #DC143C; color: white; text-align: left;\">5. Clean & Validate the Generated Code </h1>","metadata":{}},{"cell_type":"code","source":"def clean_code(raw_code: str) -> str:\n    \"\"\"Strip any accidental markdown fences the LLM might have added.\"\"\"\n  \n    fence_match = re.search(r\"```(?:python)?\\n(.*?)```\", raw_code, re.DOTALL)\n    if fence_match:\n        code = fence_match.group(1).strip()\n    else:\n    \n        code = raw_code.strip().strip(\"`\").strip()\n\n    return code\n\n\ndef validate_code(code: str) -> bool:\n    \"\"\"Check if the code compiles without syntax errors.\"\"\"\n    try:\n        compile(code, \"<generated>\", \"exec\")\n        print(\"✅ Syntax validation passed\")\n        return True\n    except SyntaxError as e:\n        print(f\"❌ Syntax error in generated code: {e}\")\n        return False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T14:14:44.926022Z","iopub.execute_input":"2026-02-19T14:14:44.926302Z","iopub.status.idle":"2026-02-19T14:14:44.938154Z","shell.execute_reply.started":"2026-02-19T14:14:44.926283Z","shell.execute_reply":"2026-02-19T14:14:44.93758Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"margin: 0; padding: 20px; background-color: #DC143C; color: white; text-align: left;\">5. Save & Execute the Generated Code </h1>","metadata":{}},{"cell_type":"code","source":"def save_code(code: str, path: str = \"/kaggle/working/generated_solution.py\"):\n    \"\"\"Save the generated code to disk for reference.\"\"\"\n    with open(path, \"w\") as f:\n        f.write(\"# ═══ AUTO-GENERATED BY KAGGLE LLM AGENT ═══\\n\")\n        f.write(f\"# Competition: {COMPETITION_ID}\\n\")\n        f.write(f\"# Model: {MODEL}\\n\\n\")\n        f.write(code)\n    print(f\"💾 Code saved to: {path}\")\n    return path\n\n\ndef execute_code(code: str):\n    \"\"\"Run the generated code using exec() in a clean namespace.\"\"\"\n    print(f\"\\n{'='*60}\")\n    print(\"🚀 EXECUTING GENERATED CODE\")\n    print(f\"{'='*60}\\n\")\n\n    namespace = {\n        \"__name__\": \"__main__\",\n        \"__file__\": \"/kaggle/working/generated_solution.py\"\n    }\n\n    exec(compile(code, \"<generated_solution>\", \"exec\"), namespace)\n    print(\"\\n✅ Execution completed successfully!\")\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T14:14:44.939701Z","iopub.execute_input":"2026-02-19T14:14:44.939979Z","iopub.status.idle":"2026-02-19T14:14:44.951713Z","shell.execute_reply.started":"2026-02-19T14:14:44.939937Z","shell.execute_reply":"2026-02-19T14:14:44.950988Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"margin: 0; padding: 20px; background-color: #DC143C; color: white; text-align: left;\">6. MAIN PIPELINE  </h1>","metadata":{}},{"cell_type":"code","source":"my_targets = [\n    \"mikhailnaumov/regression-with-an-insurance-cat-lgb-xgb-hgb-ydf\",\n    \"backpaker/rid-catboost-nonlog-as-feature\",\n    \"swagician/tps-catboost-optuna-health-score-secrets\",\n    \"mohitsharma231234/s4e12-insurance-amount-feature-eng-lgbm\"\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T14:14:44.952583Z","iopub.execute_input":"2026-02-19T14:14:44.95332Z","iopub.status.idle":"2026-02-19T14:14:44.961247Z","shell.execute_reply.started":"2026-02-19T14:14:44.953293Z","shell.execute_reply":"2026-02-19T14:14:44.960659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def main():\n    print(\"\"\"\n╔══════════════════════════════════════════════════════════════╗\n           🏆  KAGGLE LLM AGENT  🏆                                        \n   Top Notebooks → LLM Analysis → Optimized Code → exec()                   \n╚══════════════════════════════════════════════════════════════╝\n    \"\"\")\n\n    if ANTHROPIC_KEY == \"YOUR_API_KEY_HERE\" or not ANTHROPIC_KEY:\n        raise ValueError(\n            \"❌ No API key found!\\n\"\n            \"   Add your Anthropic key as a Kaggle Secret named: ANTHROPIC_API_KEY\\n\"\n            \"   Then enable it in the notebook's 'Add-ons' → 'Secrets' panel.\"\n        )\n\n    notebooks = fetch_top_notebooks(COMPETITION_ID, TOP_N_NOTEBOOKS)\n    #notebooks = fetch_specific_notebooks(my_targets)\n\n    if not notebooks:\n        raise RuntimeError(\"❌ No notebooks could be loaded. Check the competition ID.\")\n\n    prompt = build_prompt(COMPETITION_ID, notebooks)\n\n    raw_code = call_llm(prompt, ANTHROPIC_KEY, MODEL)\n\n    clean = clean_code(raw_code)\n\n    if not validate_code(clean):\n        print(\"\\n⚠️  Attempting to fix by re-prompting LLM...\")\n        fix_prompt = (\n            f\"The following Python code has a syntax error. Fix it and return \"\n            f\"ONLY valid Python code with no markdown:\\n\\n{clean}\"\n        )\n        raw_code = call_llm(fix_prompt, ANTHROPIC_KEY, MODEL)\n        clean = clean_code(raw_code)\n        if not validate_code(clean):\n            raise RuntimeError(\"❌ LLM generated invalid code even after retry.\")\n\n    saved_path = save_code(clean)\n\n    print(f\"\\n{'='*60}\")\n    print(\"📋 GENERATED CODE PREVIEW (first 50 lines):\")\n    print(f\"{'='*60}\")\n    for line in clean.split(\"\\n\")[:50]:\n        print(line)\n    print(\"...\\n\")\n\n\n    print(f\"\\n{'='*60}\")\n    print(\"⚡ Ready to execute!\")\n    print(f\"   Full code saved at: {saved_path}\")\n    print(f\"{'='*60}\")\n\n    AUTO_EXECUTE = True\n\n    if AUTO_EXECUTE:\n        execute_code(clean)\n    else:\n        print(\"ℹ️  AUTO_EXECUTE=False — Review the code then run:\")\n        print(f\"   exec(open('{saved_path}').read())\")\n\n    return clean   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T14:14:44.96218Z","iopub.execute_input":"2026-02-19T14:14:44.962485Z","iopub.status.idle":"2026-02-19T14:14:44.975076Z","shell.execute_reply.started":"2026-02-19T14:14:44.962466Z","shell.execute_reply":"2026-02-19T14:14:44.974434Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"margin: 0; padding: 20px; background-color: #DC143C; color: white; text-align: left;\">7. Entry Point  </h1>","metadata":{}},{"cell_type":"code","source":"generated_code = main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T14:14:44.975984Z","iopub.execute_input":"2026-02-19T14:14:44.976218Z"}},"outputs":[],"execution_count":null}]}