{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":69233,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":57753,"modelId":79606}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Qwen2_72b as automl solution\n\nIn this notebook we try to use an LLM (Qwen2 72B) to define the churn label.\n\nIn order to use Qwen2 72B we will need to use a TPU. Otherwise we would be restricted to the usage of weaker LLMs which usually give bad results.\nBe aware that longer prompts and expected outputs might cause even the TPU to fail. I tried to get Quantozation running, but there seems to be an issue with numpy compilation. I gave up on this. If anyone knows how to get this done, please let me know.\n\nThis notebook even has an automated reetry logic to be able to fix it's own bugs (does not always work though). \n\nLet's see what the LLM returns....","metadata":{"_uuid":"44ddaf0f-2509-47ad-b58c-b4ee310cabc4","_cell_guid":"3ed31173-e5b0-4f5b-9ba4-bdcc3ebc56bf","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"%%capture\n!pip install catboost","metadata":{"_uuid":"13cd3b12-9aa3-45b1-8d47-0a071cc17d94","_cell_guid":"45bd90b4-b41d-42c6-a84a-4d90bc42446d","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%capture\n!pip install scikit-learn==1.5.2","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%capture\n!pip uninstall transformers --y","metadata":{"_uuid":"c8090162-01ea-4099-a8c5-842c3e58d80d","_cell_guid":"ed7ecd5d-35e0-4a0c-ae97-191d51b2a05a","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install \"transformers>=4.45.1\"","metadata":{"_uuid":"ff349f62-cd64-4597-9ea1-2eac12ceedd1","_cell_guid":"1521487d-f488-4823-ae97-5cb5a4c30307","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob\nimport pandas as pd\nimport os\nimport requests\nfrom PIL import Image\nimport subprocess\nimport transformers\nfrom torch import cuda, bfloat16\nimport torch\nfrom transformers import AutoTokenizer, AutoModelForCausalLM, AutoProcessor, pipeline#, MllamaForConditionalGeneration\nfrom IPython.display import Markdown, display\n\nos.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"","metadata":{"_uuid":"db4a931d-9a28-4875-b216-4055eb7669ea","_cell_guid":"95402de4-471d-4bb6-afa2-a778fdb8e444","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nsubmission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"_uuid":"8c3e7e8f-b5bf-480f-8a3a-ef54a9d9bd0e","_cell_guid":"85233f4a-aea0-45db-abe8-57bc63acfcf1","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nUSE_MODEL = \"Qwen2_72b\"\nmodel_output_split_mapping = {\n    \"DeekSeekv2\": \"Assistant:\",\n    \"Qwen2\": \"<|im_end|>\\n<|im_start|>assistant\",\n    \"Llama3.2\": \"<|start_header_id|>assistant<|end_header_id|>\",\n    \"Qwen2_57b\": \"<|im_end|>\\n<|im_start|>assistant\",\n    \"Qwen2_72b\": \"<|im_end|>\\n<|im_start|>assistant\",\n}\nmodel_dir_mapping = {\n    \"DeekSeekv2\": \"/kaggle/input/deepseek-v2\",\n    \"Qwen2\": \"/kaggle/input/qwen2/transformers/7b-instruct/1\",\n    \"Llama3.2\": \"/kaggle/input/llama-3.2/transformers/3b-instruct/1\",\n    \"Qwen2_57b\": \"/kaggle/input/qwen2/transformers/57b-a14b-instruct/1\",\n    \"Qwen2_72b\": \"/kaggle/input/qwen2/transformers/72b-instruct/1\",\n}\n\ndef get_llm_pipeline(model_dir): \n    #bnb_config = transformers.BitsAndBytesConfig(\n    #        load_in_4bit=True,\n    #        bnb_4bit_quant_type='nf4',\n    #        bnb_4bit_use_double_quant=True,\n    #        bnb_4bit_compute_dtype=bfloat16,\n    #        llm_int8_enable_fp32_cpu_offload=True\n    #    )\n    \n    # Load the model from the local path\n    model = AutoModelForCausalLM.from_pretrained(\n        model_dir,\n        torch_dtype=torch.bfloat16,\n        device_map=\"auto\",\n        local_files_only=True,  # Ensure the model is loaded from local files\n        #quantization_config=bnb_config,\n        trust_remote_code=True\n    )\n    \n    # Load the tokenizer from the local path\n    tokenizer = AutoTokenizer.from_pretrained(model_dir, local_files_only=True)\n    \n    print(\"Model and tokenizer loaded successfully!\")\n    \n    if tokenizer.pad_token_id is None:\n        tokenizer.pad_token_id = tokenizer.eos_token_id\n    if model.config.pad_token_id is None:\n        model.config.pad_token_id = model.config.eos_token_id\n        \n    pipe = pipeline(\n        \"text-generation\",\n        model=model,\n        tokenizer=tokenizer,\n        torch_dtype=torch.float16,\n        device_map=\"auto\",\n    )\n    return pipe, tokenizer, model\n\n\npipe, tokenizer, model = get_llm_pipeline(model_dir_mapping[USE_MODEL])","metadata":{"_uuid":"f33f6c65-8506-41f9-86c7-16ef4fc32e0e","_cell_guid":"53812806-1ba5-4052-90d7-73538617c4c8","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ninput_file_path = \"/kaggle/input/playground-series-s4e12/\"\nfiles = [file_path for file_path in os.walk(input_file_path)]\ncompetition_topic = \"premium amounts in insurance\"\ntarget = \"Premium Amount\"\ncompetition_type = \"regression\"\ncompetition_metric = \"Root Mean Squared Error (RMSE)\"\ntrain_df_info = train.info()\ntest_df_info = test.info()\nimport_example = f\"\"\"\n    train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\n    test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\n    submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\n\"\"\"\nsubmission_example = \"\"\"\n    submission[target] = y_hat\n    submission.to_csv('submission.csv', index=False)\n\"\"\"\n# Provide the path to your solution file\nfile_path = \"/kaggle/working/llm_automl_solution.py\"\n\nmessages = [\n    {\n        \"role\": \"system\",\n        \"content\": f\"\"\"You are a friendly Kaggle grandmaster and participate in a competition a about {competition_topic} with the competition metric {competition_metric} (the target variable is {target}). \n        You are part of a team and have been asked to show your exceptional expertise by providing robust and highly performant code using Catboost. You are \n        expected to deliver code only without any additional explanation or example usage.\n        \"\"\",\n    },\n    {\n        \"role\": \"user\",\n        \"content\": f\"\"\"\n            Hello my friend! \\n\\n\n            \n            Could you please help me qand write code that takes a dataset and trains Catboost regressors for {competition_type} to predict the target column {target}?\\n\\n\n            \n            The files to import for this competition are under the path {input_file_path}. Here is a list of all files inside this path: {files}. Here is how train.info() looks like: {train_df_info}\\n\n            Additionally here is how test.info() looks like in comparison: {test_df_info}\\n\n            This information will tell you which columns are present in the training data, but not in the test data (i.e. {target}) and vice versa.\\n\n            Here is an example of how the datasets can be imported: {import_example}\\n\\n\n            The submission format is: {submission_example} \\n\n            \n            Instructions on the format:\\n\n            * Only use Catboost!\\n\n            * Always use triple quotes and not just one.\\n\n            * Do not show any example usage.\\n\n            * Do not return any markdown.\\n\n            * Be aware that you can only import files from the path: {input_file_path}\\n\\n\n\n            To help you I collected a summary of what needs to be done:\\n\n            * log transform the target via np.log1p\n            * Convert all category types to object type and fill missing values in object type columns only with 'None'. Here is an example:\\n\n                    categorical_cols = train.select_dtypes(include=['object', 'category']).columns\\n\n                    train[categorical_cols] = train[categorical_cols].astype('str').fillna('None').astype('category')\\n\n                    test[categorical_cols] = test[categorical_cols].astype('str').fillna('None').astype('category')\\n\n            * Convert categorical columns to a numeric representation (implement a logic for unknown categories). Remember that cat_features must be\\n\n                integer or string when being passed to the CatboostRegressor.\\n\n            * Pass the categorical columns into the Catboost regressor object during instantiation to use its inbuilt processing of categorical data.\\n\n            * Create and train the model (one model per iteration in a loop)\\n\n            * Predict on the test set\n            * reverse the log transformation of the predictions via np.expm1\n            * clean up objects and free memory after each loop iteration to release memory (CPU and GPU/TPU)\n            * Create the submission like shown above and do not change the target column name\\n\\n\n\n            Please return the raw Python code only. The code has to be runnable such that the following cell can execute it via the command: !python /kaggle/working/llm_automl_solution.py\n\n            Thank you very much!\n            \"\"\",\n    },\n]\n\nprompt = tokenizer.apply_chat_template(\n    messages, tokenize=False, add_generation_prompt=True\n)\n\noutputs = pipe(prompt, max_new_tokens=15000, do_sample=True, temperature=0.1)\noutputs","metadata":{"_uuid":"35623c8a-e54f-4215-b58e-5ef15379b6dc","_cell_guid":"e9914f70-38f5-4d2a-a472-251e5da1ab7d","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(\n    Markdown(\n            outputs[0][\"generated_text\"].split(\n                model_output_split_mapping[USE_MODEL]\n            )[1]\n        )\n    )","metadata":{"_uuid":"55a8c98d-a75b-4f79-b021-927db0a2f94f","_cell_guid":"8f60a2bf-fbb5-453b-8646-5d208318eb26","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Get the generated text content\ncontent = outputs[0][\"generated_text\"].split(model_output_split_mapping[USE_MODEL])[1]\n\n# Step 2: Remove markdown code block syntax like ```python ... ```\ncleaned_content = content.strip().replace(\"```python\", \"\").replace(\"```\", \"\").strip()\n\nwith open(\"llm_automl_solution.py\", \"a\") as file:\n    #file.write(blending_code_example)\n    #file.write(\"\\n\")\n    file.write(cleaned_content)","metadata":{"_uuid":"eb60f553-8694-4aa5-be6c-7262eadd80dd","_cell_guid":"af92f653-4e46-45ca-912f-0ed7e48b707d","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def execute_and_fix_code(file_path, retries=10):\n    global USE_MODEL\n    attempt = 0\n    while attempt < retries:\n        try:\n            # Try executing the code\n            result = subprocess.run([\"python\", file_path], capture_output=True, text=True)\n            if result.returncode == 0:\n                print(\"Code executed successfully!\")\n                break\n            else:\n                raise Exception(result.stderr)  # Raises the error for handling below\n\n        except Exception as e:\n            error_msg = str(e)\n            print(f\"Error on attempt {attempt+1}: {error_msg}\")\n            \n            # LLM prompt to fix the code\n            messages = [\n                {\n                    \"role\": \"system\",\n                    \"content\": f\"\"\"You are a friendly Kaggle grandmaster and participate in a competition a about {competition_topic} (the target variable shall be {target}). \\n\n        You are part of a team and have been asked to show your exceptional expertise by providing robust and highly performant code. You are \\n\n        expected to kindly deliver code only without any additional explanation or example usage such that automated solutions can make use of your excellent code.\n        \"\"\"\n                },\n                {\n                    \"role\": \"user\",\n                    \"content\": f\"\"\"\n                        Hi, \\n\n                        I have created the following code: {open(file_path, 'r').read()} \\n\n                        The code fails with the following error message: {error_msg} \\n\n                        The files provided for this competition are located in the path: {file_path} (Do not change the path during imports). The files inside are: {files}. \\n\n                        Here is how train.info() looks like: {train_df_info}\\n\n                        Additionally here is how test.info() looks like in comparison: {test_df_info}\\n\n                        The submission format is: {submission_example} \\n\n                        \n                        Please help me and fix the provided code such that the following cell can execute it via the command: !python /kaggle/working/llm_automl_solution.py\n                        This means the code needs a main function and __name__=='main' statement \n                        Provide code only without any additional comment or example, just the fixed code. Thank you very much!\n                        \"\"\"\n                },\n            ]\n            print(f\"Message for the debug model: {messages}\")\n            print(f\"Using model {USE_MODEL}\")\n            pipe, tokenizer, model = get_llm_pipeline(model_dir_mapping[USE_MODEL])\n\n            # Generate a new solution\n            prompt = tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)\n            # generate a new pipeline to prevent having too much context\n            pipe = pipeline(\n                \"text-generation\",\n                model=model,\n                tokenizer=tokenizer,\n                torch_dtype=torch.float16,\n                device_map=\"auto\",\n            )\n            # increasing temperature with more retries\n            outputs = pipe(prompt, max_new_tokens=15000, do_sample=True, temperature=min(0.1 + 0.1 * attempt, 1.))\n\n            # Extract and clean the generated text\n            new_code = outputs[0][\"generated_text\"].split(model_output_split_mapping[USE_MODEL])[1].strip()\n            cleaned_content = new_code.replace(\"```python\", \"\").replace(\"```\", \"\").strip()\n\n            # Overwrite the file with the new code\n            with open(file_path, \"w\") as file:\n                #file.write(blending_code_example)\n                #file.write(\"\\n\")\n                file.write(cleaned_content)\n            \n            attempt += 1\n            print(f\"Retrying... (Attempt {attempt}/{retries})\")\n    else:\n        print(f\"Failed to execute after {retries} attempts.\")\n\n# Try executing and fixing the code\nexecute_and_fix_code(file_path)","metadata":{"_uuid":"1dbe13ca-e0bd-4fc0-8db5-fdd512c7393d","_cell_guid":"d3c77b18-c75c-418f-9523-68b491635b3e","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null}]}