{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaL4","dataSources":[{"sourceId":84795,"databundleVersionId":10934030,"sourceType":"competition"},{"sourceId":218890054,"sourceType":"kernelVersion"},{"sourceId":236932,"sourceType":"modelInstanceVersion","modelInstanceId":202348,"modelId":224071}],"dockerImageVersionId":30887,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n!python -m pip install  /kaggle/input/browser-notifications-in-a-kaggle-kernel/*.whl \n!python -m pip install jupyter -q --no-index --find-links=/kaggle/input/browser-notifications-in-a-kaggle-kernel/ \n!python -m pip install -q /kaggle/input/browser-notifications-in-a-kaggle-kernel/jupyter-notify/dist/jupyternotify-0.1.15-py2.py3-none-any.whl --no-index\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-10T14:39:59.993495Z","iopub.execute_input":"2025-02-10T14:39:59.993865Z","iopub.status.idle":"2025-02-10T14:40:08.537939Z","shell.execute_reply.started":"2025-02-10T14:39:59.993827Z","shell.execute_reply":"2025-02-10T14:40:08.537031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%reload_ext jupyternotify","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-10T14:40:08.53908Z","iopub.execute_input":"2025-02-10T14:40:08.539334Z","iopub.status.idle":"2025-02-10T14:40:08.761864Z","shell.execute_reply.started":"2025-02-10T14:40:08.539311Z","shell.execute_reply":"2025-02-10T14:40:08.761255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip install vllm==0.7.2 --target=/kaggle/working","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from vllm import LLM, SamplingParams\nimport warnings\nimport os\n\nwarnings.simplefilter('ignore')\n\nos.environ[\"CUDA_VISIBLE_DEVICES\"] = \"0,1,2,3\"\nos.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"\n\n\nllm_model_pth = '/kaggle/input/deepseek-r1/transformers/deepseek-r1-distill-qwen-32b-awq/1'\n\n\nMAX_NUM_SEQS = 4\nMAX_MODEL_LEN = 32_768\nMAX_TOKENS = 32_768\n\nllm = LLM(\n    llm_model_pth,\n    # dtype=\"half\",               # The data type for the model weights and activations\n    max_num_seqs=MAX_NUM_SEQS,   # Maximum number of sequences per iteration. Default is 256\n    max_model_len=MAX_MODEL_LEN, # Model context length\n    trust_remote_code=True,      # Trust remote code (e.g., from HuggingFace) when downloading the model and tokenizer\n    tensor_parallel_size=4,      # The number of GPUs to use for distributed execution with tensor parallelism\n    gpu_memory_utilization=0.95, # The ratio (between 0 and 1) of GPU memory to reserve for the model\n    seed=2024,\n)\n\ntokenizer = llm.get_tokenizer()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def test_generate(content, temperature=0):\n\n    sampling_params = SamplingParams(\n        temperature=temperature,              # randomness of the sampling\n        min_p=0.01,\n        skip_special_tokens=True,     # Whether to skip special tokens in the output\n        max_tokens=MAX_MODEL_LEN,\n    )\n    \n    list_of_messages = [\n        [\n            {\n                \"role\": \"user\",\n                \"content\": content\n            },\n        ]\n    ]\n\n    list_of_texts = [\n        tokenizer.apply_chat_template(\n            conversation=messages,\n            tokenize=False,\n            add_generation_prompt=True\n        )\n        for messages in list_of_messages\n    ]\n    print([len(tokenizer.encode(text)) for text in list_of_texts])\n\n    request_output = llm.generate(prompts=list_of_texts, sampling_params=sampling_params)\n    if not request_output:\n        return \"\"\n    return request_output[0].outputs[0].text","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%notify\nfrom IPython.display import display, Latex\ndisplay(Latex(test_generate(\"Tell a random number with 4 decimal places between 1 to 10.\")))\nprint('---'*20)\ndisplay(Latex(test_generate(\"Tell a random number with 4 decimal places between 1 to 10.\")))\nprint('---'*20)\ndisplay(Latex(test_generate(\"Tell a random number with 4 decimal places between 1 to 10.\",temperature=1)))\nprint('---'*20)\ndisplay(Latex(test_generate(\"Tell a random number with 4 decimal places between 1 to 10.\",temperature=1)))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!wget https://raw.githubusercontent.com/vllm-project/vllm/main/collect_env.py\n# For security purposes, please feel free to check the contents of collect_env.py before running it.\n!python collect_env.py","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# transformers test","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install autoawq","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.environ[\"CUDA_VISIBLE_DEVICES\"] = \"0,1,2,3\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.cuda.empty_cache()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import AutoTokenizer, AutoModelForCausalLM\n\n# Load the model and tokenizer\nmodel_name = \"/kaggle/input/deepseek-r1/transformers/deepseek-r1-distill-qwen-32b-awq/1\"\ntokenizer = AutoTokenizer.from_pretrained(model_name, trust_remote_code=True)\nmodel = AutoModelForCausalLM.from_pretrained(model_name, trust_remote_code=True, device_map = \"sequential\" ).cuda()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-10T14:39:19.058348Z","iopub.execute_input":"2025-02-10T14:39:19.058636Z","iopub.status.idle":"2025-02-10T14:39:35.500428Z","shell.execute_reply.started":"2025-02-10T14:39:19.058613Z","shell.execute_reply":"2025-02-10T14:39:35.499706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!nvidia-smi","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ps 8505","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!chmod +x /kaggle/working/triton/backends/nvidia/bin/ptxas","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def test_generate2(content, temperature=0.0):\n    \n    inputs = tokenizer(content, return_tensors=\"pt\").to(\"cuda\")\n    import torch\n    torch.manual_seed(42)\n    with torch.no_grad():\n      outputs = model.generate(**inputs, max_length=512, temperature=temperature, do_sample=temperature)\n    \n    return (tokenizer.decode(outputs[0], skip_special_tokens=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-10T14:39:51.930893Z","iopub.execute_input":"2025-02-10T14:39:51.931495Z","iopub.status.idle":"2025-02-10T14:39:51.93551Z","shell.execute_reply.started":"2025-02-10T14:39:51.931467Z","shell.execute_reply":"2025-02-10T14:39:51.93484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%notify\nfrom IPython.display import display, Latex\ndisplay(Latex(test_generate2(\"Tell a random number with 4 decimal places between 1 to 10.\")))\nprint('---'*20)\ndisplay(Latex(test_generate2(\"Tell a random number with 4 decimal places between 1 to 10.\")))\nprint('---'*20)\ndisplay(Latex(test_generate2(\"Tell a random number with 4 decimal places between 1 to 10.\",temperature=1)))\nprint('---'*20)\ndisplay(Latex(test_generate2(\"Tell a random number with 4 decimal places between 1 to 10.\",temperature=1)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-10T14:40:08.762886Z","iopub.execute_input":"2025-02-10T14:40:08.763106Z","iopub.status.idle":"2025-02-10T15:01:21.834418Z","shell.execute_reply.started":"2025-02-10T14:40:08.763087Z","shell.execute_reply":"2025-02-10T15:01:21.833781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pkill -9 8505 ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ps 8505","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}