|
|
|
@@ -151,16 +151,14 @@
|
|
|
|
|
"metadata": {},
|
|
|
|
|
"outputs": [],
|
|
|
|
|
"source": [
|
|
|
|
|
"server_process, port = launch_server_cmd(\n",
|
|
|
|
|
" \"\"\"\n",
|
|
|
|
|
"server_process, port = launch_server_cmd(\"\"\"\n",
|
|
|
|
|
"python3 -m sglang.launch_server --model-path meta-llama/Meta-Llama-3.1-8B-Instruct \\\n",
|
|
|
|
|
" --enable-lora \\\n",
|
|
|
|
|
" --lora-paths lora0=algoprog/fact-generation-llama-3.1-8b-instruct-lora \\\n",
|
|
|
|
|
" lora1=Nutanix/Meta-Llama-3.1-8B-Instruct_lora_4_alpha_16 \\\n",
|
|
|
|
|
" --max-loras-per-batch 2 \\\n",
|
|
|
|
|
" --log-level warning \\\n",
|
|
|
|
|
"\"\"\"\n",
|
|
|
|
|
")\n",
|
|
|
|
|
"\"\"\")\n",
|
|
|
|
|
"\n",
|
|
|
|
|
"wait_for_server(f\"http://localhost:{port}\", process=server_process)"
|
|
|
|
|
]
|
|
|
|
@@ -227,8 +225,7 @@
|
|
|
|
|
"\n",
|
|
|
|
|
"# The `--target-lora-modules` param below is technically not needed, as the server will infer it from lora0 which already has all the target modules specified.\n",
|
|
|
|
|
"# We are adding it here just to demonstrate usage.\n",
|
|
|
|
|
"server_process, port = launch_server_cmd(\n",
|
|
|
|
|
" \"\"\"\n",
|
|
|
|
|
"server_process, port = launch_server_cmd(\"\"\"\n",
|
|
|
|
|
" python3 -m sglang.launch_server --model-path meta-llama/Meta-Llama-3.1-8B-Instruct \\\n",
|
|
|
|
|
" --enable-lora \\\n",
|
|
|
|
|
" --cuda-graph-max-bs 2 \\\n",
|
|
|
|
@@ -236,8 +233,7 @@
|
|
|
|
|
" --max-lora-rank 256\n",
|
|
|
|
|
" --lora-target-modules all\n",
|
|
|
|
|
" --log-level warning\n",
|
|
|
|
|
" \"\"\"\n",
|
|
|
|
|
")\n",
|
|
|
|
|
" \"\"\")\n",
|
|
|
|
|
"\n",
|
|
|
|
|
"url = f\"http://127.0.0.1:{port}\"\n",
|
|
|
|
|
"wait_for_server(url, process=server_process)"
|
|
|
|
@@ -435,8 +431,7 @@
|
|
|
|
|
"metadata": {},
|
|
|
|
|
"outputs": [],
|
|
|
|
|
"source": [
|
|
|
|
|
"server_process, port = launch_server_cmd(\n",
|
|
|
|
|
" \"\"\"\n",
|
|
|
|
|
"server_process, port = launch_server_cmd(\"\"\"\n",
|
|
|
|
|
" python3 -m sglang.launch_server --model-path meta-llama/Meta-Llama-3.1-8B-Instruct \\\n",
|
|
|
|
|
" --enable-lora \\\n",
|
|
|
|
|
" --cuda-graph-max-bs 8 \\\n",
|
|
|
|
@@ -448,8 +443,7 @@
|
|
|
|
|
" {\"lora_name\":\"lora1\",\"lora_path\":\"algoprog/fact-generation-llama-3.1-8b-instruct-lora\"} \\\n",
|
|
|
|
|
" lora2=philschmid/code-llama-3-1-8b-text-to-sql-lora\n",
|
|
|
|
|
" --log-level warning\n",
|
|
|
|
|
" \"\"\"\n",
|
|
|
|
|
")\n",
|
|
|
|
|
" \"\"\")\n",
|
|
|
|
|
"\n",
|
|
|
|
|
"\n",
|
|
|
|
|
"url = f\"http://127.0.0.1:{port}\"\n",
|
|
|
|
@@ -548,16 +542,14 @@
|
|
|
|
|
"metadata": {},
|
|
|
|
|
"outputs": [],
|
|
|
|
|
"source": [
|
|
|
|
|
"server_process, port = launch_server_cmd(\n",
|
|
|
|
|
" \"\"\"\n",
|
|
|
|
|
"server_process, port = launch_server_cmd(\"\"\"\n",
|
|
|
|
|
" python3 -m sglang.launch_server \\\n",
|
|
|
|
|
" --model-path meta-llama/Meta-Llama-3.1-8B-Instruct \\\n",
|
|
|
|
|
" --enable-lora \\\n",
|
|
|
|
|
" --lora-backend csgmv \\\n",
|
|
|
|
|
" --max-loras-per-batch 16 \\\n",
|
|
|
|
|
" --lora-paths lora1=path/to/lora1 lora2=path/to/lora2\n",
|
|
|
|
|
" \"\"\"\n",
|
|
|
|
|
")"
|
|
|
|
|
" \"\"\")"
|
|
|
|
|
]
|
|
|
|
|
},
|
|
|
|
|
{
|
|
|
|
@@ -594,8 +586,7 @@
|
|
|
|
|
"lora2 = \"philschmid/code-llama-3-1-8b-text-to-sql-lora\"\n",
|
|
|
|
|
"\n",
|
|
|
|
|
"\n",
|
|
|
|
|
"server_process, port = launch_server_cmd(\n",
|
|
|
|
|
" \"\"\"\n",
|
|
|
|
|
"server_process, port = launch_server_cmd(\"\"\"\n",
|
|
|
|
|
" python3 -m sglang.launch_server \\\n",
|
|
|
|
|
" --model-path meta-llama/Meta-Llama-3.1-8B-Instruct \\\n",
|
|
|
|
|
" --enable-lora \\\n",
|
|
|
|
@@ -606,8 +597,7 @@
|
|
|
|
|
" --max-lora-rank 256 \\\n",
|
|
|
|
|
" --max-loras-per-batch 2 \\\n",
|
|
|
|
|
" --max-loaded-loras 4\n",
|
|
|
|
|
" \"\"\"\n",
|
|
|
|
|
")\n",
|
|
|
|
|
" \"\"\")\n",
|
|
|
|
|
"\n",
|
|
|
|
|
"url = f\"http://127.0.0.1:{port}\"\n",
|
|
|
|
|
"wait_for_server(url, process=server_process)"
|
|
|
|
|