[Docs] Sync docs_new with legacy docs and update migration redirects (#23337)
Co-authored-by: Mingyi <wisclmy0611@gmail.com>
This commit is contained in:
@@ -43,7 +43,7 @@ From client side, the user needs to provide a list of strings as input batch, an
|
||||
|
||||
**Note:** SGLang supports LoRA adapters through two APIs:
|
||||
|
||||
1. **OpenAI-Compatible API** (`/v1/chat/completions`, `/v1/completions`): Use the `model:adapter-name` syntax. See [OpenAI API with LoRA](../basic_usage/openai_api_completions.ipynb#Using-LoRA-Adapters) for examples.
|
||||
1. **OpenAI-Compatible API** (`/v1/chat/completions`, `/v1/completions`): Use the `model:adapter-name` syntax. See [OpenAI API with LoRA](../basic_usage/openai_api_completions#using-lora-adapters) for examples.
|
||||
|
||||
2. **Native API** (`/generate`): Pass `lora_path` in the request body (shown below).
|
||||
|
||||
@@ -108,7 +108,7 @@ server_process, port = launch_server_cmd(
|
||||
python3 -m sglang.launch_server --model-path meta-llama/Meta-Llama-3.1-8B-Instruct \
|
||||
--enable-lora \
|
||||
--lora-paths lora0=algoprog/fact-generation-llama-3.1-8b-instruct-lora \
|
||||
lora1=Nutanix/Meta-Llama-3.1-8B-Instruct_lora_4_alpha_16 \
|
||||
lora1=Nutanix/Meta-Llama-3.1-8B-Instruct_SFT_lora_4_alpha_16_humaneval_raw_json \
|
||||
--max-loras-per-batch 2 \
|
||||
--log-level warning \
|
||||
"""
|
||||
@@ -152,7 +152,7 @@ When using dynamic LoRA loading, it's recommended to explicitly specify both `--
|
||||
|
||||
|
||||
```python Example
|
||||
lora0 = "Nutanix/Meta-Llama-3.1-8B-Instruct_lora_4_alpha_16" # rank - 4, target modules - q_proj, k_proj, v_proj, o_proj, gate_proj
|
||||
lora0 = "Nutanix/Meta-Llama-3.1-8B-Instruct_SFT_lora_4_alpha_16_humaneval_raw_json" # rank - 4, target modules - q_proj, k_proj, v_proj, o_proj, gate_proj
|
||||
lora1 = "algoprog/fact-generation-llama-3.1-8b-instruct-lora" # rank - 64, target modules - q_proj, k_proj, v_proj, o_proj, gate_proj, up_proj, down_proj
|
||||
lora0_new = "philschmid/code-llama-3-1-8b-text-to-sql-lora" # rank - 256, target modules - q_proj, k_proj, v_proj, o_proj, gate_proj, up_proj, down_proj
|
||||
|
||||
@@ -317,7 +317,7 @@ server_process, port = launch_server_cmd(
|
||||
--max-lora-rank 256 \
|
||||
--lora-target-modules all \
|
||||
--lora-paths \
|
||||
{"lora_name":"lora0","lora_path":"Nutanix/Meta-Llama-3.1-8B-Instruct_lora_4_alpha_16","pinned":true} \
|
||||
{"lora_name":"lora0","lora_path":"Nutanix/Meta-Llama-3.1-8B-Instruct_SFT_lora_4_alpha_16_humaneval_raw_json","pinned":true} \
|
||||
{"lora_name":"lora1","lora_path":"algoprog/fact-generation-llama-3.1-8b-instruct-lora"} \
|
||||
lora2=philschmid/code-llama-3-1-8b-text-to-sql-lora
|
||||
--log-level warning
|
||||
@@ -418,7 +418,7 @@ By using the `--enable-lora-overlap-loading` server argument, the SGLang engine
|
||||
|
||||
|
||||
```python Example
|
||||
lora0 = "Nutanix/Meta-Llama-3.1-8B-Instruct_lora_4_alpha_16"
|
||||
lora0 = "Nutanix/Meta-Llama-3.1-8B-Instruct_SFT_lora_4_alpha_16_humaneval_raw_json"
|
||||
lora1 = "algoprog/fact-generation-llama-3.1-8b-instruct-lora"
|
||||
lora2 = "philschmid/code-llama-3-1-8b-text-to-sql-lora"
|
||||
|
||||
@@ -429,7 +429,7 @@ server_process, port = launch_server_cmd(
|
||||
--model-path meta-llama/Meta-Llama-3.1-8B-Instruct \
|
||||
--enable-lora \
|
||||
--enable-lora-overlap-loading \
|
||||
--lora-paths lora0=Nutanix/Meta-Llama-3.1-8B-Instruct_lora_4_alpha_16 \
|
||||
--lora-paths lora0=Nutanix/Meta-Llama-3.1-8B-Instruct_SFT_lora_4_alpha_16_humaneval_raw_json \
|
||||
lora1=algoprog/fact-generation-llama-3.1-8b-instruct-lora \
|
||||
lora2=philschmid/code-llama-3-1-8b-text-to-sql-lora \
|
||||
--max-lora-rank 256 \
|
||||
|
||||
Reference in New Issue
Block a user