Upload entrypoint.sh with huggingface_hub
Browse files- entrypoint.sh +1 -1
entrypoint.sh
CHANGED
|
@@ -10,7 +10,7 @@ exec python3 -m vllm.entrypoints.openai.api_server \
|
|
| 10 |
--port 7860 \
|
| 11 |
--model google/gemma-4-E2B-it \
|
| 12 |
--dtype bfloat16 \
|
| 13 |
-
--max-model-len
|
| 14 |
--gpu-memory-utilization 0.92 \
|
| 15 |
--served-model-name google/gemma-4-E2B-it \
|
| 16 |
--max-num-seqs 8 \
|
|
|
|
| 10 |
--port 7860 \
|
| 11 |
--model google/gemma-4-E2B-it \
|
| 12 |
--dtype bfloat16 \
|
| 13 |
+
--max-model-len 65536 \
|
| 14 |
--gpu-memory-utilization 0.92 \
|
| 15 |
--served-model-name google/gemma-4-E2B-it \
|
| 16 |
--max-num-seqs 8 \
|