Dolphin/deployment/tensorrt_llm/start_dolphin_server.sh
2025-06-30 19:41:03 +08:00

10 lines
292 B
Bash

#!/usr/bin/env bash
set -ex
export MODEL_NAME="Dolphin"
python api_server.py \
--hf_model_dir tmp/hf_models/${MODEL_NAME} \
--visual_engine_dir tmp/trt_engines/${MODEL_NAME}/vision_encoder \
--llm_engine_dir tmp/trt_engines/${MODEL_NAME}/1-gpu/bfloat16 \
--max_batch_size 16