vl/serve.sh
1.6 KB · 20 lines · bash Raw
1 #!/usr/bin/env bash
2 # JEV-9B with vision: the multimodal Qwen/Qwen3.5-9B (unchanged, including its vision encoder) + JEV-9B's System 1
3 # adapter and decision head (vl/adapter_vllm: the same weights as adapter_vllm/, with the layer names moved under
4 # model.language_model). JEV-9B's language weights are bit-identical to Qwen3.5-9B's, so text decisions are unchanged.
5 # System 1: POST /v1/decide {kind, state, question, options} -> calibrated probabilities (text and images)
6 # or model "jev-decision" through /v1/completions and /v1/chat/completions
7 # System 2: model "autotrust/JEV-9B" through /v1/chat/completions (text and images)
8 # --max-num-seqs 8 is required: with more than 8 sequences in one batch, vLLM's LoRA path for this multimodal model class
9 # returns wrong System 1 probabilities. --trust-request-chat-template lets image decisions use the raw decision template.
10 set -e
11 JEV_DIR=${JEV_DIR:-JEV-9B}
12 BASE_DIR=${BASE_DIR:-Qwen3.5-9B}
13 [ -f "$JEV_DIR/vl/serve_decide.py" ] || hf download autotrust/JEV-9B --include "vl/*" --local-dir "$JEV_DIR"
14 [ -f "$BASE_DIR/config.json" ] || hf download Qwen/Qwen3.5-9B --revision c202236235762e1c871ad0ccb60c8ee5ba337b9a --local-dir "$BASE_DIR"
15 exec python3 "$JEV_DIR/vl/serve_decide.py" --model "$BASE_DIR" --served-model-name autotrust/JEV-9B \
16 --enable-lora --max-lora-rank 32 --lora-modules jev-decision="$JEV_DIR/vl/adapter_vllm" \
17 --logprobs-mode processed_logprobs --max-model-len ${MAX_MODEL_LEN:-32768} --enable-prefix-caching --mamba-cache-mode align \
18 --limit-mm-per-prompt '{"image": 8}' --max-num-seqs 8 --max-logprobs 256 --trust-request-chat-template \
19 --port ${PORT:-8000} ${EXTRA_ARGS}
20