@@ -154,20 +154,23 @@ no_dp = decode.get('no_dp', {})
154154# Decode DP config
155155print(f'DECODE_MAX_RUNNING_REQUESTS_DP=\" {dp.get(\" max_running_requests\" , 4096)}\" ')
156156print(f'DECODE_CHUNKED_PREFILL_SIZE_DP=\" {eval_formula(dp.get(\" chunked_prefill_size\" , 262144))}\" ')
157+ print(f'DECODE_CONTEXT_LENGTH_DP=\" {dp.get(\" context_length\" , \"\" )}\" ')
157158s, e = parse_range(dp.get('cuda_graph_bs_range', '1-160'), 1, 160)
158159print(f'DECODE_CUDA_GRAPH_BS_DP_START=\" {s}\" ')
159160print(f'DECODE_CUDA_GRAPH_BS_DP_END=\" {e}\" ')
160161
161162# Decode EP-only config (EP enabled but DP disabled)
162163print(f'DECODE_MAX_RUNNING_REQUESTS_EP_ONLY=\" {ep_only.get(\" max_running_requests\" , 256)}\" ')
163164print(f'DECODE_CHUNKED_PREFILL_SIZE_EP_ONLY=\" {eval_formula(ep_only.get(\" chunked_prefill_size\" , 262144))}\" ')
165+ print(f'DECODE_CONTEXT_LENGTH_EP_ONLY=\" {ep_only.get(\" context_length\" , \"\" )}\" ')
164166s, e = parse_range(ep_only.get('cuda_graph_bs_range', '1-256'), 1, 256)
165167print(f'DECODE_CUDA_GRAPH_BS_EP_ONLY_START=\" {s}\" ')
166168print(f'DECODE_CUDA_GRAPH_BS_EP_ONLY_END=\" {e}\" ')
167169
168170# Decode no-DP config
169171print(f'DECODE_MAX_RUNNING_REQUESTS_NO_DP=\" {no_dp.get(\" max_running_requests\" , 128)}\" ')
170172print(f'DECODE_CHUNKED_PREFILL_SIZE_NO_DP=\" {eval_formula(no_dp.get(\" chunked_prefill_size\" , 262144))}\" ')
173+ print(f'DECODE_CONTEXT_LENGTH_NO_DP=\" {no_dp.get(\" context_length\" , \"\" )}\" ')
171174s, e = parse_range(no_dp.get('cuda_graph_bs_range', '1-128'), 1, 128)
172175print(f'DECODE_CUDA_GRAPH_BS_NO_DP_START=\" {s}\" ')
173176print(f'DECODE_CUDA_GRAPH_BS_NO_DP_END=\" {e}\" ')
205208if [[ " $DECODE_ENABLE_DP " == " true" ]]; then
206209 decode_cuda_graph_bs=($( seq $DECODE_CUDA_GRAPH_BS_DP_START $DECODE_CUDA_GRAPH_BS_DP_END ) )
207210 decode_max_running_requests=$(( DECODE_CUDA_GRAPH_BS_DP_END * DECODE_TP_SIZE))
211+ decode_context_length=$DECODE_CONTEXT_LENGTH_DP
208212elif [[ " $DECODE_ENABLE_EP " == " true" ]]; then
209213 decode_cuda_graph_bs=($( seq $DECODE_CUDA_GRAPH_BS_EP_ONLY_START $DECODE_CUDA_GRAPH_BS_EP_ONLY_END ) )
210214 decode_max_running_requests=$DECODE_MAX_RUNNING_REQUESTS_EP_ONLY
215+ decode_context_length=$DECODE_CONTEXT_LENGTH_EP_ONLY
211216else
212217 decode_cuda_graph_bs=($( seq $DECODE_CUDA_GRAPH_BS_NO_DP_START $DECODE_CUDA_GRAPH_BS_NO_DP_END ) )
213218 decode_max_running_requests=$DECODE_MAX_RUNNING_REQUESTS_NO_DP
219+ decode_context_length=$DECODE_CONTEXT_LENGTH_NO_DP
220+ fi
221+ # In PD-disaggregation the decode must admit requests against the SAME context
222+ # length as prefill; otherwise decode accepts over-length requests that prefill
223+ # rejects, and those requests hang forever waiting for a KV transfer that never
224+ # comes (the 8k1k conc-500 straggler). Fall back to the prefill value if the
225+ # decode context_length is not set in the model config, so the two always agree.
226+ if [[ -z " $decode_context_length " ]]; then
227+ decode_context_length=$prefill_context_length
214228fi
215229
216230# When both DP and EP are enabled, override max-running-requests and dispatch tokens
@@ -251,6 +265,9 @@ DECODE_MODE_FLAGS="--mem-fraction-static ${DECODE_MEM_FRACTION_STATIC} --max-run
251265if [[ " $DECODE_PREFILL_ROUND_ROBIN_BALANCE " == " True" ]] || [[ " $DECODE_PREFILL_ROUND_ROBIN_BALANCE " == " true" ]]; then
252266 DECODE_MODE_FLAGS=" $DECODE_MODE_FLAGS --prefill-round-robin-balance"
253267fi
268+ if [[ -n " $decode_context_length " ]]; then
269+ DECODE_MODE_FLAGS=" $DECODE_MODE_FLAGS --context-length ${decode_context_length} "
270+ fi
254271
255272if [[ " $DECODE_MTP_SIZE " -gt 0 ]]; then
256273 MORI_MAX_DISPATCH_TOKENS_DECODE=$(( MORI_MAX_DISPATCH_TOKENS_DECODE * (DECODE_MTP_SIZE + 1 )) )
@@ -506,12 +523,23 @@ if [ "$NODE_RANK" -eq 0 ]; then
506523 fi
507524 echo " Congratulations!!! All prefill and decode servers are up . . ."
508525
526+ # Circuit-breaker recovery tuning (run 28696443568):
527+ # With defaults the per-worker circuit stays OPEN for cb-timeout-duration-secs=60
528+ # before a half-open retrial. In that run every request 503'd from request #1
529+ # ("all circuits open or unhealthy") for ~31s and lm_eval (max_retries=5) then gave
530+ # up -- i.e. the circuit was still open when the client budget ran out, so 0 result
531+ # files were produced. Shortening the open->half-open window (and letting the router
532+ # itself retry a failed worker selection) lets a transient trip re-close INSIDE the
533+ # client retry budget instead of nuking the whole eval. The breaker stays fully
534+ # ENABLED (thresholds unchanged); this only speeds recovery.
535+ ROUTER_CB_ARGS=" ${ROUTER_CB_ARGS:- --cb-timeout-duration-secs 15 --retry-max-retries 3} "
509536 ROUTER_CMD=" python -m sglang_router.launch_router \
510537 --pd-disaggregation \
511538 --port 30000 \
512539 --policy random \
513540 --prefill-policy random \
514541 --decode-policy random \
542+ ${ROUTER_CB_ARGS} \
515543 ${PREFILL_ARGS} \
516544 ${DECODE_ARGS} "
517545
@@ -557,6 +585,39 @@ if [ "$NODE_RANK" -eq 0 ]; then
557585 wait_or_die " $prefill0_pid " bash -c " $HEALTH_BARRIER_CMD " || exit 1
558586 fi
559587
588+ # ---- End-to-end router readiness canary (run 28696443568) ----
589+ # The /readiness barrier above only proves the router PROCESS is up; it does
590+ # NOT prove the router can reach a prefill worker and complete a generation.
591+ # In that run the eval fired the instant /readiness passed and EVERY request
592+ # 503'd ("No available prefill workers (all circuits open or unhealthy)") from
593+ # request #1 -> lm_eval gave up -> 0 result files -> "Verify eval scores" failed.
594+ # Gate the benchmark on ONE successful generation THROUGH the router so the eval
595+ # never starts against a router whose prefill path is not yet actually serving.
596+ if [[ " ${ROUTER_READINESS_CANARY:- 1} " == " 1" ]]; then
597+ CANARY_URL=" http://${NODE0_ADDR} :30000/v1/chat/completions"
598+ CANARY_MODEL=" ${MODEL_DIR} /${MODEL_NAME} "
599+ canary_ok=0
600+ canary_deadline=$(( $(date +% s) + ${ROUTER_CANARY_TIMEOUT:- 600} ))
601+ while [ " $( date +%s) " -lt " $canary_deadline " ]; do
602+ canary_code=$( curl -s -o /tmp/router_canary.out -w ' %{http_code}' \
603+ -m " ${ROUTER_CANARY_REQ_TIMEOUT:- 120} " \
604+ -X POST " $CANARY_URL " -H ' Content-Type: application/json' \
605+ -d " {\" model\" :\" ${CANARY_MODEL} \" ,\" messages\" :[{\" role\" :\" user\" ,\" content\" :\" ping\" }],\" max_tokens\" :1,\" temperature\" :0}" 2> /dev/null)
606+ if [ " $canary_code " = " 200" ] && \
607+ ! grep -qE " circuits open|server_selection_failed|No available" /tmp/router_canary.out 2> /dev/null; then
608+ canary_ok=1; break
609+ fi
610+ echo " Router readiness canary not ready yet (http=${canary_code} ); retrying in 5s . . ."
611+ sleep 5
612+ done
613+ if [ " $canary_ok " -ne 1 ]; then
614+ echo " ERROR: router readiness canary failed after ${ROUTER_CANARY_TIMEOUT:- 600} s -- the router cannot complete a generation through a prefill worker (all circuits open/unhealthy). Refusing to start the eval against a non-serving router."
615+ head -c 800 /tmp/router_canary.out 2> /dev/null
616+ exit 1
617+ fi
618+ echo " Router readiness canary passed (end-to-end generation OK)"
619+ fi
620+
560621 echo " Router is ready for benchmarking"
561622 fi
562623
0 commit comments