[Feature] Sglang Tracing: Fine-Grained Tracking for Request Latency - Part 2 (#10804)
Signed-off-by: Feng Su <sufeng@linux.alibaba.com>
This commit is contained in:
@@ -143,10 +143,13 @@ class Engine(EngineBase):
|
||||
|
||||
# Enable tracing
|
||||
if server_args.enable_trace:
|
||||
process_tracing_init(server_args.oltp_traces_endpoint, "sglang")
|
||||
if server_args.disaggregation_mode == "null":
|
||||
thread_label = "Tokenizer"
|
||||
trace_set_thread_info(thread_label)
|
||||
process_tracing_init(server_args.otlp_traces_endpoint, "sglang")
|
||||
thread_label = "Tokenizer"
|
||||
if server_args.disaggregation_mode == "prefill":
|
||||
thread_label = "Prefill Tokenizer"
|
||||
elif server_args.disaggregation_mode == "decode":
|
||||
thread_label = "Decode Tokenizer"
|
||||
trace_set_thread_info(thread_label)
|
||||
|
||||
try:
|
||||
self.loop = asyncio.get_running_loop()
|
||||
|
||||
@@ -220,9 +220,12 @@ async def lifespan(fast_api_app: FastAPI):
|
||||
|
||||
# Init tracing
|
||||
if server_args.enable_trace:
|
||||
process_tracing_init(server_args.oltp_traces_endpoint, "sglang")
|
||||
if server_args.disaggregation_mode == "null":
|
||||
trace_set_thread_info(thread_label)
|
||||
process_tracing_init(server_args.otlp_traces_endpoint, "sglang")
|
||||
if server_args.disaggregation_mode == "prefill":
|
||||
thread_label = "Prefill" + thread_label
|
||||
elif server_args.disaggregation_mode == "decode":
|
||||
thread_label = "Decode" + thread_label
|
||||
trace_set_thread_info(thread_label)
|
||||
|
||||
# Initialize OpenAI serving handlers
|
||||
fast_api_app.state.openai_serving_completion = OpenAIServingCompletion(
|
||||
|
||||
Reference in New Issue
Block a user