INNER CODE UNIT · Python
decode_tokens
invergent-ai/surogate · surogate/serve/tools/reference/qwen3_5/cli.py:218
decode_tokens = max(0, len(generated) - 1)
decode_tps = decode_tokens / timings.decode if timings.decode > 0.0 else 0.0
print(
f"TIMING: load={timings.load:.3f}s preprocess={timings.preprocess:.3f}s "
f"vision={timings.vision:.3f}s prepare={timings.prepare:.3f}s "
f"prefill={timings.prefill:.3f}s decode={timings.decode:.3f}s "
f"decode_tokens={decode_tokens} decode_tps={decode_tps:.3f}"
)
if model is not None and model.device.type == "cuda":
print(
"CUDA_PEAK: "
f"allocated={torch.cuda.max_memory_allocated(model.device) / (1 << 30):.2f}GiB "
f"reserved={torch.cuda.max_memory_reserved(model.device) / (1 << 30):.2f}GiB"
)
print("PROMPT_TOKEN_IDS:", prompt.token_ids)
print("GENERATED_TOKEN_IDS:", list(generated))
print("STOP_REASON:", stop_reason)
print("GENERATED_TEXT:")