INNER CODE UNIT · Python

decode_tokens

invergent-ai/surogate · surogate/serve/tools/reference/qwen3_5/cli.py:218

    decode_tokens = max(0, len(generated) - 1)
    decode_tps = decode_tokens / timings.decode if timings.decode > 0.0 else 0.0
    print(
        f"TIMING: load={timings.load:.3f}s preprocess={timings.preprocess:.3f}s "
        f"vision={timings.vision:.3f}s prepare={timings.prepare:.3f}s "
        f"prefill={timings.prefill:.3f}s decode={timings.decode:.3f}s "
        f"decode_tokens={decode_tokens} decode_tps={decode_tps:.3f}"
    )
    if model is not None and model.device.type == "cuda":
        print(
            "CUDA_PEAK: "
            f"allocated={torch.cuda.max_memory_allocated(model.device) / (1 << 30):.2f}GiB "
            f"reserved={torch.cuda.max_memory_reserved(model.device) / (1 << 30):.2f}GiB"
        )
    print("PROMPT_TOKEN_IDS:", prompt.token_ids)
    print("GENERATED_TOKEN_IDS:", list(generated))
    print("STOP_REASON:", stop_reason)
    print("GENERATED_TEXT:")

View source record →

📰 Research Paper
Loading…
⏳ Fetching content…