INNER CODE UNIT · Python
build_parser
invergent-ai/surogate · surogate/serve/tools/reference/qwen3_5_moe/cli.py:57
def build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--weights", required=True, help="hybrid MoE .sinfer artifact")
prompt = parser.add_mutually_exclusive_group(required=True)
prompt.add_argument("--prompt", help="single user message rendered by the artifact template")
prompt.add_argument("--ids", help="comma/space-separated prompt token IDs")
prompt.add_argument("--messages", help="Qwen messages JSON with optional image/video content")
parser.add_argument(
"--thinking",
action=argparse.BooleanOptionalAction,
default=True,
help="enable or disable the Qwen thinking generation prompt",
)
parser.add_argument("--decode", type=int, default=512, help="maximum generated tokens")
parser.add_argument("--device", default="cuda")
parser.add_argument("--gpu-memory", default="auto")
parser.add_argument("--headroom", default="2GiB")
parser.add_argument("--prefill-chunk", type=int, default=CFG.prefill_chunk)