INNER CODE UNIT · Python
build_parser
invergent-ai/surogate · surogate/serve/tools/reference/qwen3_5/cli.py:56
def build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--weights", required=True, help="a .sinfer artifact of this architecture, at any size")
prompt = parser.add_mutually_exclusive_group(required=True)
prompt.add_argument("--prompt", help="single user message rendered by the artifact template")
prompt.add_argument("--ids", help="comma/space-separated prompt token IDs")
prompt.add_argument("--messages", help="Qwen messages JSON with optional image/video content")
parser.add_argument(
"--thinking",
action=argparse.BooleanOptionalAction,
default=True,
help="enable or disable the Qwen thinking generation prompt",
)
parser.add_argument("--decode", type=int, default=512, help="maximum generated tokens")
parser.add_argument("--device", default="cuda")
parser.add_argument("--gpu-memory", default="auto")
parser.add_argument("--headroom", default="2GiB")
# Left unset so the artifact's own schedule chunk applies.