INNER CODE UNIT · Python
bot_streaming_T2T
FoundationVision/Liquid · evaluation/app.py:279
def bot_streaming_T2T(message, history,temperature):
print(message)
global stop_flag
stop_flag = True
time.sleep(0.2)
stop_flag = False
torch.cuda.empty_cache()
qs = message
conv = conv_templates['gemma'].copy()
conv.append_message(conv.roles[0], qs)
conv.append_message(conv.roles[1], None)
prompt = conv.get_prompt()
print(prompt)
with torch.no_grad():
inputs = tokenizer([prompt], return_tensors="pt").to('cuda')
streamer = TextIteratorStreamer(tokenizer, **{"skip_special_tokens": False, "skip_prompt": True})