prompt streaming working

This commit is contained in:
2026-09-27 23:08:45 -03:00
parent d01b07e5bf
commit f29c4c9435
+27 -13
View File
@@ -16,6 +16,7 @@ class bb(enum.StrEnum):
content="bold blue"
model="bold red"
prompt="bold green"
line="italic"
def write_rich(*msg, end='\n', stream=sys.stdout):
output[stream].print(*msg, end=end)
@@ -40,8 +41,8 @@ def parse_args():
"qwen3.5:4b",
])
p.add_argument("--prompt")
p.add_argument("--system-prompt",
default="")
p.add_argument("--system",
default="You are a helpful assistant. Write concisely.")
mxg = p.add_mutually_exclusive_group()
mxg.add_argument("--pull")
@@ -77,17 +78,21 @@ def parse_args():
return p.parse_args()
def s3f(ns):
return f"{(ns // 1_000_000) / 1000:.3f}"
async def do_async_turn(conf):
"""Do user->assistant turn streaming responses"""
if not conf.prompt:
prompt = Prompt.ask(f"[{bb.model}]Ask {conf.model}»[/] [{bb.prompt}]")
prompt = Prompt.ask(f"[{str(bb.model)}]Ask {conf.model}»[/] [{str(bb.prompt)}]")
if conf.response:
system_and_user = {
"input": conf.prompt or prompt,
"instructions": conf.system_prompt}
"instructions": conf.system}
else:
system_and_user = {"messages": [
{"role":"system", "content": conf.system_prompt},
{"role":"system", "content": conf.system},
{"role": "user", "content": conf.prompt or prompt}]}
client = httpx.AsyncClient(
@@ -114,23 +119,32 @@ async def do_async_turn(conf):
continue
data = json.loads(line)
if data.get("done", False):
write(data)
write(f"\n[Load: {s3f(data['load_duration'])}s | "
f"Analyze: {s3f(data['prompt_eval_duration'])}s | "
f"Generate: {s3f(data['eval_duration'])}s] "
f"Total: {s3f(data['total_duration'])}s"
f"\nToken usage: "
f"{data['prompt_eval_count']} input, "
f"{data['prompt_eval_cached_count']} cached, "
f"{data['eval_count']} output.\n")
msg['done'] = data
return msg
#write(f"[i]{data}[/i]")
for i in ["thinking","content"]:
text = data.get("message", {}).get(i, "")
msg[i][msg[i]['len']].append(text)
write(f"[i]{text}[/i]", end='')
if '\\n' in msg[i][msg[i]['len']]:
write(f"[{bb.line}]{msg[i]['len']:>3d}:[/] "
f"[{bb(i)}]{Markdown(msg[i][msg[i]['len']])}[/]")
write(f"[{str(bb[i])}]{text}[/]", end='')
if '\n' in text:
#write(f"[{str(bb.line)}]{msg[i]['len']:>3d}:[/] "
# f"[{str(bb[i])}]")
#write(Markdown("".join(msg[i][msg[i]['len']])))
msg[i]['len'] += 1
msg[i][msg[i]['len']] = []
def main():
conf = parse_args()
if conf.prompt:
write(f"[{bb.model}]Ask {conf.model}»[/] [{bb.prompt}]{conf.prompt}[/]")
write(f"[{str(bb.model)}]Ask {conf.model}»[/] [{str(bb.prompt)}]{conf.prompt}[/]")
try:
while True:
if conf.stream:
@@ -138,8 +152,8 @@ def main():
else:
data = res.json()
if "thinking" in data.get("message",{}):
write(f"[{bb.model}]{conf.model}'s thinking:[/] [{bb.thinking}] {Markdown(data['message']['thinking'])}[/]")
write(f"[{bb.model}]{conf.model}'s response: [/][{bb.content}]{Markdown(data['message']['content'])}[/]")
write(f"[{str(bb.model)}]{conf.model}'s thinking:[/] [{str(bb.thinking)}] {Markdown(data['message']['thinking'])}[/]")
write(f"[{str(bb.model)}]{conf.model}'s response: [/][{str(bb.content)}]{Markdown(data['message']['content'])}[/]")
#response = response['choices'][0]['message']
if conf.prompt:
sys.exit(0)