prompt streaming working
This commit is contained in:
@@ -16,6 +16,7 @@ class bb(enum.StrEnum):
|
||||
content="bold blue"
|
||||
model="bold red"
|
||||
prompt="bold green"
|
||||
line="italic"
|
||||
|
||||
def write_rich(*msg, end='\n', stream=sys.stdout):
|
||||
output[stream].print(*msg, end=end)
|
||||
@@ -40,8 +41,8 @@ def parse_args():
|
||||
"qwen3.5:4b",
|
||||
])
|
||||
p.add_argument("--prompt")
|
||||
p.add_argument("--system-prompt",
|
||||
default="")
|
||||
p.add_argument("--system",
|
||||
default="You are a helpful assistant. Write concisely.")
|
||||
|
||||
mxg = p.add_mutually_exclusive_group()
|
||||
mxg.add_argument("--pull")
|
||||
@@ -77,17 +78,21 @@ def parse_args():
|
||||
return p.parse_args()
|
||||
|
||||
|
||||
def s3f(ns):
|
||||
return f"{(ns // 1_000_000) / 1000:.3f}"
|
||||
|
||||
|
||||
async def do_async_turn(conf):
|
||||
"""Do user->assistant turn streaming responses"""
|
||||
if not conf.prompt:
|
||||
prompt = Prompt.ask(f"[{bb.model}]Ask {conf.model}»[/] [{bb.prompt}]")
|
||||
prompt = Prompt.ask(f"[{str(bb.model)}]Ask {conf.model}»[/] [{str(bb.prompt)}]")
|
||||
if conf.response:
|
||||
system_and_user = {
|
||||
"input": conf.prompt or prompt,
|
||||
"instructions": conf.system_prompt}
|
||||
"instructions": conf.system}
|
||||
else:
|
||||
system_and_user = {"messages": [
|
||||
{"role":"system", "content": conf.system_prompt},
|
||||
{"role":"system", "content": conf.system},
|
||||
{"role": "user", "content": conf.prompt or prompt}]}
|
||||
|
||||
client = httpx.AsyncClient(
|
||||
@@ -114,23 +119,32 @@ async def do_async_turn(conf):
|
||||
continue
|
||||
data = json.loads(line)
|
||||
if data.get("done", False):
|
||||
write(data)
|
||||
write(f"\n[Load: {s3f(data['load_duration'])}s | "
|
||||
f"Analyze: {s3f(data['prompt_eval_duration'])}s | "
|
||||
f"Generate: {s3f(data['eval_duration'])}s] "
|
||||
f"Total: {s3f(data['total_duration'])}s"
|
||||
f"\nToken usage: "
|
||||
f"{data['prompt_eval_count']} input, "
|
||||
f"{data['prompt_eval_cached_count']} cached, "
|
||||
f"{data['eval_count']} output.\n")
|
||||
msg['done'] = data
|
||||
return msg
|
||||
#write(f"[i]{data}[/i]")
|
||||
for i in ["thinking","content"]:
|
||||
text = data.get("message", {}).get(i, "")
|
||||
msg[i][msg[i]['len']].append(text)
|
||||
write(f"[i]{text}[/i]", end='')
|
||||
if '\\n' in msg[i][msg[i]['len']]:
|
||||
write(f"[{bb.line}]{msg[i]['len']:>3d}:[/] "
|
||||
f"[{bb(i)}]{Markdown(msg[i][msg[i]['len']])}[/]")
|
||||
write(f"[{str(bb[i])}]{text}[/]", end='')
|
||||
if '\n' in text:
|
||||
#write(f"[{str(bb.line)}]{msg[i]['len']:>3d}:[/] "
|
||||
# f"[{str(bb[i])}]")
|
||||
#write(Markdown("".join(msg[i][msg[i]['len']])))
|
||||
msg[i]['len'] += 1
|
||||
msg[i][msg[i]['len']] = []
|
||||
|
||||
def main():
|
||||
conf = parse_args()
|
||||
if conf.prompt:
|
||||
write(f"[{bb.model}]Ask {conf.model}»[/] [{bb.prompt}]{conf.prompt}[/]")
|
||||
write(f"[{str(bb.model)}]Ask {conf.model}»[/] [{str(bb.prompt)}]{conf.prompt}[/]")
|
||||
try:
|
||||
while True:
|
||||
if conf.stream:
|
||||
@@ -138,8 +152,8 @@ def main():
|
||||
else:
|
||||
data = res.json()
|
||||
if "thinking" in data.get("message",{}):
|
||||
write(f"[{bb.model}]{conf.model}'s thinking:[/] [{bb.thinking}] {Markdown(data['message']['thinking'])}[/]")
|
||||
write(f"[{bb.model}]{conf.model}'s response: [/][{bb.content}]{Markdown(data['message']['content'])}[/]")
|
||||
write(f"[{str(bb.model)}]{conf.model}'s thinking:[/] [{str(bb.thinking)}] {Markdown(data['message']['thinking'])}[/]")
|
||||
write(f"[{str(bb.model)}]{conf.model}'s response: [/][{str(bb.content)}]{Markdown(data['message']['content'])}[/]")
|
||||
#response = response['choices'][0]['message']
|
||||
if conf.prompt:
|
||||
sys.exit(0)
|
||||
|
||||
Reference in New Issue
Block a user