response api

This commit is contained in:
2026-09-28 00:40:44 -03:00
parent a8992b446d
commit 778845bf11
+34 -16
View File
@@ -32,6 +32,7 @@ write = write_rich
def parse_args(): def parse_args():
p = argparse.ArgumentParser() p = argparse.ArgumentParser()
p.add_argument("--debug", action="store_true")
p.add_argument("--model", p.add_argument("--model",
default="qwen3.5:4b", default="qwen3.5:4b",
choices=[ choices=[
@@ -46,19 +47,13 @@ def parse_args():
mxg = p.add_mutually_exclusive_group() mxg = p.add_mutually_exclusive_group()
mxg.add_argument("--pull") mxg.add_argument("--pull")
#mxg.add_argument("--chat", action=argparse.BooleanOptionalAction, default=True)
mxg.add_argument("--response", action="store_true") mxg.add_argument("--response", action="store_true")
mxg.add_argument("--instruct", action="store_true") mxg.add_argument("--endpoint")
mxg.add_argument("--endpoint",
default="/api/chat",
choices=[
"/api/chat",
"/v1/chat/completions",])
mxg.add_argument("--api", mxg.add_argument("--api",
default="ollama", default="ollama",
choices=["ollama","openai",]) choices=["ollama","openai",])
g = p.add_argument_group("Model Options") g = p.add_argument_group("Model Options")
g.add_argument("--reasoning", action="store_true") g.add_argument("--thinking", action=argparse.BooleanOptionalAction, default=True)
g.add_argument("--stream", action="store_true") g.add_argument("--stream", action="store_true")
g.add_argument("--num_ctx", default=128000, type=int) g.add_argument("--num_ctx", default=128000, type=int)
g.add_argument("--top_k", default=40, type=int) g.add_argument("--top_k", default=40, type=int)
@@ -87,10 +82,10 @@ async def do_async_turn(conf):
"""Do user->assistant turn streaming responses""" """Do user->assistant turn streaming responses"""
if not conf.prompt: if not conf.prompt:
prompt = Prompt.ask(f"[{str(bb.model)}]Ask {conf.model}»[/] [{str(bb.prompt)}]") prompt = Prompt.ask(f"[{str(bb.model)}]Ask {conf.model}»[/] [{str(bb.prompt)}]")
if conf.response or conf.instruct: if conf.response:
system_and_user = { system_and_user = {
"input": conf.prompt or prompt, "input" if conf.api == "openai" else "prompt": conf.prompt or prompt,
"instructions": conf.system} "instructions" if conf.api == "openai" else "system": conf.system}
else: else:
system_and_user = {"messages": [ system_and_user = {"messages": [
{"role":"system", "content": conf.system}, {"role":"system", "content": conf.system},
@@ -101,9 +96,14 @@ async def do_async_turn(conf):
headers={ headers={
"Authorization": f"Bearer {conf.apikey}",}, "Authorization": f"Bearer {conf.apikey}",},
timeout=float(conf.timeout)) timeout=float(conf.timeout))
async with client.stream('POST',"/api/chat", json=system_and_user | { async with client.stream('POST',((
"/api/generate" if conf.response else "/api/chat")
if conf.api == "ollama" else ("/v1/responses"
if conf.response else "/v1/chat/completions")),
json=system_and_user | {
"model": conf.model, "model": conf.model,
"reasoning": {"enabled": conf.reasoning}, "reasoning": {"enabled": conf.thinking},
"thinking": conf.thinking,
"stream": conf.stream, "stream": conf.stream,
"options": { "options": {
"num_ctx": conf.num_ctx, "num_ctx": conf.num_ctx,
@@ -111,6 +111,13 @@ async def do_async_turn(conf):
"top_p": conf.top_p, "top_p": conf.top_p,
"temperature": conf.temperature, "temperature": conf.temperature,
}}) as res: }}) as res:
res.raise_for_status()
if conf.debug:
write(res.request, end=' (')
write(", ".join([d for d in dir(res.request) if not d.startswith("_")]), end=")\n")
write(res.request.content)
write(res, end=' (')
write(", ".join([d for d in dir(res) if not d.startswith("_")]), end=')\n')
msg = {"thinking": {"len": 1, 1: []}, msg = {"thinking": {"len": 1, 1: []},
"content": {"len": 1, 1: []}} "content": {"len": 1, 1: []}}
async for line in res.aiter_lines(): async for line in res.aiter_lines():
@@ -119,6 +126,8 @@ async def do_async_turn(conf):
# TODO: use spinner from rich # TODO: use spinner from rich
continue continue
data = json.loads(line) data = json.loads(line)
if conf.debug:
write(line)
if data.get("done", False): if data.get("done", False):
write(f"\n[Load: {s3f(data['load_duration'])}s | " write(f"\n[Load: {s3f(data['load_duration'])}s | "
f"Analyze: {s3f(data['prompt_eval_duration'])}s | " f"Analyze: {s3f(data['prompt_eval_duration'])}s | "
@@ -131,7 +140,7 @@ async def do_async_turn(conf):
msg['done'] = data msg['done'] = data
return msg return msg
#write(f"[i]{data}[/i]") #write(f"[i]{data}[/i]")
for i in ["thinking","content"]: for i in ["thinking", "content"]:
text = data.get("message", {}).get(i, "") text = data.get("message", {}).get(i, "")
msg[i][msg[i]['len']].append(text) msg[i][msg[i]['len']].append(text)
write(f"[{str(bb[i])}]{text}[/]", end='') write(f"[{str(bb[i])}]{text}[/]", end='')
@@ -141,13 +150,21 @@ async def do_async_turn(conf):
#write(Markdown("".join(msg[i][msg[i]['len']]))) #write(Markdown("".join(msg[i][msg[i]['len']])))
msg[i]['len'] += 1 msg[i]['len'] += 1
msg[i][msg[i]['len']] = [] msg[i][msg[i]['len']] = []
write(data.get('choices',[{}])[0].get('message', ""), end='')
write(data.get('output', ""), end='')
write(f"[{str(bb.model)}]{data.get('response', '')}", end='')
write(f"[{str(bb.thinking)}]{data.get('thinking', '')}[/]", end='')
def main(): def main():
conf = parse_args() conf = parse_args()
if conf.prompt == "-": if conf.prompt == "-":
conf.prompt = sys.stdin.read() conf.prompt = sys.stdin.read()
if conf.prompt: if conf.prompt:
write(f"[{str(bb.model)}]Ask {conf.model}»[/] [{str(bb.prompt)}]{conf.prompt}[/]") write(f"[{str(bb.model)}]Ask {conf.model}»[/] ", end='')
if conf.response:
write(f"[{str(bb.prompt)}]{conf.system}[/]")
else:
write(f"[{str(bb.prompt)}]{conf.prompt}[/]")
try: try:
while True: while True:
if conf.stream: if conf.stream:
@@ -157,7 +174,8 @@ def main():
if "thinking" in data.get("message",{}): if "thinking" in data.get("message",{}):
write(f"[{str(bb.model)}]{conf.model}'s thinking:[/] [{str(bb.thinking)}] {Markdown(data['message']['thinking'])}[/]") write(f"[{str(bb.model)}]{conf.model}'s thinking:[/] [{str(bb.thinking)}] {Markdown(data['message']['thinking'])}[/]")
write(f"[{str(bb.model)}]{conf.model}'s response: [/][{str(bb.content)}]{Markdown(data['message']['content'])}[/]") write(f"[{str(bb.model)}]{conf.model}'s response: [/][{str(bb.content)}]{Markdown(data['message']['content'])}[/]")
#response = response['choices'][0]['message'] write(data['choices'][0]['message'])
write(data['output'])
if conf.prompt: if conf.prompt:
sys.exit(0) sys.exit(0)
# TODO: add turn to message list and do another turn # TODO: add turn to message list and do another turn