Compare commits
50
Commits
25c0517cf7
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
778845bf11 | ||
|
|
a8992b446d | ||
|
|
77825b7196 | ||
|
|
f29c4c9435 | ||
|
|
d01b07e5bf | ||
|
|
3ea821f585 | ||
|
|
6a23a8e43c | ||
|
|
fa7e2982aa | ||
|
|
9ff875a37b | ||
|
|
3dafcc7998 | ||
|
|
56baf9ca72 | ||
|
|
a481bb68fe | ||
|
|
1918465ada | ||
|
|
77bb8efdfa | ||
|
|
9c9bc243b3 | ||
|
|
fed4aad9f9 | ||
|
|
1e74a727f8 | ||
|
|
f88738a67f | ||
|
|
91917eb8f1 | ||
|
|
6d1baba953 | ||
|
|
6721e6208c | ||
|
|
a8a35a9919 | ||
|
|
de53a082c2 | ||
|
|
e3306d0241 | ||
|
|
3a271f2629 | ||
|
|
a387e3a224 | ||
|
|
1e063a268f | ||
|
|
2f40775233 | ||
|
|
c765b115a9 | ||
|
|
a4e805fa86 | ||
|
|
8dc600edc8 | ||
|
|
a0cb5eb9ea | ||
|
|
c14c62b297 | ||
|
|
a8a371861d | ||
|
|
0bd0695211 | ||
|
|
f8e02054b6 | ||
|
|
3ab8e2b864 | ||
|
|
69c9e0f1cc | ||
|
|
3d8267c6bf | ||
|
|
6018ddb07c | ||
|
|
eeba5793fc | ||
|
|
8a7999b98d | ||
|
|
3db415ee43 | ||
|
|
72261e42bf | ||
|
|
b213a76824 | ||
|
|
81ad5ccbd9 | ||
|
|
6eaf0856e9 | ||
|
|
4c405cb4f9 | ||
|
|
626dec8ec5 | ||
|
|
80427878f7 |
+35
-16
@@ -16,23 +16,10 @@ license-files = ["LICENSE"]
|
|||||||
#requires-python = ">=3.11,<3.13"
|
#requires-python = ">=3.11,<3.13"
|
||||||
requires-python = ">=3.14"
|
requires-python = ">=3.14"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"mutagen",
|
|
||||||
"requests",
|
|
||||||
"pandas[excel,html,output-formatting,hdf5,compression,parquet,feather]",
|
|
||||||
"numpy",
|
|
||||||
"openpyxl",
|
|
||||||
"xlrd",
|
|
||||||
"selenium",
|
|
||||||
"dotenv",
|
|
||||||
"icalendar",
|
|
||||||
"Babel",
|
|
||||||
#"gcsa",
|
|
||||||
#"caldav",
|
|
||||||
"colorlog",
|
|
||||||
"python-datauri",
|
|
||||||
"rich",
|
|
||||||
"Jinja2",
|
|
||||||
"base58",
|
"base58",
|
||||||
|
"colorlog",
|
||||||
|
"dotenv",
|
||||||
|
"rich",
|
||||||
]
|
]
|
||||||
|
|
||||||
authors = [
|
authors = [
|
||||||
@@ -54,12 +41,44 @@ dev = [
|
|||||||
"ruff>=0.4",
|
"ruff>=0.4",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
i18n = [
|
||||||
|
"Babel",
|
||||||
|
]
|
||||||
|
|
||||||
|
audio = [
|
||||||
|
"mutagen",
|
||||||
|
"selenium",
|
||||||
|
]
|
||||||
|
|
||||||
|
web = [
|
||||||
|
"httpx",
|
||||||
|
"Jinja2",
|
||||||
|
"python-datauri",
|
||||||
|
"requests",
|
||||||
|
]
|
||||||
|
|
||||||
|
webdav = [
|
||||||
|
"icalendar",
|
||||||
|
#"gcsa",
|
||||||
|
#"caldav",
|
||||||
|
]
|
||||||
|
|
||||||
|
data = [
|
||||||
|
"numpy",
|
||||||
|
"openpyxl",
|
||||||
|
"pandas[excel,html,output-formatting,hdf5,compression,parquet,feather]",
|
||||||
|
"xlrd",
|
||||||
|
]
|
||||||
|
|
||||||
[project.urls]
|
[project.urls]
|
||||||
Homepage = "https://git.silveirarosa.com/renatoxsr/monorepo"
|
Homepage = "https://git.silveirarosa.com/renatoxsr/monorepo"
|
||||||
Issues = "https://git.silveirarosa.com/renatoxsr/monorepo/issues"
|
Issues = "https://git.silveirarosa.com/renatoxsr/monorepo/issues"
|
||||||
|
|
||||||
[project.scripts]
|
[project.scripts]
|
||||||
composectl = "renatoxsr.scripts.composectl:main"
|
composectl = "renatoxsr.scripts.composectl:main"
|
||||||
|
bulk-jsonl = "renatoxsr.scripts.bulk_jsonl:main"
|
||||||
|
avaliar-dataset = "renatoxsr.scripts.avaliar_dataset:main"
|
||||||
|
prompt = "renatoxsr.scripts.prompt:main"
|
||||||
|
|
||||||
# SETUPTOOLS
|
# SETUPTOOLS
|
||||||
[tool.setuptools.packages.find]
|
[tool.setuptools.packages.find]
|
||||||
|
|||||||
@@ -16,3 +16,6 @@ logger.debug("node: %s", platform.node())
|
|||||||
# Import modules that should be exported
|
# Import modules that should be exported
|
||||||
import renatoxsr.utils
|
import renatoxsr.utils
|
||||||
import renatoxsr.scripts.composectl
|
import renatoxsr.scripts.composectl
|
||||||
|
import renatoxsr.scripts.bulk_jsonl
|
||||||
|
import renatoxsr.scripts.avaliar_dataset
|
||||||
|
import renatoxsr.scripts.prompt
|
||||||
|
|||||||
@@ -35,7 +35,6 @@ def get_formatter(format: str = FORMAT) -> logging.Formatter:
|
|||||||
|
|
||||||
|
|
||||||
def set_logfile(
|
def set_logfile(
|
||||||
*,
|
|
||||||
name: str | None = None,
|
name: str | None = None,
|
||||||
formatter: logging.Formatter | None = None,
|
formatter: logging.Formatter | None = None,
|
||||||
format: str | None = None,
|
format: str | None = None,
|
||||||
@@ -103,7 +102,6 @@ def set_loglevel(logger_instance: logging.Logger | None = None, logLevel="") ->
|
|||||||
|
|
||||||
|
|
||||||
def set_logger(
|
def set_logger(
|
||||||
*,
|
|
||||||
name: str | None = None,
|
name: str | None = None,
|
||||||
formatter: logging.Formatter | None = None,
|
formatter: logging.Formatter | None = None,
|
||||||
format: str | None = None,
|
format: str | None = None,
|
||||||
|
|||||||
@@ -0,0 +1,156 @@
|
|||||||
|
#vim: ts=4 sw=4 et ft=python :
|
||||||
|
import requests
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
from pprint import pprint
|
||||||
|
|
||||||
|
# --- CONFIGURAÇÕES ---
|
||||||
|
OLLAMA_URL = "http://192.168.14.20:11434/api/generate"
|
||||||
|
MODEL_NAME = "gemma4:e4b"
|
||||||
|
INPUT_FILE_PATH = Path("dataset.jsonl")
|
||||||
|
OUTPUT_FILE_PATH = Path("avaliado.jsonl")
|
||||||
|
|
||||||
|
# O System Prompt que criamos, definido como a regra mestra.
|
||||||
|
SYSTEM_PROMPT = """
|
||||||
|
Você é um Assistente Jurídico Sênior e especialista em Direito brasileiro. Sua função é realizar uma análise técnica, crítica e altamente estruturada de documentos jurídicos. Você deve manter o tom de voz de um profissional qualificado, objetivo e conciso.
|
||||||
|
|
||||||
|
SEU OBJETIVO: Analisar o documento jurídico fornecido e extrair quatro informações cruciais.
|
||||||
|
|
||||||
|
REGRAS OBRIGATÓRIAS:
|
||||||
|
1. Tom: O tom de todas as respostas deve ser técnico, formal e imparcial.
|
||||||
|
2. Formato: A saída deve ser ESTREITAMENTE um objeto JSON válido. Nenhum texto introdutório, explicação ou texto fora do JSON é permitido.
|
||||||
|
3. Detalhe: Mantenha a clareza e a profundidade técnica em todos os campos.
|
||||||
|
|
||||||
|
AS CHAVES E AS REGRAS DE CADA CAMPO:
|
||||||
|
|
||||||
|
(1) `tipo`: Deve identificar a categoria jurídica principal do documento (ex: Contrato de Prestação de Serviços, Petição Inicial, Escritura Pública, Contrato de Compra e Venda, Procuração Ad Judicia). Seja o mais específico possível.
|
||||||
|
(2) `resumo`: Um resumo executivo do objeto do documento, contendo no máximo 3 frases. Deve ser extremamente conciso, focado no tema central.
|
||||||
|
(3) `argumentos`: Lista em formato de array (JSON array) dos principais pontos jurídicos, premissas ou teses argumentativas que sustentam o documento.
|
||||||
|
(4) `qualidade`: Avaliação da qualidade do documento sob o ponto de vista técnico-jurídico. Você deve escolher *somente* um dos seguintes termos: "ótimo", "bom" ou "médio".
|
||||||
|
* "Ótimo": Documento juridicamente impecável, claro, completo e em conformidade com a melhor prática do direito.
|
||||||
|
* "Bom": Documento funcional, com clareza e cobertura dos pontos principais, mas pode ter pequenas imprecisões ou omissões.
|
||||||
|
* "Médio": Documento que apresenta falhas significativas de linguagem, inconsistências jurídicas graves ou que não cumpre seu propósito principal sem intervenção.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def get_llm_response(prompt_full: str) -> str:
|
||||||
|
"""
|
||||||
|
Envia o prompt para o Ollama local e retorna a resposta em string.
|
||||||
|
"""
|
||||||
|
print("🤖 Enviando requisição para o Ollama...")
|
||||||
|
try:
|
||||||
|
payload = {
|
||||||
|
"model": MODEL_NAME,
|
||||||
|
"options": {
|
||||||
|
"temperature": 0.1, # Temperatura baixa para respostas determinísticas e factuais
|
||||||
|
"num_predict": 8192, # Aumenta o limite de tokens
|
||||||
|
"num_ctx": 128000, # tamanho do contexto
|
||||||
|
},
|
||||||
|
"stream": False,
|
||||||
|
"prompt": prompt_full,
|
||||||
|
}
|
||||||
|
|
||||||
|
#print(json.dumps(payload))
|
||||||
|
response = requests.post(OLLAMA_URL, json=payload)
|
||||||
|
response.raise_for_status() # Levanta exceção para códigos de erro HTTP
|
||||||
|
|
||||||
|
# O Ollama retorna os resultados dentro do campo 'response'
|
||||||
|
return response.json()['response'].strip()
|
||||||
|
|
||||||
|
except requests.exceptions.ConnectionError:
|
||||||
|
print("\n" + "="*80)
|
||||||
|
print(f"❌ ERRO DE CONEXÃO: Não foi possível conectar ao Ollama em {OLLAMA_URL}.")
|
||||||
|
print("Por favor, verifique se o serviço Ollama está rodando e se a porta e o IP estão corretos.")
|
||||||
|
print("="*80 + "\n")
|
||||||
|
return None
|
||||||
|
except requests.exceptions.RequestException as e:
|
||||||
|
print(f"\n❌ ERRO GERAL ao comunicar com a API: {e}")
|
||||||
|
print(response.request.url)
|
||||||
|
pprint(response.request.headers, width=1)
|
||||||
|
print(response.request.body.decode("utf-8")[:120])
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
"""
|
||||||
|
Processa o arquivo e executa a análise para cada documento.
|
||||||
|
"""
|
||||||
|
if not INPUT_FILE_PATH.exists():
|
||||||
|
print(f"ERRO: Arquivo de entrada '{INPUT_FILE_PATH.name}' não encontrado. Execute o passo anterior primeiro.")
|
||||||
|
return
|
||||||
|
|
||||||
|
print(f"Iniciando a avaliação de {INPUT_FILE_PATH.name} contra o modelo {MODEL_NAME}...")
|
||||||
|
|
||||||
|
results = []
|
||||||
|
|
||||||
|
try:
|
||||||
|
with open(INPUT_FILE_PATH, 'r', encoding='utf-8') as f:
|
||||||
|
lines = f.readlines()
|
||||||
|
except Exception as e:
|
||||||
|
print(f"Erro ao ler o arquivo: {e}")
|
||||||
|
return
|
||||||
|
|
||||||
|
for i, line in enumerate(lines):
|
||||||
|
try:
|
||||||
|
data = json.loads(line.strip())
|
||||||
|
markdown_content = data.get("markdown", "")
|
||||||
|
|
||||||
|
if not markdown_content:
|
||||||
|
continue
|
||||||
|
|
||||||
|
# 1. Montar o Prompt Completo
|
||||||
|
full_prompt = f"""
|
||||||
|
DOCUMENTO JURÍDICO PARA ANÁLISE:
|
||||||
|
---
|
||||||
|
{markdown_content}
|
||||||
|
---
|
||||||
|
"""
|
||||||
|
|
||||||
|
# 2. Adicionar o prompt ao System Prompt
|
||||||
|
final_prompt = f"{SYSTEM_PROMPT}\n\n{full_prompt}"
|
||||||
|
|
||||||
|
# 3. Chamar o LLM
|
||||||
|
llm_output = get_llm_response(final_prompt)
|
||||||
|
|
||||||
|
# 4. Estruturar e Salvar Resultados
|
||||||
|
if llm_output:
|
||||||
|
result_record = {
|
||||||
|
"hash": data.get("hash", "unknown_id"),
|
||||||
|
"documento_original": markdown_content[:200] + "...", # Trunca para melhor visualização
|
||||||
|
"resultado_ollama": llm_output
|
||||||
|
}
|
||||||
|
results.append(result_record)
|
||||||
|
with open(OUTPUT_FILE_PATH, 'a', encoding='utf-8') as f:
|
||||||
|
f.write(json.dumps(result_record, ensure_ascii=False) + '\n')
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
print(f"[{i+1}/{len(lines)}] ✅ Processado com sucesso. Salvando resultado.")
|
||||||
|
else:
|
||||||
|
results.append({
|
||||||
|
"hash": data.get("hash", "unknown_id"),
|
||||||
|
"documento_original": markdown_content[:200] + "...",
|
||||||
|
"resultado_ollama": "FALHA: Verifique os logs de erro acima."
|
||||||
|
})
|
||||||
|
print(f"[{i+1}/{len(lines)}] ❌ FALHA no processamento.")
|
||||||
|
|
||||||
|
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
print(f"AVISO: Linha {i+1} inválida JSON encontrada e pulada.")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"ERRO inesperado ao processar a linha {i+1}: {e}")
|
||||||
|
|
||||||
|
# Salvando todos os resultados
|
||||||
|
#if results:
|
||||||
|
# with open(OUTPUT_FILE_PATH, 'w', encoding='utf-8') as f:
|
||||||
|
# for record in results:
|
||||||
|
# f.write(json.dumps(record, ensure_ascii=False) + '\n')
|
||||||
|
print("\n=======================================================================")
|
||||||
|
print("✅ Avaliação concluída!")
|
||||||
|
print(f"Total de documentos processados e salvos em: {OUTPUT_FILE_PATH.name}")
|
||||||
|
print("Examine este arquivo para identificar padrões de erro ou sucesso.")
|
||||||
|
print("=======================================================================")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,47 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# vim: set ft=python ts=4 sw=4 et encoding=utf-8 :
|
||||||
|
# SPDX-FileCopyrightText: 2025-present RenatoXSR <renatoxsr@gmail.com>
|
||||||
|
# SPDX-License-Identifier: BSD-3-Clause-Clear
|
||||||
|
import argparse, json, os, sys
|
||||||
|
from pathlib import Path
|
||||||
|
import datetime as dt
|
||||||
|
|
||||||
|
from renatoxsr.utils import hash_b58, pandoc
|
||||||
|
from renatoxsr.logger import set_logger
|
||||||
|
logger = set_logger(__name__)
|
||||||
|
if "--debug" in sys.argv:
|
||||||
|
logger.setLevel("DEBUG")
|
||||||
|
|
||||||
|
def parse_args():
|
||||||
|
p = argparse.ArgumentParser()
|
||||||
|
p.add_argument("--outdir", default=os.getcwd())
|
||||||
|
p.add_argument("--outfile",
|
||||||
|
default=dt.datetime.now().isoformat().replace(":", "_")+".jsonl",
|
||||||
|
help="output file (default: `datetime.now().isoformat()`.jsonl)")
|
||||||
|
p.add_argument("--input", default=os.getcwd(), help="input folder")
|
||||||
|
p.add_argument("--glob", default="*")
|
||||||
|
p.add_argument("--overwrite", default=False)
|
||||||
|
p.add_argument("--logfile", default=sys.stderr, help="Log output(default: sys.stderr)")
|
||||||
|
p.add_argument("--loglevel", default="INFO", help="Log level (default: INFO)")
|
||||||
|
p.add_argument("filename", nargs="*", help="input file or glob pattern (default: `datetime.now().isoformat()`)")
|
||||||
|
return p.parse_args()
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parsed_args = parse_args()
|
||||||
|
logger.setLevel(parsed_args.loglevel)
|
||||||
|
process_files = list()#Path(parsed_args.input).glob(parsed_args.glob))
|
||||||
|
process_files.extend(parsed_args.filename)
|
||||||
|
with open(Path(parsed_args.outdir) / parsed_args.outfile, "w" if parsed_args.overwrite else "a") as output_file:
|
||||||
|
for file in process_files:
|
||||||
|
fpath = Path(file)
|
||||||
|
data = {"hash": hash_b58(fpath, output_print=False), "fpath": str(fpath), "markdown": pandoc(fpath, to="markdown")}
|
||||||
|
if not data["markdown"]:
|
||||||
|
logger.error("No markdown output from '%s' (hash:%s)", fpath, data["hash"])
|
||||||
|
continue
|
||||||
|
logger.info("Writing markdown content (length: %d) to '%s' from '%s' (hash:%s)",
|
||||||
|
len(data["markdown"]), parsed_args.outfile, fpath, data["hash"])
|
||||||
|
output_file.write(json.dumps(data)+"\n")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -130,12 +130,6 @@ class LevelColorFormatter(logging.Formatter):
|
|||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
logger.setLevel(logging.DEBUG)
|
logger.setLevel(logging.DEBUG)
|
||||||
|
|
||||||
# Console Logger
|
|
||||||
stderr_handler = logging.StreamHandler(stream=sys.stderr)
|
|
||||||
stderr_handler.setFormatter(LevelColorFormatter())
|
|
||||||
stderr_handler.setLevel(logging.DEBUG)
|
|
||||||
logger.addHandler(stderr_handler)
|
|
||||||
#print(logger.handlers)
|
|
||||||
|
|
||||||
# ***
|
# ***
|
||||||
|
|
||||||
@@ -166,8 +160,6 @@ def check_python_version(version_info, min_version):
|
|||||||
min_version[1])
|
min_version[1])
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
|
|
||||||
# Run check
|
|
||||||
check_python_version(sys.version_info, MIN_VERSION)
|
|
||||||
|
|
||||||
# Package version
|
# Package version
|
||||||
def set_package_version(file):
|
def set_package_version(file):
|
||||||
@@ -176,8 +168,6 @@ def set_package_version(file):
|
|||||||
logger.debug("%s:%s", file, version)
|
logger.debug("%s:%s", file, version)
|
||||||
return version
|
return version
|
||||||
|
|
||||||
# Set version
|
|
||||||
__version__ = set_package_version(__file__)
|
|
||||||
|
|
||||||
# Global Functions
|
# Global Functions
|
||||||
|
|
||||||
@@ -339,6 +329,18 @@ def main():
|
|||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|
||||||
|
# Console Logger
|
||||||
|
stderr_handler = logging.StreamHandler(stream=sys.stderr)
|
||||||
|
stderr_handler.setFormatter(LevelColorFormatter())
|
||||||
|
stderr_handler.setLevel(logging.DEBUG)
|
||||||
|
logger.addHandler(stderr_handler)
|
||||||
|
#print(logger.handlers)
|
||||||
|
|
||||||
|
# Run check
|
||||||
|
check_python_version(sys.version_info, MIN_VERSION)
|
||||||
|
# Set version
|
||||||
|
__version__ = set_package_version(__file__)
|
||||||
try:
|
try:
|
||||||
sys.exit(main())
|
sys.exit(main())
|
||||||
except KeyboardInterrupt as sigint:
|
except KeyboardInterrupt as sigint:
|
||||||
|
|||||||
@@ -0,0 +1,187 @@
|
|||||||
|
#python << '#EOF'
|
||||||
|
# vim: ft=python ts=4 sw=4 et :
|
||||||
|
import requests
|
||||||
|
import asyncio, enum, os, sys, json, pathlib, argparse
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from rich import print, print_json
|
||||||
|
from rich.console import Console
|
||||||
|
from rich.prompt import Prompt
|
||||||
|
from rich.json import JSON
|
||||||
|
from rich.markdown import Markdown
|
||||||
|
|
||||||
|
output={sys.stdout: Console()}
|
||||||
|
class bb(enum.StrEnum):
|
||||||
|
thinking="bold yellow"
|
||||||
|
content="bold blue"
|
||||||
|
model="bold red"
|
||||||
|
prompt="bold green"
|
||||||
|
line="italic"
|
||||||
|
|
||||||
|
def write_rich(*msg, end='\n', stream=sys.stdout):
|
||||||
|
output[stream].print(*msg, end=end)
|
||||||
|
if end != '\n':
|
||||||
|
stream.flush()
|
||||||
|
|
||||||
|
def write_basic(*msg, sep=' ', end='\n', stream=sys.stdout):
|
||||||
|
stream.write(sep.join(*msg) + end)
|
||||||
|
if end != '\n':
|
||||||
|
stream.flush()
|
||||||
|
|
||||||
|
write = write_rich
|
||||||
|
|
||||||
|
def parse_args():
|
||||||
|
p = argparse.ArgumentParser()
|
||||||
|
p.add_argument("--debug", action="store_true")
|
||||||
|
p.add_argument("--model",
|
||||||
|
default="qwen3.5:4b",
|
||||||
|
choices=[
|
||||||
|
"gemma4:e4b",
|
||||||
|
"llama3.2:3b",
|
||||||
|
"llama3.1:8b",
|
||||||
|
"qwen3.5:4b",
|
||||||
|
])
|
||||||
|
p.add_argument("--prompt")
|
||||||
|
p.add_argument("--system",
|
||||||
|
default="You are a helpful assistant. Write concisely.")
|
||||||
|
|
||||||
|
mxg = p.add_mutually_exclusive_group()
|
||||||
|
mxg.add_argument("--pull")
|
||||||
|
mxg.add_argument("--response", action="store_true")
|
||||||
|
mxg.add_argument("--endpoint")
|
||||||
|
mxg.add_argument("--api",
|
||||||
|
default="ollama",
|
||||||
|
choices=["ollama","openai",])
|
||||||
|
g = p.add_argument_group("Model Options")
|
||||||
|
g.add_argument("--thinking", action=argparse.BooleanOptionalAction, default=True)
|
||||||
|
g.add_argument("--stream", action="store_true")
|
||||||
|
g.add_argument("--num_ctx", default=128000, type=int)
|
||||||
|
g.add_argument("--top_k", default=40, type=int)
|
||||||
|
g.add_argument("--top_p", default=0.9, type=float)
|
||||||
|
g.add_argument("--temperature", default=0.2, type=float)
|
||||||
|
|
||||||
|
url = p.add_argument_group("URL")
|
||||||
|
url.add_argument("--url",
|
||||||
|
default=os.getenv("PROMPT_URL", "http://localhost:11434"))
|
||||||
|
url.add_argument("--host",
|
||||||
|
default=os.getenv("PROMPT_HOST", "localhost"))
|
||||||
|
url.add_argument("--port",
|
||||||
|
default=os.getenv("PROMPT_PORT", "11434"))
|
||||||
|
url.add_argument("--proto",
|
||||||
|
default=os.getenv("PROMPT_PROTO", "http"))
|
||||||
|
url.add_argument("--apikey")
|
||||||
|
url.add_argument("--timeout", type=int, default=180)
|
||||||
|
return p.parse_args()
|
||||||
|
|
||||||
|
|
||||||
|
def s3f(ns):
|
||||||
|
return f"{(ns // 1_000_000) / 1000:.3f}"
|
||||||
|
|
||||||
|
|
||||||
|
async def do_async_turn(conf):
|
||||||
|
"""Do user->assistant turn streaming responses"""
|
||||||
|
if not conf.prompt:
|
||||||
|
prompt = Prompt.ask(f"[{str(bb.model)}]Ask {conf.model}»[/] [{str(bb.prompt)}]")
|
||||||
|
if conf.response:
|
||||||
|
system_and_user = {
|
||||||
|
"input" if conf.api == "openai" else "prompt": conf.prompt or prompt,
|
||||||
|
"instructions" if conf.api == "openai" else "system": conf.system}
|
||||||
|
else:
|
||||||
|
system_and_user = {"messages": [
|
||||||
|
{"role":"system", "content": conf.system},
|
||||||
|
{"role": "user", "content": conf.prompt or prompt}]}
|
||||||
|
|
||||||
|
client = httpx.AsyncClient(
|
||||||
|
base_url="http://192.168.14.20:11434",
|
||||||
|
headers={
|
||||||
|
"Authorization": f"Bearer {conf.apikey}",},
|
||||||
|
timeout=float(conf.timeout))
|
||||||
|
async with client.stream('POST',((
|
||||||
|
"/api/generate" if conf.response else "/api/chat")
|
||||||
|
if conf.api == "ollama" else ("/v1/responses"
|
||||||
|
if conf.response else "/v1/chat/completions")),
|
||||||
|
json=system_and_user | {
|
||||||
|
"model": conf.model,
|
||||||
|
"reasoning": {"enabled": conf.thinking},
|
||||||
|
"thinking": conf.thinking,
|
||||||
|
"stream": conf.stream,
|
||||||
|
"options": {
|
||||||
|
"num_ctx": conf.num_ctx,
|
||||||
|
"top_k": conf.top_k,
|
||||||
|
"top_p": conf.top_p,
|
||||||
|
"temperature": conf.temperature,
|
||||||
|
}}) as res:
|
||||||
|
res.raise_for_status()
|
||||||
|
if conf.debug:
|
||||||
|
write(res.request, end=' (')
|
||||||
|
write(", ".join([d for d in dir(res.request) if not d.startswith("_")]), end=")\n")
|
||||||
|
write(res.request.content)
|
||||||
|
write(res, end=' (')
|
||||||
|
write(", ".join([d for d in dir(res) if not d.startswith("_")]), end=')\n')
|
||||||
|
msg = {"thinking": {"len": 1, 1: []},
|
||||||
|
"content": {"len": 1, 1: []}}
|
||||||
|
async for line in res.aiter_lines():
|
||||||
|
if not line:
|
||||||
|
write('[i grey].[/]', end='')
|
||||||
|
# TODO: use spinner from rich
|
||||||
|
continue
|
||||||
|
data = json.loads(line)
|
||||||
|
if conf.debug:
|
||||||
|
write(line)
|
||||||
|
if data.get("done", False):
|
||||||
|
write(f"\n[Load: {s3f(data['load_duration'])}s | "
|
||||||
|
f"Analyze: {s3f(data['prompt_eval_duration'])}s | "
|
||||||
|
f"Generate: {s3f(data['eval_duration'])}s] "
|
||||||
|
f"Total: {s3f(data['total_duration'])}s"
|
||||||
|
f"\nToken usage: "
|
||||||
|
f"{data['prompt_eval_count']} input, "
|
||||||
|
f"{data['prompt_eval_cached_count']} cached, "
|
||||||
|
f"{data['eval_count']} output.\n")
|
||||||
|
msg['done'] = data
|
||||||
|
return msg
|
||||||
|
#write(f"[i]{data}[/i]")
|
||||||
|
for i in ["thinking", "content"]:
|
||||||
|
text = data.get("message", {}).get(i, "")
|
||||||
|
msg[i][msg[i]['len']].append(text)
|
||||||
|
write(f"[{str(bb[i])}]{text}[/]", end='')
|
||||||
|
if '\n' in text:
|
||||||
|
#write(f"[{str(bb.line)}]{msg[i]['len']:>3d}:[/] "
|
||||||
|
# f"[{str(bb[i])}]")
|
||||||
|
#write(Markdown("".join(msg[i][msg[i]['len']])))
|
||||||
|
msg[i]['len'] += 1
|
||||||
|
msg[i][msg[i]['len']] = []
|
||||||
|
write(data.get('choices',[{}])[0].get('message', ""), end='')
|
||||||
|
write(data.get('output', ""), end='')
|
||||||
|
write(f"[{str(bb.model)}]{data.get('response', '')}", end='')
|
||||||
|
write(f"[{str(bb.thinking)}]{data.get('thinking', '')}[/]", end='')
|
||||||
|
|
||||||
|
def main():
|
||||||
|
conf = parse_args()
|
||||||
|
if conf.prompt == "-":
|
||||||
|
conf.prompt = sys.stdin.read()
|
||||||
|
if conf.prompt:
|
||||||
|
write(f"[{str(bb.model)}]Ask {conf.model}»[/] ", end='')
|
||||||
|
if conf.response:
|
||||||
|
write(f"[{str(bb.prompt)}]{conf.system}[/]")
|
||||||
|
else:
|
||||||
|
write(f"[{str(bb.prompt)}]{conf.prompt}[/]")
|
||||||
|
try:
|
||||||
|
while True:
|
||||||
|
if conf.stream:
|
||||||
|
data = asyncio.run(do_async_turn(conf))
|
||||||
|
else:
|
||||||
|
data = res.json()
|
||||||
|
if "thinking" in data.get("message",{}):
|
||||||
|
write(f"[{str(bb.model)}]{conf.model}'s thinking:[/] [{str(bb.thinking)}] {Markdown(data['message']['thinking'])}[/]")
|
||||||
|
write(f"[{str(bb.model)}]{conf.model}'s response: [/][{str(bb.content)}]{Markdown(data['message']['content'])}[/]")
|
||||||
|
write(data['choices'][0]['message'])
|
||||||
|
write(data['output'])
|
||||||
|
if conf.prompt:
|
||||||
|
sys.exit(0)
|
||||||
|
# TODO: add turn to message list and do another turn
|
||||||
|
except KeyboardInterrupt as e:
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
|
#EOF
|
||||||
@@ -4,3 +4,8 @@
|
|||||||
# SPDX-License-Identifier: BSD-3-Clause-Clear
|
# SPDX-License-Identifier: BSD-3-Clause-Clear
|
||||||
"""`utils` package of RenatoXSR's Python Utilities (monorepo)"""
|
"""`utils` package of RenatoXSR's Python Utilities (monorepo)"""
|
||||||
import renatoxsr.utils.uuid7_base58
|
import renatoxsr.utils.uuid7_base58
|
||||||
|
import renatoxsr.utils.jsonl
|
||||||
|
import renatoxsr.utils.zwid
|
||||||
|
from renatoxsr.utils.pandoc import pandoc
|
||||||
|
from renatoxsr.utils.hash_b58 import hash_b58
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,32 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# vim: set ft=python ts=4 sw=4 et fenc=utf-8 :
|
||||||
|
# SPDX-FileCopyrightText: 2025-present RenatoXSR <renatoxsr@gmail.com>
|
||||||
|
# SPDX-License-Identifier: BSD-3-Clause-Clear
|
||||||
|
# vim: ts=4 tw=4 sw=4 et:
|
||||||
|
import base58
|
||||||
|
import os, sys, hashlib
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
from renatoxsr.logger import set_logger
|
||||||
|
logger = set_logger(__name__)
|
||||||
|
if "--debug" in sys.argv:
|
||||||
|
logger.setLevel("DEBUG")
|
||||||
|
|
||||||
|
def hash_b58(src_arg: str | Path, hash_algo="sha256", output_print=True, output_stream=sys.stdout):
|
||||||
|
src = Path(src_arg)
|
||||||
|
with open(src, "rb") as f:
|
||||||
|
hash = hashlib.file_digest(f, hash_algo)
|
||||||
|
hash_hex = hash.hexdigest()
|
||||||
|
hash_b58 = base58.b58encode(hash.digest()).decode("utf-8")
|
||||||
|
if output_print:
|
||||||
|
output_stream.write(f"{hash_algo}:{hash_hex} {hash_b58} '{src}'\n")
|
||||||
|
else:
|
||||||
|
logger.info(f"%s:%s %s '%s'", hash_algo, hash_hex, hash_b58, src)
|
||||||
|
return hash_b58
|
||||||
|
|
||||||
|
|
||||||
|
def hash_recursive_b58(folder: str | Path, glob="*"):
|
||||||
|
src = Path(folder)
|
||||||
|
for filename in src.glob(glob):
|
||||||
|
hash_b58(filename)
|
||||||
@@ -0,0 +1,81 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# vim: set ft=python ts=4 sw=4 et: fileencoding=utf-8
|
||||||
|
# SPDX-FileCopyrightText: 2025-present RenatoXSR <renatoxsr@gmail.com>
|
||||||
|
# SPDX-License-Identifier: BSD-3-Clause-Clear
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
from io import BytesIO
|
||||||
|
|
||||||
|
def parse_jsonl(file_path):
|
||||||
|
"""Parse a JSONL file line by line"""
|
||||||
|
with open(file_path, 'r') as file:
|
||||||
|
for line in file:
|
||||||
|
line = line.strip()
|
||||||
|
if line:
|
||||||
|
try:
|
||||||
|
yield json.loads(line)
|
||||||
|
except json.JSONDecodeError as e:
|
||||||
|
print(f"Error parsing line: {e}")
|
||||||
|
continue
|
||||||
|
|
||||||
|
|
||||||
|
def stream_jsonl(file_path: str) -> Iterator[Dict[str, Any]]:
|
||||||
|
"""Stream JSONL file for large datasets"""
|
||||||
|
with open(file_path, 'r', encoding='utf-8') as file:
|
||||||
|
for line_num, line in enumerate(file, 1):
|
||||||
|
line = line.strip()
|
||||||
|
if not line:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
yield json.loads(line)
|
||||||
|
except json.JSONDecodeError as e:
|
||||||
|
print(f"Error on line {line_num}: {e}")
|
||||||
|
continue
|
||||||
|
|
||||||
|
|
||||||
|
def jsonl_df(file_path):
|
||||||
|
"""Convert JSONL to pandas DataFrame"""
|
||||||
|
import pandas as pd
|
||||||
|
data = []
|
||||||
|
with open(file_path, 'r') as file:
|
||||||
|
for line in file:
|
||||||
|
line = line.strip()
|
||||||
|
if line:
|
||||||
|
try:
|
||||||
|
data.append(json.loads(line))
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
continue
|
||||||
|
return pd.DataFrame(data)
|
||||||
|
|
||||||
|
|
||||||
|
def get_demo_bytes(demo_file_path: str | Path = None):
|
||||||
|
if demo_file_path:
|
||||||
|
return Path(demo_file_path)
|
||||||
|
return BytesIO(
|
||||||
|
# example JSONL file:
|
||||||
|
"""{"row1":"col1","row1":"col2"}\n{"row2":"col1","row2":"col2"}""".encode("utf-8"))
|
||||||
|
|
||||||
|
|
||||||
|
def demo_parse_jsonl(demo_file_path: str | Path = None):
|
||||||
|
"""demo parse_jsonl() function"""
|
||||||
|
demo_io = get_demo_bytes(demo_file_path)
|
||||||
|
for data in parse_jsonl(demo_io):
|
||||||
|
print(data)
|
||||||
|
streaming_parser.py
|
||||||
|
import json
|
||||||
|
from typing import Iterator, Dict, Any
|
||||||
|
|
||||||
|
|
||||||
|
def demo_stream_jsonl(demo_file_path: str | Path = None):
|
||||||
|
"""Demo stream_jsonl() function"""
|
||||||
|
demo_io = get_demo_bytes(demo_file_path)
|
||||||
|
# Process large files without memory issues
|
||||||
|
for record in stream_jsonl(demo_io):
|
||||||
|
process_record(record)
|
||||||
|
pandas_parser.py
|
||||||
|
|
||||||
|
def demo_jsonl_df(demo_file_path: str | Path = None):
|
||||||
|
"""Demo jsonl_df() function"""
|
||||||
|
demo_io = get_demo_bytes(demo_file_path)
|
||||||
|
df = jsonl_df(demo_io)
|
||||||
|
print(df.head())
|
||||||
@@ -0,0 +1,45 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# vim: set ft=python ts=4 sw=4 et encoding=utf-8 :
|
||||||
|
# SPDX-FileCopyrightText: 2025-present RenatoXSR <renatoxsr@gmail.com>
|
||||||
|
# SPDX-License-Identifier: BSD-3-Clause-Clear
|
||||||
|
import errno, os, shutil, subprocess, sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from renatoxsr.logger import set_logger
|
||||||
|
logger = set_logger(__name__)
|
||||||
|
if "--debug" in sys.argv:
|
||||||
|
logger.setLevel("DEBUG")
|
||||||
|
|
||||||
|
def pandoc(
|
||||||
|
input_file: str | Path,
|
||||||
|
bin: str | Path = "pandoc",
|
||||||
|
output: str | Path = "-",
|
||||||
|
overwrite = False,
|
||||||
|
to="markdown",
|
||||||
|
) -> str:
|
||||||
|
logger.info("%s: from '%s' to '%s'", bin, input_file, output)
|
||||||
|
if not Path(bin).is_file():
|
||||||
|
which_bin = shutil.which(bin)
|
||||||
|
if not Path(which_bin).is_file():
|
||||||
|
raise FileNotFoundError(errno.ENOENT, os.strerror(errno.ENOENT), bin)
|
||||||
|
|
||||||
|
fpath = Path(input_file)
|
||||||
|
if not fpath.is_file():
|
||||||
|
raise FileNotFoundError(errno.ENOENT, os.strerror(errno.ENOENT), input_file)
|
||||||
|
|
||||||
|
outf = Path(output)
|
||||||
|
if output != "-" and outf.is_file() and not overwrite:
|
||||||
|
raise FileExistsError(errno.EEXIST, os.strerror(errno.EEXIST), outf)
|
||||||
|
cmd = [bin,
|
||||||
|
"--output", "-" if output=="-" else outf.resolve(),
|
||||||
|
"--write", to, fpath]
|
||||||
|
logger.debug("CMD %s", str(cmd))
|
||||||
|
proc = subprocess.run(cmd, capture_output=True, text=True)
|
||||||
|
|
||||||
|
if proc.stderr:
|
||||||
|
logger.error(proc.stderr)
|
||||||
|
|
||||||
|
if proc.stdout and output != "-":
|
||||||
|
logger.info(proc.stdout)
|
||||||
|
else:
|
||||||
|
return(proc.stdout)
|
||||||
Reference in New Issue
Block a user