5a26dfde2a
后端变更: - 新增 risk_config.py: 风险配置数据类,支持环境变量驱动 - 新增 risk_control.py: 风险控制控制器,管理并发和预算 - 新增 session_store.py: 匿名会话存储,基于 cookie 的 session ID - 新增 audit_store.py: API 审计日志存储,记录请求和 LLM 调用 - 新增 captcha_api.py: 验证码 API,用于验证用户操作真实性 - 新增 llm_policy.py: LLM 策略配置,管理 completion/pro/vision 模型 - main.py: 集成 middleware、risk/audit/session 模块 (+467/-7) - job_handlers.py: LLM 执行流程重构,新增 risk/audit 集成 (+207/-4) - llm.py: 异步客户端封装,新增 max_output_tokens 参数 (+78/-1) - job_system.py: stream_events 逻辑优化,支持心跳检测 (+12/-4) - pro_completions.py: SSE heartbeat 机制,防止连接超时 (+14/-4) - prompt.py: _normalize_preferences 支持 Mapping 类型 (+13/-0) - tts_asr.py: asyncio loop 初始化,router export (+10/-0) 前端变更: - src/components/CaptchaComponent.vue: 新增验证码组件 (NEW) - src/utils/cookie_policy.js: Cookie 策略工具 (NEW) - SettingsPanel.vue: 集成验证码组件,新增安全设置部分 (+59/-0) - MilkdownEditor.vue: 移除硬编码 API_KEY,新增 credentials (+32/-10) - ProBlockCrepe.vue: 样式简化,移除渐变动画 (+18/-4) - proBlockPlugin.ts: 重构 schema/serializer 引用方式,通过 Ctx 管理 (+40/-10) - api.js: 新增 credentials,重构 headers 条件逻辑 (+50/-14) - config.js: API 基址改为 https://api.imageteach.tech:8002 (+8/-4) - convert.js, docsApi.js, i18n.js: 新增 credentials 和验证码 i18n (+54/-12) - proAccept.js: 重构正则和转义处理,修复捕获组索引 (+14/-4) 配置和基础设施: - docker-compose.yml: 新增端口映射 8001:8001 (+2/-0) - docker/nginx.conf: 改为 307 redirect,优化代理配置 (+8/-6) - vite.config.js: 移除 proxy 配置,直接调用远程 API (+8/-4) - .env.example: 新增 VITE_API_BASE_URL, VITE_API_KEY (+3/-1) - backend/.env.example: 大量 RISK_*, SESSION_*, CORS_* 配置 (+54/-0) - pytest.ini: 扩展 coverage 范围到整个 backend,移除 fail_under (+3/-2) - .coveragerc: 移除 fail_under = 90 (+0/-1) - .gitignore: 新增 docker-data/ (+3/-0) - package.json: 新增 vue3-captcha 依赖 (+3/-1) - AGENTS.md, README.md: 更新 Docker 部署和前端网络约定 (+20/-5) - public/sw.js: Service Worker cache 版本从 v1 升级到 v2 (+0/-1) 测试变更: - test_main_endpoints.py: 新增 session/risk/audit reset,新增测试用例 (+63/-4) - test_main_cancel.py: 新增 reset 调用 (+6/-0) - test_pro_completions.py: 新增 preferences 序列化和测试 (+23/-0) 总计: 45 个文件变更,+1009/-280 行
622 lines
20 KiB
Python
622 lines
20 KiB
Python
import os
|
|
import time
|
|
import logging
|
|
import asyncio
|
|
import inspect
|
|
import json
|
|
import base64
|
|
from datetime import datetime
|
|
from typing import AsyncIterator, Literal
|
|
|
|
import httpx
|
|
from dotenv import load_dotenv
|
|
|
|
from prompts import get_vlm_ocr_prompt
|
|
|
|
load_dotenv()
|
|
|
|
# OpenAI-compatible endpoint config
|
|
LLM_BASE_URL = os.getenv('LLM_BASE_URL', 'http://localhost:11434/v1/')
|
|
LLM_API_KEY = os.getenv('LLM_API_KEY', 'ollama')
|
|
|
|
# Auth headers for upstream LLM service (OpenAI-compatible Bearer token)
|
|
LLM_HEADERS = {'Authorization': f'Bearer {LLM_API_KEY}'}
|
|
|
|
# Model names (backward compat: fall back to OLLAMA_MODEL if LLM_MODEL not set)
|
|
_raw_model = os.getenv('LLM_MODEL') or os.getenv('OLLAMA_MODEL', 'gpt-oss:20b')
|
|
LLM_MODEL = _raw_model.strip() if _raw_model else 'gpt-oss:20b'
|
|
PRO_LLM_MODEL = os.getenv('PRO_LLM_MODEL', LLM_MODEL)
|
|
|
|
# VLM for OCR (vision models)
|
|
VLM_MODEL = os.getenv('VLM_MODEL', 'qwen3-vl:30b')
|
|
|
|
# Fallback for legacy OLLAMA_HOST env var (auto-convert to /v1/ path)
|
|
_legacy_host = os.getenv('OLLAMA_HOST')
|
|
if _legacy_host and not os.getenv('LLM_BASE_URL'):
|
|
base = _legacy_host.rstrip('/')
|
|
if '/v1' not in base:
|
|
LLM_BASE_URL = f"{base}/v1/"
|
|
|
|
# Normalize trailing slash for base URL
|
|
LLM_BASE_URL = LLM_BASE_URL.rstrip('/') + '/'
|
|
|
|
# Timeouts in seconds (10 minutes for large model loading)
|
|
COMPLETION_TIMEOUT = int(os.getenv("LLM_COMPLETION_TIMEOUT", "600"))
|
|
OCR_TIMEOUT = int(os.getenv("LLM_OCR_TIMEOUT", "600"))
|
|
|
|
|
|
async def _maybe_await(value):
|
|
if inspect.isawaitable(value):
|
|
return await value
|
|
return value
|
|
|
|
|
|
class _AsyncClientContext:
|
|
def __init__(self, client):
|
|
self.client = client
|
|
|
|
async def __aenter__(self):
|
|
return self.client
|
|
|
|
async def __aexit__(self, *args):
|
|
close = getattr(self.client, "aclose", None) or getattr(self.client, "close", None)
|
|
if close:
|
|
await _maybe_await(close())
|
|
|
|
|
|
async def _create_async_client(timeout: httpx.Timeout):
|
|
client = await _maybe_await(
|
|
httpx.AsyncClient(base_url=LLM_BASE_URL, headers=LLM_HEADERS, timeout=timeout)
|
|
)
|
|
if hasattr(client, "__aenter__"):
|
|
return client
|
|
return _AsyncClientContext(client)
|
|
|
|
|
|
async def _client_post(client, url: str, payload: dict):
|
|
try:
|
|
return await client.post(url, json=payload)
|
|
except TypeError as exc:
|
|
raw_post = getattr(type(client), "__dict__", {}).get("post")
|
|
if raw_post is None or "multiple values for argument" not in str(exc):
|
|
raise
|
|
return await raw_post(url, json=payload)
|
|
|
|
|
|
async def _stream_line_iterator(response):
|
|
lines = await _maybe_await(response.aiter_lines())
|
|
if hasattr(lines, "__aiter__"):
|
|
return lines.__aiter__()
|
|
return lines
|
|
|
|
logger = logging.getLogger('llm')
|
|
|
|
|
|
def _extract_message(response: dict) -> tuple[str, str]:
|
|
"""Extract content and thinking from an OpenAI-compatible response dict."""
|
|
choices = response.get('choices', []) if isinstance(response, dict) else []
|
|
msg = (choices[0].get('message', {}) if choices and isinstance(choices, list) else {}).copy()
|
|
content = msg.get('content', '') or ''
|
|
thinking = (msg.get('reasoning_content') or msg.get('thinking', '') or '').strip()
|
|
return content, thinking
|
|
|
|
|
|
def _resolve_system_prompt(system_prompt: str | None) -> str:
|
|
if system_prompt and system_prompt.strip():
|
|
return system_prompt.strip()
|
|
return ''
|
|
|
|
|
|
def _resolve_model_name(model: str | None = None, *, use_pro_model: bool = False) -> str:
|
|
candidate = (model or '').strip()
|
|
if candidate:
|
|
return candidate
|
|
return PRO_LLM_MODEL if use_pro_model else LLM_MODEL
|
|
|
|
|
|
def _build_chat_payload(
|
|
prompt: str,
|
|
*,
|
|
system_prompt: str | None = None,
|
|
temperature: float = 0.7,
|
|
thinking: str | None = None,
|
|
model: str | None = None,
|
|
use_pro_model: bool = False,
|
|
prefill: str | None = None,
|
|
max_output_tokens: int | None = None,
|
|
) -> dict:
|
|
messages = []
|
|
sys_prompt = _resolve_system_prompt(system_prompt)
|
|
if sys_prompt:
|
|
messages.append({'role': 'system', 'content': sys_prompt})
|
|
messages.append({'role': 'user', 'content': prompt})
|
|
|
|
if prefill:
|
|
messages.append({'role': 'assistant', 'content': prefill})
|
|
|
|
options = {'temperature': temperature}
|
|
if thinking:
|
|
options['think'] = thinking
|
|
|
|
payload = {
|
|
'model': _resolve_model_name(model, use_pro_model=use_pro_model),
|
|
'messages': messages,
|
|
'stream': False,
|
|
'options': options,
|
|
}
|
|
if max_output_tokens and max_output_tokens > 0:
|
|
payload['max_tokens'] = int(max_output_tokens)
|
|
|
|
return payload
|
|
|
|
|
|
def _build_chat_stream_payload(
|
|
prompt: str,
|
|
*,
|
|
system_prompt: str | None = None,
|
|
temperature: float = 0.7,
|
|
thinking: str | None = None,
|
|
model: str | None = None,
|
|
use_pro_model: bool = False,
|
|
prefill: str | None = None,
|
|
max_output_tokens: int | None = None,
|
|
) -> dict:
|
|
messages = []
|
|
sys_prompt = _resolve_system_prompt(system_prompt)
|
|
if sys_prompt:
|
|
messages.append({'role': 'system', 'content': sys_prompt})
|
|
messages.append({'role': 'user', 'content': prompt})
|
|
|
|
if prefill:
|
|
messages.append({'role': 'assistant', 'content': prefill})
|
|
|
|
options = {'temperature': temperature}
|
|
if thinking:
|
|
options['think'] = thinking
|
|
|
|
payload = {
|
|
'model': _resolve_model_name(model, use_pro_model=use_pro_model),
|
|
'messages': messages,
|
|
'stream': True,
|
|
'options': options,
|
|
}
|
|
if max_output_tokens and max_output_tokens > 0:
|
|
payload['max_tokens'] = int(max_output_tokens)
|
|
|
|
return payload
|
|
|
|
|
|
def _extract_delta_text(chunk: dict) -> str:
|
|
"""Extract text delta from an OpenAI-compatible SSE chunk."""
|
|
choices = chunk.get('choices', []) if isinstance(chunk, dict) else []
|
|
delta = (choices[0].get('delta', {}) if choices and isinstance(choices, list) else {}).copy()
|
|
content = delta.get('content', '') or ''
|
|
return content
|
|
|
|
|
|
def _extract_delta_thinking(chunk: dict) -> str:
|
|
"""Extract thinking/reasoning delta from an SSE chunk."""
|
|
choices = chunk.get('choices', []) if isinstance(chunk, dict) else []
|
|
delta = (choices[0].get('delta', {}) if choices and isinstance(choices, list) else {}).copy()
|
|
return (delta.get('reasoning_content') or delta.get('thinking', '') or '').strip()
|
|
|
|
|
|
async def call_ollama(
|
|
prompt: str,
|
|
*,
|
|
system_prompt: str | None = None,
|
|
tag: str = 'default',
|
|
temperature: float = 0.7,
|
|
thinking: str | None = None,
|
|
model: str | None = None,
|
|
use_pro_model: bool = False,
|
|
prefill: str | None = None,
|
|
max_output_tokens: int | None = None,
|
|
) -> dict:
|
|
"""Call OpenAI-compatible chat completions (non-streaming) and return content/thinking."""
|
|
start = time.perf_counter()
|
|
start_dt = datetime.now()
|
|
|
|
model_name = _resolve_model_name(model, use_pro_model=use_pro_model)
|
|
log_model_name = 'pro' if (model is None and use_pro_model) else model_name
|
|
|
|
logger.info(
|
|
'[LLM][%s] request model=%s base_url=%s prompt_chars=%d system_chars=%d temp=%.2f thinking=%s',
|
|
tag, log_model_name, LLM_BASE_URL, len(prompt),
|
|
len(system_prompt or ''), temperature, thinking,
|
|
)
|
|
|
|
payload = _build_chat_payload(
|
|
prompt=prompt, system_prompt=system_prompt, temperature=temperature,
|
|
thinking=thinking, model=model, use_pro_model=use_pro_model, prefill=prefill,
|
|
max_output_tokens=max_output_tokens,
|
|
)
|
|
|
|
http_timeout = httpx.Timeout(connect=10.0, read=None, write=30.0, pool=30.0)
|
|
|
|
try:
|
|
async with await _create_async_client(http_timeout) as client:
|
|
resp = await asyncio.wait_for(
|
|
_client_post(client, '/chat/completions', payload), timeout=COMPLETION_TIMEOUT,
|
|
)
|
|
|
|
resp.raise_for_status()
|
|
response = resp.json()
|
|
|
|
except asyncio.CancelledError:
|
|
elapsed_ms = (time.perf_counter() - start) * 1000
|
|
end_dt = datetime.now()
|
|
|
|
logger.info(
|
|
'[LLM][%s] call_time [%s --> %s]', tag,
|
|
start_dt.strftime('%H:%M:%S'), end_dt.strftime('%H:%M:%S'),
|
|
)
|
|
|
|
logger.warning('[LLM][%s] request cancelled after %.1fms', tag, elapsed_ms)
|
|
raise
|
|
|
|
except Exception:
|
|
elapsed_ms = (time.perf_counter() - start) * 1000
|
|
end_dt = datetime.now()
|
|
|
|
logger.info(
|
|
'[LLM][%s] call_time [%s --> %s]', tag,
|
|
start_dt.strftime('%H:%M:%S'), end_dt.strftime('%H:%M:%S'),
|
|
)
|
|
|
|
logger.exception('[LLM][%s] request failed after %.1fms', tag, elapsed_ms)
|
|
raise
|
|
|
|
content, thinking_out = _extract_message(response)
|
|
|
|
elapsed_ms = (time.perf_counter() - start) * 1000
|
|
end_dt = datetime.now()
|
|
|
|
logger.info(
|
|
'[LLM][%s] call_time [%s --> %s]', tag,
|
|
start_dt.strftime('%H:%M:%S'), end_dt.strftime('%H:%M:%S'),
|
|
)
|
|
|
|
logger.info(
|
|
'[LLM][%s] response in %.1fms content_chars=%d thinking_chars=%d',
|
|
tag, elapsed_ms, len(content), len(thinking_out or ''),
|
|
)
|
|
|
|
if not content.strip():
|
|
logger.warning('[LLM][%s] empty content returned by model', tag)
|
|
|
|
return {'content': content, 'think': thinking_out or ''}
|
|
|
|
|
|
async def stream_ollama(
|
|
prompt: str,
|
|
*,
|
|
system_prompt: str | None = None,
|
|
tag: str = 'default-stream',
|
|
temperature: float = 0.7,
|
|
thinking: str | None = None,
|
|
model: str | None = None,
|
|
use_pro_model: bool = False,
|
|
prefill: str | None = None,
|
|
max_output_tokens: int | None = None,
|
|
) -> AsyncIterator[str]:
|
|
"""Stream text deltas from OpenAI-compatible chat completions."""
|
|
start = time.perf_counter()
|
|
start_dt = datetime.now()
|
|
|
|
model_name = _resolve_model_name(model, use_pro_model=use_pro_model)
|
|
log_model_name = 'pro' if (model is None and use_pro_model) else model_name
|
|
yielded_chars = 0
|
|
|
|
logger.info(
|
|
'[LLM][%s] stream request model=%s base_url=%s prompt_chars=%d system_chars=%d temp=%.2f thinking=%s',
|
|
tag, log_model_name, LLM_BASE_URL, len(prompt),
|
|
len(system_prompt or ''), temperature, thinking,
|
|
)
|
|
|
|
payload = _build_chat_stream_payload(
|
|
prompt=prompt, system_prompt=system_prompt, temperature=temperature,
|
|
thinking=thinking, model=model, use_pro_model=use_pro_model, prefill=prefill,
|
|
max_output_tokens=max_output_tokens,
|
|
)
|
|
|
|
http_timeout = httpx.Timeout(connect=10.0, read=None, write=30.0, pool=30.0)
|
|
|
|
try:
|
|
async with await _create_async_client(http_timeout) as client:
|
|
try:
|
|
async with client.stream('POST', '/chat/completions', json=payload) as response:
|
|
await _maybe_await(response.raise_for_status())
|
|
|
|
deadline = time.perf_counter() + COMPLETION_TIMEOUT
|
|
line_iterator = await _stream_line_iterator(response)
|
|
|
|
while True:
|
|
remaining = deadline - time.perf_counter()
|
|
if remaining <= 0:
|
|
raise TimeoutError('LLM stream timed out')
|
|
|
|
try:
|
|
line = await asyncio.wait_for(line_iterator.__anext__(), timeout=remaining)
|
|
except StopAsyncIteration:
|
|
break
|
|
|
|
if not line or line.startswith(':'):
|
|
continue
|
|
|
|
# SSE data lines: "data: {json}" or "data: [DONE]"
|
|
if line.startswith('data: '):
|
|
data_str = line[6:] # strip "data: " prefix
|
|
|
|
else:
|
|
data_str = line.strip()
|
|
|
|
if not data_str or data_str == '[DONE]':
|
|
continue
|
|
|
|
try:
|
|
chunk = json.loads(data_str)
|
|
except json.JSONDecodeError:
|
|
logger.warning('[LLM][%s] ignored invalid stream line', tag)
|
|
continue
|
|
|
|
if not isinstance(chunk, dict):
|
|
continue
|
|
|
|
text = _extract_delta_text(chunk)
|
|
if not text:
|
|
continue
|
|
|
|
yielded_chars += len(text)
|
|
yield text
|
|
|
|
except asyncio.CancelledError:
|
|
if response is not None:
|
|
await response.aclose()
|
|
raise
|
|
|
|
except asyncio.CancelledError:
|
|
elapsed_ms = (time.perf_counter() - start) * 1000
|
|
end_dt = datetime.now()
|
|
|
|
logger.info(
|
|
'[LLM][%s] stream_time [%s --> %s]', tag,
|
|
start_dt.strftime('%H:%M:%S'), end_dt.strftime('%H:%M:%S'),
|
|
)
|
|
|
|
logger.warning('[LLM][%s] stream cancelled after %.1fms', tag, elapsed_ms)
|
|
raise
|
|
|
|
except Exception:
|
|
elapsed_ms = (time.perf_counter() - start) * 1000
|
|
end_dt = datetime.now()
|
|
|
|
logger.info(
|
|
'[LLM][%s] stream_time [%s --> %s]', tag,
|
|
start_dt.strftime('%H:%M:%S'), end_dt.strftime('%H:%M:%S'),
|
|
)
|
|
|
|
logger.exception('[LLM][%s] stream failed after %.1fms', tag, elapsed_ms)
|
|
raise
|
|
|
|
elapsed_ms = (time.perf_counter() - start) * 1000
|
|
end_dt = datetime.now()
|
|
|
|
logger.info(
|
|
'[LLM][%s] stream_time [%s --> %s]', tag,
|
|
start_dt.strftime('%H:%M:%S'), end_dt.strftime('%H:%M:%S'),
|
|
)
|
|
|
|
logger.info(
|
|
'[LLM][%s] stream finished in %.1fms yielded_chars=%d',
|
|
tag, elapsed_ms, yielded_chars,
|
|
)
|
|
|
|
|
|
async def stream_ollama_events(
|
|
prompt: str,
|
|
*,
|
|
system_prompt: str | None = None,
|
|
tag: str = 'default-events',
|
|
temperature: float = 0.7,
|
|
thinking: str | None = None,
|
|
model: str | None = None,
|
|
use_pro_model: bool = False,
|
|
enable_thinking: bool = True,
|
|
prefill: str | None = None,
|
|
timeout: float | None = None,
|
|
max_output_tokens: int | None = None,
|
|
) -> AsyncIterator[tuple[Literal['thinking', 'content'], str]]:
|
|
"""Stream (event_type, payload) tuples from OpenAI-compatible chat completions."""
|
|
start = time.perf_counter()
|
|
start_dt = datetime.now()
|
|
|
|
model_name = _resolve_model_name(model, use_pro_model=use_pro_model)
|
|
log_model_name = 'pro' if (model is None and use_pro_model) else model_name
|
|
yielded_chars = 0
|
|
|
|
logger.info(
|
|
'[LLM][%s] event_stream request model=%s base_url=%s prompt_chars=%d system_chars=%d temp=%.2f thinking=%s',
|
|
tag, log_model_name, LLM_BASE_URL, len(prompt),
|
|
len(system_prompt or ''), temperature, thinking,
|
|
)
|
|
|
|
payload = _build_chat_stream_payload(
|
|
prompt=prompt, system_prompt=system_prompt, temperature=temperature,
|
|
thinking=thinking if enable_thinking else None, model=model, use_pro_model=use_pro_model, prefill=prefill,
|
|
max_output_tokens=max_output_tokens,
|
|
)
|
|
|
|
effective_timeout = timeout if timeout is not None else COMPLETION_TIMEOUT
|
|
http_timeout = httpx.Timeout(connect=10.0, read=None, write=30.0, pool=30.0)
|
|
sent_thinking = False
|
|
|
|
try:
|
|
async with await _create_async_client(http_timeout) as client:
|
|
try:
|
|
async with client.stream('POST', '/chat/completions', json=payload) as response:
|
|
await _maybe_await(response.raise_for_status())
|
|
|
|
deadline = time.perf_counter() + effective_timeout
|
|
line_iterator = await _stream_line_iterator(response)
|
|
|
|
while True:
|
|
remaining = deadline - time.perf_counter()
|
|
if remaining <= 0:
|
|
raise TimeoutError('LLM event stream timed out')
|
|
|
|
try:
|
|
line = await asyncio.wait_for(line_iterator.__anext__(), timeout=remaining)
|
|
except StopAsyncIteration:
|
|
break
|
|
|
|
if not line or line.startswith(':'):
|
|
continue
|
|
|
|
# SSE data lines: "data: {json}" or "data: [DONE]"
|
|
if line.startswith('data: '):
|
|
data_str = line[6:] # strip "data: " prefix
|
|
|
|
else:
|
|
data_str = line.strip()
|
|
|
|
if not data_str or data_str == '[DONE]':
|
|
continue
|
|
|
|
try:
|
|
chunk = json.loads(data_str)
|
|
except json.JSONDecodeError:
|
|
logger.warning('[LLM][%s] ignored invalid Ollama stream line', tag)
|
|
continue
|
|
|
|
if not isinstance(chunk, dict):
|
|
continue
|
|
|
|
error = chunk.get('error')
|
|
if error:
|
|
raise RuntimeError(str(error))
|
|
|
|
thinking_delta = _extract_delta_thinking(chunk)
|
|
if thinking_delta and not sent_thinking:
|
|
sent_thinking = True
|
|
yield 'thinking', ''
|
|
|
|
text = _extract_delta_text(chunk)
|
|
if not text:
|
|
continue
|
|
|
|
yielded_chars += len(text)
|
|
yield 'content', text
|
|
|
|
except asyncio.CancelledError:
|
|
if response is not None:
|
|
await response.aclose()
|
|
raise
|
|
|
|
except asyncio.CancelledError:
|
|
elapsed_ms = (time.perf_counter() - start) * 1000
|
|
end_dt = datetime.now()
|
|
|
|
logger.info(
|
|
'[LLM][%s] event_stream_time [%s --> %s]', tag,
|
|
start_dt.strftime('%H:%M:%S'), end_dt.strftime('%H:%M:%S'),
|
|
)
|
|
|
|
logger.warning('[LLM][%s] event stream cancelled after %.1fms', tag, elapsed_ms)
|
|
raise
|
|
|
|
except Exception:
|
|
elapsed_ms = (time.perf_counter() - start) * 1000
|
|
end_dt = datetime.now()
|
|
|
|
logger.info(
|
|
'[LLM][%s] event_stream_time [%s --> %s]', tag,
|
|
start_dt.strftime('%H:%M:%S'), end_dt.strftime('%H:%M:%S'),
|
|
)
|
|
|
|
logger.exception('[LLM][%s] event stream failed after %.1fms', tag, elapsed_ms)
|
|
raise
|
|
|
|
elapsed_ms = (time.perf_counter() - start) * 1000
|
|
end_dt = datetime.now()
|
|
|
|
logger.info(
|
|
'[LLM][%s] event_stream_time [%s --> %s]', tag,
|
|
start_dt.strftime('%H:%M:%S'), end_dt.strftime('%H:%M:%S'),
|
|
)
|
|
|
|
logger.info(
|
|
'[LLM][%s] event stream finished in %.1fms yielded_chars=%d thinking_seen=%s',
|
|
tag, elapsed_ms, yielded_chars, sent_thinking,
|
|
)
|
|
|
|
|
|
async def call_vlm_ocr(image_bytes: bytes, language: str = 'auto') -> str:
|
|
"""OCR via VLM using OpenAI-compatible vision API (image_url content part)."""
|
|
start = time.perf_counter()
|
|
start_dt = datetime.now()
|
|
|
|
logger.info(
|
|
'[VLM][ocr] request model=%s base_url=%s image_bytes=%d language=%s',
|
|
VLM_MODEL, LLM_BASE_URL, len(image_bytes), language,
|
|
)
|
|
|
|
image_b64 = base64.b64encode(image_bytes).decode('ascii')
|
|
|
|
payload = {
|
|
'model': VLM_MODEL,
|
|
'messages': [{
|
|
'role': 'user',
|
|
'content': [
|
|
{'type': 'text', 'text': get_vlm_ocr_prompt()},
|
|
{
|
|
'type': 'image_url',
|
|
'image_url': {'url': f'data:image/png;base64,{image_b64}'},
|
|
},
|
|
],
|
|
}],
|
|
'stream': False,
|
|
}
|
|
|
|
http_timeout = httpx.Timeout(connect=10.0, read=None, write=30.0, pool=30.0)
|
|
|
|
try:
|
|
async with await _create_async_client(http_timeout) as client:
|
|
resp = await asyncio.wait_for(
|
|
_client_post(client, '/chat/completions', payload), timeout=OCR_TIMEOUT,
|
|
)
|
|
|
|
resp.raise_for_status()
|
|
response = resp.json()
|
|
|
|
except Exception:
|
|
elapsed_ms = (time.perf_counter() - start) * 1000
|
|
end_dt = datetime.now()
|
|
|
|
logger.info(
|
|
'[VLM][ocr] call_time [%s --> %s]', start_dt.strftime('%H:%M:%S'),
|
|
end_dt.strftime('%H:%M:%S'),
|
|
)
|
|
|
|
logger.exception('[VLM][ocr] request failed after %.1fms', elapsed_ms)
|
|
raise
|
|
|
|
content, _ = _extract_message(response)
|
|
|
|
elapsed_ms = (time.perf_counter() - start) * 1000
|
|
end_dt = datetime.now()
|
|
|
|
logger.info(
|
|
'[VLM][ocr] call_time [%s --> %s]', start_dt.strftime('%H:%M:%S'),
|
|
end_dt.strftime('%H:%M:%S'),
|
|
)
|
|
|
|
logger.info(
|
|
'[VLM][ocr] response in %.1fms content_chars=%d', elapsed_ms, len(content),
|
|
)
|
|
|
|
if not content.strip():
|
|
logger.warning('[VLM][ocr] empty content returned by model')
|
|
|
|
return content
|