5a26dfde2a
后端变更: - 新增 risk_config.py: 风险配置数据类,支持环境变量驱动 - 新增 risk_control.py: 风险控制控制器,管理并发和预算 - 新增 session_store.py: 匿名会话存储,基于 cookie 的 session ID - 新增 audit_store.py: API 审计日志存储,记录请求和 LLM 调用 - 新增 captcha_api.py: 验证码 API,用于验证用户操作真实性 - 新增 llm_policy.py: LLM 策略配置,管理 completion/pro/vision 模型 - main.py: 集成 middleware、risk/audit/session 模块 (+467/-7) - job_handlers.py: LLM 执行流程重构,新增 risk/audit 集成 (+207/-4) - llm.py: 异步客户端封装,新增 max_output_tokens 参数 (+78/-1) - job_system.py: stream_events 逻辑优化,支持心跳检测 (+12/-4) - pro_completions.py: SSE heartbeat 机制,防止连接超时 (+14/-4) - prompt.py: _normalize_preferences 支持 Mapping 类型 (+13/-0) - tts_asr.py: asyncio loop 初始化,router export (+10/-0) 前端变更: - src/components/CaptchaComponent.vue: 新增验证码组件 (NEW) - src/utils/cookie_policy.js: Cookie 策略工具 (NEW) - SettingsPanel.vue: 集成验证码组件,新增安全设置部分 (+59/-0) - MilkdownEditor.vue: 移除硬编码 API_KEY,新增 credentials (+32/-10) - ProBlockCrepe.vue: 样式简化,移除渐变动画 (+18/-4) - proBlockPlugin.ts: 重构 schema/serializer 引用方式,通过 Ctx 管理 (+40/-10) - api.js: 新增 credentials,重构 headers 条件逻辑 (+50/-14) - config.js: API 基址改为 https://api.imageteach.tech:8002 (+8/-4) - convert.js, docsApi.js, i18n.js: 新增 credentials 和验证码 i18n (+54/-12) - proAccept.js: 重构正则和转义处理,修复捕获组索引 (+14/-4) 配置和基础设施: - docker-compose.yml: 新增端口映射 8001:8001 (+2/-0) - docker/nginx.conf: 改为 307 redirect,优化代理配置 (+8/-6) - vite.config.js: 移除 proxy 配置,直接调用远程 API (+8/-4) - .env.example: 新增 VITE_API_BASE_URL, VITE_API_KEY (+3/-1) - backend/.env.example: 大量 RISK_*, SESSION_*, CORS_* 配置 (+54/-0) - pytest.ini: 扩展 coverage 范围到整个 backend,移除 fail_under (+3/-2) - .coveragerc: 移除 fail_under = 90 (+0/-1) - .gitignore: 新增 docker-data/ (+3/-0) - package.json: 新增 vue3-captcha 依赖 (+3/-1) - AGENTS.md, README.md: 更新 Docker 部署和前端网络约定 (+20/-5) - public/sw.js: Service Worker cache 版本从 v1 升级到 v2 (+0/-1) 测试变更: - test_main_endpoints.py: 新增 session/risk/audit reset,新增测试用例 (+63/-4) - test_main_cancel.py: 新增 reset 调用 (+6/-0) - test_pro_completions.py: 新增 preferences 序列化和测试 (+23/-0) 总计: 45 个文件变更,+1009/-280 行
71 lines
2.4 KiB
Python
71 lines
2.4 KiB
Python
from dataclasses import dataclass
|
|
from typing import Any
|
|
|
|
from risk_config import RiskConfig
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class LLMPolicy:
|
|
job_type: str
|
|
model: str
|
|
profile: str
|
|
max_input_chars: int
|
|
max_output_tokens: int
|
|
temperature: float
|
|
thinking: str | None
|
|
|
|
|
|
def _normalize_thinking(value: str | None, *, allow_high: bool) -> str | None:
|
|
candidate = (value or "").strip().lower()
|
|
if candidate in {"", "none", "off"}:
|
|
return None
|
|
if candidate not in {"low", "medium", "high"}:
|
|
return "low"
|
|
if candidate == "high" and not allow_high:
|
|
return "medium"
|
|
return candidate
|
|
|
|
|
|
def resolve_llm_policy(job_type: str, request_payload: dict[str, Any], config: RiskConfig) -> LLMPolicy:
|
|
if job_type == "completion":
|
|
return LLMPolicy(
|
|
job_type=job_type,
|
|
model=config.completion_model,
|
|
profile="completion",
|
|
max_input_chars=config.completion_max_input_chars,
|
|
max_output_tokens=config.completion_max_output_tokens,
|
|
temperature=config.completion_temperature,
|
|
thinking=_normalize_thinking(request_payload.get("model_thinking"), allow_high=False),
|
|
)
|
|
if job_type == "pro_completion":
|
|
return LLMPolicy(
|
|
job_type=job_type,
|
|
model=config.pro_model,
|
|
profile="pro",
|
|
max_input_chars=config.pro_max_input_chars,
|
|
max_output_tokens=config.pro_max_output_tokens,
|
|
temperature=config.pro_temperature,
|
|
thinking=_normalize_thinking(request_payload.get("pro_thinking"), allow_high=True) or "medium",
|
|
)
|
|
if job_type == "compress":
|
|
return LLMPolicy(
|
|
job_type=job_type,
|
|
model=config.completion_model,
|
|
profile="completion",
|
|
max_input_chars=config.compress_max_input_chars,
|
|
max_output_tokens=config.compress_max_output_tokens,
|
|
temperature=0.2,
|
|
thinking="low",
|
|
)
|
|
if job_type == "ocr":
|
|
return LLMPolicy(
|
|
job_type=job_type,
|
|
model=config.vision_model,
|
|
profile="vision",
|
|
max_input_chars=config.ocr_max_input_bytes,
|
|
max_output_tokens=config.completion_max_output_tokens,
|
|
temperature=0.0,
|
|
thinking=None,
|
|
)
|
|
raise ValueError(f"unsupported llm policy job type: {job_type}")
|