Jev Recipe / 厂商对比
Jev vs Perplexity Decisions API:GA 落地、定价公布后的正面比较
Perplexity 用一个已 GA、已公布定价的决策 API 回应了 Jev。逐格核验的契约对比:noul/choice/score 三原语同名、每百万输入 token $0.04 对 $0.042、Apache 2.0 开源权重、视觉输入——以及真正决定选型的那些差异。
选定决策 API 之前的 6 项检查
{
"routing": {
"type": "choice",
"instructions": "Which queue should this ticket go to?",
"criteria": {
"billing": "Payment, invoice or refund issue",
"technical": "Product malfunction or bug",
"sales": "Buying or upgrade question",
"abuse": "Safety or abuse report"
}
},
"needs_safety_escalation": {
"type": "noul",
"instructions": "Does this ticket require a safety escalation regardless of queue?"
},
"urgency": {
"type": "score",
"instructions": "Rate how urgent a human reply is, 1 (routine) to 5 (business-stopping)"
}
}决策 API 记分牌:TypeSafe Jev vs Perplexity Decisions API(GA 发布周,逐格标注来源)
代码对比:Jev 类型化调用 vs Perplexity decisions 端点与 A/B 切换
import requests
JEV_ENDPOINT = "https://api.typesafe.ai/v1/jev/evaluate"
AUTO_ROUTE_CONFIDENCE = 0.85
QUESTIONS = {
"routing": {
"type": "choice",
"instructions": "Which queue should this ticket go to?",
"criteria": {
"billing": "Payment, invoice or refund issue",
"technical": "Product malfunction or bug",
"sales": "Buying or upgrade question",
"abuse": "Safety or abuse report",
},
},
"needs_safety_escalation": {
"type": "noul",
"instructions": "Does this ticket require a safety escalation regardless of queue?",
},
"urgency": {
"type": "score",
"instructions": "Rate how urgent a human reply is, 1 (routine) to 5 (business-stopping)",
},
}
resp = requests.post(
JEV_ENDPOINT,
json={"state": {"ticket": "..."}, "questions": QUESTIONS},
timeout=5,
)
resp.raise_for_status()
data = resp.json()
route = data["routing"] # 类型化答案 + 校准置信度,零生成
if (
route["confidence"] >= AUTO_ROUTE_CONFIDENCE
and not data["needs_safety_escalation"]["answer"]
):
lane = f"auto:{route['answer']}" # 单次前向,仅输入计费
else:
lane = "review" # 0.60~0.85 带宽或安全标记import os
import requests
# POST /v1/decisions - 末尾不能带斜杠(带斜杠返回 404)。
# 鉴权只用 Bearer:放在 x-api-key 里的 key 不会被读取,返回 401。
PPLX_ENDPOINT = "https://api.perplexity.ai/v1/decisions"
AUTO_ROUTE_CONFIDENCE = 0.85
resp = requests.post(
PPLX_ENDPOINT,
headers={"Authorization": f"Bearer {os.environ['PERPLEXITY_API_KEY']}"},
json={
# model 是每次请求的必填字段;缺失或未知返回 400
"model": "pplx-decider-v1-27b",
"state": {
"ticket": "Checkout has been failing for every customer "
"for the last hour."
},
"questions": {
"routing": {
"type": "choice",
"instructions": "Which queue should this ticket go to?",
"criteria": {
"billing": "Payment, invoice or refund issue",
"technical": "Product malfunction or bug",
"sales": "Buying or upgrade question",
"abuse": "Safety or abuse report",
},
},
"needs_safety_escalation": {
"type": "noul",
"instructions": "Does this ticket require a safety "
"escalation regardless of queue?",
},
"urgency": {
"type": "score",
"instructions": "Rate how urgent a human reply is, "
"routine to business-stopping",
# score:1~10 个有序档位,答案是档位索引(0 起)的
# 概率加权平均,可能落在两档之间
"criteria": ["Routine", "Minor", "Elevated", "Urgent",
"Business-stopping"],
},
},
},
timeout=30, # 文档口径:小请求 2 秒内;30 秒覆盖输入上限
)
resp.raise_for_status()
data = resp.json()
answers = data["answers"] # 按问题名取回,类型与问题一致
route = answers["routing"] # choice + 置信度 + 各选项概率
if (
route["confidence"] >= AUTO_ROUTE_CONFIDENCE
and answers["needs_safety_escalation"]["noul"] < 0.5 # P(yes)
):
lane = f"auto:{route['choice']}"
else:
lane = "review" # 低置信度或安全标记
# 官方文档里的诚实边界注记:
# - confidence 是「模型自己的确信估计」,不是最高概率;
# 次优选项逼近时会下降。
# - 同一请求通常返回相同数字,但可能在第二位小数上不同——
# 自动化之前先验证。
# - usage.input_tokens 按 $0.04/M 计费;output_tokens 免费。
# - 限速:全部套餐 10 请求/秒(429 带 Retry-After)。const JEV_ENDPOINT = "https://api.typesafe.ai/v1/jev/evaluate";
const AUTO_ROUTE_CONFIDENCE = 0.85;
const QUESTIONS = {
routing: {
type: "choice",
instructions: "Which queue should this ticket go to?",
criteria: {
billing: "Payment, invoice or refund issue",
technical: "Product malfunction or bug",
sales: "Buying or upgrade question",
abuse: "Safety or abuse report",
},
},
needs_safety_escalation: {
type: "noul",
instructions:
"Does this ticket require a safety escalation regardless of queue?",
},
urgency: {
type: "score",
instructions:
"Rate how urgent a human reply is, 1 (routine) to 5 (business-stopping)",
},
} as const;
const resp = await fetch(JEV_ENDPOINT, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ state: { ticket: "..." }, questions: QUESTIONS }),
});
const data = await resp.json();
const lane =
data.routing.confidence >= AUTO_ROUTE_CONFIDENCE &&
!data.needs_safety_escalation.answer
? `auto:${data.routing.answer}`
: "review";// 原语同名,契约不同——把两种形状适配进同一个接口。这个适配器
// 吸收的差异:Perplexity 每次调用必填 model: "pplx-decider-v1-27b"
// (否则 400)、答案嵌在 `answers` 下、只用 Bearer 鉴权、拒绝未知
// 字段与末尾斜杠,且全部套餐限速 10 请求/秒。它的 confidence 是
// 「模型自己的确信估计」而非最高概率——自动化之前按厂商逐个
// 验证校准。
const JEV_ENDPOINT = "https://api.typesafe.ai/v1/jev/evaluate";
const PPLX_ENDPOINT = "https://api.perplexity.ai/v1/decisions";
const QUESTIONS = {
routing: {
type: "choice",
instructions: "Which queue should this ticket go to?",
criteria: {
billing: "Payment, invoice or refund issue",
technical: "Product malfunction or bug",
sales: "Buying or upgrade question",
abuse: "Safety or abuse report",
},
},
} as const;
type TypedAnswer = { answer: string; confidence: number };
export async function routeTicket(ticket: string): Promise<TypedAnswer> {
if (process.env.DECISION_BACKEND === "perplexity") {
const resp = await fetch(PPLX_ENDPOINT, {
method: "POST",
headers: {
Authorization: `Bearer ${process.env.PERPLEXITY_API_KEY!}`,
"Content-Type": "application/json",
},
body: JSON.stringify({
model: "pplx-decider-v1-27b", // 每次请求必填
state: { ticket },
questions: QUESTIONS,
}),
});
if (!resp.ok) throw new Error("decision call failed");
const data = await resp.json();
const a = data.answers.routing as {
choice: string;
confidence: number;
probabilities: Record<string, number>;
};
return { answer: a.choice, confidence: a.confidence };
}
const resp = await fetch(JEV_ENDPOINT, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ state: { ticket }, questions: QUESTIONS }),
});
if (!resp.ok) throw new Error("decision call failed");
const data = await resp.json();
return data.routing as TypedAnswer;
}Jev vs Perplexity Decisions API 常见问题
Perplexity Decisions API 是什么?
Perplexity 的 GA 决策模型 API,发布周官宣(2026-10-05),本站于 2026-10-06 对照官方文档核验:发送一个 state(字符串、对象或数组——可以带图)加 1~128 个命名问题,拿回概率而不是生成的文本。只有一个模型服务它:pplx-decider-v1-27b;三种问题类型与 Jev 原语完全同名:noul(「是」的概率)、choice(各选项概率加置信度)、score(最多 10 档评分量表上的概率加权平均)。定价为每百万输入 token $0.04,输出 token 免费、无单次请求费;权重以 Apache 2.0 上架 Hugging Face。
它和 Jev 有什么不同?
同一品类、同名原语、不同的契约细节。Jev 返回类型化答案加 RLCD 校准置信度,单次前向、行为确定;Perplexity 的 confidence 被文档定义为「模型自己的确信估计」(不是最高概率),且同一请求可能在第二位小数上不同。Perplexity 支持图像输入、输入上限 26.2 万 token;Jev 只吃文本、上下文 32k,但公开了校准方法论(ECE,见本站基准页)并配有 SDK 与 Playground。两家都仅按输入计费、输出免费——Jev $0.042/M,Perplexity $0.04/M。
有没有 Perplexity Decisions API 的开源替代品?
Perplexity 自己就是一个:权重(Hugging Face 上的 perplexity-ai/pplx-decider-v1-27b)以 Apache 2.0 开放、附官方 Python 推理代码,可以在你自己的基础设施上跑推理——服务栈自己搭、验证归你。Jev 生态的本地路线是 OpenJev 家族克隆(Kev、SemIf、Von)跑在自己的硬件上;Cloudflare 的 Clef 则是另一个开源权重决策模型家族。诚实的保留意见:自行托管复现的是权重,不必然复现托管 API 的全部行为——图像 tile 处理、置信度语义、你机器上的校准,都要自己验证。
该选谁?
按场景选。Perplexity 优先:决策今天就要读图、需要超过 32k 的上下文、想要带厂商认可自托管路径的开源权重,或极端规模下的输入成本是决定线($0.04 对 $0.042)。Jev 优先:你要在校准置信度上做自动化、需要公开的校准方法论、需要可重放审计的确定性行为、需要 SDK 加 Playground 的工作流,或依赖审计追踪与降级链内容栈。无论哪边,拍板的数据都在本地:让约 100 条你自己的标注样本过一遍两边的闸门——发布周的厂商文档回答的是选型侦察问题,不是部署问题。
$0.04 对 $0.042 的差价是真正的成本故事吗?
不是——正常量级下两家都是几美分的事。600 个输入 token 在 Perplexity 约 $0.000024、在 Jev 约 $0.000025;5% 的差距只在月 token 量到九位数时才有意义。真正花钱的轴是:图像(Perplexity 每百万像素约 1,000 输入 token)、输出计费(两家都免费——这正是它们与 OpenAI gpt-6-luna $0.10 输入 / $0.50 输出的分界)、限速(Perplexity 全部套餐 10 请求/秒)。用成本计算器把三角算一遍,然后别再抠这几分钱,开始验证置信度。
两个 API 能共用一条请求路径吗?
加适配器可以,直接替换不行。原语名字与「state 加 questions」的哲学一致,但 Perplexity 每次调用必填 model: "pplx-decider-v1-27b"、响应嵌在 answers 下、只用 Bearer 头鉴权(x-api-key 不会被读取)、拒绝未知字段与末尾斜杠,且全部套餐限速 10 请求/秒。像上面的双后端代码一样把两者放进同一个决策接口,置信度语义按厂商分开对待,自动化之前先在标注样本上验证切换。