惯性聚合 高效追踪和阅读你感兴趣的博客、新闻、科技资讯
阅读原文 在惯性聚合中打开

推荐订阅源

P
Proofpoint News Feed
V
V2EX
WordPress大学
WordPress大学
Google DeepMind News
Google DeepMind News
Martin Fowler
Martin Fowler
小众软件
小众软件
Blog — PlanetScale
Blog — PlanetScale
月光博客
月光博客
The Cloudflare Blog
T
Tailwind CSS Blog
H
Help Net Security
腾讯CDC
爱范儿
爱范儿
人人都是产品经理
人人都是产品经理
H
Hackread – Cybersecurity News, Data Breaches, AI and More
The GitHub Blog
The GitHub Blog
Microsoft Security Blog
Microsoft Security Blog
Stack Overflow Blog
Stack Overflow Blog
D
DataBreaches.Net
C
Check Point Blog
量子位
酷 壳 – CoolShell
酷 壳 – CoolShell
美团技术团队
让小产品的独立变现更简单 - ezindie.com
让小产品的独立变现更简单 - ezindie.com

Recent Commits to openclaw:main

test: merge chat side-result checks · openclaw/openclaw@ddd2c2a test: merge cron history checks · openclaw/openclaw@f7eb746 test: merge responsive navigation shell checks · openclaw/openclaw@c2e4b47 docs(changelog): add codex oauth fixes · openclaw/openclaw@628e6cd test: merge navigation routing cases · openclaw/openclaw@5d8cecb Tests: mock channel registry bundled fallback · openclaw/openclaw@2b08233 Secrets: avoid broad web search discovery for single plugin config · openclaw/openclaw@a464f59 test: merge config view browser checks · openclaw/openclaw@20cf511 fix(status): align oauth health with runtime · openclaw/openclaw@eed7116 feat: add macOS screen snapshots for monitor preview (#67954) thanks … · openclaw/openclaw@f377db1 fix: report shared auth scopes in hello-ok (#67810) thanks @BunsDev · openclaw/openclaw@0b6c39b Auto-reply: avoid eager bundled route fallback · openclaw/openclaw@3ea1bf4 Tests: narrow session binding contract setup · openclaw/openclaw@54e4e16 fix(macOS): enable undo/redo in webchat composer text input (#34962) · openclaw/openclaw@00951dc Tests: speed up channel setup promotion · openclaw/openclaw@82b529a Docs: refresh agent instructions · openclaw/openclaw@5775fe2 fix(auth): serialize OAuth refresh across agents to fix #26322 (#67876) · openclaw/openclaw@8e79080 test: allow ollama public surface boundary test · openclaw/openclaw@7d4f1a6 Docs: add test performance guardrails · openclaw/openclaw@89706d3 Tests: restore context-engine usage proof · openclaw/openclaw@e4c4f95 Tests: slim context engine runtime coverage · openclaw/openclaw@74c198f ci: retry failed custom checkouts · openclaw/openclaw@0ee5baf test: trim duplicate provider auth onboarding cases · openclaw/openclaw@1ffc02e matrix: fix sessions_spawn --thread subagent session spawning (#67643) · openclaw/openclaw@1ce2596 test: reduce auth choice fixture churn · openclaw/openclaw@857b9cd test: mock health status config boundaries · openclaw/openclaw@9d5ab4a test: mock onboard config io boundary · openclaw/openclaw@299694d test: mock legacy state plugin boundaries · openclaw/openclaw@2713089 test: mock channel install boundaries · openclaw/openclaw@b945248 test: mock doctor preview channel boundaries · openclaw/openclaw@b1a3ad4
test: tighten qa character eval assertions · openclaw/ope...
steipete · 2026-05-11 · via Recent Commits to openclaw:main

@@ -112,6 +112,32 @@ function makeSuiteResult(params: { outputDir: string; model: string; transcript:

112112

} satisfies QaSuiteResult;

113113

}

114114115+

function requireRunSuiteParams(runSuite: ReturnType<typeof vi.fn>, index = 0) {

116+

const params = runSuite.mock.calls[index]?.[0] as CharacterRunSuiteParams | undefined;

117+

if (!params) {

118+

throw new Error(`runSuite call ${index} missing`);

119+

}

120+

return params;

121+

}

122+123+

function requireRunJudgeParams(runJudge: ReturnType<typeof vi.fn>, index = 0) {

124+

const params = runJudge.mock.calls[index]?.[0] as CharacterRunJudgeParams | undefined;

125+

if (!params) {

126+

throw new Error(`runJudge call ${index} missing`);

127+

}

128+

return params;

129+

}

130+131+

function expectFirstRunFailure(

132+

result: Awaited<ReturnType<typeof runQaCharacterEval>>,

133+

expected: { model: string; error: string },

134+

) {

135+

const run = result.runs[0];

136+

expect(run?.model).toBe(expected.model);

137+

expect(run?.status).toBe("fail");

138+

expect(run?.error).toBe(expected.error);

139+

}

140+115141

describe("runQaCharacterEval", () => {

116142

let tempRoot: string;

117143

@@ -160,24 +186,17 @@ describe("runQaCharacterEval", () => {

160186

});

161187162188

expect(runSuite).toHaveBeenCalledTimes(2);

163-

expect(runSuite).toHaveBeenNthCalledWith(

164-

1,

165-

expect.objectContaining({

166-

providerMode: "live-frontier",

167-

primaryModel: "openai/gpt-5.5",

168-

alternateModel: "openai/gpt-5.5",

169-

fastMode: true,

170-

scenarioIds: ["character-vibes-gollum"],

171-

}),

172-

);

173-

expect(runJudge).toHaveBeenCalledWith(

174-

expect.objectContaining({

175-

judgeModel: "openai/gpt-5.5",

176-

judgeThinkingDefault: "xhigh",

177-

judgeFastMode: true,

178-

timeoutMs: 300_000,

179-

}),

180-

);

189+

const firstRunParams = requireRunSuiteParams(runSuite);

190+

expect(firstRunParams.providerMode).toBe("live-frontier");

191+

expect(firstRunParams.primaryModel).toBe("openai/gpt-5.5");

192+

expect(firstRunParams.alternateModel).toBe("openai/gpt-5.5");

193+

expect(firstRunParams.fastMode).toBe(true);

194+

expect(firstRunParams.scenarioIds).toEqual(["character-vibes-gollum"]);

195+

const judgeParams = requireRunJudgeParams(runJudge);

196+

expect(judgeParams.judgeModel).toBe("openai/gpt-5.5");

197+

expect(judgeParams.judgeThinkingDefault).toBe("xhigh");

198+

expect(judgeParams.judgeFastMode).toBe(true);

199+

expect(judgeParams.timeoutMs).toBe(300_000);

181200

expect(result.judgments).toHaveLength(1);

182201

expect(result.judgments[0]?.rankings.map((ranking) => ranking.model)).toEqual([

183202

"openai/gpt-5.5",

@@ -404,9 +423,8 @@ describe("runQaCharacterEval", () => {

404423

runJudge,

405424

});

406425407-

expect(result.runs[0]).toMatchObject({

426+

expectFirstRunFailure(result, {

408427

model: "qwen/qwen3.6-plus",

409-

status: "fail",

410428

error: "model unsupported error leaked into transcript",

411429

});

412430

});

@@ -432,9 +450,8 @@ describe("runQaCharacterEval", () => {

432450

runJudge,

433451

});

434452435-

expect(result.runs[0]).toMatchObject({

453+

expectFirstRunFailure(result, {

436454

model: "qwen/qwen3.5-plus",

437-

status: "fail",

438455

error: "tool failure leaked into transcript",

439456

});

440457

});

@@ -461,9 +478,8 @@ describe("runQaCharacterEval", () => {

461478

runJudge,

462479

});

463480464-

expect(result.runs[0]).toMatchObject({

481+

expectFirstRunFailure(result, {

465482

model: "qa/generic-fallback-model",

466-

status: "fail",

467483

error: "generic request failure leaked into transcript",

468484

});

469485

});

@@ -490,9 +506,8 @@ describe("runQaCharacterEval", () => {

490506

runJudge,

491507

});

492508493-

expect(result.runs[0]).toMatchObject({

509+

expectFirstRunFailure(result, {

494510

model: "google/gemini-test",

495-

status: "fail",

496511

error: "LLM timeout leaked into transcript",

497512

});

498513

});

@@ -519,9 +534,8 @@ describe("runQaCharacterEval", () => {

519534

runJudge,

520535

});

521536522-

expect(result.runs[0]).toMatchObject({

537+

expectFirstRunFailure(result, {

523538

model: "codex/gpt-5.5",

524-

status: "fail",

525539

error: "internal harness/meta text leaked into transcript",

526540

});

527541

});