Diff
1diff --git a/src/core/providers.ts b/src/core/providers.ts
2index 206120b5d63ee0d5de2c26384c0685b99bc582e2..63628af8619201321eb93ee45d6329ee8415746b 100644
3--- a/src/core/providers.ts
4+++ b/src/core/providers.ts
5@@ -11,7 +11,6 @@ type LlamaResponse = {
6 choices?: {
7 message?: {
8 content?: unknown
9- reasoning_content?: unknown
10 }
11 }[]
12 }
13@@ -22,7 +21,7 @@ function textValue(value: unknown): string | undefined {
14
15 function llamaResponseText(data: LlamaResponse): string | undefined {
16 const choice = data.choices?.[0]
17- return textValue(choice?.message?.content) ?? textValue(choice?.message?.reasoning_content)
18+ return textValue(choice?.message?.content)
19 }
20
21 function completionUrl(baseUrl: string): string {
22diff --git a/test/core/llm.test.ts b/test/core/llm.test.ts
23index 62cab88dd17829d45a7d598bc82fdaf2540312b7..de8857d020bb3adaf47a1ba490ac27dc0b6dc515 100644
24--- a/test/core/llm.test.ts
25+++ b/test/core/llm.test.ts
26@@ -1,6 +1,6 @@
27 import { describe, expect, test } from "bun:test";
28 import { createPolicyReviewer } from "../../src/core/review";
29-import type { Decision } from "../../src/core/types";
30+import type { Decision, LLMEvaluationResult } from "../../src/core/types";
31
32 const baseUrl = process.env.POLICY_EVAL_BASE_URL?.trim();
33 const model = process.env.POLICY_EVAL_MODEL?.trim();
34@@ -91,18 +91,22 @@ EOF
35 ];
36
37 async function assertDecision(command: string, expected: ExpectedDecision) {
38- const decisions: Decision[] = [];
39+ const results: LLMEvaluationResult[] = [];
40 for (let run = 0; run < RUNS; run++) {
41- const result = await reviewer!.evaluate(
42+ results.push(await reviewer!.evaluate(
43 { toolName: "bash", input: { command } },
44 {},
45- );
46- decisions.push(result.decision);
47+ ));
48 }
49
50- const summary = decisions
51- .map(({ decision, reason }) => `${decision}: ${reason}`)
52- .join("\n");
53+ const decisions: Decision[] = results.map(({ decision }) => decision);
54+ const summary = results
55+ .map(({ decision, rawResponse, error }) => [
56+ `${decision.decision}: ${decision.reason}`,
57+ `raw response: ${rawResponse ?? "<none>"}`,
58+ ...(error ? [`error: ${error}`] : []),
59+ ].join("\n"))
60+ .join("\n\n");
61 const tooLenient = decisions.filter(
62 ({ decision }) => STRICTNESS[decision] < STRICTNESS[expected],
63 );