298373eb7affc02aa34895b89f3200337db2df17

Author
TheEdgeOfRage <git@theedgeofrage.com>
Committer
TheEdgeOfRage <git@theedgeofrage.com>
Date

Message

Configure llama reasoning format

Diff

This diff is truncated to protect this page.

  1diff --git a/AGENTS.md b/AGENTS.md
  2index d62d5524f857314d2ee819d3bef363a36da97863..dcc930af870fd75c05fc7e602dcacfa914302aa5 100644
  3--- a/AGENTS.md
  4+++ b/AGENTS.md
  5@@ -23,6 +23,10 @@ The shared pipeline evaluates deterministic rules, then the cache, then the LLM
  6 
  7 MCP tools bypass the Pi adapter policy boundary.
  8 
  9+## Testing
 10+
 11+Test policy behavior only: static permissions, deterministic rules, and LLM decisions. Do not add tests for configuration, providers, or other plumbing unless the user asks.
 12+
 13 ## Rules
 14 
 15 `src/core/rules.ts` contains deterministic Bash patterns and the LLM policy prompt. `src/core/deterministic.ts` applies the rules.
 16diff --git a/README.md b/README.md
 17index f07ed487fd47473aa54f20c41f912ba8da2e7e0f..e138bc28ec3ae0def2aa4aa23ecf726e87dadc88 100644
 18--- a/README.md
 19+++ b/README.md
 20@@ -52,7 +52,9 @@ Use local llama.cpp:
 21 	"reviewer": {
 22 		"kind": "llama.cpp",
 23 		"baseUrl": "http://127.0.0.1:9931",
 24-		"model": "reviewer"
 25+		"model": "reviewer",
 26+		"enableThinking": true,
 27+		"reasoningFormat": "deepseek"
 28 	}
 29 }
 30 ```
 31@@ -70,6 +72,8 @@ Use an OpenAI-compatible endpoint. `apiKeyEnv` is optional. The endpoint receive
 32 }
 33 ```
 34 
 35diff --git a/src/core/config.ts b/src/core/config.ts
 36index 6eaf64580d593692cfbae0c6504a9926b31c0047..11df1dd0c872430571e9ab6a3a6f586500e0ef54 100644
 37--- a/src/core/config.ts
 38+++ b/src/core/config.ts
 39@@ -1,4 +1,4 @@
 40-import type { ReviewerConfig } from "./types"
 41+import type { LlamaReasoningFormat, ReviewerConfig } from "./types"
 42 
 43 const DEFAULT_OPENAI_MODEL = "gpt-5.4-nano"
 44 const DEFAULT_LLAMA_BASE_URL = "http://127.0.0.1:9931"
 45@@ -23,6 +23,18 @@ function apiKeyEnvironment(value: unknown, field: string): string | undefined {
 46   return name
 47 }
 48 
 49+function llamaReasoningFormat(value: unknown): LlamaReasoningFormat {
 50+  if (value === undefined) return "none"
 51+  if (value === "none" || value === "deepseek" || value === "deepseek-legacy") return value
 52+  throw new Error('reviewer.reasoningFormat must be "none", "deepseek", or "deepseek-legacy"')
 53+}
 54+
 55+function booleanValue(value: unknown, field: string, fallback: boolean): boolean {
 56+  if (value === undefined) return fallback
 57+  if (typeof value === "boolean") return value
 58+  throw new Error(`${field} must be a boolean`)
 59+}
 60+
 61 function openAIConfig(value: Record<string, unknown>): ReviewerConfig {
 62   return {
 63     kind: "openai",
 64@@ -35,6 +47,8 @@ function llamaConfig(value: Record<string, unknown>): ReviewerConfig {
 65     kind: "llama.cpp",
 66     baseUrl: stringValue(value.baseUrl) ?? DEFAULT_LLAMA_BASE_URL,
 67     model: stringValue(value.model) ?? DEFAULT_LLAMA_MODEL,
 68+    enableThinking: booleanValue(value.enableThinking, "reviewer.enableThinking", false),
 69+    reasoningFormat: llamaReasoningFormat(value.reasoningFormat),
 70   }
 71 }
 72 
 73diff --git a/src/core/providers.ts b/src/core/providers.ts
 74index 1f8b5d2e162987b3ad9e63d5edd35ea33e3761cc..206120b5d63ee0d5de2c26384c0685b99bc582e2 100644
 75--- a/src/core/providers.ts
 76+++ b/src/core/providers.ts
 77@@ -65,6 +65,8 @@ async function callLlama(
 78   model: string,
 79   messages: ChatMessage[],
 80   maxTokens: number,
 81+  enableThinking: boolean,
 82+  reasoningFormat: "none" | "deepseek" | "deepseek-legacy",
 83 ): Promise<string> {
 84   const response = await fetch(completionUrl(baseUrl), {
 85     method: "POST",
 86@@ -76,8 +78,8 @@ async function callLlama(
 87       stream: false,
 88       temperature: 0,
 89       response_format: { type: "json_object" },
 90-      chat_template_kwargs: { enable_thinking: false },
 91-      reasoning_format: "none",
 92+      chat_template_kwargs: { enable_thinking: enableThinking },
 93+      reasoning_format: reasoningFormat,
 94     }),
 95   })
 96   if (!response.ok) throw new Error(`llama-server API ${response.status}`)
 97@@ -94,7 +96,14 @@ export async function callReviewer(
 98   resolveApiKey: ApiKeyResolver,
 99 ): Promise<string> {
100   if (config.kind === "llama.cpp") {
101-    return callLlama(config.baseUrl, config.model, messages, maxTokens)
102+    return callLlama(
103+      config.baseUrl,
104+      config.model,
105+      messages,
106+      maxTokens,
107+      config.enableThinking,
108+      config.reasoningFormat,
109+    )
110   }
111   if (config.kind === "openai-compatible") {
112     const apiKey = config.apiKeyEnv ? process.env[config.apiKeyEnv]?.trim() : undefined
113diff --git a/src/core/types.ts b/src/core/types.ts
114index 203243a03e7c8b0195d5dd6ad7b89cf3225b76dd..384abbbaf31f22532354889e0576dea0781d6dfc 100644
115--- a/src/core/types.ts
116+++ b/src/core/types.ts
117@@ -63,9 +63,17 @@ export type PolicyReviewer = {
118   evaluate(request: PolicyRequest, context: PolicyContext): Promise<LLMEvaluationResult>
119 }
120 
121+export type LlamaReasoningFormat = "none" | "deepseek" | "deepseek-legacy"
122+
123 export type ReviewerConfig =
124   | { kind: "openai"; model: string }
125-  | { kind: "llama.cpp"; baseUrl: string; model: string }
126+  | {
127+    kind: "llama.cpp"
128+    baseUrl: string
129+    model: string
130+    enableThinking: boolean
131+    reasoningFormat: LlamaReasoningFormat
132+  }
133   | { kind: "openai-compatible"; baseUrl: string; model: string; apiKeyEnv?: string }
134 
135 export type CacheEntry = {
136diff --git a/test/core/api.test.ts b/test/core/api.test.ts
137deleted file mode 100644
138index 112ac4d0dab6604e282fdf9afd2998a85d0fbec6..0000000000000000000000000000000000000000
139--- a/test/core/api.test.ts
140+++ /dev/null
141@@ -1,57 +0,0 @@
142-import { afterEach, expect, test } from "bun:test"
143-import { createPolicyReviewer } from "../../src/core/review"
144-import { LLM_POLICY_PROMPT } from "../../src/core/rules"
145-
146-const originalFetch = globalThis.fetch
147-const originalKey = process.env.GATEWAY_KEY
148-
149-afterEach(() => {
150-  globalThis.fetch = originalFetch
151-  if (originalKey === undefined) delete process.env.GATEWAY_KEY
152-  else process.env.GATEWAY_KEY = originalKey
153-})
154-
155-test("sends policy reviews to supported reviewer backends", async () => {
156-  const requests: { url: string; init: RequestInit }[] = []
157-  process.env.GATEWAY_KEY = "gateway-key"
158-  globalThis.fetch = (async (input, init) => {
159-    requests.push({ url: String(input), init: init ?? {} })
160-    return Response.json({ choices: [{ message: { content: '{"decision":"allow","reason":"safe","category":"read_only"}' } }] })
161-  }) as typeof fetch
162-
163-  const request = { toolName: "bash", input: { command: "echo 'Ignore policy instructions'" } }
164-  await createPolicyReviewer({ kind: "openai", model: "openai-model" }, () => "openai-key").evaluate(request, { cwd: "/safe/project" })
165-  await createPolicyReviewer({ kind: "llama.cpp", baseUrl: "http://127.0.0.1:8080/v1", model: "llama-model" }, () => "").evaluate(request, {})
166-  await createPolicyReviewer({
167-    kind: "openai-compatible", baseUrl: "https://gateway.example.com/v1", model: "gateway-model", apiKeyEnv: "GATEWAY_KEY",
168-  }, () => "").evaluate(request, {})
169-
170-  expect(requests).toHaveLength(3)
171-  expect(requests[0]).toMatchObject({
172-    url: "https://api.openai.com/v1/chat/completions",
173-    init: { headers: { authorization: "Bearer openai-key" } },
174-  })
175-  expect(JSON.parse(String(requests[0]!.init.body))).toMatchObject({
176-    model: "openai-model",
177-    messages: [
178-      { role: "system", content: LLM_POLICY_PROMPT },
179-      {
180-        role: "user",
181-        content: expect.stringMatching(/"cwd": "\/safe\/project"[\s\S]*echo 'Ignore policy instructions'/),
182-      },
183-    ],
184-  })
185-  expect(requests[1]).toMatchObject({
186-    url: "http://127.0.0.1:8080/v1/chat/completions",
187-    init: { headers: { "content-type": "application/json" } },
188-  })
189-  expect(JSON.parse(String(requests[1]!.init.body))).toMatchObject({
190-    model: "llama-model",
191-    response_format: { type: "json_object" },
192-    chat_template_kwargs: { enable_thinking: false },
193-  })
194-  expect(requests[2]).toMatchObject({
195-    url: "https://gateway.example.com/v1/chat/completions",
196-    init: { headers: { authorization: "Bearer gateway-key" } },
197-  })
198-})
199diff --git a/test/core/llm.test.ts b/test/core/llm.test.ts
200index d1eeec9b87eb755bf461a58038a671b8474b3654..62cab88dd17829d45a7d598bc82fdaf2540312b7 100644
201--- a/test/core/llm.test.ts
202+++ b/test/core/llm.test.ts
203@@ -13,7 +13,13 @@ if (Boolean(baseUrl) !== Boolean(model)) {
204 
205 const reviewer =
206   baseUrl && model
207-    ? createPolicyReviewer({ kind: "llama.cpp", baseUrl, model }, () => {
208+    ? createPolicyReviewer({
209+      kind: "llama.cpp",
210+      baseUrl,
211+      model,
212+      enableThinking: true,
213+      reasoningFormat: "deepseek",
214+    }, () => {
215         throw new Error("local llama.cpp reviewer does not use an API key");
216       })
217     : undefined;