099c26bfea222400e41f86a2c6dc128b0d7aa481

Author
célina <hanouticelina@gmail.com>
Committer
GitHub <noreply@github.com>
Date

Message

Add template-based thinking support (#2)

* first attempt

* remove startup /props probe

* remove dead code

* fixes

Diff

This diff is truncated to protect this page.

  1diff --git a/index.ts b/index.ts
  2index 20f5d11202776aaf79128c59c8c225bc73abd42d..2d77b843072766d747d799ee9c7147b29b02f10f 100644
  3--- a/index.ts
  4+++ b/index.ts
  5@@ -59,6 +59,7 @@ const PropsResponseSchema = Type.Object({
  6 			n_ctx: Type.Optional(Type.Number()),
  7 		}),
  8 	),
  9+	chat_template: Type.Optional(Type.String()),
 10 });
 11 
 12 const validatePropsResponse = Compile(PropsResponseSchema);
 13@@ -66,6 +67,33 @@ const validatePropsResponse = Compile(PropsResponseSchema);
 14 type LlamaModel = NonNullable<Parameters<ExtensionAPI["registerProvider"]>[1]["models"]>[number];
 15 type ExtensionCtx = Parameters<Parameters<ExtensionAPI["on"]>[1]>[1];
 16 
 17+// llama.cpp template thinking is boolean, so expose Pi's default off/medium toggle only.
 18+const TEMPLATE_THINKING_LEVEL_MAP = {
 19+	minimal: null,
 20+	low: null,
 21+	high: null,
 22+	xhigh: null,
 23+} satisfies NonNullable<LlamaModel["thinkingLevelMap"]>;
 24+
 25+// Minimal shape needed to update both registered models and Pi's active model snapshot.
 26+type MutableThinkingModel = {
 27+	reasoning: boolean;
 28+	thinkingLevelMap?: LlamaModel["thinkingLevelMap"];
 29+	compat?: LlamaModel["compat"];
 30+};
 31+
 32+// Mark a model as using llama.cpp's chat_template_kwargs.enable_thinking control.
 33+function applyTemplateThinkingSupport(model: MutableThinkingModel): void {
 34+	model.reasoning = true;
 35+	model.thinkingLevelMap = TEMPLATE_THINKING_LEVEL_MAP;
 36+	model.compat = {
 37+		...model.compat,
 38+		// Despite the Pi enum name, this sends llama.cpp's generic
 39+		// chat_template_kwargs.enable_thinking payload, not a Qwen-only option.
 40+		thinkingFormat: "qwen-chat-template",
 41+	};
 42+}
 43+
 44 export default async function (pi: ExtensionAPI) {
 45 	let currentModels: LlamaModel[] = [];
 46 
 47@@ -108,6 +136,7 @@ export default async function (pi: ExtensionAPI) {
 48 			const previousById = new Map(currentModels.map((m) => [m.id, m]));
 49 
 50 			currentModels = (payload.data ?? []).map((model) => {
 51+				const previous = previousById.get(model.id);
 52 				const isLoaded = model.status?.value === "loaded";
 53 				const modalities = model.architecture?.input_modalities ?? ["text"];
 54 				const input = modalities.filter(
 55@@ -123,12 +152,14 @@ export default async function (pi: ExtensionAPI) {
 56 				return {
 57 					id: model.id,
 58 					name: suffixes.length > 0 ? `${model.id} ${suffixes.join(" ")}` : model.id,
 59+					// /v1/models does not include /props-discovered capabilities, so preserve
 60+					// template thinking metadata across refreshes.
 61+					reasoning: previous?.reasoning ?? false,
 62+					thinkingLevelMap: previous?.thinkingLevelMap,
 63 					input,
 64 					cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
 65-					contextWindow:
 66-						model.meta?.n_ctx ??
 67-						previousById.get(model.id)?.contextWindow ??
 68-						DEFAULT_CONTEXT_WINDOW,
 69+					contextWindow: model.meta?.n_ctx ?? previous?.contextWindow ?? DEFAULT_CONTEXT_WINDOW,
 70+					compat: previous?.compat,
 71 				} as LlamaModel;
 72 			});
 73 
 74@@ -149,27 +180,43 @@ export default async function (pi: ExtensionAPI) {
 75 		}
 76 	}
 77 
 78-	const discoveredContext = new Set<string>();
 79-	const pendingContext = new Set<string>();
 80+	const discoveredMetadata = new Set<string>();
 81+	const pendingMetadata = new Set<string>();
 82 
 83-	async function discoverContextWindow(modelId: string, ctx: ExtensionCtx): Promise<void> {
 84-		if (discoveredContext.has(modelId) || pendingContext.has(modelId)) {
 85-			return;
 86-		}
 87+	async function discoverModelMetadata(
 88+		modelId: string,
 89+		ctx?: ExtensionCtx,
 90+		autoload = true,
 91+		timeoutMs = PROPS_TIMEOUT_MS,
 92+		selectedModel?: MutableThinkingModel,
 93+	): Promise<void> {
 94 		const model = currentModels.find((m) => m.id === modelId);
 95 		if (!model) {
 96 			return;
 97 		}
 98+		if (discoveredMetadata.has(modelId)) {
 99+			// Provider re-registration does not update Pi's active model snapshot, so copy
100+			// already-discovered thinking metadata into the selected model when available.
101+			if (selectedModel && model.reasoning) {
102+				selectedModel.reasoning = model.reasoning;
103+				selectedModel.thinkingLevelMap = model.thinkingLevelMap;
104+				selectedModel.compat = model.compat;