099c26bfea222400e41f86a2c6dc128b0d7aa481
- Author
- célina <hanouticelina@gmail.com>
- Committer
- GitHub <noreply@github.com>
- Date
Message
Diff
This diff is truncated to protect this page.
1diff --git a/index.ts b/index.ts
2index 20f5d11202776aaf79128c59c8c225bc73abd42d..2d77b843072766d747d799ee9c7147b29b02f10f 100644
3--- a/index.ts
4+++ b/index.ts
5@@ -59,6 +59,7 @@ const PropsResponseSchema = Type.Object({
6 n_ctx: Type.Optional(Type.Number()),
7 }),
8 ),
9+ chat_template: Type.Optional(Type.String()),
10 });
11
12 const validatePropsResponse = Compile(PropsResponseSchema);
13@@ -66,6 +67,33 @@ const validatePropsResponse = Compile(PropsResponseSchema);
14 type LlamaModel = NonNullable<Parameters<ExtensionAPI["registerProvider"]>[1]["models"]>[number];
15 type ExtensionCtx = Parameters<Parameters<ExtensionAPI["on"]>[1]>[1];
16
17+// llama.cpp template thinking is boolean, so expose Pi's default off/medium toggle only.
18+const TEMPLATE_THINKING_LEVEL_MAP = {
19+ minimal: null,
20+ low: null,
21+ high: null,
22+ xhigh: null,
23+} satisfies NonNullable<LlamaModel["thinkingLevelMap"]>;
24+
25+// Minimal shape needed to update both registered models and Pi's active model snapshot.
26+type MutableThinkingModel = {
27+ reasoning: boolean;
28+ thinkingLevelMap?: LlamaModel["thinkingLevelMap"];
29+ compat?: LlamaModel["compat"];
30+};
31+
32+// Mark a model as using llama.cpp's chat_template_kwargs.enable_thinking control.
33+function applyTemplateThinkingSupport(model: MutableThinkingModel): void {
34+ model.reasoning = true;
35+ model.thinkingLevelMap = TEMPLATE_THINKING_LEVEL_MAP;
36+ model.compat = {
37+ ...model.compat,
38+ // Despite the Pi enum name, this sends llama.cpp's generic
39+ // chat_template_kwargs.enable_thinking payload, not a Qwen-only option.
40+ thinkingFormat: "qwen-chat-template",
41+ };
42+}
43+
44 export default async function (pi: ExtensionAPI) {
45 let currentModels: LlamaModel[] = [];
46
47@@ -108,6 +136,7 @@ export default async function (pi: ExtensionAPI) {
48 const previousById = new Map(currentModels.map((m) => [m.id, m]));
49
50 currentModels = (payload.data ?? []).map((model) => {
51+ const previous = previousById.get(model.id);
52 const isLoaded = model.status?.value === "loaded";
53 const modalities = model.architecture?.input_modalities ?? ["text"];
54 const input = modalities.filter(
55@@ -123,12 +152,14 @@ export default async function (pi: ExtensionAPI) {
56 return {
57 id: model.id,
58 name: suffixes.length > 0 ? `${model.id} ${suffixes.join(" ")}` : model.id,
59+ // /v1/models does not include /props-discovered capabilities, so preserve
60+ // template thinking metadata across refreshes.
61+ reasoning: previous?.reasoning ?? false,
62+ thinkingLevelMap: previous?.thinkingLevelMap,
63 input,
64 cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
65- contextWindow:
66- model.meta?.n_ctx ??
67- previousById.get(model.id)?.contextWindow ??
68- DEFAULT_CONTEXT_WINDOW,
69+ contextWindow: model.meta?.n_ctx ?? previous?.contextWindow ?? DEFAULT_CONTEXT_WINDOW,
70+ compat: previous?.compat,
71 } as LlamaModel;
72 });
73
74@@ -149,27 +180,43 @@ export default async function (pi: ExtensionAPI) {
75 }
76 }
77
78- const discoveredContext = new Set<string>();
79- const pendingContext = new Set<string>();
80+ const discoveredMetadata = new Set<string>();
81+ const pendingMetadata = new Set<string>();
82
83- async function discoverContextWindow(modelId: string, ctx: ExtensionCtx): Promise<void> {
84- if (discoveredContext.has(modelId) || pendingContext.has(modelId)) {
85- return;
86- }
87+ async function discoverModelMetadata(
88+ modelId: string,
89+ ctx?: ExtensionCtx,
90+ autoload = true,
91+ timeoutMs = PROPS_TIMEOUT_MS,
92+ selectedModel?: MutableThinkingModel,
93+ ): Promise<void> {
94 const model = currentModels.find((m) => m.id === modelId);
95 if (!model) {
96 return;
97 }
98+ if (discoveredMetadata.has(modelId)) {
99+ // Provider re-registration does not update Pi's active model snapshot, so copy
100+ // already-discovered thinking metadata into the selected model when available.
101+ if (selectedModel && model.reasoning) {
102+ selectedModel.reasoning = model.reasoning;
103+ selectedModel.thinkingLevelMap = model.thinkingLevelMap;
104+ selectedModel.compat = model.compat;