Diff
1diff --git a/index.ts b/index.ts
2index 372a02715726f0479d66e7c604afe37fc24ac6d9..20f5d11202776aaf79128c59c8c225bc73abd42d 100644
3--- a/index.ts
4+++ b/index.ts
5@@ -1,8 +1,20 @@
6+/**
7+ * llama.cpp provider for pi.
8+ *
9+ * Auto-discovers models from a running `llama-server` and
10+ * registers them under the `llama-cpp` provider.
11+ *
12+ * Usage: `pi install github.com/huggingface/pi-llama`
13+ */
14+
15 import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
16 import { Type } from "typebox";
17 import { Compile } from "typebox/compile";
18
19+const PROVIDER_ID = "llama-cpp";
20 const DEFAULT_BASE_URL = "http://localhost:8080/v1";
21+// Fallback for /v1/models entries missing meta.n_ctx.
22+const DEFAULT_CONTEXT_WINDOW = 8192;
23 const PROPS_TIMEOUT_MS = 120_000;
24
25 const ModelsResponseSchema = Type.Object({
26@@ -73,7 +85,8 @@ export default async function (pi: ExtensionAPI) {
27 },
28 });
29
30- const baseUrl = (process?.env?.LLAMA_BASE_URL ?? DEFAULT_BASE_URL).replace(/\/+$/, "");
31+ const baseUrl = (process.env.LLAMA_BASE_URL ?? DEFAULT_BASE_URL).replace(/\/+$/, "");
32+ const apiKey = process.env.LLAMA_API_KEY ?? "no-key";
33
34 async function refreshProvider(): Promise<void> {
35 try {
36@@ -105,15 +118,17 @@ export default async function (pi: ExtensionAPI) {
37 suffixes.push("(image)");
38 }
39 if (isLoaded) {
40- suffixes.push("(loaded ✅)");
41+ suffixes.push("(loaded)");
42 }
43 return {
44 id: model.id,
45 name: suffixes.length > 0 ? `${model.id} ${suffixes.join(" ")}` : model.id,
46 input,
47 cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
48- contextWindow: model.meta?.n_ctx ?? previousById.get(model.id)?.contextWindow ?? 0,
49- // maxTokens: -1,
50+ contextWindow:
51+ model.meta?.n_ctx ??
52+ previousById.get(model.id)?.contextWindow ??
53+ DEFAULT_CONTEXT_WINDOW,
54 } as LlamaModel;
55 });
56
57@@ -122,10 +137,10 @@ export default async function (pi: ExtensionAPI) {
58 return;
59 }
60
61- pi.registerProvider("llama-cpp", {
62+ pi.registerProvider(PROVIDER_ID, {
63 name: "llama.cpp",
64 baseUrl,
65- apiKey: "LLAMA_API_KEY",
66+ apiKey,
67 api: "openai-completions",
68 models: currentModels,
69 });
70@@ -169,10 +184,10 @@ export default async function (pi: ExtensionAPI) {
71 if (typeof nCtx === "number" && nCtx > 0) {
72 model.contextWindow = nCtx;
73 discoveredContext.add(modelId);
74- pi.registerProvider("llama-cpp", {
75+ pi.registerProvider(PROVIDER_ID, {
76 name: "llama.cpp",
77 baseUrl,
78- apiKey: "LLAMA_API_KEY",
79+ apiKey,
80 api: "openai-completions",
81 models: currentModels,
82 });
83@@ -198,7 +213,7 @@ export default async function (pi: ExtensionAPI) {
84 });
85
86 pi.on("model_select", (event, ctx) => {
87- if (event.model.provider !== "llama-cpp") {
88+ if (event.model.provider !== PROVIDER_ID) {
89 return;
90 }
91 void discoverContextWindow(event.model.id, ctx);