505d71c929ade1dd973af7c69704660676f072d7

Author
atmosfar <atmosfar@users.noreply.github.com>
Committer
GitHub <noreply@github.com>
Date

Message

Implement llama-server /models/sse loading% indicator (#21)

* Implement model loading progress indicator

* Added completion tickmark to 'loaded' message

* addresses review comments: clear widget when finished, and use modelId to filter the sse events

* track loaded model state via SSE instead of stale snapshots

- Track currentlyLoadedModel via SSE status events for all models,
  not just the one being monitored
- Use tracked state instead of model.status?.value in
  discoverModelMetadata to decide if a model is loaded
- Clear discoveredMetadata cache when a model is unloaded so
  re-selecting it triggers a fresh /props discovery
- Set currentlyLoadedModel after autoload completes
- Abort in-flight /props requests when switching models
- Suppress stale exit code and error notifications for cancelled loads
- Use ctx.ui.notify instead of console.warn for SSE errors

* Apply suggestion from @hanouticelina

Co-authored-by: célina <hanouticelina@gmail.com>

* style: fix indentation in SSE error handler (oxfmt)

The 'Apply suggestion' commit inserted spaces instead of tabs, failing
`oxfmt --check` in CI. Reformat with oxfmt to restore tab indentation.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01VDhsHZgkG56NRpENe8xKa6

---------

Co-authored-by: célina <hanouticelina@gmail.com>
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

Diff

This diff is truncated to protect this page.

  1diff --git a/index.ts b/index.ts
  2index 42f1ddacdba79fe07ccb2bf253350f2c47742d11..9ebc7f44f6b35b249077a0589199fe9f0536dc55 100644
  3--- a/index.ts
  4+++ b/index.ts
  5@@ -10,6 +10,7 @@
  6 import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
  7 import { Type } from "typebox";
  8 import { Compile } from "typebox/compile";
  9+import { Loader, truncateToWidth, visibleWidth } from "@earendil-works/pi-tui";
 10 
 11 const PROVIDER_ID = "llama-cpp";
 12 const DEFAULT_BASE_URL = "http://localhost:8080/v1";
 13@@ -26,6 +27,7 @@ const ModelsResponseSchema = Type.Object({
 14 		Type.Array(
 15 			Type.Object({
 16 				id: Type.String(),
 17+				aliases: Type.Optional(Type.Array(Type.String())),
 18 				status: Type.Optional(
 19 					Type.Object({
 20 						value: Type.Optional(
 21@@ -69,6 +71,33 @@ const PropsResponseSchema = Type.Object({
 22 
 23 const validatePropsResponse = Compile(PropsResponseSchema);
 24 
 25+// SSE event types for model loading progress
 26+type ApiModelLoadStage = "text_model" | "spec_model" | "mmproj_model";
 27+
 28+type ApiModelsSseProgress = {
 29+	stages: ApiModelLoadStage[];
 30+	current: ApiModelLoadStage;
 31+	value: number;
 32+};
 33+
 34+type ApiModelsSseData = {
 35+	status: string;
 36+	progress?: ApiModelsSseProgress;
 37+	exit_code?: number;
 38+};
 39+
 40+type ApiModelsSseEvent = {
 41+	model: string;
 42+	event: string;
 43+	data: ApiModelsSseData;
 44+};
 45+
 46+const MODEL_LOAD_STAGE_LABELS: Record<ApiModelLoadStage, string> = {
 47+	text_model: "Loading weights",
 48+	spec_model: "Loading draft",
 49+	mmproj_model: "Loading projector",
 50+};
 51+
 52 type LlamaModel = NonNullable<Parameters<ExtensionAPI["registerProvider"]>[1]["models"]>[number];
 53 type ExtensionCtx = Parameters<Parameters<ExtensionAPI["on"]>[1]>[1];
 54 
 55@@ -170,9 +199,10 @@ export default async function (pi: ExtensionAPI) {
 56 				}
 57 				const contextWindow =
 58 					model.meta?.n_ctx ?? previous?.contextWindow ?? DEFAULT_CONTEXT_WINDOW;
 59+				const displayName = model.aliases?.[0] || model.id;
 60 				return {
 61 					id: model.id,
 62-					name: suffixes.length > 0 ? `${model.id} ${suffixes.join(" ")}` : model.id,
 63+					name: suffixes.length > 0 ? `${displayName} ${suffixes.join(" ")}` : displayName,
 64 					// /v1/models does not include /props-discovered capabilities, so preserve
 65 					// template thinking metadata across refreshes.
 66 					reasoning: previous?.reasoning ?? false,
 67@@ -182,6 +212,7 @@ export default async function (pi: ExtensionAPI) {
 68 					contextWindow,
 69 					maxTokens: Math.min(DEFAULT_MAX_TOKENS, contextWindow),
 70 					compat: previous?.compat,
 71+					status: model.status,
 72 				} as LlamaModel;
 73 			});
 74 
 75@@ -190,6 +221,10 @@ export default async function (pi: ExtensionAPI) {
 76 				return;
 77 			}
 78 
 79+			// Track which model is currently loaded on the server
 80+			const loadedModel = currentModels.find((m) => m.status?.value === "loaded");
 81+			currentlyLoadedModel = loadedModel?.id ?? null;
 82+
 83 			pi.registerProvider(PROVIDER_ID, {
 84 				name: "llama.cpp",
 85 				baseUrl,
 86@@ -204,7 +239,10 @@ export default async function (pi: ExtensionAPI) {
 87 
 88 	const discoveredMetadata = new Set<string>();
 89 	const pendingMetadata = new Set<string>();
 90+	let currentlyLoadedModel: string | null = null;
 91 	let statusTimeout: ReturnType<typeof setTimeout> | undefined;
 92+	let sseAbortController: AbortController | null = null;
 93+	let propsAbortController: AbortController | null = null;
 94 
 95 	function clearFooterStatusTimeout(): void {
 96 		if (statusTimeout !== undefined) {
 97@@ -213,6 +251,131 @@ export default async function (pi: ExtensionAPI) {
 98 		}
 99 	}
100 
101+	// Connect to SSE stream for model loading progress
102+	async function connectToLoadingProgress(
103+		modelId: string,
104+		ctx: ExtensionCtx,