1
0
Fork 0
gnu-plus-dotfiles/dot_pi/agent/extensions/llama-swap.ts

371 lines
11 KiB
TypeScript

/**
* Llama-Swap Provider Extensions
*
* Registers two OpenAI-compatible providers backed by llama-swap instances:
*
* byte.llama — byte.local GPU (localhost:9292 when running on byte)
* miche.llama — miche.local GPU (localhost:9292 when running on miche)
*
* Both discover models dynamically at startup via /v1/models.
* Context windows are read live from the llama-swap /running endpoint so pi
* always reflects the actual --ctx-size configured in llama-swap.
* Filters out embedding/reranker/image-gen/STT/TTS/video/music models.
* Deduplicates aliases — keeps canonical IDs only (longest ID per model name).
*
* NOTE: pi requires `apiKey` on providers for them to appear in --list-models,
* even if the upstream server doesn't require authentication. We use a dummy
* key ("none") for unauthenticated endpoints.
*
* Updated 2026-07-06: Refresh for current model lineup (HauhauCS + Genesis).
* Updated 2026-07-18: Dynamic context discovery from /running (was hardcoded 256K).
* Updated 2026-07-27: Added missing models to MODEL_META: kat-coder-v2.5-dev, qwen36-35b-genesis-v5, qwen36-27b-711, qwen3.5-122b-heretic, laguna-xs-2.1-apex, laguna-s-2.1-chadrock-rocmfp4-v4.
*/
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
// ── Non-chat model filtering ──────────────────────────────────────────────
const NON_CHAT_IDS = new Set([
"qwen3-embedding-0.6b",
"qwen3-reranker-0.6b",
"embedding",
"embed",
"rerank",
"reranker",
"stt_parakeet",
"stt_qwenasr",
"qwenasr",
"stt",
"asr",
"tts_qwen3",
"vid-wan2",
"music-ace",
]);
const NON_CHAT_PREFIXES = ["img-", "dall-e-"];
function isChatModel(id: string): boolean {
if (NON_CHAT_IDS.has(id)) return false;
for (const pfx of NON_CHAT_PREFIXES) {
if (id.startsWith(pfx)) return false;
}
return true;
}
// ── Model metadata (fallback when a model is not currently running) ───────
interface ModelMeta {
vision: boolean;
reasoning: boolean;
contextWindow: number;
maxTokens: number;
compat?: Record<string, unknown>;
}
const MODEL_META: Record<string, ModelMeta> = {
// Models keyed by live /v1/models IDs from both byte.local and miche.local.
// Context windows below are fallbacks — the live /running endpoint is the
// source of truth when a model is loaded.
"ds4-flash": {
vision: false,
reasoning: true,
contextWindow: 65536,
maxTokens: 16384,
compat: { supportsDeveloperRole: false },
},
"gemma4-12b-hauhau": {
vision: true,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false },
},
"gemma4-26b-hauhau": {
vision: true,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false },
},
"gemma4-26b-meromero": {
vision: true,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false },
},
"gemma4-26b-styletune": {
vision: true,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false },
},
"gemma4-31b-hauhau": {
vision: true,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false },
},
"nemotron-3-super-120b": {
vision: false,
reasoning: true,
contextWindow: 262144,
maxTokens: 32768,
compat: { supportsDeveloperRole: false, thinkingFormat: "qwen-chat-template" },
},
"qwen3.6-27b-mtp": {
vision: true,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false, thinkingFormat: "qwen-chat-template" },
},
"qwen36-35b-genesis": {
vision: true,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false, thinkingFormat: "qwen-chat-template" },
},
"step-3.7-flash": {
vision: true,
reasoning: true,
contextWindow: 131072,
maxTokens: 16384,
compat: { supportsDeveloperRole: false },
},
"chadrock-35b-ace-saber": {
vision: false,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false, thinkingFormat: "qwen-chat-template" },
},
"chadrock3.6-27b-pi-agent": {
vision: false,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false },
},
"laguna-s-2.1-uncensored-apex": {
vision: false,
reasoning: true,
contextWindow: 1048576,
maxTokens: 32768,
compat: { supportsDeveloperRole: false },
},
// Models from miche.local not yet in the table above (dynamic discovery
// from /running will override these when loaded).
"kat-coder-v2.5-dev": {
vision: false,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false },
},
"qwen36-35b-genesis-v5": {
vision: true,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false, thinkingFormat: "qwen-chat-template" },
},
"qwen36-27b-711": {
vision: true,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false, thinkingFormat: "qwen-chat-template" },
},
"qwen3.5-122b-heretic": {
vision: false,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false, thinkingFormat: "qwen-chat-template" },
},
"laguna-xs-2.1-apex": {
vision: false,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false },
},
"laguna-s-2.1-chadrock-rocmfp4-v4": {
vision: false,
reasoning: true,
contextWindow: 262144,
maxTokens: 16384,
compat: { supportsDeveloperRole: false },
},
"toriigate-0.5": {
vision: true,
reasoning: false,
contextWindow: 16384,
maxTokens: 4096,
},
};
// ── Dynamic context discovery from llama-swap /running ────────────────────
/**
* Strips the /v1 API path to get the llama-swap server root.
* e.g. "http://miche.local:9292/v1" → "http://miche.local:9292"
*/
function serverOrigin(baseUrl: string): string {
return baseUrl.replace(/\/v1\/?$/, "");
}
/**
* Parses --ctx-size or -c from a llama-server command string.
* Returns the token count, or undefined if not found.
*/
function parseCtxSize(cmd: string): number | undefined {
const match = cmd.match(/(?:^|\s)(?:--ctx-size|-c)\s*=?\s*(\d+)/);
if (!match) return undefined;
const n = Number(match[1]);
return Number.isInteger(n) && n > 0 ? n : undefined;
}
/**
* Queries GET /running on a llama-swap instance and parses --ctx-size from
* live process command lines. Returns a map of model id → context window.
*
* This is the source of truth for context windows — reads the actual
* llama-server configuration rather than relying on a static table that
* goes stale when models are reconfigured.
*/
async function discoverContextFromRunning(
baseUrl: string,
): Promise<Map<string, number>> {
const origin = serverOrigin(baseUrl);
const result = new Map<string, number>();
try {
const res = await fetch(`${origin}/running`, {
headers: { Accept: "application/json" },
});
if (!res.ok) return result;
const payload = (await res.json()) as {
running?: Array<{ model?: string; cmd?: string }>;
};
for (const proc of payload.running ?? []) {
if (!proc.model || !proc.cmd) continue;
const ctx = parseCtxSize(proc.cmd);
if (ctx) result.set(proc.model, ctx);
}
} catch {
// /running unreachable or not available — fall back to static MODEL_META
}
return result;
}
// ── Helpers ───────────────────────────────────────────────────────────────
function dedupeAliases(
models: Array<{ id: string; name?: string }>,
): Array<{ id: string; name?: string }> {
const seen = new Map<string, { id: string; name?: string }>();
for (const m of models) {
if (!isChatModel(m.id)) continue;
const key = m.name ?? m.id;
const existing = seen.get(key);
if (!existing || m.id.length > existing.id.length) seen.set(key, m);
}
return [...seen.values()];
}
function buildPiModels(
models: Array<{ id: string; name?: string }>,
discoveredCtx: Map<string, number>,
) {
return models.map((m) => {
// Live llama-server context (from /running) beats the static table.
const discovered = discoveredCtx.get(m.id);
const meta = MODEL_META[m.id] ?? {
vision: false,
reasoning: true, // default true — all llama-swap chat models have --reasoning on
contextWindow: 262144,
maxTokens: 8192,
};
const contextWindow = discovered ?? meta.contextWindow;
return {
id: m.id,
name: m.name ?? m.id,
reasoning: meta.reasoning,
input: meta.vision ? (["text", "image"] as const) : (["text"] as const),
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow,
maxTokens: meta.maxTokens,
...(meta.compat ? { compat: meta.compat } : {}),
};
});
}
async function discoverAndRegister(
pi: ExtensionAPI,
providerName: string,
displayName: string,
baseUrl: string,
apiKey: string,
) {
let models: Array<{ id: string; name?: string }>;
try {
const res = await fetch(`${baseUrl}/models`);
if (!res.ok) throw new Error(`returned ${res.status}`);
const { data } = (await res.json()) as {
data: Array<{ id: string; name?: string }>;
};
if (!data || data.length === 0) throw new Error("No models returned");
models = data;
} catch {
// Endpoint unreachable — skip silently
return;
}
const chatModels = dedupeAliases(models);
if (chatModels.length === 0) return;
// Read actual --ctx-size from live llama-swap processes so the context
// window always matches what llama-swap is really running with.
const discoveredCtx = await discoverContextFromRunning(baseUrl);
pi.registerProvider(providerName, {
name: displayName,
baseUrl,
apiKey,
api: "openai-completions",
models: buildPiModels(chatModels, discoveredCtx),
});
}
// ── Entry point ───────────────────────────────────────────────────────────
import os from "node:os";
const HOST = (os.hostname() || "").toLowerCase();
export default async function (pi: ExtensionAPI) {
// byte.llama — byte.local GPU (localhost when running on byte itself)
await discoverAndRegister(
pi,
"byte.llama",
"byte.llama (byte.local GPU)",
HOST.startsWith("byte") ? "http://localhost:9292/v1" : "http://byte.local:9292/v1",
"none",
);
// miche.llama — miche.local GPU (localhost when running on miche itself)
await discoverAndRegister(
pi,
"miche.llama",
"miche.llama (miche.local GPU)",
HOST.startsWith("miche") ? "http://localhost:9292/v1" : "http://miche.local:9292/v1",
"none",
);
}