extension: add oMLX model discovery
This commit is contained in:
118
extensions/omlx-models.ts
Normal file
118
extensions/omlx-models.ts
Normal file
@@ -0,0 +1,118 @@
|
|||||||
|
import { readFile } from "node:fs/promises";
|
||||||
|
import { homedir } from "node:os";
|
||||||
|
import { join } from "node:path";
|
||||||
|
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
||||||
|
|
||||||
|
type OmlxSettings = {
|
||||||
|
server?: { host?: string; port?: number };
|
||||||
|
auth?: { api_key?: string };
|
||||||
|
sampling?: { max_context_window?: number; max_tokens?: number };
|
||||||
|
};
|
||||||
|
|
||||||
|
type ListedModel = {
|
||||||
|
id: string;
|
||||||
|
max_model_len?: number | null;
|
||||||
|
};
|
||||||
|
|
||||||
|
type ModelStatus = {
|
||||||
|
id: string;
|
||||||
|
model_type?: string | null;
|
||||||
|
thinking_default?: boolean | null;
|
||||||
|
preserve_thinking_default?: boolean | null;
|
||||||
|
max_context_window?: number | null;
|
||||||
|
max_tokens?: number | null;
|
||||||
|
};
|
||||||
|
|
||||||
|
const ZERO_COST = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 };
|
||||||
|
|
||||||
|
function maySupportReasoning(id: string): boolean {
|
||||||
|
return /(?:thinking|reasoning|deepseek[-_. ]?r1|\b[or][13]\b|qwen3(?:[._-][56])?)/i.test(id);
|
||||||
|
}
|
||||||
|
|
||||||
|
export default async function (pi: ExtensionAPI) {
|
||||||
|
const settingsPath = join(homedir(), ".omlx", "settings.json");
|
||||||
|
|
||||||
|
try {
|
||||||
|
const settings = JSON.parse(await readFile(settingsPath, "utf8")) as OmlxSettings;
|
||||||
|
const configuredHost = settings.server?.host ?? "127.0.0.1";
|
||||||
|
const host = configuredHost === "0.0.0.0" || configuredHost === "::" ? "127.0.0.1" : configuredHost;
|
||||||
|
const port = settings.server?.port ?? 8127;
|
||||||
|
const apiKey = settings.auth?.api_key || "omlx";
|
||||||
|
const baseUrl = `http://${host}:${port}/v1`;
|
||||||
|
const headers = { Authorization: `Bearer ${apiKey}` };
|
||||||
|
const fetchOptions = { headers, signal: AbortSignal.timeout(5000) };
|
||||||
|
|
||||||
|
const modelsResponse = await fetch(`${baseUrl}/models`, fetchOptions);
|
||||||
|
if (!modelsResponse.ok) {
|
||||||
|
throw new Error(`${modelsResponse.status} ${modelsResponse.statusText}`);
|
||||||
|
}
|
||||||
|
|
||||||
|
const listed = (await modelsResponse.json()) as { data?: ListedModel[] };
|
||||||
|
|
||||||
|
let statuses = new Map<string, ModelStatus>();
|
||||||
|
try {
|
||||||
|
const statusResponse = await fetch(`${baseUrl}/models/status`, fetchOptions);
|
||||||
|
if (statusResponse.ok) {
|
||||||
|
const payload = (await statusResponse.json()) as { models?: ModelStatus[] };
|
||||||
|
statuses = new Map((payload.models ?? []).map((model) => [model.id, model]));
|
||||||
|
}
|
||||||
|
} catch {
|
||||||
|
// /v1/models is sufficient; status only enriches capabilities and limits.
|
||||||
|
}
|
||||||
|
|
||||||
|
const models = (listed.data ?? [])
|
||||||
|
.filter((model) => {
|
||||||
|
if (model.id === "MarkItDown") return false;
|
||||||
|
const type = statuses.get(model.id)?.model_type;
|
||||||
|
return type == null || type === "llm" || type === "vlm";
|
||||||
|
})
|
||||||
|
.map((model) => {
|
||||||
|
const status = statuses.get(model.id);
|
||||||
|
const reasoning = status?.thinking_default != null || maySupportReasoning(model.id);
|
||||||
|
const compat: Record<string, unknown> = {
|
||||||
|
supportsDeveloperRole: false,
|
||||||
|
supportsReasoningEffort: false,
|
||||||
|
};
|
||||||
|
|
||||||
|
if (reasoning) {
|
||||||
|
compat.thinkingFormat = "qwen-chat-template";
|
||||||
|
compat.chatTemplateKwargs = {
|
||||||
|
enable_thinking: { $var: "thinking.enabled" },
|
||||||
|
preserve_thinking: status?.preserve_thinking_default === true,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
return {
|
||||||
|
id: model.id,
|
||||||
|
name: `${model.id} (oMLX Local)`,
|
||||||
|
reasoning,
|
||||||
|
input: status?.model_type === "vlm" ? (["text", "image"] as const) : (["text"] as const),
|
||||||
|
cost: ZERO_COST,
|
||||||
|
contextWindow:
|
||||||
|
status?.max_context_window ??
|
||||||
|
model.max_model_len ??
|
||||||
|
settings.sampling?.max_context_window ??
|
||||||
|
128000,
|
||||||
|
maxTokens: status?.max_tokens ?? settings.sampling?.max_tokens ?? 16384,
|
||||||
|
compat,
|
||||||
|
};
|
||||||
|
});
|
||||||
|
|
||||||
|
if (models.length === 0) {
|
||||||
|
console.warn("oMLX model discovery returned no chat models");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
pi.registerProvider("omlx", {
|
||||||
|
name: "oMLX (Local)",
|
||||||
|
baseUrl,
|
||||||
|
api: "openai-completions",
|
||||||
|
apiKey,
|
||||||
|
authHeader: true,
|
||||||
|
models,
|
||||||
|
});
|
||||||
|
} catch (error) {
|
||||||
|
const message = error instanceof Error ? error.message : String(error);
|
||||||
|
console.warn(`Could not discover oMLX models from ~/.omlx: ${message}`);
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user