first
This commit is contained in:
+351
@@ -0,0 +1,351 @@
|
||||
// api-wrappers/lmstudio.js
|
||||
|
||||
const http = require("node:http");
|
||||
const Repetition = require("./repetition");
|
||||
|
||||
const BASE_HOST = "localhost";
|
||||
const BASE_PORT = 1234;
|
||||
const MAX_INCOMPLETE_CONTINUATIONS = 10;
|
||||
const INCOMPLETE_CONTINUE_PROMPT = [
|
||||
"Continue exactly where the previous response was cut off.",
|
||||
"Do not restart, summarize, or repeat material you already produced.",
|
||||
"Finish the original task and provide the remainder of the response."
|
||||
].join("\n");
|
||||
|
||||
function instanceMatches(instance, configured) {
|
||||
if (!instance || typeof instance !== "object") return false;
|
||||
return [instance.id, instance.key, instance.identifier, instance.model, instance.model_key, instance.modelKey]
|
||||
.some(value => typeof value === "string" && value === configured);
|
||||
}
|
||||
|
||||
async function warmUp(settings) {
|
||||
try {
|
||||
const listResponse = await fetch(`http://${BASE_HOST}:${BASE_PORT}/api/v1/models`);
|
||||
if (!listResponse.ok) throw new Error(`LM Studio model list returned ${listResponse.status}: ${await listResponse.text()}`);
|
||||
const list = await listResponse.json();
|
||||
const models = Array.isArray(list.models) ? list.models : [];
|
||||
const installed = models.find(item => item.key === settings.model);
|
||||
const loadedInstance = models.flatMap(item => item.loaded_instances || []).find(instance => instanceMatches(instance, settings.model));
|
||||
|
||||
if (loadedInstance || installed?.loaded_instances?.length) return;
|
||||
|
||||
if (!installed) {
|
||||
const openAiList = await fetch(`http://${BASE_HOST}:${BASE_PORT}/v1/models`);
|
||||
if (openAiList.ok) {
|
||||
const loaded = await openAiList.json();
|
||||
if (loaded.data?.some(item => item.id === settings.model)) return;
|
||||
}
|
||||
}
|
||||
|
||||
const modelKey = installed?.key || settings.model;
|
||||
const response = await fetch(`http://${BASE_HOST}:${BASE_PORT}/api/v1/models/load`, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ model: modelKey })
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
console.error(`Model warmup failed: LM Studio returned ${response.status}: ${await response.text()}`);
|
||||
return;
|
||||
}
|
||||
|
||||
await response.json();
|
||||
} catch (error) {
|
||||
console.error(`Model warmup failed: ${error.message}`);
|
||||
}
|
||||
}
|
||||
|
||||
function inputFor(conversation) {
|
||||
return conversation.messages.map(message => ({
|
||||
role: message.role,
|
||||
content: message.content
|
||||
}));
|
||||
}
|
||||
|
||||
function textFrom(content) {
|
||||
if (typeof content === "string") return content;
|
||||
if (!Array.isArray(content)) return "";
|
||||
return content.map(part => part?.text || "").join("");
|
||||
}
|
||||
|
||||
function replyFromResponse(result) {
|
||||
const reply = { role: "assistant", content: "" };
|
||||
const reasoning = [];
|
||||
|
||||
for (const item of result?.output || []) {
|
||||
if (item.type === "message") reply.content += textFrom(item.content);
|
||||
if (item.type === "reasoning") reasoning.push(textFrom(item.content));
|
||||
}
|
||||
|
||||
if (reasoning.length > 0) reply.reasoning = reasoning.join("\n");
|
||||
return reply;
|
||||
}
|
||||
|
||||
function deltaFromEvent(event) {
|
||||
if (!event || typeof event !== "object") return null;
|
||||
if (typeof event.delta !== "string" || !event.delta) return null;
|
||||
|
||||
const type = String(event.type || "");
|
||||
if (type.includes("reasoning")) return { channel: "reasoning", text: event.delta };
|
||||
if (type.includes("output_text") || type.includes("text.delta")) return { channel: "content", text: event.delta };
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
function pageDiagnostic(page) {
|
||||
return {
|
||||
responseId: page.responseId || null,
|
||||
status: page.status || null,
|
||||
incompleteDetails: page.incompleteDetails || null,
|
||||
usage: page.usage || null
|
||||
};
|
||||
}
|
||||
|
||||
function sendPage(conversation, settings, options = {}) {
|
||||
const body = {
|
||||
model: settings.model,
|
||||
input: options.previousResponseId && options.continuationInput ? options.continuationInput : inputFor(conversation),
|
||||
stream: true
|
||||
};
|
||||
|
||||
if (conversation.systemPrompt) body.instructions = conversation.systemPrompt;
|
||||
if (settings.reasoning && settings.reasoning !== "off") body.reasoning = { effort: settings.reasoning };
|
||||
if (options.previousResponseId) body.previous_response_id = options.previousResponseId;
|
||||
|
||||
return new Promise((resolve, reject) => {
|
||||
const payload = JSON.stringify(body);
|
||||
const request = http.request({
|
||||
hostname: BASE_HOST,
|
||||
port: BASE_PORT,
|
||||
path: "/v1/responses",
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Accept": "text/event-stream",
|
||||
"Content-Length": Buffer.byteLength(payload)
|
||||
}
|
||||
}, response => {
|
||||
if (response.statusCode < 200 || response.statusCode >= 300) {
|
||||
let errorText = "";
|
||||
response.setEncoding("utf8");
|
||||
response.on("data", chunk => errorText += chunk);
|
||||
response.on("end", () => {
|
||||
const error = new Error(`LM Studio returned ${response.statusCode}: ${errorText}`);
|
||||
error.retryable = response.statusCode >= 500;
|
||||
reject(error);
|
||||
});
|
||||
return;
|
||||
}
|
||||
|
||||
response.setEncoding("utf8");
|
||||
|
||||
let buffer = "";
|
||||
let content = "";
|
||||
let reasoning = "";
|
||||
let finalResponse = null;
|
||||
let finalEventType = null;
|
||||
let responseId = null;
|
||||
let settled = false;
|
||||
const repetition = {
|
||||
content: Repetition.createDetector(),
|
||||
reasoning: Repetition.createDetector()
|
||||
};
|
||||
|
||||
function fail(error) {
|
||||
if (settled) return;
|
||||
settled = true;
|
||||
response.destroy();
|
||||
request.destroy();
|
||||
reject(error);
|
||||
}
|
||||
|
||||
function appendDelta(channel, text) {
|
||||
if (!text) return true;
|
||||
const loop = repetition[channel].push(text);
|
||||
if (loop) {
|
||||
const error = new Error(
|
||||
`Model repetition loop detected in ${channel}: ` +
|
||||
`${loop.periodTokens}-token pattern repeated ${loop.repeatCount} times consecutively.`
|
||||
);
|
||||
error.code = "PERRY_REPETITION_LOOP";
|
||||
error.retryable = true;
|
||||
error.partial = Boolean(content || reasoning);
|
||||
error.repetition = loop;
|
||||
fail(error);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (channel === "content") {
|
||||
content += text;
|
||||
if (typeof options.onContent === "function") options.onContent(text);
|
||||
} else {
|
||||
reasoning += text;
|
||||
if (typeof options.onReasoning === "function") options.onReasoning(text);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
function consumeEvent(block) {
|
||||
const dataLines = block
|
||||
.split(/\r?\n/)
|
||||
.filter(line => line.startsWith("data:"))
|
||||
.map(line => line.slice(5).trimStart());
|
||||
|
||||
if (dataLines.length === 0) return;
|
||||
|
||||
const data = dataLines.join("\n").trim();
|
||||
if (!data || data === "[DONE]") return;
|
||||
|
||||
let event;
|
||||
try {
|
||||
event = JSON.parse(data);
|
||||
} catch {
|
||||
return;
|
||||
}
|
||||
|
||||
if (event.response?.id) responseId = event.response.id;
|
||||
|
||||
const delta = deltaFromEvent(event);
|
||||
if (delta && !appendDelta(delta.channel, delta.text)) return;
|
||||
|
||||
if ((event.type === "response.completed" || event.type === "response.incomplete") && event.response) {
|
||||
finalResponse = event.response;
|
||||
finalEventType = event.type;
|
||||
}
|
||||
|
||||
if (event.type === "response.failed" && event.response) {
|
||||
const message = event.response.error?.message || "LM Studio response failed.";
|
||||
const error = new Error(message);
|
||||
error.retryable = true;
|
||||
error.partial = Boolean(content || reasoning);
|
||||
fail(error);
|
||||
}
|
||||
}
|
||||
|
||||
function consumeBuffer(flush = false) {
|
||||
buffer = buffer.replace(/\r\n/g, "\n");
|
||||
|
||||
let boundary;
|
||||
while ((boundary = buffer.indexOf("\n\n")) !== -1) {
|
||||
const block = buffer.slice(0, boundary);
|
||||
buffer = buffer.slice(boundary + 2);
|
||||
consumeEvent(block);
|
||||
}
|
||||
|
||||
if (flush && buffer.trim()) {
|
||||
consumeEvent(buffer);
|
||||
buffer = "";
|
||||
}
|
||||
}
|
||||
|
||||
response.on("data", chunk => {
|
||||
if (settled) return;
|
||||
buffer += chunk;
|
||||
consumeBuffer();
|
||||
});
|
||||
|
||||
response.on("end", () => {
|
||||
if (settled) return;
|
||||
consumeBuffer(true);
|
||||
|
||||
if (!finalEventType) {
|
||||
const error = new Error("LM Studio stream ended without a completed or incomplete terminal response event.");
|
||||
error.retryable = true;
|
||||
error.partial = Boolean(content || reasoning);
|
||||
settled = true;
|
||||
reject(error);
|
||||
return;
|
||||
}
|
||||
|
||||
const fallback = replyFromResponse(finalResponse);
|
||||
const reply = {
|
||||
role: "assistant",
|
||||
content: content || fallback.content,
|
||||
status: finalEventType === "response.incomplete" ? "incomplete" : finalResponse?.status || "completed",
|
||||
responseId: responseId || finalResponse?.id || null,
|
||||
incompleteDetails: finalResponse?.incomplete_details || null,
|
||||
usage: finalResponse?.usage || null
|
||||
};
|
||||
|
||||
if (reasoning || fallback.reasoning) reply.reasoning = reasoning || fallback.reasoning;
|
||||
settled = true;
|
||||
resolve(reply);
|
||||
});
|
||||
|
||||
response.on("error", error => {
|
||||
if (settled) return;
|
||||
error.retryable = true;
|
||||
error.partial = Boolean(content || reasoning);
|
||||
settled = true;
|
||||
reject(error);
|
||||
});
|
||||
});
|
||||
|
||||
request.on("error", error => {
|
||||
error.retryable = true;
|
||||
reject(error);
|
||||
});
|
||||
|
||||
request.write(payload);
|
||||
request.end();
|
||||
});
|
||||
}
|
||||
|
||||
async function send(conversation, settings, options = {}) {
|
||||
const content = [];
|
||||
const reasoning = [];
|
||||
const pages = [];
|
||||
let previousResponseId = options.previousResponseId || null;
|
||||
let continuationInput = options.continuationInput || null;
|
||||
let lastReply = null;
|
||||
|
||||
for (let continuation = 0; continuation <= MAX_INCOMPLETE_CONTINUATIONS; continuation++) {
|
||||
const page = await sendPage(conversation, settings, {
|
||||
...options,
|
||||
previousResponseId,
|
||||
continuationInput
|
||||
});
|
||||
|
||||
lastReply = page;
|
||||
pages.push(pageDiagnostic(page));
|
||||
if (page.reasoning) reasoning.push(page.reasoning);
|
||||
if (page.content) content.push(page.content);
|
||||
|
||||
if (page.status !== "incomplete") {
|
||||
const reply = { role: "assistant", content: content.join("") };
|
||||
if (reasoning.length > 0) reply.reasoning = reasoning.join("");
|
||||
if (page.responseId) reply.responseId = page.responseId;
|
||||
reply.provider = {
|
||||
name: "lmstudio",
|
||||
continuationPages: pages.length - 1,
|
||||
pages
|
||||
};
|
||||
return reply;
|
||||
}
|
||||
|
||||
if (!page.responseId) {
|
||||
const error = new Error("LM Studio returned an incomplete response without a response id; it cannot be continued safely.");
|
||||
error.retryable = false;
|
||||
error.partial = Boolean(content.length || reasoning.length);
|
||||
throw error;
|
||||
}
|
||||
|
||||
if (continuation === MAX_INCOMPLETE_CONTINUATIONS) {
|
||||
const error = new Error(`LM Studio response remained incomplete after ${MAX_INCOMPLETE_CONTINUATIONS} automatic continuations.`);
|
||||
error.retryable = false;
|
||||
error.partial = Boolean(content.length || reasoning.length);
|
||||
error.provider = { name: "lmstudio", continuationPages: pages.length - 1, pages };
|
||||
throw error;
|
||||
}
|
||||
|
||||
previousResponseId = page.responseId;
|
||||
continuationInput = INCOMPLETE_CONTINUE_PROMPT;
|
||||
}
|
||||
|
||||
return lastReply;
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
warmUp,
|
||||
send
|
||||
};
|
||||
Reference in New Issue
Block a user