first
This commit is contained in:
@@ -1,3 +1,23 @@
|
|||||||
# perry
|
# Perry
|
||||||
|
|
||||||
A small agent. It doesn't do much. But somehow it does everything that is necessary to get the job done.
|
Perry is a small utility for working with persistent conversational agents.
|
||||||
|
|
||||||
|
It grew out of Chat2. Perry is not a scheduler, artifact graph, translation pipeline, or full-featured agent framework. Applications own why work happens and what the work means. Perry owns the conversational agent mechanics around a local model.
|
||||||
|
|
||||||
|
Current API:
|
||||||
|
|
||||||
|
- `warmUp()`
|
||||||
|
- `prepare(name, systemPrompt, segments)`
|
||||||
|
- `ask(agent, input, options)`
|
||||||
|
- `followUp(agent, originalInput, previousReply, input, options)`
|
||||||
|
- `model()`
|
||||||
|
|
||||||
|
Current backend: LM Studio Responses API. The extracted source contained no active Ollama backend, so none was resurrected.
|
||||||
|
|
||||||
|
## Repetition self-destruct
|
||||||
|
|
||||||
|
Perry watches reasoning and final-content streams independently for exact contiguous periodic repetition. If the tail becomes the same token sequence repeated seven complete times with no novel tokens between repetitions, Perry aborts that response and raises a retryable `PERRY_REPETITION_LOOP` error.
|
||||||
|
|
||||||
|
The detector is deliberately literal. Similar ideas with changing token sequences are allowed to continue; Perry intervenes only after the stream locks into an exact repeated orbit. Pattern periods from one through 256 tokens are checked.
|
||||||
|
|
||||||
|
The normal retry layer then starts the response again without appending the pathological partial response to the conversation.
|
||||||
|
|||||||
@@ -0,0 +1,34 @@
|
|||||||
|
function copyMessage(message, includeReasoning = true) {
|
||||||
|
const copy = { role: message.role, content: message.content };
|
||||||
|
if (includeReasoning && message.reasoning) copy.reasoning = message.reasoning;
|
||||||
|
return copy;
|
||||||
|
}
|
||||||
|
|
||||||
|
function create(systemPrompt, messages = []) {
|
||||||
|
const conversation = {
|
||||||
|
systemPrompt,
|
||||||
|
messages: messages.map(message => copyMessage(message)),
|
||||||
|
|
||||||
|
add(source) {
|
||||||
|
if (Array.isArray(source?.messages)) {
|
||||||
|
for (const message of source.messages) conversation.messages.push(copyMessage(message, false));
|
||||||
|
return conversation;
|
||||||
|
}
|
||||||
|
|
||||||
|
conversation.messages.push(copyMessage(source));
|
||||||
|
return conversation;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
return conversation;
|
||||||
|
}
|
||||||
|
|
||||||
|
function start(systemPrompt = null) {
|
||||||
|
return create(systemPrompt);
|
||||||
|
}
|
||||||
|
|
||||||
|
function from(blob, fallbackSystemPrompt = null) {
|
||||||
|
const systemPrompt = blob.systemPrompt === undefined ? fallbackSystemPrompt : blob.systemPrompt;
|
||||||
|
return create(systemPrompt, blob.messages || []);
|
||||||
|
}
|
||||||
|
|
||||||
|
module.exports = { start, from };
|
||||||
@@ -0,0 +1,139 @@
|
|||||||
|
const Conversation = require("./conversation");
|
||||||
|
const lmStudio = require("./lmstudio");
|
||||||
|
|
||||||
|
const MAX_TRANSPORT_ATTEMPTS = 10;
|
||||||
|
const MAX_CONTINUATIONS = 10;
|
||||||
|
const CONTINUE_PROMPT = [
|
||||||
|
"Your previous response ended before you produced the requested final answer.",
|
||||||
|
"Continue from where you left off. Do not restart the task or discard the work you already did.",
|
||||||
|
"Keep investigating as deeply as necessary. When you have actually finished, provide the final deliverable required by the original prompt."
|
||||||
|
].join("\n");
|
||||||
|
|
||||||
|
module.exports = utils => {
|
||||||
|
const settings = utils.getSettings("llm");
|
||||||
|
const client = String(settings.client || "lm-studio").toLowerCase();
|
||||||
|
if (client !== "lm-studio" && client !== "lmstudio") throw new Error(`Unsupported LLM client: ${settings.client}`);
|
||||||
|
const llm = lmStudio;
|
||||||
|
let warmupPromise = null;
|
||||||
|
|
||||||
|
function warmUp() {
|
||||||
|
if (!warmupPromise) warmupPromise = llm.warmUp(settings);
|
||||||
|
return warmupPromise;
|
||||||
|
}
|
||||||
|
|
||||||
|
function prepare(name, systemPrompt, segments = []) {
|
||||||
|
const conversation = Conversation.start(systemPrompt);
|
||||||
|
for (const segment of segments) conversation.add(segment);
|
||||||
|
return { name: name || "anonymous", conversation };
|
||||||
|
}
|
||||||
|
|
||||||
|
function liveOutput() {
|
||||||
|
let channel = null;
|
||||||
|
let wrote = false;
|
||||||
|
function switchTo(next) {
|
||||||
|
if (channel === next) return;
|
||||||
|
if (wrote) process.stdout.write("\n");
|
||||||
|
process.stdout.write(` --- ${next} ---\n`);
|
||||||
|
channel = next;
|
||||||
|
wrote = true;
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
reasoning(text) { switchTo("reasoning"); process.stdout.write(text); },
|
||||||
|
content(text) { switchTo("response"); process.stdout.write(text); },
|
||||||
|
end() { if (wrote) process.stdout.write("\n"); }
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
function addFallbackContinuation(conversation, reply) {
|
||||||
|
if (reply.reasoning?.trim()) {
|
||||||
|
conversation.add({
|
||||||
|
role: "assistant",
|
||||||
|
content: [
|
||||||
|
"[UNFINISHED PRIVATE WORK FROM MY PREVIOUS RESPONSE]",
|
||||||
|
reply.reasoning.trim(),
|
||||||
|
"[END UNFINISHED PRIVATE WORK]"
|
||||||
|
].join("\n")
|
||||||
|
});
|
||||||
|
}
|
||||||
|
conversation.add({ role: "user", content: CONTINUE_PROMPT });
|
||||||
|
}
|
||||||
|
|
||||||
|
async function askConversation(agent, conversation, options = {}) {
|
||||||
|
await warmUp();
|
||||||
|
const callerOwnsOutput = typeof options.onReasoning === "function" || typeof options.onContent === "function";
|
||||||
|
const reasoning = [];
|
||||||
|
let previousResponseId = null;
|
||||||
|
let useStatefulContinuation = false;
|
||||||
|
|
||||||
|
for (let continuation = 0; continuation <= MAX_CONTINUATIONS; continuation++) {
|
||||||
|
for (let attempt = 1; attempt <= MAX_TRANSPORT_ATTEMPTS; attempt++) {
|
||||||
|
const output = !callerOwnsOutput && options.live !== false ? liveOutput() : null;
|
||||||
|
const sendOptions = {
|
||||||
|
...options,
|
||||||
|
onReasoning: options.onReasoning || output?.reasoning,
|
||||||
|
onContent: options.onContent || output?.content
|
||||||
|
};
|
||||||
|
delete sendOptions.live;
|
||||||
|
if (useStatefulContinuation && previousResponseId) {
|
||||||
|
sendOptions.previousResponseId = previousResponseId;
|
||||||
|
sendOptions.continuationInput = CONTINUE_PROMPT;
|
||||||
|
}
|
||||||
|
|
||||||
|
try {
|
||||||
|
const reply = await llm.send(conversation, settings, sendOptions);
|
||||||
|
output?.end();
|
||||||
|
if (reply.reasoning?.trim()) reasoning.push(reply.reasoning.trim());
|
||||||
|
if (reply.content?.trim()) {
|
||||||
|
delete reply.responseId;
|
||||||
|
if (reasoning.length > 0) reply.reasoning = reasoning.join("\n\n--- continuation ---\n\n");
|
||||||
|
return reply;
|
||||||
|
}
|
||||||
|
if (continuation >= MAX_CONTINUATIONS) {
|
||||||
|
throw Object.assign(new Error(`LLM returned no final response content for ${agent.name} after ${MAX_CONTINUATIONS} continuations.`), { retryable: false });
|
||||||
|
}
|
||||||
|
console.error();
|
||||||
|
console.error(`LLM returned unfinished reasoning with no final response for ${agent.name}; continuing (${continuation + 1}/${MAX_CONTINUATIONS}).`);
|
||||||
|
if (reply.responseId) {
|
||||||
|
previousResponseId = reply.responseId;
|
||||||
|
useStatefulContinuation = true;
|
||||||
|
} else {
|
||||||
|
previousResponseId = null;
|
||||||
|
useStatefulContinuation = false;
|
||||||
|
addFallbackContinuation(conversation, reply);
|
||||||
|
}
|
||||||
|
break;
|
||||||
|
} catch (error) {
|
||||||
|
output?.end();
|
||||||
|
const retryable = error.retryable !== false;
|
||||||
|
console.error();
|
||||||
|
console.error(`LLM request failed for ${agent.name} (attempt ${attempt}/${MAX_TRANSPORT_ATTEMPTS}): ${error.message}`);
|
||||||
|
if (error.cause) console.error(` Cause: ${error.cause.code || error.cause.message || String(error.cause)}`);
|
||||||
|
if (error.partial) console.error(" Stream ended after partial output; retry will restart this response.");
|
||||||
|
if (!retryable || attempt === MAX_TRANSPORT_ATTEMPTS) throw error;
|
||||||
|
const delayMs = Math.min(attempt, 5) * 1000;
|
||||||
|
console.error(` Retrying in ${delayMs / 1000} second${delayMs === 1000 ? "" : "s"}...`);
|
||||||
|
await new Promise(resolve => setTimeout(resolve, delayMs));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
throw new Error(`LLM request for ${agent.name} failed unexpectedly.`);
|
||||||
|
}
|
||||||
|
|
||||||
|
async function ask(agent, input, options = {}) {
|
||||||
|
const conversation = Conversation.from(agent.conversation);
|
||||||
|
conversation.add({ role: "user", content: input });
|
||||||
|
return askConversation(agent, conversation, options);
|
||||||
|
}
|
||||||
|
|
||||||
|
async function followUp(agent, originalInput, previousReply, input, options = {}) {
|
||||||
|
const conversation = Conversation.from(agent.conversation);
|
||||||
|
conversation.add({ role: "user", content: originalInput });
|
||||||
|
conversation.add(previousReply);
|
||||||
|
conversation.add({ role: "user", content: input });
|
||||||
|
return askConversation(agent, conversation, options);
|
||||||
|
}
|
||||||
|
|
||||||
|
function model() { return settings.model; }
|
||||||
|
|
||||||
|
return { warmUp, prepare, ask, followUp, model };
|
||||||
|
};
|
||||||
+351
@@ -0,0 +1,351 @@
|
|||||||
|
// api-wrappers/lmstudio.js
|
||||||
|
|
||||||
|
const http = require("node:http");
|
||||||
|
const Repetition = require("./repetition");
|
||||||
|
|
||||||
|
const BASE_HOST = "localhost";
|
||||||
|
const BASE_PORT = 1234;
|
||||||
|
const MAX_INCOMPLETE_CONTINUATIONS = 10;
|
||||||
|
const INCOMPLETE_CONTINUE_PROMPT = [
|
||||||
|
"Continue exactly where the previous response was cut off.",
|
||||||
|
"Do not restart, summarize, or repeat material you already produced.",
|
||||||
|
"Finish the original task and provide the remainder of the response."
|
||||||
|
].join("\n");
|
||||||
|
|
||||||
|
function instanceMatches(instance, configured) {
|
||||||
|
if (!instance || typeof instance !== "object") return false;
|
||||||
|
return [instance.id, instance.key, instance.identifier, instance.model, instance.model_key, instance.modelKey]
|
||||||
|
.some(value => typeof value === "string" && value === configured);
|
||||||
|
}
|
||||||
|
|
||||||
|
async function warmUp(settings) {
|
||||||
|
try {
|
||||||
|
const listResponse = await fetch(`http://${BASE_HOST}:${BASE_PORT}/api/v1/models`);
|
||||||
|
if (!listResponse.ok) throw new Error(`LM Studio model list returned ${listResponse.status}: ${await listResponse.text()}`);
|
||||||
|
const list = await listResponse.json();
|
||||||
|
const models = Array.isArray(list.models) ? list.models : [];
|
||||||
|
const installed = models.find(item => item.key === settings.model);
|
||||||
|
const loadedInstance = models.flatMap(item => item.loaded_instances || []).find(instance => instanceMatches(instance, settings.model));
|
||||||
|
|
||||||
|
if (loadedInstance || installed?.loaded_instances?.length) return;
|
||||||
|
|
||||||
|
if (!installed) {
|
||||||
|
const openAiList = await fetch(`http://${BASE_HOST}:${BASE_PORT}/v1/models`);
|
||||||
|
if (openAiList.ok) {
|
||||||
|
const loaded = await openAiList.json();
|
||||||
|
if (loaded.data?.some(item => item.id === settings.model)) return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const modelKey = installed?.key || settings.model;
|
||||||
|
const response = await fetch(`http://${BASE_HOST}:${BASE_PORT}/api/v1/models/load`, {
|
||||||
|
method: "POST",
|
||||||
|
headers: { "Content-Type": "application/json" },
|
||||||
|
body: JSON.stringify({ model: modelKey })
|
||||||
|
});
|
||||||
|
|
||||||
|
if (!response.ok) {
|
||||||
|
console.error(`Model warmup failed: LM Studio returned ${response.status}: ${await response.text()}`);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
await response.json();
|
||||||
|
} catch (error) {
|
||||||
|
console.error(`Model warmup failed: ${error.message}`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function inputFor(conversation) {
|
||||||
|
return conversation.messages.map(message => ({
|
||||||
|
role: message.role,
|
||||||
|
content: message.content
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
function textFrom(content) {
|
||||||
|
if (typeof content === "string") return content;
|
||||||
|
if (!Array.isArray(content)) return "";
|
||||||
|
return content.map(part => part?.text || "").join("");
|
||||||
|
}
|
||||||
|
|
||||||
|
function replyFromResponse(result) {
|
||||||
|
const reply = { role: "assistant", content: "" };
|
||||||
|
const reasoning = [];
|
||||||
|
|
||||||
|
for (const item of result?.output || []) {
|
||||||
|
if (item.type === "message") reply.content += textFrom(item.content);
|
||||||
|
if (item.type === "reasoning") reasoning.push(textFrom(item.content));
|
||||||
|
}
|
||||||
|
|
||||||
|
if (reasoning.length > 0) reply.reasoning = reasoning.join("\n");
|
||||||
|
return reply;
|
||||||
|
}
|
||||||
|
|
||||||
|
function deltaFromEvent(event) {
|
||||||
|
if (!event || typeof event !== "object") return null;
|
||||||
|
if (typeof event.delta !== "string" || !event.delta) return null;
|
||||||
|
|
||||||
|
const type = String(event.type || "");
|
||||||
|
if (type.includes("reasoning")) return { channel: "reasoning", text: event.delta };
|
||||||
|
if (type.includes("output_text") || type.includes("text.delta")) return { channel: "content", text: event.delta };
|
||||||
|
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
function pageDiagnostic(page) {
|
||||||
|
return {
|
||||||
|
responseId: page.responseId || null,
|
||||||
|
status: page.status || null,
|
||||||
|
incompleteDetails: page.incompleteDetails || null,
|
||||||
|
usage: page.usage || null
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
function sendPage(conversation, settings, options = {}) {
|
||||||
|
const body = {
|
||||||
|
model: settings.model,
|
||||||
|
input: options.previousResponseId && options.continuationInput ? options.continuationInput : inputFor(conversation),
|
||||||
|
stream: true
|
||||||
|
};
|
||||||
|
|
||||||
|
if (conversation.systemPrompt) body.instructions = conversation.systemPrompt;
|
||||||
|
if (settings.reasoning && settings.reasoning !== "off") body.reasoning = { effort: settings.reasoning };
|
||||||
|
if (options.previousResponseId) body.previous_response_id = options.previousResponseId;
|
||||||
|
|
||||||
|
return new Promise((resolve, reject) => {
|
||||||
|
const payload = JSON.stringify(body);
|
||||||
|
const request = http.request({
|
||||||
|
hostname: BASE_HOST,
|
||||||
|
port: BASE_PORT,
|
||||||
|
path: "/v1/responses",
|
||||||
|
method: "POST",
|
||||||
|
headers: {
|
||||||
|
"Content-Type": "application/json",
|
||||||
|
"Accept": "text/event-stream",
|
||||||
|
"Content-Length": Buffer.byteLength(payload)
|
||||||
|
}
|
||||||
|
}, response => {
|
||||||
|
if (response.statusCode < 200 || response.statusCode >= 300) {
|
||||||
|
let errorText = "";
|
||||||
|
response.setEncoding("utf8");
|
||||||
|
response.on("data", chunk => errorText += chunk);
|
||||||
|
response.on("end", () => {
|
||||||
|
const error = new Error(`LM Studio returned ${response.statusCode}: ${errorText}`);
|
||||||
|
error.retryable = response.statusCode >= 500;
|
||||||
|
reject(error);
|
||||||
|
});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
response.setEncoding("utf8");
|
||||||
|
|
||||||
|
let buffer = "";
|
||||||
|
let content = "";
|
||||||
|
let reasoning = "";
|
||||||
|
let finalResponse = null;
|
||||||
|
let finalEventType = null;
|
||||||
|
let responseId = null;
|
||||||
|
let settled = false;
|
||||||
|
const repetition = {
|
||||||
|
content: Repetition.createDetector(),
|
||||||
|
reasoning: Repetition.createDetector()
|
||||||
|
};
|
||||||
|
|
||||||
|
function fail(error) {
|
||||||
|
if (settled) return;
|
||||||
|
settled = true;
|
||||||
|
response.destroy();
|
||||||
|
request.destroy();
|
||||||
|
reject(error);
|
||||||
|
}
|
||||||
|
|
||||||
|
function appendDelta(channel, text) {
|
||||||
|
if (!text) return true;
|
||||||
|
const loop = repetition[channel].push(text);
|
||||||
|
if (loop) {
|
||||||
|
const error = new Error(
|
||||||
|
`Model repetition loop detected in ${channel}: ` +
|
||||||
|
`${loop.periodTokens}-token pattern repeated ${loop.repeatCount} times consecutively.`
|
||||||
|
);
|
||||||
|
error.code = "PERRY_REPETITION_LOOP";
|
||||||
|
error.retryable = true;
|
||||||
|
error.partial = Boolean(content || reasoning);
|
||||||
|
error.repetition = loop;
|
||||||
|
fail(error);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (channel === "content") {
|
||||||
|
content += text;
|
||||||
|
if (typeof options.onContent === "function") options.onContent(text);
|
||||||
|
} else {
|
||||||
|
reasoning += text;
|
||||||
|
if (typeof options.onReasoning === "function") options.onReasoning(text);
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
function consumeEvent(block) {
|
||||||
|
const dataLines = block
|
||||||
|
.split(/\r?\n/)
|
||||||
|
.filter(line => line.startsWith("data:"))
|
||||||
|
.map(line => line.slice(5).trimStart());
|
||||||
|
|
||||||
|
if (dataLines.length === 0) return;
|
||||||
|
|
||||||
|
const data = dataLines.join("\n").trim();
|
||||||
|
if (!data || data === "[DONE]") return;
|
||||||
|
|
||||||
|
let event;
|
||||||
|
try {
|
||||||
|
event = JSON.parse(data);
|
||||||
|
} catch {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (event.response?.id) responseId = event.response.id;
|
||||||
|
|
||||||
|
const delta = deltaFromEvent(event);
|
||||||
|
if (delta && !appendDelta(delta.channel, delta.text)) return;
|
||||||
|
|
||||||
|
if ((event.type === "response.completed" || event.type === "response.incomplete") && event.response) {
|
||||||
|
finalResponse = event.response;
|
||||||
|
finalEventType = event.type;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (event.type === "response.failed" && event.response) {
|
||||||
|
const message = event.response.error?.message || "LM Studio response failed.";
|
||||||
|
const error = new Error(message);
|
||||||
|
error.retryable = true;
|
||||||
|
error.partial = Boolean(content || reasoning);
|
||||||
|
fail(error);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function consumeBuffer(flush = false) {
|
||||||
|
buffer = buffer.replace(/\r\n/g, "\n");
|
||||||
|
|
||||||
|
let boundary;
|
||||||
|
while ((boundary = buffer.indexOf("\n\n")) !== -1) {
|
||||||
|
const block = buffer.slice(0, boundary);
|
||||||
|
buffer = buffer.slice(boundary + 2);
|
||||||
|
consumeEvent(block);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (flush && buffer.trim()) {
|
||||||
|
consumeEvent(buffer);
|
||||||
|
buffer = "";
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
response.on("data", chunk => {
|
||||||
|
if (settled) return;
|
||||||
|
buffer += chunk;
|
||||||
|
consumeBuffer();
|
||||||
|
});
|
||||||
|
|
||||||
|
response.on("end", () => {
|
||||||
|
if (settled) return;
|
||||||
|
consumeBuffer(true);
|
||||||
|
|
||||||
|
if (!finalEventType) {
|
||||||
|
const error = new Error("LM Studio stream ended without a completed or incomplete terminal response event.");
|
||||||
|
error.retryable = true;
|
||||||
|
error.partial = Boolean(content || reasoning);
|
||||||
|
settled = true;
|
||||||
|
reject(error);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
const fallback = replyFromResponse(finalResponse);
|
||||||
|
const reply = {
|
||||||
|
role: "assistant",
|
||||||
|
content: content || fallback.content,
|
||||||
|
status: finalEventType === "response.incomplete" ? "incomplete" : finalResponse?.status || "completed",
|
||||||
|
responseId: responseId || finalResponse?.id || null,
|
||||||
|
incompleteDetails: finalResponse?.incomplete_details || null,
|
||||||
|
usage: finalResponse?.usage || null
|
||||||
|
};
|
||||||
|
|
||||||
|
if (reasoning || fallback.reasoning) reply.reasoning = reasoning || fallback.reasoning;
|
||||||
|
settled = true;
|
||||||
|
resolve(reply);
|
||||||
|
});
|
||||||
|
|
||||||
|
response.on("error", error => {
|
||||||
|
if (settled) return;
|
||||||
|
error.retryable = true;
|
||||||
|
error.partial = Boolean(content || reasoning);
|
||||||
|
settled = true;
|
||||||
|
reject(error);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
request.on("error", error => {
|
||||||
|
error.retryable = true;
|
||||||
|
reject(error);
|
||||||
|
});
|
||||||
|
|
||||||
|
request.write(payload);
|
||||||
|
request.end();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
async function send(conversation, settings, options = {}) {
|
||||||
|
const content = [];
|
||||||
|
const reasoning = [];
|
||||||
|
const pages = [];
|
||||||
|
let previousResponseId = options.previousResponseId || null;
|
||||||
|
let continuationInput = options.continuationInput || null;
|
||||||
|
let lastReply = null;
|
||||||
|
|
||||||
|
for (let continuation = 0; continuation <= MAX_INCOMPLETE_CONTINUATIONS; continuation++) {
|
||||||
|
const page = await sendPage(conversation, settings, {
|
||||||
|
...options,
|
||||||
|
previousResponseId,
|
||||||
|
continuationInput
|
||||||
|
});
|
||||||
|
|
||||||
|
lastReply = page;
|
||||||
|
pages.push(pageDiagnostic(page));
|
||||||
|
if (page.reasoning) reasoning.push(page.reasoning);
|
||||||
|
if (page.content) content.push(page.content);
|
||||||
|
|
||||||
|
if (page.status !== "incomplete") {
|
||||||
|
const reply = { role: "assistant", content: content.join("") };
|
||||||
|
if (reasoning.length > 0) reply.reasoning = reasoning.join("");
|
||||||
|
if (page.responseId) reply.responseId = page.responseId;
|
||||||
|
reply.provider = {
|
||||||
|
name: "lmstudio",
|
||||||
|
continuationPages: pages.length - 1,
|
||||||
|
pages
|
||||||
|
};
|
||||||
|
return reply;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!page.responseId) {
|
||||||
|
const error = new Error("LM Studio returned an incomplete response without a response id; it cannot be continued safely.");
|
||||||
|
error.retryable = false;
|
||||||
|
error.partial = Boolean(content.length || reasoning.length);
|
||||||
|
throw error;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (continuation === MAX_INCOMPLETE_CONTINUATIONS) {
|
||||||
|
const error = new Error(`LM Studio response remained incomplete after ${MAX_INCOMPLETE_CONTINUATIONS} automatic continuations.`);
|
||||||
|
error.retryable = false;
|
||||||
|
error.partial = Boolean(content.length || reasoning.length);
|
||||||
|
error.provider = { name: "lmstudio", continuationPages: pages.length - 1, pages };
|
||||||
|
throw error;
|
||||||
|
}
|
||||||
|
|
||||||
|
previousResponseId = page.responseId;
|
||||||
|
continuationInput = INCOMPLETE_CONTINUE_PROMPT;
|
||||||
|
}
|
||||||
|
|
||||||
|
return lastReply;
|
||||||
|
}
|
||||||
|
|
||||||
|
module.exports = {
|
||||||
|
warmUp,
|
||||||
|
send
|
||||||
|
};
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
{
|
||||||
|
"name": "perry",
|
||||||
|
"version": "0.1.0",
|
||||||
|
"private": true,
|
||||||
|
"description": "Small persistent conversational-agent utility for local LLM work.",
|
||||||
|
"main": "index.js",
|
||||||
|
"scripts": {
|
||||||
|
"test": "node --test"
|
||||||
|
},
|
||||||
|
"license": "UNLICENSED"
|
||||||
|
}
|
||||||
@@ -0,0 +1,70 @@
|
|||||||
|
const REPEAT_COUNT = 7;
|
||||||
|
const MAX_PATTERN_TOKENS = 256;
|
||||||
|
const MAX_BUFFER_CHARS = 32768;
|
||||||
|
|
||||||
|
function tokenize(text) {
|
||||||
|
return String(text ?? "").match(/<[^>\s]+>|[\p{L}\p{M}\p{N}_]+|[^\s\p{L}\p{M}\p{N}_]+/gu) || [];
|
||||||
|
}
|
||||||
|
|
||||||
|
function sameBlock(tokens, aStart, bStart, length) {
|
||||||
|
for (let i = 0; i < length; i++) {
|
||||||
|
if (tokens[aStart + i] !== tokens[bStart + i]) return false;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
function findRepeatedTail(tokens, options = {}) {
|
||||||
|
const repeatCount = options.repeatCount ?? REPEAT_COUNT;
|
||||||
|
const maxPatternTokens = options.maxPatternTokens ?? MAX_PATTERN_TOKENS;
|
||||||
|
if (repeatCount < 2) throw new Error("repeatCount must be at least 2");
|
||||||
|
|
||||||
|
const maxPeriod = Math.min(maxPatternTokens, Math.floor(tokens.length / repeatCount));
|
||||||
|
for (let periodTokens = 1; periodTokens <= maxPeriod; periodTokens++) {
|
||||||
|
const tailStart = tokens.length - (periodTokens * repeatCount);
|
||||||
|
const referenceStart = tokens.length - periodTokens;
|
||||||
|
let repeated = true;
|
||||||
|
|
||||||
|
for (let repeat = 0; repeat < repeatCount - 1; repeat++) {
|
||||||
|
const candidateStart = tailStart + (repeat * periodTokens);
|
||||||
|
if (!sameBlock(tokens, candidateStart, referenceStart, periodTokens)) {
|
||||||
|
repeated = false;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!repeated) continue;
|
||||||
|
const pattern = tokens.slice(referenceStart);
|
||||||
|
return {
|
||||||
|
repeatCount,
|
||||||
|
periodTokens,
|
||||||
|
pattern,
|
||||||
|
sample: pattern.join(" ").slice(0, 240)
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
function createDetector(options = {}) {
|
||||||
|
const maxBufferChars = options.maxBufferChars ?? MAX_BUFFER_CHARS;
|
||||||
|
let buffer = "";
|
||||||
|
|
||||||
|
return {
|
||||||
|
push(text) {
|
||||||
|
buffer += String(text ?? "");
|
||||||
|
if (buffer.length > maxBufferChars) buffer = buffer.slice(-maxBufferChars);
|
||||||
|
return findRepeatedTail(tokenize(buffer), options);
|
||||||
|
},
|
||||||
|
reset() {
|
||||||
|
buffer = "";
|
||||||
|
}
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
module.exports = {
|
||||||
|
REPEAT_COUNT,
|
||||||
|
MAX_PATTERN_TOKENS,
|
||||||
|
tokenize,
|
||||||
|
findRepeatedTail,
|
||||||
|
createDetector
|
||||||
|
};
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
const test = require("node:test");
|
||||||
|
const assert = require("node:assert/strict");
|
||||||
|
const { tokenize, findRepeatedTail, createDetector } = require("../repetition");
|
||||||
|
|
||||||
|
function detect(text, options) {
|
||||||
|
return findRepeatedTail(tokenize(text), options);
|
||||||
|
}
|
||||||
|
|
||||||
|
test("ordinary prose does not trigger", () => {
|
||||||
|
assert.equal(detect("The same word may appear again, but ordinary prose keeps moving forward."), null);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("a phrase may recur fewer than seven times", () => {
|
||||||
|
assert.equal(detect("try again. ".repeat(6)), null);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("one token repeated seven times triggers", () => {
|
||||||
|
const hit = detect("no ".repeat(7));
|
||||||
|
assert.ok(hit);
|
||||||
|
assert.equal(hit.repeatCount, 7);
|
||||||
|
assert.equal(hit.periodTokens, 1);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("special token garbage repeated seven times triggers", () => {
|
||||||
|
const hit = detect("<unused49>".repeat(7));
|
||||||
|
assert.ok(hit);
|
||||||
|
assert.equal(hit.periodTokens, 1);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("a multi-token phrase repeated seven times triggers", () => {
|
||||||
|
const hit = detect("let us try again. ".repeat(7));
|
||||||
|
assert.ok(hit);
|
||||||
|
assert.ok(hit.periodTokens > 1);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("a larger block repeated seven times triggers", () => {
|
||||||
|
const block = "I will use this wording, but no, that is wrong. Let me correct it. ";
|
||||||
|
const hit = detect(block.repeat(7));
|
||||||
|
assert.ok(hit);
|
||||||
|
assert.ok(hit.periodTokens > 5);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("meaningful token variance prevents a false positive", () => {
|
||||||
|
const text = Array.from({ length: 12 }, (_, i) => `Attempt ${i + 1}: I reconsidered the wording and chose option ${i + 1}.`).join(" ");
|
||||||
|
assert.equal(detect(text), null);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("detector works across arbitrary stream chunk boundaries", () => {
|
||||||
|
const detector = createDetector();
|
||||||
|
const text = "Okay, use אש instead. ".repeat(7);
|
||||||
|
let hit = null;
|
||||||
|
for (let i = 0; i < text.length; i += 3) {
|
||||||
|
hit = detector.push(text.slice(i, i + 3)) || hit;
|
||||||
|
}
|
||||||
|
assert.ok(hit);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("pure formatting does not accidentally tokenize as many punctuation characters", () => {
|
||||||
|
assert.deepStrictEqual(tokenize("----------"), ["----------"]);
|
||||||
|
assert.equal(detect("----------"), null);
|
||||||
|
});
|
||||||
Reference in New Issue
Block a user