Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions libraries/llm/.gitignore
Original file line number Diff line number Diff line change
@@ -1 +1,2 @@
types/
.pushwork
26 changes: 26 additions & 0 deletions libraries/llm/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -102,6 +102,32 @@ Pass a stable `sessionKey` (e.g. a doc URL) to `generate`/`stream`; after a
reload, `resume(sessionKey, { onToken, onDone })` re-attaches to the still-running
stream in the worker.

### Request preparation (`request.js`)

`generate()` is two halves: preparing the request and posting it to the worker.
The first half is exported so a host-side worker spec (one that serves this
library over a stream to sandboxed tools) prepares requests exactly the way
`generate()` does:

```js
import { prepareGenerate, buildGeneratePayload } from "@patchwork/llm"

const { cfg, config, extraSystem, builtin, input } = await prepareGenerate(cfg0, {
messages, // chat messages or a string
system, // extra system prompt, already provider-resolved
tools, // tool descriptors, already filtered
continuation, // treat a string as a raw continuation
overrides, // passed to callConfig verbatim
})
if (builtin) { /* run builtinGenerate(input, { system: effectiveSystem(cfg, extraSystem), … }) */ }
else worker.postMessage(buildGeneratePayload(config, input, { id, sessionKey }))
```

Resolving a provider-conditional `system` map, filtering tools by the user's
`toolToggles`, and choosing which `overrides` to forward are the caller's job:
`callConfig` honours provider / apiKey / model / url overrides, so a host serving
untrusted tools passes only sampling knobs.

## Events reference

| event | fields | local | openrouter | ollama |
Expand Down
72 changes: 17 additions & 55 deletions libraries/llm/client.js
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,8 @@

import {readConfig, ensureConfig, callConfig, applyPrompts, effectiveSystem} from "./config.js"
import {builtinGenerate} from "./builtin.js"
import {resolveTools, toToolSchemas, buildToolsSystem, parseToolCalls, runTool, resolveCfgPrompts, sanitizeToolName} from "./tools.js"
import {resolveTools, buildToolsSystem, parseToolCalls, runTool, resolveCfgPrompts, sanitizeToolName} from "./tools.js"
import {prepareGenerate, buildGeneratePayload, NATIVE_TOOL_PROVIDERS, TEMPLATE_TOOL_PROVIDERS} from "./request.js"

/**
* Per-call options. A superset of every option any exported function accepts;
Expand Down Expand Up @@ -66,16 +67,7 @@ import {resolveTools, toToolSchemas, buildToolsSystem, parseToolCalls, runTool,

/** @typedef {{post: (m:any)=>void}} Connection */

/** CallConfig plus the extra mutable fields the client tacks on per-call.
* @typedef {import("./config.js").CallConfig & {tools?:any, toolSystem?:string, continuation?:boolean}} CallConfigExt */

/** @typedef {{type:string, id:string, sessionKey:string, provider:import("./config.js").ProviderId, config:import("./config.js").CallConfig, text?:string, messages?:any}} GeneratePayload */

// Providers with real function-calling APIs (the worker passes tool schemas and
// parses structured tool_calls). Everything else (local transformers, Chrome
// built-in) uses the <tool_call> XML prompt convention, parsed from the text.
const NATIVE_TOOL_PROVIDERS = new Set(["openrouter", "ollama", "webllm"])
const TEMPLATE_TOOL_PROVIDERS = new Set(["local"])
/** @typedef {import("./request.js").CallConfigExt} CallConfigExt */

/** @type {Connection|null} */
let connection = null
Expand Down Expand Up @@ -196,34 +188,21 @@ function getConnection() {
*/
export async function generate(messages, opts = {}) {
const cfg0 = opts.config ?? (await ensureConfig(opts.scope))
// Resolve the selected system/pre prompt docs → their text. repo.find is
// cached, so this is cheap after first load.
const cfg = await resolveCfgPrompts(cfg0)
/** @type {CallConfigExt} */
const config = callConfig(/** @type {any} */ (cfg), /** @type {any} */ (opts))

// Tools: native providers get JSON schemas on `config.tools`; the rest get the
// <tool_call> XML convention prepended to the system prompt (parsed from text).
const hasTools = Array.isArray(opts.tools) && opts.tools.length > 0
const native = hasTools && NATIVE_TOOL_PROVIDERS.has(config.provider)
const templated = hasTools && TEMPLATE_TOOL_PROVIDERS.has(config.provider)
if (native || templated) config.tools = toToolSchemas(opts.tools)
if (templated) config.toolSystem = buildToolsSystem(opts.tools)
const extraSystem =
hasTools && !native && !templated
? [buildToolsSystem(opts.tools), opts.system].filter(Boolean).join("\n\n")
: opts.system
// Prompt resolution, CallConfig, the tools decision, the builtin detour and
// the chat-vs-continuation input shape all live in request.js, shared with
// host-side worker specs. The whole opts is passed as overrides: a same-realm
// caller legitimately owns provider/model/apiKey overrides.
const {cfg, config, extraSystem, builtin, input} = await prepareGenerate(cfg0, {
messages,
system: opts.system,
tools: opts.tools,
continuation: opts.continuation,
overrides: opts,
})

// Built-in (Chrome Prompt API) runs on the main thread, not the worker.
if (config.provider === "builtin") {
const pre = cfg.resolved?.pre || ""
const text =
typeof messages === "string"
? pre
? pre + "\n\n" + messages
: messages
: messages
return builtinGenerate(text, {
if (builtin) {
return builtinGenerate(input, {
temperature: config.temperature,
topK: config.topK,
system: effectiveSystem(/** @type {any} */ (cfg), extraSystem),
Expand All @@ -233,19 +212,6 @@ export async function generate(messages, opts = {}) {
}).then((t) => ({text: t, toolCalls: null, stats: null}))
}

// A string input is CHAT by default — wrapped as a user turn, so instruct/chat
// models respond normally and the system prompt applies. It's a raw
// CONTINUATION only when opts.continuation is set: raw-fed for
// local/webllm/ollama, and CONTINUE_SYS-framed for chat-only OpenRouter (see
// the worker). Loom passes continuation:true; other callers get plain chat.
const asContinuation = !!opts.continuation && typeof messages === "string"
const prepared = asContinuation
? messages
: typeof messages === "string"
? [{role: "user", content: messages}]
: messages
// Prepend the configured system + pre-prompt (and any tool-supplied system).
const input = applyPrompts(prepared, /** @type {any} */ (cfg), extraSystem)
const conn = getConnection()
const id = nextId()
const sessionKey = opts.sessionKey || id
Expand Down Expand Up @@ -296,11 +262,7 @@ export async function generate(messages, opts = {}) {
opts.signal.addEventListener("abort", onAbort)
}
// A string is a raw continuation prompt; an array is chat messages.
/** @type {GeneratePayload} */
const payload = {type: "generate", id, sessionKey, provider: config.provider, config}
if (typeof input === "string") payload.text = input
else payload.messages = input
conn.post(payload)
conn.post(buildGeneratePayload(config, input, {id, sessionKey}))
})
}

Expand Down
179 changes: 102 additions & 77 deletions libraries/llm/index.js
Original file line number Diff line number Diff line change
Expand Up @@ -21,92 +21,117 @@
*/

export {
// config (account doc + patchwork:llm-config provider)
readConfig,
writeConfig,
callConfig,
normalizeConfig,
subscribeConfig,
resolveConfig,
ensureSettingsDoc,
ensureConfig,
// per-tool / per-doc whole-config overrides
scopedRaw,
hasScopeOverride,
readScopedConfig,
writeScopeOverride,
clearScopeOverride,
settingsDocHandle,
applyPrompts,
effectiveSystem,
DEFAULTS,
PARAM_KEYS,
PROVIDER_CAPS,
TOOL_STORAGE_ID,
CONFIG_SELECTOR,
// catalogues / labels
LOCAL_MODELS,
WEBLLM_MODELS,
fetchOpenRouterModels,
fetchOllamaModels,
describeConfig,
} from "./config.js"
// config (account doc + patchwork:llm-config provider)
readConfig,
writeConfig,
callConfig,
normalizeConfig,
subscribeConfig,
resolveConfig,
ensureSettingsDoc,
ensureConfig,
// per-tool / per-doc whole-config overrides
scopedRaw,
hasScopeOverride,
readScopedConfig,
writeScopeOverride,
clearScopeOverride,
settingsDocHandle,
applyPrompts,
effectiveSystem,
DEFAULTS,
PARAM_KEYS,
PROVIDER_CAPS,
TOOL_STORAGE_ID,
CONFIG_SELECTOR,
// catalogues / labels
LOCAL_MODELS,
WEBLLM_MODELS,
fetchOpenRouterModels,
fetchOllamaModels,
describeConfig,
} from "./config.js";

export {
generate,
generateWithTools,
stream,
predict,
scoreTokens,
preload,
abort,
resume,
onStatus,
registerLocalModel,
computeImportance,
computeAttentionWeights,
extractFeatures,
extractCutFeatures,
decodeTokens,
probeAttention,
} from "./client.js"
generate,
generateWithTools,
stream,
predict,
scoreTokens,
preload,
abort,
resume,
onStatus,
registerLocalModel,
computeImportance,
computeAttentionWeights,
extractFeatures,
extractCutFeatures,
decodeTokens,
probeAttention,
} from "./client.js";

export {dom, popup} from "./picker.js"
export { dom, popup } from "./picker.js";

export {builtinSupported, builtinAvailability} from "./builtin.js"
export {
builtinSupported,
builtinAvailability,
builtinGenerate,
} from "./builtin.js";

// Construct this library's compute worker. This has to live in the library:
// `new URL("./worker.js", import.meta.url)` resolves against this module's own
// URL, so it finds the sibling worker.js wherever the library is served from,
// and bundlers only emit the worker chunk when they see this exact literal.
//
// import { createWorker } from "@chee/patchwork-llm";
// const worker = createWorker();
// worker.onmessage = (e) => console.log(e.data.models);
// worker.postMessage({ type: "list-local-models" });
export const createWorker = () =>
new Worker(new URL("./worker.js", import.meta.url), { type: "module" });

// Transport-neutral request preparation, shared by generate() and by
// host-side worker specs that serve this library over a stream.
export {
prepareGenerate,
buildGeneratePayload,
NATIVE_TOOL_PROVIDERS,
TEMPLATE_TOOL_PROVIDERS,
} from "./request.js";

// LLM tools (user-defined tools the model can be given)
export {
createLLMTool,
createToolFile,
LLMToolDatatype,
sanitizeToolName,
resolveTools,
toToolSchemas,
buildToolsSystem,
parseToolCalls,
loadHandler,
runTool,
runHandlerSandboxed,
// saved prompts (system + pre), same doc shape as tools
createPromptDoc,
resolvePromptDocs,
resolvePromptText,
resolveCfgPrompts,
LLMSystemPromptDatatype,
LLMPrePromptDatatype,
// folders + one-time migration
ensureFolderUrl,
addToFolder,
removeFromFolder,
migrateConfig,
} from "./tools.js"
createLLMTool,
createToolFile,
LLMToolDatatype,
sanitizeToolName,
resolveTools,
toToolSchemas,
buildToolsSystem,
parseToolCalls,
loadHandler,
runTool,
runHandlerSandboxed,
// saved prompts (system + pre), same doc shape as tools
createPromptDoc,
resolvePromptDocs,
resolvePromptText,
resolveCfgPrompts,
LLMSystemPromptDatatype,
LLMPrePromptDatatype,
// folders + one-time migration
ensureFolderUrl,
addToFolder,
removeFromFolder,
migrateConfig,
} from "./tools.js";

// Registers <patchwork-llm-config-provider> on import.
export {
PatchworkLLMConfigProvider,
definePatchworkLLMConfigProvider,
} from "./provider.js"
PatchworkLLMConfigProvider,
definePatchworkLLMConfigProvider,
} from "./provider.js";

// Built-in prompt templates
export {PROMPT_TEMPLATES} from "./templates.js"
export { PROMPT_TEMPLATES } from "./templates.js";
6 changes: 4 additions & 2 deletions libraries/llm/package.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
{
"name": "@chee/patchwork-llm",
"version": "0.2.1",
"version": "0.3.0",
"description": "LLM toolkit for Patchwork tools: a <dialog> model picker, a refresh-surviving SharedWorker that runs local (transformers.js) / OpenRouter / Ollama generation, and a streaming API that carries rich telemetry — next-token predictions, temperature, tokens/sec — alongside the text so you can build UIs that show how the model thinks.",
"type": "module",
"main": "index.js",
Expand All @@ -17,13 +17,15 @@
},
"scripts": {
"build": "pnpm build:types",
"build:types": "tsc"
"build:types": "tsc",
"push": "pnpm build && pushwork sync"
},
"files": [
"types",
"index.js",
"config.js",
"client.js",
"request.js",
"worker.js",
"picker.js",
"tools.js",
Expand Down
Loading
Loading