File size: 5,634 Bytes
1944112
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
/**
 * `CreateScheduledEngine` — the one-line migration off WebLLM.
 *
 * ```js
 * -import { CreateMLCEngine } from "@mlc-ai/web-llm";
 * -const engine = await CreateMLCEngine(modelId, { initProgressCallback });
 * +import { CreateScheduledEngine } from "everything-webgpu";
 * +const engine = await CreateScheduledEngine(modelId, { initProgressCallback });
 * ```
 *
 * Everything after that line is unchanged: `engine.chat.completions.create()`
 * takes and returns the same shapes. What the swap buys is the scheduler
 * (`session`, `priority`, `task`, `preemptible` on any request), multi-step
 * decoding, the two build-time patches, and models from disk as well as the
 * network.
 *
 * The signature mirrors `CreateMLCEngine`'s so nothing else has to move —
 * including `initProgressCallback`, which is why this is a function rather than
 * documentation telling people to construct a `ScheduledEngine` themselves.
 */
import { ScheduledEngine } from "./engine.js";
import { ERROR, EngineError } from "./errors.js";
import { ModelStore } from "./model-store.js";

/**
 * @param {string} [modelId] loaded before returning, as `CreateMLCEngine` does.
 *   Omit to get an engine that loads later.
 * @param {object} [opts]
 * @param {ModelStore | import("./model-store.js").StorageAdapter} [opts.store]
 *   defaults to IndexedDB. Pass one explicitly in a worker or a test.
 * @param {(report: {text: string, progress: number}) => void} [opts.initProgressCallback]
 * @param {string | URL} [opts.workerUrl]
 * @param {() => Promise<object>} [opts.loadWebLLM]
 * @param {boolean} [opts.prebuilt]
 * @returns {Promise<ScheduledEngine>}
 */
export async function CreateScheduledEngine(modelId, opts = {}) {
  const { store, initProgressCallback, ...rest } = opts;
  const engine = new ScheduledEngine({ ...rest, store: store ?? (await defaultStore()) });
  if (!modelId) return engine;

  // Three states, not two. `undefined` means the caller did not choose, and the
  // right default for a call that blocks for minutes on a multi-hundred-MB
  // download is to say so: the first run of the three-line example otherwise
  // looks exactly like a hung process, which is the worst possible first
  // impression and the one thing no amount of documentation undoes. `null` is
  // how you ask for silence.
  const report =
    initProgressCallback === undefined ? defaultProgressReporter(modelId) : initProgressCallback;

  if (!report) {
    await engine.load(modelId);
    return engine;
  }

  // The engine reports progress through its own lifecycle stream; WebLLM
  // reports it through a callback. Forward only the load reports, so a
  // caller's callback fires exactly when WebLLM's would have.
  const stop = engine.subscribe((state) => {
    if (state.progress) report(state.progress);
  });
  try {
    await engine.load(modelId);
  } finally {
    stop();
    report.done?.();
  }
  return engine;
}

/**
 * What you get when you say nothing: a throttled line on the console.
 *
 * Deliberately not the engine's own behaviour — `new ScheduledEngine()` stays
 * silent, because a library core that logs is wrong inside a worker, an
 * extension background page or a test. This is the getting-started facade, and
 * the person calling it has not yet decided how to render anything.
 *
 * Throttled to one line a second, because WebLLM's callback fires per shard and
 * a 58-shard model would otherwise bury the console. The 100% report is never
 * dropped, so the last line always reads as finished rather than as 97%.
 */
function defaultProgressReporter(modelId) {
  if (typeof console === "undefined") return null;

  const startedAt = Date.now();
  let announced = false;
  let lastAt = 0;
  let sawProgress = false;

  const emit = ({ text, progress }) => {
    sawProgress = true;
    if (!announced) {
      announced = true;
      console.info(
        `[everything-webgpu] loading ${modelId} — the first run downloads the weights and can take ` +
          "minutes; later loads read the cache and need no network. " +
          "Pass initProgressCallback to render this yourself, or null to silence it.",
      );
    }
    const now = Date.now();
    const finished = (progress ?? 0) >= 1;
    if (!finished && now - lastAt < 1000) return;
    lastAt = now;
    const pct = `${Math.round((progress ?? 0) * 100)}%`.padStart(4);
    console.info(`[everything-webgpu] ${pct}  ${text ?? ""}`.trimEnd());
  };

  // Called once when the load settles, so the common case ends on a line that
  // says it worked rather than trailing off mid-progress.
  emit.done = () => {
    if (!sawProgress) return;
    console.info(
      `[everything-webgpu] ${modelId} ready in ${((Date.now() - startedAt) / 1000).toFixed(1)}s.`,
    );
  };
  return emit;
}

/**
 * IndexedDB, because a registry that dies with the page would strand the
 * weights: the bytes stay in Cache Storage but nothing remembers they are
 * there. Anywhere IndexedDB is missing, the caller has to say what to use —
 * guessing would produce exactly that stranding, silently.
 */
async function defaultStore() {
  if (typeof indexedDB === "undefined") {
    throw new EngineError(
      ERROR.BAD_REQUEST,
      "No IndexedDB in this context, so there is no default store. " +
        "Pass `store` — `memoryStorage()` for a test, or your own StorageAdapter.",
    );
  }
  // Imported lazily so the engine core carries no static dependency on an
  // adapter, and a host that brings its own store never bundles this one.
  const { indexedDBStorage } = await import("../adapters/idb.js");
  return new ModelStore(await indexedDBStorage());
}