An AI that works with your Wi-Fi off
Like ChatGPT, but it runs on your own computer. Pick a quick model or a better one, load it once, switch off Wi-Fi, and keep asking. Watch the counter: nothing you type is sent anywhere.
Sandbox
Under the hood
Where it runs. When you use ChatGPT, what you type travels to a company’s servers and the answer comes back. Here it’s the other way round: you pick a quick 210 MB model or a better 830 MB one, and it downloads into your browser. It runs on your computer’s graphics chip (the quick one uses about 380 MB of memory). Nothing gets installed, and closing the tab stops it.
No Wi-Fi needed. Once it’s loaded, everything happens on your machine, so you can switch off Wi-Fi and keep asking. The counter shows every request this page makes, and it stays at 0 while the model answers. We tested it with Wi-Fi off: 3 prompts, 3 answers, counter at 0. On your next visit it loads from your browser in about a second.
The catch. The quick model is tiny next to the ones behind ChatGPT, and it shows: asked “Hey”, it introduced itself as Aliza from Cambridge University. The better 830 MB model kept all the facts in 3 of 3 email summaries in our test, at the cost of a longer first load (about 77 s on a fast connection in our test). Both can still slip.
Built with WebLLM, an open-source engine for running AI in the browser, using the SmolLM2-360M model by Hugging Face and the Qwen2.5-1.5B model by Alibaba.
Fork
private-ai.js// A small AI model running on your GPU, in this tab. Nothing downloads until you pick a model.
// mount(el) builds the whole thing inside el and returns a cleanup function.
// Each choice has a q4f16 build and a q4f32 fallback for GPUs without shader-f16.
const MODELS = {
quick: { name: 'Quick (SmolLM2-360M)', f16: 'SmolLM2-360M-Instruct-q4f16_1-MLC', f32: 'SmolLM2-360M-Instruct-q4f32_1-MLC' },
better: { name: 'Better (Qwen2.5-1.5B)', f16: 'Qwen2.5-1.5B-Instruct-q4f16_1-MLC', f32: 'Qwen2.5-1.5B-Instruct-q4f32_1-MLC' },
};
export function mount(el) {
el.innerHTML = `
<p class="msg" hidden>Your browser can't run this. Try Chrome or Safari 26 on a laptop.</p>
<div data-notice style="margin:0 0 16px;padding:12px;border:1px solid var(--line);border-radius:var(--radius-sm);font:13px/1.5 var(--font-mono)">
<div class="t-label" style="color:var(--ink-muted);margin-bottom:8px">Online first, then offline</div>
<p style="margin:0 0 4px;color:var(--ink)"><span style="color:var(--ink-muted)">1. ONLINE</span> Load a model. It needs internet once, to download it.</p>
<p style="margin:0 0 8px;color:var(--ink)"><span style="color:var(--ink-muted)">2. OFFLINE</span> Then switch off Wi-Fi and keep asking, on a plane or anywhere.</p>
<p style="margin:0;color:var(--state-poking)">Don't refresh the page while offline. A refresh needs internet, and you'll lose the model until you're back online.</p>
</div>
<div data-choices style="display:flex;flex-wrap:wrap;gap:16px">
<div style="flex:1 1 200px;min-width:0">
<button class="btn btn-primary" type="button" data-load="quick">Quick: 210 MB</button>
<p style="margin:8px 0 0;font:13px/1.5 var(--font-mono);color:var(--ink-muted)">Fast to load, often makes things up.</p>
</div>
<div style="flex:1 1 200px;min-width:0">
<button class="btn" type="button" data-load="better">Better: 830 MB</button>
<p style="margin:8px 0 0;font:13px/1.5 var(--font-mono);color:var(--ink-muted)">Slower first load, kept the facts in our tests.</p>
</div>
</div>
<p class="readout" data-prog style="margin:12px 0 0"></p>
<div data-chat hidden>
<textarea rows="4" placeholder="Ask it something" aria-label="Prompt"
style="display:block;box-sizing:border-box;width:100%;margin-top:16px;padding:8px;background:transparent;color:var(--ink);border:1px solid var(--ink-faint);border-radius:var(--radius-sm);font:13px/1.5 var(--font-mono);resize:vertical"></textarea>
<p style="margin:12px 0"><button class="btn" type="button" data-run>Run</button></p>
<pre data-out aria-live="polite" style="margin:0;min-height:3em;white-space:pre-wrap;overflow-wrap:anywhere;font:13px/1.5 var(--font-mono);color:var(--ink)"></pre>
<div class="controls">
<span class="readout">network requests since model ready: <output data-net style="min-width:0">0</output></span>
<span data-online></span>
<span data-warn style="color:var(--state-poking)" hidden>Don't refresh while offline.</span>
</div>
<p style="margin:8px 0 0;font:13px var(--font-mono);color:var(--ink-muted)">Turn off Wi-Fi and run it again. Watch the counter.</p>
</div>`;
const $ = (s) => el.querySelector(s);
const [msg, choices, prog, chat, prompt, run, out, net, online, warn, notice] =
['.msg', '[data-choices]', '[data-prog]', '[data-chat]', 'textarea', '[data-run]', '[data-out]', '[data-net]', '[data-online]', '[data-warn]', '[data-notice]'].map($);
const buttons = el.querySelectorAll('[data-load]');
let engine = null, observer = null, count = 0, cancelled = false;
const showOnline = () => {
online.textContent = navigator.onLine ? 'online' : 'offline';
warn.hidden = navigator.onLine || !engine; // reminder only once a model is loaded
};
window.addEventListener('online', showOnline);
window.addEventListener('offline', showOnline);
showOnline();
let adapter = null;
(async () => {
try { adapter = navigator.gpu && (await navigator.gpu.requestAdapter()); } catch {}
if (!adapter) {
msg.hidden = false;
choices.hidden = true;
notice.hidden = true;
}
})();
const choose = async (key) => {
const choice = MODELS[key];
buttons.forEach((b) => (b.disabled = true));
try {
// Imported only now, after the click, so nothing downloads on page load.
const webllm = await import(/* @vite-ignore */ 'https://esm.run/@mlc-ai/[email protected]');
const model = adapter.features.has('shader-f16') ? choice.f16 : choice.f32;
const e = await webllm.CreateMLCEngine(model, {
// WebLLM's own progress text is technical; show a plain percentage instead.
initProgressCallback: (p) =>
(prog.textContent = p.progress < 1 ? `Loading: ${Math.round(p.progress * 100)}%` : 'Almost ready'),
});
if (cancelled) return e.unload();
engine = e;
// From here on, every request this page makes shows up in the counter.
observer = new PerformanceObserver((list) => (net.textContent = count += list.getEntries().length));
observer.observe({ type: 'resource', buffered: false });
prog.textContent = `Running: ${choice.name}. Ready. Ask it something.`;
choices.hidden = true;
chat.hidden = false;
showOnline();
} catch (err) {
prog.textContent = `Error: ${err?.message || err}`;
buttons.forEach((b) => (b.disabled = false));
}
};
buttons.forEach((b) => b.addEventListener('click', () => choose(b.dataset.load)));
run.addEventListener('click', async () => {
run.disabled = true;
out.textContent = '';
try {
const stream = await engine.chat.completions.create({
messages: [{ role: 'user', content: prompt.value }],
stream: true,
});
for await (const chunk of stream) out.textContent += chunk.choices[0]?.delta?.content || '';
} catch (err) {
out.textContent = `Error: ${err?.message || err}`;
}
run.disabled = false;
});
return () => {
cancelled = true;
observer?.disconnect();
window.removeEventListener('online', showOnline);
window.removeEventListener('offline', showOnline);
engine?.unload();
};
}