chore: import upstream snapshot with attribution
Build / build (push) Has been cancelled
Tests / test (push) Has been cancelled
Build site and push to gh-pages / Build site (push) Has been cancelled
Linter / lint (push) Has been cancelled
Security / dependency-review (push) Has been cancelled
Security / npm-audit (push) Has been cancelled
Security / codeql (push) Has been cancelled

This commit is contained in:
wehub-resource-sync
2026-07-13 12:42:51 +08:00
commit f73e710e38
276 changed files with 36598 additions and 0 deletions
+20
View File
@@ -0,0 +1,20 @@
# Minimal makefile for Sphinx documentation
#
# You can set these variables from the command line, and also
# from the environment for the first two.
SPHINXOPTS ?=
SPHINXBUILD ?= python -m sphinx
SOURCEDIR = .
BUILDDIR = _build
# Put it first so that "make" without argument is like "make help".
help:
@$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
.PHONY: help Makefile
# Catch-all target: route all unknown targets to Sphinx using the new
# "make mode" option. $(O) is meant as a shortcut for $(SPHINXOPTS).
%: Makefile
@$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
+30
View File
@@ -0,0 +1,30 @@
# WebLLM Documentation
The documentation was built upon [Sphinx](https://www.sphinx-doc.org/en/master/).
## Dependencies
Run the following command in this directory to install dependencies first:
```bash
pip3 install -r requirements.txt
```
## Build the Documentation
Then you can build the documentation by running:
```bash
make html
```
## View the Documentation
Run the following command to start a simple HTTP server:
```bash
cd _build/html
python3 -m http.server
```
Then you can view the documentation in your browser at `http://localhost:8000` (the port can be customized by appending ` -p PORT_NUMBER` in the python command above).
+87
View File
@@ -0,0 +1,87 @@
<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<svg
xmlns:i="&amp;#38;ns_ai;"
xmlns:dc="http://purl.org/dc/elements/1.1/"
xmlns:cc="http://creativecommons.org/ns#"
xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
xmlns:svg="http://www.w3.org/2000/svg"
xmlns="http://www.w3.org/2000/svg"
xmlns:sodipodi="http://sodipodi.sourceforge.net/DTD/sodipodi-0.dtd"
xmlns:inkscape="http://www.inkscape.org/namespaces/inkscape"
version="1.1"
id="Layer_1"
x="0px"
y="0px"
viewBox="0 0 523.56958 171.35398"
xml:space="preserve"
sodipodi:docname="mlc-logo-with-text-landscape.svg"
width="523.56958"
height="171.35399"
inkscape:version="1.0.1 (1.0.1+r75)"><metadata
id="metadata23"><rdf:RDF><cc:Work
rdf:about=""><dc:format>image/svg+xml</dc:format><dc:type
rdf:resource="http://purl.org/dc/dcmitype/StillImage" /></cc:Work></rdf:RDF></metadata><defs
id="defs21" /><sodipodi:namedview
pagecolor="#ffffff"
bordercolor="#666666"
borderopacity="1"
objecttolerance="10"
gridtolerance="10"
guidetolerance="10"
inkscape:pageopacity="0"
inkscape:pageshadow="2"
inkscape:window-width="1853"
inkscape:window-height="1025"
id="namedview19"
showgrid="false"
inkscape:zoom="1.42"
inkscape:cx="201.79468"
inkscape:cy="119.53208"
inkscape:window-x="67"
inkscape:window-y="27"
inkscape:window-maximized="1"
inkscape:current-layer="Layer_1"
inkscape:document-rotation="0" />
<style
type="text/css"
id="style2">
.st0{fill-rule:evenodd;clip-rule:evenodd;fill:#062578;}
.st1{fill:#062578;}
</style>
<switch
id="switch16"
transform="translate(19.160879,-130.46791)">
<foreignObject
requiredExtensions="http://ns.adobe.com/AdobeIllustrator/10.0/"
x="0"
y="0"
width="1"
height="1">
</foreignObject>
<g
i:extraneous="self"
id="g14">
<g
id="g12">
<path
class="st0"
d="M 124.8,208.2 H 82.7 c -1.4,0 -2.6,1.2 -2.6,2.6 v 7.5 7.5 3.4 c 1.7,-0.2 3.3,0.1 4.7,0.9 v -11.8 -5.4 h 37.9 v 49.9 H 84.8 v -4.6 -11.1 c -1.1,0.8 -2.3,1.4 -3.4,2 -0.4,0.2 -0.8,0.4 -1.2,0.5 v 9.5 3.6 2.1 c 0,1.4 1.2,2.6 2.6,2.6 h 42.1 c 1.4,0 2.6,-1.1 2.6,-2.6 v -54.2 c -0.1,-1.2 -1.3,-2.4 -2.7,-2.4 z m -44.6,38.5 c -3,1.4 -8.3,2.7 -11.5,3.4 -3.6,0.8 -14.2,3.2 -15.1,-2.5 -0.7,-4.7 11.6,-9.9 14.8,-11.3 2.9,-1.3 5.9,-2.4 8.8,-3.6 3.2,-1.2 6.6,-1.7 8.2,2.3 0.7,1.7 0.9,3.2 0.9,5 v 0.2 0.2 c -0.4,3.1 -3.4,5.1 -6.1,6.3 z m 10.6,-51.3 v 0 c 1.4,-0.1 2.7,0.9 2.8,2.4 l 0.4,4.9 c 0.1,1.4 -0.9,2.7 -2.4,2.8 v 0 c -1.4,0.1 -2.7,-0.9 -2.8,-2.4 l -0.4,-4.9 c -0.1,-1.4 1,-2.6 2.4,-2.8 z m 18.8,10.9 c 0,-2.9 -0.2,-5.7 -0.4,-8.4 v 0 c -0.1,-0.9 -0.2,-1.8 -0.3,-2.7 v -0.1 -0.2 c -0.5,-3.3 -1.1,-6.6 -1.9,-10 -0.8,-3.3 -3.7,-5.4 -7,-5.5 -6.1,-0.1 -12.3,-0.2 -18.4,0 -1,-1.9 -4.2,-3.1 -7.9,-2.9 L 69.5,167 c 0.8,-0.7 1.2,-1.8 1.1,-3 -0.2,-2 -1.9,-3.5 -3.9,-3.3 -2,0.2 -3.5,1.9 -3.3,3.9 0.2,1.9 1.7,3.3 3.6,3.3 l 4,9 c -2.6,0.7 -4.6,2.2 -5.1,3.9 -5.8,0.8 -11.5,2 -17.2,3.1 -0.7,0.1 -1.3,0.4 -1.9,0.6 -6.4,1.6 -13,5.1 -13,5.1 -0.1,0.8 -0.2,1.7 -0.2,2.5 0.3,-0.1 0.6,-0.1 1,-0.2 5,-0.4 9.6,4.7 10.2,11.5 0.6,6.8 -3,12.7 -8,13.1 -0.4,0 -0.8,0 -1.1,0 0.2,0.9 0.4,1.8 0.7,2.6 4.2,2.3 9.9,3.8 13.4,4.6 v 0 c 0.8,0.3 1.7,0.5 2.6,0.5 8.6,0.2 17.3,0.3 25.9,-0.5 h 0.3 v -6.3 c -5.8,0.5 -11.8,0.5 -20.8,0.3 -1.7,0 -2.9,-1.1 -3.4,-2.5 -2,-7.2 -2.5,-14.8 -2,-22.9 0.1,-1.5 1.2,-2.8 2.9,-3.1 1.6,-0.3 3.2,-0.6 4.6,-0.9 0.7,-0.1 2.5,-0.5 3.8,-0.7 v 0 c 1,-0.2 2.1,-0.4 3.1,-0.6 3.3,0.6 5.6,5.4 10.5,4.6 5,-0.1 6.4,-5.2 9.5,-6.3 2.3,0 4.6,0 6.9,0.1 1.1,0 2.1,0 3.9,0.1 1.7,0 3,1 3.4,2.5 0.4,1.6 0.7,3.2 1,4.8 0.6,5.4 0.9,9.8 0.8,13.2 h 6.8 z m -75.1,-7.8 c 2,-0.2 3.8,2.5 4.1,6 0.3,3.5 -1,6.4 -3,6.6 -1,0.1 -1.9,-0.5 -2.6,-1.5 0.4,0.3 0.9,0.5 1.3,0.4 1.5,-0.1 2.6,-2.5 2.3,-5.3 -0.2,-2.8 -1.7,-5 -3.2,-4.8 -0.5,0 -0.9,0.3 -1.2,0.7 0.5,-1.3 1.3,-2.1 2.3,-2.1 z M 77,182.4 c 1.6,0 2.9,1.3 2.9,2.9 0,1.6 -1.3,2.9 -2.9,2.9 -1.6,0 -2.9,-1.3 -2.9,-2.9 -0.1,-1.6 1.3,-2.9 2.9,-2.9 z m -10.3,15.2 v 0 c 1.4,-0.1 2.7,0.9 2.8,2.4 l 0.4,4.9 c 0.1,1.4 -0.9,2.7 -2.4,2.8 v 0 c -1.4,0.1 -2.7,-0.9 -2.8,-2.4 l -0.4,-4.9 c -0.1,-1.4 1,-2.7 2.4,-2.8 z m -32.2,-3.1 c 3.7,-0.3 7.1,3.9 7.6,9.4 0.5,5.5 -2.1,10.2 -5.9,10.6 -3.7,0.3 -7.1,-3.9 -7.6,-9.4 -0.5,-5.6 2.1,-10.3 5.9,-10.6 z m 43.7,66.1 c -3.9,-1 -7.3,-3.6 -8.8,-7.9 2.7,-0.6 5.9,-1.3 8.8,-2.3 z m -10.9,-26.8 -0.7,-7.3 11.6,-0.6 v 3.6 c -0.6,0.2 -1.3,0.4 -1.9,0.6 -3,1.1 -6,2.3 -8.9,3.6 z m -13.1,-0.2 c 3.5,-2.6 7.9,-2.5 9.8,0.1 0.3,0.4 0.5,0.8 0.6,1.3 -4.3,1.9 -10.3,5 -12.6,8.8 -0.3,-0.2 -0.6,-0.5 -0.8,-0.8 -1.9,-2.6 -0.6,-6.8 3,-9.4 z m 48.6,10.7 c 0.7,-0.2 1.4,-0.1 2.1,0 l 0.8,-1.4 0.3,0.1 c 0.7,0.2 1.3,0.6 1.9,1.1 l 0.3,0.2 -0.8,1.4 c 0.2,0.3 0.4,0.5 0.6,0.8 0.2,0.3 0.3,0.6 0.4,0.9 h 1.6 l 0.1,0.4 c 0.1,0.7 0.1,1.5 0,2.2 l -0.1,0.4 h -1.6 c -0.2,0.7 -0.6,1.3 -1,1.8 l 0.8,1.4 -0.3,0.2 c -0.3,0.2 -0.6,0.5 -0.9,0.6 -0.3,0.2 -0.6,0.3 -1,0.5 l -0.3,0.1 -0.8,-1.4 c -0.7,0.1 -1.4,0.1 -2.1,0 l -0.8,1.4 -0.3,-0.1 c -0.7,-0.3 -1.3,-0.6 -1.9,-1.1 l -0.3,-0.2 0.8,-1.4 c -0.2,-0.3 -0.4,-0.5 -0.6,-0.8 -0.2,-0.3 -0.3,-0.6 -0.4,-0.9 h -1.6 l -0.1,-0.4 c -0.1,-0.7 -0.1,-1.5 0,-2.2 l 0.1,-0.4 h 1.6 c 0.2,-0.7 0.6,-1.3 1,-1.8 l -0.8,-1.4 0.3,-0.2 c 0.3,-0.2 0.6,-0.5 0.9,-0.6 0.3,-0.2 0.6,-0.3 1,-0.5 l 0.3,-0.1 z M 88.4,217.9 h 29.9 v 3.2 H 88.4 Z m 0,7.2 h 12.1 v 3.2 H 88.4 Z m 0,7.3 h 9.7 v 3.2 h -9.7 l -0.1,-0.3 v -2.9 z m 25.7,-4.8 c 0.9,0.3 1.7,0.8 2.5,1.4 l 1.9,-1.1 0.3,0.4 c 0.7,0.8 1.2,1.7 1.5,2.6 l 0.2,0.5 -1.9,1.1 c 0.1,0.5 0.2,0.9 0.2,1.4 0,0.5 -0.1,1 -0.2,1.4 l 1.9,1.1 -0.2,0.5 c -0.3,0.9 -0.9,1.8 -1.5,2.6 l -0.3,0.4 -1.9,-1.1 c -0.7,0.6 -1.5,1.1 -2.5,1.4 v 2.2 l -0.5,0.1 c -0.5,0.1 -1,0.1 -1.5,0.1 -0.5,0 -1,0 -1.5,-0.1 l -0.5,-0.1 v -2.2 c -0.9,-0.3 -1.7,-0.8 -2.5,-1.4 l -1.9,1.1 -0.3,-0.4 c -0.6,-0.8 -1.2,-1.7 -1.5,-2.6 l -0.2,-0.5 1.9,-1.1 c -0.1,-0.5 -0.2,-0.9 -0.2,-1.4 0,-0.5 0.1,-0.9 0.2,-1.4 l -1.9,-1.1 0.2,-0.5 c 0.3,-0.9 0.9,-1.8 1.5,-2.6 l 0.3,-0.4 1.9,1.1 c 0.7,-0.6 1.5,-1.1 2.5,-1.4 v -2.2 l 0.5,-0.1 c 0.5,-0.1 1,-0.1 1.5,-0.1 0.5,0 1,0 1.5,0.1 l 0.5,0.1 z m -2,3 c -1.8,0 -3.3,1.5 -3.3,3.3 0,1.8 1.5,3.3 3.3,3.3 1.8,0 3.3,-1.5 3.3,-3.3 0,-1.8 -1.5,-3.3 -3.3,-3.3 z m -9.5,16.3 c -1.2,0.7 -1.6,2.2 -0.9,3.3 0.7,1.2 2.2,1.6 3.3,0.9 1.2,-0.7 1.6,-2.2 0.9,-3.3 -0.7,-1.2 -2.1,-1.6 -3.3,-0.9 z"
id="path4" />
<path
class="st1"
d="m 172,209.4 v 22.9 h -4.7 v -13.2 c 0,-0.3 0,-0.6 0,-1 0,-0.4 0,-0.7 0.1,-1.1 l -6.1,11.8 c -0.2,0.4 -0.4,0.6 -0.8,0.8 -0.3,0.2 -0.7,0.3 -1.1,0.3 h -0.7 c -0.4,0 -0.8,-0.1 -1.1,-0.3 -0.3,-0.2 -0.6,-0.5 -0.8,-0.8 L 150.7,217 c 0,0.4 0.1,0.7 0.1,1.1 0,0.4 0,0.7 0,1 v 13.2 h -4.7 v -22.9 h 4.1 c 0.2,0 0.4,0 0.6,0 0.2,0 0.3,0 0.5,0.1 0.1,0.1 0.3,0.1 0.4,0.2 0.1,0.1 0.2,0.3 0.3,0.5 l 5.9,11.6 c 0.2,0.4 0.4,0.8 0.6,1.2 0.2,0.4 0.4,0.9 0.6,1.3 0.2,-0.5 0.4,-0.9 0.6,-1.4 0.2,-0.4 0.4,-0.9 0.6,-1.3 l 5.9,-11.6 c 0.1,-0.2 0.2,-0.4 0.3,-0.5 0.1,-0.1 0.2,-0.2 0.4,-0.2 0.1,-0.1 0.3,-0.1 0.5,-0.1 0.2,0 0.4,0 0.6,0 h 4 z m 16.6,14.1 -2.1,-6.3 c -0.2,-0.4 -0.3,-0.9 -0.5,-1.4 -0.2,-0.5 -0.4,-1.1 -0.5,-1.8 -0.2,0.6 -0.3,1.2 -0.5,1.8 -0.2,0.5 -0.3,1 -0.5,1.4 l -2.1,6.2 h 6.2 z m 8.4,8.8 h -4.1 c -0.5,0 -0.8,-0.1 -1.1,-0.3 -0.3,-0.2 -0.5,-0.5 -0.6,-0.8 l -1.4,-4 h -8.7 l -1.4,4 c -0.1,0.3 -0.3,0.6 -0.6,0.8 -0.3,0.2 -0.7,0.4 -1.1,0.4 h -4.2 l 8.9,-22.9 h 5.4 z m 16.5,-6 c 0.1,0 0.3,0 0.4,0.1 0.1,0 0.2,0.1 0.4,0.2 l 2.1,2.2 c -0.9,1.2 -2.1,2.1 -3.5,2.7 -1.4,0.6 -3,0.9 -4.9,0.9 -1.7,0 -3.3,-0.3 -4.7,-0.9 -1.4,-0.6 -2.5,-1.4 -3.5,-2.5 -1,-1 -1.7,-2.3 -2.2,-3.7 -0.5,-1.4 -0.8,-3 -0.8,-4.7 0,-1.7 0.3,-3.3 0.8,-4.7 0.6,-1.4 1.3,-2.7 2.3,-3.7 1,-1 2.2,-1.8 3.6,-2.4 1.4,-0.6 3,-0.9 4.6,-0.9 0.9,0 1.7,0.1 2.4,0.2 0.8,0.2 1.5,0.4 2.1,0.6 0.7,0.3 1.3,0.6 1.8,1 0.6,0.4 1,0.8 1.5,1.2 l -1.8,2.4 c -0.1,0.1 -0.3,0.3 -0.4,0.4 -0.2,0.1 -0.4,0.2 -0.7,0.2 -0.2,0 -0.4,0 -0.5,-0.1 -0.2,-0.1 -0.3,-0.2 -0.5,-0.3 -0.2,-0.1 -0.4,-0.3 -0.6,-0.4 -0.2,-0.1 -0.5,-0.3 -0.8,-0.4 -0.3,-0.1 -0.7,-0.2 -1.1,-0.3 -0.4,-0.1 -0.9,-0.1 -1.5,-0.1 -0.9,0 -1.7,0.2 -2.4,0.5 -0.7,0.3 -1.4,0.8 -1.9,1.4 -0.5,0.6 -0.9,1.4 -1.2,2.3 -0.3,0.9 -0.4,1.9 -0.4,3.1 0,1.2 0.2,2.2 0.5,3.1 0.3,0.9 0.8,1.7 1.3,2.3 0.6,0.6 1.2,1.1 1.9,1.4 0.7,0.3 1.5,0.5 2.4,0.5 0.5,0 0.9,0 1.3,-0.1 0.4,0 0.8,-0.1 1.1,-0.2 0.3,-0.1 0.7,-0.3 1,-0.4 0.3,-0.2 0.6,-0.4 0.9,-0.7 0.1,-0.1 0.3,-0.2 0.4,-0.3 0.3,0.2 0.5,0.1 0.6,0.1 z m 25.2,-16.9 v 22.9 h -5.3 v -9.7 h -9.3 v 9.7 h -5.4 v -22.9 h 5.4 v 9.6 h 9.3 v -9.6 z m 9.8,22.9 h -5.4 v -22.9 h 5.4 z M 273,209.4 v 22.9 h -2.8 c -0.4,0 -0.8,-0.1 -1,-0.2 -0.3,-0.1 -0.6,-0.4 -0.8,-0.7 l -10.8,-13.7 c 0,0.4 0.1,0.8 0.1,1.2 0,0.4 0,0.7 0,1.1 v 12.3 H 253 v -22.9 h 2.8 c 0.2,0 0.4,0 0.6,0 0.2,0 0.3,0.1 0.4,0.1 0.1,0.1 0.2,0.1 0.4,0.2 0.1,0.1 0.2,0.2 0.4,0.4 l 10.9,13.8 c -0.1,-0.4 -0.1,-0.9 -0.1,-1.3 0,-0.4 0,-0.8 0,-1.2 v -12.1 h 4.6 z m 9.8,4 v 5.4 h 7.2 v 3.9 h -7.2 v 5.5 h 9.4 v 4.1 h -14.8 v -22.9 h 14.8 v 4.1 h -9.4 z"
id="path6" />
<path
class="st1"
d="m 316.5,228 v 4.2 H 302.7 V 209.3 H 308 V 228 Z m 8.1,-14.6 v 5.4 h 7.2 v 3.9 h -7.2 v 5.5 h 9.4 v 4.1 H 319.2 V 209.4 H 334 v 4.1 h -9.4 z m 24.8,10.1 -2.1,-6.3 c -0.2,-0.4 -0.3,-0.9 -0.5,-1.4 -0.2,-0.5 -0.4,-1.1 -0.5,-1.8 -0.2,0.6 -0.3,1.2 -0.5,1.8 -0.2,0.5 -0.3,1 -0.5,1.4 l -2.1,6.2 h 6.2 z m 8.4,8.8 h -4.1 c -0.5,0 -0.8,-0.1 -1.1,-0.3 -0.3,-0.2 -0.5,-0.5 -0.6,-0.8 l -1.4,-4 h -8.7 l -1.4,4 c -0.1,0.3 -0.3,0.6 -0.6,0.8 -0.3,0.2 -0.7,0.4 -1.1,0.4 h -4.2 l 8.9,-22.9 h 5.4 z m 9.3,-12.2 c 0.7,0 1.3,-0.1 1.9,-0.3 0.5,-0.2 0.9,-0.4 1.2,-0.8 0.3,-0.3 0.6,-0.7 0.7,-1.1 0.1,-0.4 0.2,-0.9 0.2,-1.4 0,-1 -0.3,-1.8 -1,-2.4 -0.7,-0.6 -1.7,-0.8 -3,-0.8 H 365 v 6.8 z m 11.3,12.2 h -4.8 c -0.9,0 -1.5,-0.3 -1.9,-1 l -3.8,-6.7 c -0.2,-0.3 -0.4,-0.5 -0.6,-0.6 -0.2,-0.1 -0.5,-0.2 -0.9,-0.2 H 365 v 8.5 h -5.3 v -22.9 h 7.5 c 1.7,0 3.1,0.2 4.2,0.5 1.2,0.3 2.1,0.8 2.9,1.4 0.7,0.6 1.3,1.3 1.6,2.2 0.3,0.8 0.5,1.7 0.5,2.7 0,0.7 -0.1,1.4 -0.3,2.1 -0.2,0.7 -0.5,1.3 -0.9,1.8 -0.4,0.6 -0.9,1.1 -1.4,1.5 -0.6,0.4 -1.2,0.8 -2,1.1 0.3,0.2 0.7,0.4 1,0.7 0.3,0.3 0.6,0.6 0.8,1 z m 22,-22.9 v 22.9 h -2.8 c -0.4,0 -0.8,-0.1 -1,-0.2 -0.3,-0.1 -0.6,-0.4 -0.8,-0.7 L 385,217.7 c 0,0.4 0.1,0.8 0.1,1.2 0,0.4 0,0.7 0,1.1 v 12.3 h -4.7 v -22.9 h 2.8 c 0.2,0 0.4,0 0.6,0 0.2,0 0.3,0.1 0.4,0.1 0.1,0.1 0.2,0.1 0.4,0.2 0.1,0.1 0.2,0.2 0.4,0.4 l 10.9,13.8 c -0.1,-0.4 -0.1,-0.9 -0.1,-1.3 0,-0.4 0,-0.8 0,-1.2 v -12.1 h 4.6 z m 9.7,22.9 h -5.4 v -22.9 h 5.4 z m 24.5,-22.9 v 22.9 h -2.8 c -0.4,0 -0.7,-0.1 -1,-0.2 -0.3,-0.1 -0.6,-0.4 -0.8,-0.7 l -10.8,-13.7 c 0,0.4 0.1,0.8 0.1,1.2 0,0.4 0,0.7 0,1.1 v 12.3 h -4.7 v -22.9 h 2.8 c 0.2,0 0.4,0 0.6,0 0.2,0 0.3,0.1 0.4,0.1 0.1,0.1 0.2,0.1 0.4,0.2 0.1,0.1 0.2,0.2 0.4,0.4 l 10.9,13.8 c -0.1,-0.4 -0.1,-0.9 -0.1,-1.3 0,-0.4 0,-0.8 0,-1.2 v -12.1 h 4.6 z m 15.5,11 h 8.2 v 9.7 c -1.2,0.9 -2.4,1.5 -3.8,1.9 -1.3,0.4 -2.7,0.6 -4.2,0.6 -1.9,0 -3.6,-0.3 -5.2,-0.9 -1.5,-0.6 -2.9,-1.4 -4,-2.5 -1.1,-1 -2,-2.3 -2.5,-3.7 -0.6,-1.4 -0.9,-3 -0.9,-4.7 0,-1.7 0.3,-3.3 0.9,-4.7 0.6,-1.4 1.4,-2.7 2.4,-3.7 1.1,-1 2.3,-1.8 3.8,-2.4 1.5,-0.6 3.2,-0.9 5,-0.9 1,0 1.9,0.1 2.7,0.2 0.8,0.2 1.6,0.4 2.3,0.6 0.7,0.3 1.4,0.6 1.9,1 0.6,0.4 1.1,0.8 1.6,1.2 l -1.5,2.3 c -0.2,0.4 -0.6,0.6 -0.9,0.7 -0.4,0.1 -0.8,0 -1.2,-0.3 -0.4,-0.3 -0.8,-0.5 -1.2,-0.7 -0.4,-0.2 -0.8,-0.3 -1.1,-0.4 -0.4,-0.1 -0.8,-0.2 -1.2,-0.3 -0.4,-0.1 -0.9,-0.1 -1.4,-0.1 -1,0 -1.9,0.2 -2.7,0.5 -0.8,0.4 -1.5,0.9 -2,1.5 -0.6,0.6 -1,1.4 -1.3,2.3 -0.3,0.9 -0.5,1.9 -0.5,3 0,1.2 0.2,2.3 0.5,3.2 0.3,0.9 0.8,1.7 1.4,2.4 0.6,0.7 1.3,1.1 2.2,1.5 0.9,0.3 1.8,0.5 2.8,0.5 0.6,0 1.2,-0.1 1.7,-0.2 0.5,-0.1 1,-0.3 1.5,-0.5 V 224 h -2.3 c -0.3,0 -0.6,-0.1 -0.8,-0.3 -0.2,-0.2 -0.3,-0.4 -0.3,-0.7 v -2.6 z"
id="path8" />
<path
class="st1"
d="m 155.6,253.7 c 0.1,0 0.2,0 0.3,0.1 0.1,0 0.2,0.1 0.3,0.2 l 1.5,1.6 c -0.7,0.9 -1.5,1.5 -2.5,2 -1,0.4 -2.2,0.7 -3.6,0.7 -1.3,0 -2.4,-0.2 -3.4,-0.7 -1,-0.4 -1.9,-1 -2.6,-1.8 -0.7,-0.8 -1.2,-1.7 -1.6,-2.7 -0.4,-1 -0.6,-2.2 -0.6,-3.4 0,-1.2 0.2,-2.4 0.6,-3.4 0.4,-1 1,-1.9 1.7,-2.7 0.7,-0.8 1.6,-1.3 2.6,-1.8 1,-0.4 2.2,-0.6 3.4,-0.6 0.6,0 1.2,0.1 1.8,0.2 0.6,0.1 1.1,0.3 1.6,0.5 0.5,0.2 0.9,0.4 1.3,0.7 0.4,0.3 0.8,0.6 1.1,0.9 l -1.3,1.8 c -0.1,0.1 -0.2,0.2 -0.3,0.3 -0.1,0.1 -0.3,0.1 -0.5,0.1 -0.1,0 -0.3,0 -0.4,-0.1 -0.1,-0.1 -0.3,-0.1 -0.4,-0.2 -0.1,-0.1 -0.3,-0.2 -0.5,-0.3 -0.2,-0.1 -0.4,-0.2 -0.6,-0.3 -0.2,-0.1 -0.5,-0.2 -0.8,-0.2 -0.3,-0.1 -0.7,-0.1 -1.1,-0.1 -0.6,0 -1.2,0.1 -1.8,0.4 -0.5,0.2 -1,0.6 -1.4,1 -0.4,0.5 -0.7,1 -0.9,1.7 -0.2,0.7 -0.3,1.4 -0.3,2.3 0,0.9 0.1,1.6 0.4,2.3 0.2,0.7 0.6,1.2 1,1.7 0.4,0.5 0.9,0.8 1.4,1 0.5,0.2 1.1,0.4 1.7,0.4 0.4,0 0.7,0 1,-0.1 0.3,0 0.6,-0.1 0.8,-0.2 0.3,-0.1 0.5,-0.2 0.7,-0.3 0.2,-0.1 0.4,-0.3 0.7,-0.5 0.1,-0.1 0.2,-0.1 0.3,-0.2 0.2,-0.3 0.3,-0.3 0.4,-0.3 z m 36.1,-4.1 c 0,1.2 -0.2,2.3 -0.6,3.4 -0.4,1 -1,1.9 -1.8,2.7 -0.8,0.8 -1.7,1.4 -2.7,1.8 -1.1,0.4 -2.3,0.7 -3.6,0.7 -1.3,0 -2.5,-0.2 -3.6,-0.7 -1.1,-0.4 -2,-1 -2.7,-1.8 -0.8,-0.8 -1.4,-1.7 -1.8,-2.7 -0.4,-1 -0.6,-2.2 -0.6,-3.4 0,-1.2 0.2,-2.3 0.6,-3.4 0.4,-1 1,-1.9 1.8,-2.7 0.8,-0.8 1.7,-1.4 2.7,-1.8 1.1,-0.4 2.3,-0.7 3.6,-0.7 1.3,0 2.5,0.2 3.6,0.7 1.1,0.4 2,1 2.7,1.8 0.8,0.8 1.3,1.7 1.8,2.7 0.4,1.1 0.6,2.2 0.6,3.4 z m -3.9,0 c 0,-0.8 -0.1,-1.6 -0.3,-2.2 -0.2,-0.7 -0.5,-1.2 -0.9,-1.7 -0.4,-0.5 -0.9,-0.8 -1.5,-1.1 -0.6,-0.2 -1.2,-0.4 -2,-0.4 -0.8,0 -1.4,0.1 -2,0.4 -0.6,0.2 -1.1,0.6 -1.5,1.1 -0.4,0.5 -0.7,1 -0.9,1.7 -0.2,0.7 -0.3,1.4 -0.3,2.2 0,0.8 0.1,1.6 0.3,2.2 0.2,0.7 0.5,1.2 0.9,1.7 0.4,0.5 0.9,0.8 1.5,1 0.6,0.2 1.3,0.4 2,0.4 0.7,0 1.4,-0.1 2,-0.4 0.6,-0.2 1.1,-0.6 1.5,-1 0.4,-0.5 0.7,-1 0.9,-1.7 0.2,-0.6 0.3,-1.3 0.3,-2.2 z m 41,-8.3 V 258 h -3.4 v -9.6 c 0,-0.2 0,-0.5 0,-0.7 0,-0.3 0,-0.5 0.1,-0.8 l -4.4,8.6 c -0.1,0.3 -0.3,0.5 -0.6,0.6 -0.2,0.1 -0.5,0.2 -0.8,0.2 h -0.5 c -0.3,0 -0.6,-0.1 -0.8,-0.2 -0.2,-0.1 -0.4,-0.3 -0.6,-0.6 l -4.4,-8.6 c 0,0.3 0,0.5 0.1,0.8 0,0.3 0,0.5 0,0.7 v 9.6 h -3.4 v -16.7 h 3 c 0.2,0 0.3,0 0.4,0 0.1,0 0.2,0 0.3,0.1 0.1,0 0.2,0.1 0.3,0.2 0.1,0.1 0.2,0.2 0.2,0.3 l 4.3,8.5 c 0.2,0.3 0.3,0.6 0.4,0.9 0.1,0.3 0.3,0.6 0.4,1 0.1,-0.3 0.3,-0.7 0.4,-1 0.1,-0.3 0.3,-0.6 0.5,-0.9 l 4.3,-8.5 c 0.1,-0.1 0.2,-0.3 0.2,-0.3 0.1,-0.1 0.2,-0.1 0.3,-0.2 0.1,0 0.2,-0.1 0.3,-0.1 0.1,0 0.3,0 0.4,0 z m 25,8.2 c 1,0 1.7,-0.2 2.2,-0.7 0.4,-0.5 0.7,-1.2 0.7,-2 0,-0.4 -0.1,-0.7 -0.2,-1 -0.1,-0.3 -0.3,-0.6 -0.5,-0.8 -0.2,-0.2 -0.5,-0.4 -0.9,-0.5 -0.4,-0.1 -0.8,-0.2 -1.3,-0.2 h -2 v 5.3 h 2 z m 0,-8.2 c 1.2,0 2.2,0.1 3,0.4 0.8,0.3 1.5,0.7 2.1,1.2 0.5,0.5 0.9,1.1 1.2,1.7 0.3,0.7 0.4,1.4 0.4,2.2 0,0.8 -0.1,1.6 -0.4,2.3 -0.3,0.7 -0.7,1.3 -1.2,1.8 -0.6,0.5 -1.2,0.9 -2.1,1.2 -0.8,0.3 -1.8,0.4 -3,0.4 h -2 v 5.6 h -3.9 v -16.7 h 5.9 z m 27.9,16.7 h -3.9 v -16.7 h 3.9 z m 29.3,-3.1 v 3.1 h -10.1 v -16.7 h 3.9 v 13.6 z m 25.2,-3.3 -1.6,-4.6 c -0.1,-0.3 -0.2,-0.6 -0.4,-1 -0.1,-0.4 -0.3,-0.8 -0.4,-1.3 -0.1,0.5 -0.2,0.9 -0.4,1.3 -0.1,0.4 -0.3,0.7 -0.4,1 l -1.5,4.6 z m 6.2,6.4 h -3 c -0.3,0 -0.6,-0.1 -0.8,-0.2 -0.2,-0.2 -0.4,-0.4 -0.5,-0.6 l -1,-2.9 h -6.4 l -1,2.9 c -0.1,0.2 -0.2,0.4 -0.5,0.6 -0.2,0.2 -0.5,0.3 -0.8,0.3 h -3 l 6.5,-16.7 h 4 z m 27.8,-13.6 h -4.7 V 258 h -3.9 v -13.7 h -4.7 v -3.1 h 13.3 z m 21.1,13.6 h -3.9 v -16.7 h 3.9 z m 35.5,-8.4 c 0,1.2 -0.2,2.3 -0.6,3.4 -0.4,1 -1,1.9 -1.8,2.7 -0.8,0.8 -1.7,1.4 -2.7,1.8 -1.1,0.4 -2.3,0.7 -3.6,0.7 -1.3,0 -2.5,-0.2 -3.6,-0.7 -1.1,-0.4 -2,-1 -2.7,-1.8 -0.8,-0.8 -1.4,-1.7 -1.8,-2.7 -0.4,-1 -0.6,-2.2 -0.6,-3.4 0,-1.2 0.2,-2.3 0.6,-3.4 0.4,-1 1,-1.9 1.8,-2.7 0.8,-0.8 1.7,-1.4 2.7,-1.8 1.1,-0.4 2.3,-0.7 3.6,-0.7 1.3,0 2.5,0.2 3.6,0.7 1.1,0.4 2,1 2.7,1.8 0.8,0.8 1.3,1.7 1.8,2.7 0.4,1.1 0.6,2.2 0.6,3.4 z m -3.9,0 c 0,-0.8 -0.1,-1.6 -0.3,-2.2 -0.2,-0.7 -0.5,-1.2 -0.9,-1.7 -0.4,-0.5 -0.9,-0.8 -1.5,-1.1 -0.6,-0.2 -1.2,-0.4 -2,-0.4 -0.8,0 -1.4,0.1 -2,0.4 -0.6,0.2 -1.1,0.6 -1.5,1.1 -0.4,0.5 -0.7,1 -0.9,1.7 -0.2,0.7 -0.3,1.4 -0.3,2.2 0,0.8 0.1,1.6 0.3,2.2 0.2,0.7 0.5,1.2 0.9,1.7 0.4,0.5 0.9,0.8 1.5,1 0.6,0.2 1.3,0.4 2,0.4 0.7,0 1.4,-0.1 2,-0.4 0.6,-0.2 1.1,-0.6 1.5,-1 0.4,-0.5 0.7,-1 0.9,-1.7 0.2,-0.6 0.3,-1.3 0.3,-2.2 z m 36.8,-8.3 V 258 h -2 c -0.3,0 -0.5,-0.1 -0.8,-0.1 -0.2,-0.1 -0.4,-0.3 -0.6,-0.5 l -7.9,-10 c 0,0.3 0.1,0.6 0.1,0.9 0,0.3 0,0.5 0,0.8 v 9 h -3.4 v -16.7 h 2 c 0.2,0 0.3,0 0.4,0 0.1,0 0.2,0 0.3,0.1 0.1,0 0.2,0.1 0.3,0.2 0.1,0.1 0.2,0.2 0.3,0.3 l 8,10.1 c 0,-0.3 -0.1,-0.6 -0.1,-0.9 0,-0.3 0,-0.6 0,-0.9 v -8.9 h 3.4 z"
id="path10" />
</g>
</g>
</switch>
</svg>

After

Width:  |  Height:  |  Size: 16 KiB

+102
View File
@@ -0,0 +1,102 @@
# -*- coding: utf-8 -*-
import os
import sys
import tlcpack_sphinx_addon
# -- General configuration ------------------------------------------------
sys.path.insert(0, os.path.abspath("../python"))
sys.path.insert(0, os.path.abspath("../"))
autodoc_mock_imports = ["torch"]
# General information about the project.
project = "web-llm"
author = "WebLLM Contributors"
copyright = "2023, %s" % author
# Version information.
version = "0.2.84"
release = "0.2.84"
extensions = [
"sphinx_tabs.tabs",
"sphinx_toolbox.collapse",
"sphinxcontrib.httpdomain",
"sphinx.ext.autodoc",
"sphinx.ext.napoleon",
"sphinx_reredirects",
]
redirects = {"get_started/try_out": "../index.html#getting-started"}
source_suffix = [".rst"]
language = "en"
exclude_patterns = ["_build", "Thumbs.db", ".DS_Store"]
# The name of the Pygments (syntax highlighting) style to use.
pygments_style = "sphinx"
# A list of ignored prefixes for module index sorting.
# If true, `todo` and `todoList` produce output, else they produce nothing.
todo_include_todos = False
# -- Options for HTML output ----------------------------------------------
# The theme is set by the make target
import sphinx_rtd_theme
html_theme = "sphinx_rtd_theme"
html_theme_path = [sphinx_rtd_theme.get_html_theme_path()]
templates_path = []
html_static_path = []
footer_copyright = "© 2023 MLC LLM"
footer_note = " "
html_logo = "_static/img/mlc-logo-with-text-landscape.svg"
html_theme_options = {
"logo_only": True,
}
header_links = [
("Home", "https://webllm.mlc.ai/"),
("GitHub", "https://github.com/mlc-ai/web-llm"),
("Discord", "https://discord.gg/9Xpy2HGBuD"),
]
header_dropdown = {
"name": "Other Resources",
"items": [
("WebLLM Chat", "https://chat.webllm.ai/"),
("MLC Course", "https://mlc.ai/"),
("MLC Blog", "https://blog.mlc.ai/"),
("MLC LLM", "https://llm.mlc.ai/"),
],
}
html_context = {
"footer_copyright": footer_copyright,
"footer_note": footer_note,
"header_links": header_links,
"header_dropdown": header_dropdown,
"display_github": True,
"github_user": "mlc-ai",
"github_repo": "web-llm",
"github_version": "main/docs/",
"theme_vcs_pageview_mode": "edit",
# "header_logo": "/path/to/logo",
# "header_logo_link": "",
# "version_selecter": "",
}
# add additional overrides
templates_path += [tlcpack_sphinx_addon.get_templates_path()]
html_static_path += [tlcpack_sphinx_addon.get_static_path()]
+6
View File
@@ -0,0 +1,6 @@
Adding Models
=============
WebLLM allows you to compile custom language models using `MLC-LLM <https://llm.mlc.ai/>`_ and then serve the compiled model through WebLLM.
For instructions on how to compile and add custom models to WebLLM, please refer to the `MLC-LLM documentation <https://llm.mlc.ai/docs/deploy/webllm.html>`_.
+35
View File
@@ -0,0 +1,35 @@
Building From Source
====================
Clone the Repository
---------------------
.. code-block:: bash
git clone https://github.com/mlc-ai/web-llm.git
cd web-llm
Install Dependencies
---------------------
.. code-block:: bash
npm install
Build the Project
-----------------
.. code-block:: bash
npm run build
Test Changes
------------
To test your changes, you can reuse an existing example or create a new example that specifically tests the new functionality you wish to provide.
To test the effects of your code change in an example, inside ``examples/<example>/package.json``, change ``"@mlc-ai/web-llm": "^0.2.xx"`` to ``"@mlc-ai/web-llm": "../.."`` to let it reference your local code. Note that sometimes you may need to switch between ``"file:../.."`` and ``"../.."`` to trigger npm to recognize new changes.
.. code-block:: bash
cd examples/<example>
# Modify package.json as described
npm install
npm start
+35
View File
@@ -0,0 +1,35 @@
👋 Welcome to WebLLM
====================
`GitHub <https://github.com/mlc-ai/web-llm>`_ | `WebLLM Chat <https://chat.webllm.ai/>`_ | `NPM <https://www.npmjs.com/package/@mlc-ai/web-llm>`_ | `Discord <https://discord.gg/9Xpy2HGBuD>`_
WebLLM is a high-performance in-browser language model inference engine that brings large language models (LLMs) to web browsers with hardware acceleration. With WebGPU support, it allows developers to build AI-powered applications directly within the browser environment, removing the need for server-side processing and ensuring privacy.
It provides a specialized runtime for the web backend of MLCEngine, leverages
`WebGPU <https://www.w3.org/TR/webgpu/>`_ for local acceleration, offers OpenAI-compatible API,
and provides built-in support for web workers to separate heavy computation from the UI flow.
Key Features
------------
- 🌐 In-Browser Inference: Run LLMs directly in the browser
- 🚀 WebGPU Acceleration: Leverage hardware acceleration for optimal performance
- 🔄 OpenAI API Compatibility: Seamless integration with standard AI workflows
- 📦 Multiple Model Support: Works with Llama, Phi, Gemma, Mistral, and more
Start exploring WebLLM by `chatting with WebLLM Chat <https://chat.webllm.ai/>`_, and start building webapps with high-performance local LLM inference with the following guides and tutorials.
.. toctree::
:maxdepth: 2
:caption: User Guide
user/get_started.rst
user/basic_usage.rst
user/advanced_usage.rst
user/api_reference.rst
.. toctree::
:maxdepth: 2
:caption: Developer Guide
developer/building_from_source.rst
developer/add_models.rst
+35
View File
@@ -0,0 +1,35 @@
@ECHO OFF
pushd %~dp0
REM Command file for Sphinx documentation
if "%SPHINXBUILD%" == "" (
set SPHINXBUILD=sphinx-build
)
set SOURCEDIR=.
set BUILDDIR=_build
%SPHINXBUILD% >NUL 2>NUL
if errorlevel 9009 (
echo.
echo.The 'sphinx-build' command was not found. Make sure you have Sphinx
echo.installed, then set the SPHINXBUILD environment variable to point
echo.to the full path of the 'sphinx-build' executable. Alternatively you
echo.may add the Sphinx directory to PATH.
echo.
echo.If you don't have Sphinx installed, grab it from
echo.https://www.sphinx-doc.org/
exit /b 1
)
if "%1" == "" goto help
%SPHINXBUILD% -M %1 %SOURCEDIR% %BUILDDIR% %SPHINXOPTS% %O%
goto end
:help
%SPHINXBUILD% -M help %SOURCEDIR% %BUILDDIR% %SPHINXOPTS% %O%
:end
popd
+8
View File
@@ -0,0 +1,8 @@
sphinx-tabs == 3.4.1
sphinx-rtd-theme
sphinx == 5.2.3
sphinx-toolbox == 3.4.0
tlcpack-sphinx-addon==0.2.2
sphinxcontrib_httpdomain==1.8.1
sphinxcontrib-napoleon==0.7
sphinx-reredirects==0.1.2
+153
View File
@@ -0,0 +1,153 @@
Advanced Use Cases
==================
Using Workers
-------------
You can put the heavy computation in a worker script to optimize your application performance. To do so, you need to:
Create a handler in the worker thread that communicates with the frontend while handling the requests.
Create a worker engine in your main application that sends messages to the handler in the worker thread under the hood.
For detailed implementations of different kinds of workers, look at the following sections.
Using Web Workers
^^^^^^^^^^^^^^^^^
WebLLM comes with API support for `Web Workers <https://developer.mozilla.org/en-US/docs/Web/API/Web_Workers_API/Using_web_workers>`_ so you can offload the computation-heavy generation work into a separate worker thread. WebLLM has implemented cross-thread communication through messages under the hood, so manual implementation is not required.
In the worker script, import and instantiate a ``WebWorkerMLCEngineHandler``, which handles communication with other scripts and processes incoming requests.
.. code-block:: typescript
// worker.ts
import { WebWorkerMLCEngineHandler } from "@mlc-ai/web-llm";
const handler = new WebWorkerMLCEngineHandler();
self.onmessage = (msg: MessageEvent) => {
handler.onmessage(msg);
};
In the main script, import and instantiate a ``WebWorkerMLCEngine`` that implements the same ``MLCEngineInterface`` and exposes the same APIs. Then, simply use it as you would a normal ``MLCEngine``.
.. code-block:: typescript
import { CreateWebWorkerMLCEngine } from "@mlc-ai/web-llm";
async function runWorker() {
const engine = await CreateWebWorkerMLCEngine(
new Worker(new URL("./worker.ts", import.meta.url), { type: "module" }),
"Llama-3.1-8B-Instruct"
);
const messages = [{ role: "user", content: "How does WebLLM use workers?" }];
const reply = await engine.chat.completions.create({ messages });
console.log(reply.choices[0].message.content);
}
runWorker();
Under the hood, ``WebWorkerMLCEngine`` does **not** perform any computation. It translates all calls into messages and sends them to the ``WebWorkerMLCEngineHandler`` for processing. The worker thread receives these messages and processes the actual computation using a hidden engine, and returns the result to the main thread using messages.
Service Workers
^^^^^^^^^^^^^^^
WebLLM also supports offloading computation using `Service Workers <https://developer.mozilla.org/en-US/docs/Web/API/Service_Worker_API>`_. This allows you to avoid reloading the model between page refreshes and optimize your application's offline experience.
(Note, the lifecycle of a Service Worker is managed by the browser and can be killed any time without notifying the web application. WebLLM's ``ServiceWorkerMLCEngine`` attempts to keep the service worker thread alive by periodically sending heartbeat events. However, the script could still be killed at any time by Chrome, and your application should include proper error handling. Check `keepAliveMs` and `missedHeartbeat` in `ServiceWorkerMLCEngine <https://github.com/mlc-ai/web-llm/blob/main/src/service_worker.ts#L218>`_ for more details.)
In the worker script, import and instantiate ``ServiceWorkerMLCEngineHandler``, which handles communication with page scripts and processes incoming requests.
.. code-block:: typescript
// sw.ts
import { ServiceWorkerMLCEngineHandler } from "@mlc-ai/web-llm";
self.addEventListener("activate", () => {
const handler = new ServiceWorkerMLCEngineHandler();
console.log("Service Worker activated!");
});
Then, in the main page script, register the service worker and instantiate the engine using the ``CreateServiceWorkerMLCEngine`` factory function that implements the same ``MLCEngineInterface`` and exposes the same APIs. Then, simply use it as you would a normal ``MLCEngine``.
.. code-block:: typescript
// main.ts
import { MLCEngineInterface, CreateServiceWorkerMLCEngine } from "@mlc-ai/web-llm";
if ("serviceWorker" in navigator) {
navigator.serviceWorker.register(
new URL("sw.ts", import.meta.url), // worker script
{ type: "module" },
);
}
const engine: MLCEngineInterface =
await CreateServiceWorkerMLCEngine(
selectedModel,
{ initProgressCallback }, // engineConfig
);
Similar to the ``WebWorkerMLCEngine`` above, the ``ServiceWorkerMLCEngine`` is also a proxy and does not perform any actual computation. Instead, it forwards all calls to the service worker thread and receives the result through messages.
Chrome Extension
----------------
WebLLM can be used in Chrome extensions to empower local LLM inference. You can find examples of building Chrome extension using WebLLM in `examples/chrome-extension <https://github.com/mlc-ai/web-llm/blob/main/examples/chrome-extension>`_ and `examples/chrome-extension-webgpu-service-worker <https://github.com/mlc-ai/web-llm/blob/main/examples/chrome-extension-webgpu-service-worker>`_. The latter leverages Service Worker, so the extension is persistent in the background.
Additionally, we have a full Chrome extension project, `WebLLM Assistant <https://github.com/mlc-ai/web-llm-assistant>`_, which leverages WebLLM to provide a personal web browsing copilot assistant experience. Feel free to check it out and contribute if you are interested.
Additional Customization
------------------------
Using IndexedDB Cache
^^^^^^^^^^^^^^^^^^^^^
By default, WebLLM caches model artifacts using the `Cache API <https://developer.mozilla.org/en-US/docs/Web/API/Cache>`_ for faster subsequent model loads. You can alternatively use `IndexedDB caching <https://developer.mozilla.org/en-US/docs/Web/API/IndexedDB_API>`_ by setting ``appConfig.cacheBackend = "indexeddb"``. When changing only the cache backend, preserve the prebuilt model list by spreading ``prebuiltAppConfig``.
.. code-block:: typescript
import { AppConfig, CreateMLCEngine, prebuiltAppConfig } from "@mlc-ai/web-llm";
const appConfig: AppConfig = {
...prebuiltAppConfig,
cacheBackend: "indexeddb",
};
const engine = await CreateMLCEngine("Llama-3.1-8B-Instruct-q4f32_1-MLC", {
appConfig,
});
Using Cross-Origin Storage Cache
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
WebLLM also supports caching model artifacts across different origins using the experimental Cross-Origin Storage API. You can enable this cache backend by setting ``appConfig.cacheBackend = "cross-origin"``. For users with the `Cross-Origin Storage browser extension <https://chromewebstore.google.com/detail/cross-origin-storage/denpnpcgjgikjpoglpjefakmdcbmlgih>`_ installed, resources will then be cached and shared across origins. This means two independent apps opted into this cache backend using the same AI model will download and cache the required resources only once. See the `cache usage example <https://github.com/mlc-ai/web-llm/tree/main/examples/cache-usage>`_ for more details. If Cross-Origin Storage isn't available, WebLLM will automatically fall back to using the default cache.
.. code-block:: typescript
import { AppConfig, CreateMLCEngine, prebuiltAppConfig } from "@mlc-ai/web-llm";
const appConfig: AppConfig = {
...prebuiltAppConfig,
cacheBackend: "cross-origin",
};
const engine = await CreateMLCEngine("Llama-3.1-8B-Instruct-q4f32_1-MLC", {
appConfig,
});
Customizing Token Behavior
^^^^^^^^^^^^^^^^^^^^^^^^^^
You can modify `logit_bias` in `GenerationConfig` to control token likelihood. Setting a token's bias to a positive value increases its likelihood of being generated, while a negative value decreases it. A large negative value (e.g., -100) can effectively prevent the token from being generated.
.. code-block:: typescript
const messages = [
{ role: "user", content: "Describe WebLLM in detail." },
];
const response = await engine.chatCompletion({
messages,
logit_bias: { "50256": -100 }, // Example: Prevent specific token generation
});
+239
View File
@@ -0,0 +1,239 @@
.. _api-reference:
WebLLM API Reference
====================
The ``MLCEngine`` class is the core interface of WebLLM. It enables model loading, chat completions, embeddings, and other operations. Below, we document its methods, along with the associated configuration interfaces.
Interfaces
----------
The following interfaces are used as parameters or configurations within ``MLCEngine`` methods. They are linked to their respective methods for reference.
MLCEngineConfig
^^^^^^^^^^^^^^^
Optional configurations for ``CreateMLCEngine()`` and ``CreateWebWorkerMLCEngine()``.
- **Fields**:
- ``appConfig``: Configure the app, including the list of models and whether to use IndexedDB cache.
- ``initProgressCallback``: A callback for showing model loading progress.
- ``logitProcessorRegistry``: A registry for stateful logit processors (see ``webllm.LogitProcessor``).
- **Usage**:
- ``appConfig``: Contains application-specific settings, including:
- Model configurations.
- IndexedDB caching preferences.
- ``initProgressCallback``: Allows developers to visualize model loading progress by implementing a callback.
- ``logitProcessorRegistry``: A ``Map`` object for registering custom logit processors. Only applies to ``MLCEngine``.
.. note:: All fields are optional, and ``logitProcessorRegistry`` is only used in ``MLCEngine``.
Example:
.. code-block:: typescript
const engine = await CreateMLCEngine("Llama-3.1-8B-Instruct", {
appConfig: { /* app-specific config */ },
initProgressCallback: (progress) => console.log(progress),
});
GenerationConfig
^^^^^^^^^^^^^^^^
Configurations for a single generation task, primarily used in chat completions.
- **Fields**:
- ``repetition_penalty``, ``ignore_eos``: Parameters specific to MLC models.
- ``top_p``, ``temperature``, ``max_tokens``, ``stop``: Common parameters shared with OpenAI APIs.
- ``frequency_penalty``, ``presence_penalty``: Tune repetition behavior following OpenAI semantics.
- ``logit_bias``, ``n``, ``logprobs``, ``top_logprobs``: Advanced sampling controls.
- ``response_format``, ``enable_thinking``, ``enable_latency_breakdown``: Additional OpenAI-style request features.
- **Usage**:
- Fields like ``repetition_penalty`` and ``ignore_eos`` give explicit control over repetition handling and whether the model stops at the EOS token, respectively.
- Common parameters shared with OpenAI APIs (e.g., ``temperature``, ``top_p``) ensure compatibility while still falling back to the values configured during ``MLCEngine.reload()`` when omitted.
- ``frequency_penalty`` and ``presence_penalty`` mirror OpenAI's bounds ``[-2, 2]``; providing only one will default the other to ``0``.
- ``response_format`` (for JSON or other schema outputs), ``enable_thinking``, and ``enable_latency_breakdown`` pass through directly to the engine and surface enhanced telemetry or structured responses when the underlying model supports them.
Example:
.. code-block:: typescript
const messages = [
{ role: "system", content: "You are a helpful assistant." },
{ role: "user", content: "Explain WebLLM." },
];
const response = await engine.chatCompletion({
messages,
top_p: 0.9,
temperature: 0.8,
max_tokens: 150,
});
ChatConfig
^^^^^^^^^^
Model's baseline configuration loaded from ``mlc-chat-config.json`` when ``MLCEngine.reload()`` runs. ``ChatOptions`` (and therefore the ``chatOpts`` argument to ``reload``) can override any subset of these fields.
- **Fields** (subset):
- ``tokenizer_files``, ``tokenizer_info``: Files and parameters required to initialize the tokenizer.
- ``conv_template``, ``conv_config``: Conversation templates that define prompts, separators, and role formatting.
- ``context_window_size``, ``sliding_window_size``, ``attention_sink_size``: KV-cache and memory settings.
- Default generation knobs such as ``repetition_penalty``, ``frequency_penalty``, ``presence_penalty``, ``top_p``, and ``temperature``.
- **Usage**:
- Loaded automatically for each model; provides defaults that ``GenerationConfig`` falls back to when fields are omitted.
- Override selected values per model load by supplying ``chatOpts`` (``Partial<ChatConfig>``) to ``MLCEngine.reload()``.
Example:
.. code-block:: typescript
await engine.reload("Llama-3.1-8B-Instruct", {
temperature: 0.7,
repetition_penalty: 1.1,
context_window_size: 4096,
});
ChatCompletionRequest
^^^^^^^^^^^^^^^^^^^^^
Defines the structure for chat completion requests.
- **Base Interface**: ``ChatCompletionRequestBase``
- Contains parameters such as ``messages``, ``stream``, ``frequency_penalty``, and ``presence_penalty``.
- **Sub-interfaces**:
- ``ChatCompletionRequestNonStreaming``: For non-streaming completions.
- ``ChatCompletionRequestStreaming``: For streaming completions.
- **Usage**:
- Combines settings from ``GenerationConfig`` and ``ChatCompletionRequestBase`` to provide complete control over chat behavior.
- The ``stream`` parameter enables streaming responses, improving interactivity in conversational agents.
- The ``logit_bias`` feature allows controlling token generation probabilities, providing a mechanism to restrict or encourage specific outputs.
Example:
.. code-block:: typescript
const response = await engine.chatCompletion({
messages: [
{ role: "user", content: "Tell me about WebLLM." },
],
stream: true,
});
Model Loading
-------------
``MLCEngine.reload(modelId: string | string[], chatOpts?: ChatOptions | ChatOptions[]): Promise<void>``
Loads the specified model(s) into the engine. Uses ``MLCEngineConfig`` during initialization.
- Parameters:
- ``modelId``: Identifier(s) for the model(s) to load.
- ``chatOpts``: Configuration for generation (see ``ChatConfig``).
Example:
.. code-block:: typescript
await engine.reload(["Llama-3.1-8B", "Gemma-2B"], [
{ temperature: 0.7 },
{ top_p: 0.9 },
]);
``MLCEngine.unload(): Promise<void>``
Unloads all loaded models and clears their associated configurations.
Example:
.. code-block:: typescript
await engine.unload();
---
Chat Completions
----------------
``MLCEngine.chat.completions.create(request: ChatCompletionRequest): Promise<ChatCompletion | AsyncIterable<ChatCompletionChunk>>``
Generates chat-based completions using a specified request configuration.
- Parameters:
- ``request``: A ``ChatCompletionRequest`` instance.
Example:
.. code-block:: typescript
const response = await engine.chat.completions.create({
messages: [
{ role: "system", content: "You are a helpful AI assistant." },
{ role: "user", content: "What is WebLLM?" },
],
temperature: 0.8,
stream: false,
});
---
Utility Methods
^^^^^^^^^^^^^^^
``MLCEngine.getMessage(modelId?: string): Promise<string>``
Retrieves the current output message from the specified model.
- Parameters:
- ``modelId``: (Optional) Identifier of model to query. Omitting modelId only works when the engine currently has a single model loaded.
``MLCEngine.resetChat(keepStats?: boolean, modelId?: string): Promise<void>``
Resets the chat history and optionally retains usage statistics.
- Parameters:
- ``keepStats``: (Optional) If true, retains usage statistics.
- ``modelId``: (Optional) Identifier of the model to reset. Omitting modelId only works when the engine currently has a single model loaded.
GPU Information
----------------
The following methods provide detailed information about the GPU used for WebLLM computations.
``MLCEngine.getGPUVendor(): Promise<string>``
Retrieves the vendor name of the GPU used for computations. This is useful for understanding hardware capabilities during inference.
- **Returns**: A string indicating the GPU vendor (e.g., "Intel", "NVIDIA").
Example:
.. code-block:: typescript
const gpuVendor = await engine.getGPUVendor();
console.log(``GPU Vendor: ${gpuVendor}``);
``MLCEngine.getMaxStorageBufferBindingSize(): Promise<number>``
Returns the maximum storage buffer size supported by the GPU. This is important when working with larger models that require significant memory for processing.
- **Returns**: A number representing the maximum size in bytes.
Example:
.. code-block:: typescript
const maxBufferSize = await engine.getMaxStorageBufferBindingSize();
console.log(``Max Storage Buffer Binding Size: ${maxBufferSize}``);
+120
View File
@@ -0,0 +1,120 @@
Basic Usage
================
Model Records in WebLLM
-----------------------
Each of the model available WebLLM is registered as an instance of
``ModelRecord`` and can be accessed at
`webllm.prebuiltAppConfig.model_list <https://github.com/mlc-ai/web-llm/blob/main/src/config.ts#L313>`__.
Creating an MLCEngine
---------------------
WebLLM APIs are exposed through the ``MLCEngine`` interface. You can create an ``MLCEngine`` instance and load the model by calling the CreateMLCEngine() factory function.
(Note that loading models requires downloading and it can take a significant amount of time for the very first run without previous caching. You should properly handle this asynchronous call.)
``MLCEngine`` can be instantiated in two ways:
1. Using the factory function ``CreateMLCEngine``.
2. Instantiating the ``MLCEngine`` class directly and using ``reload()`` to load models.
.. code-block:: typescript
import { CreateMLCEngine, MLCEngine } from "@mlc-ai/web-llm";
// Initialize with a progress callback
const initProgressCallback = (progress) => {
console.log("Model loading progress:", progress);
};
// Using CreateMLCEngine
const engine = await CreateMLCEngine("Llama-3.1-8B-Instruct", { initProgressCallback });
// Direct instantiation
const engineInstance = new MLCEngine({ initProgressCallback });
await engineInstance.reload("Llama-3.1-8B-Instruct");
Under the hood, this factory function ``CreateMLCEngine`` does the following steps for first creating an engine instance (synchronous) and then loading the model (asynchronous). You can also do them separately in your application.
.. code-block:: typescript
import { MLCEngine } from "@mlc-ai/web-llm";
// This is a synchronous call that returns immediately
const engine = new MLCEngine({
initProgressCallback: initProgressCallback
});
// This is an asynchronous call and can take a long time to finish
await engine.reload(selectedModel);
Chat Completion
---------------
Chat completions can be invoked using OpenAI style chat APIs through the ``engine.chat.completions`` interface of an initialized ``MLCEngine``. For the full list of parameters and their descriptions, check :ref:`api-reference` for full list of parameters.
(Note: Since the model is determined during ``MLCEngine`` instantiation, the ``model`` parameter is not supported and will be **ignored**. Instead, call ``CreateMLCEngine(model)`` or ``engine.reload(model)`` to reinitialize the engine to use a specific model.)
.. code-block:: typescript
const messages = [
{ role: "system", content: "You are a helpful AI assistant." },
{ role: "user", content: "Hello!" }
];
const reply = await engine.chat.completions.create({
messages,
});
console.log(reply.choices[0].message);
console.log(reply.usage);
Streaming Chat Completion
-------------------------
Streaming chat completion could be enabled by passsing ``stream: true`` parameter to the `engine.chat.completions.create` call configuration. Check :ref:`api-reference` for full list of parameters.
.. code-block:: typescript
const messages = [
{ role: "system", content: "You are a helpful AI assistant." },
{ role: "user", content: "Hello!" },
]
// chunks is an AsyncGenerator object
const chunks = await engine.chat.completions.create({
messages,
temperature: 1,
stream: true, // <-- Enable streaming
stream_options: { include_usage: true },
});
let reply = "";
for await (const chunk of chunks) {
reply += chunk.choices[0]?.delta.content || "";
console.log(reply);
if (chunk.usage) {
console.log(chunk.usage); // only last chunk has usage
}
}
const fullReply = await engine.getMessage();
console.log(fullReply);
Chatbot Examples
----------------
Learn how to use WebLLM to integrate large language models into your applications and generate chat completions through this simple Chatbot example:
- `Example in JSFiddle <https://jsfiddle.net/neetnestor/4nmgvsa2/>`_
- `Example in CodePen <https://codepen.io/neetnestor/pen/vYwgZaG>`_
For an advanced example of a larger, more complicated project, look at `WebLLM Chat <https://github.com/mlc-ai/web-llm-chat/blob/main/app/client/webllm.ts>`_.
More examples for different use cases are available in the `WebLLM examples folder <https://github.com/mlc-ai/web-llm/tree/main/examples>`_.
+75
View File
@@ -0,0 +1,75 @@
Getting Started with WebLLM
===========================
This guide will help you set up WebLLM in your project, install necessary dependencies, and verify your setup.
WebLLM Chat
-----------
If you want to experience AI Chat supported by local LLM inference and understand how WebLLM works, try out `WebLLM Chat <https://chat.webllm.ai/>`__, which provides a great example
of integrating WebLLM into a full web application.
A WebGPU-compatible browser is needed to run WebLLM-powered web applications.
You can download the latest Google Chrome and use `WebGPU Report <https://webgpureport.org/>`__
to verify the functionality of WebGPU on your browser.
Installation
------------
WebLLM offers a minimalist and modular interface to access the chatbot in the browser. The package is designed in a modular way to hook to any of the UI components.
WebLLM is available as an `npm package <https://www.npmjs.com/package/@mlc-ai/web-llm>`_ and is also CDN-delivered. Therefore, you can install WebLLM using Node.js package managers like npm, yarn, or pnpm, or directly import the pacakge via CDN.
Using Package Managers
^^^^^^^^^^^^^^^^^^^^^^
Install WebLLM via your preferred package manager:
.. code-block:: bash
# npm
npm install @mlc-ai/web-llm
# yarn
yarn add @mlc-ai/web-llm
# pnpm
pnpm install @mlc-ai/web-llm
Import WebLLM into your project:
.. code-block:: javascript
// Import everything
import * as webllm from "@mlc-ai/web-llm";
// Or only import what you need
import { CreateMLCEngine } from "@mlc-ai/web-llm";
Using CDN
^^^^^^^^^
Thanks to `jsdelivr.com <https://www.jsdelivr.com/package/npm/@mlc-ai/web-llm>`_, WebLLM can be imported directly through URL and work out-of-the-box on cloud development platforms like `jsfiddle.net <https://jsfiddle.net/>`_, `Codepen.io <https://codepen.io/>`_, and `Scribbler <https://scribbler.live/>`_:
.. code-block:: javascript
import * as webllm from "https://esm.run/@mlc-ai/web-llm";
This method is especially useful for online environments like CodePen, JSFiddle, or local experiments.
Verifying Installation
^^^^^^^^^^^^^^^^^^^^^^
Run the following script to verify the installation:
.. code-block:: javascript
import { CreateMLCEngine } from "@mlc-ai/web-llm";
console.log("WebLLM loaded successfully!");
Online IDE Sandbox
------------------
Instead of setting WebLLM locally, you can also try it on online Javascript IDE sandboxes like:
- `Example in JSFiddle <https://jsfiddle.net/neetnestor/4nmgvsa2/>`_
- `Example in CodePen <https://codepen.io/neetnestor/pen/vYwgZaG>`_