{"data":{"node":{"slug":"ngxson-wllama","name":"wllama","tagline":"WebAssembly binding for llama.cpp - Enabling on-browser LLM inference","github_url":"https://github.com/ngxson/wllama","owner":"ngxson","repo":"wllama","owner_avatar_url":"https://avatars.githubusercontent.com/u/7702203?v=4","primary_language":"TypeScript","stars":1159,"forks":117,"topics":["llama","llamacpp","llm","wasm","webassembly"],"archived":false,"github_pushed_at":"2026-06-17T17:32:59+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/ngxson-wllama","markdown_url":"https://www.graphcanon.com/tools/ngxson-wllama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ngxson-wllama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ngxson-wllama"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"llama","name":"llama"},{"slug":"llamacpp","name":"llamacpp"},{"slug":"llm","name":"llm"},{"slug":"wasm","name":"wasm"},{"slug":"webassembly","name":"webassembly"}],"edges":[{"type":"integrates_with","direction":"in","explanation":"`wllama` provides WebAssembly bindings for `llama.cpp`, enabling on-browser LLM inference, indicating they integrate well.","successor_context":null,"tool":{"slug":"ggml-org-llama-cpp","name":"llama.cpp","tagline":"LLM inference in C/C++","github_url":"https://github.com/ggml-org/llama.cpp","owner":"ggml-org","repo":"llama.cpp","owner_avatar_url":"https://avatars.githubusercontent.com/u/134263123?v=4","primary_language":"C++","stars":122941,"forks":21406,"topics":["ggml"],"archived":false,"github_pushed_at":"2026-08-07T05:28:54+00:00","maintenance_label":"Very active","stars_delta_30d":3353,"url":"https://www.graphcanon.com/tools/ggml-org-llama-cpp","markdown_url":"https://www.graphcanon.com/tools/ggml-org-llama-cpp.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ggml-org-llama-cpp","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ggml-org-llama-cpp"}}],"neighbours":[{"slug":"ggml-org-llama-cpp","name":"llama.cpp","tagline":"LLM inference in C/C++","github_url":"https://github.com/ggml-org/llama.cpp","owner":"ggml-org","repo":"llama.cpp","owner_avatar_url":"https://avatars.githubusercontent.com/u/134263123?v=4","primary_language":"C++","stars":122941,"forks":21406,"topics":["ggml"],"archived":false,"github_pushed_at":"2026-08-07T05:28:54+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ggml-org-llama-cpp","markdown_url":"https://www.graphcanon.com/tools/ggml-org-llama-cpp.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ggml-org-llama-cpp","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ggml-org-llama-cpp","shared_categories":["inference-serving"]},{"slug":"vllm-project-vllm","name":"vllm","tagline":"A high-throughput and memory-efficient inference and serving engine for LLMs","github_url":"https://github.com/vllm-project/vllm","owner":"vllm-project","repo":"vllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/136984999?v=4","primary_language":"Python","stars":87847,"forks":20135,"topics":["amd","blackwell","cuda","deepseek","deepseek-v3","gpt","gpt-oss","inference","kimi","llama","llm","llm-serving","model-serving","moe","openai","pytorch","qwen","qwen3","tpu","transformer"],"archived":false,"github_pushed_at":"2026-08-01T11:55:36+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/vllm-project-vllm","markdown_url":"https://www.graphcanon.com/tools/vllm-project-vllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/vllm-project-vllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=vllm-project-vllm","shared_categories":["inference-serving"]},{"slug":"mozilla-ai-llamafile","name":"llamafile","tagline":"Distribute and run LLMs with a single file.","github_url":"https://github.com/mozilla-ai/llamafile","owner":"mozilla-ai","repo":"llamafile","owner_avatar_url":"https://avatars.githubusercontent.com/u/129804596?v=4","primary_language":"C++","stars":25470,"forks":1530,"topics":["cross-platform","gguf","llama-cpp","local-ai","local-inference","local-llm","open-source-ai","single-file-executable","speech-to-text"],"archived":false,"github_pushed_at":"2026-07-27T14:21:26+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/mozilla-ai-llamafile","markdown_url":"https://www.graphcanon.com/tools/mozilla-ai-llamafile.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/mozilla-ai-llamafile","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=mozilla-ai-llamafile","shared_categories":["inference-serving"]},{"slug":"jundot-omlx","name":"omlx","tagline":"LLM inference server with continuous batching and SSD caching for Apple Silicon","github_url":"https://github.com/jundot/omlx","owner":"jundot","repo":"omlx","owner_avatar_url":"https://avatars.githubusercontent.com/u/64250138?v=4","primary_language":"Python","stars":18679,"forks":1617,"topics":["apple-silicon","inference-server","llm","macos","mlx","openai-api"],"archived":false,"github_pushed_at":"2026-08-14T09:12:48+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/jundot-omlx","markdown_url":"https://www.graphcanon.com/tools/jundot-omlx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jundot-omlx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jundot-omlx","shared_categories":["inference-serving"]},{"slug":"michael-a-kuykendall-shimmy","name":"shimmy","tagline":"⚡ A Pure-Rust WebGPU Inference Engine, OpenAI-API Compatible and Native to GGUF","github_url":"https://github.com/Michael-A-Kuykendall/shimmy","owner":"Michael-A-Kuykendall","repo":"shimmy","owner_avatar_url":"https://avatars.githubusercontent.com/u/196678413?v=4","primary_language":"Rust","stars":5808,"forks":559,"topics":["api-server","command-line-tool","developer-tools","gguf","huggingface","huggingface-models","huggingface-transformers","inference-server","llama","llamacpp","llm-inference","local-ai","machine-learning","ollama-api","openai-compatible","rust","rust-crate","transformers","webgpu","webgpu-shaders"],"archived":false,"github_pushed_at":"2026-08-20T17:39:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/michael-a-kuykendall-shimmy","markdown_url":"https://www.graphcanon.com/tools/michael-a-kuykendall-shimmy.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/michael-a-kuykendall-shimmy","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=michael-a-kuykendall-shimmy","shared_categories":["inference-serving"]},{"slug":"ollama-ollama-js","name":"ollama-js","tagline":"Integrate JavaScript projects with Ollama using this TypeScript library","github_url":"https://github.com/ollama/ollama-js","owner":"ollama","repo":"ollama-js","owner_avatar_url":"https://avatars.githubusercontent.com/u/151674099?v=4","primary_language":"TypeScript","stars":4340,"forks":464,"topics":["javascript","js","ollama"],"archived":false,"github_pushed_at":"2026-02-18T22:06:18+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/ollama-ollama-js","markdown_url":"https://www.graphcanon.com/tools/ollama-ollama-js.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ollama-ollama-js","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ollama-ollama-js","shared_categories":["inference-serving"]},{"slug":"turboderp-exllama","name":"exllama","tagline":"Memory-efficient rewrite of HF transformers for Llama with quantized weights","github_url":"https://github.com/turboderp/exllama","owner":"turboderp","repo":"exllama","owner_avatar_url":"https://avatars.githubusercontent.com/u/11859846?v=4","primary_language":"Python","stars":2937,"forks":220,"topics":[],"archived":false,"github_pushed_at":"2023-09-30T19:06:04+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/turboderp-exllama","markdown_url":"https://www.graphcanon.com/tools/turboderp-exllama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/turboderp-exllama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=turboderp-exllama","shared_categories":["inference-serving"]},{"slug":"liltom-eth-llama2-webui","name":"llama2-webui","tagline":"Run Llama 2 locally with gradio UI on GPU or CPU","github_url":"https://github.com/liltom-eth/llama2-webui","owner":"liltom-eth","repo":"llama2-webui","owner_avatar_url":"https://avatars.githubusercontent.com/u/11456256?v=4","primary_language":"Jupyter Notebook","stars":1937,"forks":199,"topics":["llama-2","llama2","llm","llm-inference"],"archived":false,"github_pushed_at":"2024-03-22T09:50:24+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/liltom-eth-llama2-webui","markdown_url":"https://www.graphcanon.com/tools/liltom-eth-llama2-webui.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/liltom-eth-llama2-webui","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=liltom-eth-llama2-webui","shared_categories":["inference-serving"]},{"slug":"waybarrios-vllm-mlx","name":"vllm-mlx","tagline":"Server for LLMs and vision-language models compatible with Apple Silicon","github_url":"https://github.com/waybarrios/vllm-mlx","owner":"waybarrios","repo":"vllm-mlx","owner_avatar_url":"https://avatars.githubusercontent.com/u/6794828?v=4","primary_language":"Python","stars":1472,"forks":205,"topics":["anthropic","apple-silicon","audio-processing","claude-code","computer-vision","image-understanding","inference","llm","machine-learning","macos","mllm","mlx","multimodal-ai","speech-to-text","stt","text-to-speech","tts","video-understanding","vision-language-model","vllm"],"archived":false,"github_pushed_at":"2026-06-28T20:18:31+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/waybarrios-vllm-mlx","markdown_url":"https://www.graphcanon.com/tools/waybarrios-vllm-mlx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/waybarrios-vllm-mlx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=waybarrios-vllm-mlx","shared_categories":["inference-serving"]},{"slug":"sauravpanda-browserai","name":"BrowserAI","tagline":"Run local LLMs like llama, deepseek-distill, kokoro and more inside your browser","github_url":"https://github.com/sauravpanda/BrowserAI","owner":"sauravpanda","repo":"BrowserAI","owner_avatar_url":"https://avatars.githubusercontent.com/u/12201824?v=4","primary_language":"TypeScript","stars":1449,"forks":138,"topics":["agents","ai","llama","llm","llm-inference","local","localllm","tts","webgpu"],"archived":false,"github_pushed_at":"2026-07-21T02:47:31+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/sauravpanda-browserai","markdown_url":"https://www.graphcanon.com/tools/sauravpanda-browserai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/sauravpanda-browserai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=sauravpanda-browserai","shared_categories":["inference-serving"]},{"slug":"jmaczan-tiny-vllm","name":"tiny-vllm","tagline":"Build your own high performance LLM inference engine in C++ and CUDA - a smaller version of vLLM","github_url":"https://github.com/jmaczan/tiny-vllm","owner":"jmaczan","repo":"tiny-vllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/18054202?v=4","primary_language":"C++","stars":1075,"forks":84,"topics":["ai","attention","batching","course","cpp","cuda","hpc","inference","llm","llm-inference","pagedattention","tiny-vllm","vllm"],"archived":false,"github_pushed_at":"2026-08-23T14:20:13+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/jmaczan-tiny-vllm","markdown_url":"https://www.graphcanon.com/tools/jmaczan-tiny-vllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jmaczan-tiny-vllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jmaczan-tiny-vllm","shared_categories":["inference-serving"]},{"slug":"mukel-llama3-java","name":"llama3.java","tagline":"Llama 3+ inference in pure Java","github_url":"https://github.com/mukel/llama3.java","owner":"mukel","repo":"llama3.java","owner_avatar_url":"https://avatars.githubusercontent.com/u/1896283?v=4","primary_language":"Java","stars":815,"forks":94,"topics":["chatgpt","genai","gguf","huggingface","java","llama","llama3","llamacpp","llm","llm-inference","llms","openai","simd","transformers"],"archived":false,"github_pushed_at":"2026-04-24T16:38:39+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/mukel-llama3-java","markdown_url":"https://www.graphcanon.com/tools/mukel-llama3-java.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/mukel-llama3-java","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=mukel-llama3-java","shared_categories":["inference-serving"]}]}}