{"data":{"node":{"slug":"epistates-pmetal","name":"pmetal","tagline":"High-performance Apple Silicon framework for LLM inference and fine-tuning","github_url":"https://github.com/Epistates/pmetal","owner":"Epistates","repo":"pmetal","owner_avatar_url":"https://avatars.githubusercontent.com/u/225645584?v=4","primary_language":"Rust","stars":317,"forks":26,"topics":["ai","ane","apple-silicon","deep-learning","distillation","fine-tuning","gguf","inference-server","llm","llm-inference","llm-training","lora","machine-learning","macos","metal","mlx","qlora","quantization","transformers","tui"],"archived":false,"github_pushed_at":"2026-09-17T15:41:39+00:00","maintenance_label":"Very active","stars_delta_30d":11,"url":"https://www.graphcanon.com/tools/epistates-pmetal","markdown_url":"https://www.graphcanon.com/tools/epistates-pmetal.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/epistates-pmetal","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=epistates-pmetal"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"},{"slug":"model-training","name":"Model Training","url":"https://www.graphcanon.com/categories/model-training","markdown_url":"https://www.graphcanon.com/categories/model-training.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/model-training"}],"tags":[{"slug":"ai","name":"ai"},{"slug":"ane","name":"ane"},{"slug":"apple-silicon","name":"apple-silicon"},{"slug":"deep-learning","name":"deep-learning"},{"slug":"fine-tuning","name":"fine-tuning"},{"slug":"inference-server","name":"inference-server"},{"slug":"llm-inference","name":"llm-inference"},{"slug":"mlx","name":"mlx"}],"edges":[],"neighbours":[{"slug":"nomic-ai-gpt4all","name":"gpt4all","tagline":"Run Local LLMs on Any Device","github_url":"https://github.com/nomic-ai/gpt4all","owner":"nomic-ai","repo":"gpt4all","owner_avatar_url":"https://avatars.githubusercontent.com/u/102670180?v=4","primary_language":"C++","stars":77390,"forks":8288,"topics":["ai-chat","llm-inference"],"archived":false,"github_pushed_at":"2025-05-27T20:05:19+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/nomic-ai-gpt4all","markdown_url":"https://www.graphcanon.com/tools/nomic-ai-gpt4all.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nomic-ai-gpt4all","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nomic-ai-gpt4all","shared_categories":["inference-serving"]},{"slug":"jundot-omlx","name":"omlx","tagline":"LLM inference server with continuous batching and SSD caching for Apple Silicon","github_url":"https://github.com/jundot/omlx","owner":"jundot","repo":"omlx","owner_avatar_url":"https://avatars.githubusercontent.com/u/64250138?v=4","primary_language":"Python","stars":21934,"forks":1899,"topics":["apple-silicon","inference-server","llm","macos","mlx","openai-api"],"archived":false,"github_pushed_at":"2026-09-20T02:23:26+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/jundot-omlx","markdown_url":"https://www.graphcanon.com/tools/jundot-omlx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jundot-omlx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jundot-omlx","shared_categories":["inference-serving"]},{"slug":"tloen-alpaca-lora","name":"alpaca-lora","tagline":"Instruct-tune LLaMA on consumer hardware","github_url":"https://github.com/tloen/alpaca-lora","owner":"tloen","repo":"alpaca-lora","owner_avatar_url":"https://avatars.githubusercontent.com/u/4811103?v=4","primary_language":"Jupyter Notebook","stars":18911,"forks":2174,"topics":[],"archived":false,"github_pushed_at":"2024-07-29T13:37:49+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/tloen-alpaca-lora","markdown_url":"https://www.graphcanon.com/tools/tloen-alpaca-lora.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tloen-alpaca-lora","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tloen-alpaca-lora","shared_categories":["model-training","inference-serving"]},{"slug":"bitsandbytes-foundation-bitsandbytes","name":"bitsandbytes","tagline":"Large language model quantization toolkit for PyTorch.","github_url":"https://github.com/bitsandbytes-foundation/bitsandbytes","owner":"bitsandbytes-foundation","repo":"bitsandbytes","owner_avatar_url":"https://avatars.githubusercontent.com/u/175231607?v=4","primary_language":"Python","stars":8455,"forks":920,"topics":["llm","machine-learning","pytorch","qlora","quantization"],"archived":false,"github_pushed_at":"2026-08-27T00:16:57+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes","markdown_url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bitsandbytes-foundation-bitsandbytes","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bitsandbytes-foundation-bitsandbytes","shared_categories":["inference-serving"]},{"slug":"argmaxinc-argmax-oss-swift","name":"argmax-oss-swift","tagline":"On-device Speech AI for Apple Silicon","github_url":"https://github.com/argmaxinc/argmax-oss-swift","owner":"argmaxinc","repo":"argmax-oss-swift","owner_avatar_url":"https://avatars.githubusercontent.com/u/150409474?v=4","primary_language":"Swift","stars":6369,"forks":611,"topics":["inference","ios","macos","pyannote","qwen3-tts","speaker-diarization","speakerkit","speech-recognition","speech-to-text","swift","text-to-speech","transformers","ttskit","whisper","whisperkit"],"archived":false,"github_pushed_at":"2026-08-13T19:17:27+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/argmaxinc-argmax-oss-swift","markdown_url":"https://www.graphcanon.com/tools/argmaxinc-argmax-oss-swift.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/argmaxinc-argmax-oss-swift","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=argmaxinc-argmax-oss-swift","shared_categories":[]},{"slug":"turboderp-exllama","name":"exllama","tagline":"Memory-efficient rewrite of HF transformers for Llama with quantized weights","github_url":"https://github.com/turboderp/exllama","owner":"turboderp","repo":"exllama","owner_avatar_url":"https://avatars.githubusercontent.com/u/11859846?v=4","primary_language":"Python","stars":2943,"forks":220,"topics":[],"archived":false,"github_pushed_at":"2023-09-30T19:06:04+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/turboderp-exllama","markdown_url":"https://www.graphcanon.com/tools/turboderp-exllama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/turboderp-exllama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=turboderp-exllama","shared_categories":["inference-serving"]},{"slug":"waybarrios-vllm-mlx","name":"vllm-mlx","tagline":"Server for LLMs and vision-language models compatible with Apple Silicon","github_url":"https://github.com/waybarrios/vllm-mlx","owner":"waybarrios","repo":"vllm-mlx","owner_avatar_url":"https://avatars.githubusercontent.com/u/6794828?v=4","primary_language":"Python","stars":1588,"forks":222,"topics":["anthropic","anthropic-api","apple-silicon","claude-code","continuous-batching","inference-server","llm","local-llm","macos","mcp","mlx","multimodal-ai","openai","openai-api","openai-compatible","speech-to-text","text-to-speech","tool-calling","vision-language-model","vllm"],"archived":false,"github_pushed_at":"2026-09-19T18:11:07+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/waybarrios-vllm-mlx","markdown_url":"https://www.graphcanon.com/tools/waybarrios-vllm-mlx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/waybarrios-vllm-mlx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=waybarrios-vllm-mlx","shared_categories":["model-training","inference-serving"]},{"slug":"xllm-ai-xllm","name":"xllm","tagline":"A high-performance inference engine for LLM, VLM, DiT and REC models","github_url":"https://github.com/xLLM-AI/xllm","owner":"xLLM-AI","repo":"xllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/205719415?v=4","primary_language":"C++","stars":1577,"forks":300,"topics":["deepseek","glm","inference","inference-engine","large-language-models","llm-inference","qwen"],"archived":false,"github_pushed_at":"2026-09-19T13:15:59+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/xllm-ai-xllm","markdown_url":"https://www.graphcanon.com/tools/xllm-ai-xllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xllm-ai-xllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xllm-ai-xllm","shared_categories":["inference-serving"]},{"slug":"ddalcu-mlx-serve","name":"mlx-serve","tagline":"Native LLM inference server for Apple Silicon","github_url":"https://github.com/ddalcu/mlx-serve","owner":"ddalcu","repo":"mlx-serve","owner_avatar_url":"https://avatars.githubusercontent.com/u/869085?v=4","primary_language":"Zig","stars":1418,"forks":130,"topics":["agent","anthropic-api","apple-silicon","claude-code","deepseek-v4","diffusion","gguf","image-generation","inference","llm","local-llm","macos","macos-app","mlx","openai-api","tool-calling","video-generation","voice-agent","voice-cloning","zig"],"archived":false,"github_pushed_at":"2026-09-19T22:08:35+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/ddalcu-mlx-serve","markdown_url":"https://www.graphcanon.com/tools/ddalcu-mlx-serve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ddalcu-mlx-serve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ddalcu-mlx-serve","shared_categories":["inference-serving"]},{"slug":"arahim3-mlx-tune","name":"mlx-tune","tagline":"Fine-tune LLMs on your Mac with Apple Silicon for various tasks including SFT, DPO, GRPO, Vision, TTS, STT, Embedding, and OCR.","github_url":"https://github.com/ARahim3/mlx-tune","owner":"ARahim3","repo":"mlx-tune","owner_avatar_url":"https://avatars.githubusercontent.com/u/41390319?v=4","primary_language":"Python","stars":1412,"forks":92,"topics":["apple-silicon","deep-learning","huggingface","large-language-models","llm","llm-finetuning","local-llm","lora","machine-learning","macos","mlx","on-device-ai","peft","speech-recognition","speech-to-text","text-to-speech","transformers","unsloth","vision-language-model","whisper"],"archived":false,"github_pushed_at":"2026-06-23T12:24:30+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/arahim3-mlx-tune","markdown_url":"https://www.graphcanon.com/tools/arahim3-mlx-tune.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/arahim3-mlx-tune","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=arahim3-mlx-tune","shared_categories":["model-training"]},{"slug":"rudrankriyam-foundation-models-framework-lab","name":"Foundation-Models-Framework-Lab","tagline":"A practical lab for building, testing, and evaluating apps with Apple's Foundation Models framework","github_url":"https://github.com/rudrankriyam/Foundation-Models-Framework-Lab","owner":"rudrankriyam","repo":"Foundation-Models-Framework-Lab","owner_avatar_url":"https://avatars.githubusercontent.com/u/30552772?v=4","primary_language":"Swift","stars":1181,"forks":70,"topics":["ai","apple-foundation-models","apple-intelligence","foundation-models","foundation-models-framework","generative-ai","healthkit","ios","large-language-models","llm","macos","multilingual","on-device-ai","rag","speech-recognition","swift","swiftui","text-to-speech","tool-calling","xcode"],"archived":false,"github_pushed_at":"2026-09-18T23:34:01+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/rudrankriyam-foundation-models-framework-lab","markdown_url":"https://www.graphcanon.com/tools/rudrankriyam-foundation-models-framework-lab.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rudrankriyam-foundation-models-framework-lab","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rudrankriyam-foundation-models-framework-lab","shared_categories":[]},{"slug":"jmaczan-tiny-vllm","name":"tiny-vllm","tagline":"Build your own high performance LLM inference engine in C++ and CUDA - a smaller version of vLLM","github_url":"https://github.com/jmaczan/tiny-vllm","owner":"jmaczan","repo":"tiny-vllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/18054202?v=4","primary_language":"C++","stars":1120,"forks":92,"topics":["ai","attention","batching","course","cpp","cuda","hpc","inference","llm","llm-inference","pagedattention","tiny-vllm","vllm"],"archived":false,"github_pushed_at":"2026-09-15T19:46:09+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/jmaczan-tiny-vllm","markdown_url":"https://www.graphcanon.com/tools/jmaczan-tiny-vllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jmaczan-tiny-vllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jmaczan-tiny-vllm","shared_categories":["inference-serving"]}]}}