{"data":{"slug":"openinfer-project-openinfer","name":"openinfer","tagline":"Pure Rust CUDA LLM inference engine serving multiple models including Qwen3 and Kimi-K2","github_url":"https://github.com/openinfer-project/openinfer","owner":"openinfer-project","repo":"openinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/292134277?v=4","primary_language":"Rust","stars":657,"forks":103,"topics":["cuda","cuda-kernels","deepseek","gpu","inference","inference-engine","kimi","kimi-k2","kv-cache","llm","llm-inference","llm-serving","model-serving","moe","openai-api","paged-attention","qwen","qwen3","rust","vllm"],"archived":false,"github_pushed_at":"2026-08-25T05:44:30+00:00","maintenance_label":"Very active","stars_delta_30d":72,"url":"https://www.graphcanon.com/tools/openinfer-project-openinfer","markdown_url":"https://www.graphcanon.com/tools/openinfer-project-openinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openinfer-project-openinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openinfer-project-openinfer","description":"Pure Rust + CUDA LLM inference engine — no PyTorch, OpenAI-compatible, serves Qwen3 to Kimi-K2","homepage_url":"https://pegainfer.org/","license":"Apache-2.0","open_issues":102,"watchers":3,"ai_summary":"Provides a high-performance inference solution for large language models on GPU with no dependency on PyTorch, offering an OpenAI API-compatible service.","readme_excerpt":"## License\n\nApache-2.0 — see [LICENSE](LICENSE) and [NOTICE](NOTICE). Components ported from\nNVIDIA Dynamo (the `kvbm/kvbm-logical` crate) retain their original Apache-2.0 headers; see\n[NOTICE_DYNAMO](NOTICE_DYNAMO).","github_created_at":"2026-02-17T12:01:22+00:00","created_at":"2026-07-11T11:45:24.554146+00:00","updated_at":"2026-08-25T06:02:10.377058+00:00","categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"cuda","name":"cuda"},{"slug":"gpu","name":"gpu"},{"slug":"llm-inference","name":"llm-inference"},{"slug":"openai-api","name":"openai-api"},{"slug":"rust","name":"rust"}],"trust":{"provenance":{"is_fork":false,"github_id":1159983293,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-08-25T06:02:09.470Z","maintenance":{"label":"Very active","score":96,"methodology":"github_public_v1","releases_90d":1,"days_since_push":0,"last_release_at":"2026-06-13T12:17:37Z","stars_delta_30d":72,"open_issues_delta_30d":-26},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-11T11:45:25.741Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-08-25T06:02:09.999Z"},"languages":{"value":["rust"],"source":"github.language","observed_at":"2026-08-25T06:02:09.999Z"},"license_spdx":{"value":"Apache-2.0","source":"github.license","observed_at":"2026-08-25T06:02:09.999Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":null,"constraints":null,"when_to_use":["When you are working with large language models Qwen3 and/or Kimi-K2 specifically, and want to avoid PyTorch dependencies.","If your project requires a high-performance inference engine that is compatible with the OpenAI API, built for Rust developers."],"when_not_to_use":["Avoid if you are developing models other than Qwen3 or Kimi-K2 as support for other models might be limited.","Not recommended for projects where PyTorch integration is crucial as this tool does not depend on it and may require changes in existing workflows."],"source":"enrich:decision_facts","observed_at":"2026-07-16T23:12:55.982Z"},"constraint_facets":null,"decision_summary":[{"label":"Adopt for","value":"high-performance GPU-based inference engine for Rust developers targeting Qwen3 and Kimi-K2 using pure CUDA kernels"}]}}