[{"data":1,"prerenderedAt":111},["ShallowReactive",2],{"repo-vllm-project-semantic-router":3,"i18n-vllm-project-semantic-router-zh-CN":108},{"mirrorPool":4,"detail":13},[5,9],{"key":6,"label":7,"base":8,"type":6},"github","mirror.github","https:\u002F\u002Fgithub.com\u002F",{"key":10,"label":11,"base":12,"type":10},"auto-proxy","mirror.autoProxy","https:\u002F\u002Fgithub.xxlab.tech\u002F",{"owner":14,"repo":15,"repoId":16,"repoUrl":17,"starCount":18,"language":19,"description":20,"languages":21,"screenshots":35,"homepage":36,"pushedAt":37,"ownerAvatar":38,"ownerName":39,"ownerBio":40,"lastSyncTime":41,"lastReleaseTime":42,"tags":43,"license":54,"category":55,"subgroup":56,"releases":57},"vllm-project","semantic-router",1045247072,"https:\u002F\u002Fgithub.com\u002Fvllm-project\u002Fsemantic-router",5959,"Go","A programmable Mixture-of-Models router for heterogeneous LLM inference",{"Go":22,"Python":23,"TypeScript":24,"Rust":25,"CSS":26,"Shell":27,"ASL":28,"TeX":29,"Makefile":30,"JavaScript":31,"C++":32,"SCSS":33,"Dockerfile":33,"Svelte":33,"BibTeX Style":33,"HIP":33,"HTML":34,"C":34,"CMake":34,"Go Template":34,"MDX":34,"GLSL":34},48.3,25.3,12.3,6.6,2.6,1.4,1,0.8,0.7,0.3,0.2,0.1,0,[],"https:\u002F\u002Fvllm-sr.ai","2026-09-28","https:\u002F\u002Favatars.githubusercontent.com\u002Fu\u002F136984999?v=4","vLLM","","2026-09-28T15:18:29.427670+00:00","2026-09-27T06:59:53Z",[44,45,46,47,48,49,50,51,15,52,53],"ai-gateway","guardrails","inference","kubernetes","llm","llmrouter","mixture-of-models","pytorch","transformer","vllm","Apache-2.0","ai","LLM推理\u002F微调工具",[58,75,92],{"tag":59,"publishAt":42,"semVer":60,"prerelease":62,"assets":63},"v0.4.0",[34,61,34],4,false,[64,69],{"fileName":65,"sizeMb":66,"sha256":67,"downloadCount":61,"originUrl":68},"vllm_sr-0.4.0-py3-none-any.whl",0.94,"e027138d61918f9bc357b5918bf366248ce7b4d3aa428c20f9f04030c3a3c4ea","https:\u002F\u002Fgithub.com\u002Fvllm-project\u002Fsemantic-router\u002Freleases\u002Fdownload\u002Fv0.4.0\u002Fvllm_sr-0.4.0-py3-none-any.whl",{"fileName":70,"sizeMb":71,"sha256":72,"downloadCount":73,"originUrl":74},"vllm_sr-0.4.0.tar.gz",1.16,"19e472802aee2583bc6a49ef7239bfdd20f34cb72e0b4d9a21f0a003e622c6c0",3,"https:\u002F\u002Fgithub.com\u002Fvllm-project\u002Fsemantic-router\u002Freleases\u002Fdownload\u002Fv0.4.0\u002Fvllm_sr-0.4.0.tar.gz",{"tag":76,"publishAt":77,"semVer":78,"prerelease":62,"assets":79},"v0.3.0","2026-06-05T12:08:52Z",[34,73,34],[80,86],{"fileName":81,"sizeMb":82,"sha256":83,"downloadCount":84,"originUrl":85},"vllm_sr-0.3.0-py3-none-any.whl",0.13,"1031953be496ab8c41313cf58ca50eb76029d4dfd7a651678677251af7125cdf",35,"https:\u002F\u002Fgithub.com\u002Fvllm-project\u002Fsemantic-router\u002Freleases\u002Fdownload\u002Fv0.3.0\u002Fvllm_sr-0.3.0-py3-none-any.whl",{"fileName":87,"sizeMb":88,"sha256":89,"downloadCount":90,"originUrl":91},"vllm_sr-0.3.0.tar.gz",0.16,"6496aae846fcb14e00f95244974c681200a0edb2806a43478b49a2c6a6679448",25,"https:\u002F\u002Fgithub.com\u002Fvllm-project\u002Fsemantic-router\u002Freleases\u002Fdownload\u002Fv0.3.0\u002Fvllm_sr-0.3.0.tar.gz",{"tag":93,"publishAt":94,"semVer":95,"prerelease":62,"assets":97},"v0.2.0","2026-03-10T09:09:47Z",[34,96,34],2,[98,104],{"fileName":99,"sizeMb":100,"sha256":101,"downloadCount":102,"originUrl":103},"vllm_sr-0.2.0-py3-none-any.whl",0.08,"60b7d6e9195bedc155b27ac8cf2debe0de1135ad5644f733b8319e5a6e134406",20,"https:\u002F\u002Fgithub.com\u002Fvllm-project\u002Fsemantic-router\u002Freleases\u002Fdownload\u002Fv0.2.0\u002Fvllm_sr-0.2.0-py3-none-any.whl",{"fileName":105,"sizeMb":100,"sha256":106,"downloadCount":102,"originUrl":107},"vllm_sr-0.2.0.tar.gz","9ed8ffce14dbb52850b725b4b710f3bb748aad672e9b4043e4865b64cd173752","https:\u002F\u002Fgithub.com\u002Fvllm-project\u002Fsemantic-router\u002Freleases\u002Fdownload\u002Fv0.2.0\u002Fvllm_sr-0.2.0.tar.gz",{"description":109,"readmeHtml":110},"一个用于异构LLM推理的可编程混合模型（MoM）路由器，通过评估请求信号、用户偏好和应用策略，为每个请求选择或组合最优的模型路径，以提升质量、降低成本、延迟和增强隐私与安全。","\u003Ch2 id=\"项目概述\">项目概述\u003C\u002Fh2>\u003Cp>vLLM Semantic Router 是一个可编程的路由层，用于在异构 LLM 基础设施上构建混合模型（Mixture-of-Models）系统。它评估请求信号、用户偏好和应用策略，为每个请求选择或组合最优的模型路径，而无需将路由逻辑硬编码到应用中。\u003C\u002Fp>\n\u003Ch2 id=\"主要功能\">主要功能\u003C\u002Fh2>\u003Cul>\n\u003Cli>\u003Cstrong>可编程路由层\u003C\u002Fstrong>：支持混合模型系统的构建，根据请求信号、用户偏好和应用策略动态路由\u003C\u002Fli>\n\u003Cli>\u003Cstrong>异构计算支持\u003C\u002Fstrong>：在 GPU、加速器、边缘和云端等异构计算环境间进行路由\u003C\u002Fli>\n\u003Cli>\u003Cstrong>多维度优化\u003C\u002Fstrong>：提升质量、降低成本、延迟，并增强隐私与安全性\u003C\u002Fli>\n\u003Cli>\u003Cstrong>请求路由\u003C\u002Fstrong>：基于意图分类的智能请求路由\u003C\u002Fli>\n\u003Cli>\u003Cstrong>Envoy Proxy 集成\u003C\u002Fstrong>：支持 Envoy 代理作为路由基础设施的一部分\u003C\u002Fli>\n\u003C\u002Ful>\n\u003Ch2 id=\"适用人群\">适用人群\u003C\u002Fh2>\u003Cul>\n\u003Cli>需要构建混合模型推理系统的 AI 工程师和开发者\u003C\u002Fli>\n\u003Cli>希望在异构 LLM 基础设施上优化成本、延迟和质量的团队\u003C\u002Fli>\n\u003Cli>对多模型路由和智能调度感兴趣的 LLM 应用开发者\u003C\u002Fli>\n\u003C\u002Ful>\n\u003Ch2 id=\"快速上手\">快速上手\u003C\u002Fh2>\u003Col>\n\u003Cli>\u003Cstrong>安装 CLI\u003C\u002Fstrong>：\u003Ccode>curl -fsSL https:\u002F\u002Fvllm-sr.ai\u002Finstall.sh | bash -s -- --channel dev\u003C\u002Fcode>\u003C\u002Fli>\n\u003Cli>详细的平台配置和故障排查请参考\u003Ca href=\"https:\u002F\u002Fvllm-sr.ai\u002Fdocs\u002Finstallation\u002F\" target=\"_blank\" rel=\"noopener noreferrer\">安装指南\u003C\u002Fa>\u003C\u002Fli>\n\u003Cli>\u003Cstrong>在线 playground\u003C\u002Fstrong> 凭证：用户名 \u003Ccode>love@vllm-sr.ai\u003C\u002Fcode>，密码 \u003Ccode>vllm-sr-read\u003C\u002Fcode>\u003C\u002Fli>\n\u003C\u002Fol>\n\u003Ch2 id=\"核心优势\">核心优势\u003C\u002Fh2>\u003Cul>\n\u003Cli>\u003Cstrong>可编程的混合模型路由\u003C\u002Fstrong>：无需硬编码路由逻辑即可实现智能模型选择\u003C\u002Fli>\n\u003Cli>\u003Cstrong>跨异构计算环境\u003C\u002Fstrong>：支持 GPU、加速器、边缘和云端等多种计算资源\u003C\u002Fli>\n\u003Cli>\u003Cstrong>多维度优化\u003C\u002Fstrong>：同时优化质量、成本、延迟和隐私\u003C\u002Fli>\n\u003Cli>\u003Cstrong>信号驱动决策\u003C\u002Fstrong>：基于请求信号、用户偏好和应用策略进行智能路由\u003C\u002Fli>\n\u003C\u002Ful>\n\u003Ch2 id=\"系统要求\">系统要求\u003C\u002Fh2>\u003Cul>\n\u003Cli>操作系统：跨平台（支持 Linux、macOS）\u003C\u002Fli>\n\u003C\u002Ful>\n",1790699424834]