{"avoid_when":"需要多租户高并发 OpenAI 兼容服务集群","content_hash":"b44333c04ffa537c6089f218785ee9f88a21ee329824670bf4ebe2542af58c63","domain":"ai-agents","evidence_urls":["https://github.com/ggml-org/llama.cpp"],"id":"llama-cpp","language":"C++","license":"","name":"llama.cpp","niche":"local ggml inference","repo":"https://github.com/ggml-org/llama.cpp","source_updated_at":null,"status":"active","summary":"C/C++ 本地 LLM 推理，GGUF 生态核心。","tags":["local-llm"],"use_when":"CPU/消费级 GPU 本地跑量化模型、嵌入式部署","verification_status":"unverified","verified_at":null}
