{ "id": "local-inference", "name": "Local Inference", "description": "Unified Eliza-1 local inference provider — text generation, embeddings, voice (TTS/ASR), and image description via the elizaOS llama.cpp fork. Replaces the deprecated plugin-local-ai / plugin-local-embedding pair.", "npmName": "@elizaos/plugin-local-inference", "version": "2.0.0-beta.2", "source": "bundled", "tags": [ "ai-provider", "llm", "local-models", "self-hosted", "local", "embedding", "voice", "tts", "transcription", "eliza-1" ], "config": { "MODELS_DIR": { "type": "file-path", "required": false, "sensitive": false, "label": "Models Dir", "help": "Filesystem path to the directory where GGUF models are stored. Defaults to /models.", "advanced": false }, "CACHE_DIR": { "type": "file-path", "required": false, "sensitive": false, "label": "Cache Dir", "help": "Filesystem path to the cache directory for model assets (tokenizers, ONNX caches).", "advanced": false }, "LOCAL_SMALL_MODEL": { "type": "string", "required": false, "sensitive": false, "default": "text/eliza-1-2b-128k.gguf", "label": "Small Model", "help": "Filename of the small local model (TEXT_SMALL handler).", "advanced": false }, "LOCAL_LARGE_MODEL": { "type": "string", "required": false, "sensitive": false, "default": "text/eliza-1-4b-128k.gguf", "label": "Large Model", "help": "Filename of the large local model (TEXT_LARGE handler).", "advanced": false }, "LOCAL_EMBEDDING_MODEL": { "type": "string", "required": false, "sensitive": false, "default": "text/eliza-1-2b-128k.gguf", "label": "Embedding Model", "help": "Filename of the embedding model used for vector embeddings.", "advanced": false }, "LOCAL_EMBEDDING_DIMENSIONS": { "type": "number", "required": false, "sensitive": false, "default": "1024", "label": "Embedding Dimensions", "help": "Number of dimensions the embedding model outputs.", "advanced": false }, "CUDA_VISIBLE_DEVICES": { "type": "string", "required": false, "sensitive": false, "label": "CUDA Visible Devices", "help": "Set to restrict which CUDA-enabled GPUs the runtime sees (e.g. `0,1`).", "advanced": true }, "ELIZA_LOCAL_LLAMA": { "type": "string", "required": false, "sensitive": false, "label": "AOSP FFI Loader", "help": "Set to `1` on Android (elizaOS build) to enable the in-process bun:ffi loader (libllama.so + libeliza-llama-shim.so) via the @elizaos/plugin-aosp-local-inference companion plugin.", "advanced": true }, "ELIZA_MODEL": { "type": "select", "required": false, "sensitive": false, "label": "Eliza-1 Tier", "help": "Select a published Eliza-1 fine-tune tier. The runtime auto-picks the right quant flavor (gguf/polarquant/fp8/bf16) for your detected GPU and pulls it from HuggingFace on first use.", "options": [ { "value": "eliza-1-2b", "label": "Eliza-1 2B (mobile / entry)", "description": "Smallest/entry tier; fits Android/iOS on-device inference and 16 GB consumer GPUs at Q4_K_M." }, { "value": "eliza-1-4b", "label": "Eliza-1 4B (16-24 GB GPU)", "description": "Desktop-tier fine-tune for everyday workstation use." }, { "value": "eliza-1-9b", "label": "Eliza-1 9B (24-48 GB GPU)", "description": "Workstation-tier fine-tune; Q4_K_M for 16 GB cards, PolarQuant for 24+." }, { "value": "eliza-1-27b", "label": "Eliza-1 27B (cloud / 48 GB+)", "description": "Cloud-tier fine-tune; Q6_K GGUF on RTX 5090, fp8/bf16 on H200/B200." } ], "advanced": false } }, "render": { "visible": true, "pinTo": [], "style": "card", "icon": "Monitor", "group": "ai-provider", "groupOrder": 0, "actions": ["enable", "configure"] }, "resources": { "homepage": "https://github.com/elizaOS/eliza/tree/main/plugins/plugin-local-inference#readme", "repository": "https://github.com/elizaOS/eliza", "setupGuideUrl": "https://docs.eliza.ai/plugin-setup-guide#local-inference" }, "dependsOn": [], "kind": "plugin", "subtype": "ai-provider", "launch": { "type": "server-launch", "capabilities": [], "bootHook": { "specifier": "@elizaos/plugin-local-inference/runtime", "exportName": "registerLocalInferenceBoot" } } }