UNPKG

@huggingface/tasks

Version:
717 lines (716 loc) 26.7 kB
"use strict"; Object.defineProperty(exports, "__esModule", { value: true }); exports.LOCAL_APPS = void 0; const gguf_js_1 = require("./gguf.js"); const common_js_1 = require("./snippets/common.js"); const inputs_js_1 = require("./snippets/inputs.js"); function isAwqModel(model) { return model.config?.quantization_config?.quant_method === "awq"; } function isGptqModel(model) { return model.config?.quantization_config?.quant_method === "gptq"; } function isAqlmModel(model) { return model.config?.quantization_config?.quant_method === "aqlm"; } function isMarlinModel(model) { return model.config?.quantization_config?.quant_method === "marlin"; } function isTransformersModel(model) { return model.tags.includes("transformers"); } function isTgiModel(model) { return model.tags.includes("text-generation-inference"); } function isLlamaCppGgufModel(model) { return !!model.gguf?.context_length; } function isVllmModel(model) { return ((isAwqModel(model) || isGptqModel(model) || isAqlmModel(model) || isMarlinModel(model) || isLlamaCppGgufModel(model) || isTransformersModel(model)) && (model.pipeline_tag === "text-generation" || model.pipeline_tag === "image-text-to-text")); } function isDockerModelRunnerModel(model) { return isLlamaCppGgufModel(model) || isVllmModel(model); } function isAmdRyzenModel(model) { return model.tags.includes("ryzenai-hybrid") || model.tags.includes("ryzenai-npu"); } function isMlxModel(model) { return model.tags.includes("mlx"); } /** * Returns the model's chat template string, coalescing across sources: * GGUF metadata > chat_template_jinja file > tokenizer_config.json */ function getChatTemplate(model) { const ct = model.gguf?.chat_template ?? model.config?.chat_template_jinja ?? model.config?.tokenizer_config?.chat_template; if (typeof ct === "string") { return ct; } if (Array.isArray(ct)) { return ct[0]?.template; } return undefined; } function isUnslothModel(model) { return model.tags.includes("unsloth") || isLlamaCppGgufModel(model); } function isToolCallingLocalAgentModel(model) { return ((isLlamaCppGgufModel(model) || isMlxModel(model)) && model.tags.includes("conversational") && !!getChatTemplate(model)?.includes("tools")); } function getQuantTag(filepath) { const defaultTag = ":{{QUANT_TAG}}"; if (!filepath) { return defaultTag; } const quantLabel = (0, gguf_js_1.parseGGUFQuantLabel)(filepath); return quantLabel ? `:${quantLabel}` : defaultTag; } const snippetLlamacpp = (model, filepath) => { const serverCommand = (binary) => { const snippet = [ "# Start a local OpenAI-compatible server with a web UI:", `${binary} -hf ${model.id}${getQuantTag(filepath)}`, ]; return snippet.join("\n"); }; const cliCommand = (binary) => { const snippet = ["# Run inference directly in the terminal:", `${binary} -hf ${model.id}${getQuantTag(filepath)}`]; return snippet.join("\n"); }; return [ { title: "Install (macOS, Linux)", setup: "curl -LsSf https://llama.app/install.sh | sh", content: [serverCommand("llama serve"), cliCommand("llama cli")], }, { title: "Install from WinGet (Windows)", setup: "winget install llama.cpp", content: [serverCommand("llama serve"), cliCommand("llama cli")], }, { title: "Use pre-built binary", setup: [ // prettier-ignore "# Download pre-built binary from:", "# https://github.com/ggerganov/llama.cpp/releases", ].join("\n"), content: [serverCommand("./llama-server"), cliCommand("./llama-cli")], }, { title: "Build from source code", setup: [ "git clone https://github.com/ggerganov/llama.cpp.git", "cd llama.cpp", "cmake -B build", "cmake --build build -j --target llama-server llama-cli", ].join("\n"), content: [serverCommand("./build/bin/llama-server"), cliCommand("./build/bin/llama-cli")], }, { title: "Use Docker", content: snippetDockerModelRunner(model, filepath), }, ]; }; const snippetNodeLlamaCppCli = (model, filepath) => { const tagName = getQuantTag(filepath); return [ { title: "Chat with the model", content: `npx -y node-llama-cpp chat hf:${model.id}${tagName}`, }, { title: "Estimate the model compatibility with your hardware", content: `npx -y node-llama-cpp inspect estimate hf:${model.id}${tagName}`, }, ]; }; const snippetOllama = (model, filepath) => { return `ollama run hf.co/${model.id}${getQuantTag(filepath)}`; }; const snippetUnsloth = (model) => { const isGguf = isLlamaCppGgufModel(model); const studio_content = [ "# Run unsloth studio", "unsloth studio -H 0.0.0.0 -p 8888", "# Then open http://localhost:8888 in your browser", "# Search for " + model.id + " to start chatting", ].join("\n"); const studio_instructions = { title: "Install Unsloth Studio (macOS, Linux, WSL)", setup: "curl -fsSL https://unsloth.ai/install.sh | sh", content: studio_content, }; const studio_instructions_windows = { title: "Install Unsloth Studio (Windows)", setup: "irm https://unsloth.ai/install.ps1 | iex", content: studio_content, }; const hf_spaces_instructions = { title: "Using HuggingFace Spaces for Unsloth", setup: "# No setup required", content: "# Open https://huggingface.co/spaces/unsloth/studio in your browser\n# Search for " + model.id + " to start chatting", }; const fastmodel_instructions = { title: "Load model with FastModel", setup: "pip install unsloth", content: [ "from unsloth import FastModel", "model, tokenizer = FastModel.from_pretrained(", ' model_name="' + model.id + '",', " max_seq_length=2048,", ")", ].join("\n"), }; if (isGguf) { return [studio_instructions, studio_instructions_windows, hf_spaces_instructions]; } else { return [studio_instructions, studio_instructions_windows, hf_spaces_instructions, fastmodel_instructions]; } }; const snippetLocalAI = (model, filepath) => { const command = (binary) => ["# Load and run the model:", `${binary} huggingface://${model.id}/${filepath ?? "{{GGUF_FILE}}"}`].join("\n"); return [ { title: "Install from binary", setup: "curl https://localai.io/install.sh | sh", content: command("local-ai run"), }, { title: "Use Docker images", setup: [ // prettier-ignore "# Pull the image:", "docker pull localai/localai:latest-cpu", ].join("\n"), content: command("docker run -p 8080:8080 --name localai -v $PWD/models:/build/models localai/localai:latest-cpu"), }, ]; }; const snippetVllm = (model) => { const messages = (0, inputs_js_1.getModelInputSnippet)(model); const isMistral = model.tags.includes("mistral-common"); const mistralFlags = isMistral ? " --tokenizer_mode mistral --config_format mistral --load_format mistral --tool-call-parser mistral --enable-auto-tool-choice" : ""; const setup = isMistral ? [ "# Install vLLM from pip:", "pip install vllm", "# Install mistral-common:", "pip install --upgrade mistral-common", ].join("\n") : ["# Install vLLM from pip:", "pip install vllm"].join("\n"); const serverCommand = `# Start the vLLM server: vllm serve "${model.id}"${mistralFlags}`; const runCommandInstruct = `# Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \\ -H "Content-Type: application/json" \\ --data '{ "model": "${model.id}", "messages": ${(0, common_js_1.stringifyMessages)(messages, { indent: "\t\t", attributeKeyQuotes: true, customContentEscaper: (str) => str.replace(/'/g, "'\\''"), })} }'`; const runCommandNonInstruct = `# Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \\ -H "Content-Type: application/json" \\ --data '{ "model": "${model.id}", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'`; const runCommand = model.tags.includes("conversational") ? runCommandInstruct : runCommandNonInstruct; return [ { title: "Install from pip and serve model", setup: setup, content: [serverCommand, runCommand], }, { title: "Use Docker", content: snippetDockerModelRunner(model), }, ]; }; const snippetSglang = (model) => { const messages = (0, inputs_js_1.getModelInputSnippet)(model); const setup = ["# Install SGLang from pip:", "pip install sglang"].join("\n"); const serverCommand = `# Start the SGLang server: python3 -m sglang.launch_server \\ --model-path "${model.id}" \\ --host 0.0.0.0 \\ --port 30000`; const dockerCommand = `docker run --gpus all \\ --shm-size 32g \\ -p 30000:30000 \\ -v ~/.cache/huggingface:/root/.cache/huggingface \\ --env "HF_TOKEN=<secret>" \\ --ipc=host \\ lmsysorg/sglang:latest \\ python3 -m sglang.launch_server \\ --model-path "${model.id}" \\ --host 0.0.0.0 \\ --port 30000`; const runCommandInstruct = `# Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \\ -H "Content-Type: application/json" \\ --data '{ "model": "${model.id}", "messages": ${(0, common_js_1.stringifyMessages)(messages, { indent: "\t\t", attributeKeyQuotes: true, customContentEscaper: (str) => str.replace(/'/g, "'\\''"), })} }'`; const runCommandNonInstruct = `# Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \\ -H "Content-Type: application/json" \\ --data '{ "model": "${model.id}", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'`; const runCommand = model.tags.includes("conversational") ? runCommandInstruct : runCommandNonInstruct; return [ { title: "Install from pip and serve model", setup: setup, content: [serverCommand, runCommand], }, { title: "Use Docker images", setup: dockerCommand, content: [runCommand], }, ]; }; const snippetTgi = (model) => { const runCommand = [ "# Call the server using curl:", `curl -X POST "http://localhost:8000/v1/chat/completions" \\`, ` -H "Content-Type: application/json" \\`, ` --data '{`, ` "model": "${model.id}",`, ` "messages": [`, ` {"role": "user", "content": "What is the capital of France?"}`, ` ]`, ` }'`, ]; return [ { title: "Use Docker images", setup: [ "# Deploy with docker on Linux:", `docker run --gpus all \\`, ` -v ~/.cache/huggingface:/root/.cache/huggingface \\`, ` -e HF_TOKEN="<secret>" \\`, ` -p 8000:80 \\`, ` ghcr.io/huggingface/text-generation-inference:latest \\`, ` --model-id ${model.id}`, ].join("\n"), content: [runCommand.join("\n")], }, ]; }; const snippetMlxLm = (model) => { const openaiCurl = [ "# Calling the OpenAI-compatible server with curl", `curl -X POST "http://localhost:8000/v1/chat/completions" \\`, ` -H "Content-Type: application/json" \\`, ` --data '{`, ` "model": "${model.id}",`, ` "messages": [`, ` {"role": "user", "content": "Hello"}`, ` ]`, ` }'`, ]; return [ { title: "Generate or start a chat session", setup: ["# Install MLX LM", "uv tool install mlx-lm"].join("\n"), content: [ ...(model.tags.includes("conversational") ? ["# Interactive chat REPL", `mlx_lm.chat --model "${model.id}"`] : ["# Generate some text", `mlx_lm.generate --model "${model.id}" --prompt "Once upon a time"`]), ].join("\n"), }, ...(model.tags.includes("conversational") ? [ { title: "Run an OpenAI-compatible server", setup: ["# Install MLX LM", "uv tool install mlx-lm"].join("\n"), content: ["# Start the server", `mlx_lm.server --model "${model.id}"`, ...openaiCurl].join("\n"), }, ] : []), ]; }; const getLocalServerStep = (model, filepath) => { return isMlxModel(model) ? { title: "Start the MLX server", setup: "# Install MLX LM:\nuv tool install mlx-lm", content: `# Start a local OpenAI-compatible server:\nmlx_lm.server --model "${model.id}"`, } : { title: "Start the llama.cpp server", setup: "# Install llama.cpp:\nbrew install llama.cpp", content: `# Start a local OpenAI-compatible server:\nllama serve -hf ${model.id}${getQuantTag(filepath)}`, }; }; const snippetPi = (model, filepath) => { const isMLX = isMlxModel(model); const modelId = isMLX ? model.id : `${model.id}${getQuantTag(filepath)}`; const serverStep = getLocalServerStep(model, filepath); const modelsJson = JSON.stringify({ providers: { [isMLX ? "mlx-lm" : "llama-cpp"]: { baseUrl: "http://localhost:8080/v1", api: "openai-completions", apiKey: "none", models: [{ id: modelId }], }, }, }, null, 2); return [ serverStep, { title: "Configure the model in Pi", setup: "# Install Pi:\nnpm install -g @mariozechner/pi-coding-agent", content: `# Add to ~/.pi/agent/models.json:\n${modelsJson}`, }, { title: "Run Pi", content: "# Start Pi in your project directory:\npi", }, ]; }; const snippetHermesAgent = (model, filepath) => { const modelId = isMlxModel(model) ? model.id : `${model.id}${getQuantTag(filepath)}`; const serverStep = getLocalServerStep(model, filepath); return [ serverStep, { title: "Configure Hermes", setup: [ "# Install Hermes:", "curl -fsSL https://hermes-agent.nousresearch.com/install.sh | bash", "hermes setup", ].join("\n"), content: [ "# Point Hermes at the local server:", "hermes config set model.provider custom", "hermes config set model.base_url http://127.0.0.1:8080/v1", `hermes config set model.default ${modelId}`, ].join("\n"), }, { title: "Run Hermes", content: "hermes", }, ]; }; const snippetOpenClaw = (model, filepath) => { const isMLX = isMlxModel(model); const providerId = isMLX ? "mlx-lm" : "llama-cpp"; const modelId = isMLX ? model.id : `${model.id}${getQuantTag(filepath)}`; const serverStep = getLocalServerStep(model, filepath); return [ serverStep, { title: "Configure OpenClaw", setup: "# Install OpenClaw:\nnpm install -g openclaw@latest", content: [ "# Register the local server and set it as the default model:", "openclaw onboard --non-interactive --mode local \\", " --auth-choice custom-api-key \\", " --custom-base-url http://127.0.0.1:8080/v1 \\", ` --custom-model-id "${modelId}" \\`, ` --custom-provider-id ${providerId} \\`, " --custom-compatibility openai \\", " --custom-text-input \\", " --accept-risk \\", " --skip-health", ].join("\n"), }, { title: "Run OpenClaw", content: `openclaw agent --local --agent main --message "Hello from Hugging Face"`, }, ]; }; const snippetDockerModelRunner = (model, filepath) => { // Only add quant tag for GGUF models, not safetensors const quantTag = isLlamaCppGgufModel(model) ? getQuantTag(filepath) : ""; return `docker model run hf.co/${model.id}${quantTag}`; }; const snippetLemonade = (model, filepath) => { const modelName = model.id.includes("/") ? model.id.split("/")[1] : model.id; const isRyzenAI = model.tags.some((tag) => ["ryzenai-npu", "ryzenai-hybrid"].includes(tag)); // Lemonade auto-registers pulled models as `user.<suggested_name>[-<variant>]`. // For GGUF/llamacpp: suggested_name is the repo name and variant is the quant tag. // For RyzenAI ONNX: there is no per-variant suffix. let pullArg; let runName; let requirements; if (isRyzenAI) { pullArg = model.id; runName = `user.${modelName}`; requirements = " (requires XDNA 2 NPU)"; } else { const tagName = getQuantTag(filepath); pullArg = `${model.id}${tagName}`; runName = `user.${modelName}${tagName.replace(":", "-")}`; requirements = ""; } return [ { title: "Pull the model", setup: "# Download Lemonade from https://lemonade-server.ai/", content: `lemonade pull ${pullArg}`, }, { title: `Run and chat with the model${requirements}`, content: `lemonade run ${runName}`, }, { title: "List all available models", content: "lemonade list", }, ]; }; /** * Add your new local app here. * * This is open to new suggestions and awesome upcoming apps. * * /!\ IMPORTANT * * If possible, you need to support deeplinks and be as cross-platform as possible. * * Ping the HF team if we can help with anything! */ exports.LOCAL_APPS = { "llama.cpp": { prettyLabel: "llama.cpp", docsUrl: "https://github.com/ggerganov/llama.cpp", mainTask: "text-generation", displayOnModelPage: isLlamaCppGgufModel, snippet: snippetLlamacpp, }, "node-llama-cpp": { prettyLabel: "node-llama-cpp", docsUrl: "https://node-llama-cpp.withcat.ai", mainTask: "text-generation", displayOnModelPage: isLlamaCppGgufModel, snippet: snippetNodeLlamaCppCli, }, vllm: { prettyLabel: "vLLM", docsUrl: "https://docs.vllm.ai", mainTask: "text-generation", displayOnModelPage: isVllmModel, snippet: snippetVllm, }, sglang: { prettyLabel: "SGLang", docsUrl: "https://docs.sglang.io", mainTask: "text-generation", displayOnModelPage: (model) => (isAwqModel(model) || isGptqModel(model) || isAqlmModel(model) || isMarlinModel(model) || isTransformersModel(model)) && (model.pipeline_tag === "text-generation" || model.pipeline_tag === "image-text-to-text"), snippet: snippetSglang, }, "mlx-lm": { prettyLabel: "MLX LM", docsUrl: "https://github.com/ml-explore/mlx-lm", mainTask: "text-generation", displayOnModelPage: (model) => model.pipeline_tag === "text-generation" && isMlxModel(model), snippet: snippetMlxLm, }, tgi: { prettyLabel: "TGI", docsUrl: "https://huggingface.co/docs/text-generation-inference/", mainTask: "text-generation", displayOnModelPage: isTgiModel, snippet: snippetTgi, }, lmstudio: { prettyLabel: "LM Studio", docsUrl: "https://lmstudio.ai", mainTask: "text-generation", displayOnModelPage: (model) => isLlamaCppGgufModel(model) || isMlxModel(model), deeplink: (model, filepath) => new URL(`lmstudio://open_from_hf?model=${model.id}${filepath ? `&file=${filepath}` : ""}`), }, localai: { prettyLabel: "LocalAI", docsUrl: "https://github.com/mudler/LocalAI", mainTask: "text-generation", displayOnModelPage: isLlamaCppGgufModel, snippet: snippetLocalAI, }, jan: { prettyLabel: "Jan", docsUrl: "https://jan.ai", mainTask: "text-generation", displayOnModelPage: isLlamaCppGgufModel, deeplink: (model) => new URL(`jan://models/huggingface/${model.id}`), }, "atomic-chat": { prettyLabel: "Atomic Chat", docsUrl: "https://atomic.chat", mainTask: "text-generation", displayOnModelPage: isLlamaCppGgufModel, deeplink: (model) => new URL(`atomic-chat://models/huggingface/${model.id}`), }, backyard: { prettyLabel: "Backyard AI", docsUrl: "https://backyard.ai", mainTask: "text-generation", displayOnModelPage: isLlamaCppGgufModel, deeplink: (model) => new URL(`https://backyard.ai/hf/model/${model.id}`), }, sanctum: { prettyLabel: "Sanctum", docsUrl: "https://sanctum.ai", mainTask: "text-generation", displayOnModelPage: isLlamaCppGgufModel, deeplink: (model) => new URL(`sanctum://open_from_hf?model=${model.id}`), }, jellybox: { prettyLabel: "Jellybox", docsUrl: "https://jellybox.com", mainTask: "text-generation", displayOnModelPage: (model) => isLlamaCppGgufModel(model) || (model.library_name === "diffusers" && model.tags.includes("safetensors") && (model.pipeline_tag === "text-to-image" || model.tags.includes("lora"))), deeplink: (model) => { if (isLlamaCppGgufModel(model)) { return new URL(`jellybox://llm/models/huggingface/LLM/${model.id}`); } else if (model.tags.includes("lora")) { return new URL(`jellybox://image/models/huggingface/ImageLora/${model.id}`); } else { return new URL(`jellybox://image/models/huggingface/Image/${model.id}`); } }, }, msty: { prettyLabel: "Msty", docsUrl: "https://msty.app", mainTask: "text-generation", displayOnModelPage: isLlamaCppGgufModel, deeplink: (model) => new URL(`msty://models/search/hf/${model.id}`), }, recursechat: { prettyLabel: "RecurseChat", docsUrl: "https://recurse.chat", mainTask: "text-generation", macOSOnly: true, displayOnModelPage: isLlamaCppGgufModel, deeplink: (model) => new URL(`recursechat://new-hf-gguf-model?hf-model-id=${model.id}`), }, drawthings: { prettyLabel: "Draw Things", docsUrl: "https://drawthings.ai", mainTask: "text-to-image", macOSOnly: true, displayOnModelPage: (model) => model.library_name === "diffusers" && (model.pipeline_tag === "text-to-image" || model.tags.includes("lora")), deeplink: (model) => { if (model.tags.includes("lora")) { return new URL(`https://drawthings.ai/import/diffusers/pipeline.load_lora_weights?repo_id=${model.id}`); } else { return new URL(`https://drawthings.ai/import/diffusers/pipeline.from_pretrained?repo_id=${model.id}`); } }, }, diffusionbee: { prettyLabel: "DiffusionBee", docsUrl: "https://diffusionbee.com", mainTask: "text-to-image", macOSOnly: true, displayOnModelPage: (model) => model.library_name === "diffusers" && model.pipeline_tag === "text-to-image", deeplink: (model) => new URL(`https://diffusionbee.com/huggingface_import?model_id=${model.id}`), }, joyfusion: { prettyLabel: "JoyFusion", docsUrl: "https://joyfusion.app", mainTask: "text-to-image", macOSOnly: true, displayOnModelPage: (model) => model.tags.includes("coreml") && model.tags.includes("joyfusion") && model.pipeline_tag === "text-to-image", deeplink: (model) => new URL(`https://joyfusion.app/import_from_hf?repo_id=${model.id}`), }, ollama: { prettyLabel: "Ollama", docsUrl: "https://ollama.com", mainTask: "text-generation", displayOnModelPage: isLlamaCppGgufModel, snippet: snippetOllama, }, unsloth: { prettyLabel: "Unsloth Studio", docsUrl: "https://unsloth.ai/docs/new/studio", mainTask: "text-generation", displayOnModelPage: isUnslothModel, snippet: snippetUnsloth, }, "docker-model-runner": { prettyLabel: "Docker Model Runner", docsUrl: "https://docs.docker.com/ai/model-runner/", mainTask: "text-generation", displayOnModelPage: isDockerModelRunnerModel, snippet: snippetDockerModelRunner, }, lemonade: { prettyLabel: "Lemonade", docsUrl: "https://lemonade-server.ai", mainTask: "text-generation", displayOnModelPage: (model) => isLlamaCppGgufModel(model) || isAmdRyzenModel(model), snippet: snippetLemonade, }, pi: { prettyLabel: "Pi", docsUrl: "https://github.com/badlogic/pi-mono", mainTask: "text-generation", displayOnModelPage: isToolCallingLocalAgentModel, snippet: snippetPi, }, "hermes-agent": { prettyLabel: "Hermes Agent", docsUrl: "https://hermes-agent.nousresearch.com/", mainTask: "text-generation", displayOnModelPage: isToolCallingLocalAgentModel, snippet: snippetHermesAgent, }, openclaw: { prettyLabel: "OpenClaw", docsUrl: "https://github.com/openclaw/openclaw", mainTask: "text-generation", displayOnModelPage: isToolCallingLocalAgentModel, snippet: snippetOpenClaw, }, };