diff --git a/.agents/skills/nemoclaw-user-configure-inference/evals/evals.json b/.agents/skills/nemoclaw-user-configure-inference/evals/evals.json index f041d18a7bd..a0bd47ac294 100644 --- a/.agents/skills/nemoclaw-user-configure-inference/evals/evals.json +++ b/.agents/skills/nemoclaw-user-configure-inference/evals/evals.json @@ -3,210 +3,90 @@ "id": "docs-inference-inference-options-001", "question": "I'm choosing an inference option during onboarding. Help me compare hosted providers, local servers, and compatible endpoints so I can select a model path that fits my privacy, cost, and reliability needs.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user compare hosted providers, local servers, and compatible endpoints and gives enough concrete guidance, decision criteria, verification steps, or risk framing to select a model path that fits my privacy, cost, and reliability needs.", - "expected_behavior": [ - "The output directly addresses the user's situation: choosing an inference option during onboarding.", - "The AI coding assistant loads the expected_skill and references/inference-options.md", - "The output helps the user compare hosted providers, local servers, and compatible endpoints with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to select a model path that fits my privacy, cost, and reliability needs.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/inference-options.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user compare hosted providers, local servers, and compatible endpoints and gives enough concrete guidance, decision criteria, verification steps, or risk framing to select a model path that fits my privacy, cost, and reliability needs." }, { "id": "docs-inference-inference-options-002", "question": "I'm preparing provider credentials. Help me know which provider capabilities and secrets onboarding requires so I can complete setup without avoidable credential errors.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user know which provider capabilities and secrets onboarding requires and gives enough concrete guidance, decision criteria, verification steps, or risk framing to complete setup without avoidable credential errors.", - "expected_behavior": [ - "The output directly addresses the user's situation: preparing provider credentials.", - "The AI coding assistant loads the expected_skill and references/inference-options.md", - "The output helps the user know which provider capabilities and secrets onboarding requires with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to complete setup without avoidable credential errors.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/inference-options.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user know which provider capabilities and secrets onboarding requires and gives enough concrete guidance, decision criteria, verification steps, or risk framing to complete setup without avoidable credential errors." }, { "id": "docs-inference-inference-options-003", "question": "I'm evaluating routed inference. Help me understand how the sandbox calls models through the gateway so I can trust that model credentials stay outside the sandbox.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user understand how the sandbox calls models through the gateway and gives enough concrete guidance, decision criteria, verification steps, or risk framing to trust that model credentials stay outside the sandbox.", - "expected_behavior": [ - "The output directly addresses the user's situation: evaluating routed inference.", - "The AI coding assistant loads the expected_skill and references/inference-options.md", - "The output helps the user understand how the sandbox calls models through the gateway with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to trust that model credentials stay outside the sandbox.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/inference-options.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user understand how the sandbox calls models through the gateway and gives enough concrete guidance, decision criteria, verification steps, or risk framing to trust that model credentials stay outside the sandbox." }, { "id": "docs-inference-use-local-inference-001", "question": "I'm connecting a local inference server. Help me route NemoClaw model traffic to Ollama, vLLM, TensorRT-LLM, NIM, or another compatible endpoint so I can meet privacy, latency, or cost goals.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user route NemoClaw model traffic to Ollama, vLLM, TensorRT-LLM, NIM, or another compatible endpoint and gives enough concrete guidance, decision criteria, verification steps, or risk framing to meet privacy, latency, or cost goals.", - "expected_behavior": [ - "The output directly addresses the user's situation: connecting a local inference server.", - "The AI coding assistant loads the expected_skill and SKILL.md", - "The output helps the user route NemoClaw model traffic to Ollama, vLLM, TensorRT-LLM, NIM, or another compatible endpoint with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to meet privacy, latency, or cost goals.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the SKILL.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user route NemoClaw model traffic to Ollama, vLLM, TensorRT-LLM, NIM, or another compatible endpoint and gives enough concrete guidance, decision criteria, verification steps, or risk framing to meet privacy, latency, or cost goals." }, { "id": "docs-inference-use-local-inference-002", "question": "I'm debugging local endpoint reachability. Help me separate NemoClaw routing issues from model-server issues so I can fix the right component first.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user separate NemoClaw routing issues from model-server issues and gives enough concrete guidance, decision criteria, verification steps, or risk framing to fix the right component first.", - "expected_behavior": [ - "The output directly addresses the user's situation: debugging local endpoint reachability.", - "The AI coding assistant loads the expected_skill and SKILL.md", - "The output helps the user separate NemoClaw routing issues from model-server issues with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to fix the right component first.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the SKILL.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user separate NemoClaw routing issues from model-server issues and gives enough concrete guidance, decision criteria, verification steps, or risk framing to fix the right component first." }, { "id": "docs-inference-use-local-inference-003", "question": "I'm configuring traffic through `inference.local`. Help me understand the required host, port, and model settings so I can make sandboxed inference calls resolve to my local server.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user understand the required host, port, and model settings and gives enough concrete guidance, decision criteria, verification steps, or risk framing to make sandboxed inference calls resolve to my local server.", - "expected_behavior": [ - "The output directly addresses the user's situation: configuring traffic through `inference.local`.", - "The AI coding assistant loads the expected_skill and SKILL.md", - "The output helps the user understand the required host, port, and model settings with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to make sandboxed inference calls resolve to my local server.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the SKILL.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user understand the required host, port, and model settings and gives enough concrete guidance, decision criteria, verification steps, or risk framing to make sandboxed inference calls resolve to my local server." }, { "id": "docs-inference-switch-inference-providers-001", "question": "I'm switching inference models during a running session. Help me change model behavior without restarting the sandbox so I can adapt to task, cost, or reliability needs quickly.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user change model behavior without restarting the sandbox and gives enough concrete guidance, decision criteria, verification steps, or risk framing to adapt to task, cost, or reliability needs quickly.", - "expected_behavior": [ - "The output directly addresses the user's situation: switching inference models during a running session.", - "The AI coding assistant loads the expected_skill and references/switch-inference-providers.md", - "The output helps the user change model behavior without restarting the sandbox with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to adapt to task, cost, or reliability needs quickly.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/switch-inference-providers.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user change model behavior without restarting the sandbox and gives enough concrete guidance, decision criteria, verification steps, or risk framing to adapt to task, cost, or reliability needs quickly." }, { "id": "docs-inference-switch-inference-providers-002", "question": "I'm confirming a runtime model change. Help me verify the agent is using the new active model so I can avoid mistaking host configuration changes for live routing changes.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user verify the agent is using the new active model and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid mistaking host configuration changes for live routing changes.", - "expected_behavior": [ - "The output directly addresses the user's situation: confirming a runtime model change.", - "The AI coding assistant loads the expected_skill and references/switch-inference-providers.md", - "The output helps the user verify the agent is using the new active model with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to avoid mistaking host configuration changes for live routing changes.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/switch-inference-providers.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user verify the agent is using the new active model and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid mistaking host configuration changes for live routing changes." }, { "id": "docs-inference-switch-inference-providers-003", "question": "I'm trying a different model during active work. Help me know how to roll back to the previous model so I can experiment without disrupting the assistant workflow.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user know how to roll back to the previous model and gives enough concrete guidance, decision criteria, verification steps, or risk framing to experiment without disrupting the assistant workflow.", - "expected_behavior": [ - "The output directly addresses the user's situation: trying a different model during active work.", - "The AI coding assistant loads the expected_skill and references/switch-inference-providers.md", - "The output helps the user know how to roll back to the previous model with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to experiment without disrupting the assistant workflow.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/switch-inference-providers.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user know how to roll back to the previous model and gives enough concrete guidance, decision criteria, verification steps, or risk framing to experiment without disrupting the assistant workflow." }, { "id": "docs-inference-set-up-sub-agent-001", "question": "I'm configuring a task-specific sub-agent. Help me assign a specialized model to work the default agent should not handle so I can improve task fit without changing the whole assistant.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user assign a specialized model to work the default agent should not handle and gives enough concrete guidance, decision criteria, verification steps, or risk framing to improve task fit without changing the whole assistant.", - "expected_behavior": [ - "The output directly addresses the user's situation: configuring a task-specific sub-agent.", - "The AI coding assistant loads the expected_skill and references/set-up-sub-agent.md", - "The output helps the user assign a specialized model to work the default agent should not handle with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to improve task fit without changing the whole assistant.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/set-up-sub-agent.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user assign a specialized model to work the default agent should not handle and gives enough concrete guidance, decision criteria, verification steps, or risk framing to improve task fit without changing the whole assistant." }, { "id": "docs-inference-set-up-sub-agent-002", "question": "I'm editing sub-agent model configuration. Help me understand where files, credentials, and workspace settings live so I can avoid leaking secrets or changing the wrong agent.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user understand where files, credentials, and workspace settings live and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid leaking secrets or changing the wrong agent.", - "expected_behavior": [ - "The output directly addresses the user's situation: editing sub-agent model configuration.", - "The AI coding assistant loads the expected_skill and references/set-up-sub-agent.md", - "The output helps the user understand where files, credentials, and workspace settings live with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to avoid leaking secrets or changing the wrong agent.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/set-up-sub-agent.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user understand where files, credentials, and workspace settings live and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid leaking secrets or changing the wrong agent." }, { "id": "docs-inference-set-up-sub-agent-003", "question": "I'm testing a new sub-agent. Help me send a prompt that exercises the intended routing so I can prove it uses the expected provider and model.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user send a prompt that exercises the intended routing and gives enough concrete guidance, decision criteria, verification steps, or risk framing to prove it uses the expected provider and model.", - "expected_behavior": [ - "The output directly addresses the user's situation: testing a new sub-agent.", - "The AI coding assistant loads the expected_skill and references/set-up-sub-agent.md", - "The output helps the user send a prompt that exercises the intended routing with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to prove it uses the expected provider and model.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/set-up-sub-agent.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user send a prompt that exercises the intended routing and gives enough concrete guidance, decision criteria, verification steps, or risk framing to prove it uses the expected provider and model." }, { "id": "docs-inference-tool-calling-reliability-001", "question": "I'm seeing tool calls leak as plain text. Help me diagnose whether the model, server, or parser is incompatible so I can restore reliable tool execution.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user diagnose whether the model, server, or parser is incompatible and gives enough concrete guidance, decision criteria, verification steps, or risk framing to restore reliable tool execution.", - "expected_behavior": [ - "The output directly addresses the user's situation: seeing tool calls leak as plain text.", - "The AI coding assistant loads the expected_skill and references/tool-calling-reliability.md", - "The output helps the user diagnose whether the model, server, or parser is incompatible with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to restore reliable tool execution.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/tool-calling-reliability.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user diagnose whether the model, server, or parser is incompatible and gives enough concrete guidance, decision criteria, verification steps, or risk framing to restore reliable tool execution." }, { "id": "docs-inference-tool-calling-reliability-002", "question": "I'm comparing local inference runtimes. Help me understand whether Ollama, vLLM, or parser settings better support tool calls so I can choose a runtime that matches the agent's tool needs.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user understand whether Ollama, vLLM, or parser settings better support tool calls and gives enough concrete guidance, decision criteria, verification steps, or risk framing to choose a runtime that matches the agent's tool needs.", - "expected_behavior": [ - "The output directly addresses the user's situation: comparing local inference runtimes.", - "The AI coding assistant loads the expected_skill and references/tool-calling-reliability.md", - "The output helps the user understand whether Ollama, vLLM, or parser settings better support tool calls with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to choose a runtime that matches the agent's tool needs.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/tool-calling-reliability.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user understand whether Ollama, vLLM, or parser settings better support tool calls and gives enough concrete guidance, decision criteria, verification steps, or risk framing to choose a runtime that matches the agent's tool needs." }, { "id": "docs-inference-tool-calling-reliability-003", "question": "I'm letting an always-on assistant use tools unattended. Help me define the reliability bar for local tool calling so I can avoid silent failures or unsafe plain-text tool outputs.", "expected_skill": "nemoclaw-user-configure-inference", - "ground_truth": "A NemoClaw-specific answer that helps the user define the reliability bar for local tool calling and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid silent failures or unsafe plain-text tool outputs.", - "expected_behavior": [ - "The output directly addresses the user's situation: letting an always-on assistant use tools unattended.", - "The AI coding assistant loads the expected_skill and references/tool-calling-reliability.md", - "The output helps the user define the reliability bar for local tool calling with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to avoid silent failures or unsafe plain-text tool outputs.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/tool-calling-reliability.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user define the reliability bar for local tool calling and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid silent failures or unsafe plain-text tool outputs." } ] diff --git a/.agents/skills/nemoclaw-user-configure-security/evals/evals.json b/.agents/skills/nemoclaw-user-configure-security/evals/evals.json index 06855e83a26..9e17d649833 100644 --- a/.agents/skills/nemoclaw-user-configure-security/evals/evals.json +++ b/.agents/skills/nemoclaw-user-configure-security/evals/evals.json @@ -3,126 +3,54 @@ "id": "docs-security-best-practices-001", "question": "I'm evaluating NemoClaw security best practices. Help me understand the risk posture of each configurable control so I can justify the setup to my team or security reviewers.", "expected_skill": "nemoclaw-user-configure-security", - "ground_truth": "A NemoClaw-specific answer that helps the user understand the risk posture of each configurable control and gives enough concrete guidance, decision criteria, verification steps, or risk framing to justify the setup to my team or security reviewers.", - "expected_behavior": [ - "The output directly addresses the user's situation: evaluating NemoClaw security best practices.", - "The AI coding assistant loads the expected_skill and references/best-practices.md", - "The output helps the user understand the risk posture of each configurable control with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to justify the setup to my team or security reviewers.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/best-practices.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user understand the risk posture of each configurable control and gives enough concrete guidance, decision criteria, verification steps, or risk framing to justify the setup to my team or security reviewers." }, { "id": "docs-security-best-practices-002", "question": "I'm balancing developer convenience with lockdown. Help me compare the trade-offs of changing security controls so I can choose a posture that fits the environment.", "expected_skill": "nemoclaw-user-configure-security", - "ground_truth": "A NemoClaw-specific answer that helps the user compare the trade-offs of changing security controls and gives enough concrete guidance, decision criteria, verification steps, or risk framing to choose a posture that fits the environment.", - "expected_behavior": [ - "The output directly addresses the user's situation: balancing developer convenience with lockdown.", - "The AI coding assistant loads the expected_skill and references/best-practices.md", - "The output helps the user compare the trade-offs of changing security controls with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to choose a posture that fits the environment.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/best-practices.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user compare the trade-offs of changing security controls and gives enough concrete guidance, decision criteria, verification steps, or risk framing to choose a posture that fits the environment." }, { "id": "docs-security-best-practices-003", "question": "I'm preparing for production-like use. Help me see which defaults are acceptable and which require changes so I can avoid shipping with accidental weak spots.", "expected_skill": "nemoclaw-user-configure-security", - "ground_truth": "A NemoClaw-specific answer that helps the user see which defaults are acceptable and which require changes and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid shipping with accidental weak spots.", - "expected_behavior": [ - "The output directly addresses the user's situation: preparing for production-like use.", - "The AI coding assistant loads the expected_skill and references/best-practices.md", - "The output helps the user see which defaults are acceptable and which require changes with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to avoid shipping with accidental weak spots.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/best-practices.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user see which defaults are acceptable and which require changes and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid shipping with accidental weak spots." }, { "id": "docs-security-credential-storage-001", "question": "I'm inspecting NemoClaw credential storage. Help me verify how secrets are stored and protected so I can decide whether the setup meets my secret-handling expectations.", "expected_skill": "nemoclaw-user-configure-security", - "ground_truth": "A NemoClaw-specific answer that helps the user verify how secrets are stored and protected and gives enough concrete guidance, decision criteria, verification steps, or risk framing to decide whether the setup meets my secret-handling expectations.", - "expected_behavior": [ - "The output directly addresses the user's situation: inspecting NemoClaw credential storage.", - "The AI coding assistant loads the expected_skill and references/credential-storage.md", - "The output helps the user verify how secrets are stored and protected with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to decide whether the setup meets my secret-handling expectations.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/credential-storage.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user verify how secrets are stored and protected and gives enough concrete guidance, decision criteria, verification steps, or risk framing to decide whether the setup meets my secret-handling expectations." }, { "id": "docs-security-credential-storage-002", "question": "I'm tracing where credentials live. Help me distinguish host, gateway, and sandbox storage boundaries so I can avoid assuming secrets are available in the wrong place.", "expected_skill": "nemoclaw-user-configure-security", - "ground_truth": "A NemoClaw-specific answer that helps the user distinguish host, gateway, and sandbox storage boundaries and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid assuming secrets are available in the wrong place.", - "expected_behavior": [ - "The output directly addresses the user's situation: tracing where credentials live.", - "The AI coding assistant loads the expected_skill and references/credential-storage.md", - "The output helps the user distinguish host, gateway, and sandbox storage boundaries with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to avoid assuming secrets are available in the wrong place.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/credential-storage.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user distinguish host, gateway, and sandbox storage boundaries and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid assuming secrets are available in the wrong place." }, { "id": "docs-security-credential-storage-003", "question": "I'm rotating or inspecting credentials. Help me follow a workflow that does not print secrets in logs or docs so I can recover or update access safely.", "expected_skill": "nemoclaw-user-configure-security", - "ground_truth": "A NemoClaw-specific answer that helps the user follow a workflow that does not print secrets in logs or docs and gives enough concrete guidance, decision criteria, verification steps, or risk framing to recover or update access safely.", - "expected_behavior": [ - "The output directly addresses the user's situation: rotating or inspecting credentials.", - "The AI coding assistant loads the expected_skill and references/credential-storage.md", - "The output helps the user follow a workflow that does not print secrets in logs or docs with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to recover or update access safely.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/credential-storage.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user follow a workflow that does not print secrets in logs or docs and gives enough concrete guidance, decision criteria, verification steps, or risk framing to recover or update access safely." }, { "id": "docs-security-openclaw-controls-001", "question": "I'm reading about controls outside NemoClaw's scope. Help me understand which security responsibilities remain with OpenClaw so I can avoid treating sandbox isolation as a complete application security model.", "expected_skill": "nemoclaw-user-configure-security", - "ground_truth": "A NemoClaw-specific answer that helps the user understand which security responsibilities remain with OpenClaw and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid treating sandbox isolation as a complete application security model.", - "expected_behavior": [ - "The output directly addresses the user's situation: reading about controls outside NemoClaw's scope.", - "The AI coding assistant loads the expected_skill and references/openclaw-controls.md", - "The output helps the user understand which security responsibilities remain with OpenClaw with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to avoid treating sandbox isolation as a complete application security model.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/openclaw-controls.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user understand which security responsibilities remain with OpenClaw and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid treating sandbox isolation as a complete application security model." }, { "id": "docs-security-openclaw-controls-002", "question": "I'm assessing application-layer agent risk. Help me identify the controls NemoClaw does not add so I can plan separate mitigations for authentication, prompt handling, and agent behavior.", "expected_skill": "nemoclaw-user-configure-security", - "ground_truth": "A NemoClaw-specific answer that helps the user identify the controls NemoClaw does not add and gives enough concrete guidance, decision criteria, verification steps, or risk framing to plan separate mitigations for authentication, prompt handling, and agent behavior.", - "expected_behavior": [ - "The output directly addresses the user's situation: assessing application-layer agent risk.", - "The AI coding assistant loads the expected_skill and references/openclaw-controls.md", - "The output helps the user identify the controls NemoClaw does not add with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to plan separate mitigations for authentication, prompt handling, and agent behavior.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/openclaw-controls.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user identify the controls NemoClaw does not add and gives enough concrete guidance, decision criteria, verification steps, or risk framing to plan separate mitigations for authentication, prompt handling, and agent behavior." }, { "id": "docs-security-openclaw-controls-003", "question": "I'm documenting the security boundary. Help me explain where NemoClaw protection ends so I can set accurate expectations for reviewers and operators.", "expected_skill": "nemoclaw-user-configure-security", - "ground_truth": "A NemoClaw-specific answer that helps the user explain where NemoClaw protection ends and gives enough concrete guidance, decision criteria, verification steps, or risk framing to set accurate expectations for reviewers and operators.", - "expected_behavior": [ - "The output directly addresses the user's situation: documenting the security boundary.", - "The AI coding assistant loads the expected_skill and references/openclaw-controls.md", - "The output helps the user explain where NemoClaw protection ends with NemoClaw-specific guidance rather than generic advice.", - "The output gives enough concrete guidance, decision criteria, verification steps, or risk framing for the user to set accurate expectations for reviewers and operators.", - "The output avoids inventing unsupported NemoClaw behavior.", - "The output follows progressive disclosure: it answers the current request without dumping unrelated details other than the expected_skill and the references/openclaw-controls.md file." - ] + "ground_truth": "A NemoClaw-specific answer that helps the user explain where NemoClaw protection ends and gives enough concrete guidance, decision criteria, verification steps, or risk framing to set accurate expectations for reviewers and operators." } ] diff --git a/skills/nemoclaw-user-configure-inference/BENCHMARK.md b/skills/nemoclaw-user-configure-inference/BENCHMARK.md new file mode 100644 index 00000000000..3e1b5623ce7 --- /dev/null +++ b/skills/nemoclaw-user-configure-inference/BENCHMARK.md @@ -0,0 +1,75 @@ +# Evaluation Report + +Evaluation of the `nemoclaw-user-configure-inference` skill before publication through NVSkills-Eval. + +This benchmark summarizes 3-Tier Evaluation from NVSkills-Eval results for the skill. The goal is to document whether the skill is safe, discoverable, effective, and useful for agents before it is published for broader workflow use. + +## Evaluation Summary + +- Skill: `nemoclaw-user-configure-inference` +- Evaluation date: 2026-05-28 +- NVSkills-Eval profile: `external` +- Overall verdict: FAIL +- Tier 3 live agent evaluation: not available in this report + +## Agents Used + +- Tier 3 agent details were not available in this report. + +## Metrics Used + +Reported benchmark dimensions: + +- Security: checks whether skill-assisted execution avoids unsafe behavior such as secret leakage, destructive commands, or unauthorized access. +- Correctness: checks whether the agent follows the expected workflow and produces the correct final output. +- Discoverability: checks whether the agent loads the skill when relevant and avoids using it when irrelevant. +- Effectiveness: checks whether the agent performs measurably better with the skill than without it. +- Efficiency: checks whether the agent uses fewer tokens and avoids redundant work. + +Underlying evaluation signals used in this run: + +- No Tier 3 evaluation signal details were available in this report. + +## Test Tasks + +Tier 3 evaluation task details were not available in this report. + +## Results + +Tier 3 dimension rollup was not available in this report. + +## Tier 1: Static Validation Summary + +Tier 1 validation passed with observations. NVSkills-Eval ran 9 checks and found 13 total findings. + +Top findings: + +- MEDIUM PII/gps_coordinates: GPS coordinates (location information) (`references/inference-options.md:89`) +- MEDIUM QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.author' (`skills/nemoclaw-user-configure-inference/SKILL.md`) +- MEDIUM QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/nemoclaw-user-configure-inference/SKILL.md`) +- MEDIUM QUALITY/quality_efficiency: Deeply nested references in set-up-sub-agent.md (`skills/nemoclaw-user-configure-inference/SKILL.md`) +- MEDIUM SCHEMA/body_recommended_section: Missing recommended section: '## Instructions' (`skills/nemoclaw-user-configure-inference/SKILL.md`) + +## Tier 2: Deduplication Summary + +Tier 2 validation reported findings. NVSkills-Eval ran 2 checks and found 3 total findings. + +Top findings: + +- HIGH DUPLICATE/duplicate: Duplicate content found across SKILL.md and references/inference-options.md and references/set-up-sub-agent.md and references/switch-inference-providers.md and references/tool-calling-reliability.md and references/use-local-inference-details.md: + "(preamble)" in SKILL.md (lines 1-3) + vs "(preamble)" in references/inference-options.md (lines 1-2) + vs "(preamble)" in references/set-up-sub-agent.md (lines 1-2) + vs "(preamble)" in references/switch-inference-providers.md (lines 1-2) + vs "(preamble)" in references/tool-calling-reliability.md (lines 1-2) + vs "(preamble)" in references/use-local-inference-details.md (lines 1-2) (`SKILL.md:1`) +- HIGH DUPLICATE/duplicate: Duplicate content found across references/inference-options.md and references/tool-calling-reliability.md: + "## Next Steps" in references/inference-options.md (lines 138-142) + vs "## Next Steps" in references/tool-calling-reliability.md (lines 160-164) (`references/inference-options.md:138`) +- HIGH DUPLICATE/duplicate: Duplicate content found across references/inference-options.md and references/switch-inference-providers.md: + "## How Inference Routing Works" in references/inference-options.md (lines 9-19) + vs "## Notes" in references/switch-inference-providers.md (lines 197-204) (`references/inference-options.md:9`) + +## Publication Recommendation + +The skill should be reviewed before NVSkills-Eval publication. Skill owners should address the findings above and rerun NVSkills-Eval to refresh this benchmark. diff --git a/skills/nemoclaw-user-configure-inference/SKILL.md b/skills/nemoclaw-user-configure-inference/SKILL.md new file mode 100644 index 00000000000..0a0d57e2aea --- /dev/null +++ b/skills/nemoclaw-user-configure-inference/SKILL.md @@ -0,0 +1,285 @@ +--- +name: "nemoclaw-user-configure-inference" +description: "Connects NemoClaw to a local inference server. Use when setting up Ollama, vLLM, TensorRT-LLM, NIM, or any OpenAI-compatible local model server with NemoClaw. Trigger keywords - nemoclaw local inference, ollama nemoclaw, vllm nemoclaw, local model server, openai compatible endpoint, switch nemoclaw inference model, change inference runtime, nemoclaw additional model, nemoclaw sub-agent model, openclaw sub-agent, agents.list, sessions_spawn, vlm-demo, nemoclaw tool calling, ollama tool calls, vllm tool-call-parser, raw json in tui, nemoclaw inference options, nemoclaw onboarding providers, nemoclaw inference routing." +license: "Apache-2.0" +--- + + + + +# Use a Local Inference Server + +## Gotchas + +- Ollama is convenient for local chat, but some model/template combinations can return tool calls as plain text under realistic agent load. + +## Prerequisites + +- NemoClaw installed. +- A local model server running, or a supported Ollama, vLLM, or NIM setup that the NemoClaw onboard wizard can use, start, or install. + +NemoClaw can route inference to a model server running on your machine instead of a cloud API. +This page covers Ollama, compatible-endpoint paths for other servers, and experimental managed options for vLLM and NVIDIA NIM. + +All approaches use the same `inference.local` routing model. +The agent inside the sandbox never connects to your model server directly. +OpenShell intercepts inference traffic and forwards it to the local endpoint you configure. + +## Ollama + +Ollama is the default local inference option. +The onboard wizard detects Ollama automatically when it is installed or running on the host. + +If Ollama is installed but not running, NemoClaw starts it for you. +On macOS and Linux, the wizard can also offer to install Ollama when it is not present. +When the host Ollama is below the minimum version NemoClaw expects for its starter models (currently `0.7.0`), the wizard surfaces an explicit **Upgrade Ollama** entry in the provider menu instead of silently reusing the older daemon, and the express setup path resolves to that entry. +The wizard inspects both the CLI binary (`ollama --version`) and the locally running daemon (`/api/version` on `:11434`) so the upgrade entry still appears when only one side is stale, for example a fresh user-local binary paired with the original system daemon. +The gate skips Windows-host Ollama reached from WSL via `host.docker.internal`; the separate **Use / Start / Install Ollama on Windows host** entries handle that case and run their own actions on the Windows side. +On macOS, the wizard runs the platform install or upgrade path with `brew upgrade ollama`. +On Linux, the wizard runs the official `https://ollama.com/install.sh` path. +Upgrades on Linux always take the sudo-driven system path because the sudo-free user-local fallback would leave the existing system daemon on `:11434` serving the stale binary. +If sudo is not available in a non-interactive run, NemoClaw refuses to silently downgrade the path and asks you to rerun interactively or upgrade Ollama manually. +After an upgrade finishes, NemoClaw re-probes the running daemon's `/api/version` and fails the run if the daemon still reports below the minimum. +Fresh installs skip this re-probe because the bundled installers ship a daemon at or above the minimum. +On WSL, the wizard can use, start, restart, or install Ollama on the Windows host through PowerShell interop. + +### Linux Install Modes + +On native Linux, the install path picks between a system install (under `/usr/local`, via the official `https://ollama.com/install.sh`) and a sudo-free user-local install (under `${HOME}/.local`). +NemoClaw selects the mode automatically: + +- Running as root or with passwordless sudo (`sudo -n true` returns 0) selects the system install. +- A non-interactive run (`NEMOCLAW_NON_INTERACTIVE=1` or no TTY on stdin) without passwordless sudo selects the user-local install. + This is the path that lets headless hosts complete onboarding without prompting for a sudo password. +- An interactive shell without passwordless sudo selects the system install and lets the official installer prompt for the password as usual. + +Override the detection with `NEMOCLAW_OLLAMA_INSTALL_MODE=system` or `NEMOCLAW_OLLAMA_INSTALL_MODE=user`. + +The user-local install replicates only the binary extraction step of the official installer. +It downloads the release tarball, extracts it to `${HOME}/.local`, and launches `${HOME}/.local/bin/ollama serve` once. +It does not configure a systemd service, does not create the `ollama` system user, and does not install CUDA drivers, so the daemon must be relaunched manually after a reboot. +NemoClaw also prints a one-line `PATH` hint if `${HOME}/.local/bin` is not already on your `PATH`; you can add `export PATH="${HOME}/.local/bin:$PATH"` to your shell profile to invoke `ollama` directly. + +Both modes rely on `zstd` for archive extraction. On Debian and Ubuntu, the system path uses `sudo apt-get` to install `zstd` automatically and explains the prompt before continuing. +The user-local path cannot bootstrap system packages without elevation, so if `zstd` is missing it prints per-distro install hints and exits โ€” install `zstd` manually, then rerun onboarding. + +Run the onboard wizard. + +```console +$ nemoclaw onboard +``` + +Select **Local Ollama** from the provider list. +NemoClaw lists installed models or offers starter models if none are installed. +On hosts where the larger starter models fit the currently available GPU memory, the starter list includes `qwen3.6:35b` and selects it by default. +When another GPU workload is using most of the memory at onboard time, NemoClaw downgrades the menu to the largest model that still fits. +It pulls the selected model, loads it into memory, and validates it before continuing. +When Ollama reports a loaded-model context length, NemoClaw uses that value for the `contextWindow` baked into `openclaw.json` unless you set `NEMOCLAW_CONTEXT_WINDOW` yourself. +If the selected model declares that it does not support tool calling, onboarding stops with guidance to choose a model whose `ollama show ` capabilities include `tools`. +The validation also requires structured chat-completions tool calls. +If the model leaks tool-call JSON as plain message text, onboarding stops so you can choose a model that returns tool calls in the expected response field. +On WSL, if you choose the Windows-host Ollama path, NemoClaw uses `host.docker.internal:11434` and pulls missing models through the Ollama HTTP API instead of requiring the `ollama` CLI inside WSL. + +### WSL with Windows-Host Ollama + +When NemoClaw runs inside WSL, the provider menu can include Windows-host Ollama actions: + +- Use Ollama on Windows host when the Windows daemon is already reachable. +- Restart Ollama on Windows host when the daemon is installed but only bound to Windows loopback. +- Start Ollama on Windows host when Ollama is installed but not running. +- Install Ollama on Windows host when Windows does not have Ollama installed. + +The install and restart paths set `OLLAMA_HOST=0.0.0.0:11434` on the Windows side so Docker and WSL can reach the daemon through `host.docker.internal`. +After an install or restart action, NemoClaw relaunches Ollama from the detected Windows tray app or verified `ollama.exe` path and waits until `host.docker.internal:11434` responds. + +If the HTTP endpoint is not reachable yet, NemoClaw also checks for the Windows `ollama.exe` process through PowerShell interop so it can offer a start or restart action instead of hiding the Windows-host path. +If the daemon does not become reachable, onboarding prints PowerShell commands you can run to inspect the Windows-side process and port state. Use one Ollama instance on port `11434` at a time. +If both WSL and Windows-host Ollama are running, pick the intended menu entry during onboarding so NemoClaw validates and pulls models against the right daemon. + +**Warning:** + +Ollama is convenient for local chat, but some model/template combinations can +return tool calls as plain text under realistic agent load. If the TUI shows raw +JSON such as `{"name":"memory_search","arguments":{...}}` instead of running a +tool, switch to vLLM with `--enable-auto-tool-choice` and the correct +`--tool-call-parser`. See [Tool-Calling Reliability](references/tool-calling-reliability.md). + +### Authenticated Reverse Proxy + +On non-WSL hosts, NemoClaw keeps Ollama bound to `127.0.0.1:11434` and starts a token-gated reverse proxy on `0.0.0.0:11435`. +The native install/start paths also reset NemoClaw-managed systemd launches to the loopback binding. +Containers and other hosts on the local network reach Ollama only through the +proxy, which validates a Bearer token before forwarding requests. +On that native path, NemoClaw never exposes Ollama without authentication. + +WSL Ollama paths do not use this proxy. +Windows-host Ollama uses the Windows daemon through `host.docker.internal`. + +For non-WSL Ollama setups, the onboard wizard manages the proxy automatically: + +- Generates a random 24-byte token on first run and stores it in + `~/.nemoclaw/ollama-proxy-token` with `0600` permissions. +- Starts the proxy after Ollama and verifies it before continuing. +- Cleans up stale proxy processes from previous runs. +- Probes the sandbox Docker network path to the proxy before committing the inference route. +- Stops matching proxy processes during uninstall before deleting NemoClaw state. +- Reuses the persisted token after a host reboot so you do not need to re-run + onboard. + +On native Linux hosts, a firewall can allow the host proxy health check while still blocking sandbox containers on the OpenShell Docker bridge. +When the sandbox-side proxy probe fails with a TCP error, onboarding exits before it saves the inference route and prints a command like: + +```console +$ sudo ufw allow from to any port 11435 proto tcp +$ nemoclaw onboard +``` + +If the probe cannot run, for example because Docker Desktop or WSL uses a different host routing model, onboarding continues and relies on the regular proxy health check. + +The sandbox provider is configured to use proxy port `11435` with the generated +token as its `OPENAI_API_KEY` credential. +OpenShell's L7 proxy injects the token at egress, so the agent inside the +sandbox never sees the token directly. + +All proxy endpoints require the Bearer token, including `GET /api/tags`. +Internal health and reachability checks run via the proxy treat any HTTP +response (including `401`) as proof the proxy is alive โ€” they only fail +when nothing answers at all. + +If Ollama is already running on a non-loopback address when you start onboard, +the wizard restarts it on `127.0.0.1:11434` so the proxy is the only network +path to the model server. + +### GPU Memory Cleanup + +When you switch away from Ollama, stop host services, or destroy an Ollama-backed sandbox, NemoClaw asks Ollama to unload currently loaded models from GPU memory. +The cleanup sends `keep_alive: 0` for each model reported by Ollama and runs on a best-effort basis, so shutdown continues if Ollama is already stopped. +This does not delete downloaded model files. + +Load [references/use-local-inference-details.md](references/use-local-inference-details.md) for detailed steps on Non-Interactive Setup. + +## OpenAI-Compatible Server + +This option works with any server that implements `/v1/chat/completions`, including vLLM, TensorRT-LLM, llama.cpp, LocalAI, and others. +For compatible endpoints, NemoClaw uses `/v1/chat/completions` by default. +This avoids a class of failures where local backends accept `/v1/responses` requests but silently drop the system prompt and tool definitions. +To opt in to `/v1/responses`, set `NEMOCLAW_PREFERRED_API=openai-responses` before running onboard. + +Start your model server. +The examples below use vLLM, but any OpenAI-compatible server works. + +```console +$ vllm serve meta-llama/Llama-3.1-8B-Instruct --port 8000 +``` + +Run the onboard wizard. + +```console +$ nemoclaw onboard +``` + +When the wizard asks you to choose an inference provider, select **Other OpenAI-compatible endpoint**. +Enter the base URL of your local server, for example `http://localhost:8000/v1`. + +The wizard prompts for an API key. +If your server does not require authentication, enter any non-empty string (for example, `dummy`). + +NemoClaw validates the endpoint by sending a test inference request before continuing. +The wizard probes `/v1/chat/completions` by default for the compatible-endpoint provider. +If you set `NEMOCLAW_PREFERRED_API=openai-responses`, NemoClaw probes `/v1/responses` instead and only selects it when the response includes the streaming events OpenClaw requires. +If a reasoning model returns only reasoning content before producing a final answer, NemoClaw retries the smoke request with a larger response budget. +Route, configuration, and authentication failures still fail immediately. + +Load [references/use-local-inference-details.md](references/use-local-inference-details.md) for detailed steps on Non-Interactive Setup, Selecting the API Path. + +## Anthropic-Compatible Server + +Load [references/use-local-inference-details.md](references/use-local-inference-details.md) for detailed steps. + +## vLLM + +When vLLM is already running on `localhost:8000`, NemoClaw can detect it automatically and query the `/v1/models` endpoint to determine the loaded model. +On supported Linux hosts with NVIDIA GPUs, the onboard wizard can also install or start a managed vLLM container for you. + +For an already-running vLLM server, run `nemoclaw onboard` and select **Local vLLM [experimental]** from the provider list. + +```console +$ nemoclaw onboard +``` + +If vLLM is already running, NemoClaw detects the running model and validates the endpoint. +If vLLM is not running and your host matches a DGX Spark or DGX Station managed profile, NemoClaw shows the **Install vLLM** or **Start vLLM** entry by default. +Generic Linux NVIDIA GPU hosts still require `NEMOCLAW_EXPERIMENTAL=1` or `NEMOCLAW_PROVIDER=install-vllm` before the managed entry appears. +NemoClaw pulls the vLLM image, downloads model weights into `~/.cache/huggingface`, starts the `nemoclaw-vllm` container on `localhost:8000`, and prints progress markers while the model loads. +The first run can take 10 to 30 minutes. +Later runs reuse the cached image and model weights. + +Managed vLLM uses these profiles: + +| Host profile | Default model | +|---|---| +| DGX Spark | `Qwen/Qwen3.6-27B-FP8` | +| DGX Station | `Qwen/Qwen3.6-27B-FP8` | +| Linux with an NVIDIA GPU | `nvidia/NVIDIA-Nemotron-3-Nano-4B-FP8` | + +**Note:** + +NemoClaw forces the `chat/completions` API path for vLLM. +The vLLM `/v1/responses` endpoint does not run the `--tool-call-parser`, so tool calls arrive as raw text. + +Load [references/use-local-inference-details.md](references/use-local-inference-details.md) for detailed steps on Non-Interactive Setup, Override the Managed-vLLM Model. + +## NVIDIA NIM (Experimental) + +NemoClaw can pull, start, and manage a NIM container on hosts with a NIM-capable NVIDIA GPU. + +Set the experimental flag and run onboard. + +```console +$ NEMOCLAW_EXPERIMENTAL=1 nemoclaw onboard +``` + +Select **Local NVIDIA NIM [experimental]** from the provider list. +NemoClaw filters available models by GPU VRAM, pulls the NIM container image, starts it, and waits for it to become healthy before continuing. +On hosts with mixed NVIDIA GPU models, the preflight summary shows each detected GPU model and the total VRAM so you can confirm which device class the model selection used. + +NIM container images are hosted on `nvcr.io` and require NGC registry authentication before `docker pull` succeeds. +If Docker is not already logged in to `nvcr.io`, onboard prompts for an [NGC API key](https://org.ngc.nvidia.com/setup/api-key) and runs `docker login nvcr.io` over `--password-stdin` so the key is never written to disk or shell history. +The prompt masks the key during input and retries once on a bad key before failing. +In non-interactive mode, onboard exits with login instructions if Docker is not already authenticated; run `docker login nvcr.io` yourself, then re-run `nemoclaw onboard --non-interactive`. +If `NGC_API_KEY` or `NVIDIA_API_KEY` is already exported, NemoClaw passes it into the managed NIM container through the process environment instead of command-line arguments. +If the NIM container exits before the health endpoint becomes ready, onboarding stops early and prints the last container log lines. + +**Note:** + +NIM uses vLLM internally. +The same `chat/completions` API path restriction applies. + +Load [references/use-local-inference-details.md](references/use-local-inference-details.md) for detailed steps on Non-Interactive Setup. + +## Timeout Configuration + +Load [references/use-local-inference-details.md](references/use-local-inference-details.md) for detailed steps. + +## Verify the Configuration + +Load [references/use-local-inference-details.md](references/use-local-inference-details.md) for detailed steps. + +## Switch Models at Runtime + +Load [references/use-local-inference-details.md](references/use-local-inference-details.md) for detailed steps. + +## References + +- **Load [references/switch-inference-providers.md](references/switch-inference-providers.md)** when switching inference providers, changing the model runtime, or reconfiguring inference routing. Changes the active inference model without restarting the sandbox. +- **Load [references/set-up-sub-agent.md](references/set-up-sub-agent.md)** when users ask how to add a second model, configure a sub-agent model, use Omni for vision tasks, configure agents.list, or use sessions_spawn in NemoClaw. Shows the NemoClaw-specific file paths and update flow for adding an auxiliary OpenClaw sub-agent model. +- **[references/tool-calling-reliability.md](references/tool-calling-reliability.md)** โ€” Explains Ollama tool-call leak symptoms, when vLLM with a tool-call parser is recommended, and how to repoint NemoClaw to a parser-aware local endpoint. +- **Load [references/inference-options.md](references/inference-options.md)** when explaining which providers are available, what the onboard wizard presents, or how inference routing works. Lists all inference providers offered during NemoClaw onboarding. +- **Load [references/use-local-inference-details.md](references/use-local-inference-details.md)** when you need detailed steps for Non-Interactive Setup, Selecting the API Path, Anthropic-Compatible Server, and related details. + +## Related Skills + +- [Inference Options](references/inference-options.md) for the full list of providers available during onboarding. +- [Tool-Calling Reliability](references/tool-calling-reliability.md) for diagnosing raw JSON tool-call output with local models. +- [Switch Inference Models](references/switch-inference-providers.md) for runtime model switching. +- `nemoclaw-user-get-started` โ€” Quickstart (use the `nemoclaw-user-get-started` skill) for first-time installation diff --git a/skills/nemoclaw-user-configure-inference/evals/evals.json b/skills/nemoclaw-user-configure-inference/evals/evals.json new file mode 100644 index 00000000000..a0bd47ac294 --- /dev/null +++ b/skills/nemoclaw-user-configure-inference/evals/evals.json @@ -0,0 +1,92 @@ +[ + { + "id": "docs-inference-inference-options-001", + "question": "I'm choosing an inference option during onboarding. Help me compare hosted providers, local servers, and compatible endpoints so I can select a model path that fits my privacy, cost, and reliability needs.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user compare hosted providers, local servers, and compatible endpoints and gives enough concrete guidance, decision criteria, verification steps, or risk framing to select a model path that fits my privacy, cost, and reliability needs." + }, + { + "id": "docs-inference-inference-options-002", + "question": "I'm preparing provider credentials. Help me know which provider capabilities and secrets onboarding requires so I can complete setup without avoidable credential errors.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user know which provider capabilities and secrets onboarding requires and gives enough concrete guidance, decision criteria, verification steps, or risk framing to complete setup without avoidable credential errors." + }, + { + "id": "docs-inference-inference-options-003", + "question": "I'm evaluating routed inference. Help me understand how the sandbox calls models through the gateway so I can trust that model credentials stay outside the sandbox.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user understand how the sandbox calls models through the gateway and gives enough concrete guidance, decision criteria, verification steps, or risk framing to trust that model credentials stay outside the sandbox." + }, + { + "id": "docs-inference-use-local-inference-001", + "question": "I'm connecting a local inference server. Help me route NemoClaw model traffic to Ollama, vLLM, TensorRT-LLM, NIM, or another compatible endpoint so I can meet privacy, latency, or cost goals.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user route NemoClaw model traffic to Ollama, vLLM, TensorRT-LLM, NIM, or another compatible endpoint and gives enough concrete guidance, decision criteria, verification steps, or risk framing to meet privacy, latency, or cost goals." + }, + { + "id": "docs-inference-use-local-inference-002", + "question": "I'm debugging local endpoint reachability. Help me separate NemoClaw routing issues from model-server issues so I can fix the right component first.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user separate NemoClaw routing issues from model-server issues and gives enough concrete guidance, decision criteria, verification steps, or risk framing to fix the right component first." + }, + { + "id": "docs-inference-use-local-inference-003", + "question": "I'm configuring traffic through `inference.local`. Help me understand the required host, port, and model settings so I can make sandboxed inference calls resolve to my local server.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user understand the required host, port, and model settings and gives enough concrete guidance, decision criteria, verification steps, or risk framing to make sandboxed inference calls resolve to my local server." + }, + { + "id": "docs-inference-switch-inference-providers-001", + "question": "I'm switching inference models during a running session. Help me change model behavior without restarting the sandbox so I can adapt to task, cost, or reliability needs quickly.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user change model behavior without restarting the sandbox and gives enough concrete guidance, decision criteria, verification steps, or risk framing to adapt to task, cost, or reliability needs quickly." + }, + { + "id": "docs-inference-switch-inference-providers-002", + "question": "I'm confirming a runtime model change. Help me verify the agent is using the new active model so I can avoid mistaking host configuration changes for live routing changes.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user verify the agent is using the new active model and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid mistaking host configuration changes for live routing changes." + }, + { + "id": "docs-inference-switch-inference-providers-003", + "question": "I'm trying a different model during active work. Help me know how to roll back to the previous model so I can experiment without disrupting the assistant workflow.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user know how to roll back to the previous model and gives enough concrete guidance, decision criteria, verification steps, or risk framing to experiment without disrupting the assistant workflow." + }, + { + "id": "docs-inference-set-up-sub-agent-001", + "question": "I'm configuring a task-specific sub-agent. Help me assign a specialized model to work the default agent should not handle so I can improve task fit without changing the whole assistant.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user assign a specialized model to work the default agent should not handle and gives enough concrete guidance, decision criteria, verification steps, or risk framing to improve task fit without changing the whole assistant." + }, + { + "id": "docs-inference-set-up-sub-agent-002", + "question": "I'm editing sub-agent model configuration. Help me understand where files, credentials, and workspace settings live so I can avoid leaking secrets or changing the wrong agent.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user understand where files, credentials, and workspace settings live and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid leaking secrets or changing the wrong agent." + }, + { + "id": "docs-inference-set-up-sub-agent-003", + "question": "I'm testing a new sub-agent. Help me send a prompt that exercises the intended routing so I can prove it uses the expected provider and model.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user send a prompt that exercises the intended routing and gives enough concrete guidance, decision criteria, verification steps, or risk framing to prove it uses the expected provider and model." + }, + { + "id": "docs-inference-tool-calling-reliability-001", + "question": "I'm seeing tool calls leak as plain text. Help me diagnose whether the model, server, or parser is incompatible so I can restore reliable tool execution.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user diagnose whether the model, server, or parser is incompatible and gives enough concrete guidance, decision criteria, verification steps, or risk framing to restore reliable tool execution." + }, + { + "id": "docs-inference-tool-calling-reliability-002", + "question": "I'm comparing local inference runtimes. Help me understand whether Ollama, vLLM, or parser settings better support tool calls so I can choose a runtime that matches the agent's tool needs.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user understand whether Ollama, vLLM, or parser settings better support tool calls and gives enough concrete guidance, decision criteria, verification steps, or risk framing to choose a runtime that matches the agent's tool needs." + }, + { + "id": "docs-inference-tool-calling-reliability-003", + "question": "I'm letting an always-on assistant use tools unattended. Help me define the reliability bar for local tool calling so I can avoid silent failures or unsafe plain-text tool outputs.", + "expected_skill": "nemoclaw-user-configure-inference", + "ground_truth": "A NemoClaw-specific answer that helps the user define the reliability bar for local tool calling and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid silent failures or unsafe plain-text tool outputs." + } +] diff --git a/skills/nemoclaw-user-configure-inference/references/inference-options.md b/skills/nemoclaw-user-configure-inference/references/inference-options.md new file mode 100644 index 00000000000..5242cff46c5 --- /dev/null +++ b/skills/nemoclaw-user-configure-inference/references/inference-options.md @@ -0,0 +1,142 @@ + + +# NemoClaw Inference Options + +NemoClaw supports multiple inference providers. +During onboarding, the `nemoclaw onboard` wizard presents a numbered list of providers to choose from. +Your selection determines where the agent's inference traffic is routed. + +## How Inference Routing Works + +The agent inside the sandbox talks to `inference.local`. +It never connects to a provider directly. +OpenShell intercepts inference traffic on the host and forwards it to the provider you selected. + +Provider credentials stay on the host. +The sandbox does not receive your API key. +Local Ollama and local vLLM do not require your host `OPENAI_API_KEY`. +NemoClaw uses provider-specific local tokens for those routes, and rebuilds of legacy local-inference sandboxes migrate away from stale OpenAI credential requirements. + +## Provider Status + +| Provider | Status | Endpoint type | Notes | +|----------|--------|---------------|-------| +| NVIDIA Endpoints | Tested | OpenAI-compatible | Hosted models on integrate.api.nvidia.com | +| OpenAI | Tested | Native OpenAI-compatible | Uses OpenAI model IDs | +| Other OpenAI-compatible endpoint | Tested | Custom OpenAI-compatible | For compatible proxies and gateways | +| Anthropic | Tested | Native Anthropic | Uses anthropic-messages | +| Other Anthropic-compatible endpoint | Tested | Custom Anthropic-compatible | For Claude proxies and compatible gateways | +| Google Gemini | Tested | OpenAI-compatible | Uses Google's OpenAI-compatible endpoint | +| Hermes Provider | Hermes only | OpenAI-compatible route | Available when onboarding Hermes Agent through `nemohermes` | +| Local Ollama | Caveated | Local Ollama API | Available when Ollama is installed or running on the host | +| Local NVIDIA NIM | Experimental | Local OpenAI-compatible | Requires `NEMOCLAW_EXPERIMENTAL=1` and a NIM-capable GPU | +| Local vLLM (already running) | Caveated | Local OpenAI-compatible | Appears in the onboarding menu when NemoClaw detects a server already on `localhost:8000`. No flag required. | +| Local vLLM (managed install/start) | Caveated | Local OpenAI-compatible | Appears by default on DGX Spark and DGX Station. Generic Linux NVIDIA GPU hosts require `NEMOCLAW_EXPERIMENTAL=1` or `NEMOCLAW_PROVIDER=install-vllm`. NemoClaw pulls/starts a vLLM container on a supported NVIDIA GPU host. | + +## Provider Options + +The onboard wizard presents the following provider options by default. +The first six are always available. +Ollama appears when it is installed or running on the host. +Local vLLM appears when NemoClaw detects a running vLLM server. +The managed install/start vLLM entry appears by default on DGX Spark and DGX Station, and appears on generic Linux NVIDIA GPU hosts after opt-in. + +| Option | Description | Curated models | +|--------|-------------|----------------| +| NVIDIA Endpoints | Routes to models hosted on [build.nvidia.com](https://build.nvidia.com). You can also enter any model ID from the catalog. Set `NVIDIA_API_KEY`. | Nemotron 3 Super 120B, GLM-5.1, MiniMax M2.7, GPT-OSS 120B, DeepSeek V4 Pro | +| OpenAI | Routes to the OpenAI API. Set `OPENAI_API_KEY`. | `gpt-5.4`, `gpt-5.4-mini`, `gpt-5.4-nano`, `gpt-5.4-pro-2026-03-05` | +| Other OpenAI-compatible endpoint | Routes to any server that implements `/v1/chat/completions`. NemoClaw uses `/v1/chat/completions` at runtime by default; set `NEMOCLAW_PREFERRED_API=openai-responses` to allow `/v1/responses` for proxies that implement it, such as some llama.cpp builds. The wizard prompts for a base URL and model name. Works with OpenRouter, LocalAI, llama.cpp, or any compatible proxy. When you enable Telegram messaging, onboarding also runs a bounded sandbox-side smoke check through `https://inference.local/v1/chat/completions`. Set `COMPATIBLE_API_KEY`. | You provide the model name. | +| Anthropic | Routes to the Anthropic Messages API. Set `ANTHROPIC_API_KEY`. | `claude-sonnet-4-6`, `claude-haiku-4-5`, `claude-opus-4-6` | +| Other Anthropic-compatible endpoint | Routes to any server that implements the Anthropic Messages API (`/v1/messages`). The wizard prompts for a base URL and model name. Set `COMPATIBLE_ANTHROPIC_API_KEY`. | You provide the model name. | +| Google Gemini | Routes to Google's OpenAI-compatible chat-completions endpoint. NemoClaw skips the Responses-API probe because Gemini does not support `/v1/responses`. Set `GEMINI_API_KEY`. | `gemini-3.1-pro-preview`, `gemini-3.1-flash-lite-preview`, `gemini-3-flash-preview`, `gemini-2.5-pro`, `gemini-2.5-flash`, `gemini-2.5-flash-lite` | +| Hermes Provider | Routes Hermes Agent through the host OpenShell provider registered by NemoClaw when onboarding Hermes Agent. | Curated Hermes Provider models such as `moonshotai/kimi-k2.6`, `openai/gpt-5.4-mini`, and `z-ai/glm-5.1`. | +| Local Ollama | Routes to a local Ollama instance on `localhost:11434`. NemoClaw detects installed models, offers starter models if none are present, pulls and warms the selected model, and validates it. | Selected during onboarding. For more information, refer to [Use a Local Inference Server](../SKILL.md). | +| Model Router | Starts a host-side router on port `4000`, registers it as an OpenAI-compatible provider, and keeps the sandbox pointed at `inference.local`. Set `NEMOCLAW_PROVIDER=routed` for non-interactive setup. | The router pool defines the model names. | + +## Choosing the Right Option for Nemotron + +NVIDIA Nemotron models expose OpenAI-compatible APIs across every supported deployment surface, so two onboarding options can route to Nemotron. + +| Where Nemotron is hosted | Onboard wizard option | Why | +|---|---|---| +| `build.nvidia.com` (NVIDIA-hosted) | **Option 1: NVIDIA Endpoints** | NemoClaw sets the base URL to `https://integrate.api.nvidia.com/v1` for you and validates the model against the build catalog. | +| Self-hosted NIM container | **Option 3: Other OpenAI-compatible endpoint** | NIM exposes an OpenAI-compatible `/v1/chat/completions` route. Point the base URL at your NIM service and enter the Nemotron model ID. | +| Enterprise NVIDIA AI Enterprise gateway | **Option 3: Other OpenAI-compatible endpoint** | Enterprise gateways front Nemotron with the same OpenAI-compatible contract. Use the gateway's base URL and your enterprise token. | +| vLLM, SGLang, or TRT-LLM serving Nemotron weights | **Option 3: Other OpenAI-compatible endpoint** | Each runtime exposes Nemotron through `/v1/chat/completions`. Use the runtime's base URL and the model ID it reports. | +| Local NIM started by the wizard | **Local NVIDIA NIM** (experimental) | Requires `NEMOCLAW_EXPERIMENTAL=1` and a NIM-capable GPU. NemoClaw pulls and manages the container for you. | + +For Option 3, the API key environment variable is `COMPATIBLE_API_KEY`. Set it to whatever credential your endpoint expects, or any non-empty placeholder if your endpoint does not require auth. + +## Model Router + +The Model Router option uses the `routed` inference profile in `nemoclaw-blueprint/blueprint.yaml`. +When you select it, NemoClaw starts the router proxy on the host, waits for its health endpoint, registers the `nvidia-router` provider with OpenShell, and creates the sandbox with the same `inference.local` route the agent uses for other providers. +The sandbox does not call the router port directly. + +The router model pool lives in `nemoclaw-blueprint/router/pool-config.yaml`. +The default pool routes between NVIDIA-hosted Nemotron models and uses the `tolerance` value to choose the lowest-cost model whose predicted quality stays within the configured threshold. +To use the router in scripted setup, set: + +```console +$ NEMOCLAW_PROVIDER=routed NVIDIA_API_KEY= nemoclaw onboard --non-interactive +``` + +### Host Python requirement + +The Model Router runs in a host-side virtual environment that NemoClaw creates during onboarding. +NemoClaw probes `python3.13`, `python3.12`, `python3.11`, `python3.10`, and bare `python3`, and adopts the first interpreter that satisfies both of: + +- Version inside `[3.10, 3.14)`. +- `ensurepip`, `pyexpat`, `ssl`, and `venv` all import without error. + +If no candidate qualifies, onboarding aborts and prints the real failure for each candidate. +This surfaces issues like Homebrew `python@3.14` whose `pyexpat` extension fails to dlopen against the older system `libexpat` on macOS. + +To pin a specific interpreter, set `NEMOCLAW_MODEL_ROUTER_PYTHON` to its absolute path before running `nemoclaw onboard`: + +```console +$ NEMOCLAW_MODEL_ROUTER_PYTHON=/opt/homebrew/bin/python3.12 nemoclaw onboard +``` + +The pin is strict. +NemoClaw probes only that interpreter and aborts with the failure reason if it does not qualify, rather than silently falling back to a different python on `PATH`. +Relative command names such as `python3.12` are rejected; use `command -v python3.12` to find the absolute path. +If `python -m venv` itself fails for a probe-clean interpreter (for example, a corrupt ensurepip seed), NemoClaw retries with the next healthy candidate when no pin is set; with a pin set, the failure stops onboarding so you can fix or repoint the pinned python. + +## Caveated Local Options + +The following local inference options are caveated. +Local NIM and generic Linux managed vLLM install/start require `NEMOCLAW_EXPERIMENTAL=1`; DGX Spark and DGX Station managed vLLM entries appear by default. +An already-running vLLM server appears directly in the onboarding selection list. + +| Option | Condition | Notes | +|--------|-----------|-------| +| Local NVIDIA NIM | NIM-capable GPU detected | Pulls and manages a NIM container. | +| Local vLLM | vLLM running on `localhost:8000`, or a supported DGX Spark, DGX Station, or Linux NVIDIA GPU profile | Auto-detects the loaded model when vLLM is already running. Can install or start a managed vLLM container by default on DGX Spark/Station and after opt-in on generic Linux NVIDIA GPU hosts. | + +For setup instructions, refer to [Use a Local Inference Server](../SKILL.md). + +## Validation + +NemoClaw validates the selected provider and model before creating the sandbox. +If credential validation fails, the wizard asks whether to re-enter the API key, choose a different provider, retry, or exit. +Transient upstream validation failures are retried before the wizard reports a provider failure. +The `nvapi-` prefix check applies only to `NVIDIA_API_KEY`. +Other provider credentials, such as `OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `GEMINI_API_KEY`, and compatible endpoint keys, use provider-aware validation during retry. + +| Provider type | Validation method | +|---|---| +| OpenAI | Tries `/responses` first, then `/chat/completions`. | +| NVIDIA Endpoints | Validates via `/v1/chat/completions` only; the `/v1/responses` probe is skipped because NVIDIA Build does not expose `/v1/responses` (returns 404 for every model). | +| Google Gemini | Validates via Gemini's OpenAI-compatible chat-completions path only; the `/v1/responses` probe is skipped because Gemini does not support the Responses API. | +| Other OpenAI-compatible endpoint | Tries `/v1/responses` first with a tool-calling probe; falls back to `/v1/chat/completions`. Selected runtime API defaults to `/v1/chat/completions`; set `NEMOCLAW_PREFERRED_API=openai-responses` to allow `/v1/responses` at runtime when validation succeeds. | +| Anthropic-compatible | Tries `/v1/messages`. | +| NVIDIA Endpoints (manual model entry) | Validates the model name against the catalog API. | +| Compatible endpoints | Sends a real inference request because many proxies do not expose a `/models` endpoint. For OpenAI-compatible endpoints, the probe tries `/v1/responses` first then falls back to `/v1/chat/completions`; the selected runtime API defaults to `/v1/chat/completions`. Set `NEMOCLAW_PREFERRED_API=openai-responses` to allow `/v1/responses` at runtime when validation succeeds. | +| Local NVIDIA NIM | Validates via `/v1/chat/completions` only; the `/v1/responses` probe is skipped (same as NVIDIA Endpoints). | + +## Next Steps + +- [Use a Local Inference Server](../SKILL.md) for Ollama, vLLM, NIM, and compatible-endpoint setup details. +- [Tool-Calling Reliability](tool-calling-reliability.md) for deciding when Ollama is enough and when vLLM with a parser is safer. +- [Switch Inference Models](switch-inference-providers.md) for changing the model at runtime without re-onboarding. diff --git a/skills/nemoclaw-user-configure-inference/references/set-up-sub-agent.md b/skills/nemoclaw-user-configure-inference/references/set-up-sub-agent.md new file mode 100644 index 00000000000..148eaf0e7e2 --- /dev/null +++ b/skills/nemoclaw-user-configure-inference/references/set-up-sub-agent.md @@ -0,0 +1,120 @@ + + +# Set Up Task-Specific Sub-Agents + +OpenClaw documents the sub-agent behavior, `sessions_spawn` tool, `agents.list` configuration, tool policy, nesting, and auth model in [Sub-Agents](https://docs.openclaw.ai/tools/subagents). +Use that page as the source of truth for how OpenClaw sub-agents work. + +This NemoClaw page covers the sandbox-specific pieces: where the OpenClaw config lives, where to put per-agent credentials, which writable workspace path agents should use, and how the Omni VLM demo maps onto those paths. + +## NemoClaw Sandbox Paths + +NemoClaw runs OpenClaw inside an OpenShell sandbox. +When adapting an OpenClaw sub-agent setup, use these paths inside the sandbox: + +| Path | Purpose | +|---|---| +| `/sandbox/.openclaw/openclaw.json` | OpenClaw config, including `models.providers`, `agents.defaults`, and `agents.list`. | +| `/sandbox/.openclaw/.config-hash` | Hash for `openclaw.json`. Keep it in sync after manual config edits; it becomes a startup-enforced trust anchor only after the file is root-owned and read-only. | +| `/sandbox/.openclaw/agents//agent/auth-profiles.json` | Per-agent provider credentials. Use this when a sub-agent calls an auxiliary provider directly. | +| `/sandbox/.openclaw/workspace/` | Writable shared workspace path for files the primary agent passes to the sub-agent. | +| `/tmp/gateway.log` | OpenClaw gateway log. Use it to confirm config reloads and diagnose sub-agent failures. | + +For file-based tasks, instruct agents to use `/sandbox/.openclaw/workspace/`. +Avoid relying on legacy `.openclaw-data` paths or read-only OpenClaw paths in delegation instructions. + +## Omni Vision Sub-Agent Example + +The [`vlm-demo`](https://github.com/brevdev/nemoclaw-demos/tree/main/vlm-demo) applies the OpenClaw sub-agent pattern to a vision task. +It keeps the primary `main` agent on the normal NemoClaw inference route and adds a `vision-operator` sub-agent backed by an Omni vision model. + +| OpenClaw field | Omni example value | +|---|---| +| Primary agent | `main` | +| Primary model | `inference/nvidia/nemotron-3-super-120b-a12b` | +| Auxiliary provider | `nvidia-omni` | +| Sub-agent | `vision-operator` | +| Sub-agent model | `nvidia-omni/private/nvidia/nemotron-3-nano-omni-reasoning-30b-a3b` | +| Delegation tool | `sessions_spawn` | + +Omni is used as the specialist model for image tasks. +The primary orchestration model remains responsible for conversation, planning, and deciding when to delegate. + +## Update the Sandbox Config + +Fetch the current OpenClaw config from the sandbox, patch it with your auxiliary provider and `agents.list` changes, then upload it back. + +```console +$ export SANDBOX=my-assistant +$ export DOCKER_CTR=openshell-cluster-nemoclaw +$ docker exec "$DOCKER_CTR" kubectl exec -n openshell "$SANDBOX" -c agent -- cat /sandbox/.openclaw/openclaw.json > /tmp/openclaw.json +``` + +Create `/tmp/openclaw.updated.json` with the OpenClaw sub-agent config. +For the Omni example, the demo provides `vlm-demo/vlm-subagent/openclaw-patch.py`. + +Upload the patched config and refresh the hash. +In the default mutable state, this keeps the local hash consistent but does not make it tamper-proof; lock the config root-owned and read-only afterward if the sandbox should enforce config integrity at startup. + +```console +$ docker exec "$DOCKER_CTR" kubectl exec -n openshell "$SANDBOX" -c agent -- chmod 644 /sandbox/.openclaw/openclaw.json +$ docker exec "$DOCKER_CTR" kubectl exec -n openshell "$SANDBOX" -c agent -- chmod 644 /sandbox/.openclaw/.config-hash +$ cat /tmp/openclaw.updated.json | docker exec -i "$DOCKER_CTR" kubectl exec -i -n openshell "$SANDBOX" -c agent -- sh -c 'cat > /sandbox/.openclaw/openclaw.json' +$ docker exec "$DOCKER_CTR" kubectl exec -n openshell "$SANDBOX" -c agent -- /bin/bash -c "cd /sandbox/.openclaw && sha256sum openclaw.json > .config-hash" +$ docker exec "$DOCKER_CTR" kubectl exec -n openshell "$SANDBOX" -c agent -- chmod 444 /sandbox/.openclaw/openclaw.json +$ docker exec "$DOCKER_CTR" kubectl exec -n openshell "$SANDBOX" -c agent -- chmod 444 /sandbox/.openclaw/.config-hash +``` + +Check `/tmp/gateway.log` after upload and confirm the gateway hot-reloaded the provider or `agents.list` change. + +## Add Sub-Agent Credentials + +If the auxiliary model uses a provider key outside the normal NemoClaw inference route, put that key in the sub-agent auth profile. +For the Omni example: + +```text +/sandbox/.openclaw/agents/vision-operator/agent/auth-profiles.json +``` + +Use the same provider ID that appears in `models.providers`, such as `nvidia-omni`. +After uploading the auth profile, make sure the sub-agent directory is owned by the sandbox user: + +```console +$ docker exec "$DOCKER_CTR" kubectl exec -n openshell "$SANDBOX" -c agent -- chown -R sandbox:sandbox /sandbox/.openclaw/agents/vision-operator +``` + +## Allow Auxiliary Provider Egress + +If the sub-agent calls a provider directly, update the OpenShell network policy for the binary that makes the request. +In the Omni demo, the OpenClaw gateway runs as `/usr/local/bin/node`, so the NVIDIA endpoint policy must allow that binary. + +Refer to Customize the Network Policy (use the `nemoclaw-user-manage-policy` skill) for policy update workflows. + +## Add Delegation Instructions + +OpenClaw handles `sessions_spawn`, but the primary agent still needs task instructions. +Place those instructions in the writable workspace, for example: + +```text +/sandbox/.openclaw/workspace/TOOLS.md +``` + +The Omni demo includes `vlm-demo/vlm-subagent/TOOLS.md`, which tells `main` to delegate image tasks to `vision-operator` and tells the sub-agent to read the image path it receives. +Adapt that file for other task-specific models. + +## Demo Assets + +Use the [`vlm-demo`](https://github.com/brevdev/nemoclaw-demos/tree/main/vlm-demo) repository for runnable Omni example assets: + +- `vlm-subagent-guide.md` for a command-by-command walkthrough. +- `vlm-subagent/openclaw-patch.py` for patching `openclaw.json`. +- `vlm-subagent/auth-profiles.template.json` for the sub-agent auth profile. +- `vlm-subagent/TOOLS.md` for delegation instructions. + +## Next Steps + +Use the following resources for more information: + +- Refer to [OpenClaw Sub-Agents](https://docs.openclaw.ai/tools/subagents) for `sessions_spawn`, `agents.list`, nesting, tool policy, and auth behavior. +- Refer to [Switch Inference Providers](switch-inference-providers.md) to change the primary orchestration model instead of adding a sub-agent model. +- Refer to Workspace Files (use the `nemoclaw-user-manage-sandboxes` skill) to understand per-agent workspace directories. diff --git a/skills/nemoclaw-user-configure-inference/references/switch-inference-providers.md b/skills/nemoclaw-user-configure-inference/references/switch-inference-providers.md new file mode 100644 index 00000000000..c5a623c42e1 --- /dev/null +++ b/skills/nemoclaw-user-configure-inference/references/switch-inference-providers.md @@ -0,0 +1,207 @@ + + +# Switch Inference Models at Runtime + +Change the active inference model while the sandbox is running. +No restart is required. + +## Prerequisites + +- A running NemoClaw sandbox. +- The OpenShell CLI on your `PATH`, which NemoClaw uses under the hood. + +## Switch to a Different Model + +Use `nemoclaw inference set` with the provider and model that match the upstream you want to use. +The command updates the OpenShell inference route and synchronizes the running agent config. +For OpenClaw, it updates `agents.defaults.model.primary` and the matching provider namespace. +For Hermes, it updates `/sandbox/.hermes/config.yaml` (`model.default`, `model.base_url`, and `model.provider: custom`) without rebuilding or restarting Hermes. + +Pass `--sandbox ` when you do not want to use the default registered sandbox. +Under `nemohermes`, pass `--sandbox ` when more than one Hermes sandbox is registered. + +### NVIDIA Endpoints + +```console +$ nemoclaw inference set --provider nvidia-prod --model nvidia/nemotron-3-super-120b-a12b +``` + +### OpenAI + +```console +$ nemoclaw inference set --provider openai-api --model gpt-5.4 +``` + +### Anthropic + +```console +$ nemoclaw inference set --provider anthropic-prod --model claude-sonnet-4-6 +``` + +### Google Gemini + +```console +$ nemoclaw inference set --provider gemini-api --model gemini-2.5-flash +``` + +### Compatible Endpoints + +If you onboarded a custom compatible endpoint, switch models with the provider created for that endpoint: + +```console +$ nemoclaw inference set --provider compatible-endpoint --model +``` + +```console +$ nemoclaw inference set --provider compatible-anthropic-endpoint --model +``` + +### Hermes Provider + +For a NemoClaw-managed Hermes sandbox, use the Hermes alias with the registered Hermes Provider route: + +```console +$ nemohermes inference set --provider hermes-provider --model openai/gpt-5.4-mini +``` + +#### Switching from Responses API to Chat Completions + +If onboarding selected `/v1/responses` but the agent fails at runtime (for +example, because the backend does not emit the streaming events OpenClaw +requires), re-run onboarding so the wizard re-probes the endpoint and bakes +the correct API path into the image: + +```console +$ nemoclaw onboard +``` + +Select the same provider and endpoint again. +The updated streaming probe will detect incomplete `/v1/responses` support +and select `/v1/chat/completions` automatically. + +For the compatible-endpoint provider, NemoClaw uses `/v1/chat/completions` by +default, so no env var is required to keep the safe path. +To opt in to `/v1/responses` for a backend you have verified end to end, set +`NEMOCLAW_PREFERRED_API` before onboarding: + +```console +$ NEMOCLAW_PREFERRED_API=openai-responses nemoclaw onboard +``` + +**Note:** + +`NEMOCLAW_INFERENCE_API_OVERRIDE` patches the config at container startup but +does not update the Dockerfile ARG baked into the image. +If you recreate the sandbox without the override env var, the image reverts to +the original API path. +A fresh `nemoclaw onboard` is the reliable fix because it updates both the +session and the baked image. + +## Cross-Provider Switching + +Switching to a different provider family (for example, from NVIDIA Endpoints to Anthropic) also uses `nemoclaw inference set`. +The command updates both the gateway route and the OpenClaw provider namespace in the running sandbox config. + +```console +$ nemoclaw inference set --provider anthropic-prod --model claude-sonnet-4-6 --no-verify +``` + +Use `--no-verify` only when OpenShell cannot verify the provider at switch time but you have already confirmed the provider and credential. + +## Tune Model Metadata + +The sandbox image bakes model metadata (context window, max output tokens, reasoning mode, and accepted input modalities) into `openclaw.json` at build time. +To change these values, set the corresponding environment variables before running `nemoclaw onboard` so they patch into the Dockerfile before the image builds. + +| Variable | Values | Default | +|---|---|---| +| `NEMOCLAW_CONTEXT_WINDOW` | Positive integer (tokens) | `131072` | +| `NEMOCLAW_MAX_TOKENS` | Positive integer (tokens) | `4096` | +| `NEMOCLAW_REASONING` | `true` or `false` | `false` | +| `NEMOCLAW_INFERENCE_INPUTS` | `text` or `text,image` | `text` | +| `NEMOCLAW_AGENT_TIMEOUT` | Positive integer (seconds) | `600` | +| `NEMOCLAW_AGENT_HEARTBEAT_EVERY` | Go-style duration (`30m`, `1h`, `0m` to disable) | `unset` (OpenClaw default) | + +Invalid values are ignored, and the default bakes into the image. +For Local Ollama, onboarding loads the selected model first and uses Ollama's reported runtime context length when `NEMOCLAW_CONTEXT_WINDOW` is unset. +Use `NEMOCLAW_INFERENCE_INPUTS=text,image` only for a model that accepts image input through the selected provider. + +```console +$ export NEMOCLAW_CONTEXT_WINDOW=65536 +$ export NEMOCLAW_MAX_TOKENS=8192 +$ export NEMOCLAW_REASONING=true +$ export NEMOCLAW_INFERENCE_INPUTS=text,image +$ export NEMOCLAW_AGENT_TIMEOUT=1800 +$ export NEMOCLAW_AGENT_HEARTBEAT_EVERY=0m +$ nemoclaw onboard +``` + +`NEMOCLAW_AGENT_TIMEOUT` controls the per-request inference timeout baked into +`agents.defaults.timeoutSeconds`. Increase it for slow local inference (for +example, CPU-only Ollama or vLLM on modest hardware). NemoClaw writes this +value into `openclaw.json` during onboarding. The default sandbox may keep that +file writable for agent state, but direct in-sandbox edits are not the supported +or durable way to change NemoClaw-managed defaults. Rebuild the sandbox via +`nemoclaw onboard` to apply a new value. + +`NEMOCLAW_AGENT_HEARTBEAT_EVERY` sets `agents.defaults.heartbeat.every`. +This controls OpenClaw's periodic main-session agent turn. +Each interval, the agent wakes up to review follow-ups and read `HEARTBEAT.md` if present in the workspace. +The OpenClaw default is 30 minutes (1 hour for Anthropic OAuth / Claude CLI reuse). +Tune the cadence with a duration string like `5m` or `2h`, or set `0m` to disable the periodic turns entirely. +Disabling also drops `HEARTBEAT.md` from normal-run bootstrap context per upstream behavior, so the model no longer sees heartbeat-only instructions. +NemoClaw writes this value into `openclaw.json` during onboarding. +The in-sandbox `openclaw config set` command is not the supported path for +NemoClaw-managed build-time defaults, and direct file edits are overwritten by a +rebuild. Rebuild the sandbox via `nemoclaw onboard --resume` to apply a new value. + +These variables are build-time settings. +If you change them on an existing sandbox, recreate the sandbox so the new values bake into the image: + +```console +$ nemoclaw onboard --resume --recreate-sandbox +``` + +## Verify the Active Model + +Use `nemoclaw inference get` to print the provider and model the gateway is currently routing to. +Run it before `nemoclaw inference set` to confirm the starting state, or after a switch to verify the new route. + +```console +$ nemoclaw inference get +Provider: nvidia-prod +Model: nvidia/nemotron-3-super-120b-a12b +``` + +Pass `--json` for machine-readable output. + +```console +$ nemoclaw inference get --json +{ + "provider": "nvidia-prod", + "model": "nvidia/nemotron-3-super-120b-a12b" +} +``` + +The command exits non-zero with `OpenShell inference route is not configured.` when the gateway has no registered inference route. +Run `nemoclaw onboard` to configure one. + +Run the status command when you also need sandbox, service, and messaging health: + +```console +$ nemoclaw status +``` + +The status output includes the active provider, model, and endpoint with the rest of the sandbox state. + +## Notes + +- The host keeps provider credentials. +- The sandbox continues to use `inference.local`. +- `nemoclaw inference set` patches the selected running OpenClaw or Hermes sandbox config and recomputes its config hash. +- Use `nemoclaw onboard --resume --recreate-sandbox` for build-time settings such as context window, max tokens, reasoning mode, heartbeat cadence, or image contents. +- Local Ollama and local vLLM routes use local provider tokens rather than `OPENAI_API_KEY`. Rebuilds of older local-inference sandboxes clear the stale OpenAI credential requirement automatically. + +## Related Topics + +- [Inference Options](inference-options.md) for the full list of providers available during onboarding. diff --git a/skills/nemoclaw-user-configure-inference/references/tool-calling-reliability.md b/skills/nemoclaw-user-configure-inference/references/tool-calling-reliability.md new file mode 100644 index 00000000000..01f2c361152 --- /dev/null +++ b/skills/nemoclaw-user-configure-inference/references/tool-calling-reliability.md @@ -0,0 +1,164 @@ + + +# Tool-Calling Reliability for Local Inference + +Local inference is useful for privacy, cost control, and offline development, but +tool-calling agents place stricter demands on the model server than simple chat. +The model server must return structured `tool_calls`, not a JSON-looking string +inside normal assistant text. + +Use this page when the TUI shows raw JSON such as: + +```json +{"arguments":{"query":"robotics"},"name":"memory_search"} +``` + +If that appears as text in the assistant reply, OpenClaw cannot dispatch the +tool because the inference response did not include a structured tool call. + +## Quick Choice Guide + +| Workload | Ollama is usually sufficient | Prefer vLLM with a parser | +|---|---|---| +| Plain chat | Yes | Optional | +| Embeddings-only or retrieval setup | Yes | Optional | +| One simple tool with short prompts | Often | Optional | +| Agent loops with several tools | Risky | Yes | +| Long system prompts or sender metadata | Risky | Yes | +| Multi-turn tool dispatch | Risky | Yes | + +Ollama can work well for lightweight local chat and some simple tool surfaces. +For OpenClaw-style agent loops with multiple tools, long instructions, or +multi-turn dispatch, use a server that exposes OpenAI-compatible +`/v1/chat/completions` with a tool-call parser. vLLM is the common local choice. + +## Symptom + +The common failure mode is: + +- The model emits text that looks like a tool call. +- The response does not include a structured `tool_calls` field. +- The gateway treats the response as normal text. +- No tool runs, and the user sees raw JSON in the TUI. + +This is different from a network or policy block. `nemoclaw status`, +`nemoclaw logs`, and `nemoclaw debug --quick` can all look healthy while +tool dispatch still fails inside the conversation. + +## Recommended Fix + +For persistent NemoClaw use, start vLLM with auto tool choice and the parser that +matches your model family, then rerun onboarding and select **Local vLLM +[experimental]** or **Other OpenAI-compatible endpoint**. + +For Hermes 3 style models, a known-good vLLM command shape is: + +```console +$ vllm serve /models/Hermes-3-Llama-3.1-8B \ + --served-model-name hermes-3-llama-3.1-8b \ + --enable-auto-tool-choice \ + --tool-call-parser hermes \ + --port 8000 +``` + +For a Docker Compose setup: + +```yaml +services: + vllm-nemoclaw: + image: vllm/vllm-openai:latest + container_name: vllm-nemoclaw + restart: unless-stopped + ports: + - "8002:8000" + volumes: + - /path/to/models:/models:ro + - /path/to/hf-cache:/root/.cache/huggingface + ipc: host + deploy: + resources: + reservations: + devices: + - capabilities: [gpu] + count: all + command: > + --model /models/Hermes-3-Llama-3.1-8B + --served-model-name hermes-3-llama-3.1-8b + --enable-auto-tool-choice + --tool-call-parser hermes + --gpu-memory-utilization 0.20 + --max-model-len 32768 + --api-key ${VLLM_API_KEY} +``` + +Then onboard against that endpoint: + +```console +$ NEMOCLAW_PROVIDER=custom \ + NEMOCLAW_ENDPOINT_URL=http://localhost:8002/v1 \ + NEMOCLAW_MODEL=hermes-3-llama-3.1-8b \ + COMPATIBLE_API_KEY=$VLLM_API_KEY \ + nemoclaw onboard --non-interactive +``` + +If the endpoint does not require authentication, set `COMPATIBLE_API_KEY` to any +non-empty placeholder, such as `dummy`. + +## Advanced Temporary Repointing + +NemoClaw-managed sandboxes normally block direct `openclaw config set` writes +inside the sandbox because those edits do not survive rebuilds. Prefer rerunning +`nemoclaw onboard` for a persistent provider change. + +If you are intentionally testing a mutable OpenClaw config, prepare a batch file +like this: + +```json +{ + "models": { + "providers": { + "vllm-local": { + "baseUrl": "http://host.openshell.internal:8002/v1", + "api": "openai", + "apiKey": "${VLLM_API_KEY}" + } + } + }, + "agents": { + "defaults": { + "model": { + "primary": "vllm-local/hermes-3-llama-3.1-8b" + } + } + } +} +``` + +Apply it only in environments where OpenClaw config writes are allowed: + +```console +$ openclaw config set --batch-file /sandbox/.openclaw/vllm-tool-calls.json +``` + +After testing, persist the working provider through `nemoclaw onboard` so the +sandbox image, OpenShell inference route, and host-managed credentials stay in +sync. + +## Verify the Fix + +After switching to vLLM, ask for an action that should use a tool. Good signs: + +- The TUI does not show JSON blobs as assistant text. +- The gateway log shows tool dispatch and a follow-up answer. +- `nemoclaw status` reports the local vLLM or compatible endpoint as the + active provider. + +If JSON still appears as text, confirm that vLLM was started with both +`--enable-auto-tool-choice` and the correct `--tool-call-parser` value for your +model. + +## Next Steps + +- [Use a Local Inference Server](../SKILL.md) +- [Inference Options](inference-options.md) +- [Switch Inference Models](switch-inference-providers.md) diff --git a/skills/nemoclaw-user-configure-inference/references/use-local-inference-details.md b/skills/nemoclaw-user-configure-inference/references/use-local-inference-details.md new file mode 100644 index 00000000000..fab5e58f2be --- /dev/null +++ b/skills/nemoclaw-user-configure-inference/references/use-local-inference-details.md @@ -0,0 +1,151 @@ + + +# Use a Local Inference Server: Details + +## Non-Interactive Setup + +```console +$ NEMOCLAW_PROVIDER=ollama \ + NEMOCLAW_MODEL=qwen2.5:14b \ + nemoclaw onboard --non-interactive --yes +``` + +If `NEMOCLAW_MODEL` is not set, NemoClaw selects a default model based on available memory. +If `NEMOCLAW_MODEL` names a known bootstrap model (for example `qwen3.6:35b`) that does not fit the host's currently available GPU memory, NemoClaw warns and falls back to the largest known model that does fit. +Unknown or custom tags (any value the bootstrap registry has not seen) are still passed through; the Ollama runner validates the choice itself. + +`--yes` (or `NEMOCLAW_YES=1`) authorises the Ollama model download without an interactive confirmation prompt. +Under `--non-interactive`, `--yes` (or `NEMOCLAW_YES=1`) is required to authorise the download โ€” onboard exits otherwise, since it cannot prompt. +Run onboard without `--non-interactive` to get the interactive `[y/N]` prompt that shows the model size before downloading. + +| Variable | Purpose | +|---|---| +| `NEMOCLAW_PROVIDER` | Set to `ollama`. | +| `NEMOCLAW_MODEL` | Ollama model tag to use. Optional. | +| `NEMOCLAW_YES` | Set to `1` to auto-accept the model-download confirmation prompt. Optional. | + +### Selecting the API Path + +For the compatible-endpoint provider, `/v1/chat/completions` is the default. +NemoClaw tests streaming events during onboarding and uses chat completions +without probing the Responses API. + +To opt in to `/v1/responses`, set `NEMOCLAW_PREFERRED_API` before running onboard: + +```console +$ NEMOCLAW_PREFERRED_API=openai-responses nemoclaw onboard +``` + +The wizard then probes `/v1/responses` and only selects it when streaming +support is complete. +If the probe fails, the wizard falls back to `/v1/chat/completions` +automatically. +You can use this variable in both interactive and non-interactive mode. + +| Variable | Values | Default | +|---|---|---| +| `NEMOCLAW_PREFERRED_API` | `openai-completions`, `openai-responses` | `openai-completions` for compatible endpoints | + +If you already onboarded and the sandbox is failing at runtime, re-run +`nemoclaw onboard` to re-probe the endpoint and bake the correct API path +into the image. +Refer to [Switch Inference Models](switch-inference-providers.md) for details. + +## Anthropic-Compatible Server + +If your local server implements the Anthropic Messages API (`/v1/messages`), choose **Other Anthropic-compatible endpoint** during onboarding instead. + +```console +$ nemoclaw onboard +``` + +For non-interactive setup, use `NEMOCLAW_PROVIDER=anthropicCompatible` and set `COMPATIBLE_ANTHROPIC_API_KEY`. + +```console +$ NEMOCLAW_PROVIDER=anthropicCompatible \ + NEMOCLAW_ENDPOINT_URL=http://localhost:8080 \ + NEMOCLAW_MODEL=my-model \ + COMPATIBLE_ANTHROPIC_API_KEY=dummy \ + nemoclaw onboard --non-interactive +``` + +### Override the Managed-vLLM Model + +Managed vLLM serves the profile default unless you select a different registry entry. +Export `NEMOCLAW_VLLM_MODEL=` before invoking the installer to choose a different model from the registry. +NemoClaw uses the matching `vllm serve` flags, including the reasoning parser, tool-call parser, and `--max-model-len`. +Recognised slugs: + +| Slug | Hugging Face model | Notes | +|---|---|---| +| `qwen3.6-27b` | `Qwen/Qwen3.6-27B-FP8` | Default on DGX Spark and DGX Station profiles | +| `nemotron-3-nano-4b` | `nvidia/NVIDIA-Nemotron-3-Nano-4B-FP8` | Default on the generic Linux + NVIDIA GPU profile | +| `deepseek-r1-distill-70b` | `deepseek-ai/DeepSeek-R1-Distill-Llama-70B` | Gated. Requires Hugging Face license acceptance | + +The slug is case-insensitive; the full Hugging Face id is also accepted. +An unrecognised value fails fast with a list of valid slugs. + +Gated models require a Hugging Face token; export it before onboarding so NemoClaw can forward it into the managed vLLM container: + +```console +$ export HF_TOKEN= +$ NEMOCLAW_PROVIDER=install-vllm \ + NEMOCLAW_VLLM_MODEL=deepseek-r1-distill-70b \ + nemoclaw onboard --non-interactive +``` + +`HUGGING_FACE_HUB_TOKEN` is accepted as an alternative. +The token check runs on the host before any docker pull, so a missing or empty token aborts onboarding before bandwidth is spent on a 401. + +## Timeout Configuration + +Local inference requests use a default timeout of 180 seconds. +Large prompts on hardware such as DGX Spark can exceed shorter timeouts, so NemoClaw sets a higher default for Ollama, vLLM, NIM, and compatible-endpoint setup. + +To override the timeout, set the `NEMOCLAW_LOCAL_INFERENCE_TIMEOUT` environment variable before onboarding: + +```console +$ export NEMOCLAW_LOCAL_INFERENCE_TIMEOUT=300 +$ nemoclaw onboard +``` + +The value is in seconds. +This setting is baked into the sandbox at build time. +Changing it after onboarding requires re-running `nemoclaw onboard`. + +`NEMOCLAW_LOCAL_INFERENCE_TIMEOUT` only governs the inference-server validation probe. +The post-create readiness wait (image build, gateway upload, in-sandbox boot) has its own budget, `NEMOCLAW_SANDBOX_READY_TIMEOUT`, also defaulting to 180 seconds. +On hosts where the sandbox image takes minutes to build or upload โ€” large quantised models, DGX Station first runs, or remote VMs over a slow link โ€” raise both together: + +```console +$ export NEMOCLAW_LOCAL_INFERENCE_TIMEOUT=300 +$ export NEMOCLAW_SANDBOX_READY_TIMEOUT=600 +$ nemoclaw onboard +``` + +If onboard ends with `Sandbox '' was created but did not become ready within 180s`, refer to Troubleshooting (use the `nemoclaw-user-reference` skill). + +## Verify the Configuration + +After onboarding completes, confirm the active provider and model. + +```console +$ nemoclaw status +``` + +The output shows the provider label (for example, "Local vLLM" or "Other OpenAI-compatible endpoint") and the active model. +For Local Ollama, status also checks the authenticated proxy when a proxy token is available. +If `Inference` is healthy but `Inference (auth proxy)` is not, rerun onboarding to repair the proxy path that sandbox requests use. + +## Switch Models at Runtime + +You can change the model without re-running onboard. +Refer to [Switch Inference Models](switch-inference-providers.md) for the full procedure. + +For compatible endpoints, the command is: + +```console +$ nemoclaw inference set --provider compatible-endpoint --model +``` + +If the provider itself needs to change (for example, switching from vLLM to a cloud API), pass the new provider to `nemoclaw inference set`. diff --git a/skills/nemoclaw-user-configure-inference/skill-card.md b/skills/nemoclaw-user-configure-inference/skill-card.md new file mode 100644 index 00000000000..4a29ef3787c --- /dev/null +++ b/skills/nemoclaw-user-configure-inference/skill-card.md @@ -0,0 +1,52 @@ +## Description:
+Connects NemoClaw to a local inference server such as Ollama, vLLM, TensorRT-LLM, NIM, or any OpenAI-compatible endpoint.
+ +This skill is ready for commercial/non-commercial use.
+ +## Owner +NVIDIA
+ +### License/Terms of Use:
+Apache 2.0
+## Use Case:
+Developers and engineers configuring NemoClaw to route inference to a local model server for running AI agents inside OpenShell sandboxes.
+ +### Deployment Geography for Use:
+Global
+ +## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
+ +## Reference(s):
+- [Inference Options](references/inference-options.md)
+- [Set Up Sub-Agent](references/set-up-sub-agent.md)
+- [Switch Inference Providers](references/switch-inference-providers.md)
+- [Tool-Calling Reliability](references/tool-calling-reliability.md)
+- [Use Local Inference Details](references/use-local-inference-details.md)
+ + +## Skill Output:
+**Output Type(s):** [Configuration instructions, Shell commands]
+**Output Format:** [Markdown with inline bash code blocks]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [None]
+ +## Evaluation Metrics Used:
+Reported benchmark dimensions:
+- Security: Checks whether skill-assisted execution avoids unsafe behavior such as secret leakage, destructive commands, or unauthorized access.
+- Correctness: Checks whether the agent follows the expected workflow and produces the correct final output.
+- Discoverability: Checks whether the agent loads the skill when relevant and avoids using it when irrelevant.
+- Effectiveness: Checks whether the agent performs measurably better with the skill than without it.
+- Efficiency: Checks whether the agent uses fewer tokens and avoids redundant work.
+ + + +## Skill Version(s):
+0.1.0 (source: package.json)
+ +## Ethical Considerations:
+NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
+ +(For Release on NVIDIA Platforms Only)
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/nemoclaw-user-configure-inference/skill.oms.sig b/skills/nemoclaw-user-configure-inference/skill.oms.sig new file mode 100644 index 00000000000..f8b66a3f560 --- /dev/null +++ b/skills/nemoclaw-user-configure-inference/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAibmVtb2NsYXctdXNlci1jb25maWd1cmUtaW5mZXJlbmNlIiwKICAgICAgImRpZ2VzdCI6IHsKICAgICAgICAic2hhMjU2IjogIjU0ZWZkYWQxMDY3MzZhZWVmZWE5YmNmZmNjMjIwZDYwNGM5OWE3Yjc4ODlmYmFhNzgxYWI1ZTIxMDg2ODNlMDgiCiAgICAgIH0KICAgIH0KICBdLAogICJwcmVkaWNhdGVUeXBlIjogImh0dHBzOi8vbW9kZWxfc2lnbmluZy9zaWduYXR1cmUvdjEuMCIsCiAgInByZWRpY2F0ZSI6IHsKICAgICJzZXJpYWxpemF0aW9uIjogewogICAgICAiYWxsb3dfc3ltbGlua3MiOiBmYWxzZSwKICAgICAgIm1ldGhvZCI6ICJmaWxlcyIsCiAgICAgICJpZ25vcmVfcGF0aHMiOiBbCiAgICAgICAgIi5naXQiLAogICAgICAgICIuZ2l0YXR0cmlidXRlcyIsCiAgICAgICAgIi5naXRpZ25vcmUiLAogICAgICAgICIuZ2l0aHViIgogICAgICBdLAogICAgICAiaGFzaF90eXBlIjogInNoYTI1NiIKICAgIH0sCiAgICAicmVzb3VyY2VzIjogWwogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMDMxNTJlNGYyY2IxYzRmNjMxMTNkOTNjMWFhMTM1MTE2NGU2ODUzNzNkNzMyNjIzMjdiZmVmNTMyOWIyOTc1YiIsCiAgICAgICAgIm5hbWUiOiAiQkVOQ0hNQVJLLm1kIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiYmFkNTU3OThkNzBmZjMyODJiOTY4NGUzMzY1YTE0MGE1ZGRlYzRhYTdlNTRlMzZjMGRkNWRhMDRjZDQ3MjQ4MiIsCiAgICAgICAgIm5hbWUiOiAiU0tJTEwubWQiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICI3ZGI2ZjU2MzllNWQzYTAwMTg3NWYyOTI1Nzg0YTgzZmJjMWU2NGM0NWI3ZTI5MWE3NmU1ZDBhYTM4ZTBhMDgzIiwKICAgICAgICAibmFtZSI6ICJldmFscy9ldmFscy5qc29uIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiYzA5YmJmYjhiNjIzODI3ZGNhNWFlNzljOTEyNjUyODUxMzg4YzhhYmQzNDdiMWQxZDgxNzBkMmY2M2QwZGZhYyIsCiAgICAgICAgIm5hbWUiOiAicmVmZXJlbmNlcy9pbmZlcmVuY2Utb3B0aW9ucy5tZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogImUyYzI2NWI4Y2ZjNjliY2QxMGUxNWU1YTUzYjQ3MThmNzJmZWZiODFhMGFlMWQ1YTg3NDgyY2MyZmEwYTRiOWQiLAogICAgICAgICJuYW1lIjogInJlZmVyZW5jZXMvc2V0LXVwLXN1Yi1hZ2VudC5tZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogImY1OWFmM2IwZTc3Njk5OGY1YjZjNTU2NmViY2FjOTgzMzZhYjAwNzE2NjZmZTVkM2U3NTlmZDMzM2QwOGQ1YTgiLAogICAgICAgICJuYW1lIjogInJlZmVyZW5jZXMvc3dpdGNoLWluZmVyZW5jZS1wcm92aWRlcnMubWQiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICJjMWI1NzVjY2RhMjYzZDFjZjliYWY3NDFmZDg1YTUyMmRkNWFlMzEwMzdkY2MwOWIwZWM3NGQ5N2I0MDA2Yzk1IiwKICAgICAgICAibmFtZSI6ICJyZWZlcmVuY2VzL3Rvb2wtY2FsbGluZy1yZWxpYWJpbGl0eS5tZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogImI2MDM1MjAyNzA0MzJmZGM4MDJhZTQwZTY0Y2M0NjdkZjY3NDJmYzk4Nzk4ZGZjYmEwNGM5ODc5ZDUzMzk3N2MiLAogICAgICAgICJuYW1lIjogInJlZmVyZW5jZXMvdXNlLWxvY2FsLWluZmVyZW5jZS1kZXRhaWxzLm1kIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiZmIyOGNkYzRjYTQ4N2E5YWI1M2IxY2FhYTExODJmMmRhNzA2MTRiOGJjMWY3MWZlNTFkOTcxZDg2YWNmMDBlMSIsCiAgICAgICAgIm5hbWUiOiAic2tpbGwtY2FyZC5tZCIKICAgICAgfQogICAgXQogIH0KfQ==","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGYCMQCQ5TXuzBoc6c0Q52G4hythBwiLtiLaaKkxtDk28F8DBeqIfwWvZpcWtMzmZPXSGpwCMQDo5gZZjom06H6mzgiUqM2TsInOk/KUBwrd+Ui4sBmboS736iMYDfWVSAHWoVp74Rk=","keyid":""}]}} \ No newline at end of file diff --git a/skills/nemoclaw-user-configure-security/BENCHMARK.md b/skills/nemoclaw-user-configure-security/BENCHMARK.md new file mode 100644 index 00000000000..18b5dbe3d78 --- /dev/null +++ b/skills/nemoclaw-user-configure-security/BENCHMARK.md @@ -0,0 +1,67 @@ +# Evaluation Report + +Evaluation of the `nemoclaw-user-configure-security` skill before publication through NVSkills-Eval. + +This benchmark summarizes 3-Tier Evaluation from NVSkills-Eval results for the skill. The goal is to document whether the skill is safe, discoverable, effective, and useful for agents before it is published for broader workflow use. + +## Evaluation Summary + +- Skill: `nemoclaw-user-configure-security` +- Evaluation date: 2026-05-28 +- NVSkills-Eval profile: `external` +- Overall verdict: FAIL +- Tier 3 live agent evaluation: not available in this report + +## Agents Used + +- Tier 3 agent details were not available in this report. + +## Metrics Used + +Reported benchmark dimensions: + +- Security: checks whether skill-assisted execution avoids unsafe behavior such as secret leakage, destructive commands, or unauthorized access. +- Correctness: checks whether the agent follows the expected workflow and produces the correct final output. +- Discoverability: checks whether the agent loads the skill when relevant and avoids using it when irrelevant. +- Effectiveness: checks whether the agent performs measurably better with the skill than without it. +- Efficiency: checks whether the agent uses fewer tokens and avoids redundant work. + +Underlying evaluation signals used in this run: + +- No Tier 3 evaluation signal details were available in this report. + +## Test Tasks + +Tier 3 evaluation task details were not available in this report. + +## Results + +Tier 3 dimension rollup was not available in this report. + +## Tier 1: Static Validation Summary + +Tier 1 validation passed with observations. NVSkills-Eval ran 9 checks and found 15 total findings. + +Top findings: + +- MEDIUM QUALITY/quality_correctness: Guide-only skill has very little content (12 lines) (`skills/nemoclaw-user-configure-security/SKILL.md`) +- MEDIUM QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.author' (`skills/nemoclaw-user-configure-security/SKILL.md`) +- MEDIUM QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/nemoclaw-user-configure-security/SKILL.md`) +- MEDIUM QUALITY/quality_efficiency: Deeply nested references in credential-storage.md (`skills/nemoclaw-user-configure-security/SKILL.md`) +- MEDIUM SCHEMA/body_recommended_section: Missing recommended section: '## Instructions' (`skills/nemoclaw-user-configure-security/SKILL.md`) + +## Tier 2: Deduplication Summary + +Tier 2 validation reported findings. NVSkills-Eval ran 2 checks and found 1 total findings. + +Top findings: + +- HIGH DUPLICATE/duplicate: Duplicate content found across SKILL.md and references/best-practices.md and references/credential-storage.md and references/openclaw-controls.md: + "(preamble)" in SKILL.md (lines 1-3) + vs "(preamble)" in references/best-practices.md (lines 1-2) + vs "(preamble)" in references/credential-storage.md (lines 1-2) + vs "(preamble)" in references/openclaw-controls.md (lines 1-2) (`SKILL.md:1`) + +## Publication Recommendation + +The skill should be reviewed before NVSkills-Eval publication. Skill owners should address the findings above and rerun NVSkills-Eval to refresh this benchmark. diff --git a/skills/nemoclaw-user-configure-security/SKILL.md b/skills/nemoclaw-user-configure-security/SKILL.md new file mode 100644 index 00000000000..36df08415f6 --- /dev/null +++ b/skills/nemoclaw-user-configure-security/SKILL.md @@ -0,0 +1,16 @@ +--- +name: "nemoclaw-user-configure-security" +description: "Presents a risk framework for every configurable security control in NemoClaw. Use when evaluating security posture, reviewing sandbox security defaults, or assessing control trade-offs. Trigger keywords - nemoclaw security best practices, sandbox security controls risk framework, nemoclaw credential storage, openshell provider, api key security, openclaw security controls, nemoclaw security boundary, prompt injection, tool access control." +license: "Apache-2.0" +--- + + + + +# NemoClaw Security Best Practices: Controls, Risks, and Posture Profiles + +## References + +- **Load [references/best-practices.md](references/best-practices.md)** when evaluating security posture, reviewing sandbox security defaults, or assessing control trade-offs. Presents a risk framework for every configurable security control in NemoClaw. +- **Load [references/openclaw-controls.md](references/openclaw-controls.md)** when reviewing the security boundary between NemoClaw and OpenClaw or assessing what NemoClaw does not cover. Lists OpenClaw security controls that operate independently of NemoClaw, including prompt injection detection, tool access control, rate limiting, environment variable policy, audit framework, supply chain scanning, messaging access policy, context visibility, and safe regex. +- **Load [references/credential-storage.md](references/credential-storage.md)** when reviewing how credentials are handled, locating a stored credential, or assessing the storage threat model. Covers where NemoClaw stores provider credentials, why nothing is persisted to host disk, and how the OpenShell gateway acts as the single system of record. diff --git a/skills/nemoclaw-user-configure-security/evals/evals.json b/skills/nemoclaw-user-configure-security/evals/evals.json new file mode 100644 index 00000000000..9e17d649833 --- /dev/null +++ b/skills/nemoclaw-user-configure-security/evals/evals.json @@ -0,0 +1,56 @@ +[ + { + "id": "docs-security-best-practices-001", + "question": "I'm evaluating NemoClaw security best practices. Help me understand the risk posture of each configurable control so I can justify the setup to my team or security reviewers.", + "expected_skill": "nemoclaw-user-configure-security", + "ground_truth": "A NemoClaw-specific answer that helps the user understand the risk posture of each configurable control and gives enough concrete guidance, decision criteria, verification steps, or risk framing to justify the setup to my team or security reviewers." + }, + { + "id": "docs-security-best-practices-002", + "question": "I'm balancing developer convenience with lockdown. Help me compare the trade-offs of changing security controls so I can choose a posture that fits the environment.", + "expected_skill": "nemoclaw-user-configure-security", + "ground_truth": "A NemoClaw-specific answer that helps the user compare the trade-offs of changing security controls and gives enough concrete guidance, decision criteria, verification steps, or risk framing to choose a posture that fits the environment." + }, + { + "id": "docs-security-best-practices-003", + "question": "I'm preparing for production-like use. Help me see which defaults are acceptable and which require changes so I can avoid shipping with accidental weak spots.", + "expected_skill": "nemoclaw-user-configure-security", + "ground_truth": "A NemoClaw-specific answer that helps the user see which defaults are acceptable and which require changes and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid shipping with accidental weak spots." + }, + { + "id": "docs-security-credential-storage-001", + "question": "I'm inspecting NemoClaw credential storage. Help me verify how secrets are stored and protected so I can decide whether the setup meets my secret-handling expectations.", + "expected_skill": "nemoclaw-user-configure-security", + "ground_truth": "A NemoClaw-specific answer that helps the user verify how secrets are stored and protected and gives enough concrete guidance, decision criteria, verification steps, or risk framing to decide whether the setup meets my secret-handling expectations." + }, + { + "id": "docs-security-credential-storage-002", + "question": "I'm tracing where credentials live. Help me distinguish host, gateway, and sandbox storage boundaries so I can avoid assuming secrets are available in the wrong place.", + "expected_skill": "nemoclaw-user-configure-security", + "ground_truth": "A NemoClaw-specific answer that helps the user distinguish host, gateway, and sandbox storage boundaries and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid assuming secrets are available in the wrong place." + }, + { + "id": "docs-security-credential-storage-003", + "question": "I'm rotating or inspecting credentials. Help me follow a workflow that does not print secrets in logs or docs so I can recover or update access safely.", + "expected_skill": "nemoclaw-user-configure-security", + "ground_truth": "A NemoClaw-specific answer that helps the user follow a workflow that does not print secrets in logs or docs and gives enough concrete guidance, decision criteria, verification steps, or risk framing to recover or update access safely." + }, + { + "id": "docs-security-openclaw-controls-001", + "question": "I'm reading about controls outside NemoClaw's scope. Help me understand which security responsibilities remain with OpenClaw so I can avoid treating sandbox isolation as a complete application security model.", + "expected_skill": "nemoclaw-user-configure-security", + "ground_truth": "A NemoClaw-specific answer that helps the user understand which security responsibilities remain with OpenClaw and gives enough concrete guidance, decision criteria, verification steps, or risk framing to avoid treating sandbox isolation as a complete application security model." + }, + { + "id": "docs-security-openclaw-controls-002", + "question": "I'm assessing application-layer agent risk. Help me identify the controls NemoClaw does not add so I can plan separate mitigations for authentication, prompt handling, and agent behavior.", + "expected_skill": "nemoclaw-user-configure-security", + "ground_truth": "A NemoClaw-specific answer that helps the user identify the controls NemoClaw does not add and gives enough concrete guidance, decision criteria, verification steps, or risk framing to plan separate mitigations for authentication, prompt handling, and agent behavior." + }, + { + "id": "docs-security-openclaw-controls-003", + "question": "I'm documenting the security boundary. Help me explain where NemoClaw protection ends so I can set accurate expectations for reviewers and operators.", + "expected_skill": "nemoclaw-user-configure-security", + "ground_truth": "A NemoClaw-specific answer that helps the user explain where NemoClaw protection ends and gives enough concrete guidance, decision criteria, verification steps, or risk framing to set accurate expectations for reviewers and operators." + } +] diff --git a/skills/nemoclaw-user-configure-security/references/best-practices.md b/skills/nemoclaw-user-configure-security/references/best-practices.md new file mode 100644 index 00000000000..59e3ceee9fa --- /dev/null +++ b/skills/nemoclaw-user-configure-security/references/best-practices.md @@ -0,0 +1,511 @@ + + +# NemoClaw Security Best Practices: Controls, Risks, and Posture Profiles + +NemoClaw ships with deny-by-default security controls across four layers: network, filesystem, process, and inference. +You can tune every control, but each change shifts the risk profile. +This page documents every configurable knob, its default, what it protects, the concrete risk of relaxing it, and a recommendation for common use cases. + +For background on how the layers fit together, refer to How It Works (use the `nemoclaw-user-overview` skill). + +## Protection Layers at a Glance + +NemoClaw enforces security at four layers. +NemoClaw locks some when it creates the sandbox and requires a restart to change them. +You can hot-reload others while the sandbox runs. + +The following diagram shows the default posture immediately after `nemoclaw onboard`, before you approve any endpoints or apply any presets. + +```mermaid +flowchart TB + subgraph HOST["Your Machine: default posture after nemoclaw onboard"] + direction TB + + YOU["๐Ÿ‘ค Operator"] + + subgraph NC["NemoClaw + OpenShell"] + direction TB + + subgraph SB["Sandbox: the agent's isolated world"] + direction LR + PROC["โš™๏ธ Process Layer
Controls what the agent can execute"] + FS["๐Ÿ“ Filesystem Layer
Controls what the agent can read and write"] + AGENT["๐Ÿค– Agent"] + end + + subgraph GW["Gateway: the gatekeeper"] + direction LR + NET["๐ŸŒ Network Layer
Controls where the agent can connect"] + INF["๐Ÿง  Inference Layer
Controls which AI models the agent can use"] + end + end + end + + OUTSIDE["๐ŸŒ Outside World
Internet ยท AI Providers ยท APIs"] + + AGENT -- "all requests" --> GW + GW -- "approved only" --> OUTSIDE + YOU -. "approve / deny" .-> GW + + classDef agent fill:#76b900,stroke:#5a8f00,color:#fff,stroke-width:2px,font-weight:bold + classDef locked fill:#1a1a1a,stroke:#76b900,color:#fff,stroke-width:2px + classDef hot fill:#333,stroke:#76b900,color:#e6f2cc,stroke-width:2px + classDef external fill:#f5f5f5,stroke:#ccc,color:#1a1a1a,stroke-width:1px + classDef operator fill:#fff,stroke:#76b900,color:#1a1a1a,stroke-width:2px,font-weight:bold + + class AGENT agent + class PROC,FS locked + class NET,INF hot + class OUTSIDE external + class YOU operator + + style HOST fill:none,stroke:#76b900,stroke-width:2px,color:#1a1a1a + style NC fill:none,stroke:#76b900,stroke-width:1px,stroke-dasharray:5 5,color:#1a1a1a + style SB fill:#f5faed,stroke:#76b900,stroke-width:2px,color:#1a1a1a + style GW fill:#2a2a2a,stroke:#76b900,stroke-width:2px,color:#fff +``` + +| Layer | What it protects | Enforcement point | Changeable at runtime | +| --- | --- | --- | --- | +| Network | Unauthorized outbound connections and data exfiltration. | OpenShell gateway | Yes. Use `openshell policy set` or operator approval. | +| Filesystem | System binary tampering, credential theft, config manipulation. | Landlock LSM + container mounts | Landlock layout: no. Requires sandbox re-creation. Use host-side NemoClaw commands for durable config changes. | +| Process | Privilege escalation, fork bombs, syscall abuse. | Container runtime (Docker/K8s `securityContext`) | No. Requires sandbox re-creation. | +| Inference | Credential exposure, unauthorized model access, cost overruns. | OpenShell gateway | Yes. Use `nemoclaw inference set`. | + +## Network Controls + +NemoClaw controls which hosts, ports, and HTTP methods the sandbox can reach, and lets operators approve or deny requests in real time. + +### Deny-by-Default Egress + +The sandbox blocks all outbound connections unless you explicitly list the endpoint in the policy file `nemoclaw-blueprint/policies/openclaw-sandbox.yaml`. + +| Aspect | Detail | +|---|---| +| Default | All egress denied. Only endpoints in the baseline policy can receive traffic. | +| What you can change | Add endpoints to the policy file (static) or with `openshell policy set` (dynamic). | +| Risk if relaxed | Each allowed endpoint is a potential data exfiltration path. The agent can send workspace content, credentials, or conversation history to any reachable host. | +| Recommendation | Add only endpoints the agent needs for its task. Prefer operator approval for one-off requests over permanently widening the baseline. | + +### Binary-Scoped Endpoint Rules + +Each network policy entry restricts which executables can reach the endpoint using the `binaries` field. + +OpenShell identifies the calling binary by reading `/proc//exe` (the kernel-trusted executable path, not `argv[0]`), walking the process tree for ancestor binaries, and computing a SHA256 hash of each binary on first use. +If someone replaces a binary while the sandbox runs, the hash mismatch triggers an immediate deny. + +| Aspect | Detail | +|---|---| +| Default | Each endpoint restricts access to specific binaries. For example, the `github` preset restricts access so only `/usr/bin/git` can reach `github.com`. Binary paths support glob patterns (`*` matches one path component, `**` matches recursively). | +| What you can change | Add binaries to an endpoint entry, or omit the `binaries` field to allow any executable. | +| Risk if relaxed | Removing binary restrictions lets any process in the sandbox reach the endpoint. An agent could use `curl`, `wget`, or a Python script to exfiltrate data to an allowed host, bypassing the intended usage pattern. | +| Recommendation | Always scope endpoints to the binaries that need them. If the agent needs a host from a new binary, add that binary explicitly rather than removing the restriction. | + +### Path-Scoped HTTP Rules + +Endpoint rules restrict allowed HTTP methods and URL paths. + +| Aspect | Detail | +|---|---| +| Default | Some endpoints allow GET and POST on `/**` (for example, `clawhub.ai`). Others restrict methods and paths to specific API routes (for example, `integrate.api.nvidia.com` allows POST only to inference and embedding paths and GET to model listings). Read-only endpoints such as `docs.openclaw.ai`, the `npm_registry` baseline entry, and the `pypi` preset allow GET only (PyPI also allows HEAD). The `npm` preset is an intentional exception: npm/Yarn registry traffic uses L4 pass-through for Node 22 undici CONNECT compatibility. | +| What you can change | Add methods (PUT, DELETE, PATCH) or restrict paths to specific prefixes. | +| Risk if relaxed | Allowing all methods on an API endpoint gives the agent write and delete access. For example, allowing DELETE on `api.github.com` lets the agent delete repositories. | +| Recommendation | Use GET-only rules for endpoints that the agent only reads. Add write methods only for endpoints where the agent must create or modify resources. Restrict paths to specific API routes when possible. | + +### L4-Only vs L7 Inspection (`protocol` Field) + +All sandbox egress goes through OpenShell's CONNECT proxy. +The `protocol` field on an endpoint controls whether the proxy also inspects individual HTTP requests inside the tunnel. + +| Aspect | Detail | +|---|---| +| Default | Endpoints without a `protocol` field use L4-only enforcement: the proxy checks host, port, and binary identity, then relays the TCP stream without inspecting payloads. Setting `protocol: rest` enables L7 inspection: the proxy auto-detects and terminates TLS, then evaluates each HTTP request's method and path against the endpoint's `rules` or `access` preset. | +| What you can change | Add `protocol: rest` to an endpoint to enable per-request HTTP inspection. Use the `access` preset (`full`, `read-only`, `read-write`) or explicit `rules` to control allowed methods and paths. | +| Risk if relaxed | L4-only endpoints (no `protocol` field) allow the agent to send any data through the tunnel after the initial connection is permitted. The proxy cannot see or filter the HTTP method, path, or body. The `access: full` preset with `protocol: rest` enables inspection but allows all methods and paths, so it does not restrict what the agent can do at the HTTP level. | +| Recommendation | Use `protocol: rest` with specific `rules` for REST APIs where you want method and path control. Use `protocol: rest` with `access: read-only` for read-only endpoints. Omit `protocol` only for non-HTTP protocols (WebSocket, gRPC streaming), endpoints that do not need HTTP inspection, or documented compatibility exceptions that require a client-managed CONNECT tunnel. | + +### Operator Approval Flow + +When the agent reaches an unlisted endpoint, OpenShell blocks the request and prompts the operator in the TUI. + +| Aspect | Detail | +|---|---| +| Default | Enabled. The gateway blocks all unlisted endpoints and requires approval. | +| What you can change | The system merges approved endpoints into the sandbox's policy as a new durable revision. They persist across sandbox restarts within the same sandbox instance. However, when you destroy and recreate the sandbox (for example, by running `nemoclaw onboard`), the policy resets to the baseline defined in the blueprint. | +| Risk if relaxed | Approving an endpoint permanently widens the running sandbox's policy. If you approve a broad domain (such as a CDN that hosts arbitrary content), the agent can fetch anything from that domain until you destroy and recreate the sandbox. | +| Recommendation | Review each blocked request before approving. If you find yourself approving the same endpoint repeatedly, add it to the baseline policy with appropriate binary and path restrictions. To reset approved endpoints, destroy and recreate the sandbox. | + +### Policy Presets + +NemoClaw ships preset policy files in `nemoclaw-blueprint/policies/presets/` for common integrations. + +| Preset | What it enables | Key risk | +|---|---|---| +| `brave` | Brave Search API. | Agent can issue search queries. | +| `brew` | Homebrew (Linuxbrew) package manager. The sandbox base image includes the `brew` binary; this preset opens network egress to GitHub and the Homebrew formulae index so `brew install` can fetch bottles. | Allows installing arbitrary Homebrew packages, which may contain malicious code. | +| `discord` | Discord REST API, WebSocket gateway, CDN. | CDN endpoint (`cdn.discordapp.com`) allows GET to any path. WebSocket uses `access: full` (no inspection). | +| `github` | GitHub and GitHub REST API. | Gives agent read/write access to repositories and issues via `git`. | +| `huggingface` | Hugging Face Hub (download-only) and inference router. | Allows downloading arbitrary models and datasets. POST is restricted to the inference router only. | +| `jira` | Atlassian Jira API. | Gives agent read/write access to project issues and comments. | +| `local-inference` | Local Ollama and vLLM through the host gateway. | Allows sandbox access to host-side local inference ports covered by the preset. | +| `npm` | npm and Yarn registries via L4 pass-through. | Allows installing arbitrary npm packages, which may contain malicious code. OpenShell still gates by host, port, and binary, but does not inspect HTTP method, path, or body for this preset. | +| `outlook` | Microsoft 365, Outlook. | Gives agent access to email. | +| `pypi` | Python Package Index (GET and HEAD only). | Allows installing arbitrary Python packages, which may contain malicious code. Publishing is blocked. | +| `slack` | Slack API, Socket Mode, webhooks. | WebSocket uses `access: full`. Agent can post to any channel the bot token has access to. | +| `telegram` | Telegram Bot API. | Agent can send messages to any chat the bot token has access to. | + +**Recommendation:** Apply presets only when the agent's task requires the integration. Review the preset's YAML file before applying to understand the endpoints, methods, and binary restrictions it adds. + +## Filesystem Controls + +NemoClaw restricts which paths the agent can read and write, protecting system binaries, configuration files, and gateway credentials. + +### Read-Only System Paths + +The container mounts system directories read-only to prevent the agent from modifying binaries, libraries, or configuration files. + +| Aspect | Detail | +|---|---| +| Default | `/usr`, `/lib`, `/proc`, `/dev/urandom`, `/app`, `/etc`, `/var/log` are read-only. | +| What you can change | Add or remove paths in the `filesystem_policy.read_only` section of the policy file. | +| Risk if relaxed | Making `/usr` or `/lib` writable lets the agent replace system binaries (such as `curl` or `node`) with trojanized versions. Making `/etc` writable lets the agent modify DNS resolution, TLS trust stores, or user accounts. | +| Recommendation | Never make system paths writable. If the agent needs a writable location for generated files, use a subdirectory of `/sandbox`. | + +### Agent Config Directory + +The `/sandbox/.openclaw` directory contains the OpenClaw gateway configuration (model routing, CORS settings, channel config). +The current entrypoint reads the gateway auth token from OpenClaw config when present, exports it as `OPENCLAW_GATEWAY_TOKEN`, and writes it to `/tmp/nemoclaw-proxy-env.sh` so interactive sandbox sessions can reach the gateway through system-wide shell hooks. +In root mode, the gateway process still runs as the separate `gateway` user, but the token is intentionally available to sandbox shells for local gateway access. + +Writable agent state such as plugins, skills, hooks, and workspace metadata lives directly under `/sandbox/.openclaw`. + +By default, this directory starts writable so the agent can manage its own config, install skills, and write to standard home-directory paths natively. +For sensitive workloads, use a reviewed host-side immutability workflow after initial setup so config and writable state entry points cannot be changed by the sandbox user. + +- **DAC permissions (default).** The sandbox user owns `/sandbox/.openclaw` with mode `2770` (setgid `sandbox:sandbox`) and `openclaw.json` with mode `660`, so the agent and its group can read and write config directly. A reviewed host-side immutability workflow should compare the intended ownership and mode with the live sandbox filesystem before treating the config tree as locked. +- **Config integrity hash.** The image includes a SHA256 hash of `openclaw.json`. In the default mutable state, `.config-hash` is sandbox-owned and is not a tamper-proof trust anchor, so startup does not fail closed on that hash. When the hash is root-owned and read-only, startup enforces it and refuses to start if the hash does not match. +- **Gateway token environment.** The gateway exports `OPENCLAW_GATEWAY_TOKEN` and writes it to `/tmp/nemoclaw-proxy-env.sh` for interactive sandbox sessions. Keep this in mind when deciding whether a workload should run with mutable config or an immutable config posture. + +| Aspect | Detail | +|---|---| +| Default | The sandbox keeps `/sandbox/.openclaw` writable (`2770 sandbox:sandbox`), sets `openclaw.json` to `660 sandbox:sandbox`, lets the agent manage state directly, and has the gateway place `OPENCLAW_GATEWAY_TOKEN` in `/tmp/nemoclaw-proxy-env.sh` for interactive shells. | +| What you can change | Apply a reviewed host-side immutability workflow to lock config and state directories with DAC permissions and the immutable flag where available. | +| Risk of default | A writable `.openclaw` directory lets the agent modify its own gateway config: disabling CORS or redirecting inference to an attacker-controlled endpoint. | +| Recommendation | For always-on assistants handling sensitive workloads, lock config after initial setup. For development workflows, the writable default is appropriate. | + +### Writable Paths + +The agent has read-write access to `/sandbox`, `/tmp`, and `/dev/null`. + +| Aspect | Detail | +|---|---| +| Default | `/sandbox` (agent workspace), `/tmp` (temporary files), `/dev/null`. | +| What you can change | Add additional writable paths in `filesystem_policy.read_write`. | +| Risk if relaxed | Each additional writable path expands the agent's ability to persist data and potentially modify system behavior. Adding `/var` lets the agent write to log directories. Adding `/home` gives access to other user directories. | +| Recommendation | Keep writable paths to `/sandbox` and `/tmp`. If the agent needs a persistent working directory, create a subdirectory under `/sandbox`. | + +### Landlock LSM Enforcement + +Landlock is a Linux Security Module that enforces filesystem access rules at the kernel level. + +| Aspect | Detail | +|---|---| +| Default | `compatibility: best_effort`. The entrypoint applies Landlock rules when the kernel supports them and silently skips them on older kernels. | +| What you can change | This is a NemoClaw default, not a user-facing knob. | +| Risk if relaxed | On kernels without Landlock support (pre-5.13), filesystem restrictions rely solely on container mount configuration, which is less granular. | +| Recommendation | Run on a kernel that supports Landlock (5.13+). Ubuntu 22.04 LTS and later include Landlock support. | + +## Process Controls + +NemoClaw limits the capabilities, user privileges, and resource quotas available to processes inside the sandbox. + +### Capability Drops + +The entrypoint drops dangerous Linux capabilities from the bounding set at startup using `capsh`. +This limits what capabilities any child process (gateway, sandbox, agent) can ever acquire. +When the entrypoint switches from root to the `sandbox` and `gateway` users, it uses `setpriv` when available to remove the remaining privilege-separation capabilities from the child process at the same time as the user change. + +The initial entrypoint drop removes `cap_sys_admin`, `cap_sys_ptrace`, `cap_net_raw`, `cap_dac_override`, `cap_sys_chroot`, `cap_fsetid`, `cap_setfcap`, `cap_mknod`, `cap_audit_write`, and `cap_net_bind_service`. +During `setpriv` step-down, the child process also loses `cap_setuid`, `cap_setgid`, `cap_fowner`, `cap_chown`, and `cap_kill`. + +This is best-effort: if `capsh` is not available or `CAP_SETPCAP` is not in the bounding set, the entrypoint logs a warning and continues with the default capability set. +If `setpriv` is unavailable, the entrypoint falls back to `gosu` and logs a warning that the remaining bounding-set capabilities were retained for the child process. +For additional protection, pass `--cap-drop=ALL` with `docker run` or Compose (see Sandbox Hardening (use the `nemoclaw-user-deploy-remote` skill)). + +| Aspect | Detail | +|---|---| +| Default | The entrypoint drops dangerous capabilities at startup using `capsh`, then uses `setpriv` during user step-down when possible. Best-effort. | +| What you can change | When launching with `docker run` directly, pass `--cap-drop=ALL --cap-add=NET_BIND_SERVICE` for stricter enforcement. In the standard NemoClaw flow (with `nemoclaw onboard`), the entrypoint handles capability dropping automatically. | +| Risk if relaxed | `CAP_SYS_ADMIN` and `CAP_SYS_PTRACE` expand kernel and process attack surface. `CAP_NET_RAW` allows raw socket access for network sniffing. `CAP_DAC_OVERRIDE` bypasses filesystem permission checks. If `capsh` or `setpriv` cannot run, the container retains more of the runtime-provided capability set. | +| Recommendation | Run on an image that includes `capsh` and `setpriv` (the NemoClaw image includes them). For defense-in-depth, also pass `--cap-drop=ALL` at the container runtime level. | + +### Gateway Process Isolation + +The OpenClaw gateway runs as a separate `gateway` user, not as the `sandbox` user that runs the agent. + +| Aspect | Detail | +|---|---| +| Default | The entrypoint starts the gateway process using `gosu gateway`, isolating it from the agent's `sandbox` user. | +| What you can change | This is not a user-facing knob. The entrypoint enforces it when running as root. In non-root mode (when OpenShell sets `no-new-privileges`), gateway process isolation does not work because `gosu` cannot change users. | +| Risk if relaxed | If the gateway and agent run as the same user, the agent can kill the gateway process and restart it with a tampered configuration (the "fake-HOME" attack). | +| Recommendation | No action needed. The entrypoint handles this automatically. Be aware that non-root mode disables this isolation. | + +### No New Privileges + +The `no-new-privileges` flag prevents processes from gaining additional privileges through setuid binaries or capability inheritance. + +| Aspect | Detail | +|---|---| +| Default | OpenShell sets `PR_SET_NO_NEW_PRIVS` using `prctl()` inside the sandbox process as part of the seccomp filter setup. The NemoClaw Compose example also shows the equivalent `security_opt: no-new-privileges:true` setting. | +| What you can change | OpenShell's seccomp path enforces this inside the sandbox. It is not a user-facing knob. | +| Risk if relaxed | Without this flag, a compromised process could execute a setuid binary to escalate to root inside the container, then attempt container escape techniques. | +| Recommendation | No action needed. OpenShell enforces this automatically when the sandbox network policy is active. This flag prevents `gosu` from switching users, so non-root mode disables gateway process isolation in the NemoClaw entrypoint. | + +### Process Limit + +A process limit caps the number of processes the sandbox user can spawn. +The entrypoint sets both soft and hard limits using `ulimit -u 512`. +This is best-effort: if the container runtime restricts `ulimit` modification, the entrypoint logs a security warning and continues without the limit. + +| Aspect | Detail | +|---|---| +| Default | 512 processes (`ulimit -u 512`), best-effort. | +| What you can change | Increase or decrease the limit with `--ulimit nproc=N:N` in `docker run` or the `ulimits` section in Compose. The runtime-level ulimit takes precedence over the entrypoint's setting. | +| Risk if relaxed | Removing or raising the limit makes the sandbox vulnerable to fork-bomb attacks, where a runaway process spawns children until the host runs out of resources. If the entrypoint cannot set the limit (logs `[SECURITY] Could not set soft/hard nproc limit`), the container runs without process limits. | +| Recommendation | Keep the default at 512. If the agent runs workloads that spawn many child processes (such as parallel test runners), increase to 1024 and monitor host resource usage. If the entrypoint logs a warning about ulimit restrictions, set the limit through the container runtime instead. | + +### Non-Root User + +The sandbox runs agent processes as a dedicated `sandbox` user and group. +The entrypoint starts as root for privilege separation, then drops to the `sandbox` user for all agent commands. + +| Aspect | Detail | +|---|---| +| Default | `run_as_user: sandbox`, `run_as_group: sandbox`. A separate `gateway` user runs the gateway process. | +| What you can change | Change the `process` section in the policy file to run as a different user. | +| Risk if relaxed | Running as `root` inside the container gives the agent access to modify any file in the container filesystem and increases the impact of container escape vulnerabilities. | +| Recommendation | Never run as root. Keep the `sandbox` user. | + +### PATH Hardening + +The entrypoint locks the `PATH` environment variable to system directories, preventing the agent from injecting malicious binaries into command resolution. + +| Aspect | Detail | +|---|---| +| Default | The entrypoint sets `PATH` to `/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin` at startup. | +| What you can change | This is not a user-facing knob. The entrypoint enforces it. | +| Risk if relaxed | Without PATH hardening, the agent could create an executable named `curl` or `git` in a writable directory earlier in the PATH, intercepting commands run by the entrypoint or other processes. | +| Recommendation | No action needed. The entrypoint handles this automatically. | + +### Build Toolchain Removal + +The Dockerfile removes compilers and network probes from the runtime image. + +| Aspect | Detail | +|---|---| +| Default | The Dockerfile purges `gcc`, `gcc-12`, `g++`, `g++-12`, `cpp`, `cpp-12`, `make`, `netcat-openbsd`, `netcat-traditional`, and `ncat` from the sandbox image. | +| What you can change | Modify the Dockerfile to keep these tools, or install them at runtime if package manager access is allowed. | +| Risk if relaxed | A compiler lets the agent build arbitrary native code, including kernel exploits or custom network tools. `netcat` enables arbitrary TCP connections that bypass HTTP-level policy enforcement. | +| Recommendation | Keep build tools removed. If the agent needs to compile code, run the build in a separate, purpose-built container and copy artifacts into the sandbox. | + +### Image Digest Pinning + +The blueprint references the sandbox image by an immutable `@sha256:` digest instead of a mutable tag such as `:latest`. +A registry compromise or accidental force-push cannot silently swap the sandbox image. + +| Aspect | Detail | +|---|---| +| Default | `nemoclaw-blueprint/blueprint.yaml` pins the sandbox image by digest. A CI regression test blocks any mutable-tag reference from merging. | +| What you can change | Contributors bumping the sandbox image must update the digest in `blueprint.yaml`. Release tooling should rewrite the digest automatically. | +| Risk if relaxed | Reverting to a mutable tag (`:latest`) allows a registry-side change to replace the sandbox image without any blueprint update, which is a supply-chain risk. | +| Recommendation | Always reference the sandbox image by digest. If you build a custom image with `nemoclaw onboard --from`, the digest constraint does not apply to your local build. | + +### Auth Profile Permissions + +The entrypoint and migration flows enforce `chmod 600` on all `auth-profiles.json` files under `~/.openclaw`. +This prevents other users on the host from reading stored credentials. + +| Aspect | Detail | +|---|---| +| Default | `600` permissions applied recursively at startup and after migration restores. | +| What you can change | This is not a user-facing knob. The entrypoint enforces it. | +| Risk if relaxed | Looser permissions let other users or processes on the host read provider API keys and tokens stored in auth profiles. | +| Recommendation | No action needed. If you see a `permission denied` error when reading auth profiles, verify that you are running as the same user who created them. | + +## Gateway Authentication Controls + +The OpenClaw gateway authenticates devices that connect to the Control UI dashboard. +NemoClaw hardens these defaults at image build time. + +### Device Authentication + +Device authentication requires each connecting device to go through a pairing flow before it can interact with the gateway. + +| Aspect | Detail | +|---|---| +| Default | Enabled. The gateway requires device pairing for all connections. | +| What you can change | Set `NEMOCLAW_DISABLE_DEVICE_AUTH=1` as a Docker build argument to disable device authentication. This is a build-time setting baked into `openclaw.json` and verified by hash at startup. | +| Risk if relaxed | Disabling device auth allows any device on the network to connect to the gateway without proving identity. This is dangerous when combined with LAN-bind changes or cloudflared tunnels in remote deployments, resulting in an unauthenticated, publicly reachable dashboard. | +| Recommendation | Keep device auth enabled (the default). Only disable it for headless or development environments where no untrusted devices can reach the gateway. | + +### Gateway Bind Address + +NemoClaw binds the OpenShell gateway to loopback by default. + +| Aspect | Detail | +|---|---| +| Default | `NEMOCLAW_GATEWAY_BIND_ADDRESS=127.0.0.1`. | +| What you can change | Set `NEMOCLAW_GATEWAY_BIND_ADDRESS=0.0.0.0` before onboarding to listen on all IPv4 interfaces. | +| Risk if relaxed | Other hosts on the network may be able to reach the OpenShell gateway. | +| Recommendation | Keep the loopback default unless the gateway must be reachable from another host. | + +### Insecure Auth Derivation + +The `allowInsecureAuth` setting controls whether the gateway permits non-HTTPS authentication. + +| Aspect | Detail | +|---|---| +| Default | Derived from the `CHAT_UI_URL` scheme at build time. When the URL uses `http://` (local development), insecure auth is allowed. When it uses `https://` (remote or production), insecure auth is blocked. | +| What you can change | This is derived automatically from `CHAT_UI_URL`. Set `CHAT_UI_URL` to an `https://` URL to enforce secure auth. | +| Risk if relaxed | Allowing insecure auth over HTTPS defeats the purpose of TLS, because authentication tokens transit in cleartext. | +| Recommendation | Use `https://` for any deployment accessible beyond `localhost`. The default local URL (`http://127.0.0.1:18789`) correctly allows insecure auth for local development. | + +### Auto-Pair Client Allowlist + +The auto-pair watcher automatically approves device pairing requests from recognized clients, so you do not need to manually approve the Control UI. + +| Aspect | Detail | +|---|---| +| Default | The watcher approves devices with `clientId` set to `openclaw-control-ui` or `clientMode` set to `webchat`. All other clients are rejected and logged. | +| What you can change | This is not a user-facing knob. The allowlist is defined in the entrypoint script. | +| Risk if relaxed | Approving all device types without validation lets rogue or unexpected clients pair with the gateway unchallenged. | +| Recommendation | No action needed. The entrypoint handles this automatically. If you see `[auto-pair] rejected unknown client=...` in the logs, investigate the source of the unexpected connection. | + +### CLI Secret Redaction + +The CLI automatically redacts secret patterns (API keys, bearer tokens, provider credentials) from command output and error messages before logging them. + +| Aspect | Detail | +|---|---| +| Default | Enabled. The runner redacts secrets from stdout, stderr, and thrown error messages. | +| What you can change | This is not a user-facing knob. The CLI enforces it on all command output paths. | +| Risk if relaxed | Without redaction, secrets could appear in terminal scrollback, log files, or debug output shared in bug reports. | +| Recommendation | No action needed. If you share `nemoclaw debug` output, verify that no secrets appear in the collected diagnostics. | + +### Memory Secret Scanner + +The NemoClaw plugin blocks the agent from writing likely secrets (API keys, tokens, private keys) into persistent memory files. +The scanner intercepts Write, Edit, and similar tool calls targeting memory and workspace paths before they reach disk. + +| Aspect | Detail | +|---|---| +| Default | Enabled. The plugin registers a `before_tool_call` hook that scans for 14 high-confidence secret patterns. | +| What it covers | Examples include `.openclaw/memory/`, `.openclaw/workspace/`, `.openclaw/agents/`, `.openclaw/skills/`, `.openclaw/hooks/`, `.openclaw/credentials/`, `.openclaw/openclaw.json`, `.nemoclaw/`, and `MEMORY.md`; the exact coverage is defined by `MEMORY_PATH_SEGMENTS` and enforced through `isMemoryPath()`. | +| What you can change | This is not a user-facing knob. The plugin enforces it automatically. | +| Risk if relaxed | Without scanning, the agent could persist API keys or tokens in memory files that survive across sessions and backups. | +| Recommendation | No action needed. If a write is blocked, the agent receives an actionable error listing the detected patterns. | + +## Inference Controls + +OpenShell routes all inference traffic through the gateway to isolate provider credentials from the sandbox. + +### Routed Inference through `inference.local` + +The OpenShell gateway intercepts all inference requests from the agent and routes them to the configured provider. +The agent never receives the provider API key. + +| Aspect | Detail | +|---|---| +| Default | The agent talks to `inference.local`. The host owns the credential and upstream endpoint. | +| What you can change | You cannot configure this architecture. The system always enforces it. | +| Risk if bypassed | If the agent could reach an inference endpoint directly (by adding it to the network policy), it would need an API key. Since the sandbox does not contain credentials, this acts as defense-in-depth. However, adding an inference provider's host to the network policy without going through OpenShell routing could let the agent use a stolen or hardcoded key. | +| Recommendation | Do not add inference provider hosts (such as `api.openai.com` or `api.anthropic.com`) to the network policy. Use OpenShell inference routing instead. | + +### Provider Trust Tiers + +Different inference providers have different trust and cost profiles. + +| Provider | Trust level | Cost risk | Data handling | +|---|---|---|---| +| NVIDIA Endpoints | High. Hosted on `build.nvidia.com`. | Pay-per-token with an API key. Unattended agents can accumulate cost. | NVIDIA infrastructure processes requests. | +| OpenAI | High. Commercial API. | Pay-per-token. Same cost risk as NVIDIA Endpoints. | Subject to OpenAI data policies. | +| Anthropic | High. Commercial API. | Pay-per-token. Same cost risk as NVIDIA Endpoints. | Subject to Anthropic data policies. | +| Google Gemini | High. Commercial API. | Pay-per-token. Same cost risk as NVIDIA Endpoints. | Subject to Google data policies. | +| Local Ollama | Self-hosted. No data leaves the machine. | No per-token cost. GPU/CPU resource cost. | Data stays local. | +| Custom compatible endpoint | Varies. Depends on the proxy or gateway. | Varies. | Depends on the endpoint operator. | + +**Recommendation:** For sensitive workloads, use local Ollama to keep data on-premise. For general use, NVIDIA Endpoints provide a good balance of capability and trust. Review the data policies of any cloud provider you use. + +### Experimental Providers + +The `NEMOCLAW_EXPERIMENTAL=1` environment variable gates local NVIDIA NIM and generic Linux managed vLLM install/start. DGX Spark and DGX Station managed vLLM entries are offered by default, and an already-running vLLM server on `localhost:8000` is offered in the menu without a flag, because selecting either is an explicit user action. + +| Aspect | Detail | +|---|---| +| Default | Local NVIDIA NIM and generic Linux managed vLLM install/start are hidden. DGX Spark and DGX Station managed vLLM entries, plus already-running vLLM on `localhost:8000`, are offered when detected. | +| What you can change | Set `NEMOCLAW_EXPERIMENTAL=1` before running `nemoclaw onboard` to surface Local NIM and generic Linux managed vLLM. To request only the managed vLLM path non-interactively, set `NEMOCLAW_PROVIDER=install-vllm`. | +| Risk if selected | NemoClaw has not fully validated these providers. NIM requires a NIM-capable GPU. The managed vLLM path pulls a container image and starts it on a supported NVIDIA GPU host. Misconfiguration can cause failed inference or unexpected behavior. | +| Recommendation | Use experimental providers only for evaluation. Do not rely on them for always-on assistants. | + +## Posture Profiles + +The following profiles describe how to configure NemoClaw for different use cases. +These are not separate policy files. +They provide guidance on which controls to keep tight or relax. + +### Locked-Down (Default) + +Use for always-on assistants with minimal external access. + +- Keep all defaults. Do not add presets. +- Use operator approval for any endpoint the agent requests. +- Use NVIDIA Endpoints or local Ollama for inference. +- Monitor the TUI for unexpected network requests. + +### Development + +Use when the agent needs package registries, Docker Hub, or broader GitHub access during development tasks. + +- Apply the `pypi` and `npm` presets for package installation. +- Keep binary restrictions on all presets. +- Review the agent's network activity periodically with `openshell term`. +- Use operator approval for any endpoint not covered by a preset. + +### Integration Testing + +Use when the agent talks to internal APIs or third-party services during testing. + +- Add custom endpoint entries with tight path and method restrictions. +- Use `protocol: rest` for all HTTP APIs to maintain inspection. +- Use operator approval for unknown endpoints during test runs. +- Review and clean up the baseline policy after testing. Remove endpoints that are no longer needed. + +## Common Mistakes + +The following patterns weaken security without providing meaningful benefit. + +| Mistake | Why it matters | What to do instead | +|---------|---------------|-------------------| +| Omitting `protocol: rest` on REST API endpoints without a compatibility reason | Endpoints without a `protocol` field use L4-only enforcement. The proxy allows the TCP stream through after checking host, port, and binary, but cannot see or filter individual HTTP requests. | Add `protocol: rest` with explicit `rules` to enable per-request method and path control on REST APIs. Use L4 pass-through only for documented cases such as npm/Yarn on Node 22, where the client requires a CONNECT tunnel that L7 inspection would break. | +| Adding endpoints to the baseline policy for one-off requests | Adding an endpoint to the baseline policy makes it permanently reachable across all sandbox instances. | Use operator approval. Approved endpoints persist within the sandbox instance but reset when you destroy and recreate the sandbox. | +| Relying solely on the entrypoint for capability drops | The entrypoint drops dangerous capabilities using `capsh`, but this is best-effort. If `capsh` is unavailable or `CAP_SETPCAP` is not in the bounding set, the container runs with the default capability set. | Pass `--cap-drop=ALL` at the container runtime level as defense-in-depth. | +| Leaving `/sandbox/.openclaw` writable on sensitive workloads | This directory contains the OpenClaw gateway configuration. A writable `.openclaw` lets the agent disable CORS, redirect inference routing, or weaken gateway protections. | Lock config for always-on assistants handling sensitive data. | +| Adding inference provider hosts to the network policy | Direct network access to an inference host bypasses credential isolation and usage tracking. | Use OpenShell inference routing instead of adding hosts like `api.openai.com` or `api.anthropic.com` to the network policy. | +| Disabling device auth for remote deployments | Without device auth, any device on the network can connect to the gateway without pairing. Combined with a cloudflared tunnel, this makes the dashboard publicly accessible and unauthenticated. | Keep `NEMOCLAW_DISABLE_DEVICE_AUTH` at its default (`0`). Only set it to `1` for local headless or development environments. | + +## Known Limitations + +| Limitation | Impact | Mitigation | +|-----------|--------|------------| +| `openclaw agent --local` bypasses gateway | Secret scanning, network policy, and inference auth are not enforced when the agent runs in local mode. | A runtime warning is emitted when `--local` is detected. Avoid `--local` for production workflows. A future OpenClaw-level hook will close this gap. | +| Direct filesystem writes bypass secret scanner | The scanner intercepts OpenClaw tool calls, not raw filesystem writes (e.g., `echo secret > file`). | Landlock restricts writable paths. The scanner is application-layer defense-in-depth, not a filesystem-level control. | +| Base64/hex-encoded secrets are not detected | Content-based regex scanning cannot detect encoded or obfuscated secrets. | Use environment variables or credential stores instead of writing secrets to files. | + +## Related Topics + +- Network Policies (use the `nemoclaw-user-reference` skill) for the full baseline policy reference. +- Customize the Network Policy (use the `nemoclaw-user-manage-policy` skill) for static and dynamic policy changes. +- Approve or Deny Network Requests (use the `nemoclaw-user-manage-policy` skill) for the operator approval flow. +- Sandbox Hardening (use the `nemoclaw-user-deploy-remote` skill) for container-level security measures. +- Inference Options (use the `nemoclaw-user-configure-inference` skill) for provider configuration details. +- How It Works (use the `nemoclaw-user-overview` skill) for the protection layer architecture. diff --git a/skills/nemoclaw-user-configure-security/references/credential-storage.md b/skills/nemoclaw-user-configure-security/references/credential-storage.md new file mode 100644 index 00000000000..b40b676a206 --- /dev/null +++ b/skills/nemoclaw-user-configure-security/references/credential-storage.md @@ -0,0 +1,110 @@ + + +# Credential Storage + +NemoClaw does not persist provider credentials to host disk. +The OpenShell gateway is the only system of record for stored credentials. + +When you provide a provider credential โ€” interactively during `nemoclaw onboard` or via an environment variable โ€” NemoClaw holds the value in memory only long enough to register it with the OpenShell gateway through `openshell provider create` or `openshell provider update`. +The gateway stores the credential and the OpenShell L7 proxy substitutes it into outbound requests at egress, so sandboxed agents see placeholders instead of the raw secret. + +The sandbox-side OpenClaw gateway token is generated at container startup and is not rotated through provider credential commands. + +## Where Credentials Live + +Provider credentials live in the OpenShell gateway store. +List what is registered with: + +```console +$ openshell provider list +``` + +Or, equivalently, through NemoClaw: + +```console +$ nemoclaw credentials list +``` + +Both surface the provider names that the gateway holds credentials for. The values themselves cannot be read back from the CLI; this is a deliberate property of OpenShell. + +NemoClaw still keeps non-secret operational state under `~/.nemoclaw/` (such as the sandbox registry). +That directory is created with mode `0700` and contains no credential material. + +## Environment Variables Take Precedence + +When a NemoClaw command needs a credential value during a single run (for example to forward it to an `openshell provider` registration), it reads from `process.env` first. +This means you can: + +- Prefix any command with the credential to override the gateway-stored value: `NVIDIA_API_KEY=nvapi-... nemoclaw onboard` +- Use short-lived or rotated credentials in CI by exporting them once per pipeline run +- Avoid registering credentials in the gateway entirely if your environment supplies them + +## Deploy Reads from Environment Only + +`nemoclaw deploy` (which provisions a remote Brev box) cannot read secrets back from the gateway, so it requires every credential to be present in the host environment at invocation time. +A typical deploy invocation looks like: + +```console +$ NVIDIA_API_KEY=nvapi-... \ + HF_TOKEN=hf_... \ + TELEGRAM_BOT_TOKEN=... \ + nemoclaw deploy my-instance +``` + +For remote vLLM or Hugging Face workflows that need gated model access, `nemoclaw deploy` also forwards `HF_TOKEN` and `HUGGING_FACE_HUB_TOKEN` to the VM when either variable is present. +If a required credential is missing the deploy aborts before any remote work begins. + +## GitHub Tokens + +NemoClaw never persists `GITHUB_TOKEN` itself. +When a private repo requires authentication NemoClaw runs `gh auth token`, which returns whatever the GitHub CLI has stored โ€” without caring about the storage backend. + +The GitHub CLI prefers an OS keychain when one is reachable: macOS Keychain on macOS, Windows Credential Manager on Windows, and Linux Secret Service (libsecret + a running D-Bus session) on Linux. +On hosts where no keychain is reachable (CI runners, headless launches, WSL without a session bus, macOS contexts where Keychain access is blocked, etc.) `gh auth login` falls back to a `gh`-managed file under `~/.config/gh/` with mode `0600`. +NemoClaw treats both backends identically: `gh auth token` returns the value, and NemoClaw stages it in `process.env` for the current run only. + +If `gh` is not installed or not logged in, NemoClaw prompts for a personal access token for that single run; the prompted value is held in process memory and is not written to host disk. +Run `gh auth login` if you want a persistent backing store (whichever one applies on your host) so future runs do not prompt. + +## Migration From Earlier Releases + +Earlier NemoClaw releases stored credentials as plaintext JSON in `~/.nemoclaw/credentials.json` with mode `0600`. +On first `nemoclaw onboard` after upgrading, NemoClaw automatically: + +1. Reads the legacy file. +2. Stages allowlisted credential values into `process.env` for the rest of the run. +3. Re-registers each value with the OpenShell gateway through the normal onboarding path. +4. Securely overwrites and deletes `~/.nemoclaw/credentials.json` only after every staged value has been verified as migrated to the gateway. + +You will see a one-line stderr notice the first time this happens. +Credential lookup paths such as rebuild also stage allowlisted legacy values so interrupted upgrades can keep working, but those staging-only paths do not delete the plaintext file because they cannot prove every legacy value was registered with the gateway. +If `~/.nemoclaw/credentials.json` remains after a rebuild or other credential lookup, run `nemoclaw onboard` to complete the verified gateway migration and cleanup. + +## Rotate or Remove a Stored Credential + +The simplest way to replace a stored value is to rerun onboarding with the new value in your environment: + +```console +$ NVIDIA_API_KEY=nvapi-new-value nemoclaw onboard +``` + +To remove a credential from the gateway entirely: + +```console +$ nemoclaw credentials reset +``` + +`` is the OpenShell provider name (run `nemoclaw credentials list` first if you are not sure). +On the next run NemoClaw prompts again unless the credential is supplied through the environment. + +## Security Recommendations + +1. Prefer short-lived or low-scope provider credentials where the upstream service supports them. +2. Rotate keys after suspected exposure, machine transfer, or account changes. +3. Prefer environment variables for ephemeral automation rather than registering long-lived secrets in the gateway. +4. Do not copy any host-side NemoClaw state into container images, Git repositories, bug reports, or support bundles. Even though credentials no longer live on disk, the surrounding configuration may reveal which providers you have registered. +5. Keep your home directory private and owned by your user account. + +## Related Files + +For the broader sandbox security model and operational trade-offs, see [Security Best Practices](best-practices.md) and Architecture (use the `nemoclaw-user-reference` skill). diff --git a/skills/nemoclaw-user-configure-security/references/openclaw-controls.md b/skills/nemoclaw-user-configure-security/references/openclaw-controls.md new file mode 100644 index 00000000000..2ced0c76de2 --- /dev/null +++ b/skills/nemoclaw-user-configure-security/references/openclaw-controls.md @@ -0,0 +1,121 @@ + + +# OpenClaw Security Controls Beyond NemoClaw's Scope + +NemoClaw provides infrastructure-layer security through sandbox isolation, network policy, filesystem restrictions, SSRF validation, and credential handling. +It delegates all application-layer security to OpenClaw. +This page documents areas where NemoClaw adds no independent protection beyond what OpenClaw already provides. + +The details below reflect the OpenClaw documentation at the time of writing. +Consult the [OpenClaw Security docs](https://docs.openclaw.ai/gateway/security/index) for the current state. + +## Prompt Injection Detection and Prevention + +OpenClaw detects and neutralizes prompt injection attempts before they reach the agent. + +| Control | Detail | +|---|---| +| Regex detection | Pattern matching detects common injection vectors such as "ignore all previous instructions" and `` tag spoofing | +| Boundary wrapping | Untrusted input is wrapped in randomized XML boundary markers | +| Unicode folding | Homoglyph folding normalizes bracket variants to prevent visual spoofing | +| Invisible character stripping | Zero-width invisible characters are removed from input | +| Boundary sanitization | Fake boundary markers are sanitized to prevent marker injection | +| Auto-wrapping | Web fetch and search results are automatically wrapped as untrusted external content | + +## Tool Access Control and Policy Pipeline + +OpenClaw enforces a multi-layer tool policy pipeline that gates every tool call. + +| Control | Detail | +|---|---| +| Deny list | High-risk tools (`exec`, `spawn`, `shell`, `fs_write`, `fs_delete`, and others) are blocked from Gateway HTTP by default | +| Policy pipeline | Multi-layer pipeline evaluates tool calls through profile, provider, agent, sandbox, and per-provider policies | +| Fail-closed semantics | Tool call hooks block execution on any error | +| Loop detection | Optional guard detects and blocks repeated identical tool call patterns (disabled by default, opt-in via `tools.loopDetection.enabled`) | +| Plugin approval | Approval workflow defaults to deny on timeout | + +## Authentication Rate Limiting and Flood Protection + +OpenClaw rate-limits authentication attempts and guards against connection floods. + +| Control | Detail | +|---|---| +| Auth rate limiter | Sliding-window rate limiter tracks failed authentication attempts per IP and per scope | +| Control plane limiter | Per-device write rate limiting for control plane operations | +| WebSocket flood guard | Closes connections after repeated unauthorized attempts | +| Pre-auth budget | Limits connections before authentication completes | + +## Environment Variable Security Policy + +OpenClaw blocks environment variables that could enable code injection, privilege escalation, or credential theft. + +| Category | Detail | +|---|---| +| Always-blocked keys | Keys such as `NODE_OPTIONS`, `LD_PRELOAD`, shell injection vectors, crypto mining variables, and `GIT_*` hijacking paths | +| Override-blocked keys | Additional keys blocked unless explicitly overridden | +| Blocked prefixes | Prefixes such as `GIT_CONFIG_`, `NPM_CONFIG_`, `CARGO_REGISTRIES_`, `TF_VAR_` | +| Universal blocked prefixes | `DYLD_`, `LD_`, `BASH_FUNC_` | + +## Security Audit Framework + +OpenClaw runs automated security checks (50+ distinct check types) that cover configuration, credential handling, and sandbox posture. +Run `openclaw security audit` to see all findings for your deployment. + +These checks include: + +- Synced-folder leak detection. +- Plaintext secrets in configuration files. +- Hooks hardening verification. +- Gateway no-auth detection. +- Sandbox misconfiguration scanning. +- Weak-model susceptibility assessment. +- Multi-user exposure matrix. +- Node command policy validation. +- Dangerous config flag scanning (`allowInsecureAuth`, `dangerouslyDisableDeviceAuth`, and similar flags). + +## Skill and Extension Supply Chain Scanning + +OpenClaw scans skills and extensions with a built-in static analysis scanner before installation. +Critical findings block installation by default. + +The scanner checks for patterns including: + +- Direct process execution calls. +- Dynamic code execution (`eval`, `new Function`, and similar constructs). +- Cryptocurrency mining patterns. +- Unexpected network activity. +- Potential data exfiltration (file read combined with network calls). +- Obfuscated code. +- Environment variable harvesting combined with network calls. + +## DM and Group Messaging Access Policy + +OpenClaw controls who can interact with the agent through direct messages and group channels. + +| Control | Detail | +|---|---| +| DM policy modes | 4 modes: open, disabled, pairing, allowlist | +| Group policies | Per-group access rules | +| Per-sender authorization | Individual sender gating | +| Command authorization | Command-level access control | +| Multi-user detection | Heuristic that detects multi-user scenarios | + +## Context Visibility and Output Controls + +OpenClaw restricts what supplemental context the agent can see and how it can modify outputs. + +| Control | Detail | +|---|---| +| Mode-based restrictions | Limits visibility of history, threads, quotes, and forwarded messages based on the active mode | +| Sender-based restrictions | Limits visibility based on who sent the message | +| Plugin output hooks | Plugin hooks intercept and modify tool results before they reach the user | + +## Safe Regex (ReDoS Prevention) + +OpenClaw includes safe regex compilation to prevent Regular Expression Denial of Service (ReDoS) attacks. +The implementation detects unsafe nested quantifiers, bounds input length, and caches results. + +## Next Steps + +- [Security Best Practices](best-practices.md) for NemoClaw's own security controls and risk framework. +- [Credential Storage](credential-storage.md) for how NemoClaw stores and protects provider credentials. diff --git a/skills/nemoclaw-user-configure-security/skill-card.md b/skills/nemoclaw-user-configure-security/skill-card.md new file mode 100644 index 00000000000..ce877936ad9 --- /dev/null +++ b/skills/nemoclaw-user-configure-security/skill-card.md @@ -0,0 +1,51 @@ +## Description:
+Presents a risk framework for every configurable security control in NemoClaw.
+ +This skill is ready for commercial/non-commercial use.
+ +## Owner +NVIDIA
+ +### License/Terms of Use:
+Apache 2.0
+## Use Case:
+Developers and security engineers evaluating NemoClaw security posture, reviewing sandbox security defaults, or assessing control trade-offs for their deployment.
+ +### Deployment Geography for Use:
+Global
+ +## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
+ +## Reference(s):
+- [NemoClaw Security Best Practices](references/best-practices.md)
+- [Credential Storage](references/credential-storage.md)
+- [OpenClaw Security Controls Beyond NemoClaw's Scope](references/openclaw-controls.md)
+- [OpenClaw Security Documentation](https://docs.openclaw.ai/gateway/security/index)
+ + +## Skill Output:
+**Output Type(s):** [Analysis, Configuration instructions]
+**Output Format:** [Markdown with inline bash code blocks]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [None]
+ +## Evaluation Metrics Used:
+Reported benchmark dimensions:
+- Security: Checks whether skill-assisted execution avoids unsafe behavior such as secret leakage, destructive commands, or unauthorized access.
+- Correctness: Checks whether the agent follows the expected workflow and produces the correct final output.
+- Discoverability: Checks whether the agent loads the skill when relevant and avoids using it when irrelevant.
+- Effectiveness: Checks whether the agent performs measurably better with the skill than without it.
+- Efficiency: Checks whether the agent uses fewer tokens and avoids redundant work.
+ + + +## Skill Version(s):
+0.1.0 (source: package.json)
+ +## Ethical Considerations:
+NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
+ +(For Release on NVIDIA Platforms Only)
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/nemoclaw-user-configure-security/skill.oms.sig b/skills/nemoclaw-user-configure-security/skill.oms.sig new file mode 100644 index 00000000000..cc091b46799 --- /dev/null +++ b/skills/nemoclaw-user-configure-security/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAibmVtb2NsYXctdXNlci1jb25maWd1cmUtc2VjdXJpdHkiLAogICAgICAiZGlnZXN0IjogewogICAgICAgICJzaGEyNTYiOiAiNTRhMzkwOGI0NWNmNzEzYzczNGM0ZmJjNTAxNGRkNzhhN2YzYTIyOWQxNTBhY2QxMDIzNWZlOTBjODE1OGYzMCIKICAgICAgfQogICAgfQogIF0sCiAgInByZWRpY2F0ZVR5cGUiOiAiaHR0cHM6Ly9tb2RlbF9zaWduaW5nL3NpZ25hdHVyZS92MS4wIiwKICAicHJlZGljYXRlIjogewogICAgInNlcmlhbGl6YXRpb24iOiB7CiAgICAgICJtZXRob2QiOiAiZmlsZXMiLAogICAgICAiaWdub3JlX3BhdGhzIjogWwogICAgICAgICIuZ2l0aHViIiwKICAgICAgICAiLmdpdGlnbm9yZSIsCiAgICAgICAgIi5naXRhdHRyaWJ1dGVzIiwKICAgICAgICAiLmdpdCIKICAgICAgXSwKICAgICAgImhhc2hfdHlwZSI6ICJzaGEyNTYiLAogICAgICAiYWxsb3dfc3ltbGlua3MiOiBmYWxzZQogICAgfSwKICAgICJyZXNvdXJjZXMiOiBbCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJCRU5DSE1BUksubWQiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjEzMTUyMTJjNGVkOGZlMDQ2YTY0NDJmNTcwYTI5M2Y0YmM0MDMyOTM1ZmZkMzYxZmJiYTgyYTQyMjY0Njk1ZTQiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJTS0lMTC5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMjk0MzBmYjg5OTI2MGVlOTQ5MjIyZjU3NTQyYzBjYTU0YmI5ZTQyZjg2NjZiMTczNGU1OTQ3NWMyNmFlMTkyMiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogImV2YWxzL2V2YWxzLmpzb24iLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogImZjMTgyMWNhMjc2MjIwM2RkMzE4YjNmNDE3Nzg4ZDNhNTVhYjIyMzI0Y2ZkYWI3MDZkZTBjNzYxYWJhOWEwMmIiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJyZWZlcmVuY2VzL2Jlc3QtcHJhY3RpY2VzLm1kIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICJkNjM0ODZkN2VmNjRkYjQwNWUwYmE4ZDZkYjczOTg4Mjc3MDY2MGEwMTM3OWM4YTc2NTAyYjU5MzYxOWQ3NTMzIgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAicmVmZXJlbmNlcy9jcmVkZW50aWFsLXN0b3JhZ2UubWQiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjYyMjI4YmU3YWQyM2IzYWQ3ZWNlOGJlYmVjNGY1MTRlNDhmMWUxYTJkN2MyOWY1ZTIyOGM1MjNkMDgzZjQ4NzkiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJyZWZlcmVuY2VzL29wZW5jbGF3LWNvbnRyb2xzLm1kIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICI1Y2I4MDk2MjAwYWNiZGIwZGIwMTlhYTFlZDBhNTUxOGEwYzFhNzIxYmJiMDQwZjllZDAwZmZjNDMzODUxYTk2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAic2tpbGwtY2FyZC5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiNDRjZmU3N2RmYmIwZDUzNGFiMzEwOWMyNTA5NDRkM2M5NmViNGZjNjI3ZTMyNGNkMTY0ZjRmMGVkYzY3N2ZlNCIKICAgICAgfQogICAgXQogIH0KfQ==","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGQCMF9HXyQ/ibgDs2w4UHLfGXHFevYlXp+1Q1gYcuZPzbcqDIKW66crZahe35x17J3t9gIwKxvLiD1z3BQUc1XYkx15g6aVKV9wdpaVfvq0Dcjh8cThztcFb4+12K68Zk+n6vvI","keyid":""}]}} \ No newline at end of file