]> git.djapps.eu Git - pkg/ggml/sources/llama.cpp/commitdiff
server: add initial tool isolation support (via docker) (#26507)
authorXuan-Son Nguyen <redacted>
Sat, 8 Aug 2026 14:35:53 +0000 (16:35 +0200)
committerGitHub <redacted>
Sat, 8 Aug 2026 14:35:53 +0000 (16:35 +0200)
* server: add initial tool isolation support (via docker)

* add docs

* adapt get_info

* py: fix type check

* cont

* separate tools_io_sandbox / tools_io_docker

* rename sandbox --> isolate

* x-tool-docker --> x-tool-runtime

---------

Co-authored-by: Pascal <redacted>
common/arg.cpp
common/common.h
tools/cli/README.md
tools/completion/README.md
tools/server/README-dev.md
tools/server/README.md
tools/server/server-tools.cpp
tools/server/server-tools.h
tools/server/server.cpp
tools/server/tests/unit/test_tools_builtin.py
tools/server/tests/utils.py

index da40874740ce47994041d49e08eb10bbf411e76b..4cb853c7a4b07c6a5ac3f1eb1a8103493dd36d5d 100644 (file)
@@ -3308,6 +3308,16 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
             params.server_tools = parse_csv_row(value);
         }
     ).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_TOOLS"));
+    add_opt(common_arg(
+        {"--tools-runtime"}, "OPTION",
+        "experimental: run tools in a separate runtime environment (default: none, use host environment)\n"
+        "available options:\n"
+        "  'docker:<image>': spin up a new Docker container and reuse it for all invocations, clean up on server exit\n"
+        "  'docker-container:<id>': use an existing Docker container by ID, won't stop on server exit\n",
+        [](common_params & params, const std::string & value) {
+            params.server_tools_runtime = value;
+        }
+    ).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_TOOLS_RUNTIME"));
     add_opt(common_arg(
         {"--mcp-servers-config"}, "PATH",
         "experimental: path to JSON file with MCP server definitions (Cursor-compatible format) - do not enable in untrusted environments (default: none)\n"
index 2e15ec3f815fb1f9832ffbc112a656b4aa84196c..4811345f98641d858709346208f98134d11c65f3 100644 (file)
@@ -655,6 +655,7 @@ struct common_params {
 
     // enable built-in tools
     std::vector<std::string> server_tools;
+    std::string server_tools_runtime;
 
     // MCP server configs (Cursor-compatible JSON)
     std::string mcp_servers_config;   // path to JSON file with MCP server definitions
index 4d86ce7c0136d51a5c7656b7d846ad364ea94b35..640d4fee80e824cdac1ac2ed2b4feb12d0acb094 100644 (file)
@@ -54,7 +54,6 @@
 | `-ctv, --cache-type-v TYPE` | KV cache data type for V<br/>allowed values: f32, f16, bf16, q8_0, q4_0, q4_1, iq4_nl, q5_0, q5_1<br/>(default: f16)<br/>(env: LLAMA_ARG_CACHE_TYPE_V) |
 | `-dt, --defrag-thold N` | KV cache defragmentation threshold (DEPRECATED)<br/>(env: LLAMA_ARG_DEFRAG_THOLD) |
 | `-np, --parallel N` | number of parallel sequences to decode (default: 1)<br/>(env: LLAMA_ARG_N_PARALLEL) |
-| `--rpc SERVERS` | comma-separated list of RPC servers (host:port)<br/>(env: LLAMA_ARG_RPC) |
 | `--mlock` | DEPRECATED in favor of `--load-mode`: force system to keep model in RAM rather than swapping or compressing<br/>(env: LLAMA_ARG_MLOCK) |
 | `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
 | `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
index 2abe7aaa25bb03454ba0e7d477d7d51be1674ef7..e0923ea3005e2dd025c7152c2186603b04508885 100644 (file)
@@ -137,7 +137,6 @@ llama-completion.exe -m models\gemma-1.1-7b-it.Q4_K_M.gguf --ignore-eos -n -1
 | `-ctv, --cache-type-v TYPE` | KV cache data type for V<br/>allowed values: f32, f16, bf16, q8_0, q4_0, q4_1, iq4_nl, q5_0, q5_1<br/>(default: f16)<br/>(env: LLAMA_ARG_CACHE_TYPE_V) |
 | `-dt, --defrag-thold N` | KV cache defragmentation threshold (DEPRECATED)<br/>(env: LLAMA_ARG_DEFRAG_THOLD) |
 | `-np, --parallel N` | number of parallel sequences to decode (default: 1)<br/>(env: LLAMA_ARG_N_PARALLEL) |
-| `--rpc SERVERS` | comma-separated list of RPC servers (host:port)<br/>(env: LLAMA_ARG_RPC) |
 | `--mlock` | DEPRECATED in favor of `--load-mode`: force system to keep model in RAM rather than swapping or compressing<br/>(env: LLAMA_ARG_MLOCK) |
 | `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
 | `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
index 45bcdcca76990caa7391342711390c6db715f42b..31408f4267c3a6e550110a46c63a15174c5336d9 100644 (file)
@@ -201,6 +201,7 @@ Invoke a tool call, request body is a JSON object with:
 
 Headers:
 - `x-tool-cwd`: optional; if set, use as the CWD for tool; this is not part of tool's params because it's meant to be set by the runtime, not the LLM itself
+- `x-tool-runtime`: optional; if set, run the tool inside this isolate instead of on the host. Only `docker-container:<id>` is supported for now, using an already-running container
 
 Returns JSON object. There are two response formats (MCP tools use the same two formats: their result content is concatenated into `plain_text_response`, and RPC or tool errors are surfaced as the `error` string):
 
index 4d80f059d54c3ef256b8ecd261ac7ac3b644a498..64f0b03269d3275f40e76ece18c0a37d070069ee 100644 (file)
@@ -71,7 +71,6 @@ For the full list of features, please refer to [server's changelog](https://gith
 | `-ctk, --cache-type-k TYPE` | KV cache data type for K<br/>allowed values: f32, f16, bf16, q8_0, q4_0, q4_1, iq4_nl, q5_0, q5_1<br/>(default: f16)<br/>(env: LLAMA_ARG_CACHE_TYPE_K) |
 | `-ctv, --cache-type-v TYPE` | KV cache data type for V<br/>allowed values: f32, f16, bf16, q8_0, q4_0, q4_1, iq4_nl, q5_0, q5_1<br/>(default: f16)<br/>(env: LLAMA_ARG_CACHE_TYPE_V) |
 | `-dt, --defrag-thold N` | KV cache defragmentation threshold (DEPRECATED)<br/>(env: LLAMA_ARG_DEFRAG_THOLD) |
-| `--rpc SERVERS` | comma-separated list of RPC servers (host:port)<br/>(env: LLAMA_ARG_RPC) |
 | `--mlock` | DEPRECATED in favor of `--load-mode`: force system to keep model in RAM rather than swapping or compressing<br/>(env: LLAMA_ARG_MLOCK) |
 | `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
 | `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
@@ -198,6 +197,8 @@ For the full list of features, please refer to [server's changelog](https://gith
 | `--ui-config, --webui-config JSON` | JSON that provides default UI settings (overrides UI defaults)<br/>(env: LLAMA_ARG_UI_CONFIG) |
 | `--ui-config-file, --webui-config-file PATH` | JSON file that provides default UI settings (overrides UI defaults)<br/>(env: LLAMA_ARG_UI_CONFIG_FILE) |
 | `--ui-mcp-proxy, --webui-mcp-proxy, --no-ui-mcp-proxy, --no-webui-mcp-proxy` | experimental: whether to enable MCP CORS proxy - do not enable in untrusted environments (default: disabled)<br/>(env: LLAMA_ARG_UI_MCP_PROXY) |
+| `--tools TOOL1,TOOL2,...` | experimental: whether to enable built-in tools for AI agents - do not enable in untrusted environments (default: no tools)<br/>specify "all" to enable all tools<br/>available tools: read_file, file_glob_search, grep_search, exec_shell_command, write_file, edit_file, get_datetime<br/>note: for security reasons, this will limit --cors-origins to localhost by default<br/>(env: LLAMA_ARG_TOOLS) |
+| `--tools-runtime OPTION` | experimental: run tools in a separate runtime environment (default: none, use host environment)<br/>available options:<br/>  'docker:<image>': spin up a new Docker container and reuse it for all invocations, clean up on server exit<br/>  'docker-container:<id>': use an existing Docker container by ID, won't stop on server exit<br/><br/>(env: LLAMA_ARG_TOOLS_RUNTIME) |
 | `--tools TOOL1,TOOL2,...` | experimental: whether to enable built-in tools for AI agents - do not enable in untrusted environments (default: no tools)<br/>specify "all" to enable all tools<br/>available tools: read_file, file_glob_search, grep_search, exec_shell_command, write_file, edit_file, get_datetime, get_info<br/>note: for security reasons, this will limit --cors-origins to localhost by default<br/>(env: LLAMA_ARG_TOOLS) |
 | `--mcp-servers-config PATH` | experimental: path to JSON file with MCP server definitions (Cursor-compatible format) - do not enable in untrusted environments (default: none)<br/>note: for security reasons, this will limit --cors-origins to localhost by default<br/>(env: LLAMA_ARG_MCP_SERVERS_CONFIG) |
 | `--mcp-servers-json JSON` | experimental: inline JSON with MCP server definitions (Cursor-compatible format) - do not enable in untrusted environments (default: none)<br/>note: for security reasons, this will limit --cors-origins to localhost by default<br/>(env: LLAMA_ARG_MCP_SERVERS_JSON) |
index 2fcb2a3c88a2c7182fb5e67440f37932f69d881d..eacfbf0f74ac84d4d14c242a1af54764d76d7adb 100644 (file)
 #include <ctime>
 #include <atomic>
 #include <cstring>
+#include <cstdint>
 #include <cstdlib>
 #include <algorithm>
+#include <iterator>
 #include <unordered_set>
 #include <tuple>
 #include <functional>
 #include <memory>
+#include <mutex>
 
 #if defined(_WIN32)
 #   ifndef NOMINMAX
@@ -127,6 +130,13 @@ static int entry_depth(const std::string & rel) {
     return 1 + (int) std::count(rel.begin(), rel.end(), '/');
 }
 
+// directories that a listing reports but never descends into: they can be enormous
+// lowercase only, the local walker case-folds a name before the lookup
+static const char * const SERVER_TOOL_JUNK_DIR_NAMES[] = {
+    ".git", ".svn", ".hg", "node_modules", "__pycache__",
+    ".venv", "venv", "dist", "build", "target", ".cache", ".idea", ".vscode",
+};
+
 class tools_io {
 public:
     struct exec_result {
@@ -165,6 +175,85 @@ public:
             const std::function<bool(const std::string &)> & on_chunk = nullptr) const = 0;
 };
 
+// shared subprocess execution helper, used by both the local and the docker-backed tools_io implementations.
+// combine_stderr=false when the raw stdout bytes must not be tainted by stderr, e.g. reading file contents.
+static tools_io::exec_result run_subprocess(
+        const std::vector<std::string> & args,
+        size_t max_output,
+        int timeout_secs,
+        const std::function<bool(const std::string &)> & on_chunk,
+        bool combine_stderr,
+        const std::string & cwd = "") {
+    tools_io::exec_result res;
+
+    common_subproc proc;
+
+    int options = subprocess_option_no_window
+                | subprocess_option_inherit_environment
+                | subprocess_option_search_user_path;
+    if (combine_stderr) {
+        options |= subprocess_option_combined_stdout_stderr;
+    }
+
+    if (!proc.create(args, options, {}, cwd.empty() ? nullptr : cwd.c_str())) {
+        res.output = "failed to spawn process";
+        return res;
+    }
+
+    std::atomic<bool> done{false};
+    std::atomic<bool> timed_out{false};
+
+    std::thread timeout_thread([&]() {
+        auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(timeout_secs);
+        while (!done.load()) {
+            if (std::chrono::steady_clock::now() >= deadline) {
+                timed_out.store(true);
+                proc.terminate();
+                return;
+            }
+            std::this_thread::sleep_for(std::chrono::milliseconds(100));
+        }
+    });
+
+    FILE * f = proc.stdout_file();
+    std::string output;
+    bool truncated = false;
+    if (f) {
+        char buf[4096];
+        while (fgets(buf, sizeof(buf), f) != nullptr) {
+            if (!truncated) {
+                size_t len = strlen(buf);
+                if (output.size() + len <= max_output) {
+                    output.append(buf, len);
+                    if (on_chunk && !on_chunk(console_output_to_utf8(std::string(buf, len)))) {
+                        proc.terminate();
+                        break;
+                    }
+                } else {
+                    size_t remaining = max_output - output.size();
+                    output.append(buf, remaining);
+                    if (on_chunk && remaining > 0) on_chunk(console_output_to_utf8(std::string(buf, remaining)));
+                    truncated = true;
+                }
+            }
+        }
+    }
+
+    done.store(true);
+    if (timeout_thread.joinable()) {
+        timeout_thread.join();
+    }
+
+    res.exit_code = proc.join();
+
+    res.output    = console_output_to_utf8(output);
+    res.timed_out = timed_out.load();
+    if (truncated) {
+        res.output += "\n[output truncated]";
+    }
+    return res;
+}
+
 class tools_io_basic : public tools_io {
 public:
     // cwd, if non-empty, is used to resolve relative paths and as the working directory for run()
@@ -276,72 +365,7 @@ public:
             size_t max_output,
             int timeout_secs,
             const std::function<bool(const std::string &)> & on_chunk = nullptr) const override {
-        exec_result res;
-
-        common_subproc proc;
-
-        int options = subprocess_option_no_window
-                    | subprocess_option_combined_stdout_stderr
-                    | subprocess_option_inherit_environment
-                    | subprocess_option_search_user_path;
-
-        if (!proc.create(args, options, {}, cwd.empty() ? nullptr : cwd.c_str())) {
-            res.output = "failed to spawn process";
-            return res;
-        }
-
-        std::atomic<bool> done{false};
-        std::atomic<bool> timed_out{false};
-
-        std::thread timeout_thread([&]() {
-            auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(timeout_secs);
-            while (!done.load()) {
-                if (std::chrono::steady_clock::now() >= deadline) {
-                    timed_out.store(true);
-                    proc.terminate();
-                    return;
-                }
-                std::this_thread::sleep_for(std::chrono::milliseconds(100));
-            }
-        });
-
-        FILE * f = proc.stdout_file();
-        std::string output;
-        bool truncated = false;
-        if (f) {
-            char buf[4096];
-            while (fgets(buf, sizeof(buf), f) != nullptr) {
-                if (!truncated) {
-                    size_t len = strlen(buf);
-                    if (output.size() + len <= max_output) {
-                        output.append(buf, len);
-                        if (on_chunk && !on_chunk(console_output_to_utf8(std::string(buf, len)))) {
-                            proc.terminate();
-                            break;
-                        }
-                    } else {
-                        size_t remaining = max_output - output.size();
-                        output.append(buf, remaining);
-                        if (on_chunk && remaining > 0) on_chunk(console_output_to_utf8(std::string(buf, remaining)));
-                        truncated = true;
-                    }
-                }
-            }
-        }
-
-        done.store(true);
-        if (timeout_thread.joinable()) {
-            timeout_thread.join();
-        }
-
-        res.exit_code = proc.join();
-
-        res.output    = console_output_to_utf8(output);
-        res.timed_out = timed_out.load();
-        if (truncated) {
-            res.output += "\n[output truncated]";
-        }
-        return res;
+        return run_subprocess(args, max_output, timeout_secs, on_chunk, /*combine_stderr=*/true, cwd);
     }
 
 private:
@@ -384,10 +408,8 @@ private:
     }
 
     static const std::unordered_set<std::string> & junk_dir_names() {
-        static const std::unordered_set<std::string> names = {
-            ".git", ".svn", ".hg", "node_modules", "__pycache__",
-            ".venv", "venv", "dist", "build", "target", ".cache", ".idea", ".vscode",
-        };
+        static const std::unordered_set<std::string> names(
+            std::begin(SERVER_TOOL_JUNK_DIR_NAMES), std::end(SERVER_TOOL_JUNK_DIR_NAMES));
         return names;
     }
 
@@ -450,9 +472,274 @@ private:
     }
 };
 
+// timeout for auxiliary isolate calls (stat/mkdir/ls/cp helpers); exec_shell_command uses its own
+// caller-controlled timeout instead, enforced separately in run()
+static constexpr int SERVER_TOOL_ISOLATE_EXEC_TIMEOUT = 15; // seconds
+static constexpr size_t SERVER_TOOL_ISOLATE_READ_FILE_MAX_SIZE = 64 * 1024 * 1024; // 64 MB
+
+// runs every tools_io operation as a command inside an isolate: a container, a remote host, ...
+// the isolate is created, mounted, and torn down externally by the caller
+// it must provide a POSIX environment: sh, cat, wc, mkdir, dirname, find, timeout
+class tools_io_isolate : public tools_io {
+public:
+    // cwd, if non-empty, is used to resolve relative paths and as the working directory for run()
+    explicit tools_io_isolate(std::string cwd = "") : cwd(std::move(cwd)) {}
+
+    // resolves `path` against `cwd` if `path` is relative and `cwd` is set; otherwise returns `path` unchanged.
+    // isolate paths are always POSIX-style ('/'), regardless of host OS.
+    std::string resolve(const std::string & path) const override {
+        if (cwd.empty() || (!path.empty() && path[0] == '/')) {
+            return path;
+        }
+        return cwd + "/" + path;
+    }
+
+    bool is_directory(const std::string & path) const override {
+        return shell_test("-d", resolve(path));
+    }
+
+    bool is_regular_file(const std::string & path) const override {
+        return shell_test("-f", resolve(path));
+    }
+
+    bool file_size(const std::string & path, uintmax_t & out_size) const override {
+        auto res = exec({"sh", "-c", "wc -c < \"$1\"", "_", resolve(path)}, 64, true);
+        if (res.exit_code != 0 || res.timed_out) return false;
+        try {
+            size_t pos;
+            out_size = (uintmax_t) std::stoull(res.output, &pos);
+        } catch (...) {
+            return false;
+        }
+        return true;
+    }
+
+    bool read_file(const std::string & path, std::string & out) const override {
+        // combine_stderr=false: stderr must not be spliced into raw file bytes
+        auto res = exec({"cat", "--", resolve(path)}, SERVER_TOOL_ISOLATE_READ_FILE_MAX_SIZE, false);
+        if (res.exit_code != 0 || res.timed_out) return false;
+        out = res.output;
+        return true;
+    }
+
+    bool write_file(const std::string & path, const std::string & content) const override {
+        std::string abs_path = resolve(path);
+
+        std::error_code ec;
+        fs::path tmp_dir = fs::temp_directory_path(ec);
+        if (ec) return false;
+
+        static std::atomic<uint64_t> tmp_counter{0};
+        fs::path tmp = tmp_dir / string_format(
+            "llama-tools-io-isolate-%zu-%llu.tmp",
+            std::hash<std::thread::id>{}(std::this_thread::get_id()),
+            (unsigned long long) tmp_counter.fetch_add(1));
+
+        {
+            std::ofstream f(tmp, std::ios::binary);
+            if (!f) return false;
+            f << content;
+            if (!f) return false;
+        }
+
+        bool ok = shell_run({"sh", "-c", "mkdir -p \"$(dirname \"$1\")\"", "_", abs_path});
+        if (ok) {
+            ok = upload(tmp.string(), abs_path);
+        }
+
+        std::error_code rm_ec;
+        fs::remove(tmp, rm_ec);
+        return ok;
+    }
+
+    list_result list_entries(const std::string & base, int max_depth, list_kind kind) const override {
+        list_result out;
+
+        const std::string abs_base = resolve(base);
+        if (!is_directory(base)) {
+            out.err = "path does not exist or is not a directory";
+            return out;
+        }
+
+        // git ls-files cannot list directories; use the walker when they are requested
+        if (kind == list_kind::files) {
+            auto res = exec(
+                {"sh", "-c", "cd \"$1\" && git ls-files --cached --others --exclude-standard", "_", abs_base},
+                SERVER_TOOL_GIT_LS_FILES_MAX_OUTPUT, true);
+
+            if (res.exit_code == 0 && !res.timed_out) {
+                for (const auto & rel : split_lines(res.output, /*strip_dot_slash=*/false)) {
+                    if (max_depth > 0 && entry_depth(rel) > max_depth) continue;
+                    out.entries.push_back({rel, false});
+                }
+                return out;
+            }
+        }
+
+        if (kind == list_kind::dirs || kind == list_kind::all) {
+            for (auto & rel : find_entries(abs_base, max_depth, /*dirs=*/true, out.truncated)) {
+                out.entries.push_back({std::move(rel), true});
+            }
+        }
+        if (kind == list_kind::files || kind == list_kind::all) {
+            for (auto & rel : find_entries(abs_base, max_depth, /*dirs=*/false, out.truncated)) {
+                out.entries.push_back({std::move(rel), false});
+            }
+        }
+
+        return out;
+    }
+
+    // wraps the command with an in-isolate `timeout`, since killing the host-side client
+    // does not kill the process tree running inside the isolate
+    exec_result run(
+            const std::vector<std::string> & args,
+            size_t max_output,
+            int timeout_secs,
+            const std::function<bool(const std::string &)> & on_chunk = nullptr) const override {
+        std::vector<std::string> inner = {"timeout", std::to_string(timeout_secs) + "s"};
+        inner.insert(inner.end(), args.begin(), args.end());
+        // small buffer over timeout_secs so the in-isolate `timeout` has a chance to exit cleanly
+        // before the host-side supervisory timeout forcibly kills the client
+        return run_subprocess(
+            build_argv(with_cwd(inner), /*needs_stdin=*/true),
+            max_output, timeout_secs + 5, on_chunk, true);
+    }
+
+protected:
+    // wrap `inner` (a complete POSIX argv) into the host-side argv that runs it in the isolate
+    // a transport that re-parses its args in a remote shell (ssh) must join `inner` with shell_quote_join()
+    virtual std::vector<std::string> build_argv(const std::vector<std::string> & inner, bool needs_stdin) const = 0;
+
+    // copy a host file into the isolate, `isolate_path` is absolute and its parent already exists
+    virtual bool upload(const std::string & host_path, const std::string & isolate_path) const = 0;
+
+    // quote `argv` into a single string that a POSIX shell re-parses into exactly `argv`
+    static std::string shell_quote_join(const std::vector<std::string> & argv) {
+        std::string out;
+        for (const auto & arg : argv) {
+            if (!out.empty()) out += ' ';
+            out += '\'';
+            for (const char c : arg) {
+                // a single quote cannot be escaped inside single quotes: close, escape, reopen
+                if (c == '\'') out += "'\\''";
+                else           out += c;
+            }
+            out += '\'';
+        }
+        return out;
+    }
+
+private:
+    std::string cwd;
+
+    // set the working directory in the command itself, docker's `-w` has no equivalent on every transport
+    // auxiliary calls do not need this, they use the absolute paths from resolve()
+    std::vector<std::string> with_cwd(const std::vector<std::string> & inner) const {
+        if (cwd.empty()) {
+            return inner;
+        }
+        // 127 is what a shell reports for a command it could not run
+        std::vector<std::string> out = {"sh", "-c", "cd \"$1\" || exit 127; shift; exec \"$@\"", "_", cwd};
+        out.insert(out.end(), inner.begin(), inner.end());
+        return out;
+    }
+
+    exec_result exec(const std::vector<std::string> & inner, size_t max_output, bool combine_stderr) const {
+        return run_subprocess(
+            build_argv(inner, /*needs_stdin=*/false),
+            max_output, SERVER_TOOL_ISOLATE_EXEC_TIMEOUT, nullptr, combine_stderr);
+    }
+
+    bool shell_run(const std::vector<std::string> & inner) const {
+        auto res = exec(inner, 4096, true);
+        return res.exit_code == 0 && !res.timed_out;
+    }
+
+    bool shell_test(const char * flag, const std::string & path) const {
+        return shell_run({"sh", "-c", std::string("[ ") + flag + " \"$1\" ]", "_", path});
+    }
+
+    static std::vector<std::string> split_lines(const std::string & text, bool strip_dot_slash) {
+        std::vector<std::string> result;
+        std::istringstream iss(text);
+        std::string line;
+        while (std::getline(iss, line)) {
+            if (!line.empty() && line.back() == '\r') line.pop_back();
+            if (line.empty()) continue;
+            if (strip_dot_slash && line.rfind("./", 0) == 0) line = line.substr(2);
+            std::replace(line.begin(), line.end(), '\\', '/');
+            result.push_back(line);
+        }
+        return result;
+    }
+
+    // one `find` pass in the isolate. junk directories stay selectable but are never descended into,
+    // and -mindepth/-maxdepth keep a busybox image working as well as a GNU one
+    std::vector<std::string> find_entries(const std::string & abs_base, int max_depth, bool dirs, bool & truncated) const {
+        std::string prune_expr;
+        for (const char * n : SERVER_TOOL_JUNK_DIR_NAMES) {
+            if (!prune_expr.empty()) prune_expr += " -o ";
+            prune_expr += std::string("-name ") + n;
+        }
+
+        std::string cmd = "cd \"$1\" && find . -mindepth 1";
+        if (max_depth > 0) {
+            cmd += " -maxdepth " + std::to_string(max_depth);
+        }
+        cmd += " \\( " + prune_expr + " \\) -prune";
+        cmd += dirs ? " -print -o -type d -print" : " -o -type f -print";
+
+        auto res = exec({"sh", "-c", cmd, "_", abs_base}, SERVER_TOOL_GIT_LS_FILES_MAX_OUTPUT, true);
+        truncated = truncated || res.timed_out;
+        return split_lines(res.output, /*strip_dot_slash=*/true);
+    }
+};
+
+// an already-running docker container, driven through `docker exec` and `docker cp`
+class tools_io_docker : public tools_io_isolate {
+public:
+    tools_io_docker(std::string container_id, std::string cwd = "")
+        : tools_io_isolate(std::move(cwd)), container_id(std::move(container_id)) {}
+
+protected:
+    std::vector<std::string> build_argv(const std::vector<std::string> & inner, bool needs_stdin) const override {
+        std::vector<std::string> argv = {"docker", "exec"};
+        if (needs_stdin) {
+            argv.push_back("-i");
+        }
+        argv.push_back(container_id);
+        argv.insert(argv.end(), inner.begin(), inner.end());
+        return argv;
+    }
+
+    bool upload(const std::string & host_path, const std::string & isolate_path) const override {
+        auto res = run_subprocess(
+            {"docker", "cp", host_path, container_id + ":" + isolate_path},
+            4096, SERVER_TOOL_ISOLATE_EXEC_TIMEOUT, nullptr, true);
+        return res.exit_code == 0 && !res.timed_out;
+    }
+
+private:
+    std::string container_id;
+};
+
+// runtime spec used by --tools-runtime and the x-tool-runtime header
+// this is the only scheme for now, ssh: and podman: can be added next to it
+static const std::string SERVER_TOOL_RUNTIME_DOCKER_CONTAINER = "docker-container:";
+
+// an empty runtime runs the tools on the host
 static std::unique_ptr<tools_io> make_tools_io(const json & params) {
-    std::string cwd = json_value(params, "cwd", std::string());
-    return std::make_unique<tools_io_basic>(cwd);
+    std::string cwd     = json_value(params, "cwd", std::string());
+    std::string runtime = json_value(params, "runtime", std::string());
+    if (runtime.empty()) {
+        return std::make_unique<tools_io_basic>(cwd);
+    }
+    if (runtime.rfind(SERVER_TOOL_RUNTIME_DOCKER_CONTAINER, 0) == 0) {
+        return std::make_unique<tools_io_docker>(runtime.substr(SERVER_TOOL_RUNTIME_DOCKER_CONTAINER.size()), cwd);
+    }
+    // do not fall back to the host, the caller asked for an isolate
+    throw std::runtime_error("unknown tool runtime: " + runtime);
 }
 
 // no '/' in pattern -> match basename at any depth; else match full relative path
@@ -861,8 +1148,11 @@ struct server_tool_exec_shell_command : server_tool {
         timeout    = std::min(timeout,    SERVER_TOOL_EXEC_SHELL_COMMAND_MAX_TIMEOUT);
         max_output = std::min(max_output, SERVER_TOOL_EXEC_SHELL_COMMAND_MAX_OUTPUT_SIZE);
 
+        // an isolate is always POSIX regardless of host OS, so it always gets `sh -c`
 #ifdef _WIN32
-        std::vector<std::string> args = {"cmd", "/c", command};
+        std::vector<std::string> args = !json_value(params, "runtime", std::string()).empty()
+            ? std::vector<std::string>{"sh", "-c", command}
+            : std::vector<std::string>{"cmd", "/c", command};
 #else
         std::vector<std::string> args = {"sh", "-c", command};
 #endif
@@ -1355,11 +1645,16 @@ struct server_tool_get_info : server_tool {
     json invoke(json params, server_tool::stream *) const override {
         auto io = make_tools_io(params);
 
+        // inside an isolate, we always use the linux command
 #ifdef _WIN32
-        auto res = io->run({"cmd", "/c", "ver"}, SERVER_TOOL_GET_INFO_MAX_OUTPUT, SERVER_TOOL_GET_INFO_TIMEOUT);
+        std::vector<std::string> args = !json_value(params, "runtime", std::string()).empty()
+            ? std::vector<std::string>{"uname", "-a"}
+            : std::vector<std::string>{"cmd", "/c", "ver"};
 #else
-        auto res = io->run({"uname", "-a"}, SERVER_TOOL_GET_INFO_MAX_OUTPUT, SERVER_TOOL_GET_INFO_TIMEOUT);
+        std::vector<std::string> args = {"uname", "-a"};
 #endif
+
+        auto res = io->run(args, SERVER_TOOL_GET_INFO_MAX_OUTPUT, SERVER_TOOL_GET_INFO_TIMEOUT);
         // "ver" prints a blank line before the version, so the output is stripped on both ends;
         // a failed spawn or a timeout leaves a diagnostic in res.output, which is not an OS name
         std::string os_info = res.exit_code == 0 && !res.timed_out ? string_strip(res.output) : "unknown";
@@ -1461,6 +1756,103 @@ struct server_mcp_tool : server_tool {
     }
 };
 
+// owns the docker container used as the sandboxed runtime for tool invocations, as configured by
+// --tools-runtime. "spawned" mode starts and stops the container itself; "existing" mode just reuses
+// a container id the user already has running and never stops it.
+struct server_tools_docker_runtime {
+    server_tools_docker_runtime(const server_tools_docker_runtime &) = delete;
+
+    explicit server_tools_docker_runtime(const std::string & spec) {
+        static const std::string docker_prefix = "docker:";
+        if (spec.rfind(docker_prefix, 0) == 0) {
+            spawned = true;
+            image   = spec.substr(docker_prefix.size());
+            if (image.empty()) {
+                throw std::runtime_error("--tools-runtime docker:<image> requires an image name");
+            }
+            spawn();
+        } else if (spec.rfind(SERVER_TOOL_RUNTIME_DOCKER_CONTAINER, 0) == 0) {
+            spawned      = false;
+            container_id = spec.substr(SERVER_TOOL_RUNTIME_DOCKER_CONTAINER.size());
+            if (container_id.empty()) {
+                throw std::runtime_error("--tools-runtime docker-container:<id> requires a container id");
+            }
+        } else {
+            throw std::runtime_error("unknown --tools-runtime option: " + spec);
+        }
+    }
+
+    ~server_tools_docker_runtime() {
+        if (spawned && !container_id.empty()) {
+            // closing stdin signals the container's shell (its pid 1) to exit; --rm then removes it
+            proc.close_stdin();
+            proc.join();
+        }
+    }
+
+    // container id to use for the next tool call; respawns a spawned container that died on its own,
+    // or throws if an externally-managed one is no longer reachable
+    std::string get_container_id() {
+        std::lock_guard<std::mutex> lock(mutex);
+        if (!spawned) {
+            if (!is_running(container_id)) {
+                throw std::runtime_error(string_format(
+                    "docker container \"%s\" is no longer running, restart it to keep using tools",
+                    container_id.c_str()));
+            }
+            return container_id;
+        }
+
+        if (!proc.alive()) {
+            SRV_WRN("docker tools runtime container \"%s\" died, respawning\n", container_id.c_str());
+            spawn();
+        }
+        return container_id;
+    }
+
+private:
+    bool spawned = false;
+    std::string image; // spawned mode only
+    std::string container_id;
+    common_subproc proc; // spawned mode only: `docker run` client that keeps the container alive
+    std::mutex mutex;
+
+    // spawns "docker run --rm -i <image> sh" and keeps its stdin open; the shell blocks reading stdin,
+    // so the container stays alive until we close it (see destructor) or it is killed from the outside
+    void spawn() {
+        std::error_code ec;
+        fs::path cidfile = fs::temp_directory_path(ec) / string_format(
+            "llama-tools-runtime-cid-%zu.tmp", std::hash<std::thread::id>{}(std::this_thread::get_id()));
+        fs::remove(cidfile, ec);
+
+        std::vector<std::string> args = {"docker", "run", "--rm", "-i", "--cidfile", cidfile.string(), image, "sh"};
+        int options = subprocess_option_no_window
+                    | subprocess_option_inherit_environment
+                    | subprocess_option_search_user_path;
+        if (!proc.create(args, options)) {
+            throw std::runtime_error("failed to spawn docker container for tools runtime (image: " + image + ")");
+        }
+
+        std::string cid;
+        for (int i = 0; i < 100 && cid.empty(); i++) {
+            std::ifstream f(cidfile);
+            if (f) std::getline(f, cid);
+            if (cid.empty()) std::this_thread::sleep_for(std::chrono::milliseconds(100));
+        }
+        fs::remove(cidfile, ec);
+        if (cid.empty()) {
+            proc.terminate();
+            throw std::runtime_error("timed out waiting for docker container to start (image: " + image + ")");
+        }
+        container_id = cid;
+    }
+
+    static bool is_running(const std::string & id) {
+        auto res = run_subprocess({"docker", "inspect", "-f", "{{.State.Running}}", id}, 16, 5, nullptr, true);
+        return res.exit_code == 0 && !res.timed_out && res.output.rfind("true", 0) == 0;
+    }
+};
+
 static server_tool & find_tool(std::vector<std::unique_ptr<server_tool>> & tools, const std::string & name, bool require_stream) {
     for (auto & t : tools) {
         if (t->name == name) {
@@ -1506,8 +1898,16 @@ static std::string get_header(const std::map<std::string, std::string> & headers
     return default_value;
 }
 
+server_tools::server_tools() = default;
+server_tools::~server_tools() = default;
+
 void server_tools::setup(const std::vector<std::string> & enabled_tools,
-                         server_mcp & mcp_mgr) {
+                         server_mcp & mcp_mgr,
+                         const std::string & tools_runtime) {
+    if (!tools_runtime.empty()) {
+        docker_runtime = std::make_unique<server_tools_docker_runtime>(tools_runtime);
+    }
+
     if (!enabled_tools.empty()) {
         if (!common_subproc::is_supported()) {
             throw std::runtime_error("subprocess is not enabled on this build");
@@ -1590,11 +1990,26 @@ void server_tools::setup(const std::vector<std::string> & enabled_tools,
             bool stream = body.value("stream", false);
 
             // accept x-tool-cwd header to override of the process
+            if (params.contains("cwd")) {
+                params.erase("cwd");
+            }
             auto cwd = get_header(req.headers, "x-tool-cwd");
             if (!cwd.empty()) {
                 params["cwd"] = cwd;
             }
 
+            // accept x-tool-runtime header to route tool I/O through an isolate, e.g. "docker-container:<id>";
+            // falls back to the --tools-runtime isolate, if configured
+            if (params.contains("runtime")) {
+                params.erase("runtime");
+            }
+            auto runtime = get_header(req.headers, "x-tool-runtime");
+            if (!runtime.empty()) {
+                params["runtime"] = runtime;
+            } else if (docker_runtime) {
+                params["runtime"] = SERVER_TOOL_RUNTIME_DOCKER_CONTAINER + docker_runtime->get_container_id();
+            }
+
             server_tool & tool = find_tool(tools, tool_name, stream);
 
             if (stream) {
index 601399ee9392cc8b8e72948eb932648d87eeb205..7f70e6767e4c5418aea1d1f62d1d1662f7673fb7 100644 (file)
@@ -30,6 +30,8 @@ struct server_tool {
     json to_json() const;
 };
 
+struct server_tools_docker_runtime; // impl detail, defined in server-tools.cpp
+
 struct server_tools {
     std::vector<std::unique_ptr<server_tool>> tools;
 
@@ -37,9 +39,16 @@ struct server_tools {
     server_response queue_res;
     std::atomic<int> res_id{0};
 
+    // set when --tools-runtime is configured; owns the docker container used to run tools, if any
+    std::unique_ptr<server_tools_docker_runtime> docker_runtime;
+
     void setup(const std::vector<std::string> & enabled_tools,
-               server_mcp & mcp_mgr);
+               server_mcp & mcp_mgr,
+               const std::string & tools_runtime);
 
     server_http_context::handler_t handle_get;
     server_http_context::handler_t handle_post;
+
+    server_tools();
+    ~server_tools();
 };
index aafb1f30796605ce146a66049b0b763c399b5d95..1b2e6edb4ed517feae291dc74e2625b7237652c6 100644 (file)
@@ -338,7 +338,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
 
     if (!params.server_tools.empty() || !mcp_mgr.empty()) {
         try {
-            tools.setup(params.server_tools, mcp_mgr);
+            tools.setup(params.server_tools, mcp_mgr, params.server_tools_runtime);
         } catch (const std::exception & e) {
             SRV_ERR("tools setup failed: %s\n", e.what());
             return 1;
@@ -348,6 +348,9 @@ int llama_server(common_params & params, int argc, char ** argv) {
         if (!params.server_tools.empty()) {
             warn_names.push_back("built-in tools (experimental)");
         }
+        if (!params.server_tools_runtime.empty()) {
+            warn_names.push_back("tools runtime (experimental)");
+        }
         if (!mcp_mgr.empty()) {
             warn_names.push_back("MCP servers (experimental)");
         }
index 11c82e690adb1f5853ba84495def440bea2c9386..c651e8e72db2553a6ada0125b31cd11abba1b3ad 100755 (executable)
@@ -1,4 +1,6 @@
 import os
+import shutil
+import subprocess
 
 import pytest
 from utils import *
@@ -146,6 +148,95 @@ def test_tools_builtin_cwd_header():
             os.remove(marker_path)
 
 
+def _docker_unavailable_reason() -> str | None:
+    """None if docker can be used to run a container, otherwise the reason it can't."""
+    docker_bin = shutil.which("docker")
+    if docker_bin is None:
+        return "docker is not installed"
+    try:
+        subprocess.run([docker_bin, "info"], capture_output=True, timeout=5, check=True)
+    except Exception as e:
+        return f"docker daemon is not usable: {e}"
+    return None
+
+
+@pytest.fixture
+def docker_container():
+    reason = _docker_unavailable_reason()
+    if reason is not None:
+        pytest.skip(reason)  # ty: ignore[too-many-positional-arguments, invalid-argument-type]
+
+    proc = subprocess.run(
+        ["docker", "run", "-d", "--rm", "busybox", "sleep", "300"],
+        capture_output=True, text=True,
+    )
+    if proc.returncode != 0:
+        pytest.skip(f"failed to start docker container: {proc.stderr.strip()}")  # ty: ignore[too-many-positional-arguments, invalid-argument-type]
+
+    container_id = proc.stdout.strip()
+    try:
+        yield container_id
+    finally:
+        subprocess.run(["docker", "rm", "-f", container_id], capture_output=True)
+
+
+def test_tools_builtin_runtime_header(docker_container: str):
+    global server
+    server.start()
+
+    headers = {"x-tool-runtime": f"docker-container:{docker_container}", "x-tool-cwd": "/tmp"}
+
+    write_res = call_tool("write_file", {"path": "test.log", "content": "hello docker\n"}, headers=headers)
+    assert write_res["result"] == "file written successfully"
+
+    read_res = call_tool("read_file", {"path": "test.log"}, headers=headers)
+    assert read_res["plain_text_response"] == "hello docker\n"
+
+    exec_res = call_tool("exec_shell_command", {"command": "cat test.log"}, headers=headers)
+    assert "hello docker" in exec_res["plain_text_response"]
+
+
+def test_tools_builtin_runtime_header_unknown_scheme():
+    global server
+    server.start()
+
+    # an unknown runtime must fail, never silently fall back to running on the host
+    res = server.make_request("POST", "/tools",
+                              data={"tool": "exec_shell_command", "params": {"command": "echo hi"}},
+                              headers={"x-tool-runtime": "ssh:example.com"})
+    assert res.status_code == 500, res.body
+    assert "unknown tool runtime" in str(res.body)
+
+
+def test_tools_builtin_docker_runtime_cleans_up_spawned_container():
+    reason = _docker_unavailable_reason()
+    if reason is not None:
+        pytest.skip(reason)  # ty: ignore[too-many-positional-arguments, invalid-argument-type]
+
+    global server
+    server.server_tools_runtime = "docker:busybox"
+    server.start()
+
+    # exec_shell_command runs inside the container spawned for --tools-runtime; docker sets
+    # the container's hostname to its own short id, so this also tells us which one to check
+    res = call_tool("exec_shell_command", {"command": "hostname"})
+    container_id = res["plain_text_response"].splitlines()[0].strip()
+    assert len(container_id) >= 8, res
+
+    running = subprocess.run(
+        ["docker", "inspect", "-f", "{{.State.Running}}", container_id],
+        capture_output=True, text=True,
+    )
+    assert running.returncode == 0 and running.stdout.strip() == "true", running.stderr
+
+    server.stop()
+
+    # a clean server shutdown must stop and remove the container it spawned (it runs with --rm),
+    # not leave it behind as an abandoned child
+    leftover = subprocess.run(["docker", "inspect", container_id], capture_output=True, text=True)
+    assert leftover.returncode != 0, f"container {container_id} was not cleaned up after server exit"
+
+
 def test_tools_builtin_edit_file_rejects_overlapping_edits():
     global server
     server.start()
index 3416f0bfde4236a2865d0c46774100950efa2832..fffe07a67116f43c9beddb854db449ef4024b9bd 100644 (file)
@@ -115,6 +115,7 @@ class ServerProcess:
     backend_sampling: bool = False
     gcp_compat: bool = False
     server_tools: str | None = None
+    server_tools_runtime: str | None = None
     mcp_servers_config: str | None = None
     mcp_servers_json: str | None = None
     cors_origins: str | None = None
@@ -270,6 +271,8 @@ class ServerProcess:
             server_args.append("--ui-mcp-proxy")
         if self.server_tools:
             server_args.extend(["--tools", self.server_tools])
+        if self.server_tools_runtime:
+            server_args.extend(["--tools-runtime", self.server_tools_runtime])
         if self.mcp_servers_config:
             server_args.extend(["--mcp-servers-config", self.mcp_servers_config])
         if self.mcp_servers_json: