From f43e5b6403ac7309260111be88302e0eb40d041b Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 9 Jul 2026 16:18:29 +0800 Subject: [PATCH] docs(fetch_url): fix 3 misleading description bits MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three small but high-leverage text corrections. No code or test changes — these only affect what the LLM sees in tools/list and Pydantic schemas. 1) FetchUrlRequest.connection_name description Old: "The Connection's yarn_rm_url defines the allowed host domain." New: explicitly says url_allowlist is the host gate, NOT yarn_rm_url. yarn_rm_url is only used by get_external_* tools. Without this fix the LLM would try to control fetch scope via yarn_rm_url (a no-op) instead of url_allowlist. 2) Connection.url_allowlist description Added: "Set or change via save_connection (pass url_allowlist on create) or update_connection (PATCH the field on an existing connection)." Tells the LLM which tools populate the field, instead of leaving it to guess. 3) /fetch_url route description Old: "30s timeout, redirects followed." New: "30s timeout, redirects followed, response body capped at 1 MB (the response includes a truncated boolean when this kicks in)." LLM previously had no way to discover the 1MB cap or the truncated indicator; it would just see short responses and assume that was the full body. Tests: 394 passed, no test changes (descriptions are not asserted). Co-Authored-By: Claude Fable 5 --- spark_executor/models.py | 2 ++ spark_executor/server.py | 3 ++- spark_executor/tools/requests.py | 6 ++++-- 3 files changed, 8 insertions(+), 3 deletions(-) diff --git a/spark_executor/models.py b/spark_executor/models.py index 9e35af8..96f7bbb 100644 --- a/spark_executor/models.py +++ b/spark_executor/models.py @@ -96,6 +96,8 @@ class Connection(BaseModel): "List of fnmatch glob patterns for hosts the fetch_url tool may access. " "The list is mandatory-opt-in: an empty list (the default) denies all " "hosts, so you must populate it before fetch_url can access any URL. " + "Set or change via save_connection (pass url_allowlist on create) or " + "update_connection (PATCH the field on an existing connection). " "Useful for clusters whose hostnames do NOT share a common suffix — " "e.g. single-label hosts like 'ccam1'-'ccam99' (configure ['ccam*']) " "or HDFS namenode on a different subdomain ('*.hadoop.internal'). " diff --git a/spark_executor/server.py b/spark_executor/server.py index 4b26128..dc7510d 100644 --- a/spark_executor/server.py +++ b/spark_executor/server.py @@ -549,7 +549,8 @@ def _update_job_file(req: UpdateJobFileRequest): "guardrails — the allowlist is the only gate — so keep it tight. The " "Connection's saved auth is reused, so the agent does not need cluster " "credentials.\n\n" - "30s timeout, redirects followed." + "**Limits:** 30s timeout, redirects followed, response body capped at " + "1 MB (the response includes a `truncated` boolean when this kicks in)." ), ) def _fetch_url(req: FetchUrlRequest): diff --git a/spark_executor/tools/requests.py b/spark_executor/tools/requests.py index 9cb1f0f..39d3c46 100644 --- a/spark_executor/tools/requests.py +++ b/spark_executor/tools/requests.py @@ -408,8 +408,10 @@ class FetchUrlRequest(BaseModel): ..., description=( "Name of a saved Connection (see list_connections). The " - "Connection's yarn_rm_url defines the allowed host domain. " - "The Connection's auth_type / auth_user / auth_password / " + "Connection's url_allowlist is the host gate for this tool — " + "**not yarn_rm_url** (yarn_rm_url is only used by the " + "get_external_* tools to locate the YARN RM). The " + "Connection's auth_type / auth_user / auth_password / " "ssl_verify / ssl_ca_bundle are reused for the request." ), )