revert(yarn_client): drop HTML directory-listing fallback
The previous commit (32ee167) added a regex-based HTML directory-listing parser as a fallback when the NodeManager wraps /stdout in HTML (Knox, wrapped vendor YARN builds). On review, the parser is too brittle to trust in production: - Apache-style listings vary across Hadoop distributions; some add extra rows, icons, or anchors that change the regex hit set. - Some clusters return the listing but 4xx each individual log file (auth layer in the way), so the fallback silently returns empty. - The behavior the agent observes for the same application_id becomes environment-dependent in ways that are hard to reason about. Revert to the simple, predictable path: GET /apps/{appid}, read amContainerLogs, GET {amContainerLogs}/stdout, return its text. If the cluster wraps /stdout in HTML, the agent gets a clear 'log-aggregation-enable' error and can ask the user to enable log aggregation or use a different fetch mechanism — much easier to debug than a half-parsed directory listing. Removes: - _looks_like_html, _parse_log_listing_html helpers - The bare-URL fallback path and its try/except wrapper - 4 unit tests that exercised the parser Full suite back to 243 passed (matchesfd0036c). Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -14,12 +14,10 @@ Endpoints used (YARN 2.6+):
|
||||
|
||||
When aggregated logs are unavailable (HTTP 404/501, e.g. log aggregation
|
||||
disabled), fall back to the amContainerLogs field reported by the app
|
||||
endpoint and fetch the AM (driver) container's /stdout from the
|
||||
NodeManager. Some clusters (Knox-proxied, wrapped YARN builds) wrap
|
||||
/stdout in an HTML directory listing; in that case we fetch the bare
|
||||
{amContainerLogs} URL, parse the listing, and pull each log file. This
|
||||
returns driver logs only; executor container logs still require
|
||||
yarn.log-aggregation-enable=true (the primary /aggregated-logs path).
|
||||
endpoint and fetch the AM (driver) container's /stdout directly from
|
||||
the NodeManager. This returns driver logs only; executor container logs
|
||||
still require yarn.log-aggregation-enable=true (the primary
|
||||
/aggregated-logs path).
|
||||
|
||||
The ResourceManager URL is passed in per call (snapshotted on the Job at
|
||||
confirm_submit_job time) and falls back to the YARN_RESOURCE_MANAGER_URL env
|
||||
@@ -29,8 +27,6 @@ field was designed for, but no longer requires the `yarn` CLI to interpret it.
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
|
||||
import re
|
||||
|
||||
import httpx
|
||||
import httpx_kerberos
|
||||
|
||||
@@ -168,50 +164,11 @@ def _logs_unavailable_error(application_id: str) -> YarnError:
|
||||
)
|
||||
|
||||
|
||||
def _looks_like_html(text: str) -> bool:
|
||||
"""Cheap heuristic: True if the body starts with an HTML/XHTML prolog.
|
||||
|
||||
Intentionally lightweight (no full parser) — used only to decide whether
|
||||
a NodeManager log response is raw text or an HTML directory listing.
|
||||
"""
|
||||
head = text.lstrip()[:200].lower()
|
||||
return head.startswith(("<!doctype html", "<html", "<?xml"))
|
||||
|
||||
|
||||
def _parse_log_listing_html(html: str) -> list[str]:
|
||||
"""Extract plain log-file names from a NodeManager directory-listing page.
|
||||
|
||||
Keeps simple filenames (stdout, stderr, syslog, prelaunch.err,
|
||||
launch_container.sh, ...); drops parent-dir links, absolute URLs, and
|
||||
sort-toggle query strings. Returns hrefs in document order, deduplicated.
|
||||
"""
|
||||
hrefs = re.findall(r'href="([^"]*)"', html, re.IGNORECASE)
|
||||
seen: set[str] = set()
|
||||
out: list[str] = []
|
||||
for href in hrefs:
|
||||
if href in ("../", "/") or href.startswith("/"):
|
||||
continue
|
||||
if "://" in href or "?" in href:
|
||||
continue
|
||||
if href in seen:
|
||||
continue
|
||||
seen.add(href)
|
||||
out.append(href)
|
||||
return out
|
||||
|
||||
|
||||
def _fetch_logs_via_am_container(application_id: str, config: YarnClientConfig) -> str:
|
||||
"""
|
||||
Fetch ApplicationMaster (driver) container logs as a fallback when the
|
||||
Fetch ApplicationMaster (driver) container stdout as a fallback when the
|
||||
aggregated-logs endpoint is unavailable.
|
||||
|
||||
Fast path: GET {amContainerLogs}/stdout and return its text directly when
|
||||
the NodeManager serves raw log bytes (modern YARN).
|
||||
|
||||
Fallback path: when /stdout returns an HTML directory listing (Knox /
|
||||
wrapped YARN), GET the bare {amContainerLogs} URL, parse the listing for
|
||||
log-file names, and fetch each one individually.
|
||||
|
||||
Returns AM (driver) container logs only. Executor container logs require
|
||||
yarn.log-aggregation-enable=true, which is covered by the primary
|
||||
/aggregated-logs path.
|
||||
@@ -227,35 +184,12 @@ def _fetch_logs_via_am_container(application_id: str, config: YarnClientConfig)
|
||||
if not am_container_logs:
|
||||
raise _logs_unavailable_error(application_id)
|
||||
|
||||
am_container_logs = am_container_logs.rstrip("/")
|
||||
verify = config.verify_for_httpx()
|
||||
auth = config.auth_for_httpx()
|
||||
|
||||
# Fast path: modern YARN serves raw log bytes at /stdout.
|
||||
stdout_url = f"{am_container_logs}/stdout"
|
||||
resp = _request("GET", stdout_url, timeout=60.0, verify=verify, auth=auth)
|
||||
if resp.status_code < 400 and not _looks_like_html(resp.text):
|
||||
return resp.text
|
||||
|
||||
# Fallback path: Knox / wrapped YARN serves an HTML directory listing for
|
||||
# every URL on the NodeManager log endpoint. Parse it and fetch each file.
|
||||
orig_resp_text = resp.text
|
||||
try:
|
||||
resp = _request("GET", am_container_logs, timeout=60.0, verify=verify, auth=auth)
|
||||
if resp.status_code < 400:
|
||||
parts: list[str] = []
|
||||
for filename in _parse_log_listing_html(resp.text):
|
||||
file_url = f"{am_container_logs}/{filename}"
|
||||
file_resp = _request("GET", file_url, timeout=60.0, verify=verify, auth=auth)
|
||||
if file_resp.status_code < 400 and not _looks_like_html(file_resp.text):
|
||||
parts.append(f"=== {filename} ===\n{file_resp.text}")
|
||||
if parts:
|
||||
return "\n\n".join(parts)
|
||||
except Exception as ex:
|
||||
logger.warning(f"Parse html error: {ex}")
|
||||
return orig_resp_text
|
||||
|
||||
raise _logs_unavailable_error(application_id)
|
||||
log_url = f"{am_container_logs.rstrip('/')}/stdout"
|
||||
resp = _request("GET", log_url, timeout=60.0, verify=config.verify_for_httpx(), auth=config.auth_for_httpx())
|
||||
if resp.status_code >= 400:
|
||||
logger.error(f"YARN GET {log_url} -> {resp.status_code}: {resp.text[:500]}")
|
||||
raise _logs_unavailable_error(application_id)
|
||||
return resp.text
|
||||
|
||||
|
||||
def get_application_logs(application_id: str, config: YarnClientConfig) -> str:
|
||||
|
||||
Reference in New Issue
Block a user