Files
2026-09-04 14:58:42 +08:00

276 lines
9.6 KiBLFS
Python

#!/usr/bin/env python3
"""Validate task skill frontmatter used by the task skill catalog.
This intentionally avoids a YAML dependency. It only parses the top-level
fields needed to catch OpenHands catalog drop risks plus the repository's
stricter house style for explicit, canonical task-skill names and descriptions.
"""
from __future__ import annotations
import re
import sys
from dataclasses import dataclass
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parent.parent.parent
SKILL_NAME_RE = re.compile(r"^[a-z0-9]+(-[a-z0-9]+)*$")
TOP_LEVEL_KEY_RE = re.compile(r"^([A-Za-z0-9_-]+):(.*)$")
@dataclass(frozen=True)
class Field:
value: str
children: tuple[str, ...]
line: int
def strip_inline_comment(value: str) -> str:
"""Strip an unquoted trailing ``# ...`` comment, mirroring YAML semantics.
A ``#`` starts a comment only at the start of the value or when preceded by
whitespace, and only when it sits outside quotes and flow collections
(``[]`` / ``{}``). This keeps the hand-rolled parser in step with how
OpenHands (via PyYAML) reads the same line, so a value like
``name: my-skill # canonical`` is treated as ``my-skill``.
"""
in_single = in_double = False
depth = 0
for i, ch in enumerate(value):
if in_single:
in_single = ch != "'"
elif in_double:
in_double = ch != '"'
elif ch == "'":
in_single = True
elif ch == '"':
in_double = True
elif ch in "[{":
depth += 1
elif ch in "]}":
depth = max(0, depth - 1)
elif ch == "#" and depth == 0 and (i == 0 or value[i - 1] in " \t"):
return value[:i].rstrip()
return value
def split_top_level(text: str, sep: str) -> list[str]:
"""Split ``text`` on ``sep`` only where it sits outside quotes/collections.
Used to parse inline-flow ``{a: b, c: "x, y"}`` metadata without tripping
over commas or colons that live inside quoted strings or nested ``[]``/``{}``.
"""
parts: list[str] = []
buf: list[str] = []
in_single = in_double = False
depth = 0
for ch in text:
if in_single:
buf.append(ch)
in_single = ch != "'"
elif in_double:
buf.append(ch)
in_double = ch != '"'
elif ch == "'":
in_single = True
buf.append(ch)
elif ch == '"':
in_double = True
buf.append(ch)
elif ch in "[{":
depth += 1
buf.append(ch)
elif ch in "]}":
depth = max(0, depth - 1)
buf.append(ch)
elif ch == sep and depth == 0:
parts.append("".join(buf))
buf = []
else:
buf.append(ch)
parts.append("".join(buf))
return parts
def skill_paths(root: Path) -> list[Path]:
return sorted(root.glob("tasks/*/environment/skills/*/SKILL.md"))
def display_path(path: Path) -> Path:
return path.relative_to(REPO_ROOT) if path.is_relative_to(REPO_ROOT) else path
def frontmatter_lines(path: Path) -> tuple[list[str] | None, list[str]]:
rel = display_path(path)
lines = path.read_text(encoding="utf-8").splitlines()
if not lines or lines[0].strip() != "---":
return None, [f"{rel}: missing YAML frontmatter opening marker"]
for idx, line in enumerate(lines[1:], start=2):
# The closing marker must sit at column 0. Using ``line.strip()`` here
# would treat an indented ``---`` inside a block-scalar description as
# the terminator, truncating the frontmatter and silently skipping any
# fields that follow. ``rstrip`` tolerates trailing whitespace only.
if line.rstrip() == "---":
return lines[1 : idx - 1], []
return None, [f"{rel}: missing YAML frontmatter closing marker"]
def parse_top_level_fields(lines: list[str]) -> dict[str, Field]:
fields: dict[str, Field] = {}
i = 0
while i < len(lines):
raw = lines[i]
stripped = raw.strip()
match = TOP_LEVEL_KEY_RE.match(raw)
if not stripped or stripped.startswith("#") or not match:
i += 1
continue
key = match.group(1)
value = strip_inline_comment(match.group(2).strip())
start_line = i + 2
children: list[str] = []
i += 1
while i < len(lines):
next_raw = lines[i]
next_stripped = next_raw.strip()
if TOP_LEVEL_KEY_RE.match(next_raw) and next_stripped:
break
if next_stripped and not next_stripped.startswith("#"):
children.append(next_raw)
i += 1
fields[key] = Field(value=value, children=tuple(children), line=start_line)
return fields
def unquote(value: str) -> str:
if len(value) >= 2 and value[0] == value[-1] and value[0] in {"'", '"'}:
return value[1:-1]
return value
def is_scalar_string(value: str) -> bool:
if not value:
return False
if value in {"|", ">"}:
return True
lower = value.lower()
if lower in {"true", "false", "null", "~"}:
return False
if value.startswith(("[", "{")):
return False
try:
float(value)
except ValueError:
return True
return False
def has_nonempty_description(field: Field | None) -> bool:
if field is None:
return False
if is_scalar_string(field.value) and field.value not in {"|", ">"}:
return bool(unquote(field.value).strip())
if field.value in {"", "|", ">"}:
return any(child.strip() for child in field.children)
return False
def lint_metadata(path: Path, field: Field) -> list[str]:
"""Validate that ``metadata`` is a mapping.
Current OpenHands stringifies nested metadata values, so nested hook blocks
are safe to keep. What matters for catalog loading is that ``metadata`` is a
mapping at the top level, not a list or scalar.
"""
if field.value:
if field.value == "{}":
return []
if field.value.startswith("{") and field.value.endswith("}"):
errors: list[str] = []
for item in split_top_level(field.value[1:-1], ","):
if not item.strip():
continue
parts = split_top_level(item, ":")
if len(parts) < 2:
errors.append(f"{path}:{field.line}: metadata entry is not a key-value pair: {item.strip()!r}")
continue
key = parts[0]
if not key.strip():
errors.append(f"{path}:{field.line}: metadata contains an empty key")
return errors
return [f"{path}:{field.line}: metadata must be a mapping"]
entries = [
child
for child in field.children
if child.strip() and not child.strip().startswith("#")
]
if not entries:
return [f"{path}:{field.line}: metadata must be a mapping"]
first_level_indent = min(len(child) - len(child.lstrip(" ")) for child in entries)
errors: list[str] = []
for child in entries:
indent = len(child) - len(child.lstrip(" "))
if indent != first_level_indent:
continue
stripped = child.strip()
if stripped.startswith("- "):
errors.append(f"{path}:{field.line}: metadata must be a mapping, not a list")
elif ":" not in stripped:
errors.append(f"{path}:{field.line}: metadata entry is not a key-value pair: {stripped!r}")
key = stripped.split(":", 1)[0]
if not key.strip():
errors.append(f"{path}:{field.line}: metadata contains an empty key")
return errors
def lint_one(path: Path) -> list[str]:
rel = display_path(path)
lines, errors = frontmatter_lines(path)
if lines is None:
return errors
fields = parse_top_level_fields(lines)
name = fields.get("name")
if name is None or not is_scalar_string(name.value):
errors.append(f"{rel}: missing scalar string field: name")
else:
skill_name = unquote(name.value).strip()
directory_name = path.parent.name
if not SKILL_NAME_RE.fullmatch(skill_name):
errors.append(f"{rel}:{name.line}: name must be lowercase alphanumeric with single hyphens: {skill_name!r}")
if skill_name != directory_name:
errors.append(f"{rel}:{name.line}: name {skill_name!r} must match directory {directory_name!r}")
if not has_nonempty_description(fields.get("description")):
errors.append(f"{rel}: missing non-empty string field: description")
# ``compatibility`` and ``license`` are both typed ``str | None`` on the
# OpenHands Skill model; a dict/list value raises a ValidationError that is
# swallowed per-skill, so the skill silently vanishes from the catalog.
for scalar_field in ("compatibility", "license"):
field = fields.get(scalar_field)
if field is not None and (field.children or not is_scalar_string(field.value)):
errors.append(f"{rel}:{field.line}: {scalar_field} must be a scalar string")
metadata = fields.get("metadata")
if metadata is not None:
errors.extend(lint_metadata(rel, metadata))
return errors
def main(argv: list[str]) -> int:
targets = [Path(arg) for arg in argv] if argv else skill_paths(REPO_ROOT)
errors = [error for path in targets for error in lint_one(path)]
for error in errors:
print(f"ERROR: {error}", file=sys.stderr)
print(f"Checked {len(targets)} task skill file(s): {len(errors)} error(s).", file=sys.stderr)
return 1 if errors else 0
if __name__ == "__main__":
raise SystemExit(main(sys.argv[1:]))