#!/usr/bin/env python3 """Validate task skill frontmatter used by the task skill catalog. This intentionally avoids a YAML dependency. It only parses the top-level fields needed to catch OpenHands catalog drop risks plus the repository's stricter house style for explicit, canonical task-skill names and descriptions. """ from __future__ import annotations import re import sys from dataclasses import dataclass from pathlib import Path REPO_ROOT = Path(__file__).resolve().parent.parent.parent SKILL_NAME_RE = re.compile(r"^[a-z0-9]+(-[a-z0-9]+)*$") TOP_LEVEL_KEY_RE = re.compile(r"^([A-Za-z0-9_-]+):(.*)$") @dataclass(frozen=True) class Field: value: str children: tuple[str, ...] line: int def strip_inline_comment(value: str) -> str: """Strip an unquoted trailing ``# ...`` comment, mirroring YAML semantics. A ``#`` starts a comment only at the start of the value or when preceded by whitespace, and only when it sits outside quotes and flow collections (``[]`` / ``{}``). This keeps the hand-rolled parser in step with how OpenHands (via PyYAML) reads the same line, so a value like ``name: my-skill # canonical`` is treated as ``my-skill``. """ in_single = in_double = False depth = 0 for i, ch in enumerate(value): if in_single: in_single = ch != "'" elif in_double: in_double = ch != '"' elif ch == "'": in_single = True elif ch == '"': in_double = True elif ch in "[{": depth += 1 elif ch in "]}": depth = max(0, depth - 1) elif ch == "#" and depth == 0 and (i == 0 or value[i - 1] in " \t"): return value[:i].rstrip() return value def split_top_level(text: str, sep: str) -> list[str]: """Split ``text`` on ``sep`` only where it sits outside quotes/collections. Used to parse inline-flow ``{a: b, c: "x, y"}`` metadata without tripping over commas or colons that live inside quoted strings or nested ``[]``/``{}``. """ parts: list[str] = [] buf: list[str] = [] in_single = in_double = False depth = 0 for ch in text: if in_single: buf.append(ch) in_single = ch != "'" elif in_double: buf.append(ch) in_double = ch != '"' elif ch == "'": in_single = True buf.append(ch) elif ch == '"': in_double = True buf.append(ch) elif ch in "[{": depth += 1 buf.append(ch) elif ch in "]}": depth = max(0, depth - 1) buf.append(ch) elif ch == sep and depth == 0: parts.append("".join(buf)) buf = [] else: buf.append(ch) parts.append("".join(buf)) return parts def skill_paths(root: Path) -> list[Path]: return sorted(root.glob("tasks/*/environment/skills/*/SKILL.md")) def display_path(path: Path) -> Path: return path.relative_to(REPO_ROOT) if path.is_relative_to(REPO_ROOT) else path def frontmatter_lines(path: Path) -> tuple[list[str] | None, list[str]]: rel = display_path(path) lines = path.read_text(encoding="utf-8").splitlines() if not lines or lines[0].strip() != "---": return None, [f"{rel}: missing YAML frontmatter opening marker"] for idx, line in enumerate(lines[1:], start=2): # The closing marker must sit at column 0. Using ``line.strip()`` here # would treat an indented ``---`` inside a block-scalar description as # the terminator, truncating the frontmatter and silently skipping any # fields that follow. ``rstrip`` tolerates trailing whitespace only. if line.rstrip() == "---": return lines[1 : idx - 1], [] return None, [f"{rel}: missing YAML frontmatter closing marker"] def parse_top_level_fields(lines: list[str]) -> dict[str, Field]: fields: dict[str, Field] = {} i = 0 while i < len(lines): raw = lines[i] stripped = raw.strip() match = TOP_LEVEL_KEY_RE.match(raw) if not stripped or stripped.startswith("#") or not match: i += 1 continue key = match.group(1) value = strip_inline_comment(match.group(2).strip()) start_line = i + 2 children: list[str] = [] i += 1 while i < len(lines): next_raw = lines[i] next_stripped = next_raw.strip() if TOP_LEVEL_KEY_RE.match(next_raw) and next_stripped: break if next_stripped and not next_stripped.startswith("#"): children.append(next_raw) i += 1 fields[key] = Field(value=value, children=tuple(children), line=start_line) return fields def unquote(value: str) -> str: if len(value) >= 2 and value[0] == value[-1] and value[0] in {"'", '"'}: return value[1:-1] return value def is_scalar_string(value: str) -> bool: if not value: return False if value in {"|", ">"}: return True lower = value.lower() if lower in {"true", "false", "null", "~"}: return False if value.startswith(("[", "{")): return False try: float(value) except ValueError: return True return False def has_nonempty_description(field: Field | None) -> bool: if field is None: return False if is_scalar_string(field.value) and field.value not in {"|", ">"}: return bool(unquote(field.value).strip()) if field.value in {"", "|", ">"}: return any(child.strip() for child in field.children) return False def lint_metadata(path: Path, field: Field) -> list[str]: """Validate that ``metadata`` is a mapping. Current OpenHands stringifies nested metadata values, so nested hook blocks are safe to keep. What matters for catalog loading is that ``metadata`` is a mapping at the top level, not a list or scalar. """ if field.value: if field.value == "{}": return [] if field.value.startswith("{") and field.value.endswith("}"): errors: list[str] = [] for item in split_top_level(field.value[1:-1], ","): if not item.strip(): continue parts = split_top_level(item, ":") if len(parts) < 2: errors.append(f"{path}:{field.line}: metadata entry is not a key-value pair: {item.strip()!r}") continue key = parts[0] if not key.strip(): errors.append(f"{path}:{field.line}: metadata contains an empty key") return errors return [f"{path}:{field.line}: metadata must be a mapping"] entries = [ child for child in field.children if child.strip() and not child.strip().startswith("#") ] if not entries: return [f"{path}:{field.line}: metadata must be a mapping"] first_level_indent = min(len(child) - len(child.lstrip(" ")) for child in entries) errors: list[str] = [] for child in entries: indent = len(child) - len(child.lstrip(" ")) if indent != first_level_indent: continue stripped = child.strip() if stripped.startswith("- "): errors.append(f"{path}:{field.line}: metadata must be a mapping, not a list") elif ":" not in stripped: errors.append(f"{path}:{field.line}: metadata entry is not a key-value pair: {stripped!r}") key = stripped.split(":", 1)[0] if not key.strip(): errors.append(f"{path}:{field.line}: metadata contains an empty key") return errors def lint_one(path: Path) -> list[str]: rel = display_path(path) lines, errors = frontmatter_lines(path) if lines is None: return errors fields = parse_top_level_fields(lines) name = fields.get("name") if name is None or not is_scalar_string(name.value): errors.append(f"{rel}: missing scalar string field: name") else: skill_name = unquote(name.value).strip() directory_name = path.parent.name if not SKILL_NAME_RE.fullmatch(skill_name): errors.append(f"{rel}:{name.line}: name must be lowercase alphanumeric with single hyphens: {skill_name!r}") if skill_name != directory_name: errors.append(f"{rel}:{name.line}: name {skill_name!r} must match directory {directory_name!r}") if not has_nonempty_description(fields.get("description")): errors.append(f"{rel}: missing non-empty string field: description") # ``compatibility`` and ``license`` are both typed ``str | None`` on the # OpenHands Skill model; a dict/list value raises a ValidationError that is # swallowed per-skill, so the skill silently vanishes from the catalog. for scalar_field in ("compatibility", "license"): field = fields.get(scalar_field) if field is not None and (field.children or not is_scalar_string(field.value)): errors.append(f"{rel}:{field.line}: {scalar_field} must be a scalar string") metadata = fields.get("metadata") if metadata is not None: errors.extend(lint_metadata(rel, metadata)) return errors def main(argv: list[str]) -> int: targets = [Path(arg) for arg in argv] if argv else skill_paths(REPO_ROOT) errors = [error for path in targets for error in lint_one(path)] for error in errors: print(f"ERROR: {error}", file=sys.stderr) print(f"Checked {len(targets)} task skill file(s): {len(errors)} error(s).", file=sys.stderr) return 1 if errors else 0 if __name__ == "__main__": raise SystemExit(main(sys.argv[1:]))