mirror of
https://github.com/microsoft/BCQuality.git
synced 2026-08-06 09:26:52 +01:00
Promotes the authoring-assist prototype to a first-class BCQuality feature that
reviews knowledge-article front-matter and proposes missing routing `signals:`
(raise triggers and `effect: suppress` suppressors). The suggestion is ADVISORY
ONLY: it never fails a build; authors apply it via a normal PR under existing
R29 + CI + CODEOWNERS review.
Feature:
- tools/Suggest-ArticleSignals.ps1: suggestion engine hardened with a stable JSON
contract (schemaVersion/toolVersion) and a deterministic, effect-sensitive
per-proposal suggestionId, plus -ChangedFiles scoping for PR-diff runs. Runs
self-contained: the routing seed is now OPTIONAL (built-in domain map), so this
does not depend on the separate routing-index tuning thread.
- tools/New-AuthoringAssistComment.ps1: renders one upsertable, feedback-instrumented
advisory comment (stable anchor + aa:meta/aa:article markers carrying suggestionIds
+ reaction footer) for the downstream acceptance-measurement work.
- .github/workflows/authoring-assist*.yml: unprivileged pull_request intake +
trusted workflow_run runner (review/publish split) + in-repo self-test.
- tools/Test-SuggestArticleSignals.ps1: 23-check smoke test (contract, determinism,
scoping, suppressor/prohibition, renderer markers, seed-free graceful degradation).
Dormant schema support:
- .github/scripts/validate_frontmatter.py: adds R29, validating the optional
`signals` block shape (bare token or mapping with token + optional
pattern/domain/effect in {raise,suppress}). Reserved & dormant — no production
pipeline consumes it; the existing PR Reviewer only reads `effect: suppress`
behind its BCQ_INDEX_V2 flag and falls back safely when absent. Existing rules
(incl. R24 skill-id uniqueness, R27, R28) are unchanged.
Excludes the routing-index tuning thread entirely (separate work).
Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
Copilot-Session: 75d852d6-18a6-4c58-986f-ed5ab16618fa
703 lines
29 KiB
Python
703 lines
29 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
BCQuality content validator.
|
|
|
|
Validates frontmatter, sections, and structural rules for knowledge files,
|
|
action skills, meta-skills, and the entry-point skill. Rules derived from
|
|
/skills/read.md, /skills/write.md, /skills/do.md, and /skills/entry.md.
|
|
|
|
Usage:
|
|
python .github/scripts/validate_frontmatter.py [--root PATH]
|
|
|
|
Exit status: 0 on success (no errors), 1 on any error. Warnings do not fail.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import os
|
|
import re
|
|
import sys
|
|
from dataclasses import dataclass, field
|
|
from pathlib import Path
|
|
from typing import Any, Iterable
|
|
|
|
try:
|
|
import yaml
|
|
except ImportError:
|
|
sys.stderr.write("ERROR: PyYAML is required. Install with: pip install pyyaml\n")
|
|
sys.exit(2)
|
|
|
|
|
|
# --- Schema constants -------------------------------------------------------
|
|
|
|
KNOWLEDGE_REQUIRED_KEYS = {
|
|
"bc-version", "domain", "keywords", "technologies",
|
|
"countries", "application-area",
|
|
}
|
|
# Optional knowledge keys (do not trigger the R02 closed-key-set error).
|
|
# `signals`: OPTIONAL routing-signal declarations. RESERVED & DORMANT — the shape
|
|
# is validated (R29 below) so authors and the authoring-assist advisory can
|
|
# populate it, but no production pipeline consumes it yet. See tools/authoring-assist.md.
|
|
KNOWLEDGE_OPTIONAL_KEYS = {"signals"}
|
|
ACTION_SKILL_REQUIRED_KEYS = {
|
|
"kind", "id", "version", "title", "description", "inputs", "outputs",
|
|
}
|
|
ACTION_SKILL_OPTIONAL_KEYS = {
|
|
"bc-version", "technologies", "countries", "application-area", "sub-skills",
|
|
}
|
|
META_SKILL_REQUIRED_KEYS = {"kind", "id", "version", "title"}
|
|
ENTRY_SKILL_REQUIRED_KEYS = {"kind", "id", "version", "title"}
|
|
|
|
STANDARD_INPUTS = {
|
|
"pr-diff", "object-list", "file-path", "repository", "telemetry-query",
|
|
}
|
|
ALLOWED_OUTPUTS = {"findings-report"}
|
|
VALID_SAMPLE_KINDS = {"good", "bad"}
|
|
|
|
ACTION_SKILL_SECTIONS = ["Source", "Relevance", "Worklist", "Action", "Output"]
|
|
|
|
LAYERS = ("microsoft", "community", "custom")
|
|
META_SKILL_FILES = {"read.md", "write.md", "do.md"}
|
|
ENTRY_SKILL_FILE = "entry.md"
|
|
|
|
MAX_KNOWLEDGE_LINES = 100
|
|
|
|
KEBAB_CASE = re.compile(r"^[a-z0-9]+(-[a-z0-9]+)*$")
|
|
ISO_ALPHA2 = re.compile(r"^[a-z]{2}$")
|
|
RANGE_SHORTHAND = re.compile(r"^(\d+)\.\.(\d+)?$")
|
|
FENCED_CODE_BLOCK = re.compile(r"^```", re.MULTILINE)
|
|
HEADING_H2 = re.compile(r"^##\s+(.+?)\s*$", re.MULTILINE)
|
|
SAMPLE_REFERENCE = re.compile(r"`([a-z0-9]+(?:-[a-z0-9]+)*\.(?:good|bad)\.[a-z0-9]+)`")
|
|
|
|
|
|
# --- Diagnostics ------------------------------------------------------------
|
|
|
|
@dataclass
|
|
class Diagnostic:
|
|
level: str # "error" | "warning"
|
|
path: Path
|
|
rule: str # e.g. "R03"
|
|
message: str
|
|
line: int | None = None
|
|
|
|
def format_plain(self, root: Path) -> str:
|
|
rel = self.path.relative_to(root).as_posix()
|
|
prefix = rel if self.line is None else f"{rel}:{self.line}"
|
|
return f"{prefix}: [{self.rule}] {self.level}: {self.message}"
|
|
|
|
def format_gha(self, root: Path) -> str:
|
|
rel = self.path.relative_to(root).as_posix()
|
|
loc = f"file={rel}"
|
|
if self.line is not None:
|
|
loc += f",line={self.line}"
|
|
return f"::{self.level} {loc}::[{self.rule}] {self.message}"
|
|
|
|
|
|
@dataclass
|
|
class Report:
|
|
diagnostics: list[Diagnostic] = field(default_factory=list)
|
|
|
|
def error(self, path: Path, rule: str, message: str, line: int | None = None) -> None:
|
|
self.diagnostics.append(Diagnostic("error", path, rule, message, line))
|
|
|
|
def warn(self, path: Path, rule: str, message: str, line: int | None = None) -> None:
|
|
self.diagnostics.append(Diagnostic("warning", path, rule, message, line))
|
|
|
|
@property
|
|
def errors(self) -> list[Diagnostic]:
|
|
return [d for d in self.diagnostics if d.level == "error"]
|
|
|
|
@property
|
|
def warnings(self) -> list[Diagnostic]:
|
|
return [d for d in self.diagnostics if d.level == "warning"]
|
|
|
|
|
|
# --- Frontmatter parsing ----------------------------------------------------
|
|
|
|
@dataclass
|
|
class Parsed:
|
|
frontmatter: dict[str, Any] | None
|
|
body: str
|
|
body_start_line: int # 1-based line number where body begins
|
|
raw_lines: list[str]
|
|
frontmatter_error: str | None # yaml or delimiter issue
|
|
|
|
|
|
def parse_markdown(text: str) -> Parsed:
|
|
lines = text.splitlines()
|
|
if not lines or lines[0].rstrip() != "---":
|
|
return Parsed(None, text, 1, lines, "missing opening '---' frontmatter delimiter")
|
|
end_idx = None
|
|
for i in range(1, len(lines)):
|
|
if lines[i].rstrip() == "---":
|
|
end_idx = i
|
|
break
|
|
if end_idx is None:
|
|
return Parsed(None, text, 1, lines, "missing closing '---' frontmatter delimiter")
|
|
yaml_text = "\n".join(lines[1:end_idx])
|
|
try:
|
|
fm = yaml.safe_load(yaml_text) or {}
|
|
except yaml.YAMLError as e:
|
|
return Parsed(None, text, end_idx + 2, lines, f"YAML parse error: {e}")
|
|
if not isinstance(fm, dict):
|
|
return Parsed(None, text, end_idx + 2, lines, "frontmatter must be a YAML mapping")
|
|
body = "\n".join(lines[end_idx + 1:])
|
|
return Parsed(fm, body, end_idx + 2, lines, None)
|
|
|
|
|
|
# --- Small helpers ----------------------------------------------------------
|
|
|
|
def is_non_empty_list_of_str(value: Any) -> bool:
|
|
return isinstance(value, list) and len(value) > 0 and all(isinstance(v, str) and v for v in value)
|
|
|
|
|
|
def expand_bc_version(value: Any) -> tuple[list[int] | str | None, str | None]:
|
|
"""Return (expanded, error-message). One of the two is None.
|
|
|
|
For the universal sentinel ["all"], `expanded` is the string "all".
|
|
For an open-ended range like ["26.."], `expanded` is the normalized
|
|
string "26.." (it cannot be enumerated; consumers match target >= 26).
|
|
Otherwise it is the expanded list of version integers.
|
|
"""
|
|
if not isinstance(value, list) or not value:
|
|
return None, "must be a non-empty list"
|
|
# Case 0: universal sentinel
|
|
if len(value) == 1 and value[0] == "all":
|
|
return "all", None
|
|
if "all" in value:
|
|
return None, "'all' is mutually exclusive with explicit versions"
|
|
# Case 1: all integers
|
|
if all(isinstance(v, int) and not isinstance(v, bool) for v in value):
|
|
if any(v <= 0 for v in value):
|
|
return None, "integers must be positive"
|
|
return sorted(set(value)), None
|
|
# Case 2: single-element range shorthand — closed "[26..28]" or open-ended "[26..]"
|
|
if len(value) == 1 and isinstance(value[0], str):
|
|
m = RANGE_SHORTHAND.match(value[0].strip())
|
|
if m:
|
|
start = int(m.group(1))
|
|
if m.group(2) is None:
|
|
# Open-ended: "start.." applies from start onwards, no upper bound.
|
|
return f"{start}..", None
|
|
end = int(m.group(2))
|
|
if start > end:
|
|
return None, f"range '{value[0]}' is not ascending"
|
|
return list(range(start, end + 1)), None
|
|
return None, "must be [all], a list of integers, or a range shorthand like [26..28] or [26..]"
|
|
|
|
|
|
def headings_in_order(body: str) -> list[tuple[str, int]]:
|
|
"""Return list of (heading-text, 1-based line-number-within-body) pairs."""
|
|
out = []
|
|
for i, line in enumerate(body.splitlines(), start=1):
|
|
m = re.match(r"^##\s+(.+?)\s*$", line)
|
|
if m:
|
|
out.append((m.group(1).strip(), i))
|
|
return out
|
|
|
|
|
|
# --- Validators -------------------------------------------------------------
|
|
|
|
def validate_knowledge(path: Path, parsed: Parsed, report: Report) -> None:
|
|
# R01 frontmatter parseable
|
|
if parsed.frontmatter_error:
|
|
report.error(path, "R01", parsed.frontmatter_error, 1)
|
|
return
|
|
fm = parsed.frontmatter
|
|
assert fm is not None
|
|
|
|
# R02 required keys, no extras, none empty
|
|
missing = KNOWLEDGE_REQUIRED_KEYS - fm.keys()
|
|
extras = fm.keys() - KNOWLEDGE_REQUIRED_KEYS - KNOWLEDGE_OPTIONAL_KEYS
|
|
if missing:
|
|
report.error(path, "R02", f"missing required frontmatter keys: {sorted(missing)}", 1)
|
|
if extras:
|
|
report.error(path, "R02", f"unexpected frontmatter keys: {sorted(extras)}", 1)
|
|
for k in KNOWLEDGE_REQUIRED_KEYS & fm.keys():
|
|
v = fm[k]
|
|
if v is None or v == "" or v == []:
|
|
report.error(path, "R02", f"frontmatter key '{k}' must not be empty", 1)
|
|
|
|
# R29 optional routing `signals` block. Each entry is either a bare token
|
|
# string or a mapping with a required 'token' and optional 'pattern'/'domain'/
|
|
# 'effect' (all non-empty strings; effect in {raise, suppress}). Reserved &
|
|
# dormant: validated for shape only so the block stays well-formed for the
|
|
# authoring-assist advisory; no production pipeline consumes it yet.
|
|
if "signals" in fm:
|
|
sigs = fm["signals"]
|
|
if not isinstance(sigs, list) or not sigs:
|
|
report.error(path, "R29", "signals must be a non-empty list", 1)
|
|
else:
|
|
for entry in sigs:
|
|
if isinstance(entry, str):
|
|
if not entry.strip():
|
|
report.error(path, "R29", "signals token string must not be empty", 1)
|
|
elif isinstance(entry, dict):
|
|
tok = entry.get("token")
|
|
if not isinstance(tok, str) or not tok.strip():
|
|
report.error(path, "R29", "signals entry must have a non-empty 'token'", 1)
|
|
for opt in ("pattern", "domain"):
|
|
if opt in entry and (not isinstance(entry[opt], str) or not entry[opt].strip()):
|
|
report.error(path, "R29", f"signals '{opt}' must be a non-empty string", 1)
|
|
if "effect" in entry and entry["effect"] not in ("raise", "suppress"):
|
|
report.error(path, "R29", "signals 'effect' must be 'raise' or 'suppress'", 1)
|
|
unknown = set(entry.keys()) - {"token", "pattern", "domain", "effect"}
|
|
if unknown:
|
|
report.error(path, "R29", f"signals entry has unknown keys: {sorted(unknown)}", 1)
|
|
else:
|
|
report.error(path, "R29", "signals entry must be a string or a mapping", 1)
|
|
|
|
# R03 bc-version
|
|
if "bc-version" in fm:
|
|
_, err = expand_bc_version(fm["bc-version"])
|
|
if err:
|
|
report.error(path, "R03", f"bc-version: {err}", 1)
|
|
|
|
# R04 domain
|
|
if "domain" in fm:
|
|
if not isinstance(fm["domain"], str) or not fm["domain"].strip():
|
|
report.error(path, "R04", "domain must be a non-empty string", 1)
|
|
elif fm["domain"] != path.parent.name:
|
|
report.error(
|
|
path,
|
|
"R27",
|
|
f"frontmatter domain '{fm['domain']}' must match directory '{path.parent.name}'",
|
|
1,
|
|
)
|
|
|
|
# R05 keywords
|
|
if "keywords" in fm:
|
|
kw = fm["keywords"]
|
|
if not is_non_empty_list_of_str(kw):
|
|
report.error(path, "R05", "keywords must be a non-empty list of strings", 1)
|
|
else:
|
|
bad = [k for k in kw if not KEBAB_CASE.match(k)]
|
|
if bad:
|
|
report.error(path, "R05", f"keywords must be lowercase kebab-case: {bad}", 1)
|
|
if len(kw) > 10:
|
|
report.warn(path, "R05", f"keywords count is {len(kw)}; consider trimming toward ≤10", 1)
|
|
|
|
# R06 technologies
|
|
if "technologies" in fm:
|
|
t = fm["technologies"]
|
|
if not is_non_empty_list_of_str(t):
|
|
report.error(path, "R06", "technologies must be a non-empty list of strings", 1)
|
|
elif "all" in t:
|
|
report.error(path, "R06", "technologies must not use the 'all' sentinel; list each technology explicitly", 1)
|
|
|
|
# R07 countries
|
|
if "countries" in fm:
|
|
c = fm["countries"]
|
|
if not is_non_empty_list_of_str(c):
|
|
report.error(path, "R07", "countries must be a non-empty list of strings", 1)
|
|
elif "w1" in c and len(c) > 1:
|
|
report.error(path, "R07", "'w1' is mutually exclusive with country codes", 1)
|
|
elif "w1" not in c:
|
|
bad = [x for x in c if not ISO_ALPHA2.match(x)]
|
|
if bad:
|
|
report.error(path, "R07", f"countries must be lowercase ISO alpha-2 codes or [w1]: {bad}", 1)
|
|
|
|
# R08 application-area
|
|
if "application-area" in fm:
|
|
a = fm["application-area"]
|
|
if not is_non_empty_list_of_str(a):
|
|
report.error(path, "R08", "application-area must be a non-empty list of strings", 1)
|
|
elif "all" in a and len(a) > 1:
|
|
report.error(path, "R08", "'all' is mutually exclusive with specific application areas", 1)
|
|
|
|
# R09 has ## Description
|
|
headings = [h for h, _ in headings_in_order(parsed.body)]
|
|
if "Description" not in headings:
|
|
report.error(path, "R09", "missing required '## Description' section")
|
|
|
|
# R10 no fenced code blocks
|
|
for match in FENCED_CODE_BLOCK.finditer(parsed.body):
|
|
# offset to a 1-based line number in the original file
|
|
prefix = parsed.body[: match.start()]
|
|
body_line = prefix.count("\n") + 1
|
|
file_line = parsed.body_start_line + body_line - 1
|
|
report.error(path, "R10", "knowledge files must not contain fenced code blocks", file_line)
|
|
break # one is enough; don't spam
|
|
|
|
# R11 file size ≤ 100 lines
|
|
total_lines = len(parsed.raw_lines)
|
|
if total_lines > MAX_KNOWLEDGE_LINES:
|
|
report.error(path, "R11", f"file is {total_lines} lines; max is {MAX_KNOWLEDGE_LINES}")
|
|
|
|
|
|
def validate_action_skill(path: Path, parsed: Parsed, report: Report) -> None:
|
|
if parsed.frontmatter_error:
|
|
report.error(path, "R01", parsed.frontmatter_error, 1)
|
|
return
|
|
fm = parsed.frontmatter
|
|
assert fm is not None
|
|
|
|
# R15 required keys; warn on unknown
|
|
missing = ACTION_SKILL_REQUIRED_KEYS - fm.keys()
|
|
if missing:
|
|
report.error(path, "R15", f"missing required action-skill keys: {sorted(missing)}", 1)
|
|
unknown = fm.keys() - ACTION_SKILL_REQUIRED_KEYS - ACTION_SKILL_OPTIONAL_KEYS
|
|
if unknown:
|
|
report.warn(path, "R15", f"unknown action-skill keys: {sorted(unknown)}", 1)
|
|
for k in ACTION_SKILL_REQUIRED_KEYS & fm.keys():
|
|
v = fm[k]
|
|
if v is None or v == "" or v == []:
|
|
report.error(path, "R15", f"action-skill key '{k}' must not be empty", 1)
|
|
|
|
# R25 kind matches path
|
|
if fm.get("kind") != "action-skill":
|
|
report.error(path, "R25", f"file is in a layer skills folder but kind is '{fm.get('kind')}', expected 'action-skill'", 1)
|
|
|
|
# R16 id kebab-case, version positive int
|
|
if "id" in fm:
|
|
if not isinstance(fm["id"], str) or not KEBAB_CASE.match(fm["id"]):
|
|
report.error(path, "R16", f"id must be lowercase kebab-case: '{fm['id']}'", 1)
|
|
if "version" in fm:
|
|
v = fm["version"]
|
|
if not isinstance(v, int) or isinstance(v, bool) or v <= 0:
|
|
report.error(path, "R16", f"version must be a positive integer: {v!r}", 1)
|
|
|
|
# R17 inputs
|
|
if "inputs" in fm:
|
|
inp = fm["inputs"]
|
|
if not is_non_empty_list_of_str(inp):
|
|
report.error(path, "R17", "inputs must be a non-empty list of strings", 1)
|
|
else:
|
|
unknown_inputs = [x for x in inp if x not in STANDARD_INPUTS]
|
|
if unknown_inputs:
|
|
report.warn(path, "R17", f"inputs contains non-standard values {unknown_inputs}; standard set is {sorted(STANDARD_INPUTS)}", 1)
|
|
|
|
# R18 outputs
|
|
if "outputs" in fm:
|
|
out = fm["outputs"]
|
|
if not is_non_empty_list_of_str(out):
|
|
report.error(path, "R18", "outputs must be a non-empty list of strings", 1)
|
|
else:
|
|
bad = [x for x in out if x not in ALLOWED_OUTPUTS]
|
|
if bad:
|
|
report.error(path, "R18", f"outputs contains non-allowed values {bad}; currently only {sorted(ALLOWED_OUTPUTS)} is defined", 1)
|
|
|
|
# R19 optional filter dimensions, if present
|
|
if "bc-version" in fm:
|
|
_, err = expand_bc_version(fm["bc-version"])
|
|
if err:
|
|
report.error(path, "R19", f"bc-version: {err}", 1)
|
|
if "technologies" in fm:
|
|
t = fm["technologies"]
|
|
if not is_non_empty_list_of_str(t):
|
|
report.error(path, "R19", "technologies must be a non-empty list of strings", 1)
|
|
elif "all" in t:
|
|
report.error(path, "R19", "technologies must not use the 'all' sentinel", 1)
|
|
if "countries" in fm:
|
|
c = fm["countries"]
|
|
if not is_non_empty_list_of_str(c):
|
|
report.error(path, "R19", "countries must be a non-empty list of strings", 1)
|
|
elif "w1" in c and len(c) > 1:
|
|
report.error(path, "R19", "'w1' is mutually exclusive with country codes", 1)
|
|
elif "w1" not in c:
|
|
bad = [x for x in c if not ISO_ALPHA2.match(x)]
|
|
if bad:
|
|
report.error(path, "R19", f"countries must be ISO alpha-2 or [w1]: {bad}", 1)
|
|
if "application-area" in fm:
|
|
a = fm["application-area"]
|
|
if not is_non_empty_list_of_str(a):
|
|
report.error(path, "R19", "application-area must be a non-empty list of strings", 1)
|
|
elif "all" in a and len(a) > 1:
|
|
report.error(path, "R19", "'all' is mutually exclusive with specific application areas", 1)
|
|
|
|
# R20 sub-skills shape
|
|
if "sub-skills" in fm:
|
|
ss = fm["sub-skills"]
|
|
if not is_non_empty_list_of_str(ss):
|
|
report.error(path, "R20", "sub-skills must be a non-empty list of repo-relative paths", 1)
|
|
else:
|
|
bad = [x for x in ss if not x.endswith(".md")]
|
|
if bad:
|
|
report.error(path, "R20", f"sub-skills entries must end in '.md': {bad}", 1)
|
|
|
|
# R21 five required sections, in order, each exactly once
|
|
heads = [h for h, _ in headings_in_order(parsed.body)]
|
|
indices: list[int] = []
|
|
for required in ACTION_SKILL_SECTIONS:
|
|
occurrences = [i for i, h in enumerate(heads) if h == required]
|
|
if not occurrences:
|
|
report.error(path, "R21", f"missing required section '## {required}'")
|
|
elif len(occurrences) > 1:
|
|
report.error(path, "R21", f"section '## {required}' appears {len(occurrences)} times; must appear once")
|
|
indices.append(occurrences[0])
|
|
else:
|
|
indices.append(occurrences[0])
|
|
if len(indices) == len(ACTION_SKILL_SECTIONS) and indices != sorted(indices):
|
|
order = [heads[i] for i in indices]
|
|
report.error(path, "R21", f"required sections out of order: {order}; expected {ACTION_SKILL_SECTIONS}")
|
|
|
|
|
|
def validate_meta_skill(path: Path, parsed: Parsed, report: Report) -> None:
|
|
if parsed.frontmatter_error:
|
|
report.error(path, "R01", parsed.frontmatter_error, 1)
|
|
return
|
|
fm = parsed.frontmatter
|
|
assert fm is not None
|
|
missing = META_SKILL_REQUIRED_KEYS - fm.keys()
|
|
if missing:
|
|
report.error(path, "R22", f"missing required meta-skill keys: {sorted(missing)}", 1)
|
|
for k in META_SKILL_REQUIRED_KEYS & fm.keys():
|
|
v = fm[k]
|
|
if v is None or v == "" or v == []:
|
|
report.error(path, "R22", f"meta-skill key '{k}' must not be empty", 1)
|
|
if fm.get("kind") != "meta-skill":
|
|
report.error(path, "R25", f"file in /skills/ is a meta-skill by path but kind is '{fm.get('kind')}', expected 'meta-skill'", 1)
|
|
if "id" in fm and (not isinstance(fm["id"], str) or not KEBAB_CASE.match(fm["id"])):
|
|
report.error(path, "R22", f"id must be lowercase kebab-case: '{fm['id']}'", 1)
|
|
if "version" in fm:
|
|
v = fm["version"]
|
|
if not isinstance(v, int) or isinstance(v, bool) or v <= 0:
|
|
report.error(path, "R22", f"version must be a positive integer: {v!r}", 1)
|
|
|
|
|
|
def validate_entry_skill(path: Path, parsed: Parsed, report: Report) -> None:
|
|
if parsed.frontmatter_error:
|
|
report.error(path, "R01", parsed.frontmatter_error, 1)
|
|
return
|
|
fm = parsed.frontmatter
|
|
assert fm is not None
|
|
missing = ENTRY_SKILL_REQUIRED_KEYS - fm.keys()
|
|
if missing:
|
|
report.error(path, "R23", f"missing required entry-point keys: {sorted(missing)}", 1)
|
|
for k in ENTRY_SKILL_REQUIRED_KEYS & fm.keys():
|
|
v = fm[k]
|
|
if v is None or v == "" or v == []:
|
|
report.error(path, "R23", f"entry-point key '{k}' must not be empty", 1)
|
|
if fm.get("kind") != "entry-point":
|
|
report.error(path, "R25", f"file is /skills/entry.md but kind is '{fm.get('kind')}', expected 'entry-point'", 1)
|
|
if fm.get("id") != "entry":
|
|
report.error(path, "R23", f"entry-point id must be 'entry', got '{fm.get('id')}'", 1)
|
|
if "version" in fm:
|
|
v = fm["version"]
|
|
if not isinstance(v, int) or isinstance(v, bool) or v <= 0:
|
|
report.error(path, "R23", f"version must be a positive integer: {v!r}", 1)
|
|
|
|
|
|
# --- Path and sample checks -------------------------------------------------
|
|
|
|
def classify(path_from_root: Path) -> str | None:
|
|
"""Return 'knowledge' | 'action-skill' | 'meta' | 'entry' | None."""
|
|
parts = path_from_root.parts
|
|
if len(parts) < 2:
|
|
return None
|
|
top = parts[0]
|
|
if top == "skills":
|
|
if len(parts) == 2:
|
|
name = parts[1]
|
|
if name == ENTRY_SKILL_FILE:
|
|
return "entry"
|
|
if name in META_SKILL_FILES:
|
|
return "meta"
|
|
return None
|
|
if top in LAYERS and path_from_root.suffix == ".md":
|
|
if len(parts) >= 3 and parts[1] == "skills":
|
|
return "action-skill"
|
|
if len(parts) >= 4 and parts[1] == "knowledge":
|
|
return "knowledge"
|
|
return None
|
|
|
|
|
|
def validate_knowledge_path(path: Path, root: Path, report: Report) -> None:
|
|
rel = path.relative_to(root)
|
|
parts = rel.parts
|
|
# R13 expected shape: <layer>/knowledge/<domain>/<slug>.md
|
|
if len(parts) != 4:
|
|
report.error(path, "R13", f"knowledge file must live at <layer>/knowledge/<domain>/<slug>.md; got {rel.as_posix()}")
|
|
return
|
|
slug = path.stem
|
|
# R12 filename kebab-case
|
|
if not KEBAB_CASE.match(slug):
|
|
report.error(path, "R12", f"filename slug must be lowercase kebab-case: '{slug}'")
|
|
|
|
|
|
def validate_samples_in_domain(domain_dir: Path, root: Path, report: Report) -> None:
|
|
"""R14: every non-.md file must match <slug>.<kind>.<ext> with <slug>.md present."""
|
|
if not domain_dir.is_dir():
|
|
return
|
|
articles = {p.stem: p for p in domain_dir.glob("*.md")}
|
|
article_slugs = set(articles)
|
|
article_texts: dict[str, str] = {}
|
|
for slug, article in articles.items():
|
|
try:
|
|
article_texts[slug] = article.read_text(encoding="utf-8")
|
|
except UnicodeDecodeError:
|
|
# R01 reports this during the article pass.
|
|
continue
|
|
|
|
for entry in domain_dir.iterdir():
|
|
if not entry.is_file() or entry.suffix == ".md":
|
|
continue
|
|
name = entry.name
|
|
# Expect <slug>.<kind>.<ext>
|
|
m = re.match(r"^(?P<slug>[a-z0-9]+(?:-[a-z0-9]+)*)\.(?P<kind>[a-z0-9]+)\.(?P<ext>[a-z0-9]+)$", name)
|
|
if not m:
|
|
report.error(entry, "R14", f"sample file name must match '<slug>.<kind>.<ext>' with kebab-case slug: '{name}'")
|
|
continue
|
|
slug = m.group("slug")
|
|
kind = m.group("kind")
|
|
if slug not in article_slugs:
|
|
report.error(entry, "R14", f"orphan sample: no matching article '{slug}.md' in {domain_dir.relative_to(root).as_posix()}")
|
|
elif entry.name not in article_texts.get(slug, ""):
|
|
report.error(
|
|
entry,
|
|
"R28",
|
|
f"sample is not referenced by its article '{slug}.md'",
|
|
)
|
|
if kind not in VALID_SAMPLE_KINDS:
|
|
report.warn(entry, "R14", f"non-standard sample kind '{kind}'; standard kinds are {sorted(VALID_SAMPLE_KINDS)}")
|
|
|
|
for slug, article in articles.items():
|
|
for sample_name in SAMPLE_REFERENCE.findall(article_texts.get(slug, "")):
|
|
if not (domain_dir / sample_name).is_file():
|
|
report.error(
|
|
article,
|
|
"R28",
|
|
f"referenced sample does not exist: '{sample_name}'",
|
|
)
|
|
|
|
|
|
# --- Orchestration ----------------------------------------------------------
|
|
|
|
@dataclass
|
|
class SkillRecord:
|
|
path: Path
|
|
kind: str # frontmatter kind
|
|
skill_id: str | None
|
|
|
|
|
|
def validate_sub_skills_registry(path: Path, fm: dict[str, Any], root: Path, report: Report) -> None:
|
|
"""R26: a super-skill's declared `sub-skills` must exactly match the
|
|
`al-*-review.md` leaf files present in the same directory (set equality,
|
|
ordering-agnostic). This keeps the registered leaf list the single source
|
|
of truth and fails CI on a forgotten, stale, or missing registration.
|
|
|
|
Only applies to action-skill files declaring a non-empty list-of-str
|
|
`sub-skills`. Files whose `sub-skills` is malformed are handled by R20.
|
|
"""
|
|
ss = fm.get("sub-skills")
|
|
if not is_non_empty_list_of_str(ss):
|
|
return
|
|
|
|
declared = {s.lstrip("./") for s in ss}
|
|
|
|
# Sibling leaves on disk, excluding the super-skill file itself.
|
|
leaves = {
|
|
p.relative_to(root).as_posix()
|
|
for p in path.parent.glob("al-*-review.md")
|
|
if p.resolve() != path.resolve()
|
|
}
|
|
|
|
# Declared entries that are not real sibling leaves on disk (missing/stale).
|
|
for entry in sorted(declared - leaves):
|
|
entry_path = root / entry
|
|
if not entry_path.exists():
|
|
report.error(path, "R26", f"declared sub-skill does not exist on disk: {entry}", 1)
|
|
else:
|
|
report.error(
|
|
path, "R26",
|
|
f"sub-skills entry is not a sibling 'al-*-review.md' leaf: {entry}", 1,
|
|
)
|
|
|
|
# Sibling leaves on disk that were never registered ('forgot to wire it up').
|
|
for leaf in sorted(leaves - declared):
|
|
report.error(path, "R26", f"leaf not registered in sub-skills: {leaf}", 1)
|
|
|
|
|
|
def run(root: Path) -> Report:
|
|
report = Report()
|
|
skill_records: list[SkillRecord] = []
|
|
action_skill_fms: list[tuple[Path, dict[str, Any]]] = []
|
|
|
|
# Walk declared top-level folders only; avoid wandering into .git, etc.
|
|
walk_roots = [root / "skills"] + [root / layer for layer in LAYERS]
|
|
candidate_files: list[Path] = []
|
|
for wr in walk_roots:
|
|
if wr.exists():
|
|
candidate_files.extend(p for p in wr.rglob("*") if p.is_file())
|
|
|
|
# First pass: classify and validate each file
|
|
for path in candidate_files:
|
|
rel = path.relative_to(root)
|
|
kind = classify(rel)
|
|
if kind is None:
|
|
continue
|
|
try:
|
|
text = path.read_text(encoding="utf-8")
|
|
except UnicodeDecodeError as e:
|
|
report.error(path, "R01", f"file is not valid UTF-8: {e}")
|
|
continue
|
|
parsed = parse_markdown(text)
|
|
|
|
if kind == "knowledge":
|
|
validate_knowledge_path(path, root, report)
|
|
validate_knowledge(path, parsed, report)
|
|
elif kind == "action-skill":
|
|
validate_action_skill(path, parsed, report)
|
|
if parsed.frontmatter:
|
|
action_skill_fms.append((path, parsed.frontmatter))
|
|
if parsed.frontmatter and isinstance(parsed.frontmatter.get("id"), str):
|
|
skill_records.append(SkillRecord(path, "action-skill", parsed.frontmatter["id"]))
|
|
elif kind == "meta":
|
|
validate_meta_skill(path, parsed, report)
|
|
if parsed.frontmatter and isinstance(parsed.frontmatter.get("id"), str):
|
|
skill_records.append(SkillRecord(path, "meta-skill", parsed.frontmatter["id"]))
|
|
elif kind == "entry":
|
|
validate_entry_skill(path, parsed, report)
|
|
if parsed.frontmatter and isinstance(parsed.frontmatter.get("id"), str):
|
|
skill_records.append(SkillRecord(path, "entry-point", parsed.frontmatter["id"]))
|
|
|
|
# Second pass: sample files per knowledge domain
|
|
for layer in LAYERS:
|
|
kn_root = root / layer / "knowledge"
|
|
if not kn_root.is_dir():
|
|
continue
|
|
for domain_dir in kn_root.iterdir():
|
|
if domain_dir.is_dir():
|
|
validate_samples_in_domain(domain_dir, root, report)
|
|
|
|
# Third pass: R24 unique ids within kind
|
|
by_kind: dict[str, dict[str, list[Path]]] = {}
|
|
for rec in skill_records:
|
|
if rec.skill_id is None:
|
|
continue
|
|
by_kind.setdefault(rec.kind, {}).setdefault(rec.skill_id, []).append(rec.path)
|
|
for kind, by_id in by_kind.items():
|
|
for sid, paths in by_id.items():
|
|
if len(paths) > 1:
|
|
for p in paths:
|
|
others = [q.relative_to(root).as_posix() for q in paths if q != p]
|
|
report.error(p, "R24", f"skill id '{sid}' ({kind}) is not unique; also defined in: {others}")
|
|
|
|
# Fourth pass: R26 sub-skills registry matches leaf files on disk
|
|
for path, fm in action_skill_fms:
|
|
validate_sub_skills_registry(path, fm, root, report)
|
|
|
|
return report
|
|
|
|
|
|
def main(argv: list[str] | None = None) -> int:
|
|
parser = argparse.ArgumentParser(description="BCQuality frontmatter and structure validator.")
|
|
parser.add_argument("--root", default=".", help="Repository root (default: current directory).")
|
|
args = parser.parse_args(argv)
|
|
|
|
root = Path(args.root).resolve()
|
|
report = run(root)
|
|
|
|
gha = os.environ.get("GITHUB_ACTIONS") == "true"
|
|
for d in report.diagnostics:
|
|
line = d.format_gha(root) if gha else d.format_plain(root)
|
|
print(line)
|
|
|
|
n_err = len(report.errors)
|
|
n_warn = len(report.warnings)
|
|
print(f"\nValidator: {n_err} error(s), {n_warn} warning(s)")
|
|
return 1 if n_err else 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|