{
  "$schema": "https://json-schema.org/draft/2020-12/schema",
  "$id": "manni:evals:1.0.0",
  "title": "manni evals vocabulary v1.0.0",
  "description": "Page-level eval declarations: named, testable assertions about a page, each graded by code, an AI judge (a model or an agent), or a human. A manni common vocabulary: any grader, judge, or CI tool can implement it, and other schemas may compose on top of it. It adopts the house conventions - kebab-case keys, flat page-level facts - and claims three keys: `evals` (one assertion or the list), `eval-suite`, and `eval-skip`. There is no settings container: settings are page-level facts, and generation attribution lives in manni:ai-context:1.0.0. There, `provenance` names the machines that wrote the page's body, and `meta-provenance` the machines that proposed its metadata. Machine-proposed evals are attributed in that page-level `meta-provenance`, with the proposed ids under its `evals`. The page root stays open so sibling tools' keys pass untouched, but the `eval-` prefix is reserved and enforced here: an unrecognized `eval-*` key is rejected, giving the open root the loud-typo property a closed container would have.",
  "type": "object",
  "additionalProperties": true,
  "properties": {
    "evals": {
      "type": [
        "string",
        "array"
      ],
      "minLength": 1,
      "minItems": 1,
      "items": {
        "$ref": "#/$defs/evalEntry"
      },
      "description": "The evals for this page: one AI-judged assertion as a string, or a non-empty list of entries — each a string shorthand, a `use:` reference to a config-defined eval, or an inline definition. An empty list is not a declaration; a page with no evals omits the key. Ids are unique within a page: two entries claiming one id is an error, not a last-one-wins override. JSON Schema cannot express that over a mixed-shape array - `uniqueItems` is whole-item equality - so it is a requirement on implementations, stated here so every tool agrees on the rule.",
      "x-manni-location": "external"
    },
    "eval-suite": {
      "type": "string",
      "minLength": 1,
      "description": "Named suite from the config whose evals apply to this page, alongside anything in `evals`. Page entries win on id collision. A suite assignment is a page-level fact.",
      "x-manni-location": "external"
    },
    "eval-skip": {
      "type": "boolean",
      "default": false,
      "description": "Skip this page's evals entirely — reported as skipped, never silently ignored. Default false.",
      "x-manni-location": "external"
    }
  },
  "$defs": {
    "evalEntry": {
      "if": {
        "type": "string"
      },
      "then": {
        "type": "string",
        "minLength": 1
      },
      "else": {
        "if": {
          "type": "object",
          "required": [
            "use"
          ]
        },
        "then": {
          "$ref": "#/$defs/evalReference"
        },
        "else": {
          "$ref": "#/$defs/inlineEval"
        },
        "description": "An object entry: a `use:` reference to a config-defined eval, else an inline definition. Branched with if/then rather than oneOf so a fault inside the object form is reported once, against the key that is wrong — the same rule the starlight built-in records."
      },
      "description": "One eval: a string shorthand (an AI-judged assertion at error severity), or an object."
    },
    "evalReference": {
      "type": "object",
      "required": [
        "use"
      ],
      "properties": {
        "use": {
          "type": "string",
          "minLength": 1,
          "description": "Name of an eval defined in the config."
        },
        "type": {
          "$ref": "#/$defs/evalType"
        },
        "skip": {
          "type": "boolean"
        },
        "severity": {
          "$ref": "#/$defs/severity"
        },
        "weight": {
          "$ref": "#/$defs/weight"
        },
        "options": {
          "type": "object"
        }
      },
      "additionalProperties": false,
      "description": "A page joining an eval the config defines, with the overrides that are statements about *this page's relationship to that check*: whether it applies (`skip`), what kind of claim it is here (`type`), how badly a failure matters (`severity`), what it measures here (`options`), and how much it counts toward an aggregate (`weight`). Deliberately absent: `provider`, `model` and `runs`, which say how the tool executes rather than what this page claims - a corpus where each page picks its own judge model is one no operator can cost or reason about - and `target`, because changing which bytes a *named* eval reads makes one name mean two different things in one corpus, which is the confusion `use:` exists to prevent. An eval that needs a different subject is a different eval."
    },
    "inlineEval": {
      "type": "object",
      "required": [
        "id"
      ],
      "properties": {
        "id": {
          "type": "string",
          "pattern": "^[a-z0-9][a-z0-9-]*$",
          "description": "Unique (per page) kebab-case identifier for this eval. `id` matches the page-level `id` convention — in this family, `name` would mean a display name."
        },
        "assertion": {
          "type": "string",
          "minLength": 1,
          "description": "The testable assertion, in plain language."
        },
        "type": {
          "$ref": "#/$defs/evalType"
        },
        "grader": {
          "$ref": "#/$defs/grader"
        },
        "provider": {
          "type": "string",
          "minLength": 1,
          "description": "Which configured inference provider or agent judges an `ai`-graded eval — a model endpoint or an agent runner. Legal values are whatever the inference dependency supports, validated at run time; omit for the config default. Meaningful only on `ai` evals."
        },
        "model": {
          "type": "string",
          "minLength": 1,
          "description": "Model that judges an `ai`-graded eval, within `provider`. Legal values are whatever the inference dependency supports, validated at run time; omit for the config default. Two uses: an eval that matters can name a stronger judge than the corpus default, and an eval can name a judge other than the machines that wrote what it grades - the escape hatch for the self-preference-bias check (manni:ai-context:1.0.0), which compares the judge's model with the machines in `provenance` for `target: body`, in `meta-provenance` for `target: frontmatter`, and in both for `target: raw`. Meaningful only on `ai` evals."
        },
        "runs": {
          "type": "integer",
          "minimum": 1,
          "maximum": 50,
          "description": "Ensemble runs for this eval, overriding the tool's configured default. Lets a high-stakes or known-flaky assertion buy more agreement than a cheap one, which a corpus-wide setting cannot. Capped because it multiplies cost directly. Meaningful only on `ai` evals."
        },
        "evidence": {
          "type": "string",
          "minLength": 1,
          "description": "Hint scoping where the judge should look for evidence."
        },
        "target": {
          "$ref": "#/$defs/target"
        },
        "examples": {
          "type": "object",
          "properties": {
            "pass": {
              "$ref": "#/$defs/anchorList"
            },
            "fail": {
              "$ref": "#/$defs/anchorList"
            }
          },
          "additionalProperties": false,
          "description": "Anchor examples of passing and failing content for the judge — one or several each."
        },
        "command": {
          "type": "array",
          "minItems": 1,
          "items": {
            "type": "string",
            "minLength": 1
          },
          "description": "Command to execute for command-graded evals. {file} is replaced with the page path, and exit 0 (or `success-exit-codes`) is a pass. Omitting it is the generation contract: tooling generates a check script from the assertion, then writes `command` and `generated-assertion-hash` back. Legal only alongside `grader: command`."
        },
        "success-exit-codes": {
          "type": "array",
          "minItems": 1,
          "items": {
            "type": "integer"
          },
          "default": [
            0
          ],
          "description": "Exit codes that count as a pass."
        },
        "timeout-ms": {
          "type": "integer",
          "minimum": 1,
          "description": "Wall-clock budget for the command, in milliseconds."
        },
        "generated-assertion-hash": {
          "type": "string",
          "description": "sha256 of the assertion when the check script was generated — written back alongside `command`, and never legal without it: a hash with no command is a half write-back, and a command with a differing hash triggers regeneration."
        },
        "options": {
          "type": "object",
          "description": "Grader-specific options; validated by the grader at run time."
        },
        "severity": {
          "$ref": "#/$defs/severity"
        },
        "weight": {
          "$ref": "#/$defs/weight"
        },
        "severity-map": {
          "type": "object",
          "additionalProperties": {
            "$ref": "#/$defs/severity"
          },
          "description": "Maps a tool grader's own severities onto eval severities. Meaningful only on `tool:*` evals."
        },
        "skip": {
          "type": "boolean"
        }
      },
      "additionalProperties": false,
      "dependentRequired": {
        "generated-assertion-hash": [
          "command"
        ]
      },
      "allOf": [
        {
          "if": {
            "properties": {
              "grader": {
                "const": "ai"
              }
            },
            "required": [
              "grader"
            ]
          },
          "then": {
            "required": [
              "assertion"
            ]
          }
        },
        {
          "if": {
            "not": {
              "required": [
                "grader"
              ]
            }
          },
          "then": {
            "required": [
              "assertion"
            ]
          }
        },
        {
          "if": {
            "properties": {
              "grader": {
                "const": "command"
              }
            },
            "required": [
              "grader"
            ]
          },
          "then": {
            "anyOf": [
              {
                "required": [
                  "assertion"
                ]
              },
              {
                "required": [
                  "command"
                ]
              }
            ]
          }
        },
        {
          "if": {
            "properties": {
              "grader": {
                "const": "human"
              }
            },
            "required": [
              "grader"
            ]
          },
          "then": {
            "required": [
              "assertion"
            ]
          }
        },
        {
          "if": {
            "required": [
              "command"
            ]
          },
          "then": {
            "properties": {
              "grader": {
                "const": "command"
              }
            },
            "required": [
              "grader"
            ],
            "description": "An entry carrying an executable command must say grader: command — an execution payload on an ai or human eval is either dead weight or a surprise, depending on which field the runner keys on."
          }
        },
        {
          "if": {
            "anyOf": [
              {
                "required": [
                  "success-exit-codes"
                ]
              },
              {
                "required": [
                  "timeout-ms"
                ]
              }
            ]
          },
          "then": {
            "properties": {
              "grader": {
                "const": "command"
              }
            },
            "required": [
              "grader"
            ],
            "description": "success-exit-codes and timeout-ms configure command execution, so they are legal only alongside grader: command — with or without the command itself, since the generation contract writes the command later. On an ai or human eval they are dead weight the runner never reads."
          }
        },
        {
          "if": {
            "anyOf": [
              {
                "required": [
                  "provider"
                ]
              },
              {
                "required": [
                  "model"
                ]
              },
              {
                "required": [
                  "runs"
                ]
              }
            ]
          },
          "then": {
            "properties": {
              "grader": {
                "const": "ai"
              }
            },
            "description": "provider, model and runs configure the AI judge, so they are legal only on an `ai` eval. `grader` is not required alongside them because `ai` is the default - omitting it is the common case, and the constraint is vacuous when it is absent."
          }
        }
      ]
    },
    "evalType": {
      "enum": [
        "capability",
        "regression"
      ],
      "default": "regression",
      "description": "`capability` probes a boundary and is expected to fail sometimes; `regression` protects behavior that already works."
    },
    "severity": {
      "enum": [
        "error",
        "warning",
        "notice"
      ],
      "default": "error",
      "description": "Only error-severity failures fail the run; warning and notice report but pass. The three levels are the family scale in manni's shared severity, so a finding reads the same in every tool. Default error — the string shorthand implies it."
    },
    "weight": {
      "type": "number",
      "exclusiveMinimum": 0,
      "default": 1,
      "description": "Relative contribution to an aggregate score across evals. Does not change this eval's own pass/fail, nor its severity: an eval still passes or fails on its own terms, and weight only decides how much that outcome moves a suite or run total. Zero is excluded deliberately - a weightless eval is a silent disable, and `skip` already means that, loudly."
    },
    "grader": {
      "type": "string",
      "pattern": "^(ai|command|human|tool:[a-z0-9][a-z0-9-]*)$",
      "default": "ai",
      "description": "How the assertion is checked: an AI judge (`ai`) — a direct inference provider or an agent, selected per-eval with `provider`; an executable check (`command`), generated from the assertion when no `command` is given; a human review queue (`human`); or a named tool integration (`tool:<kebab>`) — an open family, extensible without touching this schema. The judge may be an agent rather than a bare model — hence `ai`, not `llm`. The structural kinds stay closed because the conditionals above switch on them."
    },
    "anchorList": {
      "if": {
        "type": "array"
      },
      "then": {
        "type": "array",
        "minItems": 1,
        "uniqueItems": true,
        "items": {
          "type": "string",
          "minLength": 1
        }
      },
      "else": {
        "type": "string",
        "minLength": 1
      },
      "description": "One anchor example, or a non-empty list of them."
    },
    "target": {
      "if": {
        "type": "object"
      },
      "then": {
        "type": "object",
        "additionalProperties": false,
        "required": [
          "source",
          "path"
        ],
        "properties": {
          "source": {
            "const": "file"
          },
          "path": {
            "type": "string",
            "minLength": 1,
            "description": "Path to the file to grade, relative to the page's directory."
          }
        }
      },
      "else": {
        "enum": [
          "body",
          "raw",
          "frontmatter"
        ]
      },
      "default": "body",
      "description": "What the grader receives: the page body with frontmatter stripped (`body`), the file verbatim including frontmatter (`raw`), the parsed frontmatter alone (`frontmatter`), or a companion file. Distinct from `evidence`, which hints where to look within what is graded - this selects the bytes themselves, and it is structural, so deterministic graders honour it too. Named `target` rather than `focus` for that reason: a regex grader has no focus, it has a target. Branched with if/then rather than oneOf so a fault inside the object form is reported once, against the key that is wrong."
    }
  },
  "patternProperties": {
    "^eval-(?!suite$|skip$)": false
  }
}
