mirror of
https://github.com/agentskills/agentskills.git
synced 2026-06-18 15:54:06 +08:00
863f7c2857
A how-to guide for evaluating skill output quality using structured evals. Covers the full eval workflow: designing test cases, running with-skill vs. baseline comparisons, writing assertions, LLM-based grading, aggregating benchmarks, analyzing patterns, human review, and LLM-driven iterative improvement. Derived from the workflow implemented by the `skill-creator` Skill, but written as a standalone guide that readers can follow without using that tool. Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
50 lines
907 B
JSON
50 lines
907 B
JSON
{
|
|
"$schema": "https://mintlify.com/docs.json",
|
|
"theme": "mint",
|
|
"name": "Agent Skills",
|
|
"colors": {
|
|
"primary": "#7f7f7f",
|
|
"light": "#bfbfbf",
|
|
"dark": "#404040"
|
|
},
|
|
"favicon": "/favicon.svg",
|
|
"navbar": {
|
|
"primary": {
|
|
"type": "github",
|
|
"href": "https://github.com/agentskills/agentskills"
|
|
}
|
|
},
|
|
"navigation": {
|
|
"pages": [
|
|
"home",
|
|
"what-are-skills",
|
|
"specification",
|
|
{
|
|
"group": "For skill creators",
|
|
"pages": [
|
|
"skill-creation/evaluating-skills",
|
|
"skill-creation/using-scripts"
|
|
]
|
|
},
|
|
{
|
|
"group": "For client implementors",
|
|
"pages": [
|
|
"integrate-skills"
|
|
]
|
|
}
|
|
]
|
|
},
|
|
"contextual": {
|
|
"options": [
|
|
"copy",
|
|
"view",
|
|
"chatgpt",
|
|
"claude",
|
|
"perplexity",
|
|
"mcp",
|
|
"cursor",
|
|
"vscode"
|
|
]
|
|
}
|
|
}
|