Evaluation Suite Tree
Hierarchy of evals: capability + safety + behaviour + agentic.
Rendering…
Make it your own.
digraph evals {
rankdir=TB;
graph [bgcolor=transparent];
node [shape=box, style="rounded,filled", fontname=Inter, fontsize=10];
Root [label="Eval suite", fillcolor="#fce7f3"];
Cap [label="Capability", fillcolor="#dbeafe"];
Safe [label="Safety", fillcolor="#fee2e2"];
Beh [label="Behaviour", fillcolor="#fef3c7"];
Agt [label="Agentic", fillcolor="#ede9fe"];
Cap_R [label="Reasoning (MATH, GPQA)"];
Cap_C [label="Code (HumanEval, SWE-bench)"];
Cap_K [label="Knowledge (MMLU)"];
Safe_H [label="Harms refusal"];
Safe_B [label="Bias / fairness"];
Safe_J [label="Jailbreak resistance"];
Beh_T [label="Tone + helpfulness"];
Beh_I [label="Instruction following"];
Agt_W [label="Web tasks"];
Agt_C [label="Tool use accuracy"];
Root -> Cap -> Cap_R; Cap -> Cap_C; Cap -> Cap_K;
Root -> Safe -> Safe_H; Safe -> Safe_B; Safe -> Safe_J;
Root -> Beh -> Beh_T; Beh -> Beh_I;
Root -> Agt -> Agt_W; Agt -> Agt_C;
}