ML — ablation study
ML evaluation vertical — models, datasets, evals, ablations. Typed FlowScript for machine-learning evaluation. Keywords: ablation, benchmark, evaluation, model comparison.
Make it your own.
// ML evaluation vertical — models, datasets, evals, ablations.
// The ablations view at the foot of the document draws through the
// ablation_table renderer (ablation is its registered alias).
model base {
title: "Flow-v1 base"
params: 7000000000
family: "decoder-only transformer"
}
dataset internal_eval {
title: "FlowBench v2"
n: 1240
split: "held-out"
}
eval base_score {
model: base
dataset: internal_eval
metric: "exact-match accuracy"
score: 0.612
}
ablation no_typed_attrs {
base: base_score
delta: -0.084
change: "Strip semantic types — fall back to generic boxes"
}
ablation no_propagation {
base: base_score
delta: -0.051
change: "Disable arithmetic propagation"
}
ablation no_critique {
base: base_score
delta: -0.039
change: "Skip critic step in Architect mode"
}
view ablations: ablation_table(base_score)