{% set c = comparison %}

{{ a }} vs {{ b }} raw metrics, no composite score

{% for d in c.deltas %} {% if d.metric == "success_rate" %} {% elif d.metric in ("reported_cost_usd", "estimated_cost_usd") %} {% elif d.metric == "wall_time_seconds" %} {% else %} {% endif %} {% endfor %}
metric{{ a }}{{ b }}Δ (B − A)
{{ d.label }}{% if d.lower_is_better %} (lower is better){% endif %}{{ d.a|pct }}{{ d.b|pct }} {% if d.delta is not none %}{{ (d.delta * 100)|delta(0) }} pts{% else %}—{% endif %}{{ d.a|cost }}{{ d.b|cost }} {{ d.delta|delta(4) }}{{ d.a|seconds }}{{ d.b|seconds }} {{ d.delta|delta(1) }} s{{ d.a|num }}{{ d.b|num }} {{ d.delta|delta }}
valid runs{{ c.a_aggregate.n_valid }} / {{ c.a_aggregate.n_total }}{{ c.b_aggregate.n_valid }} / {{ c.b_aggregate.n_total }}

Statistical evidence paired by task, cluster bootstrap over tasks

{% set p = c.paired %} {% include "partials/paired.html" %}

Per-task agreement

both passed: {{ c.summary.both_passed }} both failed: {{ c.summary.both_failed }} {{ a }} passed / {{ b }} failed: {{ c.summary.a_only }} {{ b }} passed / {{ a }} failed: {{ c.summary.b_only }} {% if c.summary.mixed %}mixed (repetitions disagree): {{ c.summary.mixed }}{% endif %} {% if c.summary.unverified %}unverified: {{ c.summary.unverified }}{% endif %}

{% for t in c.tasks %} {% set ca = matrix[t.task_key][a] %}{% set cb = matrix[t.task_key][b] %} {% endfor %}
task{{ a }}{{ b }}category
{{ t.task_key }} {% if ca.primary_run %}{{ ca.state|state_symbol }} {{ t.a_score|num }}{% else %}—{% endif %} {% if cb.primary_run %}{{ cb.state|state_symbol }} {{ t.b_score|num }}{% else %}—{% endif %} {{ t.category|replace("_", " ") }}