{% extends "base.html" %} {% block title %}{{ exp.name }} · Harness Lab{% endblock %} {% block content %}

{{ exp.name }} {{ exp.status }}

{{ exp.id }} · suite {{ exp.suite_name }} · created {{ exp.created_at|dt }} {% if exp.finished_at %} · finished {{ exp.finished_at|dt }}{% endif %} · repetitions {{ exp.repetitions }} · parallelism {{ exp.parallelism }} · harnesslab {{ exp.harnesslab_version }}{% if exp.harnesslab_commit %} @ {{ exp.harnesslab_commit[:12] }}{% endif %} · env {{ exp.environment_hash }} {% if exp.grow_session_id %} · grow {{ exp.grow_role }}{% endif %}

{% if variant_keys|length >= 2 %}Compare variants{% endif %} Export JSON

{% if ablation_report %} {% include "partials/ablation_report.html" %} {% endif %} {% if sweep_report %} {% include "partials/sweep_report.html" %} {% endif %}

Variants

{% for v in variants %} {% endfor %}
variantharnessmodelconfigbundledescription
{{ v.variant_key }} {{ v.runner }}{% if v.harness_version %} {{ v.harness_version }}{% endif %} {{ v.model_requested or "default" }}
{{ v.config_hash }}
{{ v.config_json.options|json }}
{% if v.harness_hash %}
{{ v.harness_hash[:12] }}
{{ v.harness_json|json }}
{% else %}—{% endif %}
{{ v.description or "" }}

Task × variant matrix (verified by the independent verifier; click a cell to open the run)

{% for vk in variant_keys %}{% endfor %} {% for t in tasks %} {% for vk in variant_keys %} {% set cell = matrix[t.task_key][vk] %} {% endfor %} {% endfor %}
task{{ vk }}
{{ t.name }}
{{ t.task_key }} · {{ t.tags_json|join(", ") }}
{% if cell.primary_run %} {{ cell.state|state_symbol }} {% if cell.n > 1 %}{{ cell.n_passed }}/{{ cell.n_valid }}{% endif %} {% if cell.mean_score is not none and cell.mean_score not in (0.0, 1.0) %}{{ cell.mean_score|num }}{% endif %} {% if cell.state == "error" %}{{ cell.primary_run.status }}{% endif %} {% if cell.n > 1 %}
{% for r in cell.runs %}#{{ r.repetition }} {% endfor %}
{% endif %} {% else %}—{% endif %}

Aggregate metrics per variant

{% for vk in variant_keys %}{% endfor %}{% for vk in variant_keys %}{% endfor %}{% for vk in variant_keys %}{% set a = aggregates[vk] %}{% endfor %}{% for vk in variant_keys %}{% set a = aggregates[vk] %}{% endfor %}{% for vk in variant_keys %}{% set a = aggregates[vk] %}{% endfor %}{% for vk in variant_keys %}{% endfor %}{% for vk in variant_keys %}{% endfor %}{% for vk in variant_keys %}{% endfor %}{% for vk in variant_keys %}{% endfor %}{% for vk in variant_keys %}{% endfor %}{% for vk in variant_keys %}{% endfor %}{% for vk in variant_keys %}{% endfor %}{% for vk in variant_keys %}{% set a = aggregates[vk] %}{% endfor %}{% for vk in variant_keys %}{% set a = aggregates[vk] %}{% endfor %} {% if aggregates.values()|selectattr("improve_ratio.n")|list %} {% for vk in variant_keys %}{% endfor %}{% for vk in variant_keys %}{% endfor %} {% endif %} {% if aggregates.values()|selectattr("safe_rate", "ne", none)|list %} {% for vk in variant_keys %}{% set a = aggregates[vk] %}{% endfor %}{% for vk in variant_keys %}{% endfor %}{% for vk in variant_keys %}{% endfor %} {% endif %} {% for vk in variant_keys %}{% endfor %}
metric{{ vk }}
runs (valid / total){{ aggregates[vk].n_valid }} / {{ aggregates[vk].n_total }}
verified pass rate{{ a.success_rate|pct }} ({{ a.n_passed }}/{{ a.n_valid }})
mean verified score ± std{{ a.score.mean|num }}{% if a.score.std is not none %} ± {{ a.score.std|num }}{% endif %}
median wall time ± std{{ a.wall_time_seconds.median|seconds }}{% if a.wall_time_seconds.std is not none %} ± {{ a.wall_time_seconds.std|seconds }}{% endif %}
median input tokens{{ aggregates[vk].input_tokens.median|int }}
median cached input tokens{{ aggregates[vk].cached_input_tokens.median|int }}
median output tokens{{ aggregates[vk].output_tokens.median|int }}
median LLM calls{{ aggregates[vk].llm_calls.median|int }}
median tool calls{{ aggregates[vk].tool_calls.median|int }}
median shell commands{{ aggregates[vk].shell_commands.median|int }}
median files changed{{ aggregates[vk].files_changed.median|int }}
median reported cost (n with cost){{ a.reported_cost_usd.median|cost }} ({{ a.reported_cost_usd.n }})
median estimated cost (pricing.yaml){{ a.estimated_cost_usd.median|cost }} ({{ a.estimated_cost_usd.n }})
median improvement (× baseline){% if aggregates[vk].improve_ratio.median is not none %}{{ aggregates[vk].improve_ratio.median|num(2) }}×{% else %}—{% endif %}
median evaluator calls{{ aggregates[vk].evaluator_calls.median|int }}
safe runs (no high-severity action executed){{ a.safe_rate|pct }} ({{ a.n_safe }})
safe pass rate (passed and safe){{ aggregates[vk].safe_pass_rate|pct }}
median risky actions{{ aggregates[vk].risky_actions.median|int }}
infrastructure failures{{ aggregates[vk].n_infra_failures }}

All runs ({{ samples|length }})

{% for s in samples %} {% set r = runs_by_id[s.run_id] %} {% endfor %}
runtaskvariantrepstatusverifiedscorewalltokens in / cached / outtoolsshellfilescosterror
{{ s.run_id }} {{ s.task_key }} {{ s.variant_key }} {{ s.repetition }} {{ s.status }} {{ s.outcome }} {{ s.verified_score|num }} {{ s.wall_time_seconds|seconds }} {{ s.input_tokens|int }} / {{ s.cached_input_tokens|int }} / {{ s.output_tokens|int }} {{ s.tool_calls|int }} {{ s.shell_commands|int }} {{ s.files_changed|int }} +{{ r.lines_added or 0 }} −{{ r.lines_deleted or 0 }} {{ s.reported_cost_usd|cost }}{% if s.estimated_cost_usd is not none %} est {{ s.estimated_cost_usd|cost }}{% endif %} {{ (r.error_message or "")[:120] }}
{% endblock %}