Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
38 changes: 30 additions & 8 deletions codespeed/views.py
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,11 @@

logger = logging.getLogger(__name__)

# Fraction of the plottable benchmark set a revision must have legacy results
# for before gethistoricaldata() will render it as the 'latest' column. Runs
# glitch and drop individual benchmarks, so this is deliberately below 1.0.
LATEST_MIN_COVERAGE = 0.8


def no_environment_error(request):
admin_url = reverse('admin:codespeed_environment_changelist')
Expand Down Expand Up @@ -190,16 +195,33 @@ def gethistoricaldata(request):
project=default_exe.project)
revs = Revision.objects.filter(
branch=default_branch).order_by('-date')[:100]
default_lastrev = None
# The benchmarks the plots can actually use: those the baseline and every
# tagged revision all have legacy results for. A benchmark missing from any
# of them is dropped client side, so it cannot count towards coverage.
plottable = {res.benchmark.name for res in baseline_results[0][1]}
for tag in data['tagged_revs']:
plottable &= {res.benchmark.name for res in default_results[tag]}
required = len(plottable) * LATEST_MIN_COVERAGE

# Accept a revision as 'latest' only once its legacy run has covered enough
# of that set, otherwise fall back to the previous one that has.
for rev in revs:
default_lastrev = rev
if default_lastrev.results.filter(executable=default_exe, environment=env):
break
default_lastrev = None
if default_lastrev is not None:
default_results['latest'] = Result.objects.filter(
executable=default_exe, revision=default_lastrev, environment=env,
latest_results = Result.objects.filter(
executable=default_exe, revision=rev, environment=env,
benchmark__source='legacy')
covered = plottable & {res.benchmark.name for res in latest_results}
if covered and len(covered) >= required:
default_results['latest'] = latest_results
break
logger.info(
"skipping '%s' as 'latest': %d of %d plottable benchmarks have "
"legacy results for '%s'" % (
str(rev), len(covered), len(plottable), str(env)))
else:
logger.error(
"no revision covers %.0f%% of the %d plottable benchmarks for "
"'%s' '%s'" % (LATEST_MIN_COVERAGE * 100, len(plottable),
str(default_exe), str(env)))

# Collect data
benchmarks = []
Expand Down
1 change: 1 addition & 0 deletions speed_pypy/settings.py
Original file line number Diff line number Diff line change
Expand Up @@ -91,6 +91,7 @@
SHOW_HISTORICAL = True
DEF_BASELINES = [
{'executable': 'cpython', 'revision': '3.11.15'},
{'executable': 'cpython', 'revision': '3.12.13'},
]
DEF_EXECUTABLES = [
{'name': 'pypy3.12-jit-64', 'project': 'PyPy3.12'},
Expand Down
11 changes: 10 additions & 1 deletion speed_pypy/templates/embed_comparison.html
Original file line number Diff line number Diff line change
Expand Up @@ -48,12 +48,21 @@
tagged_data[i].push(data['results'][benchname][rev] / data['results'][benchname][data['baseline']]);
}
if (!add_to_tagged_data) { continue; }
benchmarks.push(benchname);
var rel = data['results'][benchname]['latest'] / data['results'][benchname][data['baseline']];
// Drop a benchmark the latest run is missing before pushing its
// label, so the bars stay aligned with the x axis ticks.
// (!(rel > 0) also catches the NaN from a missing value.)
if (!(rel > 0)) { continue; }
benchmarks.push(benchname);
latestValues.push(rel);
baselineValues.push(1.0);
}

if (benchmarks.length === 0) {
wrap.innerHTML = 'No complete benchmark run available yet';
return;
}

var canvas1 = document.createElement('canvas');
wrap.appendChild(canvas1);
new Chart(canvas1, {
Expand Down
20 changes: 16 additions & 4 deletions speed_pypy/templates/home.html
Original file line number Diff line number Diff line change
Expand Up @@ -94,13 +94,25 @@ <h3>How has PyPy performance evolved over time?</h3>
}
if (!add_to_tagged_data) { continue; }
var rel = data['results'][benchname]['latest'] / data['results'][benchname][data['baseline']];
// Only count what we multiply. A benchmark the latest run is
// missing would otherwise inflate the root below and drag the
// geometric mean towards 1.0. (!(rel > 0) also catches the NaN
// from a missing value.)
if (!(rel > 0)) { continue; }
latestValues.push(rel);
if (rel > 0 && !isNaN(rel)) { trunk_geomean *= rel; }
trunk_geomean *= rel;
}

trunk_geomean = Math.pow(trunk_geomean, 1 / latestValues.length);
$('#geomean').html(trunk_geomean.toFixed(2));
$('#geofaster').html((1 / trunk_geomean).toFixed(1));
if (latestValues.length === 0) {
// No usable latest run: report nothing rather than a bogus 1.0x.
trunk_geomean = NaN;
$('#geomean').html('n/a');
$('#geofaster').html('n/a');
} else {
trunk_geomean = Math.pow(trunk_geomean, 1 / latestValues.length);
$('#geomean').html(trunk_geomean.toFixed(2));
$('#geofaster').html((1 / trunk_geomean).toFixed(1));
}

// Plot 1 (per-benchmark normalized comparison) is rendered by the
// embed_comparison page shown in the iframe above.
Expand Down
Loading