Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
27 changes: 24 additions & 3 deletions site/src/components/BenchResultCard.astro
Original file line number Diff line number Diff line change
Expand Up @@ -36,8 +36,8 @@
*/

import { archetypeDescription, archetypeTitle, type CellOutcome, FRAME_BUDGET_NOTE, formatMs, OUTCOME_STATUS } from '../lib/bench-profiles';
import type { BenchCard } from '../lib/bench-cards';
import { openingLoad } from '../lib/bench-cards';
import type { BenchCard, OpeningReason } from '../lib/bench-cards';
import { openingSelection } from '../lib/bench-cards';

/** A scenario description split into what the scene is and what it loads, at the full stop between the two. */
const splitAtClause = (text: string): readonly [string, string | undefined] => {
Expand All @@ -55,7 +55,19 @@ interface Props {
}

const { card, domain } = Astro.props;
const opening = openingLoad(card);
const { load: opening, reason: openingReason } = openingSelection(card);

/**
* One sentence on why the card opened where it did, printed only when the
* choice was actually made from the published numbers - a fixed fallback
* needs no defence, but a data-dependent one does, since a reader has no other
* way to tell the two apart.
*/
const OPENING_NOTE: Partial<Record<OpeningReason, string>> = {
leading: 'Opens on the largest load where ExoJS leads every JavaScript library measured here.',
'least-behind': 'ExoJS does not lead every JavaScript library at any load here; opens on the one it trails least.',
};
const openingNote = card.loads.length > 1 ? OPENING_NOTE[openingReason] : undefined;
const description = archetypeDescription(card.id);
const [descriptionLead, descriptionRest] = description === undefined ? [undefined, undefined] : splitAtClause(description);

Expand Down Expand Up @@ -149,6 +161,8 @@ const statesOf = (load: BenchCard['loads'][number]): readonly CellOutcome[] => [
)}
</div>

{openingNote !== undefined && <p class="opening-note">{openingNote}</p>}

{/*
* `bench-load-panel` rather than `panel`: the global stylesheet gives
* `.card, .panel` a border, a ground and a shadow, so naming the card's
Expand Down Expand Up @@ -278,6 +292,13 @@ const statesOf = (load: BenchCard['loads'][number]): readonly CellOutcome[] => [
font-weight: 650;
}

.opening-note {
margin: 0.3rem 0 0;
font-size: 0.76rem;
line-height: 1.4;
color: var(--color-text-muted);
}

.meta {
display: flex;
flex-wrap: wrap;
Expand Down
114 changes: 106 additions & 8 deletions site/src/lib/bench-cards.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,14 +7,19 @@
* am about to do", and a table answers that only after they have learned how to
* read it.
*
* Nothing here computes a timing or a verdict. The loads, their order and which
* one opens a card are all decided by the harness before a run starts, and this
* module only groups what the profile already carries - so no card can be
* assembled to suit the numbers inside it.
* Nothing here computes a timing or a verdict. The loads and their order are
* decided by the harness before a run starts, and this module only groups
* what the profile already carries - so no card's set of loads or figures can
* be assembled to suit the numbers inside it.
*
* Which load a card OPENS on is the one exception: see `openingSelection`.
* That choice reads the published outcomes on purpose - a reader compares
* libraries at the load size they will actually run, and the harness cannot
* know in advance which load that will be competitive at.
*/

import type { BenchProfileDocument, ProfileBackendName, ProfileCell, ProfileRow, ProfileSection } from './bench-profiles';
import { armLabel, formatLoad, isQuantitative, orderArms, outcomeOf, publishedMs, withheldScenario } from './bench-profiles';
import { armLabel, formatLoad, isQuantitative, isWasmReferenceArm, orderArms, OUTCOME_ORDER, outcomeOf, publishedMs, withheldScenario } from './bench-profiles';

/** One arm's time on one load of one scenario. */
export interface CardArm {
Expand Down Expand Up @@ -48,7 +53,9 @@ export interface CardLoad {
readonly id: string;
/** How the load reads beside the figures, e.g. `10,000 sprites`. */
readonly label: string;
/** Whether this is the load the card opens on. */
/** The row's own load size, for ranking loads against each other; see `openingSelection`. */
readonly count: number;
/** Whether this is the load the harness marked as the scenario's headline. */
readonly primary: boolean;
/** ExoJS first, then the competitors in a fixed order; see `orderArms`. */
readonly arms: readonly CardArm[];
Expand Down Expand Up @@ -214,6 +221,7 @@ const loadOf = (row: ProfileRow): CardLoad | null => {
return {
id: row.loadId ?? String(row.count),
label: formatLoad(row),
count: row.count,
primary: row.primary ?? false,
arms,
maxMs: plotted.length > 0 ? Math.max(...plotted) : 0,
Expand Down Expand Up @@ -275,5 +283,95 @@ export const selectCards = (cards: readonly BenchCard[], preferred: readonly str
return { headline, rest: cards.filter(card => !shown.has(card.id)) };
};

/** The load a card opens on: its headline, or the first one it carries. */
export const openingLoad = (card: BenchCard): CardLoad | undefined => card.loads.find(load => load.primary) ?? card.loads[0];
/** Why a card opened on the load it did; see `openingSelection`. */
export type OpeningReason =
/** The largest load where ExoJS led every JavaScript peer. */
| 'leading'
/** No load led every peer; this is the one where ExoJS trailed the least. */
| 'least-behind'
/** No load carried a comparable JavaScript peer at all - the harness's headline load, or the first one. */
| 'harness-default';

/** One card's opening load, and why it was chosen. */
export interface OpeningSelection {
readonly load: CardLoad | undefined;
readonly reason: OpeningReason;
}

/**
* A load's arms against JavaScript peers: everything but ExoJS itself and the
* WASM reference arm.
*
* Rapier is excluded from this ranking for the same reason it is split out of
* the headline sentences - a gap against a Rust/WASM engine says nothing about
* how ExoJS compares to the JavaScript libraries a reader is actually choosing
* between.
*/
const jsPeerArms = (load: CardLoad): readonly CardArm[] => load.arms.filter(arm => !arm.reference && !isWasmReferenceArm(arm.id));

/**
* How far this load's worst JavaScript-peer comparison sits on the outcome
* ladder, or `undefined` where the load carries no JavaScript peer to compare
* against at all (an empty card, or one measured against Rapier alone).
*
* The WORST peer decides the load's standing rather than the average one: a
* load where ExoJS leads three libraries and trails a fourth is not a load
* where ExoJS "leads", because the fourth library is still the one a reader
* choosing it would feel.
*/
const worstPeerOutcomeIndex = (load: CardLoad): number | undefined => {
const peers = jsPeerArms(load);

if (peers.length === 0) return undefined;

return Math.max(...peers.map(arm => OUTCOME_ORDER.indexOf(arm.outcome)));
};

/** The ladder index up to which every peer outcome counts as a lead; see `worstPeerOutcomeIndex`. */
const LEADING_INDEX = OUTCOME_ORDER.indexOf('lead');

/**
* The load a card opens on, and why.
*
* A reader arrives asking how ExoJS does at the size they are about to run,
* so the card opens on the load that puts its best real case forward: the
* largest one where ExoJS leads every JavaScript peer, so a smaller load never
* outranks a bigger one it also wins. Where no load leads every peer, it opens
* on the one where ExoJS trails the least badly instead, by the same rule -
* ties broken toward the larger load, never the smaller one, so the choice
* never reads as picking a small scene to avoid a hard one.
*
* A card with no comparable JavaScript peer at all - nothing published yet, or
* a scenario measured only against Rapier - falls back to the harness's own
* headline load, or its first one.
*/
export const openingSelection = (card: BenchCard): OpeningSelection => {
const ranked = card.loads
.map(load => ({ load, index: worstPeerOutcomeIndex(load) }))
.filter((entry): entry is { load: CardLoad; index: number } => entry.index !== undefined);

if (ranked.length === 0) {
return { load: card.loads.find(load => load.primary) ?? card.loads[0], reason: 'harness-default' };
}

const leading = ranked.filter(entry => entry.index <= LEADING_INDEX);
const pool = leading.length > 0 ? leading : ranked;

const best = pool.reduce(
(best, entry) => {
if (best === undefined) return entry;
// Within the leading tier every candidate already ties on the ladder, so
// only size breaks it. Outside it, a lower index is strictly better than a
// bigger load at a worse one - the ladder position is read before size.
if (leading.length === 0 && entry.index !== best.index) return entry.index < best.index ? entry : best;

return entry.load.count > best.load.count ? entry : best;
},
undefined as (typeof pool)[number] | undefined,
);

return { load: best?.load, reason: leading.length > 0 ? 'leading' : 'least-behind' };
};

/** The load a card opens on; see `openingSelection` for which one and why. */
export const openingLoad = (card: BenchCard): CardLoad | undefined => openingSelection(card).load;
113 changes: 113 additions & 0 deletions test/site/bench-opening-selection.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,113 @@
/**
* Which load a benchmark card opens on.
*
* A reader arrives asking how ExoJS does at the size they are about to run, so
* the card opens on its best real case: the largest load where ExoJS leads
* every JavaScript peer, or - failing that - the one it trails least. Rapier is
* excluded from the ranking, the same as it is from the headline sentences: a
* gap against a Rust/WASM engine is not a JavaScript-peer comparison.
*/

import { describe, expect, it } from 'vitest';

import type { BenchCard, CardArm, CardLoad } from '../../site/src/lib/bench-cards';
import { openingSelection } from '../../site/src/lib/bench-cards';
import type { CellOutcome } from '../../site/src/lib/bench-profiles';

/** One arm at the outcome a test needs, every other field the uninteresting default. */
const armOf = (id: string, outcome: CellOutcome, reference = false): CardArm => ({
id,
label: id,
ms: 1,
p95Ms: 1,
reference,
overFrameBudget: false,
outcome,
quantitative: true,
});

/** One load, named by its size, with ExoJS's outcome against each named peer. */
const loadOf = (count: number, peerOutcomes: Readonly<Record<string, CellOutcome>>): CardLoad => ({
id: String(count),
label: `${String(count)} sprites`,
count,
primary: false,
arms: [armOf('exojs', 'level', true), ...Object.entries(peerOutcomes).map(([id, outcome]) => armOf(id, outcome))],
maxMs: 1,
withheld: undefined,
});

const cardOf = (loads: readonly CardLoad[]): BenchCard => ({ id: 'test', category: 'test', loads });

describe('openingSelection', () => {
it('opens on the largest load where ExoJS leads every peer', () => {
const card = cardOf([loadOf(1_000, { pixi: 'lead' }), loadOf(10_000, { pixi: 'lead' }), loadOf(50_000, { pixi: 'loss' })]);

const selection = openingSelection(card);

expect(selection.reason).toBe('leading');
expect(selection.load?.count).toBe(10_000);
});

it('treats clear-lead and lead as the same leading tier, ranked by size alone', () => {
const card = cardOf([loadOf(1_000, { pixi: 'clear-lead' }), loadOf(10_000, { pixi: 'lead' })]);

// A small clear-lead does not outrank a bigger plain lead: both count as
// "leads", and only size breaks the tie between them.
expect(openingSelection(card).load?.count).toBe(10_000);
});

it('requires every peer to lead, not just the best one', () => {
const card = cardOf([loadOf(10_000, { pixi: 'lead', phaser: 'loss' }), loadOf(1_000, { pixi: 'lead', phaser: 'lead' })]);

const selection = openingSelection(card);

expect(selection.reason).toBe('leading');
expect(selection.load?.count).toBe(1_000);
});

it('falls back to the least-behind load when nothing leads every peer', () => {
const card = cardOf([loadOf(1_000, { pixi: 'clear-loss' }), loadOf(10_000, { pixi: 'loss' }), loadOf(50_000, { pixi: 'level' })]);

const selection = openingSelection(card);

expect(selection.reason).toBe('least-behind');
// 'level' outranks 'loss' outranks 'clear-loss' on the ladder, regardless
// of size.
expect(selection.load?.count).toBe(50_000);
});

it('breaks a least-behind tie toward the larger load', () => {
const card = cardOf([loadOf(1_000, { pixi: 'loss' }), loadOf(10_000, { pixi: 'loss' })]);

expect(openingSelection(card).load?.count).toBe(10_000);
});

it('ignores Rapier when deciding whether ExoJS leads', () => {
const card = cardOf([loadOf(1_000, { pixi: 'lead', rapier: 'clear-loss' })]);

const selection = openingSelection(card);

expect(selection.reason).toBe('leading');
expect(selection.load?.count).toBe(1_000);
});

it('falls back to the harness default where no load carries a JavaScript peer', () => {
const primary: CardLoad = { ...loadOf(1_000, { rapier: 'clear-loss' }), primary: true };
const card = cardOf([loadOf(500, { rapier: 'clear-loss' }), primary]);

const selection = openingSelection(card);

expect(selection.reason).toBe('harness-default');
expect(selection.load?.count).toBe(1_000);
});

it('falls back to the first load where the harness default carries no JavaScript peer and no primary flag', () => {
const card = cardOf([loadOf(500, {}), loadOf(1_000, {})]);

const selection = openingSelection(card);

expect(selection.reason).toBe('harness-default');
expect(selection.load?.count).toBe(500);
});
});
Loading