mirror of
https://github.com/callstack/agent-device.git
synced 2026-09-14 20:06:34 +08:00
45cfad5cc5
* feat: add e2e command perf benchmark harness + nightly CI Adds scripts/perf, a cheap end-to-end perf benchmark that drives the built CLI through an ordered Settings tour of ~24 commands for N rounds, on a fully isolated daemon/state-dir and self-cleaning device, and emits JSON + Markdown reports. Per-command timing comes from wrapping each batchable command in its own single-step batch (daemon durationMs) plus wall-clock around the process. Wires a scheduled + workflow_dispatch CI job (perf-nightly.yml) that reuses the cached iOS XCUITest runner (setup-apple-replay) and the Android replay host, and runs the CLI from source via --experimental-strip-types (no dist build). * refactor(perf): drive the harness CLI via runCmdSync, not spawnSync Review (P2): repo rule is to spawn processes through src/utils/exec.ts, not node:child_process directly. Switch the perf harness's invokeCli to runCmdSync (allowFailure so non-zero exits are recorded as samples) and add a maxBuffer option to ExecOptions/runCmdSync (snapshot payloads exceed Node's ~1MB default). * perf(harness): warm the runner after open so the first measured command is clean The first interaction after open/relaunch pays the one-time iOS XCUITest runner startup (~10s+ cold) and a per-relaunch first-AX-query settle cost (~4s). That was landing on the first measured command each round (snapshot -i), inflating it ~10x vs the next snapshot. Run an untimed warmup snapshot -i after establishSession, after each round's reset-open, and after every freshRoot relaunch, so no measured command absorbs runner startup. Noted in the report header. * refactor(perf): address review + fix Fallow CI - exec.ts: extract spawnRejectionError + commandCloseFailure helpers, deduping the error/close handler clones (Fallow duplication ✗ that surfaced once the maxBuffer change pulled exec.ts into the audit scope). - .fallowrc: exclude scripts/perf/** (non-shipped benchmark tooling, like examples/ test-app) so its naturally-moderate functions don't trip the complexity gate. - config.ts: drop unused exports CLI_BIN/DEFAULT_OUT_DIR; add readIntValue so --n/--rounds/--warmup report the actual flag + reject non-integers clearly. - harness.ts: extract toSample(); type sampleError param as CliResult. - scenario.ts: ScenarioStep is now a discriminated union on execMode (removes step.step!/ step.args ?? []). - comment/legend rewords (platform defaults are local-convenience/CI-overridden; elements = node count). check:fallow now green; typecheck/lint/unit pass. * perf(harness): downgrade sample ok when a batch step reports ok:false Defensive belt-and-suspenders for the Codex review note: stop-only batch already surfaces a failed step as a top-level failure (caught by invokeCli), but if an on-error=continue mode ever keeps the batch ok while a step fails, don't silently count that step as a successful sample — derive ok from the step's own result.ok.
68 lines
2.0 KiB
TypeScript
68 lines
2.0 KiB
TypeScript
import fs from 'node:fs';
|
|
import path from 'node:path';
|
|
import { parseConfig, REPO_ROOT, usesSourceCli } from './config.ts';
|
|
import { runScenario, setupIsolation, teardownIsolation, type IsolationContext } from './harness.ts';
|
|
import { writeReports } from './report.ts';
|
|
import type { RunResult } from './types.ts';
|
|
|
|
function readVersion(): string {
|
|
try {
|
|
const pkg = JSON.parse(fs.readFileSync(path.join(REPO_ROOT, 'package.json'), 'utf8'));
|
|
return typeof pkg.version === 'string' ? pkg.version : 'unknown';
|
|
} catch {
|
|
return 'unknown';
|
|
}
|
|
}
|
|
|
|
function main(): void {
|
|
const cfg = parseConfig(process.argv.slice(2));
|
|
// The dist binary needs a build; running from source (AGENT_DEVICE_PERF_CLI) does not.
|
|
if (!usesSourceCli() && !fs.existsSync(path.join(REPO_ROOT, 'dist', 'src'))) {
|
|
process.stderr.write('[perf] dist/ is missing — run `pnpm build` first.\n');
|
|
process.exit(1);
|
|
}
|
|
|
|
const startedAt = new Date().toISOString();
|
|
let ctx: IsolationContext | null = null;
|
|
let exitCode = 0;
|
|
|
|
const cleanup = (): void => {
|
|
if (ctx) {
|
|
teardownIsolation(ctx, cfg);
|
|
ctx = null;
|
|
}
|
|
};
|
|
process.on('SIGINT', () => {
|
|
cleanup();
|
|
process.exit(130);
|
|
});
|
|
process.on('SIGTERM', () => {
|
|
cleanup();
|
|
process.exit(143);
|
|
});
|
|
|
|
try {
|
|
ctx = setupIsolation(cfg);
|
|
const measurements = runScenario(ctx, cfg);
|
|
const run: RunResult = {
|
|
startedAt,
|
|
finishedAt: new Date().toISOString(),
|
|
platform: cfg.platform,
|
|
device: { udid: ctx.profile.udid, serial: ctx.profile.serial, name: ctx.profile.deviceName },
|
|
config: { rounds: cfg.rounds, warmup: cfg.warmup, keepArtifacts: cfg.keepArtifacts },
|
|
agentDeviceVersion: readVersion(),
|
|
measurements,
|
|
};
|
|
const { jsonPath, mdPath } = writeReports(run, cfg.outDir);
|
|
process.stderr.write(`\n[perf] report: ${mdPath}\n[perf] json: ${jsonPath}\n`);
|
|
} catch (e) {
|
|
process.stderr.write(`[perf] error: ${(e as Error).stack ?? String(e)}\n`);
|
|
exitCode = 1;
|
|
} finally {
|
|
cleanup();
|
|
}
|
|
process.exit(exitCode);
|
|
}
|
|
|
|
main();
|