mirror of
https://github.com/ComposioHQ/composio.git
synced 2026-09-22 11:46:35 +08:00
71ec523d25
This PR: - builds on merged PR #4107 - adds `examples-manifest.json`: 93 runnable entrypoints across both SDKs with tiers (unattended / provisioned / bounded / excluded), per-entry timeouts, env contracts, and labeled readiness markers for OAuth flows - adds `harness/run.mjs`: sweeps entries against staging with per-entry Composio traces, negative controls (garbage credentials must fail), a `selftest`, and a baseline/candidate client swap for parity runs - rejects every non-staging `COMPOSIO_BASE_URL` before the harness or provisioning script can send credentials - adds `--llm mock`: LLM examples run against a pinned `@copilotkit/aimock` server with scripted tool-call fixtures (`harness/llm-mock/`) — full sweeps spend zero model tokens while Composio tool execution stays live on staging; 46/48 LLM entries pass mocked, and the two `@openai/agents` entries are marked `llmMock:false` - adds `examples-live.yml`: nightly gc-provision → mock sweep → four-entry live LLM canary; dispatchable per language, client, entry subset, and LLM mode, and bound to the `staging` GitHub environment - accepts separate TypeScript tarball and Python wheel inputs for candidate sweeps, restores exact pre-run workspace files on failure, and rejects empty or partial parity comparisons The configured `COMPOSIO_API_KEY` must belong to a dedicated, disposable staging project. The workflow becomes dispatchable after it exists on the default branch. ## Context Examples historically went stale because nothing executed them. PR #4107 makes them runnable and provisionable; this PR adds the machinery that runs them on a schedule and validates the Stainless→self-managed client swap via baseline/candidate parity sweeps. ```mermaid graph TD M[examples-manifest.json] --> R[harness/run.mjs sweep] P[examples-provision.mjs --gc] -->|COMPOSIO_EXAMPLES_* ids| R R -->|tier 1-3 entries| E[TS + Python examples] A[aimock fixtures] -->|--llm mock| E E -->|live tool execution + traces| C[Composio staging] R --> N[nightly live canary: 4 entries] ``` ## Verification - `env -u COMPOSIO_BASE_URL node harness/run.mjs selftest` - `pnpm test:examples` against a clean tracked snapshot - manifest integrity: 93 unique entry IDs and 93 existing files - JavaScript, Python, JSON, YAML, and embedded Bash syntax checks - `pnpm validate:changesets`
74 lines
2.7 KiB
JavaScript
Executable File
74 lines
2.7 KiB
JavaScript
Executable File
#!/usr/bin/env node
|
|
// Trace-parity comparator: node harness/parity.mjs <baseline-run-dir> <candidate-run-dir>
|
|
// Per entry green in BOTH runs: the sets of distinct (method, path-template)
|
|
// pairs must be equal, ignoring pairs allowlisted in parity-variance.json.
|
|
// Prints a JSON report; exit 0 = all parities hold, 1 = mismatches.
|
|
|
|
import { readFileSync, readdirSync, existsSync } from 'node:fs';
|
|
import { join, resolve, dirname } from 'node:path';
|
|
import { fileURLToPath } from 'node:url';
|
|
|
|
const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..');
|
|
const [baseDir, candDir] = process.argv.slice(2);
|
|
if (!baseDir || !candDir) {
|
|
console.error('usage: parity.mjs <baseline-run-dir> <candidate-run-dir>');
|
|
process.exit(2);
|
|
}
|
|
|
|
const variance = JSON.parse(readFileSync(join(ROOT, 'parity-variance.json'), 'utf8'));
|
|
const ignored = new Set((variance.entries ?? []).map((e) => e.pair));
|
|
|
|
const loadRun = (dir) => {
|
|
const results = readFileSync(join(dir, 'results.jsonl'), 'utf8')
|
|
.trim()
|
|
.split('\n')
|
|
.filter(Boolean)
|
|
.map((l) => JSON.parse(l));
|
|
const byId = new Map();
|
|
for (const r of results) byId.set(r.id, r);
|
|
return byId;
|
|
};
|
|
|
|
const tracePairs = (dir, id) => {
|
|
const file = join(dir, 'traces', `${id.replace(/[/]/g, '__')}.jsonl`);
|
|
const pairs = new Set();
|
|
if (!existsSync(file)) return pairs;
|
|
for (const line of readFileSync(file, 'utf8').trim().split('\n').filter(Boolean)) {
|
|
const rec = JSON.parse(line);
|
|
if (rec.m && rec.p) pairs.add(`${rec.m} ${rec.p}`);
|
|
}
|
|
return pairs;
|
|
};
|
|
|
|
const baseline = loadRun(baseDir);
|
|
const candidate = loadRun(candDir);
|
|
|
|
const report = [];
|
|
const ids = new Set([...baseline.keys(), ...candidate.keys()]);
|
|
for (const id of [...ids].sort()) {
|
|
const b = baseline.get(id);
|
|
const c = candidate.get(id);
|
|
if (!b || !c) {
|
|
report.push({
|
|
id,
|
|
parity: false,
|
|
reason: `missing ${b ? 'candidate' : 'baseline'} result`,
|
|
});
|
|
continue;
|
|
}
|
|
if (b.status !== 'green' || c.status !== 'green') {
|
|
report.push({ id, parity: false, reason: `status baseline=${b.status} candidate=${c.status}` });
|
|
continue;
|
|
}
|
|
const bp = tracePairs(baseDir, id);
|
|
const cp = tracePairs(candDir, id);
|
|
const onlyBaseline = [...bp].filter((p) => !cp.has(p) && !ignored.has(p));
|
|
const onlyCandidate = [...cp].filter((p) => !bp.has(p) && !ignored.has(p));
|
|
const match = onlyBaseline.length === 0 && onlyCandidate.length === 0;
|
|
report.push({ id, parity: match, ...(match ? {} : { onlyBaseline, onlyCandidate }) });
|
|
}
|
|
|
|
const green = report.filter((r) => r.parity).length;
|
|
console.log(JSON.stringify({ compared: report.length, parityGreen: green, ignoredPairs: ignored.size, entries: report }, null, 2));
|
|
process.exit(report.length > 0 && report.every((r) => r.parity) ? 0 : 1);
|