mirror of
https://github.com/CopilotKit/CopilotKit.git
synced 2026-09-14 16:26:20 +08:00
b941298cc1
The tool-rendering, frontend-tools-async, and hitl-in-app fixtures all gated their first-leg (tool-emitting) vs. follow-up (narration) responses on `hasToolResult: false/true` and/or `turnIndex`. Those constraints count the *entire* thread, so once a user clicked a tool-using pill the thread already contained tool messages and assistant turns and subsequent pill clicks fell through to the wrong branch — d20 dropped from 5 rolls to 3, Chain tools emitted no cards, query_notes returned narration without the Notes DB card, and the second HITL pill never raised an approval dialog. Re-key every follow-up fixture on the prior step's `toolCallId` (the matcher checks `messages[last].tool_call_id`), drop the global `hasToolResult` gates from the tool-emitting fixtures, and reorder so the toolCallId-specific fixtures come first under first-match-wins. The d20 chain becomes a linear toolCallId graph (`call_tr_d20_seq_001` → `_002` → … → `_005`), Chain tools gets disambiguators for each of its three parallel tool_call_ids, and Weather/AAPL/query_notes/HITL approve+reject branches all gate on the specific request_user_approval / get_weather / query_notes / get_stock_price id that landed last. userMessage matchers are unchanged. Adds Playwright multi-pill regression tests to the four affected demos that click every pill sequentially in one thread and assert the full card counts: - tool-rendering-default-catchall: Find flights → 5 d20 rolls (with 20 last) - tool-rendering-custom-catchall: 1 flights + 5 d20 + 3 chain = 9 cards - frontend-tools-async: 3 NOTES DB cards with the right keyword per pill - hitl-in-app: refund approve then escalate, each with its own dialog
187 lines
7.6 KiB
TypeScript
187 lines
7.6 KiB
TypeScript
import { test, expect } from "@playwright/test";
|
||
|
||
// QA reference: qa/frontend-tools-async.md
|
||
// Demo source: src/app/demos/frontend-tools-async/{page.tsx, notes-card.tsx}
|
||
//
|
||
// The demo registers ONE async frontend tool via `useFrontendTool`:
|
||
// `query_notes(keyword: string)`. The handler sleeps 500ms (simulated local
|
||
// DB latency) then returns up to 5 matches from an in-memory 7-note DB.
|
||
// A custom `render` mounts `NotesCard` which exposes:
|
||
// - `data-testid="notes-card"` (outer container)
|
||
// - `data-testid="notes-keyword"` (heading: `Matching "<keyword>"`)
|
||
// - `data-testid="notes-list"` (the <ul> of matches)
|
||
// - `data-testid="note-n1"` … `note-n7` per-note rows
|
||
//
|
||
// Genuine-pass strategy: the deterministic aimock fixtures match each pill's
|
||
// verbatim prompt with a dedicated `query_notes(keyword=…)` tool call so the
|
||
// async handler runs against the real client-side NOTES_DB. The card's
|
||
// `keyword` heading is then the keyword we asserted in the fixture, and the
|
||
// `notes-list` rows reflect the actual handler-filtered results — proving
|
||
// the async tool round-trip end-to-end.
|
||
|
||
test.describe("Frontend Tools (async query_notes)", () => {
|
||
test.setTimeout(120_000);
|
||
|
||
test.beforeEach(async ({ page }) => {
|
||
await page.goto("/demos/frontend-tools-async");
|
||
});
|
||
|
||
test("page loads with composer and 3 pills", async ({ page }) => {
|
||
await expect(page.getByPlaceholder("Type a message")).toBeVisible();
|
||
await expect(
|
||
page.getByRole("button", { name: /Find project-planning notes/i }),
|
||
).toBeVisible({ timeout: 15_000 });
|
||
await expect(
|
||
page.getByRole("button", { name: /Search for 'auth'/i }),
|
||
).toBeVisible({ timeout: 15_000 });
|
||
await expect(
|
||
page.getByRole("button", { name: /What do I have about reading\?/i }),
|
||
).toBeVisible({ timeout: 15_000 });
|
||
});
|
||
|
||
test("project-planning pill → Notes DB card with project-planning notes", async ({
|
||
page,
|
||
}) => {
|
||
await page
|
||
.getByRole("button", { name: /Find project-planning notes/i })
|
||
.click();
|
||
|
||
const notesCard = page.locator('[data-testid="notes-card"]').first();
|
||
await expect(notesCard).toBeVisible({ timeout: 60_000 });
|
||
|
||
// The keyword heading proves the async handler resolved against the
|
||
// fixture-emitted `query_notes(keyword="project planning")` call.
|
||
await expect(notesCard.locator('[data-testid="notes-keyword"]')).toHaveText(
|
||
/Matching\s+"project planning"/i,
|
||
{ timeout: 30_000 },
|
||
);
|
||
|
||
// The async handler matches notes n1 ("Q2 project planning kickoff")
|
||
// and n5 ("Project planning retrospective notes") from NOTES_DB.
|
||
const list = notesCard.locator('[data-testid="notes-list"]');
|
||
await expect(list).toBeVisible({ timeout: 30_000 });
|
||
await expect(notesCard.locator('[data-testid="note-n1"]')).toBeVisible();
|
||
await expect(notesCard.locator('[data-testid="note-n5"]')).toBeVisible();
|
||
|
||
// Anti-regression: the generic-plan boilerplate from the cross-cell
|
||
// catch-all fixture must NOT appear. If it does, the d5-all.json
|
||
// fixture lost match priority to feature-parity.json's "plan" entry.
|
||
await expect(
|
||
page.getByText("Research the topic, Outline key points"),
|
||
).toHaveCount(0);
|
||
});
|
||
|
||
test("auth pill → Notes DB card with auth-related notes", async ({
|
||
page,
|
||
}) => {
|
||
await page.getByRole("button", { name: /Search for 'auth'/i }).click();
|
||
|
||
const notesCard = page.locator('[data-testid="notes-card"]').first();
|
||
await expect(notesCard).toBeVisible({ timeout: 60_000 });
|
||
|
||
await expect(notesCard.locator('[data-testid="notes-keyword"]')).toHaveText(
|
||
/Matching\s+"auth"/i,
|
||
{ timeout: 30_000 },
|
||
);
|
||
|
||
// The async handler matches note n2 ("Planning: migrate auth to
|
||
// passkeys") on the "auth" tag.
|
||
const list = notesCard.locator('[data-testid="notes-list"]');
|
||
await expect(list).toBeVisible({ timeout: 30_000 });
|
||
await expect(notesCard.locator('[data-testid="note-n2"]')).toBeVisible();
|
||
|
||
// Anti-regression: the showcase-assistant catch-all from
|
||
// feature-parity.json must NOT have intercepted this prompt.
|
||
await expect(page.getByText("I'm your showcase assistant")).toHaveCount(0);
|
||
});
|
||
|
||
test("reading pill → Notes DB card with Book recommendations + locked narration", async ({
|
||
page,
|
||
}) => {
|
||
await page
|
||
.getByRole("button", { name: /What do I have about reading\?/i })
|
||
.click();
|
||
|
||
const notesCard = page.locator('[data-testid="notes-card"]').first();
|
||
await expect(notesCard).toBeVisible({ timeout: 60_000 });
|
||
|
||
// Keyword heading + match count + per-note testid + content +
|
||
// tag chip — the full canonical shape per spec test #4.
|
||
await expect(notesCard.locator('[data-testid="notes-keyword"]')).toHaveText(
|
||
/Matching\s+"reading"/i,
|
||
{ timeout: 30_000 },
|
||
);
|
||
await expect(notesCard.getByText("1 match", { exact: false })).toBeVisible({
|
||
timeout: 30_000,
|
||
});
|
||
|
||
const note = notesCard.locator('[data-testid="note-n4"]');
|
||
await expect(note).toBeVisible({ timeout: 30_000 });
|
||
await expect(note.getByText("Book recommendations")).toBeVisible();
|
||
await expect(note.getByText(/Thinking Fast and Slow/i)).toBeVisible();
|
||
await expect(
|
||
note.getByText(/The Design of Everyday Things/i),
|
||
).toBeVisible();
|
||
await expect(note.getByText("reading", { exact: true })).toBeVisible();
|
||
|
||
// Locked narration leading phrase — proves the deterministic 2nd-turn
|
||
// fixture wired correctly through the async tool result.
|
||
await expect(
|
||
page
|
||
.locator('[data-testid="copilot-assistant-message"]')
|
||
.filter({
|
||
hasText:
|
||
'You have a note titled "Book recommendations" that is tagged with "reading',
|
||
})
|
||
.first(),
|
||
).toBeVisible({ timeout: 60_000 });
|
||
});
|
||
|
||
// Regression for the aimock multi-pill bug:
|
||
// The three frontend-tools-async fixtures used `hasToolResult: false/true`
|
||
// gates to split first-turn (emit `query_notes`) vs. follow-up (narration).
|
||
// After the user clicked a tool-using pill earlier in the same thread, the
|
||
// first-turn fixture was skipped (the thread already had a prior tool
|
||
// result), the follow-up fixture fired immediately with just narration,
|
||
// and the Notes DB card never rendered. Fix: chain via `toolCallId`, drop
|
||
// the gates. This test drives all three pills in a single thread and
|
||
// asserts every pill renders its own Notes DB card.
|
||
test("sequential pills in one thread each render their own Notes DB card", async ({
|
||
page,
|
||
}) => {
|
||
// Three pills × async-handler latency × LLM mock chain; the existing
|
||
// describe-level 120s is not enough once we drive all three in one test.
|
||
test.setTimeout(240_000);
|
||
|
||
const cards = page.locator('[data-testid="notes-card"]');
|
||
|
||
await page
|
||
.getByRole("button", { name: /Find project-planning notes/i })
|
||
.click();
|
||
await expect.poll(() => cards.count(), { timeout: 60_000 }).toBe(1);
|
||
await expect(
|
||
page.locator('[data-testid="notes-keyword"]', {
|
||
hasText: /Matching\s+"project planning"/i,
|
||
}),
|
||
).toBeVisible({ timeout: 60_000 });
|
||
|
||
await page.getByRole("button", { name: /Search for 'auth'/i }).click();
|
||
await expect.poll(() => cards.count(), { timeout: 60_000 }).toBe(2);
|
||
await expect(
|
||
page.locator('[data-testid="notes-keyword"]', {
|
||
hasText: /Matching\s+"auth"/i,
|
||
}),
|
||
).toBeVisible({ timeout: 60_000 });
|
||
|
||
await page
|
||
.getByRole("button", { name: /What do I have about reading\?/i })
|
||
.click();
|
||
await expect.poll(() => cards.count(), { timeout: 60_000 }).toBe(3);
|
||
await expect(
|
||
page.locator('[data-testid="notes-keyword"]', {
|
||
hasText: /Matching\s+"reading"/i,
|
||
}),
|
||
).toBeVisible({ timeout: 60_000 });
|
||
});
|
||
});
|