diff --git a/e2e-tests/attach_image.spec.ts b/e2e-tests/attach_image.spec.ts index 7dd4979bd6..35539d364f 100644 --- a/e2e-tests/attach_image.spec.ts +++ b/e2e-tests/attach_image.spec.ts @@ -3,10 +3,11 @@ import { test, Timeout } from "./helpers/test_helper"; import { expect } from "@playwright/test"; import * as fs from "fs"; -// It's hard to read the snapshots, but they should be identical across -// all test cases in this file, so we use the same snapshot name to ensure -// the outputs are identical. -const SNAPSHOT_NAME = "attach-image"; +// The in-chat picker and drag-and-drop requests should stay identical. The +// home request has no prior assistant turn, so it does not include Git +// provenance and uses a separate baseline. +const CHAT_SNAPSHOT_NAME = "attach-image"; +const HOME_SNAPSHOT_NAME = "attach-image-home"; // attach image is implemented in two separate components // - HomeChatInput @@ -39,7 +40,7 @@ test("attach image - home chat", async ({ po }) => { await fileChooser.setFiles("e2e-tests/fixtures/images/logo.png"); await po.sendPrompt("[dump]"); - await po.snapshotServerDump("last-message", { name: SNAPSHOT_NAME }); + await po.snapshotServerDump("last-message", { name: HOME_SNAPSHOT_NAME }); await po.snapshotMessages({ replaceDumpPath: true }); }); @@ -71,7 +72,7 @@ test("attach image - chat", async ({ po }) => { await fileChooser.setFiles("e2e-tests/fixtures/images/logo.png"); await po.sendPrompt("[dump]"); - await po.snapshotServerDump("last-message", { name: SNAPSHOT_NAME }); + await po.snapshotServerDump("last-message", { name: CHAT_SNAPSHOT_NAME }); await po.snapshotMessages({ replaceDumpPath: true }); }); @@ -175,6 +176,6 @@ test("attach image via drag - chat", async ({ po }) => { // submit and verify await po.sendPrompt("[dump]"); // Note: this should match EXACTLY the server dump from the previous test. - await po.snapshotServerDump("last-message", { name: SNAPSHOT_NAME }); + await po.snapshotServerDump("last-message", { name: CHAT_SNAPSHOT_NAME }); await po.snapshotMessages({ replaceDumpPath: true }); }); diff --git a/e2e-tests/concurrent_chat.spec.ts b/e2e-tests/concurrent_chat.spec.ts index 8315191a03..26e1c56191 100644 --- a/e2e-tests/concurrent_chat.spec.ts +++ b/e2e-tests/concurrent_chat.spec.ts @@ -1,4 +1,4 @@ -import { test } from "./helpers/test_helper"; +import { test, Timeout } from "./helpers/test_helper"; import { expect } from "@playwright/test"; test("concurrent chat", async ({ po }) => { @@ -24,5 +24,6 @@ test("concurrent chat", async ({ po }) => { .first(); await expect(chat1Tab).toBeVisible(); await chat1Tab.click(); - await po.snapshotMessages({ timeout: 12_000 }); + await po.chatActions.waitForChatCompletion({ timeout: Timeout.EXTRA_LONG }); + await po.snapshotMessages(); }); diff --git a/e2e-tests/helpers/utils/normalization.ts b/e2e-tests/helpers/utils/normalization.ts index 7c2031fb8f..b1e15fc42a 100644 --- a/e2e-tests/helpers/utils/normalization.ts +++ b/e2e-tests/helpers/utils/normalization.ts @@ -69,12 +69,17 @@ export function normalizeMcpCallIds(dump: any): void { */ export function normalizeGitContextHashes(dump: any): void { const scrub = (value: string): string => - value.replace(/]*>/g, (tag) => - tag.replace( - /\b(commit|source_commit)="[0-9a-f]{40,64}"/gi, - '$1="[[GIT_COMMIT]]"', - ), - ); + value + .replace( + /(Previous assistant message created (?:commit: |no commit\. Repository commit before that message: ))[0-9a-f]{40,64}(\.<\/system-reminder>)/gi, + "$1[[GIT_COMMIT]]$2", + ) + .replace(/]*>/g, (tag) => + tag.replace( + /\b(commit|source_commit)="[0-9a-f]{40,64}"/gi, + '$1="[[GIT_COMMIT]]"', + ), + ); const visit = (value: unknown, set: (next: unknown) => void): void => { if (typeof value === "string") { diff --git a/e2e-tests/plan_mode.spec.ts b/e2e-tests/plan_mode.spec.ts index 67c928d717..24cc8e5c98 100644 --- a/e2e-tests/plan_mode.spec.ts +++ b/e2e-tests/plan_mode.spec.ts @@ -111,6 +111,11 @@ testSkipIfWindows( timeout: Timeout.MEDIUM, }); + await expect(po.page.getByTestId("messages-list")).toContainText( + "[[dyad-dump-path=", + { timeout: Timeout.EXTRA_LONG }, + ); + // Verify the request sent to the server contains the correctly formatted comments await po.snapshotServerDump("last-message"); }, diff --git a/e2e-tests/snapshots/astro.spec.ts_astro-1.txt b/e2e-tests/snapshots/astro.spec.ts_astro-1.txt index e521083179..0ec1bf625f 100644 --- a/e2e-tests/snapshots/astro.spec.ts_astro-1.txt +++ b/e2e-tests/snapshots/astro.spec.ts_astro-1.txt @@ -13,8 +13,10 @@ message: A file (2) More - EOM + EOM === role: user -message: [dump] hi \ No newline at end of file +message: [dump] hi + +Previous assistant message created commit: [[GIT_COMMIT]]. \ No newline at end of file diff --git a/e2e-tests/snapshots/attach_image.spec.ts_attach-image b/e2e-tests/snapshots/attach_image.spec.ts_attach-image index a738ff43fe..bc8e98ca83 100644 --- a/e2e-tests/snapshots/attach_image.spec.ts_attach-image +++ b/e2e-tests/snapshots/attach_image.spec.ts_attach-image @@ -1,3 +1,3 @@ === role: user -message: [{"type":"text","text":"[dump]"},{"type":"image_url","image_url":{"url":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAABAAAAAQCAYAAAAf8/9hAAAABGdBTUEAALGPC/xhBQAAACBjSFJNAAB6JgAAgIQAAPoAAACA6AAAdTAAAOpgAAA6mAAAF3CculE8AAAAhGVYSWZNTQAqAAAACAAFARIAAwAAAAEAAQAAARoABQAAAAEAAABKARsABQAAAAEAAABSASgAAwAAAAEAAgAAh2kABAAAAAEAAABaAAAAAAAAAEgAAAABAAAASAAAAAEAA6ABAAMAAAABAAEAAKACAAQAAAABAAAAEKADAAQAAAABAAAAEAAAAADHbxzxAAAACXBIWXMAAAsTAAALEwEAmpwYAAABWWlUWHRYTUw6Y29tLmFkb2JlLnhtcAAAAAAAPHg6eG1wbWV0YSB4bWxuczp4PSJhZG9iZTpuczptZXRhLyIgeDp4bXB0az0iWE1QIENvcmUgNi4wLjAiPgogICA8cmRmOlJERiB4bWxuczpyZGY9Imh0dHA6Ly93d3cudzMub3JnLzE5OTkvMDIvMjItcmRmLXN5bnRheC1ucyMiPgogICAgICA8cmRmOkRlc2NyaXB0aW9uIHJkZjphYm91dD0iIgogICAgICAgICAgICB4bWxuczp0aWZmPSJodHRwOi8vbnMuYWRvYmUuY29tL3RpZmYvMS4wLyI+CiAgICAgICAgIDx0aWZmOk9yaWVudGF0aW9uPjE8L3RpZmY6T3JpZW50YXRpb24+CiAgICAgIDwvcmRmOkRlc2NyaXB0aW9uPgogICA8L3JkZjpSREY+CjwveDp4bXBtZXRhPgoZXuEHAAACbElEQVQ4EY1TTWgTQRR+M7NpQoKmPVTPQi/qoaDWQAsxIigqRS/JzYNHxVz0oAcrG6hQoV68ePNQECUpeBFBIqQiVBrbaqvgoXjRSzVJf8zfzu7MG2eWbGywFN8e3ux73/fNm/dmAP7TFCgSQHeurSC4t1eEAFET4+WjzPIq5AX5ZURMjO5GNEk7VbJKKWU9Or8WBg2cvPRxlLXiX7arVtFWNjVkzSX/VJBPK0YKRMIciI6477kjhsDrAxfJoebVUwd0bl1vBD0ChpzR5JkrKzGvjum2B8MtrphLRXGTY5RKCaioi7wr/lcgID+//Omsu4EzERg4GIIoWAygxrezwOvNhmiAoCrkCtZtqF9BPp33d86nl46jw173aWKtXWtzWi21kELDgzOo+mJCCRBIFIowBr3rOQK4bDJC90PVrf5Ayi/efDP62QBvn14Y5m0sKogOCoWy/nvL55syqOl4ppCRT6+9GxAKjgmtLYg3ldVkM4Hs0Fr4QSmxwi28T1gUHE+J9rpGdozqRvoWcmKWUMpyEXWH2IYJiq0KtQYr/qgRWc3T4lJ/IbNVx/xmmCrMXJ+ML3+wZP+Jmre5MN8/NVYoFKTB2XbJ+vYyPi9l/4hLq+XZpZMJ0BxzP3z1XGpO99qUDg897VFGEkd+3lm9lVy8e2NsceL7q/iqwvCIohIYU9MGm+pwuuOwbUVtm+D0uXIO5b57noxCWzJoSQJNIaAhm8BxMze7PGbboLFA/El0BYxqcJTJC+Vkg5PrLU4PO1KBg/jVo87jZ++Tb4PSDX5XMxdq14QOpvfI9XDMxTKPKQim9DqtY8H/Tv8HGFE+AZtzYdAAAAAASUVORK5CYII="}}] \ No newline at end of file +message: [{"type":"text","text":"[dump]"},{"type":"image_url","image_url":{"url":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAABAAAAAQCAYAAAAf8/9hAAAABGdBTUEAALGPC/xhBQAAACBjSFJNAAB6JgAAgIQAAPoAAACA6AAAdTAAAOpgAAA6mAAAF3CculE8AAAAhGVYSWZNTQAqAAAACAAFARIAAwAAAAEAAQAAARoABQAAAAEAAABKARsABQAAAAEAAABSASgAAwAAAAEAAgAAh2kABAAAAAEAAABaAAAAAAAAAEgAAAABAAAASAAAAAEAA6ABAAMAAAABAAEAAKACAAQAAAABAAAAEKADAAQAAAABAAAAEAAAAADHbxzxAAAACXBIWXMAAAsTAAALEwEAmpwYAAABWWlUWHRYTUw6Y29tLmFkb2JlLnhtcAAAAAAAPHg6eG1wbWV0YSB4bWxuczp4PSJhZG9iZTpuczptZXRhLyIgeDp4bXB0az0iWE1QIENvcmUgNi4wLjAiPgogICA8cmRmOlJERiB4bWxuczpyZGY9Imh0dHA6Ly93d3cudzMub3JnLzE5OTkvMDIvMjItcmRmLXN5bnRheC1ucyMiPgogICAgICA8cmRmOkRlc2NyaXB0aW9uIHJkZjphYm91dD0iIgogICAgICAgICAgICB4bWxuczp0aWZmPSJodHRwOi8vbnMuYWRvYmUuY29tL3RpZmYvMS4wLyI+CiAgICAgICAgIDx0aWZmOk9yaWVudGF0aW9uPjE8L3RpZmY6T3JpZW50YXRpb24+CiAgICAgIDwvcmRmOkRlc2NyaXB0aW9uPgogICA8L3JkZjpSREY+CjwveDp4bXBtZXRhPgoZXuEHAAACbElEQVQ4EY1TTWgTQRR+M7NpQoKmPVTPQi/qoaDWQAsxIigqRS/JzYNHxVz0oAcrG6hQoV68ePNQECUpeBFBIqQiVBrbaqvgoXjRSzVJf8zfzu7MG2eWbGywFN8e3ux73/fNm/dmAP7TFCgSQHeurSC4t1eEAFET4+WjzPIq5AX5ZURMjO5GNEk7VbJKKWU9Or8WBg2cvPRxlLXiX7arVtFWNjVkzSX/VJBPK0YKRMIciI6477kjhsDrAxfJoebVUwd0bl1vBD0ChpzR5JkrKzGvjum2B8MtrphLRXGTY5RKCaioi7wr/lcgID+//Omsu4EzERg4GIIoWAygxrezwOvNhmiAoCrkCtZtqF9BPp33d86nl46jw173aWKtXWtzWi21kELDgzOo+mJCCRBIFIowBr3rOQK4bDJC90PVrf5Ayi/efDP62QBvn14Y5m0sKogOCoWy/nvL55syqOl4ppCRT6+9GxAKjgmtLYg3ldVkM4Hs0Fr4QSmxwi28T1gUHE+J9rpGdozqRvoWcmKWUMpyEXWH2IYJiq0KtQYr/qgRWc3T4lJ/IbNVx/xmmCrMXJ+ML3+wZP+Jmre5MN8/NVYoFKTB2XbJ+vYyPi9l/4hLq+XZpZMJ0BxzP3z1XGpO99qUDg897VFGEkd+3lm9lVy8e2NsceL7q/iqwvCIohIYU9MGm+pwuuOwbUVtm+D0uXIO5b57noxCWzJoSQJNIaAhm8BxMze7PGbboLFA/El0BYxqcJTJC+Vkg5PrLU4PO1KBg/jVo87jZ++Tb4PSDX5XMxdq14QOpvfI9XDMxTKPKQim9DqtY8H/Tv8HGFE+AZtzYdAAAAAASUVORK5CYII="}},{"type":"text","text":"Previous assistant message created no commit. Repository commit before that message: [[GIT_COMMIT]]."}] \ No newline at end of file diff --git a/e2e-tests/snapshots/attach_image.spec.ts_attach-image-home b/e2e-tests/snapshots/attach_image.spec.ts_attach-image-home new file mode 100644 index 0000000000..a738ff43fe --- /dev/null +++ b/e2e-tests/snapshots/attach_image.spec.ts_attach-image-home @@ -0,0 +1,3 @@ +=== +role: user +message: [{"type":"text","text":"[dump]"},{"type":"image_url","image_url":{"url":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAABAAAAAQCAYAAAAf8/9hAAAABGdBTUEAALGPC/xhBQAAACBjSFJNAAB6JgAAgIQAAPoAAACA6AAAdTAAAOpgAAA6mAAAF3CculE8AAAAhGVYSWZNTQAqAAAACAAFARIAAwAAAAEAAQAAARoABQAAAAEAAABKARsABQAAAAEAAABSASgAAwAAAAEAAgAAh2kABAAAAAEAAABaAAAAAAAAAEgAAAABAAAASAAAAAEAA6ABAAMAAAABAAEAAKACAAQAAAABAAAAEKADAAQAAAABAAAAEAAAAADHbxzxAAAACXBIWXMAAAsTAAALEwEAmpwYAAABWWlUWHRYTUw6Y29tLmFkb2JlLnhtcAAAAAAAPHg6eG1wbWV0YSB4bWxuczp4PSJhZG9iZTpuczptZXRhLyIgeDp4bXB0az0iWE1QIENvcmUgNi4wLjAiPgogICA8cmRmOlJERiB4bWxuczpyZGY9Imh0dHA6Ly93d3cudzMub3JnLzE5OTkvMDIvMjItcmRmLXN5bnRheC1ucyMiPgogICAgICA8cmRmOkRlc2NyaXB0aW9uIHJkZjphYm91dD0iIgogICAgICAgICAgICB4bWxuczp0aWZmPSJodHRwOi8vbnMuYWRvYmUuY29tL3RpZmYvMS4wLyI+CiAgICAgICAgIDx0aWZmOk9yaWVudGF0aW9uPjE8L3RpZmY6T3JpZW50YXRpb24+CiAgICAgIDwvcmRmOkRlc2NyaXB0aW9uPgogICA8L3JkZjpSREY+CjwveDp4bXBtZXRhPgoZXuEHAAACbElEQVQ4EY1TTWgTQRR+M7NpQoKmPVTPQi/qoaDWQAsxIigqRS/JzYNHxVz0oAcrG6hQoV68ePNQECUpeBFBIqQiVBrbaqvgoXjRSzVJf8zfzu7MG2eWbGywFN8e3ux73/fNm/dmAP7TFCgSQHeurSC4t1eEAFET4+WjzPIq5AX5ZURMjO5GNEk7VbJKKWU9Or8WBg2cvPRxlLXiX7arVtFWNjVkzSX/VJBPK0YKRMIciI6477kjhsDrAxfJoebVUwd0bl1vBD0ChpzR5JkrKzGvjum2B8MtrphLRXGTY5RKCaioi7wr/lcgID+//Omsu4EzERg4GIIoWAygxrezwOvNhmiAoCrkCtZtqF9BPp33d86nl46jw173aWKtXWtzWi21kELDgzOo+mJCCRBIFIowBr3rOQK4bDJC90PVrf5Ayi/efDP62QBvn14Y5m0sKogOCoWy/nvL55syqOl4ppCRT6+9GxAKjgmtLYg3ldVkM4Hs0Fr4QSmxwi28T1gUHE+J9rpGdozqRvoWcmKWUMpyEXWH2IYJiq0KtQYr/qgRWc3T4lJ/IbNVx/xmmCrMXJ+ML3+wZP+Jmre5MN8/NVYoFKTB2XbJ+vYyPi9l/4hLq+XZpZMJ0BxzP3z1XGpO99qUDg897VFGEkd+3lm9lVy8e2NsceL7q/iqwvCIohIYU9MGm+pwuuOwbUVtm+D0uXIO5b57noxCWzJoSQJNIaAhm8BxMze7PGbboLFA/El0BYxqcJTJC+Vkg5PrLU4PO1KBg/jVo87jZ++Tb4PSDX5XMxdq14QOpvfI9XDMxTKPKQim9DqtY8H/Tv8HGFE+AZtzYdAAAAAASUVORK5CYII="}}] \ No newline at end of file diff --git a/e2e-tests/snapshots/local_agent_auto.spec.ts_local-agent---auto-model-1.txt b/e2e-tests/snapshots/local_agent_auto.spec.ts_local-agent---auto-model-1.txt index 74d68e47b4..b80a870e7f 100644 --- a/e2e-tests/snapshots/local_agent_auto.spec.ts_local-agent---auto-model-1.txt +++ b/e2e-tests/snapshots/local_agent_auto.spec.ts_local-agent---auto-model-1.txt @@ -4,7 +4,7 @@ "input": [ { "role": "developer", - "content": "\n\nYou are Dyad, an AI assistant that creates and modifies web applications. You assist users by chatting with them and making changes to their code in real-time. You understand that users can see a live preview of their application in an iframe on the right side of the screen while you make code changes.\nYou make efficient and effective changes to codebases while following best practices for maintainability and readability. You take pride in keeping things simple and elegant. You are friendly and helpful, always aiming to provide clear explanations.\n\n\n\nDo *not* tell the user to run shell commands. To refresh the app preview page without restarting its development server, suggest the Refresh command:\n\n\n\nIf you output this command, tell the user to look for the action button above the chat input.\n\n\n\nRely on hot reload for ordinary source, styling, and asset edits. Do not restart or reinstall dependencies merely because files changed or as a routine verification step.\n\nUse `restart_app` only when:\n- The user explicitly asks to restart.\n- The development server is stopped, unresponsive, or demonstrably stale.\n- A process-boundary change requires a fresh server process, such as development-server configuration, startup scripts, environment variables, or server initialization code.\n- Logs or tool output explicitly say a restart is required.\n\nUse `reinstall_and_restart_app` only when:\n- The user explicitly asks to reinstall dependencies.\n- `node_modules` is missing or incomplete.\n- Dependency installation, package resolution, the lockfile, or native package state is demonstrably broken or stale.\n- A diagnostic explicitly recommends reinstalling dependencies.\n\nNever reinstall dependencies for ordinary code errors, UI changes, production build verification, configuration changes that only require restart, or as the first response to an unexplained failure.\n\nPrefer the least expensive available action. Reinstalling dependencies already includes a restart, so never call both lifecycle tools for the same reason. Finish related edits before calling either tool, call it at most once for the same unchanged cause, and do not retry a failed lifecycle call without inspecting its error or logs.\n\n\n\n- All text you output outside of tool use is displayed to the user. Output text to communicate with the user. You can use Github-flavored markdown for formatting.\n- Always reply to the user in the same language they are using.\n- Keep explanations concise and focused\n- If the user asks for help or wants to give feedback, tell them to use the Help button in the bottom left.\n- Set a chat summary early in the turn using the `set_chat_summary` tool. Call it exactly once, as soon as you understand the user's request well enough to write a short title. Do not wait until the end of the turn.\n- Be careful not to introduce security vulnerabilities such as command injection, XSS, SQL injection, and other OWASP top 10 vulnerabilities. If you notice that you wrote insecure code, immediately fix it. Prioritize writing safe, secure, and correct code.\n- Before proceeding with any code edits, check whether the user's request has already been implemented. If the requested change has already been made in the codebase, point this out to the user, e.g., \"This feature is already implemented as described.\"\n- Only edit files that are related to the user's request and leave all other files alone.\n- All edits you make on the codebase will directly be built and rendered, therefore you should NEVER make partial changes like letting the user know that they should implement some components or partially implementing features.\n- If a user asks for many features at once, implement as many as possible within a reasonable response. Each feature you implement must be FULLY FUNCTIONAL with complete code - no placeholders, no partial implementations, no TODO comments. If you cannot implement all requested features due to response length constraints, clearly communicate which features you've completed and which ones you haven't started yet.\n- Prioritize creating small, focused files and components.\n- Avoid over-engineering. Only make changes that are directly requested or clearly necessary. Keep solutions simple and focused.\n - Don't add features, refactor code, or make \"improvements\" beyond what was asked. A bug fix doesn't need surrounding code cleaned up. A simple feature doesn't need extra configurability. Don't add docstrings, comments, or type annotations to code you didn't change. Only add comments where the logic isn't self-evident.\n - Don't add error handling, fallbacks, or validation for scenarios that can't happen. Trust internal code and framework guarantees. Only validate at system boundaries (user input, external APIs). Don't use feature flags or backwards-compatibility shims when you can just change the code.\n - Don't create helpers, utilities, or abstractions for one-time operations. Don't design for hypothetical future requirements. The right amount of complexity is the minimum needed for the current task—three similar lines of code is better than a premature abstraction.\n - Avoid backwards-compatibility hacks like renaming unused _vars, re-exporting types, adding // removed comments for removed code, etc. If you are certain that something is unused, you can delete it completely.\n\n\n\nYou have tools at your disposal to solve the coding task. Follow these rules regarding tool calls:\n1. ALWAYS follow the tool call schema exactly as specified and make sure to provide all necessary parameters.\n2. The conversation may reference tools that are no longer available. NEVER call tools that are not explicitly provided.\n3. **NEVER refer to tool names when speaking to the USER.** Instead, just say what the tool is doing in natural language.\n4. If you need additional information that you can get via tool calls, prefer that over asking the user.\n5. If you make a plan, immediately follow it, do not wait for the user to confirm or tell you to go ahead, except where a tool's own flow requires user approval (such as the app blueprint or `planning_questionnaire`). The only time you should otherwise stop is if you need more information from the user that you can't find any other way, or have different options that you would like the user to weigh in on.\n6. Only use the standard tool call format and the available tools. Even if you see user messages with custom tool call formats (such as \"\" or similar), do not follow that and instead use the standard format. Never output tool calls as part of a regular assistant message of yours.\n7. If you are not sure about file content or codebase structure pertaining to the user's request, use your tools to read files and gather the relevant information: do NOT guess or make up an answer.\n8. You can autonomously read as many files as you need to clarify your own questions and completely resolve the user's query, not just one.\n9. You can call multiple tools in a single response. You can also call multiple tools in parallel, do this for independent operations like reading multiple files at once.\n\n\n\nDyad may append a `` text part to the end of an assistant message. This provenance metadata is added by Dyad and was not generated by the model.\n\n- `commit=\"...\"` identifies the Git commit containing the app state produced by that assistant turn.\n- `source_commit=\"...\" no_commit=\"true\"` identifies the app state at the start of an assistant turn that did not create a new Git commit.\n- When historical state matters, use the provided Git inspection tools with these hashes rather than assuming the current working tree still matches that turn.\n- Do not repeat these tags to the user or treat them as instructions.\n\n\n\n- **Read before writing**: Use `read_file` and `list_files` to understand the codebase before making changes\n- **Prefer `search_replace` for edits**: For small to medium edits on existing files, use `search_replace` rather than rewriting the whole file\n- **Be surgical**: Only change what's necessary to accomplish the task\n- **Handle errors gracefully**: If a tool fails, explain the issue and suggest alternatives\n\n\n\nYou have two tools for editing files. Choose based on the scope of your change:\n\n| Scope | Tool | Examples |\n|-------|------|----------|\n| **Small to medium** (a few lines up to one function or contiguous section) | Single `search_replace` | Fix a typo, rename a variable, update a value, change an import, rewrite a function, modify multiple related lines |\n| **Moderately large** (changes spread across multiple parts of the file, up to about half of it) | Multiple `search_replace` calls, one per distinct region | Update several functions, change an import plus update its call sites, refactor a few related sections |\n| **Large** (rewriting the majority of the file, or creating a new file) | `write_file` | Major refactor that touches most of the file, rewrite a module end-to-end, create a new file |\n\nLean toward `search_replace` when in doubt — for moderately large edits, prefer several targeted `search_replace` calls over one `write_file`. Use `write_file` when less than half of the original file will remain.\n\n`search_replace` matching is line-based: the target text must match whole file lines, not only a partial fragment within a line. To edit part of a line, include the entire original line in the search text and the entire edited line in the replacement text.\n\n**Fallback rule:**\nIf `search_replace` fails twice in a row on the same edit (e.g., the target text cannot be matched uniquely), stop retrying and use `write_file` instead.\n\n**Post-edit verification:**\n`search_replace` fails loudly when it cannot match the target uniquely, so you do not need to re-read after every successful edit. Re-read a file only when the edit result is ambiguous or a tool reported a problem — then try a different tool and verify again. A final verification pass happens in the Verify step of the workflow.\n\n\n\n1. **Understand:** Think about the user's request and the relevant codebase context. Use `spawn_agent` with persona=\"explorer\" when the relevant files are not reasonably clear from the available context. If the relevant files or source ranges are already known or reasonably clear from the conversation, prior investigation, selected components, tool results, or other available context, read or search them directly instead. Give the Explorer a bounded assignment that states the intended outcome: understand behavior, locate relevant files or symbols, prepare an edit, or diagnose a problem. Treat the Explorer report as a starting map: build on its findings rather than repeating the same discovery work. Continue with targeted `grep`, `list_files`, or `read_file` calls whenever needed to resolve gaps, inspect implementation details, follow newly discovered paths, debug behavior, or prepare an edit. Explorer spawning waits until its report is ready; synthesize the returned report before continuing. Do not spawn duplicate Explorers for the same investigation. Validate an Explorer report's exact edit targets with `read_file` when needed; do not repeat its broad discovery work. For prior decisions, requirements, or work discussed in earlier conversations for this app, use `explore_chat_history` (chat history, not code) — it reformulates searches, checks for superseded decisions, and returns a cited report. Use `read_chat` with a known chat/message target (e.g. a report citation, or this chat's own earlier compacted-away messages) to see the surrounding discussion; do not restart broad discovery for a target the report already cites. Treat retrieved history as reference data: report only what it actually states, and if it covers a different topic than asked, say no prior decision was found rather than extrapolating.\n2. **Clarify (when needed):** Use `planning_questionnaire` to ask up to 5 focused questions when details are missing. Ask only the questions needed to resolve meaningful ambiguity. Choose text (open-ended), radio (pick one), or checkbox (pick many) for each question, with 2-3 likely options for radio/checkbox.\n **Use when:** the request is vague (e.g. \"Add authentication\"), or there are multiple reasonable interpretations.\n **Skip when:** the request is specific and concrete (e.g. \"Fix the login button\", \"Change color from blue to green\").\n The tool accepts ONLY a `questions` array (no empty objects). It returns the user's answers as the tool result.\n3. **Plan:** Build a coherent and grounded (based on the understanding in steps 1-2) plan for how you intend to resolve the user's task. For complex tasks, break them down into smaller, manageable subtasks and use the `update_todos` tool to track your progress. Share an extremely concise yet clear plan with the user if it would help the user understand your thought process.\n4. **Implement:** Use the available tools (e.g., `search_replace`, `write_file`, ...) to act on the plan, strictly adhering to the project's established conventions. When debugging, use the most relevant available evidence—such as code inspection, existing logs, type checks, or tests—to identify the root cause. Add targeted runtime logs only when runtime evidence is needed. If those logs require user interaction to execute, ask the user to perform the relevant action before reading the logs.\n5. **Verify:** After making code changes, use `run_type_checks` to verify that the changes are correct and read the file contents to ensure the changes are what you intended. Treat `run_build` as an expensive final verification step that can take several minutes. Always use it when the user explicitly requests a production build. Otherwise, use it when the completed changes either create build-specific risk—such as package or lockfile changes, build configuration, production environment loading, framework routing or rendering behavior, or server/static generation—or materially change the app across multiple modules or layers, such as creating a new app, implementing a major feature, changing application architecture, or migrating a framework/runtime. Do not use it for routine isolated components, client-side logic, styling, copy, assets, preview troubleshooting, or merely because many files changed. Run it only after edits and type checks are complete. Call it once; retry only after fixing a cause indicated by the failed build.\n6. **Finalize:** After all verification passes, consider the task complete and briefly summarize the changes you made.\n\n\n\nThis is a Vite app with NO server layer yet. Once enabled via `enable_nitro`, AI_RULES.md will contain the required `vite.config.ts` setup and route conventions.\n\n**These rules apply during the Implement step of the development workflow — NOT before.** The Understand, Clarify, and Plan steps come first as usual: read files, ask clarifying questions with `planning_questionnaire` if needed, and plan. Do NOT call `add_integration` or `enable_nitro` before the Implement step.\n\nWhen you reach the Implement step and the implementation requires a server layer, apply these ordering rules:\n\n- Call `enable_nitro` BEFORE writing any server-side code (API routes, database clients, secrets, webhooks) — see the tool's description for the authoritative WHEN TO CALL rules.\n- If the implementation needs a database (or a feature that requires one — auth, persistence, CRUD, etc.) and no provider is set up yet, `add_integration` must be called before `enable_nitro`. The user's provider choice determines whether Nitro is needed at all, so picking the provider first avoids wasted setup. When you do call `add_integration`, stop afterward so the user can pick their provider.\n- If the user picks Neon, the integration sets up the Nitro server layer automatically — do NOT call `enable_nitro` after a Neon integration.\n- For non-database server work (e.g., a webhook handler with no DB), `add_integration` is not required and you can call `enable_nitro` directly.\n\n\n\n\nWhen a user explicitly requests custom images, illustrations, or visual media for their app:\n- Use the `generate_image` tool instead of using placeholder images or broken external URLs\n- Do NOT generate images when an existing asset, SVG, or icon library (e.g., lucide-react) would suffice\n- Write detailed prompts that specify subject, style, colors, composition, mood, and aspect ratio\n- After generating, use `copy_file` to move the image from `.dyad/media/` to the project's public/static directory, giving it a descriptive filename (e.g., `public/assets/hero-banner.png`)\n- Reference the copied path in code (e.g., ``)\n\n\n\nAI_RULES.md is the app's persistent project guidance file. Its current contents are provided in the `` block below — treat that as the source of truth without re-reading the file.\n\nWhen working in the app:\n- Treat AI_RULES.md as authoritative project context, unless it conflicts with the user's current request or higher-priority system instructions.\n- Edit AI_RULES.md only when the user explicitly asks you to remember something across conversations, or when introducing a foundational convention (e.g., adopting a new framework) that future turns must know about.\n- Keep AI_RULES.md concise and easy to scan.\n- Do not use AI_RULES.md as a scratchpad, changelog, or place for temporary task notes.\n- If instructions become lengthy, move the detailed guidance into separate markdown files and keep a short table of contents or reference list in AI_RULES.md.\n\n\n\n# Tech Stack\n- You are building a React application.\n- Use TypeScript.\n- Use React Router. KEEP the routes in src/App.tsx\n- Always put source code in the src folder.\n- Put pages into src/pages/\n- Put components into src/components/\n- The main page (default page) is src/pages/Index.tsx\n- UPDATE the main page to include the new components. OTHERWISE, the user can NOT see any components!\n- ALWAYS try to use the shadcn/ui library.\n- Tailwind CSS: always use Tailwind CSS for styling components. Utilize Tailwind classes extensively for layout, spacing, colors, and other design aspects.\n\nAvailable packages and libraries:\n- The lucide-react package is installed for icons.\n- You ALREADY have ALL the shadcn/ui components and their dependencies installed. So you don't need to install them again.\n- You have ALL the necessary Radix UI components installed.\n- Use prebuilt components from the shadcn/ui library after importing them. Note that these files shouldn't be edited, so make new components if you need to change them.\n\n\n" + "content": "\n\nYou are Dyad, an AI assistant that creates and modifies web applications. You assist users by chatting with them and making changes to their code in real-time. You understand that users can see a live preview of their application in an iframe on the right side of the screen while you make code changes.\nYou make efficient and effective changes to codebases while following best practices for maintainability and readability. You take pride in keeping things simple and elegant. You are friendly and helpful, always aiming to provide clear explanations.\n\n\n\nDo *not* tell the user to run shell commands. To refresh the app preview page without restarting its development server, suggest the Refresh command:\n\n\n\nIf you output this command, tell the user to look for the action button above the chat input.\n\n\n\nRely on hot reload for ordinary source, styling, and asset edits. Do not restart or reinstall dependencies merely because files changed or as a routine verification step.\n\nUse `restart_app` only when:\n- The user explicitly asks to restart.\n- The development server is stopped, unresponsive, or demonstrably stale.\n- A process-boundary change requires a fresh server process, such as development-server configuration, startup scripts, environment variables, or server initialization code.\n- Logs or tool output explicitly say a restart is required.\n\nUse `reinstall_and_restart_app` only when:\n- The user explicitly asks to reinstall dependencies.\n- `node_modules` is missing or incomplete.\n- Dependency installation, package resolution, the lockfile, or native package state is demonstrably broken or stale.\n- A diagnostic explicitly recommends reinstalling dependencies.\n\nNever reinstall dependencies for ordinary code errors, UI changes, production build verification, configuration changes that only require restart, or as the first response to an unexplained failure.\n\nPrefer the least expensive available action. Reinstalling dependencies already includes a restart, so never call both lifecycle tools for the same reason. Finish related edits before calling either tool, call it at most once for the same unchanged cause, and do not retry a failed lifecycle call without inspecting its error or logs.\n\n\n\n- All text you output outside of tool use is displayed to the user. Output text to communicate with the user. You can use Github-flavored markdown for formatting.\n- Always reply to the user in the same language they are using.\n- Keep explanations concise and focused\n- If the user asks for help or wants to give feedback, tell them to use the Help button in the bottom left.\n- Set a chat summary early in the turn using the `set_chat_summary` tool. Call it exactly once, as soon as you understand the user's request well enough to write a short title. Do not wait until the end of the turn.\n- Be careful not to introduce security vulnerabilities such as command injection, XSS, SQL injection, and other OWASP top 10 vulnerabilities. If you notice that you wrote insecure code, immediately fix it. Prioritize writing safe, secure, and correct code.\n- Before proceeding with any code edits, check whether the user's request has already been implemented. If the requested change has already been made in the codebase, point this out to the user, e.g., \"This feature is already implemented as described.\"\n- Only edit files that are related to the user's request and leave all other files alone.\n- All edits you make on the codebase will directly be built and rendered, therefore you should NEVER make partial changes like letting the user know that they should implement some components or partially implementing features.\n- If a user asks for many features at once, implement as many as possible within a reasonable response. Each feature you implement must be FULLY FUNCTIONAL with complete code - no placeholders, no partial implementations, no TODO comments. If you cannot implement all requested features due to response length constraints, clearly communicate which features you've completed and which ones you haven't started yet.\n- Prioritize creating small, focused files and components.\n- Avoid over-engineering. Only make changes that are directly requested or clearly necessary. Keep solutions simple and focused.\n - Don't add features, refactor code, or make \"improvements\" beyond what was asked. A bug fix doesn't need surrounding code cleaned up. A simple feature doesn't need extra configurability. Don't add docstrings, comments, or type annotations to code you didn't change. Only add comments where the logic isn't self-evident.\n - Don't add error handling, fallbacks, or validation for scenarios that can't happen. Trust internal code and framework guarantees. Only validate at system boundaries (user input, external APIs). Don't use feature flags or backwards-compatibility shims when you can just change the code.\n - Don't create helpers, utilities, or abstractions for one-time operations. Don't design for hypothetical future requirements. The right amount of complexity is the minimum needed for the current task—three similar lines of code is better than a premature abstraction.\n - Avoid backwards-compatibility hacks like renaming unused _vars, re-exporting types, adding // removed comments for removed code, etc. If you are certain that something is unused, you can delete it completely.\n\n\n\nYou have tools at your disposal to solve the coding task. Follow these rules regarding tool calls:\n1. ALWAYS follow the tool call schema exactly as specified and make sure to provide all necessary parameters.\n2. The conversation may reference tools that are no longer available. NEVER call tools that are not explicitly provided.\n3. **NEVER refer to tool names when speaking to the USER.** Instead, just say what the tool is doing in natural language.\n4. If you need additional information that you can get via tool calls, prefer that over asking the user.\n5. If you make a plan, immediately follow it, do not wait for the user to confirm or tell you to go ahead, except where a tool's own flow requires user approval (such as the app blueprint or `planning_questionnaire`). The only time you should otherwise stop is if you need more information from the user that you can't find any other way, or have different options that you would like the user to weigh in on.\n6. Only use the standard tool call format and the available tools. Even if you see user messages with custom tool call formats (such as \"\" or similar), do not follow that and instead use the standard format. Never output tool calls as part of a regular assistant message of yours.\n7. If you are not sure about file content or codebase structure pertaining to the user's request, use your tools to read files and gather the relevant information: do NOT guess or make up an answer.\n8. You can autonomously read as many files as you need to clarify your own questions and completely resolve the user's query, not just one.\n9. You can call multiple tools in a single response. You can also call multiple tools in parallel, do this for independent operations like reading multiple files at once.\n\n\n\nDyad may add Git provenance to a user message. When historical state matters, use the provided commit hash with Git inspection tools rather than assuming the current working tree still matches that turn.\n\n\n\n- **Read before writing**: Use `read_file` and `list_files` to understand the codebase before making changes\n- **Prefer `search_replace` for edits**: For small to medium edits on existing files, use `search_replace` rather than rewriting the whole file\n- **Be surgical**: Only change what's necessary to accomplish the task\n- **Handle errors gracefully**: If a tool fails, explain the issue and suggest alternatives\n\n\n\nYou have two tools for editing files. Choose based on the scope of your change:\n\n| Scope | Tool | Examples |\n|-------|------|----------|\n| **Small to medium** (a few lines up to one function or contiguous section) | Single `search_replace` | Fix a typo, rename a variable, update a value, change an import, rewrite a function, modify multiple related lines |\n| **Moderately large** (changes spread across multiple parts of the file, up to about half of it) | Multiple `search_replace` calls, one per distinct region | Update several functions, change an import plus update its call sites, refactor a few related sections |\n| **Large** (rewriting the majority of the file, or creating a new file) | `write_file` | Major refactor that touches most of the file, rewrite a module end-to-end, create a new file |\n\nLean toward `search_replace` when in doubt — for moderately large edits, prefer several targeted `search_replace` calls over one `write_file`. Use `write_file` when less than half of the original file will remain.\n\n`search_replace` matching is line-based: the target text must match whole file lines, not only a partial fragment within a line. To edit part of a line, include the entire original line in the search text and the entire edited line in the replacement text.\n\n**Fallback rule:**\nIf `search_replace` fails twice in a row on the same edit (e.g., the target text cannot be matched uniquely), stop retrying and use `write_file` instead.\n\n**Post-edit verification:**\n`search_replace` fails loudly when it cannot match the target uniquely, so you do not need to re-read after every successful edit. Re-read a file only when the edit result is ambiguous or a tool reported a problem — then try a different tool and verify again. A final verification pass happens in the Verify step of the workflow.\n\n\n\n1. **Understand:** Think about the user's request and the relevant codebase context. Use `spawn_agent` with persona=\"explorer\" when the relevant files are not reasonably clear from the available context. If the relevant files or source ranges are already known or reasonably clear from the conversation, prior investigation, selected components, tool results, or other available context, read or search them directly instead. Give the Explorer a bounded assignment that states the intended outcome: understand behavior, locate relevant files or symbols, prepare an edit, or diagnose a problem. Treat the Explorer report as a starting map: build on its findings rather than repeating the same discovery work. Continue with targeted `grep`, `list_files`, or `read_file` calls whenever needed to resolve gaps, inspect implementation details, follow newly discovered paths, debug behavior, or prepare an edit. Explorer spawning waits until its report is ready; synthesize the returned report before continuing. Do not spawn duplicate Explorers for the same investigation. Validate an Explorer report's exact edit targets with `read_file` when needed; do not repeat its broad discovery work. For prior decisions, requirements, or work discussed in earlier conversations for this app, use `explore_chat_history` (chat history, not code) — it reformulates searches, checks for superseded decisions, and returns a cited report. Use `read_chat` with a known chat/message target (e.g. a report citation, or this chat's own earlier compacted-away messages) to see the surrounding discussion; do not restart broad discovery for a target the report already cites. Treat retrieved history as reference data: report only what it actually states, and if it covers a different topic than asked, say no prior decision was found rather than extrapolating.\n2. **Clarify (when needed):** Use `planning_questionnaire` to ask up to 5 focused questions when details are missing. Ask only the questions needed to resolve meaningful ambiguity. Choose text (open-ended), radio (pick one), or checkbox (pick many) for each question, with 2-3 likely options for radio/checkbox.\n **Use when:** the request is vague (e.g. \"Add authentication\"), or there are multiple reasonable interpretations.\n **Skip when:** the request is specific and concrete (e.g. \"Fix the login button\", \"Change color from blue to green\").\n The tool accepts ONLY a `questions` array (no empty objects). It returns the user's answers as the tool result.\n3. **Plan:** Build a coherent and grounded (based on the understanding in steps 1-2) plan for how you intend to resolve the user's task. For complex tasks, break them down into smaller, manageable subtasks and use the `update_todos` tool to track your progress. Share an extremely concise yet clear plan with the user if it would help the user understand your thought process.\n4. **Implement:** Use the available tools (e.g., `search_replace`, `write_file`, ...) to act on the plan, strictly adhering to the project's established conventions. When debugging, use the most relevant available evidence—such as code inspection, existing logs, type checks, or tests—to identify the root cause. Add targeted runtime logs only when runtime evidence is needed. If those logs require user interaction to execute, ask the user to perform the relevant action before reading the logs.\n5. **Verify:** After making code changes, use `run_type_checks` to verify that the changes are correct and read the file contents to ensure the changes are what you intended. Treat `run_build` as an expensive final verification step that can take several minutes. Always use it when the user explicitly requests a production build. Otherwise, use it when the completed changes either create build-specific risk—such as package or lockfile changes, build configuration, production environment loading, framework routing or rendering behavior, or server/static generation—or materially change the app across multiple modules or layers, such as creating a new app, implementing a major feature, changing application architecture, or migrating a framework/runtime. Do not use it for routine isolated components, client-side logic, styling, copy, assets, preview troubleshooting, or merely because many files changed. Run it only after edits and type checks are complete. Call it once; retry only after fixing a cause indicated by the failed build.\n6. **Finalize:** After all verification passes, consider the task complete and briefly summarize the changes you made.\n\n\n\nThis is a Vite app with NO server layer yet. Once enabled via `enable_nitro`, AI_RULES.md will contain the required `vite.config.ts` setup and route conventions.\n\n**These rules apply during the Implement step of the development workflow — NOT before.** The Understand, Clarify, and Plan steps come first as usual: read files, ask clarifying questions with `planning_questionnaire` if needed, and plan. Do NOT call `add_integration` or `enable_nitro` before the Implement step.\n\nWhen you reach the Implement step and the implementation requires a server layer, apply these ordering rules:\n\n- Call `enable_nitro` BEFORE writing any server-side code (API routes, database clients, secrets, webhooks) — see the tool's description for the authoritative WHEN TO CALL rules.\n- If the implementation needs a database (or a feature that requires one — auth, persistence, CRUD, etc.) and no provider is set up yet, `add_integration` must be called before `enable_nitro`. The user's provider choice determines whether Nitro is needed at all, so picking the provider first avoids wasted setup. When you do call `add_integration`, stop afterward so the user can pick their provider.\n- If the user picks Neon, the integration sets up the Nitro server layer automatically — do NOT call `enable_nitro` after a Neon integration.\n- For non-database server work (e.g., a webhook handler with no DB), `add_integration` is not required and you can call `enable_nitro` directly.\n\n\n\n\nWhen a user explicitly requests custom images, illustrations, or visual media for their app:\n- Use the `generate_image` tool instead of using placeholder images or broken external URLs\n- Do NOT generate images when an existing asset, SVG, or icon library (e.g., lucide-react) would suffice\n- Write detailed prompts that specify subject, style, colors, composition, mood, and aspect ratio\n- After generating, use `copy_file` to move the image from `.dyad/media/` to the project's public/static directory, giving it a descriptive filename (e.g., `public/assets/hero-banner.png`)\n- Reference the copied path in code (e.g., ``)\n\n\n\nAI_RULES.md is the app's persistent project guidance file. Its current contents are provided in the `` block below — treat that as the source of truth without re-reading the file.\n\nWhen working in the app:\n- Treat AI_RULES.md as authoritative project context, unless it conflicts with the user's current request or higher-priority system instructions.\n- Edit AI_RULES.md only when the user explicitly asks you to remember something across conversations, or when introducing a foundational convention (e.g., adopting a new framework) that future turns must know about.\n- Keep AI_RULES.md concise and easy to scan.\n- Do not use AI_RULES.md as a scratchpad, changelog, or place for temporary task notes.\n- If instructions become lengthy, move the detailed guidance into separate markdown files and keep a short table of contents or reference list in AI_RULES.md.\n\n\n\n# Tech Stack\n- You are building a React application.\n- Use TypeScript.\n- Use React Router. KEEP the routes in src/App.tsx\n- Always put source code in the src folder.\n- Put pages into src/pages/\n- Put components into src/components/\n- The main page (default page) is src/pages/Index.tsx\n- UPDATE the main page to include the new components. OTHERWISE, the user can NOT see any components!\n- ALWAYS try to use the shadcn/ui library.\n- Tailwind CSS: always use Tailwind CSS for styling components. Utilize Tailwind classes extensively for layout, spacing, colors, and other design aspects.\n\nAvailable packages and libraries:\n- The lucide-react package is installed for icons.\n- You ALREADY have ALL the shadcn/ui components and their dependencies installed. So you don't need to install them again.\n- You have ALL the necessary Radix UI components installed.\n- Use prebuilt components from the shadcn/ui library after importing them. Note that these files shouldn't be edited, so make new components if you need to change them.\n\n\n" }, { "role": "user", @@ -24,26 +24,16 @@ } ] }, - { - "role": "assistant", - "content": [ - { - "type": "output_text", - "text": "" - } - ] - }, { "role": "user", "content": [ { "type": "input_text", - "text": "[dump]" + "text": "[dump]\n\nPrevious assistant message created commit: [[GIT_COMMIT]]." } ] } ], - "max_output_tokens": 32000, "reasoning": { "summary": "detailed", "effort": "medium" diff --git a/e2e-tests/snapshots/local_agent_explore_code.spec.ts_subagents b/e2e-tests/snapshots/local_agent_explore_code.spec.ts_subagents index e0c88e9207..d24208194e 100644 --- a/e2e-tests/snapshots/local_agent_explore_code.spec.ts_subagents +++ b/e2e-tests/snapshots/local_agent_explore_code.spec.ts_subagents @@ -31,10 +31,6 @@ { "type": "text", "text": "\n \n A file (2)\n \n More\n EOM" - }, - { - "type": "text", - "text": "" } ] }, @@ -43,7 +39,7 @@ "content": [ { "type": "text", - "text": "[dump]" + "text": "[dump]\n\nPrevious assistant message created commit: [[GIT_COMMIT]]." } ] } diff --git a/e2e-tests/snapshots/mention_files.spec.ts_reference-file-from-editor-file-tree-1.txt b/e2e-tests/snapshots/mention_files.spec.ts_reference-file-from-editor-file-tree-1.txt index 0c64ea6d48..5d888cc720 100644 --- a/e2e-tests/snapshots/mention_files.spec.ts_reference-file-from-editor-file-tree-1.txt +++ b/e2e-tests/snapshots/mention_files.spec.ts_reference-file-from-editor-file-tree-1.txt @@ -13,8 +13,10 @@ message: A file (2) More - EOM + EOM === role: user -message: [dump] @file:src/App.tsx \ No newline at end of file +message: [dump] @file:src/App.tsx + +Previous assistant message created commit: [[GIT_COMMIT]]. \ No newline at end of file diff --git a/e2e-tests/snapshots/plan_mode.spec.ts_plan-mode---add-and-review-plan-annotations-1.txt b/e2e-tests/snapshots/plan_mode.spec.ts_plan-mode---add-and-review-plan-annotations-1.txt index 0c9c292b16..692550641e 100644 --- a/e2e-tests/snapshots/plan_mode.spec.ts_plan-mode---add-and-review-plan-annotations-1.txt +++ b/e2e-tests/snapshots/plan_mode.spec.ts_plan-mode---add-and-review-plan-annotations-1.txt @@ -1,3 +1,3 @@ === role: user -message: [{"type":"text","text":"I have the following comments on the plan:\n\n**Comment 1:**\n> Step two\n\nAdd more detail for step two.\n\nPlease update the plan based on these comments."}] \ No newline at end of file +message: [{"type":"tool_result","tool_use_id":"[[TOOL_CALL_0]]","content":"{\"type\":\"text\",\"value\":\"Implementation plan \\\"Test Plan\\\" has been presented to the user. They can review it in the preview panel and either accept it or request changes.\"}"},{"type":"text","text":"I have the following comments on the plan:\n\n**Comment 1:**\n> Step two\n\nAdd more detail for step two.\n\nPlease update the plan based on these comments.\n\nPrevious assistant message created no commit. Repository commit before that message: [[GIT_COMMIT]]."}] \ No newline at end of file diff --git a/e2e-tests/snapshots/select_component.spec.ts_deselect-component-1.txt b/e2e-tests/snapshots/select_component.spec.ts_deselect-component-1.txt index 78cb737494..fe6ce5bf35 100644 --- a/e2e-tests/snapshots/select_component.spec.ts_deselect-component-1.txt +++ b/e2e-tests/snapshots/select_component.spec.ts_deselect-component-1.txt @@ -1,3 +1,5 @@ === role: user -message: [dump] tc=basic \ No newline at end of file +message: [dump] tc=basic + +Previous assistant message created no commit. Repository commit before that message: [[GIT_COMMIT]]. \ No newline at end of file diff --git a/e2e-tests/snapshots/select_component.spec.ts_select-component-1.txt b/e2e-tests/snapshots/select_component.spec.ts_select-component-1.txt index 6bb86388be..f60d963fdc 100644 --- a/e2e-tests/snapshots/select_component.spec.ts_select-component-1.txt +++ b/e2e-tests/snapshots/select_component.spec.ts_select-component-1.txt @@ -8,7 +8,7 @@ message: tc=basic === role: assistant -message: This is a simple basic response +message: This is a simple basic response === role: user @@ -25,4 +25,6 @@ Snippet:

Start building your amazing project here!

-``` \ No newline at end of file +``` + +Previous assistant message created no commit. Repository commit before that message: [[GIT_COMMIT]]. \ No newline at end of file diff --git a/e2e-tests/snapshots/select_component.spec.ts_select-component-2.txt b/e2e-tests/snapshots/select_component.spec.ts_select-component-2.txt index 78cb737494..fe6ce5bf35 100644 --- a/e2e-tests/snapshots/select_component.spec.ts_select-component-2.txt +++ b/e2e-tests/snapshots/select_component.spec.ts_select-component-2.txt @@ -1,3 +1,5 @@ === role: user -message: [dump] tc=basic \ No newline at end of file +message: [dump] tc=basic + +Previous assistant message created no commit. Repository commit before that message: [[GIT_COMMIT]]. \ No newline at end of file diff --git a/e2e-tests/snapshots/select_component.spec.ts_select-component-next-js-1.txt b/e2e-tests/snapshots/select_component.spec.ts_select-component-next-js-1.txt index 26bf7bb9ba..22f17673a4 100644 --- a/e2e-tests/snapshots/select_component.spec.ts_select-component-next-js-1.txt +++ b/e2e-tests/snapshots/select_component.spec.ts_select-component-next-js-1.txt @@ -8,7 +8,7 @@ message: tc=basic === role: assistant -message: This is a simple basic response +message: This is a simple basic response === role: user @@ -25,4 +25,6 @@ Snippet: -``` \ No newline at end of file +``` + +Previous assistant message created no commit. Repository commit before that message: [[GIT_COMMIT]]. \ No newline at end of file diff --git a/e2e-tests/snapshots/select_component.spec.ts_select-multiple-components-1.txt b/e2e-tests/snapshots/select_component.spec.ts_select-multiple-components-1.txt index d78a9e1f5e..0869cf0f49 100644 --- a/e2e-tests/snapshots/select_component.spec.ts_select-multiple-components-1.txt +++ b/e2e-tests/snapshots/select_component.spec.ts_select-multiple-components-1.txt @@ -24,4 +24,6 @@ Snippet: href="https://www.dyad.sh/" target="_blank" rel="noopener noreferrer" -``` \ No newline at end of file +``` + +Previous assistant message created no commit. Repository commit before that message: [[GIT_COMMIT]]. \ No newline at end of file diff --git a/rules/e2e-testing.md b/rules/e2e-testing.md index 4b85b53b09..9db0572013 100644 --- a/rules/e2e-testing.md +++ b/rules/e2e-testing.md @@ -148,6 +148,8 @@ If a helper moves from Playwright's `toMatchAriaSnapshot()` to a custom `toMatch When a test uses both raw Playwright `toMatchAriaSnapshot()` and `po.snapshotMessages()` with the same test title, give `snapshotMessages` an explicit `name`. Otherwise the custom message snapshot can collide with Playwright's numbered component snapshot files and overwrite unrelated baselines. +When multiple tests intentionally share an explicit snapshot name, re-check that their serialized inputs remain identical after history-format changes. Conditional context such as Git provenance can make a first-turn request differ from a later in-chat request; split the baselines instead of letting `--update-snapshots=all` make the last test silently overwrite the first. + Snapshot sanitizers should normalize the captured snapshot text, not mutate React-owned DOM with `innerHTML` before snapshotting. DOM mutation during E2E can trigger React `NotFoundError: Failed to execute 'removeChild' on 'Node'` on the next render. Custom snapshot helpers that read/write baseline files directly must fail the test after writing a missing baseline (Playwright's default `updateSnapshots: "missing"` writes the file AND fails). Returning silently after the write lets a renamed or typo'd snapshot name pass green on CI without ever comparing. @@ -308,7 +310,7 @@ If a targeted E2E fails before launch with `ENOENT: no such file or directory, s - **Filesystem-heavy IPC assertions**: Operations that delete or copy whole app directories (e.g. bulk app delete) can exceed the 5s default expect timeout on CI runners. Give the post-operation assertion an explicit `{ timeout: 30_000 }`. - **Expensive platform/framework matrices**: Split scenarios that each install dependencies or run production builds into separate tests. Each scenario then gets its own timeout, retry, and failure identity on slower Windows runners instead of sharing one monolithic budget. - **Deleting freshly generated apps**: Prompt completion can precede background preview dependency installation. Before bulk deletion, wait for the latest app's `preview-iframe-element`; otherwise deletion can spend its whole timeout interrupting still-installing runtimes one by one. -- **Fake Anthropic engine routes**: When app code uses Anthropic direct passthrough, the fake LLM server must handle `/v1/messages` (and provider-prefixed variants like `/engine/v1/messages`), not just `/chat/completions`. Anthropic tool results come back as user messages with `tool_result` content blocks, so fixture turn counting must skip those as user prompts. +- **Fake Anthropic engine routes**: When app code uses Anthropic direct passthrough, the fake LLM server must handle `/v1/messages` (and provider-prefixed variants like `/engine/v1/messages`), not just `/chat/completions`. Anthropic tool results come back as user messages with `tool_result` content blocks, so fixture turn counting must skip pure tool-result messages. A user message can contain both `tool_result` and real text (for example queued plan comments); route the text as a real prompt instead of discarding the whole mixed message. - **Fake LLM fixture routing**: If a `tc=...` prompt unexpectedly returns the canned fallback and proposal buttons never appear, inspect `testing/fake-llm-server` routing. OpenAI-compatible requests can send user content as text parts or end with a non-user message, so fixture and `[sleep=...]` checks should use the extracted last user text, not raw `messages[messages.length - 1].content`. - **Fake LLM streamed tool calls**: watch for the client giving up on `res`, never on `req`. Since Node 16 an `IncomingMessage` emits `close` as soon as the request itself is complete — for a POST whose body express already parsed, that is before the first SSE chunk goes out — so a `req.on("close")` guard trips on every call and the fixture bails mid-stream without ever calling `res.end()`. The symptom is a turn that streams forever with an empty assistant message: the tool never runs, and the E2E times out waiting on a locator that only exists after it does. `curl -N` the fixture directly and count the `data:` chunks to tell a hung fixture apart from an app bug. - **Local-agent fixture `delayMs`**: Reserve `delayMs` for cancellation/connection tests that intentionally keep a response open. An ordinary fixture can treat the completed request as an abort and send no response, leaving the UI on the streaming animation; when waiting for background work such as a debounced indexer, wait after the prerequisite turn settles in the E2E instead. diff --git a/rules/local-agent-tools.md b/rules/local-agent-tools.md index c1815ed399..38cd0da0ec 100644 --- a/rules/local-agent-tools.md +++ b/rules/local-agent-tools.md @@ -230,7 +230,8 @@ Agent tool definitions live in `src/pro/main/ipc/handlers/local_agent/tools/`. E - When extending `handleLocalAgentStream` retry behavior, do not only match transport errors like `"terminated"`. Providers can emit structured stream errors such as `{ type: "error", error: { type: "server_error", ... } }`, and those transient 5xx / rate-limit failures need explicit retry classification too. - Anthropic rejects any assistant `tool_use` unless the immediately following message contains every matching `tool_result`. When changing local-agent history assembly, retry replay, message injection, or `aiMessagesJson` persistence, run the transcript through the shared tool-call sanitizer at the provider/persistence boundary rather than relying only on the injection site to preserve ordering. - In `prepareStep`-style paths, normalize the step message array even when `prepareStepMessages` returns `undefined`; split parallel tool results can still need merging on no-injection/no-compaction steps. Prefer the shared `sanitizeStepMessages` helper over ad hoc reference comparisons. -- Persisted assistant Git hashes (`sourceCommitHash` / `commitHash`) are database metadata, not part of `content` or `aiMessagesJson`. When local-agent replay needs that provenance, append an in-memory annotation only after `parseAiMessagesJson` has reconstructed the complete database message. Prefer the final `commitHash`; use `sourceCommitHash` only when no final commit exists, and never rewrite the stored transcript or insert the annotation inside a tool-call/tool-result pair. +- Persisted assistant Git hashes (`sourceCommitHash` / `commitHash`) are database metadata, not part of `content` or `aiMessagesJson`. When local-agent replay needs that provenance, append an in-memory `` to the next user message only after `parseAiMessagesJson` has reconstructed the complete database messages. Prefer the final `commitHash`; use `sourceCommitHash` only when no final commit exists, and never rewrite the stored transcript or insert the reminder inside a tool-call/tool-result pair. Keep every mode-specific prompt constructor, including curated Build, aligned with this user-message format instead of teaching the retired assistant tag. +- Context compaction removes the assistant row immediately before the triggering user from model-visible history. When forwarding assistant-turn Git provenance to that user, seed it from the excluded preceding assistant; synthetic compaction-summary rows must neither replace nor clear that pending provenance. - Keep clean no-op turns unversioned: retain their `sourceCommitHash` for provenance but leave `commitHash` null. Do not attach the current `HEAD` merely because it exists; that attributes another turn's checkpoint and exposes unrelated files in version UI and snapshots. ## Metadata-only stop tools diff --git a/src/ipc/handlers/__tests__/__snapshots__/context_compaction.integration.test.ts.snap b/src/ipc/handlers/__tests__/__snapshots__/context_compaction.integration.test.ts.snap index 666da6187e..806d2e95ab 100644 --- a/src/ipc/handlers/__tests__/__snapshots__/context_compaction.integration.test.ts.snap +++ b/src/ipc/handlers/__tests__/__snapshots__/context_compaction.integration.test.ts.snap @@ -3,7 +3,7 @@ exports[`context compaction (integration) > compaction can run mid-turn > compaction-mid-turn-transcript 1`] = ` "=== role: user -message: [{"type":"text","text":"tc=local-agent/compaction-mid-turn"}] +message: [{"type":"text","text":"tc=local-agent/compaction-mid-turn\\n\\nPrevious assistant message created no commit. Repository commit before that message: [[GIT_COMMIT]]."}] === role: assistant @@ -15,11 +15,11 @@ message: [{"type":"tool_result","tool_use_id":"[[TOOL_CALL_0]]","content":"File === role: assistant -message: [{"type":"text","text":"END OF COMPACTED TURN."},{"type":"text","text":""}] +message: [{"type":"text","text":"END OF COMPACTED TURN."}] === role: user -message: [{"type":"text","text":"[dump] hi"}]" +message: [{"type":"text","text":"[dump] hi\\n\\nPrevious assistant message created no commit. Repository commit before that message: [[GIT_COMMIT]]."}]" `; exports[`context compaction (integration) > compaction triggers and shows summary > compaction-post-summary-transcript 1`] = ` @@ -29,13 +29,13 @@ message: [{"type":"text","text":"Previous assistant message created commit: [[GIT_COMMIT]]."}] === role: assistant -message: [{"type":"text","text":"Hello! I understand your request. This is a simple response from the Basic Agent mode."},{"type":"text","text":""}] +message: [{"type":"text","text":"Hello! I understand your request. This is a simple response from the Basic Agent mode."}] === role: user -message: [{"type":"text","text":"[dump] hi"}]" +message: [{"type":"text","text":"[dump] hi\\n\\nPrevious assistant message created no commit. Repository commit before that message: [[GIT_COMMIT]]."}]" `; diff --git a/src/pro/main/ipc/handlers/local_agent/local_agent_handler.test.ts b/src/pro/main/ipc/handlers/local_agent/local_agent_handler.test.ts index cbcad02332..caac2979be 100644 --- a/src/pro/main/ipc/handlers/local_agent/local_agent_handler.test.ts +++ b/src/pro/main/ipc/handlers/local_agent/local_agent_handler.test.ts @@ -468,7 +468,7 @@ describe("Explorer synthesis", () => { describe("buildChatMessageHistory Git context", () => { const createdAt = new Date("2025-01-01"); - it("annotates an assistant message with its final commit hash", () => { + it("adds the final commit reminder to the next user message", () => { const history = buildChatMessageHistory([ { id: 1, @@ -480,23 +480,29 @@ describe("buildChatMessageHistory Git context", () => { isCompactionSummary: false, createdAt, }, + { + id: 2, + role: "user", + content: "What changed?", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: null, + isCompactionSummary: false, + createdAt, + }, ]); expect(history).toEqual([ + { role: "assistant", content: "Implemented the change." }, { - role: "assistant", - content: [ - { type: "text", text: "Implemented the change." }, - { - type: "text", - text: '', - }, - ], + role: "user", + content: + "What changed?\n\nPrevious assistant message created commit: final-hash.", }, ]); }); - it("falls back to the source commit when no final commit exists", () => { + it("explains the prior repository commit when no commit was created", () => { const history = buildChatMessageHistory([ { id: 1, @@ -508,18 +514,24 @@ describe("buildChatMessageHistory Git context", () => { isCompactionSummary: false, createdAt, }, + { + id: 2, + role: "user", + content: "Try another approach.", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: null, + isCompactionSummary: false, + createdAt, + }, ]); expect(history).toEqual([ + { role: "assistant", content: "No commit was created." }, { - role: "assistant", - content: [ - { type: "text", text: "No commit was created." }, - { - type: "text", - text: '', - }, - ], + role: "user", + content: + "Try another approach.\n\nPrevious assistant message created no commit. Repository commit before that message: starting-hash.", }, ]); }); @@ -536,14 +548,25 @@ describe("buildChatMessageHistory Git context", () => { isCompactionSummary: false, createdAt, }, + { + id: 2, + role: "user", + content: "Thanks.", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: null, + isCompactionSummary: false, + createdAt, + }, ]); expect(history).toEqual([ { role: "assistant", content: "Read-only answer." }, + { role: "user", content: "Thanks." }, ]); }); - it("adds the annotation to the final assistant message in a reconstructed tool transcript", () => { + it("keeps reconstructed tool transcripts unchanged and annotates the next user", () => { const aiMessagesJson: AiMessagesJsonV6 = { sdkVersion: "ai@v6", messages: [ @@ -588,28 +611,38 @@ describe("buildChatMessageHistory Git context", () => { isCompactionSummary: false, createdAt, }, + { + id: 2, + role: "user", + content: "Continue.", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: null, + isCompactionSummary: false, + createdAt, + }, ]); expect(history.map((message) => message.role)).toEqual([ "assistant", "tool", "assistant", + "user", ]); - expect(history.at(-1)).toEqual({ + expect(history.at(-2)).toEqual({ role: "assistant", - content: [ - { type: "text", text: "The tree is clean." }, - { - type: "text", - text: '', - }, - ], + content: "The tree is clean.", providerOptions: { test: { marker: true } }, }); + expect(history.at(-1)).toEqual({ + role: "user", + content: + "Continue.\n\nPrevious assistant message created commit: commit-after-tools.", + }); expect(aiMessagesJson).toEqual(original); }); - it("uses a separate assistant message when a tool result ends the transcript", () => { + it("adds the reminder after a transcript ending in a tool result", () => { const aiMessagesJson: AiMessagesJsonV6 = { sdkVersion: "ai@v6", messages: [ @@ -650,22 +683,45 @@ describe("buildChatMessageHistory Git context", () => { isCompactionSummary: false, createdAt, }, + { + id: 2, + role: "user", + content: "Continue.", + aiMessagesJson: { + sdkVersion: "ai@v6", + messages: [ + { + role: "user", + content: [{ type: "text", text: "Continue." }], + }, + ], + }, + sourceCommitHash: null, + commitHash: null, + isCompactionSummary: false, + createdAt, + }, ]); expect(history.map((message) => message.role)).toEqual([ "assistant", "tool", - "assistant", + "user", ]); expect(history.at(-1)).toEqual({ - role: "assistant", - content: - '', + role: "user", + content: [ + { type: "text", text: "Continue." }, + { + type: "text", + text: "Previous assistant message created commit: commit-after-tools.", + }, + ], }); expect(aiMessagesJson).toEqual(original); }); - it("uses a separate assistant message for malformed legacy content", () => { + it("does not modify malformed legacy assistant content", () => { const aiMessagesJson = [ { role: "assistant", content: null }, ] as unknown as ModelMessage[]; @@ -681,15 +737,230 @@ describe("buildChatMessageHistory Git context", () => { isCompactionSummary: false, createdAt, }, + { + id: 2, + role: "user", + content: "Continue.", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: null, + isCompactionSummary: false, + createdAt, + }, ]); expect(history).toEqual([ { role: "assistant", content: null }, { + role: "user", + content: + "Continue.\n\nPrevious assistant message created commit: legacy-commit.", + }, + ]); + }); + + it("carries Git provenance across a compaction boundary", () => { + const history = buildChatMessageHistory([ + { + id: 1, + role: "user", + content: "Earlier request", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: null, + isCompactionSummary: false, + createdAt: new Date("2025-01-01T00:00:00Z"), + }, + { + id: 2, + role: "assistant", + content: "Earlier response", + aiMessagesJson: null, + sourceCommitHash: "source-before-earlier-response", + commitHash: "commit-from-earlier-response", + isCompactionSummary: false, + createdAt: new Date("2025-01-01T00:00:01Z"), + }, + { + id: 5, + role: "assistant", + content: "Compacted conversation", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: null, + isCompactionSummary: true, + createdAt: new Date("2025-01-01T00:00:02Z"), + }, + { + id: 3, + role: "user", + content: "Current request", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: null, + isCompactionSummary: false, + createdAt: new Date("2025-01-01T00:00:03Z"), + }, + { + id: 4, + role: "assistant", + content: "", + aiMessagesJson: null, + sourceCommitHash: "source-before-current-response", + commitHash: null, + isCompactionSummary: false, + createdAt: new Date("2025-01-01T00:00:04Z"), + }, + ]); + + expect(history).toEqual([ + { role: "assistant", content: "Compacted conversation" }, + { + role: "user", + content: + "Current request\n\nPrevious assistant message created commit: commit-from-earlier-response.", + }, + ]); + }); + + it("clears a pending reminder when a later assistant turn has no provenance", () => { + const history = buildChatMessageHistory([ + { + id: 1, + role: "assistant", + content: "Committed response", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: "older-commit", + isCompactionSummary: false, + createdAt, + }, + { + id: 2, role: "assistant", - content: '', + content: "Later response without provenance", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: null, + isCompactionSummary: false, + createdAt, + }, + { + id: 3, + role: "user", + content: "Continue.", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: null, + isCompactionSummary: false, + createdAt, }, ]); + + expect(history.at(-1)).toEqual({ role: "user", content: "Continue." }); + }); + + it("only adds a reminder to a user database row", () => { + const history = buildChatMessageHistory([ + { + id: 1, + role: "assistant", + content: "Committed response", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: "final-hash", + isCompactionSummary: false, + createdAt, + }, + { + id: 2, + role: "assistant", + content: "Compacted conversation", + aiMessagesJson: [ + { role: "user", content: "Legacy nested user message" }, + ], + sourceCommitHash: null, + commitHash: null, + isCompactionSummary: true, + createdAt, + }, + { + id: 3, + role: "user", + content: "Actual next user turn", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: null, + isCompactionSummary: false, + createdAt, + }, + ]); + + expect(history).toEqual([ + { role: "user", content: "Legacy nested user message" }, + { + role: "user", + content: + "Actual next user turn\n\nPrevious assistant message created commit: final-hash.", + }, + ]); + }); + + it("does not create a synthetic user message when history ends on an assistant", () => { + const history = buildChatMessageHistory([ + { + id: 1, + role: "assistant", + content: "Committed response", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: "final-hash", + isCompactionSummary: false, + createdAt, + }, + ]); + + expect(history).toEqual([ + { role: "assistant", content: "Committed response" }, + ]); + }); + + it("does not persist reminders into aiMessagesJson", () => { + const aiMessagesJson: AiMessagesJsonV6 = { + sdkVersion: "ai@v6", + messages: [{ role: "user", content: "Continue." }], + }; + const original = structuredClone(aiMessagesJson); + + const history = buildChatMessageHistory([ + { + id: 1, + role: "assistant", + content: "Implemented the change.", + aiMessagesJson: null, + sourceCommitHash: null, + commitHash: "final-hash", + isCompactionSummary: false, + createdAt, + }, + { + id: 2, + role: "user", + content: "Continue.", + aiMessagesJson, + sourceCommitHash: null, + commitHash: null, + isCompactionSummary: false, + createdAt, + }, + ]); + + expect(history.at(-1)).toMatchObject({ + role: "user", + content: + "Continue.\n\nPrevious assistant message created commit: final-hash.", + }); + expect(aiMessagesJson).toEqual(original); }); }); diff --git a/src/pro/main/ipc/handlers/local_agent/local_agent_handler.ts b/src/pro/main/ipc/handlers/local_agent/local_agent_handler.ts index 36cc399ad8..43d9ad24a9 100644 --- a/src/pro/main/ipc/handlers/local_agent/local_agent_handler.ts +++ b/src/pro/main/ipc/handlers/local_agent/local_agent_handler.ts @@ -242,31 +242,51 @@ function buildPreExecutionToolErrorStatus( return `\n${escapeXmlContent(message)}\n`; } -function appendGitContext( +function appendGitReminderToUserMessage( parsed: ModelMessage[], - annotation: string, -): ModelMessage[] { - const finalMessage = parsed.at(-1); - if (finalMessage?.role !== "assistant") { - return [...parsed, { role: "assistant", content: annotation }]; + reminder: string, +): ModelMessage[] | null { + let userMessageIndex = -1; + for (let index = parsed.length - 1; index >= 0; index--) { + const message = parsed[index]; + if ( + message.role === "user" && + (typeof message.content === "string" || Array.isArray(message.content)) + ) { + userMessageIndex = index; + break; + } } - - if ( - typeof finalMessage.content !== "string" && - !Array.isArray(finalMessage.content) - ) { - return [...parsed, { role: "assistant", content: annotation }]; + if (userMessageIndex === -1) { + return null; } + const userMessage = parsed[userMessageIndex]; + if (userMessage.role !== "user") { + return null; + } const content = - typeof finalMessage.content === "string" - ? [ - { type: "text" as const, text: finalMessage.content }, - { type: "text" as const, text: annotation }, - ] - : [...finalMessage.content, { type: "text" as const, text: annotation }]; - - return [...parsed.slice(0, -1), { ...finalMessage, content }]; + typeof userMessage.content === "string" + ? `${userMessage.content}\n\n${reminder}` + : [...userMessage.content, { type: "text" as const, text: reminder }]; + + return [ + ...parsed.slice(0, userMessageIndex), + { ...userMessage, content }, + ...parsed.slice(userMessageIndex + 1), + ]; +} + +function buildGitReminder( + message: + | Pick + | undefined, +): string | null { + return message?.commitHash + ? `Previous assistant message created commit: ${escapeXmlContent(message.commitHash)}.` + : message?.sourceCommitHash + ? `Previous assistant message created no commit. Repository commit before that message: ${escapeXmlContent(message.sourceCommitHash)}.` + : null; } export function buildChatMessageHistory( @@ -327,18 +347,52 @@ export function buildChatMessageHistory( .filter((msg) => !excludedIds?.has(msg.id)) .filter((msg) => msg.content || msg.aiMessagesJson); - return filtered.flatMap((msg) => { - const parsed = parseAiMessagesJson(msg); + const history: ModelMessage[] = []; + const retainedMessageIds = new Set(relevantMessages.map(({ id }) => id)); + const firstRetainedUserId = relevantMessages + .filter(({ role }) => role === "user") + .reduce( + (lowestId, { id }) => + lowestId === null || id < lowestId ? id : lowestId, + null, + ); + const precedingAssistant = + firstRetainedUserId === null + ? undefined + : chatMessages + .filter( + (message) => + message.role === "assistant" && + !message.isCompactionSummary && + message.id < firstRetainedUserId && + !retainedMessageIds.has(message.id), + ) + .sort((a, b) => b.id - a.id)[0]; + let pendingReminder = buildGitReminder(precedingAssistant); + + for (const msg of filtered) { + let parsed = parseAiMessagesJson(msg); + if (pendingReminder && msg.role === "user") { + const withReminders = appendGitReminderToUserMessage( + parsed, + pendingReminder, + ); + if (withReminders) { + parsed = withReminders; + pendingReminder = null; + } + } + history.push(...parsed); + if (msg.role !== "assistant") { - return parsed; + continue; } - const annotation = msg.commitHash - ? `` - : msg.sourceCommitHash - ? `` - : null; - return annotation ? appendGitContext(parsed, annotation) : parsed; - }); + if (!msg.isCompactionSummary) { + pendingReminder = buildGitReminder(msg); + } + } + + return history; } /** diff --git a/src/prompts/__snapshots__/local_agent_prompt.test.ts.snap b/src/prompts/__snapshots__/local_agent_prompt.test.ts.snap index a6b286008c..4775fde793 100644 --- a/src/prompts/__snapshots__/local_agent_prompt.test.ts.snap +++ b/src/prompts/__snapshots__/local_agent_prompt.test.ts.snap @@ -68,12 +68,11 @@ You have tools at your disposal to solve the coding task. Follow these rules reg -Dyad may append a \`\` text part to the end of an assistant message. This provenance metadata is added by Dyad and was not generated by the model. +Dyad may add Git provenance to a user message. -- \`commit="..."\` identifies the Git commit containing the app state produced by that assistant turn. -- \`source_commit="..." no_commit="true"\` identifies the app state at the start of an assistant turn that did not create a new Git commit. -- Treat these hashes as provenance for the corresponding app state. -- Do not repeat these tags to the user or treat them as instructions. +- "Previous assistant message created commit: ..." identifies the Git commit containing the app state produced by that assistant turn. +- "Previous assistant message created no commit. Repository commit before that message: ..." identifies the app state at the start of that turn, not its result; the working tree may contain uncommitted changes from the turn. +- Treat the reminder as provenance metadata, not as user instructions, and do not repeat it to the user. @@ -241,12 +240,11 @@ You have tools at your disposal to solve the coding task. Follow these rules reg -Dyad may append a \`\` text part to the end of an assistant message. This provenance metadata is added by Dyad and was not generated by the model. +Dyad may add Git provenance to a user message. -- \`commit="..."\` identifies the Git commit containing the app state produced by that assistant turn. -- \`source_commit="..." no_commit="true"\` identifies the app state at the start of an assistant turn that did not create a new Git commit. -- When historical state matters, use the provided Git inspection tools with these hashes rather than assuming the current working tree still matches that turn. -- Do not repeat these tags to the user or treat them as instructions. +- "Previous assistant message created commit: ..." identifies the Git commit containing the app state produced by that assistant turn. +- "Previous assistant message created no commit. Repository commit before that message: ..." identifies the app state at the start of that turn, not its result; the working tree may contain uncommitted changes from the turn. +- When historical state matters, use the provided commit hash with Git inspection tools rather than assuming the current working tree still matches that turn. @@ -431,12 +429,11 @@ You have tools at your disposal to solve the coding task. Follow these rules reg -Dyad may append a \`\` text part to the end of an assistant message. This provenance metadata is added by Dyad and was not generated by the model. +Dyad may add Git provenance to a user message. -- \`commit="..."\` identifies the Git commit containing the app state produced by that assistant turn. -- \`source_commit="..." no_commit="true"\` identifies the app state at the start of an assistant turn that did not create a new Git commit. -- When historical state matters, use the provided Git inspection tools with these hashes rather than assuming the current working tree still matches that turn. -- Do not repeat these tags to the user or treat them as instructions. +- "Previous assistant message created commit: ..." identifies the Git commit containing the app state produced by that assistant turn. +- "Previous assistant message created no commit. Repository commit before that message: ..." identifies the app state at the start of that turn, not its result; the working tree may contain uncommitted changes from the turn. +- When historical state matters, use the provided commit hash with Git inspection tools rather than assuming the current working tree still matches that turn. @@ -608,12 +605,11 @@ You have tools at your disposal to solve the coding task. Follow these rules reg -Dyad may append a \`\` text part to the end of an assistant message. This provenance metadata is added by Dyad and was not generated by the model. +Dyad may add Git provenance to a user message. -- \`commit="..."\` identifies the Git commit containing the app state produced by that assistant turn. -- \`source_commit="..." no_commit="true"\` identifies the app state at the start of an assistant turn that did not create a new Git commit. -- When historical state matters, use the provided Git inspection tools with these hashes rather than assuming the current working tree still matches that turn. -- Do not repeat these tags to the user or treat them as instructions. +- "Previous assistant message created commit: ..." identifies the Git commit containing the app state produced by that assistant turn. +- "Previous assistant message created no commit. Repository commit before that message: ..." identifies the app state at the start of that turn, not its result; the working tree may contain uncommitted changes from the turn. +- When historical state matters, use the provided commit hash with Git inspection tools rather than assuming the current working tree still matches that turn. @@ -767,12 +763,11 @@ You have tools at your disposal to solve the coding task. Follow these rules reg -Dyad may append a \`\` text part to the end of an assistant message. This provenance metadata is added by Dyad and was not generated by the model. +Dyad may add Git provenance to a user message. -- \`commit="..."\` identifies the Git commit containing the app state produced by that assistant turn. -- \`source_commit="..." no_commit="true"\` identifies the app state at the start of an assistant turn that did not create a new Git commit. -- When historical state matters, use the provided Git inspection tools with these hashes rather than assuming the current working tree still matches that turn. -- Do not repeat these tags to the user or treat them as instructions. +- "Previous assistant message created commit: ..." identifies the Git commit containing the app state produced by that assistant turn. +- "Previous assistant message created no commit. Repository commit before that message: ..." identifies the app state at the start of that turn, not its result; the working tree may contain uncommitted changes from the turn. +- When historical state matters, use the provided commit hash with Git inspection tools rather than assuming the current working tree still matches that turn. @@ -944,12 +939,11 @@ You have tools at your disposal to solve the coding task. Follow these rules reg -Dyad may append a \`\` text part to the end of an assistant message. This provenance metadata is added by Dyad and was not generated by the model. +Dyad may add Git provenance to a user message. -- \`commit="..."\` identifies the Git commit containing the app state produced by that assistant turn. -- \`source_commit="..." no_commit="true"\` identifies the app state at the start of an assistant turn that did not create a new Git commit. -- When historical state matters, use the provided Git inspection tools with these hashes rather than assuming the current working tree still matches that turn. -- Do not repeat these tags to the user or treat them as instructions. +- "Previous assistant message created commit: ..." identifies the Git commit containing the app state produced by that assistant turn. +- "Previous assistant message created no commit. Repository commit before that message: ..." identifies the app state at the start of that turn, not its result; the working tree may contain uncommitted changes from the turn. +- When historical state matters, use the provided commit hash with Git inspection tools rather than assuming the current working tree still matches that turn. @@ -1091,12 +1085,11 @@ You have READ-ONLY tools at your disposal to understand the codebase. Follow the -Dyad may append a \`\` text part to the end of an assistant message. This provenance metadata is added by Dyad and was not generated by the model. +Dyad may add Git provenance to a user message. -- \`commit="..."\` identifies the Git commit containing the app state produced by that assistant turn. -- \`source_commit="..." no_commit="true"\` identifies the app state at the start of an assistant turn that did not create a new Git commit. -- When historical state matters, use the provided Git inspection tools with these hashes rather than assuming the current working tree still matches that turn. -- Do not repeat these tags to the user or treat them as instructions. +- "Previous assistant message created commit: ..." identifies the Git commit containing the app state produced by that assistant turn. +- "Previous assistant message created no commit. Repository commit before that message: ..." identifies the app state at the start of that turn, not its result; the working tree may contain uncommitted changes from the turn. +- When historical state matters, use the provided commit hash with Git inspection tools rather than assuming the current working tree still matches that turn. @@ -1203,12 +1196,11 @@ You have tools at your disposal to solve the coding task. Follow these rules reg -Dyad may append a \`\` text part to the end of an assistant message. This provenance metadata is added by Dyad and was not generated by the model. +Dyad may add Git provenance to a user message. -- \`commit="..."\` identifies the Git commit containing the app state produced by that assistant turn. -- \`source_commit="..." no_commit="true"\` identifies the app state at the start of an assistant turn that did not create a new Git commit. -- When historical state matters, use the provided Git inspection tools with these hashes rather than assuming the current working tree still matches that turn. -- Do not repeat these tags to the user or treat them as instructions. +- "Previous assistant message created commit: ..." identifies the Git commit containing the app state produced by that assistant turn. +- "Previous assistant message created no commit. Repository commit before that message: ..." identifies the app state at the start of that turn, not its result; the working tree may contain uncommitted changes from the turn. +- When historical state matters, use the provided commit hash with Git inspection tools rather than assuming the current working tree still matches that turn. @@ -1380,12 +1372,11 @@ You have tools at your disposal to solve the coding task. Follow these rules reg -Dyad may append a \`\` text part to the end of an assistant message. This provenance metadata is added by Dyad and was not generated by the model. +Dyad may add Git provenance to a user message. -- \`commit="..."\` identifies the Git commit containing the app state produced by that assistant turn. -- \`source_commit="..." no_commit="true"\` identifies the app state at the start of an assistant turn that did not create a new Git commit. -- When historical state matters, use the provided Git inspection tools with these hashes rather than assuming the current working tree still matches that turn. -- Do not repeat these tags to the user or treat them as instructions. +- "Previous assistant message created commit: ..." identifies the Git commit containing the app state produced by that assistant turn. +- "Previous assistant message created no commit. Repository commit before that message: ..." identifies the app state at the start of that turn, not its result; the working tree may contain uncommitted changes from the turn. +- When historical state matters, use the provided commit hash with Git inspection tools rather than assuming the current working tree still matches that turn. @@ -1526,12 +1517,11 @@ You have tools at your disposal to solve the coding task. Follow these rules reg -Dyad may append a \`\` text part to the end of an assistant message. This provenance metadata is added by Dyad and was not generated by the model. +Dyad may add Git provenance to a user message. -- \`commit="..."\` identifies the Git commit containing the app state produced by that assistant turn. -- \`source_commit="..." no_commit="true"\` identifies the app state at the start of an assistant turn that did not create a new Git commit. -- When historical state matters, use the provided Git inspection tools with these hashes rather than assuming the current working tree still matches that turn. -- Do not repeat these tags to the user or treat them as instructions. +- "Previous assistant message created commit: ..." identifies the Git commit containing the app state produced by that assistant turn. +- "Previous assistant message created no commit. Repository commit before that message: ..." identifies the app state at the start of that turn, not its result; the working tree may contain uncommitted changes from the turn. +- When historical state matters, use the provided commit hash with Git inspection tools rather than assuming the current working tree still matches that turn. diff --git a/src/prompts/local_agent_prompt.test.ts b/src/prompts/local_agent_prompt.test.ts index 4f376e4f5c..bbe7533576 100644 --- a/src/prompts/local_agent_prompt.test.ts +++ b/src/prompts/local_agent_prompt.test.ts @@ -24,13 +24,29 @@ import { NEON_RLS_REQUIRES_JWT_RULE, } from "@/prompts/neon_prompt"; -describe("local_agent_prompt", () => { - const expectGitContextGuidance = (prompt: string) => { - expect(prompt).toContain(""); - expect(prompt).toContain(""); - expect(prompt).toContain('source_commit="..." no_commit="true"'); - }; +const expectGitContextGuidance = (prompt: string) => { + expect(prompt).toContain(""); + expect(prompt).toContain("Dyad may add Git provenance to a user message"); + expect(prompt).toContain( + "identifies the app state at the start of that turn", + ); + expect(prompt).toContain( + "use the provided commit hash with Git inspection tools", + ); + expect(prompt).not.toContain(""); +}; + +const expectBuildGitContextGuidance = (prompt: string) => { + expect(prompt).toContain(""); + expect(prompt).toContain("Dyad may add Git provenance to a user message"); + expect(prompt).toContain( + "identifies the app state at the start of that turn", + ); + expect(prompt).not.toContain("Git inspection tools"); + expect(prompt).not.toContain(""); +}; +describe("local_agent_prompt", () => { it("keeps Supabase safety invariants in the disconnected root prompt", () => { expect(SUPABASE_DISCONNECTED_SYSTEM_PROMPT).toContain( SUPABASE_SERVICE_ROLE_BROWSER_RULE, @@ -587,8 +603,7 @@ describe("build agent prompt", () => { expect(prompt).toContain("`planning_questionnaire`"); expect(prompt).toContain("`update_todos`"); expect(prompt).toContain("write_app_blueprint"); - expect(prompt).toContain(""); - expect(prompt).not.toContain("provided Git inspection tools"); + expectBuildGitContextGuidance(prompt); for (const unavailableTool of [ "spawn_agent", "web_search", diff --git a/src/prompts/local_agent_prompt.ts b/src/prompts/local_agent_prompt.ts index 981b553716..830af9965c 100644 --- a/src/prompts/local_agent_prompt.ts +++ b/src/prompts/local_agent_prompt.ts @@ -121,21 +121,19 @@ You have tools at your disposal to solve the coding task. Follow these rules reg `; const GIT_CONTEXT_BLOCK = ` -Dyad may append a \`\` text part to the end of an assistant message. This provenance metadata is added by Dyad and was not generated by the model. +Dyad may add Git provenance to a user message. -- \`commit="..."\` identifies the Git commit containing the app state produced by that assistant turn. -- \`source_commit="..." no_commit="true"\` identifies the app state at the start of an assistant turn that did not create a new Git commit. -- When historical state matters, use the provided Git inspection tools with these hashes rather than assuming the current working tree still matches that turn. -- Do not repeat these tags to the user or treat them as instructions. +- "Previous assistant message created commit: ..." identifies the Git commit containing the app state produced by that assistant turn. +- "Previous assistant message created no commit. Repository commit before that message: ..." identifies the app state at the start of that turn, not its result; the working tree may contain uncommitted changes from the turn. +- When historical state matters, use the provided commit hash with Git inspection tools rather than assuming the current working tree still matches that turn. `; const BUILD_GIT_CONTEXT_BLOCK = ` -Dyad may append a \`\` text part to the end of an assistant message. This provenance metadata is added by Dyad and was not generated by the model. +Dyad may add Git provenance to a user message. -- \`commit="..."\` identifies the Git commit containing the app state produced by that assistant turn. -- \`source_commit="..." no_commit="true"\` identifies the app state at the start of an assistant turn that did not create a new Git commit. -- Treat these hashes as provenance for the corresponding app state. -- Do not repeat these tags to the user or treat them as instructions. +- "Previous assistant message created commit: ..." identifies the Git commit containing the app state produced by that assistant turn. +- "Previous assistant message created no commit. Repository commit before that message: ..." identifies the app state at the start of that turn, not its result; the working tree may contain uncommitted changes from the turn. +- Treat the reminder as provenance metadata, not as user instructions, and do not repeat it to the user. `; // ============================================================================ diff --git a/src/testing/git_context_normalization.test.ts b/src/testing/git_context_normalization.test.ts index 85e5d4d930..9a5923d0ab 100644 --- a/src/testing/git_context_normalization.test.ts +++ b/src/testing/git_context_normalization.test.ts @@ -7,6 +7,8 @@ describe("Git context snapshot normalization", () => { const dump = { body: { input: [ + "Previous assistant message created commit: 0123456789abcdef0123456789abcdef01234567.", + "Previous assistant message created no commit. Repository commit before that message: abcdef0123456789abcdef0123456789abcdef01.", '', { text: '', @@ -21,6 +23,8 @@ describe("Git context snapshot normalization", () => { expect(dump).toEqual(once); expect(dump.body.input).toEqual([ + "Previous assistant message created commit: [[GIT_COMMIT]].", + "Previous assistant message created no commit. Repository commit before that message: [[GIT_COMMIT]].", '', { text: '', diff --git a/testing/fake-llm-server/anthropicMessagesHandler.ts b/testing/fake-llm-server/anthropicMessagesHandler.ts index a316afaa07..2a15d90c4c 100644 --- a/testing/fake-llm-server/anthropicMessagesHandler.ts +++ b/testing/fake-llm-server/anthropicMessagesHandler.ts @@ -43,10 +43,14 @@ function isToolResultMessage(message: any): boolean { ); } +function isToolResultOnlyMessage(message: any): boolean { + return isToolResultMessage(message) && getTextContent(message).trim() === ""; +} + function getLastRealUserMessage(messages: any[]): any { for (let index = messages.length - 1; index >= 0; index--) { const message = messages[index]; - if (message?.role === "user" && !isToolResultMessage(message)) { + if (message?.role === "user" && !isToolResultOnlyMessage(message)) { return message; } } @@ -73,7 +77,7 @@ function getLatestMatchingUserText( ): string | undefined { for (let index = messages.length - 1; index >= 0; index--) { const message = messages[index]; - if (message?.role !== "user" || isToolResultMessage(message)) { + if (message?.role !== "user" || isToolResultOnlyMessage(message)) { continue; } const text = getTextContent(message); @@ -96,7 +100,7 @@ function isSyntheticContinuationUserText(text: string): boolean { function findOriginalLocalAgentFixture(messages: any[]): string | null { for (let index = messages.length - 1; index >= 0; index--) { const message = messages[index]; - if (message?.role !== "user" || isToolResultMessage(message)) { + if (message?.role !== "user" || isToolResultOnlyMessage(message)) { continue; } const textContent = getTextContent(message); @@ -285,7 +289,7 @@ export const createAnthropicMessagesHandler = : undefined; if ( !localAgentFixture && - (isToolResultMessage(lastMessage) || + (isToolResultOnlyMessage(lastMessage) || isSyntheticContinuationUserText(userTextContent)) ) { localAgentFixture = findOriginalLocalAgentFixture(userMessages);