From 6dd65fcd152e9fc63487b30d286c4660a372047c Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 14:43:37 +0000 Subject: [PATCH 01/17] Initial commit with task details Adding .gitkeep for PR creation (default mode). This file will be removed when the task is complete. Issue: https://github.com/link-assistant/hive-mind/issues/2119 --- .gitkeep | 1 + 1 file changed, 1 insertion(+) create mode 100644 .gitkeep diff --git a/.gitkeep b/.gitkeep new file mode 100644 index 000000000..4bd6fca96 --- /dev/null +++ b/.gitkeep @@ -0,0 +1 @@ +# .gitkeep file auto-generated at 2026-07-30T14:43:37.751Z for PR creation at branch issue-2119-ae2d4c9d7f6d for issue https://github.com/link-assistant/hive-mind/issues/2119 \ No newline at end of file From 25e7a6ad837a9c318d1be75005520d2864e55817 Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 14:51:51 +0000 Subject: [PATCH 02/17] docs(2119): preserve raw evidence for formal-ai case study Downloads the PR/issue metadata, conversation comments, review comments and full solution-draft logs of the three konard/test-hello-world-* reproductions cited in issue #2119 so the case study analysis is based on immutable local evidence. --- .../019fb330-00e1-73b9-955e-f357a1600d5b/issue-1-comments.json | 1 + .../data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/issue-1.json | 1 + .../pr-2-conversation-comments.json | 1 + .../pr-2-review-comments.json | 1 + .../data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2.json | 1 + .../019fb330-fa49-7c9d-a664-b7ea33bb698a/issue-1-comments.json | 1 + .../data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/issue-1.json | 1 + .../pr-2-conversation-comments.json | 1 + .../pr-2-review-comments.json | 1 + .../data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2.json | 1 + .../019fb331-c107-78c7-8ff6-9f127a3c593c/issue-1-comments.json | 1 + .../data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/issue-1.json | 1 + .../pr-2-conversation-comments.json | 1 + .../pr-2-review-comments.json | 1 + .../data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2.json | 1 + 15 files changed, 15 insertions(+) create mode 100644 docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/issue-1-comments.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/issue-1.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2-conversation-comments.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2-review-comments.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/issue-1-comments.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/issue-1.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2-conversation-comments.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2-review-comments.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/issue-1-comments.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/issue-1.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2-conversation-comments.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2-review-comments.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2.json diff --git a/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/issue-1-comments.json b/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/issue-1-comments.json new file mode 100644 index 000000000..0637a088a --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/issue-1-comments.json @@ -0,0 +1 @@ +[] \ No newline at end of file diff --git a/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/issue-1.json b/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/issue-1.json new file mode 100644 index 000000000..6cd8aed64 --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/issue-1.json @@ -0,0 +1 @@ +{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/1","repository_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b","labels_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/1/labels{/name}","comments_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/1/comments","events_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/1/events","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/1","id":5020167374,"node_id":"I_kwDOToRLvM8AAAABKzmszg","number":1,"title":"Implement Hello World in Scala","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"labels":[],"state":"open","locked":false,"assignees":[],"milestone":null,"comments":0,"created_at":"2026-07-30T13:21:47Z","updated_at":"2026-07-30T13:21:47Z","closed_at":null,"assignee":null,"author_association":"OWNER","active_lock_reason":null,"sub_issues_summary":{"total":0,"completed":0,"percent_completed":0},"issue_dependencies_summary":{"blocked_by":0,"total_blocked_by":0,"blocking":0,"total_blocking":0},"body":"## Task\nPlease implement a \"Hello World\" program in Scala.\n\n## Requirements\n1. Create a file with the appropriate extension for Scala\n2. The program should print exactly: `Hello, World!`\n3. Add clear comments explaining the code\n4. Ensure the code follows Scala best practices and idioms\n5. If applicable, include build/run instructions in a comment at the top of the file\n6. **Create a GitHub Actions workflow that automatically runs and tests the program on every push and pull request**\n\n## Expected Output\nWhen the program runs, it should output:\n```\nHello, World!\n```\n\n## GitHub Actions Requirements\nThe CI/CD workflow should:\n- Trigger on push to main branch and on pull requests\n- Set up the appropriate Scala runtime/compiler\n- Run the Hello World program\n- Verify the output is exactly \"Hello, World!\"\n- Show a green check mark when tests pass\n\nExample workflow structure:\n- Checkout code\n- Setup Scala environment\n- Run the program\n- Assert output matches expected string\n\n## Additional Notes\n- The implementation should be simple and straightforward\n- Focus on clarity and correctness\n- Use the standard library only (no external dependencies unless absolutely necessary for Scala)\n- The GitHub Actions workflow should be in `.github/workflows/` directory\n- The workflow should have a meaningful name like `test-hello-world.yml`\n\n## Definition of Done\n- [ ] Program file created with correct extension\n- [ ] Code prints \"Hello, World!\" exactly\n- [ ] Code is properly commented\n- [ ] Code follows Scala conventions\n- [ ] Instructions for running the program are included (if needed)\n- [ ] GitHub Actions workflow created and passing\n- [ ] CI badge showing build status (optional but recommended)","closed_by":null,"reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/1/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"timeline_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/1/timeline","performed_via_github_app":null,"state_reason":null,"pinned_comment":null} \ No newline at end of file diff --git a/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2-conversation-comments.json b/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2-conversation-comments.json new file mode 100644 index 000000000..38ebc6e7b --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2-conversation-comments.json @@ -0,0 +1 @@ +[{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131949934","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131949934","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5131949934,"node_id":"IC_kwDOToRLvM8AAAABMeNXbg","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:14:38Z","updated_at":"2026-07-30T14:14:38Z","body":"## ๐Ÿค– Solution Draft Log\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- Thinking level: off (disabled)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (961KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/465f5511052c5796806c6fc1d29c6b4c/raw/c2768221a87d91fc27cbcad2dc9bc9e45b2a6ecd/tmp-hive-mind-log-upload-TQuSSp-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131949934/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131952270","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131952270","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5131952270,"node_id":"IC_kwDOToRLvM8AAAABMeNgjg","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:14:49Z","updated_at":"2026-07-30T14:14:49Z","body":"## ๐Ÿ”„ Auto-restart 1/5\n\nDetected uncommitted changes from previous run. Starting new session to review and commit or discard them.\n\n**Uncommitted files:**\n```\n?? .formal-ai/\n?? examples\n```\n\n---\n*Auto-restart will stop after changes are committed or discarded, or after 4 more iterations. Please wait until working session will end and give your feedback.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131952270/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131959420","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131959420","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5131959420,"node_id":"IC_kwDOToRLvM8AAAABMeN8fA","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:15:24Z","updated_at":"2026-07-30T14:15:24Z","body":"## ๐Ÿ”„ Auto-restart 1/5 Log\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (1954KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/ef3290d4bc10e5d1116c6e1d79cfe9fc/raw/f4798e8bf7b0121f0e89ba82ceb233c85d55e950/tmp-hive-mind-log-upload-YU3TQL-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131959420/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131961463","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131961463","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5131961463,"node_id":"IC_kwDOToRLvM8AAAABMeOEdw","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:15:35Z","updated_at":"2026-07-30T14:15:35Z","body":"## ๐Ÿ”„ Auto-restart 2/5\n\nDetected uncommitted changes from previous run. Starting new session to review and commit or discard them.\n\n**Uncommitted files:**\n```\n?? .formal-ai/\n?? examples\n```\n\n---\n*Auto-restart will stop after changes are committed or discarded, or after 3 more iterations. Please wait until working session will end and give your feedback.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131961463/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131967541","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131967541","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5131967541,"node_id":"IC_kwDOToRLvM8AAAABMeOcNQ","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:16:07Z","updated_at":"2026-07-30T14:16:07Z","body":"## ๐Ÿ”„ Auto-restart 2/5 Log\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (2945KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/4f45c03c2bc13fabed47d8fea4a5194f/raw/08207cecd10edcfc34ff5b940a522831d92fcba1/tmp-hive-mind-log-upload-9twZ9s-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131967541/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131969808","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131969808","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5131969808,"node_id":"IC_kwDOToRLvM8AAAABMeOlEA","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:16:19Z","updated_at":"2026-07-30T14:16:19Z","body":"## ๐Ÿ”„ Auto-restart 3/5\n\nDetected uncommitted changes from previous run. Starting new session to review and commit or discard them.\n\n**Uncommitted files:**\n```\n?? .formal-ai/\n?? examples\n```\n\n---\n*Auto-restart will stop after changes are committed or discarded, or after 2 more iterations. Please wait until working session will end and give your feedback.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131969808/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131976173","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131976173","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5131976173,"node_id":"IC_kwDOToRLvM8AAAABMeO97Q","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:16:50Z","updated_at":"2026-07-30T14:16:50Z","body":"## ๐Ÿ”„ Auto-restart 3/5 Log\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (3936KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/ebda1420a27cdaac0327cf3bfa515261/raw/26108136c6e9fc928e66a55fe06e6367f0cf4fb0/tmp-hive-mind-log-upload-6bihBU-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131976173/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131978601","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131978601","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5131978601,"node_id":"IC_kwDOToRLvM8AAAABMePHaQ","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:17:03Z","updated_at":"2026-07-30T14:17:03Z","body":"## ๐Ÿ”„ Auto-restart 4/5\n\nDetected uncommitted changes from previous run. Starting new session to review and commit or discard them.\n\n**Uncommitted files:**\n```\n?? .formal-ai/\n?? examples\n```\n\n---\n*Auto-restart will stop after changes are committed or discarded, or after 1 more iteration. Please wait until working session will end and give your feedback.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131978601/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131986742","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131986742","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5131986742,"node_id":"IC_kwDOToRLvM8AAAABMePnNg","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:17:47Z","updated_at":"2026-07-30T14:17:47Z","body":"## ๐Ÿ”„ Auto-restart 4/5 Log\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (4927KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/f6a134fb73adac137e7c200ab5d3db92/raw/cee22282afc4bcbdeaba88475f26a8806cd5ece2/tmp-hive-mind-log-upload-56SSDs-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131986742/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131988832","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131988832","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5131988832,"node_id":"IC_kwDOToRLvM8AAAABMePvYA","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:17:57Z","updated_at":"2026-07-30T14:17:57Z","body":"## ๐Ÿ”„ Auto-restart 5/5\n\nDetected uncommitted changes from previous run. Starting new session to review and commit or discard them.\n\n**Uncommitted files:**\n```\n?? .formal-ai/\n?? examples\n```\n\n---\n*Auto-restart will stop after changes are committed or discarded, or after 0 more iterations. Please wait until working session will end and give your feedback.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131988832/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131996068","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131996068","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5131996068,"node_id":"IC_kwDOToRLvM8AAAABMeQLpA","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:18:35Z","updated_at":"2026-07-30T14:18:35Z","body":"## ๐Ÿ”„ Auto-restart 5/5 Log\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (5918KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/9cd2e3a7e98a634375d2eed7049fb62d/raw/570dd2d36c791629e9a68001ac94e35d36841357/tmp-hive-mind-log-upload-tWAAPV-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5131996068/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132022050","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132022050","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5132022050,"node_id":"IC_kwDOToRLvM8AAAABMeRxIg","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:20:51Z","updated_at":"2026-07-30T14:20:51Z","body":"## ๐Ÿ”„ Auto-restart triggered (iteration 1)\n\n**Reason:** Uncommitted changes detected\n\nStarting new session to address the issues.\n\n---\n*Auto-restart-until-mergeable mode is active. This run will stop after 5 restart iterations.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132022050/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132030920","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132030920","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5132030920,"node_id":"IC_kwDOToRLvM8AAAABMeSTyA","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:21:33Z","updated_at":"2026-07-30T14:21:33Z","body":"## ๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 1)\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (6913KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/385002afaadc89bd178b783864d88301/raw/1174c597b8947cbfe4ab938e802664507db3d994/tmp-hive-mind-log-upload-HRzjRr-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132030920/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132057225","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132057225","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5132057225,"node_id":"IC_kwDOToRLvM8AAAABMeT6iQ","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:23:45Z","updated_at":"2026-07-30T14:23:45Z","body":"## ๐Ÿ”„ Auto-restart triggered (iteration 2)\n\n**Reason:** Uncommitted changes detected\n\nStarting new session to address the issues.\n\n---\n*Auto-restart-until-mergeable mode is active. This run will stop after 5 restart iterations.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132057225/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132065964","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132065964","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5132065964,"node_id":"IC_kwDOToRLvM8AAAABMeUcrA","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:24:25Z","updated_at":"2026-07-30T14:24:25Z","body":"## ๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 2)\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (7877KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/037503c4af448467dd8fdc3eb3c48980/raw/3815a653380f2b28b482948233fccdc18f3e7403/tmp-hive-mind-log-upload-fEg1HP-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132065964/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132092491","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132092491","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5132092491,"node_id":"IC_kwDOToRLvM8AAAABMeWESw","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:26:37Z","updated_at":"2026-07-30T14:26:37Z","body":"## ๐Ÿ”„ Auto-restart triggered (iteration 3)\n\n**Reason:** Uncommitted changes detected\n\nStarting new session to address the issues.\n\n---\n*Auto-restart-until-mergeable mode is active. This run will stop after 5 restart iterations.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132092491/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132100370","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132100370","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5132100370,"node_id":"IC_kwDOToRLvM8AAAABMeWjEg","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:27:17Z","updated_at":"2026-07-30T14:27:17Z","body":"## ๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 3)\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (8846KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/1459e4acca0d459d305f28a6e9b215be/raw/00ea9281d3b3004012d03a99a151b4786203a546/tmp-hive-mind-log-upload-HB4zCM-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132100370/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132125072","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132125072","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5132125072,"node_id":"IC_kwDOToRLvM8AAAABMeYDkA","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:29:30Z","updated_at":"2026-07-30T14:29:30Z","body":"## ๐Ÿ”„ Auto-restart triggered (iteration 4)\n\n**Reason:** Uncommitted changes detected\n\nStarting new session to address the issues.\n\n---\n*Auto-restart-until-mergeable mode is active. This run will stop after 5 restart iterations.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132125072/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132132625","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132132625","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5132132625,"node_id":"IC_kwDOToRLvM8AAAABMeYhEQ","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:30:11Z","updated_at":"2026-07-30T14:30:11Z","body":"## ๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 4)\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (9819KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/48c447dc630ae2f8246bbe62ac835e71/raw/4f45d3bd037e82f2201771ee7d955b1b2e71af84/tmp-hive-mind-log-upload-ZzMtDW-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132132625/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132158581","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132158581","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5132158581,"node_id":"IC_kwDOToRLvM8AAAABMeaGdQ","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:32:23Z","updated_at":"2026-07-30T14:32:23Z","body":"## ๐Ÿ”„ Auto-restart triggered (iteration 5)\n\n**Reason:** Uncommitted changes detected\n\nStarting new session to address the issues.\n\n---\n*Auto-restart-until-mergeable mode is active. This run will stop after 5 restart iterations.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132158581/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132166304","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132166304","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5132166304,"node_id":"IC_kwDOToRLvM8AAAABMeakoA","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:33:02Z","updated_at":"2026-07-30T14:33:02Z","body":"## ๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 5)\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (10796KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/a5f3e9b4dea22cdffe0c699fcf32e1c5/raw/0de8fc98d9e00bf957d61ce5e3ed627a5c069c1c/tmp-hive-mind-log-upload-0JuEzF-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132166304/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132191953","html_url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132191953","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/2","id":5132191953,"node_id":"IC_kwDOToRLvM8AAAABMecI0Q","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:35:15Z","updated_at":"2026-07-30T14:35:15Z","body":"## โš ๏ธ Auto-restart limit reached\n\nHive Mind stopped auto-restart-until-mergeable after 5 restart iterations.\n\n**Configured limit:** 5\n**Remaining reason:** Uncommitted changes detected\n\nNo further AI sessions will be started automatically for this run. Please review the remaining blockers manually or rerun with a higher `--auto-restart-max-iterations` value.\n\n---\n*Auto-restart-until-mergeable stopped by the safety limit.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/comments/5132191953/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null}] \ No newline at end of file diff --git a/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2-review-comments.json b/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2-review-comments.json new file mode 100644 index 000000000..0637a088a --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2-review-comments.json @@ -0,0 +1 @@ +[] \ No newline at end of file diff --git a/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2.json b/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2.json new file mode 100644 index 000000000..f0a1dbc88 --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2.json @@ -0,0 +1 @@ +{"body":"## Summary\n\nThis pull request implements a solution for #1: Implement Hello World in Scala\n\n### Changes\n- 1 file(s) modified\n- 1 line(s) added\n- 0 line(s) removed\n\n### Issue Reference\nFixes #1\n\n---\n*This PR was created automatically by the AI issue solver*","comments":[{"id":"IC_kwDOToRLvM8AAAABMeNXbg","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿค– Solution Draft Log\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- Thinking level: off (disabled)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (961KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/465f5511052c5796806c6fc1d29c6b4c/raw/c2768221a87d91fc27cbcad2dc9bc9e45b2a6ecd/tmp-hive-mind-log-upload-TQuSSp-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","createdAt":"2026-07-30T14:14:38Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131949934","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeNgjg","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart 1/5\n\nDetected uncommitted changes from previous run. Starting new session to review and commit or discard them.\n\n**Uncommitted files:**\n```\n?? .formal-ai/\n?? examples\n```\n\n---\n*Auto-restart will stop after changes are committed or discarded, or after 4 more iterations. Please wait until working session will end and give your feedback.*","createdAt":"2026-07-30T14:14:49Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131952270","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeN8fA","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart 1/5 Log\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (1954KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/ef3290d4bc10e5d1116c6e1d79cfe9fc/raw/f4798e8bf7b0121f0e89ba82ceb233c85d55e950/tmp-hive-mind-log-upload-YU3TQL-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","createdAt":"2026-07-30T14:15:24Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131959420","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeOEdw","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart 2/5\n\nDetected uncommitted changes from previous run. Starting new session to review and commit or discard them.\n\n**Uncommitted files:**\n```\n?? .formal-ai/\n?? examples\n```\n\n---\n*Auto-restart will stop after changes are committed or discarded, or after 3 more iterations. Please wait until working session will end and give your feedback.*","createdAt":"2026-07-30T14:15:35Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131961463","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeOcNQ","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart 2/5 Log\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (2945KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/4f45c03c2bc13fabed47d8fea4a5194f/raw/08207cecd10edcfc34ff5b940a522831d92fcba1/tmp-hive-mind-log-upload-9twZ9s-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","createdAt":"2026-07-30T14:16:07Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131967541","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeOlEA","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart 3/5\n\nDetected uncommitted changes from previous run. Starting new session to review and commit or discard them.\n\n**Uncommitted files:**\n```\n?? .formal-ai/\n?? examples\n```\n\n---\n*Auto-restart will stop after changes are committed or discarded, or after 2 more iterations. Please wait until working session will end and give your feedback.*","createdAt":"2026-07-30T14:16:19Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131969808","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeO97Q","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart 3/5 Log\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (3936KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/ebda1420a27cdaac0327cf3bfa515261/raw/26108136c6e9fc928e66a55fe06e6367f0cf4fb0/tmp-hive-mind-log-upload-6bihBU-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","createdAt":"2026-07-30T14:16:50Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131976173","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMePHaQ","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart 4/5\n\nDetected uncommitted changes from previous run. Starting new session to review and commit or discard them.\n\n**Uncommitted files:**\n```\n?? .formal-ai/\n?? examples\n```\n\n---\n*Auto-restart will stop after changes are committed or discarded, or after 1 more iteration. Please wait until working session will end and give your feedback.*","createdAt":"2026-07-30T14:17:03Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131978601","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMePnNg","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart 4/5 Log\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (4927KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/f6a134fb73adac137e7c200ab5d3db92/raw/cee22282afc4bcbdeaba88475f26a8806cd5ece2/tmp-hive-mind-log-upload-56SSDs-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","createdAt":"2026-07-30T14:17:47Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131986742","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMePvYA","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart 5/5\n\nDetected uncommitted changes from previous run. Starting new session to review and commit or discard them.\n\n**Uncommitted files:**\n```\n?? .formal-ai/\n?? examples\n```\n\n---\n*Auto-restart will stop after changes are committed or discarded, or after 0 more iterations. Please wait until working session will end and give your feedback.*","createdAt":"2026-07-30T14:17:57Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131988832","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeQLpA","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart 5/5 Log\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (5918KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/9cd2e3a7e98a634375d2eed7049fb62d/raw/570dd2d36c791629e9a68001ac94e35d36841357/tmp-hive-mind-log-upload-tWAAPV-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","createdAt":"2026-07-30T14:18:35Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5131996068","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeRxIg","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart triggered (iteration 1)\n\n**Reason:** Uncommitted changes detected\n\nStarting new session to address the issues.\n\n---\n*Auto-restart-until-mergeable mode is active. This run will stop after 5 restart iterations.*","createdAt":"2026-07-30T14:20:51Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132022050","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeSTyA","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 1)\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (6913KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/385002afaadc89bd178b783864d88301/raw/1174c597b8947cbfe4ab938e802664507db3d994/tmp-hive-mind-log-upload-HRzjRr-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","createdAt":"2026-07-30T14:21:33Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132030920","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeT6iQ","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart triggered (iteration 2)\n\n**Reason:** Uncommitted changes detected\n\nStarting new session to address the issues.\n\n---\n*Auto-restart-until-mergeable mode is active. This run will stop after 5 restart iterations.*","createdAt":"2026-07-30T14:23:45Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132057225","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeUcrA","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 2)\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (7877KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/037503c4af448467dd8fdc3eb3c48980/raw/3815a653380f2b28b482948233fccdc18f3e7403/tmp-hive-mind-log-upload-fEg1HP-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","createdAt":"2026-07-30T14:24:25Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132065964","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeWESw","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart triggered (iteration 3)\n\n**Reason:** Uncommitted changes detected\n\nStarting new session to address the issues.\n\n---\n*Auto-restart-until-mergeable mode is active. This run will stop after 5 restart iterations.*","createdAt":"2026-07-30T14:26:37Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132092491","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeWjEg","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 3)\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (8846KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/1459e4acca0d459d305f28a6e9b215be/raw/00ea9281d3b3004012d03a99a151b4786203a546/tmp-hive-mind-log-upload-HB4zCM-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","createdAt":"2026-07-30T14:27:17Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132100370","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeYDkA","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart triggered (iteration 4)\n\n**Reason:** Uncommitted changes detected\n\nStarting new session to address the issues.\n\n---\n*Auto-restart-until-mergeable mode is active. This run will stop after 5 restart iterations.*","createdAt":"2026-07-30T14:29:30Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132125072","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeYhEQ","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 4)\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (9819KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/48c447dc630ae2f8246bbe62ac835e71/raw/4f45d3bd037e82f2201771ee7d955b1b2e71af84/tmp-hive-mind-log-upload-ZzMtDW-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","createdAt":"2026-07-30T14:30:11Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132132625","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeaGdQ","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart triggered (iteration 5)\n\n**Reason:** Uncommitted changes detected\n\nStarting new session to address the issues.\n\n---\n*Auto-restart-until-mergeable mode is active. This run will stop after 5 restart iterations.*","createdAt":"2026-07-30T14:32:23Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132158581","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMeakoA","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 5)\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Model: formal-ai\n- Provider: OpenCode Zen\n- Public pricing estimate: unknown\n- Calculated by OpenCode Zen: $0.00 (Free model)\n- Token usage: 0 input, 0 output\n\n### ๐Ÿค– **Models used:**\n- Tool: Agent CLI\n- Requested: `formal-ai` (`formalai/formal-ai`)\n- **Model: formalai/formal-ai** (`formalai/formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (10796KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/a5f3e9b4dea22cdffe0c699fcf32e1c5/raw/0de8fc98d9e00bf957d61ce5e3ed627a5c069c1c/tmp-hive-mind-log-upload-0JuEzF-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","createdAt":"2026-07-30T14:33:02Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132166304","viewerDidAuthor":true},{"id":"IC_kwDOToRLvM8AAAABMecI0Q","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## โš ๏ธ Auto-restart limit reached\n\nHive Mind stopped auto-restart-until-mergeable after 5 restart iterations.\n\n**Configured limit:** 5\n**Remaining reason:** Uncommitted changes detected\n\nNo further AI sessions will be started automatically for this run. Please review the remaining blockers manually or rerun with a higher `--auto-restart-max-iterations` value.\n\n---\n*Auto-restart-until-mergeable stopped by the safety limit.*","createdAt":"2026-07-30T14:35:15Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2#issuecomment-5132191953","viewerDidAuthor":true}],"commits":[{"authoredDate":"2026-07-30T14:13:41Z","authors":[{"email":"drakonard@gmail.com","id":"MDQ6VXNlcjE0MzE5MDQ=","login":"konard","name":"konard"}],"committedDate":"2026-07-30T14:13:41Z","messageBody":"Adding .gitkeep for PR creation (default mode).\nThis file will be removed when the task is complete.\n\nIssue: https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/issues/1","messageHeadline":"Initial commit with task details","oid":"fcfca1428b7cc98b0716cfe596361f77cacecc20"},{"authoredDate":"2026-07-30T14:35:17Z","authors":[{"email":"drakonard@gmail.com","id":"MDQ6VXNlcjE0MzE5MDQ=","login":"konard","name":"konard"}],"committedDate":"2026-07-30T14:35:17Z","messageBody":"This reverts commit fcfca1428b7cc98b0716cfe596361f77cacecc20.","messageHeadline":"Revert \"Initial commit with task details\"","oid":"bd620608c02a297df2e0523ba4673626c3841c26"}],"createdAt":"2026-07-30T14:13:51Z","files":[],"headRefName":"issue-1-1f3e3886bcb8","isDraft":false,"state":"OPEN","title":"'Implement Hello World in Scala'","url":"https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2"} diff --git a/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/issue-1-comments.json b/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/issue-1-comments.json new file mode 100644 index 000000000..0637a088a --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/issue-1-comments.json @@ -0,0 +1 @@ +[] \ No newline at end of file diff --git a/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/issue-1.json b/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/issue-1.json new file mode 100644 index 000000000..4ea1505d4 --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/issue-1.json @@ -0,0 +1 @@ +{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/1","repository_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a","labels_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/1/labels{/name}","comments_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/1/comments","events_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/1/events","html_url":"https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/1","id":5020176114,"node_id":"I_kwDOToRPqc8AAAABKznO8g","number":1,"title":"Implement Hello World in Kotlin","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"labels":[],"state":"open","locked":false,"assignees":[],"milestone":null,"comments":0,"created_at":"2026-07-30T13:22:50Z","updated_at":"2026-07-30T13:22:50Z","closed_at":null,"assignee":null,"author_association":"OWNER","active_lock_reason":null,"sub_issues_summary":{"total":0,"completed":0,"percent_completed":0},"issue_dependencies_summary":{"blocked_by":0,"total_blocked_by":0,"blocking":0,"total_blocking":0},"body":"## Task\nPlease implement a \"Hello World\" program in Kotlin.\n\n## Requirements\n1. Create a file with the appropriate extension for Kotlin\n2. The program should print exactly: `Hello, World!`\n3. Add clear comments explaining the code\n4. Ensure the code follows Kotlin best practices and idioms\n5. If applicable, include build/run instructions in a comment at the top of the file\n6. **Create a GitHub Actions workflow that automatically runs and tests the program on every push and pull request**\n\n## Expected Output\nWhen the program runs, it should output:\n```\nHello, World!\n```\n\n## GitHub Actions Requirements\nThe CI/CD workflow should:\n- Trigger on push to main branch and on pull requests\n- Set up the appropriate Kotlin runtime/compiler\n- Run the Hello World program\n- Verify the output is exactly \"Hello, World!\"\n- Show a green check mark when tests pass\n\nExample workflow structure:\n- Checkout code\n- Setup Kotlin environment\n- Run the program\n- Assert output matches expected string\n\n## Additional Notes\n- The implementation should be simple and straightforward\n- Focus on clarity and correctness\n- Use the standard library only (no external dependencies unless absolutely necessary for Kotlin)\n- The GitHub Actions workflow should be in `.github/workflows/` directory\n- The workflow should have a meaningful name like `test-hello-world.yml`\n\n## Definition of Done\n- [ ] Program file created with correct extension\n- [ ] Code prints \"Hello, World!\" exactly\n- [ ] Code is properly commented\n- [ ] Code follows Kotlin conventions\n- [ ] Instructions for running the program are included (if needed)\n- [ ] GitHub Actions workflow created and passing\n- [ ] CI badge showing build status (optional but recommended)","closed_by":null,"reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/1/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"timeline_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/1/timeline","performed_via_github_app":null,"state_reason":null,"pinned_comment":null} \ No newline at end of file diff --git a/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2-conversation-comments.json b/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2-conversation-comments.json new file mode 100644 index 000000000..c188ee2bf --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2-conversation-comments.json @@ -0,0 +1 @@ +[{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/comments/5132013034","html_url":"https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2#issuecomment-5132013034","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/2","id":5132013034,"node_id":"IC_kwDOToRPqc8AAAABMeRN6g","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:20:05Z","updated_at":"2026-07-30T14:20:05Z","body":"\n## Working session summary\n\nThe `pwd` command completed. Output:\n\n```text\n/tmp/gh-issue-solver-1785421161275\n```\n\n---\n*This summary was automatically extracted from the AI working session output.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/comments/5132013034/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/comments/5132015559","html_url":"https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2#issuecomment-5132015559","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/2","id":5132015559,"node_id":"IC_kwDOToRPqc8AAAABMeRXxw","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:20:18Z","updated_at":"2026-07-30T14:20:18Z","body":"## ๐Ÿค– Solution Draft Log\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Calculated by Anthropic: $0.252315\n\n### ๐Ÿค– **Models used:**\n- Tool: Anthropic Claude Code\n- Requested: `formal-ai`\n- Thinking level: off (disabled)\n- **Model: formal-ai** (`formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (137KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/16a9e51ac177a0ca8dd6f26b5148b1b4/raw/b2fb3eb4ede4871979e5443f643551b725141b90/tmp-hive-mind-log-upload-DAh7ws-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/comments/5132015559/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null},{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/comments/5132042054","html_url":"https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2#issuecomment-5132042054","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/2","id":5132042054,"node_id":"IC_kwDOToRPqc8AAAABMeS_Rg","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:22:29Z","updated_at":"2026-07-30T14:22:29Z","body":"## โœ… Ready to merge\n\nThis pull request is now ready to be merged:\n- No CI/CD checks are configured for this repository\n- No merge conflicts\n- No pending changes\n\n---\n*Monitored by hive-mind with --auto-restart-until-mergeable flag*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/comments/5132042054/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null}] \ No newline at end of file diff --git a/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2-review-comments.json b/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2-review-comments.json new file mode 100644 index 000000000..0637a088a --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2-review-comments.json @@ -0,0 +1 @@ +[] \ No newline at end of file diff --git a/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2.json b/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2.json new file mode 100644 index 000000000..fa1ca15db --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2.json @@ -0,0 +1 @@ +{"body":"## Summary\n\nThis pull request implements a solution for #1: Implement Hello World in Kotlin\n\n### Changes\n- 1 file(s) modified\n- 1 line(s) added\n- 0 line(s) removed\n\n### Issue Reference\nFixes #1\n\n---\n*This PR was created automatically by the AI issue solver*","comments":[{"id":"IC_kwDOToRPqc8AAAABMeRN6g","author":{"login":"konard"},"authorAssociation":"OWNER","body":"\n## Working session summary\n\nThe `pwd` command completed. Output:\n\n```text\n/tmp/gh-issue-solver-1785421161275\n```\n\n---\n*This summary was automatically extracted from the AI working session output.*","createdAt":"2026-07-30T14:20:05Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2#issuecomment-5132013034","viewerDidAuthor":true},{"id":"IC_kwDOToRPqc8AAAABMeRXxw","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿค– Solution Draft Log\nThis log file contains the complete execution trace of the AI solution draft process.\n\n### ๐Ÿ’ฐ **Cost estimation:**\n- Calculated by Anthropic: $0.252315\n\n### ๐Ÿค– **Models used:**\n- Tool: Anthropic Claude Code\n- Requested: `formal-ai`\n- Thinking level: off (disabled)\n- **Model: formal-ai** (`formal-ai`)\n\n### ๐Ÿ“Ž **Log file uploaded as Gist** (137KB)\n- [View complete solution draft log](https://gist.githubusercontent.com/konard/16a9e51ac177a0ca8dd6f26b5148b1b4/raw/b2fb3eb4ede4871979e5443f643551b725141b90/tmp-hive-mind-log-upload-DAh7ws-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","createdAt":"2026-07-30T14:20:18Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2#issuecomment-5132015559","viewerDidAuthor":true},{"id":"IC_kwDOToRPqc8AAAABMeS_Rg","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## โœ… Ready to merge\n\nThis pull request is now ready to be merged:\n- No CI/CD checks are configured for this repository\n- No merge conflicts\n- No pending changes\n\n---\n*Monitored by hive-mind with --auto-restart-until-mergeable flag*","createdAt":"2026-07-30T14:22:29Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2#issuecomment-5132042054","viewerDidAuthor":true}],"commits":[{"authoredDate":"2026-07-30T14:19:23Z","authors":[{"email":"drakonard@gmail.com","id":"MDQ6VXNlcjE0MzE5MDQ=","login":"konard","name":"konard"}],"committedDate":"2026-07-30T14:19:23Z","messageBody":"Adding .gitkeep for PR creation (default mode).\nThis file will be removed when the task is complete.\n\nIssue: https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/issues/1","messageHeadline":"Initial commit with task details","oid":"65a1be4da617f484cad6b0bc9629801c928973a4"},{"authoredDate":"2026-07-30T14:22:29Z","authors":[{"email":"drakonard@gmail.com","id":"MDQ6VXNlcjE0MzE5MDQ=","login":"konard","name":"konard"}],"committedDate":"2026-07-30T14:22:29Z","messageBody":"This reverts commit 65a1be4da617f484cad6b0bc9629801c928973a4.","messageHeadline":"Revert \"Initial commit with task details\"","oid":"e411d44fea1b8ba6f94118826621773ab24803f3"}],"createdAt":"2026-07-30T14:19:34Z","files":[],"headRefName":"issue-1-604f2202fd18","isDraft":false,"state":"OPEN","title":"'Implement Hello World in Kotlin'","url":"https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2"} diff --git a/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/issue-1-comments.json b/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/issue-1-comments.json new file mode 100644 index 000000000..0637a088a --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/issue-1-comments.json @@ -0,0 +1 @@ +[] \ No newline at end of file diff --git a/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/issue-1.json b/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/issue-1.json new file mode 100644 index 000000000..037735567 --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/issue-1.json @@ -0,0 +1 @@ +{"url":"https://api.github.com/repos/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/issues/1","repository_url":"https://api.github.com/repos/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c","labels_url":"https://api.github.com/repos/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/issues/1/labels{/name}","comments_url":"https://api.github.com/repos/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/issues/1/comments","events_url":"https://api.github.com/repos/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/issues/1/events","html_url":"https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/issues/1","id":5020183439,"node_id":"I_kwDOToRS7c8AAAABKznrjw","number":1,"title":"Implement Hello World in Rust","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"labels":[],"state":"open","locked":false,"assignees":[],"milestone":null,"comments":0,"created_at":"2026-07-30T13:23:41Z","updated_at":"2026-07-30T13:23:41Z","closed_at":null,"assignee":null,"author_association":"OWNER","active_lock_reason":null,"sub_issues_summary":{"total":0,"completed":0,"percent_completed":0},"issue_dependencies_summary":{"blocked_by":0,"total_blocked_by":0,"blocking":0,"total_blocking":0},"body":"## Task\nPlease implement a \"Hello World\" program in Rust.\n\n## Requirements\n1. Create a file with the appropriate extension for Rust\n2. The program should print exactly: `Hello, World!`\n3. Add clear comments explaining the code\n4. Ensure the code follows Rust best practices and idioms\n5. If applicable, include build/run instructions in a comment at the top of the file\n6. **Create a GitHub Actions workflow that automatically runs and tests the program on every push and pull request**\n\n## Expected Output\nWhen the program runs, it should output:\n```\nHello, World!\n```\n\n## GitHub Actions Requirements\nThe CI/CD workflow should:\n- Trigger on push to main branch and on pull requests\n- Set up the appropriate Rust runtime/compiler\n- Run the Hello World program\n- Verify the output is exactly \"Hello, World!\"\n- Show a green check mark when tests pass\n\nExample workflow structure:\n- Checkout code\n- Setup Rust environment\n- Run the program\n- Assert output matches expected string\n\n## Additional Notes\n- The implementation should be simple and straightforward\n- Focus on clarity and correctness\n- Use the standard library only (no external dependencies unless absolutely necessary for Rust)\n- The GitHub Actions workflow should be in `.github/workflows/` directory\n- The workflow should have a meaningful name like `test-hello-world.yml`\n\n## Definition of Done\n- [ ] Program file created with correct extension\n- [ ] Code prints \"Hello, World!\" exactly\n- [ ] Code is properly commented\n- [ ] Code follows Rust conventions\n- [ ] Instructions for running the program are included (if needed)\n- [ ] GitHub Actions workflow created and passing\n- [ ] CI badge showing build status (optional but recommended)","closed_by":null,"reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/issues/1/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"timeline_url":"https://api.github.com/repos/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/issues/1/timeline","performed_via_github_app":null,"state_reason":null,"pinned_comment":null} \ No newline at end of file diff --git a/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2-conversation-comments.json b/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2-conversation-comments.json new file mode 100644 index 000000000..a06de035b --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2-conversation-comments.json @@ -0,0 +1 @@ +[{"url":"https://api.github.com/repos/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/issues/comments/5132085159","html_url":"https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/pull/2#issuecomment-5132085159","issue_url":"https://api.github.com/repos/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/issues/2","id":5132085159,"node_id":"IC_kwDOToRS7c8AAAABMeVnpw","user":{"login":"konard","id":1431904,"node_id":"MDQ6VXNlcjE0MzE5MDQ=","avatar_url":"https://avatars.githubusercontent.com/u/1431904?v=4","gravatar_id":"","url":"https://api.github.com/users/konard","html_url":"https://github.com/konard","followers_url":"https://api.github.com/users/konard/followers","following_url":"https://api.github.com/users/konard/following{/other_user}","gists_url":"https://api.github.com/users/konard/gists{/gist_id}","starred_url":"https://api.github.com/users/konard/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/konard/subscriptions","organizations_url":"https://api.github.com/users/konard/orgs","repos_url":"https://api.github.com/users/konard/repos","events_url":"https://api.github.com/users/konard/events{/privacy}","received_events_url":"https://api.github.com/users/konard/received_events","type":"User","user_view_type":"public","site_admin":false},"created_at":"2026-07-30T14:25:59Z","updated_at":"2026-07-30T14:25:59Z","body":"## ๐Ÿšจ Solution Draft Failed\nThe automated solution draft encountered an error:\n```\nThe solver stopped while continuing pull request #2.\n\nReason: Authentication error\n```\n\n### What you can do\n- Resolve the repository, account, permissions, or environment problem described above, then rerun the solver.\n- Repository owner or Hive Mind administrator path: handle manual recreation or fix of the repository when the required action is outside the requester access.\n\n### ๐Ÿค– **Models used:**\n- Tool: OpenAI Codex\n- Requested: `formal-ai`\n- Thinking level: off (disabled)\n- **Model: formal-ai** (`formal-ai`)\n\n### ๐Ÿ“Ž **Failure log uploaded as Gist** (335KB)\n- [View complete failure log](https://gist.githubusercontent.com/konard/4e56198bee2a177e71ddc41bdf5b5294/raw/9ef0b5a8748c953eaeb2c98b9eafe6603359c855/tmp-hive-mind-log-upload-WQPa4n-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","author_association":"OWNER","reactions":{"url":"https://api.github.com/repos/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/issues/comments/5132085159/reactions","total_count":0,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"performed_via_github_app":null,"minimized":null}] \ No newline at end of file diff --git a/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2-review-comments.json b/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2-review-comments.json new file mode 100644 index 000000000..0637a088a --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2-review-comments.json @@ -0,0 +1 @@ +[] \ No newline at end of file diff --git a/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2.json b/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2.json new file mode 100644 index 000000000..53868a82d --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2.json @@ -0,0 +1 @@ +{"body":"## ๐Ÿค– AI-Powered Solution Draft\n\nThis pull request is being automatically generated to solve issue #1.\n\n### ๐Ÿ“‹ Issue Reference\nFixes #1\n\n### ๐Ÿšง Status\n**Work in Progress** - The AI assistant is currently analyzing and implementing the solution draft.\n\n### ๐Ÿ“ Implementation Details\n_Details will be added as the solution draft is developed..._\n\n---\n*This PR was created automatically by the AI issue solver*","comments":[{"id":"IC_kwDOToRS7c8AAAABMeVnpw","author":{"login":"konard"},"authorAssociation":"OWNER","body":"## ๐Ÿšจ Solution Draft Failed\nThe automated solution draft encountered an error:\n```\nThe solver stopped while continuing pull request #2.\n\nReason: Authentication error\n```\n\n### What you can do\n- Resolve the repository, account, permissions, or environment problem described above, then rerun the solver.\n- Repository owner or Hive Mind administrator path: handle manual recreation or fix of the repository when the required action is outside the requester access.\n\n### ๐Ÿค– **Models used:**\n- Tool: OpenAI Codex\n- Requested: `formal-ai`\n- Thinking level: off (disabled)\n- **Model: formal-ai** (`formal-ai`)\n\n### ๐Ÿ“Ž **Failure log uploaded as Gist** (335KB)\n- [View complete failure log](https://gist.githubusercontent.com/konard/4e56198bee2a177e71ddc41bdf5b5294/raw/9ef0b5a8748c953eaeb2c98b9eafe6603359c855/tmp-hive-mind-log-upload-WQPa4n-sanitized.log.txt)\n\n---\n*Now working session is ended, feel free to review and add any feedback on the solution draft.*","createdAt":"2026-07-30T14:25:59Z","includesCreatedEdit":false,"isMinimized":false,"minimizedReason":"","reactionGroups":[],"url":"https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/pull/2#issuecomment-5132085159","viewerDidAuthor":true}],"commits":[{"authoredDate":"2026-07-30T14:24:59Z","authors":[{"email":"drakonard@gmail.com","id":"MDQ6VXNlcjE0MzE5MDQ=","login":"konard","name":"konard"}],"committedDate":"2026-07-30T14:24:59Z","messageBody":"Adding .gitkeep for PR creation (default mode).\nThis file will be removed when the task is complete.\n\nIssue: https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/issues/1","messageHeadline":"Initial commit with task details","oid":"9d74fb354ad3ee68ea64167543be59b6604d0357"}],"createdAt":"2026-07-30T14:25:12Z","files":[{"path":".gitkeep","additions":1,"deletions":0,"changeType":"ADDED"}],"headRefName":"issue-1-09b0c76bd0e4","isDraft":true,"state":"OPEN","title":"[WIP] Implement Hello World in Rust","url":"https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/pull/2"} From 21197003f6e597951165e6f988cc48dfc15f6a47 Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 16:02:36 +0000 Subject: [PATCH 03/17] fix(2119): frame streamed JSON so token usage is not lost `--model formal-ai` runs reported "0 input, 0 output" tokens even though the tools produced real usage. Formal AI emits pretty-printed JSON objects rather than NDJSON, and codex writes partial lines, so the per-line parsers silently dropped every event that carried the usage metadata. Adds src/json-stream.lib.mjs, a shared incremental framer that accepts both NDJSON and multi-line/pretty-printed JSON and buffers partial lines, and routes the agent and codex stream readers through it. Reproduction: https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2 --- src/agent-token-usage.lib.mjs | 14 +- src/agent.lib.mjs | 256 +++++++++++--------------- src/codex.lib.mjs | 49 ++++- src/json-stream.lib.mjs | 195 ++++++++++++++++++++ tests/test-agent-stream-json-2119.mjs | 137 ++++++++++++++ tests/test-agent-token-usage.mjs | 51 +++-- 6 files changed, 518 insertions(+), 184 deletions(-) create mode 100644 src/json-stream.lib.mjs create mode 100644 tests/test-agent-stream-json-2119.mjs diff --git a/src/agent-token-usage.lib.mjs b/src/agent-token-usage.lib.mjs index 0b70fa062..b597e7667 100644 --- a/src/agent-token-usage.lib.mjs +++ b/src/agent-token-usage.lib.mjs @@ -2,6 +2,7 @@ import Decimal from 'decimal.js-light'; import { sanitizeObjectStrings } from './unicode-sanitization.lib.mjs'; +import { parseJsonRecords } from './json-stream.lib.mjs'; import { getCumulativeContextInputTokens, getRestoredContextInputTokens } from './context-fill.lib.mjs'; export const createTokenFieldAvailability = () => ({ @@ -95,15 +96,10 @@ export const accumulateAgentStepFinishUsage = (usage, data) => { export const parseAgentTokenUsage = output => { const usage = createAgentTokenUsage(); - for (const rawLine of output.split('\n')) { - const line = rawLine.trim(); - if (!line || !line.startsWith('{')) continue; - - try { - accumulateAgentStepFinishUsage(usage, sanitizeObjectStrings(JSON.parse(line))); - } catch { - continue; - } + // Issue #2119: records are framed by balanced JSON rather than by newlines, + // so pretty-printed (multi-line) and concatenated records are counted too. + for (const record of parseJsonRecords(output)) { + accumulateAgentStepFinishUsage(usage, sanitizeObjectStrings(record)); } return usage; diff --git a/src/agent.lib.mjs b/src/agent.lib.mjs index 882107b6a..2cd80766f 100644 --- a/src/agent.lib.mjs +++ b/src/agent.lib.mjs @@ -21,10 +21,12 @@ import { detectUsageLimit, formatUsageLimitMessage } from './usage-limit.lib.mjs import { sanitizeObjectStrings } from './unicode-sanitization.lib.mjs'; import Decimal from 'decimal.js-light'; import semver from 'semver'; -import { agentModels, defaultModels, freeToBaseModelMap } from './models/index.mjs'; +import { agentModels, defaultModels, freeToBaseModelMap, isFormalAiModel } from './models/index.mjs'; import { logPreparedToolCommand, resolveFormalAiToolInvocation } from './formal-ai.lib.mjs'; +import { buildFormalAiPricingInfo } from './formal-ai-pricing.lib.mjs'; // Issue #2119 import { checkPlaywrightMcpPackageAvailability, getAgentPlaywrightMcpDisableEnv } from './playwright-mcp.lib.mjs'; import { createAgentTokenUsage, accumulateAgentStepFinishUsage, parseAgentTokenUsage } from './agent-token-usage.lib.mjs'; +import { createJsonStreamScanner, parseJsonRecords } from './json-stream.lib.mjs'; import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs'; import { attachStreamingInput, finalizeBidirectionalHandler, setupBidirectionalHandler } from './bidirectional-interactive.lib.mjs'; @@ -111,6 +113,10 @@ const getBaseModelForPricing = modelName => { * - opencodeCost: Actual billed cost from OpenCode Zen (free for most models) */ export const calculateAgentPricing = async (modelId, tokenUsage) => { + // Issue #2119: Formal AI requests never reach OpenCode Zen, so neither the + // provider label nor a models.dev price lookup applies to them. + if (isFormalAiModel(modelId)) return buildFormalAiPricingInfo(modelId, tokenUsage); + // Extract the model name from provider/model format // e.g., 'opencode/grok-code' -> 'grok-code' const modelName = modelId.includes('/') ? modelId.split('/').pop() : modelId; @@ -632,80 +638,92 @@ export const executeAgentCommand = async params => { } }; - for await (const chunk of execCommand.stream()) { - if (chunk.type === 'stdout') { - const output = chunk.data.toString(); - // Split output into individual lines for NDJSON parsing - // Agent outputs NDJSON (newline-delimited JSON) format where each line is a separate JSON object - // This allows us to parse each event independently and extract structured data like session IDs - const lines = output.split('\n'); - for (const line of lines) { - if (!line.trim()) continue; - try { - const data = sanitizeObjectStrings(JSON.parse(line)); - // Issue #1968: a bare `null`/primitive NDJSON line must not abort - // event processing (any data.X access would throw on null). - if (data === null || typeof data !== 'object') continue; - // Output formatted JSON - await log(JSON.stringify(data, null, 2)); - // Capture session ID from the first message - const eventSessionId = data.sessionID || data.session_id || data.sessionId; - if (!sessionId && eventSessionId) { - sessionId = eventSessionId; - await log(`๐Ÿ“Œ Session ID: ${sessionId}`); - } - // Issue #1250: Accumulate token usage during streaming - accumulateTokenUsage(data); - await markBidirectionalStateFromAgentEvent(data); - // Issue #1201: Detect error events during streaming for reliable detection - if (data.type === 'error' || data.type === 'step_error') { - streamingErrorDetected = true; - streamingErrorMessage = data.message || data.error || line.substring(0, 100); - await log(`โš ๏ธ Error event detected in stream: ${streamingErrorMessage}`, { level: 'warning' }); - } - // Issue #1263: Track text content for result summary - // Agent outputs text via 'text', 'assistant', or 'message' type events - if (data.type === 'text' && data.text) { - lastTextContent = data.text; - } else if (data.type === 'assistant' && data.message?.content) { - // Extract text from assistant message content - const content = Array.isArray(data.message.content) ? data.message.content : [data.message.content]; - for (const item of content) { - if (item.type === 'text' && item.text) { - lastTextContent = item.text; - } - } - } else if (data.type === 'message' && data.content) { - // Direct message content - if (typeof data.content === 'string') { - lastTextContent = data.content; - } else if (Array.isArray(data.content)) { - for (const item of data.content) { - if (item.type === 'text' && item.text) { - lastTextContent = item.text; - } - } - } - } else if (data.type === 'result' && data.result) { - // Explicit result message (like Claude outputs) - lastTextContent = data.result; - } - // Issue #1276: Detect successful completion events - // When agent emits session.idle or log with "exiting loop" message, it completed successfully - // This means any previous error events were recovered from (e.g., timeout then retry) - if (isAgentSuccessfulCompletionEvent(data)) { - agentCompletedSuccessfully = true; + // Issue #2119: agentic CLIs do not all emit strict one-record-per-line + // NDJSON. `formal-ai with agent --verbose` emits pretty-printed, + // multi-line records, and records can also be concatenated without a + // separator (issue #1250) or split across process chunks. A line-based + // JSON.parse dropped every structured event in those cases, which is how + // a session that really used 21677/22834 tokens was published as + // "Token usage: 0 input, 0 output" with no session id and no result + // summary. The scanner frames records by balanced JSON instead of by + // newlines, and surfaces anything that is not JSON as plain text. + const stdoutScanner = createJsonStreamScanner(); + const stderrScanner = createJsonStreamScanner(); + + const handleAgentJsonEvent = async (raw, value) => { + const data = sanitizeObjectStrings(value); + // Issue #1968: a bare `null`/primitive record must not abort event + // processing (any data.X access would throw on null). + if (data === null || typeof data !== 'object') return; + // Output formatted JSON + await log(JSON.stringify(data, null, 2)); + // Capture session ID from the first message (agent may use stdout or stderr) + const eventSessionId = data.sessionID || data.session_id || data.sessionId; + if (!sessionId && eventSessionId) { + sessionId = eventSessionId; + await log(`๐Ÿ“Œ Session ID: ${sessionId}`); + } + // Issue #1250: Accumulate token usage during streaming + accumulateTokenUsage(data); + await markBidirectionalStateFromAgentEvent(data); + // Issue #1201: Detect error events during streaming for reliable detection + if (data.type === 'error' || data.type === 'step_error') { + streamingErrorDetected = true; + streamingErrorMessage = data.message || data.error || raw.substring(0, 100); + await log(`โš ๏ธ Error event detected in stream: ${streamingErrorMessage}`, { level: 'warning' }); + } + // Issue #1263: Track text content for result summary + // Agent outputs text via 'text', 'assistant', or 'message' type events + if (data.type === 'text' && data.text) { + lastTextContent = data.text; + } else if (data.type === 'assistant' && data.message?.content) { + // Extract text from assistant message content + const content = Array.isArray(data.message.content) ? data.message.content : [data.message.content]; + for (const item of content) { + if (item.type === 'text' && item.text) { + lastTextContent = item.text; + } + } + } else if (data.type === 'message' && data.content) { + // Direct message content + if (typeof data.content === 'string') { + lastTextContent = data.content; + } else if (Array.isArray(data.content)) { + for (const item of data.content) { + if (item.type === 'text' && item.text) { + lastTextContent = item.text; } - // Issue #1296: Detect step_finish with reason "stop" as successful completion - // This is a clear marker of success - agent finished normally, not due to error or limit - // When this event appears, we should ignore any error events that appeared earlier in the stream - // (e.g., timeout errors that were recovered from via retry logic) - if (data.type === 'step_finish' && data.part?.reason === 'stop') agentCompletedSuccessfully = true; - } catch { - // Not JSON - log as plain text - await log(line); } } + } else if (data.type === 'result' && data.result) { + // Explicit result message (like Claude outputs) + lastTextContent = data.result; + } + // Issue #1276: Detect successful completion events + // When agent emits session.idle or log with "exiting loop" message, it completed successfully + // This means any previous error events were recovered from (e.g., timeout then retry) + if (isAgentSuccessfulCompletionEvent(data)) { + agentCompletedSuccessfully = true; + } + // Issue #1296: Detect step_finish with reason "stop" as successful completion + // This is a clear marker of success - agent finished normally, not due to error or limit + // When this event appears, we should ignore any error events that appeared earlier in the stream + // (e.g., timeout errors that were recovered from via retry logic) + if (data.type === 'step_finish' && data.part?.reason === 'stop') agentCompletedSuccessfully = true; + }; + + const handleAgentStreamEvents = async events => { + for (const event of events) { + if (event.type === 'json') await handleAgentJsonEvent(event.raw, event.value); + // Not JSON - log as plain text + else await log(event.value); + } + }; + + for await (const chunk of execCommand.stream()) { + if (chunk.type === 'stdout') { + const output = chunk.data.toString(); + await handleAgentStreamEvents(stdoutScanner.write(output)); lastMessage = output; fullOutput += output; // Collect for both pricing calculation and error detection } @@ -714,69 +732,8 @@ export const executeAgentCommand = async params => { const errorOutput = chunk.data.toString(); if (errorOutput) { // Agent sends all output (including verbose logs and structured events) to stderr - // Process each line as NDJSON, same as stdout handling - const stderrLines = errorOutput.split('\n'); - for (const stderrLine of stderrLines) { - if (!stderrLine.trim()) continue; - try { - const stderrData = sanitizeObjectStrings(JSON.parse(stderrLine)); - // Issue #1968: skip bare `null`/primitive lines (see stdout handler above). - if (stderrData === null || typeof stderrData !== 'object') continue; - // Output formatted JSON (same formatting as stdout) - await log(JSON.stringify(stderrData, null, 2)); - // Capture session ID from stderr too (agent sends it via stderr) - const eventSessionId = stderrData.sessionID || stderrData.session_id || stderrData.sessionId; - if (!sessionId && eventSessionId) { - sessionId = eventSessionId; - await log(`๐Ÿ“Œ Session ID: ${sessionId}`); - } - // Issue #1250: Accumulate token usage during streaming (stderr) - accumulateTokenUsage(stderrData); - await markBidirectionalStateFromAgentEvent(stderrData); - // Issue #1201: Detect error events during streaming (stderr) for reliable detection - if (stderrData.type === 'error' || stderrData.type === 'step_error') { - streamingErrorDetected = true; - streamingErrorMessage = stderrData.message || stderrData.error || stderrLine.substring(0, 100); - await log(`โš ๏ธ Error event detected in stream: ${streamingErrorMessage}`, { level: 'warning' }); - } - // Issue #1263: Track text content for result summary (stderr) - if (stderrData.type === 'text' && stderrData.text) { - lastTextContent = stderrData.text; - } else if (stderrData.type === 'assistant' && stderrData.message?.content) { - const content = Array.isArray(stderrData.message.content) ? stderrData.message.content : [stderrData.message.content]; - for (const item of content) { - if (item.type === 'text' && item.text) { - lastTextContent = item.text; - } - } - } else if (stderrData.type === 'message' && stderrData.content) { - if (typeof stderrData.content === 'string') { - lastTextContent = stderrData.content; - } else if (Array.isArray(stderrData.content)) { - for (const item of stderrData.content) { - if (item.type === 'text' && item.text) { - lastTextContent = item.text; - } - } - } - } else if (stderrData.type === 'result' && stderrData.result) { - lastTextContent = stderrData.result; - } - // Issue #1276: Detect successful completion events (stderr) - // When agent emits session.idle or log with "exiting loop" message, it completed successfully - if (isAgentSuccessfulCompletionEvent(stderrData)) { - agentCompletedSuccessfully = true; - } - // Issue #1296: Detect step_finish with reason "stop" as successful completion (stderr) - // This is a clear marker of success - agent finished normally, not due to error or limit - if (stderrData.type === 'step_finish' && stderrData.part?.reason === 'stop') { - agentCompletedSuccessfully = true; - } - } catch { - // Not JSON - log as plain text - await log(stderrLine); - } - } + // Process it exactly like stdout so telemetry is never stream-specific + await handleAgentStreamEvents(stderrScanner.write(errorOutput)); // Also collect stderr for error detection fullOutput += errorOutput; } @@ -785,6 +742,10 @@ export const executeAgentCommand = async params => { } } + // Release any record that was still being assembled when the stream ended. + await handleAgentStreamEvents(stdoutScanner.flush()); + await handleAgentStreamEvents(stderrScanner.flush()); + // Simplified error detection for agent tool // Issue #886: Trust exit code - agent now properly returns code 1 on errors with JSON error response // Don't scan output for error patterns as this causes false positives during normal operation @@ -795,24 +756,17 @@ export const executeAgentCommand = async params => { // 2. Explicit JSON error messages from agent (type: "error") // 3. Usage limit detection (handled separately) const detectAgentErrors = stdoutOutput => { - const lines = stdoutOutput.split('\n'); + // Issue #2119: frame records by balanced JSON, not by newlines, so + // pretty-printed and concatenated records are still inspected. + for (const record of parseJsonRecords(stdoutOutput)) { + const msg = sanitizeObjectStrings(record); - for (const line of lines) { - if (!line.trim()) continue; + // Issue #1968: ignore bare `null`/primitive records (msg.type would throw on null). + if (msg === null || typeof msg !== 'object') continue; - try { - const msg = sanitizeObjectStrings(JSON.parse(line)); - - // Issue #1968: ignore bare `null`/primitive lines (msg.type would throw on null). - if (msg === null || typeof msg !== 'object') continue; - - // Check for explicit error message types from agent - if (msg.type === 'error' || msg.type === 'step_error') { - return { detected: true, type: 'AgentError', match: msg.message || msg.error || line.substring(0, 100) }; - } - } catch { - // Not JSON - ignore for error detection - continue; + // Check for explicit error message types from agent + if (msg.type === 'error' || msg.type === 'step_error') { + return { detected: true, type: 'AgentError', match: msg.message || msg.error || JSON.stringify(msg).substring(0, 100) }; } } diff --git a/src/codex.lib.mjs b/src/codex.lib.mjs index cdda69394..fdc9d29f8 100644 --- a/src/codex.lib.mjs +++ b/src/codex.lib.mjs @@ -25,13 +25,15 @@ import { detectUsageLimit, formatUsageLimitMessage } from './usage-limit.lib.mjs import { buildSolveResumeCommand } from './solve.resume-command.lib.mjs'; // Issue #942 const __codexBuildSolveResumeCmd = (argv, sessionId, tempDir) => (sessionId && argv?.url ? buildSolveResumeCommand({ issueUrl: argv.url, sessionId, tool: 'codex', model: argv.model, fallbackModel: argv.fallbackModel, tempDir }) : null); import { sanitizeObjectStrings } from './unicode-sanitization.lib.mjs'; +import { createLineBuffer } from './json-stream.lib.mjs'; // Issue #2119 import { mapModelToId, resolveCodexReasoningEffort } from './codex.options.lib.mjs'; import { createInteractiveHandler } from './interactive-mode.lib.mjs'; import { initProgressMonitoring } from './solve.progress-monitoring.lib.mjs'; import { ensureCodexPlaywrightMcpServer, getCodexPlaywrightMcpDisableConfigArgs } from './playwright-mcp.lib.mjs'; import { fetchModelInfo } from './model-info.lib.mjs'; -import { defaultModels } from './models/index.mjs'; +import { defaultModels, isFormalAiModel } from './models/index.mjs'; import { logPreparedToolCommand, resolveFormalAiToolInvocation } from './formal-ai.lib.mjs'; +import { buildFormalAiPricingInfo } from './formal-ai-pricing.lib.mjs'; // Issue #2119 import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs'; import { parseSubSessionSize, buildCodexSubSessionSizeConfigArgs, buildCodexDisable1mContextConfigArgs } from './sub-session-size.lib.mjs'; // Issue #1706 import { getCumulativeContextInputTokens } from './context-fill.lib.mjs'; @@ -581,6 +583,9 @@ export const calculateCodexPricingFromModelInfo = (modelId, tokenUsage, modelInf export const calculateCodexPricing = async (modelId, tokenUsage) => { if (!modelId) return null; + // Issue #2119: a Formal AI session is served by the local Link.Assistant + // model server, so OpenAI pricing must not be applied to it. + if (isFormalAiModel(modelId)) return buildFormalAiPricingInfo(modelId, tokenUsage); try { const modelInfo = await fetchModelInfo(modelId, { preferredProviderIds: ['openai'] }); return calculateCodexPricingFromModelInfo(modelId, tokenUsage, modelInfo); @@ -976,13 +981,21 @@ export const executeCodexCommand = async params => { observedModelDiagnosticPaths: [], }; + // Issue #2119: a process chunk boundary can fall in the middle of an + // NDJSON record. Parsing each raw chunk dropped both halves of a split + // record (token usage, session id, auth errors). Buffer whole lines so + // the line-oriented Codex parser never sees a partial record. + const codexStdoutLines = createLineBuffer(); + const codexStderrLines = createLineBuffer(); + for await (const chunk of execCommand.stream()) { if (chunk.type === 'stdout') { - const output = chunk.data.toString(); + const raw = chunk.data.toString(); if (argv.verbose) { - await log(output); + await log(raw); } - lastMessage = output; + lastMessage = raw; + const output = codexStdoutLines.write(raw); codexJsonState = parseCodexExecJsonOutput(output, codexJsonState, mappedModel); await baseBranchCommandIntervention.handleCommandExecutions(codexJsonState.commandExecutions); @@ -1022,10 +1035,11 @@ export const executeCodexCommand = async params => { } if (chunk.type === 'stderr') { - const errorOutput = chunk.data.toString(); - if (errorOutput && argv.verbose) { - await log(errorOutput, { stream: 'stderr' }); + const rawError = chunk.data.toString(); + if (rawError && argv.verbose) { + await log(rawError, { stream: 'stderr' }); } + const errorOutput = codexStderrLines.write(rawError); codexJsonState = parseCodexExecJsonOutput(errorOutput, codexJsonState, mappedModel); await baseBranchCommandIntervention.handleCommandExecutions(codexJsonState.commandExecutions); } else if (chunk.type === 'exit') { @@ -1033,6 +1047,27 @@ export const executeCodexCommand = async params => { } } + // Release any line that was still being assembled when the stream ended. + for (const remaining of [codexStdoutLines.flush(), codexStderrLines.flush()]) { + if (!remaining.trim()) continue; + codexJsonState = parseCodexExecJsonOutput(remaining, codexJsonState, mappedModel); + await baseBranchCommandIntervention.handleCommandExecutions(codexJsonState.commandExecutions); + } + + if (codexJsonState.sessionId && codexJsonState.sessionId !== sessionId) { + sessionId = codexJsonState.sessionId; + await log(`๐Ÿ“Œ Session ID: ${sessionId}`); + } + if (codexJsonState.resultSummary) { + lastTextContent = codexJsonState.resultSummary; + } + if (codexJsonState.authError && !authError) { + authError = true; + await log('\nโŒ Authentication error detected in Codex JSON stream', { level: 'error' }); + await log(' This error cannot be resolved by retrying.', { level: 'error' }); + await log(' ๐Ÿ’ก Please run: codex login', { level: 'error' }); + } + if (interactiveHandler) { await interactiveHandler.flush(); } diff --git a/src/json-stream.lib.mjs b/src/json-stream.lib.mjs new file mode 100644 index 000000000..507ce85e2 --- /dev/null +++ b/src/json-stream.lib.mjs @@ -0,0 +1,195 @@ +#!/usr/bin/env node + +/** + * Incremental JSON record scanner for agentic CLI output streams. + * + * Issue #2119: the agent stream readers split the raw output on newlines and + * called `JSON.parse` on each line. That works only when the tool emits strict + * NDJSON on line boundaries that happen to align with process chunk + * boundaries. Three real-world stream shapes break it: + * + * 1. Pretty-printed records โ€” `formal-ai with agent --verbose` emits + * multi-line, indented JSON, so *every* line fails to parse. Every + * structured event (session id, token usage, errors, result text) is then + * dropped, which is how a run with 21677 input / 22834 output tokens was + * published as "Token usage: 0 input, 0 output". + * 2. Concatenated records โ€” `{...}{...}` arriving without a separator + * (issue #1250). + * 3. Split records โ€” one record spanning two process chunks. + * + * Scanning for balanced JSON values instead of relying on line framing handles + * all three with a single mechanism, and non-JSON output is still surfaced + * verbatim as text events so plain tool logs keep flowing. + */ + +// A pending fragment that never balances (for example prose that happens to +// start with `{`) must not grow without bound. Once the buffer exceeds this +// size it is released as text. +export const DEFAULT_MAX_PENDING_BYTES = 4 * 1024 * 1024; + +const isStructuralOpener = character => character === '{' || character === '['; + +/** + * Find the index just past the JSON value starting at `start`. + * @returns {number} end index (exclusive), or -1 when the value is incomplete. + */ +const findValueEnd = (buffer, start) => { + let depth = 0; + let inString = false; + let escaped = false; + + for (let index = start; index < buffer.length; index++) { + const character = buffer[index]; + + if (inString) { + if (escaped) escaped = false; + else if (character === '\\') escaped = true; + else if (character === '"') inString = false; + continue; + } + + if (character === '"') { + inString = true; + continue; + } + if (character === '{' || character === '[') { + depth++; + continue; + } + if (character === '}' || character === ']') { + depth--; + if (depth === 0) return index + 1; + if (depth < 0) return -1; + } + } + + return -1; +}; + +/** + * Create a stateful scanner that turns a byte stream into JSON and text events. + * + * @param {Object} [options] + * @param {number} [options.maxPendingBytes] release an unbalanced buffer once it grows past this size + * @returns {{write: (chunk: string) => Array, flush: () => Array}} + */ +export const createJsonStreamScanner = (options = {}) => { + const maxPendingBytes = options.maxPendingBytes ?? DEFAULT_MAX_PENDING_BYTES; + let pending = ''; + + const emitTextLines = (text, events) => { + for (const line of text.split('\n')) { + if (line.trim()) events.push({ type: 'text', value: line }); + } + }; + + const scan = (final, events) => { + let index = 0; + + while (index < pending.length) { + const character = pending[index]; + + if (character === '\n' || character === '\r' || character === ' ' || character === '\t') { + index++; + continue; + } + + if (isStructuralOpener(character)) { + const end = findValueEnd(pending, index); + if (end < 0) break; // incomplete record: wait for more input + const raw = pending.slice(index, end); + try { + events.push({ type: 'json', value: JSON.parse(raw), raw }); + } catch { + // Balanced but not valid JSON: surface it verbatim rather than + // silently dropping tool output. + emitTextLines(raw, events); + } + index = end; + continue; + } + + const newline = pending.indexOf('\n', index); + if (newline < 0) break; // incomplete text line: wait for more input + const line = pending.slice(index, newline).replace(/\r$/, ''); + if (line.trim()) events.push({ type: 'text', value: line }); + index = newline + 1; + } + + pending = pending.slice(index); + + if (final) { + if (pending.trim()) { + const trimmed = pending.trim(); + let parsed = null; + if (isStructuralOpener(trimmed[0])) { + try { + parsed = { type: 'json', value: JSON.parse(trimmed), raw: trimmed }; + } catch { + parsed = null; + } + } + if (parsed) events.push(parsed); + else emitTextLines(pending, events); + } + pending = ''; + } else if (pending.length > maxPendingBytes) { + emitTextLines(pending, events); + pending = ''; + } + + return events; + }; + + return { + write(chunk) { + pending += String(chunk ?? ''); + return scan(false, []); + }, + flush() { + return scan(true, []); + }, + }; +}; + +/** + * Buffer a byte stream into whole lines. + * + * Line-oriented parsers (Codex NDJSON plus its interleaved OTEL diagnostics) + * stay correct as long as they never see half a line. A process chunk boundary + * can fall anywhere, so the trailing partial line is carried over to the next + * chunk and released by `flush()`. + * + * @returns {{write: (chunk: string) => string, flush: () => string}} + */ +export const createLineBuffer = () => { + let pending = ''; + + return { + write(chunk) { + pending += String(chunk ?? ''); + const boundary = pending.lastIndexOf('\n'); + if (boundary < 0) return ''; + const complete = pending.slice(0, boundary + 1); + pending = pending.slice(boundary + 1); + return complete; + }, + flush() { + const rest = pending; + pending = ''; + return rest; + }, + }; +}; + +/** + * Extract every complete JSON record from a finished output buffer. + * + * @param {string} output raw tool output + * @returns {Array} parsed JSON values in stream order + */ +export const parseJsonRecords = output => { + const scanner = createJsonStreamScanner(); + const events = [...scanner.write(String(output ?? '')), ...scanner.flush()]; + return events.filter(event => event.type === 'json').map(event => event.value); +}; diff --git a/tests/test-agent-stream-json-2119.mjs b/tests/test-agent-stream-json-2119.mjs new file mode 100644 index 000000000..12178f980 --- /dev/null +++ b/tests/test-agent-stream-json-2119.mjs @@ -0,0 +1,137 @@ +#!/usr/bin/env node +/** + * @hive-mind-test-suite default + * + * Regression coverage for issue #2119. + * + * `formal-ai with agent --verbose` emits pretty-printed, multi-line JSON + * records. The stream readers split the raw output on newlines and called + * JSON.parse per line, so every structured event was dropped: no session id, + * no result summary, no error detection and - most visibly - a published + * "Token usage: 0 input, 0 output" for a session that really used + * 21677 input / 22834 output tokens. + * + * Evidence: docs/case-studies/issue-2119/data/logs/agent-scala-solution-draft.log + * (lines 2245-2282 show the step_finish record; the PR comment reported 0/0). + * + * Framing records by balanced JSON instead of by newlines must also keep + * handling strict NDJSON, records concatenated without a separator + * (issue #1250) and records split across process chunks. + */ + +import assert from 'node:assert/strict'; + +import { createJsonStreamScanner, createLineBuffer, parseJsonRecords } from '../src/json-stream.lib.mjs'; +import { parseAgentTokenUsage } from '../src/agent-token-usage.lib.mjs'; +import { parseCodexExecJsonOutput } from '../src/codex.lib.mjs'; + +const stepFinishRecord = { + type: 'step_finish', + timestamp: 1785420850750, + sessionID: 'ses_04c9fe206ffeSfRyxn0SRerDHe', + part: { + id: 'prt_fb3602628001gT9qWhJRldfytM', + type: 'step-finish', + reason: 'tool-calls', + cost: 0, + tokens: { input: 21677, output: 22834, reasoning: 0, cache: { read: 0, write: 0 } }, + model: { providerID: 'formalai', requestedModelID: 'formal-ai', respondedModelID: 'formal-ai' }, + context: { contextLimit: 60000, outputLimit: 8192, usableContext: 51808, safeLimit: 38856 }, + }, +}; + +const prettyStream = `${JSON.stringify(stepFinishRecord, null, 2)}\n`; +const ndjsonStream = `${JSON.stringify(stepFinishRecord)}\n`; +const concatenatedStream = `${JSON.stringify(stepFinishRecord)}${JSON.stringify(stepFinishRecord)}\n`; + +// --- record framing ------------------------------------------------------- + +assert.equal(parseJsonRecords(prettyStream).length, 1, 'pretty-printed record must be recovered'); +assert.deepEqual(parseJsonRecords(prettyStream)[0], stepFinishRecord); +assert.equal(parseJsonRecords(ndjsonStream).length, 1, 'NDJSON record must still be recovered'); +assert.equal(parseJsonRecords(concatenatedStream).length, 2, 'concatenated records must be split (issue #1250)'); + +// A record split across process chunks must be assembled, not dropped. +const scanner = createJsonStreamScanner(); +const serialized = JSON.stringify(stepFinishRecord, null, 2); +const splitAt = Math.floor(serialized.length / 2); +assert.deepEqual(scanner.write(serialized.slice(0, splitAt)), [], 'incomplete record must not be emitted'); +const chunkEvents = scanner.write(`${serialized.slice(splitAt)}\n`); +assert.equal(chunkEvents.length, 1, 'record split across chunks must be emitted once complete'); +assert.deepEqual(chunkEvents[0].value, stepFinishRecord); + +// Braces inside strings must not confuse the framing. +const bracedRecords = parseJsonRecords('{"type":"text","text":"a } b { c"}\n'); +assert.equal(bracedRecords.length, 1, 'braces inside strings must not break framing'); +assert.equal(bracedRecords[0].text, 'a } b { c'); + +// Non-JSON output is still surfaced verbatim as text events. +const mixedScanner = createJsonStreamScanner(); +const mixedEvents = [...mixedScanner.write(`agent: starting\n${JSON.stringify(stepFinishRecord)}\nagent: done\n`), ...mixedScanner.flush()]; +assert.deepEqual( + mixedEvents.map(event => event.type), + ['text', 'json', 'text'], + 'plain tool output must still be surfaced' +); +assert.equal(mixedEvents[0].value, 'agent: starting'); +assert.equal(mixedEvents[2].value, 'agent: done'); + +// An unbalanced fragment must be released rather than buffered forever. +const overflowScanner = createJsonStreamScanner({ maxPendingBytes: 64 }); +const overflowEvents = overflowScanner.write(`{${'x'.repeat(200)}\n`); +assert.ok(overflowEvents.length > 0, 'unbalanced buffer must be released as text once it exceeds the cap'); +assert.ok( + overflowEvents.every(event => event.type === 'text'), + 'released fragment must be surfaced as text' +); + +// --- token accounting ----------------------------------------------------- + +for (const [label, stream] of [ + ['pretty-printed', prettyStream], + ['ndjson', ndjsonStream], +]) { + const usage = parseAgentTokenUsage(stream); + assert.equal(usage.stepCount, 1, `${label}: step must be counted`); + assert.equal(usage.inputTokens, 21677, `${label}: input tokens must be counted`); + assert.equal(usage.outputTokens, 22834, `${label}: output tokens must be counted`); + assert.equal(usage.requestedModelId, 'formal-ai', `${label}: requested model must be captured`); + assert.equal(usage.contextLimit, 60000, `${label}: context limit must be captured`); +} + +const concatenatedUsage = parseAgentTokenUsage(concatenatedStream); +assert.equal(concatenatedUsage.stepCount, 2, 'concatenated records must both be counted'); +assert.equal(concatenatedUsage.inputTokens, 21677 * 2, 'concatenated input tokens must accumulate'); + +// --- line-oriented streams (codex) ---------------------------------------- + +// The Codex parser is line-oriented, so it stays correct as long as it never +// sees half a line. A process chunk boundary can fall anywhere, which used to +// destroy both halves of the record it split. +const codexRecords = ['{"type":"thread.started","thread_id":"th_2119"}', '{"type":"turn.completed","usage":{"input_tokens":21677,"output_tokens":22834}}']; +const codexStream = `${codexRecords.join('\n')}\n`; + +// Split the stream mid-record, exactly as a process chunk boundary would. +const codexSplitAt = codexStream.indexOf('usage') + 3; +const codexChunks = [codexStream.slice(0, codexSplitAt), codexStream.slice(codexSplitAt)]; + +const unbufferedState = codexChunks.reduce((state, chunk) => parseCodexExecJsonOutput(chunk, state, 'gpt-5.5'), {}); +assert.equal(unbufferedState.tokenUsage.inputTokens, 0, 'baseline: an unbuffered split record is lost'); + +const codexLines = createLineBuffer(); +let bufferedState = {}; +for (const chunk of codexChunks) { + bufferedState = parseCodexExecJsonOutput(codexLines.write(chunk), bufferedState, 'gpt-5.5'); +} +bufferedState = parseCodexExecJsonOutput(codexLines.flush(), bufferedState, 'gpt-5.5'); +assert.equal(bufferedState.sessionId, 'th_2119', 'buffered: session id survives a split chunk'); +assert.equal(bufferedState.tokenUsage.inputTokens, 21677, 'buffered: input tokens survive a split chunk'); +assert.equal(bufferedState.tokenUsage.outputTokens, 22834, 'buffered: output tokens survive a split chunk'); + +// A stream whose final record has no trailing newline must still be released. +const trailingBuffer = createLineBuffer(); +assert.equal(trailingBuffer.write('{"type":"thread.started","thread_id":"th_tail"}'), '', 'partial line is held back'); +const tailState = parseCodexExecJsonOutput(trailingBuffer.flush(), {}, 'gpt-5.5'); +assert.equal(tailState.sessionId, 'th_tail', 'flush releases a record with no trailing newline'); + +console.log('โœ… issue #2119: agent stream records are framed by balanced JSON'); diff --git a/tests/test-agent-token-usage.mjs b/tests/test-agent-token-usage.mjs index ed7c4af15..0ec2770d2 100644 --- a/tests/test-agent-token-usage.mjs +++ b/tests/test-agent-token-usage.mjs @@ -165,14 +165,28 @@ runTest('handles token values of 0', () => { // ==== Issue #1250: Streaming accumulation tests ==== console.log('\n๐Ÿ“‹ Test Group: Issue #1250 - Streaming accumulation\n'); -runTest('Issue #1250: concatenated JSON without newlines fails parsing', () => { - // This demonstrates why the fix was needed - when NDJSON lines are concatenated - // without newlines, JSON.parse fails because it sees two objects together +runTest('Issue #1250: concatenated JSON without newlines is recovered (issue #2119)', () => { + // Concatenated records used to be lost entirely, because parsing was framed by + // newlines and `{...}{...}` is not valid JSON on a single line. Issue #2119 + // replaced line framing with balanced-JSON framing, so both records are now + // recovered by post-hoc parsing as well as by streaming accumulation. const concatenated = '{"type":"step_finish","part":{"tokens":{"input":100,"output":50}}}{"type":"step_finish","part":{"tokens":{"input":200,"output":100}}}'; const result = parseAgentTokenUsage(concatenated); - // This should fail to parse both - demonstrating the bug that the streaming fix addresses - assertEqual(result.stepCount, 0, 'Concatenated JSON without newlines fails to parse'); + assertEqual(result.stepCount, 2, 'Concatenated JSON records are both counted'); + assertEqual(result.inputTokens, 300, 'Concatenated input tokens accumulate'); + assertEqual(result.outputTokens, 150, 'Concatenated output tokens accumulate'); +}); + +runTest('Issue #2119: pretty-printed JSON records are recovered', () => { + // `formal-ai with agent --verbose` emits indented, multi-line records, so every + // line failed to parse and the whole session reported 0 input / 0 output tokens. + const pretty = `${JSON.stringify({ type: 'step_finish', part: { tokens: { input: 21677, output: 22834 } } }, null, 2)}\n`; + const result = parseAgentTokenUsage(pretty); + + assertEqual(result.stepCount, 1, 'Pretty-printed record is counted'); + assertEqual(result.inputTokens, 21677, 'Pretty-printed input tokens are counted'); + assertEqual(result.outputTokens, 22834, 'Pretty-printed output tokens are counted'); }); runTest('Issue #1250: properly newline-delimited JSON parses correctly', () => { @@ -293,16 +307,18 @@ runTest('Issue #1313: streaming accumulation correctly sums tokens like in the b assertEqual(streamingTokenUsage.stepCount, 1, 'Streaming should count 1 step'); }); -runTest('Issue #1313: concatenated JSON (old bug scenario) gives 0 tokens', () => { - // This demonstrates the EXACT scenario that caused Issue #1313: - // When NDJSON lines are concatenated without newlines, post-hoc parsing gives 0 tokens. - // The streaming accumulation fix (Issue #1250) solved this by NOT relying on post-hoc parsing. +runTest('Issue #1313: concatenated JSON is no longer lost by post-hoc parsing', () => { + // This is the EXACT scenario that caused Issue #1313: NDJSON lines concatenated + // without newlines used to make post-hoc parsing return 0 tokens. Streaming + // accumulation (issue #1250) worked around it; balanced-JSON framing + // (issue #2119) fixes the post-hoc path too, so both agree. const concatenatedOutput = '{"type":"step_finish","part":{"tokens":{"input":406,"output":353,"reasoning":281,"cache":{"read":33880,"write":0}}}}' + '{"type":"step_finish","part":{"tokens":{"input":100,"output":50,"reasoning":0,"cache":{"read":5000,"write":0}}}}'; - // parseAgentTokenUsage (post-hoc) fails because two JSON objects are concatenated const result = parseAgentTokenUsage(concatenatedOutput); - assertEqual(result.stepCount, 0, 'Post-hoc parsing of concatenated JSON returns 0 (the old bug)'); - assertEqual(result.inputTokens, 0, 'Should return 0 input (demonstrates why streaming fix was needed)'); + assertEqual(result.stepCount, 2, 'Post-hoc parsing recovers both concatenated records'); + assertEqual(result.inputTokens, 506, 'Should sum input tokens across concatenated records'); + assertEqual(result.outputTokens, 403, 'Should sum output tokens across concatenated records'); + assertEqual(result.cacheReadTokens, 38880, 'Should sum cache reads across concatenated records'); }); // ==== accumulateTokenUsage function tests ==== @@ -744,7 +760,7 @@ runTest('Issue #1313: the exact gist log data parses non-zero tokens', () => { if (usage.inputTokens === 0) throw new Error('BUG REPRODUCED: streaming accumulation returns 0 for valid data!'); }); -runTest('Issue #1313: demonstrates old post-hoc bug with concatenated JSON', () => { +runTest('Issue #1313: post-hoc parsing now agrees with streaming on concatenated JSON', () => { // Old code tried JSON.parse(fullOutput) where fullOutput had concatenated JSON objects // This is exactly what happened in the v1.21.4 code before the streaming fix const concatenatedNDJSON = '{"type":"step_finish","part":{"tokens":{"input":406,"output":353}}}' + '{"type":"step_finish","part":{"tokens":{"input":512,"output":128}}}'; @@ -759,12 +775,13 @@ runTest('Issue #1313: demonstrates old post-hoc bug with concatenated JSON', () oldApproachResult = null; // Failed to parse - returns no tokens (the bug) } - assertEqual(oldApproachResult, null, 'Old approach: JSON.parse fails on concatenated JSON'); + assertEqual(oldApproachResult, null, 'Old approach: a single JSON.parse fails on concatenated JSON'); - // New approach: parseAgentTokenUsage processes line by line - // But even this can fail if there are no newlines (the bug was that lines got concatenated) + // Issue #2119: parseAgentTokenUsage frames records by balanced JSON instead of + // by newlines, so concatenation no longer hides either record. const parseResult = parseAgentTokenUsage(concatenatedNDJSON); - assertEqual(parseResult.stepCount, 0, 'parseAgentTokenUsage also fails on concatenated JSON (no newlines)'); + assertEqual(parseResult.stepCount, 2, 'parseAgentTokenUsage recovers both concatenated records'); + assertEqual(parseResult.inputTokens, 918, 'parseAgentTokenUsage sums concatenated input tokens'); // The REAL fix: streaming accumulation processes each chunk as it arrives // So it never sees concatenated output - it processes events one by one From 2072f214e56474b401850aec22cbc30af0da971a Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 16:02:47 +0000 Subject: [PATCH 04/17] fix(2119): stop redacting token counters and JSON braces as credentials The credential sanitizer treated token telemetry (`"total_tokens":44511`, `tokens=44511`, `"maxTokens": 8192`) and JSON structural braces as secrets, so published logs and comments in the reproduction PRs showed redacted counters and mangled JSON, hiding the real usage numbers. Reproduction: https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/pull/2 --- src/credential-sanitization-core.lib.mjs | 30 ++++++- ...test-token-telemetry-sanitization-2119.mjs | 87 +++++++++++++++++++ 2 files changed, 114 insertions(+), 3 deletions(-) create mode 100644 tests/test-token-telemetry-sanitization-2119.mjs diff --git a/src/credential-sanitization-core.lib.mjs b/src/credential-sanitization-core.lib.mjs index 1110b2ff2..fe7ba49d5 100644 --- a/src/credential-sanitization-core.lib.mjs +++ b/src/credential-sanitization-core.lib.mjs @@ -71,9 +71,33 @@ const VENDOR_PATTERNS = Object.freeze([ ]); const SENSITIVE_KEY = String.raw`(?:[A-Za-z0-9_.-]*(?:api[-_]?key|account[-_]?key|client[-_]?secret|consumer[-_]?secret|webhook[-_]?secret|access[-_]?token|refresh[-_]?token|auth[-_]?token|password|passwd|pwd|private[-_]?key|secret|token|session[-_]?key|session[-_]?token|cookie|docker[-_]?auth|registry[-_]?auth|shared[-_]?access[-_]?signature|sas[-_]?token)[A-Za-z0-9_.-]*|auth|authorization)`; + +// Issue #2119: token *accounting* is not a credential. Every AI provider SDK +// spells usage telemetry with the plural "tokens" (`tokens`, `inputTokens`, +// `prompt_tokens`, `total_tokens`) or with an explicit quantity suffix +// (`token_count`, `tokenLimit`). Masking those numbers corrupted the NDJSON +// telemetry in published logs and destroyed token/cost accounting, while +// protecting nothing: a credential is never a bare number under a plural name. +// The exemption stays deliberately narrow - it requires both a counter-shaped +// key and a purely numeric value, so `access_token=123456` is still masked. +const TOKEN_COUNTER_KEY = /(?:tokens|token(?:count|limit|usage|budget|used|size|s?remaining)|(?:count|limit|usage|budget|used|size)tokens?)$/; +const NUMERIC_VALUE = /^[+-]?(?:\d+(?:\.\d+)?|\.\d+)(?:e[+-]?\d+)?$/i; + +const normalizeAssignmentKey = prefix => + String(prefix ?? '') + .replace(/\s*(?:=>|[:=])\s*$/, '') + .replace(/[^A-Za-z0-9]/g, '') + .toLowerCase(); + +const isTokenCounterAssignment = (prefix, value) => NUMERIC_VALUE.test(String(value ?? '').trim()) && TOKEN_COUNTER_KEY.test(normalizeAssignmentKey(prefix)); const SENSITIVE_ENV_NAME = /(?:API_?KEY|ACCOUNT_?KEY|CLIENT_?SECRET|CONSUMER_?SECRET|WEBHOOK_?SECRET|ACCESS_?TOKEN|REFRESH_?TOKEN|AUTH_?TOKEN|PASSWORD|PASSWD|PRIVATE_?KEY|SECRET|TOKEN|COOKIE|AUTH)$/i; const QUOTED_ASSIGNMENT = new RegExp(`((?:["']?${SENSITIVE_KEY}["']?)\\s*(?:=>|[:=])\\s*)(["'])([^"'\\r\\n]*)(\\2)`, 'gi'); -const UNQUOTED_ASSIGNMENT = new RegExp(`((?:["']?${SENSITIVE_KEY}["']?)\\s*(?:=>|[:=])\\s*)(?!["']|(?:Bearer|Basic|SharedAccessSignature)\\s)([^\\s,;}&'"\\r\\n]+)`, 'gi'); +// Issue #2119: a value that *opens* a JSON/JS structure is punctuation, not a +// secret. Without this guard `"tokens": {` was rewritten to `"tokens": [REDACTED]`, +// which silently truncated the object and made the whole record unparseable. +// The guard only rejects a structural character in first position, so a +// credential that merely contains a brace (`password=ab{cd`) is still masked whole. +const UNQUOTED_ASSIGNMENT = new RegExp(`((?:["']?${SENSITIVE_KEY}["']?)\\s*(?:=>|[:=])\\s*)(?!["']|[{[]|(?:Bearer|Basic|SharedAccessSignature)\\s)([^\\s,;}&'"\\r\\n]+)`, 'gi'); const XML_CREDENTIAL = new RegExp(`(<(${SENSITIVE_KEY})\\b[^>]*>)([\\s\\S]*?)(<\\/\\2\\s*>)`, 'gi'); const CLI_CREDENTIAL_QUOTED = new RegExp(`(--${SENSITIVE_KEY}(?:\\s+|=))(["'])([^"'\\r\\n]*)(\\2)`, 'gi'); const CLI_CREDENTIAL = new RegExp(`(--${SENSITIVE_KEY}(?:\\s+|=))(?!["'])([^\\s"'\\r\\n]+)`, 'gi'); @@ -136,8 +160,8 @@ export const sanitizeCredentialText = (input, options = {}) => { // XML and JSON/YAML/TOML/INI/shell-style assignments. output = output.replace(XML_CREDENTIAL, (_match, start, _key, value, end) => `${start}${maskValue(value.trim())}${end}`); - output = output.replace(QUOTED_ASSIGNMENT, (_match, prefix, quote, value) => `${prefix}${quote}${maskValue(value)}${quote}`); - output = output.replace(UNQUOTED_ASSIGNMENT, (_match, prefix, value) => `${prefix}${maskValue(value)}`); + output = output.replace(QUOTED_ASSIGNMENT, (match, prefix, quote, value) => (isTokenCounterAssignment(prefix, value) ? match : `${prefix}${quote}${maskValue(value)}${quote}`)); + output = output.replace(UNQUOTED_ASSIGNMENT, (match, prefix, value) => (isTokenCounterAssignment(prefix, value) ? match : `${prefix}${maskValue(value)}`)); // CLI arguments and sensitive query parameters. output = output.replace(CLI_CREDENTIAL_QUOTED, (_match, prefix, quote, value) => `${prefix}${quote}${maskValue(value)}${quote}`); diff --git a/tests/test-token-telemetry-sanitization-2119.mjs b/tests/test-token-telemetry-sanitization-2119.mjs new file mode 100644 index 000000000..85b77c5ae --- /dev/null +++ b/tests/test-token-telemetry-sanitization-2119.mjs @@ -0,0 +1,87 @@ +#!/usr/bin/env node +/** + * @hive-mind-test-suite default + * + * Regression coverage for issue #2119. + * + * The credential sanitizer treated every key containing the substring "token" + * as a credential name. Because its unquoted-value pattern also accepted the + * JSON structural opener `{`, the agent telemetry record + * + * "tokens": { + * "input": 21677, + * + * was published as `"tokens": [REDACTED]`, which truncated the object, made the + * record unparseable and destroyed token/cost accounting in the uploaded logs. + * + * Evidence: docs/case-studies/issue-2119/data/logs/agent-scala-solution-draft.log + * + * The exemption must stay narrow: only a counter-shaped key with a purely + * numeric value is telemetry. Real credentials, including numeric ones under + * singular credential names, must still be masked. + */ + +import assert from 'node:assert/strict'; + +import { sanitizeCredentialText } from '../src/credential-sanitization-core.lib.mjs'; + +const sanitize = text => sanitizeCredentialText(text, { includeEnvironmentCredentials: false }); + +// A JSON structural opener is punctuation, never a secret. +for (const opener of ['{', '[']) { + const record = ` "tokens": ${opener}`; + assert.equal(sanitize(record), record, `structural opener must survive sanitization: ${record}`); +} + +// The exact record shape emitted by the agent CLI must round-trip through the +// sanitizer and still parse as JSON. +const telemetryRecord = JSON.stringify( + { + type: 'step_finish', + sessionID: 'ses_04c9fe206ffeSfRyxn0SRerDHe', + part: { + type: 'step-finish', + reason: 'tool-calls', + cost: 0, + tokens: { input: 21677, output: 22834, reasoning: 0, cache: { read: 0, write: 0 } }, + model: { providerID: 'formalai', requestedModelID: 'formal-ai', respondedModelID: 'formal-ai' }, + context: { contextLimit: 60000, outputLimit: 8192, currentTokens: 38856, headroom: -5655 }, + }, + }, + null, + 2 +); + +const sanitizedRecord = sanitize(telemetryRecord); +assert.equal(sanitizedRecord, telemetryRecord, 'token telemetry must not be rewritten by the sanitizer'); +const reparsed = JSON.parse(sanitizedRecord); +assert.equal(reparsed.part.tokens.input, 21677, 'input token count must survive sanitization'); +assert.equal(reparsed.part.tokens.output, 22834, 'output token count must survive sanitization'); +assert.equal(reparsed.part.context.currentTokens, 38856, 'context counter must survive sanitization'); + +// Counter-shaped keys from every provider dialect we consume. +const counterAssignments = ['"prompt_tokens": 1234', '"completion_tokens": 5', '"total_tokens":44511', '"cache_read_input_tokens": 0', '"inputTokens": 21677', '"maxTokens": 8192', '"token_count": 17', '"tokenLimit": 60000', 'tokens=44511']; +for (const assignment of counterAssignments) { + assert.equal(sanitize(assignment), assignment, `token counter must survive sanitization: ${assignment}`); +} + +// Real credentials must still be masked, including numeric ones and values +// that merely contain a brace. +const mustBeMasked = [ + ['access_token=abcdef1234567890abcdef', 'abcdef1234567890abcdef'], + ['"access_token": "abcdef1234567890abcdef"', 'abcdef1234567890abcdef'], + ['token=123456', '123456'], + ['password=1234', '1234'], + ['"tokens": "sk-secretvalue1234567890"', 'sk-secretvalue1234567890'], + ['password=ab{cdefghijklmnop', 'ab{cdefghijklmnop'], + ['"api_key": "AKIAIOSFODNN7EXAMPLE"', 'AKIAIOSFODNN7EXAMPLE'], +]; +for (const [input, secret] of mustBeMasked) { + const output = sanitize(input); + assert.ok(!output.includes(secret), `credential survived sanitization: ${input}`); +} + +// Sanitization stays idempotent for telemetry. +assert.equal(sanitize(sanitize(telemetryRecord)), telemetryRecord, 'sanitization must be idempotent'); + +console.log('โœ… issue #2119: token telemetry survives credential sanitization'); From 8f7f173ac3e06b42e90183d3097f298cedb4e051 Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 16:03:00 +0000 Subject: [PATCH 05/17] fix(2119): report formal-ai as Link.Assistant at $0.00 for every tool The reproduction PRs attributed `--model formal-ai` to "OpenCode Zen" or "Anthropic" and billed it: the claude path published $0.252315 taken from Anthropic's `total_cost_usd`, and the public pricing estimate read "unknown". Formal AI is served by Link.Assistant and is free. Adds src/formal-ai-pricing.lib.mjs as the single source of the provider name and the zero-cost override, and applies it at the cost boundary of all six tools (claude, agent, opencode, codex, gemini, qwen) plus the agent commander and the cost-info renderer, which now prints "$0.00 (Free model)" instead of "unknown". Reproductions: - https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2 - https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2 --- src/agent-commander.lib.mjs | 39 +++++-- src/anthropic-cost-accumulator.lib.mjs | 36 +++++++ src/claude.lib.mjs | 28 ++--- src/formal-ai-pricing.lib.mjs | 110 ++++++++++++++++++++ src/gemini.lib.mjs | 24 +++-- src/github-cost-info.lib.mjs | 5 + src/opencode.lib.mjs | 124 +++++++++------------- src/qwen.lib.mjs | 16 ++- tests/test-formal-ai-pricing-2119.mjs | 137 +++++++++++++++++++++++++ 9 files changed, 412 insertions(+), 107 deletions(-) create mode 100644 src/formal-ai-pricing.lib.mjs create mode 100644 tests/test-formal-ai-pricing-2119.mjs diff --git a/src/agent-commander.lib.mjs b/src/agent-commander.lib.mjs index f8876a831..80328906c 100644 --- a/src/agent-commander.lib.mjs +++ b/src/agent-commander.lib.mjs @@ -11,6 +11,7 @@ import { resolveCodexReasoningEffort } from './codex.options.lib.mjs'; import { mapClaudeSubAgentModelToEnvValue, mapModelForTool } from './models/index.mjs'; import { buildCodexDisable1mContextConfigArgs, buildCodexSubSessionSizeConfigArgs, parseSubSessionSize } from './sub-session-size.lib.mjs'; import { detectUsageLimit } from './usage-limit.lib.mjs'; +import { applyFormalAiPricingOverride } from './formal-ai-pricing.lib.mjs'; // Issue #2119 import { getCacheReadTokenCount, getCumulativeContextInputTokens, getOutputTokenCount } from './context-fill.lib.mjs'; export const AGENT_COMMANDER_TOOLS = new Set(['claude', 'codex', 'opencode', 'agent', 'qwen', 'gemini']); @@ -257,25 +258,34 @@ const enrichPricingInfoWithTokenUsage = ({ pricingInfo = null, usage = null, too }; }; -export const summarizeAgentCommanderResult = ({ result, tool }) => { +export const summarizeAgentCommanderResult = ({ result, tool, model = null }) => { const plainOutput = result?.output?.plain || ''; if (result?.metadata && typeof result.metadata === 'object') { const metadata = result.metadata; const streamTokenUsage = metadata.streamTokenUsage || result.usage || null; - const pricingInfo = enrichPricingInfoWithTokenUsage({ + const enrichedPricingInfo = enrichPricingInfoWithTokenUsage({ pricingInfo: metadata.pricingInfo || null, usage: streamTokenUsage, tool, publicPricingEstimate: metadata.publicPricingEstimate ?? metadata.pricingInfo?.totalCostUSD ?? null, }); + // Issue #2119: a Formal AI session belongs to Link.Assistant at $0.00, no + // matter which agentic CLI agent-commander drove it. + const { pricingInfo, publicPricingEstimate, anthropicTotalCostUSD } = applyFormalAiPricingOverride({ + model, + pricingInfo: enrichedPricingInfo, + publicPricingEstimate: metadata.publicPricingEstimate ?? enrichedPricingInfo?.totalCostUSD ?? null, + anthropicTotalCostUSD: metadata.anthropicTotalCostUSD ?? null, + tokenUsage: streamTokenUsage, + }); return { success: metadata.success === true, sessionId: metadata.sessionId || result.sessionId || null, limitReached: !!metadata.limitReached, limitResetTime: metadata.limitResetTime || null, limitTimezone: metadata.limitTimezone || null, - anthropicTotalCostUSD: metadata.anthropicTotalCostUSD ?? null, - publicPricingEstimate: metadata.publicPricingEstimate ?? pricingInfo?.totalCostUSD ?? null, + anthropicTotalCostUSD, + publicPricingEstimate, pricingInfo, resultSummary: metadata.resultSummary || null, resultModelUsage: metadata.resultModelUsage || null, @@ -291,12 +301,19 @@ export const summarizeAgentCommanderResult = ({ result, tool }) => { const usage = result?.usage || null; const resultMessage = [...messages].reverse().find(message => message?.type === 'result') || null; const totalCost = typeof resultMessage?.total_cost_usd === 'number' ? resultMessage.total_cost_usd : null; - const publicPricingEstimate = tool === 'agent' && typeof usage?.totalCost === 'number' ? usage.totalCost : null; - const pricingInfo = enrichPricingInfoWithTokenUsage({ - pricingInfo: publicPricingEstimate !== null ? { totalCostUSD: publicPricingEstimate, source: 'agent-commander' } : null, + const rawPublicPricingEstimate = tool === 'agent' && typeof usage?.totalCost === 'number' ? usage.totalCost : null; + const enrichedPricingInfo = enrichPricingInfoWithTokenUsage({ + pricingInfo: rawPublicPricingEstimate !== null ? { totalCostUSD: rawPublicPricingEstimate, source: 'agent-commander' } : null, usage, tool, - publicPricingEstimate, + publicPricingEstimate: rawPublicPricingEstimate, + }); + const { pricingInfo, publicPricingEstimate, anthropicTotalCostUSD } = applyFormalAiPricingOverride({ + model, + pricingInfo: enrichedPricingInfo, + publicPricingEstimate: rawPublicPricingEstimate ?? enrichedPricingInfo?.totalCostUSD ?? null, + anthropicTotalCostUSD: tool === 'claude' ? totalCost : null, + tokenUsage: usage, }); return { @@ -305,8 +322,8 @@ export const summarizeAgentCommanderResult = ({ result, tool }) => { limitReached: usageLimit.isUsageLimit, limitResetTime: usageLimit.resetTime, limitTimezone: usageLimit.timezone, - anthropicTotalCostUSD: tool === 'claude' ? totalCost : null, - publicPricingEstimate: publicPricingEstimate ?? pricingInfo?.totalCostUSD ?? null, + anthropicTotalCostUSD, + publicPricingEstimate, pricingInfo, resultSummary: extractResultSummary(messages, plainOutput), resultModelUsage: null, @@ -367,7 +384,7 @@ export const executeWithAgentCommander = async params => { const result = await controller.stop(); await log(`[agent-commander] ${tool} exited with code ${result.exitCode}`); - return summarizeAgentCommanderResult({ result, tool }); + return summarizeAgentCommanderResult({ result, tool, model: argv.model }); }; export const checkForUncommittedChanges = async (tempDir, owner, repo, branchName, $, log = defaultLog, autoCommit = false, autoRestartEnabled = true) => { diff --git a/src/anthropic-cost-accumulator.lib.mjs b/src/anthropic-cost-accumulator.lib.mjs index ea6c132a7..db6f0dbcb 100644 --- a/src/anthropic-cost-accumulator.lib.mjs +++ b/src/anthropic-cost-accumulator.lib.mjs @@ -42,6 +42,8 @@ * future model. See docs/case-studies/issue-1886/ for the full analysis. */ +import { isFormalAiModel } from './models/index.mjs'; // Issue #2119 + // Module-level singleton: the cumulative Anthropic cost for the active logical // session (including anything seeded by a true resume from a prior process). let cumulativeAnthropicCostUSD = 0; @@ -120,6 +122,40 @@ export const getCumulativeAnthropicCost = () => cumulativeAnthropicCostUSD; */ export const hasCumulativeAnthropicCost = () => cumulativeAnthropicCostUSD > 0; +/** + * Interpret a Claude `result` event's `total_cost_usd`. + * + * Issue #1886: a non-success terminal event (e.g. a usage-limit hit) still + * reports this process's cost, so it is kept as an accumulation fallback rather + * than as the authoritative total. + * + * Issue #2119: `--model formal-ai` is served by the local Link.Assistant model + * server, so the session never billed Anthropic. Claude Code nevertheless + * reports a `total_cost_usd` derived from Anthropic list prices for the model + * name it sees ($0.252315 in the issue) - a false positive that must not reach + * the accumulator or the published cost comment. + * + * @param {Object} params + * @param {Object} params.data the parsed `result` stream event + * @param {string|null} params.model the model requested on the command line + * @param {Function} params.log logger + * @returns {Promise<{total?: number, fallback?: number}|null>} the captured cost, or null when none applies + */ +export const captureAnthropicResultCost = async ({ data, model, log }) => { + const cost = data?.total_cost_usd; + if (cost === undefined || cost === null) return null; + if (isFormalAiModel(model)) { + await log(`๐Ÿ’ฐ Ignoring Anthropic cost $${cost.toFixed(6)} reported for a Formal AI session (Link.Assistant, free)`, { verbose: true }); + return null; + } + if (data.subtype === 'success') { + await log(`๐Ÿ’ฐ Anthropic official cost captured from success result: $${cost.toFixed(6)}`, { verbose: true }); + return { total: cost }; + } + await log(`๐Ÿ’ฐ Anthropic cost from ${data.subtype || 'unknown'} result kept as fallback for accumulation: $${cost.toFixed(6)}`, { verbose: true }); + return { fallback: cost }; +}; + /** * Reset the accumulator. Intended for tests; production code starts scopes via * `beginAnthropicCostScope`. diff --git a/src/claude.lib.mjs b/src/claude.lib.mjs index 361884253..0bd874419 100644 --- a/src/claude.lib.mjs +++ b/src/claude.lib.mjs @@ -17,11 +17,12 @@ import { sanitizeObjectStrings } from './unicode-sanitization.lib.mjs'; import Decimal from 'decimal.js-light'; import { createEmptySubSessionUsage, accumulateModelUsage, mergeResultModelUsage, createSubAgentCallEntry, accumulateSubAgentUsage, getRawRequestInputTokens, displaySessionTokenUsage } from './claude.budget-stats.lib.mjs'; import { buildClaudeResumeCommand, buildClaudeAutonomousResumeCommand } from './claude.command-builder.lib.mjs'; -import { beginAnthropicCostScope, seedCumulativeAnthropicCost, addAnthropicRunCost } from './anthropic-cost-accumulator.lib.mjs'; // Issues #1886, #2056 +import { beginAnthropicCostScope, seedCumulativeAnthropicCost, addAnthropicRunCost, captureAnthropicResultCost } from './anthropic-cost-accumulator.lib.mjs'; // Issues #1886, #2056, #2119 import { buildSolveResumeCommand } from './solve.resume-command.lib.mjs'; // Issue #942 import { SESSION_FORCE_KILLED_MARKER, postTrackedComment } from './tool-comments.lib.mjs'; // Issue #1625 import { handleClaudeRuntimeSwitch } from './claude.runtime-switch.lib.mjs'; // see issue #1141 import { CLAUDE_MODELS as availableModels, mapClaudeSubAgentModelToEnvValue } from './models/index.mjs'; // Issue #1221, #1978 +import { applyFormalAiPricingOverride } from './formal-ai-pricing.lib.mjs'; // Issue #2119 import { logPreparedToolCommand, resolveFormalAiToolInvocation } from './formal-ai.lib.mjs'; import { buildMcpConfigWithoutPlaywright, ensureClaudePlaywrightMcpServer } from './playwright-mcp.lib.mjs'; import { resolveClaudeSessionToolFlags } from './useless-tools.lib.mjs'; @@ -876,14 +877,9 @@ export const executeClaudeCommand = async params => { } } if (data.subtype === 'success') resultSuccessReceived = true; - if (data.subtype === 'success' && data.total_cost_usd !== undefined && data.total_cost_usd !== null) { - anthropicTotalCostUSD = data.total_cost_usd; - await log(`๐Ÿ’ฐ Anthropic official cost captured from success result: $${anthropicTotalCostUSD.toFixed(6)}`, { verbose: true }); - } else if (data.total_cost_usd !== undefined && data.total_cost_usd !== null) { - // Issue #1886: non-success terminal (e.g. usage-limit hit) still reports this process's cost โ€” keep as accumulation fallback. - anthropicCostFromAnyResult = data.total_cost_usd; - await log(`๐Ÿ’ฐ Anthropic cost from ${data.subtype || 'unknown'} result kept as fallback for accumulation: $${data.total_cost_usd.toFixed(6)}`, { verbose: true }); - } + const capturedCost = await captureAnthropicResultCost({ data, model: argv.model, log }); + if (capturedCost?.total !== undefined) anthropicTotalCostUSD = capturedCost.total; + if (capturedCost?.fallback !== undefined) anthropicCostFromAnyResult = capturedCost.fallback; // Issue #1263: Extract result summary (AI's summary of work done) for --attach-solution-summary if (data.subtype === 'success' && data.result && typeof data.result === 'string') { resultSummary = data.result; @@ -1070,10 +1066,9 @@ export const executeClaudeCommand = async params => { if (data.result && typeof data.result === 'string') resultSummary = data.result; if (data.modelUsage) resultModelUsage = data.modelUsage; } - if (data.total_cost_usd != null) { - if (data.subtype === 'success') anthropicTotalCostUSD = data.total_cost_usd; - else anthropicCostFromAnyResult = data.total_cost_usd; - } + const capturedCost = await captureAnthropicResultCost({ data, model: argv.model, log }); + if (capturedCost?.total !== undefined) anthropicTotalCostUSD = capturedCost.total; + if (capturedCost?.fallback !== undefined) anthropicCostFromAnyResult = capturedCost.fallback; } // Issue #1472: Forward remaining buffer event to interactive handler (was previously missed) if (interactiveHandler) { @@ -1422,7 +1417,12 @@ export const executeClaudeCommand = async params => { } }; // End of executeWithRetry function // Start the execution with retry logic - return await executeWithRetry(); + const claudeResult = (await executeWithRetry()) || {}; + // Issue #2119: `--model formal-ai` runs against the local Link.Assistant model + // server. Claude reports no pricing record of its own, so without this the + // session was published with Anthropic's cost and no provider at all. + const formalAiPricing = applyFormalAiPricingOverride({ model: argv.model, pricingInfo: claudeResult.pricingInfo ?? null, publicPricingEstimate: claudeResult.publicPricingEstimate ?? null, anthropicTotalCostUSD: claudeResult.anthropicTotalCostUSD ?? null, tokenUsage: claudeResult.streamTokenUsage ?? null }); + return { ...claudeResult, ...formalAiPricing }; }; export const checkForUncommittedChanges = async (tempDir, owner, repo, branchName, $, log, autoCommit = false, autoRestartEnabled = true) => { await log('\n๐Ÿ” Checking for uncommitted changes...'); diff --git a/src/formal-ai-pricing.lib.mjs b/src/formal-ai-pricing.lib.mjs new file mode 100644 index 000000000..d1f50bff4 --- /dev/null +++ b/src/formal-ai-pricing.lib.mjs @@ -0,0 +1,110 @@ +#!/usr/bin/env node + +/** + * Pricing and provider identity for the Formal AI model (issue #2119). + * + * `--model formal-ai` routes every request through a local Formal AI model + * server (`formal-ai with ...`). The requests never reach OpenCode Zen, + * OpenAI, Anthropic or Google, so: + * + * - the provider is Link.Assistant, not the provider of whichever agentic CLI + * happens to be driving the session; + * - the cost is $0.00, so an inherited `total_cost_usd` (claude reported + * $0.252315 for a Formal AI session) or a models.dev price lookup for an + * unrelated base model is a false positive. + * + * Every pricing producer funnels through here so the same numbers appear in + * logs, GitHub comments and budget statistics. + */ + +import { FORMAL_AI_MODEL_ALIAS, isFormalAiModel } from './models/index.mjs'; + +export const FORMAL_AI_PROVIDER_NAME = 'Link.Assistant'; + +const ZERO_PRICING = Object.freeze({ + inputPerMillion: 0, + outputPerMillion: 0, + cacheReadPerMillion: 0, + cacheWritePerMillion: 0, + reasoningPerMillion: 0, +}); + +const ZERO_BREAKDOWN = Object.freeze({ + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + reasoning: 0, +}); + +/** + * Build the pricing record for a Formal AI session. + * + * @param {string|null} modelId model id as passed to the tool (alias or `formalai/formal-ai`) + * @param {Object|null} tokenUsage aggregated token usage, kept so token counts stay reportable + * @returns {Object} pricing info with a Link.Assistant provider and a $0.00 cost + */ +export const buildFormalAiPricingInfo = (modelId = FORMAL_AI_MODEL_ALIAS, tokenUsage = null) => ({ + modelId: modelId || FORMAL_AI_MODEL_ALIAS, + modelName: FORMAL_AI_MODEL_ALIAS, + provider: FORMAL_AI_PROVIDER_NAME, + // No third-party price applies, so there is no base model to reference. + originalProvider: null, + baseModelName: null, + tokenUsage: tokenUsage || null, + pricing: { ...ZERO_PRICING }, + breakdown: { ...ZERO_BREAKDOWN }, + totalCostUSD: 0, + isFreeModel: true, + isFormalAi: true, +}); + +/** + * Wrap a `(modelId, tokenUsage) => pricingInfo` calculator so Formal AI model + * ids short-circuit to the free Link.Assistant record instead of being priced + * against models.dev. + * + * @param {Function} calculatePricing the tool's own pricing calculator + * @returns {Function} wrapped calculator with the same signature + */ +export const withFormalAiPricing = + calculatePricing => + async (modelId, tokenUsage, ...rest) => { + if (isFormalAiModel(modelId)) return buildFormalAiPricingInfo(modelId, tokenUsage); + return calculatePricing(modelId, tokenUsage, ...rest); + }; + +/** + * Normalize a tool result's pricing fields for Formal AI sessions. + * + * Tools that report a provider cost of their own (claude's `total_cost_usd`) + * or that build a static provider record (gemini's "Google", qwen's "Alibaba") + * would otherwise attribute a Formal AI session to the wrong provider at a + * non-zero price. + * + * @param {Object} params + * @param {string|null} params.model model requested on the command line + * @param {Object|null} [params.pricingInfo] + * @param {number|null} [params.publicPricingEstimate] + * @param {number|null} [params.anthropicTotalCostUSD] + * @param {Object|null} [params.tokenUsage] fallback token usage when pricingInfo carries none + * @returns {{pricingInfo: Object|null, publicPricingEstimate: number|null, anthropicTotalCostUSD: number|null}} + */ +export const applyFormalAiPricingOverride = ({ model, pricingInfo = null, publicPricingEstimate = null, anthropicTotalCostUSD = null, tokenUsage = null }) => { + if (!isFormalAiModel(model)) return { pricingInfo, publicPricingEstimate, anthropicTotalCostUSD }; + + const usage = pricingInfo?.tokenUsage || tokenUsage || null; + // Drop provider-specific cost fields carried by the tool's own record: a + // Formal AI session was never billed by OpenCode Zen, so a + // "Calculated by OpenCode Zen" line would be a false positive. + const { opencodeCost: _opencodeCost, isOpencodeFreeModel: _isOpencodeFreeModel, ...carried } = pricingInfo || {}; + return { + pricingInfo: { ...carried, ...buildFormalAiPricingInfo(pricingInfo?.modelId || model, usage) }, + publicPricingEstimate: 0, + // The session never billed Anthropic, so any captured Anthropic cost is a + // false positive and must not be rendered as a second cost line. + anthropicTotalCostUSD: null, + }; +}; + +export { isFormalAiModel }; diff --git a/src/gemini.lib.mjs b/src/gemini.lib.mjs index 3b7e2a14c..597c03376 100644 --- a/src/gemini.lib.mjs +++ b/src/gemini.lib.mjs @@ -17,8 +17,9 @@ import { detectUsageLimit, formatUsageLimitMessage } from './usage-limit.lib.mjs import { buildSolveResumeCommand } from './solve.resume-command.lib.mjs'; // Issue #942 const __geminiBuildSolveResumeCmd = (argv, sessionId, tempDir) => (sessionId && argv?.url ? buildSolveResumeCommand({ issueUrl: argv.url, sessionId, tool: 'gemini', model: argv.model, fallbackModel: argv.fallbackModel, tempDir }) : null); import { sanitizeObjectStrings } from './unicode-sanitization.lib.mjs'; -import { defaultModels, geminiModels } from './models/index.mjs'; +import { defaultModels, geminiModels, isFormalAiModel } from './models/index.mjs'; import { logPreparedToolCommand, resolveFormalAiToolInvocation } from './formal-ai.lib.mjs'; +import { buildFormalAiPricingInfo } from './formal-ai-pricing.lib.mjs'; // Issue #2119 import { checkPlaywrightMcpPackageAvailability } from './playwright-mcp.lib.mjs'; import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs'; import { getCumulativeContextInputTokens, toTokenCount } from './context-fill.lib.mjs'; @@ -130,6 +131,15 @@ const pickTokenValue = (...values) => { return 0; }; +/** + * Issue #2119: `--model formal-ai` is served by the local Link.Assistant model + * server, so the session must not be attributed to Google. + */ +export const buildGeminiPricingInfo = mappedModel => { + if (isFormalAiModel(mappedModel)) return buildFormalAiPricingInfo(mappedModel); + return { modelId: mappedModel, modelName: mappedModel, provider: 'Google', totalCostUSD: null }; +}; + export const buildGeminiResultModelUsage = (modelId, stats = null) => { const modelStats = stats?.models && typeof stats.models === 'object' ? stats.models : null; if (modelStats) { @@ -587,8 +597,8 @@ export const executeGeminiCommand = async params => { messageCount: geminiJsonState.messageCount || 0, toolUseCount: geminiJsonState.toolUseCount || 0, resultModelUsage: geminiJsonState.resultModelUsage || buildGeminiResultModelUsage(mappedModel), - pricingInfo: { modelId: mappedModel, modelName: mappedModel, provider: 'Google', totalCostUSD: null }, - publicPricingEstimate: null, + pricingInfo: buildGeminiPricingInfo(mappedModel), + publicPricingEstimate: buildGeminiPricingInfo(mappedModel).totalCostUSD, resultSummary: geminiJsonState.resultSummary || null, // Issue #1845/#1941: surface the actual error, rejecting meaningless fragments (e.g. a lone "}") errorInfo: { message: buildToolErrorMessage({ lastMessage: errorText, exitCode, fallback: `Gemini command failed with exit code ${exitCode}`, toolLabel: 'Gemini' }), exitCode }, @@ -631,8 +641,8 @@ export const executeGeminiCommand = async params => { messageCount: geminiJsonState.messageCount || 0, toolUseCount: geminiJsonState.toolUseCount || 0, resultModelUsage: geminiJsonState.resultModelUsage || buildGeminiResultModelUsage(mappedModel), - pricingInfo: { modelId: mappedModel, modelName: mappedModel, provider: 'Google', totalCostUSD: null }, - publicPricingEstimate: null, + pricingInfo: buildGeminiPricingInfo(mappedModel), + publicPricingEstimate: buildGeminiPricingInfo(mappedModel).totalCostUSD, resultSummary: geminiJsonState.resultSummary || null, completionHealth, incompleteSession: completionHealth.incompleteSession, @@ -655,8 +665,8 @@ export const executeGeminiCommand = async params => { messageCount: geminiJsonState.messageCount || 0, toolUseCount: geminiJsonState.toolUseCount || 0, resultModelUsage: geminiJsonState.resultModelUsage || buildGeminiResultModelUsage(mappedModel), - pricingInfo: { modelId: mappedModel, modelName: mappedModel, provider: 'Google', totalCostUSD: null }, - publicPricingEstimate: null, + pricingInfo: buildGeminiPricingInfo(mappedModel), + publicPricingEstimate: buildGeminiPricingInfo(mappedModel).totalCostUSD, resultSummary: geminiJsonState.resultSummary || null, }; } catch (error) { diff --git a/src/github-cost-info.lib.mjs b/src/github-cost-info.lib.mjs index 14f881a32..a89fa9a03 100644 --- a/src/github-cost-info.lib.mjs +++ b/src/github-cost-info.lib.mjs @@ -51,6 +51,11 @@ export const buildCostInfoString = (totalCostUSD, anthropicTotalCostUSD, pricing } costInfo += `\n- Public pricing estimate: $${publicDec.toFixed(6)}${pricingRef}`; } + } else if (pricingInfo?.isFreeModel && !pricingInfo?.baseModelName) { + // Issue #2119: a free model has a known price - $0.00 - even when no + // usage-derived estimate was produced. Reporting "unknown" for it was a + // false negative (`--model formal-ai` is served free by Link.Assistant). + costInfo += '\n- Public pricing estimate: $0.00 (Free model)'; } else if (hasPricing) { costInfo += '\n- Public pricing estimate: unknown'; } diff --git a/src/opencode.lib.mjs b/src/opencode.lib.mjs index 626a23638..3b9bdb16a 100644 --- a/src/opencode.lib.mjs +++ b/src/opencode.lib.mjs @@ -23,6 +23,7 @@ import { opencodeModels, defaultModels } from './models/index.mjs'; import { logPreparedToolCommand, resolveFormalAiToolInvocation } from './formal-ai.lib.mjs'; import { checkPlaywrightMcpPackageAvailability, getOpenCodePlaywrightMcpDisableEnv } from './playwright-mcp.lib.mjs'; import { createAgentTokenUsage, accumulateAgentStepFinishUsage, parseAgentTokenUsage as parseOpenCodeTokenUsage } from './agent-token-usage.lib.mjs'; +import { createJsonStreamScanner } from './json-stream.lib.mjs'; import { calculateAgentPricing } from './agent.lib.mjs'; import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs'; @@ -336,6 +337,48 @@ export const executeOpenCodeCommand = async params => { let lastTextContent = ''; // Issue #1263: Track last text content for result summary let allOutput = ''; // Collect all output for error detection + // Issue #2119: frame records by balanced JSON instead of by newlines, so + // pretty-printed, concatenated and chunk-split records are all counted. + // The previous per-chunk try/catch also aborted parsing of the whole + // chunk as soon as one line was not JSON. + const stdoutScanner = createJsonStreamScanner(); + const stderrScanner = createJsonStreamScanner(); + + const handleOpenCodeRecords = events => { + for (const event of events) { + if (event.type !== 'json') continue; + const data = sanitizeObjectStrings(event.value); + // Issue #1968: a bare `null`/primitive record must not abort the + // rest of the chunk (data.type access would throw on null). + if (data === null || typeof data !== 'object') continue; + accumulateAgentStepFinishUsage(streamingTokenUsage, data); + // Track text content for result summary + // OpenCode outputs text via 'text', 'assistant', 'message', or 'result' type events + if (data.type === 'text' && data.text) { + lastTextContent = data.text; + } else if (data.type === 'assistant' && data.message?.content) { + const content = Array.isArray(data.message.content) ? data.message.content : [data.message.content]; + for (const item of content) { + if (item.type === 'text' && item.text) { + lastTextContent = item.text; + } + } + } else if (data.type === 'message' && data.content) { + if (typeof data.content === 'string') { + lastTextContent = data.content; + } else if (Array.isArray(data.content)) { + for (const item of data.content) { + if (item.type === 'text' && item.text) { + lastTextContent = item.text; + } + } + } + } else if (data.type === 'result' && data.result) { + lastTextContent = data.result; + } + } + }; + for await (const chunk of execCommand.stream()) { if (chunk.type === 'stdout') { const output = chunk.data.toString(); @@ -343,44 +386,8 @@ export const executeOpenCodeCommand = async params => { lastMessage = output; allOutput += output; - // Issue #1263: Try to parse JSON output to extract text content for result summary - try { - const lines = output.split('\n'); - for (const line of lines) { - if (!line.trim()) continue; - const data = sanitizeObjectStrings(JSON.parse(line)); - // Issue #1968: a bare `null`/primitive NDJSON line must not abort the - // rest of the chunk (data.type access would throw on null). - if (data === null || typeof data !== 'object') continue; - accumulateAgentStepFinishUsage(streamingTokenUsage, data); - // Track text content for result summary - // OpenCode outputs text via 'text', 'assistant', 'message', or 'result' type events - if (data.type === 'text' && data.text) { - lastTextContent = data.text; - } else if (data.type === 'assistant' && data.message?.content) { - const content = Array.isArray(data.message.content) ? data.message.content : [data.message.content]; - for (const item of content) { - if (item.type === 'text' && item.text) { - lastTextContent = item.text; - } - } - } else if (data.type === 'message' && data.content) { - if (typeof data.content === 'string') { - lastTextContent = data.content; - } else if (Array.isArray(data.content)) { - for (const item of data.content) { - if (item.type === 'text' && item.text) { - lastTextContent = item.text; - } - } - } - } else if (data.type === 'result' && data.result) { - lastTextContent = data.result; - } - } - } catch { - // Not JSON, continue - } + // Issue #1263: Parse JSON output to extract text content for result summary + handleOpenCodeRecords(stdoutScanner.write(output)); } if (chunk.type === 'stderr') { @@ -389,47 +396,18 @@ export const executeOpenCodeCommand = async params => { await log(errorOutput, { stream: 'stderr' }); allOutput += errorOutput; - // Issue #1263: Also try to parse stderr for text content - try { - const lines = errorOutput.split('\n'); - for (const line of lines) { - if (!line.trim()) continue; - const data = sanitizeObjectStrings(JSON.parse(line)); - // Issue #1968: skip bare `null`/primitive lines (see stdout handler above). - if (data === null || typeof data !== 'object') continue; - accumulateAgentStepFinishUsage(streamingTokenUsage, data); - if (data.type === 'text' && data.text) { - lastTextContent = data.text; - } else if (data.type === 'assistant' && data.message?.content) { - const content = Array.isArray(data.message.content) ? data.message.content : [data.message.content]; - for (const item of content) { - if (item.type === 'text' && item.text) { - lastTextContent = item.text; - } - } - } else if (data.type === 'message' && data.content) { - if (typeof data.content === 'string') { - lastTextContent = data.content; - } else if (Array.isArray(data.content)) { - for (const item of data.content) { - if (item.type === 'text' && item.text) { - lastTextContent = item.text; - } - } - } - } else if (data.type === 'result' && data.result) { - lastTextContent = data.result; - } - } - } catch { - // Not JSON, continue - } + // Issue #1263: Also parse stderr for text content + handleOpenCodeRecords(stderrScanner.write(errorOutput)); } } else if (chunk.type === 'exit') { exitCode = chunk.code; } } + // Release any record that was still being assembled when the stream ended. + handleOpenCodeRecords(stdoutScanner.flush()); + handleOpenCodeRecords(stderrScanner.flush()); + // Clean up the opencode.json config file to avoid polluting the repository try { await fs.unlink(opencodeConfigPath); diff --git a/src/qwen.lib.mjs b/src/qwen.lib.mjs index af767e05a..bcd3e7166 100644 --- a/src/qwen.lib.mjs +++ b/src/qwen.lib.mjs @@ -18,8 +18,9 @@ import { reportError } from './sentry.lib.mjs'; import { timeouts, retryLimits } from './config.lib.mjs'; import { detectUsageLimit, formatUsageLimitMessage } from './usage-limit.lib.mjs'; import { sanitizeObjectStrings } from './unicode-sanitization.lib.mjs'; -import { qwenModels, defaultModels } from './models/index.mjs'; +import { qwenModels, defaultModels, isFormalAiModel } from './models/index.mjs'; import { logPreparedToolCommand, resolveFormalAiToolInvocation } from './formal-ai.lib.mjs'; +import { buildFormalAiPricingInfo } from './formal-ai-pricing.lib.mjs'; // Issue #2119 import { checkPlaywrightMcpPackageAvailability } from './playwright-mcp.lib.mjs'; import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs'; import { getCumulativeContextInputTokens, getRestoredContextInputTokens, toTokenCount } from './context-fill.lib.mjs'; @@ -249,7 +250,7 @@ const applyQwenUsageToState = (state, event) => { applyQwenUsageObject(state, rawUsage, findFirstValue(event, QWEN_USAGE_PATHS.model)); }; -const buildQwenPricingInfo = (state, mappedModel) => { +export const buildQwenPricingInfo = (state, mappedModel) => { const tokenUsage = cloneQwenTokenUsage(state?.tokenUsage); if (!tokenUsage || tokenUsage.stepCount === 0) { return { @@ -264,6 +265,17 @@ const buildQwenPricingInfo = (state, mappedModel) => { tokenUsage.respondedModelId ||= tokenUsage.requestedModelId; const modelId = tokenUsage.respondedModelId || tokenUsage.requestedModelId; + // Issue #2119: `--model formal-ai` is served by the local Link.Assistant model + // server, so the session belongs to Link.Assistant at $0.00 - not to Qwen Code. + if (isFormalAiModel(mappedModel) || isFormalAiModel(modelId)) { + return { + pricingInfo: { ...buildFormalAiPricingInfo(modelId, tokenUsage), source: 'qwen-stream-json' }, + publicPricingEstimate: 0, + tokenUsage, + resultModelUsage: buildQwenResultModelUsage(tokenUsage), + }; + } + return { pricingInfo: { provider: 'Qwen Code', diff --git a/tests/test-formal-ai-pricing-2119.mjs b/tests/test-formal-ai-pricing-2119.mjs new file mode 100644 index 000000000..a7c64d03c --- /dev/null +++ b/tests/test-formal-ai-pricing-2119.mjs @@ -0,0 +1,137 @@ +#!/usr/bin/env node +/** + * @hive-mind-test-suite default + * + * Regression coverage for issue #2119 - Formal AI provider and cost identity. + * + * `--model formal-ai` routes every request through a local Formal AI model + * server started by `formal-ai with ...`. The requests never reach + * OpenCode Zen, OpenAI, Anthropic, Google or Alibaba, and they are free. + * + * The reproduction PRs reported the opposite: + * - https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2 + * "Provider: OpenCode Zen" and "Public pricing estimate: unknown" + * - https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2 + * "Calculated by Anthropic: $0.252315" + * + * Every pricing producer must funnel through formal-ai-pricing.lib.mjs so the + * same Link.Assistant / $0.00 identity appears in logs and GitHub comments. + */ + +import assert from 'node:assert'; + +import { FORMAL_AI_PROVIDER_NAME, applyFormalAiPricingOverride, buildFormalAiPricingInfo, withFormalAiPricing } from '../src/formal-ai-pricing.lib.mjs'; +import { FORMAL_AI_MODEL_ALIAS, FORMAL_AI_PROVIDER_MODEL_ID } from '../src/models/index.mjs'; +import { calculateAgentPricing } from '../src/agent.lib.mjs'; +import { calculateCodexPricing } from '../src/codex.lib.mjs'; +import { buildGeminiPricingInfo } from '../src/gemini.lib.mjs'; +import { buildQwenPricingInfo } from '../src/qwen.lib.mjs'; +import { summarizeAgentCommanderResult } from '../src/agent-commander.lib.mjs'; +import { buildCostInfoString } from '../src/github-cost-info.lib.mjs'; + +const tokenUsage = { inputTokens: 21677, outputTokens: 22834, cacheReadTokens: 0, cacheWriteTokens: 0, reasoningTokens: 0, stepCount: 1 }; + +const assertFormalAi = (pricingInfo, label) => { + assert.ok(pricingInfo, `${label}: pricing info is present`); + assert.equal(pricingInfo.provider, FORMAL_AI_PROVIDER_NAME, `${label}: provider is Link.Assistant`); + assert.equal(pricingInfo.totalCostUSD, 0, `${label}: cost is $0.00`); + assert.equal(pricingInfo.isFreeModel, true, `${label}: reported as a free model`); +}; + +// --- the shared record ------------------------------------------------------ +assertFormalAi(buildFormalAiPricingInfo(FORMAL_AI_MODEL_ALIAS, tokenUsage), 'buildFormalAiPricingInfo'); +assert.equal(buildFormalAiPricingInfo(FORMAL_AI_MODEL_ALIAS, tokenUsage).tokenUsage, tokenUsage, 'token usage stays reportable'); +assert.equal(buildFormalAiPricingInfo(FORMAL_AI_MODEL_ALIAS).modelName, FORMAL_AI_MODEL_ALIAS, 'model name is the formal-ai alias'); + +// --- per-tool pricing calculators ------------------------------------------ +assertFormalAi(await calculateAgentPricing(FORMAL_AI_MODEL_ALIAS, tokenUsage), 'agent alias'); +assertFormalAi(await calculateAgentPricing(FORMAL_AI_PROVIDER_MODEL_ID, tokenUsage), 'agent provider id'); +assertFormalAi(await calculateCodexPricing(FORMAL_AI_MODEL_ALIAS, tokenUsage), 'codex'); +assertFormalAi(buildGeminiPricingInfo(FORMAL_AI_MODEL_ALIAS), 'gemini'); +assertFormalAi(buildQwenPricingInfo({ tokenUsage }, FORMAL_AI_MODEL_ALIAS).pricingInfo, 'qwen'); +assert.equal(buildQwenPricingInfo({ tokenUsage }, FORMAL_AI_MODEL_ALIAS).publicPricingEstimate, 0, 'qwen: public estimate is $0.00'); + +// A non-formal-ai model must keep its own provider identity. +assert.notEqual(buildGeminiPricingInfo('gemini-2.5-pro').provider, FORMAL_AI_PROVIDER_NAME, 'gemini keeps Google for its own models'); + +// --- withFormalAiPricing wrapper ------------------------------------------- +let delegated = 0; +const wrapped = withFormalAiPricing(async () => { + delegated += 1; + return { provider: 'OpenCode Zen', totalCostUSD: 0.252315 }; +}); +assertFormalAi(await wrapped(FORMAL_AI_MODEL_ALIAS, tokenUsage), 'withFormalAiPricing'); +assert.equal(delegated, 0, 'formal-ai short-circuits before the tool calculator runs'); +assert.equal((await wrapped('grok-code', tokenUsage)).provider, 'OpenCode Zen', 'other models still reach the tool calculator'); +assert.equal(delegated, 1, 'the tool calculator ran exactly once'); + +// --- result override -------------------------------------------------------- +const overridden = applyFormalAiPricingOverride({ + model: FORMAL_AI_MODEL_ALIAS, + pricingInfo: { provider: 'OpenCode Zen', modelName: 'grok-code', totalCostUSD: 0.5, opencodeCost: 0.5, isOpencodeFreeModel: false, tokenUsage }, + publicPricingEstimate: 0.5, + anthropicTotalCostUSD: 0.252315, +}); +assertFormalAi(overridden.pricingInfo, 'applyFormalAiPricingOverride'); +assert.equal(overridden.publicPricingEstimate, 0, 'public estimate is $0.00'); +assert.equal(overridden.anthropicTotalCostUSD, null, 'the Anthropic cost false positive is dropped'); +assert.equal(overridden.pricingInfo.opencodeCost, undefined, 'the OpenCode Zen cost false positive is dropped'); +assert.equal(overridden.pricingInfo.isOpencodeFreeModel, undefined, 'the OpenCode Zen free-model flag is dropped'); + +const untouched = applyFormalAiPricingOverride({ + model: 'sonnet', + pricingInfo: { provider: 'Anthropic', totalCostUSD: 0.5 }, + publicPricingEstimate: 0.5, + anthropicTotalCostUSD: 0.252315, +}); +assert.equal(untouched.anthropicTotalCostUSD, 0.252315, 'non-formal-ai models keep their Anthropic cost'); +assert.equal(untouched.pricingInfo.provider, 'Anthropic', 'non-formal-ai models keep their provider'); + +// --- agent-commander summaries --------------------------------------------- +const commanderSummary = summarizeAgentCommanderResult({ + tool: 'claude', + model: FORMAL_AI_MODEL_ALIAS, + result: { + exitCode: 0, + output: { plain: 'done' }, + usage: { inputTokens: 21677, outputTokens: 22834 }, + metadata: { + success: true, + anthropicTotalCostUSD: 0.252315, + publicPricingEstimate: 0.252315, + pricingInfo: { provider: 'Anthropic', modelName: 'claude-sonnet-4-5', totalCostUSD: 0.252315 }, + streamTokenUsage: { inputTokens: 21677, outputTokens: 22834 }, + }, + }, +}); +assertFormalAi(commanderSummary.pricingInfo, 'agent-commander metadata path'); +assert.equal(commanderSummary.anthropicTotalCostUSD, null, 'agent-commander drops the Anthropic cost for formal-ai'); +assert.equal(commanderSummary.publicPricingEstimate, 0, 'agent-commander reports a $0.00 estimate for formal-ai'); + +const commanderClaudeSummary = summarizeAgentCommanderResult({ + tool: 'claude', + model: 'sonnet', + result: { + exitCode: 0, + output: { plain: '', parsed: [{ type: 'result', result: 'done', total_cost_usd: 0.25 }] }, + }, +}); +assert.equal(commanderClaudeSummary.anthropicTotalCostUSD, 0.25, 'agent-commander keeps the Anthropic cost for a real Anthropic model'); + +// --- GitHub comment rendering ---------------------------------------------- +const rendered = buildCostInfoString(0, null, buildFormalAiPricingInfo(FORMAL_AI_MODEL_ALIAS, tokenUsage)); +assert.ok(rendered.includes(`- Provider: ${FORMAL_AI_PROVIDER_NAME}`), 'comment reports the Link.Assistant provider'); +assert.ok(rendered.includes('- Public pricing estimate: $0.00 (Free model)'), 'comment reports a $0.00 estimate'); +assert.ok(!rendered.includes('Calculated by Anthropic'), 'comment has no Anthropic cost line'); +assert.ok(!rendered.includes('Calculated by OpenCode Zen'), 'comment has no OpenCode Zen cost line'); +assert.ok(rendered.includes('21,677 input, 22,834 output'), 'comment still reports token usage'); + +// A free model with no usage-derived estimate must not render "unknown". +const renderedWithoutEstimate = buildCostInfoString(null, null, { modelName: FORMAL_AI_MODEL_ALIAS, provider: FORMAL_AI_PROVIDER_NAME, isFreeModel: true }); +assert.ok(renderedWithoutEstimate.includes('- Public pricing estimate: $0.00 (Free model)'), 'a free model never renders "unknown"'); + +// A paid model with no estimate keeps reporting "unknown". +const renderedPaid = buildCostInfoString(null, null, { modelName: 'grok-code', provider: 'OpenCode Zen' }); +assert.ok(renderedPaid.includes('- Public pricing estimate: unknown'), 'a paid model without an estimate still reports "unknown"'); + +console.log('PASS: issue #2119 formal-ai pricing reports Link.Assistant at $0.00 everywhere'); From f8d4c91c63b5f45ce6305f5d8a821e417a152fa3 Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 16:03:19 +0000 Subject: [PATCH 06/17] fix(2119): one auto-restart budget with N/M labels and a hard stop Hive Mind had two independent auto-restart subsystems, each with its own counter reading the same --auto-restart-max-iterations flag. solve.mjs runs both in one process, so a limit of 5 permitted 10 AI sessions, and the reproduction PR shows the two incompatible labels the issue calls out: "Auto-restart triggered (iteration 1)" next to "Auto-restart 1/5 Log". Neither path failed at the limit, so a run that never resolved its blocker still exited 0 with the uncommitted work discarded along with the temporary clone. - src/auto-restart-budget.lib.mjs: one process-wide iteration counter both loops claim from, and one N/M label formatter (N alone when the limit is 0/unlimited). - src/auto-restart-exhaustion.lib.mjs: the single exhaustion path - log, auto- commit and push the uncommitted work via the existing critical-error recovery helper, post one "limit reached" comment, and record the failure. - solve.finalize.lib.mjs exits 1 when the budget was exhausted, so the result is actually visible instead of being reported as a success. Reproduction: https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2 --- src/auto-restart-budget.lib.mjs | 107 ++++++++++++++++ src/auto-restart-exhaustion.lib.mjs | 122 +++++++++++++++++++ src/solve.auto-merge.lib.mjs | 85 ++++++------- src/solve.finalize.lib.mjs | 15 +++ src/solve.watch.lib.mjs | 65 +++++++--- tests/test-auto-restart-budget-2119.mjs | 155 ++++++++++++++++++++++++ 6 files changed, 489 insertions(+), 60 deletions(-) create mode 100644 src/auto-restart-budget.lib.mjs create mode 100644 src/auto-restart-exhaustion.lib.mjs create mode 100644 tests/test-auto-restart-budget-2119.mjs diff --git a/src/auto-restart-budget.lib.mjs b/src/auto-restart-budget.lib.mjs new file mode 100644 index 000000000..d748c4b3c --- /dev/null +++ b/src/auto-restart-budget.lib.mjs @@ -0,0 +1,107 @@ +#!/usr/bin/env node + +/** + * Issue #2119: one auto-restart budget for the whole run. + * + * The problem + * ----------- + * Hive Mind had two independent auto-restart subsystems, each with its own + * counter reading the same `--auto-restart-max-iterations` flag: + * + * - `solve.watch.lib.mjs` restarts on uncommitted changes / feedback and + * labels its sessions `๐Ÿ”„ Auto-restart 1/5`; + * - `solve.auto-merge.lib.mjs` restarts until the PR is mergeable and labels + * its sessions `๐Ÿ”„ Auto-restart triggered (iteration 1)`. + * + * `solve.mjs` runs them one after another, so a limit of 5 allowed up to 10 AI + * sessions, and the two label formats made the published comments look like two + * unrelated features. In issue #2119 a `--model formal-ai` run that produced + * only a `.formal-ai/` scratch directory kept restarting on those "uncommitted + * changes" without ever reaching a visible failure. + * + * The fix + * ------- + * A single process-wide budget shared by both subsystems: + * + * - every AI session started by ANY auto-restart path consumes one iteration; + * - every label renders as `N/M` (or `N` when the limit is disabled with 0), + * so the limit is always visible; + * - once the budget is exhausted the run must actually fail, and the caller + * must run fail recovery (auto-commit of whatever is uncommitted) so the + * result stays visible instead of being silently discarded. + * + * The counter is a module-level singleton for the same reason + * `anthropic-cost-accumulator.lib.mjs` is: the two subsystems are separate + * modules invoked sequentially from `solve.mjs` and never see each other's + * state, and one `solve` process handles exactly one logical run. + */ + +import { DEFAULT_AUTO_ITERATION_LIMIT, formatAutoIterationLimit, hasReachedAutoIterationLimit, normalizeAutoIterationLimit } from './auto-iteration-limits.lib.mjs'; + +// Iterations consumed so far by every auto-restart subsystem in this run. +let iterationsUsed = 0; +// The active limit; 0 means "no limit" (`--auto-restart-max-iterations 0`). +let maxIterations = DEFAULT_AUTO_ITERATION_LIMIT; + +/** + * Start the shared budget for one `solve` run. + * + * Safe to call from every entry point: it is idempotent for the same limit, so + * the watch loop and the auto-merge loop can both claim the budget without the + * second one resetting the first one's progress. + * + * @param {Object} [options] + * @param {number|string|null} [options.maxIterations] raw `--auto-restart-max-iterations` value + * @param {boolean} [options.reset=false] force the counter back to zero (new logical run / tests) + * @returns {number} the normalized limit in effect + */ +export const beginAutoRestartBudget = ({ maxIterations: rawMax, reset = false } = {}) => { + const normalized = normalizeAutoIterationLimit(rawMax); + if (reset || normalized !== maxIterations) { + if (reset) iterationsUsed = 0; + maxIterations = normalized; + } + return maxIterations; +}; + +/** @returns {number} the normalized limit in effect (0 = unlimited) */ +export const getAutoRestartLimit = () => maxIterations; + +/** @returns {number} how many AI sessions auto-restart has already consumed */ +export const getAutoRestartIterationsUsed = () => iterationsUsed; + +/** @returns {number|null} iterations still available, or null when unlimited */ +export const getRemainingAutoRestartIterations = () => (maxIterations === 0 ? null : Math.max(0, maxIterations - iterationsUsed)); + +/** + * @returns {boolean} true when no further auto-restart session may be started. + * Always false when the limit is disabled (0). + */ +export const hasExhaustedAutoRestartBudget = () => hasReachedAutoIterationLimit(iterationsUsed, maxIterations); + +/** + * Claim one iteration for an AI session that is about to start. + * Call this only when a tool execution really follows, so the published `N/M` + * label matches the number of sessions that actually ran. + * @returns {number} the 1-based iteration number just claimed + */ +export const consumeAutoRestartIteration = () => { + iterationsUsed += 1; + return iterationsUsed; +}; + +/** + * Render the shared `N/M` progress label used by every auto-restart message. + * @param {number} [iteration] the iteration to render; defaults to the current count + * @returns {string} e.g. `3/5`, or `3` when the limit is disabled + */ +export const formatAutoRestartLabel = (iteration = iterationsUsed) => (maxIterations === 0 ? `${iteration}` : `${iteration}/${maxIterations}`); + +/** @returns {string} the configured limit for display (`5` or `unlimited`) */ +export const formatAutoRestartLimit = () => formatAutoIterationLimit(maxIterations); + +/** Reset the budget. Intended for tests; production code calls `beginAutoRestartBudget`. */ +export const resetAutoRestartBudget = () => { + iterationsUsed = 0; + maxIterations = DEFAULT_AUTO_ITERATION_LIMIT; +}; diff --git a/src/auto-restart-exhaustion.lib.mjs b/src/auto-restart-exhaustion.lib.mjs new file mode 100644 index 000000000..7658ae931 --- /dev/null +++ b/src/auto-restart-exhaustion.lib.mjs @@ -0,0 +1,122 @@ +#!/usr/bin/env node + +/** + * Issue #2119: what happens when the shared auto-restart budget runs out. + * + * The issue requires that after the configured number of iterations the run + * "must actually stop (fail + auto-commit on fail recovery). So the result will + * be actually visible." + * + * Before this, the two auto-restart subsystems ended differently and neither + * preserved the work: + * + * - `solve.watch.lib.mjs` logged "MAX ITERATIONS REACHED" and simply `break`ed + * out of the loop, leaving the uncommitted changes that caused every restart + * on the disposable temporary clone, where they were deleted with it; + * - `solve.auto-merge.lib.mjs` posted a comment and returned + * `auto_restart_limit_reached`, also without committing anything. + * + * This module is the one exhaustion path for both: log the failure, auto-commit + * (and push) whatever is uncommitted through the same critical-error recovery + * helper used elsewhere, and post a single comment that states the limit, the + * remaining blocker and what was preserved. + */ + +import { commitUncommittedChangesOnCriticalError } from './critical-error-commit.lib.mjs'; +import { formatAutoRestartLabel, formatAutoRestartLimit, getAutoRestartIterationsUsed } from './auto-restart-budget.lib.mjs'; +import { AUTO_RESTART_MARKER, postTrackedComment } from './tool-comments.lib.mjs'; +import { reportError } from './sentry.lib.mjs'; + +/** + * The single reason string returned by every auto-restart subsystem when the + * shared budget is exhausted, so `solve` can treat both the same way. + */ +export const AUTO_RESTART_LIMIT_REACHED_REASON = 'auto_restart_limit_reached'; + +// Module-level singleton, like the shared budget itself: `solve.mjs` runs both +// auto-restart loops sequentially and neither returns through a common result +// object, so this is what lets `finalizeSolveProcess` exit non-zero. Without it +// the run reported success even though the blocker was never resolved. +let limitFailure = null; + +/** @returns {boolean} true once any auto-restart loop exhausted the shared budget */ +export const hasAutoRestartLimitFailure = () => Boolean(limitFailure); + +/** @returns {{reason: string, iterationsUsed: number, committed: boolean, pushed: boolean}|null} */ +export const getAutoRestartLimitFailure = () => limitFailure; + +/** Clear the recorded failure. Intended for tests. */ +export const resetAutoRestartLimitFailure = () => { + limitFailure = null; +}; + +/** + * Fail the run because the shared auto-restart budget is exhausted, preserving + * any uncommitted work first. + * + * Never throws: a failure to commit or comment must not mask the limit itself. + * + * @param {Object} params + * @param {string} params.owner GitHub owner + * @param {string} params.repo GitHub repository + * @param {number|null} params.prNumber PR to comment on (comment skipped when absent) + * @param {string} params.tempDir working tree holding the uncommitted work + * @param {string|null} params.branchName branch to push the preserved work to + * @param {Function} params.$ command-stream tagged-template executor + * @param {Function} params.log async logger + * @param {Function} params.formatAligned aligned log formatter + * @param {string} params.blocker the remaining reason that kept triggering restarts + * @param {string} [params.subsystem] which loop hit the limit, for the log line + * @returns {Promise<{reason: string, iterationsUsed: number, committed: boolean, pushed: boolean}>} + */ +export const failOnAutoRestartBudgetExhausted = async ({ owner, repo, prNumber, tempDir, branchName, $, log, formatAligned, blocker = 'uncommitted changes', subsystem = 'auto-restart' }) => { + const iterationsUsed = getAutoRestartIterationsUsed(); + const label = formatAutoRestartLabel(iterationsUsed); + + await log(''); + await log(formatAligned('โŒ', 'AUTO-RESTART LIMIT REACHED', `Stopping ${subsystem} after ${label} iterations`), { level: 'error' }); + await log(formatAligned('', 'Configured limit:', formatAutoRestartLimit(), 2), { level: 'error' }); + await log(formatAligned('', 'Remaining blocker:', blocker, 2), { level: 'error' }); + await log(''); + + // Fail recovery: the work that kept triggering restarts lives in a temporary + // clone that is about to be discarded. Commit and push it so the result is + // visible in the PR instead of vanishing with the clone. + const preserved = await commitUncommittedChangesOnCriticalError({ + tempDir, + branchName, + $, + log, + reason: `auto-restart limit ${label} reached`, + push: true, + }); + + if (prNumber) { + const preservedText = preserved.committed ? `The uncommitted changes were auto-committed${preserved.pushed ? ' and pushed' : ' locally (push failed - see the log)'} so the partial result stays visible in this pull request.` : 'There were no uncommitted changes left to preserve.'; + const body = `## โŒ ${AUTO_RESTART_MARKER} ${label} - limit reached + +Hive Mind stopped after ${label} automatic restart iterations without resolving the blocker. + +**Configured limit:** ${formatAutoRestartLimit()} +**Remaining blocker:** ${blocker} + +${preservedText} + +No further AI sessions will be started automatically for this run. Review the remaining blocker manually, or rerun with a higher \`--auto-restart-max-iterations\` value. + +--- +*This run is reported as failed because the auto-restart limit was reached.*`; + try { + await postTrackedComment({ $, owner, repo, targetNumber: prNumber, body }); + await log(formatAligned('', '๐Ÿ’ฌ Posted auto-restart limit notification to PR', '', 2)); + } catch (commentError) { + reportError(commentError, { context: 'post_auto_restart_limit_comment', owner, repo, prNumber, operation: 'comment_on_pr' }); + await log(formatAligned('', 'โš ๏ธ Could not post auto-restart limit comment to PR', '', 2)); + } + } + + limitFailure = { reason: AUTO_RESTART_LIMIT_REACHED_REASON, iterationsUsed, committed: preserved.committed, pushed: preserved.pushed }; + return limitFailure; +}; + +export default { AUTO_RESTART_LIMIT_REACHED_REASON, failOnAutoRestartBudgetExhausted, hasAutoRestartLimitFailure, getAutoRestartLimitFailure, resetAutoRestartLimitFailure }; diff --git a/src/solve.auto-merge.lib.mjs b/src/solve.auto-merge.lib.mjs index e3fd345a1..47e7ecbe5 100644 --- a/src/solve.auto-merge.lib.mjs +++ b/src/solve.auto-merge.lib.mjs @@ -67,7 +67,7 @@ const { buildCancelledCIReviewComment, getRetriggerableWorkflowRuns, shouldStopF // Issue #1625: Shared marker constants + posting/tracking helpers const toolComments = await import('./tool-comments.lib.mjs'); -const { READY_TO_MERGE_MARKER, READY_FOR_REVIEW_MARKER, AUTO_RESTART_MARKER, AUTO_MERGED_MARKER, postTrackedComment } = toolComments; +const { READY_TO_MERGE_MARKER, READY_FOR_REVIEW_MARKER, AUTO_RESTART_MARKER, AUTO_RESTART_UNTIL_MERGEABLE_LOG_MARKER, AUTO_MERGED_MARKER, postTrackedComment } = toolComments; const externalReviewLimitLib = await import('./external-review-limit.lib.mjs'); const { buildReadyForReviewComment } = externalReviewLimitLib; @@ -81,6 +81,13 @@ const { maybeAttachWorkingSessionSummary, ensurePullRequestIssueLink } = results // Issue #1574: Interruptible sleep so CTRL+C is never blocked by a lingering timer const { interruptibleSleep } = await import('./interruptible-sleep.lib.mjs'); const { formatAutoIterationLimit, hasReachedAutoIterationLimit, normalizeAutoIterationLimit, shouldSyncBeforeRestart } = await import('./auto-iteration-limits.lib.mjs'); +// Issue #2119: one auto-restart budget shared with solve.watch.lib.mjs. solve.mjs +// runs both loops in the same process, so before this a limit of 5 allowed 10 AI +// sessions and the two loops published incompatible progress labels +// ("Auto-restart triggered (iteration 1)" vs "Auto-restart 1/5 Log"). +const autoRestartBudget = await import('./auto-restart-budget.lib.mjs'); +const { beginAutoRestartBudget, consumeAutoRestartIteration, formatAutoRestartLabel, formatAutoRestartLimit, hasExhaustedAutoRestartBudget } = autoRestartBudget; +const { failOnAutoRestartBudgetExhausted } = await import('./auto-restart-exhaustion.lib.mjs'); const { ensurePullRequestBaseBranch } = await import('./solve.pr-base-guard.lib.mjs'); // Issue #1895: explicitly close linked issues after merging a PR into a @@ -101,7 +108,8 @@ export const watchUntilMergeable = async params => { const MIN_CI_CHECK_INTERVAL_SECONDS = 120; const watchInterval = Math.max(rawWatchInterval, MIN_CI_CHECK_INTERVAL_SECONDS); const isAutoMerge = argv.autoMerge || false; - const maxAutoRestartIterations = normalizeAutoIterationLimit(argv.autoRestartMaxIterations); + // Issue #2119: join the shared budget instead of starting a second counter. + const maxAutoRestartIterations = beginAutoRestartBudget({ maxIterations: argv.autoRestartMaxIterations }); const maxAutoResumeIterations = normalizeAutoIterationLimit(argv.autoResumeMaxIterations); // Issue #1503/#1573/#1612: repo-wide action gating is opt-in strict mode. // The config default may be bypassed when this module is reused directly, so normalize here. @@ -112,7 +120,8 @@ export const watchUntilMergeable = async params => { let latestAnthropicCost = null; // Issue #1323: Track actual AI restarts separately from check cycle iterations - let restartCount = 0; + // Issue #2119: the count now lives in the shared budget module, so restarts + // already spent by the watch loop earlier in this run are counted here too. let limitResumeCount = 0; // Issue #1371: In-memory dedup for "Ready to merge" comment (per-session, not all-time) @@ -133,7 +142,7 @@ export const watchUntilMergeable = async params => { await log(formatAligned('', 'Mode:', isAutoMerge ? 'Auto-merge (will merge when ready)' : 'Auto-restart-until-mergeable (will NOT auto-merge)', 2)); await log(formatAligned('', 'Checking interval:', `${watchInterval} seconds (minimum: ${MIN_CI_CHECK_INTERVAL_SECONDS}s)`, 2)); await log(formatAligned('', 'Initial cooldown:', `${INITIAL_COOLDOWN_SECONDS} seconds`, 2)); - await log(formatAligned('', 'Max restart iterations:', formatAutoIterationLimit(maxAutoRestartIterations), 2)); + await log(formatAligned('', 'Max restart iterations:', formatAutoRestartLimit(), 2)); await log(formatAligned('', 'Max limit resumes:', formatAutoIterationLimit(maxAutoResumeIterations), 2)); await log(formatAligned('', 'Wait for all repo actions:', waitForAllRepoActionsFlag ? 'Yes (strict repo-wide safety)' : 'No (PR-scoped CI only)', 2)); await log(formatAligned('', 'Stop conditions:', 'PR merged, PR closed, or becomes mergeable', 2)); @@ -691,38 +700,24 @@ Once the billing issue is resolved, you can re-run the CI checks or push a new c } if (shouldRestart) { - if (hasReachedAutoIterationLimit(restartCount, maxAutoRestartIterations)) { - await log(''); - await log(formatAligned('โš ๏ธ', 'AUTO-RESTART LIMIT REACHED', `Stopping after ${restartCount} restart iteration${restartCount !== 1 ? 's' : ''}`)); - await log(formatAligned('', 'Configured limit:', formatAutoIterationLimit(maxAutoRestartIterations), 2)); - await log(formatAligned('', 'Remaining blockers:', restartReason, 2)); - await log(''); - - try { - const limitComment = `## โš ๏ธ Auto-restart limit reached - -Hive Mind stopped auto-restart-until-mergeable after ${restartCount} restart iteration${restartCount !== 1 ? 's' : ''}. - -**Configured limit:** ${formatAutoIterationLimit(maxAutoRestartIterations)} -**Remaining reason:** ${restartReason} - -No further AI sessions will be started automatically for this run. Please review the remaining blockers manually or rerun with a higher \`--auto-restart-max-iterations\` value. - ---- -*Auto-restart-until-mergeable stopped by the safety limit.*`; - await postTrackedComment({ $, owner, repo, targetNumber: prNumber, body: limitComment }); - } catch (commentError) { - reportError(commentError, { - context: 'post_auto_restart_limit_comment', - owner, - repo, - prNumber, - operation: 'comment_on_pr', - }); - await log(formatAligned('', 'โš ๏ธ Could not post auto-restart limit comment to PR', '', 2)); - } - - return { success: false, reason: 'auto_restart_limit_reached', latestSessionId, latestAnthropicCost }; + // Issue #2119: the run-wide budget is exhausted (it may already have been + // spent by the watch loop). Fail and auto-commit through the same shared + // exhaustion path the uncommitted-changes loop uses, so the outcome and + // the published comment are identical no matter which loop hit the limit. + if (hasExhaustedAutoRestartBudget()) { + const exhaustion = await failOnAutoRestartBudgetExhausted({ + owner, + repo, + prNumber, + tempDir, + branchName: prBranch || branchName, + $, + log, + formatAligned, + blocker: restartReason, + subsystem: 'auto-restart-until-mergeable', + }); + return { success: false, reason: exhaustion.reason, latestSessionId, latestAnthropicCost }; } // Add standard instructions for auto-restart-until-mergeable mode using shared utility @@ -759,17 +754,21 @@ No further AI sessions will be started automatically for this run. Please review } // Issue #1323: Increment restart count only when a tool execution is about to start. - restartCount++; + // Issue #2119: claim it from the run-wide shared budget. + const restartCount = consumeAutoRestartIteration(); await log(formatAligned('๐Ÿ”„', 'RESTART TRIGGERED:', restartReason)); - await log(formatAligned('', 'Restart iteration:', maxAutoRestartIterations === 0 ? `${restartCount}` : `${restartCount}/${maxAutoRestartIterations}`, 2)); + await log(formatAligned('', 'Restart iteration:', formatAutoRestartLabel(restartCount), 2)); await log(''); // Post a comment to PR about the restart after preflight succeeds, so every // posted restart notification corresponds to an actual tool session. try { - const limitText = maxAutoRestartIterations === 0 ? 'No automatic restart limit is configured.' : `This run will stop after ${maxAutoRestartIterations} restart iteration${maxAutoRestartIterations !== 1 ? 's' : ''}.`; - const commentBody = `## ๐Ÿ”„ ${AUTO_RESTART_MARKER} triggered (iteration ${restartCount})\n\n**Reason:** ${restartReason}\n\nStarting new session to address the issues.\n\n---\n*Auto-restart-until-mergeable mode is active. ${limitText}*`; + const limitText = maxAutoRestartIterations === 0 ? 'No automatic restart limit is configured.' : `This run will stop after ${maxAutoRestartIterations} restart iteration${maxAutoRestartIterations !== 1 ? 's' : ''} in total.`; + // Issue #2119: the same `N/M` heading the uncommitted-changes loop posts. + // "triggered (iteration N)" hid the limit and made one auto-restart + // system look like two. + const commentBody = `## ๐Ÿ”„ ${AUTO_RESTART_MARKER} ${formatAutoRestartLabel(restartCount)}\n\n**Reason:** ${restartReason}\n\nStarting new session to address the issues.\n\n---\n*Auto-restart-until-mergeable mode is active. ${limitText}*`; // Issue #1625: Track so this doesn't falsely count as an AI-authored comment await postTrackedComment({ $, owner, repo, targetNumber: prNumber, body: commentBody }); await log(formatAligned('', '๐Ÿ’ฌ Posted auto-restart notification to PR', '', 2)); @@ -1093,8 +1092,10 @@ No further AI sessions will be started automatically for this run. Please review try { const logFile = getLogFile(); if (logFile) { - // Issue #1323: Use restartCount (actual AI executions) instead of iteration (check cycles) - const customTitle = `๐Ÿ”„ Auto-restart-until-mergeable Log (iteration ${restartCount})`; + // Issue #1323: Use the restart count (actual AI executions) instead of iteration (check cycles) + // Issue #2119: `N/M` like every other auto-restart label, so the + // limit is visible in the log title too. + const customTitle = `๐Ÿ”„ ${AUTO_RESTART_UNTIL_MERGEABLE_LOG_MARKER} ${formatAutoRestartLabel()}`; await attachLogToGitHub({ logFile, targetType: 'pr', diff --git a/src/solve.finalize.lib.mjs b/src/solve.finalize.lib.mjs index dffe8cc77..285f848b7 100644 --- a/src/solve.finalize.lib.mjs +++ b/src/solve.finalize.lib.mjs @@ -1,3 +1,9 @@ +// Issue #2119: "after 5 we must actually stop (fail + auto-commit on fail +// recovery). So the result will be actually visible." Both auto-restart loops +// record their exhaustion in this shared module, so the run exits non-zero +// instead of reporting success with the blocker still unresolved. +import { getAutoRestartLimitFailure, hasAutoRestartLimitFailure } from './auto-restart-exhaustion.lib.mjs'; + export async function finalizeSolveProcess({ tempDir, argv, limitReached, path, getLogFile, log, closeSentry, logActiveHandles, cleanupTempDirectory, safeExit }) { await cleanupTempDirectory(tempDir, argv, limitReached); @@ -16,6 +22,15 @@ export async function finalizeSolveProcess({ tempDir, argv, limitReached, path, // drainHandles() inside safeExit() will unref/close these before process.exit(). await logActiveHandles(msg => log(msg)); + // Issue #2119: an exhausted auto-restart budget is a failure, not a completed run. + if (hasAutoRestartLimitFailure()) { + const failure = getAutoRestartLimitFailure(); + await log(`\nโŒ Auto-restart limit reached after ${failure.iterationsUsed} iteration${failure.iterationsUsed !== 1 ? 's' : ''} - the blocker was never resolved.`, { level: 'error' }); + await log(failure.committed ? ' Uncommitted work was auto-committed before exit, so the partial result is visible.' : ' No uncommitted work was left to preserve.', { level: 'error' }); + await safeExit(1, 'Auto-restart limit reached'); + return; + } + // Issue #1431: safeExit() unrefs handles so the event loop exits naturally, then calls process.exit(0) await safeExit(0, 'Process completed'); } diff --git a/src/solve.watch.lib.mjs b/src/solve.watch.lib.mjs index 6e4867895..e1a9f9467 100644 --- a/src/solve.watch.lib.mjs +++ b/src/solve.watch.lib.mjs @@ -46,7 +46,12 @@ const { checkGitHubTerminalState } = terminalStateLib; // Issue #1574: Interruptible sleep so CTRL+C is never blocked by a lingering timer const { interruptibleSleep } = await import('./interruptible-sleep.lib.mjs'); -const { formatAutoIterationLimit, hasReachedAutoIterationLimit, normalizeAutoIterationLimit } = await import('./auto-iteration-limits.lib.mjs'); +// Issue #2119: one auto-restart budget shared with solve.auto-merge.lib.mjs, so +// a limit of 5 means 5 AI sessions in total rather than 5 per subsystem, and +// every label renders in the same `N/M` form. +const autoRestartBudget = await import('./auto-restart-budget.lib.mjs'); +const { beginAutoRestartBudget, consumeAutoRestartIteration, formatAutoRestartLabel, formatAutoRestartLimit, getAutoRestartIterationsUsed, getRemainingAutoRestartIterations, hasExhaustedAutoRestartBudget } = autoRestartBudget; +const { failOnAutoRestartBudgetExhausted } = await import('./auto-restart-exhaustion.lib.mjs'); // Issue #1625: Central marker constants + tracked comment posting const toolComments = await import('./tool-comments.lib.mjs'); @@ -78,7 +83,9 @@ export const watchForFeedback = async params => { const watchInterval = argv.watchInterval || 60; // seconds const isTemporaryWatch = argv.temporaryWatch || false; - const maxAutoRestartIterations = normalizeAutoIterationLimit(argv.autoRestartMaxIterations); + // Issue #2119: claim the shared budget; the same limit is honoured by the + // auto-merge restart loop that solve.mjs runs afterwards. + const maxAutoRestartIterations = beginAutoRestartBudget({ maxIterations: argv.autoRestartMaxIterations }); // Track latest session data across all iterations for accurate pricing // Issue #1056: Seed from the *initial* tool execution so the first auto-restart @@ -105,7 +112,7 @@ export const watchForFeedback = async params => { await log(formatAligned('', 'Monitoring PR:', `#${prNumber}`, 2)); await log(formatAligned('', 'Mode:', 'Auto-restart (NOT --watch mode)', 2)); await log(formatAligned('', 'Stop conditions:', 'All changes committed OR PR merged OR max iterations reached', 2)); - await log(formatAligned('', 'Max iterations:', formatAutoIterationLimit(maxAutoRestartIterations), 2)); + await log(formatAligned('', 'Max iterations:', formatAutoRestartLimit(), 2)); await log(formatAligned('', 'Note:', 'No wait time between iterations in auto-restart mode', 2)); } else { await log(formatAligned('๐Ÿ‘๏ธ', 'WATCH MODE ACTIVATED', '')); @@ -118,8 +125,12 @@ export const watchForFeedback = async params => { await log(''); let iteration = 0; - let autoRestartCount = 0; + // Issue #2119: mirrors the shared budget counter so every label in this loop + // reports the run-wide iteration number, not a per-subsystem one. + let autoRestartCount = getAutoRestartIterationsUsed(); let firstIterationInTemporaryMode = isTemporaryWatch; + // Issue #2119: set when the budget runs out, so the caller learns the run failed. + let budgetExhaustion = null; while (true) { iteration++; @@ -211,13 +222,25 @@ export const watchForFeedback = async params => { break; } - // Check if we've reached max iterations - if (hasReachedAutoIterationLimit(autoRestartCount, maxAutoRestartIterations)) { - await log(''); - await log(formatAligned('โš ๏ธ', 'MAX ITERATIONS REACHED', `Exiting auto-restart mode after ${autoRestartCount} iterations`)); - await log(formatAligned('', 'Some uncommitted changes may remain', '', 2)); - await log(formatAligned('', 'Please review and commit manually if needed', '', 2)); - await log(''); + // Issue #2119: the shared budget is exhausted. Previously this logged a + // warning and broke out of the loop, leaving the very uncommitted changes + // that triggered every restart on a temporary clone that is then deleted. + // Now the run fails and the work is auto-committed first, so the result + // stays visible in the PR. + if (hasExhaustedAutoRestartBudget()) { + const changes = await getUncommittedChangesDetails(tempDir); + budgetExhaustion = await failOnAutoRestartBudgetExhausted({ + owner, + repo, + prNumber, + tempDir, + branchName: prBranch || branchName, + $, + log, + formatAligned, + blocker: changes.length > 0 ? `uncommitted changes remained: ${changes.join(', ')}` : 'uncommitted changes remained', + subsystem: 'auto-restart on uncommitted changes', + }); break; } } @@ -274,17 +297,18 @@ export const watchForFeedback = async params => { } await log(''); - // Increment auto-restart counter and log restart number - autoRestartCount++; + // Issue #2119: claim one iteration from the run-wide budget shared with + // the auto-merge restart loop. + autoRestartCount = consumeAutoRestartIteration(); autoRestartIterationsRan = true; // Issue #1290: Mark that auto-restart iterations ran lastIterationLogUploaded = false; // Reset log upload tracking for new iteration - const restartLabel = firstIterationInTemporaryMode ? 'Initial restart' : `Restart ${autoRestartCount}/${maxAutoRestartIterations}`; + const restartLabel = `Restart ${formatAutoRestartLabel(autoRestartCount)}`; await log(formatAligned('๐Ÿ”„', `${restartLabel}:`, `Running ${argv.tool.toUpperCase()} to handle uncommitted changes...`)); // Post a comment to PR about auto-restart if (prNumber) { try { - const remainingIterations = maxAutoRestartIterations === 0 ? null : maxAutoRestartIterations - autoRestartCount; + const remainingIterations = getRemainingAutoRestartIterations(); // Get uncommitted files list for the comment let uncommittedFilesList = ''; @@ -292,7 +316,7 @@ export const watchForFeedback = async params => { uncommittedFilesList = '\n\n**Uncommitted files:**\n```\n' + changes.join('\n') + '\n```'; } - const iterationLabel = maxAutoRestartIterations === 0 ? `${autoRestartCount}` : `${autoRestartCount}/${maxAutoRestartIterations}`; + const iterationLabel = formatAutoRestartLabel(autoRestartCount); const stopText = remainingIterations === null ? 'Auto-restart is configured with no iteration limit.' : `Auto-restart will stop after changes are committed or discarded, or after ${remainingIterations} more iteration${remainingIterations !== 1 ? 's' : ''}.`; const commentBody = `## ๐Ÿ”„ ${AUTO_RESTART_MARKER} ${iterationLabel}\n\nDetected uncommitted changes from previous run. Starting new session to review and commit or discard them.${uncommittedFilesList}\n\n---\n*${stopText} Please wait until working session will end and give your feedback.*`; // Issue #1625: Track so this doesn't falsely count as AI-authored. @@ -471,7 +495,7 @@ export const watchForFeedback = async params => { const logFile = getLogFile(); if (logFile) { // Use "Auto-restart X/Y Failure Log" format to distinguish from success logs - const iterationLabel = maxAutoRestartIterations === 0 ? `${autoRestartCount}` : `${autoRestartCount}/${maxAutoRestartIterations}`; + const iterationLabel = formatAutoRestartLabel(autoRestartCount); const customTitle = `โš ๏ธ Auto-restart ${iterationLabel} Failure Log`; const logUploadSuccess = await attachLogToGitHub({ logFile, @@ -607,7 +631,7 @@ export const watchForFeedback = async params => { const logFile = getLogFile(); if (logFile) { // Use "Auto-restart X/Y Log" format as requested in issue #1107 - const iterationLabel = maxAutoRestartIterations === 0 ? `${autoRestartCount}` : `${autoRestartCount}/${maxAutoRestartIterations}`; + const iterationLabel = formatAutoRestartLabel(autoRestartCount); const customTitle = `๐Ÿ”„ Auto-restart ${iterationLabel} Log`; const logUploadSuccess = await attachLogToGitHub({ logFile, @@ -733,6 +757,11 @@ export const watchForFeedback = async params => { latestAnthropicCost, autoRestartIterationsRan, // True if any auto-restart iterations actually ran lastIterationLogUploaded, // True if the last iteration's logs were uploaded + // Issue #2119: false when the shared auto-restart budget ran out, so the run + // is reported as failed instead of silently exiting with work still pending. + success: !budgetExhaustion, + reason: budgetExhaustion?.reason || null, + autoRestartLimitReached: Boolean(budgetExhaustion), }; }; diff --git a/tests/test-auto-restart-budget-2119.mjs b/tests/test-auto-restart-budget-2119.mjs new file mode 100644 index 000000000..9a4994df2 --- /dev/null +++ b/tests/test-auto-restart-budget-2119.mjs @@ -0,0 +1,155 @@ +#!/usr/bin/env node +/** + * @hive-mind-test-suite default + * + * Regression coverage for issue #2119 - a single auto-restart system. + * + * Reproduction: https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2 + * posted both `๐Ÿ”„ Auto-restart triggered (iteration 1)` and `๐Ÿ”„ Auto-restart 1/5 Log`. + * Those came from two independent subsystems, each with its own counter reading + * the same `--auto-restart-max-iterations` flag, and `solve.mjs` runs both in + * one process - so a limit of 5 permitted 10 AI sessions and the run still + * exited successfully with the blocker unresolved. + * + * The issue requires: one auto-restart system, every label in `N/M` form, a hard + * stop after the limit, and a real failure with auto-commit fail recovery so the + * result stays visible. + */ + +import assert from 'node:assert'; +import { readFile } from 'node:fs/promises'; +import { fileURLToPath } from 'node:url'; +import path from 'node:path'; + +import { beginAutoRestartBudget, consumeAutoRestartIteration, formatAutoRestartLabel, formatAutoRestartLimit, getAutoRestartIterationsUsed, getAutoRestartLimit, getRemainingAutoRestartIterations, hasExhaustedAutoRestartBudget, resetAutoRestartBudget } from '../src/auto-restart-budget.lib.mjs'; +import { AUTO_RESTART_LIMIT_REACHED_REASON, failOnAutoRestartBudgetExhausted, getAutoRestartLimitFailure, hasAutoRestartLimitFailure, resetAutoRestartLimitFailure } from '../src/auto-restart-exhaustion.lib.mjs'; +import { DEFAULT_AUTO_ITERATION_LIMIT } from '../src/auto-iteration-limits.lib.mjs'; + +const srcDir = path.join(path.dirname(fileURLToPath(import.meta.url)), '..', 'src'); +const noopLog = async () => {}; +const formatAligned = (icon, label, value) => `${icon} ${label} ${value}`; + +/** A command-stream lookalike: records commands and reports a dirty tree. */ +const makeFake$ = (statusOutput = '') => { + const calls = []; + const fake = () => async strings => { + const cmd = strings.join(' '); + calls.push(cmd); + if (cmd.includes('git status')) return { code: 0, stdout: statusOutput, stderr: '' }; + if (cmd.includes('gh api')) return { code: 0, stdout: '{"id":1}', stderr: '' }; + return { code: 0, stdout: '', stderr: '' }; + }; + fake.calls = calls; + return fake; +}; + +// --- the budget is shared, not per-subsystem -------------------------------- +resetAutoRestartBudget(); +assert.equal(getAutoRestartLimit(), DEFAULT_AUTO_ITERATION_LIMIT, 'the default limit is 5'); + +// The uncommitted-changes loop claims the budget first... +assert.equal(beginAutoRestartBudget({ maxIterations: 5 }), 5, 'the watch loop normalizes the limit'); +assert.equal(consumeAutoRestartIteration(), 1, 'first AI session is iteration 1'); +assert.equal(consumeAutoRestartIteration(), 2, 'second AI session is iteration 2'); + +// ...then the auto-restart-until-mergeable loop joins with the same flag value. +// Before the fix this restarted counting from 0, allowing 5 more sessions. +assert.equal(beginAutoRestartBudget({ maxIterations: 5 }), 5, 'the auto-merge loop joins the same budget'); +assert.equal(getAutoRestartIterationsUsed(), 2, 'joining must not reset the iterations already spent'); +assert.equal(getRemainingAutoRestartIterations(), 3, '3 of 5 iterations remain across both loops'); +assert.equal(hasExhaustedAutoRestartBudget(), false, 'not exhausted at 2/5'); + +assert.equal(consumeAutoRestartIteration(), 3, 'the auto-merge loop continues the shared count'); +assert.equal(consumeAutoRestartIteration(), 4); +assert.equal(consumeAutoRestartIteration(), 5, 'the 5th session across both loops'); +assert.equal(hasExhaustedAutoRestartBudget(), true, 'stops after the 5th iteration, not the 10th'); +assert.equal(getRemainingAutoRestartIterations(), 0, 'nothing left'); + +// --- every label uses the same N/M form ------------------------------------- +assert.equal(formatAutoRestartLabel(1), '1/5', 'labels render as N/M'); +assert.equal(formatAutoRestartLabel(), '5/5', 'the default label is the current iteration'); +assert.equal(formatAutoRestartLimit(), '5', 'the limit renders as a plain number'); + +// --- 0 disables the limit --------------------------------------------------- +resetAutoRestartBudget(); +assert.equal(beginAutoRestartBudget({ maxIterations: 0 }), 0, '0 means unlimited'); +consumeAutoRestartIteration(); +assert.equal(hasExhaustedAutoRestartBudget(), false, 'an unlimited budget is never exhausted'); +assert.equal(getRemainingAutoRestartIterations(), null, 'no remaining count when unlimited'); +assert.equal(formatAutoRestartLabel(), '1', 'an unlimited label has no denominator'); +assert.equal(formatAutoRestartLimit(), 'unlimited', 'the unlimited limit is spelled out'); + +// --- exhaustion fails the run AND preserves the work ------------------------ +resetAutoRestartBudget(); +resetAutoRestartLimitFailure(); +beginAutoRestartBudget({ maxIterations: 5, reset: true }); +for (let i = 0; i < 5; i++) consumeAutoRestartIteration(); + +const dirty$ = makeFake$(' M examples/hello.scala\n?? .formal-ai/'); +const failure = await failOnAutoRestartBudgetExhausted({ + owner: 'konard', + repo: 'test-hello-world', + prNumber: 2, + tempDir: '/tmp/none', + branchName: 'issue-1-abc', + $: dirty$, + log: noopLog, + formatAligned, + blocker: 'uncommitted changes remained', + subsystem: 'auto-restart on uncommitted changes', +}); + +assert.equal(failure.reason, AUTO_RESTART_LIMIT_REACHED_REASON, 'the run reports the limit as its failure reason'); +assert.equal(failure.iterationsUsed, 5, 'the failure reports the run-wide iteration count'); +assert.equal(failure.committed, true, 'fail recovery auto-commits the uncommitted work'); +assert.equal(failure.pushed, true, 'fail recovery pushes it so the result is visible'); +assert.ok( + dirty$.calls.some(c => c.includes('git commit')), + 'a real commit was made' +); +assert.ok( + dirty$.calls.some(c => c.includes('git push')), + 'the preserved work was pushed' +); +assert.ok( + dirty$.calls.some(c => c.includes('gh api')), + 'a limit-reached comment was posted to the PR' +); + +assert.equal(hasAutoRestartLimitFailure(), true, 'the failure is visible to finalizeSolveProcess'); +assert.equal(getAutoRestartLimitFailure().iterationsUsed, 5, 'the recorded failure carries the iteration count'); + +resetAutoRestartLimitFailure(); +assert.equal(hasAutoRestartLimitFailure(), false, 'a fresh run starts without a recorded failure'); + +// A clean tree still fails, it just has nothing to preserve. +resetAutoRestartLimitFailure(); +const clean$ = makeFake$(''); +const cleanFailure = await failOnAutoRestartBudgetExhausted({ owner: 'o', repo: 'r', prNumber: null, tempDir: '/tmp/none', branchName: 'b', $: clean$, log: noopLog, formatAligned, blocker: 'CI still failing' }); +assert.equal(cleanFailure.reason, AUTO_RESTART_LIMIT_REACHED_REASON, 'still a failure with a clean tree'); +assert.equal(cleanFailure.committed, false, 'nothing to commit on a clean tree'); +assert.ok(!clean$.calls.some(c => c.includes('gh api')), 'no comment is posted without a PR number'); +resetAutoRestartLimitFailure(); +resetAutoRestartBudget(); + +// --- the duplicated counters are gone --------------------------------------- +const watchSource = await readFile(path.join(srcDir, 'solve.watch.lib.mjs'), 'utf8'); +const autoMergeSource = await readFile(path.join(srcDir, 'solve.auto-merge.lib.mjs'), 'utf8'); +const finalizeSource = await readFile(path.join(srcDir, 'solve.finalize.lib.mjs'), 'utf8'); + +assert.ok(!/let\s+autoRestartCount\s*=\s*0/.test(watchSource), 'the watch loop no longer owns a private restart counter'); +assert.ok(!/let\s+restartCount\s*=\s*0/.test(autoMergeSource), 'the auto-merge loop no longer owns a private restart counter'); +assert.ok(watchSource.includes('consumeAutoRestartIteration()'), 'the watch loop claims iterations from the shared budget'); +assert.ok(autoMergeSource.includes('consumeAutoRestartIteration()'), 'the auto-merge loop claims iterations from the shared budget'); + +// The divergent "(iteration N)" heading from the reproduction PR must be gone. +assert.ok(!autoMergeSource.includes('triggered (iteration ${restartCount})'), 'no more "Auto-restart triggered (iteration N)" heading'); +assert.ok(!autoMergeSource.includes('Log (iteration ${restartCount})'), 'no more "(iteration N)" log title'); +assert.ok(!/\$\{autoRestartCount\}\/\$\{maxAutoRestartIterations\}/.test(watchSource), 'labels are built by the shared formatter, not inlined'); + +// Both loops must route exhaustion through the one fail + auto-commit path. +assert.ok(watchSource.includes('failOnAutoRestartBudgetExhausted'), 'the watch loop fails on exhaustion'); +assert.ok(autoMergeSource.includes('failOnAutoRestartBudgetExhausted'), 'the auto-merge loop fails on exhaustion'); +assert.ok(finalizeSource.includes('safeExit(1'), 'the process exits non-zero when the limit was reached'); + +console.log('PASS: issue #2119 one auto-restart budget, N/M labels, hard stop with fail recovery'); From 46b2df22fab53512e3d0e159e4eceb99e91770a2 Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 16:10:50 +0000 Subject: [PATCH 07/17] fix(2119): stop leaking shell quotes into PR titles and API arguments All three reproduction pull requests were titled `'Implement Hello World in Scala'` with the single quotes as literal characters. Root cause: command-stream shell-escapes every interpolated value, quoting it whenever it contains a character the shell would split on. Wrapping the placeholder in quotes - `--title "${updatedTitle}"` - therefore passes `--title "'Implement Hello World in Scala'"` and the quotes land in the title. Values without spaces were unaffected, which is why it went unnoticed. Fixed at every such site in src/ and scripts/ (PR titles and body files, the YouTrack issue sync, the claude runtime switch, reviewers-hive, the changeset and release scripts). Where the quotes belonged to an inner language the expression is now assembled in JS and passed as one argument: the fork-lookup jq filters and the issue/PR-link GraphQL queries. tests/test-shell-quoting-2119.mjs pins the command-stream behaviour that caused this and scans src/ and scripts/ so the pattern cannot come back. --- scripts/create-manual-changeset.mjs | 2 +- scripts/format-github-release.mjs | 2 +- src/claude.runtime-switch.lib.mjs | 8 +-- src/reviewers-hive.mjs | 4 +- src/solve.auto-pr.lib.mjs | 14 ++++- src/solve.repository.lib.mjs | 19 ++++-- src/solve.results.lib.mjs | 6 +- src/youtrack/youtrack-sync.mjs | 4 +- tests/test-shell-quoting-2119.mjs | 97 +++++++++++++++++++++++++++++ 9 files changed, 136 insertions(+), 20 deletions(-) create mode 100644 tests/test-shell-quoting-2119.mjs diff --git a/scripts/create-manual-changeset.mjs b/scripts/create-manual-changeset.mjs index 841f74d90..e2f2d3dbf 100644 --- a/scripts/create-manual-changeset.mjs +++ b/scripts/create-manual-changeset.mjs @@ -71,7 +71,7 @@ ${description} // Format with Prettier console.log('\nFormatting with Prettier...'); - await $`npx prettier --write "${changesetFile}"`; + await $`npx prettier --write ${changesetFile}`; console.log('\nChangeset created and formatted successfully'); } catch (error) { diff --git a/scripts/format-github-release.mjs b/scripts/format-github-release.mjs index d75489f27..a35f72696 100644 --- a/scripts/format-github-release.mjs +++ b/scripts/format-github-release.mjs @@ -69,7 +69,7 @@ try { console.log(`Formatting release notes for ${tag}...`); // Pass the trigger commit SHA for PR detection // This allows proper PR lookup even if the changelog doesn't have a commit hash - await $`node scripts/format-release-notes.mjs --release-id "${releaseId}" --release-version "${tag}" --repository "${repository}" --commit-sha "${commitSha}"`; + await $`node scripts/format-release-notes.mjs --release-id ${releaseId} --release-version ${tag} --repository ${repository} --commit-sha ${commitSha}`; console.log(`Formatted release notes for ${tag}`); } } catch (error) { diff --git a/src/claude.runtime-switch.lib.mjs b/src/claude.runtime-switch.lib.mjs index 4cea60aa8..c8a44936d 100644 --- a/src/claude.runtime-switch.lib.mjs +++ b/src/claude.runtime-switch.lib.mjs @@ -53,7 +53,7 @@ export const handleClaudeRuntimeSwitch = async argv => { process.exit(1); } // Read current shebang - const firstLine = await $`head -1 "${claudePath}"`; + const firstLine = await $`head -1 ${claudePath}`; const currentShebang = firstLine.stdout.toString().trim(); await log(` Current shebang: ${currentShebang}`); if (currentShebang.includes('bun')) { @@ -63,7 +63,7 @@ export const handleClaudeRuntimeSwitch = async argv => { // Create backup const backupPath = `${claudePath}.nodejs-backup`; - await $`cp "${claudePath}" "${backupPath}"`; + await $`cp ${claudePath} ${backupPath}`; await log(` ๐Ÿ“ฆ Backup created: ${backupPath}`); // Read file content and replace shebang @@ -126,7 +126,7 @@ export const handleClaudeRuntimeSwitch = async argv => { process.exit(1); } // Read current shebang - const firstLine = await $`head -1 "${claudePath}"`; + const firstLine = await $`head -1 ${claudePath}`; const currentShebang = firstLine.stdout.toString().trim(); await log(` Current shebang: ${currentShebang}`); if (currentShebang.includes('node') && !currentShebang.includes('bun')) { @@ -138,7 +138,7 @@ export const handleClaudeRuntimeSwitch = async argv => { try { await fs.access(backupPath); // Restore from backup - await $`cp "${backupPath}" "${claudePath}"`; + await $`cp ${backupPath} ${claudePath}`; await log(` โœ… Restored Claude from backup: ${backupPath}`); } catch (backupError) { reportError(backupError, { diff --git a/src/reviewers-hive.mjs b/src/reviewers-hive.mjs index 054a55cdd..9348415fa 100644 --- a/src/reviewers-hive.mjs +++ b/src/reviewers-hive.mjs @@ -322,10 +322,10 @@ async function reviewer(reviewerId) { await log(` ๐Ÿš€ Executing review.mjs for ${prUrl}...`); const startTime = Date.now(); - let reviewCommand = $`./review.mjs "${prUrl}" --model ${argv.model} --focus ${argv.focus}`; + let reviewCommand = $`./review.mjs ${prUrl} --model ${argv.model} --focus ${argv.focus}`; if (argv.autoApprove) { - reviewCommand = $`./review.mjs "${prUrl}" --model ${argv.model} --focus ${argv.focus} --approve`; + reviewCommand = $`./review.mjs ${prUrl} --model ${argv.model} --focus ${argv.focus} --approve`; } // Stream output and capture result diff --git a/src/solve.auto-pr.lib.mjs b/src/solve.auto-pr.lib.mjs index 392a76230..bf4e72cf5 100644 --- a/src/solve.auto-pr.lib.mjs +++ b/src/solve.auto-pr.lib.mjs @@ -1126,8 +1126,14 @@ ${prBody}`, // Link the issue to the PR in GitHub's Development section using GraphQL API await log(formatAligned('๐Ÿ”—', 'Linking:', `Issue #${issueNumber} to PR #${localPrNumber}...`)); try { + // Issue #2119: the double quotes below are GraphQL string syntax, + // not shell quoting. command-stream escapes interpolated values, so + // the queries are assembled in JS and passed as one argument. + const repositorySelector = `repository(owner: ${JSON.stringify(owner)}, name: ${JSON.stringify(repo)})`; + // First, get the node IDs for both the issue and the PR - const issueNodeResult = await $`gh api graphql -f query='query { repository(owner: "${owner}", name: "${repo}") { issue(number: ${issueNumber}) { id } } }' --jq .data.repository.issue.id`; + const issueNodeQuery = `query { ${repositorySelector} { issue(number: ${issueNumber}) { id } } }`; + const issueNodeResult = await $`gh api graphql -f query=${issueNodeQuery} --jq .data.repository.issue.id`; if (issueNodeResult.code !== 0) { throw new Error(`Failed to get issue node ID: ${issueNodeResult.stderr}`); @@ -1136,7 +1142,8 @@ ${prBody}`, const issueNodeId = issueNodeResult.stdout.toString().trim(); await log(` Issue node ID: ${issueNodeId}`, { verbose: true }); - const prNodeResult = await $`gh api graphql -f query='query { repository(owner: "${owner}", name: "${repo}") { pullRequest(number: ${localPrNumber}) { id } } }' --jq .data.repository.pullRequest.id`; + const prNodeQuery = `query { ${repositorySelector} { pullRequest(number: ${localPrNumber}) { id } } }`; + const prNodeResult = await $`gh api graphql -f query=${prNodeQuery} --jq .data.repository.pullRequest.id`; if (prNodeResult.code !== 0) { throw new Error(`Failed to get PR node ID: ${prNodeResult.stderr}`); @@ -1152,7 +1159,8 @@ ${prBody}`, // 2. For cross-repo (fork) PRs, we need "Fixes owner/repo#N" // Let's verify the link was created - const linkCheckResult = await $`gh api graphql -f query='query { repository(owner: "${owner}", name: "${repo}") { pullRequest(number: ${localPrNumber}) { closingIssuesReferences(first: 10) { nodes { number } } } } }' --jq '.data.repository.pullRequest.closingIssuesReferences.nodes[].number'`; + const linkCheckQuery = `query { ${repositorySelector} { pullRequest(number: ${localPrNumber}) { closingIssuesReferences(first: 10) { nodes { number } } } } }`; + const linkCheckResult = await $`gh api graphql -f query=${linkCheckQuery} --jq '.data.repository.pullRequest.closingIssuesReferences.nodes[].number'`; if (linkCheckResult.code === 0) { const linkedIssues = parseClosingIssueNumbers(linkCheckResult.stdout); diff --git a/src/solve.repository.lib.mjs b/src/solve.repository.lib.mjs index 76786d54e..47036d0a2 100644 --- a/src/solve.repository.lib.mjs +++ b/src/solve.repository.lib.mjs @@ -59,7 +59,11 @@ export const checkExistingForkOfRoot = async rootRepo => { const userResult = await lib.ghCmdRetry(() => $`gh api user --jq .login`, { label: 'get user (fork check)' }); if (userResult.code !== 0) return null; const currentUser = userResult.stdout.toString().trim(); - const forksResult = await lib.ghCmdRetry(() => $`gh api repos/${rootRepo}/forks --paginate --jq '.[] | select(.owner.login == "${currentUser}") | .full_name'`, { label: `check forks of ${rootRepo}` }); + // Issue #2119: build the jq expression in JS. Its double quotes belong to jq, + // not to the shell, and command-stream quotes interpolated values itself - so + // interpolating inside the quotes would leak shell quotes into the comparison. + const forkFilter = `.[] | select(.owner.login == ${JSON.stringify(currentUser)}) | .full_name`; + const forksResult = await lib.ghCmdRetry(() => $`gh api repos/${rootRepo}/forks --paginate --jq ${forkFilter}`, { label: `check forks of ${rootRepo}` }); if (forksResult.code !== 0) return null; const forks = forksResult.stdout @@ -324,9 +328,13 @@ export const tryInitializeEmptyRepository = async (owner, repo) => { const base64Content = Buffer.from(readmeContent).toString('base64'); // Try to create README.md using GitHub API + // Issue #2119: `--field content="${base64Content}"` would leak literal quotes + // into the field value as soon as command-stream decides the value needs + // quoting, so the whole `key=value` token is built in JS instead. + const contentField = `content=${base64Content}`; const createResult = await $`gh api repos/${owner}/${repo}/contents/README.md --method PUT --silent \ - --field message="Initialize repository with README" \ - --field content="${base64Content}" 2>&1`; + --field message=${'Initialize repository with README'} \ + --field ${contentField} 2>&1`; if (createResult.code === 0) { await log(`${formatAligned('โœ…', 'Success:', 'README.md created successfully')}`); @@ -1207,7 +1215,10 @@ export const setupPrForkRemote = async (tempDir, argv, prForkOwner, repo, isCont // Strategy 1: Query the upstream repo's forks to find this user's fork if (owner) { await log(`${formatAligned('๐Ÿ”', 'Discovering fork name:', `Searching ${owner}/${repo}/forks for ${prForkOwner}'s fork...`)}`); - const forksResult = await $`gh api repos/${owner}/${repo}/forks --paginate --jq '.[] | select(.owner.login == "${prForkOwner}") | .name'`; + // Issue #2119: the double quotes here are jq syntax, so the expression is + // built in JS and interpolated as one already-escaped argument. + const forkNameFilter = `.[] | select(.owner.login == ${JSON.stringify(prForkOwner)}) | .name`; + const forksResult = await $`gh api repos/${owner}/${repo}/forks --paginate --jq ${forkNameFilter}`; if (forksResult.code === 0 && forksResult.stdout) { const forkName = forksResult.stdout.toString().trim().split('\n')[0]; // Take first match if (forkName) { diff --git a/src/solve.results.lib.mjs b/src/solve.results.lib.mjs index f5b95abe0..a14c7e591 100644 --- a/src/solve.results.lib.mjs +++ b/src/solve.results.lib.mjs @@ -158,7 +158,7 @@ export const ensurePullRequestIssueLink = async ({ prNumber, issueNumber, owner, await writeSanitizedPublicationFile(tempBodyFile, linkResult.body); try { - const updateResult = await command`gh pr edit ${prNumber} --repo ${owner}/${repo} --body-file "${tempBodyFile}"`; + const updateResult = await command`gh pr edit ${prNumber} --repo ${owner}/${repo} --body-file ${tempBodyFile}`; await fs.unlink(tempBodyFile).catch(() => {}); if (updateResult.code === 0) { @@ -787,7 +787,7 @@ export const verifyResults = async (owner, repo, branchName, issueNumber, prNumb if (prTitleHasPlaceholder && !argv.autoRestartOnNonUpdatedPullRequestDescription) { const updatedTitle = await sanitizeForPublication(pr.title.replace(/^\[WIP\]\s*/, '')); await log(` ๐Ÿ“ Removing [WIP] prefix from PR title...`); - const titleResult = await $`gh pr edit ${pr.number} --repo ${owner}/${repo} --title "${updatedTitle}"`; + const titleResult = await $`gh pr edit ${pr.number} --repo ${owner}/${repo} --title ${updatedTitle}`; if (titleResult.code === 0) { await log(` โœ… Updated PR title to: "${updatedTitle}"`); } else { @@ -836,7 +836,7 @@ Fixes ${issueRef} await writeSanitizedPublicationFile(tempBodyFile, newDescription); try { - const descResult = await $`gh pr edit ${pr.number} --repo ${owner}/${repo} --body-file "${tempBodyFile}"`; + const descResult = await $`gh pr edit ${pr.number} --repo ${owner}/${repo} --body-file ${tempBodyFile}`; await fs.unlink(tempBodyFile).catch(() => {}); if (descResult.code === 0) { diff --git a/src/youtrack/youtrack-sync.mjs b/src/youtrack/youtrack-sync.mjs index b30479438..d41bc5f81 100644 --- a/src/youtrack/youtrack-sync.mjs +++ b/src/youtrack/youtrack-sync.mjs @@ -101,7 +101,7 @@ ${youTrackIssue.description || 'No description provided.'} if (needsUpdate) { await log(` ๐Ÿ“ Updating issue #${existingIssue.number} for ${youTrackId}...`); - const updateResult = await $`gh issue edit ${existingIssue.number} --repo ${owner}/${repo} --title "${ghTitle}" --body "${ghBody}"`; + const updateResult = await $`gh issue edit ${existingIssue.number} --repo ${owner}/${repo} --title ${ghTitle} --body ${ghBody}`; if (updateResult.code === 0) { await log(` โœ… Updated issue #${existingIssue.number}`); @@ -130,7 +130,7 @@ ${youTrackIssue.description || 'No description provided.'} await log(` โž• Creating GitHub issue for ${youTrackId}...`); try { - const createResult = await $`gh issue create --repo ${owner}/${repo} --title "${ghTitle}" --body "${ghBody}" --label "help wanted"`; + const createResult = await $`gh issue create --repo ${owner}/${repo} --title ${ghTitle} --body ${ghBody} --label "help wanted"`; if (createResult.code === 0) { const issueUrl = createResult.stdout.toString().trim(); diff --git a/tests/test-shell-quoting-2119.mjs b/tests/test-shell-quoting-2119.mjs new file mode 100644 index 000000000..361181276 --- /dev/null +++ b/tests/test-shell-quoting-2119.mjs @@ -0,0 +1,97 @@ +#!/usr/bin/env node + +/** + * Regression tests for issue #2119: quoted interpolations in `$` templates. + * + * The three reproduction pull requests were opened with the title + * `'Implement Hello World in Scala'` - single quotes included, as literal + * characters. For example: + * https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2 + * + * Root cause: `command-stream` shell-escapes every interpolated value, quoting + * it when it contains characters the shell would otherwise split on (a space is + * enough). Writing `$`gh pr edit ... --title "${title}"`` therefore produces + * `--title "'Implement Hello World in Scala'"`, and the extra quotes end up + * inside the title. Values without spaces slipped through unnoticed, which is + * why this survived so long. + * + * The fix is to never wrap a placeholder in quotes: interpolate bare, and when + * the quotes belong to an inner language (jq, GraphQL) build that expression in + * JS and interpolate the finished string as one argument. + * + * @hive-mind-test-suite default + */ + +import assert from 'node:assert/strict'; +import { readdir, readFile } from 'node:fs/promises'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +import { ensureUseM } from '../src/use-m-bootstrap.lib.mjs'; + +const use = await ensureUseM(); +const { $: $raw } = await use('command-stream'); +// Do not mirror probe output into the test log; only the captured value matters. +const $ = $raw({ mirror: false, capture: true }); + +const repoRoot = path.join(path.dirname(fileURLToPath(import.meta.url)), '..'); + +// --- the behaviour that caused the bug -------------------------------------- +// This is the executable version of the root cause: it must keep holding, or +// the static guard below is guarding against the wrong thing. +const title = 'Implement Hello World in Scala'; + +const doubleQuoted = await $`echo "${title}"`; +assert.equal(doubleQuoted.stdout.toString().trim(), `'${title}'`, 'wrapping a placeholder in double quotes leaks literal quotes - exactly the published PR title'); + +const bare = await $`echo ${title}`; +assert.equal(bare.stdout.toString().trim(), title, 'a bare placeholder passes the value through unchanged'); + +// Bare interpolation is also the safe form: command-stream still quotes values +// that need it, so arguments with spaces stay a single argument. +const spacedPath = path.join('/tmp', 'hive-mind-2119 quoting probe.txt'); +await $`rm -f ${spacedPath}`; +await $`touch ${spacedPath}`; +const listed = await $`ls -1 ${spacedPath}`; +assert.equal(listed.code, 0, 'a bare placeholder handles paths containing spaces'); +assert.ok(listed.stdout.toString().includes('hive-mind-2119 quoting probe.txt')); + +const listedQuoted = await $`ls -1 "${spacedPath}"`; +assert.notEqual(listedQuoted.code, 0, 'the same path wrapped in quotes is not found - the leaked quotes become part of the name'); +await $`rm -f ${spacedPath}`; + +// --- the codebase must not reintroduce the pattern -------------------------- +// Issue #2119 asks to "fully apply requirements to entire codebase", so this is +// a static guard rather than a check of the single site that was reported. +const TEMPLATE = /\$`(?:[^`\\]|\\.)*`/gs; +const QUOTED_PLACEHOLDER = /(["'])\$\{[^}]*\}\1/g; + +const sourceFiles = []; +const walk = async dir => { + for (const entry of await readdir(dir, { withFileTypes: true })) { + const entryPath = path.join(dir, entry.name); + if (entry.isDirectory()) await walk(entryPath); + else if (entry.name.endsWith('.mjs')) sourceFiles.push(entryPath); + } +}; +await walk(path.join(repoRoot, 'src')); +await walk(path.join(repoRoot, 'scripts')); + +const offenders = []; +for (const file of sourceFiles) { + const source = await readFile(file, 'utf8'); + for (const template of source.matchAll(TEMPLATE)) { + for (const hit of template[0].matchAll(QUOTED_PLACEHOLDER)) { + const line = source.slice(0, template.index).split('\n').length; + offenders.push(`${path.relative(repoRoot, file)}:${line} ${hit[0]}`); + } + } +} + +assert.deepEqual(offenders, [], `command-stream already escapes interpolated values; remove the quotes around these placeholders (or build the jq/GraphQL expression in JS):\n${offenders.join('\n')}`); + +// The specific site from the issue must use the bare form. +const resultsSource = await readFile(path.join(repoRoot, 'src', 'solve.results.lib.mjs'), 'utf8'); +assert.ok(resultsSource.includes('--title ${updatedTitle}`'), 'the PR title is set with a bare placeholder'); + +console.log(`PASS: issue #2119 shell quoting - bare placeholders, ${sourceFiles.length} source files clean`); From bd23dae689d0ebed3d303220ca1414035788615f Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 16:17:41 +0000 Subject: [PATCH 08/17] fix(2119): never report an empty pull request as changes or as ready to merge MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two of the three reproduction runs produced a pull request with an empty net diff (the AI tool wrote nothing, so the branch held only the solver's own scaffolding commit and a revert of it), yet both published ### Changes - 1 file(s) modified - 1 line(s) added and the Kotlin run additionally posted "## โœ… Ready to merge ... - No pending changes". Merging that would have closed the issue with nothing implemented. src/pull-request-changes.lib.mjs is now the single place that answers "does this pull request change anything", used by both the description writer and the mergeability watcher. A diff that could not be read is reported as unmeasured rather than as empty, so a transient API failure cannot trigger restarts. --- src/pull-request-changes.lib.mjs | 96 ++++++++++++++++++++++++++ src/solve.auto-merge.lib.mjs | 26 ++++++- src/solve.results.lib.mjs | 23 +++--- tests/test-empty-pull-request-2119.mjs | 92 ++++++++++++++++++++++++ 4 files changed, 224 insertions(+), 13 deletions(-) create mode 100644 src/pull-request-changes.lib.mjs create mode 100644 tests/test-empty-pull-request-2119.mjs diff --git a/src/pull-request-changes.lib.mjs b/src/pull-request-changes.lib.mjs new file mode 100644 index 000000000..6a6536c06 --- /dev/null +++ b/src/pull-request-changes.lib.mjs @@ -0,0 +1,96 @@ +#!/usr/bin/env node + +/** + * Issue #2119: does the pull request actually contain any changes? + * + * Nobody asked that question before, and two separate false positives followed + * from it. In the reproduction runs the AI tool produced nothing, so the branch + * ended up with the solver's own scaffolding commit and a revert of it - a net + * diff of zero files: + * + * https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2 + * https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2 + * + * Yet the published pull request bodies claimed + * + * ### Changes + * - 1 file(s) modified + * - 1 line(s) added + * + * (the stats were measured while the scaffolding commit was still in the diff + * and never revisited), and the Kotlin run went on to post "โœ… Ready to merge - + * No pending changes" for a pull request that changed nothing at all. + * + * This module is the single place that answers the question, so both the + * description writer and the mergeability watcher agree. + */ + +import { ghWithRateLimitRetry } from './github-rate-limit.lib.mjs'; + +/** + * Measure the net diff of a pull request. + * + * The counts come from the unified diff rather than from the PR's `additions` / + * `deletions` fields because those are per-commit sums: a commit and its revert + * report 1 addition and 1 deletion while the net diff is empty. + * + * @param {Object} params + * @param {string} params.owner + * @param {string} params.repo + * @param {number} params.prNumber + * @param {Function} params.$ command-stream tagged-template executor + * @returns {Promise<{hasChanges: boolean, filesChanged: number, additions: number, deletions: number, measured: boolean}>} + * `measured` is false when the diff could not be fetched, in which case + * callers must not treat the pull request as empty. + */ +export const getPullRequestChangeStats = async ({ owner, repo, prNumber, $ }) => { + let diffOutput = ''; + let measured = false; + try { + const result = await ghWithRateLimitRetry(() => $`gh pr diff ${prNumber} --repo ${owner}/${repo}`, { label: `pr diff ${owner}/${repo}#${prNumber}` }); + if (result.code === 0) { + diffOutput = result.stdout.toString(); + measured = true; + } + } catch { + // Leave measured false: an unreachable API must not read as "no changes". + } + + const filesChanged = (diffOutput.match(/^diff --git/gm) || []).length; + const additions = (diffOutput.match(/^\+[^+]/gm) || []).length; + const deletions = (diffOutput.match(/^-[^-]/gm) || []).length; + + return { hasChanges: filesChanged > 0, filesChanged, additions, deletions, measured }; +}; + +/** + * Render the "### Changes" section of a generated pull request description. + * + * When the diff is empty this says so instead of inventing a file count, so a + * reviewer reading the description learns the same thing the diff would tell + * them. + * + * @param {{hasChanges: boolean, filesChanged: number, additions: number, deletions: number, measured: boolean}} stats + * @returns {string} + */ +export const formatChangeSummary = stats => { + if (!stats.measured) { + return '- The diff could not be read, so the change summary is unavailable'; + } + if (!stats.hasChanges) { + return '- No files were changed by this pull request yet'; + } + return [`- ${stats.filesChanged} file(s) modified`, `- ${stats.additions} line(s) added`, `- ${stats.deletions} line(s) removed`].join('\n'); +}; + +/** + * The blocker to report when a pull request is otherwise mergeable but empty. + * + * Merging it would close the issue without changing anything, so this is + * treated as a reason to restart the AI rather than as success. The shared + * auto-restart budget bounds the retries and fails the run visibly once it is + * exhausted. + */ +export const EMPTY_PULL_REQUEST_BLOCKER = 'The pull request contains no changes (its net diff is empty), so there is nothing to merge'; + +export default { getPullRequestChangeStats, formatChangeSummary, EMPTY_PULL_REQUEST_BLOCKER }; diff --git a/src/solve.auto-merge.lib.mjs b/src/solve.auto-merge.lib.mjs index 47e7ecbe5..05e7c8eba 100644 --- a/src/solve.auto-merge.lib.mjs +++ b/src/solve.auto-merge.lib.mjs @@ -90,6 +90,9 @@ const { beginAutoRestartBudget, consumeAutoRestartIteration, formatAutoRestartLa const { failOnAutoRestartBudgetExhausted } = await import('./auto-restart-exhaustion.lib.mjs'); const { ensurePullRequestBaseBranch } = await import('./solve.pr-base-guard.lib.mjs'); +// Issue #2119: an empty pull request must not be reported as ready to merge. +const { EMPTY_PULL_REQUEST_BLOCKER, getPullRequestChangeStats } = await import('./pull-request-changes.lib.mjs'); + // Issue #1895: explicitly close linked issues after merging a PR into a // non-default branch, where GitHub does not auto-close them. const { ensureLinkedIssueClosedAfterMerge } = await import('./github-issue-auto-close.lib.mjs'); @@ -286,9 +289,19 @@ export const watchUntilMergeable = async params => { } } + // Issue #2119: an empty pull request is not "ready to merge". The Kotlin + // reproduction run posted "โœ… Ready to merge - No pending changes" for a + // pull request whose net diff was empty, so merging it would have closed + // the issue without implementing anything. + const changeStats = await getPullRequestChangeStats({ owner, repo, prNumber, $ }); + const isEmptyPullRequest = changeStats.measured && !changeStats.hasChanges; + if (isEmptyPullRequest) { + await log(formatAligned('โš ๏ธ', 'PR is empty:', 'net diff contains no files - not treating it as mergeable', 2), { level: 'warning' }); + } + // If PR is mergeable, no blockers, no new comments, no issue metadata - // edits, and no uncommitted changes - if (blockers.length === 0 && !hasNewComments && !hasIssueMetadataChanges && !hasUncommittedChanges) { + // edits, no uncommitted changes and it actually changes something + if (blockers.length === 0 && !hasNewComments && !hasIssueMetadataChanges && !hasUncommittedChanges && !isEmptyPullRequest) { // Issue #1503 (enhanced): Multi-mechanism consensus + repo-wide action check. // Before declaring PR mergeable, run multiple independent CI detection mechanisms // and require all to agree. This catches race conditions where CI starts between @@ -443,6 +456,15 @@ export const watchUntilMergeable = async params => { feedbackLines.push('Please review and address the feedback from these comments.'); } + // Issue #2119: Reason 1a: the pull request does not change anything yet. + if (isEmptyPullRequest) { + shouldRestart = true; + restartReason = restartReason ? `${restartReason}; ${EMPTY_PULL_REQUEST_BLOCKER}` : EMPTY_PULL_REQUEST_BLOCKER; + feedbackLines.push(`๐Ÿ“ญ ${EMPTY_PULL_REQUEST_BLOCKER}.`); + feedbackLines.push(''); + feedbackLines.push('Implement the requested change and commit it to the pull request branch. Do not report the work as done while the diff is empty.'); + } + // Issue #2007: Reason 1b: Issue title/description edited by the user. if (hasIssueMetadataChanges) { shouldRestart = true; diff --git a/src/solve.results.lib.mjs b/src/solve.results.lib.mjs index a14c7e591..15832b27a 100644 --- a/src/solve.results.lib.mjs +++ b/src/solve.results.lib.mjs @@ -67,6 +67,9 @@ const { reportError } = sentryLib; const prIssueLinking = await import('./pr-issue-linking.lib.mjs'); const { buildIssueReference, ensureIssueLinkInPullRequestBody } = prIssueLinking; +// Issue #2119: the one place that decides whether a pull request changed anything. +const { formatChangeSummary, getPullRequestChangeStats } = await import('./pull-request-changes.lib.mjs'); + /** * Placeholder patterns used to detect auto-generated PR content that was not updated by the agent. * These patterns match the initial WIP PR created by solve.auto-pr.lib.mjs. @@ -801,14 +804,14 @@ export const verifyResults = async (owner, repo, branchName, issueNumber, prNumb if (hasPlaceholder && !argv.autoRestartOnNonUpdatedPullRequestDescription) { await log(` ๐Ÿ“ Updating PR description to remove placeholder text...`); - // Build a summary of the changes from the PR diff - const diffResult = await $`gh pr diff ${pr.number} --repo ${owner}/${repo} 2>&1`; - const diffOutput = diffResult.code === 0 ? diffResult.stdout.toString() : ''; - - // Count files changed - const filesChanged = (diffOutput.match(/^diff --git/gm) || []).length; - const additions = (diffOutput.match(/^\+[^+]/gm) || []).length; - const deletions = (diffOutput.match(/^-[^-]/gm) || []).length; + // Issue #2119: measure the net diff. The reproduction PRs published + // "1 file(s) modified, 1 line(s) added" for a pull request that + // changed nothing, because the stats were never checked for being + // empty. + const changeStats = await getPullRequestChangeStats({ owner, repo, prNumber: pr.number, $ }); + if (!changeStats.hasChanges) { + await log(` โš ๏ธ PR #${pr.number} has an empty diff - the description will say so instead of claiming changes`, { level: 'warning' }); + } // Get the issue title for context const issueTitleResult = await $`gh issue view ${issueNumber} --repo ${owner}/${repo} --json title --jq .title 2>&1`; @@ -822,9 +825,7 @@ export const verifyResults = async (owner, repo, branchName, issueNumber, prNumb This pull request implements a solution for ${issueRef}: ${issueTitle} ### Changes -- ${filesChanged} file(s) modified -- ${additions} line(s) added -- ${deletions} line(s) removed +${formatChangeSummary(changeStats)} ### Issue Reference Fixes ${issueRef} diff --git a/tests/test-empty-pull-request-2119.mjs b/tests/test-empty-pull-request-2119.mjs new file mode 100644 index 000000000..3ca7c9704 --- /dev/null +++ b/tests/test-empty-pull-request-2119.mjs @@ -0,0 +1,92 @@ +#!/usr/bin/env node + +/** + * Regression tests for issue #2119: an empty pull request is not a result. + * + * Two of the three reproduction runs ended with a pull request whose net diff + * was empty - the AI tool wrote nothing, so the branch held only the solver's + * own scaffolding commit and a revert of it: + * + * https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2 + * https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2 + * + * Both published a description claiming + * + * ### Changes + * - 1 file(s) modified + * - 1 line(s) added + * + * and the Kotlin one went further and posted "## โœ… Ready to merge ... - No + * pending changes". Merging that would have closed the issue with nothing + * implemented, which is the worst kind of false positive: it looks like success. + * + * @hive-mind-test-suite default + */ + +import assert from 'node:assert/strict'; +import { readFile } from 'node:fs/promises'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +import { EMPTY_PULL_REQUEST_BLOCKER, formatChangeSummary, getPullRequestChangeStats } from '../src/pull-request-changes.lib.mjs'; + +const repoRoot = path.join(path.dirname(fileURLToPath(import.meta.url)), '..'); + +// A `$` stand-in that returns a fixed diff, so the tests never touch the network. +const fake$ = + ({ stdout = '', code = 0, throws = false }) => + () => { + if (throws) throw new Error('gh: could not reach the API'); + return Promise.resolve({ code, stdout: Buffer.from(stdout) }); + }; + +// --- measuring the diff ------------------------------------------------------ + +// The exact shape of the reproduction PRs: a commit and its revert cancel out, +// so `gh pr diff` prints nothing at all. +const empty = await getPullRequestChangeStats({ owner: 'konard', repo: 'test-hello-world', prNumber: 2, $: fake$({ stdout: '' }) }); +assert.equal(empty.measured, true, 'an empty diff was still successfully measured'); +assert.equal(empty.hasChanges, false, 'a pull request with an empty net diff has no changes'); +assert.equal(empty.filesChanged, 0); + +const realDiff = ['diff --git a/examples/hello.scala b/examples/hello.scala', 'new file mode 100644', '--- /dev/null', '+++ b/examples/hello.scala', '@@ -0,0 +1,3 @@', '+object Hello {', '+ def main(args: Array[String]): Unit = println("Hello, World!")', '+}'].join('\n'); +const changed = await getPullRequestChangeStats({ owner: 'konard', repo: 'test-hello-world', prNumber: 2, $: fake$({ stdout: realDiff }) }); +assert.equal(changed.hasChanges, true, 'a pull request that adds a file has changes'); +assert.equal(changed.filesChanged, 1); +assert.equal(changed.additions, 3, 'the `+++ b/...` header is not counted as an added line'); + +// An unreachable API must never be mistaken for "nothing changed" - that would +// turn a transient network failure into an endless restart loop. +for (const broken of [fake$({ code: 1 }), fake$({ throws: true })]) { + const stats = await getPullRequestChangeStats({ owner: 'konard', repo: 'test-hello-world', prNumber: 2, $: broken }); + assert.equal(stats.measured, false, 'a failed diff read is reported as unmeasured'); + assert.equal(stats.hasChanges, false, 'unmeasured stats claim no changes, so callers must gate on `measured`'); +} + +// --- the published description ---------------------------------------------- + +assert.equal(formatChangeSummary(empty), '- No files were changed by this pull request yet', 'the description states the diff is empty instead of inventing a file count'); +assert.ok(!formatChangeSummary(empty).includes('1 file(s) modified'), 'the false positive from the reproduction PRs is gone'); + +const summary = formatChangeSummary(changed); +assert.ok(summary.includes('- 1 file(s) modified'), summary); +assert.ok(summary.includes('- 3 line(s) added'), summary); + +const unavailable = formatChangeSummary({ measured: false, hasChanges: false, filesChanged: 0, additions: 0, deletions: 0 }); +assert.ok(unavailable.includes('could not be read'), 'an unreadable diff is reported as unknown, not as empty'); + +// --- the callers must actually use it ---------------------------------------- + +const autoMergeSource = await readFile(path.join(repoRoot, 'src', 'solve.auto-merge.lib.mjs'), 'utf8'); +assert.ok(autoMergeSource.includes("await import('./pull-request-changes.lib.mjs')"), 'the auto-merge watcher measures the diff'); +assert.ok(autoMergeSource.includes('const isEmptyPullRequest = changeStats.measured && !changeStats.hasChanges'), 'an unmeasured diff does not count as empty'); +assert.ok(/!hasUncommittedChanges && !isEmptyPullRequest/.test(autoMergeSource), 'the "ready to merge" branch is gated on the pull request not being empty'); +assert.ok(autoMergeSource.includes('EMPTY_PULL_REQUEST_BLOCKER'), 'an empty pull request is reported as a restart reason'); + +const resultsSource = await readFile(path.join(repoRoot, 'src', 'solve.results.lib.mjs'), 'utf8'); +assert.ok(resultsSource.includes('formatChangeSummary(changeStats)'), 'the generated description renders the shared change summary'); +assert.ok(!/- \$\{filesChanged\} file\(s\) modified/.test(resultsSource), 'the old unconditional file count is gone'); + +assert.ok(EMPTY_PULL_REQUEST_BLOCKER.includes('net diff is empty'), EMPTY_PULL_REQUEST_BLOCKER); + +console.log('PASS: issue #2119 empty pull requests are neither described as changes nor reported as ready to merge'); From b1065b373fe110f45369750e192e583e25d74d4b Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 16:25:59 +0000 Subject: [PATCH 09/17] fix(2119): stop AI tool scratch state from driving the auto-restart loop MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Formal AI writes a .formal-ai/ plan directory into the workspace it runs in, so every uncommitted-changes check saw ?? .formal-ai/ ๐Ÿ“ Found uncommitted changes ๐Ÿ”„ AUTO-RESTART: Restarting Agent to handle uncommitted changes... and restarting recreated the directory, so the blocker could never clear. The same state reached 'git add -A' on the auto-commit paths, which would have published a tool's private scratch files in the user's pull request. This generalizes the .playwright-mcp/ special case from #1124: the paths are written to .git/info/exclude, so git itself stops reporting them and all eight copies of checkForUncommittedChanges agree without each needing its own filter. The exclude file is local to the clone, so nothing leaks into the diff. --- experiments/apply-scratch-filter-2119.mjs | 61 +++++++++ src/agent-commander.lib.mjs | 6 +- src/agent.lib.mjs | 6 +- src/ai-tool-scratch.lib.mjs | 143 ++++++++++++++++++++++ src/claude.lib.mjs | 6 +- src/codex.lib.mjs | 5 +- src/gemini.lib.mjs | 6 +- src/opencode.lib.mjs | 6 +- src/qwen.lib.mjs | 6 +- src/solve.repository.lib.mjs | 6 + src/solve.restart-shared.lib.mjs | 8 +- tests/test-ai-tool-scratch-2119.mjs | 124 +++++++++++++++++++ 12 files changed, 374 insertions(+), 9 deletions(-) create mode 100644 experiments/apply-scratch-filter-2119.mjs create mode 100644 src/ai-tool-scratch.lib.mjs create mode 100644 tests/test-ai-tool-scratch-2119.mjs diff --git a/experiments/apply-scratch-filter-2119.mjs b/experiments/apply-scratch-filter-2119.mjs new file mode 100644 index 000000000..b6e9cd9d2 --- /dev/null +++ b/experiments/apply-scratch-filter-2119.mjs @@ -0,0 +1,61 @@ +#!/usr/bin/env node + +// Issue #2119: wire the shared AI-tool-scratch filter into every per-tool +// `checkForUncommittedChanges`. The eight implementations are near-identical +// copies, so the same two edits apply to each; done as a script so no copy is +// silently skipped. + +import { readFile, writeFile } from 'node:fs/promises'; +import path from 'node:path'; + +const repoRoot = path.join(import.meta.dirname, '..'); + +const IMPORT_LINE = "import { ensureAiToolScratchIgnored, filterAiToolScratchFromStatus } from './ai-tool-scratch.lib.mjs';\n"; +const ENSURE_CALL = ` // Issue #2119: AI tools leave scratch state (.formal-ai/, .playwright-mcp/) in + // the workspace. Ignoring it here keeps it out of both this check and 'git add -A'. + await ensureAiToolScratchIgnored(tempDir, log); +`; + +const files = ['qwen.lib.mjs', 'agent.lib.mjs', 'gemini.lib.mjs', 'opencode.lib.mjs', 'claude.lib.mjs', 'codex.lib.mjs', 'agent-commander.lib.mjs']; + +const anchorLog = " await log('\\n๐Ÿ” Checking for uncommitted changes...');\n"; +const oldStatus = ' const statusOutput = gitStatusResult.stdout.toString().trim();\n'; +const newStatus = ' const statusOutput = filterAiToolScratchFromStatus(gitStatusResult.stdout.toString().trim());\n'; + +for (const file of files) { + const filePath = path.join(repoRoot, 'src', file); + let source = await readFile(filePath, 'utf8'); + + if (source.includes('./ai-tool-scratch.lib.mjs')) { + console.log(`skip (already wired): ${file}`); + continue; + } + + // 1. import, placed after the last top-of-file static import + const importMatches = [...source.matchAll(/^import .*;\n/gm)]; + const lastImport = importMatches[importMatches.length - 1]; + source = source.slice(0, lastImport.index + lastImport[0].length) + IMPORT_LINE + source.slice(lastImport.index + lastImport[0].length); + + // 2. ensure the exclude entries exist before reading git status + const fnIndex = source.indexOf('export const checkForUncommittedChanges'); + if (fnIndex === -1) throw new Error(`no checkForUncommittedChanges in ${file}`); + const logIndex = source.indexOf(anchorLog, fnIndex); + if (logIndex === -1) throw new Error(`no log anchor in ${file}`); + source = source.slice(0, logIndex + anchorLog.length) + ENSURE_CALL + source.slice(logIndex + anchorLog.length); + + // 3. filter the status output (fallback for workspaces cloned before the fix) + const statusIndex = source.indexOf(oldStatus, fnIndex); + if (statusIndex === -1) { + // agent-commander uses a slightly different shape; handled below. + const acOld = " const statusOutput = gitStatusResult.stdout?.toString().trim() || '';\n"; + const acNew = " const statusOutput = filterAiToolScratchFromStatus(gitStatusResult.stdout?.toString().trim() || '');\n"; + const acIndex = source.indexOf(acOld, fnIndex); + if (acIndex === -1) throw new Error(`no statusOutput assignment in ${file}`); + source = source.slice(0, acIndex) + acNew + source.slice(acIndex + acOld.length); + } else { + source = source.slice(0, statusIndex) + newStatus + source.slice(statusIndex + oldStatus.length); + } + + await writeFile(filePath, source, 'utf8'); + console.log(`patched: ${file}`); +} diff --git a/src/agent-commander.lib.mjs b/src/agent-commander.lib.mjs index 80328906c..0af4d4a06 100644 --- a/src/agent-commander.lib.mjs +++ b/src/agent-commander.lib.mjs @@ -13,6 +13,7 @@ import { buildCodexDisable1mContextConfigArgs, buildCodexSubSessionSizeConfigArg import { detectUsageLimit } from './usage-limit.lib.mjs'; import { applyFormalAiPricingOverride } from './formal-ai-pricing.lib.mjs'; // Issue #2119 import { getCacheReadTokenCount, getCumulativeContextInputTokens, getOutputTokenCount } from './context-fill.lib.mjs'; +import { ensureAiToolScratchIgnored, filterAiToolScratchFromStatus } from './ai-tool-scratch.lib.mjs'; export const AGENT_COMMANDER_TOOLS = new Set(['claude', 'codex', 'opencode', 'agent', 'qwen', 'gemini']); @@ -389,8 +390,11 @@ export const executeWithAgentCommander = async params => { export const checkForUncommittedChanges = async (tempDir, owner, repo, branchName, $, log = defaultLog, autoCommit = false, autoRestartEnabled = true) => { await log('\n๐Ÿ” Checking for uncommitted changes...'); + // Issue #2119: AI tools leave scratch state (.formal-ai/, .playwright-mcp/) in + // the workspace. Ignoring it here keeps it out of both this check and 'git add -A'. + await ensureAiToolScratchIgnored(tempDir, log); const gitStatusResult = await $({ cwd: tempDir })`git status --porcelain 2>&1`; - const statusOutput = gitStatusResult.stdout?.toString().trim() || ''; + const statusOutput = filterAiToolScratchFromStatus(gitStatusResult.stdout?.toString().trim() || ''); if (!statusOutput) { await log('โœ… No uncommitted changes found'); diff --git a/src/agent.lib.mjs b/src/agent.lib.mjs index 2cd80766f..f1b5096dc 100644 --- a/src/agent.lib.mjs +++ b/src/agent.lib.mjs @@ -29,6 +29,7 @@ import { createAgentTokenUsage, accumulateAgentStepFinishUsage, parseAgentTokenU import { createJsonStreamScanner, parseJsonRecords } from './json-stream.lib.mjs'; import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs'; import { attachStreamingInput, finalizeBidirectionalHandler, setupBidirectionalHandler } from './bidirectional-interactive.lib.mjs'; +import { ensureAiToolScratchIgnored, filterAiToolScratchFromStatus } from './ai-tool-scratch.lib.mjs'; export { createAgentTokenUsage, accumulateAgentStepFinishUsage, parseAgentTokenUsage }; @@ -1066,11 +1067,14 @@ export const executeAgentCommand = async params => { export const checkForUncommittedChanges = async (tempDir, owner, repo, branchName, $, log, autoCommit = false, autoRestartEnabled = true) => { // Similar to OpenCode version, check for uncommitted changes await log('\n๐Ÿ” Checking for uncommitted changes...'); + // Issue #2119: AI tools leave scratch state (.formal-ai/, .playwright-mcp/) in + // the workspace. Ignoring it here keeps it out of both this check and 'git add -A'. + await ensureAiToolScratchIgnored(tempDir, log); try { const gitStatusResult = await $({ cwd: tempDir })`git status --porcelain 2>&1`; if (gitStatusResult.code === 0) { - const statusOutput = gitStatusResult.stdout.toString().trim(); + const statusOutput = filterAiToolScratchFromStatus(gitStatusResult.stdout.toString().trim()); if (statusOutput) { await log('๐Ÿ“ Found uncommitted changes'); diff --git a/src/ai-tool-scratch.lib.mjs b/src/ai-tool-scratch.lib.mjs new file mode 100644 index 000000000..31744619e --- /dev/null +++ b/src/ai-tool-scratch.lib.mjs @@ -0,0 +1,143 @@ +#!/usr/bin/env node + +/** + * Issue #2119: keep AI tools' own scratch state out of the solver's workspace + * bookkeeping. + * + * AI tools drop working files into the directory they are run in. Formal AI + * writes `.formal-ai/` (a `general-change-plan.lino` plan file and friends); + * Playwright MCP writes `.playwright-mcp/` (issue #1124). Neither belongs to the + * user's change, but `git status --porcelain` reports them all the same, so the + * solver read them as "the AI left uncommitted changes" and restarted the tool - + * every iteration, forever, because restarting recreates the same scratch dir: + * + * ๐Ÿ” Checking for uncommitted changes... + * ?? .formal-ai/ + * ๐Ÿ“ Found uncommitted changes + * ๐Ÿ”„ AUTO-RESTART: Restarting Agent to handle uncommitted changes... + * + * (docs/case-studies/issue-2119/data/logs/agent-scala-solution-draft.log:5425) + * + * The same state also reached `git add -A` on the auto-commit paths, which would + * have published a tool's private scratch files in the user's pull request. + * + * Every tool integration has its own `checkForUncommittedChanges` + * (claude/codex/agent/opencode/gemini/qwen/agent-commander), so filtering inside + * any one of them would fix one caller and leave the rest. Instead the paths are + * written to `.git/info/exclude`, which makes git itself stop reporting them: + * every status check, every `git add -A` and every diff agrees, without touching + * the repository's own `.gitignore` (that would show up in the pull request). + */ + +import fs from 'fs/promises'; +import path from 'path'; + +/** + * Scratch paths AI tools create inside the workspace they are run in. + * + * Keep this list to directories a tool owns entirely. Anything a user might + * legitimately want committed must not be here. + */ +export const AI_TOOL_SCRATCH_PATHS = [ + { path: '.formal-ai/', tool: 'formal-ai' }, + { path: '.playwright-mcp/', tool: 'playwright-mcp' }, +]; + +const EXCLUDE_HEADER = '# hive-mind: AI tool scratch directories (issue #2119)'; + +/** + * Does this `git status --porcelain` line describe only AI tool scratch state? + * + * @param {string} statusLine - e.g. `?? .formal-ai/` + * @returns {boolean} + */ +export const isAiToolScratchPath = statusLine => { + if (typeof statusLine !== 'string') return false; + // Porcelain v1: two status characters, a space, then the path. + const filePath = statusLine.slice(3).trim().replace(/^"|"$/g, ''); + if (!filePath) return false; + return AI_TOOL_SCRATCH_PATHS.some(({ path: scratchPath }) => { + const withoutSlash = scratchPath.replace(/\/$/, ''); + return filePath === withoutSlash || filePath.startsWith(`${withoutSlash}/`); + }); +}; + +/** + * Drop AI tool scratch entries from `git status --porcelain` output. + * + * A fallback for workspaces set up before `ensureAiToolScratchIgnored` ran (an + * existing clone, a resumed run), so a stale scratch directory cannot restart + * the restart loop. + * + * @param {string} statusOutput - raw `git status --porcelain` output + * @returns {string} the same output without scratch-only lines + */ +export const filterAiToolScratchFromStatus = statusOutput => { + if (!statusOutput) return ''; + return statusOutput + .split('\n') + .filter(line => line.trim() && !isAiToolScratchPath(line)) + .join('\n'); +}; + +/** + * Teach a cloned workspace to ignore AI tool scratch directories. + * + * Writes to `.git/info/exclude` rather than `.gitignore`: the exclude file is + * local to the clone, is never staged, and therefore never appears in the pull + * request. Idempotent - re-running leaves the file unchanged. + * + * @param {string} tempDir - the cloned workspace + * @param {(msg: string, opts?: object) => Promise} [log] + * @returns {Promise<{applied: boolean, reason?: string, added?: string[]}>} + */ +export const ensureAiToolScratchIgnored = async (tempDir, log = null) => { + const excludePath = path.join(tempDir, '.git', 'info', 'exclude'); + const report = async msg => { + if (log) await log(msg, { verbose: true }); + }; + + let existing = ''; + try { + existing = await fs.readFile(excludePath, 'utf8'); + } catch (error) { + if (error.code !== 'ENOENT') { + await report(`โš ๏ธ Could not read ${excludePath}: ${error.message}`); + return { applied: false, reason: 'unreadable' }; + } + // A worktree or a fresh clone may not have the file yet; creating it is fine. + } + + const existingLines = new Set( + existing + .split('\n') + .map(line => line.trim()) + .filter(Boolean) + ); + const missing = AI_TOOL_SCRATCH_PATHS.map(entry => entry.path).filter(scratchPath => !existingLines.has(scratchPath)); + + if (missing.length === 0) { + return { applied: true, reason: 'already_present', added: [] }; + } + + const separator = existing && !existing.endsWith('\n') ? '\n' : ''; + const block = existingLines.has(EXCLUDE_HEADER) ? `${missing.join('\n')}\n` : `${EXCLUDE_HEADER}\n${missing.join('\n')}\n`; + + try { + await fs.mkdir(path.dirname(excludePath), { recursive: true }); + await fs.writeFile(excludePath, `${existing}${separator}${block}`, 'utf8'); + } catch (error) { + await report(`โš ๏ธ Could not update ${excludePath}: ${error.message}`); + return { applied: false, reason: 'unwritable' }; + } + + await report(`๐Ÿงน Ignoring AI tool scratch directories in this workspace: ${missing.join(', ')}`); + return { applied: true, added: missing }; +}; + +export default { + AI_TOOL_SCRATCH_PATHS, + ensureAiToolScratchIgnored, + filterAiToolScratchFromStatus, + isAiToolScratchPath, +}; diff --git a/src/claude.lib.mjs b/src/claude.lib.mjs index 0bd874419..d93d25d05 100644 --- a/src/claude.lib.mjs +++ b/src/claude.lib.mjs @@ -501,6 +501,7 @@ export const calculateSessionTokens = async (sessionId, tempDir, resultModelUsag }; // Extracted to claude.stderr.lib.mjs (Issue #477, #1337) import { isStderrError } from './claude.stderr.lib.mjs'; +import { ensureAiToolScratchIgnored, filterAiToolScratchFromStatus } from './ai-tool-scratch.lib.mjs'; export { isStderrError }; export const executeClaudeCommand = async params => { const { @@ -1426,10 +1427,13 @@ export const executeClaudeCommand = async params => { }; export const checkForUncommittedChanges = async (tempDir, owner, repo, branchName, $, log, autoCommit = false, autoRestartEnabled = true) => { await log('\n๐Ÿ” Checking for uncommitted changes...'); + // Issue #2119: AI tools leave scratch state (.formal-ai/, .playwright-mcp/) in + // the workspace. Ignoring it here keeps it out of both this check and 'git add -A'. + await ensureAiToolScratchIgnored(tempDir, log); try { const gitStatusResult = await $({ cwd: tempDir })`git status --porcelain 2>&1`; if (gitStatusResult.code === 0) { - const statusOutput = gitStatusResult.stdout.toString().trim(); + const statusOutput = filterAiToolScratchFromStatus(gitStatusResult.stdout.toString().trim()); if (statusOutput) { await log('๐Ÿ“ Found uncommitted changes'); await log('Changes:'); diff --git a/src/codex.lib.mjs b/src/codex.lib.mjs index fdc9d29f8..1d6f4575e 100644 --- a/src/codex.lib.mjs +++ b/src/codex.lib.mjs @@ -41,6 +41,7 @@ import { deployHandoffSkill } from './handoff-skill.lib.mjs'; // Issue #1877 import { applyCodexCapabilityEnv, runCodexCapabilityPreflight } from './codex-capability-preflight.lib.mjs'; // Issue #2074 import { createPullRequestBaseBranchCommandIntervention } from './solve.pr-base-command-intervention.lib.mjs'; import Decimal from 'decimal.js-light'; +import { ensureAiToolScratchIgnored, filterAiToolScratchFromStatus } from './ai-tool-scratch.lib.mjs'; const CODEX_USAGE_FIELD_NAMES = ['input_tokens', 'cached_input_tokens', 'output_tokens', 'cache_write_tokens', 'cache_creation_input_tokens', 'reasoning_tokens', 'reasoning_output_tokens', 'input_tokens_details.cached_tokens', 'input_tokens_details.cache_read_tokens', 'input_tokens_details.cache_write_tokens', 'input_tokens_details.cache_creation_tokens', 'input_tokens_details.cache_creation_input_tokens', 'output_tokens_details.reasoning_tokens']; const CODEX_LONG_CONTEXT_PRICE_THRESHOLD = 272000; @@ -1404,11 +1405,13 @@ export const executeCodexCommand = async params => { export const checkForUncommittedChanges = async (tempDir, owner, repo, branchName, $, log, autoCommit = false, autoRestartEnabled = true) => { // Similar to Claude and OpenCode version, check for uncommitted changes await log('\n๐Ÿ” Checking for uncommitted changes...'); + // Issue #2119: keep AI tool scratch state out of this check and of `git add -A`. + await ensureAiToolScratchIgnored(tempDir, log); try { const gitStatusResult = await $({ cwd: tempDir })`git status --porcelain 2>&1`; if (gitStatusResult.code === 0) { - const statusOutput = gitStatusResult.stdout.toString().trim(); + const statusOutput = filterAiToolScratchFromStatus(gitStatusResult.stdout.toString().trim()); if (statusOutput) { await log('๐Ÿ“ Found uncommitted changes'); diff --git a/src/gemini.lib.mjs b/src/gemini.lib.mjs index 597c03376..eaeda1de9 100644 --- a/src/gemini.lib.mjs +++ b/src/gemini.lib.mjs @@ -23,6 +23,7 @@ import { buildFormalAiPricingInfo } from './formal-ai-pricing.lib.mjs'; // Issue import { checkPlaywrightMcpPackageAvailability } from './playwright-mcp.lib.mjs'; import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs'; import { getCumulativeContextInputTokens, toTokenCount } from './context-fill.lib.mjs'; +import { ensureAiToolScratchIgnored, filterAiToolScratchFromStatus } from './ai-tool-scratch.lib.mjs'; import { getTerminalEventCompletionHealth } from './tool-run-health.lib.mjs'; // Issue #1990 const shellQuote = value => `"${String(value).replaceAll('\\', '\\\\').replaceAll('"', '\\"')}"`; @@ -700,11 +701,14 @@ export const executeGeminiCommand = async params => { export const checkForUncommittedChanges = async (tempDir, owner, repo, branchName, $, log, autoCommit = false, autoRestartEnabled = true) => { await log('\n๐Ÿ” Checking for uncommitted changes...'); + // Issue #2119: AI tools leave scratch state (.formal-ai/, .playwright-mcp/) in + // the workspace. Ignoring it here keeps it out of both this check and 'git add -A'. + await ensureAiToolScratchIgnored(tempDir, log); try { const gitStatusResult = await $({ cwd: tempDir })`git status --porcelain 2>&1`; if (gitStatusResult.code === 0) { - const statusOutput = gitStatusResult.stdout.toString().trim(); + const statusOutput = filterAiToolScratchFromStatus(gitStatusResult.stdout.toString().trim()); if (statusOutput) { await log('๐Ÿ“ Found uncommitted changes'); diff --git a/src/opencode.lib.mjs b/src/opencode.lib.mjs index 3b9bdb16a..47d878e6b 100644 --- a/src/opencode.lib.mjs +++ b/src/opencode.lib.mjs @@ -26,6 +26,7 @@ import { createAgentTokenUsage, accumulateAgentStepFinishUsage, parseAgentTokenU import { createJsonStreamScanner } from './json-stream.lib.mjs'; import { calculateAgentPricing } from './agent.lib.mjs'; import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs'; +import { ensureAiToolScratchIgnored, filterAiToolScratchFromStatus } from './ai-tool-scratch.lib.mjs'; export { parseOpenCodeTokenUsage }; @@ -609,11 +610,14 @@ export const executeOpenCodeCommand = async params => { export const checkForUncommittedChanges = async (tempDir, owner, repo, branchName, $, log, autoCommit = false, autoRestartEnabled = true) => { // Similar to Claude version, check for uncommitted changes await log('\n๐Ÿ” Checking for uncommitted changes...'); + // Issue #2119: AI tools leave scratch state (.formal-ai/, .playwright-mcp/) in + // the workspace. Ignoring it here keeps it out of both this check and 'git add -A'. + await ensureAiToolScratchIgnored(tempDir, log); try { const gitStatusResult = await $({ cwd: tempDir })`git status --porcelain 2>&1`; if (gitStatusResult.code === 0) { - const statusOutput = gitStatusResult.stdout.toString().trim(); + const statusOutput = filterAiToolScratchFromStatus(gitStatusResult.stdout.toString().trim()); if (statusOutput) { await log('๐Ÿ“ Found uncommitted changes'); diff --git a/src/qwen.lib.mjs b/src/qwen.lib.mjs index bcd3e7166..352b22b27 100644 --- a/src/qwen.lib.mjs +++ b/src/qwen.lib.mjs @@ -24,6 +24,7 @@ import { buildFormalAiPricingInfo } from './formal-ai-pricing.lib.mjs'; // Issue import { checkPlaywrightMcpPackageAvailability } from './playwright-mcp.lib.mjs'; import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } from './tool-retry.lib.mjs'; import { getCumulativeContextInputTokens, getRestoredContextInputTokens, toTokenCount } from './context-fill.lib.mjs'; +import { ensureAiToolScratchIgnored, filterAiToolScratchFromStatus } from './ai-tool-scratch.lib.mjs'; import { getTerminalEventCompletionHealth } from './tool-run-health.lib.mjs'; // Issue #1990 export const mapModelToId = model => qwenModels[model] || model; @@ -751,11 +752,14 @@ export const executeQwenCommand = async params => { export const checkForUncommittedChanges = async (tempDir, owner, repo, branchName, $, log, autoCommit = false, autoRestartEnabled = true) => { await log('\n๐Ÿ” Checking for uncommitted changes...'); + // Issue #2119: AI tools leave scratch state (.formal-ai/, .playwright-mcp/) in + // the workspace. Ignoring it here keeps it out of both this check and 'git add -A'. + await ensureAiToolScratchIgnored(tempDir, log); try { const gitStatusResult = await $({ cwd: tempDir })`git status --porcelain 2>&1`; if (gitStatusResult.code === 0) { - const statusOutput = gitStatusResult.stdout.toString().trim(); + const statusOutput = filterAiToolScratchFromStatus(gitStatusResult.stdout.toString().trim()); if (statusOutput) { await log('๐Ÿ“ Found uncommitted changes'); diff --git a/src/solve.repository.lib.mjs b/src/solve.repository.lib.mjs index 47036d0a2..3de8c1c39 100644 --- a/src/solve.repository.lib.mjs +++ b/src/solve.repository.lib.mjs @@ -29,6 +29,7 @@ const { log, formatAligned } = lib; // Import exit handler import { safeExit } from './exit-handler.lib.mjs'; +import { ensureAiToolScratchIgnored } from './ai-tool-scratch.lib.mjs'; import { parseForkFullNameFromGhOutput } from './github-repository-names.lib.mjs'; import { checkReplacementRepositoryBranchSafety } from './solve.repository-safety.lib.mjs'; import { buildForkReplacementBlockedReason, buildForkReplacementSafetyCheckDescription } from './solve.repository-recovery-message.lib.mjs'; @@ -1070,6 +1071,11 @@ export const cloneRepository = async (repoToClone, tempDir, argv, owner, repo) = if (cloneResult.code === 0 && repoIsValid) { await log(`${formatAligned('โœ…', 'Cloned to:', tempDir)}`); + // Issue #2119: AI tools drop scratch state (`.formal-ai/`, `.playwright-mcp/`) + // into the workspace. Exclude it here, once, so every later `git status` and + // `git add -A` agrees instead of reading it as the AI's uncommitted work. + await ensureAiToolScratchIgnored(tempDir, log); + // Verify and fix remote configuration const remoteCheckResult = await $({ cwd: tempDir })`git remote -v 2>&1`; if (!remoteCheckResult.stdout || !remoteCheckResult.stdout.toString().includes('origin')) { diff --git a/src/solve.restart-shared.lib.mjs b/src/solve.restart-shared.lib.mjs index f19824c36..2316d27e8 100644 --- a/src/solve.restart-shared.lib.mjs +++ b/src/solve.restart-shared.lib.mjs @@ -32,6 +32,7 @@ const fs = (await use('fs')).promises; const lib = await import('./lib.mjs'); const { log, formatAligned, extractToolErrorCore } = lib; const { ensurePullRequestBaseBranch } = await import('./solve.pr-base-guard.lib.mjs'); +const { ensureAiToolScratchIgnored, filterAiToolScratchFromStatus } = await import('./ai-tool-scratch.lib.mjs'); const { RESOURCE_PHASE_RESTART_AFTER, RESOURCE_PHASE_RESTART_BEFORE, recordResourceSnapshot } = await import('./solve.resource-diagnostics.lib.mjs'); // Import Sentry integration @@ -126,11 +127,14 @@ export const cleanupPlaywrightMcpFolder = async (tempDir, argv = {}) => { export const checkForUncommittedChanges = async (tempDir, argv = {}) => { // First, clean up .playwright-mcp/ folder to prevent false positives (Issue #1124) await cleanupPlaywrightMcpFolder(tempDir, argv); + // Issue #2119: the same false positive, generalized - `.formal-ai/` and any + // other AI tool scratch directory must not read as the AI's uncommitted work. + await ensureAiToolScratchIgnored(tempDir, log); try { const gitStatusResult = await $({ cwd: tempDir })`git status --porcelain 2>&1`; if (gitStatusResult.code === 0) { - const statusOutput = gitStatusResult.stdout.toString().trim(); + const statusOutput = filterAiToolScratchFromStatus(gitStatusResult.stdout.toString().trim()); return statusOutput.length > 0; } } catch (error) { @@ -154,7 +158,7 @@ export const getUncommittedChangesDetails = async tempDir => { try { const gitStatusResult = await $({ cwd: tempDir })`git status --porcelain 2>&1`; if (gitStatusResult.code === 0) { - const statusOutput = gitStatusResult.stdout.toString().trim(); + const statusOutput = filterAiToolScratchFromStatus(gitStatusResult.stdout.toString().trim()); if (statusOutput) { changes.push(...statusOutput.split('\n')); } diff --git a/tests/test-ai-tool-scratch-2119.mjs b/tests/test-ai-tool-scratch-2119.mjs new file mode 100644 index 000000000..25e7a5a38 --- /dev/null +++ b/tests/test-ai-tool-scratch-2119.mjs @@ -0,0 +1,124 @@ +#!/usr/bin/env node + +/** + * Regression tests for issue #2119: AI tool scratch state must not look like + * the AI's uncommitted work. + * + * The Scala reproduction run restarted the tool on every iteration because + * Formal AI writes a `.formal-ai/` plan directory into the workspace it runs in: + * + * ๐Ÿ” Checking for uncommitted changes... + * ?? .formal-ai/ + * ๐Ÿ“ Found uncommitted changes + * ๐Ÿ”„ AUTO-RESTART: Restarting Agent to handle uncommitted changes... + * + * (docs/case-studies/issue-2119/data/logs/agent-scala-solution-draft.log:5425) + * + * Restarting recreates the directory, so the blocker could never clear. The same + * state also reached the `git add -A` on the auto-commit paths, which would have + * published a tool's private scratch files inside the user's pull request. + * + * @hive-mind-test-suite default + */ + +import assert from 'node:assert/strict'; +import { mkdtemp, readFile, readdir, rm, writeFile, mkdir } from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +import { AI_TOOL_SCRATCH_PATHS, ensureAiToolScratchIgnored, filterAiToolScratchFromStatus, isAiToolScratchPath } from '../src/ai-tool-scratch.lib.mjs'; +import { ensureUseM } from '../src/use-m-bootstrap.lib.mjs'; + +const use = await ensureUseM(); +const { $: $raw } = await use('command-stream'); +const $ = $raw({ mirror: false, capture: true }); + +const repoRoot = path.join(path.dirname(fileURLToPath(import.meta.url)), '..'); + +// --- classification ---------------------------------------------------------- + +assert.ok(isAiToolScratchPath('?? .formal-ai/'), 'the exact line from the reproduction log is recognised as scratch state'); +assert.ok(isAiToolScratchPath('?? .formal-ai/general-change-plan.lino'), 'files inside the scratch directory count too'); +assert.ok(isAiToolScratchPath('?? .playwright-mcp/'), 'the .playwright-mcp case from issue #1124 stays covered'); + +// Real work must never be filtered away - that would hide the AI's changes. +assert.ok(!isAiToolScratchPath('?? examples'), 'a new examples/ directory is the AIโ€™s actual work'); +assert.ok(!isAiToolScratchPath(' M src/solve.mjs'), 'a modified source file is real work'); +assert.ok(!isAiToolScratchPath('?? .formal-ai-notes.md'), 'a similarly named file outside the scratch directory is real work'); +assert.ok(!isAiToolScratchPath(''), 'an empty line is not a scratch path'); + +// The exact status output the Scala run saw: only `examples` survives. +assert.equal(filterAiToolScratchFromStatus('?? .formal-ai/\n?? examples'), '?? examples', 'the scratch directory is dropped and the real change is kept'); +assert.equal(filterAiToolScratchFromStatus('?? .formal-ai/'), '', 'a workspace containing nothing but scratch state reads as clean - the restart loop stops'); +assert.equal(filterAiToolScratchFromStatus(''), ''); + +// --- the git-level fix ------------------------------------------------------- +// `.git/info/exclude` is the single place that makes every git command agree, +// which matters because each tool integration reads git status for itself. +const workspace = await mkdtemp(path.join(os.tmpdir(), 'hive-mind-2119-scratch-')); +try { + await $`git -C ${workspace} init -q`; + await $`git -C ${workspace} config user.email test@example.com`; + await $`git -C ${workspace} config user.name Test`; + await writeFile(path.join(workspace, 'README.md'), '# test\n'); + await $`git -C ${workspace} add README.md`; + await $`git -C ${workspace} -c commit.gpgsign=false commit -q -m initial`; + + // Reproduce the workspace state: the tool left its plan file behind. + await mkdir(path.join(workspace, '.formal-ai'), { recursive: true }); + await writeFile(path.join(workspace, '.formal-ai', 'general-change-plan.lino'), 'plan\n'); + + const before = (await $`git -C ${workspace} status --porcelain`).stdout.toString().trim(); + assert.equal(before, '?? .formal-ai/', 'without the fix git reports the scratch directory as an untracked change'); + + const applied = await ensureAiToolScratchIgnored(workspace); + assert.equal(applied.applied, true); + assert.deepEqual( + applied.added, + AI_TOOL_SCRATCH_PATHS.map(entry => entry.path) + ); + + const after = (await $`git -C ${workspace} status --porcelain`).stdout.toString().trim(); + assert.equal(after, '', 'git itself now ignores the scratch directory, so every caller agrees'); + + // `git add -A` must not publish the tool's scratch files in the pull request. + await $`git -C ${workspace} add -A`; + const staged = (await $`git -C ${workspace} diff --cached --name-only`).stdout.toString().trim(); + assert.equal(staged, '', 'the scratch directory is not staged by git add -A'); + + // The repository's own .gitignore is untouched: an exclude entry is local to + // the clone and cannot leak into the diff. + const tracked = await readdir(workspace); + assert.ok(!tracked.includes('.gitignore'), 'no .gitignore was created in the workspace'); + + // Idempotent: a second call adds nothing, so restarts do not append forever. + const second = await ensureAiToolScratchIgnored(workspace); + assert.deepEqual(second.added, [], 're-running adds no duplicate entries'); + const excludeText = await readFile(path.join(workspace, '.git', 'info', 'exclude'), 'utf8'); + assert.equal(excludeText.match(/^\.formal-ai\/$/gm).length, 1, 'the entry appears exactly once'); + + // Real work is still detected after the fix - the check must not go blind. + await writeFile(path.join(workspace, 'hello.scala'), 'object Hello\n'); + const withWork = (await $`git -C ${workspace} status --porcelain`).stdout.toString().trim(); + assert.equal(withWork, '?? hello.scala', 'an actual new file is still reported'); +} finally { + await rm(workspace, { recursive: true, force: true }); +} + +// --- every tool integration must use it -------------------------------------- +// Issue #2119 asks to "fully apply requirements to entire codebase"; there are +// eight copies of checkForUncommittedChanges and fixing one would fix nothing. +const toolLibs = ['qwen.lib.mjs', 'agent.lib.mjs', 'gemini.lib.mjs', 'opencode.lib.mjs', 'claude.lib.mjs', 'codex.lib.mjs', 'agent-commander.lib.mjs', 'solve.restart-shared.lib.mjs']; +for (const file of toolLibs) { + const source = await readFile(path.join(repoRoot, 'src', file), 'utf8'); + assert.ok(source.includes('checkForUncommittedChanges'), `${file} still defines the check`); + assert.ok(source.includes('ensureAiToolScratchIgnored'), `${file} excludes AI tool scratch state before reading git status`); + assert.ok(source.includes('filterAiToolScratchFromStatus'), `${file} filters scratch state out of the status it acts on`); +} + +// The workspace is set up once at clone time so the very first check is clean. +const repositorySource = await readFile(path.join(repoRoot, 'src', 'solve.repository.lib.mjs'), 'utf8'); +assert.ok(repositorySource.includes('ensureAiToolScratchIgnored(tempDir, log)'), 'the freshly cloned workspace excludes AI tool scratch state'); + +console.log(`PASS: issue #2119 AI tool scratch state is ignored in ${toolLibs.length} uncommitted-change checks`); From 12f27c00c32da0f1290ce8b0fc2473e47e9d796a Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 16:29:08 +0000 Subject: [PATCH 10/17] fix(2119): publish honest working session summaries The Kotlin reproduction run posted ## Working session summary The `pwd` command completed. Output: /tmp/gh-issue-solver-1785421161275 on a pull request whose diff was empty. The AI tool returning nothing useful is an upstream problem; what Hive Mind owns is that the comment read as a report of completed work and published the solver's private workspace path. The summary now carries an explicit notice when the pull request still contains no changes, and solver workspace paths are replaced with . An unreadable diff produces no notice, so a transient API failure cannot turn into a false "nothing was implemented" claim. --- src/solve.results.lib.mjs | 16 ++++- src/working-session-summary.lib.mjs | 65 +++++++++++++++++ tests/test-working-session-summary-2119.mjs | 77 +++++++++++++++++++++ 3 files changed, 156 insertions(+), 2 deletions(-) create mode 100644 src/working-session-summary.lib.mjs create mode 100644 tests/test-working-session-summary-2119.mjs diff --git a/src/solve.results.lib.mjs b/src/solve.results.lib.mjs index 15832b27a..b4bd993c4 100644 --- a/src/solve.results.lib.mjs +++ b/src/solve.results.lib.mjs @@ -69,6 +69,7 @@ const { buildIssueReference, ensureIssueLinkInPullRequestBody } = prIssueLinking // Issue #2119: the one place that decides whether a pull request changed anything. const { formatChangeSummary, getPullRequestChangeStats } = await import('./pull-request-changes.lib.mjs'); +const { buildNoChangesNotice, redactWorkspacePaths } = await import('./working-session-summary.lib.mjs'); /** * Placeholder patterns used to detect auto-generated PR content that was not updated by the agent. @@ -1270,7 +1271,7 @@ export const buildWorkingSessionSummaryDetails = ({ publicPricingEstimate = null return `${costInfo}${budgetStats}`.trim(); }; -export const attachSolutionSummary = async ({ resultSummary, prNumber, issueNumber, owner, repo, publicPricingEstimate = null, anthropicTotalCostUSD = null, pricingInfo = null, budgetStatsData = null }) => { +export const attachSolutionSummary = async ({ resultSummary, prNumber, issueNumber, owner, repo, publicPricingEstimate = null, anthropicTotalCostUSD = null, pricingInfo = null, budgetStatsData = null, changeStats = null }) => { if (!resultSummary || typeof resultSummary !== 'string') { await log('โš ๏ธ No working session summary available to attach', { verbose: true }); return false; @@ -1291,10 +1292,16 @@ export const attachSolutionSummary = async ({ resultSummary, prNumber, issueNumb pricingInfo, budgetStatsData, }); + // Issue #2119: publish what the session actually produced. The reported + // summary said "The `pwd` command completed" and printed the solver's own + // /tmp workspace, on a pull request that was still empty. + const noChangesNotice = buildNoChangesNotice(changeStats); + const summaryBody = redactWorkspacePaths(resultSummary); + const comment = `${toolComments.WORKING_SESSION_SUMMARY_AUTOMATION_MARKER} ## ${toolComments.WORKING_SESSION_SUMMARY_MARKER} -${resultSummary}${usageDetails ? `\n\n${usageDetails}` : ''} +${summaryBody}${noChangesNotice ? `\n\n${noChangesNotice}` : ''}${usageDetails ? `\n\n${usageDetails}` : ''} --- *${toolComments.WORKING_SESSION_SUMMARY_AUTOMATED_FOOTER}*`; @@ -1396,6 +1403,10 @@ export const maybeAttachWorkingSessionSummary = async ({ argv, resultSummary, wo ...sessionUsage, }) : null); + // Issue #2119: a summary posted on a pull request that changed nothing must + // say so, instead of reading as a report of completed work. + const changeStats = prNumber ? await getPullRequestChangeStats({ owner, repo, prNumber, $ }) : null; + const ok = await attachSolutionSummary({ resultSummary, prNumber, @@ -1406,6 +1417,7 @@ export const maybeAttachWorkingSessionSummary = async ({ argv, resultSummary, wo anthropicTotalCostUSD, pricingInfo, budgetStatsData: resolvedBudgetStatsData, + changeStats, }); return { attached: !!ok, reason: ok ? 'attached' : 'post_failed', budgetStatsData: resolvedBudgetStatsData }; }; diff --git a/src/working-session-summary.lib.mjs b/src/working-session-summary.lib.mjs new file mode 100644 index 000000000..b7da14861 --- /dev/null +++ b/src/working-session-summary.lib.mjs @@ -0,0 +1,65 @@ +#!/usr/bin/env node + +/** + * Issue #2119: make the published "Working session summary" comment honest. + * + * The Kotlin reproduction run ended with the AI tool answering a single `pwd` + * and returning, and Hive Mind published exactly that as the session's result: + * + * + * ## Working session summary + * + * The `pwd` command completed. Output: + * + * ```text + * /tmp/gh-issue-solver-1785421161275 + * ``` + * + * (https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2#issuecomment-5132013034) + * + * Two things are wrong with that comment, and both are Hive Mind's to fix - the + * tool returning nothing useful is a separate, upstream problem: + * + * 1. It reads as a report of completed work. A reader has to open the diff to + * discover the pull request is still empty. Stating that in the comment + * turns a misleading summary into an accurate one. + * 2. It publishes the solver's private workspace path. That path is an + * implementation detail of the machine the run happened on; it is noise in + * a public comment and it tells readers about the host filesystem. + */ + +/** Solver workspace directories, as created by solve.repository.lib.mjs. */ +const WORKSPACE_PATH_PATTERN = /(?:\/private)?\/(?:tmp|var\/folders\/[^\s/]+\/[^\s/]+\/[^\s/]+)\/gh-issue-solver(?:-resume)?-[A-Za-z0-9._-]+/g; + +/** Replacement shown in place of a redacted workspace path. */ +export const WORKSPACE_PATH_PLACEHOLDER = ''; + +/** + * Replace solver workspace paths with a placeholder. + * + * Only the solver's own `gh-issue-solver-*` directories are touched: paths the + * user actually cares about (repository-relative paths, other absolute paths) + * are left exactly as the AI wrote them. + * + * @param {string} text + * @returns {string} + */ +export const redactWorkspacePaths = text => { + if (typeof text !== 'string' || !text) return text; + return text.replace(WORKSPACE_PATH_PATTERN, WORKSPACE_PATH_PLACEHOLDER); +}; + +/** + * The line appended when the pull request still has an empty diff. + * + * @param {{measured: boolean, hasChanges: boolean}|null} changeStats - from + * `getPullRequestChangeStats`; `null` or unmeasured stats produce no notice, + * so a failed diff read never turns into a false "no changes" claim. + * @returns {string} the notice, or an empty string when none applies + */ +export const buildNoChangesNotice = changeStats => { + if (!changeStats || !changeStats.measured || changeStats.hasChanges) return ''; + return '> โš ๏ธ This pull request still contains no changes - nothing was implemented yet.'; +}; + +export default { buildNoChangesNotice, redactWorkspacePaths, WORKSPACE_PATH_PLACEHOLDER }; diff --git a/tests/test-working-session-summary-2119.mjs b/tests/test-working-session-summary-2119.mjs new file mode 100644 index 000000000..88a4907bd --- /dev/null +++ b/tests/test-working-session-summary-2119.mjs @@ -0,0 +1,77 @@ +#!/usr/bin/env node + +/** + * Regression tests for issue #2119: the published "Working session summary". + * + * The Kotlin reproduction run posted this comment on a pull request whose diff + * was empty: + * + * + * ## Working session summary + * + * The `pwd` command completed. Output: + * + * ```text + * /tmp/gh-issue-solver-1785421161275 + * ``` + * + * https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2#issuecomment-5132013034 + * + * The AI tool doing nothing is an upstream problem. What Hive Mind owns is that + * the comment read as a report of completed work and leaked the solver's own + * workspace path into a public comment. + * + * @hive-mind-test-suite default + */ + +import assert from 'node:assert/strict'; +import { readFile } from 'node:fs/promises'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +import { buildNoChangesNotice, redactWorkspacePaths, WORKSPACE_PATH_PLACEHOLDER } from '../src/working-session-summary.lib.mjs'; + +const repoRoot = path.join(path.dirname(fileURLToPath(import.meta.url)), '..'); + +// --- workspace path redaction ------------------------------------------------ + +// The verbatim body of the reported comment. +const reportedSummary = ['The `pwd` command completed. Output:', '', '```text', '/tmp/gh-issue-solver-1785421161275', '```'].join('\n'); + +const redacted = redactWorkspacePaths(reportedSummary); +assert.ok(!redacted.includes('/tmp/gh-issue-solver-1785421161275'), 'the solver workspace path is not published'); +assert.ok(redacted.includes(WORKSPACE_PATH_PLACEHOLDER), redacted); +assert.ok(redacted.includes('The `pwd` command completed.'), 'everything else the AI wrote is preserved verbatim'); + +assert.equal(redactWorkspacePaths('cd /tmp/gh-issue-solver-resume-abc123-999 && ls'), `cd ${WORKSPACE_PATH_PLACEHOLDER} && ls`, 'resume workspaces are redacted too'); +assert.equal(redactWorkspacePaths('/private/var/folders/aa/bb/T/gh-issue-solver-42'), WORKSPACE_PATH_PLACEHOLDER, 'the macOS temp directory layout is covered'); + +// Paths that belong to the user's project must survive untouched. +for (const kept of ['src/solve.mjs', '/home/user/projects/my-repo/src/index.ts', '/tmp/my-own-scratch-file.txt', 'See ./docs/case-studies/issue-2119/README.md']) { + assert.equal(redactWorkspacePaths(kept), kept, `unrelated path is left alone: ${kept}`); +} + +for (const notText of [null, undefined, '', 42]) { + assert.equal(redactWorkspacePaths(notText), notText, 'non-string input passes through unchanged'); +} + +// --- the "nothing was implemented" notice ------------------------------------ + +const emptyStats = { measured: true, hasChanges: false, filesChanged: 0, additions: 0, deletions: 0 }; +const notice = buildNoChangesNotice(emptyStats); +assert.ok(notice.includes('no changes'), notice); +assert.ok(notice.startsWith('>'), 'the notice is rendered as a blockquote so it stands out from the AI text'); + +assert.equal(buildNoChangesNotice({ measured: true, hasChanges: true, filesChanged: 1, additions: 3, deletions: 0 }), '', 'a pull request with changes gets no notice'); +assert.equal(buildNoChangesNotice({ measured: false, hasChanges: false, filesChanged: 0, additions: 0, deletions: 0 }), '', 'an unreadable diff must not produce a false "no changes" claim'); +assert.equal(buildNoChangesNotice(null), '', 'a summary attached to an issue rather than a pull request gets no notice'); + +// --- the comment builder must use both --------------------------------------- + +const resultsSource = await readFile(path.join(repoRoot, 'src', 'solve.results.lib.mjs'), 'utf8'); +assert.ok(resultsSource.includes("await import('./working-session-summary.lib.mjs')"), 'solve.results.lib.mjs imports the helpers'); +assert.ok(resultsSource.includes('const summaryBody = redactWorkspacePaths(resultSummary);'), 'the posted body is redacted'); +assert.ok(resultsSource.includes('const noChangesNotice = buildNoChangesNotice(changeStats);'), 'the posted body carries the empty-diff notice'); +assert.ok(resultsSource.includes('const changeStats = prNumber ? await getPullRequestChangeStats({ owner, repo, prNumber, $ }) : null;'), 'the diff is measured before the summary is posted'); + +console.log('PASS: issue #2119 working session summaries are redacted and state when nothing was implemented'); From 16d8b2437fce8d69f9b1b051965bd1b60013af39 Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 16:39:44 +0000 Subject: [PATCH 11/17] fix(2119): frame gemini and qwen stream JSON by balanced values `formal-ai with --verbose` emits pretty-printed, multi-line JSON records. Commit 21197003 fixed the resulting "Token usage: 0 input, 0 output" for agent, opencode and codex by framing records on balanced JSON values instead of on newlines; gemini and qwen kept their own line-based parsers and so still had the identical defect. Both now share `takeJsonRecords` from json-stream.lib.mjs, which also handles records concatenated without a separator (issue #1250) and records split across process chunks. tests/test-formal-ai-uniform-tools-2119.mjs pins the uniformity down: same stream shapes, same token accounting and the same Link.Assistant $0.00 pricing across all six formal-ai tools. --- .../find-quoted-interpolations-2119.mjs | 40 +++++ experiments/import-smoke-2119.mjs | 4 + experiments/jq-quote-probe-2119.mjs | 18 ++ experiments/path-quote-probe-2119.mjs | 39 ++++ experiments/quote-probe-2119.mjs | 27 +++ src/gemini.lib.mjs | 47 ++--- src/json-stream.lib.mjs | 24 +++ src/qwen.lib.mjs | 43 ++--- tests/test-formal-ai-uniform-tools-2119.mjs | 168 ++++++++++++++++++ 9 files changed, 346 insertions(+), 64 deletions(-) create mode 100644 experiments/find-quoted-interpolations-2119.mjs create mode 100644 experiments/import-smoke-2119.mjs create mode 100644 experiments/jq-quote-probe-2119.mjs create mode 100644 experiments/path-quote-probe-2119.mjs create mode 100644 experiments/quote-probe-2119.mjs create mode 100644 tests/test-formal-ai-uniform-tools-2119.mjs diff --git a/experiments/find-quoted-interpolations-2119.mjs b/experiments/find-quoted-interpolations-2119.mjs new file mode 100644 index 000000000..5af0a0133 --- /dev/null +++ b/experiments/find-quoted-interpolations-2119.mjs @@ -0,0 +1,40 @@ +#!/usr/bin/env node +/** + * Issue #2119: find `"${x}"` / `'${x}'` inside command-stream `$` templates. + * + * command-stream already shell-quotes every interpolated value, so a manual + * quote around the placeholder leaks literal quotes into the argument whenever + * the value needs quoting (i.e. contains a space). That is how the PR title + * became `'Implement Hello World in Scala'`. + */ +import { readdir, readFile } from 'node:fs/promises'; +import path from 'node:path'; + +const roots = ['src', 'scripts']; +const files = []; +const walk = async dir => { + for (const e of await readdir(dir, { withFileTypes: true })) { + const p = path.join(dir, e.name); + if (e.isDirectory()) await walk(p); + else if (e.name.endsWith('.mjs')) files.push(p); + } +}; +for (const r of roots) await walk(r); + +// Match `$` followed by a backtick template, tolerating nested ${...} braces. +const TEMPLATE = /\$`(?:[^`\\]|\\.)*`/gs; +const QUOTED = /(["'])\$\{[^}]*\}\1/g; + +let total = 0; +for (const file of files.sort()) { + const src = await readFile(file, 'utf8'); + for (const m of src.matchAll(TEMPLATE)) { + const hits = [...m[0].matchAll(QUOTED)]; + if (!hits.length) continue; + const line = src.slice(0, m.index).split('\n').length; + total += hits.length; + console.log(`${file}:${line} ${hits.map(h => h[0]).join(' ')}`); + console.log(` ${m[0].replace(/\s+/g, ' ').slice(0, 190)}`); + } +} +console.log(`\n${total} quoted interpolation(s)`); diff --git a/experiments/import-smoke-2119.mjs b/experiments/import-smoke-2119.mjs new file mode 100644 index 000000000..5ba3c2bad --- /dev/null +++ b/experiments/import-smoke-2119.mjs @@ -0,0 +1,4 @@ +await import('../src/codex.lib.mjs'); +await import('../src/opencode.lib.mjs'); +await import('../src/agent.lib.mjs'); +console.log('imports ok'); diff --git a/experiments/jq-quote-probe-2119.mjs b/experiments/jq-quote-probe-2119.mjs new file mode 100644 index 000000000..6766f5058 --- /dev/null +++ b/experiments/jq-quote-probe-2119.mjs @@ -0,0 +1,18 @@ +#!/usr/bin/env node +// Issue #2119: `--jq '... == "${x}"'` inside a command-stream template. +import { ensureUseM } from '../src/use-m-bootstrap.lib.mjs'; +const use = await ensureUseM(); +const { $ } = await use('command-stream'); +const login = 'konard'; +const broken = await $`gh api /users/${login} --jq 'select(.login == "${login}") | .login'`; +console.log( + 'interpolated inside jq quotes:', + JSON.stringify(broken.stdout.toString().trimEnd()), + 'code', + broken.code, + String(broken.stderr || '') + .trim() + .slice(0, 160) +); +const fixed = await $`gh api /users/${login} --jq ${`select(.login == "${login}") | .login`}`; +console.log('pre-built jq expression: ', JSON.stringify(fixed.stdout.toString().trimEnd()), 'code', fixed.code); diff --git a/experiments/path-quote-probe-2119.mjs b/experiments/path-quote-probe-2119.mjs new file mode 100644 index 000000000..183c9d80f --- /dev/null +++ b/experiments/path-quote-probe-2119.mjs @@ -0,0 +1,39 @@ +#!/usr/bin/env node +// Issue #2119: `"${path}"` inside a command-stream template - does the command find the file? +import { ensureUseM } from '../src/use-m-bootstrap.lib.mjs'; +const use = await ensureUseM(); +const { $ } = await use('command-stream'); +const file = '/tmp/quote-probe-2119-file.txt'; +await $`rm -f ${file}`; +await $`sh -c ${`printf 'first-line\n' > ${file}`}`; +const quoted = await $`head -1 "${file}"`; +const bare = await $`head -1 ${file}`; +console.log( + 'head -1 "${file}":', + JSON.stringify(quoted.stdout.toString().trimEnd()), + 'code', + quoted.code, + String(quoted.stderr || '') + .trim() + .slice(0, 120) +); +console.log('head -1 ${file} :', JSON.stringify(bare.stdout.toString().trimEnd()), 'code', bare.code); +await $`rm -f ${file}`; + +// Same test with a space in the path - this is where the extra quotes leak. +const spaced = '/tmp/quote probe 2119 file.txt'; +await $`rm -f ${spaced}`; +await $`sh -c ${`printf 'first-line\n' > '${spaced}'`}`; +const q2 = await $`head -1 "${spaced}"`; +const b2 = await $`head -1 ${spaced}`; +console.log( + 'spaced quoted:', + JSON.stringify(q2.stdout.toString().trimEnd()), + 'code', + q2.code, + String(q2.stderr || '') + .trim() + .slice(0, 120) +); +console.log('spaced bare :', JSON.stringify(b2.stdout.toString().trimEnd()), 'code', b2.code); +await $`rm -f ${spaced}`; diff --git a/experiments/quote-probe-2119.mjs b/experiments/quote-probe-2119.mjs new file mode 100644 index 000000000..10144588c --- /dev/null +++ b/experiments/quote-probe-2119.mjs @@ -0,0 +1,27 @@ +#!/usr/bin/env node +// Issue #2119: how does command-stream quote interpolated values in each context? +import { ensureUseM } from '../src/use-m-bootstrap.lib.mjs'; +const use = await ensureUseM(); +const { $ } = await use('command-stream'); +const v = 'Hello World'; +const path = '/tmp/quote probe 2119.txt'; + +const cases = { + 'double-quoted "${v}"': await $`echo "${v}"`, + 'bare ${v} ': await $`echo ${v}`, + "single-quoted '${v}'": await $`echo '${v}'`, + 'inside jq-ish --arg \'x == "${v}"\'': await $`echo 'x == "${v}"'`, +}; +for (const [k, r] of Object.entries(cases)) console.log(k.padEnd(38), JSON.stringify(r.stdout.toString().trimEnd())); + +// A path with a space must survive bare interpolation. +await $`rm -f ${path}`; +const w = await $`touch ${path}`; +const ls = await $`ls -1 ${path}`; +console.log('bare path with space:'.padEnd(38), w.code, JSON.stringify(ls.stdout.toString().trimEnd())); +await $`rm -f ${path}`; + +// The jq-inside-single-quotes form used by the fork lookups: does it still match? +const login = 'konard'; +const broken = await $`echo '{"owner":{"login":"konard"},"full_name":"konard/x"}' | jq -r 'select(.owner.login == "${login}") | .full_name'`; +console.log('jq select with "${login}":'.padEnd(38), JSON.stringify(broken.stdout.toString().trimEnd()), 'code', broken.code, (broken.stderr || '').toString().trim().slice(0, 120)); diff --git a/src/gemini.lib.mjs b/src/gemini.lib.mjs index eaeda1de9..b79566109 100644 --- a/src/gemini.lib.mjs +++ b/src/gemini.lib.mjs @@ -25,6 +25,7 @@ import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } fro import { getCumulativeContextInputTokens, toTokenCount } from './context-fill.lib.mjs'; import { ensureAiToolScratchIgnored, filterAiToolScratchFromStatus } from './ai-tool-scratch.lib.mjs'; import { getTerminalEventCompletionHealth } from './tool-run-health.lib.mjs'; // Issue #1990 +import { takeJsonRecords } from './json-stream.lib.mjs'; // Issue #2119 const shellQuote = value => `"${String(value).replaceAll('\\', '\\\\').replaceAll('"', '\\"')}"`; @@ -232,41 +233,17 @@ export const parseGeminiJsonOutput = (output, state = {}, modelId = null) => { partialLine: state.partialLine || '', }; - const trimmedOutput = output.trim(); - if (trimmedOutput && !nextState.partialLine) { - try { - const parsed = JSON.parse(trimmedOutput); - for (const event of Array.isArray(parsed) ? parsed : [parsed]) { - applyGeminiJsonEvent(event, nextState, modelId); - } - return nextState; - } catch { - // stream-json emits one JSON object per line; fall through to JSONL parsing. - } - } - - const bufferedOutput = `${nextState.partialLine}${output}`; - nextState.partialLine = ''; - const lines = bufferedOutput.split(/\r?\n/); - const hasTrailingLineBreak = /\r?\n$/.test(bufferedOutput); - const completeLines = hasTrailingLineBreak ? lines : lines.slice(0, -1); - const possiblePartialLine = hasTrailingLineBreak ? '' : lines.at(-1) || ''; - - for (const line of completeLines) { - if (!line.trim()) continue; - - try { - applyGeminiJsonEvent(JSON.parse(line), nextState, modelId); - } catch { - continue; - } - } - - if (possiblePartialLine.trim()) { - try { - applyGeminiJsonEvent(JSON.parse(possiblePartialLine), nextState, modelId); - } catch { - nextState.partialLine = possiblePartialLine; + // Issue #2119: frame the stream by balanced JSON values instead of by lines. + // `formal-ai with gemini` emits pretty-printed, multi-line records, so every + // line failed to parse and every event - including the token usage - was + // dropped. Scanning for balanced values also covers records concatenated + // without a separator and records split across two process chunks. + const { records, rest } = takeJsonRecords(`${nextState.partialLine}${String(output ?? '')}`); + nextState.partialLine = rest; + + for (const record of records) { + for (const event of Array.isArray(record) ? record : [record]) { + applyGeminiJsonEvent(event, nextState, modelId); } } diff --git a/src/json-stream.lib.mjs b/src/json-stream.lib.mjs index 507ce85e2..e1f238094 100644 --- a/src/json-stream.lib.mjs +++ b/src/json-stream.lib.mjs @@ -149,6 +149,30 @@ export const createJsonStreamScanner = (options = {}) => { flush() { return scan(true, []); }, + /** The unconsumed tail: an incomplete record or text line. */ + pending() { + return pending; + }, + }; +}; + +/** + * Split a buffered stream into complete JSON records plus the unconsumed tail. + * + * Stateless counterpart of `createJsonStreamScanner`, for the parsers that keep + * their buffer inside a plain state object they hand back to their caller + * (`parseGeminiJsonOutput`, `parseQwenStreamJsonOutput`) instead of holding a + * closure across chunks. + * + * @param {string} buffered carried-over tail followed by the new chunk + * @returns {{records: Array, rest: string}} + */ +export const takeJsonRecords = buffered => { + const scanner = createJsonStreamScanner(); + const events = scanner.write(String(buffered ?? '')); + return { + records: events.filter(event => event.type === 'json').map(event => event.value), + rest: scanner.pending(), }; }; diff --git a/src/qwen.lib.mjs b/src/qwen.lib.mjs index 352b22b27..f697cf0cc 100644 --- a/src/qwen.lib.mjs +++ b/src/qwen.lib.mjs @@ -26,6 +26,7 @@ import { classifyRetryableError, prepareRetryAfterError, waitWithCountdown } fro import { getCumulativeContextInputTokens, getRestoredContextInputTokens, toTokenCount } from './context-fill.lib.mjs'; import { ensureAiToolScratchIgnored, filterAiToolScratchFromStatus } from './ai-tool-scratch.lib.mjs'; import { getTerminalEventCompletionHealth } from './tool-run-health.lib.mjs'; // Issue #1990 +import { takeJsonRecords } from './json-stream.lib.mjs'; // Issue #2119 export const mapModelToId = model => qwenModels[model] || model; @@ -328,35 +329,19 @@ export const parseQwenStreamJsonOutput = (output, state = {}) => { const text = output?.toString?.() ?? String(output || ''); nextState.plainText += text; - const parseCandidate = value => { - const trimmed = value.trim(); - if (!trimmed) return true; - - try { - const parsed = JSON.parse(trimmed); - if (Array.isArray(parsed)) { - for (const item of parsed) addQwenEventToState(nextState, item); - } else { - addQwenEventToState(nextState, parsed); - } - return true; - } catch { - return false; - } - }; - - const combined = `${nextState.buffer}${text}`; - nextState.buffer = ''; - - const lines = combined.split(/\r?\n/); - for (let index = 0; index < lines.length; index++) { - const line = lines[index]; - const isLastLine = index === lines.length - 1; - if (!line.trim()) continue; - - const parsed = parseCandidate(line); - if (!parsed && isLastLine) { - nextState.buffer = line; + // Issue #2119: frame the stream by balanced JSON values instead of by lines. + // `formal-ai with qwen` emits pretty-printed, multi-line records, so every + // line failed to parse and every event - including the token usage - was + // dropped. Scanning for balanced values also covers records concatenated + // without a separator and records split across two process chunks. + const { records, rest } = takeJsonRecords(`${nextState.buffer}${text}`); + nextState.buffer = rest; + + for (const record of records) { + if (Array.isArray(record)) { + for (const item of record) addQwenEventToState(nextState, item); + } else { + addQwenEventToState(nextState, record); } } diff --git a/tests/test-formal-ai-uniform-tools-2119.mjs b/tests/test-formal-ai-uniform-tools-2119.mjs new file mode 100644 index 000000000..1a70212ce --- /dev/null +++ b/tests/test-formal-ai-uniform-tools-2119.mjs @@ -0,0 +1,168 @@ +#!/usr/bin/env node +/** + * @hive-mind-test-suite default + * + * Regression coverage for issue #2119: "We should also ensure uniform support + * for gemini, and qwen for the Formal AI and Hive Mind." + * + * The reproduction runs used `--tool agent`, `--tool codex` and `--tool claude`, + * and the agent run published "Token usage: 0 input, 0 output" for a session + * that really used 21677 input / 22834 output tokens, because the stream reader + * split the output on newlines while `formal-ai with --verbose` emits + * pretty-printed, multi-line JSON records + * (docs/case-studies/issue-2119/data/logs/agent-scala-solution-draft.log). + * + * Commit 21197003 fixed that for agent, opencode and codex. Gemini and Qwen + * kept their own line-based parsers, so the very same defect was still present + * on those two tools - invisible only because nobody had run them yet. This + * test pins all of it down: the same stream shapes must produce the same token + * accounting and the same Link.Assistant $0.00 pricing on every tool. + */ + +import assert from 'node:assert/strict'; +import { readFile } from 'node:fs/promises'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +import { calculateAgentPricing } from '../src/agent.lib.mjs'; +import { buildGeminiPricingInfo, parseGeminiJsonOutput } from '../src/gemini.lib.mjs'; +import { buildQwenPricingInfo, parseQwenStreamJsonOutput } from '../src/qwen.lib.mjs'; +import { FORMAL_AI_SUPPORTED_TOOLS } from '../src/formal-ai.lib.mjs'; + +const repoRoot = path.join(path.dirname(fileURLToPath(import.meta.url)), '..'); + +/** The model alias that routes a tool through the local Link.Assistant server. */ +const FORMAL_AI_MODEL = 'formal-ai'; + +// --- gemini ------------------------------------------------------------------ + +const geminiResult = { + type: 'result', + session_id: 'gemini-formal-ai-session', + response: 'implemented hello world', + stats: { + models: { + [FORMAL_AI_MODEL]: { tokens: { input: 21677, output: 22834, total: 44511, contextLimit: 60000, outputLimit: 8192 } }, + }, + }, +}; + +const assertGeminiUsage = (state, label) => { + const usage = state.resultModelUsage?.[FORMAL_AI_MODEL]; + assert.ok(usage, `${label}: the result event must be parsed`); + assert.equal(usage.inputTokens, 21677, `${label}: input tokens must be counted`); + assert.equal(usage.outputTokens, 22834, `${label}: output tokens must be counted`); + assert.equal(state.sessionId, 'gemini-formal-ai-session', `${label}: session id must be captured`); + assert.equal(state.resultSummary, 'implemented hello world', `${label}: result text must be captured`); +}; + +// The shape `formal-ai with gemini --verbose` emits: indented, multi-line JSON. +assertGeminiUsage(parseGeminiJsonOutput(`${JSON.stringify(geminiResult, null, 2)}\n`, {}, FORMAL_AI_MODEL), 'gemini pretty-printed'); + +// Strict NDJSON, as gemini-cli emits on its own, must keep working. +assertGeminiUsage(parseGeminiJsonOutput(`${JSON.stringify(geminiResult)}\n`, {}, FORMAL_AI_MODEL), 'gemini ndjson'); + +// A pretty-printed record split across two process chunks must be assembled. +const geminiSerialized = JSON.stringify(geminiResult, null, 2); +const geminiSplitAt = geminiSerialized.indexOf('"stats"'); +let geminiChunked = parseGeminiJsonOutput(geminiSerialized.slice(0, geminiSplitAt), {}, FORMAL_AI_MODEL); +assert.equal(geminiChunked.resultModelUsage, null, 'gemini: an incomplete record must not be applied yet'); +geminiChunked = parseGeminiJsonOutput(`${geminiSerialized.slice(geminiSplitAt)}\n`, geminiChunked, FORMAL_AI_MODEL); +assertGeminiUsage(geminiChunked, 'gemini split across chunks'); + +// Records concatenated without a separator (issue #1250) must both be applied. +const geminiConcatenated = parseGeminiJsonOutput([JSON.stringify({ type: 'message', content: 'working' }), JSON.stringify({ type: 'tool_use', toolCall: { name: 'write_file' } }), JSON.stringify(geminiResult)].join(''), {}, FORMAL_AI_MODEL); +assert.equal(geminiConcatenated.messageCount, 2, 'gemini: concatenated message events must be counted'); +assert.equal(geminiConcatenated.toolUseCount, 1, 'gemini: concatenated tool events must be counted'); +assertGeminiUsage(geminiConcatenated, 'gemini concatenated'); + +// Plain, non-JSON tool chatter around the records must not break the framing. +assertGeminiUsage(parseGeminiJsonOutput(`Loaded cached credentials.\n${geminiSerialized}\nDone.\n`, {}, FORMAL_AI_MODEL), 'gemini mixed with plain text'); + +// The session belongs to Link.Assistant and costs nothing. +const geminiPricing = buildGeminiPricingInfo(FORMAL_AI_MODEL); +assert.equal(geminiPricing.provider, 'Link.Assistant', 'gemini: formal-ai runs are attributed to Link.Assistant'); +assert.equal(geminiPricing.totalCostUSD, 0, 'gemini: formal-ai runs are free'); + +// --- qwen -------------------------------------------------------------------- + +const qwenResult = { + type: 'result', + session_id: 'qwen-formal-ai-session', + result: 'implemented hello world', + usage: { model: FORMAL_AI_MODEL, inputTokens: 21677, outputTokens: 22834, contextLimit: 60000, outputLimit: 8192 }, +}; + +const assertQwenUsage = (state, label) => { + assert.equal(state.tokenUsage.stepCount, 1, `${label}: the result event must be parsed`); + assert.equal(state.tokenUsage.inputTokens, 21677, `${label}: input tokens must be counted`); + assert.equal(state.tokenUsage.outputTokens, 22834, `${label}: output tokens must be counted`); + assert.equal(state.sessionId, 'qwen-formal-ai-session', `${label}: session id must be captured`); + assert.equal(state.lastTextContent, 'implemented hello world', `${label}: result text must be captured`); +}; + +assertQwenUsage(parseQwenStreamJsonOutput(`${JSON.stringify(qwenResult, null, 2)}\n`), 'qwen pretty-printed'); +assertQwenUsage(parseQwenStreamJsonOutput(`${JSON.stringify(qwenResult)}\n`), 'qwen ndjson'); + +const qwenSerialized = JSON.stringify(qwenResult, null, 2); +const qwenSplitAt = qwenSerialized.indexOf('"usage"'); +let qwenChunked = parseQwenStreamJsonOutput(qwenSerialized.slice(0, qwenSplitAt)); +assert.equal(qwenChunked.tokenUsage.stepCount, 0, 'qwen: an incomplete record must not be applied yet'); +qwenChunked = parseQwenStreamJsonOutput(`${qwenSerialized.slice(qwenSplitAt)}\n`, qwenChunked); +assertQwenUsage(qwenChunked, 'qwen split across chunks'); + +const qwenConcatenated = parseQwenStreamJsonOutput([JSON.stringify({ type: 'session.started', session_id: 'ignored' }), JSON.stringify(qwenResult)].join('')); +assert.equal(qwenConcatenated.eventCounts['session.started'], 1, 'qwen: concatenated records must both be applied'); +assert.equal(qwenConcatenated.tokenUsage.stepCount, 1, 'qwen: concatenated usage must be counted'); + +assertQwenUsage(parseQwenStreamJsonOutput(`Loaded cached credentials.\n${qwenSerialized}\nDone.\n`), 'qwen mixed with plain text'); + +const qwenPricing = buildQwenPricingInfo(parseQwenStreamJsonOutput(`${qwenSerialized}\n`), FORMAL_AI_MODEL); +assert.equal(qwenPricing.pricingInfo.provider, 'Link.Assistant', 'qwen: formal-ai runs are attributed to Link.Assistant'); +assert.equal(qwenPricing.pricingInfo.totalCostUSD, 0, 'qwen: formal-ai runs are free'); +assert.equal(qwenPricing.publicPricingEstimate, 0, 'qwen: the public estimate for a free model is $0.00, not "unknown"'); + +// A tool run on its own model keeps its own provider - the fix must not make +// every session look like Link.Assistant. +assert.equal(buildGeminiPricingInfo('gemini-2.5-pro').provider, 'Google'); +assert.equal(buildQwenPricingInfo(parseQwenStreamJsonOutput('{"type":"result","result":"ok","usage":{"model":"qwen3-coder-plus","inputTokens":5,"outputTokens":7}}\n'), 'qwen3-coder-plus').pricingInfo.provider, 'Qwen Code'); + +// --- uniformity across every formal-ai tool ---------------------------------- + +assert.deepEqual(FORMAL_AI_SUPPORTED_TOOLS, ['claude', 'agent', 'opencode', 'codex', 'qwen', 'gemini'], 'the supported tool list is the contract this test covers'); + +// Every tool that parses a JSON event stream must share one framing +// implementation, so a fix lands on all of them at once instead of being +// rediscovered per tool - which is exactly how gemini and qwen were missed. +const streamParsingLibs = ['gemini.lib.mjs', 'qwen.lib.mjs', 'agent.lib.mjs', 'opencode.lib.mjs', 'codex.lib.mjs']; +for (const file of streamParsingLibs) { + const source = await readFile(path.join(repoRoot, 'src', file), 'utf8'); + assert.ok(source.includes("from './json-stream.lib.mjs'"), `${file} frames its stream with the shared scanner`); +} + +// Every tool must route `--model formal-ai` through the same dispatcher and +// reach the same pricing helper, so provider and cost are reported identically. +// `opencode` reaches it indirectly: it prices through agent.lib.mjs, which is +// where the formal-ai short-circuit lives for both tools. +const PRICING_ENTRY_POINT_BY_TOOL = { + claude: 'applyFormalAiPricingOverride', + agent: 'buildFormalAiPricingInfo', + opencode: 'calculateAgentPricing', + codex: 'buildFormalAiPricingInfo', + qwen: 'buildFormalAiPricingInfo', + gemini: 'buildFormalAiPricingInfo', +}; + +for (const tool of FORMAL_AI_SUPPORTED_TOOLS) { + const source = await readFile(path.join(repoRoot, 'src', `${tool}.lib.mjs`), 'utf8'); + assert.ok(source.includes('resolveFormalAiToolInvocation'), `${tool}.lib.mjs dispatches through formal-ai.lib.mjs`); + assert.ok(source.includes(PRICING_ENTRY_POINT_BY_TOOL[tool]), `${tool}.lib.mjs prices formal-ai sessions through ${PRICING_ENTRY_POINT_BY_TOOL[tool]}`); +} + +// The shared entry point opencode relies on must itself be free of the +// OpenCode Zen attribution the issue reported. +const opencodePricing = await calculateAgentPricing(FORMAL_AI_MODEL, { inputTokens: 21677, outputTokens: 22834, stepCount: 1 }); +assert.equal(opencodePricing.provider, 'Link.Assistant', 'opencode/agent: formal-ai runs are not attributed to OpenCode Zen'); +assert.equal(opencodePricing.totalCostUSD, 0, 'opencode/agent: formal-ai runs are free'); + +console.log(`โœ… issue #2119: formal-ai stream parsing and pricing are uniform across ${FORMAL_AI_SUPPORTED_TOOLS.length} tools`); From a40825e2da2876ab69177ba24c7975f61784c82f Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 16:49:23 +0000 Subject: [PATCH 12/17] fix(2119): do not count the solver's own placeholder as a change The Rust reproduction PR contains exactly one file - hive-mind's own auto-generated .gitkeep placeholder - so the change stats reported "1 file changed" for a run that implemented nothing. pull-request-changes.lib.mjs now splits the diff per file and drops the sections whose content matches what solve.auto-pr.lib.mjs writes into .gitkeep / CLAUDE.md, exposes placeholderOnly, and offers buildEmptyPullRequestBlocker() so the restart reason names the placeholder. Matching is on the generated content, so a repository's own .gitkeep stays a real change. --- src/pull-request-changes.lib.mjs | 86 +++++++++++++++++++++++--- src/solve.auto-merge.lib.mjs | 9 +-- tests/test-empty-pull-request-2119.mjs | 55 +++++++++++++++- 3 files changed, 136 insertions(+), 14 deletions(-) diff --git a/src/pull-request-changes.lib.mjs b/src/pull-request-changes.lib.mjs index 6a6536c06..61506289d 100644 --- a/src/pull-request-changes.lib.mjs +++ b/src/pull-request-changes.lib.mjs @@ -21,12 +21,57 @@ * and never revisited), and the Kotlin run went on to post "โœ… Ready to merge - * No pending changes" for a pull request that changed nothing at all. * + * The third reproduction run failed before the AI committed anything, so its + * pull request kept the scaffolding file itself: + * + * https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/pull/2 + * .gitkeep | 1 + + * + * That is the same "nothing was implemented" state wearing a file count, so the + * solver's own placeholder is excluded from the counts here rather than being + * reported as the AI's work. + * * This module is the single place that answers the question, so both the * description writer and the mergeability watcher agree. */ import { ghWithRateLimitRetry } from './github-rate-limit.lib.mjs'; +/** + * The solver's own scaffolding files, recognised by the content it writes into + * them (`src/solve.auto-pr.lib.mjs`). A pull request whose whole diff is one of + * these contains no solution: the placeholder exists only to give an empty + * branch something to open a pull request from, and is reverted once the AI + * commits real work. + * + * Matching on content, not on the file name, keeps a repository's own + * `.gitkeep` or `CLAUDE.md` edits counted as the real changes they are. + */ +const PLACEHOLDER_CONTENT_PATTERNS = new Map([ + ['.gitkeep', [/^\+#\s*\.gitkeep file auto-generated at .+ for PR creation at branch /m]], + ['CLAUDE.md', [/^\+Issue to solve: \S+/m, /^\+Your prepared branch: \S+/m]], +]); + +/** Split a unified diff into one section per file. */ +const splitDiffByFile = diff => { + const sections = []; + for (const line of diff.split('\n')) { + if (line.startsWith('diff --git ')) { + const match = /^diff --git a\/(.+) b\/(.+)$/.exec(line); + sections.push({ path: match ? match[2] : '', body: '' }); + continue; + } + if (sections.length > 0) sections[sections.length - 1].body += `${line}\n`; + } + return sections; +}; + +/** True when this file section is nothing but the solver's own placeholder. */ +const isPlaceholderSection = section => { + const patterns = PLACEHOLDER_CONTENT_PATTERNS.get(section.path); + return Boolean(patterns) && patterns.every(pattern => pattern.test(section.body)); +}; + /** * Measure the net diff of a pull request. * @@ -39,9 +84,11 @@ import { ghWithRateLimitRetry } from './github-rate-limit.lib.mjs'; * @param {string} params.repo * @param {number} params.prNumber * @param {Function} params.$ command-stream tagged-template executor - * @returns {Promise<{hasChanges: boolean, filesChanged: number, additions: number, deletions: number, measured: boolean}>} - * `measured` is false when the diff could not be fetched, in which case - * callers must not treat the pull request as empty. + * @returns {Promise<{hasChanges: boolean, filesChanged: number, additions: number, deletions: number, placeholderOnly: boolean, measured: boolean}>} + * The counts cover the AI's own work: the solver's placeholder file is + * excluded and reported through `placeholderOnly` instead. `measured` is + * false when the diff could not be fetched, in which case callers must not + * treat the pull request as empty. */ export const getPullRequestChangeStats = async ({ owner, repo, prNumber, $ }) => { let diffOutput = ''; @@ -56,11 +103,23 @@ export const getPullRequestChangeStats = async ({ owner, repo, prNumber, $ }) => // Leave measured false: an unreachable API must not read as "no changes". } - const filesChanged = (diffOutput.match(/^diff --git/gm) || []).length; - const additions = (diffOutput.match(/^\+[^+]/gm) || []).length; - const deletions = (diffOutput.match(/^-[^-]/gm) || []).length; + const sections = splitDiffByFile(diffOutput); + const placeholderSections = sections.filter(isPlaceholderSection); + const realSections = sections.filter(section => !isPlaceholderSection(section)); - return { hasChanges: filesChanged > 0, filesChanged, additions, deletions, measured }; + const countMatches = (pattern, text) => (text.match(pattern) || []).length; + const filesChanged = realSections.length; + const additions = realSections.reduce((total, section) => total + countMatches(/^\+[^+]/gm, section.body), 0); + const deletions = realSections.reduce((total, section) => total + countMatches(/^-[^-]/gm, section.body), 0); + + return { + hasChanges: filesChanged > 0, + filesChanged, + additions, + deletions, + placeholderOnly: filesChanged === 0 && placeholderSections.length > 0, + measured, + }; }; /** @@ -78,6 +137,9 @@ export const formatChangeSummary = stats => { return '- The diff could not be read, so the change summary is unavailable'; } if (!stats.hasChanges) { + if (stats.placeholderOnly) { + return '- No files were changed by this pull request yet (it contains only the placeholder file the solver commits to open a pull request)'; + } return '- No files were changed by this pull request yet'; } return [`- ${stats.filesChanged} file(s) modified`, `- ${stats.additions} line(s) added`, `- ${stats.deletions} line(s) removed`].join('\n'); @@ -93,4 +155,12 @@ export const formatChangeSummary = stats => { */ export const EMPTY_PULL_REQUEST_BLOCKER = 'The pull request contains no changes (its net diff is empty), so there is nothing to merge'; -export default { getPullRequestChangeStats, formatChangeSummary, EMPTY_PULL_REQUEST_BLOCKER }; +/** + * The same blocker, naming the placeholder when that is all the diff contains. + * + * @param {{placeholderOnly?: boolean}|null} stats + * @returns {string} + */ +export const buildEmptyPullRequestBlocker = (stats = null) => (stats?.placeholderOnly ? 'The pull request contains only the placeholder file the solver commits to open a pull request, so there is nothing to merge' : EMPTY_PULL_REQUEST_BLOCKER); + +export default { getPullRequestChangeStats, formatChangeSummary, EMPTY_PULL_REQUEST_BLOCKER, buildEmptyPullRequestBlocker }; diff --git a/src/solve.auto-merge.lib.mjs b/src/solve.auto-merge.lib.mjs index 05e7c8eba..96258ef8b 100644 --- a/src/solve.auto-merge.lib.mjs +++ b/src/solve.auto-merge.lib.mjs @@ -91,7 +91,7 @@ const { failOnAutoRestartBudgetExhausted } = await import('./auto-restart-exhaus const { ensurePullRequestBaseBranch } = await import('./solve.pr-base-guard.lib.mjs'); // Issue #2119: an empty pull request must not be reported as ready to merge. -const { EMPTY_PULL_REQUEST_BLOCKER, getPullRequestChangeStats } = await import('./pull-request-changes.lib.mjs'); +const { buildEmptyPullRequestBlocker, getPullRequestChangeStats } = await import('./pull-request-changes.lib.mjs'); // Issue #1895: explicitly close linked issues after merging a PR into a // non-default branch, where GitHub does not auto-close them. @@ -295,8 +295,9 @@ export const watchUntilMergeable = async params => { // the issue without implementing anything. const changeStats = await getPullRequestChangeStats({ owner, repo, prNumber, $ }); const isEmptyPullRequest = changeStats.measured && !changeStats.hasChanges; + const emptyPullRequestBlocker = buildEmptyPullRequestBlocker(changeStats); if (isEmptyPullRequest) { - await log(formatAligned('โš ๏ธ', 'PR is empty:', 'net diff contains no files - not treating it as mergeable', 2), { level: 'warning' }); + await log(formatAligned('โš ๏ธ', 'PR is empty:', changeStats.placeholderOnly ? 'only the solver placeholder file is in the diff - not treating it as mergeable' : 'net diff contains no files - not treating it as mergeable', 2), { level: 'warning' }); } // If PR is mergeable, no blockers, no new comments, no issue metadata @@ -459,8 +460,8 @@ export const watchUntilMergeable = async params => { // Issue #2119: Reason 1a: the pull request does not change anything yet. if (isEmptyPullRequest) { shouldRestart = true; - restartReason = restartReason ? `${restartReason}; ${EMPTY_PULL_REQUEST_BLOCKER}` : EMPTY_PULL_REQUEST_BLOCKER; - feedbackLines.push(`๐Ÿ“ญ ${EMPTY_PULL_REQUEST_BLOCKER}.`); + restartReason = restartReason ? `${restartReason}; ${emptyPullRequestBlocker}` : emptyPullRequestBlocker; + feedbackLines.push(`๐Ÿ“ญ ${emptyPullRequestBlocker}.`); feedbackLines.push(''); feedbackLines.push('Implement the requested change and commit it to the pull request branch. Do not report the work as done while the diff is empty.'); } diff --git a/tests/test-empty-pull-request-2119.mjs b/tests/test-empty-pull-request-2119.mjs index 3ca7c9704..b4a91b982 100644 --- a/tests/test-empty-pull-request-2119.mjs +++ b/tests/test-empty-pull-request-2119.mjs @@ -20,6 +20,14 @@ * pending changes". Merging that would have closed the issue with nothing * implemented, which is the worst kind of false positive: it looks like success. * + * The third run stopped before the AI committed anything, so its pull request + * kept the solver's own placeholder file - `.gitkeep | 1 +` and nothing else: + * + * https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/pull/2 + * + * That is the same "nothing was implemented" state wearing a file count, so the + * placeholder is excluded from the counts too. + * * @hive-mind-test-suite default */ @@ -28,7 +36,7 @@ import { readFile } from 'node:fs/promises'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; -import { EMPTY_PULL_REQUEST_BLOCKER, formatChangeSummary, getPullRequestChangeStats } from '../src/pull-request-changes.lib.mjs'; +import { EMPTY_PULL_REQUEST_BLOCKER, buildEmptyPullRequestBlocker, formatChangeSummary, getPullRequestChangeStats } from '../src/pull-request-changes.lib.mjs'; const repoRoot = path.join(path.dirname(fileURLToPath(import.meta.url)), '..'); @@ -55,6 +63,49 @@ assert.equal(changed.hasChanges, true, 'a pull request that adds a file has chan assert.equal(changed.filesChanged, 1); assert.equal(changed.additions, 3, 'the `+++ b/...` header is not counted as an added line'); +// --- the solver's own placeholder is not a change ---------------------------- + +// The third reproduction run never got as far as a commit, so its pull request +// still holds the placeholder `src/solve.auto-pr.lib.mjs` writes to make an +// empty branch openable: +// https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/pull/2 +const gitkeepPlaceholderDiff = ['diff --git a/.gitkeep b/.gitkeep', 'new file mode 100644', 'index 0000000..b5cf1f4', '--- /dev/null', '+++ b/.gitkeep', '@@ -0,0 +1 @@', '+# .gitkeep file auto-generated at 2026-07-30T14:24:59.267Z for PR creation at branch issue-1-09b0c76bd0e4 for issue https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/issues/1'].join('\n'); +const claudeMdPlaceholderDiff = ['diff --git a/CLAUDE.md b/CLAUDE.md', 'new file mode 100644', '--- /dev/null', '+++ b/CLAUDE.md', '@@ -0,0 +1,3 @@', '+Issue to solve: https://github.com/konard/test-hello-world/issues/1', '+Your prepared branch: issue-1-09b0c76bd0e4', '+Proceed.'].join('\n'); + +for (const [label, diff] of [ + ['.gitkeep', gitkeepPlaceholderDiff], + ['CLAUDE.md', claudeMdPlaceholderDiff], +]) { + const stats = await getPullRequestChangeStats({ owner: 'konard', repo: 'test-hello-world', prNumber: 2, $: fake$({ stdout: diff }) }); + assert.equal(stats.hasChanges, false, `${label}: a pull request holding only the solver placeholder implements nothing`); + assert.equal(stats.filesChanged, 0, `${label}: the placeholder is not counted as the AI's work`); + assert.equal(stats.additions, 0, `${label}: the placeholder's lines are not counted either`); + assert.equal(stats.placeholderOnly, true, `${label}: the caller can tell an empty diff from a placeholder-only one`); + assert.ok(formatChangeSummary(stats).includes('placeholder'), `${label}: the description names the placeholder instead of claiming a file was modified`); + assert.ok(buildEmptyPullRequestBlocker(stats).includes('placeholder'), `${label}: the restart reason names the placeholder`); +} + +// Matching is on the generated content, so a repository's own `.gitkeep` or +// `CLAUDE.md` stays the real change it is. +const ownGitkeepDiff = ['diff --git a/docs/.gitkeep b/.gitkeep', 'new file mode 100644', '--- /dev/null', '+++ b/.gitkeep', '@@ -0,0 +1 @@', '+keep this directory'].join('\n'); +const ownGitkeep = await getPullRequestChangeStats({ owner: 'konard', repo: 'test-hello-world', prNumber: 2, $: fake$({ stdout: ownGitkeepDiff }) }); +assert.equal(ownGitkeep.hasChanges, true, 'a `.gitkeep` without the auto-generated marker is a real change'); +assert.equal(ownGitkeep.filesChanged, 1); +assert.equal(ownGitkeep.placeholderOnly, false); + +// A real file next to the placeholder counts once: the placeholder drops out, +// the solution stays. +const mixed = await getPullRequestChangeStats({ owner: 'konard', repo: 'test-hello-world', prNumber: 2, $: fake$({ stdout: `${gitkeepPlaceholderDiff}\n${realDiff}` }) }); +assert.equal(mixed.hasChanges, true, 'the placeholder does not hide real work'); +assert.equal(mixed.filesChanged, 1, 'only the real file is counted'); +assert.equal(mixed.additions, 3, 'the placeholder line is excluded from the addition count'); +assert.equal(mixed.placeholderOnly, false, 'a pull request with real work is not placeholder-only'); + +// An empty diff is empty, not placeholder-only - the two get different wording. +assert.equal(empty.placeholderOnly, false); +assert.equal(buildEmptyPullRequestBlocker(empty), EMPTY_PULL_REQUEST_BLOCKER); +assert.equal(buildEmptyPullRequestBlocker(), EMPTY_PULL_REQUEST_BLOCKER, 'the blocker has a sensible default'); + // An unreachable API must never be mistaken for "nothing changed" - that would // turn a transient network failure into an endless restart loop. for (const broken of [fake$({ code: 1 }), fake$({ throws: true })]) { @@ -81,7 +132,7 @@ const autoMergeSource = await readFile(path.join(repoRoot, 'src', 'solve.auto-me assert.ok(autoMergeSource.includes("await import('./pull-request-changes.lib.mjs')"), 'the auto-merge watcher measures the diff'); assert.ok(autoMergeSource.includes('const isEmptyPullRequest = changeStats.measured && !changeStats.hasChanges'), 'an unmeasured diff does not count as empty'); assert.ok(/!hasUncommittedChanges && !isEmptyPullRequest/.test(autoMergeSource), 'the "ready to merge" branch is gated on the pull request not being empty'); -assert.ok(autoMergeSource.includes('EMPTY_PULL_REQUEST_BLOCKER'), 'an empty pull request is reported as a restart reason'); +assert.ok(autoMergeSource.includes('buildEmptyPullRequestBlocker(changeStats)'), 'an empty pull request is reported as a restart reason, naming the placeholder when that is all there is'); const resultsSource = await readFile(path.join(repoRoot, 'src', 'solve.results.lib.mjs'), 'utf8'); assert.ok(resultsSource.includes('formatChangeSummary(changeStats)'), 'the generated description renders the shared change summary'); From 75110d52ec456996cae7f9683e50dfb767e9473e Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 16:58:19 +0000 Subject: [PATCH 13/17] docs(2119): deep case study of the formal-ai reproduction runs Compiles every artifact of the three 2026-07-30 reproduction runs into docs/case-studies/issue-2119: the issue and its comments, all six test-hello-world repositories (three 2025 baselines for contrast), the three session logs, the PR/issue JSON, the net diffs and the URLs of all 13 published session logs. README.md reconstructs the timeline of each run, lists every requirement from the issue with where it is addressed, gives the root cause and fix of all 13 Hive Mind defects, records the two deliberate non-changes and why, reviews the existing libraries considered for each fix, and states where the data was insufficient and what instrumentation was added. upstream-formal-ai.md is the consolidated Formal AI report: five defects with reproductions, workarounds and code-level suggestions, framed as a request for generalization and self-healing rather than per-symptom patches. --- docs/case-studies/issue-2119/README.md | 431 ++++++++++++++++++ .../issue-2119/data/issue-2119-comments.json | 1 + .../issue-2119/data/issue-2119.json | 11 + .../issue-2119/data/log-gists.json | 17 + .../pr-2.diff | 0 .../pr-2.diff | 0 .../pr-2.diff | 8 + ...1e68-059d-749e-900f-af9e7aa44451-pr-2.json | 24 + ...1e7d-f3e2-713e-a22a-0e69a326dc93-pr-2.json | 12 + ...2020-00f8-7cf2-9bb6-a1c2a7718de5-pr-2.json | 24 + .../data/test-hello-world-repos.json | 8 + .../issue-2119/upstream-formal-ai.md | 212 +++++++++ 12 files changed, 748 insertions(+) create mode 100644 docs/case-studies/issue-2119/README.md create mode 100644 docs/case-studies/issue-2119/data/issue-2119-comments.json create mode 100644 docs/case-studies/issue-2119/data/issue-2119.json create mode 100644 docs/case-studies/issue-2119/data/log-gists.json create mode 100644 docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2.diff create mode 100644 docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2.diff create mode 100644 docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2.diff create mode 100644 docs/case-studies/issue-2119/data/prs/baseline-2025/01991e68-059d-749e-900f-af9e7aa44451-pr-2.json create mode 100644 docs/case-studies/issue-2119/data/prs/baseline-2025/01991e7d-f3e2-713e-a22a-0e69a326dc93-pr-2.json create mode 100644 docs/case-studies/issue-2119/data/prs/baseline-2025/01992020-00f8-7cf2-9bb6-a1c2a7718de5-pr-2.json create mode 100644 docs/case-studies/issue-2119/data/test-hello-world-repos.json create mode 100644 docs/case-studies/issue-2119/upstream-formal-ai.md diff --git a/docs/case-studies/issue-2119/README.md b/docs/case-studies/issue-2119/README.md new file mode 100644 index 000000000..d309db85c --- /dev/null +++ b/docs/case-studies/issue-2119/README.md @@ -0,0 +1,431 @@ +# Case study โ€” issue #2119: `--model formal-ai` not working on the simplest hello world + +- Issue: https://github.com/link-assistant/hive-mind/issues/2119 +- Pull request: https://github.com/link-assistant/hive-mind/pull/2120 +- Raw evidence: [`data/`](./data) (logs, PR/issue JSON, diffs, repository inventory) +- Date of the reproduction runs: **2026-07-30**, 13:21โ€“14:35 UTC + +## 1. What happened, in one paragraph + +Three `hive-mind solve` runs were started against three freshly generated hello-world +repositories, each with a different `--tool` but the same `--model formal-ai`. None of +them produced a line of source code. That part is a Formal AI problem, reported +upstream. What this case study is about is the second, larger failure: **Hive Mind +described those three empty runs as successes.** It billed a free model at $0.252315, +attributed Link.Assistant's model to "OpenCode Zen" and "Anthropic", reported +`0 input, 0 output` tokens for a session that used 44 511, published `1 file(s) +modified` for an empty diff, posted `โœ… Ready to merge` on a pull request that changed +nothing, ran eleven AI sessions on a hello-world because two independent restart loops +shared one limit flag, and finally exited `0`. Every visible signal said "done". + +Thirteen defects follow from that, all of them in Hive Mind, all of them fixed in +PR #2120 with a regression test each. + +## 2. The reproduction inventory + +Six `konard/test-hello-world-*` repositories have ever existed +([`data/test-hello-world-repos.json`](./data/test-hello-world-repos.json)). Three are +the 2025 baseline, three are the 2026 formal-ai runs. The contrast is the whole story: + +| Created | Repository (UUIDv7 suffix) | Tool | Language | PR #2 title | Files | + | +| ---------------- | -------------------------------------- | -------- | ----------- | ---------------------------------------------------------- | ----- | --- | +| 2025-09-06 09:42 | `01991e68-059d-749e-900f-af9e7aa44451` | baseline | Rust | Implement Hello World in Rust with GitHub Actions | 2 | 69 | +| 2025-09-06 10:06 | `01991e7d-f3e2-713e-a22a-0e69a326dc93` | baseline | Common Lisp | Implement Hello World in Common Lisp | 2 | 45 | +| 2025-09-06 17:43 | `01992020-00f8-7cf2-9bb6-a1c2a7718de5` | baseline | COBOL | Implement Hello World program in COBOL with CI/CD workflow | 2 | 81 | +| 2026-07-30 13:21 | `019fb330-00e1-73b9-955e-f357a1600d5b` | `agent` | Scala | `'Implement Hello World in Scala'` | **0** | 0 | +| 2026-07-30 13:22 | `019fb330-fa49-7c9d-a664-b7ea33bb698a` | `claude` | Kotlin | `'Implement Hello World in Kotlin'` | **0** | 0 | +| 2026-07-30 13:23 | `019fb331-c107-78c7-8ff6-9f127a3c593c` | `codex` | Rust | `[WIP] Implement Hello World in Rust` | **1** | 1 | + +The 2025 baseline shows the pipeline works end to end when the AI tool answers: two +files, a real title, tens of added lines. The 2026 runs produce zero โ€” and the one file +in the Rust run is Hive Mind's own `.gitkeep` placeholder, not code +([`data/prs/019fb331-.../pr-2.diff`](./data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2.diff)). + +The repositories are generated by `create-test-repo.mjs` in the repository root (UUIDv7 +name, language drawn at random from a 40-entry list) and removed by +`cleanup-test-repos.mjs`. + +## 3. Timeline of events + +All times UTC, 2026-07-30. Sources: the PR conversation comments archived under +[`data/prs/`](./data/prs), and the three session logs under [`data/logs/`](./data/logs). + +### 3.1 Run A โ€” `--tool agent`, Scala (`019fb330-00e1-...`) + +| Time | Event | +| ----------- | ------------------------------------------------------------------------------------------------------------------ | +| 13:21 | Test repository created | +| 14:13 | PR #2 opened, titled `'Implement Hello World in Scala'` (quotes literal), body claims `1 file(s) modified` | +| 14:14 | Solution draft log: `Provider: OpenCode Zen`, `Public pricing estimate: unknown`, `Token usage: 0 input, 0 output` | +| 14:14 | `๐Ÿ”„ Auto-restart 1/5` โ€” "Detected uncommitted changes", listing `?? .formal-ai/` and `?? examples` | +| 14:15 | `๐Ÿ”„ Auto-restart 1/5 Log` โ€” same zeroed cost block | +| 14:15โ€“14:18 | Auto-restart 2/5, 3/5, 4/5, 5/5 โ€” each restart re-creates `.formal-ai/`, so the blocker never clears | +| 14:20 | The **second** restart system starts: `๐Ÿ”„ Auto-restart triggered (iteration 1)` | +| 14:21 | `๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 1)` | +| 14:23โ€“14:33 | iterations 2, 3, 4, 5 of the second system | +| 14:35 | `โš ๏ธ Auto-restart limit reached` โ€” "Remaining reason: Uncommitted changes detected". Process exits **0** | + +**Eleven** AI sessions, 21 minutes, `--auto-restart-max-iterations 5`, zero lines of +Scala, and an exit status that says success. + +### 3.2 Run B โ€” `--tool claude`, Kotlin (`019fb330-fa49-...`) + +| Time | Event | +| ----- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 14:19 | PR #2 opened, titled `'Implement Hello World in Kotlin'` | +| 14:20 | `Working session summary`: "The `pwd` command completed. Output: `/tmp/gh-issue-solver-1785421161275`" โ€” the whole session was one `pwd` (`num_turns: 2`), and the solver's private workspace path was published | +| 14:20 | Solution draft log: `Calculated by Anthropic: $0.252315` for a model that is free | +| 14:22 | `## โœ… Ready to merge` โ€” "No CI/CD checks are configured", "No merge conflicts", "No pending changes" | + +The branch holds `65a1be4d Initial commit with task details` and +`e411d44f Revert "Initial commit with task details"`, so `gh pr diff` prints nothing. +Merging on that recommendation would have closed the issue with nothing implemented โ€” +the most expensive kind of false positive, because it looks like success. + +### 3.3 Run C โ€” `--tool codex`, Rust (`019fb331-c107-...`) + +| Time | Event | +| ----- | ---------------------------------------------------------------------------------------- | +| 14:25 | PR #2 opened as `[WIP] Implement Hello World in Rust`, containing only `.gitkeep \| 1 +` | +| 14:25 | `๐Ÿšจ Solution Draft Failed` โ€” "Reason: Authentication error" | + +The failure log ([`data/logs/codex-rust-failure.log`](./data/logs/codex-rust-failure.log)) +shows codex talking to `https://api.openai.com/v1/responses` and receiving +`401 Missing bearer or basic authentication in header`: the `formal-ai` alias never +reached the local Link.Assistant server. This run is the one whose published logs also +showed redacted token counters, because the credential sanitizer treated +`"total_tokens":44511` as a secret. + +## 4. Requirements extracted from the issue + +| # | Requirement (verbatim intent) | Where it is addressed | +| --- | --------------------------------------------------------------------------------------------- | ----------------------------------------------------------------- | +| R1 | Provider must be `Link.Assistant`, not `OpenCode Zen` / `Anthropic` | D1 โ€” `8f7f173a` | +| R2 | The model is free: cost must be $0 | D2, D3 โ€” `8f7f173a` | +| R3 | Fix all other false positives, false negatives, errors and warnings in the comments and logs | D4โ€“D13 | +| R4 | One auto-restart system, `N/M` labelled, hard stop after the limit | D5 โ€” `f8d4c91c` | +| R5 | After the limit: actually fail, with auto-commit on fail recovery, so the result is visible | D6 โ€” `f8d4c91c` | +| R6 | Solve everything belonging to Hive Mind in this pull request | ยง5, all 13 defects | +| R7 | Report everything belonging to Formal AI upstream, asking for generalization and self-healing | ยง8, [`upstream-formal-ai.md`](./upstream-formal-ai.md) | +| R8 | Find all other `test-hello-world-*` examples; guarantee quality with tests | ยง2 (all six found), ยง7 (test inventory) | +| R9 | Uniform support for gemini and qwen | D4b โ€” `16d8b243`, `tests/test-formal-ai-uniform-tools-2119.mjs` | +| R10 | Collect all logs; single big upstream issue with all materials linked | [`data/`](./data), ยง8 | +| R11 | Compile the data into `./docs/case-studies/issue-2119` and do a deep analysis | this document | +| R12 | Where data is insufficient, add debug output / verbose mode for the next iteration | ยง9 | +| R13 | Report to other repositories where applicable, with reproductions and fix suggestions | ยง8 (Formal AI; `link-assistant/agent` covered by the same report) | +| R14 | Apply every fix across the entire codebase, not just where it was observed | ยง5 โ€” each fix names its sweep | + +## 5. The thirteen defects: evidence, root cause, fix + +### D1 โ€” The provider of a Link.Assistant model was reported as someone else + +**Evidence.** `Provider: OpenCode Zen` (agent run), `Tool: Anthropic Claude Code / +Calculated by Anthropic` (claude run). + +**Root cause.** Pricing and attribution were derived from the _tool_ that ran, not from +the _model_ that answered. Each tool library had its own hard-coded provider string: +`agent.lib.mjs` and `opencode.lib.mjs` said "OpenCode Zen", `claude.lib.mjs` took +Anthropic's own `total_cost_usd` field verbatim. `formal-ai` is an alias that routes any +of the six tools through a local Link.Assistant server, so the tool name says nothing +about who served the tokens. + +**Fix (`8f7f173a`).** `src/formal-ai-pricing.lib.mjs` is the single source of the +provider name and the zero-cost override, applied at the cost boundary of all six tools +plus `agent-commander` and the cost-info renderer. `isFormalAiModel(modelId)` +short-circuits before any tool-specific pricing runs. + +### D2 โ€” A free model was billed $0.252315 + +**Root cause.** Same as D1: `claude.lib.mjs` published Anthropic's `total_cost_usd` +without asking which model produced it. + +**Fix.** `applyFormalAiPricingOverride` zeroes the cost for formal-ai sessions; +`buildFormalAiPricingInfo` builds the equivalent for the other five tools. + +### D3 โ€” "Public pricing estimate: unknown" for a model whose price is known to be zero + +**Root cause.** The renderer printed `unknown` for any model missing from the public +price table. A free model is not an unknown-price model. + +**Fix.** `src/github-cost-info.lib.mjs` prints `$0.00 (Free model)`. + +### D4 โ€” `Token usage: 0 input, 0 output` for a session that used 44 511 tokens + +**Evidence.** [`data/logs/agent-scala-solution-draft.log`](./data/logs/agent-scala-solution-draft.log) +contains the real usage โ€” 21 677 input, 22 834 output โ€” inside a pretty-printed JSON +record; the published comment says `0 input, 0 output`. + +**Root cause.** Every stream reader split the tool's stdout on `\n` and called +`JSON.parse` on each line. That is correct only for strict NDJSON whose line boundaries +happen to align with process chunk boundaries. `formal-ai with --verbose` emits +**indented, multi-line** JSON, so _every_ line fails to parse and every structured event +โ€” session id, token usage, errors, result text โ€” is dropped silently. Two neighbouring +shapes break the same way: records concatenated without a separator (issue #1250) and a +single record split across two process chunks. + +**Fix.** + +- `21197003` adds `src/json-stream.lib.mjs`: `createJsonStreamScanner` frames records by + **balanced JSON values** instead of by lines, releasing non-JSON output verbatim as + text events, with a 4 MiB guard so an unbalanced fragment cannot grow without bound. + Routed `agent`, `opencode` and `codex` through it. +- **D4b, `16d8b243`** โ€” the sweep required by R14. `gemini.lib.mjs` and `qwen.lib.mjs` + kept their own line-based parsers and therefore still had the identical defect, + invisible only because nobody had run those two tools yet. Both now share + `takeJsonRecords`, the stateless counterpart for parsers that carry their buffer in a + plain state object. + +### D5 โ€” Two duplicate auto-restart systems sharing one limit flag + +**Evidence.** The Scala PR carries both `๐Ÿ”„ Auto-restart 1/5` and +`๐Ÿ”„ Auto-restart triggered (iteration 1)` โ€” the two incompatible labels the issue calls +out โ€” and eleven sessions ran under `--auto-restart-max-iterations 5`. + +**Root cause.** Hive Mind had two independent restart subsystems, each with its own +counter reading the same flag: the uncommitted-changes loop in `solve.watch.lib.mjs` and +the mergeability loop in `solve.auto-merge.lib.mjs`. `solve.mjs` runs both in one +process, so a limit of 5 permitted 10 AI sessions. Only one of the two printed `N/M`. + +**Fix (`f8d4c91c`).** `src/auto-restart-budget.lib.mjs` holds one process-wide iteration +counter that both loops claim from, and one label formatter (`N/M`, or bare `N` when the +limit is 0 = unlimited). + +### D6 โ€” Exhausting the limit was reported as success + +**Evidence.** `โš ๏ธ Auto-restart limit reached โ€ฆ Remaining reason: Uncommitted changes +detected`, then exit `0`, and the uncommitted work discarded with the temporary clone. + +**Root cause.** Neither restart path had a failure branch. "Limit reached" was a comment, +not an outcome. + +**Fix (`f8d4c91c`).** `src/auto-restart-exhaustion.lib.mjs` is the single exhaustion +path: log, auto-commit and push the uncommitted work through the existing critical-error +recovery helper, post one "limit reached" comment, record the failure โ€” and +`solve.finalize.lib.mjs` exits `1`. The result is now visible, and the work is preserved +instead of deleted. + +### D7 โ€” Shell quotes leaked into pull request titles + +**Evidence.** All three PRs are titled `'Implement Hello World in Scala'` with the single +quotes as literal characters. + +**Root cause.** `command-stream` shell-escapes every interpolated value, quoting it +whenever it contains a character the shell would split on. Writing +`` $`gh pr edit --title "${updatedTitle}"` `` therefore passes +`--title "'Implement Hello World in Scala'"` and the quotes land in the title. Values +without spaces were unaffected, which is why it survived so long. + +**Fix (`46b2df22`).** Every such site in `src/` and `scripts/` โ€” PR titles and body +files, the YouTrack sync, the claude runtime switch, `reviewers-hive`, the changeset and +release scripts. Where quotes belonged to an inner language (jq filters, GraphQL +queries), the expression is assembled in JS and passed as one argument. +`tests/test-shell-quoting-2119.mjs` pins the `command-stream` behaviour _and_ scans +`src/` and `scripts/` so the pattern cannot come back. + +### D8 โ€” An empty diff described as `1 file(s) modified, 1 line(s) added` + +**Root cause.** The stats were measured while the scaffolding commit was still in the +diff and never revisited. Nothing in the codebase asked "does this pull request change +anything?" after the AI session ended. + +**Fix (`bd23dae6`).** `src/pull-request-changes.lib.mjs` is the single place that answers +it, used by both the description writer and the mergeability watcher. Counts come from +the unified diff rather than the PR's `additions`/`deletions` fields, because those are +per-commit sums โ€” a commit and its revert report 1 addition and 1 deletion while the net +diff is empty. + +### D9 โ€” `โœ… Ready to merge` for a pull request that changes nothing + +**Fix (`bd23dae6`).** The mergeable branch is gated on `!isEmptyPullRequest`, and an +empty pull request becomes a restart reason with an explicit instruction not to report +the work as done while the diff is empty. Crucially, a diff that _could not be read_ is +reported as `measured: false` rather than as empty, so a transient API failure cannot +manufacture an endless restart loop โ€” a false negative introduced while fixing a false +positive would have been the worst possible trade. + +### D10 โ€” A working session summary reporting `pwd`, leaking the workspace path + +**Root cause.** The summary extractor published whatever the last tool result was, with +no notion of whether the session accomplished anything, and did not sanitize +`/tmp/gh-issue-solver-*` paths. + +**Fix (`12f27c00`).** `src/working-session-summary.lib.mjs` adds an explicit notice when +the pull request still contains no changes, and replaces solver workspace paths with +``. An unreadable diff produces no notice. + +### D11 โ€” `.formal-ai/` scratch state driving the restart loop forever + +**Evidence.** Every restart comment lists `?? .formal-ai/`, and restarting re-creates it. + +**Root cause.** Formal AI writes a `.formal-ai/` plan directory into the workspace. Eight +copies of `checkForUncommittedChanges` saw it as user work. Worse, the same state reached +`git add -A` on the auto-commit paths, which would have published a tool's private +scratch files in the user's pull request. + +**Fix (`b1065b37`).** Generalizes the `.playwright-mcp/` special case from #1124: the +paths are written to `.git/info/exclude`, so **git itself** stops reporting them and all +eight call sites agree without each needing its own filter. `.git/info/exclude` is local +to the clone, so nothing leaks into the repository or the diff. + +### D12 โ€” The credential sanitizer redacting token counters and JSON braces + +**Evidence.** Published logs showed redacted `"total_tokens":44511`, `tokens=44511`, +`"maxTokens": 8192` and mangled JSON structure โ€” hiding exactly the numbers needed to +diagnose D4. + +**Root cause.** The secret patterns matched `token`-adjacent key/value pairs regardless +of whether the value was a number, and a brace-balancing bug consumed structural +characters. + +**Fix (`2072f214`).** Numeric token telemetry is excluded from the credential patterns +and JSON structure is preserved. + +### D13 โ€” The solver's own placeholder counted as a change + +**Evidence.** The Rust PR's entire diff is + +``` +diff --git a/.gitkeep b/.gitkeep ++# .gitkeep file auto-generated at 2026-07-30T14:24:59.267Z for PR creation at branch issue-1-09b0c76bd0e4 for issue https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/issues/1 +``` + +**Root cause.** `src/solve.auto-pr.lib.mjs` commits a `.gitkeep` (or `CLAUDE.md`) +placeholder so an empty branch has something to open a pull request from; it is reverted +once real work lands. When the run fails before that, the placeholder stays โ€” and D8's +new change counter honestly reported "1 file changed" for a run that implemented nothing. + +**Fix (`a40825e2`).** The diff is split per file and sections whose **content** matches +what `solve.auto-pr.lib.mjs` writes are dropped from the counts; +`placeholderOnly` is exposed so `formatChangeSummary` and `buildEmptyPullRequestBlocker` +can name the placeholder explicitly. Matching on content rather than on the file name +keeps a repository's own `.gitkeep` or `CLAUDE.md` edits counted as the real changes they +are. + +## 6. Deliberate non-changes + +Two things that look like they belong in the sweep and were left alone on purpose. + +**`src/solve.keep-working.lib.mjs` is not merged into the shared restart budget.** It is +a third loop, but it is not a restart-on-failure system: it is the opt-in +`--keep-working-until-all-requirements-are-fully-done` completeness feature, explicitly +marked EXPERIMENTAL. It already prints `iteration N/M`, has its own bounded limit and its +own `MAX_CONSECUTIVE_ERRORS = 3`. Folding an opt-in "keep going until complete" loop into +the automatic "something is broken, try again" budget would make one flag silently +consume the other's allowance โ€” the exact defect D5 is about. + +**`src/claude.lib.mjs` keeps its line-buffered NDJSON reader** (lines 800โ€“807). Claude +Code's `--output-format stream-json` is strict NDJSON by contract, its usage _was_ +reported correctly in the Kotlin run, and the file already sits at the 1500-line +`max-lines` cap. Changing a working parser on a tool that does not exhibit the defect +would be risk without benefit. If claude ever emits pretty-printed records, the shared +`takeJsonRecords` is one import away. + +## 7. Tests that pin this down + +Nine test files, all in the `default` suite (`npm test`): + +| Test | Guards | +| -------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------- | +| `test-agent-stream-json-2119.mjs` | pretty-printed / concatenated / chunk-split framing for agent and codex | +| `test-formal-ai-uniform-tools-2119.mjs` | R9 โ€” identical token accounting and Link.Assistant $0.00 pricing on all six tools; every stream-parsing lib imports the shared framer | +| `test-formal-ai-pricing-2119.mjs` | D1โ€“D3, and that non-formal-ai models keep their own provider | +| `test-auto-restart-budget-2119.mjs` | D5, D6 โ€” one budget, `N/M` labels, exhaustion fails and auto-commits | +| `test-shell-quoting-2119.mjs` | D7 โ€” the `command-stream` behaviour plus a repository-wide scan | +| `test-empty-pull-request-2119.mjs` | D8, D9, D13 โ€” empty vs placeholder-only vs unmeasured diffs | +| `test-working-session-summary-2119.mjs` | D10 โ€” the no-changes notice and workspace-path redaction | +| `test-ai-tool-scratch-2119.mjs` | D11 โ€” scratch paths excluded via `.git/info/exclude` | +| `test-token-telemetry-sanitization-2119.mjs` | D12 โ€” counters survive sanitization, real secrets still do not | + +The uniformity test deserves a note, because it is the answer to "how do we stop +rediscovering the same bug per tool": it asserts that every one of the five +stream-parsing libraries imports `json-stream.lib.mjs`, and that every one of the six +tools reaches the formal-ai pricing helper (directly, or through +`calculateAgentPricing` as `opencode` does). A seventh tool added tomorrow fails the test +until it is wired the same way. + +## 8. What belongs upstream + +The three runs also demonstrate five Formal AI defects, none of which Hive Mind can fix. +They are collected in [`upstream-formal-ai.md`](./upstream-formal-ai.md) and filed as a +single issue against `github.com/link-assistant/formal-ai`, framed โ€” as the issue asks โ€” +as a request for **generalization and self-healing**, not per-symptom patches: + +1. The `agent` run reinterpreted "implement hello world in Scala" as a general change + request against `./examples` and wrote no Scala. +2. The `codex` run never reached the local server: `401 Missing bearer or basic +authentication in header` from `https://api.openai.com/v1/responses`. +3. The `claude` run returned after a single `pwd` (`num_turns: 2`). +4. `.formal-ai/` scratch state is left in the caller's workspace. +5. Output is pretty-printed rather than NDJSON, and carries no usage metadata + (`rawMetadata: "{\"formalai\":{}}"`) โ€” the shape that caused D4. + +Issues in `link-assistant/agent` itself are covered by the same report, since the `agent` +run is the one that produced defect 1. + +## 9. Where the data was insufficient (R12) + +Two questions the archived logs cannot answer, and the instrumentation added so the next +run can: + +- **Why did the `agent` run target `./examples`?** The log records the final tool calls + but not the prompt Formal AI actually received. The shared JSON framer now surfaces + every record โ€” including the ones the line parser was silently dropping โ€” so the next + run's log contains the request as well as the response. +- **Where was the `formal-ai` alias lost on the codex path?** The failure log shows the + outbound request to `api.openai.com` but not the resolution step that chose it. + `resolveFormalAiToolInvocation` is now the single dispatch point for all six tools, so + one log line there covers every tool instead of six divergent code paths. + +## 10. Existing components considered + +For the JSON framing (D4), the ecosystem does have libraries that handle concatenated and +multi-document JSON: +[jsonhilo](https://github.com/xtao-org/jsonhilo) (accepts multiple consecutive top-level +values, JSONL and concatenated JSON), +[concatjson](https://github.com/manidlou/concatjson), +[clarinet](https://github.com/dscape/clarinet) (SAX-style evented parser), +[json-stream-es](https://github.com/cdauth/json-stream-es) and +[stream-json](https://github.com/uhop/stream-json). + +We implemented a ~220-line scanner instead, for three reasons worth recording: + +1. **Text passthrough is a requirement, not a nicety.** Agentic CLIs interleave plain log + lines with JSON records; a strict JSON parser errors on them, while + `createJsonStreamScanner` emits them as `text` events so tool chatter keeps flowing to + the user. +2. **Two call shapes are needed.** Some readers hold a closure across chunks + (`write`/`flush`), others hand their buffer back to the caller in a state object + (`takeJsonRecords`). None of the libraries above offers both. +3. **Dependency surface.** `package.json` carries 13 runtime dependencies; the scanner is + ~220 lines with no transitive tree, and the defect it fixes is in the log path of every + AI session. + +For D7 the equivalent question is `shell-quote` โ€” and here the answer was the opposite: +the fix is to _stop_ quoting, because `command-stream` already escapes. Adding a quoting +library would have compounded the bug. + +For D11, `.git/info/exclude` is the existing component: git's own per-clone ignore +mechanism, which no library can improve on and which makes all eight +`checkForUncommittedChanges` call sites agree for free. + +## 11. Files in this case study + +``` +docs/case-studies/issue-2119/ +โ”œโ”€โ”€ README.md this document +โ”œโ”€โ”€ upstream-formal-ai.md the consolidated Formal AI report +โ””โ”€โ”€ data/ + โ”œโ”€โ”€ issue-2119.json the issue as filed + โ”œโ”€โ”€ issue-2119-comments.json its comments + โ”œโ”€โ”€ test-hello-world-repos.json all six reproduction repositories + โ”œโ”€โ”€ log-gists.json public URLs of all 13 published session logs + โ”œโ”€โ”€ logs/ + โ”‚ โ”œโ”€โ”€ agent-scala-solution-draft.log + โ”‚ โ”œโ”€โ”€ claude-kotlin-solution-draft.log + โ”‚ โ””โ”€โ”€ codex-rust-failure.log + โ””โ”€โ”€ prs/ + โ”œโ”€โ”€ 019fb330-00e1-.../ Scala run: issue, PR, comments, diff + โ”œโ”€โ”€ 019fb330-fa49-.../ Kotlin run + โ”œโ”€โ”€ 019fb331-c107-.../ Rust run + โ””โ”€โ”€ baseline-2025/ the three 2025 runs for comparison +``` diff --git a/docs/case-studies/issue-2119/data/issue-2119-comments.json b/docs/case-studies/issue-2119/data/issue-2119-comments.json new file mode 100644 index 000000000..fe51488c7 --- /dev/null +++ b/docs/case-studies/issue-2119/data/issue-2119-comments.json @@ -0,0 +1 @@ +[] diff --git a/docs/case-studies/issue-2119/data/issue-2119.json b/docs/case-studies/issue-2119/data/issue-2119.json new file mode 100644 index 000000000..485278437 --- /dev/null +++ b/docs/case-studies/issue-2119/data/issue-2119.json @@ -0,0 +1,11 @@ +{ + "author": { "id": "MDQ6VXNlcjE0MzE5MDQ=", "is_bot": false, "login": "konard", "name": "Konstantin Diachenko" }, + "body": "Provider of the model is `Link.Assistant`, not `OpenCode Zen` or `Anthropic`.\n\nThe cost of the model tokens is 0$ (free).\n\nCarefully check all the GitHub comments and logs, and fix all other false positives, false negatives, errors and warnings.\n\nAuto-restart/resume on uncommitted changes and any other reason must stop after 5-th iteration, otherwise it will stuck infinitely, that must be applied to any model, as not only `formal-ai` may fail that way. Also as `Auto-restart triggered (iteration 1)` and `Auto-restart 1/5 Log` are different, that means we have some critical error in duplication of code. We should only use single auto-restart/resume system where it provides N/M as in `Auto-restart 1/5 Log`, so it is clear what is the limit. And after 5 we must actually stop (fail + auto-commit on fail recovery).\n\nSo the result will be actually visible.\n\nWe must solve all issues related to Hive Mind itself in this pull request, and report all issues related to Formal AI itself to github.com/link-assistant/formal-ai, and as always we should ask to solve it by generalization, not specialization, improving self learning and self healing meta algorithm. As Formal AI must be capable of coding and computer tasks.\n\nPull requests tested with `--model formal-ai`:\n- https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2\n- https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/pull/2\n- https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2\n\nAlso find all other previous `konard/test-hello-world-*` examples, and we should make sure we have tests in Formal AI, that guarantee highest possible quality of implementation. So based on these examples and tests we will be able to build quality product out of Formal AI. Each good example now is the strongest foundation for the future development.\n\nEach pull request targeted different `--tool`: codex, claude, agent (our own https://github.com/link-assistant/agent if any issues with it also report).\n\nWe should also ensure uniform support for gemini, and qwen for the Formal AI and Hive Mind.\n\nWe need to collect all the logs, and use them to report issue to Formal AI (it should fit single big issue with all listed requirements to be fixed) with all the materials linked.\n\nWe need to download all logs and data related about the issue to this repository, make sure we compile that data to `./docs/case-studies/issue-{id}` folder, and use it to do deep case study analysis (also make sure to search online for additional facts and data), in which we will reconstruct timeline/sequence of events, list of each and all requirements from the issue, find root causes of the each problem, and propose possible solutions and solution plans for each requirement (we should also check known existing components/libraries, that solve similar problem or can help in solutions).\n\nIf there is not enough data to find actual root cause, add debug output and verbose mode if not present, that will allow us to find root cause on next iteration.\n\nIf issue related to any other repository/project, where we can report issues on GitHub, please do so. Each issue must contain reproducible examples, workarounds and suggestions for fix the issue in code. Also double check to fully apply requirements to entire codebase, so if we have issue in multiple places, it should be fixed in all them.\n\nPlease plan and execute everything in this single pull request, you have unlimited time and context, as context auto-compacts and you can continue indefinitely, until it is each and every requirement fully addressed, and everything is totally done.", + "createdAt": "2026-07-30T14:29:29Z", + "labels": [{ "id": "LA_kwDOPUU0qc8AAAACGYm6iw", "name": "bug", "description": "Something isn't working", "color": "d73a4a" }], + "number": 2119, + "state": "OPEN", + "title": "`--model formal-ai` not working on simplest hello world", + "updatedAt": "2026-07-30T14:44:56Z", + "url": "https://github.com/link-assistant/hive-mind/issues/2119" +} diff --git a/docs/case-studies/issue-2119/data/log-gists.json b/docs/case-studies/issue-2119/data/log-gists.json new file mode 100644 index 000000000..3e4d96858 --- /dev/null +++ b/docs/case-studies/issue-2119/data/log-gists.json @@ -0,0 +1,17 @@ +{ + "019fb330-00e1-73b9-955e-f357a1600d5b": [ + { "createdAt": "2026-07-30T14:14:38Z", "gist": "https://gist.githubusercontent.com/konard/465f5511052c5796806c6fc1d29c6b4c/raw/c2768221a87d91fc27cbcad2dc9bc9e45b2a6ecd/tmp-hive-mind-log-upload-TQuSSp-sanitized.log.txt", "title": "๐Ÿค– Solution Draft Log" }, + { "createdAt": "2026-07-30T14:15:24Z", "gist": "https://gist.githubusercontent.com/konard/ef3290d4bc10e5d1116c6e1d79cfe9fc/raw/f4798e8bf7b0121f0e89ba82ceb233c85d55e950/tmp-hive-mind-log-upload-YU3TQL-sanitized.log.txt", "title": "๐Ÿ”„ Auto-restart 1/5 Log" }, + { "createdAt": "2026-07-30T14:16:07Z", "gist": "https://gist.githubusercontent.com/konard/4f45c03c2bc13fabed47d8fea4a5194f/raw/08207cecd10edcfc34ff5b940a522831d92fcba1/tmp-hive-mind-log-upload-9twZ9s-sanitized.log.txt", "title": "๐Ÿ”„ Auto-restart 2/5 Log" }, + { "createdAt": "2026-07-30T14:16:50Z", "gist": "https://gist.githubusercontent.com/konard/ebda1420a27cdaac0327cf3bfa515261/raw/26108136c6e9fc928e66a55fe06e6367f0cf4fb0/tmp-hive-mind-log-upload-6bihBU-sanitized.log.txt", "title": "๐Ÿ”„ Auto-restart 3/5 Log" }, + { "createdAt": "2026-07-30T14:17:47Z", "gist": "https://gist.githubusercontent.com/konard/f6a134fb73adac137e7c200ab5d3db92/raw/cee22282afc4bcbdeaba88475f26a8806cd5ece2/tmp-hive-mind-log-upload-56SSDs-sanitized.log.txt", "title": "๐Ÿ”„ Auto-restart 4/5 Log" }, + { "createdAt": "2026-07-30T14:18:35Z", "gist": "https://gist.githubusercontent.com/konard/9cd2e3a7e98a634375d2eed7049fb62d/raw/570dd2d36c791629e9a68001ac94e35d36841357/tmp-hive-mind-log-upload-tWAAPV-sanitized.log.txt", "title": "๐Ÿ”„ Auto-restart 5/5 Log" }, + { "createdAt": "2026-07-30T14:21:33Z", "gist": "https://gist.githubusercontent.com/konard/385002afaadc89bd178b783864d88301/raw/1174c597b8947cbfe4ab938e802664507db3d994/tmp-hive-mind-log-upload-HRzjRr-sanitized.log.txt", "title": "๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 1)" }, + { "createdAt": "2026-07-30T14:24:25Z", "gist": "https://gist.githubusercontent.com/konard/037503c4af448467dd8fdc3eb3c48980/raw/3815a653380f2b28b482948233fccdc18f3e7403/tmp-hive-mind-log-upload-fEg1HP-sanitized.log.txt", "title": "๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 2)" }, + { "createdAt": "2026-07-30T14:27:17Z", "gist": "https://gist.githubusercontent.com/konard/1459e4acca0d459d305f28a6e9b215be/raw/00ea9281d3b3004012d03a99a151b4786203a546/tmp-hive-mind-log-upload-HB4zCM-sanitized.log.txt", "title": "๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 3)" }, + { "createdAt": "2026-07-30T14:30:11Z", "gist": "https://gist.githubusercontent.com/konard/48c447dc630ae2f8246bbe62ac835e71/raw/4f45d3bd037e82f2201771ee7d955b1b2e71af84/tmp-hive-mind-log-upload-ZzMtDW-sanitized.log.txt", "title": "๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 4)" }, + { "createdAt": "2026-07-30T14:33:02Z", "gist": "https://gist.githubusercontent.com/konard/a5f3e9b4dea22cdffe0c699fcf32e1c5/raw/0de8fc98d9e00bf957d61ce5e3ed627a5c069c1c/tmp-hive-mind-log-upload-0JuEzF-sanitized.log.txt", "title": "๐Ÿ”„ Auto-restart-until-mergeable Log (iteration 5)" } + ], + "019fb330-fa49-7c9d-a664-b7ea33bb698a": [{ "createdAt": "2026-07-30T14:20:18Z", "gist": "https://gist.githubusercontent.com/konard/16a9e51ac177a0ca8dd6f26b5148b1b4/raw/b2fb3eb4ede4871979e5443f643551b725141b90/tmp-hive-mind-log-upload-DAh7ws-sanitized.log.txt", "title": "๐Ÿค– Solution Draft Log" }], + "019fb331-c107-78c7-8ff6-9f127a3c593c": [{ "createdAt": "2026-07-30T14:25:59Z", "gist": "https://gist.githubusercontent.com/konard/4e56198bee2a177e71ddc41bdf5b5294/raw/9ef0b5a8748c953eaeb2c98b9eafe6603359c855/tmp-hive-mind-log-upload-WQPa4n-sanitized.log.txt", "title": "๐Ÿšจ Solution Draft Failed" }] +} diff --git a/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2.diff b/docs/case-studies/issue-2119/data/prs/019fb330-00e1-73b9-955e-f357a1600d5b/pr-2.diff new file mode 100644 index 000000000..e69de29bb diff --git a/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2.diff b/docs/case-studies/issue-2119/data/prs/019fb330-fa49-7c9d-a664-b7ea33bb698a/pr-2.diff new file mode 100644 index 000000000..e69de29bb diff --git a/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2.diff b/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2.diff new file mode 100644 index 000000000..e3f90b3fc --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/019fb331-c107-78c7-8ff6-9f127a3c593c/pr-2.diff @@ -0,0 +1,8 @@ +diff --git a/.gitkeep b/.gitkeep +new file mode 100644 +index 0000000..3cf0f01 +--- /dev/null ++++ b/.gitkeep +@@ -0,0 +1 @@ ++# .gitkeep file auto-generated at 2026-07-30T14:24:59.267Z for PR creation at branch issue-1-09b0c76bd0e4 for issue https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/issues/1 +\ No newline at end of file diff --git a/docs/case-studies/issue-2119/data/prs/baseline-2025/01991e68-059d-749e-900f-af9e7aa44451-pr-2.json b/docs/case-studies/issue-2119/data/prs/baseline-2025/01991e68-059d-749e-900f-af9e7aa44451-pr-2.json new file mode 100644 index 000000000..a481eb68f --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/baseline-2025/01991e68-059d-749e-900f-af9e7aa44451-pr-2.json @@ -0,0 +1,24 @@ +{ + "additions": 69, + "body": "## Summary\n- Implemented Hello World program in Rust (`main.rs`) with proper comments and build instructions\n- Created GitHub Actions workflow (`.github/workflows/test-hello-world.yml`) that automatically tests the program\n- Program outputs exactly \"Hello, World!\" as required\n- Follows Rust best practices and idioms\n\n## Test plan\n- [x] Program compiles successfully with `rustc main.rs` \n- [x] Program outputs exactly \"Hello, World!\" when run\n- [x] GitHub Actions workflow triggers on push/PR to main branch\n- [x] Workflow sets up Rust environment, compiles, runs, and verifies output\n- [x] All requirements from issue #1 are satisfied\n\n## Files Added\n- `main.rs` - Hello World program in Rust with comprehensive comments\n- `.github/workflows/test-hello-world.yml` - CI/CD workflow for automated testing\n\nResolves #1\n\n๐Ÿค– Generated with [Claude Code](https://claude.ai/code)", + "changedFiles": 2, + "commits": [ + { + "authoredDate": "2025-09-06T09:49:40Z", + "authors": [ + { "email": "drakonard@gmail.com", "id": "MDQ6VXNlcjE0MzE5MDQ=", "login": "konard", "name": "Konard" }, + { "email": "noreply@anthropic.com", "id": "MDQ6VXNlcjgxODQ3", "login": "claude", "name": "Claude" } + ], + "committedDate": "2025-09-06T09:49:40Z", + "messageBody": "- Add main.rs with Hello World implementation following Rust best practices\n- Include comprehensive comments explaining the code and build instructions\n- Create GitHub Actions workflow that tests the program on push/PR\n- Workflow verifies output matches exactly \"Hello, World!\"\n- All requirements from issue #1 have been implemented\n\n๐Ÿค– Generated with [Claude Code](https://claude.ai/code)\n\nCo-Authored-By: Claude ", + "messageHeadline": "Implement Hello World program in Rust with GitHub Actions workflow", + "oid": "fa664d649aaa8feff15097252dd43643b6b6234c" + } + ], + "createdAt": "2025-09-06T09:49:54Z", + "deletions": 0, + "number": 2, + "state": "OPEN", + "title": "Implement Hello World in Rust with GitHub Actions", + "url": "https://github.com/konard/test-hello-world-01991e68-059d-749e-900f-af9e7aa44451/pull/2" +} diff --git a/docs/case-studies/issue-2119/data/prs/baseline-2025/01991e7d-f3e2-713e-a22a-0e69a326dc93-pr-2.json b/docs/case-studies/issue-2119/data/prs/baseline-2025/01991e7d-f3e2-713e-a22a-0e69a326dc93-pr-2.json new file mode 100644 index 000000000..1106d962a --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/baseline-2025/01991e7d-f3e2-713e-a22a-0e69a326dc93-pr-2.json @@ -0,0 +1,12 @@ +{ + "additions": 45, + "body": "This PR implements the Hello World program in Common Lisp as requested in issue #1.\n\n## Changes\n- Added with a simple Hello World program\n- Included proper comments and run instructions\n- Created GitHub Actions workflow in \n- The workflow installs SBCL, runs the program, and verifies the output\n\n## Testing\n- Program tested locally with SBCL\n- CI workflow will run on every push and PR to ensure correctness\n\nCloses #1", + "changedFiles": 2, + "commits": [{ "authoredDate": "2025-09-06T10:18:54Z", "authors": [{ "email": "drakonard@gmail.com", "id": "MDQ6VXNlcjE0MzE5MDQ=", "login": "konard", "name": "Konard" }], "committedDate": "2025-09-06T10:18:54Z", "messageBody": "- Add hello-world.lisp with proper comments and run instructions\n- Create GitHub Actions workflow to test the program\n- Program outputs 'Hello, World!' as required", "messageHeadline": "Implement Hello World in Common Lisp with CI workflow", "oid": "675e08b1a8ca55ef11bfc61a91000a8badf7d461" }], + "createdAt": "2025-09-06T10:19:01Z", + "deletions": 0, + "number": 2, + "state": "OPEN", + "title": "Implement Hello World in Common Lisp", + "url": "https://github.com/konard/test-hello-world-01991e7d-f3e2-713e-a22a-0e69a326dc93/pull/2" +} diff --git a/docs/case-studies/issue-2119/data/prs/baseline-2025/01992020-00f8-7cf2-9bb6-a1c2a7718de5-pr-2.json b/docs/case-studies/issue-2119/data/prs/baseline-2025/01992020-00f8-7cf2-9bb6-a1c2a7718de5-pr-2.json new file mode 100644 index 000000000..b0f4a252e --- /dev/null +++ b/docs/case-studies/issue-2119/data/prs/baseline-2025/01992020-00f8-7cf2-9bb6-a1c2a7718de5-pr-2.json @@ -0,0 +1,24 @@ +{ + "additions": 81, + "body": "## Summary\n- โœ… Implemented COBOL Hello World program that prints exactly \"Hello, World!\"\n- โœ… Added comprehensive GitHub Actions workflow for automated testing\n- โœ… Program follows COBOL best practices with proper structure and comments\n- โœ… Includes build/run instructions in comments\n\n## Files Added\n- `hello.cob` - COBOL program with proper divisions, clear comments, and compilation instructions\n- `.github/workflows/test-hello-world.yml` - CI/CD workflow that installs GnuCOBOL, compiles the program, runs it, and verifies output\n\n## Features\n- **Program Structure**: Uses standard COBOL divisions (IDENTIFICATION, ENVIRONMENT, DATA, PROCEDURE)\n- **Clear Comments**: Explains each section and includes compilation/execution instructions\n- **CI/CD Integration**: Automated workflow runs on every push to main and pull requests\n- **Output Verification**: Workflow verifies the program outputs exactly \"Hello, World!\"\n\n## Test Plan\n- [x] Program compiles successfully with GnuCOBOL\n- [x] Program runs and outputs \"Hello, World!\" correctly\n- [x] GitHub Actions workflow passes all steps\n- [x] Output verification logic works correctly\n- [x] Code follows COBOL conventions and best practices\n\n๐Ÿค– Generated with [Claude Code](https://claude.ai/code)", + "changedFiles": 2, + "commits": [ + { + "authoredDate": "2025-09-06T17:46:10Z", + "authors": [ + { "email": "drakonard@gmail.com", "id": "MDQ6VXNlcjE0MzE5MDQ=", "login": "konard", "name": "Konstantin Dyachenko" }, + { "email": "noreply@anthropic.com", "id": "MDQ6VXNlcjgxODQ3", "login": "claude", "name": "Claude" } + ], + "committedDate": "2025-09-06T17:46:10Z", + "messageBody": "- Add hello.cob: COBOL program that prints \"Hello, World!\"\n- Add .github/workflows/test-hello-world.yml: CI workflow that compiles and tests the program\n- Program follows COBOL best practices with proper divisions and clear comments\n- Workflow runs on push to main and pull requests, installs GnuCOBOL, compiles and verifies output\n\n๐Ÿค– Generated with [Claude Code](https://claude.ai/code)\n\nCo-Authored-By: Claude ", + "messageHeadline": "Implement Hello World program in COBOL with CI/CD workflow", + "oid": "4c16a61c41b5545e8f85477189d6cbff8a15169b" + } + ], + "createdAt": "2025-09-06T17:46:25Z", + "deletions": 0, + "number": 2, + "state": "OPEN", + "title": "Implement Hello World program in COBOL with CI/CD workflow", + "url": "https://github.com/konard/test-hello-world-01992020-00f8-7cf2-9bb6-a1c2a7718de5/pull/2" +} diff --git a/docs/case-studies/issue-2119/data/test-hello-world-repos.json b/docs/case-studies/issue-2119/data/test-hello-world-repos.json new file mode 100644 index 000000000..42b83f658 --- /dev/null +++ b/docs/case-studies/issue-2119/data/test-hello-world-repos.json @@ -0,0 +1,8 @@ +[ + { "createdAt": "2025-09-06T09:42:38Z", "fullName": "konard/test-hello-world-01991e68-059d-749e-900f-af9e7aa44451", "url": "https://github.com/konard/test-hello-world-01991e68-059d-749e-900f-af9e7aa44451" }, + { "createdAt": "2025-09-06T10:06:36Z", "fullName": "konard/test-hello-world-01991e7d-f3e2-713e-a22a-0e69a326dc93", "url": "https://github.com/konard/test-hello-world-01991e7d-f3e2-713e-a22a-0e69a326dc93" }, + { "createdAt": "2025-09-06T17:43:12Z", "fullName": "konard/test-hello-world-01992020-00f8-7cf2-9bb6-a1c2a7718de5", "url": "https://github.com/konard/test-hello-world-01992020-00f8-7cf2-9bb6-a1c2a7718de5" }, + { "createdAt": "2026-07-30T13:21:37Z", "fullName": "konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b", "url": "https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b" }, + { "createdAt": "2026-07-30T13:22:40Z", "fullName": "konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a", "url": "https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a" }, + { "createdAt": "2026-07-30T13:23:31Z", "fullName": "konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c", "url": "https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c" } +] diff --git a/docs/case-studies/issue-2119/upstream-formal-ai.md b/docs/case-studies/issue-2119/upstream-formal-ai.md new file mode 100644 index 000000000..30893a75b --- /dev/null +++ b/docs/case-studies/issue-2119/upstream-formal-ai.md @@ -0,0 +1,212 @@ +# Upstream report โ€” Formal AI + +This is the consolidated report filed against +[`link-assistant/formal-ai`](https://github.com/link-assistant/formal-ai) from the +evidence in [`README.md`](./README.md). It is kept in the repository so the Hive Mind +side of the investigation and the upstream side stay linked. + +Everything below concerns Formal AI itself. The thirteen defects Hive Mind owns are +fixed in https://github.com/link-assistant/hive-mind/pull/2120 and are **not** part of +this report. + +--- + +## Title + +`formal-ai with ` produces no work on the simplest possible task across agent, claude and codex + +## Summary + +Three `hive-mind solve --model formal-ai` runs were started on 2026-07-30 against three +freshly generated hello-world repositories, one per tool (`agent`, `claude`, `codex`). +The task in each was a single sentence: _implement Hello World in ``_. All +three produced zero lines of source code, each failing in a different way. The three +failures share one root: **the run has no way to notice that it accomplished nothing and +no mechanism to recover.** + +The request is not three patches. It is the meta-capability the issue asks for: +generalization, self-learning and self-healing, so that a run which produces no artifact +detects that fact itself, diagnoses why, and retries differently โ€” rather than exiting +successfully with an empty workspace. + +## Reproduction + +Common setup โ€” a repository with one issue, `Implement Hello World in `: + +```bash +# hive-mind (any version; 2.10.3 was used) +solve https://github.com///issues/1 \ + --model formal-ai \ + --tool \ + --auto-continue --attach-logs --verbose +``` + +| Run | Tool | Repository | Pull request | +| --- | -------- | ----------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------- | +| A | `agent` | [`test-hello-world-019fb330-00e1-...`](https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b) | [PR #2](https://github.com/konard/test-hello-world-019fb330-00e1-73b9-955e-f357a1600d5b/pull/2) | +| B | `claude` | [`test-hello-world-019fb330-fa49-...`](https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a) | [PR #2](https://github.com/konard/test-hello-world-019fb330-fa49-7c9d-a664-b7ea33bb698a/pull/2) | +| C | `codex` | [`test-hello-world-019fb331-c107-...`](https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c) | [PR #2](https://github.com/konard/test-hello-world-019fb331-c107-78c7-8ff6-9f127a3c593c/pull/2) | + +Full session logs (13 of them, including all ten restart iterations of run A) are +archived in +[`docs/case-studies/issue-2119/data/`](https://github.com/link-assistant/hive-mind/tree/main/docs/case-studies/issue-2119/data) +with their public gist URLs in +[`data/log-gists.json`](https://github.com/link-assistant/hive-mind/blob/main/docs/case-studies/issue-2119/data/log-gists.json). + +For comparison, three runs of the same generator in 2025 with a conventional model each +produced a two-file, 45โ€“81 line pull request. The pipeline works when the model answers. + +## Defect 1 โ€” the task is reinterpreted into something else (`agent`) + +Run A never wrote Scala. The log +([`agent-scala-solution-draft.log`](https://github.com/link-assistant/hive-mind/blob/main/docs/case-studies/issue-2119/data/logs/agent-scala-solution-draft.log)) +shows the session concluding that it + +> applied the general change request for `./examples` and verified it with `cat ./examples` + +for an issue that says _Implement Hello World in Scala_. `./examples` did not exist; the +verification command was run against a path that is not a file, and the session ended +reporting success. + +**Why this is a generalization problem, not a prompt problem.** The failure is not that +one prompt was misread โ€” it is that the run had no acceptance criterion. "Implement X" +implies an artifact exists afterwards. A self-checking agent would derive that criterion +from the request itself, verify it (a `.scala` file exists, it compiles, it prints +`Hello, World!`), and treat its absence as a failed attempt rather than a completed one. + +**Suggested direction.** Before declaring completion, re-derive the task's observable +postcondition from the original request and check it. Where the check fails, that is +training signal: record the (request, plan, outcome) triple and try a different +decomposition. This is the self-learning loop the request asks for, and it subsumes every +individual "the model misunderstood" bug. + +**Workaround.** None on the caller side. Hive Mind can now _detect_ the empty result and +refuse to call it success (PR #2120), but it cannot make the tool produce work. + +## Defect 2 โ€” `formal-ai` does not route through the local server (`codex`) + +Run C failed with + +``` +401 Missing bearer or basic authentication in header +POST https://api.openai.com/v1/responses +``` + +([`codex-rust-failure.log`](https://github.com/link-assistant/hive-mind/blob/main/docs/case-studies/issue-2119/data/logs/codex-rust-failure.log)). + +The `formal-ai` model alias is supposed to resolve to the local Link.Assistant server. +On the codex path the alias was accepted but the request still went to OpenAI's public +API, where no credential exists โ€” so the run failed for a reason entirely unrelated to +the task. + +**Suggested direction.** Fail closed: if the resolved base URL for a `formal-ai` +invocation is not the local server, abort with a diagnostic naming the expected and the +actual endpoint, instead of issuing the request and surfacing a generic authentication +error. A single resolution step shared by all six tools removes the class of defect +rather than this instance of it. + +**Workaround.** Set the tool's own base-URL environment variable explicitly before +invoking, so codex cannot fall back to the public endpoint. + +## Defect 3 โ€” the session ends after a single no-op command (`claude`) + +Run B's entire session was one `pwd`. The result record reports `"num_turns": 2` and +`"stop_reason": "end_turn"` +([`claude-kotlin-solution-draft.log`](https://github.com/link-assistant/hive-mind/blob/main/docs/case-studies/issue-2119/data/logs/claude-kotlin-solution-draft.log), line 1045). +The workspace was untouched, and the session reported normal termination. + +**Suggested direction.** `end_turn` after a single orientation command is +indistinguishable from a crash from the caller's side. A completion gate โ€” "did this +session change anything, and if not, is that consistent with the request?" โ€” turns this +into a retryable state instead of a silent success. Emitting a structured +`completion_state` in the result record would let callers act on it without heuristics. + +**Workaround.** None; the caller sees a well-formed successful result. + +## Defect 4 โ€” `.formal-ai/` scratch state is left in the caller's workspace + +Every run wrote a `.formal-ai/` plan directory into the working tree it was invoked in +and left it there. On the caller's side this shows up as untracked user changes: + +``` +?? .formal-ai/ +๐Ÿ“ Found uncommitted changes +๐Ÿ”„ AUTO-RESTART: Restarting Agent to handle uncommitted changes... +``` + +Restarting re-creates the directory, so the condition never clears. Run A spent ten +restart iterations on this โ€” visible in the PR's `Auto-restart 1/5` โ€ฆ `5/5` comments. +Worse, an auto-commit path doing `git add -A` would publish the tool's private planning +state in the user's pull request. + +**Suggested direction.** Keep session scratch state outside the user's working tree (an +XDG state directory, or a temporary directory keyed by session id). If it must live in +the workspace, write the ignore entry along with it. + +**Workaround (implemented caller-side).** Hive Mind now adds `.formal-ai/` to +`.git/info/exclude` so git itself stops reporting it +([`b1065b37`](https://github.com/link-assistant/hive-mind/commit/b1065b37)). This is a +local mitigation, not a fix โ€” anyone invoking Formal AI without that mitigation still +gets the scratch directory in their tree. + +## Defect 5 โ€” the output stream is unparseable and carries no usage metadata + +Two problems in one stream. + +**5a. Pretty-printed instead of NDJSON.** `formal-ai with --verbose` emits +indented, multi-line JSON records. Every consumer that treats the stream as NDJSON โ€” +which is what the tools' own `--output-format stream-json` contracts promise โ€” fails to +parse every single record. Concretely: a session that used 21 677 input and 22 834 output +tokens was published as `Token usage: 0 input, 0 output`, because the record carrying the +usage never parsed. + +**5b. Empty metadata.** The records that do carry a metadata field carry nothing in it: + +``` +rawMetadata": "{\"formalai\":{}}" +``` + +so even a correct parser learns no model, no usage and no cost. + +**Suggested direction.** Emit one JSON value per line (NDJSON) when a machine-readable +output format is requested โ€” pretty-printing is a human affordance and belongs behind a +separate flag. Populate `rawMetadata` with at least model id, input/output token counts +and the served endpoint, so callers can attribute and account for a session without +scraping prose. + +**Workaround (implemented caller-side).** Hive Mind now frames records by balanced JSON +values instead of by lines +([`json-stream.lib.mjs`](https://github.com/link-assistant/hive-mind/blob/main/src/json-stream.lib.mjs)), +which tolerates pretty-printed, concatenated and chunk-split records. That recovers the +token counts, but it cannot invent the metadata that is not sent. + +## Defect 6 โ€” uniform behaviour across the six supported tools + +`FORMAL_AI_SUPPORTED_TOOLS` is `claude, agent, opencode, codex, qwen, gemini`. Only three +were exercised, and each failed differently โ€” one misread the task, one bypassed the +server, one no-opped. `qwen` and `gemini` were never run at all. + +**Request.** A conformance suite that runs the identical trivial task through every +supported tool and asserts the same observable contract: an artifact exists, the model +alias resolved to the local server, the output stream is NDJSON, usage metadata is +populated, and the workspace is left clean. Anything a single tool can do that the others +cannot is a divergence to be closed, not documented. + +Hive Mind added its half of this on 2026-07-30: +[`tests/test-formal-ai-uniform-tools-2119.mjs`](https://github.com/link-assistant/hive-mind/blob/main/tests/test-formal-ai-uniform-tools-2119.mjs) +asserts that all six tools share one stream framer and one pricing path, so a seventh +tool cannot be added with a divergent implementation. The equivalent on the Formal AI +side is what this request is for. + +## What is being asked for, in one sentence + +Not six patches โ€” a run-level self-check that notices "I produced no artifact", explains +why, and retries differently; the individual defects above are what that check would have +caught. + +## Related + +- Hive Mind issue: https://github.com/link-assistant/hive-mind/issues/2119 +- Hive Mind fixes (13 defects, 9 regression tests): https://github.com/link-assistant/hive-mind/pull/2120 +- Full case study, timeline and raw evidence: + https://github.com/link-assistant/hive-mind/tree/main/docs/case-studies/issue-2119 From 0a7196881303af3900fd82025e00f75e6a06817b Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 17:08:11 +0000 Subject: [PATCH 14/17] docs(2119): link the filed upstream issues (formal-ai#879, agent#285) Files the consolidated Formal AI report as formal-ai#879 and, from the same logs, three Agent CLI defects as agent#285: unknown model ids fall through to the without-todo prompt, that prompt tells the model it is opencode and routes feedback to sst/opencode, and session records carry a hard-coded agent-cli-1.0.0 version. Section 9 is corrected too - the system prompt *is* in the archived log, in the request bodyPreview records, which is how the prompt-selection defect was found. --- docs/case-studies/issue-2119/README.md | 84 ++++++++++++++----- .../issue-2119/upstream-formal-ai.md | 5 ++ 2 files changed, 67 insertions(+), 22 deletions(-) diff --git a/docs/case-studies/issue-2119/README.md b/docs/case-studies/issue-2119/README.md index d309db85c..10922feaa 100644 --- a/docs/case-studies/issue-2119/README.md +++ b/docs/case-studies/issue-2119/README.md @@ -98,22 +98,22 @@ showed redacted token counters, because the credential sanitizer treated ## 4. Requirements extracted from the issue -| # | Requirement (verbatim intent) | Where it is addressed | -| --- | --------------------------------------------------------------------------------------------- | ----------------------------------------------------------------- | -| R1 | Provider must be `Link.Assistant`, not `OpenCode Zen` / `Anthropic` | D1 โ€” `8f7f173a` | -| R2 | The model is free: cost must be $0 | D2, D3 โ€” `8f7f173a` | -| R3 | Fix all other false positives, false negatives, errors and warnings in the comments and logs | D4โ€“D13 | -| R4 | One auto-restart system, `N/M` labelled, hard stop after the limit | D5 โ€” `f8d4c91c` | -| R5 | After the limit: actually fail, with auto-commit on fail recovery, so the result is visible | D6 โ€” `f8d4c91c` | -| R6 | Solve everything belonging to Hive Mind in this pull request | ยง5, all 13 defects | -| R7 | Report everything belonging to Formal AI upstream, asking for generalization and self-healing | ยง8, [`upstream-formal-ai.md`](./upstream-formal-ai.md) | -| R8 | Find all other `test-hello-world-*` examples; guarantee quality with tests | ยง2 (all six found), ยง7 (test inventory) | -| R9 | Uniform support for gemini and qwen | D4b โ€” `16d8b243`, `tests/test-formal-ai-uniform-tools-2119.mjs` | -| R10 | Collect all logs; single big upstream issue with all materials linked | [`data/`](./data), ยง8 | -| R11 | Compile the data into `./docs/case-studies/issue-2119` and do a deep analysis | this document | -| R12 | Where data is insufficient, add debug output / verbose mode for the next iteration | ยง9 | -| R13 | Report to other repositories where applicable, with reproductions and fix suggestions | ยง8 (Formal AI; `link-assistant/agent` covered by the same report) | -| R14 | Apply every fix across the entire codebase, not just where it was observed | ยง5 โ€” each fix names its sweep | +| # | Requirement (verbatim intent) | Where it is addressed | +| --- | --------------------------------------------------------------------------------------------- | --------------------------------------------------------------- | +| R1 | Provider must be `Link.Assistant`, not `OpenCode Zen` / `Anthropic` | D1 โ€” `8f7f173a` | +| R2 | The model is free: cost must be $0 | D2, D3 โ€” `8f7f173a` | +| R3 | Fix all other false positives, false negatives, errors and warnings in the comments and logs | D4โ€“D13 | +| R4 | One auto-restart system, `N/M` labelled, hard stop after the limit | D5 โ€” `f8d4c91c` | +| R5 | After the limit: actually fail, with auto-commit on fail recovery, so the result is visible | D6 โ€” `f8d4c91c` | +| R6 | Solve everything belonging to Hive Mind in this pull request | ยง5, all 13 defects | +| R7 | Report everything belonging to Formal AI upstream, asking for generalization and self-healing | ยง8, [`upstream-formal-ai.md`](./upstream-formal-ai.md) | +| R8 | Find all other `test-hello-world-*` examples; guarantee quality with tests | ยง2 (all six found), ยง7 (test inventory) | +| R9 | Uniform support for gemini and qwen | D4b โ€” `16d8b243`, `tests/test-formal-ai-uniform-tools-2119.mjs` | +| R10 | Collect all logs; single big upstream issue with all materials linked | [`data/`](./data), ยง8 | +| R11 | Compile the data into `./docs/case-studies/issue-2119` and do a deep analysis | this document | +| R12 | Where data is insufficient, add debug output / verbose mode for the next iteration | ยง9 | +| R13 | Report to other repositories where applicable, with reproductions and fix suggestions | ยง8 โ€” formal-ai#879, agent#285 | +| R14 | Apply every fix across the entire codebase, not just where it was observed | ยง5 โ€” each fix names its sweep | ## 5. The thirteen defects: evidence, root cause, fix @@ -345,6 +345,8 @@ until it is wired the same way. ## 8. What belongs upstream +### 8.1 Formal AI โ€” [link-assistant/formal-ai#879](https://github.com/link-assistant/formal-ai/issues/879) + The three runs also demonstrate five Formal AI defects, none of which Hive Mind can fix. They are collected in [`upstream-formal-ai.md`](./upstream-formal-ai.md) and filed as a single issue against `github.com/link-assistant/formal-ai`, framed โ€” as the issue asks โ€” @@ -359,18 +361,56 @@ authentication in header` from `https://api.openai.com/v1/responses`. 5. Output is pretty-printed rather than NDJSON, and carries no usage metadata (`rawMetadata: "{\"formalai\":{}}"`) โ€” the shape that caused D4. -Issues in `link-assistant/agent` itself are covered by the same report, since the `agent` -run is the one that produced defect 1. +It cross-references the existing Formal AI issues it touches: #859 (hello world fails in +Codex while it works in Claude Code and OpenCode), #848 (the coding-task ladder), #864 +(proactive issue reporting on a failure of thinking) and #847 (task decomposition). + +### 8.2 Agent CLI โ€” [link-assistant/agent#285](https://github.com/link-assistant/agent/issues/285) + +Re-reading the `agent` run's log for the prompt Formal AI actually received turned up +three defects that belong to `link-assistant/agent` itself, all verified against that +repository's `main`: + +1. **Unknown models get the weakest prompt, silently.** + `js/src/session/system.ts` selects the system prompt by substring-matching the model + id (`gpt-5`, `gpt-`/`o1`/`o3`, `gemini-`, `claude`, `polaris-alpha`, `grok-code`) and + falls through to `PROMPT_ANTHROPIC_WITHOUT_TODO` (`prompt/qwen.txt`) for everything + else. `formalai/formal-ai` matches nothing, so the run that produced no Scala was + driven by a prompt with **zero** todo/task-tracking instructions (`grep -ci todo`: 0 in + `qwen.txt`, 12 in `anthropic.txt`) โ€” and nothing in the log says which prompt was + chosen. +2. **The agent tells the model it is `opencode`.** `prompt/qwen.txt:1` opens with `You are +opencode, โ€ฆ` and lines 8-11 route feedback to `https://github.com/sst/opencode/issues` + and docs to `https://opencode.ai`. Eight prompt files carry the string (9 hits in + `anthropic.txt` and `polaris.txt`, 8 in `qwen.txt`). Any harness that asks the agent to + report its own failure โ€” hive-mind's report-issue flow, formal-ai#864 โ€” is being aimed + at the wrong repository. +3. **Session records carry a hard-coded version.** The same session logs + `"service": "default", "version": "0.25.3"` and + `"service": "session", "version": "agent-cli-1.0.0"`; the latter is the literal at + `js/src/session/index.ts:229`, while `js/package.json` says `0.25.3`. Stored sessions + are therefore unattributable to a release. + +Each is reported with a code-level fix framed as generalization: resolve the prompt from +declared model capabilities with a logged default instead of substring matching, keep the +product identity in one interpolated place instead of eight text files, and derive the +version from the manifest so the three reported values cannot diverge. ## 9. Where the data was insufficient (R12) Two questions the archived logs cannot answer, and the instrumentation added so the next run can: -- **Why did the `agent` run target `./examples`?** The log records the final tool calls - but not the prompt Formal AI actually received. The shared JSON framer now surfaces - every record โ€” including the ones the line parser was silently dropping โ€” so the next - run's log contains the request as well as the response. +- **Why did the `agent` run target `./examples`?** Partly answered on the second reading. + The log does carry the system prompt, in the request `bodyPreview` records at lines + 1452 and 1563 of + [`agent-scala-solution-draft.log`](./data/logs/agent-scala-solution-draft.log), and it + is the without-todo prompt (`prompt/qwen.txt`) that Agent CLI selects for any model id + it does not recognise โ€” see ยง8.2. That explains the missing plan/verify discipline; it + does not explain the specific choice of `./examples`, for which the user-turn content + would be needed. The shared JSON framer now surfaces every record โ€” including the ones + the line parser was silently dropping โ€” so the next run's log carries the full request + alongside the response rather than only the fragments that happened to parse. - **Where was the `formal-ai` alias lost on the codex path?** The failure log shows the outbound request to `api.openai.com` but not the resolution step that chose it. `resolveFormalAiToolInvocation` is now the single dispatch point for all six tools, so diff --git a/docs/case-studies/issue-2119/upstream-formal-ai.md b/docs/case-studies/issue-2119/upstream-formal-ai.md index 30893a75b..a53b70e06 100644 --- a/docs/case-studies/issue-2119/upstream-formal-ai.md +++ b/docs/case-studies/issue-2119/upstream-formal-ai.md @@ -5,6 +5,11 @@ This is the consolidated report filed against evidence in [`README.md`](./README.md). It is kept in the repository so the Hive Mind side of the investigation and the upstream side stay linked. +**Filed as [link-assistant/formal-ai#879](https://github.com/link-assistant/formal-ai/issues/879)** +on 2026-07-30. The three Agent CLI defects found in the same logs are reported separately, +where they belong: [link-assistant/agent#285](https://github.com/link-assistant/agent/issues/285) +(see ยง8.2 of the case study). + Everything below concerns Formal AI itself. The thirteen defects Hive Mind owns are fixed in https://github.com/link-assistant/hive-mind/pull/2120 and are **not** part of this report. From 6b3df3c3cf4bf0ee2f6379cb96963c7b2161a45b Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 17:18:05 +0000 Subject: [PATCH 15/17] chore(2119): add the release changeset and keep archived evidence verbatim The release workflow requires exactly one changeset per pull request and forbids manual package.json version edits, so the bump is declared here. The archived case-study data is raw evidence, so it is added to .prettierignore alongside every earlier case study's data directory rather than reformatted. --- .changeset/formal-ai-empty-run-false-positives.md | 5 +++++ .prettierignore | 1 + 2 files changed, 6 insertions(+) create mode 100644 .changeset/formal-ai-empty-run-false-positives.md diff --git a/.changeset/formal-ai-empty-run-false-positives.md b/.changeset/formal-ai-empty-run-false-positives.md new file mode 100644 index 000000000..d9e4da31b --- /dev/null +++ b/.changeset/formal-ai-empty-run-false-positives.md @@ -0,0 +1,5 @@ +--- +'@link-assistant/hive-mind': patch +--- + +Stop reporting an empty `--model formal-ai` run as a success. Formal AI sessions are now attributed to Link.Assistant at $0.00, token usage is parsed for all six tools, the two duplicate auto-restart loops are one N/M budget that fails visibly when it is exhausted, and a pull request whose net diff is empty (or holds only the solver's own placeholder) is neither described as changed nor announced as ready to merge. diff --git a/.prettierignore b/.prettierignore index 9cf69a297..863cc2958 100644 --- a/.prettierignore +++ b/.prettierignore @@ -36,6 +36,7 @@ docs/case-studies/issue-1881/data/ docs/case-studies/issue-2080/data/ docs/case-studies/issue-2084/data/ docs/case-studies/issue-2085/data/ +docs/case-studies/issue-2119/data/ docs/case-studies/issue-2109/source/*.json docs/case-studies/issue-2109/research/*.json docs/case-studies/issue-1724/data/use-m-source.js From 5c74bef6fc24b5183e0dd28db592c5aa4d0ee627 Mon Sep 17 00:00:00 2001 From: konard Date: Thu, 30 Jul 2026 17:18:36 +0000 Subject: [PATCH 16/17] chore(2119): drop the solver's own PR placeholder .gitkeep was committed to open this pull request; the solution replaced it, so it is not part of the change. This is the same placeholder D13 teaches the change counter to ignore. --- .gitkeep | 1 - 1 file changed, 1 deletion(-) delete mode 100644 .gitkeep diff --git a/.gitkeep b/.gitkeep deleted file mode 100644 index 4bd6fca96..000000000 --- a/.gitkeep +++ /dev/null @@ -1 +0,0 @@ -# .gitkeep file auto-generated at 2026-07-30T14:43:37.751Z for PR creation at branch issue-2119-ae2d4c9d7f6d for issue https://github.com/link-assistant/hive-mind/issues/2119 \ No newline at end of file From bf3dee3acb727d06961ddbdd939768d606d978fc Mon Sep 17 00:00:00 2001 From: konard Date: Sat, 1 Aug 2026 07:34:32 +0000 Subject: [PATCH 17/17] fix(2119): ship released Formal AI recovery fixes --- .../formal-ai-empty-run-false-positives.md | 2 +- Dockerfile | 2 +- Dockerfile.dind | 2 +- Dockerfile.formal-ai | 2 +- coolify/Dockerfile | 2 +- docs/case-studies/issue-2119/README.md | 33 ++++++++++++------- tests/test-formal-ai-uniform-tools-2119.mjs | 10 ++++++ 7 files changed, 37 insertions(+), 16 deletions(-) diff --git a/.changeset/formal-ai-empty-run-false-positives.md b/.changeset/formal-ai-empty-run-false-positives.md index d9e4da31b..78d1f0d7a 100644 --- a/.changeset/formal-ai-empty-run-false-positives.md +++ b/.changeset/formal-ai-empty-run-false-positives.md @@ -2,4 +2,4 @@ '@link-assistant/hive-mind': patch --- -Stop reporting an empty `--model formal-ai` run as a success. Formal AI sessions are now attributed to Link.Assistant at $0.00, token usage is parsed for all six tools, the two duplicate auto-restart loops are one N/M budget that fails visibly when it is exhausted, and a pull request whose net diff is empty (or holds only the solver's own placeholder) is neither described as changed nor announced as ready to merge. +Stop reporting an empty `--model formal-ai` run as a success. Formal AI sessions are now attributed to Link.Assistant at $0.00, token usage is parsed for all six tools, the two duplicate auto-restart loops are one N/M budget that fails visibly when it is exhausted, and a pull request whose net diff is empty (or holds only the solver's own placeholder) is neither described as changed nor announced as ready to merge. Docker images now pin Formal AI 0.317.0 so the upstream workspace-effect and self-healing fixes are distributed with Hive Mind. diff --git a/Dockerfile b/Dockerfile index bd95f10a0..931c83163 100644 --- a/Dockerfile +++ b/Dockerfile @@ -18,7 +18,7 @@ # # Build: docker build -t konard/hive-mind . -ARG FORMAL_AI_VERSION=0.305.0 +ARG FORMAL_AI_VERSION=0.317.0 # Bookworm's glibc 2.36 remains compatible with the Ubuntu 24.04 Box runtime. FROM rust:1.96-slim-bookworm AS formal-ai-builder ARG FORMAL_AI_VERSION diff --git a/Dockerfile.dind b/Dockerfile.dind index 8dc6867b5..4350eef30 100644 --- a/Dockerfile.dind +++ b/Dockerfile.dind @@ -21,7 +21,7 @@ # We pin 2.3.5 (latest patch on top of that fix). # Latest Box releases: https://github.com/link-foundation/box/releases -ARG FORMAL_AI_VERSION=0.305.0 +ARG FORMAL_AI_VERSION=0.317.0 # Bookworm's glibc 2.36 remains compatible with the Ubuntu 24.04 Box runtime. FROM rust:1.96-slim-bookworm AS formal-ai-builder ARG FORMAL_AI_VERSION diff --git a/Dockerfile.formal-ai b/Dockerfile.formal-ai index 4a474fe34..6e9e7b980 100644 --- a/Dockerfile.formal-ai +++ b/Dockerfile.formal-ai @@ -4,7 +4,7 @@ # isolated solve jobs have the same agentic CLIs. The pinned builder keeps this # file buildable before a new Hive Mind tag reaches every registry mirror. -ARG FORMAL_AI_VERSION=0.305.0 +ARG FORMAL_AI_VERSION=0.317.0 ARG HIVE_MIND_VERSION=latest # Bookworm's glibc 2.36 remains compatible with the Ubuntu 24.04 Box runtime. FROM rust:1.96-slim-bookworm AS formal-ai-builder diff --git a/coolify/Dockerfile b/coolify/Dockerfile index 9b448e478..fe2f61dcb 100644 --- a/coolify/Dockerfile +++ b/coolify/Dockerfile @@ -18,7 +18,7 @@ # # Build: docker build -f coolify/Dockerfile -t konard/hive-mind . -ARG FORMAL_AI_VERSION=0.305.0 +ARG FORMAL_AI_VERSION=0.317.0 # Bookworm's glibc 2.36 remains compatible with the Ubuntu 24.04 Box runtime. FROM rust:1.96-slim-bookworm AS formal-ai-builder ARG FORMAL_AI_VERSION diff --git a/docs/case-studies/issue-2119/README.md b/docs/case-studies/issue-2119/README.md index 10922feaa..9b13f376e 100644 --- a/docs/case-studies/issue-2119/README.md +++ b/docs/case-studies/issue-2119/README.md @@ -324,17 +324,17 @@ would be risk without benefit. If claude ever emits pretty-printed records, the Nine test files, all in the `default` suite (`npm test`): -| Test | Guards | -| -------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------- | -| `test-agent-stream-json-2119.mjs` | pretty-printed / concatenated / chunk-split framing for agent and codex | -| `test-formal-ai-uniform-tools-2119.mjs` | R9 โ€” identical token accounting and Link.Assistant $0.00 pricing on all six tools; every stream-parsing lib imports the shared framer | -| `test-formal-ai-pricing-2119.mjs` | D1โ€“D3, and that non-formal-ai models keep their own provider | -| `test-auto-restart-budget-2119.mjs` | D5, D6 โ€” one budget, `N/M` labels, exhaustion fails and auto-commits | -| `test-shell-quoting-2119.mjs` | D7 โ€” the `command-stream` behaviour plus a repository-wide scan | -| `test-empty-pull-request-2119.mjs` | D8, D9, D13 โ€” empty vs placeholder-only vs unmeasured diffs | -| `test-working-session-summary-2119.mjs` | D10 โ€” the no-changes notice and workspace-path redaction | -| `test-ai-tool-scratch-2119.mjs` | D11 โ€” scratch paths excluded via `.git/info/exclude` | -| `test-token-telemetry-sanitization-2119.mjs` | D12 โ€” counters survive sanitization, real secrets still do not | +| Test | Guards | +| -------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------- | +| `test-agent-stream-json-2119.mjs` | pretty-printed / concatenated / chunk-split framing for agent and codex | +| `test-formal-ai-uniform-tools-2119.mjs` | R9 โ€” identical token accounting and Link.Assistant $0.00 pricing on all six tools; shared stream framing; every image pins Formal AI 0.317.0 | +| `test-formal-ai-pricing-2119.mjs` | D1โ€“D3, and that non-formal-ai models keep their own provider | +| `test-auto-restart-budget-2119.mjs` | D5, D6 โ€” one budget, `N/M` labels, exhaustion fails and auto-commits | +| `test-shell-quoting-2119.mjs` | D7 โ€” the `command-stream` behaviour plus a repository-wide scan | +| `test-empty-pull-request-2119.mjs` | D8, D9, D13 โ€” empty vs placeholder-only vs unmeasured diffs | +| `test-working-session-summary-2119.mjs` | D10 โ€” the no-changes notice and workspace-path redaction | +| `test-ai-tool-scratch-2119.mjs` | D11 โ€” scratch paths excluded via `.git/info/exclude` | +| `test-token-telemetry-sanitization-2119.mjs` | D12 โ€” counters survive sanitization, real secrets still do not | The uniformity test deserves a note, because it is the answer to "how do we stop rediscovering the same bug per tool": it asserts that every one of the five @@ -396,6 +396,17 @@ declared model capabilities with a logged default instead of substring matching, product identity in one interpolated place instead of eight text files, and derive the version from the manifest so the three reported values cannot diverge. +### 8.3 Upstream resolution and distribution verification + +Both upstream reports are resolved. Formal AI +[PR #881](https://github.com/link-assistant/formal-ai/pull/881) shipped the generalized +workspace-effect completion contract in v0.316.1; v0.317.0 is the current release at +final verification. All four Hive Mind image definitions now pin v0.317.0, and the +uniformity regression test prevents an image variant from silently retaining an older +wrapper. Agent [PR #286](https://github.com/link-assistant/agent/pull/286) shipped the +prompt, branding, and version fixes in npm package 0.25.4; the current `latest` package +installed by Hive Mind images is 0.25.5. + ## 9. Where the data was insufficient (R12) Two questions the archived logs cannot answer, and the instrumentation added so the next diff --git a/tests/test-formal-ai-uniform-tools-2119.mjs b/tests/test-formal-ai-uniform-tools-2119.mjs index 1a70212ce..94aaf3b68 100644 --- a/tests/test-formal-ai-uniform-tools-2119.mjs +++ b/tests/test-formal-ai-uniform-tools-2119.mjs @@ -33,6 +33,7 @@ const repoRoot = path.join(path.dirname(fileURLToPath(import.meta.url)), '..'); /** The model alias that routes a tool through the local Link.Assistant server. */ const FORMAL_AI_MODEL = 'formal-ai'; +const FORMAL_AI_RELEASE_WITH_ISSUE_2119_FIXES = '0.317.0'; // --- gemini ------------------------------------------------------------------ @@ -165,4 +166,13 @@ const opencodePricing = await calculateAgentPricing(FORMAL_AI_MODEL, { inputToke assert.equal(opencodePricing.provider, 'Link.Assistant', 'opencode/agent: formal-ai runs are not attributed to OpenCode Zen'); assert.equal(opencodePricing.totalCostUSD, 0, 'opencode/agent: formal-ai runs are free'); +// Formal AI v0.316.1 shipped the upstream half of issue #2119: workspace-effect +// validation and recovery, scratch exclusion, endpoint validation, and strict +// completion telemetry for all six clients. Keep every distributed Hive Mind +// image on the same current release so users actually receive those fixes. +for (const file of ['Dockerfile', 'Dockerfile.dind', 'Dockerfile.formal-ai', 'coolify/Dockerfile']) { + const source = await readFile(path.join(repoRoot, file), 'utf8'); + assert.match(source, new RegExp(`^ARG FORMAL_AI_VERSION=${FORMAL_AI_RELEASE_WITH_ISSUE_2119_FIXES}$`, 'm'), `${file} installs the Formal AI release containing the upstream issue #2119 fixes`); +} + console.log(`โœ… issue #2119: formal-ai stream parsing and pricing are uniform across ${FORMAL_AI_SUPPORTED_TOOLS.length} tools`);