Skip to content

Commit 4f367fd

Browse files
Merge pull request #477 from salesforcecli/chore/fix-failing-nuts
fix: report in-progress status when agent test run --wait times out @W-23915028@
2 parents 2a71f54 + c858714 commit 4f367fd

5 files changed

Lines changed: 106 additions & 49 deletions

File tree

‎src/commands/agent/test/resume.ts‎

Lines changed: 12 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -141,10 +141,18 @@ export default class AgentTestResume extends SfCommand<AgentTestRunResult> {
141141
);
142142
}
143143

144-
if (completed) await agentTestCache.removeCacheEntry(runId);
145-
146144
this.mso.stop();
147145

146+
// A client-side poll timeout does not mean the run finished — it is still running
147+
// server-side. Leave the cache entry in place so the run can be resumed again (poll() has
148+
// already printed the resume hint) and report the in-progress status rather than masking the
149+
// timeout as COMPLETED with no results.
150+
if (!completed || !response) {
151+
return { status: 'IN_PROGRESS', runId };
152+
}
153+
154+
await agentTestCache.removeCacheEntry(runId);
155+
148156
await handleTestResults({
149157
id: runId,
150158
format: resultFormat ?? flags['result-format'],
@@ -157,16 +165,11 @@ export default class AgentTestResume extends SfCommand<AgentTestRunResult> {
157165
// Set exit code to 1 only for execution errors (tests couldn't run properly)
158166
// Test assertion failures are business logic and should not affect exit code
159167
// Only applicable to legacy responses (Agentforce Studio doesn't have test case status)
160-
if (
161-
response &&
162-
'subjectName' in response &&
163-
response.testCases.some((tc) => 'status' in tc && tc.status === 'ERROR')
164-
) {
168+
if ('subjectName' in response && response.testCases.some((tc) => 'status' in tc && tc.status === 'ERROR')) {
165169
process.exitCode = 1;
166170
}
167171

168-
// eslint-disable-next-line @typescript-eslint/no-unnecessary-type-assertion
169-
return { ...response!, runId, status: 'COMPLETED' } as AgentTestRunResult;
172+
return { ...response, runId, status: 'COMPLETED' } as AgentTestRunResult;
170173
}
171174

172175
protected catch(error: Error | SfError | CLIError): Promise<never> {

‎src/commands/agent/test/run.ts‎

Lines changed: 10 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -178,10 +178,18 @@ export default class AgentTestRun extends SfCommand<AgentTestRunResult> {
178178
);
179179
}
180180

181-
if (completed) await agentTestCache.removeCacheEntry(response.runId);
182-
183181
this.mso.stop();
184182

183+
// A client-side poll timeout does not mean the run finished — it is still running
184+
// server-side. Leave the cache entry in place so `agent test resume` works (poll() has
185+
// already printed the resume hint) and report the in-progress status rather than masking
186+
// the timeout as COMPLETED with no results.
187+
if (!completed || !detailsResponse) {
188+
return { status: 'IN_PROGRESS', runId: response.runId };
189+
}
190+
191+
await agentTestCache.removeCacheEntry(response.runId);
192+
185193
await handleTestResults({
186194
id: response.runId,
187195
format: flags['result-format'],
@@ -195,7 +203,6 @@ export default class AgentTestRun extends SfCommand<AgentTestRunResult> {
195203
// Test assertion failures are business logic and should not affect exit code
196204
// Only applicable to legacy responses (Agentforce Studio doesn't have test case status)
197205
if (
198-
detailsResponse &&
199206
'subjectName' in detailsResponse &&
200207
detailsResponse.testCases.some((tc) => 'status' in tc && tc.status === 'ERROR')
201208
) {

‎src/testStages.ts‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -113,7 +113,10 @@ export class TestStages {
113113
this.stop('async');
114114
this.ux.log(`Client timed out after ${wait.minutes} minutes.`);
115115
this.ux.log(`Run ${colorize('dim', `sf agent test resume --job-id ${id}`)} to resuming watching this test.`);
116-
return { completed: true };
116+
// The client stopped watching, but the run is still going server-side — it did NOT
117+
// complete. Report completed:false so the caller preserves the cache entry (for
118+
// `agent test resume`) and does not mask the timeout as a COMPLETED result.
119+
return { completed: false };
117120
} else {
118121
this.error();
119122
throw e;

‎test/nuts/z2.agent.publish.nut.ts‎

Lines changed: 19 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -87,6 +87,25 @@ describe('agent publish authoring-bundle NUTs', function () {
8787
connection = org.getConnection();
8888
});
8989

90+
// Per-attempt timing so a slow publish is visible in CI. This fires on EVERY attempt
91+
// (including the ones mocha's this.retries() would otherwise hide by reporting only the
92+
// final pass), so a stalled first attempt shows up here as a large elapsed with a failed
93+
// state. Paired with the bounded publish-poll timeout, a genuine stall also surfaces which
94+
// poll hung via its error (agentRetrievalError vs authoringBundleDeploymentError).
95+
let attemptStart = 0;
96+
beforeEach(() => {
97+
attemptStart = Date.now();
98+
});
99+
afterEach(function () {
100+
const elapsedSec = Math.round((Date.now() - attemptStart) / 1000);
101+
// eslint-disable-next-line no-console
102+
console.log(
103+
`[z2 publish timing] "${this.currentTest?.title ?? 'unknown'}" attempt took ${elapsedSec}s (${
104+
this.currentTest?.state ?? 'unknown'
105+
})`
106+
);
107+
});
108+
90109
it('should publish a new agent (first version)', async function () {
91110
// Increase timeout to 30 minutes since deployment can take a long time
92111
this.timeout(30 * 60 * 1000); // 30 minutes

‎test/nuts/z4.agent.test.AFS.nut.ts‎

Lines changed: 61 additions & 36 deletions
Original file line numberDiff line numberDiff line change
@@ -123,9 +123,11 @@ describe('agent test (agentforce-studio)', function () {
123123
writeFileSync(join(metaDir, `${afsTestName}.aiTestingDefinition-meta.xml`), metaXml, 'utf8');
124124
console.log(`Wrote AiTestingDefinition metadata to ${metaDir}`);
125125

126-
// Deploy the definition
126+
// Deploy the definition. Scope to just the file we authored — deploying the whole
127+
// aiTestingDefinitions dir would also sweep in static fixtures (e.g. ReturnsCheckoutSuite,
128+
// which targets a non-deployed multi-agent subject) and fail server-side validation.
127129
const cs = await ComponentSetBuilder.build({
128-
sourcepath: [metaDir],
130+
sourcepath: [join(metaDir, `${afsTestName}.aiTestingDefinition-meta.xml`)],
129131
});
130132
const deploy = await cs.deploy({ usernameOrConnection: getUsername() });
131133
const deployResult = await deploy.pollStatus({ frequency: Duration.seconds(10), timeout: Duration.minutes(10) });
@@ -139,8 +141,10 @@ describe('agent test (agentforce-studio)', function () {
139141
console.log(`Deployed AiTestingDefinition '${afsTestName}'`);
140142
});
141143

142-
// Set by the run test, consumed by the results tests (Mocha runs describes sequentially)
143-
let completedRunId: string;
144+
// Set by the run test, consumed by the results/resume tests (Mocha runs describes sequentially).
145+
// The AFS eval can outlast the client --wait window, so the run may still be IN_PROGRESS here.
146+
let sharedRunId: string;
147+
let primaryRunCompleted = false;
144148

145149
describe('agent test list', () => {
146150
it('should include the AFS test definition in list', async () => {
@@ -162,20 +166,26 @@ describe('agent test (agentforce-studio)', function () {
162166
{ ensureExitCode: 0 }
163167
).jsonOutput;
164168

165-
expect(output?.result.status).to.equal('COMPLETED');
166-
expect(output?.result.runId.startsWith('3A2')).to.be.true;
167169
const result = output?.result as AgentTestRunResult & { testCases?: unknown[] };
168-
expect(result?.testCases).to.be.an('array');
170+
// The AFS eval can take longer than the --wait window. A client-side timeout is a valid
171+
// outcome that now reports IN_PROGRESS (rather than masking as COMPLETED), so accept either
172+
// and assert only the run shape. testCases is only guaranteed once the run has COMPLETED.
173+
expect(result?.status).to.be.oneOf(['COMPLETED', 'IN_PROGRESS']);
174+
expect(result?.runId.startsWith('3A2')).to.be.true;
169175
expect(result).to.not.have.property('subjectName');
176+
if (result?.status === 'COMPLETED') {
177+
expect(result?.testCases).to.be.an('array');
178+
primaryRunCompleted = true;
179+
}
170180

171-
completedRunId = output!.result.runId;
181+
sharedRunId = output!.result.runId;
172182
});
173183
});
174184

175185
describe('agent test results', () => {
176186
it('should fetch AFS results by job ID (json)', async () => {
177187
const output = execCmd<AgentTestResultsResult>(
178-
`agent test results --job-id ${completedRunId} --target-org ${getUsername()} --json`,
188+
`agent test results --job-id ${sharedRunId} --target-org ${getUsername()} --json`,
179189
{ ensureExitCode: 0 }
180190
).jsonOutput;
181191

@@ -187,15 +197,15 @@ describe('agent test (agentforce-studio)', function () {
187197

188198
it('should support human result format', () => {
189199
const output = execCmd(
190-
`agent test results --job-id ${completedRunId} --result-format human --target-org ${getUsername()}`,
200+
`agent test results --job-id ${sharedRunId} --result-format human --target-org ${getUsername()}`,
191201
{ ensureExitCode: 0 }
192202
);
193203
expect(output.shellOutput.stdout).to.be.a('string').with.length.greaterThan(0);
194204
});
195205

196206
it('should support junit result format', () => {
197207
const output = execCmd(
198-
`agent test results --job-id ${completedRunId} --result-format junit --target-org ${getUsername()}`,
208+
`agent test results --job-id ${sharedRunId} --result-format junit --target-org ${getUsername()}`,
199209
{ ensureExitCode: 0 }
200210
);
201211
expect(output.shellOutput.stdout).to.include('<?xml');
@@ -204,44 +214,59 @@ describe('agent test (agentforce-studio)', function () {
204214

205215
it('should support tap result format', () => {
206216
const output = execCmd(
207-
`agent test results --job-id ${completedRunId} --result-format tap --target-org ${getUsername()}`,
217+
`agent test results --job-id ${sharedRunId} --result-format tap --target-org ${getUsername()}`,
208218
{ ensureExitCode: 0 }
209219
);
210220
expect(output.shellOutput.stdout).to.include('TAP version 13');
211221
});
212222
});
213223

214224
describe('agent test resume', () => {
215-
it('should start async then resume by job ID, and support --use-most-recent', async () => {
216-
// Clear any stale entries before the run
217-
const cacheBefore = await AgentTestCache.create();
218-
cacheBefore.clear();
219-
await cacheBefore.write();
225+
it('should resume the in-flight run (or a fresh async run) and honor the cache', async function () {
226+
this.timeout(30 * 60 * 1000);
220227

221-
// One async start covers both resume paths
222-
const runResult = execCmd<AgentTestRunResult>(
223-
`agent test run --api-name ${afsTestName} --target-org ${getUsername()} --json`,
224-
{ ensureExitCode: 0 }
225-
).jsonOutput;
228+
if (primaryRunCompleted) {
229+
// The primary --wait run already finished, so its cache entry was removed and no run is
230+
// in flight. Start a fresh async run to exercise NEW + cache-write, then resume by job-id.
231+
const cacheBefore = await AgentTestCache.create();
232+
cacheBefore.clear();
233+
await cacheBefore.write();
226234

227-
expect(runResult?.result.runId.startsWith('3A2')).to.be.true;
228-
expect(runResult?.result.status).to.equal('NEW');
235+
const runResult = execCmd<AgentTestRunResult>(
236+
`agent test run --api-name ${afsTestName} --target-org ${getUsername()} --json`,
237+
{ ensureExitCode: 0 }
238+
).jsonOutput;
229239

230-
// Re-read from disk — the run command wrote the cache entry in a subprocess
231-
const cache = await AgentTestCache.create();
232-
expect(cache.resolveFromCache().runnerType).to.equal('agentforce-studio');
240+
expect(runResult?.result.runId.startsWith('3A2')).to.be.true;
241+
expect(runResult?.result.status).to.equal('NEW');
233242

234-
const output = execCmd<AgentTestRunResult>(
235-
`agent test resume --job-id ${runResult?.result.runId} --target-org ${getUsername()} --json`,
236-
{ ensureExitCode: 0 }
237-
).jsonOutput;
243+
// Re-read from disk — the run command wrote the cache entry in a subprocess
244+
const cache = await AgentTestCache.create();
245+
expect(cache.resolveFromCache().runnerType).to.equal('agentforce-studio');
246+
247+
const output = execCmd<AgentTestRunResult>(
248+
`agent test resume --job-id ${runResult?.result.runId} --target-org ${getUsername()} --json`,
249+
{ ensureExitCode: 0 }
250+
).jsonOutput;
251+
252+
// Resume may itself time out (the eval can outlast the poll window); both are valid.
253+
expect(output?.result.status).to.be.oneOf(['COMPLETED', 'IN_PROGRESS']);
254+
expect(output?.result.runId.startsWith('3A2')).to.be.true;
255+
} else {
256+
// The primary --wait run timed out client-side and is still running server-side. Our fix
257+
// preserves the cache entry on a timeout, so resume --use-most-recent picks it up. Starting
258+
// a NEW run here would collide — AFS allows only one run per definition at a time.
259+
const cache = await AgentTestCache.create();
260+
expect(cache.resolveFromCache().runnerType).to.equal('agentforce-studio');
238261

239-
expect(output?.result.status).to.equal('COMPLETED');
240-
expect(output?.result.runId.startsWith('3A2')).to.be.true;
262+
const output = execCmd<AgentTestRunResult>(
263+
`agent test resume --use-most-recent --target-org ${getUsername()} --json`,
264+
{ ensureExitCode: 0 }
265+
).jsonOutput;
241266

242-
// Re-read from disk — resume removes the entry in a subprocess
243-
const cacheAfter = await AgentTestCache.create();
244-
expect(() => cacheAfter.resolveFromCache()).to.throw('Could not find a runId to resume');
267+
expect(output?.result.status).to.be.oneOf(['COMPLETED', 'IN_PROGRESS']);
268+
expect(output?.result.runId).to.equal(sharedRunId);
269+
}
245270
});
246271
});
247272

0 commit comments

Comments
 (0)