Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
21 commits
Select commit Hold shift + click to select a range
0edbb22
feat(webapp): pin the query boundary end-to-end and cap the query retry
kathiekiwi Aug 10, 2026
3fa21ef
Merge remote-tracking branch 'origin/fix/watch-mode-keepalive-tri-130…
kathiekiwi Aug 10, 2026
acd54e0
Merge remote-tracking branch 'origin/fix/watch-mode-keepalive-tri-130…
kathiekiwi Aug 10, 2026
25fe809
merge: propagate review fixes from fix/watch-mode-keepalive-tri-13065
kathiekiwi Aug 10, 2026
2361556
merge: propagate wave-2 review fixes from fix/watch-mode-keepalive-tr…
kathiekiwi Aug 10, 2026
4ed6b99
merge: propagate org-purge best-effort from fix/watch-mode-keepalive-…
kathiekiwi Aug 10, 2026
614a7a9
fix(dashboard-agent): only count SQL errors toward the query-failure cap
kathiekiwi Aug 10, 2026
24444be
merge: query-failure cap SQL errors only review-comment fixes
kathiekiwi Aug 10, 2026
9a78579
merge: propagate review-comment fixes from fix/watch-mode-keepalive-t…
kathiekiwi Aug 10, 2026
50df6c3
merge: propagate second-pass fixes from fix/watch-mode-keepalive-tri-…
kathiekiwi Aug 11, 2026
2026a4e
merge: propagate server-changes consolidation from fix/watch-mode-kee…
kathiekiwi Aug 11, 2026
ed2c5c2
merge: propagate changeset consolidation and note restoration from fi…
kathiekiwi Aug 11, 2026
dbd216b
merge: propagate base UI relocation + drizzle attribution
kathiekiwi Aug 11, 2026
cf7fa9a
merge: propagate the tsql linter test fix from fix/watch-mode-keepali…
kathiekiwi Aug 11, 2026
dca2118
merge: propagate card-test relocation
kathiekiwi Aug 11, 2026
bf6a051
chore: merge fix/watch-mode-keepalive-tri-13065 (main sync)
kathiekiwi Aug 11, 2026
dd22919
chore: merge fix/watch-mode-keepalive-tri-13065 (review fixes)
kathiekiwi Aug 11, 2026
76560bb
chore: merge fix/watch-mode-keepalive-tri-13065 (review fixes round 2)
kathiekiwi Aug 11, 2026
14ea61c
chore: merge fix/watch-mode-keepalive-tri-13065 (review fixes round 3)
kathiekiwi Aug 11, 2026
66b6dfa
chore: merge fix/watch-mode-keepalive-tri-13065 (review fixes round 3)
kathiekiwi Aug 11, 2026
d329967
chore: merge fix/watch-mode-keepalive-tri-13065 (composer escape foll…
kathiekiwi Aug 11, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions .server-changes/query-boundary-and-retry-cap.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
---
area: webapp
type: improvement
---

Queries stay read-only, and the agent now stops after a few failed queries in a row and answers with what it found instead of spending the whole reply retrying.
1 change: 1 addition & 0 deletions apps/webapp/app/services/queryService.server.ts
Original file line number Diff line number Diff line change
Expand Up @@ -397,6 +397,7 @@ export async function executeQuery<TOut extends z.ZodSchema>(
...getDefaultClickhouseSettings(),
...queryCacheSettings,
...baseOptions.clickhouseSettings, // Allow caller overrides if needed
readonly: "1", // Not overridable: every query through here is read-only.
Comment thread
kathiekiwi marked this conversation as resolved.
},
querySettings: {
maxRows: env.QUERY_CLICKHOUSE_MAX_RETURNED_ROWS,
Expand Down
168 changes: 168 additions & 0 deletions apps/webapp/test/queryRouteReadOnly.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,168 @@
import { generateJWT } from "@trigger.dev/core/v3/jwt";
import { beforeEach, describe, expect, it, vi } from "vitest";

/**
* The query API is read-only, and the grammar is what enforces it. A parser test alone would
* stay green if the route ever compiled agent SQL somewhere else, so these drive the real route
* with a real signed environment JWT and stub only the ClickHouse client. A write must be
* refused before anything reaches ClickHouse.
*/

const ENVIRONMENT_ID = "env_1234";
const API_KEY = "tr_dev_abcdefghijklmnop";

const environment = {
id: ENVIRONMENT_ID,
type: "DEVELOPMENT",
slug: "dev",
branchName: null,
apiKey: API_KEY,
organizationId: "org_1",
projectId: "proj_1",
archivedAt: null,
concurrencyLimitBurstFactor: { toNumber: () => 1 },
maximumConcurrencyLimit: 10,
project: { id: "proj_1", externalRef: "proj_ref", deletedAt: null },
organization: { id: "org_1" },
orgMember: null,
parentEnvironment: null,
};

const mocks = vi.hoisted(() => ({
runtimeEnvironmentFindFirst: vi.fn(),
queryWithStats: vi.fn(),
customerQueryCreate: vi.fn(),
}));

vi.mock("~/db.server", () => {
const client = {
runtimeEnvironment: {
findFirst: mocks.runtimeEnvironmentFindFirst,
findMany: async () => [],
},
revokedApiKey: { findMany: async () => [], findFirst: async () => null },
project: { findMany: async () => [] },
customerQuery: { findFirst: async () => null, create: mocks.customerQueryCreate },
};
return { prisma: client, $replica: client };
});
Comment thread
kathiekiwi marked this conversation as resolved.
vi.mock("~/env.server", () => ({
env: {
SESSION_SECRET: "test-session-secret",
QUERY_CLICKHOUSE_MAX_EXECUTION_TIME: "30",
QUERY_CLICKHOUSE_MAX_MEMORY_USAGE: 1000000,
QUERY_CLICKHOUSE_MAX_AST_ELEMENTS: 50000,
QUERY_CLICKHOUSE_MAX_EXPANDED_AST_ELEMENTS: 500000,
QUERY_CLICKHOUSE_MAX_BYTES_BEFORE_EXTERNAL_GROUP_BY: 1000000,
QUERY_CLICKHOUSE_MAX_RETURNED_ROWS: 1000,
},
}));
vi.mock("~/services/clickhouse/clickhouseFactoryInstance.server", () => ({
clickhouseFactory: {
getClickhouseForOrganization: async () => ({
reader: { queryWithStats: mocks.queryWithStats },
}),
},
}));
vi.mock("~/services/platform.v3.server", () => ({ getLimit: async () => 30 }));
vi.mock("~/services/queryConcurrencyLimiter.server", () => ({
queryConcurrencyLimiter: {
acquire: async () => ({ success: true }),
release: async () => {},
},
DEFAULT_ORG_CONCURRENCY_LIMIT: 10,
GLOBAL_CONCURRENCY_LIMIT: 100,
}));
vi.mock("~/services/logger.server", () => ({
logger: { debug: vi.fn(), error: vi.fn(), warn: vi.fn(), info: vi.fn() },
}));
vi.mock("~/v3/services/worker/workerGroupTokenService.server", () => ({
WorkerGroupTokenService: class {},
}));
vi.mock("~/v3/services/common.server", () => ({ ServiceValidationError: class extends Error {} }));
vi.mock("@internal/run-engine", () => ({ EngineServiceValidationError: class extends Error {} }));
Comment thread
coderabbitai[bot] marked this conversation as resolved.
Comment on lines +37 to +83

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🟡 New tests rely on mocking, which the repository's test guidance forbids

The new test suites stub out the database, environment config, ClickHouse factory, logger and concurrency limiter with mock objects (vi.mock("~/db.server", ...) at apps/webapp/test/queryRouteReadOnly.test.ts:37-83), whereas the repository's testing guidance requires real containers instead of mocks.
Impact: The tests can keep passing while the real database, environment and ClickHouse wiring drifts, so the read-only guarantee they claim to pin is only proven against stand-ins.

Which rule is violated and where

AGENTS.md ("Testing") states: "We use vitest exclusively. Never mock anything - use testcontainers instead." Both new files rely on mocking: apps/webapp/test/queryRouteReadOnly.test.ts:31-83 mocks ~/db.server, ~/env.server, the ClickHouse factory instance, the platform limits service, the concurrency limiter and the logger; internal-packages/dashboard-agent/src/tool-query-retry-cap.test.ts:12-26 hand-builds a fake DashboardAgentApiClient. The webapp suite has an existing test/ layout with container-based e2e configs (apps/webapp/test/README.md) that these route-level assertions could use instead.

Open in Devin Review

Was this helpful? React with 👍 or 👎 to provide feedback.


import { action } from "~/routes/api.v1.query";
import { executeQuery } from "~/services/queryService.server";

/** The claims the env-JWT exchange mints (api.v1.projects.$projectRef.$env.jwt.ts). */
function mintEnvJwt(scopes: string[]) {
return generateJWT({
secretKey: API_KEY,
payload: {
sub: ENVIRONMENT_ID,
pub: true,
scopes,
act: { sub: "usr_1", client: "dashboard-agent" },
},
expirationTime: "1h",
});
}

async function runQuery(query: string): Promise<{ status: number; body: any }> {
const jwt = await mintEnvJwt(["read:query"]);
const response = await action({
request: new Request("https://api.trigger.dev/api/v1/query", {
method: "POST",
headers: { Authorization: `Bearer ${jwt}`, "Content-Type": "application/json" },
body: JSON.stringify({ query }),
}),
params: {},
context: {},
} as any);
return { status: response.status, body: await response.json() };
}

describe("the query API route", () => {
beforeEach(() => {
vi.clearAllMocks();
mocks.runtimeEnvironmentFindFirst.mockResolvedValue(environment);
mocks.customerQueryCreate.mockResolvedValue({ id: "cq_1" });
mocks.queryWithStats.mockReturnValue(async () => [null, { rows: [], stats: {} }]);
});

// Pins the seam the two refusals assert against: a read really does reach ClickHouse here,
// so `not.toHaveBeenCalled()` below means refused, not unreachable.
it("runs a read against ClickHouse", async () => {
const result = await runQuery("SELECT count() FROM runs");

expect(result.status).toBe(200);
expect(mocks.queryWithStats).toHaveBeenCalled();
});

it("refuses a write smuggled in as a second statement", async () => {
const result = await runQuery("SELECT 1 FROM runs; DROP TABLE runs");

expect(result.status).toBe(400);
expect(mocks.queryWithStats).not.toHaveBeenCalled();
});

it("refuses a mutating statement", async () => {
const result = await runQuery("INSERT INTO runs (task_identifier) VALUES ('x')");

expect(result.status).toBe(400);
expect(mocks.queryWithStats).not.toHaveBeenCalled();
});
});

describe("the query service", () => {
beforeEach(() => {
vi.clearAllMocks();
mocks.queryWithStats.mockReturnValue(async () => [null, { rows: [], stats: {} }]);
});

it("keeps ClickHouse read-only when a caller overrides the settings", async () => {
await executeQuery({
name: "test-query",
query: "SELECT count() FROM runs",
scope: "environment",
organizationId: "org_1",
projectId: "proj_1",
environmentId: ENVIRONMENT_ID,
clickhouseSettings: { readonly: "0" },
} as any);

expect(mocks.queryWithStats).toHaveBeenCalled();
expect(mocks.queryWithStats.mock.calls[0][0].settings.readonly).toBe("1");
});
});
23 changes: 22 additions & 1 deletion internal-packages/dashboard-agent/src/tool-api.ts
Original file line number Diff line number Diff line change
Expand Up @@ -177,6 +177,9 @@ export function withLiveState(metrics: unknown, queueType: "task" | "custom", li
};
}

/** Failed `run_query` calls in a row before the tool tells the model to stop and answer. */
export const MAX_CONSECUTIVE_QUERY_FAILURES = 3;

export function buildApiTools(args: {
ctx: DashboardAgentToolContext;
client: DashboardAgentApiClient;
Expand All @@ -186,6 +189,12 @@ export function buildApiTools(args: {
const { userActorToken, projectRef, environmentName, environmentBranch } = ctx;
const { origin, hasAuth, envApiGet, postQuery, validateChartQuery } = client;

// A failed query hands the model the database error to fix, and it usually does. When it
// doesn't, the only other limit is the turn's 10 steps, so one broken query can eat the
// whole turn and leave the user with no answer at all. This tool set is built per turn,
// so the counter caps consecutive failures within one turn.
let consecutiveQueryFailures = 0;
Comment thread
kathiekiwi marked this conversation as resolved.
Comment thread
kathiekiwi marked this conversation as resolved.

return {
list_projects: tool({
...listProjectsSchema,
Expand Down Expand Up @@ -341,7 +350,19 @@ export function buildApiTools(args: {
execute: async ({ query, period }) => {
const result = await postQuery(query, period);
if (isEnvUnavailable(result)) return envUnavailableError(result, "query");
if (!result.ok) return { error: result.error };
if (!result.ok) {
// Only SQL errors count toward the cap; transport errors are transient.
if (result.kind === "query") {
consecutiveQueryFailures++;
Comment thread
kathiekiwi marked this conversation as resolved.
if (consecutiveQueryFailures >= MAX_CONSECUTIVE_QUERY_FAILURES) {
return {
error: `${result.error} That is ${consecutiveQueryFailures} queries in a row that failed. Stop querying and answer the user with what you already have.`,
};
}
}
return { error: result.error };
}
Comment thread
devin-ai-integration[bot] marked this conversation as resolved.
consecutiveQueryFailures = 0;
Comment thread
devin-ai-integration[bot] marked this conversation as resolved.
Comment on lines +353 to +365

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🔍 Chart-query validation failures bypass the retry cap

render_view validates chart queries through validateChartQuery, which also posts to the query endpoint (internal-packages/dashboard-agent/src/tool-api-client.ts), but a failure there returns an error prompting the model to "Fix the query ... and render the chart again" (internal-packages/dashboard-agent/src/tool-api.ts:407-412) without touching consecutiveQueryFailures. So the failure mode the cap is meant to prevent — a model burning the turn's step budget rewriting a query it can't fix — is still reachable via repeated render_view calls.

Open in Devin Review

Was this helpful? React with 👍 or 👎 to provide feedback.

const cap = 200;
const rows = result.rows;
return { rows: rows.slice(0, cap), rowCount: rows.length, truncated: rows.length > cap };
Expand Down
90 changes: 90 additions & 0 deletions internal-packages/dashboard-agent/src/tool-query-retry-cap.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,90 @@
import { describe, expect, it, vi } from "vitest";
import { buildApiTools, MAX_CONSECUTIVE_QUERY_FAILURES } from "./tool-api";
import type { DashboardAgentApiClient } from "./tool-api-client";

/**
* A failed query hands the model the database error to fix. Without a cap, the only other
* limit is the turn's step budget, so a model that keeps rewriting the same broken query
* burns the whole turn and the user gets no answer. After three failures in a row the tool
* tells it to stop and answer.
*/

function queryTool(postQuery: DashboardAgentApiClient["postQuery"]) {
const client = {
origin: "https://api.example.com",
hasAuth: true,
envApiGet: async () => ({ ok: false as const, status: 500 }),
postQuery,
validateChartQuery: async () => null,
} as unknown as DashboardAgentApiClient;
const tools = buildApiTools({
ctx: { userActorToken: "uat", apiOrigin: client.origin },
client,
renderInvestigations: (() => []) as any,
});
return (query: string) => (tools.run_query as any).execute({ query }, {} as any);
}

const failure = {
ok: false as const,
kind: "query" as const,
error: "Unknown expression identifier 'createdAt'.",
};
const transportFailure = {
ok: false as const,
kind: "transport" as const,
error: "The environment is temporarily unavailable.",
};
const success = { ok: true as const, rows: [{ n: 1 }] };

describe("run_query's consecutive-failure cap", () => {
it("keeps handing back the plain error until the cap", async () => {
const run = queryTool(async () => failure);

for (let attempt = 1; attempt < MAX_CONSECUTIVE_QUERY_FAILURES; attempt++) {
const result = await run("SELECT createdAt FROM runs");
expect(result.error).toBe(failure.error);
}
});

it("tells the model to stop and answer at the cap", async () => {
const run = queryTool(async () => failure);

let result: { error: string } = { error: "" };
for (let attempt = 0; attempt < MAX_CONSECUTIVE_QUERY_FAILURES; attempt++) {
result = await run("SELECT createdAt FROM runs");
}

expect(result.error).toContain(failure.error);
expect(result.error).toContain("answer the user with what you already have");
});

it("counts consecutive failures only, so a good query clears the count", async () => {
const postQuery = vi
.fn()
.mockResolvedValueOnce(failure)
.mockResolvedValueOnce(failure)
.mockResolvedValueOnce(success)
.mockResolvedValue(failure);
const run = queryTool(postQuery as any);

await run("bad");
await run("bad");
await run("good");
const result = await run("bad");

expect(result.error).toBe(failure.error);
});

it("does not count transport errors toward the cap", async () => {
const run = queryTool(async () => transportFailure);

let result: { error: string } = { error: "" };
for (let attempt = 0; attempt < MAX_CONSECUTIVE_QUERY_FAILURES + 2; attempt++) {
result = await run("SELECT createdAt FROM runs");
}

expect(result.error).toBe(transportFailure.error);
expect(result.error).not.toContain("answer the user with what you already have");
});
});
Loading