Skip to content

Commit b9e80bd

Browse files
carderneTrigger.dev RepoOps
authored andcommitted
perf(webapp,clickhouse): faster logs list and search queries
Mono-RevId: 7fd0191ab689a6a51bac2069f8297ff7e996e25a
1 parent fc69d15 commit b9e80bd

7 files changed

Lines changed: 64 additions & 8 deletions

File tree

Lines changed: 6 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,6 @@
1+
---
2+
area: webapp
3+
type: fix
4+
---
5+
6+
The Logs page loads and searches faster, and busy environments no longer fail to load it. Searches that are still too expensive now show a message suggesting a narrower range instead of a generic error.

‎apps/webapp/app/env.server.ts‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -2300,8 +2300,8 @@ const EnvironmentSchema = z
23002300
.int()
23012301
.default(256_000_000),
23022302
CLICKHOUSE_LOGS_LIST_MAX_THREADS: z.coerce.number().int().default(2),
2303-
CLICKHOUSE_LOGS_LIST_MAX_ROWS_TO_READ: z.coerce.number().int().default(10_000_000),
2304-
CLICKHOUSE_LOGS_LIST_MAX_EXECUTION_TIME: z.coerce.number().int().default(120),
2303+
// The HTTP request timeout is derived from this so ClickHouse gives up before the client does.
2304+
CLICKHOUSE_LOGS_LIST_MAX_EXECUTION_TIME: z.coerce.number().int().default(15),
23052305
// Bound read-in-order memory on object-storage reads: each part opens a per-column read
23062306
// stream, and the default ~1 MiB+ S3 buffers dominate peak memory. These two byte sizes
23072307
// cap the per-stream buffers and exist on every supported ClickHouse, so they are always on.

‎apps/webapp/app/presenters/v3/LogsListPresenter.server.ts‎

Lines changed: 21 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,9 @@
1-
import { type ClickHouse, type WhereCondition } from "@internal/clickhouse";
1+
import {
2+
type ClickHouse,
3+
isClickhouseResourceLimitError,
4+
TASK_EVENT_SEARCH_MAX_TRIGGERED_AFTER_INSERT_MS,
5+
type WhereCondition,
6+
} from "@internal/clickhouse";
27
import { type PrismaClientOrTransaction } from "@trigger.dev/database";
38
import { z } from "zod";
49
import { EVENT_STORE_TYPES, getConfiguredEventRepository } from "~/v3/eventRepository/index.server";
@@ -274,6 +279,14 @@ export class LogsListPresenter extends BasePresenter {
274279
queryBuilder.where("triggered_timestamp >= {triggeredAtStart: DateTime64(3)}", {
275280
triggeredAtStart: convertDateToClickhouseDateTime(effectiveFrom),
276281
});
282+
// The table is partitioned by inserted_at, not event time. Writers clamp
283+
// triggered_timestamp to at most inserted_at plus a fixed delay, so this bound only
284+
// prunes partitions that cannot hold a matching row.
285+
queryBuilder.where("inserted_at >= {insertedAtStart: DateTime64(3)}", {
286+
insertedAtStart: convertDateToClickhouseDateTime(
287+
new Date(effectiveFrom.getTime() - TASK_EVENT_SEARCH_MAX_TRIGGERED_AFTER_INSERT_MS)
288+
),
289+
});
277290
}
278291

279292
// Task filter (applies directly to ClickHouse)
@@ -354,6 +367,13 @@ export class LogsListPresenter extends BasePresenter {
354367

355368
const [queryError, queryResult] = await runQuery();
356369
if (queryError) {
370+
if (isClickhouseResourceLimitError(queryError)) {
371+
throw new ServiceValidationError(
372+
searchTerm === undefined
373+
? "These logs took too long to load. Try a shorter time range or add a filter."
374+
: "This search took too long. Try a shorter time range, a more specific search, or add a filter."
375+
);
376+
}
357377
throw queryError;
358378
}
359379

‎apps/webapp/app/services/clickhouse/clickhouseFactory.server.ts‎

Lines changed: 16 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -7,6 +7,9 @@ import { singleton } from "~/utils/singleton";
77
import type { OrganizationDataStoresRegistry } from "~/services/dataStores/organizationDataStoresRegistry.server";
88
import { type IEventRepository } from "~/v3/eventRepository/eventRepository.types";
99

10+
// Declared before the singletons below, which initialize clients at module load.
11+
const LOGS_LIST_REQUEST_TIMEOUT_MARGIN_SECONDS = 5;
12+
1013
// ---------------------------------------------------------------------------
1114
// Default clients (singleton per process)
1215
// ---------------------------------------------------------------------------
@@ -70,15 +73,23 @@ function getLogsListClickhouseSettings() {
7073
filesystem_cache_prefer_bigger_buffer_size:
7174
env.CLICKHOUSE_LOGS_LIST_FILESYSTEM_CACHE_PREFER_BIGGER_BUFFER_SIZE,
7275
}),
73-
...(env.CLICKHOUSE_LOGS_LIST_MAX_ROWS_TO_READ && {
74-
max_rows_to_read: env.CLICKHOUSE_LOGS_LIST_MAX_ROWS_TO_READ.toString(),
75-
}),
76+
// No max_rows_to_read: ClickHouse checks it against the estimated rows of every selected
77+
// granule before reading, which rejects newest-first LIMIT queries on large environments that
78+
// would stop after a few thousand rows. max_execution_time bounds the work instead.
7679
...(env.CLICKHOUSE_LOGS_LIST_MAX_EXECUTION_TIME && {
7780
max_execution_time: env.CLICKHOUSE_LOGS_LIST_MAX_EXECUTION_TIME,
7881
}),
7982
};
8083
}
8184

85+
/** Outlasts the server-side limit so ClickHouse cancels the query and reports why. */
86+
function getLogsListRequestTimeoutMs() {
87+
if (!env.CLICKHOUSE_LOGS_LIST_MAX_EXECUTION_TIME) return undefined;
88+
return (
89+
(env.CLICKHOUSE_LOGS_LIST_MAX_EXECUTION_TIME + LOGS_LIST_REQUEST_TIMEOUT_MARGIN_SECONDS) * 1000
90+
);
91+
}
92+
8293
function initializeLogsClickhouseClient() {
8394
const url = new URL(
8495
env.LOGS_SEARCH_READER_CLICKHOUSE_URL ?? env.CLICKHOUSE_READER_URL ?? env.CLICKHOUSE_URL
@@ -95,6 +106,7 @@ function initializeLogsClickhouseClient() {
95106
logLevel: env.CLICKHOUSE_LOG_LEVEL,
96107
compression: { request: true },
97108
maxOpenConnections: env.CLICKHOUSE_MAX_OPEN_CONNECTIONS,
109+
requestTimeoutMs: getLogsListRequestTimeoutMs(),
98110
clickhouseSettings: getLogsListClickhouseSettings(),
99111
});
100112
}
@@ -592,6 +604,7 @@ function buildOrgClickhouseClient(url: string, clientType: ClientType): ClickHou
592604
logLevel: env.CLICKHOUSE_LOG_LEVEL,
593605
compression: { request: true },
594606
maxOpenConnections: env.CLICKHOUSE_MAX_OPEN_CONNECTIONS,
607+
requestTimeoutMs: getLogsListRequestTimeoutMs(),
595608
clickhouseSettings: getLogsListClickhouseSettings(),
596609
});
597610
case "engine":

‎internal-packages/clickhouse/src/taskEvents.ts‎

Lines changed: 6 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -386,6 +386,12 @@ export function getLogsSearchListQueryBuilder(ch: ClickhouseReader) {
386386
],
387387
settings: {
388388
use_query_condition_cache: 1,
389+
// The ngram text index covers every organization's rows in a part, so reading it costs more
390+
// than scanning one environment's time range, often by seconds when it is not cached.
391+
ignore_data_skipping_indices: "idx_search_text",
392+
// Hold the response until the query finishes so a limit error arrives as an HTTP error the
393+
// client turns into a QueryError, not mid-stream. Pages are a few hundred small rows.
394+
wait_end_of_query: 1,
389395
},
390396
});
391397

‎internal-packages/clickhouse/src/taskEventsSearch.ts‎

Lines changed: 9 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -25,6 +25,12 @@ export const TaskEventSearchV2Input = z.object({
2525

2626
export type TaskEventSearchV2Input = z.input<typeof TaskEventSearchV2Input>;
2727

28+
/**
29+
* `triggered_timestamp` is clamped to at most `inserted_at` plus this delay, so readers can derive a
30+
* safe `inserted_at` lower bound (the partition key) from an event-time lower bound.
31+
*/
32+
export const TASK_EVENT_SEARCH_MAX_TRIGGERED_AFTER_INSERT_MS = 5 * 60_000;
33+
2834
export const TASK_EVENT_SEARCH_V2_INSERT_COLUMNS = [
2935
"environment_id",
3036
"organization_id",
@@ -138,7 +144,9 @@ function errorMessage(attributes: unknown): string {
138144

139145
function triggeredTimestamp(startTime: string, duration: string, insertedAt: string): string {
140146
const completedAt = parseNanoseconds(startTime) + BigInt(duration);
141-
const latestAllowed = parseMilliseconds(insertedAt) + BigInt(5 * 60_000) * BigInt(1_000_000);
147+
const latestAllowed =
148+
parseMilliseconds(insertedAt) +
149+
BigInt(TASK_EVENT_SEARCH_MAX_TRIGGERED_AFTER_INSERT_MS) * BigInt(1_000_000);
142150
return formatNanoseconds(completedAt < latestAllowed ? completedAt : latestAllowed);
143151
}
144152

‎internal-packages/clickhouse/src/taskEventsSearchProjector.ts‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -3,6 +3,7 @@ import type { Result } from "@trigger.dev/core/v3";
33
import { z } from "zod";
44
import type { QueryError } from "./client/errors.js";
55
import type { ClickhouseWriter } from "./client/types.js";
6+
import { TASK_EVENT_SEARCH_MAX_TRIGGERED_AFTER_INSERT_MS } from "./taskEventsSearch.js";
67

78
export type TaskEventsSearchV2ProjectionWindow = {
89
start: Date;
@@ -74,7 +75,9 @@ FROM
7475
least(
7576
toInt128(toUnixTimestamp64Nano(start_time)) + toInt128(duration),
7677
toInt128(
77-
toUnixTimestamp64Nano(inserted_at + INTERVAL 5 MINUTE)
78+
toUnixTimestamp64Nano(
79+
inserted_at + INTERVAL ${TASK_EVENT_SEARCH_MAX_TRIGGERED_AFTER_INSERT_MS / 1000} SECOND
80+
)
7881
)
7982
)
8083
)

0 commit comments

Comments
 (0)