diff --git a/packages/mcp-core/src/skillDefinitions.json b/packages/mcp-core/src/skillDefinitions.json index 44b41ac87..4d56c16f0 100644 --- a/packages/mcp-core/src/skillDefinitions.json +++ b/packages/mcp-core/src/skillDefinitions.json @@ -209,7 +209,7 @@ }, { "name": "search_logs", - "description": "Search Sentry logs: application log entries, including error- and warning-severity log messages. Use for log counts, statistics, trends, and individual log lines.\n\n`query` is natural language (preferred) or Sentry search syntax; a configured agent translates it into query, fields, and sort.\n\nSupports aggregations ('warning logs by service'), individual entries ('error logs from the last hour'), and time series ('error logs per hour').\n\nNOT for exceptions/crashes (use search_errors). For requests or spans whose trace also has a matching log, use search_traces.\n\n\nsearch_logs(organizationSlug='my-org', query='error logs from the last hour')\nsearch_logs(organizationSlug='my-org', query='logs mentioning payment timeout in production')\nsearch_logs(organizationSlug='my-org', query='count warning logs by service over 7 days')\n\n\n\n- name/otherName notation means /; parse it directly, don't call find_organizations/find_projects.\n- Natural language is usually enough. Only pass fields/sort when you need exact columns or ordering.\n", + "description": "Search Sentry logs: application log entries, including error- and warning-severity log messages. Use for log counts, statistics, trends, and individual log lines.\n\n`query` is natural language (preferred) or Sentry search syntax; a configured agent translates it into query, fields, and sort.\n\nSupports aggregations ('warning logs by service'), individual entries ('error logs from the last hour'), and time series ('error logs per hour').\n\nNOT for exceptions/crashes (use search_errors). For requests or spans whose trace also has a matching log, use search_traces.\n\n\nsearch_logs(organizationSlug='my-org', query='error logs from the last hour')\nsearch_logs(organizationSlug='my-org', query='logs mentioning payment timeout in production')\nsearch_logs(organizationSlug='my-org', query='count warning logs by service over 7 days')\n\n\n\n- name/otherName notation means /; parse it directly, don't call find_organizations/find_projects.\n- Natural language is usually enough. Only pass fields/sort when you need exact columns or ordering.\n- Sentry search syntax for logs includes RE2 regex filters, never quoted: `message://^Timeout after \\d+ms//`.\n", "requiredScopes": ["event:read"] }, { @@ -304,7 +304,7 @@ }, { "name": "search_logs", - "description": "Search Sentry logs: application log entries, including error- and warning-severity log messages. Use for log counts, statistics, trends, and individual log lines.\n\n`query` is natural language (preferred) or Sentry search syntax; a configured agent translates it into query, fields, and sort.\n\nSupports aggregations ('warning logs by service'), individual entries ('error logs from the last hour'), and time series ('error logs per hour').\n\nNOT for exceptions/crashes (use search_errors). For requests or spans whose trace also has a matching log, use search_traces.\n\n\nsearch_logs(organizationSlug='my-org', query='error logs from the last hour')\nsearch_logs(organizationSlug='my-org', query='logs mentioning payment timeout in production')\nsearch_logs(organizationSlug='my-org', query='count warning logs by service over 7 days')\n\n\n\n- name/otherName notation means /; parse it directly, don't call find_organizations/find_projects.\n- Natural language is usually enough. Only pass fields/sort when you need exact columns or ordering.\n", + "description": "Search Sentry logs: application log entries, including error- and warning-severity log messages. Use for log counts, statistics, trends, and individual log lines.\n\n`query` is natural language (preferred) or Sentry search syntax; a configured agent translates it into query, fields, and sort.\n\nSupports aggregations ('warning logs by service'), individual entries ('error logs from the last hour'), and time series ('error logs per hour').\n\nNOT for exceptions/crashes (use search_errors). For requests or spans whose trace also has a matching log, use search_traces.\n\n\nsearch_logs(organizationSlug='my-org', query='error logs from the last hour')\nsearch_logs(organizationSlug='my-org', query='logs mentioning payment timeout in production')\nsearch_logs(organizationSlug='my-org', query='count warning logs by service over 7 days')\n\n\n\n- name/otherName notation means /; parse it directly, don't call find_organizations/find_projects.\n- Natural language is usually enough. Only pass fields/sort when you need exact columns or ordering.\n- Sentry search syntax for logs includes RE2 regex filters, never quoted: `message://^Timeout after \\d+ms//`.\n", "requiredScopes": ["event:read"] }, { @@ -470,7 +470,7 @@ }, { "name": "search_logs", - "description": "Search Sentry logs: application log entries, including error- and warning-severity log messages. Use for log counts, statistics, trends, and individual log lines.\n\n`query` is natural language (preferred) or Sentry search syntax; a configured agent translates it into query, fields, and sort.\n\nSupports aggregations ('warning logs by service'), individual entries ('error logs from the last hour'), and time series ('error logs per hour').\n\nNOT for exceptions/crashes (use search_errors). For requests or spans whose trace also has a matching log, use search_traces.\n\n\nsearch_logs(organizationSlug='my-org', query='error logs from the last hour')\nsearch_logs(organizationSlug='my-org', query='logs mentioning payment timeout in production')\nsearch_logs(organizationSlug='my-org', query='count warning logs by service over 7 days')\n\n\n\n- name/otherName notation means /; parse it directly, don't call find_organizations/find_projects.\n- Natural language is usually enough. Only pass fields/sort when you need exact columns or ordering.\n", + "description": "Search Sentry logs: application log entries, including error- and warning-severity log messages. Use for log counts, statistics, trends, and individual log lines.\n\n`query` is natural language (preferred) or Sentry search syntax; a configured agent translates it into query, fields, and sort.\n\nSupports aggregations ('warning logs by service'), individual entries ('error logs from the last hour'), and time series ('error logs per hour').\n\nNOT for exceptions/crashes (use search_errors). For requests or spans whose trace also has a matching log, use search_traces.\n\n\nsearch_logs(organizationSlug='my-org', query='error logs from the last hour')\nsearch_logs(organizationSlug='my-org', query='logs mentioning payment timeout in production')\nsearch_logs(organizationSlug='my-org', query='count warning logs by service over 7 days')\n\n\n\n- name/otherName notation means /; parse it directly, don't call find_organizations/find_projects.\n- Natural language is usually enough. Only pass fields/sort when you need exact columns or ordering.\n- Sentry search syntax for logs includes RE2 regex filters, never quoted: `message://^Timeout after \\d+ms//`.\n", "requiredScopes": ["event:read"] }, { diff --git a/packages/mcp-core/src/toolDefinitions.json b/packages/mcp-core/src/toolDefinitions.json index 9fa64538c..eb0fef4a2 100644 --- a/packages/mcp-core/src/toolDefinitions.json +++ b/packages/mcp-core/src/toolDefinitions.json @@ -7554,7 +7554,7 @@ }, { "name": "search_logs", - "description": "Search Sentry logs: application log entries, including error- and warning-severity log messages. Use for log counts, statistics, trends, and individual log lines.\n\n`query` is natural language (preferred) or Sentry search syntax; a configured agent translates it into query, fields, and sort.\n\nSupports aggregations ('warning logs by service'), individual entries ('error logs from the last hour'), and time series ('error logs per hour').\n\nNOT for exceptions/crashes (use search_errors). For requests or spans whose trace also has a matching log, use search_traces.\n\n\nsearch_logs(organizationSlug='my-org', query='error logs from the last hour')\nsearch_logs(organizationSlug='my-org', query='logs mentioning payment timeout in production')\nsearch_logs(organizationSlug='my-org', query='count warning logs by service over 7 days')\n\n\n\n- name/otherName notation means /; parse it directly, don't call find_organizations/find_projects.\n- Natural language is usually enough. Only pass fields/sort when you need exact columns or ordering.\n", + "description": "Search Sentry logs: application log entries, including error- and warning-severity log messages. Use for log counts, statistics, trends, and individual log lines.\n\n`query` is natural language (preferred) or Sentry search syntax; a configured agent translates it into query, fields, and sort.\n\nSupports aggregations ('warning logs by service'), individual entries ('error logs from the last hour'), and time series ('error logs per hour').\n\nNOT for exceptions/crashes (use search_errors). For requests or spans whose trace also has a matching log, use search_traces.\n\n\nsearch_logs(organizationSlug='my-org', query='error logs from the last hour')\nsearch_logs(organizationSlug='my-org', query='logs mentioning payment timeout in production')\nsearch_logs(organizationSlug='my-org', query='count warning logs by service over 7 days')\n\n\n\n- name/otherName notation means /; parse it directly, don't call find_organizations/find_projects.\n- Natural language is usually enough. Only pass fields/sort when you need exact columns or ordering.\n- Sentry search syntax for logs includes RE2 regex filters, never quoted: `message://^Timeout after \\d+ms//`.\n", "inputSchema": { "type": "object", "properties": { diff --git a/packages/mcp-core/src/tools/catalog/search-events-environment-note.test.ts b/packages/mcp-core/src/tools/catalog/search-events-environment-note.test.ts index b8b4a05a1..612be897e 100644 --- a/packages/mcp-core/src/tools/catalog/search-events-environment-note.test.ts +++ b/packages/mcp-core/src/tools/catalog/search-events-environment-note.test.ts @@ -59,6 +59,21 @@ describe("collectRequestedEnvironments", () => { ), ).toEqual(["qa"]); }); + + it("keeps regex values with spaces and apostrophes as one token", () => { + expect( + collectRequestedEnvironments( + null, + "message://can't connect// environment:qa", + ), + ).toEqual(["qa"]); + expect( + collectRequestedEnvironments(null, "message://a environment:foo b//"), + ).toEqual([]); + expect( + collectRequestedEnvironments(null, "(message://a environment:foo b//)"), + ).toEqual([]); + }); }); describe("formatUnknownEnvironmentNote", () => { diff --git a/packages/mcp-core/src/tools/catalog/search-events.test.ts b/packages/mcp-core/src/tools/catalog/search-events.test.ts index 7de937048..885d81153 100644 --- a/packages/mcp-core/src/tools/catalog/search-events.test.ts +++ b/packages/mcp-core/src/tools/catalog/search-events.test.ts @@ -3073,6 +3073,115 @@ describe("search_events", () => { expect(result).toContain("TimeoutError"); }); + describe("with regex log queries", () => { + const regexSearchParams = { + organizationSlug: "test-org", + regionUrl: null, + projectSlug: null, + dataset: "logs" as const, + fields: null, + sort: null, + period: "24h", + limit: 10, + includeExplanation: false, + }; + const regexSearchContext = { + constraints: { + organizationSlug: null, + regionUrl: null, + projectSlug: null, + }, + accessToken: "test-token", + userId: "1", + }; + + const captureEventsQueries = () => { + const queries: Array = []; + mswServer.use( + http.get( + "https://sentry.io/api/0/organizations/test-org/events/", + ({ request }) => { + queries.push(new URL(request.url).searchParams.get("query")); + return HttpResponse.json({ data: [] }); + }, + ), + ); + return queries; + }; + + it("runs a regex-only query as written without the agent when fields and sort are explicit", async () => { + const queries = captureEventsQueries(); + + await searchEvents.handler( + { + ...regexSearchParams, + query: "message://^Timeout after \\d+ms//", + fields: ["timestamp", "message"], + sort: "-timestamp", + }, + regexSearchContext, + ); + + expect(mockGenerateText).not.toHaveBeenCalled(); + expect(queries).toEqual(["message://^Timeout after \\d+ms//"]); + }); + + it("keeps the regex filter when the agent rewrites it into a wildcard", async () => { + mockGenerateText.mockResolvedValueOnce( + mockAIResponse("logs", 'message:"*Timeout after*"'), + ); + const queries = captureEventsQueries(); + + await searchEvents.handler( + { ...regexSearchParams, query: "message://^Timeout after \\d+ms//" }, + regexSearchContext, + ); + + expect(mockGenerateText).toHaveBeenCalledTimes(1); + expect(queries).toEqual(["message://^Timeout after \\d+ms//"]); + }); + + it("accepts agent repairs that keep a regex value with spaces and apostrophes", async () => { + mockGenerateText.mockResolvedValueOnce( + mockAIResponse( + "logs", + "severity:error message://can't connect to \\w+// has:trace", + ), + ); + const queries = captureEventsQueries(); + + await searchEvents.handler( + { + ...regexSearchParams, + query: "message://can't connect to \\w+// severity:error", + }, + regexSearchContext, + ); + + expect(queries).toEqual([ + "severity:error message://can't connect to \\w+// has:trace", + ]); + }); + + it("uses the agent's rewrite of regex syntax outside logs", async () => { + mockGenerateText.mockResolvedValueOnce( + mockAIResponse("spans", 'span.description:"GET /api/*"'), + ); + const queries = captureEventsQueries(); + + await searchEvents.handler( + { + ...regexSearchParams, + dataset: "spans", + query: "span.description://^GET \\/api\\/\\d+//", + }, + regexSearchContext, + ); + + expect(queries).toEqual(['span.description:"GET /api/*"']); + }); + }); + it("keeps caller fields when the agent returns an empty fields array", async () => { let eventsRequestUrl: URL | undefined; @@ -3977,6 +4086,10 @@ describe("search_events", () => { it.each([ ["a structured query", { query: "span.op:http.client" }], + [ + "a regex-only logs query", + { dataset: "logs" as const, query: "message://^Timeout//" }, + ], ["explicit fields", { fields: ["span.description", "count()"] }], ["an explicit sort", { sort: "-count()" }], ])("should skip Seer for %s", async (_, overrides) => { diff --git a/packages/mcp-core/src/tools/catalog/search-logs.ts b/packages/mcp-core/src/tools/catalog/search-logs.ts index 1a60689ed..bd98ca29d 100644 --- a/packages/mcp-core/src/tools/catalog/search-logs.ts +++ b/packages/mcp-core/src/tools/catalog/search-logs.ts @@ -27,6 +27,7 @@ export default defineTool({ "", "- name/otherName notation means /; parse it directly, don't call find_organizations/find_projects.", "- Natural language is usually enough. Only pass fields/sort when you need exact columns or ordering.", + "- Sentry search syntax for logs includes RE2 regex filters, never quoted: `message://^Timeout after \\d+ms//`.", "", ].join("\n"), inputSchema: buildDatasetSearchInputSchema(), diff --git a/packages/mcp-core/src/tools/support/search-events/search.ts b/packages/mcp-core/src/tools/support/search-events/search.ts index 3879ccf45..a1c248af9 100644 --- a/packages/mcp-core/src/tools/support/search-events/search.ts +++ b/packages/mcp-core/src/tools/support/search-events/search.ts @@ -52,6 +52,7 @@ import { isAggregateQuery, isSemanticFilterDowngrade, looksLikeSentrySearchSyntax, + readRegexFilterValue, recordEventsSearchValidationTelemetry, validateEventsSearch, } from "./utils"; @@ -262,7 +263,8 @@ function tokenizeSearchQuery(query: string): string[] { let quote: '"' | "'" | null = null; let escaped = false; - for (const char of query) { + for (let i = 0; i < query.length; i += 1) { + const char = query[i]; if (escaped) { currentToken += char; escaped = false; @@ -283,6 +285,13 @@ function tokenizeSearchQuery(query: string): string[] { continue; } + const regexValue = readRegexFilterValue(query, i); + if (regexValue) { + currentToken += regexValue; + i += regexValue.length - 1; + continue; + } + if (char === '"' || char === "'") { currentToken += char; quote = char; @@ -322,6 +331,7 @@ function choosePreservingRepairedQuery(params: { originalQuery: string; repairedQuery?: string | null; filter?: string; + dataset: PublicEventsDataset | "replays"; }): string { const originalQuery = params.originalQuery.trim(); const repairedQuery = params.repairedQuery?.trim(); @@ -329,7 +339,7 @@ function choosePreservingRepairedQuery(params: { return appendSearchFilter(originalQuery, params.filter); } - if (isSemanticFilterDowngrade(originalQuery, repairedQuery)) { + if (isSemanticFilterDowngrade(originalQuery, repairedQuery, params.dataset)) { return appendSearchFilter(originalQuery, params.filter); } @@ -611,7 +621,10 @@ export async function runSearchEvents( setTargetTagsAndAttributes(params); const inputDataset = params.dataset ?? "errors"; - const hasStructuredQuery = looksLikeSentrySearchSyntax(params.query); + const hasStructuredQuery = looksLikeSentrySearchSyntax( + params.query, + inputDataset, + ); const canApplyEnvironmentFilter = inputDataset !== "replays" && isTraceItemDataset(inputDataset) && @@ -781,9 +794,14 @@ export async function runSearchEvents( originalQuery: params.query ?? "", repairedQuery: parsed.query, filter: environmentFilter, + dataset, }) - : looksLikeSentrySearchSyntax(params.query) && - isSemanticFilterDowngrade(params.query ?? "", parsed.query || "") + : looksLikeSentrySearchSyntax(params.query, dataset) && + isSemanticFilterDowngrade( + params.query ?? "", + parsed.query || "", + dataset, + ) ? (params.query ?? "") : parsed.query || ""; sortParam = diff --git a/packages/mcp-core/src/tools/support/search-events/utils.test.ts b/packages/mcp-core/src/tools/support/search-events/utils.test.ts index 4dbb0c158..b4d695726 100644 --- a/packages/mcp-core/src/tools/support/search-events/utils.test.ts +++ b/packages/mcp-core/src/tools/support/search-events/utils.test.ts @@ -297,6 +297,156 @@ describe("search query helpers", () => { expect(looksLikeSentrySearchSyntax("ERROR: service is down")).toBe(false); }); + it("should detect regex filters in logs queries", () => { + expect(looksLikeSentrySearchSyntax("message://^Timeout//", "logs")).toBe( + true, + ); + expect( + looksLikeSentrySearchSyntax("message://^Timeout after \\d+ms//", "logs"), + ).toBe(true); + expect( + looksLikeSentrySearchSyntax("!message://(?i)healthcheck//", "logs"), + ).toBe(true); + expect( + looksLikeSentrySearchSyntax("open http://example.com/a//b", "logs"), + ).toBe(false); + }); + + it("should detect regex filters inside parentheses and on Sentry's other key shapes", () => { + expect(looksLikeSentrySearchSyntax("(message://a b//)", "logs")).toBe(true); + expect( + looksLikeSentrySearchSyntax("tags[App:Region]://^us-//", "logs"), + ).toBe(true); + expect(looksLikeSentrySearchSyntax("flags[Feature:X]://y//", "logs")).toBe( + true, + ); + expect( + looksLikeSentrySearchSyntax("tags[foo, string]://a b//", "logs"), + ).toBe(true); + expect(looksLikeSentrySearchSyntax('"mykey"://a b//', "logs")).toBe(true); + expect(looksLikeSentrySearchSyntax("arr[*]://a b//", "logs")).toBe(true); + expect(looksLikeSentrySearchSyntax('"Note"message://a b//', "logs")).toBe( + true, + ); + expect(looksLikeSentrySearchSyntax("(level:error)", "logs")).toBe(false); + }); + + it("should not treat regex filters as search syntax outside logs", () => { + expect(looksLikeSentrySearchSyntax("message://^Timeout//")).toBe(false); + expect( + looksLikeSentrySearchSyntax("span.description://^GET \\/api//", "spans"), + ).toBe(false); + expect(looksLikeSentrySearchSyntax("message://^Timeout//", "errors")).toBe( + false, + ); + }); + + it("detects regex filters rewritten into plain or wildcard filters", () => { + expect( + isSemanticFilterDowngrade( + "message://^Timeout//", + 'message:"*Timeout*"', + "logs", + ), + ).toBe(true); + expect( + isSemanticFilterDowngrade( + "message://can't connect to \\w+//", + "message:*connect*", + "logs", + ), + ).toBe(true); + expect( + isSemanticFilterDowngrade( + "custom://^order-\\d+$// level:error", + 'level:error message:"*order-*"', + "logs", + ), + ).toBe(true); + expect( + isSemanticFilterDowngrade( + "custom://^order-\\d+$//", + "custom:order-*", + "logs", + ), + ).toBe(true); + + expect( + isSemanticFilterDowngrade( + "message://^Timeout//", + "message://^Timeout// severity:error", + "logs", + ), + ).toBe(false); + expect( + isSemanticFilterDowngrade( + "custom://^order-\\d+$//", + "tags[custom]://^order-\\d+$//", + "logs", + ), + ).toBe(false); + }); + + it("detects regex filter downgrades inside parentheses and on colon keys", () => { + expect( + isSemanticFilterDowngrade( + "severity:error (message://^Timeout after \\d+ms//)", + 'severity:error message:"*Timeout after*"', + "logs", + ), + ).toBe(true); + expect( + isSemanticFilterDowngrade( + "tags[sentry:user]://^id \\d+//", + "tags[sentry:user]://^id//", + "logs", + ), + ).toBe(true); + }); + + it("detects regex filters rewritten into wildcards on array, quoted, and spaced keys", () => { + expect( + isSemanticFilterDowngrade("arr[*]://^a b//", "arr[*]:*a*", "logs"), + ).toBe(true); + expect( + isSemanticFilterDowngrade('"mykey"://^a b//', '"mykey":*a*', "logs"), + ).toBe(true); + expect( + isSemanticFilterDowngrade( + "tags[x, string]://^a b//", + 'tags[x, string]:"*a*"', + "logs", + ), + ).toBe(true); + }); + + it("detects regex filters whose pattern changed case", () => { + expect( + isSemanticFilterDowngrade( + "message://^Timeout//", + "message://^timeout//", + "logs", + ), + ).toBe(true); + }); + + it("allows rewriting regex filters outside logs", () => { + expect( + isSemanticFilterDowngrade( + "span.description://^GET \\/api\\/\\d+//", + 'span.description:"GET /api/*"', + "spans", + ), + ).toBe(false); + expect( + isSemanticFilterDowngrade( + "message://^Timeout//", + 'message:"*Timeout*"', + "errors", + ), + ).toBe(false); + }); + it("detects message full-text downgrades but allows real field renames", () => { expect( isSemanticFilterDowngrade("conv_id:ZYGC-86ZR", 'message:"*ZYGC-86ZR*"'), diff --git a/packages/mcp-core/src/tools/support/search-events/utils.ts b/packages/mcp-core/src/tools/support/search-events/utils.ts index 74ffd2b68..d312d4241 100644 --- a/packages/mcp-core/src/tools/support/search-events/utils.ts +++ b/packages/mcp-core/src/tools/support/search-events/utils.ts @@ -27,8 +27,22 @@ export type FlexibleEventData = Record; const DEFAULT_MAX_VALUE_LENGTH = 200; const DEFAULT_MAX_ARRAY_ITEMS = 20; -const SENTRY_SEARCH_TOKEN_PATTERN = - /(^|\s)!?([A-Za-z_][A-Za-z0-9_.[\],-]*):(?=\S)(?!\/\/)/g; +const REGEX_FILTER_VALUE_SOURCE = String.raw`\/\/(?!\/\/(?:[\t\n )]|$))[^\n]{1,1024}?\/\/(?=[\t\n )]|$)`; +const REGEX_FILTER_VALUE_PATTERN = new RegExp(`^${REGEX_FILTER_VALUE_SOURCE}`); +const SEARCH_FILTER_KEY_SOURCE = String.raw`(^|\s)!?(?[A-Za-z_][A-Za-z0-9_.[\],-]*):`; +const REGEX_FILTER_KEY_SOURCE = String.raw`(^|[\s()"])!?(?(?:tags|flags)\[[\w.:-]+(?: *, *(?:string|number|boolean|array))?\](?:\[\*\])?|"[\w.:-]+"(?:\[\*\])?|[A-Za-z_][A-Za-z0-9_.[\],-]*(?:\[\*\])?):`; +const REGEX_FILTER_KEY_BEFORE_PATTERN = new RegExp( + `${REGEX_FILTER_KEY_SOURCE}$`, +); +const REGEX_FILTER_KEY_SCAN_LIMIT = 256; +const SENTRY_SEARCH_TOKEN_PATTERN = new RegExp( + String.raw`${SEARCH_FILTER_KEY_SOURCE}(?=\S)(?!\/\/)`, + "g", +); +const SENTRY_SEARCH_TOKEN_WITH_REGEX_PATTERN = new RegExp( + String.raw`${REGEX_FILTER_KEY_SOURCE}(?=${REGEX_FILTER_VALUE_SOURCE})|${SEARCH_FILTER_KEY_SOURCE}(?=\S)(?!\/\/)`, + "g", +); const KNOWN_SENTRY_SEARCH_KEYS = new Set([ "browser", "device", @@ -84,14 +98,31 @@ export function isAggregateQuery(fields: string[]): boolean { return fields.some((field) => field.includes("(") && field.includes(")")); } -export function looksLikeSentrySearchSyntax(query?: string): boolean { +type SearchSyntaxDataset = EventsDataset | "replays"; + +// Sentry only honors key://pattern// as a regex on logs; other datasets match +// the literal text //pattern//. +function searchTokenPattern(dataset?: SearchSyntaxDataset): RegExp { + return dataset === "logs" + ? SENTRY_SEARCH_TOKEN_WITH_REGEX_PATTERN + : SENTRY_SEARCH_TOKEN_PATTERN; +} + +export function looksLikeSentrySearchSyntax( + query?: string, + dataset?: SearchSyntaxDataset, +): boolean { const trimmedQuery = query?.trim(); if (!trimmedQuery) { return false; } - for (const match of trimmedQuery.matchAll(SENTRY_SEARCH_TOKEN_PATTERN)) { - const key = match[2]; + for (const match of trimmedQuery.matchAll(searchTokenPattern(dataset))) { + if (match.groups?.regexKey) { + return true; + } + + const key = match.groups?.key; if (!key) { continue; } @@ -116,19 +147,38 @@ export function looksLikeSentrySearchSyntax(query?: string): boolean { return false; } +export function readRegexFilterValue( + query: string, + index: number, +): string | undefined { + if (!query.startsWith("//", index) || query[index - 1] !== ":") { + return undefined; + } + // Callers probe every index, so only scan a bounded key-sized prefix; the + // NUL stands in for truncated text so `^` can't match mid-query. + const start = Math.max(0, index - REGEX_FILTER_KEY_SCAN_LIMIT); + const prefix = `${start > 0 ? "\0" : ""}${query.slice(start, index)}`; + if (!REGEX_FILTER_KEY_BEFORE_PATTERN.test(prefix)) { + return undefined; + } + return REGEX_FILTER_VALUE_PATTERN.exec(query.slice(index))?.[0]; +} + const FULL_TEXT_SEARCH_KEYS = new Set(["message", "log.body"]); /** - * Replace quoted regions with same-length placeholders so colons inside quotes - * are not treated as filter-key separators (e.g. transaction:"handle message:hello"). - * Length is preserved so offsets still map back to the original query. + * Replace quoted regions and regex values with same-length placeholders so + * colons inside them are not treated as filter-key separators (e.g. + * transaction:"handle message:hello"). Length is preserved so offsets still + * map back to the original query. */ function maskQuotedRegions(query: string): string { let out = ""; let quote: '"' | "'" | null = null; let escaped = false; - for (const char of query) { + for (let i = 0; i < query.length; i += 1) { + const char = query[i]; if (escaped) { escaped = false; out += quote ? "x" : char; @@ -149,6 +199,12 @@ function maskQuotedRegions(query: string): string { } continue; } + const regexValue = readRegexFilterValue(query, i); + if (regexValue) { + out += `//${regexValue.slice(2, -2).replace(/[:\s]/g, "x")}//`; + i += regexValue.length - 1; + continue; + } if (char === '"' || char === "'") { quote = char; out += char; @@ -163,6 +219,7 @@ function maskQuotedRegions(query: string): string { type SearchFilterOccurrence = { key: string; value: string; + regex: boolean; }; function normalizeFilterValue(rawValue: string): string { @@ -215,19 +272,32 @@ function readRawFilterValue( * boundaries. Values are normalized for comparison (strip wrapping quotes and * leading/trailing wildcards). */ -function searchFilterOccurrences(query: string): SearchFilterOccurrence[] { +function searchFilterOccurrences( + query: string, + dataset?: SearchSyntaxDataset, +): SearchFilterOccurrence[] { const occurrences: SearchFilterOccurrence[] = []; const masked = maskQuotedRegions(query); - for (const match of masked.matchAll(SENTRY_SEARCH_TOKEN_PATTERN)) { - const key = match[2]?.toLowerCase(); + for (const match of masked.matchAll(searchTokenPattern(dataset))) { + const key = (match.groups?.regexKey ?? match.groups?.key)?.toLowerCase(); if (!key || match.index === undefined) { continue; } // Read the real value from the original query at the same offset so quotes // and multi-word quoted values are preserved before normalization. - const valueStart = match.index + match[0].indexOf(":") + 1; + const valueStart = match.index + match[0].length; + const regexValue = readRegexFilterValue(query, valueStart); + if (regexValue) { + occurrences.push({ + key, + value: regexValue.slice(2, -2), + regex: true, + }); + continue; + } + const rawValue = readRawFilterValue(query, valueStart); if (!rawValue) { continue; @@ -238,26 +308,32 @@ function searchFilterOccurrences(query: string): SearchFilterOccurrence[] { continue; } - occurrences.push({ key, value }); + occurrences.push({ key, value, regex: false }); } return occurrences; } -function structuredFilterOccurrences(query: string): SearchFilterOccurrence[] { - return searchFilterOccurrences(query).filter( +function structuredFilterOccurrences( + query: string, + dataset?: SearchSyntaxDataset, +): SearchFilterOccurrence[] { + return searchFilterOccurrences(query, dataset).filter( (occurrence) => !FULL_TEXT_SEARCH_KEYS.has(occurrence.key), ); } -function fullTextFilterValues(query: string): string[] { - return searchFilterOccurrences(query) +function fullTextFilterValues( + query: string, + dataset?: SearchSyntaxDataset, +): string[] { + return searchFilterOccurrences(query, dataset) .filter((occurrence) => FULL_TEXT_SEARCH_KEYS.has(occurrence.key)) .map((occurrence) => occurrence.value); } function filterOccurrenceIdentity(occurrence: SearchFilterOccurrence): string { - return `${occurrence.key}\0${occurrence.value}`; + return `${occurrence.key}\0${occurrence.regex}\0${occurrence.value}`; } function escapeRegExp(value: string): string { @@ -316,9 +392,43 @@ function unmatchedStructuredFilters( return unmatched; } +function isRegexFilterDowngrade( + originalQuery: string, + repairedQuery: string, + dataset?: SearchSyntaxDataset, +): boolean { + const originalRegexFilters = searchFilterOccurrences( + originalQuery, + dataset, + ).filter((occurrence) => occurrence.regex); + if (originalRegexFilters.length === 0) { + return false; + } + + const repairedFilters = searchFilterOccurrences(repairedQuery, dataset); + const maskedRepairedQuery = maskQuotedRegions(repairedQuery); + return unmatchedStructuredFilters(originalRegexFilters, repairedFilters).some( + (filter) => + new RegExp( + String.raw`(?:^|[\s()"])!?${escapeRegExp(filter.key)}:(?!\/\/)`, + "i", + ).test(maskedRepairedQuery) || + repairedFilters.some( + (repaired) => + repaired.key === filter.key || + (FULL_TEXT_SEARCH_KEYS.has(repaired.key) && + isRelatedFilterValue( + filter.value.toLowerCase(), + repaired.value.toLowerCase(), + )), + ), + ); +} + /** * True when a structured field:value filter was replaced with message/log.body - * full-text matching (false-success path). Allows real attribute renames. + * full-text matching (false-success path), or a logs `key://pattern//` regex + * filter was replaced with a plain or wildcard filter. Allows real attribute renames. * * Uses multiset key+value matching so dropping one of several identical keys * (e.g. `custom:foo custom:bar` → `custom:bar message:"*foo*"`) is still caught. @@ -328,17 +438,22 @@ function unmatchedStructuredFilters( export function isSemanticFilterDowngrade( originalQuery: string, repairedQuery: string, + dataset?: SearchSyntaxDataset, ): boolean { - if (!looksLikeSentrySearchSyntax(originalQuery)) { + if (!looksLikeSentrySearchSyntax(originalQuery, dataset)) { return false; } - const originalFilters = structuredFilterOccurrences(originalQuery); + if (isRegexFilterDowngrade(originalQuery, repairedQuery, dataset)) { + return true; + } + + const originalFilters = structuredFilterOccurrences(originalQuery, dataset); if (originalFilters.length === 0) { return false; } - const repairedFilters = structuredFilterOccurrences(repairedQuery); + const repairedFilters = structuredFilterOccurrences(repairedQuery, dataset); const droppedFilters = unmatchedStructuredFilters( originalFilters, repairedFilters, @@ -347,7 +462,7 @@ export function isSemanticFilterDowngrade( return false; } - const repairedFullTextValues = fullTextFilterValues(repairedQuery); + const repairedFullTextValues = fullTextFilterValues(repairedQuery, dataset); if (repairedFullTextValues.length === 0) { // Renames keep values on non-full-text attributes and do not need this guard. return false;