diff --git a/apps/sim/lib/mothership/agent-cli/engines/universal-grep.test.ts b/apps/sim/lib/mothership/agent-cli/engines/universal-grep.test.ts index db4aef39b85..d29e49c6878 100644 --- a/apps/sim/lib/mothership/agent-cli/engines/universal-grep.test.ts +++ b/apps/sim/lib/mothership/agent-cli/engines/universal-grep.test.ts @@ -1,3 +1,4 @@ +import { flushMicrotasks } from '@sim/testing/helpers/async' import { sleep } from '@sim/utils/helpers' import { generateId } from '@sim/utils/id' import { beforeEach, describe, expect, it, vi } from 'vitest' @@ -71,6 +72,87 @@ function runtimeWithFilePages(count: number, requested: string[]): AgentCliRunti } describe('universal grep', () => { + it('bounds matching time for adversarial patterns across file lines', async () => { + const runtime = runtimeWith({ + '/api/v2/files': { data: [{ id: 'input' }], nextCursor: null }, + '/api/v2/files/input/text': { + data: { text: Array.from({ length: 12 }, () => `${'a'.repeat(26)}!`).join('\n') }, + }, + }) + const started = performance.now() + const result = await runEngine('grep', ['(a+)+$'], runtime, { scope: 'files', count: true }) + expect(performance.now() - started).toBeLessThan(2_000) + expect(result).toMatchObject({ exitCode: 0, stdout: '0' }) + }, 15_000) + + it('matches unsupported regex syntax literally without falling back to backtracking', async () => { + const result = await runEngine( + 'grep', + ['(?=NEEDLE)'], + runtimeWith({ + '/api/v2/files': { data: [{ id: 'input' }], nextCursor: null }, + '/api/v2/files/input/text': { data: { text: '(?=needle)\nneedle' } }, + }), + { scope: 'files', count: true, i: true } + ) + expect(result).toMatchObject({ exitCode: 0, stdout: '1 (files=1)' }) + }) + + it('bounds overlapping context windows while counting matches beyond the output limit', async () => { + const result = await runEngine( + 'grep', + ['needle'], + runtimeWith({ + '/api/v2/files': { data: [{ id: 'input' }], nextCursor: null }, + '/api/v2/files/input/text': { + data: { text: ['header', ...Array.from({ length: 30_000 }, () => 'needle')].join('\n') }, + }, + }), + { scope: 'files', C: '30000', limit: '3' } + ) + expect(result).toMatchObject({ + exitCode: 0, + stdout: + 'files/input:1: header\nfiles/input:2: needle\nfiles/input:3: needle\n' + + '[2 of 30000 matching lines shown — narrow with --scope, --in, or a tighter pattern]', + }) + }, 15_000) + + it('fails an exhausted scan budget instead of returning a complete count', async () => { + vi.spyOn(performance, 'now').mockReturnValueOnce(0).mockReturnValue(6_000) + const result = await runEngine('grep', ['id'], runtimeWith(CATALOG), { + scope: 'blocks', + count: true, + }) + expect(result.exitCode).toBe(1) + expect(result.stdout).toBe('') + expect(result.stderr).toMatch(/incomplete.*budget/i) + }) + + it('yields during scanning so a pending cancellation can stop the search', async () => { + vi.useFakeTimers() + try { + vi.spyOn(performance, 'now').mockReturnValueOnce(0).mockReturnValue(20) + const controller = new AbortController() + setTimeout(() => controller.abort(new Error('Search stopped')), 0) + const pending = runEngine( + 'grep', + ['id'], + { ...runtimeWith(CATALOG), signal: controller.signal }, + { scope: 'blocks' } + ) + await flushMicrotasks() + await vi.runAllTimersAsync() + expect(await pending).toMatchObject({ + exitCode: 1, + stdout: '', + stderr: expect.stringContaining('Search stopped'), + }) + } finally { + vi.useRealTimers() + } + }) + it('finds field ids inside block definitions and names the path-shaped line', async () => { const result = await runEngine('grep', ['stream'], runtimeWith(CATALOG), { scope: 'blocks', diff --git a/apps/sim/lib/mothership/agent-cli/engines/universal-grep.ts b/apps/sim/lib/mothership/agent-cli/engines/universal-grep.ts index 2dd33e90258..a0038cca685 100644 --- a/apps/sim/lib/mothership/agent-cli/engines/universal-grep.ts +++ b/apps/sim/lib/mothership/agent-cli/engines/universal-grep.ts @@ -1,7 +1,9 @@ +import { sleep } from '@sim/utils/helpers' import { isRecordLike } from '@sim/utils/object' import type { ReadFileTextResponse } from 'sim/embed' import { listCatalogTools } from '@/lib/catalog/application/list-tools' import { readBlockCatalog } from '@/lib/catalog/application/read-block-catalog' +import { compileLinearRegex, isPlainText, literalRegex } from '@/lib/core/security/linear-regex' import { enginePrincipal } from '@/lib/mothership/agent-cli/engine-principal' import { type AgentCliEngine, @@ -46,6 +48,9 @@ const CATALOG_ALL = 100_000 const MAX_FILES = 300 const FILE_READ_CONCURRENCY = 5 const MAX_BYTES_PER_FILE = 262_144 +/** Bound scanning after materialization without charging backend read latency. */ +const MAX_SCAN_TIME_MS = 5_000 +const SCAN_YIELD_INTERVAL_MS = 10 /** `workflow:` — a prefixed form no world or resource ever prints as its path. */ const PREFIX_SELECTOR = /^\w+:/ const PLATFORM_SCOPES: ReadonlySet = new Set(['blocks', 'tools']) @@ -416,13 +421,12 @@ async function materializeWithin( } function compilePattern(raw: string, ignoreCase: boolean): (line: string) => boolean { - try { - const regex = new RegExp(raw, ignoreCase ? 'i' : '') - return (line) => regex.test(line) - } catch { - const needle = ignoreCase ? raw.toLowerCase() : raw - return (line) => (ignoreCase ? line.toLowerCase() : line).includes(needle) - } + const regex = isPlainText(raw) + ? literalRegex(raw, { ignoreCase }) + : compileLinearRegex(raw, { ignoreCase }) + if (regex) return (line) => regex.test(line) + const needle = ignoreCase ? raw.toLowerCase() : raw + return (line) => (ignoreCase ? line.toLowerCase() : line).includes(needle) } function clip(line: string): string { @@ -570,31 +574,51 @@ export const universalGrepCommand: AgentCliEngine = { let total = 0 let shownMatches = 0 const perScope = new Map() + const scanStartedAt = performance.now() + let lastYieldAt = scanStartedAt + const checkScan = () => { + runtime.signal?.throwIfAborted() + const now = performance.now() + if (now - scanStartedAt >= MAX_SCAN_TIME_MS) { + throw new Error( + 'Search incomplete: scanning exceeded the time budget. Narrow with --scope or --in.' + ) + } + return now + } for (const resource of candidates) { if (resource.text === null) continue const lines = resource.text.split('\n') - const selected = new Set() + const selected = new Map() + const remainingLines = limit - out.length + let nextContextLine = 0 for (let i = 0; i < lines.length; i++) { + if (checkScan() - lastYieldAt >= SCAN_YIELD_INTERVAL_MS) { + await sleep(0) + lastYieldAt = checkScan() + } if (!matches(lines[i])) continue total++ perScope.set(resource.scope, (perScope.get(resource.scope) ?? 0) + 1) if (countOnly) continue for ( - let j = Math.max(0, i - context.before); - j <= Math.min(lines.length - 1, i + context.after); + let j = Math.max(nextContextLine, i - context.before); + j <= Math.min(lines.length - 1, i + context.after) && selected.size < remainingLines; j++ ) { - selected.add(j) + selected.set(j, false) + nextContextLine = j + 1 } + if (selected.has(i)) selected.set(i, true) } if (countOnly || selected.size === 0 || out.length >= limit) continue const header = `${resource.scope}/${resource.label}${resource.label === resource.id ? '' : ` (${resource.id})`}` - for (const i of [...selected].sort((a, b) => a - b)) { - if (out.length >= limit) break + for (const [i, isMatch] of selected) { out.push(`${header}:${i + 1}: ${clip(lines[i])}`) - if (matches(lines[i])) shownMatches++ + if (isMatch) shownMatches++ } } + checkScan() if (countOnly) { const breakdown = [...perScope.entries()].map(([s, n]) => `${s}=${n}`).join(' ')