Skip to content

Commit dadc698

Browse files
committed
fix(search): bound document conversion and preserve complete reads
1 parent 1c9d7a3 commit dadc698

6 files changed

Lines changed: 315 additions & 37 deletions

File tree

‎apps/sim/lib/file-parsers/docx-parser.ts‎

Lines changed: 106 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,6 @@
11
import { readFile } from 'fs/promises'
22
import { createLogger } from '@sim/logger'
3+
import { isRecordLike, toRecord } from '@sim/utils/object'
34
import mammoth from 'mammoth'
45
import {
56
FileParserError,
@@ -19,6 +20,79 @@ import { assertOoxmlArchiveWithinLimits } from '@/lib/file-parsers/zip-guard'
1920

2021
const logger = createLogger('DocxParser')
2122

23+
/** Bounds repeated notes and generated markup independently of the normalized text budget. */
24+
const MAX_DOCX_CONVERSION_NODES = 50_000
25+
const MAX_DOCX_CONVERSION_BYTES = 2 * 1024 * 1024
26+
27+
/**
28+
* Mammoth's supported transform hook runs before HTML generation. Count every reference
29+
* expansion, including repeated notes, without retaining the expanded graph. The node ceiling
30+
* also stops cyclic references. The separate 2 MiB ceiling charges raw UTF-8 model strings,
31+
* including attributes, before HTML escaping. Fixed default styles and omitted image data
32+
* bound conversion amplification; embedded style maps could add arbitrary wrapper markup.
33+
*/
34+
function assertDocxConversionWithinLimits(document: unknown, signal?: AbortSignal): void {
35+
const notes = toRecord(toRecord(document).notes)
36+
const pending: unknown[] = [document]
37+
let nodes = 0
38+
let bytes = 0
39+
const complexity = () =>
40+
new FileParserError('complexity_limit', 'DOCX conversion exceeds its graph budget')
41+
const append = (children: unknown) => {
42+
if (!Array.isArray(children))
43+
throw new FileParserError('invalid_format', 'DOCX conversion has invalid children')
44+
if (nodes + pending.length + children.length > MAX_DOCX_CONVERSION_NODES) throw complexity()
45+
for (let index = children.length - 1; index >= 0; index--) pending.push(children[index])
46+
}
47+
while (pending.length) {
48+
signal?.throwIfAborted()
49+
const node = pending.pop()
50+
if (!isRecordLike(node) || typeof node.type !== 'string')
51+
throw new FileParserError('invalid_format', 'DOCX conversion has an invalid node')
52+
if (++nodes > MAX_DOCX_CONVERSION_NODES) throw complexity()
53+
for (const value of Object.values(node)) {
54+
if (typeof value === 'string') bytes += Buffer.byteLength(value, 'utf8')
55+
}
56+
if (bytes > MAX_DOCX_CONVERSION_BYTES) throw complexity()
57+
switch (node.type) {
58+
case 'document':
59+
case 'paragraph':
60+
case 'run':
61+
case 'hyperlink':
62+
case 'table':
63+
case 'tableRow':
64+
case 'tableCell':
65+
append(node.children)
66+
break
67+
case 'note':
68+
case 'comment':
69+
append(node.body)
70+
break
71+
case 'noteReference': {
72+
if (typeof notes.resolve !== 'function')
73+
throw new FileParserError('invalid_format', 'DOCX note resolver is unavailable')
74+
const note: unknown = Reflect.apply(notes.resolve, notes, [node])
75+
if (!note) throw new FileParserError('invalid_format', 'DOCX references a missing note')
76+
append([note])
77+
break
78+
}
79+
case 'text':
80+
if (typeof node.value !== 'string')
81+
throw new FileParserError('invalid_format', 'DOCX text is malformed')
82+
break
83+
case 'image':
84+
case 'tab':
85+
case 'checkbox':
86+
case 'break':
87+
case 'bookmarkStart':
88+
case 'commentReference':
89+
break
90+
default:
91+
throw new FileParserError('invalid_format', 'DOCX conversion has an unsupported node')
92+
}
93+
}
94+
}
95+
2296
/**
2397
* Extracts DOCX text by rendering the document to HTML with mammoth and walking
2498
* that HTML with the shared structured-text walker. mammoth's HTML keeps the
@@ -45,25 +119,49 @@ export class DocxParser implements FileParser {
45119
}
46120

47121
assertOoxmlArchiveWithinLimits(buffer)
122+
const maxTextBytes =
123+
options.docxTextMode === 'complete'
124+
? (options.maxTextBytes ?? MAX_DOCX_CONVERSION_BYTES)
125+
: undefined
126+
if (maxTextBytes !== undefined && (!Number.isSafeInteger(maxTextBytes) || maxTextBytes <= 0))
127+
throw new FileParserError('complexity_limit', 'Invalid DOCX text byte budget')
48128

49129
const extractionErrors: unknown[] = []
50130
let parserReturnedEmpty = false
51131

52132
try {
53-
const htmlResult = await mammoth.convertToHtml({ buffer })
133+
const htmlResult = await mammoth.convertToHtml(
134+
{ buffer },
135+
maxTextBytes === undefined
136+
? undefined
137+
: {
138+
includeEmbeddedStyleMap: false,
139+
convertImage: mammoth.images.imgElement(async () => ({ src: '' })),
140+
transformDocument: (document: unknown) => {
141+
assertDocxConversionWithinLimits(document, options.signal)
142+
return document
143+
},
144+
}
145+
)
54146
options.signal?.throwIfAborted()
55147

56-
const structured = this.structuredTextFromHtml(htmlResult.value)
148+
const structured = this.structuredTextFromHtml(htmlResult.value, maxTextBytes !== undefined)
57149
if (structured) {
150+
const content = sanitizeTextForUTF8(structured)
151+
if (maxTextBytes !== undefined && Buffer.byteLength(content, 'utf8') > maxTextBytes)
152+
throw new FileParserError('complexity_limit', 'DOCX text exceeds its byte budget')
58153
return {
59-
content: sanitizeTextForUTF8(structured),
154+
content,
60155
metadata: {
61156
extractionMethod: 'mammoth-html',
62157
messages: htmlResult.messages,
63158
},
64159
}
65160
}
66161

162+
if (maxTextBytes !== undefined)
163+
throw new FileParserError('no_extractable_text', 'No complete DOCX text was extracted')
164+
67165
const rawResult = await mammoth.extractRawText({ buffer })
68166
options.signal?.throwIfAborted()
69167

@@ -79,6 +177,7 @@ export class DocxParser implements FileParser {
79177
parserReturnedEmpty = true
80178
} catch (mammothError) {
81179
options.signal?.throwIfAborted()
180+
if (maxTextBytes !== undefined) throw mammothError
82181
logger.warn('mammoth failed, trying officeparser:', mammothError)
83182
extractionErrors.push(mammothError)
84183
}
@@ -161,14 +260,16 @@ export class DocxParser implements FileParser {
161260
/**
162261
* Walks mammoth's HTML rendering under the HTML parser's size caps. A rendering
163262
* too large to walk safely falls back to the raw-text path by returning empty,
164-
* since mammoth has already materialised the document once at that point.
263+
* since mammoth has already materialised the document once at that point. Budgeted reads
264+
* reject instead: the fallback cannot preserve the conversion and completeness bounds.
165265
*/
166-
private structuredTextFromHtml(html: string): string {
266+
private structuredTextFromHtml(html: string, bounded = false): string {
167267
if (!html || html.trim().length === 0) return ''
168268
try {
169269
assertHtmlStringWithinLimits(html)
170270
} catch (error) {
171271
if (isHtmlComplexityError(error)) {
272+
if (bounded) throw error
172273
logger.warn('mammoth HTML exceeds walker limits, using raw text:', error.message)
173274
return ''
174275
}

‎apps/sim/lib/file-parsers/pdf-parser.ts‎

Lines changed: 29 additions & 24 deletions
Original file line numberDiff line numberDiff line change
@@ -36,17 +36,18 @@ export const MAX_PDF_TEXT_CHARS = 10_000_000
3636
/** Complete extraction shares the ingestion pipeline's bounded text-output envelope. */
3737
export const MAX_COMPLETE_PDF_TEXT_BYTES = 20 * 1024 * 1024
3838

39+
/** Independent retained-text ceiling, including conservative layout overhead. */
40+
const MAX_COMPLETE_PDF_RETAINED_TEXT_BYTES = MAX_COMPLETE_PDF_TEXT_BYTES
41+
3942
/** Bounds expansion on one page independently of a long document's output budget. */
4043
export const MAX_COMPLETE_PDF_PAGE_CHARS = 250_000
4144

4245
/** Wall-clock ceiling for extracting text from a whole document. */
4346
const PDF_EXTRACTION_TIMEOUT_MS = 60_000
4447

4548
/**
46-
* Upper bound on what line reconstruction adds per line after the budget is
47-
* spent: a two-character paragraph break plus a three-character heading marker.
48-
* The complete-mode byte ceiling therefore trips slightly earlier than it did
49-
* when pages were flattened to one line, by at most this many bytes per line.
49+
* Conservative retained-text allowance for a paragraph break and heading marker.
50+
* This protects parser state independently of the caller's rendered-output budget.
5051
*/
5152
const MAX_LINE_DECORATION_BYTES = 5
5253

@@ -249,8 +250,8 @@ function readPageHeight(page: PDFPageProxy): number | undefined {
249250
}
250251
}
251252

252-
/** Bytes a page's lines can occupy in the output once joined and decorated. */
253-
function estimatePageBytes(lines: readonly PdfLine[]): number {
253+
/** Conservative text storage for retained lines and their possible decorations. */
254+
function estimateRetainedPageBytes(lines: readonly PdfLine[]): number {
254255
let bytes = 0
255256
for (const line of lines) {
256257
bytes += Buffer.byteLength(line.text, 'utf8') + MAX_LINE_DECORATION_BYTES
@@ -267,7 +268,8 @@ function estimatePageBytes(lines: readonly PdfLine[]): number {
267268
async function assemblePages(
268269
pages: readonly PdfPageLines[],
269270
complete: boolean,
270-
signal: AbortSignal | undefined
271+
signal: AbortSignal | undefined,
272+
maxTextBytes: number
271273
): Promise<string> {
272274
signal?.throwIfAborted()
273275
const filteredPages = suppressFurniture(pages)
@@ -280,14 +282,25 @@ async function assemblePages(
280282
headingMarkers: PDF_HEADING_MARKERS_ENABLED && headingMarkersViable(allLines, bodyHeight),
281283
}
282284
const pageTexts: string[] = []
285+
let outputBytes = 0
283286
for (const [index, lines] of filteredPages.entries()) {
284287
if (index > 0 && index % ASSEMBLY_YIELD_EVERY_PAGES === 0) {
285288
await sleep(0)
286289
signal?.throwIfAborted()
287290
}
288291
const joined = joinLines(lines, options)
289292
const text = complete ? normalizePdfWhitespace(sanitizeTextForUTF8(joined)).trim() : joined
290-
if (text.length > 0) pageTexts.push(text)
293+
if (text.length > 0) {
294+
if (complete) {
295+
outputBytes +=
296+
Buffer.byteLength(text, 'utf8') + (pageTexts.length > 0 ? PAGE_SEPARATOR.length : 0)
297+
if (outputBytes > maxTextBytes)
298+
throw completeExtractionLimit(
299+
`PDF text exceeds the safe ${maxTextBytes.toLocaleString()}-byte output limit.`
300+
)
301+
}
302+
pageTexts.push(text)
303+
}
291304
}
292305
const text = pageTexts.join(PAGE_SEPARATOR)
293306
return complete ? text : normalizePdfWhitespace(text).trim()
@@ -320,7 +333,7 @@ async function extractTextWithinBudget(
320333
const pages: PdfPageLines[] = []
321334

322335
let remainingChars = MAX_PDF_TEXT_CHARS
323-
let outputBytes = 0
336+
let retainedTextBytes = 0
324337
let pagesRead = 0
325338
let truncated = totalPages > pageLimit
326339

@@ -352,9 +365,7 @@ async function extractTextWithinBudget(
352365
const page = pageResult
353366
const pageHeight = readPageHeight(page)
354367
let extraction: PageExtraction
355-
const pageCharLimit = complete
356-
? Math.min(MAX_COMPLETE_PDF_PAGE_CHARS, maxTextBytes - outputBytes)
357-
: remainingChars
368+
const pageCharLimit = complete ? MAX_COMPLETE_PDF_PAGE_CHARS : remainingChars
358369
try {
359370
extraction = await readPageWithinBudget(page, pageCharLimit, deadline, signal)
360371
} finally {
@@ -370,10 +381,11 @@ async function extractTextWithinBudget(
370381
pagesRead++
371382
if (complete) {
372383
if (lines.length > 0) {
373-
outputBytes += estimatePageBytes(lines) + (pages.length > 0 ? PAGE_SEPARATOR.length : 0)
374-
if (outputBytes > maxTextBytes) {
384+
retainedTextBytes +=
385+
estimateRetainedPageBytes(lines) + (pages.length > 0 ? PAGE_SEPARATOR.length : 0)
386+
if (retainedTextBytes > MAX_COMPLETE_PDF_RETAINED_TEXT_BYTES) {
375387
throw completeExtractionLimit(
376-
`PDF text exceeds the safe ${maxTextBytes.toLocaleString()}-byte output limit.`
388+
`PDF text exceeds the safe ${MAX_COMPLETE_PDF_RETAINED_TEXT_BYTES.toLocaleString()}-byte output limit.`
377389
)
378390
}
379391
pages.push({ lines, pageHeight })
@@ -388,22 +400,15 @@ async function extractTextWithinBudget(
388400
throw completeExtractionLimit(
389401
extraction.deadlineReached
390402
? 'PDF text extraction exceeded its time limit.'
391-
: pageCharLimit < MAX_COMPLETE_PDF_PAGE_CHARS
392-
? `PDF text exceeds the safe ${maxTextBytes.toLocaleString()}-byte output limit.`
393-
: `PDF page ${pageNumber} exceeds the safe expansion limit of ${MAX_COMPLETE_PDF_PAGE_CHARS.toLocaleString()} characters per page.`
403+
: `PDF page ${pageNumber} exceeds the safe expansion limit of ${MAX_COMPLETE_PDF_PAGE_CHARS.toLocaleString()} characters per page.`
394404
)
395405
}
396406
truncated = true
397407
break
398408
}
399409
}
400410

401-
let text = await assemblePages(pages, complete, signal)
402-
if (complete && Buffer.byteLength(text, 'utf8') > maxTextBytes) {
403-
throw completeExtractionLimit(
404-
`PDF text exceeds the safe ${maxTextBytes.toLocaleString()}-byte output limit.`
405-
)
406-
}
411+
let text = await assemblePages(pages, complete, signal, maxTextBytes)
407412

408413
/** Paragraph breaks land after the budget is spent; trimming that overflow is a truncation too. */
409414
if (!complete && text.length > MAX_PDF_TEXT_CHARS) {

‎apps/sim/lib/file-parsers/types.ts‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -39,11 +39,15 @@ export interface FileParseOptions {
3939
* Complete text byte budget for CSV/XLSX (default 25 MiB) and PDF (default 20 MiB).
4040
* PDF applies this when pdfTextMode is 'complete' and never exceeds its safe default.
4141
* Exceeding it throws complexity_limit instead of returning a truncated prefix.
42+
* DOCX applies this only when docxTextMode is 'complete' (default 2 MiB); separate
43+
* conversion graph limits can reject structurally complex inputs before normalization.
4244
* Preview mode retains its own limits; other formats use their parser-specific budgets.
4345
*/
4446
maxTextBytes?: number
4547
/** Preserve textual markup in a canonical .txt artifact instead of interpreting it as HTML or RTF. */
4648
textMode?: 'literal'
49+
/** Opt into bounded DOCX conversion without lossy or unchecked parser fallbacks. */
50+
docxTextMode?: 'complete'
4751
/** Complete PDF extraction rejects safety limits instead of returning preview text. */
4852
pdfTextMode?: 'preview' | 'complete'
4953
/** Lower page ceiling for complete PDF extraction; defaults to the parser's safe limit. */

‎apps/sim/lib/sim-search/live/README.md‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -32,7 +32,7 @@ For an explicit source user list, source-side verification tries those users' de
3232

3333
The source credential must be able to read the file. Selected folders, accessible subfolders, shared-drive IDs, and file types are checked using source-side metadata. Folder queries narrow candidate retrieval; metadata verification remains authoritative. A member's personal file outside the configured source scope is excluded even if their OAuth token can read it. Folder expansion and ancestry traversal are bounded and may report partial coverage.
3434

35-
Text-bearing PDF and DOCX files are downloaded with the member credential after checking `capabilities.canDownload`, then parsed with the shared document parsers. The download is capped at 4 MiB of decoded bytes; PDF extraction is complete or unavailable, with a 100-page and 1 MiB text limit. DOCX archives are checked before parsing (10 MiB total expanded, 4 MiB per entry), with a 1 MiB extracted-text limit. MIME/signature mismatches, password protection, malformed files, and documents with no extractable text fail explicitly; this path does not perform OCR. Existing source links and bounded comments/replies remain in successful reads. Other unsupported binary formats return labeled metadata only.
35+
Text-bearing PDF and DOCX files are downloaded with the member credential after checking `capabilities.canDownload`, then parsed with the shared document parsers. The download is capped at 4 MiB of decoded bytes; PDF extraction is complete or unavailable, with a 100-page and 1 MiB text limit. DOCX archives are checked before parsing (10 MiB total expanded, 4 MiB per entry), with a 1 MiB extracted-text limit. Complete DOCX reads additionally bound the expanded conversion graph to 50,000 nodes and 2 MiB of model strings before HTML rendering, use built-in styles without embedding image bytes, and reject conversion failures instead of retrying a lossy fallback. MIME/signature mismatches, password protection, malformed files, and documents with no extractable text fail explicitly; this path does not perform OCR. Existing source links and bounded comments/replies remain in successful reads. Other unsupported binary formats return labeled metadata only.
3636

3737
### Gmail
3838

‎apps/sim/lib/sim-search/live/drive-content.ts‎

Lines changed: 21 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,6 @@
11
import { toStringOrNull } from '@sim/utils/coerce'
22
import { toRecord } from '@sim/utils/object'
3+
import { isPayloadSizeLimitError } from '@/lib/core/utils/stream-limits'
34
import { FileParserError, getFileParserErrorCode } from '@/lib/file-parsers/errors'
45
import { sniffFileKind } from '@/lib/file-parsers/sniff'
56
import type { FileParseResult } from '@/lib/file-parsers/types'
@@ -62,10 +63,21 @@ export async function readDriveFileContent(
6263
)
6364
}
6465

65-
const bytes = await client.bytes(`/drive/v3/files/${segment(id)}`, {
66-
alt: 'media',
67-
supportsAllDrives: 'true',
68-
})
66+
let bytes: Buffer
67+
try {
68+
bytes = await client.bytes(`/drive/v3/files/${segment(id)}`, {
69+
alt: 'media',
70+
supportsAllDrives: 'true',
71+
})
72+
} catch (error) {
73+
signal?.throwIfAborted()
74+
if (isPayloadSizeLimitError(error))
75+
throw new NativeSearchError(
76+
'unavailable',
77+
'This file exceeds the 4 MiB live-read limit. Split it into smaller files or open the source.'
78+
)
79+
throw error
80+
}
6981
signal?.throwIfAborted()
7082
try {
7183
if (bytes.byteLength === 0) throw new FileParserError('empty_input', 'Empty file')
@@ -89,7 +101,11 @@ export async function readDriveFileContent(
89101
} else {
90102
assertOoxmlArchiveWithinLimits(bytes, DRIVE_DOCX_LIMITS)
91103
const { DocxParser } = await import('@/lib/file-parsers/docx-parser')
92-
parsed = await new DocxParser().parseBuffer(bytes, { signal, contentMode: 'complete' })
104+
parsed = await new DocxParser().parseBuffer(bytes, {
105+
signal,
106+
docxTextMode: 'complete',
107+
maxTextBytes: MAX_DRIVE_TEXT_BYTES,
108+
})
93109
}
94110
signal?.throwIfAborted()
95111
if (parsed.metadata?.degraded || parsed.metadata?.truncated)

0 commit comments

Comments
 (0)