Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 18 additions & 0 deletions .github/workflows/test-build.yml
Original file line number Diff line number Diff line change
Expand Up @@ -129,6 +129,24 @@ jobs:
if-no-files-found: warn
retention-days: 14

- name: Verify Google document reads over real HTTP
if: matrix.provision == 'push'
working-directory: apps/sim
env:
NEXT_PUBLIC_APP_URL: http://127.0.0.1:3040
NEXT_PUBLIC_FORCE_HOSTED: 'false'
SEARCH_GOOGLE_CONTENT_REPORT_PATH: ${{ runner.temp }}/search-google-content.json
run: bun scripts/test-search-google-content-e2e.ts

- name: Upload Google content acceptance report
if: failure() && matrix.provision == 'push'
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: search-google-content
path: ${{ runner.temp }}/search-google-content.json
if-no-files-found: ignore
retention-days: 7

- name: Verify SCIM and administration over real HTTP
working-directory: apps/sim
env:
Expand Down
111 changes: 106 additions & 5 deletions apps/sim/lib/file-parsers/docx-parser.ts
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import { readFile } from 'fs/promises'
import { createLogger } from '@sim/logger'
import { isRecordLike, toRecord } from '@sim/utils/object'
import mammoth from 'mammoth'
import {
FileParserError,
Expand All @@ -19,6 +20,79 @@ import { assertOoxmlArchiveWithinLimits } from '@/lib/file-parsers/zip-guard'

const logger = createLogger('DocxParser')

/** Bounds repeated notes and generated markup independently of the normalized text budget. */
const MAX_DOCX_CONVERSION_NODES = 50_000
const MAX_DOCX_CONVERSION_BYTES = 2 * 1024 * 1024

/**
* Mammoth's supported transform hook runs before HTML generation. Count every reference
* expansion, including repeated notes, without retaining the expanded graph. The node ceiling
* also stops cyclic references. The separate 2 MiB ceiling charges raw UTF-8 model strings,
* including attributes, before HTML escaping. Fixed default styles and omitted image data
* bound conversion amplification; embedded style maps could add arbitrary wrapper markup.
*/
function assertDocxConversionWithinLimits(document: unknown, signal?: AbortSignal): void {
const notes = toRecord(toRecord(document).notes)
const pending: unknown[] = [document]
let nodes = 0
let bytes = 0
const complexity = () =>
new FileParserError('complexity_limit', 'DOCX conversion exceeds its graph budget')
const append = (children: unknown) => {
if (!Array.isArray(children))
throw new FileParserError('invalid_format', 'DOCX conversion has invalid children')
if (nodes + pending.length + children.length > MAX_DOCX_CONVERSION_NODES) throw complexity()
for (let index = children.length - 1; index >= 0; index--) pending.push(children[index])
}
while (pending.length) {
signal?.throwIfAborted()
const node = pending.pop()
if (!isRecordLike(node) || typeof node.type !== 'string')
throw new FileParserError('invalid_format', 'DOCX conversion has an invalid node')
if (++nodes > MAX_DOCX_CONVERSION_NODES) throw complexity()
for (const value of Object.values(node)) {
if (typeof value === 'string') bytes += Buffer.byteLength(value, 'utf8')
}
if (bytes > MAX_DOCX_CONVERSION_BYTES) throw complexity()
switch (node.type) {
case 'document':
case 'paragraph':
case 'run':
case 'hyperlink':
case 'table':
case 'tableRow':
case 'tableCell':
append(node.children)
break
case 'note':
case 'comment':
append(node.body)
break
case 'noteReference': {
if (typeof notes.resolve !== 'function')
throw new FileParserError('invalid_format', 'DOCX note resolver is unavailable')
const note: unknown = Reflect.apply(notes.resolve, notes, [node])
if (!note) throw new FileParserError('invalid_format', 'DOCX references a missing note')
append([note])
break
}
case 'text':
if (typeof node.value !== 'string')
throw new FileParserError('invalid_format', 'DOCX text is malformed')
break
case 'image':
case 'tab':
case 'checkbox':
case 'break':
case 'bookmarkStart':
case 'commentReference':
break
default:
throw new FileParserError('invalid_format', 'DOCX conversion has an unsupported node')
}
}
}

/**
* Extracts DOCX text by rendering the document to HTML with mammoth and walking
* that HTML with the shared structured-text walker. mammoth's HTML keeps the
Expand All @@ -45,25 +119,49 @@ export class DocxParser implements FileParser {
}

assertOoxmlArchiveWithinLimits(buffer)
const maxTextBytes =
options.docxTextMode === 'complete'
? (options.maxTextBytes ?? MAX_DOCX_CONVERSION_BYTES)
: undefined
if (maxTextBytes !== undefined && (!Number.isSafeInteger(maxTextBytes) || maxTextBytes <= 0))
throw new FileParserError('complexity_limit', 'Invalid DOCX text byte budget')

const extractionErrors: unknown[] = []
let parserReturnedEmpty = false

try {
const htmlResult = await mammoth.convertToHtml({ buffer })
const htmlResult = await mammoth.convertToHtml(
{ buffer },
maxTextBytes === undefined
? undefined
: {
includeEmbeddedStyleMap: false,
convertImage: mammoth.images.imgElement(async () => ({ src: '' })),
transformDocument: (document: unknown) => {
assertDocxConversionWithinLimits(document, options.signal)
return document
},
}
)
options.signal?.throwIfAborted()

const structured = this.structuredTextFromHtml(htmlResult.value)
const structured = this.structuredTextFromHtml(htmlResult.value, maxTextBytes !== undefined)
if (structured) {
const content = sanitizeTextForUTF8(structured)
if (maxTextBytes !== undefined && Buffer.byteLength(content, 'utf8') > maxTextBytes)
throw new FileParserError('complexity_limit', 'DOCX text exceeds its byte budget')
return {
content: sanitizeTextForUTF8(structured),
content,
metadata: {
extractionMethod: 'mammoth-html',
messages: htmlResult.messages,
},
}
}

if (maxTextBytes !== undefined)
throw new FileParserError('no_extractable_text', 'No complete DOCX text was extracted')

const rawResult = await mammoth.extractRawText({ buffer })
options.signal?.throwIfAborted()

Expand All @@ -79,6 +177,7 @@ export class DocxParser implements FileParser {
parserReturnedEmpty = true
} catch (mammothError) {
options.signal?.throwIfAborted()
if (maxTextBytes !== undefined) throw mammothError
logger.warn('mammoth failed, trying officeparser:', mammothError)
extractionErrors.push(mammothError)
}
Expand Down Expand Up @@ -161,14 +260,16 @@ export class DocxParser implements FileParser {
/**
* Walks mammoth's HTML rendering under the HTML parser's size caps. A rendering
* too large to walk safely falls back to the raw-text path by returning empty,
* since mammoth has already materialised the document once at that point.
* since mammoth has already materialised the document once at that point. Budgeted reads
* reject instead: the fallback cannot preserve the conversion and completeness bounds.
*/
private structuredTextFromHtml(html: string): string {
private structuredTextFromHtml(html: string, bounded = false): string {
if (!html || html.trim().length === 0) return ''
try {
assertHtmlStringWithinLimits(html)
} catch (error) {
if (isHtmlComplexityError(error)) {
if (bounded) throw error
logger.warn('mammoth HTML exceeds walker limits, using raw text:', error.message)
return ''
}
Expand Down
63 changes: 42 additions & 21 deletions apps/sim/lib/file-parsers/pdf-parser.ts
Original file line number Diff line number Diff line change
Expand Up @@ -36,17 +36,18 @@ export const MAX_PDF_TEXT_CHARS = 10_000_000
/** Complete extraction shares the ingestion pipeline's bounded text-output envelope. */
export const MAX_COMPLETE_PDF_TEXT_BYTES = 20 * 1024 * 1024

/** Independent retained-text ceiling, including conservative layout overhead. */
const MAX_COMPLETE_PDF_RETAINED_TEXT_BYTES = MAX_COMPLETE_PDF_TEXT_BYTES

/** Bounds expansion on one page independently of a long document's output budget. */
export const MAX_COMPLETE_PDF_PAGE_CHARS = 250_000

/** Wall-clock ceiling for extracting text from a whole document. */
const PDF_EXTRACTION_TIMEOUT_MS = 60_000

/**
* Upper bound on what line reconstruction adds per line after the budget is
* spent: a two-character paragraph break plus a three-character heading marker.
* The complete-mode byte ceiling therefore trips slightly earlier than it did
* when pages were flattened to one line, by at most this many bytes per line.
* Conservative retained-text allowance for a paragraph break and heading marker.
* This protects parser state independently of the caller's rendered-output budget.
*/
const MAX_LINE_DECORATION_BYTES = 5

Expand Down Expand Up @@ -249,8 +250,8 @@ function readPageHeight(page: PDFPageProxy): number | undefined {
}
}

/** Bytes a page's lines can occupy in the output once joined and decorated. */
function estimatePageBytes(lines: readonly PdfLine[]): number {
/** Conservative text storage for retained lines and their possible decorations. */
function estimateRetainedPageBytes(lines: readonly PdfLine[]): number {
let bytes = 0
for (const line of lines) {
bytes += Buffer.byteLength(line.text, 'utf8') + MAX_LINE_DECORATION_BYTES
Expand All @@ -267,7 +268,8 @@ function estimatePageBytes(lines: readonly PdfLine[]): number {
async function assemblePages(
pages: readonly PdfPageLines[],
complete: boolean,
signal: AbortSignal | undefined
signal: AbortSignal | undefined,
maxTextBytes: number
): Promise<string> {
signal?.throwIfAborted()
const filteredPages = suppressFurniture(pages)
Expand All @@ -280,14 +282,25 @@ async function assemblePages(
headingMarkers: PDF_HEADING_MARKERS_ENABLED && headingMarkersViable(allLines, bodyHeight),
}
const pageTexts: string[] = []
let outputBytes = 0
for (const [index, lines] of filteredPages.entries()) {
if (index > 0 && index % ASSEMBLY_YIELD_EVERY_PAGES === 0) {
await sleep(0)
signal?.throwIfAborted()
}
const joined = joinLines(lines, options)
const text = complete ? normalizePdfWhitespace(sanitizeTextForUTF8(joined)).trim() : joined
if (text.length > 0) pageTexts.push(text)
if (text.length > 0) {
if (complete) {
outputBytes +=
Buffer.byteLength(text, 'utf8') + (pageTexts.length > 0 ? PAGE_SEPARATOR.length : 0)
if (outputBytes > maxTextBytes)
throw completeExtractionLimit(
`PDF text exceeds the safe ${maxTextBytes.toLocaleString()}-byte output limit.`
)
}
pageTexts.push(text)
}
}
const text = pageTexts.join(PAGE_SEPARATOR)
return complete ? text : normalizePdfWhitespace(text).trim()
Expand All @@ -297,6 +310,13 @@ function completeExtractionLimit(message: string): FileParserError {
return new FileParserError('complexity_limit', `${message} Split or simplify the PDF and retry.`)
}

function completeBudget(value: number | undefined, ceiling: number): number {
if (value === undefined) return ceiling
if (!Number.isSafeInteger(value) || value < 1)
throw completeExtractionLimit('PDF extraction limits must be positive safe integers.')
return Math.min(value, ceiling)
}

async function extractTextWithinBudget(
pdf: PDFDocumentProxy,
options: FileParseOptions,
Expand All @@ -305,17 +325,21 @@ async function extractTextWithinBudget(
const { signal } = options
const complete = options.pdfTextMode === 'complete'
const totalPages = pdf.numPages
const pageLimit = Math.min(totalPages, MAX_PDF_PAGES)
const maxPages = complete ? completeBudget(options.pdfMaxPages, MAX_PDF_PAGES) : MAX_PDF_PAGES
const maxTextBytes = complete
? completeBudget(options.maxTextBytes, MAX_COMPLETE_PDF_TEXT_BYTES)
: MAX_COMPLETE_PDF_TEXT_BYTES
const pageLimit = Math.min(totalPages, maxPages)
const pages: PdfPageLines[] = []

let remainingChars = MAX_PDF_TEXT_CHARS
let outputBytes = 0
let retainedTextBytes = 0
let pagesRead = 0
let truncated = totalPages > pageLimit

if (complete && truncated) {
throw completeExtractionLimit(
`PDF exceeds the safe limit of ${MAX_PDF_PAGES.toLocaleString()} pages.`
`PDF exceeds the safe limit of ${maxPages.toLocaleString()} pages.`
)
}

Expand All @@ -341,13 +365,9 @@ async function extractTextWithinBudget(
const page = pageResult
const pageHeight = readPageHeight(page)
let extraction: PageExtraction
const pageCharLimit = complete ? MAX_COMPLETE_PDF_PAGE_CHARS : remainingChars
try {
extraction = await readPageWithinBudget(
page,
complete ? MAX_COMPLETE_PDF_PAGE_CHARS : remainingChars,
deadline,
signal
)
extraction = await readPageWithinBudget(page, pageCharLimit, deadline, signal)
} finally {
page.cleanup()
}
Expand All @@ -361,10 +381,11 @@ async function extractTextWithinBudget(
pagesRead++
if (complete) {
if (lines.length > 0) {
outputBytes += estimatePageBytes(lines) + (pages.length > 0 ? PAGE_SEPARATOR.length : 0)
if (outputBytes > MAX_COMPLETE_PDF_TEXT_BYTES) {
retainedTextBytes +=
estimateRetainedPageBytes(lines) + (pages.length > 0 ? PAGE_SEPARATOR.length : 0)
if (retainedTextBytes > MAX_COMPLETE_PDF_RETAINED_TEXT_BYTES) {
Comment thread
waleedlatif1 marked this conversation as resolved.
throw completeExtractionLimit(
`PDF text exceeds the safe ${MAX_COMPLETE_PDF_TEXT_BYTES.toLocaleString()}-byte output limit.`
`PDF text exceeds the safe ${MAX_COMPLETE_PDF_RETAINED_TEXT_BYTES.toLocaleString()}-byte output limit.`
)
}
pages.push({ lines, pageHeight })
Expand All @@ -387,7 +408,7 @@ async function extractTextWithinBudget(
}
}

let text = await assemblePages(pages, complete, signal)
let text = await assemblePages(pages, complete, signal, maxTextBytes)

/** Paragraph breaks land after the budget is spent; trimming that overflow is a truncation too. */
if (!complete && text.length > MAX_PDF_TEXT_CHARS) {
Expand Down
9 changes: 8 additions & 1 deletion apps/sim/lib/file-parsers/types.ts
Original file line number Diff line number Diff line change
Expand Up @@ -36,15 +36,22 @@ export interface FileParseOptions {
/** Indexing callers require complete extraction; preview row limits must not discard content. */
contentMode?: 'preview' | 'complete'
/**
* CSV/XLSX byte budget when contentMode is 'complete' (default 25 MiB).
* Complete text byte budget for CSV/XLSX (default 25 MiB) and PDF (default 20 MiB).
* PDF applies this when pdfTextMode is 'complete' and never exceeds its safe default.
* Exceeding it throws complexity_limit instead of returning a truncated prefix.
* DOCX applies this only when docxTextMode is 'complete' (default 2 MiB); separate
* conversion graph limits can reject structurally complex inputs before normalization.
* Preview mode retains its own limits; other formats use their parser-specific budgets.
*/
maxTextBytes?: number
/** Preserve textual markup in a canonical .txt artifact instead of interpreting it as HTML or RTF. */
textMode?: 'literal'
/** Opt into bounded DOCX conversion without lossy or unchecked parser fallbacks. */
docxTextMode?: 'complete'
/** Complete PDF extraction rejects safety limits instead of returning preview text. */
pdfTextMode?: 'preview' | 'complete'
/** Lower page ceiling for complete PDF extraction; defaults to the parser's safe limit. */
pdfMaxPages?: number
}

export interface FileParser {
Expand Down
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
import { setEnv } from '@sim/testing/mocks/env.mock'
import { describe, expect, it } from 'vitest'
import { resetEnvMock, setEnv } from '@sim/testing/mocks/env.mock'
import { afterAll, describe, expect, it } from 'vitest'
import {
getIntegrationAvailability,
getOAuthServiceAvailability,
Expand All @@ -8,7 +8,10 @@ import {
setEnv({
GITHUB_APP_CLIENT_ID: 'repository-client',
GITHUB_APP_CLIENT_SECRET: 'repository-secret',
GOOGLE_CLIENT_ID: undefined,
GOOGLE_CLIENT_SECRET: undefined,
})
afterAll(resetEnvMock)

describe('OAuth service availability projection', () => {
it('uses the repository App while the workflow block keeps its API-key path', () => {
Expand Down
Loading
Loading