Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion apps/sim/lib/file-parsers/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -176,7 +176,7 @@ export async function parseBuffer(
}

const kind = sniffFileKind(buffer, normalizedExtension)
const route = reconcileParserRoute(normalizedExtension, kind)
const route = reconcileParserRoute(normalizedExtension, kind, options)
const parser = PARSERS.get(route.extension)

if (!parser?.parseBuffer) {
Expand Down
32 changes: 32 additions & 0 deletions apps/sim/lib/file-parsers/sniff.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -308,6 +308,38 @@ describe('parseBuffer reconciles the extension with the sniffed bytes', () => {
expect(result.metadata?.detectedType).toBe('html')
})

it.each([
'<!doctype html><html><head><script src="./app.js"></script></head><body><div id="root"></div></body></html>',
'{\\rtf1\\ansi Source-format example}',
])(
'preserves textual markup when the caller supplies a canonical text artifact',
async (content) => {
const result = await parseBuffer(Buffer.from(content), 'txt', { textMode: 'literal' })

expect(result.content).toBe(content)
expect(result.metadata?.detectedType).toBeUndefined()
}
)

it('keeps literal-text handling scoped to txt artifacts', async () => {
const result = await parseBuffer(
Buffer.from('<!doctype html><html><body><p>Readable page</p></body></html>'),
'html',
{ textMode: 'literal' }
)

expect(result.content).toContain('Readable page')
expect(result.content).not.toContain('<html>')
await expect(
parseBuffer(Buffer.from('<!doctype html><html><body>403 Forbidden</body></html>'), 'json', {
textMode: 'literal',
})
).rejects.toMatchObject({ code: 'invalid_format' })
await expect(parseBuffer(oleBinary(), 'txt', { textMode: 'literal' })).rejects.toMatchObject({
code: 'invalid_format',
})
})

it('extracts a docx labelled .xlsx through the Word parser', async () => {
const result = await parseBuffer(await buildDocx('Office Relocation'), 'xlsx')

Expand Down
14 changes: 13 additions & 1 deletion apps/sim/lib/file-parsers/sniff.ts
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import { FileParserError } from '@/lib/file-parsers/errors'
import { isEncryptedOoxmlContainer } from '@/lib/file-parsers/ooxml-encryption'
import type { FileParseOptions } from '@/lib/file-parsers/types'
import { decodeTextBuffer, detectBomlessUtf16 } from '@/lib/file-parsers/utils'
import { isZipShaped } from '@/lib/file-parsers/zip-guard'

Expand Down Expand Up @@ -307,7 +308,18 @@ function invalidFormat(extension: string, kind: SniffedKind): FileParserError {
* text (as CSV under a spreadsheet extension), and an OLE2 file under a modern
* Word extension is the legacy `.doc` parser's job. Legacy `.ppt` has no reader.
*/
export function reconcileParserRoute(extension: string, kind: SniffedKind): ParserRoute {
export function reconcileParserRoute(
extension: string,
kind: SniffedKind,
options: Pick<FileParseOptions, 'textMode'> = {}
): ParserRoute {
if (
extension === 'txt' &&
options.textMode === 'literal' &&
(kind === 'html' || kind === 'rtf')
) {
return { extension }
}
if (kind === 'rtf') {
throw new FileParserError(
'unsupported_type',
Expand Down
2 changes: 2 additions & 0 deletions apps/sim/lib/file-parsers/types.ts
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,8 @@ export interface FileParseResult {

export interface FileParseOptions {
signal?: AbortSignal
/** Preserve textual markup in a canonical .txt artifact instead of interpreting it as HTML or RTF. */
textMode?: 'literal'
/** Complete PDF extraction rejects safety limits instead of returning preview text. */
pdfTextMode?: 'preview' | 'complete'
}
Expand Down
Loading
Loading