Skip to content

Commit 9aa51ee

Browse files
waleedlatif1claude
andcommitted
fix(parsers): keep date fields, drop file-name image alt text, fast-path plain cells
Only the slide-number field is layout text; date and time fields outside a dt placeholder are content, and skipping them emptied a deck made of them. Image alternative text that is a bare file name or an auto caption is noise, so the HTML, PresentationML, and OpenDocument walkers share one filter. A table cell with no element children is read directly instead of running the block-spacing and nested-table queries. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
1 parent db10f79 commit 9aa51ee

7 files changed

Lines changed: 73 additions & 18 deletions

File tree

‎apps/sim/lib/file-parsers/html-parser.test.ts‎

Lines changed: 12 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -159,6 +159,18 @@ describe('HtmlParser', () => {
159159
expect(result.metadata?.tableCount).toBe(2)
160160
})
161161

162+
it('keeps descriptive image alt text and drops file-name alt text', async () => {
163+
const buffer = Buffer.from(
164+
`<body><img alt="Org chart"><img alt="python-logo.gif"><img alt="Image 2"><p>Body</p></body>`
165+
)
166+
167+
const result = await parser.parseBuffer(buffer)
168+
169+
expect(result.content).toContain('[Image: Org chart]')
170+
expect(result.content).not.toContain('python-logo')
171+
expect(result.content).not.toContain('Image 2')
172+
})
173+
162174
it('separates block elements inside a list item', async () => {
163175
const buffer = Buffer.from(
164176
`<body><ul><li><div><p>Versions</p><p>Release Information</p></div></li></ul></body>`

‎apps/sim/lib/file-parsers/html-parser.ts‎

Lines changed: 9 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -3,6 +3,7 @@ import { createLogger } from '@sim/logger'
33
import { getErrorMessage } from '@sim/utils/errors'
44
import * as cheerio from 'cheerio'
55
import { FileParserError } from '@/lib/file-parsers/errors'
6+
import { imageAltText } from '@/lib/file-parsers/office-text'
67
import type { FileParseResult, FileParser } from '@/lib/file-parsers/types'
78
import { decodeTextBuffer, sanitizeTextForUTF8 } from '@/lib/file-parsers/utils'
89

@@ -259,10 +260,8 @@ function processElement(
259260
}
260261

261262
case 'img': {
262-
const alt = $node.attr('alt')
263-
if (alt) {
264-
contentParts.push(`[Image: ${alt}]`)
265-
}
263+
const image = imageAltText($node.attr('alt'))
264+
if (image) contentParts.push(image)
266265
break
267266
}
268267

@@ -372,14 +371,19 @@ function flattenedTableCells($: cheerio.CheerioAPI, table: cheerio.Cheerio<AnyNo
372371
}
373372

374373
/**
375-
* One cell's text on a single line. Block elements inside the cell get a space
374+
* One cell's text on a single line. A text-only cell — the common case in a
375+
* data table — is read directly. Block elements inside the cell get a space
376376
* so adjacent paragraphs do not glue together — this mutates the live DOM, and
377377
* runs before `extractHeadings`/`extractLinks`, so heading or link text inside a
378378
* cell gains those spaces too. A nested table is rendered on a clone of the cell
379379
* as its cells joined with ` / `, without `[Table]` markers or pipes, so the
380380
* outer row stays one line and the inner text appears exactly once.
381381
*/
382382
function cellText($: cheerio.CheerioAPI, cell: cheerio.Cheerio<AnyNode>): string {
383+
if (cell.children().length === 0) {
384+
return cell.text().replace(/\s+/g, ' ').trim()
385+
}
386+
383387
const nested = topLevelNestedTables($, cell)
384388
if (nested.length === 0) {
385389
cell.find(CELL_BLOCK_SELECTOR).after(' ')

‎apps/sim/lib/file-parsers/odf-text.test.ts‎

Lines changed: 8 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -127,6 +127,14 @@ describe('extractOpenDocumentText', () => {
127127
expect(await extractOpenDocumentText(buffer)).toBe('[Image: Org chart]')
128128
})
129129

130+
it('drops a file-name image title', async () => {
131+
const buffer = await text(
132+
`<text:p><draw:frame draw:name="Image1"><draw:image xlink:href="Pictures/a.png" xmlns:xlink="http://www.w3.org/1999/xlink"/><svg:title xmlns:svg="urn:oasis:names:tc:opendocument:xmlns:svg-compatible:1.0">python-icon.jpeg</svg:title></draw:frame>after</text:p>`
133+
)
134+
135+
expect(await extractOpenDocumentText(buffer)).toBe('after')
136+
})
137+
130138
it('rejects a content part above the per-part size cap before parsing it', async () => {
131139
const buffer = await text('<text:p>Small</text:p>')
132140
const zip = await JSZip.loadAsync(buffer)

‎apps/sim/lib/file-parsers/odf-text.ts‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -5,6 +5,7 @@ import {
55
collapseWhitespace,
66
findFirst,
77
formatTableRow,
8+
imageAltText,
89
isXmlElement,
910
joinBlocks,
1011
NOTES_MARKER,
@@ -246,8 +247,7 @@ function imageFrameText(frame: XmlElement, state: WalkState): string | null {
246247
const children = childElements(frame)
247248
if (!children.some((child) => child.name === 'draw:image')) return null
248249
const alt = children.find((child) => child.name === 'svg:title' || child.name === 'svg:desc')
249-
const text = alt ? collapseWhitespace(inlineText(alt, state)) : ''
250-
return text ? `[Image: ${text}]` : ''
250+
return alt ? (imageAltText(inlineText(alt, state)) ?? '') : ''
251251
}
252252

253253
function emitNotes(notes: XmlElement, state: WalkState): void {

‎apps/sim/lib/file-parsers/office-text.ts‎

Lines changed: 16 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -88,6 +88,22 @@ export function findAll(node: XmlDocument | XmlElement, tagName: string): XmlEle
8888
return DomUtils.findAll((element) => element.name === tagName, node.children)
8989
}
9090

91+
/** A bare file name (`python-logo.gif`) rather than a description. */
92+
const FILENAME_LIKE = /^[^\s]+\.[a-z0-9]{2,4}$/i
93+
94+
/** Auto-generated captions that name the object, not its content (`Picture 3`). */
95+
const AUTO_CAPTION = /^(?:picture|image|graphic|photo|figure|chart|diagram|screenshot)\s*\d*$/i
96+
97+
/**
98+
* Renders an image's alternative text as `[Image: …]`, or `null` when the text
99+
* is a file name or an auto-generated caption, which would only add noise.
100+
*/
101+
export function imageAltText(raw: string | undefined): string | null {
102+
const text = raw ? collapseWhitespace(raw) : ''
103+
if (!text || FILENAME_LIKE.test(text) || AUTO_CAPTION.test(text)) return null
104+
return `[Image: ${text}]`
105+
}
106+
91107
/** Collapses internal whitespace so a cell or list item occupies a single line. */
92108
export function collapseWhitespace(text: string): string {
93109
return text.replace(/\s+/g, ' ').trim()

‎apps/sim/lib/file-parsers/ooxml-presentation.test.ts‎

Lines changed: 15 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -145,14 +145,24 @@ describe('extractPresentationText', () => {
145145
)
146146
})
147147

148-
it('skips slide-number and date fields outside their placeholders', async () => {
149-
const spTree = `<p:sp><p:nvSpPr><p:cNvPr id="1" name="s"/><p:cNvSpPr/></p:nvSpPr><p:txBody><a:p><a:r><a:t>Page </a:t></a:r><a:fld type="slidenum"><a:t>369</a:t></a:fld><a:fld type="datetime1"><a:t>1/1/2026</a:t></a:fld><a:fld type="custom"><a:t>kept</a:t></a:fld></a:p></p:txBody></p:sp>`
148+
it('skips slide-number fields outside their placeholder but keeps date fields', async () => {
149+
const spTree = `<p:sp><p:nvSpPr><p:cNvPr id="1" name="s"/><p:cNvSpPr/></p:nvSpPr><p:txBody><a:p><a:r><a:t>Page </a:t></a:r><a:fld type="slidenum"><a:t>369</a:t></a:fld><a:fld type="datetime1"><a:t>6/29/2021</a:t></a:fld><a:fld type="custom"><a:t>kept</a:t></a:fld></a:p></p:txBody></p:sp>`
150150

151-
expect(await extractPresentationText(await buildDeck([{ index: 1, spTree }]))).toBe('Page kept')
151+
expect(await extractPresentationText(await buildDeck([{ index: 1, spTree }]))).toBe(
152+
'Page 6/29/2021kept'
153+
)
154+
})
155+
156+
it('still drops a date field inside a dt placeholder', async () => {
157+
const spTree = `<p:sp><p:nvSpPr><p:cNvPr id="1" name="s"/><p:cNvSpPr/><p:nvPr><p:ph type="dt"/></p:nvPr></p:nvSpPr><p:txBody><a:p><a:fld type="datetime1"><a:t>6/29/2021</a:t></a:fld></a:p></p:txBody></p:sp>${shape('Body')}`
158+
159+
expect(await extractPresentationText(await buildDeck([{ index: 1, spTree }]))).toBe('Body')
152160
})
153161

154-
it('emits a picture as its alternative text', async () => {
155-
const spTree = `<p:pic><p:nvPicPr><p:cNvPr id="4" name="Picture 3" descr="Org chart"/><p:cNvPicPr/><p:nvPr/></p:nvPicPr></p:pic>${shape('Caption')}`
162+
it('emits a picture as its alternative text unless it is a file name or auto caption', async () => {
163+
const pic = (descr: string) =>
164+
`<p:pic><p:nvPicPr><p:cNvPr id="4" name="Picture 3" descr="${descr}"/><p:cNvPicPr/><p:nvPr/></p:nvPicPr></p:pic>`
165+
const spTree = pic('Org chart') + pic('python-logo.gif') + pic('Picture 2') + shape('Caption')
156166

157167
expect(await extractPresentationText(await buildDeck([{ index: 1, spTree }]))).toBe(
158168
'[Image: Org chart]\nCaption'

‎apps/sim/lib/file-parsers/ooxml-presentation.ts‎

Lines changed: 11 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -4,6 +4,7 @@ import {
44
findAll,
55
findFirst,
66
formatTableRow,
7+
imageAltText,
78
isXmlElement,
89
joinBlocks,
910
NOTES_MARKER,
@@ -31,8 +32,12 @@ const SLIDE_PART = /^ppt\/slides\/slide(\d+)\.xml$/
3132
const NOTES_RELATIONSHIP_SUFFIX = '/notesSlide'
3233
const NOTES_PART_PREFIX = 'ppt/notesSlides/'
3334

34-
/** Fields PowerPoint fills at render time; their cached text is the layout's, not the author's. */
35-
const RENDER_TIME_FIELD_TYPES = /^(slidenum|datetime)/i
35+
/**
36+
* The slide-number field's cached text is the layout's, not the author's. Date
37+
* fields keep their text: outside a `dt` placeholder (already skipped) a deck's
38+
* dates are content, and python-pptx keeps them too.
39+
*/
40+
const SLIDE_NUMBER_FIELD_TYPE = 'slidenum'
3641

3742
/** Layout-chrome placeholders whose text is a field, not slide content. */
3843
const SKIPPED_PLACEHOLDER_TYPES = new Set(['sldNum', 'dt', 'ftr', 'hdr'])
@@ -49,7 +54,7 @@ function placeholderType(shape: XmlElement): string | null {
4954

5055
/**
5156
* Concatenates a DrawingML paragraph's runs, turning `<a:br/>` into a newline
52-
* and skipping slide-number and date fields wherever they appear.
57+
* and skipping slide-number fields wherever they appear.
5358
*/
5459
function paragraphText(paragraph: XmlElement): string {
5560
const pieces: string[] = []
@@ -58,7 +63,7 @@ function paragraphText(paragraph: XmlElement): string {
5863
pieces.push('\n')
5964
return
6065
}
61-
if (element.name === 'a:fld' && RENDER_TIME_FIELD_TYPES.test(element.attribs.type ?? '')) {
66+
if (element.name === 'a:fld' && element.attribs.type === SLIDE_NUMBER_FIELD_TYPE) {
6267
return
6368
}
6469
if (element.name === 'a:t') {
@@ -124,8 +129,8 @@ function graphicFrameBlocks(frame: XmlElement): string[] {
124129
function pictureBlocks(picture: XmlElement): string[] {
125130
const nonVisual = childElements(picture).find((child) => child.name === 'p:nvPicPr')
126131
const properties = nonVisual ? findFirst(nonVisual, 'p:cNvPr') : null
127-
const description = properties?.attribs.descr?.trim()
128-
return description ? [`[Image: ${description}]`] : []
132+
const image = imageAltText(properties?.attribs.descr)
133+
return image ? [image] : []
129134
}
130135

131136
/**

0 commit comments

Comments
 (0)