From 241dc342990339f459d13686528b0a8b4dcb9775 Mon Sep 17 00:00:00 2001 From: Etherlink Intern Date: Tue, 2 Jun 2026 21:59:37 +0800 Subject: [PATCH] Add Apple PDF text conversion --- .github/workflows/ci.yml | 10 ++- Docs/NativeConverterBackends.md | 14 ++-- README.md | 10 +-- Scripts/smoke-test.sh | 10 ++- .../Converters/PDFConverter.swift | 44 ++++++++++++ Sources/SwiftMarkItDown/MarkItDown.swift | 3 +- Tests/Expected/sample.md | 2 + Tests/Fixtures/empty.pdf | Bin 0 -> 460 bytes Tests/Fixtures/sample.pdf | Bin 54 -> 659 bytes .../SwiftMarkItDownTests.swift | 63 ++++++++++++++++-- 10 files changed, 135 insertions(+), 21 deletions(-) create mode 100644 Sources/SwiftMarkItDown/Converters/PDFConverter.swift create mode 100644 Tests/Expected/sample.md diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 6f34ce4..3a9184a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -9,8 +9,14 @@ on: jobs: test: - name: Swift tests and smoke tests - runs-on: macos-15 + name: Swift tests and smoke tests (${{ matrix.os }}) + runs-on: ${{ matrix.os }} + strategy: + fail-fast: false + matrix: + os: + - macos-15 + - ubuntu-latest steps: - name: Checkout uses: actions/checkout@v5 diff --git a/Docs/NativeConverterBackends.md b/Docs/NativeConverterBackends.md index 61c8e79..ff966db 100644 --- a/Docs/NativeConverterBackends.md +++ b/Docs/NativeConverterBackends.md @@ -1,6 +1,6 @@ # Native and FOSS Converter Backend Plan -This document records the likely native/FOSS backends for the reserved document formats in `DocumentFormat`. These are **not package dependencies yet**; the current release still returns `unsupportedFormat` for PDF, DOCX, PPTX, and XLSX. The goal is to keep the app's import UI honest while making the implementation path explicit. +This document records the native/FOSS backend status for the reserved document formats in `DocumentFormat`. PDFKit-backed embedded-text PDF extraction is now implemented on Apple platforms; DOCX, PPTX, and XLSX still return `unsupportedFormat` until their native converter modules are implemented. The goal is to keep the app's import UI honest while making the implementation path explicit. ## Short answer: are these simple to add? @@ -8,7 +8,7 @@ They are straightforward to integrate as Swift Package / Apple-framework buildin | Format | Proposed backend | Integration effort | Why | | --- | --- | --- | --- | -| PDF | Apple's PDFKit | Low for embedded text; medium when OCR fallback is included. | PDFKit is built into Apple platforms and can extract text from many PDFs, but scanned/image-only PDFs still need page rendering plus Vision OCR. | +| PDF | Apple's PDFKit | Shipped for embedded text; medium remaining work for OCR fallback. | PDFKit is built into Apple platforms and now extracts embedded text from PDFs, but scanned/image-only PDFs still need page rendering plus Vision OCR. | | DOCX | ZIPFoundation + OOXML parsing | Medium. | DOCX is a ZIP of XML parts, but useful Markdown needs document body parsing, relationships, styles, numbering, tables, hyperlinks, and images. | | PPTX | ZIPFoundation + OOXML parsing | Medium-high. | PPTX uses the same OpenXML ZIP structure, but slide ordering, shapes, notes, and layout-driven reading order make Markdown extraction more involved than DOCX. | | XLSX | CoreXLSX | Medium-low for worksheet tables; medium for richer workbooks. | CoreXLSX already parses XLSX structure in Swift, but Markdown output still needs shared strings, sheet selection, empty-cell handling, formulas, merged cells, and table shaping decisions. | @@ -26,11 +26,11 @@ When one of these backends is implemented, the PR that adds it should also add t ## Recommended implementation order -1. **PDF text extraction with PDFKit** - - Add a `PDFConverter` behind `#if canImport(PDFKit)`. - - Extract embedded page text first. - - If a page has no embedded text, optionally render the page and reuse the Vision OCR path already used by image ingestion. - - Keep non-Apple platforms returning `unsupportedFormat` unless a separate cross-platform PDF backend is added. +1. **PDF text extraction with PDFKit** — shipped for embedded text + - `PDFConverter` is compiled behind `#if canImport(PDFKit)`. + - Embedded page text is extracted first. + - Remaining work: if a page has no embedded text, optionally render the page and reuse the Vision OCR path already used by image ingestion. + - Non-Apple platforms continue returning `unsupportedFormat` unless a separate cross-platform PDF backend is added. 2. **Shared OpenXML ZIP infrastructure** - Add ZIPFoundation as a SwiftPM dependency only when DOCX or PPTX work starts. diff --git a/README.md b/README.md index 81f28b1..09ce7f5 100644 --- a/README.md +++ b/README.md @@ -14,6 +14,7 @@ The current MVP is intentionally small and deterministic. These formats have con | CSV | `csv` | `text/csv`, `application/csv` | Converts rows to GitHub-Flavored Markdown tables, including quoted fields and escaped pipes. | All package platforms. | | JSON | `json` | `application/json`, `text/json` | Converts objects and arrays to nested Markdown bullets with stable key ordering. | All package platforms. | | Images | `png`, `jpg`, `jpeg`, `heic`, `heif`, `tif`, `tiff`, `gif` | `image/png`, `image/jpeg`, `image/heic`, `image/heif`, `image/tiff`, `image/gif` | Uses Apple Vision OCR and returns recognized text lines as Markdown text. GIF OCR uses the decoded first image. | Apple platforms that provide Vision, CoreGraphics, and ImageIO. Other platforms recognize the formats but return `unsupportedFormat`. | +| PDF | `pdf` | `application/pdf` | Uses Apple PDFKit to extract embedded page text into Markdown paragraphs and records page-count metadata. | Apple platforms that provide PDFKit. Other platforms recognize PDF but return `unsupportedFormat`. | The package also includes: @@ -22,12 +23,13 @@ The package also includes: - `csv` to GitHub-Flavored Markdown tables, including quoted fields and escaped pipes. - `json` to nested Markdown bullets with stable key ordering. - Apple-platform image OCR for `png`, `jpg`/`jpeg`, `heic`, `tiff`, and `gif` inputs using Vision text recognition, returning recognized lines as Markdown text. +- Apple-platform PDF text extraction for embedded-text PDFs using PDFKit. - A CLI wrapper for local/manual conversion checks. - A SwiftUI iOS demo app for editing sample input and converting it to Markdown in the simulator. Image OCR is available when the package is built on platforms that provide Vision, CoreGraphics, and ImageIO. On other platforms, image formats are recognized but return `unsupportedFormat`. -PDF, DOCX, PPTX, and XLSX are represented in the format model but still return `unsupportedFormat` until their native converter modules are implemented. +DOCX, PPTX, and XLSX are represented in the format model but still return `unsupportedFormat` until their native converter modules are implemented. PDF conversion is implemented on Apple platforms that provide PDFKit; non-Apple platforms still return `unsupportedFormat` for PDF. ## Repository layout @@ -87,7 +89,7 @@ SwiftMarkItDown is available under the [MIT License](LICENSE). 1. Expand the text/HTML/CSV/JSON converters with richer Markdown normalization and metadata extraction. 2. Improve OCR layout reconstruction for headings, lists, tables, and multi-column scans. -3. Add a ZIP/OpenXML package reader as shared infrastructure for DOCX, PPTX, and XLSX. -4. Implement DOCX paragraph, heading, table, hyperlink, and image-reference extraction. -5. Add PDFKit/Vision-backed PDF text and OCR extraction for Apple platforms behind conditional compilation. +3. Add OCR fallback for scanned/image-only PDF pages on Apple platforms. +4. Add a ZIP/OpenXML package reader as shared infrastructure for DOCX, PPTX, and XLSX. +5. Implement DOCX paragraph, heading, table, hyperlink, and image-reference extraction. 6. Evolve the demo into a more complete iOS MVP with document picker import, share/export flows, progress reporting, and a pluggable backend escape hatch for heavyweight conversions. diff --git a/Scripts/smoke-test.sh b/Scripts/smoke-test.sh index 9b5d8a7..c77052c 100755 --- a/Scripts/smoke-test.sh +++ b/Scripts/smoke-test.sh @@ -70,11 +70,17 @@ run_case empty.html run_case empty.csv run_error_case empty.json "swift-markitdown: JSON input is empty." -run_error_case sample.pdf "swift-markitdown: No converter is registered for pdf." +if swift -e 'import PDFKit' >/dev/null 2>&1; then + run_case sample.pdf + run_case empty.pdf +else + run_error_case sample.pdf "swift-markitdown: No converter is registered for pdf." + run_error_case empty.pdf "swift-markitdown: No converter is registered for pdf." +fi + run_error_case sample.docx "swift-markitdown: No converter is registered for docx." run_error_case sample.pptx "swift-markitdown: No converter is registered for pptx." run_error_case sample.xlsx "swift-markitdown: No converter is registered for xlsx." -run_error_case empty.pdf "swift-markitdown: No converter is registered for pdf." run_error_case empty.docx "swift-markitdown: No converter is registered for docx." run_error_case empty.pptx "swift-markitdown: No converter is registered for pptx." run_error_case empty.xlsx "swift-markitdown: No converter is registered for xlsx." diff --git a/Sources/SwiftMarkItDown/Converters/PDFConverter.swift b/Sources/SwiftMarkItDown/Converters/PDFConverter.swift new file mode 100644 index 0000000..cc004d1 --- /dev/null +++ b/Sources/SwiftMarkItDown/Converters/PDFConverter.swift @@ -0,0 +1,44 @@ +import Foundation +#if canImport(PDFKit) +import PDFKit +#endif + +/// Extracts Markdown-ready text from PDF documents using Apple PDFKit when available. +public struct PDFConverter: DocumentConverter { + public let supportedFormats: Set = [.pdf] + + public init() {} + + public func convert(_ request: ConversionRequest, format: DocumentFormat) throws -> MarkdownDocument { + #if canImport(PDFKit) + guard let document = PDFDocument(data: request.data) else { + throw ConversionError.malformedInput("The input could not be parsed as a PDF document.") + } + + var extractedPageCount = 0 + let pages = (0..lWQx7mEA()spZWD(<-4YlHN>Z)x^*9}DMa?OGe$VgO z(R#7GNH231!SD5QffJ7;G#O_2?6?uu}5Mbg)Z^KS?=E z3r_Pk=uB&0C=ewr%V3~A=^W(2ao`i`Kc9QljejM8@;h%)euOp7zrgD()+_<_JY(NsN=jVAr3sEq81fFZm0*7I5a%s%5e(q*Vh{Y4W(8N_U!D; zZgxAfQPgD*F6qGHkY{?z)pnvs@=$<@K$ahMr37JMfn04!uLm^#`Vc^eT=Sh=1D)}; zD8V*_IrRHP6e$w%eVuUP(dRrZVd)_K&8Wz#(7;-W7`aVE8zl_&L?VySQy@5 z%r#BnY4Pm9ti>P9aa)*HJl_dbZ_OhA7T}u+Nr?k*wx!fu>AFrg@JXjLevQZvkjJog delta 33 ocmbQtYBoVy+SyZ~G_Sa{pdi1fBsE1Lz{O1=EwiGev?!Ge0J&`oH2?qr diff --git a/Tests/SwiftMarkItDownTests/SwiftMarkItDownTests.swift b/Tests/SwiftMarkItDownTests/SwiftMarkItDownTests.swift index 973b260..f749850 100644 --- a/Tests/SwiftMarkItDownTests/SwiftMarkItDownTests.swift +++ b/Tests/SwiftMarkItDownTests/SwiftMarkItDownTests.swift @@ -2,6 +2,10 @@ import Foundation import Testing @testable import SwiftMarkItDown +#if canImport(PDFKit) +import PDFKit +#endif + #if canImport(Vision) && canImport(CoreGraphics) && canImport(CoreText) && canImport(ImageIO) import CoreGraphics import CoreText @@ -61,13 +65,52 @@ struct SwiftMarkItDownTests { } - @Test("throws for reserved but unimplemented formats") + @Test("throws for reserved but unimplemented non-PDF formats") func throwsForUnimplementedFormats() throws { - let request = ConversionRequest(data: Data(), fileName: "paper.pdf") + let cases: [(String, DocumentFormat)] = [ + ("document.docx", .docx), + ("deck.pptx", .pptx), + ("workbook.xlsx", .xlsx) + ] + + for (fileName, format) in cases { + let request = ConversionRequest(data: Data(), fileName: fileName) + #expect(throws: ConversionError.unsupportedFormat(format)) { + try MarkItDown().convert(request) + } + } + } + + #if canImport(PDFKit) + @Test("converts embedded-text PDF fixtures to Markdown") + func convertsPDF() throws { + let document = try MarkItDown().convert(contentsOf: fixtureURL("sample.pdf")) + let expected = try String(contentsOf: expectedURL("sample.md"), encoding: .utf8) + .smid_testTrimmedTrailingNewline + + #expect(document.markdown == expected) + #expect(document.sourceFormat == .pdf) + #expect(document.metadata["pageCount"] == "1") + #expect(document.metadata["extractedTextPageCount"] == "1") + } + + @Test("converts blank PDF fixtures to empty Markdown") + func convertsBlankPDF() throws { + let document = try MarkItDown().convert(contentsOf: fixtureURL("empty.pdf")) + + #expect(document.markdown == "") + #expect(document.sourceFormat == .pdf) + #expect(document.metadata["pageCount"] == "1") + #expect(document.metadata["extractedTextPageCount"] == "0") + } + #else + @Test("throws unsupported for PDFs when PDFKit is unavailable") + func throwsForPDFsWhenPDFKitIsUnavailable() throws { #expect(throws: ConversionError.unsupportedFormat(.pdf)) { - try MarkItDown().convert(request) + try MarkItDown().convert(contentsOf: fixtureURL("sample.pdf")) } } + #endif #if canImport(Vision) && canImport(CoreGraphics) && canImport(CoreText) && canImport(ImageIO) @Test("uses Vision OCR to convert rendered images to Markdown") @@ -105,11 +148,9 @@ struct SwiftMarkItDownTests { @Test("throws unsupported for every reserved document path, including empty documents") func throwsUnsupportedForReservedDocumentPaths() throws { let cases: [(String, DocumentFormat)] = [ - ("sample.pdf", .pdf), ("sample.docx", .docx), ("sample.pptx", .pptx), ("sample.xlsx", .xlsx), - ("empty.pdf", .pdf), ("empty.docx", .docx), ("empty.pptx", .pptx), ("empty.xlsx", .xlsx) @@ -136,6 +177,18 @@ private func fixtureURL(_ fileName: String) -> URL { .appendingPathComponent(fileName) } +private func expectedURL(_ fileName: String) -> URL { + URL(fileURLWithPath: FileManager.default.currentDirectoryPath) + .appendingPathComponent("Tests/Expected") + .appendingPathComponent(fileName) +} + +private extension String { + var smid_testTrimmedTrailingNewline: String { + trimmingCharacters(in: .newlines) + } +} + private func blankPNGFixtureData() throws -> Data { #if canImport(Vision) && canImport(CoreGraphics) && canImport(ImageIO) let width = 100