From 35d155ba6bcd78f7a6f4257450027282d76872b0 Mon Sep 17 00:00:00 2001 From: Himanshu Shekhar Date: Fri, 29 May 2026 03:11:01 +0530 Subject: [PATCH 1/2] feat(files): Docling extraction + MINT document serialization Adds a Docling-based file extractor (python/docling_extract.py, spawned by docling-client.service.ts) as the default, with graceful fallback to the existing Gemini extractor on any failure. Extracted document structure is serialized to MINT via the new mint-mapper (encodeDocument from @q1k-oss/mint-format >=1.1.0) and stored in files.mint_content for token-efficient prompt injection. - file.types: DocumentStructure gains figures[] + section pageRef - schema: files.mint_content column (apply via db:push) - config: extractor option ('docling' | 'gemini') - adds CLAUDE.md (extractor flow, Docling/uv setup, MINT usage) - scripts/smoke-docling.ts verifies the structure -> MINT path --- CLAUDE.md | 50 +++++++ package.json | 2 +- pyproject.toml | 11 ++ python/docling_extract.py | 134 +++++++++++++++++++ scripts/smoke-docling.ts | 61 +++++++++ src/config.ts | 2 + src/db/schema/files.ts | 2 + src/services/files/docling-client.service.ts | 65 +++++++++ src/services/files/file-processor.service.ts | 50 +++++-- src/services/files/mint-mapper.ts | 80 +++++++++++ src/types/file.types.ts | 7 + 11 files changed, 451 insertions(+), 13 deletions(-) create mode 100644 CLAUDE.md create mode 100644 pyproject.toml create mode 100644 python/docling_extract.py create mode 100644 scripts/smoke-docling.ts create mode 100644 src/services/files/docling-client.service.ts create mode 100644 src/services/files/mint-mapper.ts diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 0000000..d133aea --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1,50 @@ +# CLAUDE.md + +Guidance for Claude Code when working in the **context-engine** repo. + +## What this is + +`@q1k-oss/context-engine` — a TypeScript SDK + HTTP service for the customer-knowledge layer: file ingestion, knowledge-graph extraction, and prioritized-context retrieval for LLM prompts. It is consumed by `q1k-controlplane` as a vendored npm dependency (with a patch). Initialize with `initContextEngine({ databaseUrl, ... })` before using any service. + +## Commands + +```bash +npm run dev # tsx watch src/server.ts +npm run build # tsc -> dist/ +npm run start # node dist/server.js +npm run db:push # apply schema to the DB (this repo uses push, not migrations) +npm run db:studio +``` + +## File extraction (upload → knowledge graph) + +`POST /api/files/upload` → `fileProcessorService.saveFile` (writes to `uploadDir`, default `./uploads`) → async `processFile`: + +1. **Extract** via the configured extractor: + - `extractor: 'docling'` (default) — `doclingClientService` spawns `python/docling_extract.py` (Docling). On any failure it **falls back to Gemini**, so a missing Python/Docling runtime degrades gracefully. + - `extractor: 'gemini'` — `geminiClientService` (extraction only, never reasoning). + Both return the same `ExtractedContent` shape (`src/types/file.types.ts`). +2. **Serialize to MINT** — `structureToMint` (`src/services/files/mint-mapper.ts`) encodes the document `structure` via `@q1k-oss/mint-format`'s `encodeDocument` and stores it in `files.mint_content` for token-efficient prompt injection. +3. **Integrate into the graph** — `integrateIntoGraph` creates `Artifact` + `Entity` nodes/edges. + +### Docling (Python) setup + +Docling is a Python dependency (`pyproject.toml`), not bundled with the JS package. To run the default extractor locally: + +```bash +uv sync # installs Docling (first run pulls ML models, ~hundreds of MB) +npx tsx scripts/smoke-docling.ts # MINT-mapping smoke (no Docling needed) +npx tsx scripts/smoke-docling.ts file.pdf # real Docling extraction on a file +``` + +Override the Python command with `DOCLING_CMD` (default `uv run python`). The script must emit valid JSON to stdout or exit non-zero (the TS caller falls back to Gemini on failure). + +## MINT usage + +MINT (`@q1k-oss/mint-format`) is used to pack data into LLM prompts token-efficiently: graph nodes/edges (`claude-client.service.ts`, `encode`) and now parsed documents (`encodeDocument`, see the mapper above). Requires `@q1k-oss/mint-format` ≥ 1.1.0 (adds `encodeDocument`/`decodeDocument`). + +## Conventions + +- ESM, NodeNext module resolution — **import with `.js` extensions** even from `.ts`. +- Schema in `src/db/schema/*.ts`; apply with `db:push`. +- HTTP routes currently use a trusted-caller pattern (no JWT yet) — q1k-auth middleware is a planned hardening item. diff --git a/package.json b/package.json index ef582a9..3972905 100644 --- a/package.json +++ b/package.json @@ -86,7 +86,7 @@ "dependencies": { "@anthropic-ai/sdk": "^0.39.0", "@google/generative-ai": "^0.21.0", - "@q1k-oss/mint-format": "^1.0.2", + "@q1k-oss/mint-format": "^1.1.0", "cors": "^2.8.5", "drizzle-orm": "^0.38.3", "express": "^4.21.2", diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..c090259 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,11 @@ +[project] +name = "context-engine-docling" +version = "0.1.0" +description = "Docling extraction adapter for context-engine file ingestion" +requires-python = ">=3.11" +dependencies = [ + "docling>=2.0.0", +] + +[tool.uv] +package = false diff --git a/python/docling_extract.py b/python/docling_extract.py new file mode 100644 index 0000000..002c2fa --- /dev/null +++ b/python/docling_extract.py @@ -0,0 +1,134 @@ +#!/usr/bin/env python3 +"""Docling extraction adapter for context-engine. + +Reads a single file path (argv[1]), parses it with Docling, and prints a JSON +object to stdout matching the TypeScript `ExtractedContent` shape: + + { + "textContent": str, + "structure": { + "title": str | None, + "sections": [{"heading": str, "content": str, "level": int, "pageRef": int | None}], + "tables": [{"caption": str | None, "headers": [str], "rows": [[str]]}], + "lists": [{"type": "ordered" | "unordered", "items": [str]}], + "figures": [{"caption": str | None, "pageRef": int | None}] + }, + "metadata": {"pageCount": int, "wordCount": int, "characterCount": int} + } + +The TS caller (`docling-client.service.ts`) spawns this and falls back to the +Gemini extractor on a non-zero exit, so this script must fail loudly (exit 1 + +stderr) rather than emit partial JSON. + +Run: `uv run python python/docling_extract.py ` +""" + +import json +import sys + + +def _page_of(item) -> "int | None": + prov = getattr(item, "prov", None) + if prov: + first = prov[0] if isinstance(prov, (list, tuple)) else prov + return getattr(first, "page_no", None) + return None + + +def extract(path: str) -> dict: + from docling.document_converter import DocumentConverter + + doc = DocumentConverter().convert(path).document + + text_content = doc.export_to_markdown() + + sections: list[dict] = [] + lists: list[dict] = [] + current_list: "dict | None" = None + + for item in getattr(doc, "texts", []) or []: + label = str(getattr(item, "label", "")).lower() + text = (getattr(item, "text", "") or "").strip() + if not text: + continue + + if "list_item" in label: + if current_list is None: + current_list = {"type": "unordered", "items": []} + lists.append(current_list) + current_list["items"].append(text) + continue + current_list = None # any non-list item closes the run + + if "section_header" in label or "title" in label: + sections.append( + { + "heading": text, + "content": "", + "level": int(getattr(item, "level", 1) or 1), + "pageRef": _page_of(item), + } + ) + else: + if not sections: + sections.append({"heading": "", "content": "", "level": 1, "pageRef": _page_of(item)}) + sections[-1]["content"] = (sections[-1]["content"] + "\n" + text).strip() + + tables: list[dict] = [] + for tbl in getattr(doc, "tables", []) or []: + try: + df = tbl.export_to_dataframe() + headers = [str(c) for c in df.columns.tolist()] + rows = [[("" if v is None else str(v)) for v in row] for row in df.values.tolist()] + except Exception: + headers, rows = [], [] + caption = None + try: + caption = tbl.caption_text(doc) or None + except Exception: + pass + tables.append({"caption": caption, "headers": headers, "rows": rows}) + + figures: list[dict] = [] + for pic in getattr(doc, "pictures", []) or []: + caption = None + try: + caption = pic.caption_text(doc) or None + except Exception: + pass + figures.append({"caption": caption, "pageRef": _page_of(pic)}) + + title = sections[0]["heading"] if sections and sections[0]["heading"] else None + + return { + "textContent": text_content, + "structure": { + "title": title, + "sections": sections, + "tables": tables, + "lists": lists, + "figures": figures, + }, + "metadata": { + "pageCount": len(getattr(doc, "pages", []) or []), + "wordCount": len(text_content.split()), + "characterCount": len(text_content), + }, + } + + +def main() -> int: + if len(sys.argv) < 2: + print("usage: docling_extract.py ", file=sys.stderr) + return 2 + try: + result = extract(sys.argv[1]) + except Exception as exc: # noqa: BLE001 — surface any failure to the TS caller + print(f"docling extraction failed: {exc}", file=sys.stderr) + return 1 + json.dump(result, sys.stdout, ensure_ascii=False) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/smoke-docling.ts b/scripts/smoke-docling.ts new file mode 100644 index 0000000..4aa62ee --- /dev/null +++ b/scripts/smoke-docling.ts @@ -0,0 +1,61 @@ +/** + * Smoke test for the Docling → MINT path. + * + * Always runs the deterministic MINT-mapping check (no Docling needed): + * npx tsx scripts/smoke-docling.ts + * + * Optionally runs the real Docling extractor on a file (requires `uv sync` + * to have installed Docling + first-run model download): + * npx tsx scripts/smoke-docling.ts path/to/sample.pdf + */ +import { decodeDocument } from '@q1k-oss/mint-format'; +import { structureToMint } from '../src/services/files/mint-mapper.js'; +import { doclingClientService } from '../src/services/files/docling-client.service.js'; +import type { DocumentStructure } from '../src/types/index.js'; + +const sample: DocumentStructure = { + title: 'Sample SOP', + sections: [ + { heading: 'Overview', content: 'Covers returns within 30 days.', level: 1, pageRef: 1 }, + { heading: 'Eligibility', content: 'Items must be unused.', level: 2, pageRef: 2 }, + ], + tables: [{ caption: 'Windows', headers: ['Category', 'Days'], rows: [['Apparel', '30']] }], + lists: [{ type: 'ordered', items: ['Inspect item', 'Issue refund'] }], + figures: [{ caption: 'Refund flow', pageRef: 2 }], +}; + +function assert(cond: boolean, msg: string): void { + if (!cond) throw new Error(`SMOKE FAIL: ${msg}`); +} + +async function main(): Promise { + const mint = structureToMint(sample, { pageCount: 4, author: 'Acme' }); + console.log('--- MINT ---\n' + mint + '\n'); + + assert(mint.includes('§1 Overview'), 'heading not rendered'); + assert(mint.includes('| Category | Days |'), 'table not rendered'); + assert(mint.includes('1. Inspect item'), 'ordered list not rendered'); + assert(mint.includes('@fig:'), 'figure not rendered'); + + const decoded = decodeDocument(mint); + assert(decoded.sections.length >= 2, 'sections lost on decode round-trip'); + console.log(`MINT mapping OK — ${decoded.sections.length} sections round-tripped.\n`); + + const file = process.argv[2]; + if (file) { + console.log(`Running Docling on ${file} ...`); + const r = await doclingClientService.extractFromFile({ storagePath: file, mimeType: 'application/pdf' }); + if (!r.success) { + console.log(`Docling failed (expected if not installed): ${r.error}`); + return; + } + const s = r.extractedContent?.structure; + assert(!!s, 'docling returned no structure'); + console.log('Docling OK. MINT preview:\n' + structureToMint(s!, r.extractedContent?.metadata).slice(0, 500)); + } +} + +main().catch((err) => { + console.error(err); + process.exit(1); +}); diff --git a/src/config.ts b/src/config.ts index 3205da1..8d52eff 100644 --- a/src/config.ts +++ b/src/config.ts @@ -19,6 +19,8 @@ export interface ContextEngineConfig { claudeModel?: string; /** Gemini model for file extraction (default: 'gemini-2.0-flash-exp') */ geminiModel?: string; + /** File extractor: 'docling' (Python subprocess, default) or 'gemini'. */ + extractor?: 'docling' | 'gemini'; /** PostgreSQL connection pool size (default: 10) */ dbPoolSize?: number; /** PostgreSQL connection idle timeout in seconds (default: 30) */ diff --git a/src/db/schema/files.ts b/src/db/schema/files.ts index b9471d8..a7c4126 100644 --- a/src/db/schema/files.ts +++ b/src/db/schema/files.ts @@ -10,6 +10,8 @@ export const files = pgTable('files', { fileSize: integer('file_size').notNull(), storagePath: text('storage_path').notNull(), extractedContent: jsonb('extracted_content'), + // MINT-serialized document (encodeDocument output) for token-efficient prompt injection. + mintContent: text('mint_content'), processingStatus: varchar('processing_status', { length: 20 }).default('pending').notNull(), processingError: text('processing_error'), createdAt: timestamp('created_at', { withTimezone: true }).defaultNow().notNull(), diff --git a/src/services/files/docling-client.service.ts b/src/services/files/docling-client.service.ts new file mode 100644 index 0000000..fc5cbb6 --- /dev/null +++ b/src/services/files/docling-client.service.ts @@ -0,0 +1,65 @@ +import { spawn } from 'child_process'; +import * as path from 'path'; +import { fileURLToPath } from 'url'; +import type { ExtractedContent } from '../../types/index.js'; + +export interface DoclingExtractionRequest { + /** Absolute or relative path to the file on disk. */ + storagePath: string; + mimeType: string; +} + +export interface DoclingExtractionResponse { + success: boolean; + extractedContent?: ExtractedContent; + error?: string; +} + +// This file lives at /src/services/files (dev) or /dist/services/files +// (built) — both are three levels under the package root, where python/ sits. +const PACKAGE_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../../../'); +const SCRIPT_PATH = path.join(PACKAGE_ROOT, 'python', 'docling_extract.py'); + +/** + * Docling-backed file extraction. Spawns the Python adapter (`docling_extract.py`) + * and parses its JSON into `ExtractedContent`. The Python command defaults to + * `uv run python` (resolves the pyproject env) and can be overridden via + * `DOCLING_CMD` (space-separated, e.g. "python3"). + * + * Returns `{ success: false }` on any failure so the caller can fall back to the + * Gemini extractor rather than throwing. + */ +export const doclingClientService = { + async extractFromFile(request: DoclingExtractionRequest): Promise { + const parts = (process.env.DOCLING_CMD ?? 'uv run python').split(/\s+/).filter(Boolean); + const cmd = parts[0] ?? 'python3'; + const args = [...parts.slice(1), SCRIPT_PATH, request.storagePath]; + + return new Promise((resolve) => { + let stdout = ''; + let stderr = ''; + // spawn does not throw synchronously for a missing binary — it emits an + // 'error' event, handled below — so no try/catch is needed here. + const child = spawn(cmd, args, { cwd: PACKAGE_ROOT }); + + child.stdout.on('data', (d: Buffer) => (stdout += d.toString())); + child.stderr.on('data', (d: Buffer) => (stderr += d.toString())); + child.on('error', (err: Error) => resolve({ success: false, error: err.message })); + child.on('close', (code: number | null) => { + if (code !== 0) { + resolve({ success: false, error: stderr.trim() || `docling exited with code ${code}` }); + return; + } + try { + const extractedContent = JSON.parse(stdout) as ExtractedContent; + resolve({ success: true, extractedContent }); + } catch (err) { + resolve({ + success: false, + error: `failed to parse docling output: ${err instanceof Error ? err.message : err}`, + }); + } + }); + }); + }, +}; diff --git a/src/services/files/file-processor.service.ts b/src/services/files/file-processor.service.ts index 2297876..134d1a0 100644 --- a/src/services/files/file-processor.service.ts +++ b/src/services/files/file-processor.service.ts @@ -6,6 +6,8 @@ import * as fs from 'fs/promises'; import * as path from 'path'; import type { FileMetadata, ExtractedContent, NodeType } from '../../types/index.js'; import { geminiClientService } from '../llm/gemini-client.service.js'; +import { doclingClientService } from './docling-client.service.js'; +import { structureToMint } from './mint-mapper.js'; import { relationshipInferrerService } from '../knowledge-graph/relationship-inferrer.service.js'; import { AppError } from '../../middleware/error-handler.js'; import { getConfig } from '../../config.js'; @@ -89,32 +91,56 @@ export const fileProcessorService = { throw new AppError('FILE_NOT_FOUND', 'File not found', 404); } - // Read file content - const fileBuffer = await fs.readFile(fileRecord.storagePath); - const base64Content = fileBuffer.toString('base64'); + // Extract content (EXTRACTION ONLY — no reasoning). Docling is the + // default; on any failure we fall back to the Gemini extractor so a + // missing Python/Docling runtime degrades gracefully. + let extractionResult: { success: boolean; extractedContent?: ExtractedContent; error?: string } | undefined; - // Extract content using Gemini (EXTRACTION ONLY) - const extractionResult = await geminiClientService.extractFromFile({ - fileId, - mimeType: fileRecord.mimeType, - fileContent: base64Content, - }); + if ((getConfig().extractor ?? 'docling') === 'docling') { + const docling = await doclingClientService.extractFromFile({ + storagePath: fileRecord.storagePath, + mimeType: fileRecord.mimeType, + }); + if (docling.success) { + extractionResult = docling; + } else { + console.warn(`Docling extraction failed for ${fileId}, falling back to Gemini: ${docling.error}`); + } + } - if (!extractionResult.success) { + if (!extractionResult) { + const fileBuffer = await fs.readFile(fileRecord.storagePath); + const base64Content = fileBuffer.toString('base64'); + extractionResult = await geminiClientService.extractFromFile({ + fileId, + mimeType: fileRecord.mimeType, + fileContent: base64Content, + }); + } + + if (!extractionResult.success || !extractionResult.extractedContent) { throw new Error(extractionResult.error || 'Extraction failed'); } + const extracted = extractionResult.extractedContent; + + // Serialize the document structure to MINT for token-efficient prompt use. + const mintContent = extracted.structure + ? structureToMint(extracted.structure, extracted.metadata) + : null; + // Update file record with extracted content await getDb() .update(files) .set({ - extractedContent: extractionResult.extractedContent, + extractedContent: extracted, + mintContent, processingStatus: 'completed', }) .where(eq(files.id, fileId)); // Create knowledge graph nodes from extracted content - await this.integrateIntoGraph(fileRecord.sessionId, fileId, extractionResult.extractedContent!); + await this.integrateIntoGraph(fileRecord.sessionId, fileId, extracted); } catch (error) { console.error('File processing error:', error); diff --git a/src/services/files/mint-mapper.ts b/src/services/files/mint-mapper.ts new file mode 100644 index 0000000..339936b --- /dev/null +++ b/src/services/files/mint-mapper.ts @@ -0,0 +1,80 @@ +import { + encodeDocument, + type MintDocument, + type MintSection, +} from '@q1k-oss/mint-format'; +import type { DocumentStructure, DocumentMetadata } from '../../types/index.js'; + +/** + * Map an extracted document (Docling/Gemini `DocumentStructure`) onto a + * `MintDocument`. `DocumentStructure` keeps sections / tables / lists / figures + * as parallel arrays, so tables/lists/figures are grouped into trailing + * sections — enough for token-efficient, lossless prompt injection. + */ +export function toMintDocument( + structure: DocumentStructure, + metadata?: DocumentMetadata +): MintDocument { + const sections: MintSection[] = []; + + for (const s of structure.sections ?? []) { + sections.push({ + heading: s.heading, + level: s.level ?? 1, + blocks: s.content ? [{ type: 'text', text: s.content }] : [], + }); + } + + if (structure.tables?.length) { + sections.push({ + heading: 'Tables', + level: 1, + blocks: structure.tables.map((t) => ({ + type: 'table' as const, + table: { caption: t.caption, headers: t.headers, rows: t.rows }, + })), + }); + } + + if (structure.lists?.length) { + sections.push({ + heading: 'Lists', + level: 1, + blocks: structure.lists.map((l) => ({ + type: 'list' as const, + list: { ordered: l.type === 'ordered', items: l.items }, + })), + }); + } + + if (structure.figures?.length) { + sections.push({ + heading: 'Figures', + level: 1, + blocks: structure.figures.map((f) => ({ + type: 'figure' as const, + figure: { caption: f.caption, page: f.pageRef }, + })), + }); + } + + const meta: Record = {}; + if (metadata?.pageCount != null) meta.pages = metadata.pageCount; + if (metadata?.wordCount != null) meta.words = metadata.wordCount; + if (metadata?.author) meta.author = metadata.author; + if (metadata?.language) meta.language = metadata.language; + + return { + title: structure.title, + metadata: Object.keys(meta).length ? meta : undefined, + sections, + }; +} + +/** Convenience: structure → MINT string. */ +export function structureToMint( + structure: DocumentStructure, + metadata?: DocumentMetadata +): string { + return encodeDocument(toMintDocument(structure, metadata)); +} diff --git a/src/types/file.types.ts b/src/types/file.types.ts index 1b34a99..f07d1db 100644 --- a/src/types/file.types.ts +++ b/src/types/file.types.ts @@ -48,6 +48,8 @@ export interface DocumentStructure { heading: string; content: string; level?: number; + /** 1-based page the section starts on (Docling provenance), if known. */ + pageRef?: number; }>; tables?: Array<{ caption?: string; @@ -58,6 +60,11 @@ export interface DocumentStructure { type: 'ordered' | 'unordered'; items: string[]; }>; + /** Figures/images extracted from the document (Docling pictures). */ + figures?: Array<{ + caption?: string; + pageRef?: number; + }>; } export interface ExtractedEntities { From 9376d51f37b32b919c22c4562ec6f40b30ceee62 Mon Sep 17 00:00:00 2001 From: Himanshu Shekhar Date: Fri, 29 May 2026 04:16:40 +0530 Subject: [PATCH 2/2] ci: add minimal CI (npm ci + build) on PRs --- .github/workflows/ci.yml | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) create mode 100644 .github/workflows/ci.yml diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..056fa1a --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,21 @@ +name: CI + +on: + pull_request: + push: + branches: [main] + +jobs: + build: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-node@v4 + with: + node-version: 20 + registry-url: https://registry.npmjs.org + + - run: npm ci + + - run: npm run build