Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
21 changes: 21 additions & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
name: CI

on:
pull_request:
push:
branches: [main]

jobs:
build:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4

- uses: actions/setup-node@v4
with:
node-version: 20
registry-url: https://registry.npmjs.org

- run: npm ci

- run: npm run build
50 changes: 50 additions & 0 deletions CLAUDE.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,50 @@
# CLAUDE.md

Guidance for Claude Code when working in the **context-engine** repo.

## What this is

`@q1k-oss/context-engine` — a TypeScript SDK + HTTP service for the customer-knowledge layer: file ingestion, knowledge-graph extraction, and prioritized-context retrieval for LLM prompts. It is consumed by `q1k-controlplane` as a vendored npm dependency (with a patch). Initialize with `initContextEngine({ databaseUrl, ... })` before using any service.

## Commands

```bash
npm run dev # tsx watch src/server.ts
npm run build # tsc -> dist/
npm run start # node dist/server.js
npm run db:push # apply schema to the DB (this repo uses push, not migrations)
npm run db:studio
```

## File extraction (upload → knowledge graph)

`POST /api/files/upload` → `fileProcessorService.saveFile` (writes to `uploadDir`, default `./uploads`) → async `processFile`:

1. **Extract** via the configured extractor:
- `extractor: 'docling'` (default) — `doclingClientService` spawns `python/docling_extract.py` (Docling). On any failure it **falls back to Gemini**, so a missing Python/Docling runtime degrades gracefully.
- `extractor: 'gemini'` — `geminiClientService` (extraction only, never reasoning).
Both return the same `ExtractedContent` shape (`src/types/file.types.ts`).
2. **Serialize to MINT** — `structureToMint` (`src/services/files/mint-mapper.ts`) encodes the document `structure` via `@q1k-oss/mint-format`'s `encodeDocument` and stores it in `files.mint_content` for token-efficient prompt injection.
3. **Integrate into the graph** — `integrateIntoGraph` creates `Artifact` + `Entity` nodes/edges.

### Docling (Python) setup

Docling is a Python dependency (`pyproject.toml`), not bundled with the JS package. To run the default extractor locally:

```bash
uv sync # installs Docling (first run pulls ML models, ~hundreds of MB)
npx tsx scripts/smoke-docling.ts # MINT-mapping smoke (no Docling needed)
npx tsx scripts/smoke-docling.ts file.pdf # real Docling extraction on a file
```

Override the Python command with `DOCLING_CMD` (default `uv run python`). The script must emit valid JSON to stdout or exit non-zero (the TS caller falls back to Gemini on failure).

## MINT usage

MINT (`@q1k-oss/mint-format`) is used to pack data into LLM prompts token-efficiently: graph nodes/edges (`claude-client.service.ts`, `encode`) and now parsed documents (`encodeDocument`, see the mapper above). Requires `@q1k-oss/mint-format` ≥ 1.1.0 (adds `encodeDocument`/`decodeDocument`).

## Conventions

- ESM, NodeNext module resolution — **import with `.js` extensions** even from `.ts`.
- Schema in `src/db/schema/*.ts`; apply with `db:push`.
- HTTP routes currently use a trusted-caller pattern (no JWT yet) — q1k-auth middleware is a planned hardening item.
2 changes: 1 addition & 1 deletion package.json
Original file line number Diff line number Diff line change
Expand Up @@ -86,7 +86,7 @@
"dependencies": {
"@anthropic-ai/sdk": "^0.39.0",
"@google/generative-ai": "^0.21.0",
"@q1k-oss/mint-format": "^1.0.2",
"@q1k-oss/mint-format": "^1.1.0",
"cors": "^2.8.5",
"drizzle-orm": "^0.38.3",
"express": "^4.21.2",
Expand Down
11 changes: 11 additions & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,11 @@
[project]
name = "context-engine-docling"
version = "0.1.0"
description = "Docling extraction adapter for context-engine file ingestion"
requires-python = ">=3.11"
dependencies = [
"docling>=2.0.0",
]

[tool.uv]
package = false
134 changes: 134 additions & 0 deletions python/docling_extract.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,134 @@
#!/usr/bin/env python3
"""Docling extraction adapter for context-engine.

Reads a single file path (argv[1]), parses it with Docling, and prints a JSON
object to stdout matching the TypeScript `ExtractedContent` shape:

{
"textContent": str,
"structure": {
"title": str | None,
"sections": [{"heading": str, "content": str, "level": int, "pageRef": int | None}],
"tables": [{"caption": str | None, "headers": [str], "rows": [[str]]}],
"lists": [{"type": "ordered" | "unordered", "items": [str]}],
"figures": [{"caption": str | None, "pageRef": int | None}]
},
"metadata": {"pageCount": int, "wordCount": int, "characterCount": int}
}

The TS caller (`docling-client.service.ts`) spawns this and falls back to the
Gemini extractor on a non-zero exit, so this script must fail loudly (exit 1 +
stderr) rather than emit partial JSON.

Run: `uv run python python/docling_extract.py <file>`
"""

import json
import sys


def _page_of(item) -> "int | None":
prov = getattr(item, "prov", None)
if prov:
first = prov[0] if isinstance(prov, (list, tuple)) else prov
return getattr(first, "page_no", None)
return None


def extract(path: str) -> dict:
from docling.document_converter import DocumentConverter

doc = DocumentConverter().convert(path).document

text_content = doc.export_to_markdown()

sections: list[dict] = []
lists: list[dict] = []
current_list: "dict | None" = None

for item in getattr(doc, "texts", []) or []:
label = str(getattr(item, "label", "")).lower()
text = (getattr(item, "text", "") or "").strip()
if not text:
continue

if "list_item" in label:
if current_list is None:
current_list = {"type": "unordered", "items": []}
lists.append(current_list)
current_list["items"].append(text)
continue
current_list = None # any non-list item closes the run

if "section_header" in label or "title" in label:
sections.append(
{
"heading": text,
"content": "",
"level": int(getattr(item, "level", 1) or 1),
"pageRef": _page_of(item),
}
)
else:
if not sections:
sections.append({"heading": "", "content": "", "level": 1, "pageRef": _page_of(item)})
sections[-1]["content"] = (sections[-1]["content"] + "\n" + text).strip()

tables: list[dict] = []
for tbl in getattr(doc, "tables", []) or []:
try:
df = tbl.export_to_dataframe()
headers = [str(c) for c in df.columns.tolist()]
rows = [[("" if v is None else str(v)) for v in row] for row in df.values.tolist()]
except Exception:
headers, rows = [], []
caption = None
try:
caption = tbl.caption_text(doc) or None
except Exception:
pass
tables.append({"caption": caption, "headers": headers, "rows": rows})

figures: list[dict] = []
for pic in getattr(doc, "pictures", []) or []:
caption = None
try:
caption = pic.caption_text(doc) or None
except Exception:
pass
figures.append({"caption": caption, "pageRef": _page_of(pic)})

title = sections[0]["heading"] if sections and sections[0]["heading"] else None

return {
"textContent": text_content,
"structure": {
"title": title,
"sections": sections,
"tables": tables,
"lists": lists,
"figures": figures,
},
"metadata": {
"pageCount": len(getattr(doc, "pages", []) or []),
"wordCount": len(text_content.split()),
"characterCount": len(text_content),
},
}


def main() -> int:
if len(sys.argv) < 2:
print("usage: docling_extract.py <file>", file=sys.stderr)
return 2
try:
result = extract(sys.argv[1])
except Exception as exc: # noqa: BLE001 — surface any failure to the TS caller
print(f"docling extraction failed: {exc}", file=sys.stderr)
return 1
json.dump(result, sys.stdout, ensure_ascii=False)
return 0


if __name__ == "__main__":
sys.exit(main())
61 changes: 61 additions & 0 deletions scripts/smoke-docling.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,61 @@
/**
* Smoke test for the Docling → MINT path.
*
* Always runs the deterministic MINT-mapping check (no Docling needed):
* npx tsx scripts/smoke-docling.ts
*
* Optionally runs the real Docling extractor on a file (requires `uv sync`
* to have installed Docling + first-run model download):
* npx tsx scripts/smoke-docling.ts path/to/sample.pdf
*/
import { decodeDocument } from '@q1k-oss/mint-format';
import { structureToMint } from '../src/services/files/mint-mapper.js';
import { doclingClientService } from '../src/services/files/docling-client.service.js';
import type { DocumentStructure } from '../src/types/index.js';

const sample: DocumentStructure = {
title: 'Sample SOP',
sections: [
{ heading: 'Overview', content: 'Covers returns within 30 days.', level: 1, pageRef: 1 },
{ heading: 'Eligibility', content: 'Items must be unused.', level: 2, pageRef: 2 },
],
tables: [{ caption: 'Windows', headers: ['Category', 'Days'], rows: [['Apparel', '30']] }],
lists: [{ type: 'ordered', items: ['Inspect item', 'Issue refund'] }],
figures: [{ caption: 'Refund flow', pageRef: 2 }],
};

function assert(cond: boolean, msg: string): void {
if (!cond) throw new Error(`SMOKE FAIL: ${msg}`);
}

async function main(): Promise<void> {
const mint = structureToMint(sample, { pageCount: 4, author: 'Acme' });
console.log('--- MINT ---\n' + mint + '\n');

assert(mint.includes('§1 Overview'), 'heading not rendered');
assert(mint.includes('| Category | Days |'), 'table not rendered');
assert(mint.includes('1. Inspect item'), 'ordered list not rendered');
assert(mint.includes('@fig:'), 'figure not rendered');

const decoded = decodeDocument(mint);
assert(decoded.sections.length >= 2, 'sections lost on decode round-trip');
console.log(`MINT mapping OK — ${decoded.sections.length} sections round-tripped.\n`);

const file = process.argv[2];
if (file) {
console.log(`Running Docling on ${file} ...`);
const r = await doclingClientService.extractFromFile({ storagePath: file, mimeType: 'application/pdf' });
if (!r.success) {
console.log(`Docling failed (expected if not installed): ${r.error}`);
return;
}
const s = r.extractedContent?.structure;
assert(!!s, 'docling returned no structure');
console.log('Docling OK. MINT preview:\n' + structureToMint(s!, r.extractedContent?.metadata).slice(0, 500));
}
}

main().catch((err) => {
console.error(err);
process.exit(1);
});
2 changes: 2 additions & 0 deletions src/config.ts
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,8 @@ export interface ContextEngineConfig {
claudeModel?: string;
/** Gemini model for file extraction (default: 'gemini-2.0-flash-exp') */
geminiModel?: string;
/** File extractor: 'docling' (Python subprocess, default) or 'gemini'. */
extractor?: 'docling' | 'gemini';
/** PostgreSQL connection pool size (default: 10) */
dbPoolSize?: number;
/** PostgreSQL connection idle timeout in seconds (default: 30) */
Expand Down
2 changes: 2 additions & 0 deletions src/db/schema/files.ts
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,8 @@ export const files = pgTable('files', {
fileSize: integer('file_size').notNull(),
storagePath: text('storage_path').notNull(),
extractedContent: jsonb('extracted_content'),
// MINT-serialized document (encodeDocument output) for token-efficient prompt injection.
mintContent: text('mint_content'),
processingStatus: varchar('processing_status', { length: 20 }).default('pending').notNull(),
processingError: text('processing_error'),
createdAt: timestamp('created_at', { withTimezone: true }).defaultNow().notNull(),
Expand Down
65 changes: 65 additions & 0 deletions src/services/files/docling-client.service.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,65 @@
import { spawn } from 'child_process';
import * as path from 'path';
import { fileURLToPath } from 'url';
import type { ExtractedContent } from '../../types/index.js';

export interface DoclingExtractionRequest {
/** Absolute or relative path to the file on disk. */
storagePath: string;
mimeType: string;
}

export interface DoclingExtractionResponse {
success: boolean;
extractedContent?: ExtractedContent;
error?: string;
}

// This file lives at <pkg>/src/services/files (dev) or <pkg>/dist/services/files
// (built) — both are three levels under the package root, where python/ sits.
const PACKAGE_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../../../');
const SCRIPT_PATH = path.join(PACKAGE_ROOT, 'python', 'docling_extract.py');

/**
* Docling-backed file extraction. Spawns the Python adapter (`docling_extract.py`)
* and parses its JSON into `ExtractedContent`. The Python command defaults to
* `uv run python` (resolves the pyproject env) and can be overridden via
* `DOCLING_CMD` (space-separated, e.g. "python3").
*
* Returns `{ success: false }` on any failure so the caller can fall back to the
* Gemini extractor rather than throwing.
*/
export const doclingClientService = {
async extractFromFile(request: DoclingExtractionRequest): Promise<DoclingExtractionResponse> {
const parts = (process.env.DOCLING_CMD ?? 'uv run python').split(/\s+/).filter(Boolean);
const cmd = parts[0] ?? 'python3';
const args = [...parts.slice(1), SCRIPT_PATH, request.storagePath];

return new Promise<DoclingExtractionResponse>((resolve) => {
let stdout = '';
let stderr = '';
// spawn does not throw synchronously for a missing binary — it emits an
// 'error' event, handled below — so no try/catch is needed here.
const child = spawn(cmd, args, { cwd: PACKAGE_ROOT });

child.stdout.on('data', (d: Buffer) => (stdout += d.toString()));
child.stderr.on('data', (d: Buffer) => (stderr += d.toString()));
child.on('error', (err: Error) => resolve({ success: false, error: err.message }));
child.on('close', (code: number | null) => {
if (code !== 0) {
resolve({ success: false, error: stderr.trim() || `docling exited with code ${code}` });
return;
}
try {
const extractedContent = JSON.parse(stdout) as ExtractedContent;
resolve({ success: true, extractedContent });
} catch (err) {
resolve({
success: false,
error: `failed to parse docling output: ${err instanceof Error ? err.message : err}`,
});
}
});
});
},
};
Loading
Loading