|
9 | 9 |
|
10 | 10 | import contextlib |
11 | 11 | import json as _json |
| 12 | +from dataclasses import dataclass |
12 | 13 | from pathlib import Path, PurePosixPath |
| 14 | +from typing import Literal |
13 | 15 |
|
| 16 | +from openkb import frontmatter |
14 | 17 | from openkb.locks import atomic_write_text |
15 | 18 |
|
| 19 | +# Maps a taxonomy "kind" to its wiki subdirectory. Single source of truth for |
| 20 | +# list_taxonomy_items/get_taxonomy_item below. |
| 21 | +_TAXONOMY_DIRS: dict[str, str] = {"concept": "concepts", "entity": "entities"} |
| 22 | + |
16 | 23 |
|
17 | 24 | def list_wiki_files(directory: str, wiki_root: str) -> str: |
18 | 25 | """List all Markdown files in a wiki subdirectory. |
@@ -135,6 +142,103 @@ def get_wiki_page_content(doc_name: str, pages: str, wiki_root: str) -> str: |
135 | 142 | return "\n\n".join(parts) + "\n\n" |
136 | 143 |
|
137 | 144 |
|
| 145 | +@dataclass(frozen=True) |
| 146 | +class TaxonomyItem: |
| 147 | + """One persisted concept or entity page (never a pending candidate). |
| 148 | +
|
| 149 | + ``PendingTopicsStore`` (see ``openkb.pending``) buffers not-yet-paged |
| 150 | + concept/entity candidates separately from the compiled ``.md`` pages |
| 151 | + under ``concepts/``/``entities/`` — this dataclass, and |
| 152 | + :func:`list_taxonomy_items`, only ever surface the latter, so a caller |
| 153 | + never sees an in-progress candidate as if it were a real page. |
| 154 | + """ |
| 155 | + |
| 156 | + kind: Literal["concept", "entity"] |
| 157 | + slug: str |
| 158 | + path: str # wiki-root-relative, e.g. "concepts/attention.md" |
| 159 | + brief: str |
| 160 | + # Entity type (e.g. "person", "organization"); always None for concepts. |
| 161 | + type: str | None = None |
| 162 | + |
| 163 | + |
| 164 | +def list_taxonomy_items(wiki_root: str, kind: str | None = None) -> list[TaxonomyItem]: |
| 165 | + """List persisted concept and/or entity pages with their one-line briefs. |
| 166 | +
|
| 167 | + Intended as the first step of the search strategy: browse this compact, |
| 168 | + semantically-scannable list and let the caller (an LLM) pick the |
| 169 | + relevant slug(s) by meaning — this is deliberately not a keyword search |
| 170 | + (see ``search_wiki`` for that, over summaries/sources only). |
| 171 | +
|
| 172 | + Args: |
| 173 | + wiki_root: Absolute path to the wiki root directory. |
| 174 | + kind: Restrict to ``"concept"`` or ``"entity"``; ``None`` returns both. |
| 175 | +
|
| 176 | + Returns: |
| 177 | + Items sorted by kind, then slug. Empty list if the KB has neither |
| 178 | + directory yet or both are empty. |
| 179 | +
|
| 180 | + Raises: |
| 181 | + ValueError: *kind* is neither ``None``, ``"concept"``, nor ``"entity"``. |
| 182 | + """ |
| 183 | + root = Path(wiki_root).resolve() |
| 184 | + kinds = [kind] if kind else ["concept", "entity"] |
| 185 | + for k in kinds: |
| 186 | + if k not in _TAXONOMY_DIRS: |
| 187 | + raise ValueError(f"Unknown kind {k!r}; expected 'concept' or 'entity'.") |
| 188 | + |
| 189 | + items: list[TaxonomyItem] = [] |
| 190 | + for k in kinds: |
| 191 | + directory = root / _TAXONOMY_DIRS[k] |
| 192 | + if not directory.is_dir(): |
| 193 | + continue |
| 194 | + for md_file in sorted(directory.glob("*.md")): |
| 195 | + text = md_file.read_text(encoding="utf-8") |
| 196 | + fm = frontmatter.parse(text) |
| 197 | + brief = frontmatter.resolve_description(fm) |
| 198 | + etype = None |
| 199 | + if k == "entity": |
| 200 | + etype = str(fm.get("type") or "").strip().lower() or "other" |
| 201 | + items.append( |
| 202 | + TaxonomyItem( |
| 203 | + kind=k, # type: ignore[arg-type] # validated against _TAXONOMY_DIRS above |
| 204 | + slug=md_file.stem, |
| 205 | + path=f"{_TAXONOMY_DIRS[k]}/{md_file.name}", |
| 206 | + brief=brief, |
| 207 | + type=etype, |
| 208 | + ) |
| 209 | + ) |
| 210 | + return items |
| 211 | + |
| 212 | + |
| 213 | +def get_taxonomy_item(slug: str, wiki_root: str, kind: str | None = None) -> str: |
| 214 | + """Read a persisted concept or entity page's full Markdown content. |
| 215 | +
|
| 216 | + Args: |
| 217 | + slug: Page slug (filename without ``.md``), e.g. ``"attention"``. |
| 218 | + wiki_root: Absolute path to the wiki root directory. |
| 219 | + kind: ``"concept"`` or ``"entity"`` to disambiguate a same-named |
| 220 | + slug; ``None`` checks ``concepts/`` first, then ``entities/``. |
| 221 | +
|
| 222 | + Returns: |
| 223 | + Full file content, or a "not found" message if no match exists in |
| 224 | + the requested (or either) directory. |
| 225 | +
|
| 226 | + Raises: |
| 227 | + ValueError: *kind* is neither ``None``, ``"concept"``, nor ``"entity"``. |
| 228 | + """ |
| 229 | + root = Path(wiki_root).resolve() |
| 230 | + kinds = [kind] if kind else ["concept", "entity"] |
| 231 | + for k in kinds: |
| 232 | + if k not in _TAXONOMY_DIRS: |
| 233 | + raise ValueError(f"Unknown kind {k!r}; expected 'concept' or 'entity'.") |
| 234 | + |
| 235 | + for k in kinds: |
| 236 | + path = (root / _TAXONOMY_DIRS[k] / f"{slug}.md").resolve() |
| 237 | + if path.is_relative_to(root) and path.exists(): |
| 238 | + return path.read_text(encoding="utf-8") |
| 239 | + return f"Taxonomy item not found: {slug}" |
| 240 | + |
| 241 | + |
138 | 242 | def search_wiki(query: str, wiki_root: str, top_k: int = 5) -> str: |
139 | 243 | """Full-text (BM25) search over concepts/entities/summaries wiki pages. |
140 | 244 |
|
|
0 commit comments