@@ -724,14 +724,117 @@ def _resolve_description(fm: dict) -> str:
724724 return ""
725725
726726
727- def _read_concept_briefs (wiki_dir : Path ) -> str :
727+ DEFAULT_BRIEFS_BUDGET_CHARS = 120_000
728+ """Per-list character cap for the concept/entity briefs in the plan prompt.
729+
730+ Roughly 30k tokens each, so a KB has to grow well past a few hundred pages
731+ before anything is trimmed — the cap is a ceiling against unbounded growth, not
732+ a working limit. Overridable with the ``briefs_budget_chars`` config key; set it
733+ to 0 to restore the previous uncapped behavior.
734+
735+ Why a cap at all: ``_read_concept_briefs`` and ``_read_entity_briefs`` read
736+ *every* page in the KB and the result is rebuilt into the plan prompt for each
737+ compiled document. The size is therefore O(KB), and a large KB eventually
738+ exceeds the model's context window — which surfaces as a hard failure partway
739+ through a recompile, not as degraded output.
740+ """
741+
742+
743+ def _resolve_briefs_budget (config : dict ) -> int :
744+ """Read ``briefs_budget_chars`` from config, falling back to the default.
745+
746+ A non-integer or negative value falls back rather than raising: a typo in a
747+ config file should not abort a compile, and a negative budget has no sane
748+ reading. ``0`` is meaningful and passes through — it disables trimming.
749+ """
750+ raw = config .get ("briefs_budget_chars" , DEFAULT_BRIEFS_BUDGET_CHARS )
751+ if isinstance (raw , bool ) or not isinstance (raw , int ) or raw < 0 :
752+ logger .warning (
753+ "briefs_budget_chars=%r is not a non-negative integer — using %d" ,
754+ raw ,
755+ DEFAULT_BRIEFS_BUDGET_CHARS ,
756+ )
757+ return DEFAULT_BRIEFS_BUDGET_CHARS
758+ return raw
759+
760+
761+ def _n_sources (fm_dict : dict ) -> int :
762+ """How many documents a page cites, from its ``sources:`` frontmatter list.
763+
764+ This is the cross-document recurrence signal already used in the entity
765+ brief line. It doubles as the salience ranking when the brief list has to
766+ be trimmed to fit a budget: a concept seen in many documents is the one the
767+ planner most needs to know about, because it is the one most likely to be
768+ updated rather than created.
769+ """
770+ sources = fm_dict .get ("sources" )
771+ return len (sources ) if isinstance (sources , list ) else 0
772+
773+
774+ def _fit_briefs (ranked : list [tuple [int , str ]], budget_chars : int , noun : str ) -> str :
775+ """Join brief lines under a character budget, ranked by salience.
776+
777+ ``ranked`` is ``[(n_sources, line), ...]``. Lines are emitted most-cited
778+ first, then alphabetically (the caller sorts), until the budget is spent.
779+
780+ **The truncation is announced in the returned text.** A shortened list that
781+ looks complete is worse than a long one: the planner reads these briefs to
782+ decide create-vs-update, so a silently dropped concept comes back as a
783+ duplicate page for something the KB already has. The trailing marker tells
784+ the model the list is partial, so "not listed" stops meaning "not present".
785+
786+ A budget of 0 or less disables trimming, which keeps the previous behavior
787+ available for callers that want the full list.
788+ """
789+ if budget_chars <= 0 :
790+ return "\n " .join (line for _ , line in ranked ) or "(none yet)"
791+
792+ kept : list [str ] = []
793+ used = 0
794+ for _ , line in ranked :
795+ if used + len (line ) + 1 > budget_chars :
796+ break
797+ kept .append (line )
798+ used += len (line ) + 1
799+
800+ dropped = len (ranked ) - len (kept )
801+ if not dropped :
802+ return "\n " .join (kept ) or "(none yet)"
803+ if not kept :
804+ # Budget smaller than a single line. Say so rather than return an empty
805+ # list that reads like an empty KB.
806+ return f"(list omitted: { len (ranked )} { noun } exceed the brief budget)"
807+
808+ logger .info (
809+ "brief list trimmed to fit budget: kept %d of %d %s (%d chars)" ,
810+ len (kept ),
811+ len (ranked ),
812+ noun ,
813+ budget_chars ,
814+ )
815+ kept .append (
816+ f"- (… { dropped } more { noun } not listed — this list is truncated to the "
817+ f"{ len (kept ) - 1 } most-cited; absence here does NOT mean the page is "
818+ f"missing, so prefer 'update' over 'create' when unsure)"
819+ )
820+ return "\n " .join (kept )
821+
822+
823+ def _read_concept_briefs (wiki_dir : Path , budget_chars : int = 0 ) -> str :
728824 """Read existing concept pages and return compact one-line summaries.
729825
730826 For each concept, reads the ``description:`` field (falling back to legacy
731827 ``brief:``) from YAML frontmatter if present; otherwise falls back to
732828 truncating the first 150 chars of the body (newlines collapsed to spaces).
733829 Formats each as ``- {slug}: {description}``.
734830
831+ With ``budget_chars`` > 0 the list is capped at that many characters, most-
832+ cited concepts first (see :func:`_fit_briefs`). The cap exists because this
833+ list grows with the KB and lands in the plan prompt on every compiled
834+ document: an unbounded list eventually exceeds the model's context window,
835+ and the failure arrives as a hard error mid-compile rather than as degraded
836+ output.
837+
735838 Returns "(none yet)" if the concepts directory is missing or empty.
736839 """
737840 concepts_dir = wiki_dir / "concepts"
@@ -742,7 +845,7 @@ def _read_concept_briefs(wiki_dir: Path) -> str:
742845 if not md_files :
743846 return "(none yet)"
744847
745- lines : list [str ] = []
848+ ranked : list [tuple [ int , str ] ] = []
746849 for path in md_files :
747850 text = path .read_text (encoding = "utf-8" )
748851 fm_dict = frontmatter .parse (text )
@@ -752,17 +855,22 @@ def _read_concept_briefs(wiki_dir: Path) -> str:
752855 body = parts [1 ] if parts is not None else text
753856 brief = body .strip ().replace ("\n " , " " )[:150 ]
754857 if brief :
755- lines .append (f"- { path .stem } : { brief } " )
858+ ranked .append (( _n_sources ( fm_dict ), f"- { path .stem } : { brief } " ) )
756859
757- return "\n " .join (lines ) or "(none yet)"
860+ # Most-cited first; alphabetical within a tier so the prompt stays stable
861+ # across runs (prompt caching depends on byte-identical prefixes).
862+ ranked .sort (key = lambda item : (- item [0 ], item [1 ]))
863+ return _fit_briefs (ranked , budget_chars , "concepts" )
758864
759865
760- def _read_entity_briefs (wiki_dir : Path ) -> str :
866+ def _read_entity_briefs (wiki_dir : Path , budget_chars : int = 0 ) -> str :
761867 """Read existing entity pages as compact lines for the plan call.
762868
763869 Formats each as ``- {slug} ({type}, {n} sources) — {brief}``. The source
764870 count is the cross-document recurrence signal the LLM uses to decide
765- create-vs-update and salience. Returns "(none yet)" when empty.
871+ create-vs-update and salience — and, with ``budget_chars`` > 0, the ranking
872+ used to decide what stays when the list is capped. Returns "(none yet)"
873+ when empty.
766874 """
767875 entities_dir = wiki_dir / "entities"
768876 if not entities_dir .exists ():
@@ -772,21 +880,22 @@ def _read_entity_briefs(wiki_dir: Path) -> str:
772880 if not md_files :
773881 return "(none yet)"
774882
775- lines : list [str ] = []
883+ ranked : list [tuple [ int , str ] ] = []
776884 for path in md_files :
777885 text = path .read_text (encoding = "utf-8" )
778886 fm_dict = frontmatter .parse (text )
779887 brief = _resolve_description (fm_dict )
780888 etype = str (fm_dict .get ("type" ) or "" ).strip ().lower () or "other"
781- n_sources = len (fm_dict [ "sources" ]) if isinstance ( fm_dict . get ( "sources" ), list ) else 0
889+ n_sources = _n_sources (fm_dict )
782890 if not brief :
783891 parts = frontmatter .split (text )
784892 body = parts [1 ] if parts is not None else text
785893 brief = body .strip ().replace ("\n " , " " )[:150 ]
786894 suffix = f" — { brief } " if brief else ""
787- lines .append (f"- { path .stem } ({ etype } , { n_sources } sources){ suffix } " )
895+ ranked .append (( n_sources , f"- { path .stem } ({ etype } , { n_sources } sources){ suffix } " ) )
788896
789- return "\n " .join (lines ) or "(none yet)"
897+ ranked .sort (key = lambda item : (- item [0 ], item [1 ]))
898+ return _fit_briefs (ranked , budget_chars , "entities" )
790899
791900
792901def _iter_h2_headings (lines : list [str ]) -> list [tuple [int , str ]]:
@@ -1603,6 +1712,7 @@ async def _compile_concepts(
16031712 doc_type : str = "short" ,
16041713 rewrite_summary : bool = False ,
16051714 entity_types : list [str ] | None = None ,
1715+ briefs_budget_chars : int | None = None ,
16061716 bundle = None ,
16071717) -> None :
16081718 """Shared Steps 2-4: concepts plan → generate/update → index.
@@ -1624,8 +1734,14 @@ async def _compile_concepts(
16241734 valid_types = frozenset (entity_types )
16251735
16261736 # --- Step 2: Get concepts plan (A cached) ---
1627- concept_briefs = _read_concept_briefs (wiki_dir )
1628- entity_briefs = _read_entity_briefs (wiki_dir )
1737+ # Both lists grow with the KB and are rebuilt into the plan prompt for every
1738+ # compiled document, so their combined size is what eventually pushes this
1739+ # call past the model's context window. The budget caps each one; see
1740+ # ``_fit_briefs`` for why the trim is announced rather than silent.
1741+ if briefs_budget_chars is None :
1742+ briefs_budget_chars = DEFAULT_BRIEFS_BUDGET_CHARS
1743+ concept_briefs = _read_concept_briefs (wiki_dir , briefs_budget_chars )
1744+ entity_briefs = _read_entity_briefs (wiki_dir , briefs_budget_chars )
16291745
16301746 # Second cache breakpoint: end of the assistant summary message. Covers
16311747 # (system + doc + summary) for the plan call and every concept call.
@@ -2220,6 +2336,7 @@ async def compile_short_doc(
22202336 config = resolve_effective_config (kb_dir )[0 ]
22212337 language : str = config .get ("language" , "en" )
22222338 entity_types = resolve_entity_types (config )
2339+ briefs_budget = _resolve_briefs_budget (config )
22232340
22242341 wiki_dir = kb_dir / "wiki"
22252342 schema_md = get_agents_md (wiki_dir )
@@ -2280,6 +2397,7 @@ async def compile_short_doc(
22802397 doc_type = "short" ,
22812398 rewrite_summary = True ,
22822399 entity_types = entity_types ,
2400+ briefs_budget_chars = briefs_budget ,
22832401 bundle = bundle ,
22842402 )
22852403 finally :
@@ -2308,6 +2426,7 @@ async def compile_long_doc(
23082426 config = resolve_effective_config (kb_dir )[0 ]
23092427 language : str = config .get ("language" , "en" )
23102428 entity_types = resolve_entity_types (config )
2429+ briefs_budget = _resolve_briefs_budget (config )
23112430
23122431 wiki_dir = kb_dir / "wiki"
23132432 schema_md = get_agents_md (wiki_dir )
@@ -2364,6 +2483,7 @@ async def compile_long_doc(
23642483 doc_brief = doc_description ,
23652484 doc_type = "pageindex" ,
23662485 entity_types = entity_types ,
2486+ briefs_budget_chars = briefs_budget ,
23672487 bundle = bundle ,
23682488 )
23692489 finally :
0 commit comments