From bba1f03231f616dda5fd1422186157adf953d592 Mon Sep 17 00:00:00 2001 From: Riyan Dhiman Date: Thu, 20 Aug 2026 00:05:25 +0530 Subject: [PATCH 1/7] =?UTF-8?q?flow:=20harden=20UAF/DF=20capping=20?= =?UTF-8?q?=E2=80=94=20sticky=20widening,=20lead=20dedup,=20unplaced-USE?= =?UTF-8?q?=20tolerance?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit object_state: make widening sticky per successor (collapse once past the disjunct cap or already-widened, re-queue only on key-set change) so the worklist converges instead of oscillating on disjunct-heavy functions. object_lifetime: dedup leads to one per (pattern, root+suffix) keeping the earliest line plus a site count; treat a dropped USE op as false-negative-safe (only state-changing unplaced ops mark a function untrustworthy) so an unresolved USE no longer suppresses the whole object flow — recovers the openjpeg image double-free the earlier suppression dropped. --- lachesis/flow/object_lifetime.py | 35 +++++++++++++++++++++++++------- lachesis/flow/object_state.py | 23 ++++++++++++++------- 2 files changed, 44 insertions(+), 14 deletions(-) diff --git a/lachesis/flow/object_lifetime.py b/lachesis/flow/object_lifetime.py index 3092b7c6..0a51519e 100644 --- a/lachesis/flow/object_lifetime.py +++ b/lachesis/flow/object_lifetime.py @@ -574,21 +574,42 @@ def analyze_object_lifetimes(store, functions, call_successors, *, lang="c", gra # second time just to collect the same local findings. diagnostics["analyzed"] += 1 diagnostics["unplaced"] += len(result.unplaced) - if result.unplaced: - diagnostics["unplaced_functions"][name] = len(result.unplaced) + # Only a dropped *state-changing* op (a free/alloc/reset we could not place) + # leaves the summary untrustworthy: it may hide a free (missed double-free in a + # caller) or a reset (a live object we would wrongly call freed). A dropped USE + # can only cost a read — a false negative — so it must not mark the function + # unsafe and poison every caller that passes an object through it. + state_changing = sum(1 for op in result.unplaced if op.kind is not OpKind.USE) + if state_changing: + diagnostics["unplaced_functions"][name] = state_changing diagnostics["widenings"] += result.widenings diagnostics["transfers"] += result.transfers if result.capped: diagnostics["capped"].append(name) + # One lead per (object, pattern): a loop-carried free floods the freed state + # around the back-edge, so the same bug otherwise surfaces at every later + # free/use of that object (187 findings for one image double-free in a decoder + # main). Collapse to the earliest site -- the representative, root-cause-nearest + # occurrence -- and keep a count so the volume is visible without the noise. + best: dict[tuple, dict] = {} for finding in sorted(result.findings): root_id = finding.path.root.removeprefix("decl:") root = sub.label(root_id) or root_id suffix = "".join(finding.path.selectors) - leads.append({ - "pattern": finding.pattern, "var": root + suffix, "root": root, - "entry": name, "line": finding.line, "node": finding.node, - "engine": "object-identity", - }) + key = (finding.pattern, root + suffix) + existing = best.get(key) + if existing is None: + best[key] = { + "pattern": finding.pattern, "var": root + suffix, "root": root, + "entry": name, "line": finding.line, "node": finding.node, + "engine": "object-identity", "sites": 1, + } + else: + existing["sites"] += 1 + if finding.line is not None and ( + existing["line"] is None or finding.line < existing["line"]): + existing["line"], existing["node"] = finding.line, finding.node + leads.extend(best[key] for key in sorted(best)) # A function is *seed*-unsafe when its own analysis is untrustworthy (no CFG, # capped worklist, dropped ops). That set propagates UP the call graph, because a diff --git a/lachesis/flow/object_state.py b/lachesis/flow/object_state.py index 95287fa5..be54edd7 100644 --- a/lachesis/flow/object_state.py +++ b/lachesis/flow/object_state.py @@ -424,6 +424,7 @@ def analyze( incoming[nodes[0]][seed.key()] = seed work = deque([nodes[0]]) queued = {nodes[0]} + widened: set[Hashable] = set() findings: set[Finding] = set() transfers = widenings = 0 cap = self.transfer_cap or max(10000, len(nodes) * 500) @@ -436,17 +437,25 @@ def analyze( for successor in successors.get(node, ()): if successor not in incoming: continue - changed = False + before = set(incoming[successor]) for key, state in outgoing.items(): - if key not in incoming[successor]: - incoming[successor][key] = state - changed = True - if len(incoming[successor]) > self.max_disjuncts: + incoming[successor].setdefault(key, state) + # Sticky widening: once a node's disjunct budget is exceeded it stays + # collapsed to a single joined state, and every later update joins into + # that state instead of re-expanding. Without this a loop node oscillates + # -- widen to one state, re-expand past the budget from the back-edge, + # widen again -- burning the whole transfer budget and capping the + # function. Collapsing monotonically bounds the lattice height so the + # fixpoint terminates; the join is a sound may-approximation. + if len(incoming[successor]) > self.max_disjuncts or successor in widened: merged = join_states(incoming[successor].values(), successor) incoming[successor] = {merged.key(): merged} + if successor not in widened: + widened.add(successor) widenings += 1 - changed = True - if changed and successor not in queued: + # Re-queue only when the successor's state set actually changed; a + # collapsed node that re-joins to the same key has reached its fixpoint. + if set(incoming[successor]) != before and successor not in queued: work.append(successor) queued.add(successor) From 2f0c0d31094e1ee9e8f93e306c19a0b0e09042bc Mon Sep 17 00:00:00 2001 From: Riyan Dhiman Date: Thu, 20 Aug 2026 00:29:32 +0530 Subject: [PATCH 2/7] manifest: lachesis.toml schema + strict tomllib loader (P1) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit New lachesis/manifest/ package. schema.py: typed two-block manifest — project facts (core: source/build/memory/surface/trust; advanced: functions/alias/ dispatch/typedefs) plus analysis run config. loader.py: stdlib tomllib (lazy, with tomli backport on 3.10), strict validation that rejects unknown keys and wrong types with the dotted key path, walk-up discovery. Pure-additive: importing the package forces no new dep or Python floor; only reading a manifest needs 3.11+. 12 unit tests. --- lachesis/manifest/__init__.py | 53 ++++++ lachesis/manifest/loader.py | 287 +++++++++++++++++++++++++++++ lachesis/manifest/schema.py | 161 ++++++++++++++++ lachesis/manifest/test_manifest.py | 178 ++++++++++++++++++ 4 files changed, 679 insertions(+) create mode 100644 lachesis/manifest/__init__.py create mode 100644 lachesis/manifest/loader.py create mode 100644 lachesis/manifest/schema.py create mode 100644 lachesis/manifest/test_manifest.py diff --git a/lachesis/manifest/__init__.py b/lachesis/manifest/__init__.py new file mode 100644 index 00000000..4a5e0cf8 --- /dev/null +++ b/lachesis/manifest/__init__.py @@ -0,0 +1,53 @@ +"""Project manifest (``lachesis.toml``): declared facts + run configuration. + +A manifest lets a target declare ground truth the analysis cannot reliably infer — +its build variant, alloc/free vocabulary, entrypoints, trust boundaries, and expert +facts about opaque functions and dispatch seams — so the pipeline runs +deterministically instead of guessing. Facts are validated against the graph where +possible (:mod:`lachesis.manifest.validate`); run configuration is applied and logged. +""" +from __future__ import annotations + +from .loader import ( + MANIFEST_NAME, + ManifestError, + discover_manifest, + load_manifest, + load_or_discover, + parse_manifest, +) +from .schema import ( + AliasFacts, + AnalysisConfig, + Build, + FunctionContract, + Manifest, + Memory, + Ownership, + ProjectFacts, + Source, + Surface, + Trust, + UntrustedInput, +) + +__all__ = [ + "MANIFEST_NAME", + "ManifestError", + "discover_manifest", + "load_manifest", + "load_or_discover", + "parse_manifest", + "AliasFacts", + "AnalysisConfig", + "Build", + "FunctionContract", + "Manifest", + "Memory", + "Ownership", + "ProjectFacts", + "Source", + "Surface", + "Trust", + "UntrustedInput", +] diff --git a/lachesis/manifest/loader.py b/lachesis/manifest/loader.py new file mode 100644 index 00000000..5b8a74fc --- /dev/null +++ b/lachesis/manifest/loader.py @@ -0,0 +1,287 @@ +"""Load and validate a ``lachesis.toml`` manifest. + +Parsing is deliberately strict: an unknown key or a wrong-typed value raises +:class:`ManifestError` naming the offending path. A manifest is a *facts* file that +silently changes what the analysis sees, so a typo (``fre`` for ``free``) must fail +loudly rather than drop a fact without a trace. + +TOML is read with the standard library ``tomllib`` (Python 3.11+); on 3.10 the +``tomli`` backport is used if installed. The import is lazy so that importing this +package never forces a 3.11 floor on the rest of lachesis — only *using* a manifest +does. +""" +from __future__ import annotations + +from pathlib import Path + +from .schema import ( + AliasFacts, + AnalysisConfig, + Build, + FunctionContract, + Manifest, + Memory, + Ownership, + ProjectFacts, + Source, + Surface, + Trust, + UntrustedInput, +) + +MANIFEST_NAME = "lachesis.toml" + + +class ManifestError(ValueError): + """A manifest that is present but malformed (unknown key, wrong type, ...).""" + + +# --------------------------------------------------------------------------- # +# TOML backend (lazy) +# --------------------------------------------------------------------------- # +def _read_toml(path: Path) -> dict: + try: + import tomllib as _toml # py3.11+ + except ModuleNotFoundError: # pragma: no cover - exercised on 3.10 only + try: + import tomli as _toml # backport + except ModuleNotFoundError: + raise ManifestError( + "reading a manifest needs Python 3.11+ (stdlib tomllib) or the " + "'tomli' backport on 3.10; neither is importable" + ) + try: + with open(path, "rb") as fh: + return _toml.load(fh) + except OSError as exc: + raise ManifestError(f"cannot read {path}: {exc}") from exc + except Exception as exc: # tomllib.TOMLDecodeError and friends + raise ManifestError(f"{path}: invalid TOML: {exc}") from exc + + +# --------------------------------------------------------------------------- # +# small typed extractors (all raise ManifestError with the dotted key path) +# --------------------------------------------------------------------------- # +def _reject_unknown(table: dict, allowed: set[str], where: str) -> None: + extra = sorted(set(table) - allowed) + if extra: + raise ManifestError( + f"{where}: unknown key(s) {extra}; allowed: {sorted(allowed)}" + ) + + +def _table(value, where: str) -> dict: + if not isinstance(value, dict): + raise ManifestError(f"{where}: expected a table, got {type(value).__name__}") + return value + + +def _str(value, where: str) -> str: + if not isinstance(value, str): + raise ManifestError(f"{where}: expected a string, got {type(value).__name__}") + return value + + +def _str_list(value, where: str) -> tuple[str, ...]: + if not isinstance(value, list) or not all(isinstance(x, str) for x in value): + raise ManifestError(f"{where}: expected a list of strings") + return tuple(value) + + +def _int(value, where: str) -> int: + if isinstance(value, bool) or not isinstance(value, int): + raise ManifestError(f"{where}: expected an integer, got {type(value).__name__}") + return value + + +def _duration_seconds(value, where: str) -> float: + """Accept a number of seconds or a string like ``"30s"`` / ``"5m"``.""" + if isinstance(value, bool): + raise ManifestError(f"{where}: expected a duration, got a boolean") + if isinstance(value, (int, float)): + return float(value) + if isinstance(value, str): + units = {"s": 1.0, "m": 60.0, "h": 3600.0, "ms": 0.001} + text = value.strip() + for suffix, scale in sorted(units.items(), key=lambda kv: -len(kv[0])): + if text.endswith(suffix): + try: + return float(text[: -len(suffix)]) * scale + except ValueError: + break + try: + return float(text) + except ValueError: + pass + raise ManifestError(f"{where}: expected a duration (seconds or '30s'/'5m')") + + +# --------------------------------------------------------------------------- # +# block parsers +# --------------------------------------------------------------------------- # +def _parse_source(t: dict) -> Source: + _reject_unknown(t, {"roots", "exclude"}, "project.source") + return Source( + roots=_str_list(t.get("roots", []), "project.source.roots"), + exclude=_str_list(t.get("exclude", []), "project.source.exclude"), + ) + + +def _parse_build(t: dict) -> Build: + _reject_unknown(t, {"config", "include", "defines"}, "project.build") + defines = _table(t.get("defines", {}), "project.build.defines") + return Build( + config=_str_list(t.get("config", []), "project.build.config"), + include=_str_list(t.get("include", []), "project.build.include"), + defines=dict(defines), + ) + + +def _parse_memory(t: dict) -> Memory: + _reject_unknown(t, {"alloc", "free"}, "project.memory") + return Memory( + alloc=_str_list(t.get("alloc", []), "project.memory.alloc"), + free=_str_list(t.get("free", []), "project.memory.free"), + ) + + +def _parse_surface(t: dict) -> Surface: + _reject_unknown(t, {"entrypoints", "untrusted"}, "project.surface") + raw = t.get("untrusted", []) + if not isinstance(raw, list): + raise ManifestError("project.surface.untrusted: expected an array of tables") + untrusted = [] + for i, item in enumerate(raw): + where = f"project.surface.untrusted[{i}]" + item = _table(item, where) + _reject_unknown(item, {"fn", "at"}, where) + if "fn" not in item or "at" not in item: + raise ManifestError(f"{where}: requires both 'fn' and 'at'") + untrusted.append( + UntrustedInput(fn=_str(item["fn"], f"{where}.fn"), + at=_str(item["at"], f"{where}.at")) + ) + return Surface( + entrypoints=_str_list(t.get("entrypoints", []), "project.surface.entrypoints"), + untrusted=tuple(untrusted), + ) + + +def _parse_trust(t: dict) -> Trust: + _reject_unknown(t, {"sanitizers"}, "project.trust") + return Trust(sanitizers=_str_list(t.get("sanitizers", []), "project.trust.sanitizers")) + + +def _parse_functions(t: dict) -> tuple[FunctionContract, ...]: + out = [] + for name, spec in t.items(): + where = f"project.functions.{name}" + spec = _table(spec, where) + _reject_unknown(spec, {"frees", "allocs", "uses", "returns"}, where) + ret_raw = spec.get("returns", "unknown") + try: + returns = Ownership(_str(ret_raw, f"{where}.returns")) + except ValueError: + raise ManifestError( + f"{where}.returns: expected one of " + f"{[o.value for o in Ownership]}, got {ret_raw!r}" + ) + out.append( + FunctionContract( + name=name, + frees=_str_list(spec.get("frees", []), f"{where}.frees"), + allocs=_str_list(spec.get("allocs", []), f"{where}.allocs"), + uses=_str_list(spec.get("uses", []), f"{where}.uses"), + returns=returns, + ) + ) + return tuple(out) + + +def _parse_alias(t: dict) -> AliasFacts: + _reject_unknown(t, {"noalias"}, "project.alias") + raw = t.get("noalias", []) + if not isinstance(raw, list): + raise ManifestError("project.alias.noalias: expected a list of groups") + groups = tuple( + _str_list(g, f"project.alias.noalias[{i}]") for i, g in enumerate(raw) + ) + return AliasFacts(noalias=groups) + + +def _parse_str_map(t: dict, where: str) -> dict[str, str]: + out = {} + for k, v in t.items(): + out[k] = _str(v, f"{where}.{k}") + return out + + +def _parse_project(t: dict) -> ProjectFacts: + allowed = {"name", "language", "source", "build", "memory", "surface", "trust", + "functions", "alias", "dispatch", "typedefs"} + _reject_unknown(t, allowed, "project") + return ProjectFacts( + name=_str(t.get("name", ""), "project.name"), + language=_str(t.get("language", "c"), "project.language"), + source=_parse_source(_table(t.get("source", {}), "project.source")), + build=_parse_build(_table(t.get("build", {}), "project.build")), + memory=_parse_memory(_table(t.get("memory", {}), "project.memory")), + surface=_parse_surface(_table(t.get("surface", {}), "project.surface")), + trust=_parse_trust(_table(t.get("trust", {}), "project.trust")), + functions=_parse_functions(_table(t.get("functions", {}), "project.functions")), + alias=_parse_alias(_table(t.get("alias", {}), "project.alias")), + dispatch=_parse_str_map(_table(t.get("dispatch", {}), "project.dispatch"), + "project.dispatch"), + typedefs=_parse_str_map(_table(t.get("typedefs", {}), "project.typedefs"), + "project.typedefs"), + ) + + +def _parse_analysis(t: dict) -> AnalysisConfig: + allowed = {"engine", "graph", "disjunct_cap", "timeout_per_fn"} + _reject_unknown(t, allowed, "analysis") + return AnalysisConfig( + engine=_str(t["engine"], "analysis.engine") if "engine" in t else None, + graph=_str(t["graph"], "analysis.graph") if "graph" in t else None, + disjunct_cap=(_int(t["disjunct_cap"], "analysis.disjunct_cap") + if "disjunct_cap" in t else None), + timeout_per_fn=(_duration_seconds(t["timeout_per_fn"], "analysis.timeout_per_fn") + if "timeout_per_fn" in t else None), + ) + + +# --------------------------------------------------------------------------- # +# public API +# --------------------------------------------------------------------------- # +def parse_manifest(data: dict, *, path: str | None = None) -> Manifest: + """Build a :class:`Manifest` from an already-parsed TOML mapping.""" + _reject_unknown(data, {"project", "analysis"}, "") + return Manifest( + project=_parse_project(_table(data.get("project", {}), "project")), + analysis=_parse_analysis(_table(data.get("analysis", {}), "analysis")), + path=path, + ) + + +def load_manifest(path) -> Manifest: + """Read, parse and validate the ``lachesis.toml`` at *path*.""" + path = Path(path) + return parse_manifest(_read_toml(path), path=str(path)) + + +def discover_manifest(start=".") -> Path | None: + """Walk up from *start* (a file or directory) to the nearest ``lachesis.toml``.""" + p = Path(start).resolve() + if p.is_file(): + p = p.parent + for directory in (p, *p.parents): + candidate = directory / MANIFEST_NAME + if candidate.is_file(): + return candidate + return None + + +def load_or_discover(start=".") -> Manifest | None: + """Load the manifest nearest *start*, or ``None`` if there is none.""" + found = discover_manifest(start) + return load_manifest(found) if found is not None else None diff --git a/lachesis/manifest/schema.py b/lachesis/manifest/schema.py new file mode 100644 index 00000000..fa064127 --- /dev/null +++ b/lachesis/manifest/schema.py @@ -0,0 +1,161 @@ +"""Typed shape of a ``lachesis.toml`` project manifest. + +A manifest is a per-target file, checked in beside the code, that declares *facts +about the project* so the analysis pipeline runs deterministically instead of +guessing. Two blocks: + +* ``project`` — facts about the code. Every entry is a true statement that can, in + principle, be checked against the graph (see :mod:`lachesis.manifest.validate`). + Split into a *core* tier (things any maintainer knows: source layout, build + variant, alloc/free vocabulary, trust boundaries) and an *advanced* tier (expert + facts about hard internals: function contracts, aliasing, dispatch seams). +* ``analysis`` — run configuration (engine, caps, graph path). Applied, not + validated; every cap that drops coverage is reported by the runner. + +The dataclasses here are pure data with no graph dependency; the loader +(:mod:`lachesis.manifest.loader`) builds them from parsed TOML and the rest of the +pipeline consumes them. Leaf *facts* are frozen (hashable, comparable); the +container blocks are plain dataclasses so defaults compose cleanly. +""" +from __future__ import annotations + +from dataclasses import dataclass, field +from enum import Enum + + +class Ownership(str, Enum): + """What a function's return value obliges the caller to do.""" + + OWNED = "owned" # caller receives ownership and must free + BORROWED = "borrowed" # caller must NOT free (aliases live state) + UNKNOWN = "unknown" + + +# --------------------------------------------------------------------------- # +# core tier — facts any maintainer can state in five minutes +# --------------------------------------------------------------------------- # +@dataclass +class Source: + """Where the project's own code lives, and what to skip.""" + + roots: tuple[str, ...] = () + exclude: tuple[str, ...] = () + + +@dataclass +class Build: + """The variant actually shipped, so the right branches are analyzed.""" + + config: tuple[str, ...] = () # active #ifdef / feature flags + include: tuple[str, ...] = () # header search paths + defines: dict[str, object] = field(default_factory=dict) # macro -> value + + +@dataclass +class Memory: + """The project's allocation vocabulary (its custom alloc/free wrappers).""" + + alloc: tuple[str, ...] = () + free: tuple[str, ...] = () + + +@dataclass(frozen=True) +class UntrustedInput: + """A point where attacker-controlled data enters the program.""" + + fn: str # function that introduces the input + at: str # "return" | "argN" | a parameter name + + +@dataclass +class Surface: + """Where execution starts and where the outside world touches the program.""" + + entrypoints: tuple[str, ...] = () + untrusted: tuple[UntrustedInput, ...] = () + + +@dataclass +class Trust: + """Things the project has already made safe.""" + + sanitizers: tuple[str, ...] = () # inputs through here are validated + + +# --------------------------------------------------------------------------- # +# advanced tier — expert facts about hard internals (all optional) +# --------------------------------------------------------------------------- # +@dataclass(frozen=True) +class FunctionContract: + """A behavioural summary a maintainer supplies for an opaque/cross-TU function. + + ``frees``/``allocs``/``uses`` are access paths rooted at a parameter, e.g. + ``"arg0"`` or ``"arg0.data"``. ``returns`` records return ownership. + """ + + name: str + frees: tuple[str, ...] = () + allocs: tuple[str, ...] = () + uses: tuple[str, ...] = () + returns: Ownership = Ownership.UNKNOWN + + +@dataclass +class AliasFacts: + """Heap facts the points-to model is too conservative to derive.""" + + # each group is a set of access paths declared NOT to alias one another + noalias: tuple[tuple[str, ...], ...] = () + + +@dataclass +class ProjectFacts: + """The ``[project]`` block: facts about the code, core + advanced tiers.""" + + name: str = "" + language: str = "c" + # core + source: Source = field(default_factory=Source) + build: Build = field(default_factory=Build) + memory: Memory = field(default_factory=Memory) + surface: Surface = field(default_factory=Surface) + trust: Trust = field(default_factory=Trust) + # advanced + functions: tuple[FunctionContract, ...] = () + alias: AliasFacts = field(default_factory=AliasFacts) + dispatch: dict[str, str] = field(default_factory=dict) # "struct.field" -> handler + typedefs: dict[str, str] = field(default_factory=dict) # alias -> concrete struct + + +# --------------------------------------------------------------------------- # +# run configuration — applied, not validated +# --------------------------------------------------------------------------- # +@dataclass +class AnalysisConfig: + """The ``[analysis]`` block: how to run, not what the code is.""" + + engine: str | None = None # lifetime engine, e.g. "object" + graph: str | None = None # path to a prebuilt .kuzu, if any + disjunct_cap: int | None = None # object-state disjunct ceiling + timeout_per_fn: float | None = None # seconds; None = engine default + + +@dataclass +class Manifest: + """A parsed ``lachesis.toml``: project facts plus run configuration.""" + + project: ProjectFacts = field(default_factory=ProjectFacts) + analysis: AnalysisConfig = field(default_factory=AnalysisConfig) + path: str | None = None # source file, for diagnostics + + @property + def is_empty(self) -> bool: + """True when neither block declares anything (a no-op manifest).""" + p, a = self.project, self.analysis + return ( + not (p.name or p.source.roots or p.build.config or p.memory.alloc + or p.memory.free or p.surface.entrypoints or p.surface.untrusted + or p.trust.sanitizers or p.functions or p.alias.noalias + or p.dispatch or p.typedefs) + and a == AnalysisConfig() + ) diff --git a/lachesis/manifest/test_manifest.py b/lachesis/manifest/test_manifest.py new file mode 100644 index 00000000..cb1c2bca --- /dev/null +++ b/lachesis/manifest/test_manifest.py @@ -0,0 +1,178 @@ +"""Tests for the ``lachesis.toml`` loader (P1): parsing, defaults, strictness, +discovery. Pure data — no graph involved.""" +from __future__ import annotations + +import textwrap + +import pytest + +from lachesis.manifest import ( + ManifestError, + Ownership, + discover_manifest, + load_manifest, + load_or_discover, + parse_manifest, +) +from lachesis.manifest.loader import _duration_seconds + +FULL = textwrap.dedent( + """ + [project] + name = "curl" + language = "c" + + [project.source] + roots = ["lib", "src"] + exclude = ["tests", "**/vendor/**"] + + [project.build] + config = ["USE_OPENSSL", "ENABLE_IPV6"] + include = ["include", "lib"] + defines = { CURL_DISABLE_FTP = 0 } + + [project.memory] + alloc = ["curl_malloc", "Curl_saferealloc"] + free = ["Curl_safefree"] + + [project.surface] + entrypoints = ["curl_easy_perform"] + [[project.surface.untrusted]] + fn = "Curl_read" + at = "return" + [[project.surface.untrusted]] + fn = "curl_easy_setopt" + at = "arg2" + + [project.trust] + sanitizers = ["Curl_urldecode"] + + [project.functions.Curl_close] + frees = ["arg0.data"] + returns = "borrowed" + + [project.functions.Curl_dup] + returns = "owned" + + [project.alias] + noalias = [["Curl_easy.state", "Curl_easy.set"]] + + [project.dispatch] + "Curl_handler.disconnect" = "ossl_disconnect" + + [project.typedefs] + Curl_easy = "SessionHandle" + + [analysis] + engine = "object" + graph = "~/.lachesis/graphs/curl.kuzu" + disjunct_cap = 64 + timeout_per_fn = "30s" + """ +) + + +def _parse(text: str): + import tomllib + return parse_manifest(tomllib.loads(text), path="") + + +def test_full_manifest_parses(): + m = _parse(FULL) + p = m.project + assert p.name == "curl" and p.language == "c" + assert p.source.roots == ("lib", "src") + assert p.source.exclude == ("tests", "**/vendor/**") + assert p.build.config == ("USE_OPENSSL", "ENABLE_IPV6") + assert p.build.defines == {"CURL_DISABLE_FTP": 0} + assert p.memory.alloc == ("curl_malloc", "Curl_saferealloc") + assert p.memory.free == ("Curl_safefree",) + assert p.surface.entrypoints == ("curl_easy_perform",) + assert [(u.fn, u.at) for u in p.surface.untrusted] == [ + ("Curl_read", "return"), ("curl_easy_setopt", "arg2") + ] + assert p.trust.sanitizers == ("Curl_urldecode",) + by_name = {f.name: f for f in p.functions} + assert by_name["Curl_close"].frees == ("arg0.data",) + assert by_name["Curl_close"].returns is Ownership.BORROWED + assert by_name["Curl_dup"].returns is Ownership.OWNED + assert p.alias.noalias == (("Curl_easy.state", "Curl_easy.set"),) + assert p.dispatch == {"Curl_handler.disconnect": "ossl_disconnect"} + assert p.typedefs == {"Curl_easy": "SessionHandle"} + assert m.analysis.engine == "object" + assert m.analysis.disjunct_cap == 64 + assert m.analysis.timeout_per_fn == 30.0 + assert not m.is_empty + + +def test_empty_manifest_defaults(): + m = parse_manifest({}, path="") + assert m.is_empty + assert m.project.language == "c" + assert m.project.memory.free == () + assert m.analysis.engine is None + + +def test_unknown_top_level_key_rejected(): + with pytest.raises(ManifestError, match="unknown key"): + _parse("[projekt]\nname='x'\n") + + +def test_unknown_nested_key_rejected(): + # a typo that would silently drop a fact must fail loudly + with pytest.raises(ManifestError, match=r"project\.memory.*unknown key"): + _parse("[project.memory]\nfre = ['x']\n") + + +def test_wrong_type_rejected(): + with pytest.raises(ManifestError, match="list of strings"): + _parse("[project.memory]\nfree = 'Curl_safefree'\n") + + +def test_bad_ownership_value_rejected(): + with pytest.raises(ManifestError, match="returns"): + _parse("[project.functions.f]\nreturns = 'leased'\n") + + +def test_untrusted_requires_fn_and_at(): + with pytest.raises(ManifestError, match="requires both"): + _parse("[[project.surface.untrusted]]\nfn = 'x'\n") + + +def test_duration_parsing(): + assert _duration_seconds(30, "x") == 30.0 + assert _duration_seconds("30s", "x") == 30.0 + assert _duration_seconds("5m", "x") == 300.0 + assert _duration_seconds("250ms", "x") == 0.25 + with pytest.raises(ManifestError): + _duration_seconds(True, "x") + + +def test_discovery_walks_up(tmp_path): + (tmp_path / "lachesis.toml").write_text("[project]\nname='root'\n") + nested = tmp_path / "a" / "b" + nested.mkdir(parents=True) + found = discover_manifest(nested) + assert found == tmp_path / "lachesis.toml" + m = load_or_discover(nested) + assert m is not None and m.project.name == "root" + + +def test_discovery_absent_returns_none(tmp_path): + assert discover_manifest(tmp_path) is None + assert load_or_discover(tmp_path) is None + + +def test_load_manifest_from_file(tmp_path): + f = tmp_path / "lachesis.toml" + f.write_text(FULL) + m = load_manifest(f) + assert m.project.name == "curl" + assert m.path == str(f) + + +def test_invalid_toml_reports_path(tmp_path): + f = tmp_path / "lachesis.toml" + f.write_text("[project\nname = ") + with pytest.raises(ManifestError, match="invalid TOML"): + load_manifest(f) From a1723e9ef7656ffe28b59f962549f9ae520abc00 Mon Sep 17 00:00:00 2001 From: Riyan Dhiman Date: Thu, 20 Aug 2026 00:33:10 +0530 Subject: [PATCH 3/7] manifest: graph-validation with warn-on-contradiction (P2) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit validate.py: resolve every declared symbol against the graph and sort into validated (defined, right kind) / external (declared-only, unverifiable — the manifest's job) / warning (unresolved or wrong-kind — typo or stale). Only warnings surface as problems; the external/warning split is what keeps the check honest. This is the anti-gaming keystone: facts are checkable, verdicts are not. Depends only on store.resolve, so tests use a stub (6 tests); verified end-to-end against a real graph. Semantic 'does it actually free' contradiction lands in P3 with solver effects; access-path and build facts are left to consuming stages. --- lachesis/manifest/test_validate.py | 101 +++++++++++++++++++ lachesis/manifest/validate.py | 153 +++++++++++++++++++++++++++++ 2 files changed, 254 insertions(+) create mode 100644 lachesis/manifest/test_validate.py create mode 100644 lachesis/manifest/validate.py diff --git a/lachesis/manifest/test_validate.py b/lachesis/manifest/test_validate.py new file mode 100644 index 00000000..290721bb --- /dev/null +++ b/lachesis/manifest/test_validate.py @@ -0,0 +1,101 @@ +"""Tests for manifest graph-validation (P2). Uses a stub store exposing only +``resolve(name)`` — the sole surface :func:`validate_manifest` depends on.""" +from __future__ import annotations + +from lachesis.manifest import parse_manifest +from lachesis.manifest.validate import Status, validate_manifest + + +class FakeStore: + """A store whose resolve() returns canned entries by exact name.""" + + def __init__(self, entries: dict[str, list[dict]]): + self._entries = entries + + def resolve(self, name: str) -> list[dict]: + # include a fuzzy near-miss to prove _exact() filters on name equality + out = list(self._entries.get(name, [])) + out.append({"name": name + "_similar", "kind": "function"}) + return out + + +def _fn(name, file="lib/x.c", line=10, decl=False): + return {"name": name, "kind": "function", "file": file, "line": line, + "declaration_only": decl, "node_id": f"id::{name}"} + + +def _parse(text): + import tomllib + return parse_manifest(tomllib.loads(text)) + + +def test_validated_external_and_warning_buckets(): + store = FakeStore({ + "Curl_safefree": [_fn("Curl_safefree")], # defined -> validated + "talloc_free": [_fn("talloc_free", decl=True)], # decl-only -> external + # "typo_free" absent -> warning (not found) + "some_global": [{"name": "some_global", "kind": "variable"}], # wrong kind + }) + m = _parse( + """ + [project.memory] + free = ["Curl_safefree", "talloc_free", "typo_free", "some_global"] + """ + ) + report = validate_manifest(m, store) + by_sym = {c.symbol: c.status for c in report.checks} + assert by_sym["Curl_safefree"] is Status.VALIDATED + assert by_sym["talloc_free"] is Status.EXTERNAL + assert by_sym["typo_free"] is Status.WARNING + assert by_sym["some_global"] is Status.WARNING + assert not report.ok # warnings present + assert len(report.validated) == 1 + assert len(report.external) == 1 + assert len(report.warnings) == 2 + + +def test_clean_manifest_is_ok(): + store = FakeStore({"curl_easy_perform": [_fn("curl_easy_perform")]}) + m = _parse('[project.surface]\nentrypoints = ["curl_easy_perform"]\n') + report = validate_manifest(m, store) + assert report.ok and len(report.validated) == 1 + + +def test_typedef_expects_type_kind(): + store = FakeStore({ + "SessionHandle": [{"name": "SessionHandle", "kind": "struct", + "declaration_only": False, "file": "h.h", "line": 3}], + "NotAType": [_fn("NotAType")], # a function, not a type -> warning + }) + m = _parse( + """ + [project.typedefs] + Curl_easy = "SessionHandle" + Bad = "NotAType" + """ + ) + report = validate_manifest(m, store) + by_sym = {c.symbol: c.status for c in report.checks} + assert by_sym["SessionHandle"] is Status.VALIDATED + assert by_sym["NotAType"] is Status.WARNING + + +def test_dispatch_handler_checked_as_function(): + store = FakeStore({"ossl_disconnect": [_fn("ossl_disconnect")]}) + m = _parse('[project.dispatch]\n"Curl_handler.disconnect" = "ossl_disconnect"\n') + report = validate_manifest(m, store) + assert report.checks[0].status is Status.VALIDATED + assert report.checks[0].symbol == "ossl_disconnect" + + +def test_empty_manifest_no_checks(): + report = validate_manifest(parse_manifest({}), FakeStore({})) + assert report.checks == () and report.ok + assert "no symbol facts" in report.format() + + +def test_format_lists_warnings_first(): + store = FakeStore({"ok_fn": [_fn("ok_fn")]}) + m = _parse('[project.memory]\nfree = ["ok_fn", "ghost"]\n') + text = validate_manifest(m, store).format() + assert text.index("ghost") < text.index("ok_fn") diff --git a/lachesis/manifest/validate.py b/lachesis/manifest/validate.py new file mode 100644 index 00000000..776b85e8 --- /dev/null +++ b/lachesis/manifest/validate.py @@ -0,0 +1,153 @@ +"""Check a manifest's *facts* against the graph — the anti-gaming keystone. + +Every ``[project]`` entry is a statement about the code, so it can be checked where +the graph can see the truth. This module resolves each declared symbol and sorts it +into three honest buckets: + +* **validated** — the symbol is defined in the graph, with the expected kind. The + fact is grounded. +* **external** — the graph knows the symbol only as a declaration with no body + (a cross-TU or library function). It *cannot* be verified — which is exactly what + the manifest exists to supply — so it is accepted quietly but counted. +* **warning** — the symbol does not resolve at all, or resolves only to the wrong + kind (a ``free`` name that is actually a variable, a ``typedef`` target that is not + a type). This is the typo / stale-manifest signal. + +The distinction between *external* and *warning* is what keeps the check honest: a +name the graph has never heard of is a probable mistake; a name it knows but cannot +open is a legitimate declared fact. Only warnings are surfaced as problems. + +Scope of P2: existence and kind. The deeper semantic contradiction — "you declared +this frees, but its body never frees" — needs the object solver's per-function +effects and is checked in P3, where those effects are already computed. Access-path +facts (``frees = ["arg0.data"]``, ``noalias``) and non-symbol facts (build flags, +source roots) are not symbol lookups and are left to their consuming stages. + +Depends only on ``store.resolve(name) -> list[entry]``; any object exposing that +works (the real :class:`GraphStore`, or a stub in tests). +""" +from __future__ import annotations + +from dataclasses import dataclass +from enum import Enum + +from .schema import Manifest + +FUNCTION_KINDS = frozenset({"function", "method"}) +TYPE_KINDS = frozenset({"struct", "union", "enum", "class", "typedef", "type", "interface"}) + + +class Status(str, Enum): + VALIDATED = "validated" + EXTERNAL = "external" + WARNING = "warning" + + +@dataclass(frozen=True) +class FactCheck: + """The outcome of checking one declared symbol against the graph.""" + + location: str # dotted manifest location, e.g. "project.memory.free" + symbol: str # the declared name + status: Status + detail: str # human-readable explanation + + def __str__(self) -> str: + mark = {Status.VALIDATED: "✓", Status.EXTERNAL: "~", Status.WARNING: "!"} + return f" {mark[self.status]} {self.location} '{self.symbol}' — {self.detail}" + + +@dataclass +class ManifestReport: + """The verdict over every symbol a manifest declares.""" + + checks: tuple[FactCheck, ...] = () + + @property + def validated(self) -> tuple[FactCheck, ...]: + return tuple(c for c in self.checks if c.status is Status.VALIDATED) + + @property + def external(self) -> tuple[FactCheck, ...]: + return tuple(c for c in self.checks if c.status is Status.EXTERNAL) + + @property + def warnings(self) -> tuple[FactCheck, ...]: + return tuple(c for c in self.checks if c.status is Status.WARNING) + + @property + def ok(self) -> bool: + """True when no declared fact contradicts the graph.""" + return not self.warnings + + def format(self) -> str: + if not self.checks: + return "manifest: no symbol facts to validate" + lines = [ + f"manifest validation: {len(self.validated)} grounded, " + f"{len(self.external)} external (unverifiable), " + f"{len(self.warnings)} warning(s)" + ] + # warnings first — they are what a reader must act on + for c in (*self.warnings, *self.external, *self.validated): + lines.append(str(c)) + return "\n".join(lines) + + +# --------------------------------------------------------------------------- # +def _exact(store, name: str) -> list[dict]: + """Candidates whose name matches *name* exactly (resolve() also fuzzes).""" + return [e for e in store.resolve(name) if e.get("name") == name] + + +def _classify(store, name: str, kinds: frozenset[str]) -> tuple[Status, str]: + cands = _exact(store, name) + if not cands: + return Status.WARNING, "no such symbol in the graph (typo or stale?)" + + typed = [e for e in cands if e.get("kind") in kinds] + if not typed: + found = sorted({str(e.get("kind")) for e in cands}) + want = "function" if kinds is FUNCTION_KINDS else "type" + return Status.WARNING, f"resolves only to kind(s) {found}, expected a {want}" + + defined = [e for e in typed if not e.get("declaration_only")] + if defined: + e = defined[0] + where = f"{e.get('file')}:{e.get('line')}" if e.get("file") else "graph" + return Status.VALIDATED, f"defined ({e.get('kind')}) at {where}" + + # known to the graph but only as a bodiless declaration — the manifest's job + return Status.EXTERNAL, "declared-only in graph (no body to verify)" + + +def _check(store, checks: list, location: str, name: str, kinds: frozenset[str]) -> None: + status, detail = _classify(store, name, kinds) + checks.append(FactCheck(location=location, symbol=name, status=status, detail=detail)) + + +def validate_manifest(manifest: Manifest, store) -> ManifestReport: + """Resolve every declared symbol in *manifest* against *store*.""" + p = manifest.project + checks: list[FactCheck] = [] + + for name in p.memory.alloc: + _check(store, checks, "project.memory.alloc", name, FUNCTION_KINDS) + for name in p.memory.free: + _check(store, checks, "project.memory.free", name, FUNCTION_KINDS) + for name in p.surface.entrypoints: + _check(store, checks, "project.surface.entrypoints", name, FUNCTION_KINDS) + for u in p.surface.untrusted: + _check(store, checks, "project.surface.untrusted", u.fn, FUNCTION_KINDS) + for name in p.trust.sanitizers: + _check(store, checks, "project.trust.sanitizers", name, FUNCTION_KINDS) + for fc in p.functions: + _check(store, checks, "project.functions", fc.name, FUNCTION_KINDS) + for field, handler in p.dispatch.items(): + # the seam target must be a real function; the "struct.field" key is a + # structural fact checked more deeply once the seam binder consumes it. + _check(store, checks, f"project.dispatch[{field}]", handler, FUNCTION_KINDS) + for alias, struct in p.typedefs.items(): + _check(store, checks, f"project.typedefs[{alias}]", struct, TYPE_KINDS) + + return ManifestReport(checks=tuple(checks)) From 620a629fba72b5ba708cb582740aecd318dffa6d Mon Sep 17 00:00:00 2001 From: Riyan Dhiman Date: Thu, 20 Aug 2026 00:42:26 +0530 Subject: [PATCH 4/7] P3: wire manifest memory-roles + analysis config into the pipeline The manifest's per-target facts now reach the object-lifetime engine, closing a false negative every off-the-shelf analyzer shares: a project that frees through its own wrapper (Curl_safefree, my_release) instead of libc free was invisible to the default lifecycle catalog, so no free event was emitted and no double-free/UAF was reported. - normalize.py: Normalizer takes extra_alloc/extra_dealloc; normalizer_with() returns the cached per-lang instance when there are no extras, else a fresh UNCACHED one so a run's custom vocabulary never leaks into the global cache. Manifest names join the post-canonicalization alloc/dealloc sets (canon of an unmapped name is itself). - object_lifetime.py: analyze_object_lifetimes() accepts extra_alloc/extra_dealloc and max_disjuncts; the solver disjunct cap rides inside the pickled prepared tuple so it survives the spawn to a worker (a module global would reset on re-import). Default stays 32 (the tuned value) when unset. - pipeline.py: run_pass(manifest=...) extracts the run knobs via _manifest_config and records every one in lifetime['applied_config'] -- including analysis.timeout_per_fn, surfaced as declared-but-not-enforced rather than silently dropped, so config can never become a hidden verdict override. Precedence: explicit arg > manifest > env. Proof (test_pipeline_manifest.py): the same C graph reports no double-free without a manifest and fires the double-free once memory.free declares the wrapper; plus the config-extraction/audit unit. 22 flow + 18 manifest tests green. The semantic "declared frees but body never frees" contradiction needs per-function effects (only computed inside the pass), so it lands in P5's run summary, not the pure P2 symbol validator. --- lachesis/flow/normalize.py | 28 +++++- lachesis/flow/object_lifetime.py | 42 ++++++--- lachesis/flow/pipeline.py | 53 +++++++++-- lachesis/flow/test_pipeline_manifest.py | 111 ++++++++++++++++++++++++ 4 files changed, 214 insertions(+), 20 deletions(-) create mode 100644 lachesis/flow/test_pipeline_manifest.py diff --git a/lachesis/flow/normalize.py b/lachesis/flow/normalize.py index 8b122130..18d922de 100644 --- a/lachesis/flow/normalize.py +++ b/lachesis/flow/normalize.py @@ -47,7 +47,7 @@ def _is_symbol(s): class Normalizer: """Canonicalizes IR names from one language's form profile. Cheap; build once per lang.""" - def __init__(self, lang="c"): + def __init__(self, lang="c", *, extra_alloc=(), extra_dealloc=()): self.lang = lang prof = atropos.normalization_profile(lang) or {} # Merge EVERY '*_aliases' section into one callee-rewrite table (convention, not @@ -80,6 +80,16 @@ def __init__(self, lang="c"): cat = atropos.sink_catalog(lang) self.alloc_names = {m for m, c in cat.items() if c.get("family") in alloc_kinds} | alloc_extra + # Per-target manifest facts (project.memory.alloc/free) extend the catalog + # vocabulary for this run only. A target's own allocator/free wrappers + # (Curl_safefree, talloc_free, ...) are named here so the typestate skeleton + # emits their lifecycle events -- the highest-recall lever the manifest adds. + # They are canonical surface names, so they join the post-canonicalization sets + # directly (canon of an unmapped name is itself). + self.alloc_names |= {n for n in extra_alloc if _is_symbol(n)} + self.dealloc_names |= {n for n in extra_dealloc if _is_symbol(n)} + self.manifest_alloc = tuple(n for n in extra_alloc if _is_symbol(n)) + self.manifest_dealloc = tuple(n for n in extra_dealloc if _is_symbol(n)) def is_alloc(self, callee): """True if `callee` allocates an owned object (a lifecycle alloc event source).""" @@ -106,7 +116,9 @@ def summary(self): """Small dict describing what this normalizer will apply -- for a coverage line.""" return {"lang": self.lang, "alias_sections": sorted(self.alias_sections), "callee_rewrites": len(self.callee_rewrites), "opaque_kinds": len(self.opaque), - "alloc_names": len(self.alloc_names), "dealloc_names": len(self.dealloc_names)} + "alloc_names": len(self.alloc_names), "dealloc_names": len(self.dealloc_names), + "manifest_alloc": list(self.manifest_alloc), + "manifest_dealloc": list(self.manifest_dealloc)} _CACHE = {} @@ -117,3 +129,15 @@ def normalizer(lang="c"): if lang not in _CACHE: _CACHE[lang] = Normalizer(lang) return _CACHE[lang] + + +def normalizer_with(lang="c", extra_alloc=(), extra_dealloc=()): + """A Normalizer for *lang*, extended with per-target manifest alloc/free names. + + With no extras this is the shared cached instance. With extras it is a fresh, + UNCACHED instance -- the global cache stays keyed on language only, so one run's + manifest never leaks its custom vocabulary into another run's normalizer.""" + if not extra_alloc and not extra_dealloc: + return normalizer(lang) + return Normalizer(lang, extra_alloc=tuple(extra_alloc), + extra_dealloc=tuple(extra_dealloc)) diff --git a/lachesis/flow/object_lifetime.py b/lachesis/flow/object_lifetime.py index 0a51519e..2f5419ed 100644 --- a/lachesis/flow/object_lifetime.py +++ b/lachesis/flow/object_lifetime.py @@ -20,7 +20,7 @@ from lachesis.nav.dataflow.reaching_def import ReachingDef from lachesis.nav.dataflow.substrate import Substrate -from .normalize import normalizer +from .normalize import normalizer, normalizer_with from .object_state import ( AbstractState, AccessPath, @@ -39,6 +39,11 @@ } _NULL_KINDS = {"GNUNullExpr", "CXXNullPtrLiteralExpr"} +# Per-node disjunct cap for the object-state solver. 32 (not 64) is tuned: see +# ``_analyze_prepared`` for why earlier widening raises recall on looping functions. +# A manifest ``analysis.disjunct_cap`` overrides it per run. +_DEFAULT_MAX_DISJUNCTS = 32 + def _props(item): return (item or {}).get("properties") or {} @@ -350,21 +355,25 @@ def _initial_state(cfg, operations): return initial -def _summary_for(sub, norm, function_id, function_ir, all_functions, summaries, cfg): +def _summary_for(sub, norm, function_id, function_ir, all_functions, summaries, cfg, + max_disjuncts=_DEFAULT_MAX_DISJUNCTS): prepared = _prepare_summary( - sub, norm, function_id, function_ir, all_functions, summaries, cfg) + sub, norm, function_id, function_ir, all_functions, summaries, cfg, max_disjuncts) return _analyze_prepared(prepared) -def _prepare_summary(sub, norm, function_id, function_ir, all_functions, summaries, cfg): +def _prepare_summary(sub, norm, function_id, function_ir, all_functions, summaries, cfg, + max_disjuncts=_DEFAULT_MAX_DISJUNCTS): operations = extract_operations( sub, norm, function_id, function_ir, all_functions, summaries, cfg) - return cfg["nodes"], cfg["succ"], operations, _initial_state(cfg, operations) + # max_disjuncts rides INSIDE the prepared tuple so it survives the pickle to a worker + # process -- a module global would reset to its default on spawn re-import. + return cfg["nodes"], cfg["succ"], operations, _initial_state(cfg, operations), max_disjuncts def _analyze_prepared(prepared): """Pure, pickleable solver boundary used by process workers.""" - nodes, successors, operations, initial = prepared + nodes, successors, operations, initial, max_disjuncts = prepared # 32 disjuncts/node (not 64): a fully-wired CFG closes every loop's def-use cycle, # so a looping function accumulates disjuncts across the back-edge until widening # fires. At 64 the widening fired so late that small pipeline-walk functions blew @@ -372,7 +381,8 @@ def _analyze_prepared(prepared): # Widening sooner makes them converge within budget; the join is an over- # approximation, so recall (uncapped functions) rises at a marginal precision cost, # which is the right trade for a finder (capping is a guaranteed false negative). - result = ObjectStateAnalyzer(max_disjuncts=32).analyze( + # A manifest may override the cap; the default stays 32 for the reasons above. + result = ObjectStateAnalyzer(max_disjuncts=max_disjuncts).analyze( nodes, successors, operations, initial=initial) alternatives = {state.trace for state in result.exit_states} return tuple(sorted(alternatives, key=repr)), result @@ -438,16 +448,23 @@ class ObjectLifetimeResult: diagnostics: dict -def analyze_object_lifetimes(store, functions, call_successors, *, lang="c", graph=None): - """Run object-identity lifetime analysis over all defined functions in ``functions``.""" +def analyze_object_lifetimes(store, functions, call_successors, *, lang="c", graph=None, + extra_alloc=(), extra_dealloc=(), max_disjuncts=None): + """Run object-identity lifetime analysis over all defined functions in ``functions``. + + ``extra_alloc`` / ``extra_dealloc`` are per-target manifest ``memory.alloc`` / + ``memory.free`` names: they extend the lifecycle vocabulary this run recognizes so a + project's own allocator/free wrappers emit typestate events. ``max_disjuncts`` overrides + the solver's per-node disjunct cap (``None`` keeps the tuned default).""" started = perf_counter() + disjunct_cap = _DEFAULT_MAX_DISJUNCTS if max_disjuncts is None else max_disjuncts if graph is not None and graph is not store.graph: from lachesis.nav.graph_store import GraphStore analysis_store = GraphStore(graph) else: analysis_store = store sub = Substrate(analysis_store.index).load().load_initializers() - norm = normalizer(lang) + norm = normalizer_with(lang, extra_alloc, extra_dealloc) function_node_ids = [node_id for kind in ("function", "method", "constructor") for node_id in getattr(analysis_store.index, "by_kind", {}).get(kind, ())] sub.warm_nodes(function_node_ids) @@ -496,7 +513,8 @@ def analyze_object_lifetimes(store, functions, call_successors, *, lang="c", gra continue name = analysable[0] prepared = _prepare_summary( - sub, norm, by_name[name], functions[name], functions, summaries, cfgs[name]) + sub, norm, by_name[name], functions[name], functions, summaries, + cfgs[name], disjunct_cap) future = (executor.submit(_analyze_prepared, prepared) if executor is not None else None) pending.append((name, prepared, future)) @@ -527,7 +545,7 @@ def analyze_object_lifetimes(store, functions, call_successors, *, lang="c", gra break summary, result = _summary_for( sub, norm, by_name[name], functions[name], functions, - summaries, cfgs[name]) + summaries, cfgs[name], disjunct_cap) summary_runs[name] += 1 summary_transfers += result.transfers artifacts[name] = result diff --git a/lachesis/flow/pipeline.py b/lachesis/flow/pipeline.py index 15b25662..12e4d69d 100644 --- a/lachesis/flow/pipeline.py +++ b/lachesis/flow/pipeline.py @@ -82,7 +82,37 @@ def _match_object_mode_legacy(skels, cfg, fallback_entries): return leads -def run_pass(store, lang="c", lifetime_engine=None): +def _manifest_config(manifest): + """Extract the run knobs a manifest contributes, and an audit of each. + + Returns ``(engine, extra_alloc, extra_dealloc, max_disjuncts, applied)`` where + ``applied`` is a per-knob log so a run summary can show what the manifest changed + and -- critically -- what it declared that has no effect yet (no silent truncation: + a config knob that quietly does nothing is exactly the backdoor a facts-file must + not become).""" + applied = {} + if manifest is None: + return None, (), (), None, applied + mem = manifest.project.memory + analysis = manifest.analysis + if mem.alloc: + applied["memory.alloc"] = f"+{len(mem.alloc)} allocator name(s): {list(mem.alloc)}" + if mem.free: + applied["memory.free"] = f"+{len(mem.free)} free name(s): {list(mem.free)}" + if analysis.engine: + applied["analysis.engine"] = f"engine={analysis.engine} (manifest)" + if analysis.disjunct_cap is not None: + applied["analysis.disjunct_cap"] = f"solver disjunct cap -> {analysis.disjunct_cap}" + if analysis.timeout_per_fn is not None: + # Declared but not consumed: the object engine caps by disjunct/run counts, not + # wall-clock. Surface it rather than silently ignore it. + applied["analysis.timeout_per_fn"] = ( + f"declared {analysis.timeout_per_fn}s but NOT enforced " + "(engine caps by disjunct/run count, no wall-clock timeout yet)") + return (analysis.engine, mem.alloc, mem.free, analysis.disjunct_cap, applied) + + +def run_pass(store, lang="c", lifetime_engine=None, manifest=None): """Return {F, succ, summaries, skeletons, leads, lifetime} for an opened GraphStore. The store's whole-graph value-flow tier is ensured once (cached to disk), then every @@ -93,15 +123,23 @@ def run_pass(store, lang="c", lifetime_engine=None): C double-free/UAF leads use object identity by default. ``lifetime`` includes the bounded legacy differential and coverage diagnostics; functions with no complete object analysis retain legacy leads. Set ``LACHESIS_LIFETIME_ENGINE=shadow`` to run - both without changing output, or ``legacy`` for an operational rollback.""" + both without changing output, or ``legacy`` for an operational rollback. + + A ``manifest`` (:class:`lachesis.manifest.Manifest`) contributes per-target facts and + run config: ``memory.alloc``/``free`` extend the lifecycle vocabulary, and the + ``analysis`` block can pick the engine and the solver disjunct cap. Every knob it + applies is recorded in ``lifetime['applied_config']`` -- nothing is silently dropped. + Explicit arguments still win over the manifest, which wins over the environment.""" started = perf_counter() store.ensure_dataflow_tier() tier_done = perf_counter() - requested = lifetime_engine or os.environ.get( + (cfg_engine, extra_alloc, extra_dealloc, + cfg_disjunct, applied_config) = _manifest_config(manifest) + requested = lifetime_engine or cfg_engine or os.environ.get( "LACHESIS_LIFETIME_ENGINE", _DEFAULT_LIFETIME_ENGINE) if requested not in {"legacy", "shadow", "object"}: raise ValueError( - "LACHESIS_LIFETIME_ENGINE must be one of legacy, shadow, or object") + "lifetime engine must be one of legacy, shadow, or object") object_requested = lang.lower() == "c" and requested != "legacy" if object_requested: F, succ, analysis_graph = build_F(store, lang=lang, return_graph=True) @@ -114,12 +152,15 @@ def run_pass(store, lang="c", lifetime_engine=None): skeletons = build_skeletons(F, summaries, lang=lang) skeletons_done = perf_counter() - lifetime = {"requested": requested, "active": "legacy", "available": False} + lifetime = {"requested": requested, "active": "legacy", "available": False, + "applied_config": applied_config} legacy_leads = None leads = [] if object_requested: object_result = analyze_object_lifetimes( - store, F, succ, lang=lang, graph=analysis_graph) + store, F, succ, lang=lang, graph=analysis_graph, + extra_alloc=extra_alloc, extra_dealloc=extra_dealloc, + max_disjuncts=cfg_disjunct) # The projection already paid to materialize the disk graph. Reuse that same # in-memory index for the legacy coverage fallback instead of issuing another # whole-graph set of Kuzu scans merely to project CFG edges. diff --git a/lachesis/flow/test_pipeline_manifest.py b/lachesis/flow/test_pipeline_manifest.py new file mode 100644 index 00000000..3180e734 --- /dev/null +++ b/lachesis/flow/test_pipeline_manifest.py @@ -0,0 +1,111 @@ +"""P3 proof: a project manifest's ``memory.free`` vocabulary changes what the object +engine finds. + +A project that frees through its own wrapper (``my_release``, not the libc ``free``) +is invisible to the default lifecycle catalog -- so the engine sees no free event and +reports no double-free, a false negative every off-the-shelf analyzer shares. Declaring +that wrapper in ``lachesis.toml`` is the highest-recall lever the manifest adds: the same +graph, run with the manifest, now fires the double-free. + +Also covers the ``analysis`` config plumbing (:func:`_manifest_config`): which knobs are +applied, and the honest record of one declared-but-not-yet-enforced (``timeout_per_fn``) +so config can never become a silent verdict override. +""" +import subprocess +import sys +import tempfile +import unittest +from collections import defaultdict +from pathlib import Path + +from lachesis.core.snapshot import load_snapshot +from lachesis.manifest.schema import ( + AnalysisConfig, + Manifest, + Memory, + ProjectFacts, +) +from lachesis.nav.graph_store import GraphStore +from lachesis.pipeline import semantic_snapshot_graph + +from .pipeline import _manifest_config, run_pass + +# A double-free routed entirely through a project-specific free wrapper. The libc +# ``free`` never appears, so nothing in the default catalog marks a lifecycle event. +SOURCE = r""" +void *malloc(unsigned long); +void my_release(void *); + +void custom_double_free(void) { + char *first = malloc(8); + char *second = first; + my_release(first); + my_release(second); +} +""" + + +def _patterns_by_function(result): + by_function = defaultdict(set) + for lead in result["leads"]: + if lead["pattern"] in {"double-free", "use-after-free"}: + by_function[lead["entry"]].add(lead["pattern"]) + return by_function + + +class ManifestDrivenLifetimeTests(unittest.TestCase): + def test_manifest_free_wrapper_makes_double_free_fire(self): + with tempfile.TemporaryDirectory() as source_dir, tempfile.TemporaryDirectory() as output: + Path(source_dir, "custom.c").write_text(SOURCE) + completed = subprocess.run( + [sys.executable, "-m", "lachesis.frontends.c.build_graph", source_dir, output], + text=True, capture_output=True, check=False, + ) + self.assertEqual(completed.returncode, 0, completed.stderr) + snapshot = load_snapshot(output) + + # Baseline: no manifest. my_release is not a known free -> no double-free. + store = GraphStore(semantic_snapshot_graph(snapshot)) + baseline = run_pass(store, lang="c", lifetime_engine="object") + self.assertEqual( + _patterns_by_function(baseline)["custom_double_free"], set(), + "custom free wrapper must be invisible without a manifest", + ) + + # With a manifest declaring the wrapper, the same graph fires the double-free. + manifest = Manifest(project=ProjectFacts(memory=Memory(free=("my_release",)))) + store2 = GraphStore(semantic_snapshot_graph(snapshot)) + declared = run_pass(store2, lang="c", lifetime_engine="object", manifest=manifest) + self.assertEqual( + _patterns_by_function(declared)["custom_double_free"], {"double-free"}, + "declaring the free wrapper must recover the double-free", + ) + + # The applied-config audit records the recall lever the manifest pulled. + applied = declared["lifetime"]["applied_config"] + self.assertIn("memory.free", applied) + self.assertIn("my_release", applied["memory.free"]) + + def test_manifest_config_extraction_and_audit(self): + # No manifest -> empty knobs, empty audit. + engine, alloc, dealloc, cap, applied = _manifest_config(None) + self.assertEqual((engine, alloc, dealloc, cap, applied), (None, (), (), None, {})) + + # A full analysis block: applied knobs are recorded; timeout is surfaced as + # declared-but-unenforced rather than silently dropped. + manifest = Manifest( + project=ProjectFacts(memory=Memory(alloc=("xmalloc",), free=("xfree",))), + analysis=AnalysisConfig(engine="object", disjunct_cap=16, timeout_per_fn=30.0), + ) + engine, alloc, dealloc, cap, applied = _manifest_config(manifest) + self.assertEqual(engine, "object") + self.assertEqual(alloc, ("xmalloc",)) + self.assertEqual(dealloc, ("xfree",)) + self.assertEqual(cap, 16) + self.assertIn("memory.alloc", applied) + self.assertIn("analysis.disjunct_cap", applied) + self.assertIn("NOT enforced", applied["analysis.timeout_per_fn"]) + + +if __name__ == "__main__": + unittest.main() From 352aa4c37d6a2c204ac5165def3b63ff7e40a2db Mon Sep 17 00:00:00 2001 From: Riyan Dhiman Date: Thu, 20 Aug 2026 00:47:01 +0530 Subject: [PATCH 5/7] P4a: manifest function contracts -> summaries (cross-TU effect composition) A FunctionContract in the manifest lets a maintainer state what an opaque / cross-TU callee does (frees/allocs/uses of an argument access path). Those facts are converted to engine summaries and seeded for callees the solver cannot see into, so a free or use composes across a library boundary exactly like a summary the solver derived from a body. This is the interprocedural lever: without it every off-the-shelf analyzer treats the opaque call as a blind arg-dereference and misses the lifecycle event entirely. - object_lifetime.py: _parse_arg_path (argN[.sel]* -> position+selectors; non-positional forms skipped, never mis-bound) and _contracts_to_summaries (contract -> one summary alternative of ParamEffects, seeded ONLY for names with no analyzable body so a real body always stays authoritative). The summary-instantiation gate drops its `callee in all_functions` requirement: a summary is replayed whether the solver derived it or a contract supplied it. - pipeline.py: _manifest_config surfaces project.functions as contracts and records them in applied_config; run_pass threads them into analyze_object_lifetimes. Proof (test_pipeline_manifest.py): an opaque `ext_release` frees its arg only once a contract declares frees=["arg0"] -- the later *p=1 then fires a use-after-free that is invisible without the manifest. Plus converter units (arg-path parse, analyzable-name exclusion, effect mapping). 25 flow tests green. returns-ownership is a return-binding effect, not a param effect, and is deferred to a later P4 sub-step. --- lachesis/flow/object_lifetime.py | 66 ++++++++++++++++++-- lachesis/flow/pipeline.py | 16 +++-- lachesis/flow/test_pipeline_manifest.py | 81 ++++++++++++++++++++++++- 3 files changed, 150 insertions(+), 13 deletions(-) diff --git a/lachesis/flow/object_lifetime.py b/lachesis/flow/object_lifetime.py index 2f5419ed..a04d58f3 100644 --- a/lachesis/flow/object_lifetime.py +++ b/lachesis/flow/object_lifetime.py @@ -285,7 +285,10 @@ def extract_operations(sub, norm, function_id, function_ir, all_functions, summa continue callee_summary = summaries.get(callee) - if callee in all_functions and callee_summary is not None: + # A summary is instantiated whether the solver derived it (the callee is an + # analyzable function) or a manifest contract supplied it for an opaque callee + # the solver cannot see into; both are trustworthy replays of the callee's effects. + if callee_summary is not None: alternatives = [] for alternative in callee_summary: effects = [] @@ -448,14 +451,64 @@ class ObjectLifetimeResult: diagnostics: dict +def _parse_arg_path(spec): + """Parse a contract access path ``arg0.data`` -> ``(position, selectors)``. + + Only the positional ``argN`` form is resolvable, because a summary instantiates by + binding ``effect.position`` to the actual argument at a call site -- a named formal + would need the (often bodiless) callee's signature we do not have. Returns ``None`` + for any other spelling so a non-positional path is skipped, not silently mis-bound.""" + head, _, rest = spec.partition(".") + if not head.startswith("arg"): + return None + try: + position = int(head[3:]) + except ValueError: + return None + selectors = tuple(s for s in rest.split(".") if s) + return position, selectors + + +def _contracts_to_summaries(contracts, exclude): + """Convert manifest ``FunctionContract`` facts into engine summaries. + + A contract is ground truth a maintainer supplies for a function the engine cannot see + into (cross-TU / library). It is therefore seeded ONLY for names not in ``exclude`` + (the analyzable set): a function with a real body stays body-authoritative, so a + contract never overrides what the solver can observe directly. Each contract becomes + one summary alternative whose ``ParamEffect``s replay at every call site, exactly like + a computed summary. ``returns`` ownership is not a parameter effect and is handled at + the return-binding site, not here.""" + summaries = {} + for contract in contracts: + if contract.name in exclude: + continue + effects = [] + for kind, paths in ((OpKind.ALLOC, contract.allocs), + (OpKind.USE, contract.uses), + (OpKind.FREE, contract.frees)): + for spec in paths: + parsed = _parse_arg_path(spec) + if parsed is None: + continue + position, selectors = parsed + effects.append(ParamEffect(kind, position, selectors)) + if effects: + summaries[contract.name] = (tuple(effects),) + return summaries + + def analyze_object_lifetimes(store, functions, call_successors, *, lang="c", graph=None, - extra_alloc=(), extra_dealloc=(), max_disjuncts=None): + extra_alloc=(), extra_dealloc=(), max_disjuncts=None, + contracts=()): """Run object-identity lifetime analysis over all defined functions in ``functions``. ``extra_alloc`` / ``extra_dealloc`` are per-target manifest ``memory.alloc`` / ``memory.free`` names: they extend the lifecycle vocabulary this run recognizes so a project's own allocator/free wrappers emit typestate events. ``max_disjuncts`` overrides - the solver's per-node disjunct cap (``None`` keeps the tuned default).""" + the solver's per-node disjunct cap (``None`` keeps the tuned default). ``contracts`` are + manifest ``FunctionContract`` facts, seeded as summaries for opaque/cross-TU callees the + solver cannot analyze -- letting a free/use effect compose across a library boundary.""" started = perf_counter() disjunct_cap = _DEFAULT_MAX_DISJUNCTS if max_disjuncts is None else max_disjuncts if graph is not None and graph is not store.graph: @@ -491,7 +544,12 @@ def analyze_object_lifetimes(store, functions, call_successors, *, lang="c", gra # Absence means "no analyzable summary", not "proven to have no effects". That # distinction makes callers of a CFG failure take the conservative external-call # path instead of silently treating the callee as pure. - summaries = {} + # + # Manifest contracts seed the summary table for opaque callees (names with no body + # here). They persist through the worklist because no computed summary ever overwrites + # a name that was never analyzed -- so a maintainer-declared free/use composes across + # the library boundary exactly like a summary the solver derived itself. + summaries = _contracts_to_summaries(contracts, exclude=set(functions)) artifacts = {} summary_capped = set() summary_runs = Counter() diff --git a/lachesis/flow/pipeline.py b/lachesis/flow/pipeline.py index 12e4d69d..32f52797 100644 --- a/lachesis/flow/pipeline.py +++ b/lachesis/flow/pipeline.py @@ -85,20 +85,24 @@ def _match_object_mode_legacy(skels, cfg, fallback_entries): def _manifest_config(manifest): """Extract the run knobs a manifest contributes, and an audit of each. - Returns ``(engine, extra_alloc, extra_dealloc, max_disjuncts, applied)`` where - ``applied`` is a per-knob log so a run summary can show what the manifest changed + Returns ``(engine, extra_alloc, extra_dealloc, max_disjuncts, contracts, applied)`` + where ``applied`` is a per-knob log so a run summary can show what the manifest changed and -- critically -- what it declared that has no effect yet (no silent truncation: a config knob that quietly does nothing is exactly the backdoor a facts-file must not become).""" applied = {} if manifest is None: - return None, (), (), None, applied + return None, (), (), None, (), applied mem = manifest.project.memory analysis = manifest.analysis + contracts = manifest.project.functions if mem.alloc: applied["memory.alloc"] = f"+{len(mem.alloc)} allocator name(s): {list(mem.alloc)}" if mem.free: applied["memory.free"] = f"+{len(mem.free)} free name(s): {list(mem.free)}" + if contracts: + applied["functions"] = ( + f"{len(contracts)} function contract(s): {[c.name for c in contracts]}") if analysis.engine: applied["analysis.engine"] = f"engine={analysis.engine} (manifest)" if analysis.disjunct_cap is not None: @@ -109,7 +113,7 @@ def _manifest_config(manifest): applied["analysis.timeout_per_fn"] = ( f"declared {analysis.timeout_per_fn}s but NOT enforced " "(engine caps by disjunct/run count, no wall-clock timeout yet)") - return (analysis.engine, mem.alloc, mem.free, analysis.disjunct_cap, applied) + return (analysis.engine, mem.alloc, mem.free, analysis.disjunct_cap, contracts, applied) def run_pass(store, lang="c", lifetime_engine=None, manifest=None): @@ -134,7 +138,7 @@ def run_pass(store, lang="c", lifetime_engine=None, manifest=None): store.ensure_dataflow_tier() tier_done = perf_counter() (cfg_engine, extra_alloc, extra_dealloc, - cfg_disjunct, applied_config) = _manifest_config(manifest) + cfg_disjunct, contracts, applied_config) = _manifest_config(manifest) requested = lifetime_engine or cfg_engine or os.environ.get( "LACHESIS_LIFETIME_ENGINE", _DEFAULT_LIFETIME_ENGINE) if requested not in {"legacy", "shadow", "object"}: @@ -160,7 +164,7 @@ def run_pass(store, lang="c", lifetime_engine=None, manifest=None): object_result = analyze_object_lifetimes( store, F, succ, lang=lang, graph=analysis_graph, extra_alloc=extra_alloc, extra_dealloc=extra_dealloc, - max_disjuncts=cfg_disjunct) + max_disjuncts=cfg_disjunct, contracts=contracts) # The projection already paid to materialize the disk graph. Reuse that same # in-memory index for the legacy coverage fallback instead of issuing another # whole-graph set of Kuzu scans merely to project CFG edges. diff --git a/lachesis/flow/test_pipeline_manifest.py b/lachesis/flow/test_pipeline_manifest.py index 3180e734..c9b6e149 100644 --- a/lachesis/flow/test_pipeline_manifest.py +++ b/lachesis/flow/test_pipeline_manifest.py @@ -21,13 +21,17 @@ from lachesis.core.snapshot import load_snapshot from lachesis.manifest.schema import ( AnalysisConfig, + FunctionContract, Manifest, Memory, + Ownership, ProjectFacts, ) from lachesis.nav.graph_store import GraphStore from lachesis.pipeline import semantic_snapshot_graph +from .object_lifetime import _contracts_to_summaries, _parse_arg_path +from .object_state import OpKind, ParamEffect from .pipeline import _manifest_config, run_pass # A double-free routed entirely through a project-specific free wrapper. The libc @@ -88,8 +92,9 @@ def test_manifest_free_wrapper_makes_double_free_fire(self): def test_manifest_config_extraction_and_audit(self): # No manifest -> empty knobs, empty audit. - engine, alloc, dealloc, cap, applied = _manifest_config(None) - self.assertEqual((engine, alloc, dealloc, cap, applied), (None, (), (), None, {})) + engine, alloc, dealloc, cap, contracts, applied = _manifest_config(None) + self.assertEqual((engine, alloc, dealloc, cap, contracts, applied), + (None, (), (), None, (), {})) # A full analysis block: applied knobs are recorded; timeout is surfaced as # declared-but-unenforced rather than silently dropped. @@ -97,7 +102,7 @@ def test_manifest_config_extraction_and_audit(self): project=ProjectFacts(memory=Memory(alloc=("xmalloc",), free=("xfree",))), analysis=AnalysisConfig(engine="object", disjunct_cap=16, timeout_per_fn=30.0), ) - engine, alloc, dealloc, cap, applied = _manifest_config(manifest) + engine, alloc, dealloc, cap, contracts, applied = _manifest_config(manifest) self.assertEqual(engine, "object") self.assertEqual(alloc, ("xmalloc",)) self.assertEqual(dealloc, ("xfree",)) @@ -107,5 +112,75 @@ def test_manifest_config_extraction_and_audit(self): self.assertIn("NOT enforced", applied["analysis.timeout_per_fn"]) +# An opaque (declaration-only, cross-TU) free that releases its whole argument. Only a +# manifest contract can tell the engine what it does; without one the call is a blind USE. +CONTRACT_SOURCE = r""" +void *malloc(unsigned long); +void ext_release(void *p); + +void frees_via_opaque(void) { + char *p = malloc(8); + ext_release(p); + *p = 1; +} +""" + + +class ContractConversionTests(unittest.TestCase): + def test_parse_arg_path(self): + self.assertEqual(_parse_arg_path("arg0"), (0, ())) + self.assertEqual(_parse_arg_path("arg1.data"), (1, ("data",))) + self.assertEqual(_parse_arg_path("arg2.next.data"), (2, ("next", "data"))) + # non-positional forms are unresolvable and skipped, not mis-bound + self.assertIsNone(_parse_arg_path("value")) + self.assertIsNone(_parse_arg_path("argX")) + + def test_contract_to_summary_skips_analyzable_and_maps_effects(self): + contracts = ( + FunctionContract(name="ext_free_data", frees=("arg0.data",)), + FunctionContract(name="ext_use", uses=("arg0",)), + FunctionContract(name="local_fn", frees=("arg0",)), # excluded: has a body + ) + summaries = _contracts_to_summaries(contracts, exclude={"local_fn"}) + self.assertNotIn("local_fn", summaries) # body-authoritative, never overridden + self.assertEqual(summaries["ext_free_data"], + ((ParamEffect(OpKind.FREE, 0, ("data",)),),)) + self.assertEqual(summaries["ext_use"], ((ParamEffect(OpKind.USE, 0, ()),),)) + + +class ContractDrivenLifetimeTests(unittest.TestCase): + def test_opaque_free_contract_composes_use_after_free(self): + with tempfile.TemporaryDirectory() as source_dir, tempfile.TemporaryDirectory() as output: + Path(source_dir, "contract.c").write_text(CONTRACT_SOURCE) + completed = subprocess.run( + [sys.executable, "-m", "lachesis.frontends.c.build_graph", source_dir, output], + text=True, capture_output=True, check=False, + ) + self.assertEqual(completed.returncode, 0, completed.stderr) + snapshot = load_snapshot(output) + + # Baseline: ext_release is opaque, so the engine cannot know it frees p. + store = GraphStore(semantic_snapshot_graph(snapshot)) + baseline = run_pass(store, lang="c", lifetime_engine="object") + self.assertEqual( + _patterns_by_function(baseline)["frees_via_opaque"], set(), + "an opaque free must be invisible without a contract", + ) + + # A contract declaring ext_release frees arg0 composes across the boundary: + # the later *p = 1 becomes a use-after-free. + manifest = Manifest(project=ProjectFacts(functions=( + FunctionContract(name="ext_release", frees=("arg0",), + returns=Ownership.UNKNOWN), + ))) + store2 = GraphStore(semantic_snapshot_graph(snapshot)) + declared = run_pass(store2, lang="c", lifetime_engine="object", manifest=manifest) + self.assertEqual( + _patterns_by_function(declared)["frees_via_opaque"], {"use-after-free"}, + "the free contract must compose a cross-TU use-after-free", + ) + self.assertIn("functions", declared["lifetime"]["applied_config"]) + + if __name__ == "__main__": unittest.main() From 006b9c5d25a629e7e2acfc11c0017ea119c94931 Mon Sep 17 00:00:00 2001 From: Riyan Dhiman Date: Thu, 20 Aug 2026 08:53:09 +0530 Subject: [PATCH 6/7] P4b: manifest-aware run command and audited lead scoping --- examples/curl/lachesis.toml | 50 ++++++ lachesis/cli/main.py | 15 ++ lachesis/cli/manifest_run.py | 230 ++++++++++++++++++++++++ lachesis/flow/pipeline.py | 8 + lachesis/flow/test_pipeline_manifest.py | 76 ++++++++ lachesis/manifest/validate.py | 28 +++ 6 files changed, 407 insertions(+) create mode 100644 examples/curl/lachesis.toml create mode 100644 lachesis/cli/manifest_run.py diff --git a/examples/curl/lachesis.toml b/examples/curl/lachesis.toml new file mode 100644 index 00000000..b8d605d7 --- /dev/null +++ b/examples/curl/lachesis.toml @@ -0,0 +1,50 @@ +# Curl-shaped example manifest. Copy this beside a checkout and replace the symbols, +# paths, and graph with facts for the build variant you actually ship. +[project] +name = "curl" +language = "c" + +[project.source] +roots = ["lib", "src"] +exclude = ["tests", "**/vendor/**"] + +[project.build] +config = ["USE_OPENSSL", "ENABLE_IPV6"] +include = ["include", "lib"] +defines = { CURL_DISABLE_FTP = 0 } + +[project.memory] +alloc = ["curl_malloc", "Curl_saferealloc"] +free = ["Curl_safefree"] + +[project.surface] +entrypoints = ["curl_easy_perform"] +untrusted = [ + { fn = "Curl_read", at = "return" }, + { fn = "curl_easy_setopt", at = "arg2" }, +] + +[project.trust] +sanitizers = ["Curl_urldecode"] + +[project.functions.Curl_close] +frees = ["arg0.data"] +returns = "borrowed" + +[project.functions.Curl_dup] +returns = "owned" + +[project.alias] +noalias = [["Curl_easy.state", "Curl_easy.set"]] + +[project.dispatch] +"Curl_handler.disconnect" = "ossl_disconnect" + +[project.typedefs] +Curl_easy = "SessionHandle" + +[analysis] +engine = "object" +graph = "~/.lachesis/graphs/curl.kuzu" +disjunct_cap = 64 +timeout_per_fn = "30s" diff --git a/lachesis/cli/main.py b/lachesis/cli/main.py index 8d99901c..239660d2 100644 --- a/lachesis/cli/main.py +++ b/lachesis/cli/main.py @@ -28,6 +28,7 @@ EPILOG = """\ examples: + lachesis run run the checked-in lachesis.toml contract lachesis scan analyse the current directory and report findings lachesis scan ~/src/app --json the same, as JSON, for a script lachesis mcp serve the current directory to an AI agent @@ -275,6 +276,20 @@ def build_parser() -> argparse.ArgumentParser: help="print the installed version and exit") subcommands = root.add_subparsers(dest="command", metavar="") + run = subcommands.add_parser( + "run", help="run the project's checked-in lachesis.toml contract", + description="Load and validate lachesis.toml, run the flow pass, then report " + "applied configuration and any excluded coverage.") + _add_source_flags(run) + run.add_argument("--manifest", default=None, metavar="PATH", + help="manifest path (default: nearest lachesis.toml)") + run.add_argument("--json", action="store_true", + help="write the run summary and leads as JSON") + run.add_argument("--quiet", "-q", action="store_true", + help="suppress progress; the run summary is still printed") + from lachesis.cli.manifest_run import command_run + run.set_defaults(handler=command_run) + scan = subcommands.add_parser( "scan", help="report what an attacker could reach in a codebase", description="Index the tree if needed, then rank the reachable sensitive " diff --git a/lachesis/cli/manifest_run.py b/lachesis/cli/manifest_run.py new file mode 100644 index 00000000..e816dbda --- /dev/null +++ b/lachesis/cli/manifest_run.py @@ -0,0 +1,230 @@ +"""Manifest-aware end-to-end flow runner used by ``lachesis run``. + +This is deliberately a product-layer adapter. The graph builder remains unchanged: +the manifest is loaded beside the target, graph-checkable facts are validated, the +existing flow pass consumes the facts it understands, and source exclusions are +applied only after a lead's entry function has been resolved back to its file. +""" +from __future__ import annotations + +import fnmatch +import json +from pathlib import Path + +from lachesis.manifest.loader import discover_manifest, load_manifest +from lachesis.manifest.validate import validate_manifest + + +def _manifest_path(start: Path, explicit: str | None) -> Path: + if explicit: + path = Path(explicit).expanduser() + if not path.is_absolute(): + path = start / path + path = path.resolve() + else: + path = discover_manifest(start) + if path is None: + raise ValueError( + f"no lachesis.toml found at or above {start}; add one or pass --manifest" + ) + return path + + +def _graph_path(value: str, manifest_path: Path) -> Path: + path = Path(value).expanduser() + if not path.is_absolute(): + path = manifest_path.parent / path + return path.resolve() + + +def _relative_file(file: str, project_root: Path) -> str: + path = Path(file) + if path.is_absolute(): + try: + path = path.resolve().relative_to(project_root) + except ValueError: + return path.as_posix() + return path.as_posix().lstrip("./") + + +def _matches_exclude(file: str, patterns: tuple[str, ...]) -> bool: + """Match both directory shorthand (``tests``) and portable glob spelling. + + ``fnmatch`` does not make a leading ``**/`` optional, so test both the declared + pattern and its root-level form. This makes ``**/vendor/**`` cover ``vendor/x.c`` + as well as ``lib/vendor/x.c`` without introducing a filesystem walk. + """ + file = file.replace("\\", "/").lstrip("./") + for raw in patterns: + pattern = raw.replace("\\", "/").strip().lstrip("./").rstrip("/") + if not pattern: + continue + if not any(mark in pattern for mark in "*?["): + if file == pattern or file.startswith(pattern + "/"): + return True + continue + variants = {pattern} + if pattern.startswith("**/"): + variants.add(pattern[3:]) + if any(fnmatch.fnmatchcase(file, variant) for variant in variants): + return True + return False + + +def _entry_files(store, entry: str, project_root: Path) -> list[str]: + candidates = [item for item in store.resolve(entry) + if item.get("name") == entry and item.get("file")] + definitions = [item for item in candidates if not item.get("declaration_only")] + chosen = definitions or candidates + return sorted({_relative_file(str(item["file"]), project_root) for item in chosen}) + + +def scope_leads(leads, store, project_root: Path, exclude: tuple[str, ...]): + """Attach entry files and return ``(kept, excluded)``. + + A homonymous entry may resolve to more than one file. Such a lead is excluded only + when every possible definition is excluded; uncertainty must not suppress signal. + """ + kept, dropped = [], [] + for original in leads: + lead = dict(original) + files = _entry_files(store, str(lead.get("entry", "")), project_root) + if files: + lead["files"] = files + if len(files) == 1: + lead["file"] = files[0] + if files and all(_matches_exclude(file, exclude) for file in files): + dropped.append(lead) + else: + kept.append(lead) + return kept, dropped + + +def _report_dict(report) -> dict: + def item(check): + return {"location": check.location, "symbol": check.symbol, + "status": check.status.value, "detail": check.detail} + return { + "validated": len(report.validated), + "external": len(report.external), + "warnings": len(report.warnings), + "checks": [item(check) for check in report.checks], + } + + +def execute_manifest_run(source: Path, manifest_path: Path, graph_path: Path, store) -> dict: + """Validate and run an already-opened store; split out for focused tests.""" + from lachesis.flow.pipeline import run_pass + + manifest = load_manifest(manifest_path) + validation = validate_manifest(manifest, store) + bundle = run_pass(store, lang=manifest.project.language, manifest=manifest) + exclude = manifest.project.source.exclude + leads, excluded = scope_leads( + bundle["leads"], store, manifest_path.parent.resolve(), exclude) + + applied = bundle["lifetime"].setdefault("applied_config", {}) + applied["analysis.graph"] = str(graph_path) + if exclude: + applied["project.source.exclude"] = ( + f"{list(exclude)} excluded {len(excluded)} of " + f"{len(leads) + len(excluded)} lead(s) after entry-to-file resolution" + ) + + diagnostics = bundle["lifetime"].get("diagnostics", {}) + semantic = bundle["lifetime"].get("semantic_warnings", []) + summary = { + "project": manifest.project.name or source.name, + "manifest": str(manifest_path), + "graph": str(graph_path), + "language": manifest.project.language, + "functions": len(bundle["F"]), + "skeletons": len(bundle["skeletons"]), + "leads": len(leads), + "excluded_leads": len(excluded), + "excluded_patterns": list(exclude), + "applied_config": dict(applied), + "manifest_validation": _report_dict(validation), + "semantic_warnings": semantic, + "coverage": { + key: diagnostics[key] for key in ( + "functions", "analyzed", "capped", "summary_capped", + "cfg_failures", "unplaced", "unsafe_functions", + ) if key in diagnostics + }, + "timings": bundle.get("timings", {}), + } + return {"run_summary": summary, "leads": leads} + + +def command_run(args) -> int: + """CLI handler. Imports the Kuzu-backed pieces lazily for loader-only installs.""" + from lachesis.cli.indexer import (EnvironmentProblem, NoSourceFound, + ensure_graph) + from lachesis.cli.main import (_report_environment, _stderr, EXIT_OK, + EXIT_USAGE) + from lachesis.cli.progress import Progress + from lachesis.nav.graph_store import GraphStore + + source = Path(args.path or ".").expanduser().resolve() + manifest_path = _manifest_path(source, args.manifest) + manifest = load_manifest(manifest_path) + progress = Progress(enabled=not args.quiet) + if not args.quiet: + _stderr(f"lachesis run: {source}") + _stderr(f"manifest: {manifest_path}") + + if manifest.analysis.graph: + graph_path = _graph_path(manifest.analysis.graph, manifest_path) + if not graph_path.exists(): + raise ValueError(f"analysis.graph does not exist: {graph_path}") + else: + try: + graph_path, _ = ensure_graph( + source, refresh=args.refresh, progress=progress, + timeout_seconds=args.timeout) + except EnvironmentProblem as error: + return _report_environment(error) + except NoSourceFound as error: + _stderr(f"lachesis run: {error}") + return EXIT_USAGE + + payload = execute_manifest_run( + source, manifest_path, graph_path, GraphStore.load(str(graph_path))) + if args.json: + print(json.dumps(payload, indent=2, ensure_ascii=False)) + else: + _render(payload) + return EXIT_OK + + +def _render(payload: dict) -> None: + summary = payload["run_summary"] + validation = summary["manifest_validation"] + print( + f"run summary: {summary['leads']} lead(s) over {summary['functions']} " + f"function(s); {summary['excluded_leads']} excluded" + ) + print( + f"manifest: {validation['validated']} grounded, " + f"{validation['external']} external, {validation['warnings']} warning(s)" + ) + for check in validation["checks"]: + if check["status"] == "warning": + print(f" ! {check['location']} '{check['symbol']}' — {check['detail']}") + for warning in summary["semantic_warnings"]: + print(f" ! {warning['location']} '{warning['symbol']}' — {warning['detail']}") + print("applied config:") + if summary["applied_config"]: + for key, value in summary["applied_config"].items(): + print(f" {key}: {value}") + else: + print(" (defaults)") + coverage = summary["coverage"] + if coverage: + print("coverage: " + ", ".join(f"{key}={value}" for key, value in coverage.items())) + for lead in payload["leads"]: + where = lead.get("file") or ",".join(lead.get("files", ())) or lead.get("entry", "?") + if lead.get("line") is not None: + where += f":{lead['line']}" + print(f" {lead['pattern']}: {where} ({lead.get('entry', '?')})") diff --git a/lachesis/flow/pipeline.py b/lachesis/flow/pipeline.py index 32f52797..fcd7c60d 100644 --- a/lachesis/flow/pipeline.py +++ b/lachesis/flow/pipeline.py @@ -174,6 +174,14 @@ def run_pass(store, lang="c", lifetime_engine=None, manifest=None): else: fallback_store = store diagnostics = object_result.diagnostics + if manifest is not None: + from lachesis.manifest.validate import validate_contract_effects + effect_report = validate_contract_effects(manifest, object_result.summaries) + lifetime["semantic_warnings"] = [ + {"location": check.location, "symbol": check.symbol, + "status": check.status.value, "detail": check.detail} + for check in effect_report.warnings + ] unsafe = set(diagnostics.get("unsafe_functions", ())) # Object mode is fully untrusted only where the function's OWN analysis failed # (seed-unsafe); propagation-only-unsafe functions keep their object leads and are diff --git a/lachesis/flow/test_pipeline_manifest.py b/lachesis/flow/test_pipeline_manifest.py index c9b6e149..8596cc71 100644 --- a/lachesis/flow/test_pipeline_manifest.py +++ b/lachesis/flow/test_pipeline_manifest.py @@ -19,6 +19,7 @@ from pathlib import Path from lachesis.core.snapshot import load_snapshot +from lachesis.cli.manifest_run import execute_manifest_run from lachesis.manifest.schema import ( AnalysisConfig, FunctionContract, @@ -182,5 +183,80 @@ def test_opaque_free_contract_composes_use_after_free(self): self.assertIn("functions", declared["lifetime"]["applied_config"]) +RUN_SOURCE = r""" +void *malloc(unsigned long); +void free(void *); + +void kept_double_free(void) { + char *p = malloc(8); + free(p); + free(p); +} + +void claims_to_free(char *p) { + (void)p; +} +""" + +EXCLUDED_SOURCE = r""" +void *malloc(unsigned long); +void free(void *); + +void excluded_double_free(void) { + char *p = malloc(8); + free(p); + free(p); +} +""" + +RUN_MANIFEST = r""" +[project] +name = "manifest-run-proof" +language = "c" + +[project.source] +exclude = ["tests", "**/vendor/**"] + +[project.functions.claims_to_free] +frees = ["arg0"] + +[analysis] +engine = "object" +""" + + +class ManifestRunTests(unittest.TestCase): + def test_end_to_end_summary_scopes_excludes_and_warns_on_body_contradiction(self): + with tempfile.TemporaryDirectory() as source_dir, tempfile.TemporaryDirectory() as output: + Path(source_dir, "src").mkdir() + Path(source_dir, "tests").mkdir() + Path(source_dir, "src", "kept.c").write_text(RUN_SOURCE) + Path(source_dir, "tests", "excluded.c").write_text(EXCLUDED_SOURCE) + manifest_path = Path(source_dir, "lachesis.toml") + manifest_path.write_text(RUN_MANIFEST) + completed = subprocess.run( + [sys.executable, "-m", "lachesis.frontends.c.build_graph", source_dir, output], + text=True, capture_output=True, check=False, + ) + self.assertEqual(completed.returncode, 0, completed.stderr) + snapshot = load_snapshot(output) + store = GraphStore(semantic_snapshot_graph(snapshot)) + + payload = execute_manifest_run( + Path(source_dir), manifest_path, Path(output), store) + summary = payload["run_summary"] + + entries = {lead["entry"] for lead in payload["leads"]} + self.assertIn("kept_double_free", entries) + self.assertNotIn("excluded_double_free", entries) + self.assertGreaterEqual(summary["excluded_leads"], 1) + self.assertIn("project.source.exclude", summary["applied_config"]) + self.assertEqual(summary["manifest_validation"]["warnings"], 0) + self.assertEqual( + [warning["symbol"] for warning in summary["semantic_warnings"]], + ["claims_to_free"], + ) + + if __name__ == "__main__": unittest.main() diff --git a/lachesis/manifest/validate.py b/lachesis/manifest/validate.py index 776b85e8..fcd5e153 100644 --- a/lachesis/manifest/validate.py +++ b/lachesis/manifest/validate.py @@ -151,3 +151,31 @@ def validate_manifest(manifest: Manifest, store) -> ManifestReport: _check(store, checks, f"project.typedefs[{alias}]", struct, TYPE_KINDS) return ManifestReport(checks=tuple(checks)) + + +def validate_contract_effects(manifest: Manifest, summaries) -> ManifestReport: + """Warn when a body-visible function contract claims a free the solver never saw. + + Opaque functions are intentionally absent from ``summaries`` and remain unverifiable. + For a visible body we keep the check conservative: any observed free effect satisfies + the declaration. Exact access-path equivalence is a deeper claim than the current + summary representation can prove reliably. + """ + checks = [] + for contract in manifest.project.functions: + if not contract.frees or contract.name not in summaries: + continue + alternatives = summaries.get(contract.name) or () + observed = any( + getattr(getattr(effect, "kind", None), "value", + getattr(effect, "kind", None)) == "free" + for alternative in alternatives for effect in alternative + ) + if not observed: + checks.append(FactCheck( + location=f"project.functions.{contract.name}.frees", + symbol=contract.name, + status=Status.WARNING, + detail="contract declares a free, but the analyzed body has no free effect", + )) + return ManifestReport(checks=tuple(checks)) From 38a79e779546a3493654e49da311c2a7b624501f Mon Sep 17 00:00:00 2001 From: Riyan Dhiman Date: Thu, 20 Aug 2026 12:59:24 +0530 Subject: [PATCH 7/7] draft --- lachesis/flow/pipeline.py | 40 ++++++++++++- lachesis/flow/test_pipeline_manifest.py | 77 +++++++++++++++++++++++++ lachesis/flow/translate.py | 9 ++- 3 files changed, 120 insertions(+), 6 deletions(-) diff --git a/lachesis/flow/pipeline.py b/lachesis/flow/pipeline.py index fcd7c60d..e435957a 100644 --- a/lachesis/flow/pipeline.py +++ b/lachesis/flow/pipeline.py @@ -18,6 +18,35 @@ _DEFAULT_LIFETIME_ENGINE = "object" +def _apply_owned_return_contracts(F, contracts): + """Turn opaque ``returns=owned`` facts into alloc events on assigned results. + + Function bodies remain authoritative: a contract whose name is defined in ``F`` is + ignored here, matching the object summary rule. The graph itself is untouched; this + augments only the per-run flow IR consumed by summaries and skeletons. + """ + owned = { + contract.name for contract in contracts + if getattr(contract.returns, "value", contract.returns) == "owned" + and contract.name not in F + } + if not owned: + return () + for function in F.values(): + events = function["events"] + seen = {(event.get("kind"), event.get("var"), event.get("node")) + for event in events} + for assign in function.get("assigns", ()): + if assign.get("callee") not in owned: + continue + key = ("alloc", assign.get("var"), assign.get("node")) + if key not in seen: + events.append({"kind": "alloc", "var": assign.get("var"), + "line": assign.get("line"), "node": assign.get("node")}) + seen.add(key) + return tuple(sorted(owned)) + + def _lead_key(lead): # The two engines encode object display names differently; a differential is # about whether they agree on the finding site, not renderer spelling. @@ -146,10 +175,14 @@ def run_pass(store, lang="c", lifetime_engine=None, manifest=None): "lifetime engine must be one of legacy, shadow, or object") object_requested = lang.lower() == "c" and requested != "legacy" if object_requested: - F, succ, analysis_graph = build_F(store, lang=lang, return_graph=True) + F, succ, analysis_graph = build_F( + store, lang=lang, return_graph=True, + extra_alloc=extra_alloc, extra_dealloc=extra_dealloc) else: - F, succ = build_F(store, lang=lang) + F, succ = build_F( + store, lang=lang, extra_alloc=extra_alloc, extra_dealloc=extra_dealloc) analysis_graph = None + owned_alloc = _apply_owned_return_contracts(F, contracts) projection_done = perf_counter() summaries = _summaries_for(F, succ) legacy_summaries_done = perf_counter() @@ -163,7 +196,8 @@ def run_pass(store, lang="c", lifetime_engine=None, manifest=None): if object_requested: object_result = analyze_object_lifetimes( store, F, succ, lang=lang, graph=analysis_graph, - extra_alloc=extra_alloc, extra_dealloc=extra_dealloc, + extra_alloc=tuple(extra_alloc) + owned_alloc, + extra_dealloc=extra_dealloc, max_disjuncts=cfg_disjunct, contracts=contracts) # The projection already paid to materialize the disk graph. Reuse that same # in-memory index for the legacy coverage fallback instead of issuing another diff --git a/lachesis/flow/test_pipeline_manifest.py b/lachesis/flow/test_pipeline_manifest.py index 8596cc71..ccf3d573 100644 --- a/lachesis/flow/test_pipeline_manifest.py +++ b/lachesis/flow/test_pipeline_manifest.py @@ -91,6 +91,83 @@ def test_manifest_free_wrapper_makes_double_free_fire(self): self.assertIn("memory.free", applied) self.assertIn("my_release", applied["memory.free"]) + def test_manifest_allocator_makes_unreleased_assignment_a_leak(self): + source = r""" +void *xmalloc(unsigned long); +void unreleased_custom_alloc(void) { + char *p = xmalloc(8); + (void)p; +} +""" + with tempfile.TemporaryDirectory() as source_dir, tempfile.TemporaryDirectory() as output: + Path(source_dir, "alloc.c").write_text(source) + completed = subprocess.run( + [sys.executable, "-m", "lachesis.frontends.c.build_graph", source_dir, output], + text=True, capture_output=True, check=False, + ) + self.assertEqual(completed.returncode, 0, completed.stderr) + snapshot = load_snapshot(output) + + baseline = run_pass( + GraphStore(semantic_snapshot_graph(snapshot)), + lang="c", lifetime_engine="object") + self.assertFalse(any( + lead["pattern"] == "leak" + and lead["entry"] == "unreleased_custom_alloc" + for lead in baseline["leads"]), + "an unknown allocator must not manufacture an allocation origin", + ) + + manifest = Manifest(project=ProjectFacts(memory=Memory(alloc=("xmalloc",)))) + declared = run_pass( + GraphStore(semantic_snapshot_graph(snapshot)), + lang="c", lifetime_engine="object", manifest=manifest) + self.assertTrue(any( + lead["pattern"] == "leak" + and lead["entry"] == "unreleased_custom_alloc" + for lead in declared["leads"]), + "declaring xmalloc must emit an alloc event and expose the leak", + ) + + def test_owned_return_contract_makes_unreleased_result_a_leak(self): + source = r""" +void *external_owned(void); +void unreleased_owned_result(void) { + char *p = external_owned(); + (void)p; +} +""" + with tempfile.TemporaryDirectory() as source_dir, tempfile.TemporaryDirectory() as output: + Path(source_dir, "owned.c").write_text(source) + completed = subprocess.run( + [sys.executable, "-m", "lachesis.frontends.c.build_graph", source_dir, output], + text=True, capture_output=True, check=False, + ) + self.assertEqual(completed.returncode, 0, completed.stderr) + snapshot = load_snapshot(output) + + baseline = run_pass( + GraphStore(semantic_snapshot_graph(snapshot)), + lang="c", lifetime_engine="object") + self.assertFalse(any( + lead["pattern"] == "leak" + and lead["entry"] == "unreleased_owned_result" + for lead in baseline["leads"]), + ) + + manifest = Manifest(project=ProjectFacts(functions=( + FunctionContract(name="external_owned", returns=Ownership.OWNED), + ))) + declared = run_pass( + GraphStore(semantic_snapshot_graph(snapshot)), + lang="c", lifetime_engine="object", manifest=manifest) + self.assertTrue(any( + lead["pattern"] == "leak" + and lead["entry"] == "unreleased_owned_result" + for lead in declared["leads"]), + "an opaque owned return must seed caller ownership", + ) + def test_manifest_config_extraction_and_audit(self): # No manifest -> empty knobs, empty audit. engine, alloc, dealloc, cap, contracts, applied = _manifest_config(None) diff --git a/lachesis/flow/translate.py b/lachesis/flow/translate.py index 7cbb3177..a5495443 100644 --- a/lachesis/flow/translate.py +++ b/lachesis/flow/translate.py @@ -43,7 +43,7 @@ from lachesis.planner.unbounded_copy import BranchRegions from . import atropos -from .normalize import normalizer +from .normalize import normalizer_with # --- small helpers --------------------------------------------------------------- @@ -346,7 +346,7 @@ def _walk_function(ix, regions, nest, sinks, norm, fnode): "assigns": assigns, "returns": returns, "callees": callees} -def build_F(store, lang="c", *, return_graph=False): +def build_F(store, lang="c", *, return_graph=False, extra_alloc=(), extra_dealloc=()): """Build the whole-graph F dict + succ (callee-edge) map from an enriched store. Reproduces order.load's return so the pass is input-source agnostic. Taxonomy / @@ -367,7 +367,10 @@ def build_F(store, lang="c", *, return_graph=False): nest = ControlNesting(graph) # loop/branch nesting from AST containment sinks = atropos.sink_catalog(lang) sink_names = set(sinks) - norm = normalizer(lang) # form oracle: canonicalize callee names + # Manifest lifecycle names are a run-local extension. ``normalizer_with`` returns + # the shared catalog instance when both are empty, and an uncached instance otherwise, + # so one target's vocabulary can never leak into another analysis. + norm = normalizer_with(lang, extra_alloc, extra_dealloc) fnodes = list(ix.nodes_of_kind("function", "method", "constructor")) defined = {f.get("label") for f in fnodes if not _props(f).get("declaration_only")}