feat(reflect): work-memory overlay — surface learned verdicts as a graph sidecar (#1441)

Projects the verdicts `graphify reflect` already distills (preferred /
tentative / contested, exponential time-decayed) into a derived
experiential layer the read surfaces consume, so accumulated agent
experience actually shows up where you look — without polluting the
structural graph.

Design (grounded in agent-memory + provenance literature; a redesign of
the #1542 approach):
- SIDECAR, not graph.json stamping. `reflect` writes `.graphify_learning.json`
  next to graph.json (an additional output, so the git hooks produce it
  automatically). graph.json stays purely structural; nothing leaks into
  GraphML; no graph.json churn. Mirrors the named-graph / event-sourcing
  separation of durable truth from a derived layer.
- Reuses the existing reflect aggregate (its `_decay` is the
  recency-weighted exponential model; `_finalize_sources` the
  classification) — no new scoring.
- PROVENANCE: each verdict carries the source questions/dates that produced
  it (cap 5, most-recent first).
- STALENESS: each verdict stores the node's file fingerprint; on read, a
  changed source file flags the verdict stale ("code changed since —
  re-verify") rather than presenting a confident lesson on rewritten code.
- CONTESTED surfaced distinctly (useful N / dead-end M), not averaged away.
- DEAD-ENDS stay QUERY-SCOPED — never a node-level status; they appear only
  in the report as question -> nodes.
- Read surfaces (explain / query+MCP / GRAPH_REPORT / graph.html) merge the
  overlay at read time, sanitized; un-annotated graphs are byte-identical.

Deferred (logged): letting verdicts influence query/seed traversal — the
recommender feedback-loop / Matthew-effect risk means that needs
propensity correction + exploration, not naive biasing.

Builds on the idea in #1441/#1542 (thanks @TPAteeq).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
safishamsi
2026-06-30 11:19:54 +01:00
co-authored by Claude Opus 4.8
parent 0792b419fc
commit 5779767fd3
11 changed files with 804 additions and 9 deletions
+27 -1
View File
@@ -3186,6 +3186,30 @@ def main() -> None:
)
print(f" Type: {d.get('file_type', '')}")
print(f" Community: {d.get('community_name') or d.get('community', '')}")
# Work-memory overlay: a derived experiential hint from `graphify reflect`,
# merged in display-only from the .graphify_learning.json sidecar next to
# graph.json. No line when the node has no overlay entry.
try:
from graphify.reflect import load_learning_overlay as _llo
from graphify.security import sanitize_label as _sl
_overlay = _llo(gp)
_entry = _overlay.get(str(nid))
if _entry:
_status = _sl(str(_entry.get("status", "")))
if _status == "contested":
_line = (f" Lesson: contested (useful {_entry.get('uses', 0)} / "
f"dead-end {_entry.get('neg', 0)})")
elif _status == "preferred":
_line = (f" Lesson: preferred source (start here) — "
f"{_entry.get('uses', 0)} useful, score={_entry.get('score', 0)}")
else:
_line = (f" Lesson: {_status or 'tentative'} — "
f"{_entry.get('uses', 0)} useful, score={_entry.get('score', 0)}")
if _entry.get("stale"):
_line += " [code changed since — re-verify]"
print(_line)
except Exception:
pass
print(f" Degree: {G.degree(nid)}")
from graphify.build import edge_data
connections: list[tuple[str, str, dict]] = [] # (direction, neighbor_id, edge_data)
@@ -3529,10 +3553,12 @@ def main() -> None:
tokens = {"input": 0, "output": 0}
from graphify.export import _git_head as _gh
_commit = _gh()
from graphify.report import load_learning_for_report as _llfr
report = generate(G, communities, cohesion, labels, gods, surprises,
{"warning": "cluster-only mode — file stats not available"},
tokens, str(watch_path), suggested_questions=questions,
min_community_size=min_community_size, built_at_commit=_commit)
min_community_size=min_community_size, built_at_commit=_commit,
learning=_llfr(out / "graph.json"))
(out / "GRAPH_REPORT.md").write_text(report, encoding="utf-8")
stages.mark("report")
from graphify.export import backup_if_protected as _backup
+49 -2
View File
@@ -636,6 +636,7 @@ def to_html(
community_labels: dict[int, str] | None = None,
member_counts: dict[int, int] | None = None,
node_limit: int | None = None,
learning_overlay: dict | None = None,
) -> None:
"""Generate an interactive vis.js HTML visualization of the graph.
@@ -713,6 +714,21 @@ def to_html(
max_deg = max(degree.values(), default=1) or 1
max_mc = (max(member_counts.values(), default=1) or 1) if member_counts else 1
# Work-memory overlay (derived sidecar). When not passed explicitly, load it
# best-effort from the sibling .graphify_learning.json next to the output
# graph.html (which lives beside graph.json). Empty/missing => no learning
# fields, so the un-annotated render is byte-identical to pre-feature.
if learning_overlay is None:
learning_overlay = {}
try:
from graphify.reflect import load_learning_overlay as _llo
learning_overlay = _llo(Path(output_path))
except Exception:
learning_overlay = {}
# Status -> ring color. preferred=green, contested=amber. Tentative gets no
# ring (it's not yet trustworthy enough to highlight in the map).
_RING = {"preferred": "#22c55e", "contested": "#f59e0b"}
# Build nodes list for vis.js
vis_nodes = []
for node_id, data in G.nodes(data=True):
@@ -728,7 +744,7 @@ def to_html(
size = 10 + 30 * (deg / max_deg)
# Only show label for high-degree nodes by default; others show on hover
font_size = 12 if deg >= max_deg * 0.15 else 0
vis_nodes.append({
node = {
"id": node_id,
"label": label,
"color": {"background": color, "border": color, "highlight": {"background": "#ffffff", "border": color}},
@@ -740,7 +756,38 @@ def to_html(
"source_file": sanitize_label(str(data.get("source_file") or "")),
"file_type": data.get("file_type", ""),
"degree": deg,
})
}
# Conditional learning fields — only present for annotated nodes, so
# un-annotated output keeps the exact pre-feature node dict shape.
entry = learning_overlay.get(str(node_id)) if learning_overlay else None
if entry:
status = sanitize_label(str(entry.get("status", "")))
stale = bool(entry.get("stale"))
node["learning_status"] = status
node["learning_stale"] = stale
ring = _RING.get(status)
if ring:
# Status-colored ring via the border; stale => desaturated +
# dashed (vis.js supports per-node `shapeProperties.borderDashes`).
if stale:
ring = "#9ca3af"
node["shapeProperties"] = {"borderDashes": [4, 4]}
node["borderWidth"] = 3
node["color"] = {
"background": color, "border": ring,
"highlight": {"background": "#ffffff", "border": ring},
}
# Lesson line appended to the hover title.
if status == "contested":
lesson = f"Lesson: contested (useful {entry.get('uses', 0)} / dead-end {entry.get('neg', 0)})"
elif status == "preferred":
lesson = f"Lesson: preferred source ({entry.get('uses', 0)} useful, score={entry.get('score', 0)})"
else:
lesson = f"Lesson: {status} ({entry.get('uses', 0)} useful)"
if stale:
lesson += " [code changed — re-verify]"
node["title"] = _html.escape(label) + "\n" + _html.escape(sanitize_label(lesson))
vis_nodes.append(node)
# Build edges list. Restore original edge direction from _src/_tgt
# (stashed by build.py for exactly this reason): undirected NetworkX
+264 -2
View File
@@ -38,6 +38,14 @@ from graphify.ingest import OUTCOMES
_UNCATEGORIZED = "Uncategorized"
# Derived experiential layer written alongside graph.json (a SIDECAR, kept
# separate from the durable structural truth in graph.json — no learning_*
# fields are ever stamped into the graph itself). Read-surface annotations are
# merged in at display time from this file.
LEARNING_SIDECAR_NAME = ".graphify_learning.json"
_LEARNING_SCHEMA_VERSION = 1
_PROVENANCE_CAP = 5 # most-recent (question, date, outcome) entries per node
# Scoring defaults (both exposed as CLI flags).
_DEFAULT_HALF_LIFE_DAYS = 30.0 # a signal's weight halves every 30 days
_DEFAULT_MIN_CORROBORATION = 2 # distinct useful results needed to "prefer" a node
@@ -289,13 +297,18 @@ def _empty_bucket() -> dict[str, Any]:
"node_neg": Counter(),
# node -> most recent event date seen (for the contested verdict line)
"node_last": {},
# node -> list of (date, question, outcome) for useful/corrected citations.
# Feeds the sidecar overlay's per-node provenance; never read by LESSONS.md,
# so it doesn't touch the aggregate's public shape.
"node_provenance": {},
"dead_ends": [],
"corrections": [],
}
def _record_node(bucket: dict[str, Any], node: str, sign: int,
weight: float, date: str) -> None:
weight: float, date: str, *, outcome: str | None = None,
question: str = "") -> None:
bucket["node_score"][node] = bucket["node_score"].get(node, 0.0) + sign * weight
if sign > 0:
bucket["node_pos"][node] += 1
@@ -303,6 +316,11 @@ def _record_node(bucket: dict[str, Any], node: str, sign: int,
bucket["node_neg"][node] += 1
if date > bucket["node_last"].get(node, ""):
bucket["node_last"][node] = date
# Provenance: only useful/corrected events are recorded (the experiential
# trail an agent cares about — what cited this node, and how it turned out).
if outcome in ("useful", "corrected"):
bucket["node_provenance"].setdefault(node, []).append(
(date, question, outcome))
def _finalize_sources(bucket: dict[str, Any],
@@ -382,7 +400,8 @@ def aggregate_lessons(docs: list[dict[str, Any]],
target["counts"][outcome if outcome in OUTCOMES else "unmarked"] += 1
if sign:
for n in nodes:
_record_node(target, n, sign, weight, date)
_record_node(target, n, sign, weight, date,
outcome=outcome, question=doc.get("question", ""))
if outcome == "dead_end":
target["dead_ends"].append(
{"question": doc.get("question", ""), "nodes": nodes, "date": date})
@@ -411,6 +430,10 @@ def aggregate_lessons(docs: list[dict[str, Any]],
"dead_ends": _dedupe_by_question(overall["dead_ends"]),
"corrections": _dedupe_by_question(overall["corrections"]),
"by_community": community_out,
# Private: per-node (date, question, outcome) trail for the sidecar
# overlay's provenance. Underscore-prefixed and not rendered by
# render_lessons_md, so the public aggregate shape is unchanged.
"_node_provenance": overall["node_provenance"],
}
@@ -574,4 +597,243 @@ def reflect(memory_dir: Path, out_path: Path,
out_path = Path(out_path)
out_path.parent.mkdir(parents=True, exist_ok=True)
out_path.write_text(render_lessons_md(agg), encoding="utf-8")
# Also project a derived experiential sidecar next to graph.json when a graph
# is in hand. Best-effort: a sidecar failure must never break LESSONS.md.
if graph_path is not None:
try:
write_learning_sidecar(agg, Path(graph_path), now=now)
except Exception:
pass
return out_path, agg
# --- work-memory overlay sidecar ------------------------------------------------
#
# A derived, experiential projection of the reflect aggregate, written next to
# graph.json as ``.graphify_learning.json``. It carries which nodes have proven
# preferred/tentative/contested, a code fingerprint for staleness detection, and
# a short provenance trail. graph.json (durable structural truth) is never
# touched — read surfaces merge this overlay in only at display time.
def _build_id_label_maps(graph_path: Path) -> tuple[dict[str, str], dict[str, list[str]],
dict[str, dict[str, Any]]]:
"""From graph.json build:
- ``id_set``: id -> id (every node id, so an id-form citation resolves to itself)
- ``label_to_ids``: label -> [ids] (so a label-form citation can be resolved,
and ambiguity — one label, many ids — can be detected and skipped)
- ``node_by_id``: id -> node dict (for source_file lookup)
Best-effort; an unreadable/garbage graph yields empty maps.
"""
id_set: dict[str, str] = {}
label_to_ids: dict[str, list[str]] = {}
node_by_id: dict[str, dict[str, Any]] = {}
try:
data = json.loads(Path(graph_path).read_text(encoding="utf-8"))
except (OSError, ValueError):
return id_set, label_to_ids, node_by_id
for n in data.get("nodes", []):
if not isinstance(n, dict) or n.get("id") is None:
continue
nid = str(n["id"])
id_set[nid] = nid
node_by_id[nid] = n
label = n.get("label")
if label is not None:
label_to_ids.setdefault(str(label), []).append(nid)
return id_set, label_to_ids, node_by_id
def _resolve_canonical_id(cited: str, id_set: dict[str, str],
label_to_ids: dict[str, list[str]]) -> str | None:
"""Resolve a cited node (a label OR an id) to a single canonical node id.
Returns None if the citation is unresolved (stale — gone from the graph) or
ambiguous (a label shared by >1 node id). Such citations can't be displayed
against a single node, so the caller skips them.
"""
if cited in id_set:
return id_set[cited]
ids = label_to_ids.get(cited)
if ids and len(ids) == 1:
return ids[0]
return None
def _code_fingerprint(node: dict[str, Any] | None, root: Path) -> str:
"""File-level content hash of the node's ``source_file``, or '' if unavailable.
Coarse on purpose — a file-level hash over-flags (any edit to the file marks
every node in it stale) rather than under-flags, which is the safe direction
for a "re-verify" hint.
"""
if not node:
return ""
src = node.get("source_file")
if not src:
return ""
try:
from graphify.cache import file_hash
p = Path(src)
if not p.is_absolute():
p = (Path(root) / p)
return file_hash(p, root)
except Exception:
return ""
def _provenance_for(node: str, prov_map: dict[str, list],
fallback_outcome: str) -> list[dict[str, str]]:
"""Most-recent-first, capped provenance entries for a node.
``prov_map`` is the aggregate's private per-node (date, question, outcome)
trail. ``fallback_outcome`` covers an entry with no recorded trail (shouldn't
happen for preferred/tentative/contested, which all have ≥1 positive event).
"""
events = prov_map.get(node, [])
# Sort recent-first; (date desc, then question for a stable tiebreak).
ordered = sorted(events, key=lambda e: (e[0], e[1]), reverse=True)
out: list[dict[str, str]] = []
for date, question, outcome in ordered[:_PROVENANCE_CAP]:
out.append({"q": question, "date": date, "outcome": outcome})
return out
def build_learning_overlay(agg: dict[str, Any], graph_path: Path,
*, now: datetime | None = None) -> dict[str, Any]:
"""Project the reflect aggregate into the sidecar's ``{version, generated_at,
nodes}`` structure, keyed by canonical node id.
Built from preferred + tentative + contested (NOT dead_ends — those stay
query-scoped, surfaced only in the report). Citations that don't resolve to
exactly one node id are skipped.
"""
if now is None:
now = datetime.now(timezone.utc)
elif now.tzinfo is None:
now = now.replace(tzinfo=timezone.utc)
graph_path = Path(graph_path)
root = graph_path.parent
id_set, label_to_ids, node_by_id = _build_id_label_maps(graph_path)
prov_map = agg.get("_node_provenance", {})
# id -> entry; a canonical id can be cited under both its id and label form,
# but the aggregate dedups per node string, so collisions here are benign and
# resolved deterministically by iteration order (preferred, tentative, contested).
nodes_out: dict[str, dict[str, Any]] = {}
def _add(entry_src: dict[str, Any], status: str) -> None:
cited = entry_src["node"]
cid = _resolve_canonical_id(cited, id_set, label_to_ids)
if cid is None:
return # ambiguous or stale — can't display against a single node
if cid in nodes_out:
return # first status wins (preferred > tentative > contested order)
node = node_by_id.get(cid)
out: dict[str, Any] = {
"status": status,
"score": entry_src["score"],
"uses": entry_src.get("n", entry_src.get("pos", 0)),
"last": entry_src.get("last", ""),
"label": str(node.get("label", cited)) if node else str(cited),
"source_file": str(node.get("source_file") or "") if node else "",
"code_fingerprint": _code_fingerprint(node, root),
"provenance": _provenance_for(cited, prov_map, status),
}
if status == "contested":
out["verdict"] = entry_src.get("verdict", "even")
out["neg"] = entry_src.get("neg", 0)
else:
# preferred/tentative carry no contested verdict; derive `last` from
# provenance if the finalizer didn't (positive-only buckets do track it
# via node_last for contested only).
if not out["last"] and out["provenance"]:
out["last"] = out["provenance"][0]["date"]
nodes_out[cid] = out
for e in agg.get("preferred", []):
_add(e, "preferred")
for e in agg.get("tentative", []):
_add(e, "tentative")
for e in agg.get("contested", []):
_add(e, "contested")
return {
"version": _LEARNING_SCHEMA_VERSION,
"generated_at": now.isoformat(),
"nodes": nodes_out,
}
def write_learning_sidecar(agg: dict[str, Any], graph_path: Path,
*, now: datetime | None = None) -> Path:
"""Write ``.graphify_learning.json`` next to ``graph_path`` deterministically.
Sorted keys + indent=2 so re-runs on identical input (and a fixed ``now``)
are byte-identical. Returns the sidecar path.
"""
overlay = build_learning_overlay(agg, graph_path, now=now)
sidecar = Path(graph_path).parent / LEARNING_SIDECAR_NAME
sidecar.write_text(
json.dumps(overlay, indent=2, sort_keys=True, ensure_ascii=False) + "\n",
encoding="utf-8",
)
return sidecar
def load_learning_overlay(graph_path: Path) -> dict[str, dict[str, Any]]:
"""Load the sidecar next to ``graph_path`` and return ``{node_id -> entry}``
with a recomputed ``stale: bool`` per entry. Best-effort -> {} on any error.
Staleness: recompute ``file_hash(source_file)`` and compare to the entry's
stored ``code_fingerprint``. Matching fingerprints -> not stale. Differing,
missing-but-recomputable, or a vanished file -> stale (the safe, over-flagging
direction). An entry with no stored fingerprint AND no current file is not
marked stale (nothing to re-verify).
"""
sidecar = Path(graph_path).parent / LEARNING_SIDECAR_NAME
try:
data = json.loads(sidecar.read_text(encoding="utf-8"))
except (OSError, ValueError):
return {}
nodes = data.get("nodes")
if not isinstance(nodes, dict):
return {}
root = Path(graph_path).parent
out: dict[str, dict[str, Any]] = {}
for nid, entry in nodes.items():
if not isinstance(entry, dict):
continue
merged = dict(entry)
merged["stale"] = _is_stale(entry, root)
out[str(nid)] = merged
return out
def _is_stale(entry: dict[str, Any], root: Path) -> bool:
"""True if the node's source file changed (or vanished) since the fingerprint
was taken."""
stored = entry.get("code_fingerprint", "")
src = entry.get("source_file", "")
if not src:
# No file to track. Stale only if a fingerprint was stored yet there's
# nothing to compare against — treat as not stale (nothing to re-verify).
return False
p = Path(src)
if not p.is_absolute():
p = root / p
if not p.exists():
return True # file gone — definitely re-verify
try:
from graphify.cache import file_hash
current = file_hash(p, root)
except Exception:
return bool(stored) # couldn't recompute; flag iff we had something to compare
if not stored:
return True # had a file but never fingerprinted it -> can't trust -> stale
return current != stored
+64
View File
@@ -12,6 +12,62 @@ def _safe_community_name(label: str) -> str:
return cleaned or "unnamed"
def load_learning_for_report(graph_path) -> dict | None:
"""Assemble the report's work-memory inputs from sibling artifacts.
Reads the ``.graphify_learning.json`` overlay (preferred sources) next to
``graph_path`` and re-aggregates the memory docs for the query-scoped
dead-ends. Best-effort: returns None if neither is available, so the report
simply omits the section. Never raises.
"""
from pathlib import Path as _Path
try:
gp = _Path(graph_path)
from graphify.reflect import load_learning_overlay, load_memory_docs, aggregate_lessons
overlay = load_learning_overlay(gp)
dead_ends: list[dict] = []
mem = gp.parent / "memory"
if mem.is_dir():
agg = aggregate_lessons(load_memory_docs(mem))
dead_ends = agg.get("dead_ends", [])
if not overlay and not dead_ends:
return None
return {"overlay": overlay, "dead_ends": dead_ends}
except Exception:
return None
def _learning_section(lines: list, learning: dict | None, top_n: int = 10) -> None:
"""Append the ``## Work-memory lessons`` section, or nothing when empty."""
if not learning:
return
overlay = learning.get("overlay") or {}
dead_ends = learning.get("dead_ends") or []
preferred = [
(nid, e) for nid, e in overlay.items()
if isinstance(e, dict) and e.get("status") == "preferred"
]
# Most-corroborated first (uses desc), then by score, then id for stability.
preferred.sort(key=lambda kv: (-kv[1].get("uses", 0),
-float(kv[1].get("score", 0) or 0), kv[0]))
if not preferred and not dead_ends:
return
lines += ["", "## Work-memory lessons"]
if preferred:
lines += ["", "**Preferred sources** — corroborated by past sessions; start here."]
for nid, e in preferred[:top_n]:
label = e.get("label") or nid
stale = " _(code changed — re-verify)_" if e.get("stale") else ""
lines.append(f"- `{label}` ({e.get('uses', 0)}× useful, "
f"score={e.get('score', 0)}){stale}")
if dead_ends:
lines += ["", "**Known dead ends** — questions that led nowhere; don't re-derive."]
for d in dead_ends:
nodes = ", ".join(f"`{n}`" for n in d.get("nodes", []))
lines.append(f"- \"{d.get('question', '')}\""
+ (f" -> {nodes}" if nodes else ""))
def generate(
G: nx.Graph,
communities: dict[int, list[str]],
@@ -25,6 +81,7 @@ def generate(
suggested_questions: list[dict] | None = None,
min_community_size: int = 3,
built_at_commit: str | None = None,
learning: dict | None = None,
) -> str:
today = date.today().isoformat()
@@ -202,6 +259,13 @@ def generate(
if amb_pct > 20:
lines.append(f"- **High ambiguity: {amb_pct}% of edges are AMBIGUOUS.** Review the Ambiguous Edges section above.")
# --- Work-memory lessons (derived overlay) ---
# Preferred sources come from the .graphify_learning.json sidecar; the
# query-scoped dead-ends come from the reflect aggregate. Section omitted
# entirely when neither is present, so a graph with no work-memory is
# byte-identical to the pre-feature report.
_learning_section(lines, learning)
if suggested_questions:
lines += ["", "## Suggested Questions"]
no_signal = len(suggested_questions) == 1 and suggested_questions[0].get("type") == "no_signal"
+24 -3
View File
@@ -42,9 +42,18 @@ def _load_graph(graph_path: str) -> nx.Graph:
except Exception:
pass
try:
return json_graph.node_link_graph(data, edges="links")
G = json_graph.node_link_graph(data, edges="links")
except TypeError:
return json_graph.node_link_graph(data)
G = json_graph.node_link_graph(data)
# Attach the work-memory overlay (derived sidecar next to graph.json) so
# the query/MCP read surface can annotate NODE lines display-only. Empty
# when no sidecar exists, leaving un-annotated output byte-identical.
try:
from graphify.reflect import load_learning_overlay as _llo
G.graph["_learning_overlay"] = _llo(resolved)
except Exception:
G.graph["_learning_overlay"] = {}
return G
except (ValueError, FileNotFoundError) as exc:
print(f"error: {exc}", file=sys.stderr)
sys.exit(1)
@@ -503,6 +512,9 @@ def _subgraph_to_text(G: nx.Graph, nodes: set[str], edges: list[tuple], token_bu
"""
char_budget = token_budget * 3
lines = []
# Work-memory overlay (derived sidecar) stashed on the graph at load time.
# Empty when no sidecar exists, so un-annotated output stays byte-identical.
overlay = getattr(G, "graph", {}).get("_learning_overlay", {}) or {}
seed_set = set(seeds or [])
ordered = [n for n in (seeds or []) if n in nodes] + \
sorted(nodes - seed_set, key=lambda n: G.degree(n), reverse=True)
@@ -513,11 +525,20 @@ def _subgraph_to_text(G: nx.Graph, nodes: set[str], edges: list[tuple], token_bu
# corpus document can otherwise inject ANSI escapes, fake graphify-out
# log lines, or prompt-injection markup into the model's context via
# source_file / source_location / community.
# The learning= suffix is appended INSIDE the bracket and BEFORE the
# budget check below, so it counts in char_budget accounting.
entry = overlay.get(str(nid))
learning_suffix = ""
if entry:
status = sanitize_label(str(entry.get("status", "")))
if status:
learning_suffix = f" learning={status}{':stale' if entry.get('stale') else ''}"
line = (
f"NODE {sanitize_label(d.get('label', nid))} "
f"[src={sanitize_label(str(d.get('source_file', '')))} "
f"loc={sanitize_label(str(d.get('source_location', '')))} "
f"community={sanitize_label(str(d.get('community_name') or d.get('community', '')))}]"
f"community={sanitize_label(str(d.get('community_name') or d.get('community', '')))}"
f"{learning_suffix}]"
)
lines.append(line)
for u, v in edges:
+2 -1
View File
@@ -807,9 +807,10 @@ def _rebuild_code(
if cid not in labels:
labels[cid] = "Community " + str(cid)
questions = suggest_questions(G, communities, labels)
from graphify.report import load_learning_for_report as _llfr
report = generate(G, communities, cohesion, labels, gods, surprises, detection,
{"input": 0, "output": 0}, report_root, suggested_questions=questions,
built_at_commit=commit)
built_at_commit=commit, learning=_llfr(out / "graph.json"))
report_path = out / "GRAPH_REPORT.md"
labels_json = json.dumps({str(k): v for k, v in sorted(labels.items())}, ensure_ascii=False, indent=2) + "\n"
graph_tmp = out / ".graph.tmp.json"
+43
View File
@@ -80,3 +80,46 @@ def test_explain_source_file_path_prefers_file_level_node(monkeypatch, tmp_path,
assert "ID: example_route" in out
assert f"Source: {source_file} L1" in out
assert "Node: GET()" not in out
# --- work-memory overlay Lesson line ------------------------------------------
def _write_sidecar(tmp_path, nodes):
(tmp_path / ".graphify_learning.json").write_text(
json.dumps({"version": 1, "generated_at": "2026-06-01T00:00:00+00:00",
"nodes": nodes}),
encoding="utf-8",
)
def test_explain_shows_preferred_lesson_line(monkeypatch, tmp_path, capsys):
p = _write_graph(tmp_path)
_write_sidecar(tmp_path, {
"validate": {"status": "preferred", "score": 2.4, "uses": 3,
"label": "validateSanitySession()", "source_file": "",
"code_fingerprint": "", "provenance": []},
})
out = _run(monkeypatch, p, "validateSanitySession", capsys)
assert "Lesson: preferred source (start here) — 3 useful, score=2.4" in out
assert "code changed" not in out
def test_explain_shows_contested_and_stale_lesson(monkeypatch, tmp_path, capsys):
p = _write_graph(tmp_path)
# source_file points at a path that does not exist -> loader marks it stale.
_write_sidecar(tmp_path, {
"validate": {"status": "contested", "score": -0.1, "uses": 2, "neg": 1,
"verdict": "dead end", "label": "validateSanitySession()",
"source_file": "server/sanity-validate-session.ts",
"code_fingerprint": "deadbeef", "provenance": []},
})
out = _run(monkeypatch, p, "validateSanitySession", capsys)
assert "Lesson: contested (useful 2 / dead-end 1)" in out
assert "[code changed since — re-verify]" in out
def test_explain_no_lesson_line_for_unannotated_node(monkeypatch, tmp_path, capsys):
"""No sidecar => no Lesson line; output identical to pre-feature."""
p = _write_graph(tmp_path)
out = _run(monkeypatch, p, "validateSanitySession", capsys)
assert "Lesson:" not in out
+69
View File
@@ -184,6 +184,75 @@ def test_to_html_member_counts_accepted():
assert out.exists()
def _vis_nodes_from_html(content: str) -> list:
"""Extract the RAW_NODES JSON array embedded in the generated HTML."""
m = re.search(r"const RAW_NODES = (\[.*?\]);", content, re.DOTALL)
assert m, "RAW_NODES not found in HTML"
return json.loads(m.group(1).replace("<\\/", "</"))
def test_to_html_annotated_node_gets_learning_status_and_ring():
"""A node with an overlay entry gets learning_status + learning_stale fields,
a status-colored ring (border), and a Lesson line in its hover title."""
G = make_graph()
communities = cluster(G)
overlay = {
"n_transformer": {"status": "preferred", "uses": 3, "score": 2.4,
"stale": False, "neg": 0},
}
with tempfile.TemporaryDirectory() as tmp:
out = Path(tmp) / "graph.html"
to_html(G, communities, str(out), learning_overlay=overlay)
content = out.read_text()
nodes = {n["id"]: n for n in _vis_nodes_from_html(content)}
ann = nodes["n_transformer"]
assert ann["learning_status"] == "preferred"
assert ann["learning_stale"] is False
assert ann["color"]["border"] == "#22c55e" # green ring for preferred
assert ann.get("borderWidth") == 3
assert "Lesson: preferred source" in ann["title"]
# An un-annotated node carries no learning fields.
other = next(n for nid, n in nodes.items() if nid != "n_transformer")
assert "learning_status" not in other
assert "learning_stale" not in other
def test_to_html_contested_stale_node_gets_dashed_desaturated_ring():
G = make_graph()
communities = cluster(G)
overlay = {
"n_transformer": {"status": "contested", "uses": 2, "neg": 1,
"verdict": "dead end", "stale": True},
}
with tempfile.TemporaryDirectory() as tmp:
out = Path(tmp) / "graph.html"
to_html(G, communities, str(out), learning_overlay=overlay)
content = out.read_text()
ann = {n["id"]: n for n in _vis_nodes_from_html(content)}["n_transformer"]
assert ann["learning_status"] == "contested"
assert ann["learning_stale"] is True
assert ann["color"]["border"] == "#9ca3af" # desaturated when stale
assert ann["shapeProperties"]["borderDashes"] == [4, 4]
assert "code changed" in ann["title"]
def test_to_html_unannotated_identical_to_pre_feature():
"""With no overlay, the HTML is byte-identical whether learning_overlay is
omitted or passed empty — no learning fields leak into the un-annotated render."""
G = make_graph()
communities = cluster(G)
with tempfile.TemporaryDirectory() as tmp:
a = Path(tmp) / "a.html"
b = Path(tmp) / "b.html"
to_html(G, communities, str(a))
to_html(G, communities, str(b), learning_overlay={})
# Output path appears in the title, so compare with paths normalized out.
ca = a.read_text().replace("a.html", "X.html")
cb = b.read_text().replace("b.html", "X.html")
assert ca == cb
assert "learning_status" not in ca
def test_to_canvas_file_paths_relative_to_vault():
"""Node file paths in canvas must be vault-root-relative (just fname.md), not hardcoded."""
G = make_graph()
+170
View File
@@ -710,3 +710,173 @@ def test_dead_ends_and_corrections_dedupe_by_question():
assert [d["question"] for d in agg["dead_ends"]] == ["ws server?"]
assert len(agg["corrections"]) == 1
assert agg["corrections"][0]["correction"] == "SHA-256" # recency wins
# --- work-memory overlay sidecar (.graphify_learning.json) --------------------
#
# The sidecar is a DERIVED experiential layer written next to graph.json; the
# durable structural truth in graph.json is never stamped with learning_* fields.
# It projects the reflect aggregate (preferred/tentative/contested) into a
# per-node-id map with a code fingerprint for staleness and a provenance trail.
from graphify.reflect import ( # noqa: E402
LEARNING_SIDECAR_NAME,
build_learning_overlay,
load_learning_overlay,
write_learning_sidecar,
)
def _overlay_graph(out: Path, nodes: list[dict]) -> None:
"""Write a minimal graph.json under ``out`` with the given node dicts."""
out.mkdir(parents=True, exist_ok=True)
graph = {"directed": True, "multigraph": False, "graph": {},
"nodes": nodes, "links": []}
(out / "graph.json").write_text(json.dumps(graph), encoding="utf-8")
def _overlay_corpus(mem: Path) -> None:
"""A corpus with: a PREFERRED node (2 useful), a TENTATIVE node (1 useful),
a CONTESTED node (useful + dead_end), and a DEAD-END-ONLY node."""
_write_raw_doc(mem, "p1.md", "2026-05-01", outcome="useful",
question="how do I auth?", nodes=["login()"])
_write_raw_doc(mem, "p2.md", "2026-05-10", outcome="useful",
question="auth again", nodes=["login()"])
_write_raw_doc(mem, "t1.md", "2026-05-02", outcome="useful",
question="cache?", nodes=["RedisClient"])
_write_raw_doc(mem, "c1.md", "2026-05-03", outcome="useful",
question="contested useful", nodes=["Contested"])
_write_raw_doc(mem, "c2.md", "2026-05-04", outcome="dead_end",
question="contested dead", nodes=["Contested"])
_write_raw_doc(mem, "d1.md", "2026-05-05", outcome="dead_end",
question="led nowhere", nodes=["DeadEnd"])
def test_sidecar_write_classifies_and_keys_by_canonical_id(tmp_path):
"""reflect with a graph writes .graphify_learning.json next to graph.json with
the preferred/tentative/contested nodes keyed by canonical node id; the
dead-end-only node is NOT present; score/uses/provenance are carried."""
out = tmp_path / "graphify-out"
src = tmp_path / "auth.py"
src.write_text("def login(): pass\n", encoding="utf-8")
_overlay_graph(out, [
{"id": "auth_login", "label": "login()", "source_file": str(src), "community": 0},
{"id": "redis_client", "label": "RedisClient", "source_file": "", "community": 0},
{"id": "contested_node", "label": "Contested", "source_file": "", "community": 0},
{"id": "deadend_node", "label": "DeadEnd", "source_file": "", "community": 0},
])
mem = out / "memory"
_overlay_corpus(mem)
reflect(mem, out / "reflections" / "LESSONS.md",
graph_path=out / "graph.json", now=_NOW)
sidecar = json.loads((out / LEARNING_SIDECAR_NAME).read_text(encoding="utf-8"))
assert sidecar["version"] == 1
assert sidecar["generated_at"] == _NOW.isoformat()
nodes = sidecar["nodes"]
# Keyed by canonical node id, not label.
assert nodes["auth_login"]["status"] == "preferred"
assert nodes["auth_login"]["uses"] == 2
assert nodes["auth_login"]["label"] == "login()"
assert isinstance(nodes["auth_login"]["score"], float)
assert nodes["auth_login"]["provenance"] # captured during aggregation
assert nodes["redis_client"]["status"] == "tentative"
assert nodes["contested_node"]["status"] == "contested"
assert nodes["contested_node"]["verdict"] in ("useful", "dead end", "even")
# Dead-end-only node stays query-scoped — never in the overlay.
assert "deadend_node" not in nodes
# And learning_* is NOT stamped into graph.json (durable truth untouched).
graph = json.loads((out / "graph.json").read_text(encoding="utf-8"))
for n in graph["nodes"]:
assert not any(k.startswith("learning") for k in n)
def test_sidecar_is_byte_identical_across_runs(tmp_path):
"""Two reflect runs on identical input + fixed `now` produce a byte-identical
sidecar (sorted keys, stable indent)."""
out = tmp_path / "graphify-out"
src = tmp_path / "auth.py"
src.write_text("def login(): pass\n", encoding="utf-8")
_overlay_graph(out, [
{"id": "auth_login", "label": "login()", "source_file": str(src), "community": 0},
])
mem = out / "memory"
_write_raw_doc(mem, "a.md", "2026-05-01", outcome="useful", nodes=["login()"])
_write_raw_doc(mem, "b.md", "2026-05-10", outcome="useful", nodes=["login()"])
reflect(mem, out / "reflections" / "LESSONS.md",
graph_path=out / "graph.json", now=_NOW)
first = (out / LEARNING_SIDECAR_NAME).read_bytes()
reflect(mem, out / "reflections" / "LESSONS.md",
graph_path=out / "graph.json", now=_NOW)
second = (out / LEARNING_SIDECAR_NAME).read_bytes()
assert first == second
def test_loader_marks_entry_stale_when_source_file_changes(tmp_path):
"""load_learning_overlay recomputes the file fingerprint: unchanged source =>
stale=False; an edit to that source => stale=True."""
out = tmp_path / "graphify-out"
src = tmp_path / "auth.py"
src.write_text("def login(): pass\n", encoding="utf-8")
_overlay_graph(out, [
{"id": "auth_login", "label": "login()", "source_file": str(src), "community": 0},
])
mem = out / "memory"
_write_raw_doc(mem, "a.md", "2026-05-01", outcome="useful", nodes=["login()"])
_write_raw_doc(mem, "b.md", "2026-05-10", outcome="useful", nodes=["login()"])
reflect(mem, out / "reflections" / "LESSONS.md",
graph_path=out / "graph.json", now=_NOW)
fresh = load_learning_overlay(out / "graph.json")
assert fresh["auth_login"]["stale"] is False
src.write_text("def login(): return 1 # changed\n", encoding="utf-8")
after = load_learning_overlay(out / "graph.json")
assert after["auth_login"]["stale"] is True
def test_provenance_capped_to_five_most_recent(tmp_path):
"""A node cited by >5 useful results keeps exactly the 5 most-recent in
provenance (recent-first)."""
out = tmp_path / "graphify-out"
src = tmp_path / "auth.py"
src.write_text("x\n", encoding="utf-8")
_overlay_graph(out, [
{"id": "auth_login", "label": "login()", "source_file": str(src), "community": 0},
])
mem = out / "memory"
for i in range(7):
_write_raw_doc(mem, f"u{i}.md", f"2026-05-{10 + i:02d}",
outcome="useful", question=f"q{i}", nodes=["login()"])
reflect(mem, out / "reflections" / "LESSONS.md",
graph_path=out / "graph.json", now=_NOW)
sidecar = json.loads((out / LEARNING_SIDECAR_NAME).read_text(encoding="utf-8"))
prov = sidecar["nodes"]["auth_login"]["provenance"]
assert len(prov) == 5
# Most-recent first.
assert prov[0]["date"] == "2026-05-16"
assert prov[-1]["date"] == "2026-05-12"
def test_ambiguous_or_unresolved_citation_is_skipped(tmp_path):
"""A label shared by >1 node id (ambiguous) or absent from the graph
(unresolved) is skipped — it can't be displayed against a single node."""
out = tmp_path / "graphify-out"
_overlay_graph(out, [
{"id": "dup_a", "label": "Dup", "source_file": "", "community": 0},
{"id": "dup_b", "label": "Dup", "source_file": "", "community": 0},
{"id": "solo", "label": "Solo", "source_file": "", "community": 0},
])
mem = out / "memory"
_write_raw_doc(mem, "a.md", "2026-05-01", outcome="useful", nodes=["Dup"])
_write_raw_doc(mem, "b.md", "2026-05-02", outcome="useful", nodes=["Dup"])
_write_raw_doc(mem, "c.md", "2026-05-03", outcome="useful", nodes=["Solo"])
_write_raw_doc(mem, "d.md", "2026-05-04", outcome="useful", nodes=["Solo"])
reflect(mem, out / "reflections" / "LESSONS.md",
graph_path=out / "graph.json", now=_NOW)
nodes = json.loads((out / LEARNING_SIDECAR_NAME).read_text(encoding="utf-8"))["nodes"]
# Ambiguous "Dup" skipped; only the unambiguous "Solo" survives.
assert "dup_a" not in nodes and "dup_b" not in nodes
assert "solo" in nodes
+44
View File
@@ -61,3 +61,47 @@ def test_report_shows_raw_cohesion_scores():
assert "Cohesion:" in report
assert "✓" not in report
assert "⚠" not in report
# --- work-memory lessons section ----------------------------------------------
def test_report_work_memory_section_present_with_overlay_and_dead_ends():
"""When a work-memory overlay (preferred sources) and query-scoped dead-ends
are supplied, the report grows a `## Work-memory lessons` section listing the
preferred sources and, separately, the dead-ends as question -> nodes."""
G, communities, cohesion, labels, gods, surprises, detection, tokens = make_inputs()
learning = {
"overlay": {
"auth_login": {"status": "preferred", "uses": 3, "score": 2.4,
"label": "login()", "stale": False},
"redis": {"status": "tentative", "uses": 1, "score": 0.5,
"label": "RedisClient", "stale": False},
},
"dead_ends": [
{"question": "does it use websockets?", "nodes": ["WSServer"], "date": "2026-05-01"},
],
}
report = generate(G, communities, cohesion, labels, gods, surprises, detection,
tokens, "./project", learning=learning)
assert "## Work-memory lessons" in report
assert "**Preferred sources**" in report
assert "`login()`" in report
# Tentative is not listed in the report's preferred block.
assert "RedisClient" not in report
# Dead-ends are query-scoped: question -> nodes, NOT a node-level status.
assert "**Known dead ends**" in report
assert "does it use websockets?" in report
assert "`WSServer`" in report
def test_report_work_memory_section_absent_without_overlay():
"""No learning input => no section; report identical to pre-feature."""
G, communities, cohesion, labels, gods, surprises, detection, tokens = make_inputs()
before = generate(G, communities, cohesion, labels, gods, surprises, detection,
tokens, "./project")
assert "## Work-memory lessons" not in before
# Explicit empty learning also omits the section.
empty = generate(G, communities, cohesion, labels, gods, surprises, detection,
tokens, "./project", learning={"overlay": {}, "dead_ends": []})
assert "## Work-memory lessons" not in empty
assert before == empty
+48
View File
@@ -379,6 +379,54 @@ def test_subgraph_to_text_includes_edge_context():
assert "context=call" in text
# --- work-memory overlay annotation on NODE lines -----------------------------
def test_subgraph_to_text_annotates_node_with_learning_status():
"""An annotated node gets a `learning=<status>` suffix inside its NODE
bracket; an un-annotated node gets none."""
G = _make_graph()
G.graph["_learning_overlay"] = {
"n1": {"status": "preferred", "stale": False},
}
text = _subgraph_to_text(G, {"n1", "n2"}, [("n1", "n2")])
lines = {l.split()[1]: l for l in text.splitlines() if l.startswith("NODE ")}
assert "learning=preferred]" in lines["extract"]
assert "learning=" not in lines["cluster"] # un-annotated node
def test_subgraph_to_text_marks_stale_status():
G = _make_graph()
G.graph["_learning_overlay"] = {"n1": {"status": "contested", "stale": True}}
text = _subgraph_to_text(G, {"n1"}, [])
assert "learning=contested:stale]" in text
def test_subgraph_to_text_learning_suffix_counts_against_budget():
"""The learning= suffix is part of the NODE line BEFORE the budget cut, so it
is included in the char_budget accounting (a budget tight enough to fit the
bare line but not the suffixed line forces truncation)."""
G = _make_graph()
bare = _subgraph_to_text(G, {"n1", "n2", "n3"}, [])
# token_budget chosen so the un-annotated render fits without truncation...
budget = (len(bare) // 3) + 1
assert "truncated" not in _subgraph_to_text(G, {"n1", "n2", "n3"}, [],
token_budget=budget)
# ...but once every node carries a learning= suffix, the same budget overflows.
G.graph["_learning_overlay"] = {
n: {"status": "preferred", "stale": False} for n in ("n1", "n2", "n3")
}
annotated = _subgraph_to_text(G, {"n1", "n2", "n3"}, [], token_budget=budget)
assert "learning=preferred" in annotated
assert "truncated" in annotated
def test_subgraph_to_text_no_overlay_is_unchanged():
"""With no overlay on the graph, NODE lines carry no learning= suffix."""
G = _make_graph()
text = _subgraph_to_text(G, {"n1", "n2"}, [("n1", "n2")])
assert "learning=" not in text
def test_query_graph_text_explicit_context_filter_changes_traversal():
G = _make_graph()
text = _query_graph_text(G, "extract", mode="bfs", depth=2, token_budget=2000, context_filters=["call"])