diff --git a/CHANGELOG.md b/CHANGELOG.md index a7dbe18..8516c00 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,7 @@ Full release notes with details on each version: [GitHub Releases](https://githu ## Unreleased +- Feat: a first deterministic slice of self-improving "work memory" (#1441). `graphify save-result` gains optional `--outcome useful|dead_end|corrected` and `--correction TEXT` flags that record how a saved Q&A turned out — written to the memory doc's frontmatter and an `## Outcome` body section so the signal both stays machine-readable and round-trips into the graph on the next semantic re-extraction. A new `graphify reflect` command then scans `graphify-out/memory/` and writes a deterministic `graphify-out/reflections/LESSONS.md` an agent can load at the start of the next session — grouped by community when a `graph.json` is present, flat otherwise. Source nodes are scored, not counted: each citation is a signed, time-decayed value (`useful` positive, `dead_end`/`corrected` negative, configurable half-life via `--half-life-days`, default 30), so a fresh dead end outweighs a months-old useful. A node is only promoted to "preferred" once corroborated by ≥`--min-corroboration` distinct results (default 2) — one save can't mint a trusted lesson; the rest render as "tentative", and nodes with both positive and negative signals render once as "contested" with a recency-wins verdict. Source nodes are matched to the graph by label or id, and citations whose node no longer exists are dropped so stale lessons don't linger. Deterministic, no LLM; bare `graphify save-result` and all existing behavior are unchanged. - Fix: `graphify update` now prunes a function/symbol removed from a still-present file without needing `--force`. The build already dropped the stale node (#1116), but the shrink-guard then refused to write the smaller graph ("new graph has N nodes but existing has M … Refusing to overwrite"), so the deletion silently never persisted unless you passed `--force` — leaving stale nodes (and the work-memory node-existence gate) lagging until a forced rebuild. The guard is now file-aware: a net shrink is allowed when every lost node belongs to a file re-extracted this run (or deleted), and still refused when a node disappears from a file that was *not* touched (the failed/partial-extraction case it exists to catch). - Fix: `validate_extraction` and `build_from_json` no longer crash on a non-hashable node `id` or edge `source`/`target` (e.g. a list emitted by a malformed LLM extraction) — previously a single bad node raised `TypeError: unhashable type` and aborted the entire build of an otherwise-complete corpus. The validator now reports the bad id/endpoint as an error string (its documented contract), and the build skips the malformed entry with a stderr warning while keeping every well-formed node/edge; non-dict nodes are still left to raise so shape diagnostics are unchanged (#1447, thanks @dschwartzi). - Fix: Python qualified class-method calls (`ClassName.method(...)`) now produce an EXTRACTED `calls` edge to the class-qualified method node (#1446). Previously these cross-class static/qualified calls were dropped: the shared cross-file pass skips all member calls (the #543/#1219 god-node guard against bare `obj.method()` collisions), and when the called method shared its name with an in-file node — e.g. a viewset action `approve()` delegating to a service `Service.approve()` — the bare-name lookup matched the caller's own node and silently dropped it. The Python extractor now captures a simple-identifier receiver, defers capitalized-receiver member calls to a new receiver-based resolver (`_resolve_python_member_calls`, mirroring the Swift pass), and emits the edge only when the receiver resolves to exactly one class that owns the method (single-definition god-node guard); instance/module calls (`self.x()`, `obj.x()`, lowercase receivers) are unaffected. diff --git a/README.md b/README.md index 3573af9..ed4f4f6 100644 --- a/README.md +++ b/README.md @@ -538,6 +538,10 @@ graphify install # overwrites the skill file /graphify path "DigestAuth" "Response" /graphify explain "SwinTransformer" +graphify reflect # aggregate graphify-out/memory/ outcomes into reflections/LESSONS.md +graphify reflect --out docs/LESSONS.md # write the lessons doc somewhere else +graphify reflect --graph graphify-out/graph.json # also group lessons by community + graphify uninstall # remove from all platforms in one shot graphify uninstall --purge # also delete graphify-out/ graphify uninstall --project --platform codex # remove project-scoped install files only diff --git a/graphify/__init__.py b/graphify/__init__.py index e34c938..41d4b1f 100644 --- a/graphify/__init__.py +++ b/graphify/__init__.py @@ -19,6 +19,8 @@ def __getattr__(name): "to_svg": ("graphify.export", "to_svg"), "to_canvas": ("graphify.export", "to_canvas"), "to_wiki": ("graphify.wiki", "to_wiki"), + "reflect": ("graphify.reflect", "reflect"), + "save_query_result": ("graphify.ingest", "save_query_result"), } if name in _map: import importlib diff --git a/graphify/__main__.py b/graphify/__main__.py index 3f7b6e2..9434519 100644 --- a/graphify/__main__.py +++ b/graphify/__main__.py @@ -2246,7 +2246,17 @@ def main() -> None: " --type T query type: query|path_query|explain (default: query)" ) print(" --nodes N1 N2 ... source node labels cited in the answer") + print(" --outcome O work-memory signal: useful|dead_end|corrected") + print(" --correction TEXT what the right answer was (pairs with --outcome corrected)") print(" --memory-dir DIR memory directory (default: graphify-out/memory)") + print(" reflect aggregate graphify-out/memory/ outcomes into a deterministic lessons doc") + print(" --memory-dir DIR memory directory (default: graphify-out/memory)") + print(" --out FILE output path (default: graphify-out/reflections/LESSONS.md)") + print(" --graph PATH graph.json, for community grouping + dropping stale nodes (optional)") + print(" --analysis PATH .graphify_analysis.json (optional, auto-detected next to --graph)") + print(" --labels PATH .graphify_labels.json (optional, auto-detected next to --graph)") + print(" --half-life-days N signal weight halves every N days (default 30)") + print(" --min-corroboration N distinct useful results to prefer a node (default 2)") print(" check-update check needs_update flag and notify if semantic re-extraction is pending (cron-safe)") print(" tree emit a D3 v7 collapsible-tree HTML for graph.json") print(" --graph PATH path to graph.json (default graphify-out/graph.json)") @@ -2906,7 +2916,8 @@ def main() -> None: ) ) elif cmd == "save-result": - # graphify save-result --question Q --answer A --type T [--nodes N1 N2 ...] + # graphify save-result --question Q --answer A [--type T] [--nodes N1 N2 ...] + # [--outcome useful|dead_end|corrected] [--correction TEXT] import argparse as _ap p = _ap.ArgumentParser(prog="graphify save-result") @@ -2914,6 +2925,8 @@ def main() -> None: p.add_argument("--answer", required=True) p.add_argument("--type", dest="query_type", default="query") p.add_argument("--nodes", nargs="*", default=[]) + p.add_argument("--outcome", choices=("useful", "dead_end", "corrected"), default=None) + p.add_argument("--correction", default=None) p.add_argument("--memory-dir", default=str(Path(_GRAPHIFY_OUT) / "memory")) opts = p.parse_args(sys.argv[2:]) from graphify.ingest import save_query_result as _sqr @@ -2924,8 +2937,50 @@ def main() -> None: memory_dir=Path(opts.memory_dir), query_type=opts.query_type, source_nodes=opts.nodes or None, + outcome=opts.outcome, + correction=opts.correction, ) print(f"Saved to {out}") + elif cmd == "reflect": + import argparse as _ap + + p = _ap.ArgumentParser(prog="graphify reflect") + p.add_argument("--memory-dir", default=str(Path(_GRAPHIFY_OUT) / "memory")) + p.add_argument( + "--out", + default=str(Path(_GRAPHIFY_OUT) / "reflections" / "LESSONS.md"), + ) + p.add_argument("--graph", default=None) + p.add_argument("--analysis", default=None) + p.add_argument("--labels", default=None) + p.add_argument("--half-life-days", type=float, default=30.0, + help="signal weight halves every N days (default 30)") + p.add_argument("--min-corroboration", type=int, default=2, + help="distinct useful results to promote a node to preferred (default 2)") + opts = p.parse_args(sys.argv[2:]) + from graphify.reflect import reflect as _reflect + + graph_arg = opts.graph + if graph_arg is None: + default_graph = Path(_GRAPHIFY_OUT) / "graph.json" + if default_graph.exists(): + graph_arg = str(default_graph) + + out_path, agg = _reflect( + memory_dir=Path(opts.memory_dir), + out_path=Path(opts.out), + graph_path=Path(graph_arg) if graph_arg else None, + analysis_path=Path(opts.analysis) if opts.analysis else None, + labels_path=Path(opts.labels) if opts.labels else None, + half_life_days=opts.half_life_days, + min_corroboration=opts.min_corroboration, + ) + c = agg["counts"] + print( + f"Reflected {agg['total']} memories " + f"({c['useful']} useful, {c['dead_end']} dead ends, " + f"{c['corrected']} corrected) -> {out_path}" + ) elif cmd == "path": if len(sys.argv) < 4: print( diff --git a/graphify/ingest.py b/graphify/ingest.py index 93e69bd..86b9b75 100644 --- a/graphify/ingest.py +++ b/graphify/ingest.py @@ -268,6 +268,8 @@ def ingest(url: str, target_dir: Path, author: str | None = None, contributor: s print(f"Saved {url_type}: {out_path.name}") return out_path +OUTCOMES = ("useful", "dead_end", "corrected") + def save_query_result( question: str, @@ -275,13 +277,23 @@ def save_query_result( memory_dir: Path, query_type: str = "query", source_nodes: list[str] | None = None, + outcome: str | None = None, + correction: str | None = None, ) -> Path: """Save a Q&A result as markdown so it gets extracted into the graph on next --update. Files are stored in memory_dir (typically graphify-out/memory/) with YAML frontmatter that graphify's extractor reads as node metadata. This closes the feedback loop: the system grows smarter from both what you add AND what you ask. + + ``outcome`` (one of :data:`OUTCOMES`) and ``correction`` are optional work-memory + signals: they are written both to the frontmatter (so `graphify reflect` can + aggregate them deterministically) and to an ``## Outcome`` body section (so the + signal round-trips into the graph on the next semantic re-extraction). """ + if outcome is not None and outcome not in OUTCOMES: + raise ValueError(f"outcome must be one of {OUTCOMES}, got {outcome!r}") + memory_dir = Path(memory_dir) memory_dir.mkdir(parents=True, exist_ok=True) @@ -296,8 +308,12 @@ def save_query_result( f'question: "{_yaml_str(question)}"', 'contributor: "graphify"', ] + if outcome: + frontmatter_lines.append(f'outcome: "{_yaml_str(outcome)}"') + if correction: + frontmatter_lines.append(f'correction: "{_yaml_str(correction)}"') if source_nodes: - nodes_str = ", ".join(f'"{n}"' for n in source_nodes[:10]) + nodes_str = ", ".join(f'"{_yaml_str(n)}"' for n in source_nodes[:10]) frontmatter_lines.append(f"source_nodes: [{nodes_str}]") frontmatter_lines.append("---") @@ -309,6 +325,12 @@ def save_query_result( "", answer, ] + if outcome or correction: + body_lines += ["", "## Outcome", ""] + if outcome: + body_lines.append(f"- Signal: {outcome}") + if correction: + body_lines.append(f"- Correction: {correction}") if source_nodes: body_lines += ["", "## Source Nodes", ""] body_lines += [f"- {n}" for n in source_nodes] diff --git a/graphify/reflect.py b/graphify/reflect.py new file mode 100644 index 0000000..1c697a3 --- /dev/null +++ b/graphify/reflect.py @@ -0,0 +1,525 @@ +"""Deterministic "work memory" reflection over graphify-out/memory/. + +`graphify reflect` reads the Q&A memory docs that `graphify save-result` files back +into the graph, aggregates their outcome signals (useful / dead_end / corrected), and +writes a single lessons artifact an agent can load at the start of the next session: + + - **Preferred sources** — nodes corroborated by multiple ``useful`` answers. + - **Tentative** — nodes seen useful only once (not yet corroborated). + - **Contested** — nodes with both positive and negative signals; recency decides. + - **Known dead ends** — questions/sources marked ``dead_end``; don't re-derive them. + - **Corrections** — answers the user corrected, and what the right answer was. + +Source nodes are scored, not counted: each citation contributes a signed, +time-decayed value (``useful`` positive, ``dead_end``/``corrected`` negative, with a +half-life so a fresh dead end outweighs a months-old useful). A node is only promoted +to "preferred" once corroborated by enough distinct results; one save can't mint a +trusted lesson. When a graph is in hand, source nodes that no longer exist are dropped. + +It is deterministic: no LLM, stable sort orders, byte-stable output for a given input +and a given ``now``. When a graph (`graph.json` + `.graphify_analysis.json`) is available +the lessons are also grouped by community label; without it they degrade to a single +flat section. + +The artifact lands at ``graphify-out/reflections/LESSONS.md`` rather than inside the wiki +because ``graphify export wiki`` deletes every ``wiki/*.md`` on each run — a lessons file +written there would be clobbered on the next export. +""" +from __future__ import annotations + +import json +import re +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + +from graphify.ingest import OUTCOMES + +_UNCATEGORIZED = "Uncategorized" + +# Scoring defaults (both exposed as CLI flags). +_DEFAULT_HALF_LIFE_DAYS = 30.0 # a signal's weight halves every 30 days +_DEFAULT_MIN_CORROBORATION = 2 # distinct useful results needed to "prefer" a node + +# Rounding for the signed score keeps sort order and the contested verdict stable +# across platforms (C pow can differ in the last ULP). +_SCORE_NDIGITS = 9 + + +# --- frontmatter parsing ------------------------------------------------------- +# +# save_query_result writes a tiny, hand-built YAML subset (no PyYAML dependency), +# so we parse the same subset by hand rather than adding a dependency: scalar +# `key: "value"` lines and a `source_nodes: ["a", "b"]` flow list. Anything we +# don't recognise is ignored, so foreign .md files in memory/ are skipped cleanly. + +_SCALAR_RE = re.compile(r'^([A-Za-z_][\w-]*):\s*"(.*)"\s*$') +_LIST_RE = re.compile(r"^([A-Za-z_][\w-]*):\s*\[(.*)\]\s*$") +_DQ_ITEM_RE = re.compile(r'"((?:[^"\\]|\\.)*)"') + + +def _yaml_unescape(s: str) -> str: + """Reverse the double-quoted escaping that ingest._yaml_str applies.""" + out: list[str] = [] + i = 0 + simple = {"n": "\n", "r": "\r", "t": "\t", "0": "\0", '"': '"', "\\": "\\", + "L": "\u2028", "P": "\u2029"} # YAML line/paragraph separators + while i < len(s): + ch = s[i] + if ch == "\\" and i + 1 < len(s): + nxt = s[i + 1] + if nxt in simple: + out.append(simple[nxt]) + i += 2 + continue + if nxt == "x" and i + 3 < len(s): + try: + out.append(chr(int(s[i + 2:i + 4], 16))) + i += 4 + continue + except ValueError: + pass + if nxt == "u" and i + 5 < len(s): + try: + out.append(chr(int(s[i + 2:i + 6], 16))) + i += 6 + continue + except ValueError: + pass + out.append(ch) + i += 1 + return "".join(out) + + +def parse_memory_doc(text: str) -> dict[str, Any] | None: + """Parse the frontmatter of a memory doc into a dict, or None if it has none. + + Returns the recognised fields (``type``, ``date``, ``question``, ``outcome``, + ``correction``, ``source_nodes``). ``source_nodes`` is always a list. + """ + if not text.startswith("---"): + return None + lines = text.splitlines() + if not lines or lines[0].strip() != "---": + return None + fields: dict[str, Any] = {"source_nodes": []} + for line in lines[1:]: + if line.strip() == "---": + break + m = _LIST_RE.match(line) + if m and m.group(1) == "source_nodes": + fields["source_nodes"] = [ + _yaml_unescape(item) for item in _DQ_ITEM_RE.findall(m.group(2)) + ] + continue + m = _SCALAR_RE.match(line) + if m: + key, val = m.group(1), _yaml_unescape(m.group(2)) + if key in ("type", "date", "question", "outcome", "correction", "contributor"): + fields[key] = val + return fields + + +def load_memory_docs(memory_dir: Path) -> list[dict[str, Any]]: + """Parse every memory doc under ``memory_dir``, sorted by date then filename. + + Each record is the parsed frontmatter plus ``_path`` (the source file). Docs + without recognisable frontmatter (foreign .md files, the LESSONS.md artifact) + are skipped. + """ + memory_dir = Path(memory_dir) + if not memory_dir.exists(): + return [] + docs: list[dict[str, Any]] = [] + for path in sorted(memory_dir.glob("*.md")): + try: + text = path.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError): + continue + parsed = parse_memory_doc(text) + if parsed is None: + continue + parsed["_path"] = path.name + docs.append(parsed) + # Stable order: by (date, filename) so output is deterministic across runs. + docs.sort(key=lambda d: (d.get("date", ""), d["_path"])) + return docs + + +# --- graph / community lookup (optional) --------------------------------------- + + +def _load_node_community(graph_path: Path, analysis_path: Path, + labels_path: Path) -> dict[str, str] | None: + """Build a lookup from node id AND node label -> community label, or None if the + graph isn't available. + + Mirrors how `graphify export wiki` reads graph.json + .graphify_analysis.json + + .graphify_labels.json. Community membership in the analysis sidecar is keyed by + node id, but `save-result` cites nodes by label, so both are mapped — otherwise a + cited ``build_from_json()`` never finds its community and every lesson collapses + into Uncategorized. Best-effort: any missing/unparseable artifact disables grouping. + """ + if not graph_path.exists() or not analysis_path.exists(): + return None + try: + analysis = json.loads(analysis_path.read_text(encoding="utf-8")) + except (OSError, ValueError): + return None + communities = analysis.get("communities", {}) + if not communities: + return None + labels: dict[str, str] = {} + if labels_path.exists(): + try: + labels = json.loads(labels_path.read_text(encoding="utf-8")) + except (OSError, ValueError): + labels = {} + # id -> label from the graph, so a label-form citation resolves to a community too. + id_to_label: dict[str, str] = {} + try: + gdata = json.loads(graph_path.read_text(encoding="utf-8")) + for n in gdata.get("nodes", []): + if isinstance(n, dict) and n.get("id") is not None and n.get("label") is not None: + id_to_label[str(n["id"])] = str(n["label"]) + except (OSError, ValueError): + id_to_label = {} + # Sorted cid iteration + setdefault makes any label collision resolve + # deterministically (smallest community id wins). + node_community: dict[str, str] = {} + for cid in sorted(communities, key=str): + label = labels.get(str(cid)) or labels.get(cid) or f"Community {cid}" + for nid in communities[cid]: + nid = str(nid) + node_community.setdefault(nid, label) + nlabel = id_to_label.get(nid) + if nlabel is not None: + node_community.setdefault(nlabel, label) + return node_community + + +def _load_known_nodes(graph_path: Path) -> set[str] | None: + """The set of node ids AND labels in the current graph, or None if unavailable. + + Used to drop source nodes from lessons once the code they pointed at is gone + (deleted/renamed) — a stale lesson shouldn't keep getting recommended. Both ids + and labels are collected because `save-result` records source nodes by their + human-readable label (what an agent cites, e.g. ``build_from_json()``), while + graph nodes are keyed by id (e.g. ``module_build_from_json``). Matching on either + keeps a still-present node and only drops one that survives under neither name — + indexing ids alone silently dropped every label-form citation (the common case). + """ + try: + data = json.loads(Path(graph_path).read_text(encoding="utf-8")) + except (OSError, ValueError): + return None + nodes = data.get("nodes") + if not isinstance(nodes, list): + return None + known: set[str] = set() + for n in nodes: + if not isinstance(n, dict): + continue + if n.get("id") is not None: + known.add(str(n["id"])) + if n.get("label") is not None: + known.add(str(n["label"])) + return known or None + + +def _doc_community(nodes: list[str], + node_community: dict[str, str] | None) -> str: + """The community a doc belongs to: the plurality community of its source nodes. + + Ties break to the lexicographically-smallest label, so the result is + deterministic regardless of source-node order. Docs with no resolvable + community (no source nodes, or no graph) fall into the Uncategorized bucket. + """ + if not node_community: + return _UNCATEGORIZED + labels = [node_community[n] for n in nodes if n in node_community] + if not labels: + return _UNCATEGORIZED + counts = Counter(labels) + # Highest count wins; on a tie, the smaller label (most-negative count first, + # then ascending label) — a plain min() over (-count, label). + return min(counts.items(), key=lambda kv: (-kv[1], kv[0]))[0] + + +# --- scoring helpers ----------------------------------------------------------- + + +def _parse_dt(date_str: str) -> datetime | None: + """Parse an ISO date/datetime to an aware UTC datetime, or None if unparseable.""" + if not date_str: + return None + try: + dt = datetime.fromisoformat(date_str) + except ValueError: + return None + if dt.tzinfo is None: + dt = dt.replace(tzinfo=timezone.utc) + return dt + + +def _decay(date_str: str, now: datetime, half_life_days: float) -> float: + """Time-decay weight in (0, 1]: halves every ``half_life_days``. + + Undated/unparseable signals keep full weight (1.0); future-dated ones are + clamped to age 0 (also 1.0). + """ + dt = _parse_dt(date_str) + if dt is None or half_life_days <= 0: + return 1.0 + age_days = max(0.0, (now - dt).total_seconds() / 86400.0) + return 0.5 ** (age_days / half_life_days) + + +# --- aggregation --------------------------------------------------------------- + + +def _empty_bucket() -> dict[str, Any]: + return { + "counts": {k: 0 for k in (*OUTCOMES, "unmarked")}, + # node -> running signed, time-decayed score + "node_score": {}, + # node -> distinct positive / negative result counts (for corroboration) + "node_pos": Counter(), + "node_neg": Counter(), + # node -> most recent event date seen (for the contested verdict line) + "node_last": {}, + "dead_ends": [], + "corrections": [], + } + + +def _record_node(bucket: dict[str, Any], node: str, sign: int, + weight: float, date: str) -> None: + bucket["node_score"][node] = bucket["node_score"].get(node, 0.0) + sign * weight + if sign > 0: + bucket["node_pos"][node] += 1 + elif sign < 0: + bucket["node_neg"][node] += 1 + if date > bucket["node_last"].get(node, ""): + bucket["node_last"][node] = date + + +def _finalize_sources(bucket: dict[str, Any], + min_corroboration: int) -> dict[str, list]: + """Split a bucket's scored nodes into preferred / tentative / contested lists.""" + preferred, tentative, contested = [], [], [] + for node in bucket["node_score"]: + pos = bucket["node_pos"][node] + neg = bucket["node_neg"][node] + score = round(bucket["node_score"][node], _SCORE_NDIGITS) + if pos and neg: + verdict = "useful" if score > 0 else "dead end" if score < 0 else "even" + contested.append({"node": node, "pos": pos, "neg": neg, + "score": score, "verdict": verdict, + "last": bucket["node_last"].get(node, "")}) + elif pos: # positive-only + entry = {"node": node, "n": pos, "score": score} + (preferred if pos >= min_corroboration else tentative).append(entry) + # negative-only nodes are surfaced via the dead-ends questions, not here. + preferred.sort(key=lambda e: (-e["score"], e["node"])) + tentative.sort(key=lambda e: (-e["score"], e["node"])) + contested.sort(key=lambda e: (-e["score"], e["node"])) + return {"preferred": preferred, "tentative": tentative, "contested": contested} + + +def aggregate_lessons(docs: list[dict[str, Any]], + node_community: dict[str, str] | None = None, + *, + now: datetime | None = None, + half_life_days: float = _DEFAULT_HALF_LIFE_DAYS, + min_corroboration: int = _DEFAULT_MIN_CORROBORATION, + known_nodes: set[str] | None = None) -> dict[str, Any]: + """Aggregate parsed memory docs into a deterministic lessons structure. + + ``now`` anchors the time-decay (pass it explicitly for byte-stable output). + ``known_nodes`` (when given) gates out source nodes no longer in the graph. + Returns ``{"total", "counts", "min_corroboration", "preferred", "tentative", + "contested", "dead_ends", "corrections", "by_community"}``; ``by_community`` is + empty unless a graph is supplied. + """ + if now is None: + now = datetime.now(timezone.utc) + elif now.tzinfo is None: + now = now.replace(tzinfo=timezone.utc) + + overall = _empty_bucket() + by_community: dict[str, dict[str, Any]] = {} + + for doc in docs: + outcome = doc.get("outcome") + date = doc.get("date", "") + # One event per node per doc; drop nodes the graph no longer knows about. + raw = doc.get("source_nodes", []) + nodes = list(dict.fromkeys( + n for n in raw if known_nodes is None or n in known_nodes)) + community = _doc_community(nodes, node_community) + bucket = by_community.setdefault(community, _empty_bucket()) + + sign = 1 if outcome == "useful" else -1 if outcome in ("dead_end", "corrected") else 0 + weight = _decay(date, now, half_life_days) if sign else 0.0 + + for target in (overall, bucket): + target["counts"][outcome if outcome in OUTCOMES else "unmarked"] += 1 + if sign: + for n in nodes: + _record_node(target, n, sign, weight, date) + if outcome == "dead_end": + target["dead_ends"].append( + {"question": doc.get("question", ""), "nodes": nodes, "date": date}) + elif outcome == "corrected": + target["corrections"].append( + {"question": doc.get("question", ""), + "correction": doc.get("correction", ""), "date": date}) + + # Only surface per-community grouping when a graph was actually supplied; + # without one every doc falls into Uncategorized and the section would just + # duplicate the flat "Lessons" block. + community_out: dict[str, dict[str, Any]] = {} + if node_community: + community_out = { + label: {"counts": b["counts"], **_finalize_sources(b, min_corroboration), + "dead_ends": b["dead_ends"], "corrections": b["corrections"]} + for label, b in by_community.items() + } + + return { + "total": len(docs), + "counts": overall["counts"], + "min_corroboration": min_corroboration, + **_finalize_sources(overall, min_corroboration), + "dead_ends": overall["dead_ends"], + "corrections": overall["corrections"], + "by_community": community_out, + } + + +# --- rendering ----------------------------------------------------------------- + + +def _render_bucket(out: list[str], data: dict[str, Any], k: int) -> None: + preferred = data["preferred"] + tentative = data["tentative"] + contested = data["contested"] + dead_ends = data["dead_ends"] + corrections = data["corrections"] + + if preferred: + out += [f"**Preferred sources** — corroborated by ≥{k} useful results; " + "start here.", ""] + for e in preferred: + out.append(f"- `{e['node']}` ({e['n']}× useful)") + out.append("") + if tentative: + out += [f"**Tentative** — useful in fewer than {k} results; verify before " + "relying.", ""] + for e in tentative: + out.append(f"- `{e['node']}` ({e['n']}× useful)") + out.append("") + if contested: + out += ["**Contested** — mixed signals; recency decides.", ""] + for e in contested: + day = e["last"][:10] + verdict = ("evenly split" if e["verdict"] == "even" + else f"recency leans **{e['verdict']}**") + out.append( + f"- `{e['node']}` — {e['pos']}× useful, {e['neg']}× " + f"dead end/corrected → {verdict}" + + (f" (latest {day})" if day else "")) + out.append("") + if dead_ends: + out += ["**Known dead ends** — led nowhere; don't re-derive.", ""] + for d in dead_ends: + nodes = ", ".join(f"`{n}`" for n in d["nodes"]) + out.append(f"- \"{d['question']}\"" + (f" — {nodes}" if nodes else "")) + out.append("") + if corrections: + out += ["**Corrections** — do these differently.", ""] + for c in corrections: + out.append(f"- \"{c['question']}\" → {c['correction']}") + out.append("") + if not (preferred or tentative or contested or dead_ends or corrections): + out += ["_No marked outcomes yet._", ""] + + +def render_lessons_md(agg: dict[str, Any]) -> str: + """Render the aggregate into the deterministic LESSONS.md markdown body.""" + c = agg["counts"] + k = agg.get("min_corroboration", _DEFAULT_MIN_CORROBORATION) + out: list[str] = [ + "# Lessons", + "", + f"_Auto-generated by `graphify reflect` from {agg['total']} session " + f"{'memory' if agg['total'] == 1 else 'memories'} in graphify-out/memory/. " + "Deterministic; no LLM. Use for orientation — verify before relying, and " + "revisit dead ends if the code has changed since._", + "", + "## Summary", + "", + f"- {c['useful']} useful · {c['dead_end']} dead ends · " + f"{c['corrected']} corrected · {c['unmarked']} unmarked", + "", + "## Lessons", + "", + ] + _render_bucket(out, agg, k) + + if agg["by_community"]: + out += ["## By topic", ""] + # Uncategorized sorts last; everything else alphabetically. + def _topic_key(label: str) -> tuple[int, str]: + return (1 if label == _UNCATEGORIZED else 0, label) + for label in sorted(agg["by_community"], key=_topic_key): + out += [f"### {label}", ""] + _render_bucket(out, agg["by_community"][label], k) + + # Single trailing newline, no trailing whitespace lines. + return "\n".join(out).rstrip("\n") + "\n" + + +# --- orchestrator -------------------------------------------------------------- + + +def reflect(memory_dir: Path, out_path: Path, + graph_path: Path | None = None, + analysis_path: Path | None = None, + labels_path: Path | None = None, + *, + now: datetime | None = None, + half_life_days: float = _DEFAULT_HALF_LIFE_DAYS, + min_corroboration: int = _DEFAULT_MIN_CORROBORATION, + ) -> tuple[Path, dict[str, Any]]: + """Scan ``memory_dir``, write the lessons doc to ``out_path``, return (path, agg). + + If ``graph_path`` is given lessons are grouped by community and source nodes no + longer in the graph are dropped; otherwise the doc is a single flat section. + """ + docs = load_memory_docs(memory_dir) + + node_community = None + known_nodes = None + if graph_path is not None: + graph_path = Path(graph_path) + analysis_path = Path(analysis_path) if analysis_path else ( + graph_path.parent / ".graphify_analysis.json") + labels_path = Path(labels_path) if labels_path else ( + graph_path.parent / ".graphify_labels.json") + node_community = _load_node_community(graph_path, analysis_path, labels_path) + known_nodes = _load_known_nodes(graph_path) + + if now is None: + now = datetime.now(timezone.utc) + + agg = aggregate_lessons(docs, node_community, now=now, + half_life_days=half_life_days, + min_corroboration=min_corroboration, + known_nodes=known_nodes) + out_path = Path(out_path) + out_path.parent.mkdir(parents=True, exist_ok=True) + out_path.write_text(render_lessons_md(agg), encoding="utf-8") + return out_path, agg diff --git a/tests/test_ingest.py b/tests/test_ingest.py index 41128ee..dd9e17e 100644 --- a/tests/test_ingest.py +++ b/tests/test_ingest.py @@ -66,3 +66,35 @@ def test_answer_in_body(tmp_path): out = save_query_result("what is the answer?", answer, mem) content = out.read_text() assert answer in content + +def test_outcome_in_frontmatter_and_body(tmp_path): + """An outcome signal is written to both frontmatter (for `reflect`) and an + ## Outcome body section (so it round-trips into the graph on re-extraction).""" + out = save_query_result("q", "a", tmp_path / "memory", outcome="useful") + content = out.read_text() + assert 'outcome: "useful"' in content + assert "## Outcome" in content + assert "- Signal: useful" in content + + +def test_correction_in_frontmatter_and_body(tmp_path): + out = save_query_result( + "what hashes passwords?", "MD5", tmp_path / "memory", + outcome="corrected", correction="It's bcrypt, see PasswordHasher", + ) + content = out.read_text() + assert 'correction: "It\'s bcrypt, see PasswordHasher"' in content + assert "- Correction: It's bcrypt, see PasswordHasher" in content + + +def test_no_outcome_means_no_outcome_section(tmp_path): + """Backward compatible: a result without an outcome looks exactly as before.""" + out = save_query_result("q", "a", tmp_path / "memory") + content = out.read_text() + assert "outcome:" not in content + assert "## Outcome" not in content + + +def test_invalid_outcome_rejected(tmp_path): + with pytest.raises(ValueError): + save_query_result("q", "a", tmp_path / "memory", outcome="great") diff --git a/tests/test_reflect.py b/tests/test_reflect.py new file mode 100644 index 0000000..2a7660a --- /dev/null +++ b/tests/test_reflect.py @@ -0,0 +1,566 @@ +"""Tests for `graphify reflect` and the work-memory reflection layer. + +`graphify reflect` reads the outcome-tagged Q&A docs that `graphify save-result` +files into graphify-out/memory/ and writes a deterministic lessons artifact +(graphify-out/reflections/LESSONS.md) an agent can load next session: preferred +sources, known dead ends, and corrections — optionally grouped by community. + +Covers the pure aggregation/rendering helpers (deterministic, no LLM, no graph +required) and the end-to-end CLI, including the "second session benefits from the +first" worked example from the issue. +""" +from __future__ import annotations + +import json +import subprocess +import sys +from datetime import datetime, timedelta, timezone +from pathlib import Path + +import pytest + +from graphify.ingest import save_query_result +from graphify.reflect import ( + aggregate_lessons, + load_memory_docs, + parse_memory_doc, + reflect, + render_lessons_md, +) + +PYTHON = sys.executable +FIXTURES = Path(__file__).parent / "fixtures" + +# Fixed clock so time-decay scoring is byte-stable in tests (reflect/aggregate take `now`). +_NOW = datetime(2026, 6, 1, tzinfo=timezone.utc) + + +def _days_before(n: int) -> str: + return (_NOW - timedelta(days=n)).isoformat() + + +def _run(args: list[str], cwd: Path) -> subprocess.CompletedProcess: + return subprocess.run( + [PYTHON, "-m", "graphify"] + args, + cwd=cwd, capture_output=True, text=True, + ) + + +# --- frontmatter parsing ------------------------------------------------------- + + +def test_parse_round_trips_a_saved_doc(tmp_path): + """parse_memory_doc reads back exactly what save_query_result wrote, including + an escaped question and the source_nodes flow list.""" + out = save_query_result( + 'what is "attention"?', "softmax", tmp_path / "memory", + query_type="explain", source_nodes=["AttentionLayer", "SoftmaxFunc"], + outcome="useful", + ) + parsed = parse_memory_doc(out.read_text(encoding="utf-8")) + assert parsed is not None + assert parsed["type"] == "explain" + assert parsed["question"] == 'what is "attention"?' + assert parsed["outcome"] == "useful" + assert parsed["source_nodes"] == ["AttentionLayer", "SoftmaxFunc"] + + +def test_parse_returns_none_for_foreign_doc(): + """A plain markdown file with no frontmatter is skipped, not crashed on.""" + assert parse_memory_doc("# just a note\n\nno frontmatter here\n") is None + assert parse_memory_doc("") is None + + +def test_round_trip_survives_backslash_newline_and_quoted_node(tmp_path): + """save -> parse preserves tricky characters in the question, the correction, + and (the previously-unescaped) source-node names exactly.""" + out = save_query_result( + r'path is C:\Users and a "quote"', "a", tmp_path / "memory", + source_nodes=[r'Node"With\Quote'], + outcome="corrected", correction="line1\nline2", + ) + parsed = parse_memory_doc(out.read_text(encoding="utf-8")) + assert parsed is not None + assert parsed["question"] == r'path is C:\Users and a "quote"' + assert parsed["correction"] == "line1\nline2" + assert parsed["source_nodes"] == [r'Node"With\Quote'] + + +def test_parse_handles_crlf(): + doc = "---\r\ntype: \"query\"\r\noutcome: \"useful\"\r\nsource_nodes: [\"A\"]\r\n---\r\n# body\r\n" + parsed = parse_memory_doc(doc) + assert parsed is not None + assert parsed["outcome"] == "useful" + assert parsed["source_nodes"] == ["A"] + + +def test_load_memory_docs_skips_foreign_and_sorts(tmp_path): + mem = tmp_path / "memory" + mem.mkdir() + (mem / "foreign.md").write_text("# not a memory doc\n", encoding="utf-8") + save_query_result("first", "a", mem, outcome="useful") + save_query_result("second", "b", mem, outcome="dead_end") + docs = load_memory_docs(mem) + # Foreign doc dropped; the two real docs survive. + assert len(docs) == 2 + assert {d["outcome"] for d in docs} == {"useful", "dead_end"} + + +def test_load_memory_docs_missing_dir_is_empty(tmp_path): + assert load_memory_docs(tmp_path / "nope") == [] + + +def _write_raw_doc(mem: Path, filename: str, date: str, *, outcome="dead_end", + question="q", nodes=None): + """Write a memory doc with a controlled date so ordering is deterministic to assert.""" + mem.mkdir(parents=True, exist_ok=True) + nodes = nodes or [] + lines = ["---", 'type: "query"', f'date: "{date}"', f'question: "{question}"', + 'contributor: "graphify"', f'outcome: "{outcome}"'] + if nodes: + lines.append("source_nodes: [" + ", ".join(f'"{n}"' for n in nodes) + "]") + lines += ["---", "", f"# Q: {question}", ""] + (mem / filename).write_text("\n".join(lines), encoding="utf-8") + + +def test_load_memory_docs_orders_by_date_then_filename(tmp_path): + """Determinism hinges on this sort: docs come back oldest-first, filename as tiebreak.""" + mem = tmp_path / "memory" + _write_raw_doc(mem, "z.md", "2026-03-01", question="march") + _write_raw_doc(mem, "a.md", "2026-01-01", question="january") + _write_raw_doc(mem, "b.md", "2026-02-01", question="february") + # Same date, two filenames -> filename tiebreak. + _write_raw_doc(mem, "c.md", "2026-01-01", question="january-2") + dates = [d["date"] for d in load_memory_docs(mem)] + assert dates == ["2026-01-01", "2026-01-01", "2026-02-01", "2026-03-01"] + # Within the tied date, "a.md" precedes "c.md". + tied = [d["_path"] for d in load_memory_docs(mem) if d["date"] == "2026-01-01"] + assert tied == ["a.md", "c.md"] + + +# --- aggregation --------------------------------------------------------------- + + +def _doc(outcome=None, nodes=None, question="q", correction="", date="2026-01-01"): + return { + "outcome": outcome, "source_nodes": nodes or [], + "question": question, "correction": correction, "date": date, + } + + +def test_aggregate_counts_each_outcome(): + docs = [ + _doc("useful", ["A"]), _doc("useful", ["A", "B"]), + _doc("dead_end", ["C"]), _doc("corrected", correction="use D"), + _doc(None), + ] + agg = aggregate_lessons(docs) + assert agg["total"] == 5 + assert agg["counts"] == {"useful": 2, "dead_end": 1, "corrected": 1, "unmarked": 1} + + +def test_sources_split_into_preferred_tentative_contested(): + """Corroboration (k>=2) + sign decide the bucket, not raw frequency: + A is useful twice but also a dead end -> contested; B twice-useful -> preferred; + C once-useful -> tentative.""" + docs = [ + _doc("useful", ["A", "B"]), _doc("useful", ["A", "B"]), + _doc("useful", ["C"]), + _doc("dead_end", ["A"]), # gives A a negative signal + ] + agg = aggregate_lessons(docs, now=_NOW, min_corroboration=2) + preferred = [e["node"] for e in agg["preferred"]] + tentative = [e["node"] for e in agg["tentative"]] + contested = [e["node"] for e in agg["contested"]] + assert preferred == ["B"] # 2 useful, no negatives + assert tentative == ["C"] # 1 useful only + assert contested == ["A"] # 2 useful + 1 dead end + # A never silently appears as a plain preferred/tentative source. + assert "A" not in preferred and "A" not in tentative + + +def test_corroboration_threshold_promotes_only_repeated_nodes(): + """One save can't mint a 'preferred' lesson; a second distinct result promotes it.""" + one = aggregate_lessons([_doc("useful", ["A"])], now=_NOW, min_corroboration=2) + assert [e["node"] for e in one["tentative"]] == ["A"] + assert one["preferred"] == [] + + two = aggregate_lessons( + [_doc("useful", ["A"]), _doc("useful", ["A"])], now=_NOW, min_corroboration=2) + assert [e["node"] for e in two["preferred"]] == ["A"] + assert two["tentative"] == [] + + +def test_recency_decides_contested_verdict(): + """A fresh dead_end outweighs a stale useful (30d half-life), so the contested + node leans 'dead end'; flip the dates and it leans 'useful'.""" + stale_useful = _doc("useful", ["N"], date=_days_before(120)) + fresh_deadend = _doc("dead_end", ["N"], date=_days_before(1)) + agg = aggregate_lessons([stale_useful, fresh_deadend], now=_NOW) + contested = agg["contested"] + assert len(contested) == 1 and contested[0]["node"] == "N" + assert contested[0]["verdict"] == "dead end" + + flipped = aggregate_lessons( + [_doc("useful", ["N"], date=_days_before(1)), + _doc("dead_end", ["N"], date=_days_before(120))], now=_NOW) + assert flipped["contested"][0]["verdict"] == "useful" + + +def test_node_existence_gate_drops_stale_nodes(): + """A cited node no longer in the graph is dropped from lessons entirely.""" + docs = [_doc("useful", ["Alive", "Deleted"]), _doc("useful", ["Alive", "Deleted"])] + agg = aggregate_lessons(docs, now=_NOW, known_nodes={"Alive"}) + names = [e["node"] for e in agg["preferred"] + agg["tentative"] + agg["contested"]] + assert "Deleted" not in names + assert "Alive" in names + + +def test_corroboration_counts_distinct_docs_not_citations(): + """A node cited twice *within one doc* counts as ONE corroborating result, so it + stays tentative under k=2 — guards the dict.fromkeys per-doc dedup.""" + agg = aggregate_lessons([_doc("useful", ["A", "A"])], now=_NOW, min_corroboration=2) + assert agg["preferred"] == [] + assert [e["node"] for e in agg["tentative"]] == ["A"] + assert agg["tentative"][0]["n"] == 1 + + +def test_min_corroboration_is_honored_not_hardcoded(): + """Two distinct useful results -> preferred at k=2, but only tentative at k=3.""" + docs = [_doc("useful", ["A"]), _doc("useful", ["A"])] + assert [e["node"] for e in aggregate_lessons(docs, now=_NOW, min_corroboration=2)["preferred"]] == ["A"] + at_k3 = aggregate_lessons(docs, now=_NOW, min_corroboration=3) + assert at_k3["preferred"] == [] + assert [e["node"] for e in at_k3["tentative"]] == ["A"] + + +def test_half_life_actually_feeds_decay(): + """Two stale useful + one fresh dead_end: a long half-life (≈no decay) lets the 2 + useful win; a short half-life lets the fresh dead end win. Proves the flag feeds + the decay, not just the default.""" + docs = [ + _doc("useful", ["N"], date=_days_before(90)), + _doc("useful", ["N"], date=_days_before(90)), + _doc("dead_end", ["N"], date=_days_before(1)), + ] + long_hl = aggregate_lessons(docs, now=_NOW, half_life_days=100000) + short_hl = aggregate_lessons(docs, now=_NOW, half_life_days=10) + assert long_hl["contested"][0]["verdict"] == "useful" + assert short_hl["contested"][0]["verdict"] == "dead end" + + +def test_evenly_split_verdict_when_signals_cancel(): + """A same-date useful + dead_end on one node cancel to score 0 -> 'evenly split'.""" + day = _days_before(5) + agg = aggregate_lessons( + [_doc("useful", ["N"], date=day), _doc("dead_end", ["N"], date=day)], now=_NOW) + assert agg["contested"][0]["verdict"] == "even" + assert "evenly split" in render_lessons_md(agg) + + +def test_nonpositive_half_life_disables_decay(): + """half_life<=0 turns decay off (full weight), so a stale useful and a fresh + dead_end weigh equally and cancel.""" + docs = [_doc("useful", ["N"], date=_days_before(365)), + _doc("dead_end", ["N"], date=_days_before(1))] + agg = aggregate_lessons(docs, now=_NOW, half_life_days=0) + assert agg["contested"][0]["verdict"] == "even" + + +def test_negative_only_node_absent_from_sources(): + """A node seen only in dead_end docs never appears as a source bucket entry, but + its dead-end question still renders.""" + agg = aggregate_lessons([_doc("dead_end", ["Bad"], question="why?")], now=_NOW) + names = [e["node"] for e in agg["preferred"] + agg["tentative"] + agg["contested"]] + assert "Bad" not in names + assert agg["dead_ends"][0]["nodes"] == ["Bad"] + + +def test_dead_ends_and_corrections_collected(): + docs = [ + _doc("dead_end", ["RedisClient"], question="where is the cache?"), + _doc("corrected", question="what hashes pw?", correction="bcrypt"), + ] + agg = aggregate_lessons(docs) + assert agg["dead_ends"][0]["question"] == "where is the cache?" + assert agg["dead_ends"][0]["nodes"] == ["RedisClient"] + assert agg["corrections"][0]["correction"] == "bcrypt" + + +def test_dead_ends_and_corrections_follow_doc_order(tmp_path): + """dead_ends/corrections are appended in doc order, so their determinism rides on + load_memory_docs' (date, filename) sort — assert that, not just their presence.""" + mem = tmp_path / "memory" + _write_raw_doc(mem, "later.md", "2026-02-01", outcome="dead_end", question="second") + _write_raw_doc(mem, "earlier.md", "2026-01-01", outcome="dead_end", question="first") + agg = aggregate_lessons(load_memory_docs(mem)) + assert [d["question"] for d in agg["dead_ends"]] == ["first", "second"] + + +def test_no_community_grouping_without_graph(): + agg = aggregate_lessons([_doc("useful", ["A"])]) + assert agg["by_community"] == {} + + +def test_doc_community_tie_breaks_to_smallest_label(): + """A doc whose source nodes split evenly across communities lands in the + lexicographically-smallest one — deterministically, regardless of node order.""" + nc = {"x": "Zeta", "y": "Alpha"} + agg1 = aggregate_lessons([_doc("useful", ["x", "y"])], nc) + agg2 = aggregate_lessons([_doc("useful", ["y", "x"])], nc) + assert "Alpha" in agg1["by_community"] and "Zeta" not in agg1["by_community"] + assert agg1["by_community"].keys() == agg2["by_community"].keys() + + +def test_community_grouping_uses_plurality_community(): + node_community = {"A": "Auth", "B": "Auth", "C": "Cache"} + docs = [ + _doc("useful", ["A", "B", "C"]), # plurality Auth (2 vs 1) + _doc("dead_end", ["C"]), # Cache + _doc("useful", ["Z"]), # unknown node -> Uncategorized + ] + agg = aggregate_lessons(docs, node_community) + assert set(agg["by_community"]) == {"Auth", "Cache", "Uncategorized"} + assert agg["by_community"]["Auth"]["counts"]["useful"] == 1 + assert agg["by_community"]["Cache"]["counts"]["dead_end"] == 1 + assert agg["by_community"]["Uncategorized"]["counts"]["useful"] == 1 + + +# --- rendering ----------------------------------------------------------------- + + +def test_render_is_deterministic(): + docs = [_doc("useful", ["A", "B"]), _doc("dead_end", ["C"], question="dead?")] + agg = aggregate_lessons(docs) + assert render_lessons_md(agg) == render_lessons_md(agg) + + +def test_render_has_summary_and_sections(): + docs = [ + _doc("useful", ["AuthMiddleware"]), + _doc("dead_end", ["RedisClient"], question="where is the cache?"), + _doc("corrected", question="pw?", correction="bcrypt"), + ] + md = render_lessons_md(aggregate_lessons(docs)) + assert "# Lessons" in md + assert "1 useful · 1 dead ends · 1 corrected" in md + assert "`AuthMiddleware`" in md + assert "where is the cache?" in md + assert "bcrypt" in md + # No graph -> no per-topic section. + assert "## By topic" not in md + + +def test_render_includes_by_topic_when_graph_present(): + node_community = {"A": "Auth"} + md = render_lessons_md(aggregate_lessons([_doc("useful", ["A"])], node_community)) + assert "## By topic" in md + assert "### Auth" in md + + +def test_topic_sections_alpha_with_uncategorized_last(): + """Topic headers render alphabetically, with Uncategorized always last.""" + nc = {"a": "Zeta", "b": "Alpha"} + docs = [_doc("useful", ["a"]), _doc("useful", ["b"]), _doc("useful", ["unknown"])] + md = render_lessons_md(aggregate_lessons(docs, nc)) + headers = [line[4:] for line in md.splitlines() if line.startswith("### ")] + assert headers == ["Alpha", "Zeta", "Uncategorized"] + + +def test_render_byte_stable_across_independent_aggregations(tmp_path): + """The headline guarantee: identical memory/ contents + same `now` -> byte-identical + output, built from scratch twice (not just render(agg)==render(agg)).""" + mem = tmp_path / "memory" + _write_raw_doc(mem, "a.md", "2026-01-01", outcome="useful", nodes=["A", "B"]) + _write_raw_doc(mem, "b.md", "2026-01-02", outcome="dead_end", question="dead?") + first = render_lessons_md(aggregate_lessons(load_memory_docs(mem), now=_NOW)) + second = render_lessons_md(aggregate_lessons(load_memory_docs(mem), now=_NOW)) + assert first == second + + +def test_contested_node_renders_once_under_contested(): + """A mixed-signal node appears in a single Contested line, not silently in both + a positive bucket and elsewhere.""" + docs = [_doc("useful", ["N"]), _doc("dead_end", ["N"], question="bad?")] + md = render_lessons_md(aggregate_lessons(docs, now=_NOW)) + assert "**Contested**" in md + # Exactly one rendered line carries the node as a contested source. + contested_lines = [l for l in md.splitlines() + if l.startswith("- `N` —") and "useful" in l and "dead end" in l] + assert len(contested_lines) == 1 + + +def test_header_is_cautious(): + """The header nudges verification, not blind reuse.""" + md = render_lessons_md(aggregate_lessons([_doc("useful", ["A"])], now=_NOW)) + assert "verify before relying" in md + assert "reuse what worked" not in md + + +def test_lessons_artifact_cannot_be_globbed_back_into_memory(tmp_path): + """Regression guard: the LESSONS.md output must never be re-ingested as a memory + doc. It has no frontmatter, so parse_memory_doc rejects it and load_memory_docs + skips it even if it lands inside memory/.""" + md = render_lessons_md(aggregate_lessons([_doc("useful", ["A"])], now=_NOW)) + assert parse_memory_doc(md) is None + mem = tmp_path / "memory" + mem.mkdir() + (mem / "LESSONS.md").write_text(md, encoding="utf-8") + save_query_result("real", "a", mem, outcome="useful") + docs = load_memory_docs(mem) + assert len(docs) == 1 and docs[0]["question"] == "real" + + +def test_render_empty_memory_is_graceful(): + md = render_lessons_md(aggregate_lessons([], now=_NOW)) + assert "from 0 session memories" in md + assert "_No marked outcomes yet._" in md + + +# --- orchestrator + CLI -------------------------------------------------------- + + +def test_reflect_writes_lessons_file(tmp_path): + mem = tmp_path / "memory" + save_query_result("q1", "a1", mem, source_nodes=["A"], outcome="useful") + out_path, agg = reflect(mem, tmp_path / "reflections" / "LESSONS.md") + assert out_path.exists() + assert agg["total"] == 1 + assert "`A`" in out_path.read_text(encoding="utf-8") + + +def test_second_session_benefits_from_the_first(tmp_path): + """The issue's worked example: session 1 records a win and a dead end; session 2 + loads LESSONS.md and sees both.""" + out = tmp_path / "graphify-out" + mem = out / "memory" + + # Session 1: one useful answer, one dead end. + save_query_result( + "how does auth work?", "JWT in middleware", mem, + source_nodes=["AuthMiddleware"], outcome="useful", + ) + save_query_result( + "where is the cache?", "looked at RedisClient, not it", mem, + source_nodes=["RedisClient"], outcome="dead_end", + ) + + # End of session 1 -> reflect. + lessons = out / "reflections" / "LESSONS.md" + reflect(mem, lessons) + + # Session 2 loads the lessons doc. + body = lessons.read_text(encoding="utf-8") + assert "`AuthMiddleware`" in body # start here next time + assert "where is the cache?" in body # don't re-derive this dead end + + +def test_cli_reflect_end_to_end(tmp_path): + cwd = tmp_path + r1 = _run(["save-result", "--question", "how does auth work?", + "--answer", "JWT", "--nodes", "AuthMiddleware", + "--outcome", "useful"], cwd) + assert r1.returncode == 0, r1.stderr + r2 = _run(["reflect"], cwd) + assert r2.returncode == 0, r2.stderr + assert "Reflected 1 memories" in r2.stdout + lessons = cwd / "graphify-out" / "reflections" / "LESSONS.md" + assert lessons.exists() + assert "`AuthMiddleware`" in lessons.read_text(encoding="utf-8") + + +def test_cli_save_result_rejects_bad_outcome(tmp_path): + """argparse `choices` rejects an unknown outcome before save_query_result runs.""" + r = _run(["save-result", "--question", "q", "--answer", "a", + "--outcome", "great"], tmp_path) + assert r.returncode != 0 + assert "great" in (r.stderr + r.stdout) + + +def test_cli_reflect_cold_start_writes_empty_lessons(tmp_path): + """First run with no graphify-out/memory/ still succeeds and writes a valid doc.""" + r = _run(["reflect"], tmp_path) + assert r.returncode == 0, r.stderr + assert "Reflected 0 memories" in r.stdout + lessons = tmp_path / "graphify-out" / "reflections" / "LESSONS.md" + assert lessons.exists() + assert "from 0 session memories" in lessons.read_text(encoding="utf-8") + + +def test_cli_reflect_respects_out_flag(tmp_path): + cwd = tmp_path + _run(["save-result", "--question", "q", "--answer", "a", + "--outcome", "useful", "--nodes", "X"], cwd) + dest = cwd / "custom" / "lessons.md" + r = _run(["reflect", "--out", str(dest)], cwd) + assert r.returncode == 0, r.stderr + assert dest.exists() + + +def test_cli_reflect_groups_by_community_when_graph_present(tmp_path): + """With a real graph.json present, reflect auto-detects it and groups lessons + under the community of the cited node — including when the node is cited by its + LABEL (what save-result records), not its id (regression guard: keying community + lookup on ids alone collapsed every lesson into Uncategorized).""" + out = _make_graph(tmp_path) + graph = json.loads((out / "graph.json").read_text()) + node_label = graph["nodes"][0]["label"] + + _run(["save-result", "--question", "q", "--answer", "a", + "--nodes", node_label, "--outcome", "useful"], tmp_path) + r = _run(["reflect"], tmp_path) + assert r.returncode == 0, r.stderr + body = (out / "reflections" / "LESSONS.md").read_text(encoding="utf-8") + assert "## By topic" in body + # The label-cited node must land in a real community, not Uncategorized. + assert "### Uncategorized" not in body + + +def test_cli_node_existence_gate_drops_stale_node_end_to_end(tmp_path): + """Through reflect()/CLI with a real graph.json: a cited node that isn't in the + graph is dropped from LESSONS.md; a real one stays. Exercises _load_known_nodes + + the wiring, not just the known_nodes param.""" + out = _make_graph(tmp_path) + # Cite the node by its LABEL — what an agent/`save-result` actually records — + # not its id. The gate must match labels too, else every real citation is + # silently dropped whenever a graph is present (regression guard). + real = json.loads((out / "graph.json").read_text())["nodes"][0]["label"] + + _run(["save-result", "--question", "q", "--answer", "a", + "--nodes", real, "GhostNode", "--outcome", "useful"], tmp_path) + r = _run(["reflect"], tmp_path) + assert r.returncode == 0, r.stderr + body = (out / "reflections" / "LESSONS.md").read_text(encoding="utf-8") + assert "GhostNode" not in body + assert f"`{real}`" in body + + +def _make_graph(tmp_path: Path) -> Path: + """Build a minimal graph.json + analysis/labels in tmp_path/graphify-out/. + + Mirrors tests/test_cli_export.py::_make_graph so reflect can be exercised with a + real community structure. + """ + out = tmp_path / "graphify-out" + out.mkdir() + extraction = json.loads((FIXTURES / "extraction.json").read_text()) + from graphify.build import build_from_json + from graphify.cluster import cluster, score_all + from graphify.analyze import god_nodes, surprising_connections + from graphify.export import to_json + + G = build_from_json(extraction) + communities = cluster(G) + cohesion = score_all(G, communities) + gods = god_nodes(G) + surprises = surprising_connections(G, communities) + to_json(G, communities, str(out / "graph.json")) + (out / ".graphify_analysis.json").write_text(json.dumps({ + "communities": {str(k): v for k, v in communities.items()}, + "cohesion": {str(k): v for k, v in cohesion.items()}, + "gods": gods, "surprises": surprises, + })) + (out / ".graphify_labels.json").write_text( + json.dumps({str(cid): f"Community {cid}" for cid in communities}) + ) + return out