fix: guard label/text normalizers against None node labels (#1195)

Guards _norm, _norm_label, and _strip_diacritics against None node labels that cause TypeError in unicodedata.normalize(). Fixes #1194. Consistent with existing security.py:270 precedent.

Co-authored-by: freiit <freiit@users.noreply.github.com>
This commit is contained in:
freiit
2026-06-08 23:22:15 +01:00
committed by GitHub
parent 7477b469ee
commit 3602c8031e
4 changed files with 12 additions and 4 deletions
+3 -1
View File
@@ -311,8 +311,10 @@ def build(
return build_from_json(combined, directed=directed, root=root)
def _norm_label(label: str) -> str:
def _norm_label(label: str | None) -> str:
"""Canonical dedup key — Unicode-aware, preserves CJK/word characters."""
if not isinstance(label, str):
label = "" if label is None else str(label)
label = unicodedata.normalize("NFKC", label)
return re.sub(r"[\W_ ]+", " ", label.casefold(), flags=re.UNICODE).strip()
+3 -1
View File
@@ -15,8 +15,10 @@ from rapidfuzz.distance import JaroWinkler
# ── helpers ───────────────────────────────────────────────────────────────────
def _norm(label: str) -> str:
def _norm(label: str | None) -> str:
"""Lowercase + collapse non-alphanumeric runs to space (Unicode-aware)."""
if not isinstance(label, str):
label = "" if label is None else str(label)
label = unicodedata.normalize("NFKC", label)
return re.sub(r"[\W_]+", " ", label.casefold(), flags=re.UNICODE).strip()
+3 -1
View File
@@ -102,8 +102,10 @@ def _obsidian_tag(name: str) -> str:
return re.sub(r"[^a-zA-Z0-9_\-/]", "", name.replace(" ", "_"))
def _strip_diacritics(text: str) -> str:
def _strip_diacritics(text: str | None) -> str:
import unicodedata
if not isinstance(text, str):
text = "" if text is None else str(text)
nfkd = unicodedata.normalize("NFKD", text)
return "".join(c for c in nfkd if not unicodedata.combining(c))
+3 -1
View File
@@ -51,8 +51,10 @@ def _communities_from_graph(G: nx.Graph) -> dict[int, list[str]]:
return communities
def _strip_diacritics(text: str) -> str:
def _strip_diacritics(text: str | None) -> str:
import unicodedata
if not isinstance(text, str):
text = "" if text is None else str(text)
nfkd = unicodedata.normalize("NFKD", text)
return "".join(c for c in nfkd if not unicodedata.combining(c))