fix: guard label/text normalizers against None node labels (#1195)
Guards _norm, _norm_label, and _strip_diacritics against None node labels that cause TypeError in unicodedata.normalize(). Fixes #1194. Consistent with existing security.py:270 precedent. Co-authored-by: freiit <freiit@users.noreply.github.com>
This commit is contained in:
+3
-1
@@ -311,8 +311,10 @@ def build(
|
||||
return build_from_json(combined, directed=directed, root=root)
|
||||
|
||||
|
||||
def _norm_label(label: str) -> str:
|
||||
def _norm_label(label: str | None) -> str:
|
||||
"""Canonical dedup key — Unicode-aware, preserves CJK/word characters."""
|
||||
if not isinstance(label, str):
|
||||
label = "" if label is None else str(label)
|
||||
label = unicodedata.normalize("NFKC", label)
|
||||
return re.sub(r"[\W_ ]+", " ", label.casefold(), flags=re.UNICODE).strip()
|
||||
|
||||
|
||||
+3
-1
@@ -15,8 +15,10 @@ from rapidfuzz.distance import JaroWinkler
|
||||
|
||||
# ── helpers ───────────────────────────────────────────────────────────────────
|
||||
|
||||
def _norm(label: str) -> str:
|
||||
def _norm(label: str | None) -> str:
|
||||
"""Lowercase + collapse non-alphanumeric runs to space (Unicode-aware)."""
|
||||
if not isinstance(label, str):
|
||||
label = "" if label is None else str(label)
|
||||
label = unicodedata.normalize("NFKC", label)
|
||||
return re.sub(r"[\W_]+", " ", label.casefold(), flags=re.UNICODE).strip()
|
||||
|
||||
|
||||
+3
-1
@@ -102,8 +102,10 @@ def _obsidian_tag(name: str) -> str:
|
||||
return re.sub(r"[^a-zA-Z0-9_\-/]", "", name.replace(" ", "_"))
|
||||
|
||||
|
||||
def _strip_diacritics(text: str) -> str:
|
||||
def _strip_diacritics(text: str | None) -> str:
|
||||
import unicodedata
|
||||
if not isinstance(text, str):
|
||||
text = "" if text is None else str(text)
|
||||
nfkd = unicodedata.normalize("NFKD", text)
|
||||
return "".join(c for c in nfkd if not unicodedata.combining(c))
|
||||
|
||||
|
||||
+3
-1
@@ -51,8 +51,10 @@ def _communities_from_graph(G: nx.Graph) -> dict[int, list[str]]:
|
||||
return communities
|
||||
|
||||
|
||||
def _strip_diacritics(text: str) -> str:
|
||||
def _strip_diacritics(text: str | None) -> str:
|
||||
import unicodedata
|
||||
if not isinstance(text, str):
|
||||
text = "" if text is None else str(text)
|
||||
nfkd = unicodedata.normalize("NFKD", text)
|
||||
return "".join(c for c in nfkd if not unicodedata.combining(c))
|
||||
|
||||
|
||||
Reference in New Issue
Block a user