From 341f6c4229ed246e42a71ae0662907802598b592 Mon Sep 17 00:00:00 2001 From: himanshupatro-334 Date: Thu, 10 Sep 2026 22:26:24 +0530 Subject: [PATCH] Add CSS and SCSS design token extraction --- graphify/cache.py | 4 +- graphify/detect.py | 2 +- graphify/extract.py | 125 +++++++++ graphify/extractors/__init__.py | 3 + graphify/extractors/css.py | 322 +++++++++++++++++++++++ tests/test_css.py | 450 ++++++++++++++++++++++++++++++++ tests/test_detect.py | 18 ++ 7 files changed, 921 insertions(+), 3 deletions(-) create mode 100644 graphify/extractors/css.py create mode 100644 tests/test_css.py diff --git a/graphify/cache.py b/graphify/cache.py index d6060ed2b7..f8191b90e6 100644 --- a/graphify/cache.py +++ b/graphify/cache.py @@ -609,7 +609,7 @@ def _relativize_source_files_in(payload: dict, root: Path) -> None: # definition_file (#2990) is a path into the scanned tree exactly like # source_file; a cache entry keeping it absolute replayed the build host's # layout on every warm hit (#3223). - for bucket in ("nodes", "edges", "hyperedges", "raw_calls"): + for bucket in ("nodes", "edges", "hyperedges", "raw_calls", "raw_token_uses"): for item in payload.get(bucket, []): if not isinstance(item, dict): continue @@ -911,7 +911,7 @@ def _absolutize_source_files_in(payload: dict, root: Path) -> None: root_resolved = Path(root).resolve() except OSError: return - for bucket in ("nodes", "edges", "hyperedges", "raw_calls"): + for bucket in ("nodes", "edges", "hyperedges", "raw_calls", "raw_token_uses"): for item in payload.get(bucket, []): if not isinstance(item, dict): continue diff --git a/graphify/detect.py b/graphify/detect.py index 340b595767..41dfd55b0c 100644 --- a/graphify/detect.py +++ b/graphify/detect.py @@ -41,7 +41,7 @@ class FileType(str, Enum): _MTIME_COARSE_S = 2.0 _MTIME_SUBSECOND_S = 0.05 -CODE_EXTENSIONS = {'.py', '.ts', '.tsx', '.mts', '.cts', '.js', '.jsx', '.mjs', '.cjs', '.ejs', '.ets', '.go', '.rs', '.java', '.groovy', '.gradle', '.cpp', '.cc', '.cxx', '.c', '.h', '.hpp', '.cu', '.cuh', '.metal', '.rb', '.rake', '.swift', '.kt', '.kts', '.cs', '.scala', '.php', '.lua', '.luau', '.toc', '.zig', '.ps1', '.psm1', '.psd1', '.ex', '.exs', '.m', '.mm', '.ml', '.mli', '.jl', '.vue', '.svelte', '.astro', '.dart', '.v', '.sv', '.svh', '.sql', '.r', '.f', '.F', '.f90', '.F90', '.f95', '.F95', '.f03', '.F03', '.f08', '.F08', '.pas', '.pp', '.dpr', '.dpk', '.lpr', '.inc', '.dfm', '.lfm', '.lpk', '.sh', '.bash', '.json', '.tf', '.tfvars', '.hcl', '.dm', '.dme', '.dmi', '.dmm', '.dmf', '.sln', '.slnx', '.csproj', '.fsproj', '.vbproj', '.xaml', '.razor', '.cshtml', '.cls', '.trigger', '.lisp', '.cl', '.lsp', '.asd', '.robot', '.resource'} +CODE_EXTENSIONS = {'.py', '.ts', '.tsx', '.mts', '.cts', '.js', '.jsx', '.mjs', '.cjs', '.ejs', '.ets', '.go', '.rs', '.java', '.groovy', '.gradle', '.cpp', '.cc', '.cxx', '.c', '.h', '.hpp', '.cu', '.cuh', '.metal', '.rb', '.rake', '.swift', '.kt', '.kts', '.cs', '.scala', '.php', '.lua', '.luau', '.toc', '.zig', '.ps1', '.psm1', '.psd1', '.ex', '.exs', '.m', '.mm', '.ml', '.mli', '.jl', '.vue', '.svelte', '.astro', '.dart', '.v', '.sv', '.svh', '.sql', '.r', '.f', '.F', '.f90', '.F90', '.f95', '.F95', '.f03', '.F03', '.f08', '.F08', '.pas', '.pp', '.dpr', '.dpk', '.lpr', '.inc', '.dfm', '.lfm', '.lpk', '.sh', '.bash', '.json', '.tf', '.tfvars', '.hcl', '.dm', '.dme', '.dmi', '.dmm', '.dmf', '.sln', '.slnx', '.csproj', '.fsproj', '.vbproj', '.xaml', '.razor', '.cshtml', '.cls', '.trigger', '.lisp', '.cl', '.lsp', '.asd', '.robot', '.resource', '.css', '.scss'} DOC_EXTENSIONS = {'.md', '.mdx', '.qmd', '.skill', '.txt', '.rst', '.html', '.yaml', '.yml'} PAPER_EXTENSIONS = {'.pdf'} IMAGE_EXTENSIONS = {'.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg'} diff --git a/graphify/extract.py b/graphify/extract.py index 99300f4303..0ef57dcde4 100644 --- a/graphify/extract.py +++ b/graphify/extract.py @@ -41,6 +41,7 @@ _resolve_cross_file_csharp_imports, _resolve_csharp_type_references, ) +from graphify.extractors.css import extract_css # noqa: F401 from graphify.extractors.dart import extract_dart # noqa: F401 from graphify.extractors.dm import extract_dm, extract_dmf, extract_dmi, extract_dmm # noqa: F401 from graphify.extractors.elixir import extract_elixir # noqa: F401 @@ -5870,6 +5871,8 @@ def add_existing_edge(edge: dict) -> None: ".vue": extract_vue, ".svelte": extract_svelte, ".astro": extract_astro, + ".css": extract_css, + ".scss": extract_css, ".dart": extract_dart, ".ml": extract_ocaml, ".mli": extract_ocaml, @@ -6318,6 +6321,110 @@ def _extract_sequential( print(f" AST extraction: {_done}/{_done} uncached files (100%)", flush=True) +def _resolve_css_tokens( + raw_token_uses: list[dict], + resolution_nodes: list[dict], + all_edges: list[dict], + *, + resolution_context_edges: list[dict] | None = None, +) -> None: + """Resolve CSS custom property references (var(--name)) to design-token definitions. + + Emits: + consumer stylesheet -> uses_token -> token node + with confidence 'EXTRACTED'. + + Resolution rules: + - same-file matching definition -> resolve; + - exactly one matching definition across available graph/resolution nodes -> resolve; + - multiple matching definitions -> emit nothing (conservative, no fan-out); + - no matching definition -> emit nothing. + """ + if not raw_token_uses or not resolution_nodes: + return + + def _norm_sf(sf: str | Path | None) -> str: + if not sf: + return "" + try: + return str(Path(sf).resolve()) + except Exception: + return str(sf).replace("\\", "/") + + # Map source_file -> file node id + sf_to_file_nid: dict[str, str] = {} + for n in resolution_nodes: + sf = n.get("source_file") + if sf and n.get("label") == Path(str(sf)).name: + norm = _norm_sf(sf) + sf_to_file_nid.setdefault(str(sf), n["id"]) + sf_to_file_nid.setdefault(norm, n["id"]) + + # Index token definition nodes + tokens_by_label: dict[str, list[dict]] = {} + tokens_by_file_and_label: dict[tuple[str, str], list[dict]] = {} + for n in resolution_nodes: + if n.get("node_kind") == "token": + lbl = n.get("label", "") + if lbl.startswith("--"): + tokens_by_label.setdefault(lbl, []).append(n) + sf = n.get("source_file") + if sf: + tokens_by_file_and_label.setdefault((_norm_sf(sf), lbl), []).append(n) + + seen_edges = { + (e["source"], e["target"], e.get("relation")) + for e in (all_edges + list(resolution_context_edges or [])) + } + + for tu in raw_token_uses: + token_name = tu.get("token_name", "") + if not token_name: + continue + + source_file = tu.get("source_file", "") + norm_sf = _norm_sf(source_file) + + # 1. Check same-file match first + target_node: dict | None = None + same_file_matches = tokens_by_file_and_label.get((norm_sf, token_name), []) + if same_file_matches: + target_node = same_file_matches[0] + else: + # 2. Exactly one matching definition across available graph/resolution nodes + global_matches = tokens_by_label.get(token_name, []) + if len(global_matches) == 1: + target_node = global_matches[0] + else: + # 0 matches or multiple ambiguous matches -> emit nothing + continue + + target_nid = target_node.get("id") + if not target_nid: + continue + + consumer_nid = sf_to_file_nid.get(source_file) or sf_to_file_nid.get(norm_sf) or tu.get("source_nid") + if not consumer_nid: + continue + + edge_key = (consumer_nid, target_nid, "uses_token") + if edge_key in seen_edges: + continue + seen_edges.add(edge_key) + + line = tu.get("line") + all_edges.append({ + "source": consumer_nid, + "target": target_nid, + "relation": "uses_token", + "confidence": "EXTRACTED", + "confidence_score": 1.0, + "source_file": source_file, + "source_location": f"L{line}" if line else None, + "weight": 1.0, + }) + + _PARALLEL_THRESHOLD = 20 @@ -6634,10 +6741,12 @@ def _describe_syntax_error(rel: str, line: "int | None", kept: int) -> str: all_nodes: list[dict] = [] all_edges: list[dict] = [] all_raw_calls: list[dict] = [] + all_raw_token_uses: list[dict] = [] for result in per_file: all_nodes.extend(result.get("nodes", [])) all_edges.extend(result.get("edges", [])) all_raw_calls.extend(result.get("raw_calls", [])) + all_raw_token_uses.extend(result.get("raw_token_uses", [])) # Function / method / class def ids for the cross-file indirect_call callable # guard. Built from the `_callable` node marker AFTER the id-remap / disambiguation # passes below (which rewrite node ids), so it can never go stale — see the @@ -6858,6 +6967,10 @@ def _portable_out_of_root_sf(p: Path) -> str: cn = rc.get("caller_nid") if cn in id_remap: rc["caller_nid"] = id_remap[cn] + for tu in all_raw_token_uses: + sn = tu.get("source_nid") + if sn in id_remap: + tu["source_nid"] = id_remap[sn] # swift_extensions[].nid is the same kind of id carrier as caller_nid # above (cache.py remaps both), consumed by _merge_swift_extensions far # below. Left stale it matches no node, so whether the extension merge @@ -6933,6 +7046,10 @@ def _portable_out_of_root_sf(p: Path) -> str: cn = rc.get("caller_nid") if cn in sym_remap: rc["caller_nid"] = sym_remap[cn] + for tu in all_raw_token_uses: + sn = tu.get("source_nid") + if sn in sym_remap: + tu["source_nid"] = sym_remap[sn] # Same for swift_extensions[].nid (see the id_remap pass above). for result in per_file: for ext in result.get("swift_extensions", []) or []: @@ -7556,6 +7673,14 @@ def _has_import_evidence(candidate_id: str) -> bool: else: run_language_resolvers(paths, per_file, all_nodes, all_edges) + if all_raw_token_uses: + _resolve_css_tokens( + all_raw_token_uses, + resolution_nodes, + all_edges, + resolution_context_edges=resolution_context_edges, + ) + # Relativize source_file fields so paths are portable across machines (#555). # When the node's id was itself minted from the absolute path, remap it to a # portable id and rewrite the edge endpoints that reference it. diff --git a/graphify/extractors/__init__.py b/graphify/extractors/__init__.py index 68ff3340c3..baa4fc9c44 100644 --- a/graphify/extractors/__init__.py +++ b/graphify/extractors/__init__.py @@ -14,6 +14,7 @@ from graphify.extractors.bash import extract_bash from graphify.extractors.blade import extract_blade from graphify.extractors.commonlisp import extract_commonlisp +from graphify.extractors.css import extract_css from graphify.extractors.dart import extract_dart from graphify.extractors.dm import extract_dm, extract_dmf, extract_dmi, extract_dmm from graphify.extractors.elixir import extract_elixir @@ -39,6 +40,7 @@ "bash": extract_bash, "blade": extract_blade, "commonlisp": extract_commonlisp, + "css": extract_css, "dart": extract_dart, "delphi_form": extract_delphi_form, "dm": extract_dm, @@ -58,6 +60,7 @@ "powershell_manifest": extract_powershell_manifest, "razor": extract_razor, "rust": extract_rust, + "scss": extract_css, "sln": extract_sln, "sql": extract_sql, "terraform": extract_terraform, diff --git a/graphify/extractors/css.py b/graphify/extractors/css.py new file mode 100644 index 0000000000..3a8775d6f5 --- /dev/null +++ b/graphify/extractors/css.py @@ -0,0 +1,322 @@ +"""CSS and SCSS design-token extractor. + +Extracts design-token definitions (--name: value;) declared within root or +theme contexts (:root, :host, html, body, [data-theme...], [data-mode...], +.dark, .theme-*, @theme, including when enclosed by @media/@layer/@supports), +and collects var(--name) references for post-extraction cross-file resolution. +""" +from __future__ import annotations + +import re +from pathlib import Path + +from graphify.extractors.base import _file_stem, _make_id + +_ROOT_START_RE = re.compile( + r'^(?::(?:root|host)(?![a-zA-Z0-9_-])|html(?![a-zA-Z0-9_-])|body(?![a-zA-Z0-9_-])|@theme(?![a-zA-Z0-9_-])|\[data-(?:theme|mode)[^\]]*\]|\.dark(?![a-zA-Z0-9_-])|\.theme-[a-zA-Z0-9_-]+)', + re.IGNORECASE, +) +_SCSS_NESTED_ROOT_RE = re.compile( + r'^&(?:(?::(?:root|host)(?![a-zA-Z0-9_-]))|(?:\[data-(?:theme|mode)[^\]]*\])|\.dark(?![a-zA-Z0-9_-])|\.theme-[a-zA-Z0-9_-]+)', + re.IGNORECASE, +) +_TRANSPARENT_AT_RULES = ("@media", "@layer", "@supports") +_URL_START_RE = re.compile(r"^url\s*\(", re.IGNORECASE) +_VAR_RE = re.compile(r"var\(\s*(--[a-zA-Z0-9_-]+)") +_DECL_RE = re.compile(r"^\s*(--[a-zA-Z0-9_-]+)\s*:\s*([^;]+?)\s*(?:;|\Z)") + + +def _strip_comments_preserving_lines(text: str) -> str: + """Replace comment characters with spaces, preserving newlines so line numbers remain exact.""" + out: list[str] = [] + i = 0 + n = len(text) + in_str: str | None = None + in_url: bool = False + while i < n: + c = text[i] + nxt = text[i + 1] if i + 1 < n else "" + if in_str: + out.append(c) + if c == "\\" and i + 1 < n: + out.append(text[i + 1]) + i += 2 + continue + elif c == in_str: + in_str = None + i += 1 + continue + + if c in ('"', "'"): + in_str = c + out.append(c) + i += 1 + continue + + # Track unquoted url(...) so protocol slashes (http://, https://) don't trigger SCSS comments + if not in_url and c in ("u", "U") and _URL_START_RE.match(text[i:]): + in_url = True + out.append(c) + i += 1 + continue + + if in_url and (c == ")" or c == "\n"): + in_url = False + out.append(c) + i += 1 + continue + + # Block comment /* ... */ + if c == "/" and nxt == "*": + out.append(" ") + out.append(" ") + i += 2 + while i < n: + if text[i] == "*" and i + 1 < n and text[i + 1] == "/": + out.append(" ") + out.append(" ") + i += 2 + break + else: + out.append("\n" if text[i] == "\n" else " ") + i += 1 + continue + + # SCSS single-line comment // ... (only outside unquoted url(...)) + if not in_url and c == "/" and nxt == "/": + out.append(" ") + out.append(" ") + i += 2 + while i < n and text[i] != "\n": + out.append(" ") + i += 1 + continue + + out.append(c) + i += 1 + + return "".join(out) + + +def _blank_strings_preserving_lines(text: str) -> str: + """Replace characters inside quoted string literals with spaces, preserving newlines.""" + out: list[str] = [] + i = 0 + n = len(text) + in_str: str | None = None + while i < n: + c = text[i] + if in_str: + if c == "\\" and i + 1 < n: + out.append(" ") + out.append("\n" if text[i + 1] == "\n" else " ") + i += 2 + continue + elif c == in_str: + in_str = None + out.append(c) + i += 1 + continue + else: + out.append("\n" if c == "\n" else " ") + i += 1 + continue + + if c in ('"', "'"): + in_str = c + out.append(c) + i += 1 + continue + + out.append(c) + i += 1 + + return "".join(out) + + +def _is_root_branch(branch: str) -> bool: + """Check whether a single selector branch is rooted and contains only root/theme selectors.""" + segments = [seg.strip() for seg in re.split(r'[\s>]+', branch) if seg.strip()] + if not segments: + return False + return all(_ROOT_START_RE.match(seg) for seg in segments) + + +def _is_root_context(stack: list[str]) -> bool: + """Return True if the current selector stack represents a root or theme context.""" + non_at = [s.strip() for s in stack if not s.strip().startswith(_TRANSPARENT_AT_RULES)] + if not non_at: + return False + base_branches = [b.strip() for b in non_at[0].split(",") if b.strip()] + if not any(_is_root_branch(b) for b in base_branches): + return False + for s in non_at[1:]: + branches = [b.strip() for b in s.split(",") if b.strip()] + for b in branches: + if not (_SCSS_NESTED_ROOT_RE.match(b) or _is_root_branch(b)): + return False + return True + + +def extract_css(path: Path) -> dict: + """Extract design-token definitions and references from a .css or .scss file.""" + str_path = str(path) + try: + src = path.read_text(encoding="utf-8", errors="replace") + except OSError as exc: + return {"nodes": [], "edges": [], "error": str(exc)} + + clean = _strip_comments_preserving_lines(src) + code_only = _blank_strings_preserving_lines(clean) + + file_nid = _make_id(str_path) + stem = _file_stem(path) + + nodes: list[dict] = [ + { + "id": file_nid, + "label": path.name, + "file_type": "code", + "source_file": str_path, + "source_location": "L1", + } + ] + edges: list[dict] = [] + raw_token_uses: list[dict] = [] + + # 1. Collect var(--token) references outside string literals + seen_uses: set[tuple[str, int]] = set() + for m in _VAR_RE.finditer(code_only): + token_name = m.group(1) + line = code_only[: m.start()].count("\n") + 1 + key = (token_name, line) + if key in seen_uses: + continue + seen_uses.add(key) + raw_token_uses.append( + { + "source_nid": file_nid, + "token_name": token_name, + "source_file": str_path, + "line": line, + } + ) + + # 2. Collect token definitions inside root/theme contexts + seen_tokens: set[str] = set() + stack: list[str] = [] + buf: list[str] = [] + i = 0 + n = len(clean) + in_str: str | None = None + + while i < n: + c = clean[i] + if in_str: + buf.append(c) + if c == "\\" and i + 1 < n: + buf.append(clean[i + 1]) + i += 2 + continue + elif c == in_str: + in_str = None + i += 1 + continue + + if c in ('"', "'"): + in_str = c + buf.append(c) + i += 1 + continue + + if c == "{": + selector = "".join(buf).strip() + stack.append(selector) + buf = [] + i += 1 + continue + + if c == "}": + stmt = "".join(buf).strip() + if stmt and stack and _is_root_context(stack): + m = _DECL_RE.match(stmt) + if m: + token_name = m.group(1) + if token_name not in seen_tokens: + seen_tokens.add(token_name) + pos = i - len(stmt) + line = clean[:pos].count("\n") + 1 + token_nid = _make_id(stem, token_name) + nodes.append( + { + "id": token_nid, + "label": token_name, + "file_type": "code", + "node_kind": "token", + "source_file": str_path, + "source_location": f"L{line}", + } + ) + edges.append( + { + "source": file_nid, + "target": token_nid, + "relation": "defines_token", + "confidence": "EXTRACTED", + "confidence_score": 1.0, + "source_file": str_path, + "source_location": f"L{line}", + "weight": 1.0, + } + ) + if stack: + stack.pop() + buf = [] + i += 1 + continue + + if c == ";": + stmt = "".join(buf).strip() + if stmt and stack and _is_root_context(stack): + m = _DECL_RE.match(stmt) + if m: + token_name = m.group(1) + if token_name not in seen_tokens: + seen_tokens.add(token_name) + pos = i - len(stmt) + line = clean[:pos].count("\n") + 1 + token_nid = _make_id(stem, token_name) + nodes.append( + { + "id": token_nid, + "label": token_name, + "file_type": "code", + "node_kind": "token", + "source_file": str_path, + "source_location": f"L{line}", + } + ) + edges.append( + { + "source": file_nid, + "target": token_nid, + "relation": "defines_token", + "confidence": "EXTRACTED", + "confidence_score": 1.0, + "source_file": str_path, + "source_location": f"L{line}", + "weight": 1.0, + } + ) + buf = [] + i += 1 + continue + + buf.append(c) + i += 1 + + return { + "nodes": nodes, + "edges": edges, + "raw_token_uses": raw_token_uses, + } diff --git a/tests/test_css.py b/tests/test_css.py new file mode 100644 index 0000000000..61a784cfe6 --- /dev/null +++ b/tests/test_css.py @@ -0,0 +1,450 @@ +"""Focused tests for CSS and SCSS design-token extraction and resolution (#3473).""" +from pathlib import Path +import pytest + +from graphify.detect import FileType, classify_file, _is_sensitive +from graphify.extract import extract, _file_node_id, _make_id +from graphify.extractors.css import extract_css + + +# ── 1. Detection ───────────────────────────────────────────────────────────── + +def test_detect_css_classified_as_code(): + assert classify_file(Path("styles.css")) == FileType.CODE + assert classify_file(Path("src/components/button.css")) == FileType.CODE + + +def test_detect_scss_classified_as_code(): + assert classify_file(Path("styles.scss")) == FileType.CODE + assert classify_file(Path("src/styles/_tokens.scss")) == FileType.CODE + + +def test_tokens_stylesheet_not_flagged_as_sensitive(): + assert not _is_sensitive(Path("tokens.css")) + assert not _is_sensitive(Path("tokens.scss")) + assert not _is_sensitive(Path("design-tokens.css")) + assert not _is_sensitive(Path("styles/tokens.scss")) + + +# ── 2. Extraction: Root & Theme Contexts ────────────────────────────────────── + +def test_extract_root_tokens(tmp_path): + f = tmp_path / "root.css" + f.write_text( + """:root { + --color-primary: #0070f3; + --spacing-sm: 8px; +} +""", + encoding="utf-8", + ) + res = extract_css(f) + token_nodes = [n for n in res["nodes"] if n.get("node_kind") == "token"] + assert len(token_nodes) == 2 + labels = {n["label"] for n in token_nodes} + assert labels == {"--color-primary", "--spacing-sm"} + + edges = [e for e in res["edges"] if e.get("relation") == "defines_token"] + assert len(edges) == 2 + for e in edges: + assert e["confidence"] == "EXTRACTED" + assert e["confidence_score"] == 1.0 + + +def test_extract_host_tokens(tmp_path): + f = tmp_path / "shadow.css" + f.write_text( + """:host { + --widget-bg: #fff; +} +""", + encoding="utf-8", + ) + res = extract_css(f) + tokens = [n["label"] for n in res["nodes"] if n.get("node_kind") == "token"] + assert tokens == ["--widget-bg"] + + +def test_extract_html_and_body_tokens(tmp_path): + f = tmp_path / "base.css" + f.write_text( + """html { + --font-base: sans-serif; +} +body { + --text-color: #333; +} +""", + encoding="utf-8", + ) + res = extract_css(f) + tokens = [n["label"] for n in res["nodes"] if n.get("node_kind") == "token"] + assert set(tokens) == {"--font-base", "--text-color"} + + +def test_extract_theme_selectors(tmp_path): + f = tmp_path / "themes.css" + f.write_text( + """[data-theme="dark"] { + --bg-dark: #121212; +} +[data-mode="dim"] { + --bg-dim: #222; +} +.dark { + --dark-accent: #f00; +} +.theme-dracula { + --dracula-pink: #ff79c6; +} +""", + encoding="utf-8", + ) + res = extract_css(f) + tokens = {n["label"] for n in res["nodes"] if n.get("node_kind") == "token"} + assert tokens == {"--bg-dark", "--bg-dim", "--dark-accent", "--dracula-pink"} + + +def test_extract_at_theme(tmp_path): + f = tmp_path / "tailwind.css" + f.write_text( + """@theme { + --font-display: 'Inter', sans-serif; +} +""", + encoding="utf-8", + ) + res = extract_css(f) + tokens = [n["label"] for n in res["nodes"] if n.get("node_kind") == "token"] + assert tokens == ["--font-display"] + + +def test_extract_nested_under_transparent_at_rules(tmp_path): + f = tmp_path / "media.css" + f.write_text( + """@media (prefers-color-scheme: dark) { + :root { + --media-dark: #000; + } +} +@layer base { + :root { + --layer-base-token: #fff; + } +} +@supports (display: grid) { + :root { + --grid-gap: 16px; + } +} +""", + encoding="utf-8", + ) + res = extract_css(f) + tokens = {n["label"] for n in res["nodes"] if n.get("node_kind") == "token"} + assert tokens == {"--media-dark", "--layer-base-token", "--grid-gap"} + + +def test_ordinary_component_selectors_rejected(tmp_path): + f = tmp_path / "components.css" + f.write_text( + """.card { + --card-bg: #fff; +} +.button { + --btn-padding: 4px; +} +#header { + --header-h: 60px; +} +table tr:hover { + --hover-bg: #f5f5f5; +} +.p-4 { + --spacing: 16px; +} +.card .dark { + --scoped-dark: #000; +} +:root .card { + --nested-card-override: #111; +} +""", + encoding="utf-8", + ) + res = extract_css(f) + tokens = [n["label"] for n in res["nodes"] if n.get("node_kind") == "token"] + assert tokens == [], f"Expected no token nodes from component selectors, got {tokens}" + + +# ── 3. SCSS Handling ───────────────────────────────────────────────────────── + +def test_scss_comments_and_nested_root(tmp_path): + f = tmp_path / "theme.scss" + f.write_text( + """// Single-line SCSS comment with --fake-token: 1; +/* Block comment with :root { --commented-token: 2; } */ +$sass-var: #ff0000; +$primary-color: #fff; + +:root { + // line comment inside root + /* block comment inside root: --inside-comment: 3; */ + --real-token: #123456; + + &.dark { + --nested-dark-token: #000; + } +} +""", + encoding="utf-8", + ) + res = extract_css(f) + tokens = {n["label"] for n in res["nodes"] if n.get("node_kind") == "token"} + assert tokens == {"--real-token", "--nested-dark-token"} + assert "--fake-token" not in tokens + assert "--commented-token" not in tokens + assert "--inside-comment" not in tokens + assert "$sass-var" not in tokens + assert "$primary-color" not in tokens + + +def test_scss_partials_extracted_normally(tmp_path): + f = tmp_path / "_tokens.scss" + f.write_text( + """:root { + --brand: #abc; +} +""", + encoding="utf-8", + ) + res = extract_css(f) + tokens = [n["label"] for n in res["nodes"] if n.get("node_kind") == "token"] + assert tokens == ["--brand"] + + +# ── 4. Identity & Schema ───────────────────────────────────────────────────── + +def test_token_identity_and_schema(tmp_path): + f_a = tmp_path / "theme-a.css" + f_b = tmp_path / "theme-b.css" + f_a.write_text(":root { --color-card: #aaa; }", encoding="utf-8") + f_b.write_text(":root { --color-card: #bbb; }", encoding="utf-8") + + res_a = extract_css(f_a) + res_b = extract_css(f_b) + + token_a = next(n for n in res_a["nodes"] if n.get("node_kind") == "token") + token_b = next(n for n in res_b["nodes"] if n.get("node_kind") == "token") + + # Distinct file-scoped IDs + assert token_a["id"] != token_b["id"] + + # Schema verification + assert token_a["file_type"] == "code" + assert token_a["node_kind"] == "token" + assert token_a["label"] == "--color-card" + assert token_a["source_file"] == str(f_a) + assert token_a["source_location"] == "L1" + + edge_a = next(e for e in res_a["edges"] if e.get("relation") == "defines_token") + assert edge_a["source"] == res_a["nodes"][0]["id"] + assert edge_a["target"] == token_a["id"] + assert edge_a["confidence"] == "EXTRACTED" + assert edge_a["confidence_score"] == 1.0 + + +# ── 5. Post-Extraction Resolution ──────────────────────────────────────────── + +def test_same_file_unique_token_resolves(tmp_path): + f = tmp_path / "styles.css" + f.write_text( + """:root { + --brand: #123; +} +.btn { + color: var(--brand); +} +""", + encoding="utf-8", + ) + g = extract([f], root=tmp_path) + edges = [e for e in g["edges"] if e.get("relation") == "uses_token"] + assert len(edges) == 1 + edge = edges[0] + token_node = next(n for n in g["nodes"] if n.get("label") == "--brand") + file_node = next(n for n in g["nodes"] if n.get("label") == "styles.css") + assert edge["source"] == file_node["id"] + assert edge["target"] == token_node["id"] + assert edge["confidence"] == "EXTRACTED" + assert edge["confidence_score"] == 1.0 + + +def test_unique_cross_file_token_resolves(tmp_path): + tokens_file = tmp_path / "tokens.css" + app_file = tmp_path / "app.css" + tokens_file.write_text(":root { --accent: #ff0; }", encoding="utf-8") + app_file.write_text(".sidebar { background: var(--accent); }", encoding="utf-8") + + g = extract([tokens_file, app_file], root=tmp_path) + uses_edges = [e for e in g["edges"] if e.get("relation") == "uses_token"] + assert len(uses_edges) == 1 + edge = uses_edges[0] + token_node = next(n for n in g["nodes"] if n.get("label") == "--accent") + app_file_node = next(n for n in g["nodes"] if n.get("label") == "app.css") + assert edge["source"] == app_file_node["id"] + assert edge["target"] == token_node["id"] + assert edge["confidence"] == "EXTRACTED" + + +def test_multiple_same_name_definitions_produce_no_uses_edge(tmp_path): + theme_a = tmp_path / "theme_a.css" + theme_b = tmp_path / "theme_b.css" + consumer = tmp_path / "consumer.css" + + theme_a.write_text(":root { --ambiguous-token: #111; }", encoding="utf-8") + theme_b.write_text(":root { --ambiguous-token: #222; }", encoding="utf-8") + consumer.write_text(".box { color: var(--ambiguous-token); }", encoding="utf-8") + + g = extract([theme_a, theme_b, consumer], root=tmp_path) + consumer_node = next(n for n in g["nodes"] if n.get("label") == "consumer.css") + consumer_uses = [ + e for e in g["edges"] + if e.get("relation") == "uses_token" and e.get("source") == consumer_node["id"] + ] + # Ambiguous cross-file references must NOT resolve (conservative, no fan-out) + assert consumer_uses == [] + + +def test_same_file_precedence_over_cross_file_ambiguity(tmp_path): + theme_a = tmp_path / "theme_a.css" + theme_b = tmp_path / "theme_b.css" + + # Both define --local-token, but theme_a also uses it + theme_a.write_text( + """:root { --local-token: #111; } +.my-el { color: var(--local-token); } +""", + encoding="utf-8", + ) + theme_b.write_text(":root { --local-token: #222; }", encoding="utf-8") + + g = extract([theme_a, theme_b], root=tmp_path) + theme_a_node = next(n for n in g["nodes"] if n.get("label") == "theme_a.css") + token_a = next( + n for n in g["nodes"] + if n.get("label") == "--local-token" and "theme_a" in n["id"] + ) + uses = [ + e for e in g["edges"] + if e.get("relation") == "uses_token" and e.get("source") == theme_a_node["id"] + ] + # Same-file match must resolve to theme_a's own token definition + assert len(uses) == 1 + assert uses[0]["target"] == token_a["id"] + + +def test_nonexistent_token_produces_no_uses_edge(tmp_path): + f = tmp_path / "orphan.css" + f.write_text(".box { color: var(--nonexistent-var); }", encoding="utf-8") + g = extract([f], root=tmp_path) + uses = [e for e in g["edges"] if e.get("relation") == "uses_token"] + assert uses == [] + + +# ── 6. Incremental / Context Resolution ────────────────────────────────────── + +def test_incremental_resolution_with_context_nodes(tmp_path): + # Simulate an incremental build where consumer.css is freshly extracted + # while the token definition in tokens.css is supplied via resolution_context_nodes + consumer = tmp_path / "consumer.css" + consumer.write_text(".panel { background: var(--theme-color); }", encoding="utf-8") + + # Mock token node from unchanged tokens.css + token_id = "tokens_theme_color" + context_node = { + "id": token_id, + "label": "--theme-color", + "file_type": "code", + "node_kind": "token", + "source_file": str(tmp_path / "tokens.css"), + "source_location": "L1", + } + + g = extract( + [consumer], + root=tmp_path, + resolution_context_nodes=[context_node], + ) + consumer_node = next(n for n in g["nodes"] if n.get("label") == "consumer.css") + uses = [ + e for e in g["edges"] + if e.get("relation") == "uses_token" and e.get("source") == consumer_node["id"] + ] + assert len(uses) == 1 + assert uses[0]["target"] == token_id + assert uses[0]["confidence"] == "EXTRACTED" + + +# ── 7. Regression Tests for Edge Cases ──────────────────────────────────────── + +def test_unquoted_url_and_scss_comments(tmp_path): + f = tmp_path / "urls.scss" + f.write_text( + """:root { + --font-url: url(https://example.com/font.woff); + --valid: 123; + --http-url: url(http://example.com/image.png); +} +// --ignored: value; +""", + encoding="utf-8", + ) + res = extract_css(f) + tokens = {n["label"] for n in res["nodes"] if n.get("node_kind") == "token"} + assert "--font-url" in tokens + assert "--valid" in tokens + assert "--http-url" in tokens + assert "--ignored" not in tokens + + +def test_var_inside_quoted_strings_ignored(tmp_path): + f = tmp_path / "content.css" + f.write_text( + """:root { + --brand: blue; +} + +.example::before { + content: "Use var(--brand) here"; + color: var(--brand); +} +""", + encoding="utf-8", + ) + res = extract_css(f) + # Only the real outside-string color: var(--brand) should be captured + assert len(res["raw_token_uses"]) == 1 + use = res["raw_token_uses"][0] + assert use["token_name"] == "--brand" + assert use["line"] == 7 # line of `color: var(--brand);` + + +def test_warm_cache_roundtrip(tmp_path): + t = tmp_path / "tokens.css" + t.write_text(":root { --cached-token: #123; }", encoding="utf-8") + c = tmp_path / "consumer.css" + c.write_text(".btn { color: var(--cached-token); }", encoding="utf-8") + + # Cold extraction + g1 = extract([t, c], root=tmp_path) + uses1 = [e for e in g1["edges"] if e.get("relation") == "uses_token"] + assert len(uses1) == 1 + + # Warm extraction from cache + g2 = extract([t, c], root=tmp_path) + uses2 = [e for e in g2["edges"] if e.get("relation") == "uses_token"] + assert len(uses2) == 1 + assert uses1[0]["source"] == uses2[0]["source"] + assert uses1[0]["target"] == uses2[0]["target"] + diff --git a/tests/test_detect.py b/tests/test_detect.py index 1bf6b056bc..a45fae400f 100644 --- a/tests/test_detect.py +++ b/tests/test_detect.py @@ -42,6 +42,15 @@ def test_classify_powershell_manifest(): # #1331: .psd1 manifests must be classified as CODE so the manifest extractor runs. assert classify_file(Path("MyModule.psd1")) == FileType.CODE +def test_classify_css(): + # #3473: .css files must be classified as CODE for token extraction + assert classify_file(Path("styles.css")) == FileType.CODE + +def test_classify_scss(): + # #3473: .scss files must be classified as CODE for token extraction + assert classify_file(Path("styles.scss")) == FileType.CODE + + def test_classify_markdown(): assert classify_file(Path("README.md")) == FileType.DOCUMENT @@ -1740,6 +1749,15 @@ def test_sensitive_does_not_flag_ruby_code_modules(): assert not _is_sensitive(Path("app/controllers/api/v1/passwords_controller.rb")) +def test_sensitive_does_not_flag_stylesheet_tokens(): + # #3473: stylesheet token files are code, not sensitive secret stores + assert not _is_sensitive(Path("tokens.css")) + assert not _is_sensitive(Path("tokens.scss")) + assert not _is_sensitive(Path("design-tokens.css")) + assert not _is_sensitive(Path("styles/tokens.scss")) + + + def test_sensitive_still_flags_data_secret_stores(): # #1666 guard: the exemption is ONLY for real source code, not data/config # formats — credentials.json / oauth_token.json / secrets.yaml are the secret