Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 16 additions & 0 deletions src/cgis/extractors/base.py
Original file line number Diff line number Diff line change
@@ -1,10 +1,26 @@
"""Base Extractor Interface"""

from abc import ABC, abstractmethod
from typing import Protocol, runtime_checkable

from cgis.core.models import Edge, Node


@runtime_checkable
class ModuleNamer(Protocol):
"""An extractor that can say which module FQN it gives a file path.

Structural rather than a base-class method: only some extractors can answer,
and a caller that needs one — the workspace-package map naming a package's
directory (#504) — can then ask for it without importing any language's
extractor, which would load that language's grammar for every run (#506).
"""

def module_fqn(self, file_path: str) -> str:
"""The module FQN this extractor gives `file_path`."""
...


class BaseExtractor(ABC):
"""
Abstract Base Class for all language-specific extractors.
Expand Down
11 changes: 10 additions & 1 deletion src/cgis/extractors/typescript_extractor.py
Original file line number Diff line number Diff line change
Expand Up @@ -113,11 +113,20 @@ def __init__(self, tsx: bool = False, source_roots: list[str] | None = None) ->
lang = tsts.language_tsx() if tsx else tsts.language_typescript()
self._parser = Parser(Language(lang))

def module_fqn(self, file_path: str) -> str:
"""The module FQN this extractor gives `file_path`, with its source root applied.

`parse` names every file with this, so a caller that needs to spell a path
the way the graph's node ids spell it — the workspace-package map naming a
package's directory (#504) — gets the same rule rather than a copy of it.
"""
return file_path_to_module_fqn(file_path, self._pick_source_root(file_path))

def parse(self, code: str, file_path: str) -> tuple[list[Node], list[Edge]]:
"""Extract nodes and edges from TypeScript source code."""
code_bytes = code.encode("utf-8")
tree = self._parser.parse(code_bytes)
module_fqn = file_path_to_module_fqn(file_path, self._pick_source_root(file_path))
module_fqn = self.module_fqn(file_path)

file_node = Node(
id=module_fqn,
Expand Down
23 changes: 20 additions & 3 deletions src/cgis/pipeline.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@
from cgis.extractors.base import BaseExtractor
from cgis.resolver.engine import ResolverEngine
from cgis.resolver.uplift import SemanticUpliftEngine
from cgis.workspaces import PACKAGE_MANIFEST, WorkspacePackages

if TYPE_CHECKING:
from cgis.storage.sqlite_store import SQLiteStore
Expand Down Expand Up @@ -96,7 +97,7 @@
# `cgis ingest /abs/path/src` produce identical file_paths and FQNs.
return path.resolve()

def run(

Check failure on line 100 in src/cgis/pipeline.py

View check run for this annotation

SonarQubeCloud / SonarCloud Code Analysis

Refactor this function to reduce its Cognitive Complexity from 20 to the 15 allowed.

See more on https://sonarcloud.io/project/issues?id=zaebee_codegraph-brain&issues=AaDZGAc36R5Iaf-qfiUh&open=AaDZGAc36R5Iaf-qfiUh&pullRequest=506
self,
repo_path: str,
store: "SQLiteStore | None" = None,
Expand All @@ -122,8 +123,8 @@
# file_path -> new hash, only for files that were re-extracted
changed_files: dict[str, str] = {}
found_file_paths: set[str] = set()

workspace_root = self.workspace_root(repo_path)
packages = WorkspacePackages(workspace_root, self._extractors)

with Progress(
SpinnerColumn(),
Expand All @@ -137,6 +138,8 @@
for root, dirs, files in workspace_root.walk():
dirs[:] = [d for d in dirs if not d.startswith(".") and d not in self._excluded]
for file in files:
if file == PACKAGE_MANIFEST:
packages.note(root / file)
extractor = self._get_extractor(file)
if not extractor:
continue
Expand Down Expand Up @@ -168,14 +171,24 @@
# nothing went stale, the persisted graph is already correct.
# Re-running the resolver + persistence + uplift would rebuild the
# whole graph from the DB for zero benefit, so skip them entirely.
if self._is_noop_incremental(store, changed_files, found_file_paths):
workspace_packages = packages.unambiguous()
# A renamed or moved package changes what unchanged files' imports mean,
# and an incremental run re-resolves only changed files (#504).
packages_changed = (
store is not None
and not rebuild
and (store.get_workspace_packages() or {}) != workspace_packages
)
if not packages_changed and self._is_noop_incremental(
store, changed_files, found_file_paths
):
logger.info("No changes detected — skipping resolution and persistence.")
return all_nodes, all_edges, []

# Task 2: Resolution
resolve_task = progress.add_task(description="Resolving semantic links...", total=None)
logger.info("Starting resolution phase...")
resolver = ResolverEngine(all_nodes, all_edges)
resolver = ResolverEngine(all_nodes, all_edges, workspace_packages=workspace_packages)
resolved_edges, virtual_nodes = resolver.resolve()
all_nodes.extend(virtual_nodes)
progress.update(resolve_task, advance=1)
Expand All @@ -187,6 +200,9 @@

if store is None:
return all_nodes, all_edges, resolved_edges
if packages_changed:
logger.info("Workspace packages changed — rebuilding the graph.")
return self.run(repo_path, store=store, rebuild=True)
if self._cross_file_inputs_changed(
store, all_nodes, resolved_edges, changed_files, found_file_paths, rebuild
):
Expand All @@ -204,6 +220,7 @@
virtual_nodes,
rebuild,
)
store.record_workspace_packages(workspace_packages)
logger.info("Running semantic uplift...")
SemanticUpliftEngine(store, self._domains_config).execute_uplift()
logger.info("Semantic uplift complete.")
Expand Down
19 changes: 16 additions & 3 deletions src/cgis/resolver/engine.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,7 @@
"""Implements ResolverEngine class."""

from collections.abc import Mapping

from cgis.core.models import (
RAW_CLASS_PREFIX,
SELF_PREFIX,
Expand Down Expand Up @@ -55,10 +57,21 @@ class ResolverEngine:
creation (spec §2.5).
"""

def __init__(self, nodes: list[Node], edges: list[Edge]) -> None:
"""Build the symbol resolver (which builds the index) from the extracted graph."""
def __init__(
self,
nodes: list[Node],
edges: list[Edge],
*,
workspace_packages: Mapping[str, str] | None = None,
) -> None:
"""Build the symbol resolver (which builds the index) from the extracted graph.

`workspace_packages` maps dotted package names to directory FQNs, from the
`package.json` files the pipeline walked, so TypeScript imports of a
workspace package resolve to its modules (#504).
"""
self.edges = edges
self._resolver = SymbolResolver(nodes, edges)
self._resolver = SymbolResolver(nodes, edges, workspace_packages)
self._index = self._resolver.index

def resolve(self) -> tuple[list[Edge], list[Node]]:
Expand Down
47 changes: 45 additions & 2 deletions src/cgis/resolver/indices.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,8 @@

#: Sources whose references read against Python's stdlib, builtins and import roots.
_PYTHON_SUFFIXES: tuple[str, ...] = (".py",)
#: The suffixes the TypeScript extractor handles; `.js`/`.jsx` are not ingested (registry.py).
_TYPESCRIPT_SUFFIXES = (".ts", ".tsx")


@dataclass(frozen=True)
Expand Down Expand Up @@ -66,6 +68,12 @@ class SymbolIndex:
# import maps, never guessed — kept apart from `internal_roots` so a target can
# be reconciled by stripping *these* and nothing else (#494).
layout_prefixes: frozenset[str] = frozenset()
# dotted workspace package name -> the directory FQN it lives at, longest name
# first: `@calcom.lib` -> `packages.lib`. Read from the `package.json` files the
# pipeline walked, never guessed, and consulted only for TypeScript sources
# (#504). Longest first because npm names may contain dots, so once `/` is
# dotted `a.b` is both "package a, subpath b" and "package a.b".
workspace_packages: Mapping[str, str] = MappingProxyType({})

def resolve_import_target(self, fqn: str, source_file: str | None = None) -> str | None:
"""Reconcile a module import against the node ids, or None to leave it alone.
Expand All @@ -92,6 +100,8 @@ def resolve_import_target(self, fqn: str, source_file: str | None = None) -> str
"""
if fqn in self.nodes:
return fqn
if source_file is not None and source_file.endswith(_TYPESCRIPT_SUFFIXES):
return self._resolve_workspace_import(fqn)
if source_file is not None and not source_file.endswith(_PYTHON_SUFFIXES):
return None
parts = fqn.split(".")
Expand All @@ -103,6 +113,30 @@ def resolve_import_target(self, fqn: str, source_file: str | None = None) -> str
return candidate
return None

def _resolve_workspace_import(self, fqn: str) -> str | None:
"""Map a TypeScript import of a workspace package onto the module it names.

`@calcom.lib.hooks.useLocale` becomes `packages.lib.hooks.useLocale` when the
pipeline found a `package.json` named `@calcom/lib` in `packages/lib`. The
package root is tried before `src/`, a common layout for a package's sources.

The name must match on a segment boundary, so `@x/a` never claims
`@x/a-b`. The first — longest — package that matches decides: falling back
to a shorter one would read `a.b.util` as package `a` when package `a.b`
was meant. And the rewrite happens only onto a node that exists; otherwise
None, leaving the edge visibly unresolved rather than on a plausible name.
"""
for name, directory in self.workspace_packages.items():
if fqn != name and not fqn.startswith(name + "."):
continue
subpath = fqn[len(name) + 1 :]
for base in (directory, f"{directory}.src"):
candidate = f"{base}.{subpath}" if subpath else base
if candidate in self.nodes:
return candidate
return None
return None

def map_to_node_fqn(self, imported_fqn: str) -> str | None:
"""Resolve an imported FQN to an actual node in the graph.

Expand Down Expand Up @@ -293,8 +327,14 @@ class IndexBuilder:
so it belongs to SymbolResolver (spec §2.4).
"""

def build(self, nodes: list[Node]) -> SymbolIndex:
"""Index all nodes for fast resolution and return the frozen index."""
def build(
self, nodes: list[Node], workspace_packages: Mapping[str, str] | None = None
) -> SymbolIndex:
"""Index all nodes for fast resolution and return the frozen index.

`workspace_packages` maps dotted package names to directory FQNs (#504); it
is stored longest name first, the order `_resolve_workspace_import` relies on.
"""
nodes_by_id = {n.id: n for n in nodes}
global_symbols: dict[str, list[str]] = {}
file_global_symbols: dict[tuple[str, str], list[str]] = {}
Expand Down Expand Up @@ -357,6 +397,9 @@ def build(self, nodes: list[Node]) -> SymbolIndex:
internal_roots=frozenset(internal_roots | first_party),
external_roots=frozenset(external_roots),
layout_prefixes=frozenset(corroborated - internal_roots),
workspace_packages=MappingProxyType(
dict(sorted((workspace_packages or {}).items(), key=lambda item: -len(item[0])))
),
)

@staticmethod
Expand Down
11 changes: 9 additions & 2 deletions src/cgis/resolver/symbols.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,7 @@
"""Symbol resolution strategies over a SymbolIndex."""

from collections.abc import Mapping

from cgis.core.models import RAW_CLASS_PREFIX, Edge, EdgeType, Node, NodeNamespace
from cgis.resolver.indices import IndexBuilder, SymbolIndex

Expand Down Expand Up @@ -78,9 +80,14 @@ class SymbolResolver:
without building it themselves.
"""

def __init__(self, nodes: list[Node], edges: list[Edge]) -> None:
def __init__(
self,
nodes: list[Node],
edges: list[Edge],
workspace_packages: Mapping[str, str] | None = None,
) -> None:
"""Build the symbol index from nodes, then the inheritance tree from EXTENDS edges."""
self.index: SymbolIndex = IndexBuilder().build(nodes)
self.index: SymbolIndex = IndexBuilder().build(nodes, workspace_packages)
# class_fqn -> [resolved parent FQNs] built from EXTENDS edges
self._inheritance_tree: dict[str, list[str]] = {}
for edge in edges:
Expand Down
36 changes: 35 additions & 1 deletion src/cgis/storage/sqlite_store.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@
import sqlite3
import time
from collections import Counter
from collections.abc import Iterable
from collections.abc import Iterable, Mapping
from dataclasses import dataclass, field
from pathlib import Path

Expand All @@ -20,6 +20,9 @@
)
from cgis.core.paths import is_excluded_dir, is_test_path

#: `ingest_state` key for the workspace packages imports were resolved against (#504).
_WORKSPACE_PACKAGES_KEY = "workspace_packages"

RAW_CALL_PREFIX = "raw_call:"
_DELETE_ALL_NODES = "DELETE FROM nodes"
_DELETE_ALL_EDGES = "DELETE FROM edges"
Expand Down Expand Up @@ -1065,6 +1068,37 @@ def get_ingest_state(self) -> tuple[str, float] | None:
return None
return rows["root"], float(rows["ingested_at"])

def record_workspace_packages(self, packages: Mapping[str, str]) -> None:
"""Record the workspace packages TypeScript imports were resolved against (#504).

Stored as sorted JSON in `ingest_state`, so an unchanged mapping reads back
equal on the next run and only a real change triggers a rebuild.
"""
if not self._conn:
raise RuntimeError(self._error_message)
self._conn.execute(
"INSERT OR REPLACE INTO ingest_state (key, value) VALUES (?, ?)",
(_WORKSPACE_PACKAGES_KEY, json.dumps(dict(sorted(packages.items())))),
)
self._conn.commit()

def get_workspace_packages(self) -> dict[str, str] | None:
"""The recorded workspace packages, or None on a graph that predates recording them.

None rather than an empty mapping: a graph built before #504 resolved no
workspace imports at all, and reading its silence as "no packages" would
leave those edges unresolved on a repository that does have them.
"""
if not self._conn:
raise RuntimeError(self._error_message)
row = self._conn.execute(
"SELECT value FROM ingest_state WHERE key = ?", (_WORKSPACE_PACKAGES_KEY,)
).fetchone()
if row is None:
return None
loaded = json.loads(row["value"])
return {str(k): str(v) for k, v in loaded.items()} if isinstance(loaded, dict) else None

def get_all_tracked_files(self) -> set[str]:
"""Return the set of all file paths currently tracked in files_state."""
if not self._conn:
Expand Down
Loading
Loading