diff --git a/src/cgis/extractors/base.py b/src/cgis/extractors/base.py index d47a8001..37680977 100644 --- a/src/cgis/extractors/base.py +++ b/src/cgis/extractors/base.py @@ -1,10 +1,26 @@ """Base Extractor Interface""" from abc import ABC, abstractmethod +from typing import Protocol, runtime_checkable from cgis.core.models import Edge, Node +@runtime_checkable +class ModuleNamer(Protocol): + """An extractor that can say which module FQN it gives a file path. + + Structural rather than a base-class method: only some extractors can answer, + and a caller that needs one — the workspace-package map naming a package's + directory (#504) — can then ask for it without importing any language's + extractor, which would load that language's grammar for every run (#506). + """ + + def module_fqn(self, file_path: str) -> str: + """The module FQN this extractor gives `file_path`.""" + ... + + class BaseExtractor(ABC): """ Abstract Base Class for all language-specific extractors. diff --git a/src/cgis/extractors/typescript_extractor.py b/src/cgis/extractors/typescript_extractor.py index 45a46915..b9811537 100644 --- a/src/cgis/extractors/typescript_extractor.py +++ b/src/cgis/extractors/typescript_extractor.py @@ -113,11 +113,20 @@ def __init__(self, tsx: bool = False, source_roots: list[str] | None = None) -> lang = tsts.language_tsx() if tsx else tsts.language_typescript() self._parser = Parser(Language(lang)) + def module_fqn(self, file_path: str) -> str: + """The module FQN this extractor gives `file_path`, with its source root applied. + + `parse` names every file with this, so a caller that needs to spell a path + the way the graph's node ids spell it — the workspace-package map naming a + package's directory (#504) — gets the same rule rather than a copy of it. + """ + return file_path_to_module_fqn(file_path, self._pick_source_root(file_path)) + def parse(self, code: str, file_path: str) -> tuple[list[Node], list[Edge]]: """Extract nodes and edges from TypeScript source code.""" code_bytes = code.encode("utf-8") tree = self._parser.parse(code_bytes) - module_fqn = file_path_to_module_fqn(file_path, self._pick_source_root(file_path)) + module_fqn = self.module_fqn(file_path) file_node = Node( id=module_fqn, diff --git a/src/cgis/pipeline.py b/src/cgis/pipeline.py index ccd78214..02369813 100644 --- a/src/cgis/pipeline.py +++ b/src/cgis/pipeline.py @@ -17,6 +17,7 @@ from cgis.extractors.base import BaseExtractor from cgis.resolver.engine import ResolverEngine from cgis.resolver.uplift import SemanticUpliftEngine +from cgis.workspaces import PACKAGE_MANIFEST, WorkspacePackages if TYPE_CHECKING: from cgis.storage.sqlite_store import SQLiteStore @@ -122,8 +123,8 @@ def run( # file_path -> new hash, only for files that were re-extracted changed_files: dict[str, str] = {} found_file_paths: set[str] = set() - workspace_root = self.workspace_root(repo_path) + packages = WorkspacePackages(workspace_root, self._extractors) with Progress( SpinnerColumn(), @@ -137,6 +138,8 @@ def run( for root, dirs, files in workspace_root.walk(): dirs[:] = [d for d in dirs if not d.startswith(".") and d not in self._excluded] for file in files: + if file == PACKAGE_MANIFEST: + packages.note(root / file) extractor = self._get_extractor(file) if not extractor: continue @@ -168,14 +171,24 @@ def run( # nothing went stale, the persisted graph is already correct. # Re-running the resolver + persistence + uplift would rebuild the # whole graph from the DB for zero benefit, so skip them entirely. - if self._is_noop_incremental(store, changed_files, found_file_paths): + workspace_packages = packages.unambiguous() + # A renamed or moved package changes what unchanged files' imports mean, + # and an incremental run re-resolves only changed files (#504). + packages_changed = ( + store is not None + and not rebuild + and (store.get_workspace_packages() or {}) != workspace_packages + ) + if not packages_changed and self._is_noop_incremental( + store, changed_files, found_file_paths + ): logger.info("No changes detected — skipping resolution and persistence.") return all_nodes, all_edges, [] # Task 2: Resolution resolve_task = progress.add_task(description="Resolving semantic links...", total=None) logger.info("Starting resolution phase...") - resolver = ResolverEngine(all_nodes, all_edges) + resolver = ResolverEngine(all_nodes, all_edges, workspace_packages=workspace_packages) resolved_edges, virtual_nodes = resolver.resolve() all_nodes.extend(virtual_nodes) progress.update(resolve_task, advance=1) @@ -187,6 +200,9 @@ def run( if store is None: return all_nodes, all_edges, resolved_edges + if packages_changed: + logger.info("Workspace packages changed — rebuilding the graph.") + return self.run(repo_path, store=store, rebuild=True) if self._cross_file_inputs_changed( store, all_nodes, resolved_edges, changed_files, found_file_paths, rebuild ): @@ -204,6 +220,7 @@ def run( virtual_nodes, rebuild, ) + store.record_workspace_packages(workspace_packages) logger.info("Running semantic uplift...") SemanticUpliftEngine(store, self._domains_config).execute_uplift() logger.info("Semantic uplift complete.") diff --git a/src/cgis/resolver/engine.py b/src/cgis/resolver/engine.py index ec256884..39a8318b 100644 --- a/src/cgis/resolver/engine.py +++ b/src/cgis/resolver/engine.py @@ -1,5 +1,7 @@ """Implements ResolverEngine class.""" +from collections.abc import Mapping + from cgis.core.models import ( RAW_CLASS_PREFIX, SELF_PREFIX, @@ -55,10 +57,21 @@ class ResolverEngine: creation (spec §2.5). """ - def __init__(self, nodes: list[Node], edges: list[Edge]) -> None: - """Build the symbol resolver (which builds the index) from the extracted graph.""" + def __init__( + self, + nodes: list[Node], + edges: list[Edge], + *, + workspace_packages: Mapping[str, str] | None = None, + ) -> None: + """Build the symbol resolver (which builds the index) from the extracted graph. + + `workspace_packages` maps dotted package names to directory FQNs, from the + `package.json` files the pipeline walked, so TypeScript imports of a + workspace package resolve to its modules (#504). + """ self.edges = edges - self._resolver = SymbolResolver(nodes, edges) + self._resolver = SymbolResolver(nodes, edges, workspace_packages) self._index = self._resolver.index def resolve(self) -> tuple[list[Edge], list[Node]]: diff --git a/src/cgis/resolver/indices.py b/src/cgis/resolver/indices.py index 09a953bd..d3d5e421 100644 --- a/src/cgis/resolver/indices.py +++ b/src/cgis/resolver/indices.py @@ -14,6 +14,8 @@ #: Sources whose references read against Python's stdlib, builtins and import roots. _PYTHON_SUFFIXES: tuple[str, ...] = (".py",) +#: The suffixes the TypeScript extractor handles; `.js`/`.jsx` are not ingested (registry.py). +_TYPESCRIPT_SUFFIXES = (".ts", ".tsx") @dataclass(frozen=True) @@ -66,6 +68,12 @@ class SymbolIndex: # import maps, never guessed — kept apart from `internal_roots` so a target can # be reconciled by stripping *these* and nothing else (#494). layout_prefixes: frozenset[str] = frozenset() + # dotted workspace package name -> the directory FQN it lives at, longest name + # first: `@calcom.lib` -> `packages.lib`. Read from the `package.json` files the + # pipeline walked, never guessed, and consulted only for TypeScript sources + # (#504). Longest first because npm names may contain dots, so once `/` is + # dotted `a.b` is both "package a, subpath b" and "package a.b". + workspace_packages: Mapping[str, str] = MappingProxyType({}) def resolve_import_target(self, fqn: str, source_file: str | None = None) -> str | None: """Reconcile a module import against the node ids, or None to leave it alone. @@ -92,6 +100,8 @@ def resolve_import_target(self, fqn: str, source_file: str | None = None) -> str """ if fqn in self.nodes: return fqn + if source_file is not None and source_file.endswith(_TYPESCRIPT_SUFFIXES): + return self._resolve_workspace_import(fqn) if source_file is not None and not source_file.endswith(_PYTHON_SUFFIXES): return None parts = fqn.split(".") @@ -103,6 +113,30 @@ def resolve_import_target(self, fqn: str, source_file: str | None = None) -> str return candidate return None + def _resolve_workspace_import(self, fqn: str) -> str | None: + """Map a TypeScript import of a workspace package onto the module it names. + + `@calcom.lib.hooks.useLocale` becomes `packages.lib.hooks.useLocale` when the + pipeline found a `package.json` named `@calcom/lib` in `packages/lib`. The + package root is tried before `src/`, a common layout for a package's sources. + + The name must match on a segment boundary, so `@x/a` never claims + `@x/a-b`. The first — longest — package that matches decides: falling back + to a shorter one would read `a.b.util` as package `a` when package `a.b` + was meant. And the rewrite happens only onto a node that exists; otherwise + None, leaving the edge visibly unresolved rather than on a plausible name. + """ + for name, directory in self.workspace_packages.items(): + if fqn != name and not fqn.startswith(name + "."): + continue + subpath = fqn[len(name) + 1 :] + for base in (directory, f"{directory}.src"): + candidate = f"{base}.{subpath}" if subpath else base + if candidate in self.nodes: + return candidate + return None + return None + def map_to_node_fqn(self, imported_fqn: str) -> str | None: """Resolve an imported FQN to an actual node in the graph. @@ -293,8 +327,14 @@ class IndexBuilder: so it belongs to SymbolResolver (spec §2.4). """ - def build(self, nodes: list[Node]) -> SymbolIndex: - """Index all nodes for fast resolution and return the frozen index.""" + def build( + self, nodes: list[Node], workspace_packages: Mapping[str, str] | None = None + ) -> SymbolIndex: + """Index all nodes for fast resolution and return the frozen index. + + `workspace_packages` maps dotted package names to directory FQNs (#504); it + is stored longest name first, the order `_resolve_workspace_import` relies on. + """ nodes_by_id = {n.id: n for n in nodes} global_symbols: dict[str, list[str]] = {} file_global_symbols: dict[tuple[str, str], list[str]] = {} @@ -357,6 +397,9 @@ def build(self, nodes: list[Node]) -> SymbolIndex: internal_roots=frozenset(internal_roots | first_party), external_roots=frozenset(external_roots), layout_prefixes=frozenset(corroborated - internal_roots), + workspace_packages=MappingProxyType( + dict(sorted((workspace_packages or {}).items(), key=lambda item: -len(item[0]))) + ), ) @staticmethod diff --git a/src/cgis/resolver/symbols.py b/src/cgis/resolver/symbols.py index dcea00d3..2416a13d 100644 --- a/src/cgis/resolver/symbols.py +++ b/src/cgis/resolver/symbols.py @@ -1,5 +1,7 @@ """Symbol resolution strategies over a SymbolIndex.""" +from collections.abc import Mapping + from cgis.core.models import RAW_CLASS_PREFIX, Edge, EdgeType, Node, NodeNamespace from cgis.resolver.indices import IndexBuilder, SymbolIndex @@ -78,9 +80,14 @@ class SymbolResolver: without building it themselves. """ - def __init__(self, nodes: list[Node], edges: list[Edge]) -> None: + def __init__( + self, + nodes: list[Node], + edges: list[Edge], + workspace_packages: Mapping[str, str] | None = None, + ) -> None: """Build the symbol index from nodes, then the inheritance tree from EXTENDS edges.""" - self.index: SymbolIndex = IndexBuilder().build(nodes) + self.index: SymbolIndex = IndexBuilder().build(nodes, workspace_packages) # class_fqn -> [resolved parent FQNs] built from EXTENDS edges self._inheritance_tree: dict[str, list[str]] = {} for edge in edges: diff --git a/src/cgis/storage/sqlite_store.py b/src/cgis/storage/sqlite_store.py index 79393269..cbbdc30a 100644 --- a/src/cgis/storage/sqlite_store.py +++ b/src/cgis/storage/sqlite_store.py @@ -5,7 +5,7 @@ import sqlite3 import time from collections import Counter -from collections.abc import Iterable +from collections.abc import Iterable, Mapping from dataclasses import dataclass, field from pathlib import Path @@ -20,6 +20,9 @@ ) from cgis.core.paths import is_excluded_dir, is_test_path +#: `ingest_state` key for the workspace packages imports were resolved against (#504). +_WORKSPACE_PACKAGES_KEY = "workspace_packages" + RAW_CALL_PREFIX = "raw_call:" _DELETE_ALL_NODES = "DELETE FROM nodes" _DELETE_ALL_EDGES = "DELETE FROM edges" @@ -1065,6 +1068,37 @@ def get_ingest_state(self) -> tuple[str, float] | None: return None return rows["root"], float(rows["ingested_at"]) + def record_workspace_packages(self, packages: Mapping[str, str]) -> None: + """Record the workspace packages TypeScript imports were resolved against (#504). + + Stored as sorted JSON in `ingest_state`, so an unchanged mapping reads back + equal on the next run and only a real change triggers a rebuild. + """ + if not self._conn: + raise RuntimeError(self._error_message) + self._conn.execute( + "INSERT OR REPLACE INTO ingest_state (key, value) VALUES (?, ?)", + (_WORKSPACE_PACKAGES_KEY, json.dumps(dict(sorted(packages.items())))), + ) + self._conn.commit() + + def get_workspace_packages(self) -> dict[str, str] | None: + """The recorded workspace packages, or None on a graph that predates recording them. + + None rather than an empty mapping: a graph built before #504 resolved no + workspace imports at all, and reading its silence as "no packages" would + leave those edges unresolved on a repository that does have them. + """ + if not self._conn: + raise RuntimeError(self._error_message) + row = self._conn.execute( + "SELECT value FROM ingest_state WHERE key = ?", (_WORKSPACE_PACKAGES_KEY,) + ).fetchone() + if row is None: + return None + loaded = json.loads(row["value"]) + return {str(k): str(v) for k, v in loaded.items()} if isinstance(loaded, dict) else None + def get_all_tracked_files(self) -> set[str]: """Return the set of all file paths currently tracked in files_state.""" if not self._conn: diff --git a/src/cgis/workspaces.py b/src/cgis/workspaces.py new file mode 100644 index 00000000..bf7812ce --- /dev/null +++ b/src/cgis/workspaces.py @@ -0,0 +1,83 @@ +"""Workspace packages a TypeScript monorepo names in its `package.json` files (#504). + +A monorepo imports its own packages by name: `@calcom/lib/hooks/useLocale` means +`packages/lib/hooks/useLocale.ts`. The resolver can map such an import onto the +module only when told which directory each name lives in, and this collects that +from the manifests the pipeline walks past. +""" + +import json +from collections.abc import Mapping +from pathlib import Path + +import structlog + +from cgis.extractors.base import BaseExtractor, ModuleNamer + +logger = structlog.getLogger(__name__) + +#: The manifest a workspace package is named in. +PACKAGE_MANIFEST = "package.json" + + +class WorkspacePackages: + """Collects package names during a walk and hands the resolver the unambiguous ones.""" + + def __init__(self, workspace_root: Path, extractors: Mapping[str, BaseExtractor]) -> None: + """Collect for `workspace_root`, naming directories with the TypeScript extractor's rule. + + The extractor is the one registered for `.ts`, chosen by extension rather + than by class, so this module imports no language's extractor (#506). With + none configured, or one that cannot name modules, every manifest is ignored. + """ + self._root = workspace_root + extractor = extractors.get(".ts") or extractors.get(".tsx") + self._extractor = extractor if isinstance(extractor, ModuleNamer) else None + # dotted package name -> every directory FQN that claims it + self._claims: dict[str, set[str]] = {} + + def note(self, manifest: Path) -> None: + """Record which directory a `package.json` names. + + Never the repository root: its manifest names the monorepo itself, and + mapping it would send an import of that name to the empty directory FQN. + The directory FQN comes from the extractor's own `module_fqn`, so it is + spelled exactly as that package's node ids are. An unreadable or nameless + manifest is skipped: one bad file must not stop the ingest. + """ + if self._extractor is None: + return + try: + directory = manifest.resolve().parent.relative_to(self._root).as_posix() + except ValueError: + return + if directory in ("", "."): + return + try: + data = json.loads(manifest.read_text(encoding="utf-8")) + except (OSError, ValueError) as exc: + logger.warning("Skipping unreadable package.json", path=str(manifest), error=str(exc)) + return + name = data.get("name") if isinstance(data, dict) else None + if not isinstance(name, str) or not name.strip(): + return + directory_fqn = self._extractor.module_fqn(f"{directory}/index.ts") + self._claims.setdefault(name.replace("/", "."), set()).add(directory_fqn) + + def unambiguous(self) -> dict[str, str]: + """Dotted package names claimed by exactly one directory; a name claimed twice is dropped. + + Two manifests with one name cannot both be what an import means, and + choosing one would wire every importer to a package it may not use. + """ + packages: dict[str, str] = {} + for name, directories in sorted(self._claims.items()): + if len(directories) == 1: + packages[name] = next(iter(directories)) + else: + logger.warning( + "Package name claimed by several directories; not resolving it", + name=name, + directories=sorted(directories), + ) + return packages diff --git a/tests/unit/test_pipeline_ts_workspaces.py b/tests/unit/test_pipeline_ts_workspaces.py new file mode 100644 index 00000000..2922c2f2 --- /dev/null +++ b/tests/unit/test_pipeline_ts_workspaces.py @@ -0,0 +1,133 @@ +"""The pipeline learns workspace packages from `package.json` and hands them to the resolver (#504). + +The resolver rewrites `@x/lib/util` onto `packages/lib/util.ts` only when told that +a `package.json` named `@x/lib` lives in `packages/lib`. These tests cover where +that knowledge comes from: which manifests count, which are refused, and that a +change to it rebuilds the graph — an incremental run re-resolves only the files +that changed, so without a rebuild an unchanged importer would keep an edge onto a +package that no longer carries that name. +""" + +import json +import sqlite3 +import subprocess +import sys +from pathlib import Path + +from cgis.core.models import EdgeType +from cgis.extractors.typescript_extractor import TypeScriptExtractor +from cgis.pipeline import IngestionPipeline +from cgis.storage.sqlite_store import SQLiteStore + +IMPORTER = "apps.web.page" + + +def _write(root: Path, files: dict[str, str]) -> None: + for rel, text in files.items(): + path = root / rel + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(text, encoding="utf-8") + + +def _manifest(name: str) -> str: + return json.dumps({"name": name, "version": "0.0.0"}) + + +def _monorepo(root: Path, lib_name: str = "@x/lib") -> None: + _write( + root, + { + "packages/lib/package.json": _manifest(lib_name), + "packages/lib/util.ts": "export function helper() {\n return 1;\n}\n", + "apps/web/package.json": _manifest("web"), + "apps/web/page.ts": ( + 'import { helper } from "@x/lib/util";\n' + "export function Page() {\n return helper();\n}\n" + ), + }, + ) + + +def _pipeline() -> IngestionPipeline: + return IngestionPipeline({".ts": TypeScriptExtractor()}) + + +def _import_target(root: Path) -> str: + _, _, resolved = _pipeline().run(str(root)) + (edge,) = (e for e in resolved if e.type == EdgeType.IMPORTS and e.source == IMPORTER) + return edge.target + + +def test_a_workspace_import_reaches_the_package_module(tmp_path: Path) -> None: + _monorepo(tmp_path) + assert _import_target(tmp_path) == "packages.lib.util" + + +def test_the_repository_root_package_is_not_a_workspace_package(tmp_path: Path) -> None: + # The root manifest names the monorepo itself. Mapping it would send any import + # of that name to the empty directory FQN. + _monorepo(tmp_path) + _write(tmp_path, {"package.json": _manifest("@x/lib")}) + assert _import_target(tmp_path) == "packages.lib.util" + + +def test_a_name_claimed_by_two_packages_is_not_mapped(tmp_path: Path) -> None: + _monorepo(tmp_path) + _write( + tmp_path, + { + "packages/lib-copy/package.json": _manifest("@x/lib"), + "packages/lib-copy/util.ts": "export function helper() {\n return 2;\n}\n", + }, + ) + assert _import_target(tmp_path) == "@x.lib.util" + + +def test_a_package_linked_into_node_modules_is_not_a_second_claim(tmp_path: Path) -> None: + # Package managers place workspace packages under node_modules too. The walk + # never enters node_modules, so the copy there must not make the name ambiguous. + _monorepo(tmp_path) + _write(tmp_path, {"node_modules/@x/lib/package.json": _manifest("@x/lib")}) + assert _import_target(tmp_path) == "packages.lib.util" + + +def test_an_unreadable_manifest_is_skipped_not_fatal(tmp_path: Path) -> None: + _monorepo(tmp_path) + _write(tmp_path, {"packages/broken/package.json": "{ not json"}) + assert _import_target(tmp_path) == "packages.lib.util" + + +def test_renaming_a_package_rebuilds_so_an_unchanged_importer_follows(tmp_path: Path) -> None: + _monorepo(tmp_path) + db = str(tmp_path / "graph.db") + work = tmp_path + with SQLiteStore(db) as store: + _pipeline().run(str(work), store=store, rebuild=True) + + # Only the manifest changes. page.ts is untouched, so an incremental run would + # otherwise not re-resolve its import and would keep it on packages.lib.util. + _write(work, {"packages/lib/package.json": _manifest("@y/lib")}) + with SQLiteStore(db) as store: + _pipeline().run(str(work), store=store) + + targets = [ + row[0] + for row in sqlite3.connect(db).execute( + "select target from edges where source = ? and type = ?", + (IMPORTER, EdgeType.IMPORTS.value), + ) + ] + assert targets == ["@x.lib.util"] + + +def test_importing_the_pipeline_does_not_load_a_language_grammar() -> None: + """The pipeline is language-agnostic; workspace support must not change that (#506 review). + + Run in a fresh interpreter: in this one, other tests have already imported the + TypeScript extractor, so `sys.modules` would say nothing about the pipeline. + """ + probe = "import sys, cgis.pipeline; print('tree_sitter_typescript' in sys.modules)" + result = subprocess.run( + [sys.executable, "-c", probe], capture_output=True, text=True, check=True + ) + assert result.stdout.strip() == "False" diff --git a/tests/unit/test_resolver_ts_workspaces.py b/tests/unit/test_resolver_ts_workspaces.py new file mode 100644 index 00000000..233a7ab9 --- /dev/null +++ b/tests/unit/test_resolver_ts_workspaces.py @@ -0,0 +1,116 @@ +"""TypeScript imports of a workspace package must reach the package's own modules (#504). + +A monorepo imports its packages by name: `@calcom/lib/hooks/useLocale` means +`packages/lib/hooks/useLocale.ts`. The extractor writes that target dotted, +`@calcom.lib.hooks.useLocale`, and nothing mapped it to the node. On cal.com that +left 86% of IMPORTS edges on names no node bears, and the impact graph of a module +imported in 256 places was the module alone. + +The mapping is not a guess. It comes from the `package.json` files the pipeline +walked, and a target is rewritten only when the module it names exists — anything +else stays visibly unresolved, the same rule `resolve_import_target` applies to +Python layout prefixes (#494). +""" + +from cgis.core.models import Edge, EdgeType, Node, NodeNamespace, NodeType +from cgis.resolver.engine import ResolverEngine + +PACKAGES = {"@calcom.lib": "packages.lib", "@calcom.ui": "packages.ui"} +PAGE = "apps.web.page" + + +def _file(node_id: str, file_path: str) -> Node: + return Node( + id=node_id, + type=NodeType.FILE, + name=node_id.rsplit(".", 1)[-1], + file_path=file_path, + start_line=1, + end_line=1, + namespace=NodeNamespace.INTERNAL, + ) + + +def _import(source: str, target: str) -> Edge: + return Edge( + id=f"{source}->import:{target}", source=source, target=target, type=EdgeType.IMPORTS + ) + + +def _resolved_target( + nodes: list[Node], + target: str, + packages: dict[str, str] | None = PACKAGES, + source: str = PAGE, +) -> str: + page = _file(source, "apps/web/page.tsx") if source == PAGE else None + graph = [*nodes, page] if page is not None else nodes + engine = ( + ResolverEngine(graph, [_import(source, target)]) + if packages is None + else ResolverEngine(graph, [_import(source, target)], workspace_packages=packages) + ) + resolved, _ = engine.resolve() + (edge,) = (e for e in resolved if e.type == EdgeType.IMPORTS) + return edge.target + + +def test_a_workspace_package_import_reaches_the_module() -> None: + nodes = [_file("packages.lib.hooks.useLocale", "packages/lib/hooks/useLocale.ts")] + assert _resolved_target(nodes, "@calcom.lib.hooks.useLocale") == "packages.lib.hooks.useLocale" + + +def test_the_package_name_alone_reaches_its_index() -> None: + nodes = [_file("packages.lib", "packages/lib/index.ts")] + assert _resolved_target(nodes, "@calcom.lib") == "packages.lib" + + +def test_a_src_layout_is_tried_after_the_package_root() -> None: + nodes = [_file("packages.ui.src.Button", "packages/ui/src/Button.tsx")] + assert _resolved_target(nodes, "@calcom.ui.Button") == "packages.ui.src.Button" + + +def test_a_package_name_matches_only_on_a_segment_boundary() -> None: + # `@x/a` must not claim `@x/a-b/util`: without the boundary it would rewrite the + # target to `packages.a` followed by `-b.util`, a name no node bears. + packages = {"@x.a": "packages.a", "@x.a-b": "packages.ab"} + nodes = [_file("packages.ab.util", "packages/ab/util.ts")] + assert _resolved_target(nodes, "@x.a-b.util", packages) == "packages.ab.util" + + +def test_the_longest_package_name_wins() -> None: + # npm names may contain dots (`socket.io`), so once `/` is dotted, `a.b` is both + # "package a, subpath b" and "package a.b". The more specific package is meant. + packages = {"a": "packages.a", "a.b": "packages.ab"} + nodes = [ + _file("packages.a.b.util", "packages/a/b/util.ts"), + _file("packages.ab.util", "packages/ab/util.ts"), + ] + assert _resolved_target(nodes, "a.b.util", packages) == "packages.ab.util" + + +def test_a_module_the_package_does_not_have_stays_unresolved() -> None: + nodes = [_file("packages.lib.hooks.useLocale", "packages/lib/hooks/useLocale.ts")] + assert _resolved_target(nodes, "@calcom.lib.hooks.gone") == "@calcom.lib.hooks.gone" + + +def test_an_external_package_is_left_alone() -> None: + nodes = [_file("packages.lib.hooks.useLocale", "packages/lib/hooks/useLocale.ts")] + assert _resolved_target(nodes, "@prisma.client") == "@prisma.client" + + +def test_a_python_source_is_not_rewritten() -> None: + # The mapping is learned from package.json and scoped to the language that + # declares it, as the Python layout prefixes are scoped the other way (#454). + nodes = [ + _file("packages.lib.hooks.useLocale", "packages/lib/hooks/useLocale.ts"), + _file("tools.seed", "tools/seed.py"), + ] + target = "@calcom.lib.hooks.useLocale" + assert _resolved_target(nodes, target, source="tools.seed") == target + + +def test_without_workspace_packages_nothing_changes() -> None: + nodes = [_file("packages.lib.hooks.useLocale", "packages/lib/hooks/useLocale.ts")] + target = "@calcom.lib.hooks.useLocale" + assert _resolved_target(nodes, target, packages=None) == target diff --git a/tests/unit/test_typescript_extractor.py b/tests/unit/test_typescript_extractor.py index 63f04648..a03b31d1 100644 --- a/tests/unit/test_typescript_extractor.py +++ b/tests/unit/test_typescript_extractor.py @@ -335,3 +335,14 @@ def test_typescript_extractor_source_roots_unmatched_root() -> None: nodes, _ = ext.parse(code, "src/api/handler.ts") fqns = [n.id for n in nodes] assert any(fqn.startswith("src.api.") for fqn in fqns) + + +def test_module_fqn_is_the_fqn_parse_gives_the_file() -> None: + """The workspace-package map names directories with this (#504). + + So it must be the rule `parse` itself uses, not a copy of it. + """ + extractor = TypeScriptExtractor(source_roots=["src"]) + nodes, _ = extractor.parse("export const x = 1;\n", "src/pkg/index.ts") + (file_node,) = (n for n in nodes if n.type == NodeType.FILE) + assert extractor.module_fqn("src/pkg/index.ts") == file_node.id == "pkg"