diff --git a/README.md b/README.md index 8215234..5cc2ecc 100644 --- a/README.md +++ b/README.md @@ -192,7 +192,8 @@ on how to use Serena's tools. We also recommend that you index your code once before starting (especially for larger projects), it will accelerate the symbolic operations. ```shell -uvx --from git+https://github.com/oraios/serena index-project $(pwd) +# from the project directory, or pass the path to the project as argument +uvx --from git+https://github.com/oraios/serena index-project ``` diff --git a/src/multilspy/language_server.py b/src/multilspy/language_server.py index e1124cc..dd123c9 100644 --- a/src/multilspy/language_server.py +++ b/src/multilspy/language_server.py @@ -1210,7 +1210,6 @@ class LanguageServer: symbol_body = symbol_body[symbol_start_column:] return symbol_body - async def request_parsed_files(self) -> list[str]: """Retrieves relative paths of all files analyzed by the Language Server.""" if not self.server_started: @@ -1221,13 +1220,9 @@ class LanguageServer: raise MultilspyException("Language Server not started") rel_file_paths = [] for root, dirs, files in os.walk(self.repository_root_path): - # Don't go into directories that are ignored by modifying dirs inplace - # Explanation for the + "/" part: - # pathspec can't handle the matching of directories if they don't end with a slash! - # see https://github.com/cpburnz/python-pathspec/issues/89 - dirs[:] = [d for d in dirs if not self.is_ignored_path(os.path.join(root, d) + "/")] + dirs[:] = [d for d in dirs if not self.is_ignored_path(os.path.join(root, d))] for file in files: - rel_file_path = os.path.join(root, file) + rel_file_path = os.path.relpath(os.path.join(root, file), start=self.repository_root_path) if not self.is_ignored_path(rel_file_path): rel_file_paths.append(rel_file_path) return rel_file_paths @@ -1596,7 +1591,10 @@ class LanguageServer: return defining_symbol @property - def _cache_path(self) -> Path: + def cache_path(self) -> Path: + """ + The path to the cache file for the document symbols. + """ return Path(self.repository_root_path) / ".serena" / "cache" / self.language_id / "document_symbols_cache_v20-05-25.pkl" async def index_repository(self, progress_bar: bool = True, save_after_n_files: int = 10) -> None: @@ -1623,33 +1621,33 @@ class LanguageServer: self.logger.log("No changes to document symbols cache, skipping save", logging.DEBUG) return - self.logger.log(f"Saving updated document symbols cache to {self._cache_path}", logging.INFO) - self._cache_path.parent.mkdir(parents=True, exist_ok=True) + self.logger.log(f"Saving updated document symbols cache to {self.cache_path}", logging.INFO) + self.cache_path.parent.mkdir(parents=True, exist_ok=True) try: - with open(self._cache_path, "wb") as f: + with open(self.cache_path, "wb") as f: pickle.dump(self._document_symbols_cache, f) self._cache_has_changed = False except Exception as e: self.logger.log( - f"Failed to save document symbols cache to {self._cache_path}: {e}. " + f"Failed to save document symbols cache to {self.cache_path}: {e}. " "Note: this may have resulted in a corrupted cache file.", logging.ERROR, ) def load_cache(self): - if not self._cache_path.exists(): + if not self.cache_path.exists(): return with self._cache_lock: - self.logger.log(f"Loading document symbols cache from {self._cache_path}", logging.INFO) + self.logger.log(f"Loading document symbols cache from {self.cache_path}", logging.INFO) try: - with open(self._cache_path, "rb") as f: + with open(self.cache_path, "rb") as f: self._document_symbols_cache = pickle.load(f) self.logger.log(f"Loaded {len(self._document_symbols_cache)} document symbols from cache.", logging.INFO) except Exception as e: # cache often becomes corrupt, so just skip loading it self.logger.log( - f"Failed to load document symbols cache from {self._cache_path}: {e}. Possible cause: the cache file is corrupted. " + f"Failed to load document symbols cache from {self.cache_path}: {e}. Possible cause: the cache file is corrupted. " "Check for any errors related to saving the cache in the logs.", logging.ERROR, ) @@ -1704,6 +1702,13 @@ class SyncLanguageServer: self._shutdown_lock = threading.Lock() self._is_shutting_down = False + @property + def cache_path(self) -> Path: + """ + The path to the cache file for the document symbols. + """ + return self.language_server.cache_path + @classmethod def create( cls, config: MultilspyConfig, logger: MultilspyLogger, repository_root_path: str, diff --git a/src/serena/agent.py b/src/serena/agent.py index 580b447..6b16a9a 100644 --- a/src/serena/agent.py +++ b/src/serena/agent.py @@ -667,20 +667,21 @@ def create_ls_for_project( @click.command() -@click.argument("project", type=click.Path(exists=True)) +@click.argument("project", type=click.Path(exists=True), required=False, default=os.getcwd()) @click.option("--log-level", type=click.Choice(["DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL"]), default="WARNING") def index_project(project: str, log_level: str = "INFO") -> None: """ Index a project by saving the symbols of files to Serena's language server cache. - :param project: the project to index + :param project: the project to index. By default, the current working directory is used. """ log_level_int = logging.getLevelNamesMapping()[log_level.upper()] project = os.path.abspath(project) - log.info(f"Indexing project {project}") + print(f"Indexing symbols in project {project}") ls = create_ls_for_project(project, log_level=log_level_int) with ls.start_server(): ls.index_repository() + print(f"Symbols saved to {ls.cache_path}") class SerenaAgent: @@ -2138,21 +2139,19 @@ class SearchForPatternTool(Tool): ) else: # we walk through all files in the project starting from the root - files_to_search = [] - for root, dirs, files in os.walk(self.get_project_root()): - # Don't go into directories that are ignored by modifying dirs inplace - # Explanation for the + "/" part: - # pathspec can't handle the matching of directories if they don't end with a slash! - # see https://github.com/cpburnz/python-pathspec/issues/89 + project_root = self.get_project_root() + rel_paths_to_search = [] + for root, dirs, files in os.walk(project_root): dirs[:] = [d for d in dirs if not self.agent.path_is_gitignored(os.path.join(root, d))] for file in files: file_path = os.path.join(root, file) if not self.agent.path_is_gitignored(file_path): - files_to_search.append(file_path) + relative_path = os.path.relpath(file_path, project_root) + rel_paths_to_search.append(relative_path) # TODO (maybe): not super efficient to walk through the files again and filter if glob patterns are provided # but it probably never matters and this version required no further refactoring matches = search_files( - files_to_search, + rel_paths_to_search, pattern, paths_include_glob=paths_include_glob, paths_exclude_glob=paths_exclude_glob, diff --git a/src/serena/text_utils.py b/src/serena/text_utils.py index 4f40aa0..729f170 100644 --- a/src/serena/text_utils.py +++ b/src/serena/text_utils.py @@ -271,6 +271,8 @@ def glob_match(pattern: str, path: str) -> bool: :param path: File path to match against :return: True if path matches pattern """ + pattern = pattern.replace("\\", "/") # Normalize backslashes to forward slashes + # Handle ** patterns that should match zero or more directories if "**" in pattern: # Method 1: Standard fnmatch (matches one or more directories)