feat(indexing): add persistent index cache and incremental reparse

- Pass extension storage path to the server (client/ changes) so the
  server can persist a workspace index.
- Introduce IndexCache (server/tools/index_cache.py) and load/save it on
  initialization and after background indexing. Index entries are stored
  only when the file's stat hasn't changed while being read.
- Add incremental reparse logic (server/tools/incremental_parse.py) and
  use a per-file _last_parse cache in the language server to reparse only
  the top-level Tcl commands touched by an edit, falling back to a full
  parse when necessary.
- Use a new _FileIndex dataclass and _build_file_index helper to unify
  what is stored/loaded for a file; update update_poco_completion_for_file
  to use the persistent cache for disk-read files (from_disk/source_stat).
- Keep background indexing non-blocking and persist the index at the
  end of the run. Add basic unit tests for incremental parse and index cache.

Before: edits and background work always required full parsing of files
and no persistent cross-restart index. After: some edits reuse previous
ASTs and files read from disk can use a persisted index to skip
re-indexing across restarts.
This commit is contained in:
Christoph Brandau
2026-09-23 13:20:09 +02:00
parent b2e6e9d250
commit 01e8670cc1
10 changed files with 674 additions and 42 deletions
+156
View File
@@ -0,0 +1,156 @@
"""Reparse only the top-level commands touched by an edit.
Tcl top-level commands are independent once the previous command ended on its
own line: the parser keeps no state between them. An edit is therefore
reparsed from the first to the last top-level command it touches, commands
before it are reused as-is and commands after it are reused with their line
numbers shifted. Whenever that assumption could break, the caller falls back
to a full parse.
"""
from __future__ import annotations
import copy
from collections.abc import Callable
from tclint.syntax_tree import Node, Script
from tclint.violations import Violation
ParseChunk = Callable[[str, tuple[int, int]], tuple[Script, list[Violation]]]
def normalize_newlines(source: str) -> str:
"""Match the universal newline handling of the tclint parser."""
return source.replace("\r\n", "\n").replace("\r", "\n")
def _shifted_tree(node: Node, delta: int) -> Node:
"""Copy a subtree with its line numbers moved by `delta`.
Cached trees may still be read by other requests, so nodes are never
mutated. Attributes such as `Command.routine` alias entries of `children`,
so every reference is remapped to the same copy.
"""
copies: dict[int, Node] = {}
def shifted(original: Node) -> Node:
existing = copies.get(id(original))
if existing is not None:
return existing
clone = object.__new__(type(original))
copies[id(original)] = clone
state = dict(original.__dict__)
if state.get("line") is not None:
state["line"] += delta
end = state.get("end_pos")
if end is not None:
state["end_pos"] = (end[0] + delta, end[1])
for key, value in state.items():
if isinstance(value, Node):
state[key] = shifted(value)
elif isinstance(value, (list, tuple)) and value and isinstance(value[0], Node):
state[key] = type(value)(shifted(item) for item in value)
clone.__dict__.update(state)
return clone
return shifted(node)
def _shifted_violation(violation: Violation, delta: int) -> Violation:
clone = copy.copy(violation)
clone.start = (violation.start[0] + delta, violation.start[1])
clone.end = (violation.end[0] + delta, violation.end[1])
return clone
def reparse(
old_source: str,
old_tree: Script,
old_violations: list[Violation],
new_source: str,
parse_chunk: ParseChunk,
) -> tuple[Script, list[Violation]] | None:
"""Return the tree of `new_source`, or None when a full parse is needed.
Both sources must already be newline-normalized. `parse_chunk` parses a
top-level fragment starting at the given (line, column) and may raise
TclSyntaxError, which the caller handles like any failed parse.
"""
if old_source == new_source:
return old_tree, list(old_violations)
# Changed line range (1-indexed); lines outside it are identical. A pure
# insertion leaves last_changed_old == first_changed_line - 1.
old_lines = old_source.split("\n")
new_lines = new_source.split("\n")
limit = min(len(old_lines), len(new_lines))
same_before = 0
while same_before < limit and old_lines[same_before] == new_lines[same_before]:
same_before += 1
same_after = 0
while (
same_after < limit - same_before
and old_lines[-1 - same_after] == new_lines[-1 - same_after]
):
same_after += 1
first_changed_line = same_before + 1
last_changed_old = len(old_lines) - same_after
delta = len(new_lines) - len(old_lines)
commands = old_tree.children
if any(command.line is None or command.end_pos is None for command in commands):
return None
# Commands overlapping the changed lines, widened so that no reused
# command shares a line with the reparsed range.
first = next(
(index for index, command in enumerate(commands) if command.end_pos[0] >= first_changed_line),
len(commands),
)
start_line = first_changed_line
if first < len(commands):
start_line = min(start_line, commands[first].line)
while first > 0 and commands[first - 1].end_pos[0] >= start_line:
first -= 1
start_line = min(start_line, commands[first].line)
last = first - 1
end_line_old = last_changed_old
while last + 1 < len(commands) and commands[last + 1].line <= end_line_old:
last += 1
end_line_old = max(end_line_old, commands[last].end_pos[0])
end_line_new = end_line_old + delta
if end_line_new < start_line - 1 or end_line_new > len(new_lines):
return None
# A trailing backslash joins a line with the next one across the boundary.
if start_line > 1 and new_lines[start_line - 2].endswith("\\"):
return None
if end_line_new >= start_line and new_lines[end_line_new - 1].endswith("\\"):
return None
chunk_commands: list[Node] = []
chunk_violations: list[Violation] = []
if end_line_new >= start_line:
chunk = "\n".join(new_lines[start_line - 1 : end_line_new])
chunk_tree, chunk_violations = parse_chunk(chunk, (start_line, 1))
chunk_commands = chunk_tree.children
reused_after = [_shifted_tree(command, delta) for command in commands[last + 1 :]]
tree = Script(
*commands[:first],
*chunk_commands,
*reused_after,
pos=(old_tree.line, old_tree.col),
)
tree.end_pos = (len(new_lines), len(new_lines[-1]) + 1)
violations = [violation for violation in old_violations if violation.start[0] < start_line]
violations += chunk_violations
violations += [
_shifted_violation(violation, delta)
for violation in old_violations
if violation.start[0] > end_line_old
]
return tree, violations
+128
View File
@@ -0,0 +1,128 @@
"""Persist per-file index results across server restarts.
Entries are keyed by path and validated by the file's size and mtime. The
whole cache is tied to a fingerprint of the code that produced it: the
indexing sources of this server, the bundled tclint sources and the versions
of all bundled libraries. Any change to them, including a tclint update or a
local patch, discards the cache instead of loading stale results.
"""
from __future__ import annotations
import hashlib
import logging
import os
import pathlib
import pickle
import sys
import tempfile
import threading
import zlib
from typing import Any
LOGGER = logging.getLogger(__name__)
# Bump when the cached data layout changes without a source change above.
CACHE_FORMAT = 1
CACHE_FILE = "index-cache.pickle.z"
_SRC_DIR = pathlib.Path(__file__).resolve().parent.parent
_LIBS_DIR = _SRC_DIR.parent / "libs"
FileStat = tuple[int, int]
def code_fingerprint() -> str:
digest = hashlib.sha256()
digest.update(f"{CACHE_FORMAT}|{sys.version}".encode())
sources = [
_SRC_DIR / "lsp_tclserver.py",
*sorted((_SRC_DIR / "tools").glob("*.py")),
*sorted((_SRC_DIR / "plugins").glob("*.py")),
*sorted((_LIBS_DIR / "tclint").rglob("*.py")),
]
for source in sources:
digest.update(source.relative_to(_SRC_DIR.parent).as_posix().encode())
digest.update(source.read_bytes())
for dist_info in sorted(_LIBS_DIR.glob("*.dist-info")):
digest.update(dist_info.name.encode())
return digest.hexdigest()
def file_stat(path: str) -> FileStat | None:
try:
stat = os.stat(path)
except OSError:
return None
return stat.st_mtime_ns, stat.st_size
class IndexCache:
def __init__(self, directory: pathlib.Path | None = None, fingerprint: str = ""):
self._path = directory / CACHE_FILE if directory is not None else None
self._fingerprint = fingerprint
self._entries: dict[str, tuple[FileStat, Any]] = {}
self._used: set[str] = set()
self._dirty = False
self._lock = threading.Lock()
@classmethod
def load(cls, directory: pathlib.Path | str | None) -> IndexCache:
"""Open the cache in `directory`; without one, nothing is persisted."""
if not directory:
return cls()
cache = cls(pathlib.Path(directory), code_fingerprint())
try:
with open(cache._path, "rb") as file:
fingerprint, entries = pickle.loads(zlib.decompress(file.read()))
except FileNotFoundError:
return cache
except Exception as error: # A damaged cache must never stop indexing.
LOGGER.warning("Ignoring unreadable index cache %s: %s", cache._path, error)
cache._dirty = True
return cache
if fingerprint == cache._fingerprint:
cache._entries = entries
else:
cache._dirty = True
return cache
def get(self, path: str, stat: FileStat) -> Any | None:
with self._lock:
entry = self._entries.get(path)
if entry is None or entry[0] != stat:
return None
self._used.add(path)
return entry[1]
def put(self, path: str, stat: FileStat, data: Any) -> None:
if self._path is None:
return
with self._lock:
self._entries[path] = (stat, data)
self._used.add(path)
self._dirty = True
def save(self) -> None:
"""Write entries used in this session atomically; others are dropped."""
if self._path is None:
return
with self._lock:
if not self._dirty and self._used == self._entries.keys():
return
entries = {path: self._entries[path] for path in self._used if path in self._entries}
self._entries = entries
self._dirty = False
try:
self._path.parent.mkdir(parents=True, exist_ok=True)
with tempfile.NamedTemporaryFile(dir=self._path.parent, delete=False) as file:
data = pickle.dumps((self._fingerprint, entries), protocol=pickle.HIGHEST_PROTOCOL)
# Pickled indexes are very repetitive; fast compression cuts ~90%.
file.write(zlib.compress(data, 1))
os.replace(file.name, self._path)
except Exception as error:
LOGGER.warning("Could not write index cache %s: %s", self._path, error)
try:
os.unlink(file.name)
except (OSError, NameError):
pass
+10 -4
View File
@@ -265,6 +265,12 @@ def build_file_symbol_index(
filepath: str, uri: str, tree: Node
) -> FileSymbolIndex:
occurrences: list[SymbolOccurrence] = []
# Most occurrences repeat a few identities; sharing one object per identity
# keeps the index (and its persistent cache) small.
identities: dict[SymbolIdentity, SymbolIdentity] = {}
def shared(identity: SymbolIdentity | None) -> SymbolIdentity | None:
return None if identity is None else identities.setdefault(identity, identity)
def add_proc(
node: Node,
@@ -274,7 +280,7 @@ def build_file_symbol_index(
is_definition: bool,
declaration_range: lsp.Range | None = None,
) -> None:
identity = _proc_identity(raw_name, scope.namespace)
identity = shared(_proc_identity(raw_name, scope.namespace))
caller = None
if not is_definition:
caller = (
@@ -288,14 +294,14 @@ def build_file_symbol_index(
fallback_identity=(
None
if is_definition
else _proc_fallback(raw_name, scope.namespace)
else shared(_proc_fallback(raw_name, scope.namespace))
),
range=_name_range(node, raw_name),
placeholder=_basename(raw_name),
is_definition=is_definition,
symbol_kind=lsp.SymbolKind.Function,
container_name=_container_name(identity),
caller=caller,
caller=shared(caller),
declaration_range=declaration_range,
)
)
@@ -309,7 +315,7 @@ def build_file_symbol_index(
variable_sub: bool = False,
identity: SymbolIdentity | None = None,
) -> None:
symbol_identity = identity or _variable_identity(raw_name, scope)
symbol_identity = shared(identity or _variable_identity(raw_name, scope))
occurrences.append(
SymbolOccurrence(
identity=symbol_identity,