Files
nuwiki/crates/nuwiki-lsp/src/lib.rs
T
gffranco 6efe154f1a
CI / cargo fmt --check (push) Successful in 30s
CI / cargo clippy (push) Successful in 1m16s
CI / cargo test (push) Successful in 1m20s
phase 7: semantic tokens for syntax highlighting
`textDocument/semanticTokens/full` + `/range` per SPEC §6.9. Token
type scheme settled in P7: custom `vimwiki*` types
(`vimwikiHeading`, `vimwikiBold`, `vimwikiCodeBlock`, `vimwikiKeyword`,
`vimwikiWikilink`, `vimwikiTransclusion`, …) plus `level1`..`level6`
and `centered` modifiers. Default highlight groups for these arrive
in Phase 9.

Pipeline (`semantic_tokens.rs`):

- `legend()` builds the `SemanticTokensLegend` advertised in
  `initialize` capabilities.
- `LineIndex` precomputes byte offsets per line so multi-line span
  splitting and UTF-16 conversion stay linear.
- `collect_tokens` walks the AST. Blocks that are visually distinct
  (heading, preformatted, math-block, comment) emit one token over
  their span; multi-line ones get split per line because the LSP wire
  format can't represent a token that crosses a newline. Container
  blocks (paragraph, list item, table cell, blockquote) recurse into
  inlines instead of wrapping them, so we end up with a non-overlapping
  tile of inline tokens. Headings shadow nested inlines (one heading
  token covers the line, no Bold-inside-Heading).
- `encode` produces the LSP delta-quintuple stream
  `(deltaLine, deltaStart, length, type, mods)`. UTF-16 mode converts
  byte offsets/lengths to code-unit counts on the fly (BMP + surrogate
  pairs).
- `build_data` is the entry point used by both handlers; the `range`
  variant filters tokens by overlap in the same encoding the client
  requested.

Capability declared as `SemanticTokensOptions { full = true,
range = true, legend = … }`. Both handlers reuse the document store.

P7 resolved in SPEC §11 — custom vimwiki types with rationale and a
note that Phase 9 will ship default highlight-group definitions in
`syntax/nuwiki.vim` and the Lua glue.

Tests (21 new): legend integrity, per-construct emission for every
inline + block kind, heading shadowing of inner inlines, multi-line
code/math block splitting, sort order, delta encoding (same line and
crossing lines), UTF-16 length conversion for é, range filter,
empty-doc edge case, end-to-end `build_data` parity.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-10 21:14:01 +00:00

472 lines
16 KiB
Rust

//! LSP protocol bridge for nuwiki.
//!
//! `tower-lsp` server that maintains a `DocumentStore` (SPEC §6.10) and
//! exposes the Phase 6 feature set:
//!
//! - `initialize` / `initialized` / `shutdown`
//! - `textDocument/didOpen`, `didChange` (full sync), `didClose`
//! - `textDocument/publishDiagnostics` from `BlockNode::Error` nodes
//! - `textDocument/documentSymbol` — nested outline from headings
//!
//! Position encoding is negotiated as UTF-8 when the client supports LSP
//! 3.17+ encodings (SPEC §6.11). When the client only supports the legacy
//! UTF-16 default, positions get translated through a per-line lookup.
//!
//! Re-parses are full per change (incremental parsing deferred post-v1).
pub mod semantic_tokens;
use std::io;
use std::sync::atomic::{AtomicBool, Ordering};
use std::sync::Arc;
use dashmap::DashMap;
use tokio::io::{AsyncRead, AsyncWrite};
use tower_lsp::jsonrpc::Result as LspResult;
use tower_lsp::lsp_types::{
Diagnostic, DiagnosticSeverity, DidChangeTextDocumentParams, DidCloseTextDocumentParams,
DidOpenTextDocumentParams, DocumentSymbol, DocumentSymbolParams, DocumentSymbolResponse,
InitializeParams, InitializeResult, InitializedParams, MessageType, OneOf,
PositionEncodingKind, SemanticTokens, SemanticTokensFullOptions, SemanticTokensOptions,
SemanticTokensParams, SemanticTokensRangeParams, SemanticTokensRangeResult,
SemanticTokensResult, SemanticTokensServerCapabilities, ServerCapabilities, ServerInfo,
SymbolKind, TextDocumentSyncCapability, TextDocumentSyncKind, Url, WorkDoneProgressOptions,
};
use tower_lsp::{Client, LanguageServer, LspService, Server};
use nuwiki_core::ast::{
BlockNode, BlockquoteNode, DocumentNode, ErrorNode, HeadingNode, InlineNode, ListItemNode,
ListNode, Span,
};
use nuwiki_core::syntax::vimwiki::VimwikiSyntax;
use nuwiki_core::syntax::SyntaxRegistry;
/// Run the LSP server over the given async reader/writer pair. The binary
/// (`nuwiki-ls`) just calls this with stdin/stdout.
pub async fn run<I, O>(reader: I, writer: O)
where
I: AsyncRead + Unpin,
O: AsyncWrite + Unpin,
{
let (service, socket) = LspService::new(Backend::new);
Server::new(reader, writer, socket).serve(service).await;
}
/// Convenience: run on stdin/stdout. Fails only if stdio handles are
/// unavailable.
pub async fn run_stdio() -> io::Result<()> {
let stdin = tokio::io::stdin();
let stdout = tokio::io::stdout();
run(stdin, stdout).await;
Ok(())
}
#[derive(Debug)]
struct DocumentState {
text: String,
ast: DocumentNode,
/// Last version we observed from the client. Held per SPEC §6.10 even
/// though the foundation phase doesn't yet need to read it back.
#[allow(dead_code)]
version: i32,
}
struct Backend {
client: Client,
documents: Arc<DashMap<Url, DocumentState>>,
registry: Arc<SyntaxRegistry>,
/// True after `initialize` if the client opted into UTF-8 encoding.
/// Otherwise we keep the LSP 3.16 default of UTF-16.
use_utf8: Arc<AtomicBool>,
}
impl Backend {
fn new(client: Client) -> Self {
let mut registry = SyntaxRegistry::new();
registry.register(VimwikiSyntax::new());
Self {
client,
documents: Arc::new(DashMap::new()),
registry: Arc::new(registry),
use_utf8: Arc::new(AtomicBool::new(false)),
}
}
async fn update_document(&self, uri: Url, text: String, version: i32) {
// For Phase 6 we always treat unknown buffers as vimwiki — if/when
// multi-syntax dispatch lands the pick should follow the URI's ext.
let plugin = match self.registry.get("vimwiki") {
Some(p) => p,
None => {
self.client
.log_message(MessageType::ERROR, "vimwiki plugin missing")
.await;
return;
}
};
let ast = plugin.parse(&text);
let diagnostics = ast_diagnostics(&ast, &text, self.use_utf8.load(Ordering::Relaxed));
self.documents
.insert(uri.clone(), DocumentState { text, ast, version });
self.client
.publish_diagnostics(uri, diagnostics, Some(version))
.await;
}
}
#[tower_lsp::async_trait]
impl LanguageServer for Backend {
async fn initialize(&self, params: InitializeParams) -> LspResult<InitializeResult> {
// SPEC §6.11: prefer UTF-8 if the client advertises support.
let supports_utf8 = params
.capabilities
.general
.as_ref()
.and_then(|g| g.position_encodings.as_ref())
.map(|encs| encs.iter().any(|e| *e == PositionEncodingKind::UTF8))
.unwrap_or(false);
self.use_utf8.store(supports_utf8, Ordering::Relaxed);
let position_encoding = if supports_utf8 {
Some(PositionEncodingKind::UTF8)
} else {
None // server defaults to UTF-16 per LSP 3.16.
};
Ok(InitializeResult {
capabilities: ServerCapabilities {
position_encoding,
text_document_sync: Some(TextDocumentSyncCapability::Kind(
TextDocumentSyncKind::FULL,
)),
document_symbol_provider: Some(OneOf::Left(true)),
semantic_tokens_provider: Some(
SemanticTokensServerCapabilities::SemanticTokensOptions(
SemanticTokensOptions {
work_done_progress_options: WorkDoneProgressOptions::default(),
legend: semantic_tokens::legend(),
range: Some(true),
full: Some(SemanticTokensFullOptions::Bool(true)),
},
),
),
..ServerCapabilities::default()
},
server_info: Some(ServerInfo {
name: env!("CARGO_PKG_NAME").into(),
version: Some(env!("CARGO_PKG_VERSION").into()),
}),
})
}
async fn initialized(&self, _params: InitializedParams) {
let enc = if self.use_utf8.load(Ordering::Relaxed) {
"utf-8"
} else {
"utf-16"
};
self.client
.log_message(
MessageType::INFO,
format!("nuwiki-lsp initialized (positionEncoding={enc})"),
)
.await;
}
async fn shutdown(&self) -> LspResult<()> {
Ok(())
}
async fn did_open(&self, params: DidOpenTextDocumentParams) {
let uri = params.text_document.uri;
let text = params.text_document.text;
let version = params.text_document.version;
self.update_document(uri, text, version).await;
}
async fn did_change(&self, params: DidChangeTextDocumentParams) {
let uri = params.text_document.uri;
let version = params.text_document.version;
// Full sync: the *last* content change carries the whole document.
if let Some(change) = params.content_changes.into_iter().last() {
self.update_document(uri, change.text, version).await;
}
}
async fn did_close(&self, params: DidCloseTextDocumentParams) {
self.documents.remove(&params.text_document.uri);
// Clear lingering diagnostics so closed buffers don't keep red squiggles.
self.client
.publish_diagnostics(params.text_document.uri, Vec::new(), None)
.await;
}
async fn document_symbol(
&self,
params: DocumentSymbolParams,
) -> LspResult<Option<DocumentSymbolResponse>> {
let uri = &params.text_document.uri;
let Some(doc) = self.documents.get(uri) else {
return Ok(None);
};
let symbols =
headings_to_symbols(&doc.ast, &doc.text, self.use_utf8.load(Ordering::Relaxed));
Ok(Some(DocumentSymbolResponse::Nested(symbols)))
}
async fn semantic_tokens_full(
&self,
params: SemanticTokensParams,
) -> LspResult<Option<SemanticTokensResult>> {
let uri = &params.text_document.uri;
let Some(doc) = self.documents.get(uri) else {
return Ok(None);
};
let utf8 = self.use_utf8.load(Ordering::Relaxed);
let data = semantic_tokens::build_data(&doc.ast, &doc.text, utf8, None);
Ok(Some(SemanticTokensResult::Tokens(SemanticTokens {
result_id: None,
data: pack_data(&data),
})))
}
async fn semantic_tokens_range(
&self,
params: SemanticTokensRangeParams,
) -> LspResult<Option<SemanticTokensRangeResult>> {
let uri = &params.text_document.uri;
let Some(doc) = self.documents.get(uri) else {
return Ok(None);
};
let utf8 = self.use_utf8.load(Ordering::Relaxed);
let data = semantic_tokens::build_data(&doc.ast, &doc.text, utf8, Some(params.range));
Ok(Some(SemanticTokensRangeResult::Tokens(SemanticTokens {
result_id: None,
data: pack_data(&data),
})))
}
}
fn pack_data(flat: &[u32]) -> Vec<tower_lsp::lsp_types::SemanticToken> {
flat.chunks_exact(5)
.map(|c| tower_lsp::lsp_types::SemanticToken {
delta_line: c[0],
delta_start: c[1],
length: c[2],
token_type: c[3],
token_modifiers_bitset: c[4],
})
.collect()
}
// ===== Pure helpers =====
/// Convert one of our `Position` values into an LSP `Position`.
///
/// In UTF-8 encoding mode (negotiated during `initialize`) the byte column
/// is the LSP `character`. In UTF-16 mode the LSP `character` is a UTF-16
/// code-unit offset, so we walk the source line to translate.
pub fn to_lsp_position(
pos: &nuwiki_core::ast::Position,
text: &str,
utf8: bool,
) -> tower_lsp::lsp_types::Position {
let character = if utf8 {
pos.column
} else {
utf16_column(text, pos.line, pos.column)
};
tower_lsp::lsp_types::Position {
line: pos.line,
character,
}
}
pub fn to_lsp_range(span: &Span, text: &str, utf8: bool) -> tower_lsp::lsp_types::Range {
tower_lsp::lsp_types::Range {
start: to_lsp_position(&span.start, text, utf8),
end: to_lsp_position(&span.end, text, utf8),
}
}
/// Convert a byte-offset column on `line` into the corresponding UTF-16
/// code-unit count from the start of that line. Out-of-range inputs are
/// clamped to the line length, matching the LSP spec's "if the character
/// value is greater than the length of the line it is clipped to the
/// length".
fn utf16_column(text: &str, line: u32, byte_col: u32) -> u32 {
let mut current_line = 0u32;
let mut line_start = 0usize;
let bytes = text.as_bytes();
while current_line < line && line_start < bytes.len() {
if let Some(nl) = bytes[line_start..].iter().position(|b| *b == b'\n') {
line_start += nl + 1;
current_line += 1;
} else {
return 0;
}
}
let line_end = bytes[line_start..]
.iter()
.position(|b| *b == b'\n')
.map(|i| line_start + i)
.unwrap_or(bytes.len());
let target = (line_start + byte_col as usize).min(line_end);
text[line_start..target]
.chars()
.map(char::len_utf16)
.sum::<usize>() as u32
}
pub fn ast_diagnostics(ast: &DocumentNode, text: &str, utf8: bool) -> Vec<Diagnostic> {
let mut out = Vec::new();
walk_blocks_for_errors(&ast.children, text, utf8, &mut out);
out
}
fn walk_blocks_for_errors(blocks: &[BlockNode], text: &str, utf8: bool, out: &mut Vec<Diagnostic>) {
for block in blocks {
match block {
BlockNode::Error(err) => out.push(error_to_diagnostic(err, text, utf8)),
BlockNode::Blockquote(BlockquoteNode { children, .. }) => {
walk_blocks_for_errors(children, text, utf8, out);
}
BlockNode::List(ListNode { items, .. }) => {
for item in items {
walk_list_item_for_errors(item, text, utf8, out);
}
}
_ => {}
}
}
}
fn walk_list_item_for_errors(
item: &ListItemNode,
text: &str,
utf8: bool,
out: &mut Vec<Diagnostic>,
) {
if let Some(sub) = &item.sublist {
for sub_item in &sub.items {
walk_list_item_for_errors(sub_item, text, utf8, out);
}
}
let _ = (text, utf8, out);
}
fn error_to_diagnostic(err: &ErrorNode, text: &str, utf8: bool) -> Diagnostic {
Diagnostic {
range: to_lsp_range(&err.span, text, utf8),
severity: Some(DiagnosticSeverity::ERROR),
message: err.message.clone(),
source: Some("nuwiki".into()),
..Diagnostic::default()
}
}
pub fn headings_to_symbols(ast: &DocumentNode, text: &str, utf8: bool) -> Vec<DocumentSymbol> {
let headings: Vec<&HeadingNode> = ast
.children
.iter()
.filter_map(|b| match b {
BlockNode::Heading(h) => Some(h),
_ => None,
})
.collect();
#[allow(deprecated)]
fn build<'a>(
headings: &mut std::slice::Iter<'a, &'a HeadingNode>,
text: &str,
utf8: bool,
parent_level: u8,
peeked: &mut Option<&'a HeadingNode>,
) -> Vec<DocumentSymbol> {
let mut out = Vec::new();
loop {
let h = match peeked.take().or_else(|| headings.next().copied()) {
Some(h) => h,
None => return out,
};
if h.level <= parent_level {
*peeked = Some(h);
return out;
}
let children = build(headings, text, utf8, h.level, peeked);
let title = inline_to_text(&h.children);
out.push(DocumentSymbol {
name: if title.is_empty() {
"(empty heading)".into()
} else {
title
},
detail: None,
kind: SymbolKind::STRING,
tags: None,
deprecated: None,
range: to_lsp_range(&h.span, text, utf8),
selection_range: to_lsp_range(&h.span, text, utf8),
children: if children.is_empty() {
None
} else {
Some(children)
},
});
}
}
let mut iter = headings.iter();
let mut peeked: Option<&HeadingNode> = None;
build(&mut iter, text, utf8, 0, &mut peeked)
}
fn inline_to_text(inlines: &[InlineNode]) -> String {
let mut out = String::new();
inline_to_text_into(inlines, &mut out);
out.trim().to_string()
}
fn inline_to_text_into(inlines: &[InlineNode], out: &mut String) {
for n in inlines {
match n {
InlineNode::Text(t) => out.push_str(&t.content),
InlineNode::Bold(b) => inline_to_text_into(&b.children, out),
InlineNode::Italic(i) => inline_to_text_into(&i.children, out),
InlineNode::BoldItalic(bi) => inline_to_text_into(&bi.children, out),
InlineNode::Strikethrough(s) => inline_to_text_into(&s.children, out),
InlineNode::Code(c) => out.push_str(&c.content),
InlineNode::Superscript(s) => inline_to_text_into(&s.children, out),
InlineNode::Subscript(s) => inline_to_text_into(&s.children, out),
InlineNode::MathInline(m) => out.push_str(&m.content),
InlineNode::Keyword(k) => out.push_str(keyword_str(k.keyword)),
InlineNode::Color(c) => inline_to_text_into(&c.children, out),
InlineNode::WikiLink(w) => match &w.description {
Some(d) => inline_to_text_into(d, out),
None => {
if let Some(p) = &w.target.path {
out.push_str(p);
}
}
},
InlineNode::ExternalLink(e) => match &e.description {
Some(d) => inline_to_text_into(d, out),
None => out.push_str(&e.url),
},
InlineNode::Transclusion(t) => out.push_str(t.alt.as_deref().unwrap_or(&t.url)),
InlineNode::RawUrl(r) => out.push_str(&r.url),
}
}
}
fn keyword_str(k: nuwiki_core::ast::Keyword) -> &'static str {
match k {
nuwiki_core::ast::Keyword::Todo => "TODO",
nuwiki_core::ast::Keyword::Done => "DONE",
nuwiki_core::ast::Keyword::Started => "STARTED",
nuwiki_core::ast::Keyword::Fixme => "FIXME",
nuwiki_core::ast::Keyword::Fixed => "FIXED",
nuwiki_core::ast::Keyword::Xxx => "XXX",
}
}