8 files changed,
276 insertions(+),
24 deletions(-)
Author:
Oleksandr Smirnov
olexsmir@gmail.com
Committed at:
2026-08-13 13:40:52 +0300
Authored at:
2026-08-13 13:23:58 +0300
Change ID:
qumwxvszquouyroorypovowvwquonnpz
Parent:
33b67bd
jump to
A
internal/lsp/lsputil/lineindex.go
··· 1 +package lsputil 2 + 3 +import ( 4 + "sort" 5 + "unicode/utf8" 6 + 7 + "go.lsp.dev/protocol" 8 + 9 + "olexsmir.xyz/clerk/journal/token" 10 +) 11 + 12 +// LineIndex resolves byte offsets to LSP positions (and back) in O(log n) from 13 +// a precomputed table of line starts, avoiding a per-request content scan. 14 +type LineIndex struct { 15 + content string 16 + starts []int // byte offset of each line's first byte; starts[0] == 0 17 +} 18 + 19 +// NewLineIndex builds the line-start table for content. 20 +func NewLineIndex(content string) *LineIndex { 21 + starts := make([]int, 1, len(content)/20+1) 22 + for i := 0; i < len(content); i++ { 23 + if content[i] == '\n' { 24 + starts = append(starts, i+1) 25 + } 26 + } 27 + return &LineIndex{content: content, starts: starts} 28 +} 29 + 30 +// Position converts a byte offset to a 0-based LSP position. 31 +func (l *LineIndex) Position(offset int) protocol.Position { 32 + if offset < 0 { 33 + offset = 0 34 + } 35 + if offset > len(l.content) { 36 + offset = len(l.content) 37 + } 38 + idx := sort.Search(len(l.starts), func(i int) bool { return l.starts[i] > offset }) - 1 39 + lineStart := l.starts[idx] 40 + return protocol.Position{ 41 + Line: uint32(idx), 42 + Character: uint32(Utf16Col(l.content[lineStart:offset], offset-lineStart)), 43 + } 44 +} 45 + 46 +// Offset converts a 0-based line and UTF-16 code unit column to a byte offset, 47 +// clamped to the content bounds. The inverse of Position; matches the 48 +// standalone Offset on the same content. 49 +func (l *LineIndex) Offset(line, col int) int { 50 + if line < 0 { 51 + line = 0 52 + } 53 + if line >= len(l.starts) { 54 + return len(l.content) 55 + } 56 + lineStart := l.starts[line] 57 + lineEnd := len(l.content) 58 + if line+1 < len(l.starts) { 59 + lineEnd = l.starts[line+1] 60 + } 61 + for lineEnd > lineStart && (l.content[lineEnd-1] == '\n' || l.content[lineEnd-1] == '\r') { 62 + lineEnd-- 63 + } 64 + off := lineStart 65 + units := 0 66 + for off < lineEnd && units < col { 67 + r, size := utf8.DecodeRuneInString(l.content[off:lineEnd]) 68 + off += size 69 + units += utf16Len(r) 70 + } 71 + return off 72 +} 73 + 74 +// SpanRange converts a span to a protocol range, trimming trailing whitespace and newlines from the end. 75 +func (l *LineIndex) SpanRange(span token.Span) protocol.Range { 76 + return protocol.Range{ 77 + Start: l.Position(span.Start.Offset), 78 + End: l.Position(l.clampEnd(span.End.Offset)), 79 + } 80 +} 81 + 82 +func (l *LineIndex) clampEnd(end int) int { 83 + for end > 0 { 84 + switch l.content[end-1] { 85 + case ' ', '\t', '\r', '\n': 86 + end-- 87 + default: 88 + return end 89 + } 90 + } 91 + return end 92 +}
A
internal/lsp/lsputil/lineindex_test.go
··· 1 +package lsputil 2 + 3 +import ( 4 + "testing" 5 + 6 + "olexsmir.xyz/clerk/journal/token" 7 +) 8 + 9 +// TestLineIndexParity asserts Offset/Position agree with the standalone 10 +// converters, including clamped out-of-range positions. 11 +func TestLineIndexParity(t *testing.T) { 12 + contents := []string{"abc\ndef\n", "abc\ndef", "abc", "a\nb\nc", "", "один\nдва\n"} 13 + for _, content := range contents { 14 + li := NewLineIndex(content) 15 + for _, line := range []int{0, 1, 2, 3, 100} { 16 + for _, col := range []int{0, 1, 5} { 17 + off := li.Offset(line, col) 18 + if want := Offset(content, line, col); off != want { 19 + t.Errorf("content %q: Offset(%d,%d) = %d, want %d", content, line, col, off, want) 20 + } 21 + pos := li.Position(off) 22 + if want := Position(content, off); pos != want { 23 + t.Errorf("content %q: Position(%d) = %+v, want %+v", content, off, pos, want) 24 + } 25 + } 26 + } 27 + } 28 +} 29 + 30 +func TestLineIndexSpanRange(t *testing.T) { 31 + content := "2024-01-15 grocery\n expenses:food $50\n assets:cash\n" 32 + li := NewLineIndex(content) 33 + // span ending at the start of the next line clamps to the line end 34 + span := token.Span{ 35 + Start: token.Pos{Offset: 11, Line: 1, Col: 12}, 36 + End: token.Pos{Offset: 18, Line: 2, Col: 0}, 37 + } 38 + rng := li.SpanRange(span) 39 + if rng.Start.Line != 0 || rng.Start.Character != 11 || rng.End.Line != 0 || rng.End.Character != 18 { 40 + t.Errorf("grocery span = %+v, want 0:11-0:18", rng) 41 + } 42 +}
M
internal/lsp/lsputil/utf16.go
··· 27 27 return col 28 28 } 29 29 30 +// Utf16ColBytes returns the UTF-16 code unit column of b (without newline). 31 +func Utf16ColBytes(b []byte) int { 32 + col := 0 33 + for i := 0; i < len(b); { 34 + r, size := utf8.DecodeRune(b[i:]) 35 + if r == utf8.RuneError && size <= 1 { 36 + break 37 + } 38 + col += utf16Len(r) 39 + i += size 40 + } 41 + return col 42 +} 43 + 30 44 // Utf16Len returns the UTF-16 code unit length of content[offset:end]. 31 45 func Utf16Len(content string, offset, end int) int { 32 46 if offset < 0 {
M
internal/lsp/textdocument_definition.go
··· 3 3 import ( 4 4 "context" 5 5 "slices" 6 + "sort" 6 7 7 8 "go.lsp.dev/protocol" 8 9 "go.lsp.dev/uri" ··· 20 21 } 21 22 22 23 an := s.analysis() 23 - cursor := lsputil.Offset(state.text, int(params.Position.Line), int(params.Position.Character)) 24 + cursor := state.lineIdx.Offset(int(params.Position.Line), int(params.Position.Character)) 24 25 return findDefinitionUnderCursor(an, params.TextDocument.URI.Path(), state.text, cursor), nil 25 26 } 26 27 ··· 113 114 pf := a.Files[fileIdx] 114 115 return &protocol.Location{ 115 116 URI: uri.File(pf.Path), 116 - Range: spanToProtocolRange(string(pf.Src), span), 117 + Range: spanRangeFromSrc(pf.Src, span), 117 118 } 118 119 } 119 120 120 -func spanToProtocolRange(content string, span token.Span) protocol.Range { 121 +// spanRangeFromSrc converts a span to a protocol range. Parsed spans carry 122 +// 1-based Line/Col and are converted directly; spans whose end runs into 123 +// trailing whitespace or uses the next-line-start convention (Col == 0) get a 124 +// one-line scan back from the end offset. 125 +func spanRangeFromSrc(src []byte, span token.Span) protocol.Range { 126 + if span.Start.Line == 0 || span.End.Line == 0 { 127 + // spans built from offsets without Line/Col: full line index 128 + return lsputil.NewLineIndex(string(src)).SpanRange(span) 129 + } 130 + start := protocol.Position{Line: uint32(span.Start.Line - 1), Character: uint32(span.Start.Col - 1)} 131 + end := span.End.Offset 132 + if span.End.Col > 0 && end > span.Start.Offset && !isSpanSpace(src[end-1]) { 133 + // the span's stored end position matches its offset 134 + return protocol.Range{Start: start, End: protocol.Position{Line: uint32(span.End.Line - 1), Character: uint32(span.End.Col - 1)}} 135 + } 136 + // Trim trailing whitespace back from the end offset; both scans are 137 + // bounded by the one line the span ends on. 138 + line := span.End.Line - 1 // 1-based line holding the end 139 + for end > span.Start.Offset && isSpanSpace(src[end-1]) { 140 + if src[end-1] == '\n' { 141 + line-- 142 + } 143 + end-- 144 + } 145 + lineStart := end 146 + for lineStart > 0 && src[lineStart-1] != '\n' { 147 + lineStart-- 148 + } 121 149 return protocol.Range{ 122 - Start: lsputil.Position(content, span.Start.Offset), 123 - End: lsputil.Position(content, spanEndClamped(content, span.End.Offset)), 150 + Start: start, 151 + End: protocol.Position{ 152 + Line: uint32(line - 1), 153 + Character: uint32(lsputil.Utf16ColBytes(src[lineStart:end])), 154 + }, 155 + } 156 +} 157 + 158 +func isSpanSpace(b byte) bool { 159 + switch b { 160 + case ' ', '\t', '\r', '\n': 161 + return true 124 162 } 163 + return false 125 164 } 126 165 127 166 func spanContains(content string, span token.Span, offset int) bool { ··· 130 169 } 131 170 end := spanEndClamped(content, span.End.Offset) 132 171 return span.Start.Offset <= offset && offset <= end 172 +} 173 + 174 +// entryAt returns the entry whose start offset is at or before cursor, the 175 +// only entry whose tokens can contain it. Entries are stored in file order, 176 +// so a binary search replaces a linear scan for late-cursor requests. 177 +func entryAt(entries []ast.Entry, cursor int) ast.Entry { 178 + idx := sort.Search(len(entries), func(i int) bool { return entryStart(entries[i]) > cursor }) - 1 179 + if idx < 0 { 180 + return nil 181 + } 182 + return entries[idx] 183 +} 184 + 185 +func entryStart(e ast.Entry) int { 186 + switch e := e.(type) { 187 + case *ast.BlankLine: 188 + return e.Span.Start.Offset 189 + case *ast.Transaction: 190 + return e.Span.Start.Offset 191 + case *ast.PeriodicTransaction: 192 + return e.Span.Start.Offset 193 + case *ast.AutomatedTransaction: 194 + return e.Span.Start.Offset 195 + case *ast.Comment: 196 + return e.Span.Start.Offset 197 + case *ast.AccountDirective: 198 + return e.Span.Start.Offset 199 + case *ast.CommodityDirective: 200 + return e.Span.Start.Offset 201 + case *ast.PayeeDirective: 202 + return e.Span.Start.Offset 203 + case *ast.TagDirective: 204 + return e.Span.Start.Offset 205 + case *ast.IncludeDirective: 206 + return e.Span.Start.Offset 207 + case *ast.AliasDirective: 208 + return e.Span.Start.Offset 209 + case *ast.YearDirective: 210 + return e.Span.Start.Offset 211 + case *ast.DecimalMarkDirective: 212 + return e.Span.Start.Offset 213 + case *ast.DefaultCommodityDirective: 214 + return e.Span.Start.Offset 215 + case *ast.MarketPriceDirective: 216 + return e.Span.Start.Offset 217 + case *ast.ConversionDirective: 218 + return e.Span.Start.Offset 219 + case *ast.ApplyDirective: 220 + return e.Span.Start.Offset 221 + case *ast.EndDirective: 222 + return e.Span.Start.Offset 223 + case *ast.CommentBlockDirective: 224 + return e.Span.Start.Offset 225 + case *ast.IgnoredDirective: 226 + return e.Span.Start.Offset 227 + } 228 + return 0 133 229 } 134 230 135 231 func spanEndClamped(content string, end int) int {
M
internal/lsp/textdocument_definition_test.go
··· 80 80 srv.server.current = a 81 81 82 82 for tname, tt := range map[string]int{ 83 - "1k txns, account": strings.Index(content, "\n 1:2:3 ") + len("\n ") + 2, 84 - "1k txns, payee":strings.Index(content, "transaction 1") + len("transaction "), 85 - "1k txns, commodity":strings.Index(content, "2 B @@") + len("2 B"), 83 + "1k txns, account": strings.Index(content, "\n 1:2:3 ") + len("\n ") + 2, 84 + "1k txns, payee": strings.Index(content, "transaction 1") + len("transaction "), 85 + "1k txns, commodity": strings.Index(content, "2 B @@") + len("2 B"), 86 86 } { 87 87 b.Run(tname, func(b *testing.B) { 88 88 line, col := lsputil.LineCol(content, tt)
M
internal/lsp/textdocument_hover.go
··· 8 8 "go.lsp.dev/protocol" 9 9 10 10 "olexsmir.xyz/clerk/internal/analyzer" 11 - "olexsmir.xyz/clerk/internal/lsp/lsputil" 12 11 "olexsmir.xyz/clerk/journal/ast" 13 12 "olexsmir.xyz/clerk/journal/token" 14 13 ) ··· 20 19 } 21 20 22 21 an := s.analysis() 23 - cursor := lsputil.Offset(state.text, int(params.Position.Line), int(params.Position.Character)) 22 + cursor := state.lineIdx.Offset(int(params.Position.Line), int(params.Position.Character)) 24 23 el := hoverAt(an, params.TextDocument.URI.Path(), state.text, cursor) 25 24 if el == nil { 26 25 return nil, nil ··· 31 30 Kind: protocol.MarkupKindMarkdown, 32 31 Value: buildHoverContent(an, state.text, el), 33 32 }, 34 - Range: new(spanToProtocolRange(state.text, el.span)), 33 + Range: new(state.lineIdx.SpanRange(el.span)), 35 34 }, nil 36 35 } 37 36 ··· 58 57 tx *ast.Transaction // date hover 59 58 } 60 59 60 +// hoverAt finds the hover target under the cursor in the document matching 61 +// docPath. Only the entry containing the cursor can match; entries are in 62 +// file order, so [entryAt] finds the containing entry in O(log n). 61 63 func hoverAt(an *analyzer.Analysis, docPath, content string, cursor int) *hoverElement { 62 64 for _, pf := range an.Files { 63 65 if pf.Path != docPath { 64 66 continue 65 67 } 66 - for _, entry := range pf.Ast.Entries { 67 - if el := hoverInEntry(content, entry, cursor); el != nil { 68 - return el 69 - } 68 + if entry := entryAt(pf.Ast.Entries, cursor); entry != nil { 69 + return hoverInEntry(content, entry, cursor) 70 70 } 71 71 return nil 72 72 }
M
internal/lsp/textdocument_rename.go
··· 22 22 } 23 23 24 24 an := s.analysis() 25 - cursor := lsputil.Offset(state.text, int(params.Position.Line), int(params.Position.Character)) 25 + cursor := state.lineIdx.Offset(int(params.Position.Line), int(params.Position.Character)) 26 26 ref := findSymbolUnderCursor(an, params.TextDocument.URI.Path(), state.text, cursor) 27 27 if ref == nil { 28 28 return nil, nil 29 29 } 30 30 31 31 return &protocol.PrepareRenamePlaceholder{ 32 - Range: spanToProtocolRange(state.text, ref.span), 32 + Range: state.lineIdx.SpanRange(ref.span), 33 33 Placeholder: ref.name, 34 34 }, nil 35 35 } ··· 41 41 } 42 42 43 43 an := s.analysis() 44 - cursor := lsputil.Offset(state.text, int(params.Position.Line), int(params.Position.Character)) 44 + cursor := state.lineIdx.Offset(int(params.Position.Line), int(params.Position.Character)) 45 45 ref := findSymbolUnderCursor(an, params.TextDocument.URI.Path(), state.text, cursor) 46 46 if ref == nil { 47 47 return nil, nil ··· 80 80 if pf.Path != docPath { 81 81 continue 82 82 } 83 - for _, entry := range pf.Ast.Entries { 84 - if ref := symbolInEntry(content, entry, cursor); ref != nil { 85 - return ref 86 - } 83 + if entry := entryAt(pf.Ast.Entries, cursor); entry != nil { 84 + return symbolInEntry(content, entry, cursor) 87 85 } 88 86 return nil 89 87 } ··· 248 246 changes := make(map[uri.URI][]protocol.TextEdit) 249 247 for _, pf := range an.Files { 250 248 content := string(pf.Src) 249 + // LineIndex is built lazily: most files have no matching edits, and 250 + // each edit needs an O(log n) offset-to-position lookup, not a scan. 251 + var li *lsputil.LineIndex 251 252 var edits []protocol.TextEdit 252 253 add := func(span token.Span, text string) { 254 + if li == nil { 255 + li = lsputil.NewLineIndex(content) 256 + } 253 257 edits = append(edits, protocol.TextEdit{ 254 - Range: spanToProtocolRange(content, span), 258 + Range: li.SpanRange(span), 255 259 NewText: text, 256 260 }) 257 261 }
M
internal/lsp/textdocument_sync.go
··· 6 6 "go.lsp.dev/protocol" 7 7 "go.lsp.dev/uri" 8 8 9 + "olexsmir.xyz/clerk/internal/lsp/lsputil" 9 10 "olexsmir.xyz/clerk/journal/ast" 10 11 "olexsmir.xyz/clerk/journal/lexer" 11 12 "olexsmir.xyz/clerk/journal/parser" ··· 39 40 text string 40 41 version int32 41 42 languageID protocol.LanguageKind 42 - semTokens []semanticToken // cached semantic tokens 43 + semTokens []semanticToken // cached semantic tokens 44 + lineIdx *lsputil.LineIndex // cached line index for the text 43 45 } 44 46 45 47 func (s *server) openDoc(u uri.URI, text string, version int32, langID protocol.LanguageKind) { ··· 48 50 text: text, 49 51 version: version, 50 52 languageID: langID, 53 + lineIdx: lsputil.NewLineIndex(text), 51 54 } 52 55 s.mu.Unlock() 53 56 } ··· 65 68 case *protocol.TextDocumentContentChangeWholeDocument: 66 69 state.text = ev.Text 67 70 state.semTokens = nil 71 + state.lineIdx = lsputil.NewLineIndex(ev.Text) 68 72 case *protocol.TextDocumentContentChangePartial: 69 73 _ = ev // TODO: incremental edit support 70 74 }