all repos

clerk @ 961356c

missing tooling for ledger/hledger

clerk/internal/lsp/lsputil/lineindex.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
lsp: textdocument/selectionRange..., 1 month ago
1
package lsputil
2
3
import (
4
	"sort"
5
	"unicode/utf8"
6
7
	"go.lsp.dev/protocol"
8
9
	"olexsmir.xyz/clerk/journal/token"
10
)
11
12
// LineIndex resolves byte offsets to LSP positions (and back) in O(log n) from
13
// a precomputed table of line starts, avoiding a per-request content scan.
14
type LineIndex struct {
15
	content string
16
	starts  []int // byte offset of each line's first byte; starts[0] == 0
17
}
18
19
// NewLineIndex builds the line-start table for content.
20
func NewLineIndex(content string) *LineIndex {
21
	starts := make([]int, 1, len(content)/20+1)
22
	for i := 0; i < len(content); i++ {
23
		if content[i] == '\n' {
24
			starts = append(starts, i+1)
25
		}
26
	}
27
	return &LineIndex{content: content, starts: starts}
28
}
29
30
// Position converts a byte offset to a 0-based LSP position.
31
func (l *LineIndex) Position(offset int) protocol.Position {
32
	if offset < 0 {
33
		offset = 0
34
	}
35
	if offset > len(l.content) {
36
		offset = len(l.content)
37
	}
38
	idx := sort.Search(len(l.starts), func(i int) bool { return l.starts[i] > offset }) - 1
39
	lineStart := l.starts[idx]
40
	return protocol.Position{
41
		Line:      uint32(idx),
42
		Character: uint32(Utf16Col(l.content[lineStart:offset], offset-lineStart)),
43
	}
44
}
45
46
// Offset converts a 0-based line and UTF-16 code unit column to a byte offset,
47
// clamped to the content bounds. The inverse of Position; matches the
48
// standalone Offset on the same content.
49
func (l *LineIndex) Offset(line, col int) int {
50
	if line < 0 {
51
		line = 0
52
	}
53
	if line >= len(l.starts) {
54
		return len(l.content)
55
	}
56
	lineStart := l.starts[line]
57
	lineEnd := len(l.content)
58
	if line+1 < len(l.starts) {
59
		lineEnd = l.starts[line+1]
60
	}
61
	for lineEnd > lineStart && (l.content[lineEnd-1] == '\n' || l.content[lineEnd-1] == '\r') {
62
		lineEnd--
63
	}
64
	seg := l.content[lineStart:lineEnd]
65
	ascii := true
66
	for i := range seg {
67
		if seg[i] >= utf8.RuneSelf {
68
			ascii = false
69
			break
70
		}
71
	}
72
	if ascii {
73
		if col >= len(seg) {
74
			return lineEnd
75
		}
76
		return lineStart + col
77
	}
78
	off := lineStart
79
	units := 0
80
	for off < lineEnd && units < col {
81
		r, size := utf8.DecodeRuneInString(l.content[off:lineEnd])
82
		off += size
83
		units += utf16Len(r)
84
	}
85
	return off
86
}
87
88
// SpanRange converts a span to a protocol range, trimming trailing whitespace and newlines from the end.
89
func (l *LineIndex) SpanRange(span token.Span) protocol.Range {
90
	return protocol.Range{
91
		Start: l.Position(span.Start.Offset),
92
		End:   l.Position(l.clampEnd(span.End.Offset)),
93
	}
94
}
95
96
func (l *LineIndex) clampEnd(end int) int {
97
	for end > 0 {
98
		switch l.content[end-1] {
99
		case ' ', '\t', '\r', '\n':
100
			end--
101
		default:
102
			return end
103
		}
104
	}
105
	return end
106
}