all repos

clerk @ 7ca4635

missing tooling for ledger/hledger

clerk/internal/lsp/textdocument_completion.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
lsp: completion, 2 months ago
1
package lsp
2
3
import (
4
	"context"
5
	"fmt"
6
	"math"
7
	"sort"
8
	"strings"
9
	"time"
10
11
	"go.lsp.dev/protocol"
12
13
	"olexsmir.xyz/clerk/internal/analyzer"
14
	"olexsmir.xyz/clerk/internal/lsp/fuzzy"
15
	"olexsmir.xyz/clerk/internal/lsp/lsputil"
16
	"olexsmir.xyz/clerk/journal/ast"
17
	"olexsmir.xyz/clerk/journal/lexer"
18
	"olexsmir.xyz/clerk/journal/token"
19
)
20
21
func (s *server) Completion(ctx context.Context, params *protocol.CompletionParams) (protocol.CompletionResult, error) {
22
	state, ok := s.getDocState(params.TextDocument.URI)
23
	if !ok {
24
		return &protocol.CompletionList{}, nil
25
	}
26
	cursor := lsputil.Offset(state.text, int(params.Position.Line), int(params.Position.Character))
27
	if cursor > len(state.text) {
28
		return &protocol.CompletionList{}, nil
29
	}
30
	detectedCtx, start := detectCompletionCtx(state.text, cursor)
31
	if detectedCtx == cmplNone {
32
		return &protocol.CompletionList{}, nil
33
	}
34
	a := s.analysis()
35
	if a == nil {
36
		return &protocol.CompletionList{}, nil
37
	}
38
	return &protocol.CompletionList{
39
		IsIncomplete: true,
40
		Items:        cmplItems(a, detectedCtx, state.text, start, cursor),
41
	}, nil
42
}
43
44
const maxCompletionItems = 50
45
46
type cmplCtx int
47
48
const (
49
	cmplNone cmplCtx = iota
50
	cmplAccount
51
	cmplPayee
52
	cmplCommodity
53
	cmplTagName
54
	cmplDirective
55
)
56
57
var directiveKeywords = []string{
58
	"account", "include", "commodity", "payee", "decimal-mark", "alias",
59
	"apply", "end", "tag", "year", "D", "P", "N", "C", "Y",
60
}
61
62
func detectCompletionCtx(content string, cursor int) (cmplCtx, int) {
63
	toks := lexLine(content, cursor)
64
	lineStart, _ := lineBounds(content, cursor)
65
66
	if m := commentMarker(toks, cursor); m != -1 {
67
		return cmplTagContext(content, toks[m].Span.End.Offset, cursor)
68
	}
69
	if len(toks) == 0 {
70
		return cmplDirective, lineStart
71
	}
72
73
	switch toks[0].Type {
74
	case token.INDENT:
75
		return cmplPostingCtx(content, cursor, toks)
76
	case token.DATE:
77
		return cmplHeaderCtx(content, cursor, toks)
78
	case token.ACCOUNT, token.COMMODITY, token.PAYEE, token.TAG:
79
		return cmplDirectiveContext(cursor, lineStart, toks)
80
	case token.TEXT:
81
		return cmplDirective, lineStart // half-typed keyword or unparseable line
82
	}
83
	return cmplNone, cursor
84
}
85
86
func cmplPostingCtx(content string, cursor int, toks []token.Token) (cmplCtx, int) {
87
	if inDirectiveBody(content, cursor) {
88
		return cmplNone, cursor
89
	}
90
	fieldStart := toks[0].Span.End.Offset
91
	i := 1
92
	for i < len(toks) {
93
		switch toks[i].Type {
94
		case token.STAR, token.BANG, token.LPAREN, token.LBRACKET, token.WHITESPACE:
95
			fieldStart = toks[i].Span.End.Offset
96
			i++
97
		default:
98
			goto run
99
		}
100
	}
101
run:
102
	// account run: consecutive account-name segments and colons
103
	fieldEnd := fieldStart
104
	for ; i < len(toks); i++ {
105
		if toks[i].Type != token.TEXT && toks[i].Type != token.COLON {
106
			break
107
		}
108
		fieldEnd = toks[i].Span.End.Offset
109
	}
110
	if cursor <= fieldEnd && cursor >= fieldStart {
111
		return cmplAccount, fieldStart
112
	}
113
	if cursor > fieldEnd {
114
		if t := tokenUnder(toks, cursor); t != nil && (t.Type == token.COMMODITYMARK || t.Type == token.STRING) {
115
			start := t.Span.Start.Offset
116
			if t.Type == token.STRING {
117
				start++ // skip opening quote
118
			}
119
			return cmplCommodity, start
120
		}
121
		if strings.TrimSpace(content[fieldEnd:cursor]) == "" {
122
			return cmplCommodity, cursor
123
		}
124
	}
125
	return cmplNone, cursor
126
}
127
128
func cmplHeaderCtx(content string, cursor int, toks []token.Token) (cmplCtx, int) {
129
	// skip date, status, code, and whitespace - where the payee beginds
130
	fieldStart := toks[0].Span.End.Offset
131
	fieldEnd := fieldStart
132
	seen := false
133
	for i := 1; i < len(toks); i++ {
134
		t := toks[i]
135
		switch t.Type {
136
		case token.WHITESPACE, token.STAR, token.BANG, token.DATE, token.TIME,
137
			token.EQ, token.EQEQ, token.EQEQEQ:
138
			fieldStart = t.Span.End.Offset
139
			fieldEnd = t.Span.End.Offset
140
		case token.TEXT:
141
			lit := content[t.Span.Start.Offset:t.Span.End.Offset]
142
			if !seen && len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
143
				fieldStart = t.Span.End.Offset // parenthesized code
144
				fieldEnd = t.Span.End.Offset
145
				continue
146
			}
147
			if !seen {
148
				fieldStart = t.Span.Start.Offset
149
				seen = true
150
			}
151
			fieldEnd = t.Span.End.Offset
152
			if p := strings.IndexByte(lit, '|'); p >= 0 {
153
				fieldEnd = t.Span.Start.Offset + p // "payee|note" keeps the pipe in the token
154
				return payeeAt(cursor, fieldStart, fieldEnd)
155
			}
156
		case token.STRING:
157
			if !seen {
158
				fieldStart = t.Span.Start.Offset + 1 // skip opening quote
159
				seen = true
160
			}
161
			fieldEnd = t.Span.End.Offset
162
		default:
163
			// PIPE, SEMICOLON, ...
164
			return payeeAt(cursor, fieldStart, fieldEnd)
165
		}
166
	}
167
	if seen {
168
		return payeeAt(cursor, fieldStart, fieldEnd)
169
	}
170
	// no payee yet: the payee field is the whitespace after the header meta
171
	if cursor >= fieldStart && strings.TrimSpace(content[fieldStart:cursor]) == "" {
172
		return cmplPayee, cursor
173
	}
174
	return cmplNone, cursor
175
}
176
177
func payeeAt(cursor, start, end int) (cmplCtx, int) {
178
	if cursor >= start && cursor <= end {
179
		return cmplPayee, start
180
	}
181
	return cmplNone, cursor
182
}
183
184
// cmplDirectiveContext classifies a directive line. keyword completion before the keyword ends, symbol completion in the value field after
185
func cmplDirectiveContext(cursor, lineStart int, toks []token.Token) (cmplCtx, int) {
186
	kwEnd := toks[0].Span.End.Offset
187
	if cursor <= kwEnd {
188
		return cmplDirective, lineStart
189
	}
190
	start := kwEnd
191
	for i := 1; i < len(toks); i++ {
192
		if toks[i].Type == token.WHITESPACE || toks[i].Span.End.Offset <= kwEnd {
193
			continue
194
		}
195
		start = toks[i].Span.Start.Offset
196
		if toks[i].Type == token.STRING {
197
			start++ // skip opening quote
198
		}
199
		break
200
	}
201
	if start > cursor {
202
		start = cursor
203
	}
204
	switch toks[0].Type {
205
	case token.ACCOUNT:
206
		return cmplAccount, start
207
	case token.COMMODITY:
208
		return cmplCommodity, start
209
	case token.PAYEE:
210
		return cmplPayee, start
211
	case token.TAG:
212
		return cmplTagName, start
213
	}
214
	return cmplNone, cursor
215
}
216
217
// cmplTagContext completes tag names at the start of a comment, stopping at first ':'.
218
func cmplTagContext(content string, commentStart, cursor int) (cmplCtx, int) {
219
	if strings.ContainsRune(content[commentStart:cursor], ':') {
220
		return cmplNone, cursor
221
	}
222
	keyStart := commentStart + lastSeparator(content[commentStart:cursor]) + 1
223
	return cmplTagName, keyStart
224
}
225
226
// commentMarker returns index of the first comment marker token at or before the cursor, or -1.
227
func commentMarker(toks []token.Token, cursor int) int {
228
	for i, t := range toks {
229
		switch t.Type {
230
		case token.SEMICOLON, token.HASH, token.PERCENT:
231
			if t.Span.Start.Offset <= cursor {
232
				return i
233
			}
234
		case token.STAR:
235
			if i == 0 && t.Span.Start.Offset <= cursor {
236
				return i
237
			}
238
		}
239
	}
240
	return -1
241
}
242
243
func inDirectiveBody(content string, cursor int) bool {
244
	lineStart, _ := lineBounds(content, cursor)
245
	if toks := lexLine(content, lineStart); len(toks) == 0 || toks[0].Type != token.INDENT {
246
		return false
247
	}
248
	for lineStart > 0 {
249
		lineStart, _ = lineBounds(content, lineStart-1)
250
		toks := lexLine(content, lineStart)
251
		if len(toks) == 0 || toks[0].Type != token.INDENT {
252
			return len(toks) > 0 && (toks[0].Type == token.ACCOUNT || toks[0].Type == token.COMMODITY)
253
		}
254
	}
255
	return false
256
}
257
258
type cmplCand struct {
259
	label    string
260
	score    float64
261
	count    int
262
	lastUsed ast.Date
263
}
264
265
// cmplItems ranks candidates for the content against typed pattern
266
func cmplItems(a *analyzer.Analysis, ctx cmplCtx, content string, start, cursor int) []protocol.CompletionItem {
267
	pattern := content[start:cursor]
268
269
	var kind protocol.CompletionItemKind
270
	var cands []cmplCand
271
	switch ctx {
272
	case cmplAccount:
273
		kind = protocol.CompletionItemKindClass
274
		for name, info := range a.Accounts {
275
			cands = append(cands, cmplCand{label: name, count: info.UsedCount, lastUsed: info.LastUsed})
276
		}
277
	case cmplPayee:
278
		kind = protocol.CompletionItemKindVariable
279
		for name, info := range a.Payees {
280
			cands = append(cands, cmplCand{label: name, count: info.UsedCount, lastUsed: info.LastUsed})
281
		}
282
	case cmplCommodity:
283
		kind = protocol.CompletionItemKindValue
284
		for name, info := range a.Commodities {
285
			cands = append(cands, cmplCand{label: name, count: info.UsedCount, lastUsed: info.LastUsed})
286
		}
287
	case cmplTagName:
288
		kind = protocol.CompletionItemKindProperty
289
		for _, name := range a.TagNames {
290
			cands = append(cands, cmplCand{label: name})
291
		}
292
	case cmplDirective:
293
		kind = protocol.CompletionItemKindKeyword
294
		for _, name := range directiveKeywords {
295
			cands = append(cands, cmplCand{label: name})
296
		}
297
	default:
298
		return nil
299
	}
300
301
	var newest ast.Date
302
	if n := len(a.Dates); n > 0 {
303
		newest = a.Dates[n-1]
304
	}
305
	ranked := cands[:0]
306
	for i := range cands {
307
		sc := fuzzy.Score(pattern, cands[i].label)
308
		if sc != 0 {
309
			sc *= 1 + math.Log1p(float64(cands[i].count))
310
			if cands[i].count > 0 && cands[i].lastUsed.Year != 0 && newest.Year != 0 {
311
				days := daysBetween(cands[i].lastUsed, newest)
312
				sc *= 1 + 0.5*max(0, 1-float64(days)/365)
313
			}
314
		}
315
		cands[i].score = sc
316
		if sc != 0 {
317
			ranked = append(ranked, cands[i])
318
		}
319
	}
320
	sort.Slice(ranked, func(i, j int) bool {
321
		if ranked[i].score != ranked[j].score {
322
			return ranked[i].score > ranked[j].score
323
		}
324
		if ranked[i].lastUsed != ranked[j].lastUsed {
325
			return ranked[i].lastUsed.Compare(ranked[j].lastUsed) > 0
326
		}
327
		return ranked[i].label < ranked[j].label
328
	})
329
	if len(ranked) > maxCompletionItems {
330
		ranked = ranked[:maxCompletionItems]
331
	}
332
333
	replace := protocol.Range{
334
		Start: lsputil.Position(content, start),
335
		End:   lsputil.Position(content, cursor),
336
	}
337
	items := make([]protocol.CompletionItem, len(ranked))
338
	for i, r := range ranked {
339
		it := protocol.CompletionItem{
340
			Label:      r.label,
341
			Kind:       kind,
342
			SortText:   protocol.NewOptional(fmt.Sprintf("%04d", i)),
343
			FilterText: protocol.NewOptional(r.label),
344
			TextEdit: &protocol.TextEdit{
345
				Range:   replace,
346
				NewText: r.label,
347
			},
348
		}
349
		if r.count > 0 {
350
			it.Detail = protocol.NewOptional(fmt.Sprintf("%d uses", r.count))
351
		}
352
		items[i] = it
353
	}
354
	return items
355
}
356
357
func daysBetween(a, b ast.Date) int {
358
	return int(time.Date(b.Year, time.Month(b.Month), b.Day, 0, 0, 0, 0, time.UTC).
359
		Sub(time.Date(a.Year, time.Month(a.Month), a.Day, 0, 0, 0, 0, time.UTC)).
360
		Hours() / 24)
361
}
362
363
// lineBounds returns the byte offsets of the line containing cursor
364
func lineBounds(content string, cursor int) (start, end int) {
365
	start = cursor
366
	for start > 0 && content[start-1] != '\n' && content[start-1] != '\r' {
367
		start--
368
	}
369
	end = start
370
	for end < len(content) && content[end] != '\n' && content[end] != '\r' {
371
		end++
372
	}
373
	return start, end
374
}
375
376
func lastSeparator(s string) int {
377
	for i := len(s) - 1; i >= 0; i-- {
378
		switch s[i] {
379
		case ' ', '\t', ',':
380
			return i
381
		}
382
	}
383
	return -1
384
}
385
386
func lexLine(content string, cursor int) []token.Token {
387
	lineStart, lineEnd := lineBounds(content, cursor)
388
	l := lexer.New("", []byte(content[lineStart:lineEnd]))
389
	var out []token.Token
390
	for {
391
		t := l.Next()
392
		if t.Type == token.EOF || t.Type == token.NEWLINE {
393
			break
394
		}
395
		t.Span.Start.Offset += lineStart
396
		t.Span.End.Offset += lineStart
397
		out = append(out, t)
398
	}
399
	return out
400
}
401
402
func tokenUnder(toks []token.Token, cursor int) *token.Token {
403
	for i := range toks {
404
		t := &toks[i]
405
		if t.Span.Start.Offset <= cursor && cursor <= t.Span.End.Offset {
406
			return t
407
		}
408
	}
409
	return nil
410
}