all repos

clerk @ c958816

missing tooling for ledger/hledger

clerk/internal/lsp/textdocument_completion.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
lsp: latin to cyrillic transliteration, 1 month ago
1
package lsp
2
3
import (
4
	"context"
5
	"fmt"
6
	"math"
7
	"sort"
8
	"strings"
9
	"time"
10
11
	"go.lsp.dev/protocol"
12
13
	"olexsmir.xyz/clerk/internal/analyzer"
14
	"olexsmir.xyz/clerk/internal/lsp/fuzzy"
15
	"olexsmir.xyz/clerk/internal/lsp/lsputil"
16
	"olexsmir.xyz/clerk/journal/ast"
17
	"olexsmir.xyz/clerk/journal/lexer"
18
	"olexsmir.xyz/clerk/journal/token"
19
)
20
21
func (s *server) Completion(ctx context.Context, params *protocol.CompletionParams) (protocol.CompletionResult, error) {
22
	state, ok := s.getDocState(params.TextDocument.URI)
23
	if !ok {
24
		return &protocol.CompletionList{}, nil
25
	}
26
	cursor := state.lineIdx.Offset(int(params.Position.Line), int(params.Position.Character))
27
	if cursor > len(state.text) {
28
		return &protocol.CompletionList{}, nil
29
	}
30
	detectedCtx, start := detectCompletionCtx(state.text, cursor)
31
	if detectedCtx == cmplNone {
32
		return &protocol.CompletionList{}, nil
33
	}
34
	an := s.analysisFor(params.TextDocument.URI)
35
	if an == nil {
36
		return &protocol.CompletionList{}, nil
37
	}
38
	return &protocol.CompletionList{
39
		IsIncomplete: true,
40
		Items:        cmplItems(an, detectedCtx, state.text, state.lineIdx, start, cursor, s.latinToCyrillicCompletionEnabled()),
41
	}, nil
42
}
43
44
const maxCompletionItems = 50
45
46
type cmplCtx int
47
48
const (
49
	cmplNone cmplCtx = iota
50
	cmplAccount
51
	cmplPayee
52
	cmplCommodity
53
	cmplDate
54
	cmplTagName
55
	cmplTagValue
56
	cmplDirective
57
)
58
59
var directiveKeywords = []string{
60
	"account", "include", "commodity", "payee", "decimal-mark", "alias",
61
	"apply", "end", "tag", "year", "D", "P", "N", "C", "Y",
62
}
63
64
func detectCompletionCtx(content string, cursor int) (cmplCtx, int) {
65
	toks := lexLine(content, cursor)
66
	lineStart, _ := lineBounds(content, cursor)
67
	if len(toks) == 0 {
68
		return cmplDate, lineStart
69
	}
70
	if m := commentMarker(toks, cursor); m != -1 {
71
		return cmplTagContext(content, toks[m].Span.End.Offset, cursor)
72
	}
73
74
	switch toks[0].Type {
75
	case token.INDENT:
76
		return cmplPostingCtx(content, cursor, toks)
77
	case token.DATE:
78
		return cmplHeaderCtx(content, cursor, toks)
79
	case token.ILLEGAL:
80
		// a digit that is not lexed yet, as a date: 20, 2026-0
81
		if len(toks[0].Literal) > 0 && isDigitByte(toks[0].Literal[0]) {
82
			return cmplHeaderCtx(content, cursor, toks)
83
		}
84
		return cmplNone, cursor
85
	case token.ACCOUNT, token.COMMODITY, token.PAYEE, token.TAG:
86
		return cmplDirectiveContext(cursor, lineStart, toks)
87
	case token.P:
88
		return cmplPriceContext(cursor, toks)
89
	case token.TEXT:
90
		return cmplDirective, lineStart // half-typed keyword or unparseable line
91
	default:
92
		return cmplNone, cursor
93
	}
94
}
95
96
func cmplPostingCtx(content string, cursor int, toks []token.Token) (cmplCtx, int) {
97
	if inDirectiveBody(content, cursor) {
98
		return cmplNone, cursor
99
	}
100
	fieldStart := toks[0].Span.End.Offset
101
	i := 1
102
	for i < len(toks) {
103
		switch toks[i].Type {
104
		case token.STAR, token.BANG, token.LPAREN, token.LBRACKET, token.WHITESPACE:
105
			fieldStart = toks[i].Span.End.Offset
106
			i++
107
		default:
108
			goto run
109
		}
110
	}
111
run:
112
	// account run: consecutive account-name segments and colons
113
	fieldEnd := fieldStart
114
	for ; i < len(toks); i++ {
115
		if toks[i].Type != token.TEXT && toks[i].Type != token.COLON {
116
			break
117
		}
118
		fieldEnd = toks[i].Span.End.Offset
119
	}
120
	if cursor <= fieldEnd && cursor >= fieldStart {
121
		return cmplAccount, fieldStart
122
	}
123
	if cursor > fieldEnd {
124
		if t := tokenUnder(toks, cursor); t != nil && (t.Type == token.COMMODITYMARK || t.Type == token.STRING) {
125
			start := t.Span.Start.Offset
126
			if t.Type == token.STRING {
127
				start++ // skip opening quote
128
			}
129
			return cmplCommodity, start
130
		}
131
		if strings.TrimSpace(content[fieldEnd:cursor]) == "" {
132
			return cmplCommodity, cursor
133
		}
134
	}
135
	return cmplNone, cursor
136
}
137
138
func cmplHeaderCtx(content string, cursor int, toks []token.Token) (cmplCtx, int) {
139
	// cursor inside date token
140
	if cursor >= toks[0].Span.Start.Offset && cursor <= toks[0].Span.End.Offset {
141
		return cmplDate, toks[0].Span.Start.Offset
142
	}
143
144
	// skip date, status, code, and whitespace - where the payee beginds
145
	fieldStart := toks[0].Span.End.Offset
146
	fieldEnd := fieldStart
147
	seen := false
148
	for _, t := range toks[1:] {
149
		switch t.Type {
150
		case token.WHITESPACE, token.STAR, token.BANG, token.DATE, token.TIME,
151
			token.EQ, token.EQEQ, token.EQEQEQ:
152
			fieldStart = t.Span.End.Offset
153
			fieldEnd = t.Span.End.Offset
154
		case token.TEXT:
155
			lit := content[t.Span.Start.Offset:t.Span.End.Offset]
156
			if !seen && len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
157
				fieldStart = t.Span.End.Offset // parenthesized code
158
				fieldEnd = t.Span.End.Offset
159
				continue
160
			}
161
			if !seen {
162
				fieldStart = t.Span.Start.Offset
163
				seen = true
164
			}
165
			fieldEnd = t.Span.End.Offset
166
			if p := strings.IndexByte(lit, '|'); p >= 0 {
167
				fieldEnd = t.Span.Start.Offset + p // "payee|note" keeps the pipe in the token
168
				return payeeAt(cursor, fieldStart, fieldEnd)
169
			}
170
		case token.STRING:
171
			if !seen {
172
				fieldStart = t.Span.Start.Offset + 1 // skip opening quote
173
				seen = true
174
			}
175
			fieldEnd = t.Span.End.Offset
176
		default: // PIPE, SEMICOLON, ...
177
			return payeeAt(cursor, fieldStart, fieldEnd)
178
		}
179
	}
180
	if seen {
181
		return payeeAt(cursor, fieldStart, fieldEnd)
182
	}
183
	// no payee yet: the payee field is the whitespace after the header meta
184
	if cursor >= fieldStart && strings.TrimSpace(content[fieldStart:cursor]) == "" {
185
		return cmplPayee, cursor
186
	}
187
	return cmplNone, cursor
188
}
189
190
func payeeAt(cursor, start, end int) (cmplCtx, int) {
191
	if cursor >= start && cursor <= end {
192
		return cmplPayee, start
193
	}
194
	return cmplNone, cursor
195
}
196
197
// cmplDirectiveContext classifies a directive line. keyword completion before the keyword ends, symbol completion in the value field after
198
func cmplDirectiveContext(cursor, lineStart int, toks []token.Token) (cmplCtx, int) {
199
	kwEnd := toks[0].Span.End.Offset
200
	if cursor <= kwEnd {
201
		return cmplDirective, lineStart
202
	}
203
	start := kwEnd
204
	for i := 1; i < len(toks); i++ {
205
		if toks[i].Type == token.WHITESPACE || toks[i].Span.End.Offset <= kwEnd {
206
			continue
207
		}
208
		start = toks[i].Span.Start.Offset
209
		if toks[i].Type == token.STRING {
210
			start++ // skip opening quote
211
		}
212
		break
213
	}
214
	if start > cursor {
215
		start = cursor
216
	}
217
	switch toks[0].Type {
218
	case token.ACCOUNT:
219
		return cmplAccount, start
220
	case token.COMMODITY:
221
		return cmplCommodity, start
222
	case token.PAYEE:
223
		return cmplPayee, start
224
	case token.TAG:
225
		return cmplTagName, start
226
	}
227
	return cmplNone, cursor
228
}
229
230
func cmplPriceContext(cursor int, toks []token.Token) (cmplCtx, int) {
231
	kwEnd := toks[0].Span.End.Offset
232
	if cursor > toks[0].Span.Start.Offset && cursor < kwEnd {
233
		return cmplNone, cursor // on the P keyword
234
	}
235
236
	dateStart := kwEnd           // start of the date field, for replacing partials whole
237
	dateDone := false            // a DATE, TIME, or symbol has ended the date field
238
	for _, t := range toks[1:] { // skip the P keyword
239
		if t.Type == token.WHITESPACE {
240
			continue
241
		}
242
		if !dateDone && dateStart == kwEnd {
243
			dateStart = t.Span.Start.Offset
244
		}
245
		if cursor > t.Span.End.Offset {
246
			switch t.Type {
247
			case token.DATE, token.TIME, token.COMMODITYMARK, token.STRING:
248
				dateDone = true
249
			}
250
			continue
251
		}
252
		if cursor < t.Span.Start.Offset {
253
			if (t.Type == token.INT || t.Type == token.DECIMAL) && !dateDone {
254
				return cmplDate, dateStart // partial date fragments ahead
255
			}
256
			if t.Type == token.DATE || t.Type == token.TIME {
257
				return cmplNone, cursor
258
			}
259
			return cmplCommodity, cursor // symbol or price-commodity slot
260
		}
261
262
		switch t.Type {
263
		case token.DATE:
264
			return cmplDate, t.Span.Start.Offset
265
		case token.COMMODITYMARK:
266
			return cmplCommodity, t.Span.Start.Offset
267
		case token.STRING:
268
			return cmplCommodity, min(t.Span.Start.Offset+1, cursor) // skip the opening quote
269
		case token.TIME:
270
			return cmplNone, cursor
271
		case token.INT, token.DECIMAL: // a partial date fragment, or the amount quantity
272
			if !dateDone {
273
				return cmplDate, dateStart
274
			}
275
			if cursor == t.Span.End.Offset {
276
				return cmplCommodity, cursor // suffix price-commodity slot
277
			}
278
			return cmplNone, cursor
279
		}
280
	}
281
	// the cursor sits in whitespace after the last token, or only after the keyword
282
	if !dateDone {
283
		start := dateStart
284
		if start == kwEnd {
285
			start = cursor // `P `: the date comes next
286
		}
287
		return cmplDate, start
288
	}
289
	return cmplCommodity, cursor
290
}
291
292
// commentStart completes tag names before ':' of the current tag and tag values after it
293
func cmplTagContext(content string, commentStart, cursor int) (cmplCtx, int) {
294
	prefix := content[commentStart:cursor]
295
	segStart := commentStart
296
	seg := prefix
297
	if comma := strings.LastIndexByte(prefix, ','); comma >= 0 {
298
		segStart = commentStart + comma + 1
299
		seg = prefix[comma+1:]
300
	}
301
	if colon := strings.IndexByte(seg, ':'); colon >= 0 {
302
		start := segStart + colon + 1
303
		for start < cursor && (content[start] == ' ' || content[start] == '\t') {
304
			start++
305
		}
306
		return cmplTagValue, start
307
	}
308
	keyStart := commentStart + lastSeparator(prefix) + 1
309
	return cmplTagName, keyStart
310
}
311
312
// tagKeyAt returns the key of tag whose value region starts at start
313
func tagKeyAt(content string, start int) (string, bool) {
314
	lineStart, _ := lineBounds(content, start)
315
	segStart := lineStart
316
	for i := start - 1; i >= lineStart; i-- {
317
		switch content[i] {
318
		case ',', ';', '#', '%':
319
			segStart = i + 1
320
			i = lineStart - 1 // stop at the separator closest to start
321
		}
322
	}
323
	colon := strings.IndexByte(content[segStart:start], ':')
324
	if colon < 0 {
325
		return "", false
326
	}
327
	colon += segStart
328
	keyStart := lastSeparator(content[segStart:colon]) + 1
329
	key := content[segStart+keyStart : colon]
330
	if key == "" {
331
		return "", false
332
	}
333
	return key, true
334
}
335
336
// commentMarker returns index of the first comment marker token at or before the cursor, or -1.
337
func commentMarker(toks []token.Token, cursor int) int {
338
	for i, t := range toks {
339
		switch t.Type {
340
		case token.SEMICOLON, token.HASH, token.PERCENT:
341
			if t.Span.Start.Offset <= cursor {
342
				return i
343
			}
344
		case token.STAR:
345
			if i == 0 && t.Span.Start.Offset <= cursor {
346
				return i
347
			}
348
		}
349
	}
350
	return -1
351
}
352
353
func inDirectiveBody(content string, cursor int) bool {
354
	lineStart, _ := lineBounds(content, cursor)
355
	if toks := lexLine(content, lineStart); len(toks) == 0 || toks[0].Type != token.INDENT {
356
		return false
357
	}
358
	for lineStart > 0 {
359
		lineStart, _ = lineBounds(content, lineStart-1)
360
		toks := lexLine(content, lineStart)
361
		if len(toks) == 0 || toks[0].Type != token.INDENT {
362
			return len(toks) > 0 && (toks[0].Type == token.ACCOUNT || toks[0].Type == token.COMMODITY)
363
		}
364
	}
365
	return false
366
}
367
368
type cmplCand struct {
369
	label        string
370
	score        float64
371
	count        int
372
	lastUsedDays int64 // days since 1970-01-01; 0 when unset
373
	rank         int   // lower sorts first among equal scores; 0 except for date completions
374
}
375
376
// cmplItems ranks candidates for the content against typed pattern
377
func cmplItems(
378
	a *analyzer.Analysis,
379
	ctx cmplCtx,
380
	content string,
381
	li *lsputil.LineIndex,
382
	start,
383
	cursor int,
384
	transliterate bool,
385
) []protocol.CompletionItem {
386
	pattern := content[start:cursor]
387
388
	var kind protocol.CompletionItemKind
389
	var cands []cmplCand
390
	switch ctx {
391
	case cmplAccount:
392
		kind = protocol.CompletionItemKindClass
393
		cands = make([]cmplCand, 0, len(a.Accounts))
394
		for name, info := range a.Accounts {
395
			cands = append(cands, cmplCand{label: name, count: info.UsedCount, lastUsedDays: dateToDays(info.LastUsed)})
396
		}
397
	case cmplPayee:
398
		kind = protocol.CompletionItemKindVariable
399
		cands = make([]cmplCand, 0, len(a.Payees))
400
		for name, info := range a.Payees {
401
			cands = append(cands, cmplCand{label: name, count: info.UsedCount, lastUsedDays: dateToDays(info.LastUsed)})
402
		}
403
	case cmplCommodity:
404
		kind = protocol.CompletionItemKindValue
405
		cands = make([]cmplCand, 0, len(a.Commodities))
406
		for name, info := range a.Commodities {
407
			cands = append(cands, cmplCand{label: name, count: info.UsedCount, lastUsedDays: dateToDays(info.LastUsed)})
408
		}
409
	case cmplTagName:
410
		kind = protocol.CompletionItemKindProperty
411
		cands = make([]cmplCand, 0, len(a.Tags))
412
		for name, info := range a.Tags {
413
			cands = append(cands, cmplCand{label: name, count: info.UsedCount, lastUsedDays: dateToDays(info.LastUsed)})
414
		}
415
	case cmplTagValue:
416
		kind = protocol.CompletionItemKindProperty
417
		if key, ok := tagKeyAt(content, start); ok {
418
			if info, ok := a.Tags[key]; ok {
419
				for _, v := range info.Values {
420
					cands = append(cands, cmplCand{label: v})
421
				}
422
			}
423
		}
424
	case cmplDirective:
425
		kind = protocol.CompletionItemKindKeyword
426
		for _, name := range directiveKeywords {
427
			cands = append(cands, cmplCand{label: name})
428
		}
429
	case cmplDate:
430
		kind = protocol.CompletionItemKindConstant
431
		now := time.Now()
432
		sep, hasYear := dateStyle(a.DateStrings)
433
		today := renderDate(now, sep, hasYear)
434
		yesterday := renderDate(now.AddDate(0, 0, -1), sep, hasYear)
435
		twoDaysAgo := renderDate(now.AddDate(0, 0, -2), sep, hasYear)
436
		cands = append(cands,
437
			cmplCand{label: today, rank: 0},
438
			cmplCand{label: yesterday, rank: 1},
439
			cmplCand{label: twoDaysAgo, rank: 2})
440
441
		seen := map[string]bool{today: true, yesterday: true, twoDaysAgo: true}
442
		for i, d := range a.Dates {
443
			if !datePatternMatch(pattern, a.DateStrings[i]) {
444
				continue
445
			}
446
			label := a.DateStrings[i]
447
			if seen[label] {
448
				continue
449
			}
450
			seen[label] = true
451
			cands = append(cands, cmplCand{label: label, rank: 3, lastUsedDays: dateToDays(d)})
452
		}
453
	default:
454
		return nil
455
	}
456
457
	var newest int64 // days since epoch of the newest transaction; 0 when none
458
	if n := len(a.Dates); n > 0 {
459
		newest = dateToDays(a.Dates[n-1])
460
	}
461
462
	matcher := fuzzy.Compile(pattern)
463
	translMatcher, hasTransl := fuzzy.Matcher{}, false
464
	if transliterate {
465
		if t := latinToCyrillic(pattern); t != pattern {
466
			translMatcher, hasTransl = fuzzy.Compile(t), true
467
		}
468
	}
469
	ranked := cands[:0]
470
	for i := range cands {
471
		sc := matcher.Score(cands[i].label)
472
		if hasTransl {
473
			sc = max(sc, translMatcher.Score(cands[i].label))
474
		}
475
		if sc != 0 {
476
			sc *= 1 + math.Log1p(float64(cands[i].count))
477
			if cands[i].count > 0 && cands[i].lastUsedDays != 0 && newest != 0 {
478
				days := newest - cands[i].lastUsedDays
479
				sc *= 1 + 0.5*max(0, 1-float64(days)/365)
480
			}
481
		}
482
		cands[i].score = sc
483
		if sc != 0 {
484
			ranked = append(ranked, cands[i])
485
		}
486
	}
487
	sort.Slice(ranked, func(i, j int) bool {
488
		if ranked[i].score != ranked[j].score {
489
			return ranked[i].score > ranked[j].score
490
		}
491
		if ranked[i].rank != ranked[j].rank {
492
			return ranked[i].rank < ranked[j].rank
493
		}
494
		if ranked[i].lastUsedDays != ranked[j].lastUsedDays {
495
			return ranked[i].lastUsedDays > ranked[j].lastUsedDays
496
		}
497
		return ranked[i].label < ranked[j].label
498
	})
499
	if len(ranked) > maxCompletionItems {
500
		ranked = ranked[:maxCompletionItems]
501
	}
502
503
	end := cursor
504
	if ctx == cmplDate {
505
		end = dateTokenEnd(content, start)
506
	}
507
	items := make([]protocol.CompletionItem, len(ranked))
508
	for i, r := range ranked {
509
		// nvim re-filters against the typed prefix, which a transliterated label never starts with; present the typed pattern as filterText.
510
		filterText := r.label
511
		if hasTransl && translMatcher.Score(r.label) > 0 {
512
			filterText = pattern
513
		}
514
		it := protocol.CompletionItem{
515
			Label:      r.label,
516
			Kind:       kind,
517
			SortText:   protocol.NewOptional(fmt.Sprintf("%04d", i)),
518
			FilterText: protocol.NewOptional(filterText),
519
			TextEdit: &protocol.TextEdit{
520
				Range: protocol.Range{
521
					Start: li.Position(start),
522
					End:   li.Position(end),
523
				},
524
				NewText: r.label,
525
			},
526
		}
527
		if r.count > 0 {
528
			it.Detail = protocol.NewOptional(fmt.Sprintf("%d uses", r.count))
529
		}
530
		items[i] = it
531
	}
532
	return items
533
}
534
535
// dateToDays converts a date to days since 1970-01-01; zero dates map to 0.
536
func dateToDays(d ast.Date) int64 {
537
	if d.Year == 0 {
538
		return 0
539
	}
540
	return daysFromCivil(d.Year, d.Month, d.Day)
541
}
542
543
// daysFromCivil converts a proleptic Gregorian date to days since 1970-01-01
544
// (Howard Hinnant's algorithm).
545
func daysFromCivil(y, m, d int) int64 {
546
	if m <= 2 {
547
		y--
548
	}
549
	era := y / 400
550
	yoe := y - era*400
551
	mp := (m + 9) % 12
552
	doy := (153*mp+2)/5 + d - 1
553
	doe := yoe*365 + yoe/4 - yoe/100 + doy
554
	return int64(era)*146097 + int64(doe) - 719468
555
}
556
557
// dateStyle returns reparator and yesr-ness of the most recent history date.
558
// Defualts to '-'/true when history is empty.
559
func dateStyle(history []string) (sep byte, hasYear bool) {
560
	sep, hasYear = '-', true
561
	if n := len(history); n > 0 {
562
		if i := strings.IndexAny(history[n-1], "-/."); i >= 0 {
563
			return history[n-1][i], i == 4
564
		}
565
	}
566
	return
567
}
568
569
func renderDate(t time.Time, sep byte, hasYear bool) string {
570
	d := ast.Date{Month: int(t.Month()), Day: t.Day(), Sep: sep}
571
	if hasYear {
572
		d.Year = t.Year()
573
	}
574
	return d.String()
575
}
576
577
func datePatternMatch(pattern, canonical string) bool {
578
	pi, ci := 0, 0
579
	for pi < len(pattern) && ci < len(canonical) {
580
		if pattern[pi] != canonical[ci] {
581
			ci++
582
			continue
583
		}
584
		pi++
585
		ci++
586
	}
587
	return pi == len(pattern)
588
}
589
590
func isDateSepByte(b byte) bool { return b == '-' || b == '/' || b == '.' }
591
func isDigitByte(b byte) bool   { return b >= '0' && b <= '9' }
592
593
func dateTokenEnd(content string, start int) int {
594
	i := start
595
	for i < len(content) && (isDigitByte(content[i]) || isDateSepByte(content[i])) {
596
		i++
597
	}
598
	return i
599
}
600
601
// lineBounds returns the byte offsets of the line containing cursor
602
func lineBounds(content string, cursor int) (start, end int) {
603
	start = cursor
604
	for start > 0 && content[start-1] != '\n' && content[start-1] != '\r' {
605
		start--
606
	}
607
	end = start
608
	for end < len(content) && content[end] != '\n' && content[end] != '\r' {
609
		end++
610
	}
611
	return start, end
612
}
613
614
func lastSeparator(s string) int {
615
	for i := len(s) - 1; i >= 0; i-- {
616
		switch s[i] {
617
		case ' ', '\t', ',':
618
			return i
619
		}
620
	}
621
	return -1
622
}
623
624
func lexLine(content string, cursor int) []token.Token {
625
	lineStart, lineEnd := lineBounds(content, cursor)
626
	l := lexer.New("", []byte(content[lineStart:lineEnd]))
627
	var out []token.Token
628
	for {
629
		t := l.Next()
630
		if t.Type == token.EOF || t.Type == token.NEWLINE {
631
			break
632
		}
633
		t.Span.Start.Offset += lineStart
634
		t.Span.End.Offset += lineStart
635
		out = append(out, t)
636
	}
637
	return out
638
}
639
640
func tokenUnder(toks []token.Token, cursor int) *token.Token {
641
	for i := range toks {
642
		t := &toks[i]
643
		if t.Span.Start.Offset <= cursor && cursor <= t.Span.End.Offset {
644
			return t
645
		}
646
	}
647
	return nil
648
}