all repos

clerk @ 4047965bca274b556119f4bdb74e9b725a7dfd35

missing tooling for ledger/hledger

clerk/internal/lsp/textdocument_semantic_tokens.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
lsp: implement textDocument/semanticTokens/delta, 1 month ago
1
package lsp
2
3
import (
4
	"context"
5
	"slices"
6
	"unicode/utf8"
7
8
	"go.lsp.dev/protocol"
9
	"go.lsp.dev/uri"
10
11
	"olexsmir.xyz/clerk/internal/lsp/lsputil"
12
	"olexsmir.xyz/clerk/journal/ast"
13
	"olexsmir.xyz/clerk/journal/lexer"
14
	"olexsmir.xyz/clerk/journal/token"
15
)
16
17
func (s *server) SemanticTokensFull(ctx context.Context, params *protocol.SemanticTokensParams) (*protocol.SemanticTokens, error) {
18
	if !s.semanticHighlightingEnabled() {
19
		return &protocol.SemanticTokens{}, nil
20
	}
21
	return s.semanticTokensFullResult(params.TextDocument.URI), nil
22
}
23
24
func (s *server) SemanticTokensFullDelta(ctx context.Context, params *protocol.SemanticTokensDeltaParams) (protocol.SemanticTokensDeltaResult, error) {
25
	if !s.semanticHighlightingEnabled() {
26
		return &protocol.SemanticTokens{}, nil
27
	}
28
29
	u := params.TextDocument.URI
30
	s.mu.RLock()
31
	st, ok := s.openDocs[u]
32
	s.mu.RUnlock()
33
	if !ok || st.semGen == 0 || params.PreviousResultID != st.resultID() {
34
		return s.semanticTokensFullResult(u), nil
35
	}
36
37
	data, ok := s.semanticTokensData(u)
38
	if !ok {
39
		return &protocol.SemanticTokens{}, nil
40
	}
41
	edits := semanticTokensEdits(st.semBaseline, data)
42
	if len(edits) == 0 {
43
		return &protocol.SemanticTokensDelta{ResultID: new(st.resultID()), Edits: []protocol.SemanticTokensEdit{}}, nil
44
	}
45
46
	rid, ok := s.storeSemResult(u, data)
47
	if !ok {
48
		return &protocol.SemanticTokens{Data: data}, nil
49
	}
50
	return &protocol.SemanticTokensDelta{ResultID: &rid, Edits: edits}, nil
51
}
52
53
func (s *server) SemanticTokensRange(ctx context.Context, params *protocol.SemanticTokensRangeParams) (*protocol.SemanticTokens, error) {
54
	if !s.semanticHighlightingEnabled() {
55
		return &protocol.SemanticTokens{}, nil
56
	}
57
58
	tokens, ok := s.tokensForDoc(params.TextDocument.URI)
59
	if !ok {
60
		return &protocol.SemanticTokens{}, nil
61
	}
62
	start := int(params.Range.Start.Line)
63
	end := int(params.Range.End.Line)
64
	var filtered []semanticToken
65
	for _, t := range tokens {
66
		if t.line >= uint32(start) && t.line <= uint32(end) {
67
			filtered = append(filtered, t)
68
		}
69
	}
70
	return &protocol.SemanticTokens{Data: encodeSemTokens(filtered)}, nil
71
}
72
73
func (s *server) semanticTokensFullResult(u uri.URI) *protocol.SemanticTokens {
74
	data, ok := s.semanticTokensData(u)
75
	if !ok {
76
		return &protocol.SemanticTokens{}
77
	}
78
	res := &protocol.SemanticTokens{Data: data}
79
	if rid, ok := s.storeSemResult(u, data); ok {
80
		res.ResultID = &rid
81
	}
82
	return res
83
}
84
85
func (s *server) semanticTokensData(u uri.URI) ([]uint32, bool) {
86
	tokens, ok := s.tokensForDoc(u)
87
	if !ok {
88
		return nil, false
89
	}
90
	return encodeSemTokens(tokens), true
91
}
92
93
func (s *server) storeSemResult(u uri.URI, data []uint32) (string, bool) {
94
	s.mu.Lock()
95
	defer s.mu.Unlock()
96
	st, ok := s.openDocs[u]
97
	if !ok {
98
		return "", false
99
	}
100
	rid := st.nextResultID()
101
	st.semBaseline = data
102
	s.openDocs[u] = st
103
	return rid, true
104
}
105
106
func (s *server) tokensForDoc(doc uri.URI) ([]semanticToken, bool) {
107
	s.mu.RLock()
108
	st, ok := s.openDocs[doc]
109
	s.mu.RUnlock()
110
	if !ok {
111
		return nil, false
112
	}
113
	if st.semTokens != nil {
114
		return st.semTokens, true
115
	}
116
	// Tokenize outside the lock: a full tokenization of a large journal is
117
	// milliseconds, during which didChange/didOpen would otherwise stall.
118
	tokens := tokenizeForSemantics(st.text, parseJournalStr(st.text))
119
	s.mu.Lock()
120
	defer s.mu.Unlock()
121
	cur, ok := s.openDocs[doc]
122
	if !ok {
123
		return nil, false
124
	}
125
	if cur.text == st.text { // unchanged during tokenization
126
		cur.semTokens = tokens
127
		s.openDocs[doc] = cur
128
	}
129
	return tokens, true
130
}
131
132
const (
133
	semDirective uint32 = iota
134
	semDate
135
	semAccount
136
	semCommodity
137
	semAmount
138
	semStatus
139
	semComment
140
	semString
141
	semOperator
142
	semProperty
143
)
144
145
var tokenTypeStrings = []string{
146
	string(protocol.SemanticTokenTypesKeyword),   // directive
147
	string(protocol.SemanticTokenTypesClass),     // date
148
	string(protocol.SemanticTokenTypesNamespace), // account
149
	string(protocol.SemanticTokenTypesType),      // commodity
150
	string(protocol.SemanticTokenTypesNumber),    // amount
151
	string(protocol.SemanticTokenTypesOperator),  // status
152
	string(protocol.SemanticTokenTypesComment),   // comment
153
	string(protocol.SemanticTokenTypesString),    // string
154
	string(protocol.SemanticTokenTypesOperator),  // operator
155
	string(protocol.SemanticTokenTypesProperty),  // property
156
}
157
158
const (
159
	modifierAbstract = 1 << 0 // virtual account
160
	modifierNegative = 1 << 1 // negative amount
161
)
162
163
var modifierStrings = []string{
164
	"abstract", // bit 0
165
	"negative", // bit 1
166
}
167
168
func getSemanticTokensLegend() protocol.SemanticTokensLegend {
169
	return protocol.SemanticTokensLegend{
170
		TokenTypes:     tokenTypeStrings,
171
		TokenModifiers: modifierStrings,
172
	}
173
}
174
175
type semanticToken struct {
176
	line, col uint32 // 0-based
177
	length    uint32
178
	tokenType uint32
179
	modifiers uint32
180
}
181
182
func tokenizeForSemantics(content string, j *ast.Journal) []semanticToken {
183
	var raw []rawSpan
184
	emit := func(s token.Span, tokType, mods uint32) {
185
		if s.Start.Offset >= s.End.Offset {
186
			return
187
		}
188
		raw = append(raw, rawSpan{s, tokType, mods})
189
	}
190
	for _, e := range j.Entries {
191
		visitEntry(content, e, emit)
192
	}
193
	if len(j.Errors) > 0 {
194
		// parser recovers per line; lexer fills the unparsed regions, keeping ast tokens where the parser succeeded
195
		semLexerFallback(content, raw, emit)
196
	}
197
	return rawToSemanticTokens(content, raw)
198
}
199
200
// rawSpan is a source span tagged with semantic token
201
type rawSpan struct {
202
	span      token.Span
203
	tok, mods uint32
204
}
205
206
func rawToSemanticTokens(content string, raw []rawSpan) []semanticToken {
207
	if len(raw) == 0 {
208
		return nil
209
	}
210
	slices.SortFunc(raw, func(a, b rawSpan) int { return a.span.Start.Offset - b.span.Start.Offset })
211
	out := make([]semanticToken, len(raw))
212
	line, col, cursor := 0, 0, 0
213
	advance := func(end int) {
214
		for cursor < end {
215
			r, size := utf8.DecodeRuneInString(content[cursor:])
216
			if r == utf8.RuneError && size <= 1 {
217
				break
218
			}
219
			if r == '\r' {
220
				cursor += size
221
				if cursor < len(content) && content[cursor] == '\n' {
222
					cursor++
223
				}
224
				line++
225
				col = 0
226
				continue
227
			}
228
			if r == '\n' {
229
				cursor += size
230
				line++
231
				col = 0
232
				continue
233
			}
234
			cursor += size
235
			col += utf16Units(r)
236
		}
237
	}
238
	for i, t := range raw {
239
		if cursor < t.span.Start.Offset {
240
			advance(t.span.Start.Offset)
241
		}
242
		out[i] = semanticToken{
243
			line:      uint32(line),
244
			col:       uint32(col),
245
			length:    uint32(lsputil.Utf16Len(content, t.span.Start.Offset, t.span.End.Offset)),
246
			tokenType: t.tok,
247
			modifiers: t.mods,
248
		}
249
		advance(t.span.End.Offset)
250
	}
251
	return out
252
}
253
254
func utf16Units(r rune) int {
255
	if r >= 0x10000 && r <= 0x10FFFF {
256
		return 2
257
	}
258
	return 1
259
}
260
261
func visitEntry(content string, e ast.Entry, emit semEmitFunc) {
262
	switch e := e.(type) {
263
	case *ast.Transaction:
264
		visitTransaction(content, e, emit)
265
	case *ast.PeriodicTransaction:
266
		visitPeriodicTransaction(content, e, emit)
267
	case *ast.AutomatedTransaction:
268
		visitAutomatedTransaction(content, e, emit)
269
	case *ast.AccountDirective:
270
		emit(directiveKeyword(e.Span, "account"), semDirective, 0)
271
		emit(e.Account.Span, semAccount, 0)
272
		for _, sd := range e.Subdirectives {
273
			if sd.Kind == ast.SubdirectiveComment {
274
				emitComment(sd.Comment, emit)
275
				continue
276
			}
277
			emit(sd.NameSpan, semDirective, 0)
278
			switch sd.Kind {
279
			case ast.SubdirectiveAlias:
280
				emit(sd.ValueSpan, semAccount, 0)
281
			case ast.SubdirectiveType, ast.SubdirectiveNote:
282
				emit(sd.ValueSpan, semProperty, 0)
283
			}
284
			emitComment(sd.Comment, emit)
285
		}
286
		emitComment(e.Comment, emit)
287
	case *ast.CommodityDirective:
288
		emit(directiveKeyword(e.Span, "commodity"), semDirective, 0)
289
		if e.FormatSub != nil {
290
			if e.FormatSub.KeywordSpan.End.Offset > 0 {
291
				emit(e.FormatSub.KeywordSpan, semDirective, 0)
292
			}
293
			semEmitAmount(content, &e.FormatSub.Amount, emit)
294
			emitComment(e.FormatSub.Comment, emit)
295
		} else if e.CommoditySpan.Start.Offset > 0 && e.CommoditySpan.End.Offset > 0 {
296
			emit(e.CommoditySpan, semCommodity, 0)
297
		}
298
		emitBlockComments(e.BlockComments, emit)
299
		emitComment(e.Comment, emit)
300
	case *ast.IncludeDirective:
301
		emitDirective(content, e.Span, len("include"), semString, e.Comment, emit)
302
	case *ast.PayeeDirective:
303
		emitDirective(content, e.Span, len("payee"), semProperty, e.Comment, emit)
304
	case *ast.TagDirective:
305
		emitDirective(content, e.Span, len("tag"), semProperty, e.Comment, emit)
306
	case *ast.AliasDirective:
307
		emit(directiveKeyword(e.Span, "alias"), semDirective, 0)
308
		emit(e.From.Span, semAccount, 0)
309
		if op, ok := betweenSpan(content, e.Span.Start.File, e.From.Span.End.Offset, e.To.Span.Start.Offset); ok {
310
			emit(op, semOperator, 0)
311
		}
312
		emit(e.To.Span, semAccount, 0)
313
		emitComment(e.Comment, emit)
314
	case *ast.YearDirective:
315
		kwLen := len("year")
316
		if content[e.Span.Start.Offset] == 'Y' {
317
			kwLen = 1
318
		}
319
		emitDirective(content, e.Span, kwLen, semProperty, e.Comment, emit)
320
	case *ast.DecimalMarkDirective:
321
		emitDirective(content, e.Span, len("decimal-mark"), semProperty, e.Comment, emit)
322
	case *ast.DefaultCommodityDirective:
323
		emit(directiveKeyword(e.Span, "D"), semDirective, 0)
324
		semEmitAmount(content, &e.Amount, emit)
325
		emitComment(e.Comment, emit)
326
	case *ast.MarketPriceDirective:
327
		emit(directiveKeyword(e.Span, "P"), semDirective, 0)
328
		emit(e.DateTime.Date.Span, semDate, 0)
329
		if e.DateTime.Time != nil {
330
			emit(e.DateTime.Time.Span, semDate, 0)
331
		}
332
		// commodity: text between the date (or time) and the amount
333
		commStart := e.DateTime.Date.Span.End.Offset
334
		if e.DateTime.Time != nil {
335
			commStart = e.DateTime.Time.Span.End.Offset
336
		}
337
		if comm, ok := betweenSpan(content, e.Span.Start.File, commStart, e.Amount.Span.Start.Offset); ok {
338
			emit(comm, semCommodity, 0)
339
		}
340
		semEmitAmount(content, &e.Amount, emit)
341
		emitComment(e.Comment, emit)
342
	case *ast.ConversionDirective:
343
		emit(directiveKeyword(e.Span, "C"), semDirective, 0)
344
		semEmitAmount(content, &e.From, emit)
345
		// = operator: text between the two amounts
346
		if op, ok := betweenSpan(content, e.Span.Start.File, e.From.Span.End.Offset, e.To.Span.Start.Offset); ok {
347
			emit(op, semOperator, 0)
348
		}
349
		semEmitAmount(content, &e.To, emit)
350
		emitComment(e.Comment, emit)
351
	case *ast.Comment:
352
		emitComment(e, emit)
353
	case *ast.CommentBlockDirective:
354
		emit(e.Span, semComment, 0)
355
	case *ast.IgnoredDirective:
356
		emitDirective(content, e.Span, len("N"), semProperty, e.Comment, emit)
357
	case *ast.ApplyDirective:
358
		emitDirective(content, e.Span, len("apply"), semProperty, e.Comment, emit)
359
	case *ast.EndDirective:
360
		emitDirective(content, e.Span, len("end"), semProperty, e.Comment, emit)
361
	case *ast.BlankLine:
362
	}
363
}
364
365
func visitTransaction(content string, t *ast.Transaction, emit semEmitFunc) {
366
	emit(t.Date.Span, semDate, 0)
367
	if t.SecondDate != nil {
368
		emit(t.SecondDate.Span, semDate, 0)
369
	}
370
	if t.Status.Value != ast.StatusNone {
371
		emit(t.Status.Span, semStatus, 0)
372
	}
373
	if t.Code != nil {
374
		emit(t.Code.Span, semString, 0)
375
	}
376
	if t.Payee != nil {
377
		emit(t.Payee.Span, semProperty, 0)
378
	}
379
	if t.Note != nil {
380
		emit(t.Note.Span, semProperty, 0)
381
	}
382
	emitComment(t.Comment, emit)
383
	for i := range t.HeaderComments {
384
		emitComment(t.HeaderComments[i], emit)
385
	}
386
	for _, p := range t.Postings {
387
		visitPosting(content, p, emit)
388
	}
389
}
390
391
func visitPeriodicTransaction(content string, pt *ast.PeriodicTransaction, emit semEmitFunc) {
392
	// ~ operator is at the start of the period span
393
	emit(offsetSpan(pt.Span.Start.File, pt.Span.Start.Offset, pt.Span.Start.Offset+1), semOperator, 0)
394
395
	// The period span covers the whole expr, including any "from ... to ..." dates
396
	if pt.Period.Span.End.Offset > pt.Period.Span.Start.Offset {
397
		var dates []*ast.Date
398
		if pt.Period.From != nil {
399
			dates = append(dates, pt.Period.From)
400
		}
401
		if pt.Period.To != nil {
402
			dates = append(dates, pt.Period.To)
403
		}
404
		pos := pt.Period.Span.Start.Offset
405
		for _, d := range dates {
406
			if d.Span.Start.Offset > pos {
407
				emit(offsetSpan(pt.Period.Span.Start.File, pos, d.Span.Start.Offset), semProperty, 0)
408
			}
409
			emit(d.Span, semDate, 0)
410
			pos = d.Span.End.Offset
411
		}
412
		if pos < pt.Period.Span.End.Offset {
413
			emit(offsetSpan(pt.Period.Span.Start.File, pos, pt.Period.Span.End.Offset), semProperty, 0)
414
		}
415
	}
416
	if pt.Description != nil {
417
		emit(pt.Description.Span, semProperty, 0)
418
	}
419
	emitComment(pt.Comment, emit)
420
	for i := range pt.HeaderComments {
421
		emitComment(pt.HeaderComments[i], emit)
422
	}
423
	for _, p := range pt.Postings {
424
		visitPosting(content, p, emit)
425
	}
426
}
427
428
func visitAutomatedTransaction(content string, at *ast.AutomatedTransaction, emit semEmitFunc) {
429
	// = operator is at the start of the expression span
430
	emit(offsetSpan(at.Span.Start.File, at.Span.Start.Offset, at.Span.Start.Offset+1), semOperator, 0)
431
432
	if at.Expr.Value != "" {
433
		emit(at.Expr.Span, semString, 0)
434
	}
435
	emitComment(at.Comment, emit)
436
	for i := range at.HeaderComments {
437
		emitComment(at.HeaderComments[i], emit)
438
	}
439
	for _, p := range at.Postings {
440
		visitPosting(content, p, emit)
441
	}
442
}
443
444
func visitPosting(content string, p *ast.Posting, emit semEmitFunc) {
445
	if p.Status.Value != ast.StatusNone {
446
		emit(p.Status.Span, semStatus, 0)
447
	}
448
449
	// virtual brackets
450
	if p.Type == ast.PostingVirtualUnbalanced || p.Type == ast.PostingVirtualBalanced {
451
		// opening bracket
452
		for off := p.Span.Start.Offset; off < p.Account.Span.Start.Offset && off < p.Span.End.Offset; off++ {
453
			if content[off] == '(' || content[off] == '[' {
454
				brSpan := token.Span{Start: offsetPos(p.Span.Start.File, off), End: offsetPos(p.Span.Start.File, off+1)}
455
				emit(brSpan, semOperator, modifierAbstract)
456
				break
457
			}
458
		}
459
		// closing bracket
460
		for off := p.Account.Span.End.Offset; off < p.Span.End.Offset; off++ {
461
			if content[off] == ')' || content[off] == ']' {
462
				brSpan := token.Span{Start: offsetPos(p.Span.Start.File, off), End: offsetPos(p.Span.Start.File, off+1)}
463
				emit(brSpan, semOperator, modifierAbstract)
464
				break
465
			}
466
		}
467
	}
468
469
	emit(p.Account.Span, semAccount, 0)
470
471
	if p.Amount != nil {
472
		semEmitAmount(content, p.Amount, emit)
473
	}
474
	if p.Cost != nil {
475
		semEmitCost(content, p.Cost, emit)
476
	}
477
	if p.Balance != nil {
478
		semEmitBalanceAssertion(content, p.Balance, emit)
479
	}
480
	emitComment(p.Comment, emit)
481
	for i := range p.Comments {
482
		emitComment(&p.Comments[i], emit)
483
	}
484
}
485
486
// directiveKeyword returns the span of the leading keyword on a directive line.
487
func directiveKeyword(e token.Span, kw string) token.Span {
488
	return token.Span{Start: e.Start, End: offsetPos(e.Start.File, e.Start.Offset+len(kw))}
489
}
490
491
// directiveValue returns the trimmed span of the text after the keyword end
492
// offset, up to the inline comment or the end of the line.
493
func directiveValue(content string, e token.Span, comment *ast.Comment, kwEnd int) (token.Span, bool) {
494
	end := e.End.Offset
495
	if comment != nil {
496
		end = comment.Span.Start.Offset
497
	}
498
	return betweenSpan(content, e.Start.File, kwEnd, end)
499
}
500
501
func semEmitAmount(content string, a *ast.Amount, emit semEmitFunc) {
502
	if a == nil {
503
		return
504
	}
505
	hasCommodity := a.CommoditySpan.Start.Offset > 0 && a.CommoditySpan.End.Offset > 0
506
	if hasCommodity && a.CommodityPos == ast.CommodityBefore {
507
		emit(a.CommoditySpan, semCommodity, 0)
508
		semEmitQuantity(content, a, emit)
509
		return
510
	}
511
	semEmitQuantity(content, a, emit)
512
	if hasCommodity {
513
		emit(a.CommoditySpan, semCommodity, 0)
514
	}
515
}
516
517
func semEmitQuantity(content string, a *ast.Amount, emit semEmitFunc) {
518
	qStart, qEnd := quantitySpan(content, a)
519
	if qEnd <= qStart {
520
		return
521
	}
522
	mods := uint32(0)
523
	if a.IsNegative {
524
		mods |= modifierNegative
525
	}
526
	emit(offsetSpan(a.Span.Start.File, qStart, qEnd), semAmount, mods)
527
}
528
529
func quantitySpan(content string, a *ast.Amount) (int, int) {
530
	start, end := a.Span.Start.Offset, a.Span.End.Offset
531
	switch {
532
	case a.Commodity == "":
533
		// bare quantity
534
	case a.CommodityPos == ast.CommodityBefore:
535
		// "$50.00" or "$   50.00": quantity follows the commodity span
536
		start = a.CommoditySpan.End.Offset
537
	default: // CommodityAfter
538
		// "50.00 USD" or "50.00     USD": quantity precedes the commodity
539
		end = a.CommoditySpan.Start.Offset
540
	}
541
	for start < end && (content[start] == ' ' || content[start] == '\t') {
542
		start++
543
	}
544
	for end > start && (content[end-1] == ' ' || content[end-1] == '\t') {
545
		end--
546
	}
547
	return start, end
548
}
549
550
type semEmitFunc func(tok token.Span, tokKind, modifier uint32)
551
552
func emitBlockComments(cs []*ast.Comment, emit semEmitFunc) {
553
	for _, c := range cs {
554
		emitComment(c, emit)
555
	}
556
}
557
558
func emitComment(c *ast.Comment, emit semEmitFunc) {
559
	if c == nil {
560
		return
561
	}
562
	if len(c.Tags) == 0 {
563
		emit(c.Span, semComment, 0)
564
		return
565
	}
566
	pos := c.Span.Start.Offset
567
	for _, t := range c.Tags {
568
		if t.Span.Start.Offset > pos {
569
			emit(offsetSpan(c.Span.Start.File, pos, t.Span.Start.Offset), semComment, 0)
570
		}
571
		emit(t.Span, semProperty, 0)
572
		pos = t.Span.End.Offset
573
	}
574
	if pos < c.Span.End.Offset {
575
		emit(offsetSpan(c.Span.Start.File, pos, c.Span.End.Offset), semComment, 0)
576
	}
577
}
578
579
func emitDirective(content string, e token.Span, kwLen int, valType uint32, comment *ast.Comment, emit semEmitFunc) {
580
	kwEnd := e.Start.Offset + kwLen
581
	emit(token.Span{Start: e.Start, End: offsetPos(e.Start.File, kwEnd)}, semDirective, 0)
582
	if v, ok := directiveValue(content, e, comment, kwEnd); ok {
583
		emit(v, valType, 0)
584
	}
585
	emitComment(comment, emit)
586
}
587
588
func semEmitCost(content string, c *ast.Cost, emit semEmitFunc) {
589
	if c.IsTotal {
590
		emit(token.Span{Start: c.Span.Start, End: offsetPos(c.Span.Start.File, c.Span.Start.Offset+2)}, semOperator, 0)
591
	} else {
592
		emit(token.Span{Start: c.Span.Start, End: offsetPos(c.Span.Start.File, c.Span.Start.Offset+1)}, semOperator, 0)
593
	}
594
	semEmitAmount(content, &c.Amount, emit)
595
}
596
597
func semEmitBalanceAssertion(content string, ba *ast.BalanceAssertion, emit semEmitFunc) {
598
	// The operator is the run of '=', ':', '*' chars from the span start.
599
	// The ':' of ':=' precedes the '=' token, so back up one offset.
600
	opStart := ba.Span.Start.Offset
601
	if ba.IsAssignment && opStart > 0 {
602
		opStart--
603
	}
604
	opEnd := opStart
605
	for opEnd < ba.Span.End.Offset && (content[opEnd] == '=' || content[opEnd] == ':' || content[opEnd] == '*') {
606
		opEnd++
607
	}
608
	emit(token.Span{Start: offsetPos(ba.Span.Start.File, opStart), End: offsetPos(ba.Span.Start.File, opEnd)}, semOperator, 0)
609
	semEmitAmount(content, &ba.Amount, emit)
610
	if ba.Cost != nil {
611
		semEmitCost(content, ba.Cost, emit)
612
	}
613
}
614
615
func semLexerFallback(content string, base []rawSpan, emit semEmitFunc) {
616
	l := lexer.New("", []byte(content))
617
618
	var commentStart, commentEnd int // 0 = not inside a comment line
619
	lineStart := true                // the next significant token starts a line
620
	skipLine := false                // the line starts with an unclassifiable token; emit nothing
621
	i := 0                           // next base span to compare against
622
623
	take := func(span token.Span, tokType uint32, mods uint32) {
624
		for i < len(base) && base[i].span.End.Offset <= span.Start.Offset {
625
			i++
626
		}
627
		if i < len(base) && base[i].span.Start.Offset < span.End.Offset {
628
			return // overlaps an AST token; AST wins
629
		}
630
		emit(span, tokType, mods)
631
	}
632
633
	for {
634
		tok := l.Next()
635
		if tok.Type == token.EOF {
636
			if commentStart > 0 {
637
				take(token.Span{Start: offsetPos("", commentStart), End: offsetPos("", commentEnd)}, semComment, 0)
638
			}
639
			break
640
		}
641
		if tok.Type == token.NEWLINE {
642
			if commentStart > 0 {
643
				take(token.Span{Start: offsetPos("", commentStart), End: offsetPos("", commentEnd)}, semComment, 0)
644
				commentStart, commentEnd = 0, 0
645
			}
646
			lineStart, skipLine = true, false
647
			continue
648
		}
649
		if tok.Type == token.WHITESPACE || tok.Type == token.INDENT {
650
			continue
651
		}
652
		if lineStart {
653
			lineStart = false
654
			if !isLineStartToken(tok.Type) {
655
				skipLine = true
656
			}
657
		}
658
		if skipLine {
659
			continue
660
		}
661
		tokType := semProperty
662
		if commentStart > 0 {
663
			tokType = semComment
664
			if tok.Span.End.Offset > commentEnd {
665
				commentEnd = tok.Span.End.Offset
666
			}
667
			continue
668
		}
669
670
		switch tok.Type {
671
		case token.SEMICOLON, token.HASH, token.PERCENT, token.STAR:
672
			tokType = semComment
673
			commentStart = tok.Span.Start.Offset
674
			commentEnd = tok.Span.End.Offset
675
			continue
676
		case token.STRING:
677
			tokType = semString
678
		case token.DATE:
679
			tokType = semDate
680
		case token.INT, token.DECIMAL:
681
			tokType = semAmount
682
		case token.COMMODITYMARK:
683
			tokType = semCommodity
684
		case token.BANG:
685
			tokType = semStatus
686
		case token.AT, token.ATAT, token.EQ, token.EQEQ, token.EQEQEQ, token.EQSTAR:
687
			tokType = semOperator
688
		case token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE,
689
			token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END,
690
			token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C:
691
			tokType = semDirective
692
		}
693
		take(tok.Span, tokType, 0)
694
	}
695
}
696
697
func isLineStartToken(t token.Type) bool {
698
	switch t {
699
	case token.DATE, token.TILDE, token.EQ, token.BANG, token.AT,
700
		token.SEMICOLON, token.HASH, token.PERCENT, token.STAR,
701
		token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE,
702
		token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END,
703
		token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C:
704
		return true
705
	}
706
	return false
707
}
708
709
// semanticTokensEdits returns the single edit turning old into new, or nil when
710
// identical. Relative delta encoding keeps the common prefix and suffix unchanged.
711
func semanticTokensEdits(old, new []uint32) []protocol.SemanticTokensEdit {
712
	p := 0
713
	for p < len(old) && p < len(new) && old[p] == new[p] {
714
		p++
715
	}
716
	s := 0
717
	for s < len(old)-p && s < len(new)-p && old[len(old)-1-s] == new[len(new)-1-s] {
718
		s++
719
	}
720
	delCount := len(old) - p - s
721
	ins := new[p : len(new)-s]
722
	if delCount == 0 && len(ins) == 0 {
723
		return nil
724
	}
725
	return []protocol.SemanticTokensEdit{{
726
		Start:       uint32(p),
727
		DeleteCount: uint32(delCount),
728
		Data:        ins,
729
	}}
730
}
731
732
// encodeSemTokens encodes tokens into LSP delta form. Input must be sorted by
733
// line and column; [rawToSemanticTokens] produces such order.
734
func encodeSemTokens(tokens []semanticToken) []uint32 {
735
	if len(tokens) == 0 {
736
		return nil
737
	}
738
	data := make([]uint32, 0, len(tokens)*5)
739
	var prevLine, prevCol uint32
740
	for _, t := range tokens {
741
		var deltaLine, deltaCol uint32
742
		if t.line == prevLine {
743
			deltaLine = 0
744
			deltaCol = t.col - prevCol
745
		} else {
746
			deltaLine = t.line - prevLine
747
			deltaCol = t.col
748
		}
749
		data = append(data, deltaLine, deltaCol, t.length, t.tokenType, t.modifiers)
750
		prevLine = t.line
751
		prevCol = t.col
752
	}
753
	return data
754
}
755
756
// betweenSpan returns the span of the text between two offsets, trimmed of surrounding whitespace.
757
func betweenSpan(content, file string, start, end int) (token.Span, bool) {
758
	for start < end && (content[start] == ' ' || content[start] == '\t') {
759
		start++
760
	}
761
	for end > start && (content[end-1] == ' ' || content[end-1] == '\t' || content[end-1] == '\n' || content[end-1] == '\r') {
762
		end--
763
	}
764
	if end <= start {
765
		return token.Span{}, false
766
	}
767
	return token.Span{Start: offsetPos(file, start), End: offsetPos(file, end)}, true
768
}
769
770
func offsetPos(file string, offset int) token.Pos { return token.Pos{File: file, Offset: offset} }
771
func offsetSpan(file string, start, end int) token.Span {
772
	return token.Span{Start: offsetPos(file, start), End: offsetPos(file, end)}
773
}