all repos

clerk @ 7b8e468208b6ef8bf9db2b5f2f8c1f01b22cb31a

missing tooling for ledger/hledger

clerk/internal/lsp/textdocument_semantic_tokens.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
lsp: parse semantic tokens from loader to cache results, 1 month ago
1
package lsp
2
3
import (
4
	"context"
5
	"slices"
6
	"unicode/utf8"
7
8
	"go.lsp.dev/protocol"
9
	"go.lsp.dev/uri"
10
11
	"olexsmir.xyz/clerk/internal/lsp/lsputil"
12
	"olexsmir.xyz/clerk/journal/ast"
13
	"olexsmir.xyz/clerk/journal/lexer"
14
	"olexsmir.xyz/clerk/journal/token"
15
)
16
17
func (s *server) SemanticTokensFull(ctx context.Context, params *protocol.SemanticTokensParams) (*protocol.SemanticTokens, error) {
18
	if !s.semanticHighlightingEnabled() {
19
		return &protocol.SemanticTokens{}, nil
20
	}
21
	return s.semanticTokensFullResult(params.TextDocument.URI), nil
22
}
23
24
func (s *server) SemanticTokensFullDelta(ctx context.Context, params *protocol.SemanticTokensDeltaParams) (protocol.SemanticTokensDeltaResult, error) {
25
	if !s.semanticHighlightingEnabled() {
26
		return &protocol.SemanticTokens{}, nil
27
	}
28
29
	u := params.TextDocument.URI
30
	s.mu.RLock()
31
	st, ok := s.openDocs[u]
32
	s.mu.RUnlock()
33
	if !ok || st.semGen == 0 || params.PreviousResultID != st.resultID() {
34
		return s.semanticTokensFullResult(u), nil
35
	}
36
37
	data, ok := s.semanticTokensData(u)
38
	if !ok {
39
		return &protocol.SemanticTokens{}, nil
40
	}
41
	edits := semanticTokensEdits(st.semBaseline, data)
42
	if len(edits) == 0 {
43
		return &protocol.SemanticTokensDelta{ResultID: new(st.resultID()), Edits: []protocol.SemanticTokensEdit{}}, nil
44
	}
45
46
	rid, ok := s.storeSemResult(u, data)
47
	if !ok {
48
		return &protocol.SemanticTokens{Data: data}, nil
49
	}
50
	return &protocol.SemanticTokensDelta{ResultID: &rid, Edits: edits}, nil
51
}
52
53
func (s *server) SemanticTokensRange(ctx context.Context, params *protocol.SemanticTokensRangeParams) (*protocol.SemanticTokens, error) {
54
	if !s.semanticHighlightingEnabled() {
55
		return &protocol.SemanticTokens{}, nil
56
	}
57
58
	tokens, ok := s.tokensForDoc(params.TextDocument.URI)
59
	if !ok {
60
		return &protocol.SemanticTokens{}, nil
61
	}
62
	start := int(params.Range.Start.Line)
63
	end := int(params.Range.End.Line)
64
	var filtered []semanticToken
65
	for _, t := range tokens {
66
		if t.line >= uint32(start) && t.line <= uint32(end) {
67
			filtered = append(filtered, t)
68
		}
69
	}
70
	return &protocol.SemanticTokens{Data: encodeSemTokens(filtered)}, nil
71
}
72
73
func (s *server) semanticTokensFullResult(u uri.URI) *protocol.SemanticTokens {
74
	data, ok := s.semanticTokensData(u)
75
	if !ok {
76
		return &protocol.SemanticTokens{}
77
	}
78
	res := &protocol.SemanticTokens{Data: data}
79
	if rid, ok := s.storeSemResult(u, data); ok {
80
		res.ResultID = &rid
81
	}
82
	return res
83
}
84
85
func (s *server) semanticTokensData(u uri.URI) ([]uint32, bool) {
86
	tokens, ok := s.tokensForDoc(u)
87
	if !ok {
88
		return nil, false
89
	}
90
	return encodeSemTokens(tokens), true
91
}
92
93
func (s *server) storeSemResult(u uri.URI, data []uint32) (string, bool) {
94
	s.mu.Lock()
95
	defer s.mu.Unlock()
96
	st, ok := s.openDocs[u]
97
	if !ok {
98
		return "", false
99
	}
100
	rid := st.nextResultID()
101
	st.semBaseline = data
102
	s.openDocs[u] = st
103
	return rid, true
104
}
105
106
func (s *server) tokensForDoc(doc uri.URI) ([]semanticToken, bool) {
107
	s.mu.RLock()
108
	st, ok := s.openDocs[doc]
109
	s.mu.RUnlock()
110
	if !ok {
111
		return nil, false
112
	}
113
	if st.semTokens != nil {
114
		return st.semTokens, true
115
	}
116
	// Tokenize outside the lock: a full tokenization of a large journal is
117
	// milliseconds, during which didChange/didOpen would otherwise stall.
118
	rj := s.loader.ResolveBytes(doc.Path(), []byte(st.text))
119
	tokens := tokenizeForSemantics(st.text, rj.Occurrences[0].Ast)
120
	s.mu.Lock()
121
	defer s.mu.Unlock()
122
	cur, ok := s.openDocs[doc]
123
	if !ok {
124
		return nil, false
125
	}
126
	if cur.text == st.text { // unchanged during tokenization
127
		cur.semTokens = tokens
128
		s.openDocs[doc] = cur
129
	}
130
	return tokens, true
131
}
132
133
const (
134
	semDirective uint32 = iota
135
	semDate
136
	semAccount
137
	semCommodity
138
	semAmount
139
	semStatus
140
	semComment
141
	semString
142
	semOperator
143
	semProperty
144
)
145
146
var tokenTypeStrings = []string{
147
	string(protocol.SemanticTokenTypesKeyword),   // directive
148
	string(protocol.SemanticTokenTypesClass),     // date
149
	string(protocol.SemanticTokenTypesNamespace), // account
150
	string(protocol.SemanticTokenTypesType),      // commodity
151
	string(protocol.SemanticTokenTypesNumber),    // amount
152
	string(protocol.SemanticTokenTypesOperator),  // status
153
	string(protocol.SemanticTokenTypesComment),   // comment
154
	string(protocol.SemanticTokenTypesString),    // string
155
	string(protocol.SemanticTokenTypesOperator),  // operator
156
	string(protocol.SemanticTokenTypesProperty),  // property
157
}
158
159
const (
160
	modifierAbstract = 1 << 0 // virtual account
161
	modifierNegative = 1 << 1 // negative amount
162
)
163
164
var modifierStrings = []string{
165
	"abstract", // bit 0
166
	"negative", // bit 1
167
}
168
169
func getSemanticTokensLegend() protocol.SemanticTokensLegend {
170
	return protocol.SemanticTokensLegend{
171
		TokenTypes:     tokenTypeStrings,
172
		TokenModifiers: modifierStrings,
173
	}
174
}
175
176
type semanticToken struct {
177
	line, col uint32 // 0-based
178
	length    uint32
179
	tokenType uint32
180
	modifiers uint32
181
}
182
183
func tokenizeForSemantics(content string, j *ast.Journal) []semanticToken {
184
	var raw []rawSpan
185
	emit := func(s token.Span, tokType, mods uint32) {
186
		if s.Start.Offset >= s.End.Offset {
187
			return
188
		}
189
		raw = append(raw, rawSpan{s, tokType, mods})
190
	}
191
	for _, e := range j.Entries {
192
		visitEntry(content, e, emit)
193
	}
194
	if len(j.Errors) > 0 {
195
		// parser recovers per line; lexer fills the unparsed regions, keeping ast tokens where the parser succeeded
196
		semLexerFallback(content, raw, emit)
197
	}
198
	return rawToSemanticTokens(content, raw)
199
}
200
201
// rawSpan is a source span tagged with semantic token
202
type rawSpan struct {
203
	span      token.Span
204
	tok, mods uint32
205
}
206
207
func rawToSemanticTokens(content string, raw []rawSpan) []semanticToken {
208
	if len(raw) == 0 {
209
		return nil
210
	}
211
	slices.SortFunc(raw, func(a, b rawSpan) int { return a.span.Start.Offset - b.span.Start.Offset })
212
	out := make([]semanticToken, len(raw))
213
	line, col, cursor := 0, 0, 0
214
	advance := func(end int) {
215
		for cursor < end {
216
			r, size := utf8.DecodeRuneInString(content[cursor:])
217
			if r == utf8.RuneError && size <= 1 {
218
				break
219
			}
220
			if r == '\r' {
221
				cursor += size
222
				if cursor < len(content) && content[cursor] == '\n' {
223
					cursor++
224
				}
225
				line++
226
				col = 0
227
				continue
228
			}
229
			if r == '\n' {
230
				cursor += size
231
				line++
232
				col = 0
233
				continue
234
			}
235
			cursor += size
236
			col += utf16Units(r)
237
		}
238
	}
239
	for i, t := range raw {
240
		if cursor < t.span.Start.Offset {
241
			advance(t.span.Start.Offset)
242
		}
243
		out[i] = semanticToken{
244
			line:      uint32(line),
245
			col:       uint32(col),
246
			length:    uint32(lsputil.Utf16Len(content, t.span.Start.Offset, t.span.End.Offset)),
247
			tokenType: t.tok,
248
			modifiers: t.mods,
249
		}
250
		advance(t.span.End.Offset)
251
	}
252
	return out
253
}
254
255
func utf16Units(r rune) int {
256
	if r >= 0x10000 && r <= 0x10FFFF {
257
		return 2
258
	}
259
	return 1
260
}
261
262
func visitEntry(content string, e ast.Entry, emit semEmitFunc) {
263
	switch e := e.(type) {
264
	case *ast.Transaction:
265
		visitTransaction(content, e, emit)
266
	case *ast.PeriodicTransaction:
267
		visitPeriodicTransaction(content, e, emit)
268
	case *ast.AutomatedTransaction:
269
		visitAutomatedTransaction(content, e, emit)
270
	case *ast.AccountDirective:
271
		emit(directiveKeyword(e.Span, "account"), semDirective, 0)
272
		emit(e.Account.Span, semAccount, 0)
273
		for _, sd := range e.Subdirectives {
274
			if sd.Kind == ast.SubdirectiveComment {
275
				emitComment(sd.Comment, emit)
276
				continue
277
			}
278
			emit(sd.NameSpan, semDirective, 0)
279
			switch sd.Kind {
280
			case ast.SubdirectiveAlias:
281
				emit(sd.ValueSpan, semAccount, 0)
282
			case ast.SubdirectiveType, ast.SubdirectiveNote:
283
				emit(sd.ValueSpan, semProperty, 0)
284
			}
285
			emitComment(sd.Comment, emit)
286
		}
287
		emitComment(e.Comment, emit)
288
	case *ast.CommodityDirective:
289
		emit(directiveKeyword(e.Span, "commodity"), semDirective, 0)
290
		if e.FormatSub != nil {
291
			if e.FormatSub.KeywordSpan.End.Offset > 0 {
292
				emit(e.FormatSub.KeywordSpan, semDirective, 0)
293
			}
294
			semEmitAmount(content, &e.FormatSub.Amount, emit)
295
			emitComment(e.FormatSub.Comment, emit)
296
		} else if e.CommoditySpan.Start.Offset > 0 && e.CommoditySpan.End.Offset > 0 {
297
			emit(e.CommoditySpan, semCommodity, 0)
298
		}
299
		emitBlockComments(e.BlockComments, emit)
300
		emitComment(e.Comment, emit)
301
	case *ast.IncludeDirective:
302
		emitDirective(content, e.Span, len("include"), semString, e.Comment, emit)
303
	case *ast.PayeeDirective:
304
		emitDirective(content, e.Span, len("payee"), semProperty, e.Comment, emit)
305
	case *ast.TagDirective:
306
		emitDirective(content, e.Span, len("tag"), semProperty, e.Comment, emit)
307
	case *ast.AliasDirective:
308
		emit(directiveKeyword(e.Span, "alias"), semDirective, 0)
309
		emit(e.From.Span, semAccount, 0)
310
		if op, ok := betweenSpan(content, e.Span.File, e.From.Span.End.Offset, e.To.Span.Start.Offset); ok {
311
			emit(op, semOperator, 0)
312
		}
313
		emit(e.To.Span, semAccount, 0)
314
		emitComment(e.Comment, emit)
315
	case *ast.YearDirective:
316
		kwLen := len("year")
317
		if content[e.Span.Start.Offset] == 'Y' {
318
			kwLen = 1
319
		}
320
		emitDirective(content, e.Span, kwLen, semProperty, e.Comment, emit)
321
	case *ast.DecimalMarkDirective:
322
		emitDirective(content, e.Span, len("decimal-mark"), semProperty, e.Comment, emit)
323
	case *ast.DefaultCommodityDirective:
324
		emit(directiveKeyword(e.Span, "D"), semDirective, 0)
325
		semEmitAmount(content, &e.Amount, emit)
326
		emitComment(e.Comment, emit)
327
	case *ast.MarketPriceDirective:
328
		emit(directiveKeyword(e.Span, "P"), semDirective, 0)
329
		emit(e.DateTime.Date.Span, semDate, 0)
330
		if e.DateTime.Time != nil {
331
			emit(e.DateTime.Time.Span, semDate, 0)
332
		}
333
		// commodity: text between the date (or time) and the amount
334
		commStart := e.DateTime.Date.Span.End.Offset
335
		if e.DateTime.Time != nil {
336
			commStart = e.DateTime.Time.Span.End.Offset
337
		}
338
		if comm, ok := betweenSpan(content, e.Span.File, commStart, e.Amount.Span.Start.Offset); ok {
339
			emit(comm, semCommodity, 0)
340
		}
341
		semEmitAmount(content, &e.Amount, emit)
342
		emitComment(e.Comment, emit)
343
	case *ast.ConversionDirective:
344
		emit(directiveKeyword(e.Span, "C"), semDirective, 0)
345
		semEmitAmount(content, &e.From, emit)
346
		// = operator: text between the two amounts
347
		if op, ok := betweenSpan(content, e.Span.File, e.From.Span.End.Offset, e.To.Span.Start.Offset); ok {
348
			emit(op, semOperator, 0)
349
		}
350
		semEmitAmount(content, &e.To, emit)
351
		emitComment(e.Comment, emit)
352
	case *ast.Comment:
353
		emitComment(e, emit)
354
	case *ast.CommentBlockDirective:
355
		emit(e.Span, semComment, 0)
356
	case *ast.IgnoredDirective:
357
		emitDirective(content, e.Span, len("N"), semProperty, e.Comment, emit)
358
	case *ast.ApplyDirective:
359
		emitDirective(content, e.Span, len("apply"), semProperty, e.Comment, emit)
360
	case *ast.EndDirective:
361
		emitDirective(content, e.Span, len("end"), semProperty, e.Comment, emit)
362
	case *ast.BlankLine:
363
	}
364
}
365
366
func visitTransaction(content string, t *ast.Transaction, emit semEmitFunc) {
367
	emit(t.Date.Span, semDate, 0)
368
	if t.SecondDate != nil {
369
		emit(t.SecondDate.Span, semDate, 0)
370
	}
371
	if t.Status.Value != ast.StatusNone {
372
		emit(t.Status.Span, semStatus, 0)
373
	}
374
	if t.Code != nil {
375
		emit(t.Code.Span, semString, 0)
376
	}
377
	if t.Payee != nil {
378
		emit(t.Payee.Span, semProperty, 0)
379
	}
380
	if t.Note != nil {
381
		emit(t.Note.Span, semProperty, 0)
382
	}
383
	emitComment(t.Comment, emit)
384
	for i := range t.HeaderComments {
385
		emitComment(t.HeaderComments[i], emit)
386
	}
387
	for _, p := range t.Postings {
388
		visitPosting(content, p, emit)
389
	}
390
}
391
392
func visitPeriodicTransaction(content string, pt *ast.PeriodicTransaction, emit semEmitFunc) {
393
	// ~ operator is at the start of the period span
394
	emit(offsetSpan(pt.Span.File, pt.Span.Start.Offset, pt.Span.Start.Offset+1), semOperator, 0)
395
396
	// The period span covers the whole expr, including any "from ... to ..." dates
397
	if pt.Period.Span.End.Offset > pt.Period.Span.Start.Offset {
398
		var dates []*ast.Date
399
		if pt.Period.From != nil {
400
			dates = append(dates, pt.Period.From)
401
		}
402
		if pt.Period.To != nil {
403
			dates = append(dates, pt.Period.To)
404
		}
405
		pos := pt.Period.Span.Start.Offset
406
		for _, d := range dates {
407
			if d.Span.Start.Offset > pos {
408
				emit(offsetSpan(pt.Period.Span.File, pos, d.Span.Start.Offset), semProperty, 0)
409
			}
410
			emit(d.Span, semDate, 0)
411
			pos = d.Span.End.Offset
412
		}
413
		if pos < pt.Period.Span.End.Offset {
414
			emit(offsetSpan(pt.Period.Span.File, pos, pt.Period.Span.End.Offset), semProperty, 0)
415
		}
416
	}
417
	if pt.Description != nil {
418
		emit(pt.Description.Span, semProperty, 0)
419
	}
420
	emitComment(pt.Comment, emit)
421
	for i := range pt.HeaderComments {
422
		emitComment(pt.HeaderComments[i], emit)
423
	}
424
	for _, p := range pt.Postings {
425
		visitPosting(content, p, emit)
426
	}
427
}
428
429
func visitAutomatedTransaction(content string, at *ast.AutomatedTransaction, emit semEmitFunc) {
430
	// = operator is at the start of the expression span
431
	emit(offsetSpan(at.Span.File, at.Span.Start.Offset, at.Span.Start.Offset+1), semOperator, 0)
432
433
	if at.Expr.Value != "" {
434
		emit(at.Expr.Span, semString, 0)
435
	}
436
	emitComment(at.Comment, emit)
437
	for i := range at.HeaderComments {
438
		emitComment(at.HeaderComments[i], emit)
439
	}
440
	for _, p := range at.Postings {
441
		visitPosting(content, p, emit)
442
	}
443
}
444
445
func visitPosting(content string, p ast.Posting, emit semEmitFunc) {
446
	if p.Status.Value != ast.StatusNone {
447
		emit(p.Status.Span, semStatus, 0)
448
	}
449
450
	// virtual brackets
451
	if p.Type == ast.PostingVirtualUnbalanced || p.Type == ast.PostingVirtualBalanced {
452
		// opening bracket
453
		for off := p.Span.Start.Offset; off < p.Account.Span.Start.Offset && off < p.Span.End.Offset; off++ {
454
			if content[off] == '(' || content[off] == '[' {
455
				emit(offsetSpan(p.Span.File, off, off+1), semOperator, modifierAbstract)
456
				break
457
			}
458
		}
459
		// closing bracket
460
		for off := p.Account.Span.End.Offset; off < p.Span.End.Offset; off++ {
461
			if content[off] == ')' || content[off] == ']' {
462
				emit(offsetSpan(p.Span.File, off, off+1), semOperator, modifierAbstract)
463
				break
464
			}
465
		}
466
	}
467
468
	emit(p.Account.Span, semAccount, 0)
469
470
	if p.Amount != nil {
471
		semEmitAmount(content, p.Amount, emit)
472
	}
473
	if p.Cost != nil {
474
		semEmitCost(content, p.Cost, emit)
475
	}
476
	if p.Balance != nil {
477
		semEmitBalanceAssertion(content, p.Balance, emit)
478
	}
479
	emitComment(p.Comment, emit)
480
	for i := range p.Comments {
481
		emitComment(&p.Comments[i], emit)
482
	}
483
}
484
485
// directiveKeyword returns the span of the leading keyword on a directive line.
486
func directiveKeyword(e token.Span, kw string) token.Span {
487
	return token.Span{File: e.File, Start: e.Start, End: token.Pos{Offset: e.Start.Offset + len(kw)}}
488
}
489
490
// directiveValue returns the trimmed span of the text after the keyword end
491
// offset, up to the inline comment or the end of the line.
492
func directiveValue(content string, e token.Span, comment *ast.Comment, kwEnd int) (token.Span, bool) {
493
	end := e.End.Offset
494
	if comment != nil {
495
		end = comment.Span.Start.Offset
496
	}
497
	return betweenSpan(content, e.File, kwEnd, end)
498
}
499
500
func semEmitAmount(content string, a *ast.Amount, emit semEmitFunc) {
501
	if a == nil {
502
		return
503
	}
504
	hasCommodity := a.CommoditySpan.Start.Offset > 0 && a.CommoditySpan.End.Offset > 0
505
	if hasCommodity && a.CommodityPos == ast.CommodityBefore {
506
		emit(a.CommoditySpan, semCommodity, 0)
507
		semEmitQuantity(content, a, emit)
508
		return
509
	}
510
	semEmitQuantity(content, a, emit)
511
	if hasCommodity {
512
		emit(a.CommoditySpan, semCommodity, 0)
513
	}
514
}
515
516
func semEmitQuantity(content string, a *ast.Amount, emit semEmitFunc) {
517
	qStart, qEnd := quantitySpan(content, a)
518
	if qEnd <= qStart {
519
		return
520
	}
521
	mods := uint32(0)
522
	if a.IsNegative {
523
		mods |= modifierNegative
524
	}
525
	emit(offsetSpan(a.Span.File, qStart, qEnd), semAmount, mods)
526
}
527
528
func quantitySpan(content string, a *ast.Amount) (int, int) {
529
	start, end := a.Span.Start.Offset, a.Span.End.Offset
530
	switch {
531
	case a.Commodity == "":
532
		// bare quantity
533
	case a.CommodityPos == ast.CommodityBefore:
534
		// "$50.00" or "$   50.00": quantity follows the commodity span
535
		start = a.CommoditySpan.End.Offset
536
	default: // CommodityAfter
537
		// "50.00 USD" or "50.00     USD": quantity precedes the commodity
538
		end = a.CommoditySpan.Start.Offset
539
	}
540
	for start < end && (content[start] == ' ' || content[start] == '\t') {
541
		start++
542
	}
543
	for end > start && (content[end-1] == ' ' || content[end-1] == '\t') {
544
		end--
545
	}
546
	return start, end
547
}
548
549
type semEmitFunc func(tok token.Span, tokKind, modifier uint32)
550
551
func emitBlockComments(cs []*ast.Comment, emit semEmitFunc) {
552
	for _, c := range cs {
553
		emitComment(c, emit)
554
	}
555
}
556
557
func emitComment(c *ast.Comment, emit semEmitFunc) {
558
	if c == nil {
559
		return
560
	}
561
	if len(c.Tags) == 0 {
562
		emit(c.Span, semComment, 0)
563
		return
564
	}
565
	pos := c.Span.Start.Offset
566
	for _, t := range c.Tags {
567
		if t.Span.Start.Offset > pos {
568
			emit(offsetSpan(c.Span.File, pos, t.Span.Start.Offset), semComment, 0)
569
		}
570
		emit(t.Span, semProperty, 0)
571
		pos = t.Span.End.Offset
572
	}
573
	if pos < c.Span.End.Offset {
574
		emit(offsetSpan(c.Span.File, pos, c.Span.End.Offset), semComment, 0)
575
	}
576
}
577
578
func emitDirective(content string, e token.Span, kwLen int, valType uint32, comment *ast.Comment, emit semEmitFunc) {
579
	kwEnd := e.Start.Offset + kwLen
580
	emit(token.Span{File: e.File, Start: e.Start, End: token.Pos{Offset: kwEnd}}, semDirective, 0)
581
	if v, ok := directiveValue(content, e, comment, kwEnd); ok {
582
		emit(v, valType, 0)
583
	}
584
	emitComment(comment, emit)
585
}
586
587
func semEmitCost(content string, c *ast.Cost, emit semEmitFunc) {
588
	if c.IsTotal {
589
		emit(token.Span{File: c.Span.File, Start: c.Span.Start, End: token.Pos{Offset: c.Span.Start.Offset + 2}}, semOperator, 0)
590
	} else {
591
		emit(token.Span{File: c.Span.File, Start: c.Span.Start, End: token.Pos{Offset: c.Span.Start.Offset + 1}}, semOperator, 0)
592
	}
593
	semEmitAmount(content, &c.Amount, emit)
594
}
595
596
func semEmitBalanceAssertion(content string, ba *ast.BalanceAssertion, emit semEmitFunc) {
597
	// The operator is the run of '=', ':', '*' chars from the span start.
598
	// The ':' of ':=' precedes the '=' token, so back up one offset.
599
	opStart := ba.Span.Start.Offset
600
	if ba.IsAssignment && opStart > 0 {
601
		opStart--
602
	}
603
	opEnd := opStart
604
	for opEnd < ba.Span.End.Offset && (content[opEnd] == '=' || content[opEnd] == ':' || content[opEnd] == '*') {
605
		opEnd++
606
	}
607
	emit(token.Span{File: ba.Span.File, Start: token.Pos{Offset: opStart}, End: token.Pos{Offset: opEnd}}, semOperator, 0)
608
	semEmitAmount(content, &ba.Amount, emit)
609
	if ba.Cost != nil {
610
		semEmitCost(content, ba.Cost, emit)
611
	}
612
}
613
614
func semLexerFallback(content string, base []rawSpan, emit semEmitFunc) {
615
	l := lexer.New("", []byte(content))
616
617
	var commentStart, commentEnd int // 0 = not inside a comment line
618
	lineStart := true                // the next significant token starts a line
619
	skipLine := false                // the line starts with an unclassifiable token; emit nothing
620
	i := 0                           // next base span to compare against
621
622
	take := func(span token.Span, tokType uint32, mods uint32) {
623
		for i < len(base) && base[i].span.End.Offset <= span.Start.Offset {
624
			i++
625
		}
626
		if i < len(base) && base[i].span.Start.Offset < span.End.Offset {
627
			return // overlaps an AST token; AST wins
628
		}
629
		emit(span, tokType, mods)
630
	}
631
632
	for {
633
		tok := l.Next()
634
		if tok.Type == token.EOF {
635
			if commentStart > 0 {
636
				take(token.Span{Start: token.Pos{Offset: commentStart}, End: token.Pos{Offset: commentEnd}}, semComment, 0)
637
			}
638
			break
639
		}
640
		if tok.Type == token.NEWLINE {
641
			if commentStart > 0 {
642
				take(token.Span{Start: token.Pos{Offset: commentStart}, End: token.Pos{Offset: commentEnd}}, semComment, 0)
643
				commentStart, commentEnd = 0, 0
644
			}
645
			lineStart, skipLine = true, false
646
			continue
647
		}
648
		if tok.Type == token.WHITESPACE || tok.Type == token.INDENT {
649
			continue
650
		}
651
		if lineStart {
652
			lineStart = false
653
			if !isLineStartToken(tok.Type) {
654
				skipLine = true
655
			}
656
		}
657
		if skipLine {
658
			continue
659
		}
660
		tokType := semProperty
661
		if commentStart > 0 {
662
			tokType = semComment
663
			if tok.Span.End.Offset > commentEnd {
664
				commentEnd = tok.Span.End.Offset
665
			}
666
			continue
667
		}
668
669
		switch tok.Type {
670
		case token.SEMICOLON, token.HASH, token.PERCENT, token.STAR:
671
			tokType = semComment
672
			commentStart = tok.Span.Start.Offset
673
			commentEnd = tok.Span.End.Offset
674
			continue
675
		case token.STRING:
676
			tokType = semString
677
		case token.DATE:
678
			tokType = semDate
679
		case token.INT, token.DECIMAL:
680
			tokType = semAmount
681
		case token.COMMODITYMARK:
682
			tokType = semCommodity
683
		case token.BANG:
684
			tokType = semStatus
685
		case token.AT, token.ATAT, token.EQ, token.EQEQ, token.EQEQEQ, token.EQSTAR:
686
			tokType = semOperator
687
		case token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE,
688
			token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END,
689
			token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C:
690
			tokType = semDirective
691
		}
692
		take(tok.Span, tokType, 0)
693
	}
694
}
695
696
func isLineStartToken(t token.Type) bool {
697
	switch t {
698
	case token.DATE, token.TILDE, token.EQ, token.BANG, token.AT,
699
		token.SEMICOLON, token.HASH, token.PERCENT, token.STAR,
700
		token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE,
701
		token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END,
702
		token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C:
703
		return true
704
	}
705
	return false
706
}
707
708
// semanticTokensEdits returns the single edit turning old into new, or nil when
709
// identical. Relative delta encoding keeps the common prefix and suffix unchanged.
710
func semanticTokensEdits(old, new []uint32) []protocol.SemanticTokensEdit {
711
	p := 0
712
	for p < len(old) && p < len(new) && old[p] == new[p] {
713
		p++
714
	}
715
	s := 0
716
	for s < len(old)-p && s < len(new)-p && old[len(old)-1-s] == new[len(new)-1-s] {
717
		s++
718
	}
719
	delCount := len(old) - p - s
720
	ins := new[p : len(new)-s]
721
	if delCount == 0 && len(ins) == 0 {
722
		return nil
723
	}
724
	return []protocol.SemanticTokensEdit{{
725
		Start:       uint32(p),
726
		DeleteCount: uint32(delCount),
727
		Data:        ins,
728
	}}
729
}
730
731
// encodeSemTokens encodes tokens into LSP delta form. Input must be sorted by
732
// line and column; [rawToSemanticTokens] produces such order.
733
func encodeSemTokens(tokens []semanticToken) []uint32 {
734
	if len(tokens) == 0 {
735
		return nil
736
	}
737
	data := make([]uint32, 0, len(tokens)*5)
738
	var prevLine, prevCol uint32
739
	for _, t := range tokens {
740
		var deltaLine, deltaCol uint32
741
		if t.line == prevLine {
742
			deltaLine = 0
743
			deltaCol = t.col - prevCol
744
		} else {
745
			deltaLine = t.line - prevLine
746
			deltaCol = t.col
747
		}
748
		data = append(data, deltaLine, deltaCol, t.length, t.tokenType, t.modifiers)
749
		prevLine = t.line
750
		prevCol = t.col
751
	}
752
	return data
753
}
754
755
// betweenSpan returns the span of the text between two offsets, trimmed of surrounding whitespace.
756
func betweenSpan(content, file string, start, end int) (token.Span, bool) {
757
	for start < end && (content[start] == ' ' || content[start] == '\t') {
758
		start++
759
	}
760
	for end > start && (content[end-1] == ' ' || content[end-1] == '\t' || content[end-1] == '\n' || content[end-1] == '\r') {
761
		end--
762
	}
763
	if end <= start {
764
		return token.Span{}, false
765
	}
766
	return offsetSpan(file, start, end), true
767
}
768
769
func offsetSpan(file string, start, end int) token.Span {
770
	return token.Span{File: file, Start: token.Pos{Offset: start}, End: token.Pos{Offset: end}}
771
}