all repos

clerk @ 3d2b11f

missing tooling for ledger/hledger

clerk/internal/lsp/textdocument_semantic.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
lsp: tokenize semantic tokens outside the document lock, 1 month ago
1
package lsp
2
3
import (
4
	"context"
5
	"slices"
6
	"unicode/utf8"
7
8
	"go.lsp.dev/protocol"
9
	"go.lsp.dev/uri"
10
11
	"olexsmir.xyz/clerk/internal/lsp/lsputil"
12
	"olexsmir.xyz/clerk/journal/ast"
13
	"olexsmir.xyz/clerk/journal/lexer"
14
	"olexsmir.xyz/clerk/journal/token"
15
)
16
17
func (s *server) SemanticTokensFull(ctx context.Context, params *protocol.SemanticTokensParams) (*protocol.SemanticTokens, error) {
18
	if !s.semanticHighlightingEnabled() {
19
		return &protocol.SemanticTokens{}, nil
20
	}
21
22
	tokens, ok := s.tokensForDoc(params.TextDocument.URI)
23
	if !ok {
24
		return &protocol.SemanticTokens{}, nil
25
	}
26
	return &protocol.SemanticTokens{Data: encodeSemTokens(tokens)}, nil
27
}
28
29
func (s *server) SemanticTokensRange(ctx context.Context, params *protocol.SemanticTokensRangeParams) (*protocol.SemanticTokens, error) {
30
	if !s.semanticHighlightingEnabled() {
31
		return &protocol.SemanticTokens{}, nil
32
	}
33
34
	tokens, ok := s.tokensForDoc(params.TextDocument.URI)
35
	if !ok {
36
		return &protocol.SemanticTokens{}, nil
37
	}
38
	start := int(params.Range.Start.Line)
39
	end := int(params.Range.End.Line)
40
	var filtered []semanticToken
41
	for _, t := range tokens {
42
		if t.line >= uint32(start) && t.line <= uint32(end) {
43
			filtered = append(filtered, t)
44
		}
45
	}
46
	return &protocol.SemanticTokens{Data: encodeSemTokens(filtered)}, nil
47
}
48
49
func (s *server) tokensForDoc(doc uri.URI) ([]semanticToken, bool) {
50
	s.mu.Lock()
51
	st, ok := s.openDocs[doc]
52
	s.mu.Unlock()
53
	if !ok {
54
		return nil, false
55
	}
56
	if st.semTokens != nil {
57
		return st.semTokens, true
58
	}
59
	// Tokenize outside the lock: a full tokenization of a large journal is
60
	// milliseconds, during which didChange/didOpen would otherwise stall.
61
	tokens := tokenizeForSemantics(st.text, parseJournalStr(st.text))
62
	s.mu.Lock()
63
	defer s.mu.Unlock()
64
	cur, ok := s.openDocs[doc]
65
	if !ok {
66
		return nil, false
67
	}
68
	if cur.text == st.text { // unchanged during tokenization
69
		cur.semTokens = tokens
70
		s.openDocs[doc] = cur
71
	}
72
	return tokens, true
73
}
74
75
// Implementation
76
77
const (
78
	semDirective uint32 = iota
79
	semDate
80
	semAccount
81
	semCommodity
82
	semAmount
83
	semStatus
84
	semComment
85
	semString
86
	semOperator
87
	semProperty
88
)
89
90
var tokenTypeStrings = []string{
91
	string(protocol.SemanticTokenTypesKeyword),   // directive
92
	string(protocol.SemanticTokenTypesClass),     // date
93
	string(protocol.SemanticTokenTypesNamespace), // account
94
	string(protocol.SemanticTokenTypesType),      // commodity
95
	string(protocol.SemanticTokenTypesNumber),    // amount
96
	string(protocol.SemanticTokenTypesOperator),  // status
97
	string(protocol.SemanticTokenTypesComment),   // comment
98
	string(protocol.SemanticTokenTypesString),    // string
99
	string(protocol.SemanticTokenTypesOperator),  // operator
100
	string(protocol.SemanticTokenTypesProperty),  // property
101
}
102
103
const (
104
	modifierAbstract = 1 << 0 // virtual account
105
	modifierNegative = 1 << 1 // negative amount
106
)
107
108
var modifierStrings = []string{
109
	"abstract", // bit 0
110
	"negative", // bit 1
111
}
112
113
func getSemanticTokensLegend() protocol.SemanticTokensLegend {
114
	return protocol.SemanticTokensLegend{
115
		TokenTypes:     tokenTypeStrings,
116
		TokenModifiers: modifierStrings,
117
	}
118
}
119
120
type semanticToken struct {
121
	line, col uint32 // 0-based
122
	length    uint32
123
	tokenType uint32
124
	modifiers uint32
125
}
126
127
func tokenizeForSemantics(content string, j *ast.Journal) []semanticToken {
128
	var raw []rawSpan
129
	emit := func(s token.Span, tokType, mods uint32) {
130
		if s.Start.Offset >= s.End.Offset {
131
			return
132
		}
133
		raw = append(raw, rawSpan{s, tokType, mods})
134
	}
135
	for _, e := range j.Entries {
136
		visitEntry(content, e, emit)
137
	}
138
	if len(j.Errors) > 0 {
139
		// parser recovers per line; lexer fills the unparsed regions, keeping ast tokens where the parser succeeded
140
		semLexerFallback(content, raw, emit)
141
	}
142
	return rawToSemanticTokens(content, raw)
143
}
144
145
// rawSpan is a source span tagged with semantic token
146
type rawSpan struct {
147
	span      token.Span
148
	tok, mods uint32
149
}
150
151
func rawToSemanticTokens(content string, raw []rawSpan) []semanticToken {
152
	if len(raw) == 0 {
153
		return nil
154
	}
155
	slices.SortFunc(raw, func(a, b rawSpan) int { return a.span.Start.Offset - b.span.Start.Offset })
156
	out := make([]semanticToken, len(raw))
157
	line, col, cursor := 0, 0, 0
158
	advance := func(end int) {
159
		for cursor < end {
160
			r, size := utf8.DecodeRuneInString(content[cursor:])
161
			if r == utf8.RuneError && size <= 1 {
162
				break
163
			}
164
			if r == '\r' {
165
				cursor += size
166
				if cursor < len(content) && content[cursor] == '\n' {
167
					cursor++
168
				}
169
				line++
170
				col = 0
171
				continue
172
			}
173
			if r == '\n' {
174
				cursor += size
175
				line++
176
				col = 0
177
				continue
178
			}
179
			cursor += size
180
			col += utf16Units(r)
181
		}
182
	}
183
	for i, t := range raw {
184
		if cursor < t.span.Start.Offset {
185
			advance(t.span.Start.Offset)
186
		}
187
		out[i] = semanticToken{
188
			line:      uint32(line),
189
			col:       uint32(col),
190
			length:    uint32(lsputil.Utf16Len(content, t.span.Start.Offset, t.span.End.Offset)),
191
			tokenType: t.tok,
192
			modifiers: t.mods,
193
		}
194
		advance(t.span.End.Offset)
195
	}
196
	return out
197
}
198
199
func utf16Units(r rune) int {
200
	if r >= 0x10000 && r <= 0x10FFFF {
201
		return 2
202
	}
203
	return 1
204
}
205
206
func visitEntry(content string, e ast.Entry, emit semEmitFunc) {
207
	switch e := e.(type) {
208
	case *ast.Transaction:
209
		visitTransaction(content, e, emit)
210
	case *ast.PeriodicTransaction:
211
		visitPeriodicTransaction(content, e, emit)
212
	case *ast.AutomatedTransaction:
213
		visitAutomatedTransaction(content, e, emit)
214
	case *ast.AccountDirective:
215
		emit(directiveKeyword(e.Span, "account"), semDirective, 0)
216
		emit(e.Account.Span, semAccount, 0)
217
		for _, sd := range e.Subdirectives {
218
			if sd.Name == "" {
219
				emitComment(sd.Comment, emit)
220
				continue
221
			}
222
			emit(sd.NameSpan, semDirective, 0)
223
			switch sd.Name {
224
			case "alias":
225
				emit(sd.ValueSpan, semAccount, 0)
226
			case "type", "note":
227
				emit(sd.ValueSpan, semProperty, 0)
228
			}
229
			emitComment(sd.Comment, emit)
230
		}
231
		emitComment(e.Comment, emit)
232
	case *ast.CommodityDirective:
233
		emit(directiveKeyword(e.Span, "commodity"), semDirective, 0)
234
		if e.FormatSub != nil {
235
			if e.FormatSub.KeywordSpan.End.Offset > 0 {
236
				emit(e.FormatSub.KeywordSpan, semDirective, 0)
237
			}
238
			semEmitAmount(content, &e.FormatSub.Amount, emit)
239
			emitComment(e.FormatSub.Comment, emit)
240
		} else if e.CommoditySpan.Start.Offset > 0 && e.CommoditySpan.End.Offset > 0 {
241
			emit(e.CommoditySpan, semCommodity, 0)
242
		}
243
		emitBlockComments(e.BlockComments, emit)
244
		emitComment(e.Comment, emit)
245
	case *ast.IncludeDirective:
246
		emitDirective(content, e.Span, len("include"), semString, e.Comment, emit)
247
	case *ast.PayeeDirective:
248
		emitDirective(content, e.Span, len("payee"), semProperty, e.Comment, emit)
249
	case *ast.TagDirective:
250
		emitDirective(content, e.Span, len("tag"), semProperty, e.Comment, emit)
251
	case *ast.AliasDirective:
252
		emit(directiveKeyword(e.Span, "alias"), semDirective, 0)
253
		emit(e.From.Span, semAccount, 0)
254
		if op, ok := betweenSpan(content, e.Span.Start.File, e.From.Span.End.Offset, e.To.Span.Start.Offset); ok {
255
			emit(op, semOperator, 0)
256
		}
257
		emit(e.To.Span, semAccount, 0)
258
		emitComment(e.Comment, emit)
259
	case *ast.YearDirective:
260
		kwLen := len("year")
261
		if content[e.Span.Start.Offset] == 'Y' {
262
			kwLen = 1
263
		}
264
		emitDirective(content, e.Span, kwLen, semProperty, e.Comment, emit)
265
	case *ast.DecimalMarkDirective:
266
		emitDirective(content, e.Span, len("decimal-mark"), semProperty, e.Comment, emit)
267
	case *ast.DefaultCommodityDirective:
268
		emit(directiveKeyword(e.Span, "D"), semDirective, 0)
269
		semEmitAmount(content, &e.Amount, emit)
270
		emitComment(e.Comment, emit)
271
	case *ast.MarketPriceDirective:
272
		emit(directiveKeyword(e.Span, "P"), semDirective, 0)
273
		emit(e.DateTime.Date.Span, semDate, 0)
274
		if e.DateTime.Time != nil {
275
			emit(e.DateTime.Time.Span, semDate, 0)
276
		}
277
		// commodity: text between the date (or time) and the amount
278
		commStart := e.DateTime.Date.Span.End.Offset
279
		if e.DateTime.Time != nil {
280
			commStart = e.DateTime.Time.Span.End.Offset
281
		}
282
		if comm, ok := betweenSpan(content, e.Span.Start.File, commStart, e.Amount.Span.Start.Offset); ok {
283
			emit(comm, semCommodity, 0)
284
		}
285
		semEmitAmount(content, &e.Amount, emit)
286
		emitComment(e.Comment, emit)
287
	case *ast.ConversionDirective:
288
		emit(directiveKeyword(e.Span, "C"), semDirective, 0)
289
		semEmitAmount(content, &e.From, emit)
290
		// = operator: text between the two amounts
291
		if op, ok := betweenSpan(content, e.Span.Start.File, e.From.Span.End.Offset, e.To.Span.Start.Offset); ok {
292
			emit(op, semOperator, 0)
293
		}
294
		semEmitAmount(content, &e.To, emit)
295
		emitComment(e.Comment, emit)
296
	case *ast.Comment:
297
		emitComment(e, emit)
298
	case *ast.CommentBlockDirective:
299
		emit(e.Span, semComment, 0)
300
	case *ast.IgnoredDirective:
301
		emitDirective(content, e.Span, len("N"), semProperty, e.Comment, emit)
302
	case *ast.ApplyDirective:
303
		emitDirective(content, e.Span, len("apply"), semProperty, e.Comment, emit)
304
	case *ast.EndDirective:
305
		emitDirective(content, e.Span, len("end"), semProperty, e.Comment, emit)
306
	case *ast.BlankLine:
307
	}
308
}
309
310
func visitTransaction(content string, t *ast.Transaction, emit semEmitFunc) {
311
	emit(t.Date.Span, semDate, 0)
312
	if t.SecondDate != nil {
313
		emit(t.SecondDate.Span, semDate, 0)
314
	}
315
	if t.Status.Value != ast.StatusNone {
316
		emit(t.Status.Span, semStatus, 0)
317
	}
318
	if t.Code != nil {
319
		emit(t.Code.Span, semString, 0)
320
	}
321
	if t.Payee != nil {
322
		emit(t.Payee.Span, semProperty, 0)
323
	}
324
	if t.Note != nil {
325
		emit(t.Note.Span, semProperty, 0)
326
	}
327
	emitComment(t.Comment, emit)
328
	for i := range t.HeaderComments {
329
		emitComment(t.HeaderComments[i], emit)
330
	}
331
	for _, p := range t.Postings {
332
		visitPosting(content, p, emit)
333
	}
334
}
335
336
func visitPeriodicTransaction(content string, pt *ast.PeriodicTransaction, emit semEmitFunc) {
337
	// ~ operator is at the start of the period span
338
	emit(offsetSpan(pt.Span.Start.File, pt.Span.Start.Offset, pt.Span.Start.Offset+1), semOperator, 0)
339
340
	// The period span covers the whole expr, including any "from ... to ..." dates
341
	if pt.Period.Span.End.Offset > pt.Period.Span.Start.Offset {
342
		var dates []*ast.Date
343
		if pt.Period.From != nil {
344
			dates = append(dates, pt.Period.From)
345
		}
346
		if pt.Period.To != nil {
347
			dates = append(dates, pt.Period.To)
348
		}
349
		pos := pt.Period.Span.Start.Offset
350
		for _, d := range dates {
351
			if d.Span.Start.Offset > pos {
352
				emit(offsetSpan(pt.Period.Span.Start.File, pos, d.Span.Start.Offset), semProperty, 0)
353
			}
354
			emit(d.Span, semDate, 0)
355
			pos = d.Span.End.Offset
356
		}
357
		if pos < pt.Period.Span.End.Offset {
358
			emit(offsetSpan(pt.Period.Span.Start.File, pos, pt.Period.Span.End.Offset), semProperty, 0)
359
		}
360
	}
361
	if pt.Description != nil {
362
		emit(pt.Description.Span, semProperty, 0)
363
	}
364
	emitComment(pt.Comment, emit)
365
	for i := range pt.HeaderComments {
366
		emitComment(pt.HeaderComments[i], emit)
367
	}
368
	for _, p := range pt.Postings {
369
		visitPosting(content, p, emit)
370
	}
371
}
372
373
func visitAutomatedTransaction(content string, at *ast.AutomatedTransaction, emit semEmitFunc) {
374
	// = operator is at the start of the expression span
375
	emit(offsetSpan(at.Span.Start.File, at.Span.Start.Offset, at.Span.Start.Offset+1), semOperator, 0)
376
377
	if at.Expr.Value != "" {
378
		emit(at.Expr.Span, semString, 0)
379
	}
380
	emitComment(at.Comment, emit)
381
	for i := range at.HeaderComments {
382
		emitComment(at.HeaderComments[i], emit)
383
	}
384
	for _, p := range at.Postings {
385
		visitPosting(content, p, emit)
386
	}
387
}
388
389
func visitPosting(content string, p *ast.Posting, emit semEmitFunc) {
390
	if p.Status.Value != ast.StatusNone {
391
		emit(p.Status.Span, semStatus, 0)
392
	}
393
394
	// virtual brackets
395
	if p.Type == ast.PostingVirtualUnbalanced || p.Type == ast.PostingVirtualBalanced {
396
		// opening bracket
397
		for off := p.Span.Start.Offset; off < p.Account.Span.Start.Offset && off < p.Span.End.Offset; off++ {
398
			if content[off] == '(' || content[off] == '[' {
399
				brSpan := token.Span{Start: offsetPos(p.Span.Start.File, off), End: offsetPos(p.Span.Start.File, off+1)}
400
				emit(brSpan, semOperator, modifierAbstract)
401
				break
402
			}
403
		}
404
		// closing bracket
405
		for off := p.Account.Span.End.Offset; off < p.Span.End.Offset; off++ {
406
			if content[off] == ')' || content[off] == ']' {
407
				brSpan := token.Span{Start: offsetPos(p.Span.Start.File, off), End: offsetPos(p.Span.Start.File, off+1)}
408
				emit(brSpan, semOperator, modifierAbstract)
409
				break
410
			}
411
		}
412
	}
413
414
	emit(p.Account.Span, semAccount, 0)
415
416
	if p.Amount != nil {
417
		semEmitAmount(content, p.Amount, emit)
418
	}
419
	if p.Cost != nil {
420
		semEmitCost(content, p.Cost, emit)
421
	}
422
	if p.Balance != nil {
423
		semEmitBalanceAssertion(content, p.Balance, emit)
424
	}
425
	emitComment(p.Comment, emit)
426
	for i := range p.Comments {
427
		emitComment(&p.Comments[i], emit)
428
	}
429
}
430
431
// directiveKeyword returns the span of the leading keyword on a directive line.
432
func directiveKeyword(e token.Span, kw string) token.Span {
433
	return token.Span{Start: e.Start, End: offsetPos(e.Start.File, e.Start.Offset+len(kw))}
434
}
435
436
// directiveValue returns the trimmed span of the text after the keyword end
437
// offset, up to the inline comment or the end of the line.
438
func directiveValue(content string, e token.Span, comment *ast.Comment, kwEnd int) (token.Span, bool) {
439
	end := e.End.Offset
440
	if comment != nil {
441
		end = comment.Span.Start.Offset
442
	}
443
	return betweenSpan(content, e.Start.File, kwEnd, end)
444
}
445
446
func semEmitAmount(content string, a *ast.Amount, emit semEmitFunc) {
447
	if a == nil {
448
		return
449
	}
450
	hasCommodity := a.CommoditySpan.Start.Offset > 0 && a.CommoditySpan.End.Offset > 0
451
	if hasCommodity && a.CommodityPos == ast.CommodityBefore {
452
		emit(a.CommoditySpan, semCommodity, 0)
453
		semEmitQuantity(content, a, emit)
454
		return
455
	}
456
	semEmitQuantity(content, a, emit)
457
	if hasCommodity {
458
		emit(a.CommoditySpan, semCommodity, 0)
459
	}
460
}
461
462
func semEmitQuantity(content string, a *ast.Amount, emit semEmitFunc) {
463
	qStart, qEnd := quantitySpan(content, a)
464
	if qEnd <= qStart {
465
		return
466
	}
467
	mods := uint32(0)
468
	if a.IsNegative {
469
		mods |= modifierNegative
470
	}
471
	emit(offsetSpan(a.Span.Start.File, qStart, qEnd), semAmount, mods)
472
}
473
474
func quantitySpan(content string, a *ast.Amount) (int, int) {
475
	start, end := a.Span.Start.Offset, a.Span.End.Offset
476
	switch {
477
	case a.Commodity == "":
478
		// bare quantity
479
	case a.CommodityPos == ast.CommodityBefore:
480
		// "$50.00" or "$   50.00": quantity follows the commodity span
481
		start = a.CommoditySpan.End.Offset
482
	default: // CommodityAfter
483
		// "50.00 USD" or "50.00     USD": quantity precedes the commodity
484
		end = a.CommoditySpan.Start.Offset
485
	}
486
	for start < end && (content[start] == ' ' || content[start] == '\t') {
487
		start++
488
	}
489
	for end > start && (content[end-1] == ' ' || content[end-1] == '\t') {
490
		end--
491
	}
492
	return start, end
493
}
494
495
type semEmitFunc func(tok token.Span, tokKind, modifier uint32)
496
497
func emitBlockComments(cs []*ast.Comment, emit semEmitFunc) {
498
	for _, c := range cs {
499
		emitComment(c, emit)
500
	}
501
}
502
503
func emitComment(c *ast.Comment, emit semEmitFunc) {
504
	if c == nil {
505
		return
506
	}
507
	if len(c.Tags) == 0 {
508
		emit(c.Span, semComment, 0)
509
		return
510
	}
511
	pos := c.Span.Start.Offset
512
	for _, t := range c.Tags {
513
		if t.Span.Start.Offset > pos {
514
			emit(offsetSpan(c.Span.Start.File, pos, t.Span.Start.Offset), semComment, 0)
515
		}
516
		emit(t.Span, semProperty, 0)
517
		pos = t.Span.End.Offset
518
	}
519
	if pos < c.Span.End.Offset {
520
		emit(offsetSpan(c.Span.Start.File, pos, c.Span.End.Offset), semComment, 0)
521
	}
522
}
523
524
func emitDirective(content string, e token.Span, kwLen int, valType uint32, comment *ast.Comment, emit semEmitFunc) {
525
	kwEnd := e.Start.Offset + kwLen
526
	emit(token.Span{Start: e.Start, End: offsetPos(e.Start.File, kwEnd)}, semDirective, 0)
527
	if v, ok := directiveValue(content, e, comment, kwEnd); ok {
528
		emit(v, valType, 0)
529
	}
530
	emitComment(comment, emit)
531
}
532
533
func semEmitCost(content string, c *ast.Cost, emit semEmitFunc) {
534
	if c.IsTotal {
535
		emit(token.Span{Start: c.Span.Start, End: offsetPos(c.Span.Start.File, c.Span.Start.Offset+2)}, semOperator, 0)
536
	} else {
537
		emit(token.Span{Start: c.Span.Start, End: offsetPos(c.Span.Start.File, c.Span.Start.Offset+1)}, semOperator, 0)
538
	}
539
	semEmitAmount(content, &c.Amount, emit)
540
}
541
542
func semEmitBalanceAssertion(content string, ba *ast.BalanceAssertion, emit semEmitFunc) {
543
	// The operator is the run of '=', ':', '*' chars from the span start.
544
	// The ':' of ':=' precedes the '=' token, so back up one offset.
545
	opStart := ba.Span.Start.Offset
546
	if ba.IsAssignment && opStart > 0 {
547
		opStart--
548
	}
549
	opEnd := opStart
550
	for opEnd < ba.Span.End.Offset && (content[opEnd] == '=' || content[opEnd] == ':' || content[opEnd] == '*') {
551
		opEnd++
552
	}
553
	emit(token.Span{Start: offsetPos(ba.Span.Start.File, opStart), End: offsetPos(ba.Span.Start.File, opEnd)}, semOperator, 0)
554
	semEmitAmount(content, &ba.Amount, emit)
555
	if ba.Cost != nil {
556
		semEmitCost(content, ba.Cost, emit)
557
	}
558
}
559
560
func semLexerFallback(content string, base []rawSpan, emit semEmitFunc) {
561
	l := lexer.New("", []byte(content))
562
563
	var commentStart, commentEnd int // 0 = not inside a comment line
564
	lineStart := true                // the next significant token starts a line
565
	skipLine := false                // the line starts with an unclassifiable token; emit nothing
566
	i := 0                           // next base span to compare against
567
568
	take := func(span token.Span, tokType uint32, mods uint32) {
569
		for i < len(base) && base[i].span.End.Offset <= span.Start.Offset {
570
			i++
571
		}
572
		if i < len(base) && base[i].span.Start.Offset < span.End.Offset {
573
			return // overlaps an AST token; AST wins
574
		}
575
		emit(span, tokType, mods)
576
	}
577
578
	for {
579
		tok := l.Next()
580
		if tok.Type == token.EOF {
581
			if commentStart > 0 {
582
				take(token.Span{Start: offsetPos("", commentStart), End: offsetPos("", commentEnd)}, semComment, 0)
583
			}
584
			break
585
		}
586
		if tok.Type == token.NEWLINE {
587
			if commentStart > 0 {
588
				take(token.Span{Start: offsetPos("", commentStart), End: offsetPos("", commentEnd)}, semComment, 0)
589
				commentStart, commentEnd = 0, 0
590
			}
591
			lineStart, skipLine = true, false
592
			continue
593
		}
594
		if tok.Type == token.WHITESPACE || tok.Type == token.INDENT {
595
			continue
596
		}
597
		if lineStart {
598
			lineStart = false
599
			if !isLineStartToken(tok.Type) {
600
				skipLine = true
601
			}
602
		}
603
		if skipLine {
604
			continue
605
		}
606
		tokType := semProperty
607
		if commentStart > 0 {
608
			tokType = semComment
609
			if tok.Span.End.Offset > commentEnd {
610
				commentEnd = tok.Span.End.Offset
611
			}
612
			continue
613
		}
614
615
		switch tok.Type {
616
		case token.SEMICOLON, token.HASH, token.PERCENT, token.STAR:
617
			tokType = semComment
618
			commentStart = tok.Span.Start.Offset
619
			commentEnd = tok.Span.End.Offset
620
			continue
621
		case token.STRING:
622
			tokType = semString
623
		case token.DATE:
624
			tokType = semDate
625
		case token.INT, token.DECIMAL:
626
			tokType = semAmount
627
		case token.COMMODITYMARK:
628
			tokType = semCommodity
629
		case token.BANG:
630
			tokType = semStatus
631
		case token.AT, token.ATAT, token.EQ, token.EQEQ, token.EQEQEQ, token.EQSTAR:
632
			tokType = semOperator
633
		case token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE,
634
			token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END,
635
			token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C:
636
			tokType = semDirective
637
		}
638
		take(tok.Span, tokType, 0)
639
	}
640
}
641
642
func isLineStartToken(t token.Type) bool {
643
	switch t {
644
	case token.DATE, token.TILDE, token.EQ, token.BANG, token.AT,
645
		token.SEMICOLON, token.HASH, token.PERCENT, token.STAR,
646
		token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE,
647
		token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END,
648
		token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C:
649
		return true
650
	}
651
	return false
652
}
653
654
// encodeSemTokens encodes tokens into LSP delta form. Input must be sorted by
655
// line and column; [rawToSemanticTokens] produces such order.
656
func encodeSemTokens(tokens []semanticToken) []uint32 {
657
	if len(tokens) == 0 {
658
		return nil
659
	}
660
	data := make([]uint32, 0, len(tokens)*5)
661
	var prevLine, prevCol uint32
662
	for _, t := range tokens {
663
		var deltaLine, deltaCol uint32
664
		if t.line == prevLine {
665
			deltaLine = 0
666
			deltaCol = t.col - prevCol
667
		} else {
668
			deltaLine = t.line - prevLine
669
			deltaCol = t.col
670
		}
671
		data = append(data, deltaLine, deltaCol, t.length, t.tokenType, t.modifiers)
672
		prevLine = t.line
673
		prevCol = t.col
674
	}
675
	return data
676
}
677
678
// betweenSpan returns the span of the text between two offsets, trimmed of surrounding whitespace.
679
func betweenSpan(content, file string, start, end int) (token.Span, bool) {
680
	for start < end && (content[start] == ' ' || content[start] == '\t') {
681
		start++
682
	}
683
	for end > start && (content[end-1] == ' ' || content[end-1] == '\t' || content[end-1] == '\n' || content[end-1] == '\r') {
684
		end--
685
	}
686
	if end <= start {
687
		return token.Span{}, false
688
	}
689
	return token.Span{Start: offsetPos(file, start), End: offsetPos(file, end)}, true
690
}
691
692
func offsetPos(file string, offset int) token.Pos { return token.Pos{File: file, Offset: offset} }
693
func offsetSpan(file string, start, end int) token.Span {
694
	return token.Span{Start: offsetPos(file, start), End: offsetPos(file, end)}
695
}