4 files changed,
75 insertions(+),
37 deletions(-)
Author:
Oleksandr Smirnov
olexsmir@gmail.com
Committed at:
2026-08-04 20:02:00 +0300
Authored at:
2026-08-04 17:47:41 +0300
Change ID:
nnykzlqnqykrvkkmxwzmvkmzvsnxqnyn
Parent:
f58ca0d
M
internal/lsp/testdata/semantic-unparseable.txtar
··· 15 15 garbage $$$$ 123 16 16 17 17 -- expect -- 18 -0:0+7 string 19 -0:8+7 string 20 18 1:0+20 comment 21 19 2:0+11 comment 22 20 3:0+14 comment 23 21 4:0+11 comment 24 22 5:0+10 class 25 -5:11+1 operator 23 +5:11+2 operator 26 24 5:13+6 string 27 -5:20+5 string 28 -6:2+8 string 29 -6:10+1 string 30 -6:11+4 string 25 +5:20+5 property 26 +6:2+13 namespace 31 27 6:17+1 type 32 28 6:18+5 number 33 29 6:24+1 operator 34 30 6:26+1 number 35 31 6:28+3 type 36 -7:2+6 string 37 -7:8+1 string 38 -7:9+4 string 39 -7:15+2 string 32 +7:2+15 namespace 40 33 8:0+1 operator 41 34 8:2+4 string 42 -9:0+1 string 43 -9:2+7 string 35 +9:0+1 operator 36 +9:2+7 property 44 37 10:0+7 keyword 45 -10:8+3 string 38 +10:8+3 namespace 46 39 11:0+1 keyword 47 -11:2+1 type 40 +11:2+1 property 48 41 12:0+1 keyword 49 42 12:2+10 class 50 43 12:13+3 type 51 44 12:17+4 number 52 -13:0+7 string 53 -13:8+8 string
A
internal/lsp/testdata/semantic-with-errors.txtar
··· 1 +-- in.journal -- 2 +2026-01-04 salary 3 + income:salary $543.22 4 + assets:bank $-543.22 5 + 6 +some broken input 7 + 8 +-- expect -- 9 +0:0+10 class 10 +0:11+6 property 11 +1:2+13 namespace 12 +1:17+1 type 13 +1:18+6 number 14 +2:2+11 namespace 15 +2:17+1 type 16 +2:18+7 number negative
M
internal/lsp/textdocument_semantic.go
··· 64 64 // Implementation 65 65 66 66 const ( 67 - semDirective = iota 67 + semDirective uint32 = iota 68 68 semDate 69 69 semAccount 70 70 semCommodity ··· 121 121 } 122 122 raw = append(raw, rawSpan{s, tokType, mods}) 123 123 } 124 - if j == nil || len(j.Errors) > 0 { 125 - semLexerFallback(content, emit) 126 - } else { 127 - for _, e := range j.Entries { 128 - visitEntry(content, e, emit) 129 - } 124 + for _, e := range j.Entries { 125 + visitEntry(content, e, emit) 126 + } 127 + if len(j.Errors) > 0 { 128 + // parser recovers per line; lexer fills the unparsed regions, keeping ast tokens where the parser succeeded 129 + semLexerFallback(content, raw, emit) 130 130 } 131 131 return rawToSemanticTokens(content, raw) 132 132 } ··· 522 522 } 523 523 } 524 524 525 -// semLexerFallback produces semantic tokens using only the lexer (for unparseable documents). 526 -func semLexerFallback(content string, emit semEmitFn) { 525 +func semLexerFallback(content string, base []rawSpan, emit semEmitFn) { 527 526 l := lexer.New("", []byte(content)) 528 527 529 528 var commentStart, commentEnd int // 0 = not inside a comment line 529 + lineStart := true // the next significant token starts a line 530 + skipLine := false // the line starts with an unclassifiable token; emit nothing 531 + i := 0 // next base span to compare against 532 + 530 533 take := func(span token.Span, tokType uint32, mods uint32) { 534 + for i < len(base) && base[i].span.End.Offset <= span.Start.Offset { 535 + i++ 536 + } 537 + if i < len(base) && base[i].span.Start.Offset < span.End.Offset { 538 + return // overlaps an AST token; AST wins 539 + } 531 540 emit(span, tokType, mods) 532 541 } 542 + 533 543 for { 534 544 tok := l.Next() 535 545 if tok.Type == token.EOF { ··· 543 553 take(token.Span{Start: offsetPos("", commentStart), End: offsetPos("", commentEnd)}, semComment, 0) 544 554 commentStart, commentEnd = 0, 0 545 555 } 556 + lineStart, skipLine = true, false 546 557 continue 547 558 } 548 559 if tok.Type == token.WHITESPACE || tok.Type == token.INDENT { 549 560 continue 550 561 } 551 - tokType := uint32(semString) 562 + if lineStart { 563 + lineStart = false 564 + if !isLineStartToken(tok.Type) { 565 + skipLine = true 566 + } 567 + } 568 + if skipLine { 569 + continue 570 + } 571 + tokType := semProperty 552 572 if commentStart > 0 { 553 573 tokType = semComment 554 574 if tok.Span.End.Offset > commentEnd { ··· 556 576 } 557 577 continue 558 578 } 579 + 559 580 switch tok.Type { 560 - case token.SEMICOLON, token.HASH, token.PERCENT: 581 + case token.SEMICOLON, token.HASH, token.PERCENT, token.STAR: 561 582 tokType = semComment 562 583 commentStart = tok.Span.Start.Offset 563 584 commentEnd = tok.Span.End.Offset 564 585 continue 565 - case token.STAR: 566 - tokType = semComment // * at col 0 is comment marker 567 - commentStart = tok.Span.Start.Offset 568 - commentEnd = tok.Span.End.Offset 569 - continue 570 - case token.ACCOUNT, token.COMMODITY, token.INCLUDE, token.ALIAS, 571 - token.PAYEE, token.TAG, token.APPLY, token.END, token.COMMENTKW, 572 - token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C: 573 - tokType = semDirective 586 + case token.STRING: 587 + tokType = semString 574 588 case token.DATE: 575 589 tokType = semDate 576 590 case token.INT, token.DECIMAL: ··· 581 595 tokType = semStatus 582 596 case token.AT, token.ATAT, token.EQ, token.EQEQ, token.EQEQEQ, token.EQSTAR: 583 597 tokType = semOperator 598 + case token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE, 599 + token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END, 600 + token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C: 601 + tokType = semDirective 584 602 } 585 603 take(tok.Span, tokType, 0) 586 604 } 605 +} 606 + 607 +func isLineStartToken(t token.Type) bool { 608 + switch t { 609 + case token.DATE, token.TILDE, token.EQ, token.BANG, token.AT, 610 + token.SEMICOLON, token.HASH, token.PERCENT, token.STAR, 611 + token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE, 612 + token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END, 613 + token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C: 614 + return true 615 + } 616 + return false 587 617 } 588 618 589 619 func encodeSemTokens(tokens []semanticToken) []uint32 {