clerk/internal/lsp/textdocument_semantic_test.go (view raw)
Oleksandr Smirnov
Oleksandr Smirnov
olexsmir@gmail.com lsp: implement hybrid(ast and lexer at the same time) semantic tokenizer, 2 months ago
olexsmir@gmail.com lsp: implement hybrid(ast and lexer at the same time) semantic tokenizer, 2 months ago
| 1 | package lsp |
| 2 | |
| 3 | import ( |
| 4 | "fmt" |
| 5 | "slices" |
| 6 | "strings" |
| 7 | "testing" |
| 8 | |
| 9 | "go.lsp.dev/protocol" |
| 10 | "go.lsp.dev/uri" |
| 11 | |
| 12 | "olexsmir.xyz/clerk/internal/testutil/golden" |
| 13 | ) |
| 14 | |
| 15 | func TestEncodeSemTokens(t *testing.T) { |
| 16 | tests := map[string]struct { |
| 17 | tokens []semanticToken |
| 18 | want []uint32 |
| 19 | }{ |
| 20 | "nil": {nil, nil}, |
| 21 | "empty": {[]semanticToken{}, nil}, |
| 22 | "single": {[]semanticToken{ |
| 23 | {line: 0, col: 0, length: 4, tokenType: semDate}, |
| 24 | }, []uint32{0, 0, 4, semDate, 0}}, |
| 25 | "line 0 col 0": {[]semanticToken{ |
| 26 | {line: 0, col: 0, length: 1, tokenType: semDirective}, |
| 27 | }, []uint32{0, 0, 1, semDirective, 0}}, |
| 28 | "multiple": {[]semanticToken{ |
| 29 | {line: 0, col: 0, length: 10, tokenType: semDate}, |
| 30 | {line: 0, col: 11, length: 5, tokenType: semString}, |
| 31 | {line: 1, col: 4, length: 10, tokenType: semAccount}, |
| 32 | }, []uint32{ |
| 33 | 0, 0, 10, semDate, 0, |
| 34 | 0, 11, 5, semString, 0, |
| 35 | 1, 4, 10, semAccount, 0, |
| 36 | }}, |
| 37 | // the wire format requires non-negative deltas; unsorted input must |
| 38 | // be sorted first (commodity at col 36 comes after amount at col 32) |
| 39 | "unsorted input": {[]semanticToken{ |
| 40 | {line: 0, col: 0, length: 10, tokenType: semDate}, |
| 41 | {line: 0, col: 36, length: 3, tokenType: semCommodity}, |
| 42 | {line: 0, col: 32, length: 2, tokenType: semAmount}, |
| 43 | {line: 1, col: 4, length: 6, tokenType: semAccount}, |
| 44 | }, []uint32{ |
| 45 | 0, 0, 10, semDate, 0, |
| 46 | 0, 32, 2, semAmount, 0, |
| 47 | 0, 4, 3, semCommodity, 0, |
| 48 | 1, 4, 6, semAccount, 0, |
| 49 | }}, |
| 50 | } |
| 51 | |
| 52 | for tname, tt := range tests { |
| 53 | t.Run(tname, func(t *testing.T) { |
| 54 | if got := encodeSemTokens(tt.tokens); !slices.Equal(got, tt.want) { |
| 55 | t.Errorf("encodeSemTokens() = %v, want %v", got, tt.want) |
| 56 | } |
| 57 | }) |
| 58 | } |
| 59 | } |
| 60 | |
| 61 | func TestServer_Semantic_SimpleTransaction(t *testing.T) { |
| 62 | content := `2024-01-15 test |
| 63 | expenses:food $50 |
| 64 | assets:cash |
| 65 | ` |
| 66 | |
| 67 | srv := NewServer("test") |
| 68 | srv.server.openDoc(uri.URI("file:///test.journal"), content, 1, "journal") |
| 69 | |
| 70 | result, err := srv.server.SemanticTokensFull(t.Context(), &protocol.SemanticTokensParams{ |
| 71 | TextDocument: protocol.TextDocumentIdentifier{URI: uri.URI("file:///test.journal")}, |
| 72 | }) |
| 73 | if err != nil { |
| 74 | t.Fatal(err) |
| 75 | } |
| 76 | if result == nil { |
| 77 | t.Fatal("result is nil") |
| 78 | } |
| 79 | if len(result.Data) == 0 { |
| 80 | t.Fatal("expected non-empty token data") |
| 81 | } |
| 82 | if len(result.Data)%5 != 0 { |
| 83 | t.Fatalf("token data length %d is not a multiple of 5", len(result.Data)) |
| 84 | } |
| 85 | // first token is the transaction date at line 0, col 0: deltas are 0, 0 |
| 86 | if result.Data[0] != 0 || result.Data[1] != 0 { |
| 87 | t.Errorf("first token deltas = %d,%d, want 0,0", result.Data[0], result.Data[1]) |
| 88 | } |
| 89 | } |
| 90 | |
| 91 | func TestServer_Semantic_EmptyDocument(t *testing.T) { |
| 92 | srv := NewServer("test") |
| 93 | srv.server.openDoc(uri.URI("file:///empty.journal"), "", 1, "journal") |
| 94 | |
| 95 | result, err := srv.server.SemanticTokensFull(t.Context(), &protocol.SemanticTokensParams{ |
| 96 | TextDocument: protocol.TextDocumentIdentifier{URI: uri.URI("file:///empty.journal")}, |
| 97 | }) |
| 98 | if err != nil { |
| 99 | t.Fatal(err) |
| 100 | } |
| 101 | if len(result.Data) != 0 { |
| 102 | t.Errorf("expected empty data for empty doc, got %d values", len(result.Data)) |
| 103 | } |
| 104 | } |
| 105 | |
| 106 | func TestServer_Semantic_DocumentNotFound(t *testing.T) { |
| 107 | result, err := NewServer("test").server.SemanticTokensFull(t.Context(), &protocol.SemanticTokensParams{ |
| 108 | TextDocument: protocol.TextDocumentIdentifier{URI: uri.URI("file:///unknown.journal")}, |
| 109 | }) |
| 110 | if err != nil { |
| 111 | t.Fatal(err) |
| 112 | } |
| 113 | if result == nil { |
| 114 | t.Fatal("result is nil") |
| 115 | } |
| 116 | if result.Data != nil { |
| 117 | t.Errorf("expected nil Data for unknown doc, got %v", result.Data) |
| 118 | } |
| 119 | } |
| 120 | |
| 121 | func TestServer_Semantic_Range(t *testing.T) { |
| 122 | content := `2024-01-15 test |
| 123 | expenses:food $50 |
| 124 | |
| 125 | 2024-01-16 other |
| 126 | expenses:drinks $20 |
| 127 | ` |
| 128 | |
| 129 | srv := NewServer("test") |
| 130 | srv.server.openDoc(uri.URI("file:///test.journal"), content, 1, "journal") |
| 131 | |
| 132 | result, err := srv.server.SemanticTokensRange(t.Context(), &protocol.SemanticTokensRangeParams{ |
| 133 | TextDocument: protocol.TextDocumentIdentifier{URI: uri.URI("file:///test.journal")}, |
| 134 | Range: protocol.Range{ |
| 135 | Start: protocol.Position{Line: 0, Character: 0}, |
| 136 | End: protocol.Position{Line: 1, Character: 50}, |
| 137 | }, |
| 138 | }) |
| 139 | if err != nil { |
| 140 | t.Fatal(err) |
| 141 | } |
| 142 | if result == nil || len(result.Data) == 0 { |
| 143 | t.Fatal("expected non-empty tokens for range") |
| 144 | } |
| 145 | |
| 146 | // decode and assert every token is inside the requested line range |
| 147 | line, col := 0, 0 |
| 148 | for i := 0; i+4 < len(result.Data); i += 5 { |
| 149 | dl, dc := result.Data[i], result.Data[i+1] |
| 150 | if dl > 0 { |
| 151 | col = 0 |
| 152 | } |
| 153 | line += int(dl) |
| 154 | col += int(dc) |
| 155 | if line > 1 { |
| 156 | t.Fatalf("token at line %d outside requested range [0,1]", line) |
| 157 | } |
| 158 | } |
| 159 | } |
| 160 | |
| 161 | // Golden |
| 162 | |
| 163 | func TestSemanticTokensTxtar(t *testing.T) { |
| 164 | tests := []string{ |
| 165 | "semantic-empty", |
| 166 | "semantic-journal", |
| 167 | "semantic-directives", |
| 168 | "semantic-unparseable", |
| 169 | "semantic-with-errors", |
| 170 | } |
| 171 | |
| 172 | for _, tt := range tests { |
| 173 | ar := golden.Read(t, tt) |
| 174 | |
| 175 | t.Run(tt+"_golden", func(t *testing.T) { |
| 176 | toks := renderSemanticTokens(tokSem(ar.Get("in.journal"))) |
| 177 | golden.Assert(t, ar, toks) |
| 178 | }) |
| 179 | |
| 180 | t.Run(tt+"_no-overlap", func(t *testing.T) { |
| 181 | toks := tokSem(ar.Get("in.journal")) |
| 182 | slices.SortFunc(toks, func(a, b semanticToken) int { |
| 183 | if a.line != b.line { |
| 184 | return int(a.line) - int(b.line) |
| 185 | } |
| 186 | return int(a.col) - int(b.col) |
| 187 | }) |
| 188 | |
| 189 | for i := 1; i < len(toks); i++ { |
| 190 | prev, cur := toks[i-1], toks[i] |
| 191 | if prev.line != cur.line { |
| 192 | continue |
| 193 | } |
| 194 | if cur.col < prev.col+prev.length { |
| 195 | t.Errorf("%s: overlapping tokens on line %d: %s@%d+%d then %s@%d+%d", |
| 196 | tt, prev.line, tokenTypeStrings[prev.tokenType], prev.col, prev.length, |
| 197 | tokenTypeStrings[cur.tokenType], cur.col, cur.length) |
| 198 | } |
| 199 | } |
| 200 | }) |
| 201 | } |
| 202 | } |
| 203 | |
| 204 | func renderSemanticTokens(tokens []semanticToken) string { |
| 205 | slices.SortFunc(tokens, func(a, b semanticToken) int { |
| 206 | if a.line != b.line { |
| 207 | return int(a.line) - int(b.line) |
| 208 | } |
| 209 | return int(a.col) - int(b.col) |
| 210 | }) |
| 211 | var b strings.Builder |
| 212 | for _, tok := range tokens { |
| 213 | fmt.Fprintf(&b, "%d:%d+%d %s", tok.line, tok.col, tok.length, tokenTypeStrings[tok.tokenType]) |
| 214 | for i, m := range modifierStrings { |
| 215 | if tok.modifiers&(1<<uint(i)) != 0 { |
| 216 | b.WriteByte(' ') |
| 217 | b.WriteString(m) |
| 218 | } |
| 219 | } |
| 220 | b.WriteByte('\n') |
| 221 | } |
| 222 | return b.String() |
| 223 | } |
| 224 | |
| 225 | func tokSem(content []byte) []semanticToken { |
| 226 | c := string(content) |
| 227 | return tokenizeForSemantics(c, parseJournalStr(c)) |
| 228 | } |