clerk/internal/lsp/lsputil/utf16_test.go (view raw)
| 1 | package lsputil |
| 2 | |
| 3 | import ( |
| 4 | "testing" |
| 5 | "unicode/utf8" |
| 6 | ) |
| 7 | |
| 8 | func TestUtf16Col_Basic(t *testing.T) { |
| 9 | col := Utf16Col("hello", 3) |
| 10 | if col != 3 { |
| 11 | t.Errorf("Utf16Col for ASCII = %d, want 3", col) |
| 12 | } |
| 13 | } |
| 14 | |
| 15 | func TestUtf16Col_Cyrillic(t *testing.T) { |
| 16 | // "Привет" is 6 Cyrillic chars = 12 bytes, 6 UTF-16 code units |
| 17 | line := "Привет" |
| 18 | col := Utf16Col(line, len(line)) |
| 19 | if col != 6 { |
| 20 | t.Errorf("Utf16Col for 'Привет' = %d, want 6", col) |
| 21 | } |
| 22 | } |
| 23 | |
| 24 | func TestUtf16Len_ASCII(t *testing.T) { |
| 25 | l := Utf16Len("hello world", 0, 5) |
| 26 | if l != 5 { |
| 27 | t.Errorf("Utf16Len = %d, want 5", l) |
| 28 | } |
| 29 | } |
| 30 | |
| 31 | func TestUtf16Len_Cyrillic(t *testing.T) { |
| 32 | // "Привет" = 6 UTF-16 code units |
| 33 | l := Utf16Len("Привет мир", 0, 12) |
| 34 | if l != 6 { |
| 35 | t.Errorf("Utf16Len for 'Привет' = %d, want 6", l) |
| 36 | } |
| 37 | } |
| 38 | |
| 39 | func TestLineCol_Basic(t *testing.T) { |
| 40 | line, col := LineCol("hello\nworld", 11) |
| 41 | if line != 1 || col != 5 { |
| 42 | t.Errorf("LineCol = (%d,%d), want (1,5)", line, col) |
| 43 | } |
| 44 | } |
| 45 | |
| 46 | func TestLineCol_FirstLine(t *testing.T) { |
| 47 | line, col := LineCol("hello world", 5) |
| 48 | if line != 0 || col != 5 { |
| 49 | t.Errorf("LineCol = (%d,%d), want (0,5)", line, col) |
| 50 | } |
| 51 | } |
| 52 | |
| 53 | func TestLineCol_LineStart(t *testing.T) { |
| 54 | // offset at the first char of line 1 must be (1,0), not the tail of line 0 |
| 55 | line, col := LineCol("ab\ncd", 3) |
| 56 | if line != 1 || col != 0 { |
| 57 | t.Errorf("LineCol(3) = (%d,%d), want (1,0)", line, col) |
| 58 | } |
| 59 | // newline char itself belongs to the line it ends |
| 60 | line, col = LineCol("ab\ncd", 2) |
| 61 | if line != 0 || col != 2 { |
| 62 | t.Errorf("LineCol(2) = (%d,%d), want (0,2)", line, col) |
| 63 | } |
| 64 | } |
| 65 | |
| 66 | func TestLineCol_BlankLine(t *testing.T) { |
| 67 | // "a\n\nb": line 1 is blank; offset 3 is the start of line 2 |
| 68 | line, col := LineCol("a\n\nb", 3) |
| 69 | if line != 2 || col != 0 { |
| 70 | t.Errorf("LineCol(3) = (%d,%d), want (2,0)", line, col) |
| 71 | } |
| 72 | line, col = LineCol("a\n\nb", 2) |
| 73 | if line != 1 || col != 0 { |
| 74 | t.Errorf("LineCol(2) = (%d,%d), want (1,0)", line, col) |
| 75 | } |
| 76 | } |
| 77 | |
| 78 | func TestLineCol_CRLF(t *testing.T) { |
| 79 | line, col := LineCol("ab\r\ncd", 4) |
| 80 | if line != 1 || col != 0 { |
| 81 | t.Errorf("LineCol(4) = (%d,%d), want (1,0)", line, col) |
| 82 | } |
| 83 | line, col = LineCol("ab\r\ncd", 2) |
| 84 | if line != 0 || col != 2 { |
| 85 | t.Errorf("LineCol(2) = (%d,%d), want (0,2)", line, col) |
| 86 | } |
| 87 | } |
| 88 | |
| 89 | func TestOffset_Basic(t *testing.T) { |
| 90 | if got := Offset("hello\nworld", 1, 5); got != 11 { |
| 91 | t.Errorf("Offset(1,5) = %d, want 11", got) |
| 92 | } |
| 93 | } |
| 94 | |
| 95 | func TestOffset_RoundTrip(t *testing.T) { |
| 96 | content := "first\nПривет мир\r\nlast\n" |
| 97 | // only rune-boundary offsets round-trip (LineCol clamps mid-rune offsets) |
| 98 | var boundaries []int |
| 99 | for i := 0; i < len(content); { |
| 100 | _, size := utf8.DecodeRuneInString(content[i:]) |
| 101 | if !(content[i] == '\n' && i > 0 && content[i-1] == '\r') { |
| 102 | boundaries = append(boundaries, i) |
| 103 | } |
| 104 | i += size |
| 105 | } |
| 106 | for _, off := range boundaries { |
| 107 | line, col := LineCol(content, off) |
| 108 | if got := Offset(content, line, col); got != off { |
| 109 | t.Errorf("round trip at %d: Offset(LineCol(%d)) = %d", off, off, got) |
| 110 | } |
| 111 | } |
| 112 | } |
| 113 | |
| 114 | func TestOffset_Clamps(t *testing.T) { |
| 115 | content := "ab\ncd" |
| 116 | if got := Offset(content, 5, 0); got != len(content) { |
| 117 | t.Errorf("past EOF line: Offset = %d, want %d", got, len(content)) |
| 118 | } |
| 119 | // column beyond line end clamps to line end |
| 120 | if got := Offset(content, 0, 99); got != 2 { |
| 121 | t.Errorf("past EOL col: Offset = %d, want 2", got) |
| 122 | } |
| 123 | } |
| 124 | |
| 125 | func TestOffset_UTF16(t *testing.T) { |
| 126 | // Cyrillic chars are 1 UTF-16 unit each; emoji are 2 |
| 127 | content := "Привет😀" |
| 128 | if got := Offset(content, 0, 6); got != 12 { |
| 129 | t.Errorf("Offset after Cyrillic = %d, want 12", got) |
| 130 | } |
| 131 | if got := Offset(content, 0, 7); got != 16 { |
| 132 | t.Errorf("Offset after emoji (2 UTF-16 units) = %d, want 16", got) |
| 133 | } |
| 134 | } |
| 135 | |
| 136 | func TestPosition_RoundTrip(t *testing.T) { |
| 137 | content := "2024-01-15 Супермаркет\r\n Витрати:Продукти ¥50\n" |
| 138 | for _, off := range []int{0, 10, len("2024-01-15 Супермаркет"), len(content)} { |
| 139 | p := Position(content, off) |
| 140 | if got := Offset(content, int(p.Line), int(p.Character)); got != off { |
| 141 | t.Errorf("round trip of %d = %d (pos %d:%d), want identity", off, got, p.Line, p.Character) |
| 142 | } |
| 143 | } |
| 144 | } |