all repos

clerk @ a925423

missing tooling for ledger/hledger

clerk/journal/lexer/lexer.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
add support of subdirectives and it's highlights, 1 month ago
1
package lexer
2
3
import (
4
	"strings"
5
	"unicode"
6
	"unicode/utf8"
7
8
	"olexsmir.xyz/clerk/journal/token"
9
)
10
11
type mode uint
12
13
const (
14
	// start of a line, nothing consumed
15
	modeDefault mode = iota
16
17
	// after ; # * % ; at start of line, or anywhere inline
18
	// everything until \n is comment text
19
	modeComment
20
21
	// after lexing a date at column 0
22
	// expects: optional status, optional code, description, comment
23
	modeTransaction
24
25
	// after lexing an indent at start of line
26
	// expects: account name, then two spaces, then amount
27
	modePosting
28
29
	// after ~, period expression
30
	// expects: period, optional description (after 2+ spaces), optional comment
31
	modePeriodic
32
33
	// after =, automates transaction
34
	// expects: expression
35
	modeAutomated
36
37
	// after a directive keyword like account, commodity, include
38
	// expects: rest of directive content
39
	modeDirective
40
)
41
42
type Lexer struct {
43
	file  string
44
	input []byte
45
	mode  mode
46
47
	ch     rune // current rune (0 = EOF/sentinel)
48
	chSize int  // byte size of current rune
49
	pos    int  // current byte offset (points at ch)
50
	rpos   int  // next byte offset to read (one ahead of pos)
51
	col    int  // current column (1-based)
52
	line   int  // current line (1-based)
53
54
	transactionPastStatus bool
55
	postingExpectAccount  bool
56
	readingNoteAfterPipe  bool
57
58
	includePath bool // true after [token.INCLUDE] is omitted
59
60
	// subdirective is set while inside an account/commodity directive block;
61
	// every indented line then lexes as directive content until a
62
	// non-indented line start clears it in lexDefault
63
	subdirective bool
64
}
65
66
func New(file string, input []byte) *Lexer {
67
	l := &Lexer{
68
		file:  file,
69
		input: input,
70
		line:  1,
71
	}
72
	l.advance()
73
	if l.ch == '\uFEFF' { // start of the input
74
		l.advance()
75
	}
76
	return l
77
}
78
79
// Next returns next token in the input
80
func (l *Lexer) Next() token.Token {
81
	switch l.mode {
82
	case modeDefault:
83
		return l.lexDefault()
84
	case modeComment:
85
		return l.lexComment()
86
	case modeTransaction:
87
		return l.lexTransaction()
88
	case modePosting:
89
		return l.lexPosting()
90
	case modePeriodic:
91
		return l.lexPeriodic()
92
	case modeAutomated:
93
		return l.lexAutomated()
94
	case modeDirective:
95
		return l.lexDirective()
96
	}
97
	panic("unreachable")
98
}
99
100
func (l *Lexer) lexDefault() token.Token {
101
	if l.includePath {
102
		l.includePath = false
103
	}
104
	if l.subdirective && l.ch != ' ' && l.ch != '\t' {
105
		l.subdirective = false
106
	}
107
	switch {
108
	case l.ch == 0:
109
		return l.token(token.EOF, "")
110
	case l.ch == '\n':
111
		return l.lexNewline()
112
	case l.ch == '\r':
113
		l.col = 0
114
		l.advance()
115
		return l.lexNewline()
116
	case l.ch == ' ' || l.ch == '\t':
117
		tok := l.lexIndent()
118
		if l.subdirective {
119
			// keep directive mode for every line of a directive block; the
120
			// flag is cleared at the next non-indented line start in lexDefault
121
			l.mode = modeDirective
122
		} else {
123
			l.mode = modePosting
124
		}
125
		l.postingExpectAccount = true
126
		return tok
127
	case l.ch == ';' || l.ch == '#' || l.ch == '%':
128
		l.mode = modeComment
129
		return l.lexSingle(token.SEMICOLON) // todo: ??
130
	case l.ch == '*': // * at col 0 == comment
131
		l.mode = modeComment
132
		return l.lexSingle(token.STAR)
133
	case l.ch == '~':
134
		l.mode = modePeriodic
135
		return l.lexSingle(token.TILDE)
136
	case l.ch == '=':
137
		l.mode = modeAutomated
138
		return l.lexSingle(token.EQ)
139
	case l.ch == '+':
140
		return l.lexSingle(token.PLUS)
141
	case l.ch == '-':
142
		return l.lexSingle(token.MINUS)
143
	case l.ch == '.':
144
		return l.lexSingle(token.TEXT)
145
	case l.ch == '!':
146
		return l.lexSingle(token.BANG)
147
	case l.ch == '@':
148
		return l.lexSingle(token.AT)
149
	case l.isAlpha():
150
		return l.lexKeyword()
151
	case l.isDigit():
152
		if !l.isDate() {
153
			s := l.save()
154
			for l.isDigit() || l.ch == '-' || l.ch == '/' || l.ch == '.' {
155
				l.advance()
156
			}
157
			return token.Token{Type: token.ILLEGAL, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
158
		}
159
		tok := l.lexDate()
160
		l.mode = modeTransaction
161
		l.transactionPastStatus = false
162
		return tok
163
	default:
164
		s := l.save()
165
		l.advance()
166
		return token.Token{Type: token.ILLEGAL, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
167
	}
168
}
169
170
func (l *Lexer) lexComment() token.Token {
171
	if l.ch == '\n' || l.ch == '\r' || l.ch == 0 {
172
		l.mode = modeDefault
173
		if l.ch == '\r' {
174
			l.col = 0
175
			l.advance()
176
		}
177
		return l.lexNewline()
178
	}
179
180
	for l.ch == ' ' || l.ch == '\t' {
181
		l.lexWhitespace()
182
	}
183
184
	if l.ch == '\n' || l.ch == '\r' || l.ch == 0 {
185
		l.mode = modeDefault
186
		if l.ch == '\r' {
187
			l.col = 0
188
			l.advance()
189
		}
190
		return l.lexNewline()
191
	}
192
193
	s := l.save()
194
	for l.ch != '\n' && l.ch != '\r' && l.ch != 0 {
195
		l.advance()
196
	}
197
	return token.Token{Type: token.TEXT, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
198
}
199
200
func (l *Lexer) lexTransaction() token.Token {
201
	if l.readingNoteAfterPipe {
202
		l.readingNoteAfterPipe = false
203
		return l.lexNote()
204
	}
205
206
	switch l.ch {
207
	case 0:
208
		return l.token(token.EOF, "")
209
	case '\n':
210
		l.mode = modeDefault
211
		return l.lexNewline()
212
	case '\r':
213
		l.col = 0
214
		l.advance()
215
		return l.lexNewline()
216
	case ' ', '\t':
217
		return l.lexWhitespace()
218
	case ';':
219
		l.mode = modeComment
220
		return l.lexSingle(token.SEMICOLON)
221
	case '*':
222
		if !l.transactionPastStatus {
223
			l.transactionPastStatus = true
224
			return l.lexSingle(token.STAR)
225
		}
226
		return l.lexText()
227
	case '!':
228
		if !l.transactionPastStatus {
229
			l.transactionPastStatus = true
230
			return l.lexSingle(token.BANG)
231
		}
232
		return l.lexText()
233
	case '|':
234
		l.transactionPastStatus = true
235
		l.readingNoteAfterPipe = true
236
		return l.lexSingle(token.PIPE)
237
	case '+':
238
		return l.lexSingle(token.PLUS)
239
	case '-':
240
		return l.lexSingle(token.MINUS)
241
	case '=':
242
		return l.lexEquals()
243
	case '"', '\'':
244
		return l.lexString()
245
	default: // description / payee
246
		if l.isDate() { // secondsry date after =
247
			return l.lexDate()
248
		}
249
		return l.lexText()
250
	}
251
}
252
253
func (l *Lexer) lexNote() token.Token {
254
	for l.ch == ' ' || l.ch == '\t' {
255
		l.advance()
256
	}
257
	if l.ch == 0 || l.ch == '\n' || l.ch == '\r' || l.ch == ';' {
258
		return l.Next()
259
	}
260
	s := l.save()
261
	for l.ch != 0 && l.ch != '\n' && l.ch != '\r' && l.ch != ';' {
262
		l.advance()
263
	}
264
	lit := string(l.input[s.offset:l.pos])
265
	for len(lit) > 0 && (lit[len(lit)-1] == ' ' || lit[len(lit)-1] == '\t') {
266
		lit = lit[:len(lit)-1]
267
	}
268
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
269
}
270
271
func (l *Lexer) lexPeriodic() token.Token {
272
	switch l.ch {
273
	case 0:
274
		return l.token(token.EOF, "")
275
	case '\n':
276
		l.mode = modeDefault
277
		return l.lexNewline()
278
	case '\r':
279
		l.col = 0
280
		l.advance()
281
		return l.lexNewline()
282
	case ';':
283
		l.mode = modeComment
284
		return l.lexSingle(token.SEMICOLON)
285
	case ' ', '\t':
286
		return l.lexWhitespace()
287
	default:
288
		return l.lexText()
289
	}
290
}
291
292
func (l *Lexer) lexAutomated() token.Token {
293
	switch l.ch {
294
	case 0:
295
		return l.token(token.EOF, "")
296
	case '\n':
297
		l.mode = modeDefault
298
		return l.lexNewline()
299
	case '\r':
300
		l.col = 0
301
		l.advance()
302
		return l.lexNewline()
303
	case ' ', '\t':
304
		return l.lexWhitespace()
305
	case ';':
306
		l.mode = modeComment
307
		return l.lexSingle(token.SEMICOLON)
308
	default:
309
		return l.lexText()
310
	}
311
}
312
313
func (l *Lexer) lexPosting() token.Token {
314
	switch {
315
	case l.ch == 0:
316
		l.postingExpectAccount = false
317
		return l.token(token.EOF, "")
318
	case l.ch == '\n':
319
		l.postingExpectAccount = false
320
		l.mode = modeDefault
321
		return l.lexNewline()
322
	case l.ch == '\r':
323
		l.postingExpectAccount = false
324
		l.col = 0
325
		l.advance()
326
		l.mode = modeDefault
327
		return l.lexNewline()
328
	case l.ch == ';':
329
		l.postingExpectAccount = false
330
		l.mode = modeComment
331
		return l.lexSingle(token.SEMICOLON)
332
	case l.ch == ' ' || l.ch == '\t':
333
		return l.lexWhitespace()
334
	case l.postingExpectAccount && l.ch == '*':
335
		return l.lexSingle(token.STAR)
336
	case l.postingExpectAccount && l.ch == '!':
337
		return l.lexSingle(token.BANG)
338
	case l.ch == '=':
339
		return l.lexEquals()
340
	case l.ch == '@':
341
		return l.lexAt()
342
	case l.ch == '{':
343
		return l.lexLBrace()
344
	case l.ch == '}':
345
		return l.lexRBrace()
346
	case l.ch == '(':
347
		if !l.postingExpectAccount {
348
			return l.lexParenExpr()
349
		}
350
		return l.lexSingle(token.LPAREN)
351
	case l.ch == ')':
352
		return l.lexSingle(token.RPAREN)
353
	case l.ch == '[':
354
		return l.lexSingle(token.LBRACKET)
355
	case l.ch == ']':
356
		return l.lexSingle(token.RBRACKET)
357
	case l.ch == ':':
358
		return l.lexSingle(token.COLON)
359
	case l.postingExpectAccount && l.ch != '*' && l.ch != '!' && l.ch != '(' && l.ch != '[':
360
		return l.lexAccountNamePosting()
361
	case l.ch == '*': // after account name
362
		return l.lexSingle(token.STAR)
363
	case l.isDigit(), l.ch == '.':
364
		return l.lexNumber()
365
	case l.ch == '-':
366
		return l.lexSingle(token.MINUS)
367
	case l.ch == '+':
368
		return l.lexSingle(token.PLUS)
369
	case l.isCommodityStart():
370
		return l.lexCommodityMark()
371
	case l.ch == '"' || l.ch == '\'':
372
		return l.lexString()
373
	case l.ch >= 'a' && l.ch <= 'z':
374
		return l.lexCommodityMark()
375
	default:
376
		return l.lexAccountNamePosting()
377
	}
378
}
379
380
func (l *Lexer) lexDirective() token.Token {
381
	switch {
382
	case l.ch == '\n', l.ch == 0:
383
		l.mode = modeDefault
384
		return l.lexNewline()
385
	case l.ch == '\r':
386
		l.mode = modeDefault
387
		l.col = 0
388
		l.advance()
389
		return l.lexNewline()
390
	case l.ch == ';':
391
		l.mode = modeComment
392
		return l.lexSingle(token.SEMICOLON)
393
	case l.ch == ' ', l.ch == '\t':
394
		return l.lexWhitespace()
395
	case l.ch == '=':
396
		return l.lexSingle(token.EQ)
397
	case l.ch == '+':
398
		return l.lexSingle(token.PLUS)
399
	case l.ch == '-':
400
		return l.lexSingle(token.MINUS)
401
	case l.ch == '"', l.ch == '\'':
402
		return l.lexString()
403
	case l.ch == ':':
404
		return l.lexSingle(token.COLON)
405
	case l.ch == ')':
406
		return l.lexSingle(token.RPAREN)
407
	case l.ch == ']':
408
		return l.lexSingle(token.RBRACKET)
409
	case l.includePath:
410
		return l.lexPath()
411
	case l.isCommodityStart():
412
		return l.lexCommodityMark()
413
	case l.isTime():
414
		return l.lexTime()
415
	case l.isDate():
416
		return l.lexDate()
417
	case l.isDigit():
418
		return l.lexNumber()
419
	default:
420
		return l.lexAccountNameDirective()
421
	}
422
}
423
424
func (l *Lexer) lexPath() token.Token {
425
	s := l.save()
426
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ' ' && l.ch != '\t' {
427
		l.advance()
428
	}
429
	return token.Token{
430
		Type:    token.TEXT,
431
		Literal: string(l.input[s.offset:l.pos]),
432
		Span:    l.span(s),
433
	}
434
}
435
436
func (l *Lexer) lexSingle(kind token.Type) token.Token {
437
	s := l.save()
438
	l.advance()
439
	return token.Token{
440
		Type:    kind,
441
		Literal: string(l.input[s.offset:l.pos]),
442
		Span:    l.span(s),
443
	}
444
}
445
446
func (l *Lexer) lexNewline() token.Token {
447
	s := l.save()
448
	l.advance()
449
	l.mode = modeDefault
450
	return token.Token{Type: token.NEWLINE, Literal: "\n", Span: l.span(s)}
451
}
452
453
func (l *Lexer) lexWhitespace() token.Token {
454
	s := l.save()
455
	for l.ch == ' ' || l.ch == '\t' {
456
		l.advance()
457
	}
458
	lit := string(l.input[s.offset:l.pos])
459
	return token.Token{Type: token.WHITESPACE, Literal: lit, Span: l.span(s)}
460
}
461
462
func (l *Lexer) lexIndent() token.Token {
463
	s := l.save()
464
	for l.ch == ' ' || l.ch == '\t' {
465
		l.advance()
466
	}
467
	lit := string(l.input[s.offset:l.pos])
468
	return token.Token{Type: token.INDENT, Literal: lit, Span: l.span(s)}
469
}
470
471
func (l *Lexer) lexEquals() token.Token {
472
	s := l.save()
473
	l.advance()
474
	if l.ch == '=' {
475
		l.advance()
476
		switch l.ch {
477
		case '=':
478
			l.advance()
479
			return token.Token{Type: token.EQEQEQ, Literal: "===", Span: l.span(s)}
480
		case '*':
481
			l.advance()
482
			return token.Token{Type: token.EQEQEQ, Literal: "==*", Span: l.span(s)}
483
		default:
484
			return token.Token{Type: token.EQEQ, Literal: "==", Span: l.span(s)}
485
		}
486
	}
487
	if l.ch == '*' {
488
		l.advance()
489
		return token.Token{Type: token.EQSTAR, Literal: "=*", Span: l.span(s)}
490
	}
491
	return token.Token{Type: token.EQ, Literal: "=", Span: l.span(s)}
492
}
493
494
func (l *Lexer) lexAt() token.Token {
495
	s := l.save()
496
	l.advance()
497
	if l.ch == '@' {
498
		l.advance()
499
		return token.Token{Type: token.ATAT, Literal: "@@", Span: l.span(s)}
500
	}
501
	return token.Token{Type: token.AT, Literal: "@", Span: l.span(s)}
502
}
503
504
func (l *Lexer) lexText() token.Token {
505
	s := l.save()
506
	l.advance()
507
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ' ' && l.ch != '\t' {
508
		l.advance()
509
	}
510
	lit := string(l.input[s.offset:l.pos])
511
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
512
}
513
514
// lexAccountNameDirective reads accout name in directive context.
515
// stops at any whitespace, supports multi-word names("Taxi Fare").
516
func (l *Lexer) lexAccountNameDirective() token.Token {
517
	s := l.save()
518
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ')' && l.ch != ']' && l.ch != ':' && l.ch != ' ' && l.ch != '\t' {
519
		l.advance()
520
	}
521
	lit := string(l.input[s.offset:l.pos])
522
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
523
}
524
525
// lexAccountNamePosting reads an account name in posting context.
526
// stops at two consecutive spaces.
527
func (l *Lexer) lexAccountNamePosting() token.Token {
528
	s := l.save()
529
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ')' && l.ch != ']' && l.ch != ':' {
530
		if l.isTwoSpaces() {
531
			break
532
		}
533
		l.advance()
534
	}
535
	if l.ch != ':' {
536
		l.postingExpectAccount = false
537
	}
538
	lit := string(l.input[s.offset:l.pos])
539
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
540
}
541
542
func (l *Lexer) lexParenExpr() token.Token {
543
	s := l.save()
544
	depth := 0
545
	for l.ch != '\n' && l.ch != '\r' && l.ch != 0 {
546
		if l.ch == '(' {
547
			depth++
548
		} else if l.ch == ')' {
549
			depth--
550
			if depth == 0 {
551
				l.advance()
552
				break
553
			}
554
		}
555
		l.advance()
556
	}
557
	lit := string(l.input[s.offset:l.pos])
558
	return token.Token{Type: token.PARENEXPR, Literal: lit, Span: l.span(s)}
559
}
560
561
func (l *Lexer) lexNumber() token.Token {
562
	s := l.save()
563
	for {
564
		if l.isDigit() || l.ch == '.' || l.ch == ',' || l.ch == '_' || l.ch == '\'' {
565
			l.advance()
566
		} else if l.ch == ' ' && (l.peek() >= '0' && l.peek() <= '9') {
567
			l.advance()
568
		} else if l.ch == 'e' || l.ch == 'E' {
569
			// exponent: consume only when E[sign]digit, so `10E ` stays an integer
570
			p := l.pos + 1
571
			if p < len(l.input) && (l.input[p] == '+' || l.input[p] == '-') {
572
				p++
573
			}
574
			if p >= len(l.input) || l.input[p] < '0' || l.input[p] > '9' {
575
				break
576
			}
577
			l.advance() // e/E
578
			if l.ch == '+' || l.ch == '-' {
579
				l.advance()
580
			}
581
			for l.isDigit() {
582
				l.advance()
583
			}
584
		} else {
585
			break
586
		}
587
	}
588
	lit := string(l.input[s.offset:l.pos])
589
	kind := token.INT
590
	if strings.ContainsAny(lit, "., eE") {
591
		kind = token.DECIMAL
592
	}
593
	return token.Token{Type: kind, Literal: lit, Span: l.span(s)}
594
}
595
596
func (l *Lexer) lexKeyword() token.Token {
597
	s := l.save()
598
	for l.ch != 0 && l.ch != '\n' && l.ch != '\r' && l.ch != ' ' && l.ch != '\t' && l.ch != ';' {
599
		l.advance()
600
	}
601
	lit := string(l.input[s.offset:l.pos])
602
	kind := l.keyword(lit)
603
	if kind == token.ILLEGAL { // todo: report an error ??
604
		kind = token.TEXT
605
	} else {
606
		l.mode = modeDirective
607
		l.subdirective = kind == token.ACCOUNT || kind == token.COMMODITY
608
		l.includePath = kind == token.INCLUDE
609
	}
610
	return token.Token{Type: kind, Literal: lit, Span: l.span(s)}
611
}
612
613
func (l *Lexer) lexDate() token.Token {
614
	s := l.save()
615
	for l.isDigit() || (l.isDateSep() && l.peekIsDigit()) {
616
		l.advance()
617
	}
618
	return token.Token{Type: token.DATE, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
619
}
620
621
func isSymbolChar(r rune) bool {
622
	return r == '$' || unicode.In(r, unicode.Sc)
623
}
624
625
func (l *Lexer) lexString() token.Token {
626
	s := l.save()
627
	quote := l.ch
628
	l.advance() // consume the quote character
629
	for l.ch != quote && l.ch != '\n' && l.ch != 0 {
630
		l.advance()
631
	}
632
	if l.ch == quote {
633
		l.advance()
634
	}
635
	return token.Token{Type: token.STRING, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
636
}
637
638
func (l *Lexer) lexCommodityMark() token.Token {
639
	s := l.save()
640
641
	if l.ch == '"' {
642
		l.advance()
643
		for l.ch != '"' && l.ch != '\n' && l.ch != 0 {
644
			l.advance()
645
		}
646
		if l.ch == '"' {
647
			l.advance()
648
		}
649
		return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
650
	}
651
652
	if unicode.IsLetter(l.ch) {
653
		for unicode.IsLetter(l.ch) {
654
			l.advance()
655
		}
656
		return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
657
	}
658
659
	if isSymbolChar(l.ch) {
660
		for isSymbolChar(l.ch) {
661
			l.advance()
662
		}
663
		return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
664
	}
665
666
	l.advance()
667
	return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
668
}
669
670
func (l *Lexer) lexLBrace() token.Token {
671
	s := l.save()
672
	l.advance()
673
	if l.ch == '{' {
674
		l.advance()
675
		return token.Token{Type: token.LBRACELBRACE, Literal: "{{", Span: l.span(s)}
676
	}
677
	return token.Token{Type: token.LBRACE, Literal: "{", Span: l.span(s)}
678
}
679
680
func (l *Lexer) lexRBrace() token.Token {
681
	s := l.save()
682
	l.advance()
683
	if l.ch == '}' {
684
		l.advance()
685
		return token.Token{Type: token.RBRACERBRACE, Literal: "}}", Span: l.span(s)}
686
	}
687
	return token.Token{Type: token.RBRACE, Literal: "}", Span: l.span(s)}
688
}
689
690
func (l *Lexer) advance() {
691
	if l.rpos >= len(l.input) {
692
		l.ch = 0
693
		l.chSize = 0
694
	} else {
695
		r, size := utf8.DecodeRune(l.input[l.rpos:])
696
		l.ch = r
697
		l.chSize = size
698
	}
699
	l.pos = l.rpos
700
	l.rpos += l.chSize
701
	if l.ch == '\n' || l.ch == '\r' {
702
		l.line++
703
		l.col = 0
704
	} else {
705
		l.col++
706
	}
707
}
708
709
func (l *Lexer) peek() rune {
710
	r, _ := utf8.DecodeRune(l.input[l.rpos:])
711
	return r
712
}
713
714
func (l *Lexer) peekN(n int) byte {
715
	if l.pos+n >= len(l.input) {
716
		return 0
717
	}
718
	return l.input[l.pos+n]
719
}
720
721
func (l *Lexer) isDigit() bool { return l.ch >= '0' && l.ch <= '9' }
722
func (l *Lexer) isAlpha() bool {
723
	return (l.ch >= 'a' && l.ch <= 'z') ||
724
		(l.ch >= 'A' && l.ch <= 'Z')
725
}
726
727
func (l *Lexer) isTwoSpaces() bool { return l.ch == ' ' && l.peek() == ' ' }
728
729
func (l *Lexer) isDateSep() bool { return l.ch == '-' || l.ch == '/' || l.ch == '.' }
730
731
func (l *Lexer) peekIsDigit() bool {
732
	r := l.peek()
733
	return r >= '0' && r <= '9'
734
}
735
736
func (l *Lexer) isCommodityStart() bool {
737
	if l.ch == '$' || (l.ch >= 'A' && l.ch <= 'Z') {
738
		return true
739
	}
740
	if l.ch < utf8.RuneSelf {
741
		return false
742
	}
743
	return unicode.In(l.ch, unicode.Sc) || unicode.IsLetter(l.ch)
744
}
745
746
func (l *Lexer) isDate() bool {
747
	if !l.isDigit() {
748
		return false
749
	}
750
	// YYYY/M/D or YYYY/MM/DD
751
	if l.peekN(1) >= '0' && l.peekN(1) <= '9' &&
752
		l.peekN(2) >= '0' && l.peekN(2) <= '9' &&
753
		l.peekN(3) >= '0' && l.peekN(3) <= '9' {
754
		sep := l.peekN(4)
755
		if sep == '/' || sep == '-' || sep == '.' {
756
			if l.peekN(5) >= '0' && l.peekN(5) <= '9' {
757
				if l.peekN(6) == sep {
758
					return l.peekN(7) >= '0' && l.peekN(7) <= '9'
759
				}
760
				if l.peekN(7) == sep {
761
					return l.peekN(8) >= '0' && l.peekN(8) <= '9'
762
				}
763
			}
764
		}
765
		return false
766
	}
767
	// M/D or MM/DD(year inferred, only / and - separators; . is ambiguous with decimal numbers like 1.01)
768
	if (l.peekN(1) == '/' || l.peekN(1) == '-') &&
769
		l.peekN(2) >= '0' && l.peekN(2) <= '9' &&
770
		l.ch >= '1' && l.ch <= '9' {
771
		return validDay(l.peekN(2), l.peekN(3))
772
	}
773
	if (l.peekN(2) == '/' || l.peekN(2) == '-') &&
774
		l.peekN(3) >= '0' && l.peekN(3) <= '9' {
775
		m := int(l.ch-'0')*10 + int(l.peekN(1)-'0')
776
		return m >= 1 && m <= 12 && validDay(l.peekN(3), l.peekN(4))
777
	}
778
	return false
779
}
780
781
func validDay(first, second byte) bool {
782
	d := int(first - '0')
783
	if second >= '0' && second <= '9' {
784
		d = d*10 + int(second-'0')
785
	}
786
	return d >= 1 && d <= 31
787
}
788
789
func (l *Lexer) isTime() bool {
790
	if !l.isDigit() {
791
		return false
792
	}
793
	return l.peekN(2) == ':'
794
}
795
796
func (l *Lexer) lexTime() token.Token {
797
	s := l.save()
798
	for l.isDigit() || l.ch == ':' {
799
		l.advance()
800
	}
801
	return token.Token{Type: token.TIME, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
802
}
803
804
type savedPos struct{ offset, line, col int }
805
806
func (l *Lexer) save() savedPos {
807
	return savedPos{l.pos, l.line, l.col}
808
}
809
810
func (l *Lexer) span(s savedPos) token.Span {
811
	return token.Span{
812
		Start: token.Pos{File: l.file, Offset: s.offset, Line: s.line, Col: s.col},
813
		End:   token.Pos{File: l.file, Offset: l.pos, Line: l.line, Col: l.col},
814
	}
815
}
816
817
func (l *Lexer) token(kind token.Type, literal string) token.Token {
818
	s := savedPos{l.pos, l.line, l.col}
819
	return token.Token{Type: kind, Literal: literal, Span: l.span(s)}
820
}
821
822
func (l *Lexer) keyword(s string) token.Type {
823
	switch s {
824
	case "comment":
825
		return token.COMMENTKW
826
	case "account":
827
		return token.ACCOUNT
828
	case "commodity":
829
		return token.COMMODITY
830
	case "include":
831
		return token.INCLUDE
832
	case "alias":
833
		return token.ALIAS
834
	case "payee":
835
		return token.PAYEE
836
	case "tag":
837
		return token.TAG
838
	case "apply":
839
		return token.APPLY
840
	case "end":
841
		return token.END
842
	case "Y", "year":
843
		return token.YEAR
844
	case "decimal-mark":
845
		return token.DECIMALMARK
846
	case "D":
847
		return token.D
848
	case "P":
849
		return token.P
850
	case "N":
851
		return token.N
852
	case "C":
853
		return token.C
854
	default:
855
		return token.ILLEGAL
856
	}
857
}