all repos

clerk @ b08266897cc230f2a3323a2a70705116f7cfe2e1

missing tooling for ledger/hledger

clerk/journal/lexer/lexer.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
syntax: respect commodity format; =*; ===; :=, 2 months ago
1
package lexer
2
3
import (
4
	"strings"
5
	"unicode"
6
	"unicode/utf8"
7
8
	"olexsmir.xyz/clerk/journal/token"
9
)
10
11
type Mode uint
12
13
const (
14
	// start of a line, nothing consumed
15
	ModeDefault Mode = iota
16
17
	// after ; # * % ; at start of line, or anywhere inline
18
	// everything until \n is comment text
19
	ModeComment
20
21
	// after lexing a date at column 0
22
	// expects: optional status, optional code, description, comment
23
	ModeTransaction
24
25
	// after lexing an indent at start of line
26
	// expects: account name, then two spaces, then amount
27
	ModePosting
28
29
	// after ~, period expression
30
	// expects: period, optional description (after 2+ spaces), optional comment
31
	ModePeriodic
32
33
	// after =, automates transaction
34
	// expects: expression
35
	ModeAutomated
36
37
	// after a directive keyword like account, commodity, include
38
	// expects: rest of directive content
39
	ModeDirective
40
)
41
42
type Lexer struct {
43
	file  string
44
	input []byte
45
	mode  Mode
46
47
	ch     rune // current rune (0 = EOF/sentinel)
48
	chSize int  // byte size of current rune
49
	pos    int  // current byte offset (points at ch)
50
	rpos   int  // next byte offset to read (one ahead of pos)
51
	col    int  // current column (1-based)
52
	line   int  // current line (1-based)
53
54
	transactionPastStatus bool
55
	postingExpectAccount  bool
56
	readingNoteAfterPipe  bool
57
58
	// subdirective is set when the current line is an account/commodity
59
	// directive; the next indented line then lexes as directive content
60
	subdirective bool
61
}
62
63
func New(file string, input []byte) *Lexer {
64
	l := &Lexer{
65
		file:  file,
66
		input: input,
67
		line:  1,
68
	}
69
	l.advance()
70
	if l.ch == '\uFEFF' { // start of the input
71
		l.advance()
72
	}
73
	return l
74
}
75
76
// Next returns next token in the input
77
func (l *Lexer) Next() token.Token {
78
	switch l.mode {
79
	case ModeDefault:
80
		return l.lexDefault()
81
	case ModeComment:
82
		return l.lexComment()
83
	case ModeTransaction:
84
		return l.lexTransaction()
85
	case ModePosting:
86
		return l.lexPosting()
87
	case ModePeriodic:
88
		return l.lexPeriodic()
89
	case ModeAutomated:
90
		return l.lexAutomated()
91
	case ModeDirective:
92
		return l.lexDirective()
93
	}
94
	panic("unreachable")
95
}
96
97
func (l *Lexer) lexDefault() token.Token {
98
	if l.subdirective && l.ch != ' ' && l.ch != '\t' {
99
		l.subdirective = false
100
	}
101
	switch {
102
	case l.ch == 0:
103
		return l.token(token.EOF, "")
104
	case l.ch == '\n':
105
		return l.lexNewline()
106
	case l.ch == '\r':
107
		l.col = 0
108
		l.advance()
109
		return l.lexNewline()
110
	case l.ch == ' ' || l.ch == '\t':
111
		tok := l.lexIndent()
112
		if l.subdirective {
113
			l.mode = ModeDirective
114
			l.subdirective = false
115
		} else {
116
			l.mode = ModePosting
117
		}
118
		l.postingExpectAccount = true
119
		return tok
120
	case l.ch == ';' || l.ch == '#' || l.ch == '%':
121
		l.mode = ModeComment
122
		return l.lexSingle(token.SEMICOLON) // todo: ??
123
	case l.ch == '*': // * at col 0 == comment
124
		l.mode = ModeComment
125
		return l.lexSingle(token.STAR)
126
	case l.ch == '~':
127
		l.mode = ModePeriodic
128
		return l.lexSingle(token.TILDE)
129
	case l.ch == '=':
130
		l.mode = ModeAutomated
131
		return l.lexSingle(token.EQ)
132
	case l.ch == '+':
133
		return l.lexSingle(token.PLUS)
134
	case l.ch == '-':
135
		return l.lexSingle(token.MINUS)
136
	case l.ch == '.':
137
		return l.lexSingle(token.TEXT)
138
	case l.ch == '!':
139
		return l.lexSingle(token.BANG)
140
	case l.ch == '@':
141
		return l.lexSingle(token.AT)
142
	case l.isAlpha():
143
		return l.lexKeyword()
144
	case l.isDigit():
145
		if !l.isDate() {
146
			s := l.save()
147
			for l.isDigit() || l.ch == '-' || l.ch == '/' || l.ch == '.' {
148
				l.advance()
149
			}
150
			return token.Token{Type: token.ILLEGAL, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
151
		}
152
		tok := l.lexDate()
153
		l.mode = ModeTransaction
154
		l.transactionPastStatus = false
155
		return tok
156
	default:
157
		s := l.save()
158
		l.advance()
159
		return token.Token{Type: token.ILLEGAL, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
160
	}
161
}
162
163
func (l *Lexer) lexComment() token.Token {
164
	if l.ch == '\n' || l.ch == 0 {
165
		l.mode = ModeDefault
166
		return l.lexNewline()
167
	}
168
169
	for l.ch == ' ' || l.ch == '\t' {
170
		l.lexWhitespace()
171
	}
172
173
	if l.ch == '\n' || l.ch == 0 {
174
		l.mode = ModeDefault
175
		return l.lexNewline()
176
	}
177
178
	s := l.save()
179
	for l.ch != '\n' && l.ch != 0 {
180
		l.advance()
181
	}
182
	return token.Token{Type: token.TEXT, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
183
}
184
185
func (l *Lexer) lexTransaction() token.Token {
186
	if l.readingNoteAfterPipe {
187
		l.readingNoteAfterPipe = false
188
		return l.lexNote()
189
	}
190
191
	switch l.ch {
192
	case 0:
193
		return l.token(token.EOF, "")
194
	case '\n':
195
		l.mode = ModeDefault
196
		return l.lexNewline()
197
	case '\r':
198
		l.col = 0
199
		l.advance()
200
		return l.lexNewline()
201
	case ' ', '\t':
202
		return l.lexWhitespace()
203
	case ';':
204
		l.mode = ModeComment
205
		return l.lexSingle(token.SEMICOLON)
206
	case '*':
207
		if !l.transactionPastStatus {
208
			l.transactionPastStatus = true
209
			return l.lexSingle(token.STAR)
210
		}
211
		return l.lexText()
212
	case '!':
213
		if !l.transactionPastStatus {
214
			l.transactionPastStatus = true
215
			return l.lexSingle(token.BANG)
216
		}
217
		return l.lexText()
218
	case '|':
219
		l.transactionPastStatus = true
220
		l.readingNoteAfterPipe = true
221
		return l.lexSingle(token.PIPE)
222
	case '+':
223
		return l.lexSingle(token.PLUS)
224
	case '-':
225
		return l.lexSingle(token.MINUS)
226
	case '=':
227
		return l.lexEquals()
228
	case '"', '\'':
229
		return l.lexString()
230
	default: // description / payee
231
		if l.isDate() { // secondsry date after =
232
			return l.lexDate()
233
		}
234
		return l.lexText()
235
	}
236
}
237
238
func (l *Lexer) lexNote() token.Token {
239
	for l.ch == ' ' || l.ch == '\t' {
240
		l.advance()
241
	}
242
	if l.ch == 0 || l.ch == '\n' || l.ch == ';' {
243
		return l.Next()
244
	}
245
	s := l.save()
246
	for l.ch != 0 && l.ch != '\n' && l.ch != ';' {
247
		l.advance()
248
	}
249
	lit := string(l.input[s.offset:l.pos])
250
	for len(lit) > 0 && (lit[len(lit)-1] == ' ' || lit[len(lit)-1] == '\t') {
251
		lit = lit[:len(lit)-1]
252
	}
253
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
254
}
255
256
func (l *Lexer) lexPeriodic() token.Token {
257
	switch l.ch {
258
	case 0:
259
		return l.token(token.EOF, "")
260
	case '\n':
261
		l.mode = ModeDefault
262
		return l.lexNewline()
263
	case '\r':
264
		l.col = 0
265
		l.advance()
266
		return l.lexNewline()
267
	case ';':
268
		l.mode = ModeComment
269
		return l.lexSingle(token.SEMICOLON)
270
	case ' ', '\t':
271
		return l.lexWhitespace()
272
	default:
273
		return l.lexText()
274
	}
275
}
276
277
func (l *Lexer) lexAutomated() token.Token {
278
	switch l.ch {
279
	case 0:
280
		return l.token(token.EOF, "")
281
	case '\n':
282
		l.mode = ModeDefault
283
		return l.lexNewline()
284
	case '\r':
285
		l.col = 0
286
		l.advance()
287
		return l.lexNewline()
288
	case ' ', '\t':
289
		return l.lexWhitespace()
290
	case ';':
291
		l.mode = ModeComment
292
		return l.lexSingle(token.SEMICOLON)
293
	default:
294
		return l.lexText()
295
	}
296
}
297
298
func (l *Lexer) lexPosting() token.Token {
299
	switch {
300
	case l.ch == 0:
301
		l.postingExpectAccount = false
302
		return l.token(token.EOF, "")
303
	case l.ch == '\n':
304
		l.postingExpectAccount = false
305
		l.mode = ModeDefault
306
		return l.lexNewline()
307
	case l.ch == ';':
308
		l.postingExpectAccount = false
309
		l.mode = ModeComment
310
		return l.lexSingle(token.SEMICOLON)
311
	case l.ch == ' ' || l.ch == '\t':
312
		return l.lexWhitespace()
313
	case l.postingExpectAccount && l.ch == '*':
314
		return l.lexSingle(token.STAR)
315
	case l.postingExpectAccount && l.ch == '!':
316
		return l.lexSingle(token.BANG)
317
	case l.ch == '=':
318
		return l.lexEquals()
319
	case l.ch == '@':
320
		return l.lexAt()
321
	case l.ch == '{':
322
		return l.lexLBrace()
323
	case l.ch == '}':
324
		return l.lexRBrace()
325
	case l.ch == '(':
326
		if !l.postingExpectAccount {
327
			return l.lexParenExpr()
328
		}
329
		return l.lexSingle(token.LPAREN)
330
	case l.ch == ')':
331
		return l.lexSingle(token.RPAREN)
332
	case l.ch == '[':
333
		return l.lexSingle(token.LBRACKET)
334
	case l.ch == ']':
335
		return l.lexSingle(token.RBRACKET)
336
	case l.ch == ':':
337
		return l.lexSingle(token.COLON)
338
	case l.postingExpectAccount && l.ch != '*' && l.ch != '!' && l.ch != '(' && l.ch != '[':
339
		return l.lexAccountNamePosting()
340
	case l.ch == '*': // after account name
341
		return l.lexSingle(token.STAR)
342
	case l.isDigit(), l.ch == '.':
343
		return l.lexNumber()
344
	case l.ch == '-':
345
		return l.lexSingle(token.MINUS)
346
	case l.ch == '+':
347
		return l.lexSingle(token.PLUS)
348
	case l.isCommodityStart():
349
		return l.lexCommodityMark()
350
	case l.ch == '"' || l.ch == '\'':
351
		return l.lexString()
352
	case l.ch >= 'a' && l.ch <= 'z':
353
		return l.lexCommodityMark()
354
	default:
355
		return l.lexAccountNamePosting()
356
	}
357
}
358
359
func (l *Lexer) lexDirective() token.Token {
360
	switch l.ch {
361
	case '\n', 0:
362
		l.mode = ModeDefault
363
		return l.lexNewline()
364
	case ';':
365
		l.mode = ModeComment
366
		return l.lexSingle(token.SEMICOLON)
367
	case ' ', '\t':
368
		return l.lexWhitespace()
369
	case '=':
370
		return l.lexSingle(token.EQ)
371
	case '+':
372
		return l.lexSingle(token.PLUS)
373
	case '-':
374
		return l.lexSingle(token.MINUS)
375
	case '"', '\'':
376
		return l.lexString()
377
	case ':':
378
		return l.lexSingle(token.COLON)
379
	case ')':
380
		return l.lexSingle(token.RPAREN)
381
	case ']':
382
		return l.lexSingle(token.RBRACKET)
383
	default:
384
		if l.isCommodityStart() {
385
			return l.lexCommodityMark()
386
		}
387
		if l.isTime() {
388
			return l.lexTime()
389
		}
390
		if l.isDate() {
391
			return l.lexDate()
392
		}
393
		if l.isDigit() {
394
			return l.lexNumber()
395
		}
396
		return l.lexAccountNameDirective()
397
	}
398
}
399
400
func (l *Lexer) lexSingle(kind token.Type) token.Token {
401
	s := l.save()
402
	l.advance()
403
	return token.Token{
404
		Type:    kind,
405
		Literal: string(l.input[s.offset:l.pos]),
406
		Span:    l.span(s),
407
	}
408
}
409
410
func (l *Lexer) lexNewline() token.Token {
411
	s := l.save()
412
	l.advance()
413
	l.mode = ModeDefault
414
	return token.Token{Type: token.NEWLINE, Literal: "\n", Span: l.span(s)}
415
}
416
417
func (l *Lexer) lexWhitespace() token.Token {
418
	s := l.save()
419
	for l.ch == ' ' || l.ch == '\t' {
420
		l.advance()
421
	}
422
	lit := string(l.input[s.offset:l.pos])
423
	return token.Token{Type: token.WHITESPACE, Literal: lit, Span: l.span(s)}
424
}
425
426
func (l *Lexer) lexIndent() token.Token {
427
	s := l.save()
428
	for l.ch == ' ' || l.ch == '\t' {
429
		l.advance()
430
	}
431
	lit := string(l.input[s.offset:l.pos])
432
	return token.Token{Type: token.INDENT, Literal: lit, Span: l.span(s)}
433
}
434
435
func (l *Lexer) lexEquals() token.Token {
436
	s := l.save()
437
	l.advance()
438
	if l.ch == '=' {
439
		l.advance()
440
		switch l.ch {
441
		case '=':
442
			l.advance()
443
			return token.Token{Type: token.EQEQEQ, Literal: "===", Span: l.span(s)}
444
		case '*':
445
			l.advance()
446
			return token.Token{Type: token.EQEQEQ, Literal: "==*", Span: l.span(s)}
447
		default:
448
			return token.Token{Type: token.EQEQ, Literal: "==", Span: l.span(s)}
449
		}
450
	}
451
	if l.ch == '*' {
452
		l.advance()
453
		return token.Token{Type: token.EQSTAR, Literal: "=*", Span: l.span(s)}
454
	}
455
	return token.Token{Type: token.EQ, Literal: "=", Span: l.span(s)}
456
}
457
458
func (l *Lexer) lexAt() token.Token {
459
	s := l.save()
460
	l.advance()
461
	if l.ch == '@' {
462
		l.advance()
463
		return token.Token{Type: token.ATAT, Literal: "@@", Span: l.span(s)}
464
	}
465
	return token.Token{Type: token.AT, Literal: "@", Span: l.span(s)}
466
}
467
468
func (l *Lexer) lexText() token.Token {
469
	s := l.save()
470
	l.advance()
471
	for l.ch != '\n' && l.ch != ';' && l.ch != 0 && l.ch != ' ' && l.ch != '\t' {
472
		l.advance()
473
	}
474
	lit := string(l.input[s.offset:l.pos])
475
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
476
}
477
478
// lexAccountNameDirective reads accout name in directive context.
479
// stops at any whitespace, supports multi-word names("Taxi Fare").
480
func (l *Lexer) lexAccountNameDirective() token.Token {
481
	s := l.save()
482
	for l.ch != '\n' && l.ch != ';' && l.ch != 0 && l.ch != ')' && l.ch != ']' && l.ch != ':' && l.ch != ' ' && l.ch != '\t' {
483
		l.advance()
484
	}
485
	lit := string(l.input[s.offset:l.pos])
486
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
487
}
488
489
// lexAccountNamePosting reads an account name in posting context.
490
// stops at two consecutive spaces.
491
func (l *Lexer) lexAccountNamePosting() token.Token {
492
	s := l.save()
493
	for l.ch != '\n' && l.ch != ';' && l.ch != 0 && l.ch != ')' && l.ch != ']' && l.ch != ':' {
494
		if l.isTwoSpaces() {
495
			break
496
		}
497
		l.advance()
498
	}
499
	if l.ch != ':' {
500
		l.postingExpectAccount = false
501
	}
502
	lit := string(l.input[s.offset:l.pos])
503
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
504
}
505
506
func (l *Lexer) lexParenExpr() token.Token {
507
	s := l.save()
508
	depth := 0
509
	for l.ch != '\n' && l.ch != 0 {
510
		if l.ch == '(' {
511
			depth++
512
		} else if l.ch == ')' {
513
			depth--
514
			if depth == 0 {
515
				l.advance()
516
				break
517
			}
518
		}
519
		l.advance()
520
	}
521
	lit := string(l.input[s.offset:l.pos])
522
	return token.Token{Type: token.PARENEXPR, Literal: lit, Span: l.span(s)}
523
}
524
525
func (l *Lexer) lexNumber() token.Token {
526
	s := l.save()
527
	for {
528
		if l.isDigit() || l.ch == '.' || l.ch == ',' || l.ch == '_' || l.ch == '\'' {
529
			l.advance()
530
		} else if l.ch == ' ' && (l.peek() >= '0' && l.peek() <= '9') {
531
			l.advance()
532
		} else if l.ch == 'e' || l.ch == 'E' {
533
			// exponent: consume only when E[sign]digit, so `10E ` stays an integer
534
			p := l.pos + 1
535
			if p < len(l.input) && (l.input[p] == '+' || l.input[p] == '-') {
536
				p++
537
			}
538
			if p >= len(l.input) || l.input[p] < '0' || l.input[p] > '9' {
539
				break
540
			}
541
			l.advance() // e/E
542
			if l.ch == '+' || l.ch == '-' {
543
				l.advance()
544
			}
545
			for l.isDigit() {
546
				l.advance()
547
			}
548
		} else {
549
			break
550
		}
551
	}
552
	lit := string(l.input[s.offset:l.pos])
553
	kind := token.INT
554
	if strings.ContainsAny(lit, "., eE") {
555
		kind = token.DECIMAL
556
	}
557
	return token.Token{Type: kind, Literal: lit, Span: l.span(s)}
558
}
559
560
func (l *Lexer) lexKeyword() token.Token {
561
	s := l.save()
562
	for l.ch != 0 && l.ch != '\n' && l.ch != '\r' && l.ch != ' ' && l.ch != '\t' && l.ch != ';' {
563
		l.advance()
564
	}
565
	lit := string(l.input[s.offset:l.pos])
566
	kind := l.keyword(lit)
567
	if kind == token.ILLEGAL { // todo: report an error ??
568
		kind = token.TEXT
569
	} else {
570
		l.mode = ModeDirective
571
		l.subdirective = lit == "account" || lit == "commodity"
572
	}
573
	return token.Token{Type: kind, Literal: lit, Span: l.span(s)}
574
}
575
576
func (l *Lexer) lexDate() token.Token {
577
	s := l.save()
578
	for l.isDigit() || (l.isDateSep() && l.peekIsDigit()) {
579
		l.advance()
580
	}
581
	return token.Token{Type: token.DATE, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
582
}
583
584
func isSymbolChar(r rune) bool {
585
	return r == '$' || unicode.In(r, unicode.Sc)
586
}
587
588
func (l *Lexer) lexString() token.Token {
589
	s := l.save()
590
	quote := l.ch
591
	l.advance() // consume the quote character
592
	for l.ch != quote && l.ch != '\n' && l.ch != 0 {
593
		l.advance()
594
	}
595
	if l.ch == quote {
596
		l.advance()
597
	}
598
	return token.Token{Type: token.STRING, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
599
}
600
601
func (l *Lexer) lexCommodityMark() token.Token {
602
	s := l.save()
603
604
	if l.ch == '"' {
605
		l.advance()
606
		for l.ch != '"' && l.ch != '\n' && l.ch != 0 {
607
			l.advance()
608
		}
609
		if l.ch == '"' {
610
			l.advance()
611
		}
612
		return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
613
	}
614
615
	if unicode.IsLetter(l.ch) {
616
		for unicode.IsLetter(l.ch) {
617
			l.advance()
618
		}
619
		return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
620
	}
621
622
	if isSymbolChar(l.ch) {
623
		for isSymbolChar(l.ch) {
624
			l.advance()
625
		}
626
		return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
627
	}
628
629
	l.advance()
630
	return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
631
}
632
633
func (l *Lexer) lexLBrace() token.Token {
634
	s := l.save()
635
	l.advance()
636
	if l.ch == '{' {
637
		l.advance()
638
		return token.Token{Type: token.LBRACELBRACE, Literal: "{{", Span: l.span(s)}
639
	}
640
	return token.Token{Type: token.LBRACE, Literal: "{", Span: l.span(s)}
641
}
642
643
func (l *Lexer) lexRBrace() token.Token {
644
	s := l.save()
645
	l.advance()
646
	if l.ch == '}' {
647
		l.advance()
648
		return token.Token{Type: token.RBRACERBRACE, Literal: "}}", Span: l.span(s)}
649
	}
650
	return token.Token{Type: token.RBRACE, Literal: "}", Span: l.span(s)}
651
}
652
653
func (l *Lexer) advance() {
654
	if l.rpos >= len(l.input) {
655
		l.ch = 0
656
		l.chSize = 0
657
	} else {
658
		r, size := utf8.DecodeRune(l.input[l.rpos:])
659
		l.ch = r
660
		l.chSize = size
661
	}
662
	l.pos = l.rpos
663
	l.rpos += l.chSize
664
	if l.ch == '\n' || l.ch == '\r' {
665
		l.line++
666
		l.col = 0
667
	} else {
668
		l.col++
669
	}
670
}
671
672
func (l *Lexer) peek() rune {
673
	r, _ := utf8.DecodeRune(l.input[l.rpos:])
674
	return r
675
}
676
677
func (l *Lexer) peekN(n int) byte {
678
	if l.pos+n >= len(l.input) {
679
		return 0
680
	}
681
	return l.input[l.pos+n]
682
}
683
684
func (l *Lexer) isDigit() bool { return l.ch >= '0' && l.ch <= '9' }
685
func (l *Lexer) isAlpha() bool {
686
	return (l.ch >= 'a' && l.ch <= 'z') ||
687
		(l.ch >= 'A' && l.ch <= 'Z')
688
}
689
690
func (l *Lexer) isTwoSpaces() bool { return l.ch == ' ' && l.peek() == ' ' }
691
692
func (l *Lexer) isDateSep() bool { return l.ch == '-' || l.ch == '/' || l.ch == '.' }
693
694
func (l *Lexer) peekIsDigit() bool {
695
	r := l.peek()
696
	return r >= '0' && r <= '9'
697
}
698
699
func (l *Lexer) isCommodityStart() bool {
700
	if l.ch == '$' || (l.ch >= 'A' && l.ch <= 'Z') {
701
		return true
702
	}
703
	if l.ch < utf8.RuneSelf {
704
		return false
705
	}
706
	return unicode.In(l.ch, unicode.Sc) || unicode.IsLetter(l.ch)
707
}
708
709
func (l *Lexer) isDate() bool {
710
	if !l.isDigit() {
711
		return false
712
	}
713
	// YYYY/M/D or YYYY/MM/DD
714
	if l.peekN(1) >= '0' && l.peekN(1) <= '9' &&
715
		l.peekN(2) >= '0' && l.peekN(2) <= '9' &&
716
		l.peekN(3) >= '0' && l.peekN(3) <= '9' {
717
		sep := l.peekN(4)
718
		if sep == '/' || sep == '-' || sep == '.' {
719
			if l.peekN(5) >= '0' && l.peekN(5) <= '9' {
720
				if l.peekN(6) == sep {
721
					return l.peekN(7) >= '0' && l.peekN(7) <= '9'
722
				}
723
				if l.peekN(7) == sep {
724
					return l.peekN(8) >= '0' && l.peekN(8) <= '9'
725
				}
726
			}
727
		}
728
		return false
729
	}
730
	// M/D or MM/DD(year inferred, only / and - separators; . is ambiguous with decimal numbers like 1.01)
731
	if (l.peekN(1) == '/' || l.peekN(1) == '-') &&
732
		l.peekN(2) >= '0' && l.peekN(2) <= '9' &&
733
		l.ch >= '1' && l.ch <= '9' {
734
		return validDay(l.peekN(2), l.peekN(3))
735
	}
736
	if (l.peekN(2) == '/' || l.peekN(2) == '-') &&
737
		l.peekN(3) >= '0' && l.peekN(3) <= '9' {
738
		m := int(l.ch-'0')*10 + int(l.peekN(1)-'0')
739
		return m >= 1 && m <= 12 && validDay(l.peekN(3), l.peekN(4))
740
	}
741
	return false
742
}
743
744
func validDay(first, second byte) bool {
745
	d := int(first - '0')
746
	if second >= '0' && second <= '9' {
747
		d = d*10 + int(second-'0')
748
	}
749
	return d >= 1 && d <= 31
750
}
751
752
func (l *Lexer) isTime() bool {
753
	if !l.isDigit() {
754
		return false
755
	}
756
	return l.peekN(2) == ':'
757
}
758
759
func (l *Lexer) lexTime() token.Token {
760
	s := l.save()
761
	for l.isDigit() || l.ch == ':' {
762
		l.advance()
763
	}
764
	return token.Token{Type: token.TIME, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
765
}
766
767
type savedPos struct{ offset, line, col int }
768
769
func (l *Lexer) save() savedPos {
770
	return savedPos{l.pos, l.line, l.col}
771
}
772
773
func (l *Lexer) span(s savedPos) token.Span {
774
	return token.Span{
775
		Start: token.Pos{File: l.file, Offset: s.offset, Line: s.line, Col: s.col},
776
		End:   token.Pos{File: l.file, Offset: l.pos, Line: l.line, Col: l.col},
777
	}
778
}
779
780
func (l *Lexer) token(kind token.Type, literal string) token.Token {
781
	s := savedPos{l.pos, l.line, l.col}
782
	return token.Token{Type: kind, Literal: literal, Span: l.span(s)}
783
}
784
785
func (l *Lexer) keyword(s string) token.Type {
786
	switch s {
787
	case "comment":
788
		return token.COMMENTKW
789
	case "account":
790
		return token.ACCOUNT
791
	case "commodity":
792
		return token.COMMODITY
793
	case "include":
794
		return token.INCLUDE
795
	case "alias":
796
		return token.ALIAS
797
	case "payee":
798
		return token.PAYEE
799
	case "tag":
800
		return token.TAG
801
	case "apply":
802
		return token.APPLY
803
	case "end":
804
		return token.END
805
	case "Y", "year":
806
		return token.YEAR
807
	case "decimal-mark":
808
		return token.DECIMALMARK
809
	case "D":
810
		return token.D
811
	case "P":
812
		return token.P
813
	case "N":
814
		return token.N
815
	case "C":
816
		return token.C
817
	default:
818
		return token.ILLEGAL
819
	}
820
}