all repos

clerk @ 1f6bd77

missing tooling for ledger/hledger

clerk/journal/lexer/lexer.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
lsp: completion, 2 months ago
1
package lexer
2
3
import (
4
	"strings"
5
	"unicode"
6
	"unicode/utf8"
7
8
	"olexsmir.xyz/clerk/journal/token"
9
)
10
11
type Mode uint
12
13
const (
14
	// start of a line, nothing consumed
15
	ModeDefault Mode = iota
16
17
	// after ; # * % ; at start of line, or anywhere inline
18
	// everything until \n is comment text
19
	ModeComment
20
21
	// after lexing a date at column 0
22
	// expects: optional status, optional code, description, comment
23
	ModeTransaction
24
25
	// after lexing an indent at start of line
26
	// expects: account name, then two spaces, then amount
27
	ModePosting
28
29
	// after ~, period expression
30
	// expects: period, optional description (after 2+ spaces), optional comment
31
	ModePeriodic
32
33
	// after =, automates transaction
34
	// expects: expression
35
	ModeAutomated
36
37
	// after a directive keyword like account, commodity, include
38
	// expects: rest of directive content
39
	ModeDirective
40
)
41
42
type Lexer struct {
43
	file  string
44
	input []byte
45
	mode  Mode
46
47
	ch     rune // current rune (0 = EOF/sentinel)
48
	chSize int  // byte size of current rune
49
	pos    int  // current byte offset (points at ch)
50
	rpos   int  // next byte offset to read (one ahead of pos)
51
	col    int  // current column (1-based)
52
	line   int  // current line (1-based)
53
54
	transactionPastStatus bool
55
	postingExpectAccount  bool
56
	readingNoteAfterPipe  bool
57
58
	// subdirective is set when the current line is an account/commodity
59
	// directive; the next indented line then lexes as directive content
60
	subdirective bool
61
}
62
63
func New(file string, input []byte) *Lexer {
64
	l := &Lexer{
65
		file:  file,
66
		input: input,
67
		line:  1,
68
	}
69
	l.advance()
70
	if l.ch == '\uFEFF' { // start of the input
71
		l.advance()
72
	}
73
	return l
74
}
75
76
// Next returns next token in the input
77
func (l *Lexer) Next() token.Token {
78
	switch l.mode {
79
	case ModeDefault:
80
		return l.lexDefault()
81
	case ModeComment:
82
		return l.lexComment()
83
	case ModeTransaction:
84
		return l.lexTransaction()
85
	case ModePosting:
86
		return l.lexPosting()
87
	case ModePeriodic:
88
		return l.lexPeriodic()
89
	case ModeAutomated:
90
		return l.lexAutomated()
91
	case ModeDirective:
92
		return l.lexDirective()
93
	}
94
	panic("unreachable")
95
}
96
97
func (l *Lexer) lexDefault() token.Token {
98
	if l.subdirective && l.ch != ' ' && l.ch != '\t' {
99
		l.subdirective = false
100
	}
101
	switch {
102
	case l.ch == 0:
103
		return l.token(token.EOF, "")
104
	case l.ch == '\n':
105
		return l.lexNewline()
106
	case l.ch == '\r':
107
		l.col = 0
108
		l.advance()
109
		return l.lexNewline()
110
	case l.ch == ' ' || l.ch == '\t':
111
		tok := l.lexIndent()
112
		if l.subdirective {
113
			l.mode = ModeDirective
114
			l.subdirective = false
115
		} else {
116
			l.mode = ModePosting
117
		}
118
		l.postingExpectAccount = true
119
		return tok
120
	case l.ch == ';' || l.ch == '#' || l.ch == '%':
121
		l.mode = ModeComment
122
		return l.lexSingle(token.SEMICOLON) // todo: ??
123
	case l.ch == '*': // * at col 0 == comment
124
		l.mode = ModeComment
125
		return l.lexSingle(token.STAR)
126
	case l.ch == '~':
127
		l.mode = ModePeriodic
128
		return l.lexSingle(token.TILDE)
129
	case l.ch == '=':
130
		l.mode = ModeAutomated
131
		return l.lexSingle(token.EQ)
132
	case l.ch == '+':
133
		return l.lexSingle(token.PLUS)
134
	case l.ch == '-':
135
		return l.lexSingle(token.MINUS)
136
	case l.ch == '.':
137
		return l.lexSingle(token.TEXT)
138
	case l.ch == '!':
139
		return l.lexSingle(token.BANG)
140
	case l.ch == '@':
141
		return l.lexSingle(token.AT)
142
	case l.isAlpha():
143
		return l.lexKeyword()
144
	case l.isDigit():
145
		if !l.isDate() {
146
			s := l.save()
147
			for l.isDigit() || l.ch == '-' || l.ch == '/' || l.ch == '.' {
148
				l.advance()
149
			}
150
			return token.Token{Type: token.ILLEGAL, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
151
		}
152
		tok := l.lexDate()
153
		l.mode = ModeTransaction
154
		l.transactionPastStatus = false
155
		return tok
156
	default:
157
		s := l.save()
158
		l.advance()
159
		return token.Token{Type: token.ILLEGAL, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
160
	}
161
}
162
163
func (l *Lexer) lexComment() token.Token {
164
	if l.ch == '\n' || l.ch == '\r' || l.ch == 0 {
165
		l.mode = ModeDefault
166
		if l.ch == '\r' {
167
			l.col = 0
168
			l.advance()
169
		}
170
		return l.lexNewline()
171
	}
172
173
	for l.ch == ' ' || l.ch == '\t' {
174
		l.lexWhitespace()
175
	}
176
177
	if l.ch == '\n' || l.ch == '\r' || l.ch == 0 {
178
		l.mode = ModeDefault
179
		if l.ch == '\r' {
180
			l.col = 0
181
			l.advance()
182
		}
183
		return l.lexNewline()
184
	}
185
186
	s := l.save()
187
	for l.ch != '\n' && l.ch != '\r' && l.ch != 0 {
188
		l.advance()
189
	}
190
	return token.Token{Type: token.TEXT, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
191
}
192
193
func (l *Lexer) lexTransaction() token.Token {
194
	if l.readingNoteAfterPipe {
195
		l.readingNoteAfterPipe = false
196
		return l.lexNote()
197
	}
198
199
	switch l.ch {
200
	case 0:
201
		return l.token(token.EOF, "")
202
	case '\n':
203
		l.mode = ModeDefault
204
		return l.lexNewline()
205
	case '\r':
206
		l.col = 0
207
		l.advance()
208
		return l.lexNewline()
209
	case ' ', '\t':
210
		return l.lexWhitespace()
211
	case ';':
212
		l.mode = ModeComment
213
		return l.lexSingle(token.SEMICOLON)
214
	case '*':
215
		if !l.transactionPastStatus {
216
			l.transactionPastStatus = true
217
			return l.lexSingle(token.STAR)
218
		}
219
		return l.lexText()
220
	case '!':
221
		if !l.transactionPastStatus {
222
			l.transactionPastStatus = true
223
			return l.lexSingle(token.BANG)
224
		}
225
		return l.lexText()
226
	case '|':
227
		l.transactionPastStatus = true
228
		l.readingNoteAfterPipe = true
229
		return l.lexSingle(token.PIPE)
230
	case '+':
231
		return l.lexSingle(token.PLUS)
232
	case '-':
233
		return l.lexSingle(token.MINUS)
234
	case '=':
235
		return l.lexEquals()
236
	case '"', '\'':
237
		return l.lexString()
238
	default: // description / payee
239
		if l.isDate() { // secondsry date after =
240
			return l.lexDate()
241
		}
242
		return l.lexText()
243
	}
244
}
245
246
func (l *Lexer) lexNote() token.Token {
247
	for l.ch == ' ' || l.ch == '\t' {
248
		l.advance()
249
	}
250
	if l.ch == 0 || l.ch == '\n' || l.ch == ';' {
251
		return l.Next()
252
	}
253
	s := l.save()
254
	for l.ch != 0 && l.ch != '\n' && l.ch != '\r' && l.ch != ';' {
255
		l.advance()
256
	}
257
	lit := string(l.input[s.offset:l.pos])
258
	for len(lit) > 0 && (lit[len(lit)-1] == ' ' || lit[len(lit)-1] == '\t') {
259
		lit = lit[:len(lit)-1]
260
	}
261
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
262
}
263
264
func (l *Lexer) lexPeriodic() token.Token {
265
	switch l.ch {
266
	case 0:
267
		return l.token(token.EOF, "")
268
	case '\n':
269
		l.mode = ModeDefault
270
		return l.lexNewline()
271
	case '\r':
272
		l.col = 0
273
		l.advance()
274
		return l.lexNewline()
275
	case ';':
276
		l.mode = ModeComment
277
		return l.lexSingle(token.SEMICOLON)
278
	case ' ', '\t':
279
		return l.lexWhitespace()
280
	default:
281
		return l.lexText()
282
	}
283
}
284
285
func (l *Lexer) lexAutomated() token.Token {
286
	switch l.ch {
287
	case 0:
288
		return l.token(token.EOF, "")
289
	case '\n':
290
		l.mode = ModeDefault
291
		return l.lexNewline()
292
	case '\r':
293
		l.col = 0
294
		l.advance()
295
		return l.lexNewline()
296
	case ' ', '\t':
297
		return l.lexWhitespace()
298
	case ';':
299
		l.mode = ModeComment
300
		return l.lexSingle(token.SEMICOLON)
301
	default:
302
		return l.lexText()
303
	}
304
}
305
306
func (l *Lexer) lexPosting() token.Token {
307
	switch {
308
	case l.ch == 0:
309
		l.postingExpectAccount = false
310
		return l.token(token.EOF, "")
311
	case l.ch == '\n':
312
		l.postingExpectAccount = false
313
		l.mode = ModeDefault
314
		return l.lexNewline()
315
	case l.ch == '\r':
316
		l.postingExpectAccount = false
317
		l.col = 0
318
		l.advance()
319
		l.mode = ModeDefault
320
		return l.lexNewline()
321
	case l.ch == ';':
322
		l.postingExpectAccount = false
323
		l.mode = ModeComment
324
		return l.lexSingle(token.SEMICOLON)
325
	case l.ch == ' ' || l.ch == '\t':
326
		return l.lexWhitespace()
327
	case l.postingExpectAccount && l.ch == '*':
328
		return l.lexSingle(token.STAR)
329
	case l.postingExpectAccount && l.ch == '!':
330
		return l.lexSingle(token.BANG)
331
	case l.ch == '=':
332
		return l.lexEquals()
333
	case l.ch == '@':
334
		return l.lexAt()
335
	case l.ch == '{':
336
		return l.lexLBrace()
337
	case l.ch == '}':
338
		return l.lexRBrace()
339
	case l.ch == '(':
340
		if !l.postingExpectAccount {
341
			return l.lexParenExpr()
342
		}
343
		return l.lexSingle(token.LPAREN)
344
	case l.ch == ')':
345
		return l.lexSingle(token.RPAREN)
346
	case l.ch == '[':
347
		return l.lexSingle(token.LBRACKET)
348
	case l.ch == ']':
349
		return l.lexSingle(token.RBRACKET)
350
	case l.ch == ':':
351
		return l.lexSingle(token.COLON)
352
	case l.postingExpectAccount && l.ch != '*' && l.ch != '!' && l.ch != '(' && l.ch != '[':
353
		return l.lexAccountNamePosting()
354
	case l.ch == '*': // after account name
355
		return l.lexSingle(token.STAR)
356
	case l.isDigit(), l.ch == '.':
357
		return l.lexNumber()
358
	case l.ch == '-':
359
		return l.lexSingle(token.MINUS)
360
	case l.ch == '+':
361
		return l.lexSingle(token.PLUS)
362
	case l.isCommodityStart():
363
		return l.lexCommodityMark()
364
	case l.ch == '"' || l.ch == '\'':
365
		return l.lexString()
366
	case l.ch >= 'a' && l.ch <= 'z':
367
		return l.lexCommodityMark()
368
	default:
369
		return l.lexAccountNamePosting()
370
	}
371
}
372
373
func (l *Lexer) lexDirective() token.Token {
374
	switch l.ch {
375
	case '\n', 0:
376
		l.mode = ModeDefault
377
		return l.lexNewline()
378
	case '\r':
379
		l.mode = ModeDefault
380
		l.col = 0
381
		l.advance()
382
		return l.lexNewline()
383
	case ';':
384
		l.mode = ModeComment
385
		return l.lexSingle(token.SEMICOLON)
386
	case ' ', '\t':
387
		return l.lexWhitespace()
388
	case '=':
389
		return l.lexSingle(token.EQ)
390
	case '+':
391
		return l.lexSingle(token.PLUS)
392
	case '-':
393
		return l.lexSingle(token.MINUS)
394
	case '"', '\'':
395
		return l.lexString()
396
	case ':':
397
		return l.lexSingle(token.COLON)
398
	case ')':
399
		return l.lexSingle(token.RPAREN)
400
	case ']':
401
		return l.lexSingle(token.RBRACKET)
402
	default:
403
		if l.isCommodityStart() {
404
			return l.lexCommodityMark()
405
		}
406
		if l.isTime() {
407
			return l.lexTime()
408
		}
409
		if l.isDate() {
410
			return l.lexDate()
411
		}
412
		if l.isDigit() {
413
			return l.lexNumber()
414
		}
415
		return l.lexAccountNameDirective()
416
	}
417
}
418
419
func (l *Lexer) lexSingle(kind token.Type) token.Token {
420
	s := l.save()
421
	l.advance()
422
	return token.Token{
423
		Type:    kind,
424
		Literal: string(l.input[s.offset:l.pos]),
425
		Span:    l.span(s),
426
	}
427
}
428
429
func (l *Lexer) lexNewline() token.Token {
430
	s := l.save()
431
	l.advance()
432
	l.mode = ModeDefault
433
	return token.Token{Type: token.NEWLINE, Literal: "\n", Span: l.span(s)}
434
}
435
436
func (l *Lexer) lexWhitespace() token.Token {
437
	s := l.save()
438
	for l.ch == ' ' || l.ch == '\t' {
439
		l.advance()
440
	}
441
	lit := string(l.input[s.offset:l.pos])
442
	return token.Token{Type: token.WHITESPACE, Literal: lit, Span: l.span(s)}
443
}
444
445
func (l *Lexer) lexIndent() token.Token {
446
	s := l.save()
447
	for l.ch == ' ' || l.ch == '\t' {
448
		l.advance()
449
	}
450
	lit := string(l.input[s.offset:l.pos])
451
	return token.Token{Type: token.INDENT, Literal: lit, Span: l.span(s)}
452
}
453
454
func (l *Lexer) lexEquals() token.Token {
455
	s := l.save()
456
	l.advance()
457
	if l.ch == '=' {
458
		l.advance()
459
		switch l.ch {
460
		case '=':
461
			l.advance()
462
			return token.Token{Type: token.EQEQEQ, Literal: "===", Span: l.span(s)}
463
		case '*':
464
			l.advance()
465
			return token.Token{Type: token.EQEQEQ, Literal: "==*", Span: l.span(s)}
466
		default:
467
			return token.Token{Type: token.EQEQ, Literal: "==", Span: l.span(s)}
468
		}
469
	}
470
	if l.ch == '*' {
471
		l.advance()
472
		return token.Token{Type: token.EQSTAR, Literal: "=*", Span: l.span(s)}
473
	}
474
	return token.Token{Type: token.EQ, Literal: "=", Span: l.span(s)}
475
}
476
477
func (l *Lexer) lexAt() token.Token {
478
	s := l.save()
479
	l.advance()
480
	if l.ch == '@' {
481
		l.advance()
482
		return token.Token{Type: token.ATAT, Literal: "@@", Span: l.span(s)}
483
	}
484
	return token.Token{Type: token.AT, Literal: "@", Span: l.span(s)}
485
}
486
487
func (l *Lexer) lexText() token.Token {
488
	s := l.save()
489
	l.advance()
490
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ' ' && l.ch != '\t' {
491
		l.advance()
492
	}
493
	lit := string(l.input[s.offset:l.pos])
494
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
495
}
496
497
// lexAccountNameDirective reads accout name in directive context.
498
// stops at any whitespace, supports multi-word names("Taxi Fare").
499
func (l *Lexer) lexAccountNameDirective() token.Token {
500
	s := l.save()
501
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ')' && l.ch != ']' && l.ch != ':' && l.ch != ' ' && l.ch != '\t' {
502
		l.advance()
503
	}
504
	lit := string(l.input[s.offset:l.pos])
505
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
506
}
507
508
// lexAccountNamePosting reads an account name in posting context.
509
// stops at two consecutive spaces.
510
func (l *Lexer) lexAccountNamePosting() token.Token {
511
	s := l.save()
512
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ')' && l.ch != ']' && l.ch != ':' {
513
		if l.isTwoSpaces() {
514
			break
515
		}
516
		l.advance()
517
	}
518
	if l.ch != ':' {
519
		l.postingExpectAccount = false
520
	}
521
	lit := string(l.input[s.offset:l.pos])
522
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
523
}
524
525
func (l *Lexer) lexParenExpr() token.Token {
526
	s := l.save()
527
	depth := 0
528
	for l.ch != '\n' && l.ch != '\r' && l.ch != 0 {
529
		if l.ch == '(' {
530
			depth++
531
		} else if l.ch == ')' {
532
			depth--
533
			if depth == 0 {
534
				l.advance()
535
				break
536
			}
537
		}
538
		l.advance()
539
	}
540
	lit := string(l.input[s.offset:l.pos])
541
	return token.Token{Type: token.PARENEXPR, Literal: lit, Span: l.span(s)}
542
}
543
544
func (l *Lexer) lexNumber() token.Token {
545
	s := l.save()
546
	for {
547
		if l.isDigit() || l.ch == '.' || l.ch == ',' || l.ch == '_' || l.ch == '\'' {
548
			l.advance()
549
		} else if l.ch == ' ' && (l.peek() >= '0' && l.peek() <= '9') {
550
			l.advance()
551
		} else if l.ch == 'e' || l.ch == 'E' {
552
			// exponent: consume only when E[sign]digit, so `10E ` stays an integer
553
			p := l.pos + 1
554
			if p < len(l.input) && (l.input[p] == '+' || l.input[p] == '-') {
555
				p++
556
			}
557
			if p >= len(l.input) || l.input[p] < '0' || l.input[p] > '9' {
558
				break
559
			}
560
			l.advance() // e/E
561
			if l.ch == '+' || l.ch == '-' {
562
				l.advance()
563
			}
564
			for l.isDigit() {
565
				l.advance()
566
			}
567
		} else {
568
			break
569
		}
570
	}
571
	lit := string(l.input[s.offset:l.pos])
572
	kind := token.INT
573
	if strings.ContainsAny(lit, "., eE") {
574
		kind = token.DECIMAL
575
	}
576
	return token.Token{Type: kind, Literal: lit, Span: l.span(s)}
577
}
578
579
func (l *Lexer) lexKeyword() token.Token {
580
	s := l.save()
581
	for l.ch != 0 && l.ch != '\n' && l.ch != '\r' && l.ch != ' ' && l.ch != '\t' && l.ch != ';' {
582
		l.advance()
583
	}
584
	lit := string(l.input[s.offset:l.pos])
585
	kind := l.keyword(lit)
586
	if kind == token.ILLEGAL { // todo: report an error ??
587
		kind = token.TEXT
588
	} else {
589
		l.mode = ModeDirective
590
		l.subdirective = lit == "account" || lit == "commodity"
591
	}
592
	return token.Token{Type: kind, Literal: lit, Span: l.span(s)}
593
}
594
595
func (l *Lexer) lexDate() token.Token {
596
	s := l.save()
597
	for l.isDigit() || (l.isDateSep() && l.peekIsDigit()) {
598
		l.advance()
599
	}
600
	return token.Token{Type: token.DATE, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
601
}
602
603
func isSymbolChar(r rune) bool {
604
	return r == '$' || unicode.In(r, unicode.Sc)
605
}
606
607
func (l *Lexer) lexString() token.Token {
608
	s := l.save()
609
	quote := l.ch
610
	l.advance() // consume the quote character
611
	for l.ch != quote && l.ch != '\n' && l.ch != 0 {
612
		l.advance()
613
	}
614
	if l.ch == quote {
615
		l.advance()
616
	}
617
	return token.Token{Type: token.STRING, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
618
}
619
620
func (l *Lexer) lexCommodityMark() token.Token {
621
	s := l.save()
622
623
	if l.ch == '"' {
624
		l.advance()
625
		for l.ch != '"' && l.ch != '\n' && l.ch != 0 {
626
			l.advance()
627
		}
628
		if l.ch == '"' {
629
			l.advance()
630
		}
631
		return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
632
	}
633
634
	if unicode.IsLetter(l.ch) {
635
		for unicode.IsLetter(l.ch) {
636
			l.advance()
637
		}
638
		return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
639
	}
640
641
	if isSymbolChar(l.ch) {
642
		for isSymbolChar(l.ch) {
643
			l.advance()
644
		}
645
		return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
646
	}
647
648
	l.advance()
649
	return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
650
}
651
652
func (l *Lexer) lexLBrace() token.Token {
653
	s := l.save()
654
	l.advance()
655
	if l.ch == '{' {
656
		l.advance()
657
		return token.Token{Type: token.LBRACELBRACE, Literal: "{{", Span: l.span(s)}
658
	}
659
	return token.Token{Type: token.LBRACE, Literal: "{", Span: l.span(s)}
660
}
661
662
func (l *Lexer) lexRBrace() token.Token {
663
	s := l.save()
664
	l.advance()
665
	if l.ch == '}' {
666
		l.advance()
667
		return token.Token{Type: token.RBRACERBRACE, Literal: "}}", Span: l.span(s)}
668
	}
669
	return token.Token{Type: token.RBRACE, Literal: "}", Span: l.span(s)}
670
}
671
672
func (l *Lexer) advance() {
673
	if l.rpos >= len(l.input) {
674
		l.ch = 0
675
		l.chSize = 0
676
	} else {
677
		r, size := utf8.DecodeRune(l.input[l.rpos:])
678
		l.ch = r
679
		l.chSize = size
680
	}
681
	l.pos = l.rpos
682
	l.rpos += l.chSize
683
	if l.ch == '\n' || l.ch == '\r' {
684
		l.line++
685
		l.col = 0
686
	} else {
687
		l.col++
688
	}
689
}
690
691
func (l *Lexer) peek() rune {
692
	r, _ := utf8.DecodeRune(l.input[l.rpos:])
693
	return r
694
}
695
696
func (l *Lexer) peekN(n int) byte {
697
	if l.pos+n >= len(l.input) {
698
		return 0
699
	}
700
	return l.input[l.pos+n]
701
}
702
703
func (l *Lexer) isDigit() bool { return l.ch >= '0' && l.ch <= '9' }
704
func (l *Lexer) isAlpha() bool {
705
	return (l.ch >= 'a' && l.ch <= 'z') ||
706
		(l.ch >= 'A' && l.ch <= 'Z')
707
}
708
709
func (l *Lexer) isTwoSpaces() bool { return l.ch == ' ' && l.peek() == ' ' }
710
711
func (l *Lexer) isDateSep() bool { return l.ch == '-' || l.ch == '/' || l.ch == '.' }
712
713
func (l *Lexer) peekIsDigit() bool {
714
	r := l.peek()
715
	return r >= '0' && r <= '9'
716
}
717
718
func (l *Lexer) isCommodityStart() bool {
719
	if l.ch == '$' || (l.ch >= 'A' && l.ch <= 'Z') {
720
		return true
721
	}
722
	if l.ch < utf8.RuneSelf {
723
		return false
724
	}
725
	return unicode.In(l.ch, unicode.Sc) || unicode.IsLetter(l.ch)
726
}
727
728
func (l *Lexer) isDate() bool {
729
	if !l.isDigit() {
730
		return false
731
	}
732
	// YYYY/M/D or YYYY/MM/DD
733
	if l.peekN(1) >= '0' && l.peekN(1) <= '9' &&
734
		l.peekN(2) >= '0' && l.peekN(2) <= '9' &&
735
		l.peekN(3) >= '0' && l.peekN(3) <= '9' {
736
		sep := l.peekN(4)
737
		if sep == '/' || sep == '-' || sep == '.' {
738
			if l.peekN(5) >= '0' && l.peekN(5) <= '9' {
739
				if l.peekN(6) == sep {
740
					return l.peekN(7) >= '0' && l.peekN(7) <= '9'
741
				}
742
				if l.peekN(7) == sep {
743
					return l.peekN(8) >= '0' && l.peekN(8) <= '9'
744
				}
745
			}
746
		}
747
		return false
748
	}
749
	// M/D or MM/DD(year inferred, only / and - separators; . is ambiguous with decimal numbers like 1.01)
750
	if (l.peekN(1) == '/' || l.peekN(1) == '-') &&
751
		l.peekN(2) >= '0' && l.peekN(2) <= '9' &&
752
		l.ch >= '1' && l.ch <= '9' {
753
		return validDay(l.peekN(2), l.peekN(3))
754
	}
755
	if (l.peekN(2) == '/' || l.peekN(2) == '-') &&
756
		l.peekN(3) >= '0' && l.peekN(3) <= '9' {
757
		m := int(l.ch-'0')*10 + int(l.peekN(1)-'0')
758
		return m >= 1 && m <= 12 && validDay(l.peekN(3), l.peekN(4))
759
	}
760
	return false
761
}
762
763
func validDay(first, second byte) bool {
764
	d := int(first - '0')
765
	if second >= '0' && second <= '9' {
766
		d = d*10 + int(second-'0')
767
	}
768
	return d >= 1 && d <= 31
769
}
770
771
func (l *Lexer) isTime() bool {
772
	if !l.isDigit() {
773
		return false
774
	}
775
	return l.peekN(2) == ':'
776
}
777
778
func (l *Lexer) lexTime() token.Token {
779
	s := l.save()
780
	for l.isDigit() || l.ch == ':' {
781
		l.advance()
782
	}
783
	return token.Token{Type: token.TIME, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
784
}
785
786
type savedPos struct{ offset, line, col int }
787
788
func (l *Lexer) save() savedPos {
789
	return savedPos{l.pos, l.line, l.col}
790
}
791
792
func (l *Lexer) span(s savedPos) token.Span {
793
	return token.Span{
794
		Start: token.Pos{File: l.file, Offset: s.offset, Line: s.line, Col: s.col},
795
		End:   token.Pos{File: l.file, Offset: l.pos, Line: l.line, Col: l.col},
796
	}
797
}
798
799
func (l *Lexer) token(kind token.Type, literal string) token.Token {
800
	s := savedPos{l.pos, l.line, l.col}
801
	return token.Token{Type: kind, Literal: literal, Span: l.span(s)}
802
}
803
804
func (l *Lexer) keyword(s string) token.Type {
805
	switch s {
806
	case "comment":
807
		return token.COMMENTKW
808
	case "account":
809
		return token.ACCOUNT
810
	case "commodity":
811
		return token.COMMODITY
812
	case "include":
813
		return token.INCLUDE
814
	case "alias":
815
		return token.ALIAS
816
	case "payee":
817
		return token.PAYEE
818
	case "tag":
819
		return token.TAG
820
	case "apply":
821
		return token.APPLY
822
	case "end":
823
		return token.END
824
	case "Y", "year":
825
		return token.YEAR
826
	case "decimal-mark":
827
		return token.DECIMALMARK
828
	case "D":
829
		return token.D
830
	case "P":
831
		return token.P
832
	case "N":
833
		return token.N
834
	case "C":
835
		return token.C
836
	default:
837
		return token.ILLEGAL
838
	}
839
}