all repos

clerk @ 23d0a0ddf451ff6ac76c6dd2ffda2cd7d4fe34fc

missing tooling for ledger/hledger

clerk/journal/lexer/lexer.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
fix typos and grammar, 9 days ago
1
package lexer
2
3
import (
4
	"unicode"
5
	"unicode/utf8"
6
	"unsafe"
7
8
	"olexsmir.xyz/clerk/journal/token"
9
)
10
11
type mode uint
12
13
const (
14
	// start of a line, nothing consumed
15
	modeDefault mode = iota
16
17
	// after ; # * % ; at start of line, or anywhere inline
18
	// everything until \n is comment text
19
	modeComment
20
21
	// after lexing a date at column 0
22
	// expects: optional status, optional code, description, comment
23
	modeTransaction
24
25
	// after ~ or =, at the head of a periodic or automated transaction
26
	// expects: period or expression, then optional description, then optional comment
27
	modeExpr
28
29
	// after lexing an indent at start of line
30
	// expects: account name, then two spaces, then amount
31
	modePosting
32
33
	// after a directive keyword like account, commodity, include
34
	// expects: rest of directive content
35
	modeDirective
36
)
37
38
type Lexer struct {
39
	file  string
40
	input []byte
41
	mode  mode
42
43
	ch     rune // current rune (0 = EOF/sentinel)
44
	chSize int  // byte size of current rune
45
	pos    int  // current byte offset (points at ch)
46
	rpos   int  // next byte offset to read (one ahead of pos)
47
	col    int  // current column (1-based)
48
	line   int  // current line (1-based)
49
50
	transactionPastStatus bool
51
	postingExpectAccount  bool
52
	readingNoteAfterPipe  bool
53
54
	includePath bool // true after [token.INCLUDE] is omitted
55
56
	// subdirective is set while inside an account/commodity directive block;
57
	// every indented line then lexes as directive content until a
58
	// non-indented line start clears it in lexDefault
59
	subdirective bool
60
}
61
62
func New(file string, input []byte) *Lexer {
63
	l := &Lexer{
64
		file:  file,
65
		input: input,
66
		line:  1,
67
	}
68
	l.advance()
69
	if l.ch == '\uFEFF' { // start of the input
70
		l.advance()
71
	}
72
	return l
73
}
74
75
// Next returns next token in the input
76
func (l *Lexer) Next() token.Token {
77
	switch l.mode {
78
	case modeDefault:
79
		return l.lexDefault()
80
	case modeComment:
81
		return l.lexComment()
82
	case modeTransaction:
83
		return l.lexTransaction()
84
	case modePosting:
85
		return l.lexPosting()
86
	case modeExpr:
87
		return l.lexExpr()
88
	case modeDirective:
89
		return l.lexDirective()
90
	}
91
	panic("unreachable")
92
}
93
94
func (l *Lexer) lexDefault() token.Token {
95
	if l.includePath {
96
		l.includePath = false
97
	}
98
	if l.subdirective && l.ch != ' ' && l.ch != '\t' {
99
		l.subdirective = false
100
	}
101
	switch {
102
	case l.ch == 0:
103
		return l.token(token.EOF, "")
104
	case l.ch == '\n':
105
		return l.lexNewline()
106
	case l.ch == '\r':
107
		l.col = 0
108
		l.advance()
109
		return l.lexNewline()
110
	case l.ch == ' ' || l.ch == '\t':
111
		tok := l.lexIndent()
112
		if l.subdirective {
113
			// keep directive mode for every line of a directive block; the
114
			// flag is cleared at the next non-indented line start in lexDefault
115
			l.mode = modeDirective
116
		} else {
117
			l.mode = modePosting
118
		}
119
		l.postingExpectAccount = true
120
		return tok
121
	case l.ch == ';' || l.ch == '#' || l.ch == '%':
122
		l.mode = modeComment
123
		return l.lexSingle(token.SEMICOLON)
124
	case l.ch == '*': // * at col 0 == comment
125
		l.mode = modeComment
126
		return l.lexSingle(token.STAR)
127
	case l.ch == '~':
128
		l.mode = modeExpr
129
		return l.lexSingle(token.TILDE)
130
	case l.ch == '=':
131
		l.mode = modeExpr
132
		return l.lexSingle(token.EQ)
133
	case l.ch == '+':
134
		return l.lexSingle(token.PLUS)
135
	case l.ch == '-':
136
		return l.lexSingle(token.MINUS)
137
	case l.ch == '.':
138
		return l.lexSingle(token.TEXT)
139
	case l.ch == '!':
140
		return l.lexSingle(token.BANG)
141
	case l.ch == '@':
142
		return l.lexSingle(token.AT)
143
	case l.isAlpha():
144
		return l.lexKeyword()
145
	case l.isDigit():
146
		if !l.isDate() {
147
			s := l.save()
148
			for l.isDigit() || l.ch == '-' || l.ch == '/' || l.ch == '.' {
149
				l.advance()
150
			}
151
			return token.Token{Type: token.ILLEGAL, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
152
		}
153
		tok := l.lexDate()
154
		l.mode = modeTransaction
155
		l.transactionPastStatus = false
156
		return tok
157
	default:
158
		s := l.save()
159
		l.advance()
160
		return token.Token{Type: token.ILLEGAL, Literal: l.lit(s), Span: l.span(s)}
161
	}
162
}
163
164
func (l *Lexer) lexComment() token.Token {
165
	if l.ch == '\n' || l.ch == '\r' || l.ch == 0 {
166
		l.mode = modeDefault
167
		if l.ch == '\r' {
168
			l.col = 0
169
			l.advance()
170
		}
171
		return l.lexNewline()
172
	}
173
174
	for l.ch == ' ' || l.ch == '\t' {
175
		l.lexWhitespace()
176
	}
177
178
	if l.ch == '\n' || l.ch == '\r' || l.ch == 0 {
179
		l.mode = modeDefault
180
		if l.ch == '\r' {
181
			l.col = 0
182
			l.advance()
183
		}
184
		return l.lexNewline()
185
	}
186
187
	s := l.save()
188
	for l.ch != '\n' && l.ch != '\r' && l.ch != 0 {
189
		l.advance()
190
	}
191
	return token.Token{Type: token.TEXT, Literal: l.lit(s), Span: l.span(s)}
192
}
193
194
func (l *Lexer) lexTransaction() token.Token {
195
	if l.readingNoteAfterPipe {
196
		l.readingNoteAfterPipe = false
197
		return l.lexNote()
198
	}
199
200
	switch l.ch {
201
	case 0:
202
		return l.token(token.EOF, "")
203
	case '\n':
204
		l.mode = modeDefault
205
		return l.lexNewline()
206
	case '\r':
207
		l.col = 0
208
		l.advance()
209
		return l.lexNewline()
210
	case ' ', '\t':
211
		return l.lexWhitespace()
212
	case ';':
213
		l.mode = modeComment
214
		return l.lexSingle(token.SEMICOLON)
215
	case '*':
216
		if !l.transactionPastStatus {
217
			l.transactionPastStatus = true
218
			return l.lexSingle(token.STAR)
219
		}
220
		return l.lexText()
221
	case '!':
222
		if !l.transactionPastStatus {
223
			l.transactionPastStatus = true
224
			return l.lexSingle(token.BANG)
225
		}
226
		return l.lexText()
227
	case '|':
228
		l.transactionPastStatus = true
229
		l.readingNoteAfterPipe = true
230
		return l.lexSingle(token.PIPE)
231
	case '+':
232
		return l.lexSingle(token.PLUS)
233
	case '-':
234
		return l.lexSingle(token.MINUS)
235
	case '=':
236
		return l.lexEquals()
237
	case '"', '\'':
238
		return l.lexString()
239
	default: // description / payee
240
		if l.isDate() { // secondary date after =
241
			return l.lexDate()
242
		}
243
		return l.lexText()
244
	}
245
}
246
247
func (l *Lexer) lexNote() token.Token {
248
	for l.ch == ' ' || l.ch == '\t' {
249
		l.advance()
250
	}
251
	if l.ch == 0 || l.ch == '\n' || l.ch == '\r' || l.ch == ';' {
252
		return l.Next()
253
	}
254
	s := l.save()
255
	for l.ch != 0 && l.ch != '\n' && l.ch != '\r' && l.ch != ';' {
256
		l.advance()
257
	}
258
	lit := l.lit(s)
259
	for len(lit) > 0 && (lit[len(lit)-1] == ' ' || lit[len(lit)-1] == '\t') {
260
		lit = lit[:len(lit)-1]
261
	}
262
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
263
}
264
265
func (l *Lexer) lexExpr() token.Token {
266
	switch l.ch {
267
	case 0:
268
		return l.token(token.EOF, "")
269
	case '\n':
270
		l.mode = modeDefault
271
		return l.lexNewline()
272
	case '\r':
273
		l.col = 0
274
		l.advance()
275
		return l.lexNewline()
276
	case ';':
277
		l.mode = modeComment
278
		return l.lexSingle(token.SEMICOLON)
279
	case ' ', '\t':
280
		return l.lexWhitespace()
281
	default:
282
		return l.lexText()
283
	}
284
}
285
286
func (l *Lexer) lexPosting() token.Token {
287
	switch {
288
	case l.ch == 0:
289
		l.postingExpectAccount = false
290
		return l.token(token.EOF, "")
291
	case l.ch == '\n':
292
		l.postingExpectAccount = false
293
		l.mode = modeDefault
294
		return l.lexNewline()
295
	case l.ch == '\r':
296
		l.postingExpectAccount = false
297
		l.col = 0
298
		l.advance()
299
		l.mode = modeDefault
300
		return l.lexNewline()
301
	case l.ch == ';':
302
		l.postingExpectAccount = false
303
		l.mode = modeComment
304
		return l.lexSingle(token.SEMICOLON)
305
	case l.ch == ' ' || l.ch == '\t':
306
		return l.lexWhitespace()
307
	case l.postingExpectAccount && l.ch == '*':
308
		return l.lexSingle(token.STAR)
309
	case l.postingExpectAccount && l.ch == '!':
310
		return l.lexSingle(token.BANG)
311
	case l.ch == '=':
312
		return l.lexEquals()
313
	case l.ch == '@':
314
		return l.lexAt()
315
	case l.ch == '{':
316
		return l.lexLBrace()
317
	case l.ch == '}':
318
		return l.lexRBrace()
319
	case l.ch == '(':
320
		if !l.postingExpectAccount {
321
			return l.lexParenExpr()
322
		}
323
		return l.lexSingle(token.LPAREN)
324
	case l.ch == ')':
325
		return l.lexSingle(token.RPAREN)
326
	case l.ch == '[':
327
		return l.lexSingle(token.LBRACKET)
328
	case l.ch == ']':
329
		return l.lexSingle(token.RBRACKET)
330
	case l.ch == ':':
331
		return l.lexSingle(token.COLON)
332
	case l.postingExpectAccount && l.ch != '*' && l.ch != '!' && l.ch != '(' && l.ch != '[':
333
		return l.lexAccountNamePosting()
334
	case l.ch == '*': // after account name
335
		return l.lexSingle(token.STAR)
336
	case l.isDigit(), l.ch == '.':
337
		return l.lexNumber()
338
	case l.ch == '-':
339
		return l.lexSingle(token.MINUS)
340
	case l.ch == '+':
341
		return l.lexSingle(token.PLUS)
342
	case l.isCommodityStart():
343
		return l.lexCommodityMark()
344
	case l.ch == '"' || l.ch == '\'':
345
		return l.lexString()
346
	case l.ch >= 'a' && l.ch <= 'z':
347
		return l.lexCommodityMark()
348
	default:
349
		return l.lexAccountNamePosting()
350
	}
351
}
352
353
func (l *Lexer) lexDirective() token.Token {
354
	switch {
355
	case l.ch == '\n', l.ch == 0:
356
		l.mode = modeDefault
357
		return l.lexNewline()
358
	case l.ch == '\r':
359
		l.mode = modeDefault
360
		l.col = 0
361
		l.advance()
362
		return l.lexNewline()
363
	case l.ch == ';':
364
		l.mode = modeComment
365
		return l.lexSingle(token.SEMICOLON)
366
	case l.ch == ' ', l.ch == '\t':
367
		return l.lexWhitespace()
368
	case l.ch == '=':
369
		return l.lexSingle(token.EQ)
370
	case l.ch == '+':
371
		return l.lexSingle(token.PLUS)
372
	case l.ch == '-':
373
		return l.lexSingle(token.MINUS)
374
	case l.ch == '"', l.ch == '\'':
375
		return l.lexString()
376
	case l.ch == ':':
377
		return l.lexSingle(token.COLON)
378
	case l.ch == ')':
379
		return l.lexSingle(token.RPAREN)
380
	case l.ch == ']':
381
		return l.lexSingle(token.RBRACKET)
382
	case l.includePath:
383
		return l.lexPath()
384
	case l.isCommodityStart():
385
		return l.lexCommodityMark()
386
	case l.isTime():
387
		return l.lexTime()
388
	case l.isDate():
389
		return l.lexDate()
390
	case l.isDigit():
391
		return l.lexNumber()
392
	default:
393
		return l.lexAccountNameDirective()
394
	}
395
}
396
397
func (l *Lexer) lexPath() token.Token {
398
	s := l.save()
399
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ' ' && l.ch != '\t' {
400
		l.advance()
401
	}
402
	return token.Token{
403
		Type:    token.TEXT,
404
		Literal: l.lit(s),
405
		Span:    l.span(s),
406
	}
407
}
408
409
func (l *Lexer) lexSingle(kind token.Type) token.Token {
410
	s := l.save()
411
	l.advance()
412
	return token.Token{
413
		Type:    kind,
414
		Literal: l.lit(s),
415
		Span:    l.span(s),
416
	}
417
}
418
419
func (l *Lexer) lexNewline() token.Token {
420
	s := l.save()
421
	l.advance()
422
	l.mode = modeDefault
423
	return token.Token{Type: token.NEWLINE, Literal: "\n", Span: l.span(s)}
424
}
425
426
func (l *Lexer) lexWhitespace() token.Token {
427
	s := l.save()
428
	for l.ch == ' ' || l.ch == '\t' {
429
		l.advance()
430
	}
431
	return token.Token{Type: token.WHITESPACE, Literal: l.lit(s), Span: l.span(s)}
432
}
433
434
func (l *Lexer) lexIndent() token.Token {
435
	s := l.save()
436
	for l.ch == ' ' || l.ch == '\t' {
437
		l.advance()
438
	}
439
	return token.Token{Type: token.INDENT, Literal: l.lit(s), Span: l.span(s)}
440
}
441
442
func (l *Lexer) lexEquals() token.Token {
443
	s := l.save()
444
	l.advance()
445
	if l.ch == '=' {
446
		l.advance()
447
		switch l.ch {
448
		case '=':
449
			l.advance()
450
			return token.Token{Type: token.EQEQEQ, Literal: "===", Span: l.span(s)}
451
		case '*':
452
			l.advance()
453
			return token.Token{Type: token.EQEQEQ, Literal: "==*", Span: l.span(s)}
454
		default:
455
			return token.Token{Type: token.EQEQ, Literal: "==", Span: l.span(s)}
456
		}
457
	}
458
	if l.ch == '*' {
459
		l.advance()
460
		return token.Token{Type: token.EQSTAR, Literal: "=*", Span: l.span(s)}
461
	}
462
	return token.Token{Type: token.EQ, Literal: "=", Span: l.span(s)}
463
}
464
465
func (l *Lexer) lexAt() token.Token {
466
	s := l.save()
467
	l.advance()
468
	if l.ch == '@' {
469
		l.advance()
470
		return token.Token{Type: token.ATAT, Literal: "@@", Span: l.span(s)}
471
	}
472
	return token.Token{Type: token.AT, Literal: "@", Span: l.span(s)}
473
}
474
475
func (l *Lexer) lexText() token.Token {
476
	s := l.save()
477
	l.advance()
478
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ' ' && l.ch != '\t' {
479
		l.advance()
480
	}
481
	lit := string(l.input[s.offset:l.pos])
482
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
483
}
484
485
// lexAccountNameDirective reads an account name in directive context.
486
// stops at any whitespace, supports multi-word names("Taxi Fare").
487
func (l *Lexer) lexAccountNameDirective() token.Token {
488
	s := l.save()
489
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ')' && l.ch != ']' && l.ch != ':' && l.ch != ' ' && l.ch != '\t' {
490
		l.advance()
491
	}
492
	return token.Token{Type: token.TEXT, Literal: l.lit(s), Span: l.span(s)}
493
}
494
495
// lexAccountNamePosting reads an account name in posting context.
496
// stops at two consecutive spaces.
497
func (l *Lexer) lexAccountNamePosting() token.Token {
498
	s := l.save()
499
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ')' && l.ch != ']' && l.ch != ':' {
500
		if l.isTwoSpaces() {
501
			break
502
		}
503
		l.advance()
504
	}
505
	if l.ch != ':' {
506
		l.postingExpectAccount = false
507
	}
508
	return token.Token{Type: token.TEXT, Literal: l.lit(s), Span: l.span(s)}
509
}
510
511
func (l *Lexer) lexParenExpr() token.Token {
512
	s := l.save()
513
	depth := 0
514
	for l.ch != '\n' && l.ch != '\r' && l.ch != 0 {
515
		if l.ch == '(' {
516
			depth++
517
		} else if l.ch == ')' {
518
			depth--
519
			if depth == 0 {
520
				l.advance()
521
				break
522
			}
523
		}
524
		l.advance()
525
	}
526
	return token.Token{Type: token.PARENEXPR, Literal: l.lit(s), Span: l.span(s)}
527
}
528
529
func (l *Lexer) lexNumber() token.Token {
530
	s := l.save()
531
	isDecimal := false
532
	for {
533
		if l.isDigit() || l.ch == '.' || l.ch == ',' || l.ch == '_' || l.ch == '\'' {
534
			if l.ch == '.' || l.ch == ',' {
535
				isDecimal = true
536
			}
537
			l.advance()
538
		} else if l.ch == ' ' && (l.peek() >= '0' && l.peek() <= '9') {
539
			isDecimal = true
540
			l.advance()
541
		} else if l.ch == 'e' || l.ch == 'E' {
542
			// exponent: consume only when E[sign]digit, so `10E ` stays an integer
543
			p := l.pos + 1
544
			if p < len(l.input) && (l.input[p] == '+' || l.input[p] == '-') {
545
				p++
546
			}
547
			if p >= len(l.input) || l.input[p] < '0' || l.input[p] > '9' {
548
				break
549
			}
550
			isDecimal = true
551
			l.advance() // e/E
552
			if l.ch == '+' || l.ch == '-' {
553
				l.advance()
554
			}
555
			for l.isDigit() {
556
				l.advance()
557
			}
558
		} else {
559
			break
560
		}
561
	}
562
	lit := l.lit(s)
563
	kind := token.INT
564
	if isDecimal {
565
		kind = token.DECIMAL
566
	}
567
	return token.Token{Type: kind, Literal: lit, Span: l.span(s)}
568
}
569
570
func (l *Lexer) lexKeyword() token.Token {
571
	s := l.save()
572
	for l.ch != 0 && l.ch != '\n' && l.ch != '\r' && l.ch != ' ' && l.ch != '\t' && l.ch != ';' {
573
		l.advance()
574
	}
575
	lit := l.lit(s)
576
	kind := l.keyword(lit)
577
	if kind == token.ILLEGAL {
578
		kind = token.TEXT
579
	} else {
580
		l.mode = modeDirective
581
		l.subdirective = kind == token.ACCOUNT || kind == token.COMMODITY
582
		l.includePath = kind == token.INCLUDE
583
	}
584
	return token.Token{Type: kind, Literal: lit, Span: l.span(s)}
585
}
586
587
func (l *Lexer) lexDate() token.Token {
588
	s := l.save()
589
	for l.isDigit() || (l.isDateSep() && l.peekIsDigit()) {
590
		l.advance()
591
	}
592
	return token.Token{Type: token.DATE, Literal: l.lit(s), Span: l.span(s)}
593
}
594
595
func isSymbolChar(r rune) bool {
596
	return r == '$' || unicode.In(r, unicode.Sc)
597
}
598
599
func (l *Lexer) lexString() token.Token {
600
	s := l.save()
601
	quote := l.ch
602
	l.advance() // consume the quote character
603
	for l.ch != quote && l.ch != '\n' && l.ch != 0 {
604
		l.advance()
605
	}
606
	if l.ch == quote {
607
		l.advance()
608
	}
609
	return token.Token{Type: token.STRING, Literal: l.lit(s), Span: l.span(s)}
610
}
611
612
func (l *Lexer) lexCommodityMark() token.Token {
613
	s := l.save()
614
615
	if l.ch == '"' {
616
		l.advance()
617
		for l.ch != '"' && l.ch != '\n' && l.ch != 0 {
618
			l.advance()
619
		}
620
		if l.ch == '"' {
621
			l.advance()
622
		}
623
		return token.Token{Type: token.COMMODITYMARK, Literal: l.lit(s), Span: l.span(s)}
624
	}
625
626
	if unicode.IsLetter(l.ch) {
627
		for unicode.IsLetter(l.ch) {
628
			l.advance()
629
		}
630
		return token.Token{Type: token.COMMODITYMARK, Literal: l.lit(s), Span: l.span(s)}
631
	}
632
633
	if isSymbolChar(l.ch) {
634
		for isSymbolChar(l.ch) {
635
			l.advance()
636
		}
637
		return token.Token{Type: token.COMMODITYMARK, Literal: l.lit(s), Span: l.span(s)}
638
	}
639
640
	l.advance()
641
	return token.Token{Type: token.COMMODITYMARK, Literal: l.lit(s), Span: l.span(s)}
642
}
643
644
func (l *Lexer) lexLBrace() token.Token {
645
	s := l.save()
646
	l.advance()
647
	if l.ch == '{' {
648
		l.advance()
649
		return token.Token{Type: token.LBRACELBRACE, Literal: "{{", Span: l.span(s)}
650
	}
651
	return token.Token{Type: token.LBRACE, Literal: "{", Span: l.span(s)}
652
}
653
654
func (l *Lexer) lexRBrace() token.Token {
655
	s := l.save()
656
	l.advance()
657
	if l.ch == '}' {
658
		l.advance()
659
		return token.Token{Type: token.RBRACERBRACE, Literal: "}}", Span: l.span(s)}
660
	}
661
	return token.Token{Type: token.RBRACE, Literal: "}", Span: l.span(s)}
662
}
663
664
func (l *Lexer) advance() {
665
	if l.rpos >= len(l.input) {
666
		l.ch = 0
667
		l.chSize = 0
668
	} else if b := l.input[l.rpos]; b < utf8.RuneSelf { // ASCII fast path
669
		l.ch = rune(b)
670
		l.chSize = 1
671
	} else {
672
		l.ch, l.chSize = utf8.DecodeRune(l.input[l.rpos:])
673
	}
674
	l.pos = l.rpos
675
	l.rpos += l.chSize
676
	if l.ch == '\n' || l.ch == '\r' {
677
		l.line++
678
		l.col = 0
679
	} else {
680
		l.col++
681
	}
682
}
683
684
func (l *Lexer) peek() rune {
685
	if l.rpos < len(l.input) && l.input[l.rpos] < utf8.RuneSelf {
686
		return rune(l.input[l.rpos])
687
	}
688
	r, _ := utf8.DecodeRune(l.input[l.rpos:])
689
	return r
690
}
691
692
func (l *Lexer) peekN(n int) byte {
693
	if l.pos+n >= len(l.input) {
694
		return 0
695
	}
696
	return l.input[l.pos+n]
697
}
698
699
func (l *Lexer) isDigit() bool { return l.ch >= '0' && l.ch <= '9' }
700
func (l *Lexer) isAlpha() bool {
701
	return (l.ch >= 'a' && l.ch <= 'z') ||
702
		(l.ch >= 'A' && l.ch <= 'Z')
703
}
704
705
func (l *Lexer) isTwoSpaces() bool {
706
	return l.ch == ' ' && l.rpos < len(l.input) && l.input[l.rpos] == ' '
707
}
708
709
func (l *Lexer) isDateSep() bool { return l.ch == '-' || l.ch == '/' || l.ch == '.' }
710
711
func (l *Lexer) peekIsDigit() bool {
712
	r := l.peek()
713
	return r >= '0' && r <= '9'
714
}
715
716
func (l *Lexer) isCommodityStart() bool {
717
	if l.ch == '$' || (l.ch >= 'A' && l.ch <= 'Z') {
718
		return true
719
	}
720
	if l.ch < utf8.RuneSelf {
721
		return false
722
	}
723
	return unicode.In(l.ch, unicode.Sc) || unicode.IsLetter(l.ch)
724
}
725
726
func (l *Lexer) isDate() bool {
727
	if !l.isDigit() {
728
		return false
729
	}
730
	// YYYY/M/D or YYYY/MM/DD
731
	if l.peekN(1) >= '0' && l.peekN(1) <= '9' &&
732
		l.peekN(2) >= '0' && l.peekN(2) <= '9' &&
733
		l.peekN(3) >= '0' && l.peekN(3) <= '9' {
734
		sep := l.peekN(4)
735
		if sep == '/' || sep == '-' || sep == '.' {
736
			if l.peekN(5) >= '0' && l.peekN(5) <= '9' {
737
				if l.peekN(6) == sep {
738
					return l.peekN(7) >= '0' && l.peekN(7) <= '9'
739
				}
740
				if l.peekN(7) == sep {
741
					return l.peekN(8) >= '0' && l.peekN(8) <= '9'
742
				}
743
			}
744
		}
745
		return false
746
	}
747
	// M/D or MM/DD(year inferred, only / and - separators; . is ambiguous with decimal numbers like 1.01)
748
	if (l.peekN(1) == '/' || l.peekN(1) == '-') &&
749
		l.peekN(2) >= '0' && l.peekN(2) <= '9' &&
750
		l.ch >= '1' && l.ch <= '9' {
751
		return validDay(l.peekN(2), l.peekN(3))
752
	}
753
	if (l.peekN(2) == '/' || l.peekN(2) == '-') &&
754
		l.peekN(3) >= '0' && l.peekN(3) <= '9' {
755
		m := int(l.ch-'0')*10 + int(l.peekN(1)-'0')
756
		return m >= 1 && m <= 12 && validDay(l.peekN(3), l.peekN(4))
757
	}
758
	return false
759
}
760
761
func validDay(first, second byte) bool {
762
	d := int(first - '0')
763
	if second >= '0' && second <= '9' {
764
		d = d*10 + int(second-'0')
765
	}
766
	return d >= 1 && d <= 31
767
}
768
769
func (l *Lexer) isTime() bool {
770
	if !l.isDigit() {
771
		return false
772
	}
773
	return l.peekN(2) == ':'
774
}
775
776
func (l *Lexer) lexTime() token.Token {
777
	s := l.save()
778
	for l.isDigit() || l.ch == ':' {
779
		l.advance()
780
	}
781
	return token.Token{Type: token.TIME, Literal: l.lit(s), Span: l.span(s)}
782
}
783
784
type savedPos struct{ offset, line, col int }
785
786
func (l *Lexer) save() savedPos {
787
	return savedPos{l.pos, l.line, l.col}
788
}
789
790
func (l *Lexer) span(s savedPos) token.Span {
791
	return token.Span{
792
		File:  l.file,
793
		Start: token.Pos{Offset: s.offset, Line: s.line, Col: s.col},
794
		End:   token.Pos{Offset: l.pos, Line: l.line, Col: l.col},
795
	}
796
}
797
798
func (l *Lexer) token(kind token.Type, literal string) token.Token {
799
	s := savedPos{l.pos, l.line, l.col}
800
	return token.Token{Type: kind, Literal: literal, Span: l.span(s)}
801
}
802
803
func (l *Lexer) lit(s savedPos) string {
804
	if s.offset == l.pos {
805
		return ""
806
	}
807
	return unsafe.String(&l.input[s.offset], l.pos-s.offset)
808
}
809
810
func (l *Lexer) keyword(s string) token.Type {
811
	switch s {
812
	case "comment":
813
		return token.COMMENTKW
814
	case "account":
815
		return token.ACCOUNT
816
	case "commodity":
817
		return token.COMMODITY
818
	case "include":
819
		return token.INCLUDE
820
	case "alias":
821
		return token.ALIAS
822
	case "payee":
823
		return token.PAYEE
824
	case "tag":
825
		return token.TAG
826
	case "apply":
827
		return token.APPLY
828
	case "end":
829
		return token.END
830
	case "Y", "year":
831
		return token.YEAR
832
	case "decimal-mark":
833
		return token.DECIMALMARK
834
	case "D":
835
		return token.D
836
	case "P":
837
		return token.P
838
	case "N":
839
		return token.N
840
	case "C":
841
		return token.C
842
	default:
843
		return token.ILLEGAL
844
	}
845
}