all repos

clerk @ a2ef079f64a5ad301b0e6f4425931ebbb2657a79

missing tooling for ledger/hledger

clerk/journal/lexer/lexer.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
lexer: return empty literal for empty ranges in l.lit, 1 month ago
1
package lexer
2
3
import (
4
	"strings"
5
	"unicode"
6
	"unicode/utf8"
7
	"unsafe"
8
9
	"olexsmir.xyz/clerk/journal/token"
10
)
11
12
type mode uint
13
14
const (
15
	// start of a line, nothing consumed
16
	modeDefault mode = iota
17
18
	// after ; # * % ; at start of line, or anywhere inline
19
	// everything until \n is comment text
20
	modeComment
21
22
	// after lexing a date at column 0
23
	// expects: optional status, optional code, description, comment
24
	modeTransaction
25
26
	// after lexing an indent at start of line
27
	// expects: account name, then two spaces, then amount
28
	modePosting
29
30
	// after ~, period expression
31
	// expects: period, optional description (after 2+ spaces), optional comment
32
	modePeriodic
33
34
	// after =, automates transaction
35
	// expects: expression
36
	modeAutomated
37
38
	// after a directive keyword like account, commodity, include
39
	// expects: rest of directive content
40
	modeDirective
41
)
42
43
type Lexer struct {
44
	file  string
45
	input []byte
46
	mode  mode
47
48
	ch     rune // current rune (0 = EOF/sentinel)
49
	chSize int  // byte size of current rune
50
	pos    int  // current byte offset (points at ch)
51
	rpos   int  // next byte offset to read (one ahead of pos)
52
	col    int  // current column (1-based)
53
	line   int  // current line (1-based)
54
55
	transactionPastStatus bool
56
	postingExpectAccount  bool
57
	readingNoteAfterPipe  bool
58
59
	includePath bool // true after [token.INCLUDE] is omitted
60
61
	// subdirective is set while inside an account/commodity directive block;
62
	// every indented line then lexes as directive content until a
63
	// non-indented line start clears it in lexDefault
64
	subdirective bool
65
}
66
67
func New(file string, input []byte) *Lexer {
68
	l := &Lexer{
69
		file:  file,
70
		input: input,
71
		line:  1,
72
	}
73
	l.advance()
74
	if l.ch == '\uFEFF' { // start of the input
75
		l.advance()
76
	}
77
	return l
78
}
79
80
// Next returns next token in the input
81
func (l *Lexer) Next() token.Token {
82
	switch l.mode {
83
	case modeDefault:
84
		return l.lexDefault()
85
	case modeComment:
86
		return l.lexComment()
87
	case modeTransaction:
88
		return l.lexTransaction()
89
	case modePosting:
90
		return l.lexPosting()
91
	case modePeriodic:
92
		return l.lexPeriodic()
93
	case modeAutomated:
94
		return l.lexAutomated()
95
	case modeDirective:
96
		return l.lexDirective()
97
	}
98
	panic("unreachable")
99
}
100
101
func (l *Lexer) lexDefault() token.Token {
102
	if l.includePath {
103
		l.includePath = false
104
	}
105
	if l.subdirective && l.ch != ' ' && l.ch != '\t' {
106
		l.subdirective = false
107
	}
108
	switch {
109
	case l.ch == 0:
110
		return l.token(token.EOF, "")
111
	case l.ch == '\n':
112
		return l.lexNewline()
113
	case l.ch == '\r':
114
		l.col = 0
115
		l.advance()
116
		return l.lexNewline()
117
	case l.ch == ' ' || l.ch == '\t':
118
		tok := l.lexIndent()
119
		if l.subdirective {
120
			// keep directive mode for every line of a directive block; the
121
			// flag is cleared at the next non-indented line start in lexDefault
122
			l.mode = modeDirective
123
		} else {
124
			l.mode = modePosting
125
		}
126
		l.postingExpectAccount = true
127
		return tok
128
	case l.ch == ';' || l.ch == '#' || l.ch == '%':
129
		l.mode = modeComment
130
		return l.lexSingle(token.SEMICOLON) // todo: ??
131
	case l.ch == '*': // * at col 0 == comment
132
		l.mode = modeComment
133
		return l.lexSingle(token.STAR)
134
	case l.ch == '~':
135
		l.mode = modePeriodic
136
		return l.lexSingle(token.TILDE)
137
	case l.ch == '=':
138
		l.mode = modeAutomated
139
		return l.lexSingle(token.EQ)
140
	case l.ch == '+':
141
		return l.lexSingle(token.PLUS)
142
	case l.ch == '-':
143
		return l.lexSingle(token.MINUS)
144
	case l.ch == '.':
145
		return l.lexSingle(token.TEXT)
146
	case l.ch == '!':
147
		return l.lexSingle(token.BANG)
148
	case l.ch == '@':
149
		return l.lexSingle(token.AT)
150
	case l.isAlpha():
151
		return l.lexKeyword()
152
	case l.isDigit():
153
		if !l.isDate() {
154
			s := l.save()
155
			for l.isDigit() || l.ch == '-' || l.ch == '/' || l.ch == '.' {
156
				l.advance()
157
			}
158
			return token.Token{Type: token.ILLEGAL, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
159
		}
160
		tok := l.lexDate()
161
		l.mode = modeTransaction
162
		l.transactionPastStatus = false
163
		return tok
164
	default:
165
		s := l.save()
166
		l.advance()
167
		return token.Token{Type: token.ILLEGAL, Literal: l.lit(s), Span: l.span(s)}
168
	}
169
}
170
171
func (l *Lexer) lexComment() token.Token {
172
	if l.ch == '\n' || l.ch == '\r' || l.ch == 0 {
173
		l.mode = modeDefault
174
		if l.ch == '\r' {
175
			l.col = 0
176
			l.advance()
177
		}
178
		return l.lexNewline()
179
	}
180
181
	for l.ch == ' ' || l.ch == '\t' {
182
		l.lexWhitespace()
183
	}
184
185
	if l.ch == '\n' || l.ch == '\r' || l.ch == 0 {
186
		l.mode = modeDefault
187
		if l.ch == '\r' {
188
			l.col = 0
189
			l.advance()
190
		}
191
		return l.lexNewline()
192
	}
193
194
	s := l.save()
195
	for l.ch != '\n' && l.ch != '\r' && l.ch != 0 {
196
		l.advance()
197
	}
198
	return token.Token{Type: token.TEXT, Literal: l.lit(s), Span: l.span(s)}
199
}
200
201
func (l *Lexer) lexTransaction() token.Token {
202
	if l.readingNoteAfterPipe {
203
		l.readingNoteAfterPipe = false
204
		return l.lexNote()
205
	}
206
207
	switch l.ch {
208
	case 0:
209
		return l.token(token.EOF, "")
210
	case '\n':
211
		l.mode = modeDefault
212
		return l.lexNewline()
213
	case '\r':
214
		l.col = 0
215
		l.advance()
216
		return l.lexNewline()
217
	case ' ', '\t':
218
		return l.lexWhitespace()
219
	case ';':
220
		l.mode = modeComment
221
		return l.lexSingle(token.SEMICOLON)
222
	case '*':
223
		if !l.transactionPastStatus {
224
			l.transactionPastStatus = true
225
			return l.lexSingle(token.STAR)
226
		}
227
		return l.lexText()
228
	case '!':
229
		if !l.transactionPastStatus {
230
			l.transactionPastStatus = true
231
			return l.lexSingle(token.BANG)
232
		}
233
		return l.lexText()
234
	case '|':
235
		l.transactionPastStatus = true
236
		l.readingNoteAfterPipe = true
237
		return l.lexSingle(token.PIPE)
238
	case '+':
239
		return l.lexSingle(token.PLUS)
240
	case '-':
241
		return l.lexSingle(token.MINUS)
242
	case '=':
243
		return l.lexEquals()
244
	case '"', '\'':
245
		return l.lexString()
246
	default: // description / payee
247
		if l.isDate() { // secondsry date after =
248
			return l.lexDate()
249
		}
250
		return l.lexText()
251
	}
252
}
253
254
func (l *Lexer) lexNote() token.Token {
255
	for l.ch == ' ' || l.ch == '\t' {
256
		l.advance()
257
	}
258
	if l.ch == 0 || l.ch == '\n' || l.ch == '\r' || l.ch == ';' {
259
		return l.Next()
260
	}
261
	s := l.save()
262
	for l.ch != 0 && l.ch != '\n' && l.ch != '\r' && l.ch != ';' {
263
		l.advance()
264
	}
265
	lit := l.lit(s)
266
	for len(lit) > 0 && (lit[len(lit)-1] == ' ' || lit[len(lit)-1] == '\t') {
267
		lit = lit[:len(lit)-1]
268
	}
269
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
270
}
271
272
func (l *Lexer) lexPeriodic() token.Token {
273
	switch l.ch {
274
	case 0:
275
		return l.token(token.EOF, "")
276
	case '\n':
277
		l.mode = modeDefault
278
		return l.lexNewline()
279
	case '\r':
280
		l.col = 0
281
		l.advance()
282
		return l.lexNewline()
283
	case ';':
284
		l.mode = modeComment
285
		return l.lexSingle(token.SEMICOLON)
286
	case ' ', '\t':
287
		return l.lexWhitespace()
288
	default:
289
		return l.lexText()
290
	}
291
}
292
293
func (l *Lexer) lexAutomated() token.Token {
294
	switch l.ch {
295
	case 0:
296
		return l.token(token.EOF, "")
297
	case '\n':
298
		l.mode = modeDefault
299
		return l.lexNewline()
300
	case '\r':
301
		l.col = 0
302
		l.advance()
303
		return l.lexNewline()
304
	case ' ', '\t':
305
		return l.lexWhitespace()
306
	case ';':
307
		l.mode = modeComment
308
		return l.lexSingle(token.SEMICOLON)
309
	default:
310
		return l.lexText()
311
	}
312
}
313
314
func (l *Lexer) lexPosting() token.Token {
315
	switch {
316
	case l.ch == 0:
317
		l.postingExpectAccount = false
318
		return l.token(token.EOF, "")
319
	case l.ch == '\n':
320
		l.postingExpectAccount = false
321
		l.mode = modeDefault
322
		return l.lexNewline()
323
	case l.ch == '\r':
324
		l.postingExpectAccount = false
325
		l.col = 0
326
		l.advance()
327
		l.mode = modeDefault
328
		return l.lexNewline()
329
	case l.ch == ';':
330
		l.postingExpectAccount = false
331
		l.mode = modeComment
332
		return l.lexSingle(token.SEMICOLON)
333
	case l.ch == ' ' || l.ch == '\t':
334
		return l.lexWhitespace()
335
	case l.postingExpectAccount && l.ch == '*':
336
		return l.lexSingle(token.STAR)
337
	case l.postingExpectAccount && l.ch == '!':
338
		return l.lexSingle(token.BANG)
339
	case l.ch == '=':
340
		return l.lexEquals()
341
	case l.ch == '@':
342
		return l.lexAt()
343
	case l.ch == '{':
344
		return l.lexLBrace()
345
	case l.ch == '}':
346
		return l.lexRBrace()
347
	case l.ch == '(':
348
		if !l.postingExpectAccount {
349
			return l.lexParenExpr()
350
		}
351
		return l.lexSingle(token.LPAREN)
352
	case l.ch == ')':
353
		return l.lexSingle(token.RPAREN)
354
	case l.ch == '[':
355
		return l.lexSingle(token.LBRACKET)
356
	case l.ch == ']':
357
		return l.lexSingle(token.RBRACKET)
358
	case l.ch == ':':
359
		return l.lexSingle(token.COLON)
360
	case l.postingExpectAccount && l.ch != '*' && l.ch != '!' && l.ch != '(' && l.ch != '[':
361
		return l.lexAccountNamePosting()
362
	case l.ch == '*': // after account name
363
		return l.lexSingle(token.STAR)
364
	case l.isDigit(), l.ch == '.':
365
		return l.lexNumber()
366
	case l.ch == '-':
367
		return l.lexSingle(token.MINUS)
368
	case l.ch == '+':
369
		return l.lexSingle(token.PLUS)
370
	case l.isCommodityStart():
371
		return l.lexCommodityMark()
372
	case l.ch == '"' || l.ch == '\'':
373
		return l.lexString()
374
	case l.ch >= 'a' && l.ch <= 'z':
375
		return l.lexCommodityMark()
376
	default:
377
		return l.lexAccountNamePosting()
378
	}
379
}
380
381
func (l *Lexer) lexDirective() token.Token {
382
	switch {
383
	case l.ch == '\n', l.ch == 0:
384
		l.mode = modeDefault
385
		return l.lexNewline()
386
	case l.ch == '\r':
387
		l.mode = modeDefault
388
		l.col = 0
389
		l.advance()
390
		return l.lexNewline()
391
	case l.ch == ';':
392
		l.mode = modeComment
393
		return l.lexSingle(token.SEMICOLON)
394
	case l.ch == ' ', l.ch == '\t':
395
		return l.lexWhitespace()
396
	case l.ch == '=':
397
		return l.lexSingle(token.EQ)
398
	case l.ch == '+':
399
		return l.lexSingle(token.PLUS)
400
	case l.ch == '-':
401
		return l.lexSingle(token.MINUS)
402
	case l.ch == '"', l.ch == '\'':
403
		return l.lexString()
404
	case l.ch == ':':
405
		return l.lexSingle(token.COLON)
406
	case l.ch == ')':
407
		return l.lexSingle(token.RPAREN)
408
	case l.ch == ']':
409
		return l.lexSingle(token.RBRACKET)
410
	case l.includePath:
411
		return l.lexPath()
412
	case l.isCommodityStart():
413
		return l.lexCommodityMark()
414
	case l.isTime():
415
		return l.lexTime()
416
	case l.isDate():
417
		return l.lexDate()
418
	case l.isDigit():
419
		return l.lexNumber()
420
	default:
421
		return l.lexAccountNameDirective()
422
	}
423
}
424
425
func (l *Lexer) lexPath() token.Token {
426
	s := l.save()
427
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ' ' && l.ch != '\t' {
428
		l.advance()
429
	}
430
	return token.Token{
431
		Type:    token.TEXT,
432
		Literal: l.lit(s),
433
		Span:    l.span(s),
434
	}
435
}
436
437
func (l *Lexer) lexSingle(kind token.Type) token.Token {
438
	s := l.save()
439
	l.advance()
440
	return token.Token{
441
		Type:    kind,
442
		Literal: l.lit(s),
443
		Span:    l.span(s),
444
	}
445
}
446
447
func (l *Lexer) lexNewline() token.Token {
448
	s := l.save()
449
	l.advance()
450
	l.mode = modeDefault
451
	return token.Token{Type: token.NEWLINE, Literal: "\n", Span: l.span(s)}
452
}
453
454
func (l *Lexer) lexWhitespace() token.Token {
455
	s := l.save()
456
	for l.ch == ' ' || l.ch == '\t' {
457
		l.advance()
458
	}
459
	return token.Token{Type: token.WHITESPACE, Literal: l.lit(s), Span: l.span(s)}
460
}
461
462
func (l *Lexer) lexIndent() token.Token {
463
	s := l.save()
464
	for l.ch == ' ' || l.ch == '\t' {
465
		l.advance()
466
	}
467
	return token.Token{Type: token.INDENT, Literal: l.lit(s), Span: l.span(s)}
468
}
469
470
func (l *Lexer) lexEquals() token.Token {
471
	s := l.save()
472
	l.advance()
473
	if l.ch == '=' {
474
		l.advance()
475
		switch l.ch {
476
		case '=':
477
			l.advance()
478
			return token.Token{Type: token.EQEQEQ, Literal: "===", Span: l.span(s)}
479
		case '*':
480
			l.advance()
481
			return token.Token{Type: token.EQEQEQ, Literal: "==*", Span: l.span(s)}
482
		default:
483
			return token.Token{Type: token.EQEQ, Literal: "==", Span: l.span(s)}
484
		}
485
	}
486
	if l.ch == '*' {
487
		l.advance()
488
		return token.Token{Type: token.EQSTAR, Literal: "=*", Span: l.span(s)}
489
	}
490
	return token.Token{Type: token.EQ, Literal: "=", Span: l.span(s)}
491
}
492
493
func (l *Lexer) lexAt() token.Token {
494
	s := l.save()
495
	l.advance()
496
	if l.ch == '@' {
497
		l.advance()
498
		return token.Token{Type: token.ATAT, Literal: "@@", Span: l.span(s)}
499
	}
500
	return token.Token{Type: token.AT, Literal: "@", Span: l.span(s)}
501
}
502
503
func (l *Lexer) lexText() token.Token {
504
	s := l.save()
505
	l.advance()
506
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ' ' && l.ch != '\t' {
507
		l.advance()
508
	}
509
	lit := string(l.input[s.offset:l.pos])
510
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
511
}
512
513
// lexAccountNameDirective reads accout name in directive context.
514
// stops at any whitespace, supports multi-word names("Taxi Fare").
515
func (l *Lexer) lexAccountNameDirective() token.Token {
516
	s := l.save()
517
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ')' && l.ch != ']' && l.ch != ':' && l.ch != ' ' && l.ch != '\t' {
518
		l.advance()
519
	}
520
	return token.Token{Type: token.TEXT, Literal: l.lit(s), Span: l.span(s)}
521
}
522
523
// lexAccountNamePosting reads an account name in posting context.
524
// stops at two consecutive spaces.
525
func (l *Lexer) lexAccountNamePosting() token.Token {
526
	s := l.save()
527
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ')' && l.ch != ']' && l.ch != ':' {
528
		if l.isTwoSpaces() {
529
			break
530
		}
531
		l.advance()
532
	}
533
	if l.ch != ':' {
534
		l.postingExpectAccount = false
535
	}
536
	return token.Token{Type: token.TEXT, Literal: l.lit(s), Span: l.span(s)}
537
}
538
539
func (l *Lexer) lexParenExpr() token.Token {
540
	s := l.save()
541
	depth := 0
542
	for l.ch != '\n' && l.ch != '\r' && l.ch != 0 {
543
		if l.ch == '(' {
544
			depth++
545
		} else if l.ch == ')' {
546
			depth--
547
			if depth == 0 {
548
				l.advance()
549
				break
550
			}
551
		}
552
		l.advance()
553
	}
554
	return token.Token{Type: token.PARENEXPR, Literal: l.lit(s), Span: l.span(s)}
555
}
556
557
func (l *Lexer) lexNumber() token.Token {
558
	s := l.save()
559
	for {
560
		if l.isDigit() || l.ch == '.' || l.ch == ',' || l.ch == '_' || l.ch == '\'' {
561
			l.advance()
562
		} else if l.ch == ' ' && (l.peek() >= '0' && l.peek() <= '9') {
563
			l.advance()
564
		} else if l.ch == 'e' || l.ch == 'E' {
565
			// exponent: consume only when E[sign]digit, so `10E ` stays an integer
566
			p := l.pos + 1
567
			if p < len(l.input) && (l.input[p] == '+' || l.input[p] == '-') {
568
				p++
569
			}
570
			if p >= len(l.input) || l.input[p] < '0' || l.input[p] > '9' {
571
				break
572
			}
573
			l.advance() // e/E
574
			if l.ch == '+' || l.ch == '-' {
575
				l.advance()
576
			}
577
			for l.isDigit() {
578
				l.advance()
579
			}
580
		} else {
581
			break
582
		}
583
	}
584
	lit := l.lit(s)
585
	kind := token.INT
586
	if strings.ContainsAny(lit, "., eE") {
587
		kind = token.DECIMAL
588
	}
589
	return token.Token{Type: kind, Literal: lit, Span: l.span(s)}
590
}
591
592
func (l *Lexer) lexKeyword() token.Token {
593
	s := l.save()
594
	for l.ch != 0 && l.ch != '\n' && l.ch != '\r' && l.ch != ' ' && l.ch != '\t' && l.ch != ';' {
595
		l.advance()
596
	}
597
	lit := l.lit(s)
598
	kind := l.keyword(lit)
599
	if kind == token.ILLEGAL { // todo: report an error ??
600
		kind = token.TEXT
601
	} else {
602
		l.mode = modeDirective
603
		l.subdirective = kind == token.ACCOUNT || kind == token.COMMODITY
604
		l.includePath = kind == token.INCLUDE
605
	}
606
	return token.Token{Type: kind, Literal: lit, Span: l.span(s)}
607
}
608
609
func (l *Lexer) lexDate() token.Token {
610
	s := l.save()
611
	for l.isDigit() || (l.isDateSep() && l.peekIsDigit()) {
612
		l.advance()
613
	}
614
	return token.Token{Type: token.DATE, Literal: l.lit(s), Span: l.span(s)}
615
}
616
617
func isSymbolChar(r rune) bool {
618
	return r == '$' || unicode.In(r, unicode.Sc)
619
}
620
621
func (l *Lexer) lexString() token.Token {
622
	s := l.save()
623
	quote := l.ch
624
	l.advance() // consume the quote character
625
	for l.ch != quote && l.ch != '\n' && l.ch != 0 {
626
		l.advance()
627
	}
628
	if l.ch == quote {
629
		l.advance()
630
	}
631
	return token.Token{Type: token.STRING, Literal: l.lit(s), Span: l.span(s)}
632
}
633
634
func (l *Lexer) lexCommodityMark() token.Token {
635
	s := l.save()
636
637
	if l.ch == '"' {
638
		l.advance()
639
		for l.ch != '"' && l.ch != '\n' && l.ch != 0 {
640
			l.advance()
641
		}
642
		if l.ch == '"' {
643
			l.advance()
644
		}
645
		return token.Token{Type: token.COMMODITYMARK, Literal: l.lit(s), Span: l.span(s)}
646
	}
647
648
	if unicode.IsLetter(l.ch) {
649
		for unicode.IsLetter(l.ch) {
650
			l.advance()
651
		}
652
		return token.Token{Type: token.COMMODITYMARK, Literal: l.lit(s), Span: l.span(s)}
653
	}
654
655
	if isSymbolChar(l.ch) {
656
		for isSymbolChar(l.ch) {
657
			l.advance()
658
		}
659
		return token.Token{Type: token.COMMODITYMARK, Literal: l.lit(s), Span: l.span(s)}
660
	}
661
662
	l.advance()
663
	return token.Token{Type: token.COMMODITYMARK, Literal: l.lit(s), Span: l.span(s)}
664
}
665
666
func (l *Lexer) lexLBrace() token.Token {
667
	s := l.save()
668
	l.advance()
669
	if l.ch == '{' {
670
		l.advance()
671
		return token.Token{Type: token.LBRACELBRACE, Literal: "{{", Span: l.span(s)}
672
	}
673
	return token.Token{Type: token.LBRACE, Literal: "{", Span: l.span(s)}
674
}
675
676
func (l *Lexer) lexRBrace() token.Token {
677
	s := l.save()
678
	l.advance()
679
	if l.ch == '}' {
680
		l.advance()
681
		return token.Token{Type: token.RBRACERBRACE, Literal: "}}", Span: l.span(s)}
682
	}
683
	return token.Token{Type: token.RBRACE, Literal: "}", Span: l.span(s)}
684
}
685
686
func (l *Lexer) advance() {
687
	if l.rpos >= len(l.input) {
688
		l.ch = 0
689
		l.chSize = 0
690
	} else if b := l.input[l.rpos]; b < utf8.RuneSelf { // ASCII fast path
691
		l.ch = rune(b)
692
		l.chSize = 1
693
	} else {
694
		l.ch, l.chSize = utf8.DecodeRune(l.input[l.rpos:])
695
	}
696
	l.pos = l.rpos
697
	l.rpos += l.chSize
698
	if l.ch == '\n' || l.ch == '\r' {
699
		l.line++
700
		l.col = 0
701
	} else {
702
		l.col++
703
	}
704
}
705
706
func (l *Lexer) peek() rune {
707
	r, _ := utf8.DecodeRune(l.input[l.rpos:])
708
	return r
709
}
710
711
func (l *Lexer) peekN(n int) byte {
712
	if l.pos+n >= len(l.input) {
713
		return 0
714
	}
715
	return l.input[l.pos+n]
716
}
717
718
func (l *Lexer) isDigit() bool { return l.ch >= '0' && l.ch <= '9' }
719
func (l *Lexer) isAlpha() bool {
720
	return (l.ch >= 'a' && l.ch <= 'z') ||
721
		(l.ch >= 'A' && l.ch <= 'Z')
722
}
723
724
func (l *Lexer) isTwoSpaces() bool { return l.ch == ' ' && l.peek() == ' ' }
725
726
func (l *Lexer) isDateSep() bool { return l.ch == '-' || l.ch == '/' || l.ch == '.' }
727
728
func (l *Lexer) peekIsDigit() bool {
729
	r := l.peek()
730
	return r >= '0' && r <= '9'
731
}
732
733
func (l *Lexer) isCommodityStart() bool {
734
	if l.ch == '$' || (l.ch >= 'A' && l.ch <= 'Z') {
735
		return true
736
	}
737
	if l.ch < utf8.RuneSelf {
738
		return false
739
	}
740
	return unicode.In(l.ch, unicode.Sc) || unicode.IsLetter(l.ch)
741
}
742
743
func (l *Lexer) isDate() bool {
744
	if !l.isDigit() {
745
		return false
746
	}
747
	// YYYY/M/D or YYYY/MM/DD
748
	if l.peekN(1) >= '0' && l.peekN(1) <= '9' &&
749
		l.peekN(2) >= '0' && l.peekN(2) <= '9' &&
750
		l.peekN(3) >= '0' && l.peekN(3) <= '9' {
751
		sep := l.peekN(4)
752
		if sep == '/' || sep == '-' || sep == '.' {
753
			if l.peekN(5) >= '0' && l.peekN(5) <= '9' {
754
				if l.peekN(6) == sep {
755
					return l.peekN(7) >= '0' && l.peekN(7) <= '9'
756
				}
757
				if l.peekN(7) == sep {
758
					return l.peekN(8) >= '0' && l.peekN(8) <= '9'
759
				}
760
			}
761
		}
762
		return false
763
	}
764
	// M/D or MM/DD(year inferred, only / and - separators; . is ambiguous with decimal numbers like 1.01)
765
	if (l.peekN(1) == '/' || l.peekN(1) == '-') &&
766
		l.peekN(2) >= '0' && l.peekN(2) <= '9' &&
767
		l.ch >= '1' && l.ch <= '9' {
768
		return validDay(l.peekN(2), l.peekN(3))
769
	}
770
	if (l.peekN(2) == '/' || l.peekN(2) == '-') &&
771
		l.peekN(3) >= '0' && l.peekN(3) <= '9' {
772
		m := int(l.ch-'0')*10 + int(l.peekN(1)-'0')
773
		return m >= 1 && m <= 12 && validDay(l.peekN(3), l.peekN(4))
774
	}
775
	return false
776
}
777
778
func validDay(first, second byte) bool {
779
	d := int(first - '0')
780
	if second >= '0' && second <= '9' {
781
		d = d*10 + int(second-'0')
782
	}
783
	return d >= 1 && d <= 31
784
}
785
786
func (l *Lexer) isTime() bool {
787
	if !l.isDigit() {
788
		return false
789
	}
790
	return l.peekN(2) == ':'
791
}
792
793
func (l *Lexer) lexTime() token.Token {
794
	s := l.save()
795
	for l.isDigit() || l.ch == ':' {
796
		l.advance()
797
	}
798
	return token.Token{Type: token.TIME, Literal: l.lit(s), Span: l.span(s)}
799
}
800
801
type savedPos struct{ offset, line, col int }
802
803
func (l *Lexer) save() savedPos {
804
	return savedPos{l.pos, l.line, l.col}
805
}
806
807
func (l *Lexer) span(s savedPos) token.Span {
808
	return token.Span{
809
		File:  l.file,
810
		Start: token.Pos{Offset: s.offset, Line: s.line, Col: s.col},
811
		End:   token.Pos{Offset: l.pos, Line: l.line, Col: l.col},
812
	}
813
}
814
815
func (l *Lexer) token(kind token.Type, literal string) token.Token {
816
	s := savedPos{l.pos, l.line, l.col}
817
	return token.Token{Type: kind, Literal: literal, Span: l.span(s)}
818
}
819
820
func (l *Lexer) lit(s savedPos) string {
821
	if s.offset == l.pos {
822
		return ""
823
	}
824
	return unsafe.String(&l.input[s.offset], l.pos-s.offset)
825
}
826
827
func (l *Lexer) keyword(s string) token.Type {
828
	switch s {
829
	case "comment":
830
		return token.COMMENTKW
831
	case "account":
832
		return token.ACCOUNT
833
	case "commodity":
834
		return token.COMMODITY
835
	case "include":
836
		return token.INCLUDE
837
	case "alias":
838
		return token.ALIAS
839
	case "payee":
840
		return token.PAYEE
841
	case "tag":
842
		return token.TAG
843
	case "apply":
844
		return token.APPLY
845
	case "end":
846
		return token.END
847
	case "Y", "year":
848
		return token.YEAR
849
	case "decimal-mark":
850
		return token.DECIMALMARK
851
	case "D":
852
		return token.D
853
	case "P":
854
		return token.P
855
	case "N":
856
		return token.N
857
	case "C":
858
		return token.C
859
	default:
860
		return token.ILLEGAL
861
	}
862
}