all repos

clerk @ 1efa5a1

missing tooling for ledger/hledger

clerk/journal/lexer/lexer.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
lexer: minimize public surface of the package, 2 months ago
1
package lexer
2
3
import (
4
	"strings"
5
	"unicode"
6
	"unicode/utf8"
7
8
	"olexsmir.xyz/clerk/journal/token"
9
)
10
11
type mode uint
12
13
const (
14
	// start of a line, nothing consumed
15
	modeDefault mode = iota
16
17
	// after ; # * % ; at start of line, or anywhere inline
18
	// everything until \n is comment text
19
	modeComment
20
21
	// after lexing a date at column 0
22
	// expects: optional status, optional code, description, comment
23
	modeTransaction
24
25
	// after lexing an indent at start of line
26
	// expects: account name, then two spaces, then amount
27
	modePosting
28
29
	// after ~, period expression
30
	// expects: period, optional description (after 2+ spaces), optional comment
31
	modePeriodic
32
33
	// after =, automates transaction
34
	// expects: expression
35
	modeAutomated
36
37
	// after a directive keyword like account, commodity, include
38
	// expects: rest of directive content
39
	modeDirective
40
)
41
42
type Lexer struct {
43
	file  string
44
	input []byte
45
	mode  mode
46
47
	ch     rune // current rune (0 = EOF/sentinel)
48
	chSize int  // byte size of current rune
49
	pos    int  // current byte offset (points at ch)
50
	rpos   int  // next byte offset to read (one ahead of pos)
51
	col    int  // current column (1-based)
52
	line   int  // current line (1-based)
53
54
	transactionPastStatus bool
55
	postingExpectAccount  bool
56
	readingNoteAfterPipe  bool
57
58
	includePath bool // true after [token.INCLUDE] is omitted
59
60
	// subdirective is set when the current line is an account/commodity
61
	// directive; the next indented line then lexes as directive content
62
	subdirective bool
63
}
64
65
func New(file string, input []byte) *Lexer {
66
	l := &Lexer{
67
		file:  file,
68
		input: input,
69
		line:  1,
70
	}
71
	l.advance()
72
	if l.ch == '\uFEFF' { // start of the input
73
		l.advance()
74
	}
75
	return l
76
}
77
78
// Next returns next token in the input
79
func (l *Lexer) Next() token.Token {
80
	switch l.mode {
81
	case modeDefault:
82
		return l.lexDefault()
83
	case modeComment:
84
		return l.lexComment()
85
	case modeTransaction:
86
		return l.lexTransaction()
87
	case modePosting:
88
		return l.lexPosting()
89
	case modePeriodic:
90
		return l.lexPeriodic()
91
	case modeAutomated:
92
		return l.lexAutomated()
93
	case modeDirective:
94
		return l.lexDirective()
95
	}
96
	panic("unreachable")
97
}
98
99
func (l *Lexer) lexDefault() token.Token {
100
	if l.includePath {
101
		l.includePath = false
102
	}
103
	if l.subdirective && l.ch != ' ' && l.ch != '\t' {
104
		l.subdirective = false
105
	}
106
	switch {
107
	case l.ch == 0:
108
		return l.token(token.EOF, "")
109
	case l.ch == '\n':
110
		return l.lexNewline()
111
	case l.ch == '\r':
112
		l.col = 0
113
		l.advance()
114
		return l.lexNewline()
115
	case l.ch == ' ' || l.ch == '\t':
116
		tok := l.lexIndent()
117
		if l.subdirective {
118
			l.mode = modeDirective
119
			l.subdirective = false
120
		} else {
121
			l.mode = modePosting
122
		}
123
		l.postingExpectAccount = true
124
		return tok
125
	case l.ch == ';' || l.ch == '#' || l.ch == '%':
126
		l.mode = modeComment
127
		return l.lexSingle(token.SEMICOLON) // todo: ??
128
	case l.ch == '*': // * at col 0 == comment
129
		l.mode = modeComment
130
		return l.lexSingle(token.STAR)
131
	case l.ch == '~':
132
		l.mode = modePeriodic
133
		return l.lexSingle(token.TILDE)
134
	case l.ch == '=':
135
		l.mode = modeAutomated
136
		return l.lexSingle(token.EQ)
137
	case l.ch == '+':
138
		return l.lexSingle(token.PLUS)
139
	case l.ch == '-':
140
		return l.lexSingle(token.MINUS)
141
	case l.ch == '.':
142
		return l.lexSingle(token.TEXT)
143
	case l.ch == '!':
144
		return l.lexSingle(token.BANG)
145
	case l.ch == '@':
146
		return l.lexSingle(token.AT)
147
	case l.isAlpha():
148
		return l.lexKeyword()
149
	case l.isDigit():
150
		if !l.isDate() {
151
			s := l.save()
152
			for l.isDigit() || l.ch == '-' || l.ch == '/' || l.ch == '.' {
153
				l.advance()
154
			}
155
			return token.Token{Type: token.ILLEGAL, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
156
		}
157
		tok := l.lexDate()
158
		l.mode = modeTransaction
159
		l.transactionPastStatus = false
160
		return tok
161
	default:
162
		s := l.save()
163
		l.advance()
164
		return token.Token{Type: token.ILLEGAL, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
165
	}
166
}
167
168
func (l *Lexer) lexComment() token.Token {
169
	if l.ch == '\n' || l.ch == '\r' || l.ch == 0 {
170
		l.mode = modeDefault
171
		if l.ch == '\r' {
172
			l.col = 0
173
			l.advance()
174
		}
175
		return l.lexNewline()
176
	}
177
178
	for l.ch == ' ' || l.ch == '\t' {
179
		l.lexWhitespace()
180
	}
181
182
	if l.ch == '\n' || l.ch == '\r' || l.ch == 0 {
183
		l.mode = modeDefault
184
		if l.ch == '\r' {
185
			l.col = 0
186
			l.advance()
187
		}
188
		return l.lexNewline()
189
	}
190
191
	s := l.save()
192
	for l.ch != '\n' && l.ch != '\r' && l.ch != 0 {
193
		l.advance()
194
	}
195
	return token.Token{Type: token.TEXT, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
196
}
197
198
func (l *Lexer) lexTransaction() token.Token {
199
	if l.readingNoteAfterPipe {
200
		l.readingNoteAfterPipe = false
201
		return l.lexNote()
202
	}
203
204
	switch l.ch {
205
	case 0:
206
		return l.token(token.EOF, "")
207
	case '\n':
208
		l.mode = modeDefault
209
		return l.lexNewline()
210
	case '\r':
211
		l.col = 0
212
		l.advance()
213
		return l.lexNewline()
214
	case ' ', '\t':
215
		return l.lexWhitespace()
216
	case ';':
217
		l.mode = modeComment
218
		return l.lexSingle(token.SEMICOLON)
219
	case '*':
220
		if !l.transactionPastStatus {
221
			l.transactionPastStatus = true
222
			return l.lexSingle(token.STAR)
223
		}
224
		return l.lexText()
225
	case '!':
226
		if !l.transactionPastStatus {
227
			l.transactionPastStatus = true
228
			return l.lexSingle(token.BANG)
229
		}
230
		return l.lexText()
231
	case '|':
232
		l.transactionPastStatus = true
233
		l.readingNoteAfterPipe = true
234
		return l.lexSingle(token.PIPE)
235
	case '+':
236
		return l.lexSingle(token.PLUS)
237
	case '-':
238
		return l.lexSingle(token.MINUS)
239
	case '=':
240
		return l.lexEquals()
241
	case '"', '\'':
242
		return l.lexString()
243
	default: // description / payee
244
		if l.isDate() { // secondsry date after =
245
			return l.lexDate()
246
		}
247
		return l.lexText()
248
	}
249
}
250
251
func (l *Lexer) lexNote() token.Token {
252
	for l.ch == ' ' || l.ch == '\t' {
253
		l.advance()
254
	}
255
	if l.ch == 0 || l.ch == '\n' || l.ch == '\r' || l.ch == ';' {
256
		return l.Next()
257
	}
258
	s := l.save()
259
	for l.ch != 0 && l.ch != '\n' && l.ch != '\r' && l.ch != ';' {
260
		l.advance()
261
	}
262
	lit := string(l.input[s.offset:l.pos])
263
	for len(lit) > 0 && (lit[len(lit)-1] == ' ' || lit[len(lit)-1] == '\t') {
264
		lit = lit[:len(lit)-1]
265
	}
266
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
267
}
268
269
func (l *Lexer) lexPeriodic() token.Token {
270
	switch l.ch {
271
	case 0:
272
		return l.token(token.EOF, "")
273
	case '\n':
274
		l.mode = modeDefault
275
		return l.lexNewline()
276
	case '\r':
277
		l.col = 0
278
		l.advance()
279
		return l.lexNewline()
280
	case ';':
281
		l.mode = modeComment
282
		return l.lexSingle(token.SEMICOLON)
283
	case ' ', '\t':
284
		return l.lexWhitespace()
285
	default:
286
		return l.lexText()
287
	}
288
}
289
290
func (l *Lexer) lexAutomated() token.Token {
291
	switch l.ch {
292
	case 0:
293
		return l.token(token.EOF, "")
294
	case '\n':
295
		l.mode = modeDefault
296
		return l.lexNewline()
297
	case '\r':
298
		l.col = 0
299
		l.advance()
300
		return l.lexNewline()
301
	case ' ', '\t':
302
		return l.lexWhitespace()
303
	case ';':
304
		l.mode = modeComment
305
		return l.lexSingle(token.SEMICOLON)
306
	default:
307
		return l.lexText()
308
	}
309
}
310
311
func (l *Lexer) lexPosting() token.Token {
312
	switch {
313
	case l.ch == 0:
314
		l.postingExpectAccount = false
315
		return l.token(token.EOF, "")
316
	case l.ch == '\n':
317
		l.postingExpectAccount = false
318
		l.mode = modeDefault
319
		return l.lexNewline()
320
	case l.ch == '\r':
321
		l.postingExpectAccount = false
322
		l.col = 0
323
		l.advance()
324
		l.mode = modeDefault
325
		return l.lexNewline()
326
	case l.ch == ';':
327
		l.postingExpectAccount = false
328
		l.mode = modeComment
329
		return l.lexSingle(token.SEMICOLON)
330
	case l.ch == ' ' || l.ch == '\t':
331
		return l.lexWhitespace()
332
	case l.postingExpectAccount && l.ch == '*':
333
		return l.lexSingle(token.STAR)
334
	case l.postingExpectAccount && l.ch == '!':
335
		return l.lexSingle(token.BANG)
336
	case l.ch == '=':
337
		return l.lexEquals()
338
	case l.ch == '@':
339
		return l.lexAt()
340
	case l.ch == '{':
341
		return l.lexLBrace()
342
	case l.ch == '}':
343
		return l.lexRBrace()
344
	case l.ch == '(':
345
		if !l.postingExpectAccount {
346
			return l.lexParenExpr()
347
		}
348
		return l.lexSingle(token.LPAREN)
349
	case l.ch == ')':
350
		return l.lexSingle(token.RPAREN)
351
	case l.ch == '[':
352
		return l.lexSingle(token.LBRACKET)
353
	case l.ch == ']':
354
		return l.lexSingle(token.RBRACKET)
355
	case l.ch == ':':
356
		return l.lexSingle(token.COLON)
357
	case l.postingExpectAccount && l.ch != '*' && l.ch != '!' && l.ch != '(' && l.ch != '[':
358
		return l.lexAccountNamePosting()
359
	case l.ch == '*': // after account name
360
		return l.lexSingle(token.STAR)
361
	case l.isDigit(), l.ch == '.':
362
		return l.lexNumber()
363
	case l.ch == '-':
364
		return l.lexSingle(token.MINUS)
365
	case l.ch == '+':
366
		return l.lexSingle(token.PLUS)
367
	case l.isCommodityStart():
368
		return l.lexCommodityMark()
369
	case l.ch == '"' || l.ch == '\'':
370
		return l.lexString()
371
	case l.ch >= 'a' && l.ch <= 'z':
372
		return l.lexCommodityMark()
373
	default:
374
		return l.lexAccountNamePosting()
375
	}
376
}
377
378
func (l *Lexer) lexDirective() token.Token {
379
	switch {
380
	case l.ch == '\n', l.ch == 0:
381
		l.mode = modeDefault
382
		return l.lexNewline()
383
	case l.ch == '\r':
384
		l.mode = modeDefault
385
		l.col = 0
386
		l.advance()
387
		return l.lexNewline()
388
	case l.ch == ';':
389
		l.mode = modeComment
390
		return l.lexSingle(token.SEMICOLON)
391
	case l.ch == ' ', l.ch == '\t':
392
		return l.lexWhitespace()
393
	case l.ch == '=':
394
		return l.lexSingle(token.EQ)
395
	case l.ch == '+':
396
		return l.lexSingle(token.PLUS)
397
	case l.ch == '-':
398
		return l.lexSingle(token.MINUS)
399
	case l.ch == '"', l.ch == '\'':
400
		return l.lexString()
401
	case l.ch == ':':
402
		return l.lexSingle(token.COLON)
403
	case l.ch == ')':
404
		return l.lexSingle(token.RPAREN)
405
	case l.ch == ']':
406
		return l.lexSingle(token.RBRACKET)
407
	case l.includePath:
408
		return l.lexPath()
409
	case l.isCommodityStart():
410
		return l.lexCommodityMark()
411
	case l.isTime():
412
		return l.lexTime()
413
	case l.isDate():
414
		return l.lexDate()
415
	case l.isDigit():
416
		return l.lexNumber()
417
	default:
418
		return l.lexAccountNameDirective()
419
	}
420
}
421
422
func (l *Lexer) lexPath() token.Token {
423
	s := l.save()
424
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ' ' && l.ch != '\t' {
425
		l.advance()
426
	}
427
	return token.Token{
428
		Type:    token.TEXT,
429
		Literal: string(l.input[s.offset:l.pos]),
430
		Span:    l.span(s),
431
	}
432
}
433
434
func (l *Lexer) lexSingle(kind token.Type) token.Token {
435
	s := l.save()
436
	l.advance()
437
	return token.Token{
438
		Type:    kind,
439
		Literal: string(l.input[s.offset:l.pos]),
440
		Span:    l.span(s),
441
	}
442
}
443
444
func (l *Lexer) lexNewline() token.Token {
445
	s := l.save()
446
	l.advance()
447
	l.mode = modeDefault
448
	return token.Token{Type: token.NEWLINE, Literal: "\n", Span: l.span(s)}
449
}
450
451
func (l *Lexer) lexWhitespace() token.Token {
452
	s := l.save()
453
	for l.ch == ' ' || l.ch == '\t' {
454
		l.advance()
455
	}
456
	lit := string(l.input[s.offset:l.pos])
457
	return token.Token{Type: token.WHITESPACE, Literal: lit, Span: l.span(s)}
458
}
459
460
func (l *Lexer) lexIndent() token.Token {
461
	s := l.save()
462
	for l.ch == ' ' || l.ch == '\t' {
463
		l.advance()
464
	}
465
	lit := string(l.input[s.offset:l.pos])
466
	return token.Token{Type: token.INDENT, Literal: lit, Span: l.span(s)}
467
}
468
469
func (l *Lexer) lexEquals() token.Token {
470
	s := l.save()
471
	l.advance()
472
	if l.ch == '=' {
473
		l.advance()
474
		switch l.ch {
475
		case '=':
476
			l.advance()
477
			return token.Token{Type: token.EQEQEQ, Literal: "===", Span: l.span(s)}
478
		case '*':
479
			l.advance()
480
			return token.Token{Type: token.EQEQEQ, Literal: "==*", Span: l.span(s)}
481
		default:
482
			return token.Token{Type: token.EQEQ, Literal: "==", Span: l.span(s)}
483
		}
484
	}
485
	if l.ch == '*' {
486
		l.advance()
487
		return token.Token{Type: token.EQSTAR, Literal: "=*", Span: l.span(s)}
488
	}
489
	return token.Token{Type: token.EQ, Literal: "=", Span: l.span(s)}
490
}
491
492
func (l *Lexer) lexAt() token.Token {
493
	s := l.save()
494
	l.advance()
495
	if l.ch == '@' {
496
		l.advance()
497
		return token.Token{Type: token.ATAT, Literal: "@@", Span: l.span(s)}
498
	}
499
	return token.Token{Type: token.AT, Literal: "@", Span: l.span(s)}
500
}
501
502
func (l *Lexer) lexText() token.Token {
503
	s := l.save()
504
	l.advance()
505
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ' ' && l.ch != '\t' {
506
		l.advance()
507
	}
508
	lit := string(l.input[s.offset:l.pos])
509
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
510
}
511
512
// lexAccountNameDirective reads accout name in directive context.
513
// stops at any whitespace, supports multi-word names("Taxi Fare").
514
func (l *Lexer) lexAccountNameDirective() token.Token {
515
	s := l.save()
516
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ')' && l.ch != ']' && l.ch != ':' && l.ch != ' ' && l.ch != '\t' {
517
		l.advance()
518
	}
519
	lit := string(l.input[s.offset:l.pos])
520
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
521
}
522
523
// lexAccountNamePosting reads an account name in posting context.
524
// stops at two consecutive spaces.
525
func (l *Lexer) lexAccountNamePosting() token.Token {
526
	s := l.save()
527
	for l.ch != '\n' && l.ch != '\r' && l.ch != ';' && l.ch != 0 && l.ch != ')' && l.ch != ']' && l.ch != ':' {
528
		if l.isTwoSpaces() {
529
			break
530
		}
531
		l.advance()
532
	}
533
	if l.ch != ':' {
534
		l.postingExpectAccount = false
535
	}
536
	lit := string(l.input[s.offset:l.pos])
537
	return token.Token{Type: token.TEXT, Literal: lit, Span: l.span(s)}
538
}
539
540
func (l *Lexer) lexParenExpr() token.Token {
541
	s := l.save()
542
	depth := 0
543
	for l.ch != '\n' && l.ch != '\r' && l.ch != 0 {
544
		if l.ch == '(' {
545
			depth++
546
		} else if l.ch == ')' {
547
			depth--
548
			if depth == 0 {
549
				l.advance()
550
				break
551
			}
552
		}
553
		l.advance()
554
	}
555
	lit := string(l.input[s.offset:l.pos])
556
	return token.Token{Type: token.PARENEXPR, Literal: lit, Span: l.span(s)}
557
}
558
559
func (l *Lexer) lexNumber() token.Token {
560
	s := l.save()
561
	for {
562
		if l.isDigit() || l.ch == '.' || l.ch == ',' || l.ch == '_' || l.ch == '\'' {
563
			l.advance()
564
		} else if l.ch == ' ' && (l.peek() >= '0' && l.peek() <= '9') {
565
			l.advance()
566
		} else if l.ch == 'e' || l.ch == 'E' {
567
			// exponent: consume only when E[sign]digit, so `10E ` stays an integer
568
			p := l.pos + 1
569
			if p < len(l.input) && (l.input[p] == '+' || l.input[p] == '-') {
570
				p++
571
			}
572
			if p >= len(l.input) || l.input[p] < '0' || l.input[p] > '9' {
573
				break
574
			}
575
			l.advance() // e/E
576
			if l.ch == '+' || l.ch == '-' {
577
				l.advance()
578
			}
579
			for l.isDigit() {
580
				l.advance()
581
			}
582
		} else {
583
			break
584
		}
585
	}
586
	lit := string(l.input[s.offset:l.pos])
587
	kind := token.INT
588
	if strings.ContainsAny(lit, "., eE") {
589
		kind = token.DECIMAL
590
	}
591
	return token.Token{Type: kind, Literal: lit, Span: l.span(s)}
592
}
593
594
func (l *Lexer) lexKeyword() token.Token {
595
	s := l.save()
596
	for l.ch != 0 && l.ch != '\n' && l.ch != '\r' && l.ch != ' ' && l.ch != '\t' && l.ch != ';' {
597
		l.advance()
598
	}
599
	lit := string(l.input[s.offset:l.pos])
600
	kind := l.keyword(lit)
601
	if kind == token.ILLEGAL { // todo: report an error ??
602
		kind = token.TEXT
603
	} else {
604
		l.mode = modeDirective
605
		l.subdirective = kind == token.ACCOUNT || kind == token.COMMODITY
606
		l.includePath = kind == token.INCLUDE
607
	}
608
	return token.Token{Type: kind, Literal: lit, Span: l.span(s)}
609
}
610
611
func (l *Lexer) lexDate() token.Token {
612
	s := l.save()
613
	for l.isDigit() || (l.isDateSep() && l.peekIsDigit()) {
614
		l.advance()
615
	}
616
	return token.Token{Type: token.DATE, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
617
}
618
619
func isSymbolChar(r rune) bool {
620
	return r == '$' || unicode.In(r, unicode.Sc)
621
}
622
623
func (l *Lexer) lexString() token.Token {
624
	s := l.save()
625
	quote := l.ch
626
	l.advance() // consume the quote character
627
	for l.ch != quote && l.ch != '\n' && l.ch != 0 {
628
		l.advance()
629
	}
630
	if l.ch == quote {
631
		l.advance()
632
	}
633
	return token.Token{Type: token.STRING, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
634
}
635
636
func (l *Lexer) lexCommodityMark() token.Token {
637
	s := l.save()
638
639
	if l.ch == '"' {
640
		l.advance()
641
		for l.ch != '"' && l.ch != '\n' && l.ch != 0 {
642
			l.advance()
643
		}
644
		if l.ch == '"' {
645
			l.advance()
646
		}
647
		return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
648
	}
649
650
	if unicode.IsLetter(l.ch) {
651
		for unicode.IsLetter(l.ch) {
652
			l.advance()
653
		}
654
		return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
655
	}
656
657
	if isSymbolChar(l.ch) {
658
		for isSymbolChar(l.ch) {
659
			l.advance()
660
		}
661
		return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
662
	}
663
664
	l.advance()
665
	return token.Token{Type: token.COMMODITYMARK, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
666
}
667
668
func (l *Lexer) lexLBrace() token.Token {
669
	s := l.save()
670
	l.advance()
671
	if l.ch == '{' {
672
		l.advance()
673
		return token.Token{Type: token.LBRACELBRACE, Literal: "{{", Span: l.span(s)}
674
	}
675
	return token.Token{Type: token.LBRACE, Literal: "{", Span: l.span(s)}
676
}
677
678
func (l *Lexer) lexRBrace() token.Token {
679
	s := l.save()
680
	l.advance()
681
	if l.ch == '}' {
682
		l.advance()
683
		return token.Token{Type: token.RBRACERBRACE, Literal: "}}", Span: l.span(s)}
684
	}
685
	return token.Token{Type: token.RBRACE, Literal: "}", Span: l.span(s)}
686
}
687
688
func (l *Lexer) advance() {
689
	if l.rpos >= len(l.input) {
690
		l.ch = 0
691
		l.chSize = 0
692
	} else {
693
		r, size := utf8.DecodeRune(l.input[l.rpos:])
694
		l.ch = r
695
		l.chSize = size
696
	}
697
	l.pos = l.rpos
698
	l.rpos += l.chSize
699
	if l.ch == '\n' || l.ch == '\r' {
700
		l.line++
701
		l.col = 0
702
	} else {
703
		l.col++
704
	}
705
}
706
707
func (l *Lexer) peek() rune {
708
	r, _ := utf8.DecodeRune(l.input[l.rpos:])
709
	return r
710
}
711
712
func (l *Lexer) peekN(n int) byte {
713
	if l.pos+n >= len(l.input) {
714
		return 0
715
	}
716
	return l.input[l.pos+n]
717
}
718
719
func (l *Lexer) isDigit() bool { return l.ch >= '0' && l.ch <= '9' }
720
func (l *Lexer) isAlpha() bool {
721
	return (l.ch >= 'a' && l.ch <= 'z') ||
722
		(l.ch >= 'A' && l.ch <= 'Z')
723
}
724
725
func (l *Lexer) isTwoSpaces() bool { return l.ch == ' ' && l.peek() == ' ' }
726
727
func (l *Lexer) isDateSep() bool { return l.ch == '-' || l.ch == '/' || l.ch == '.' }
728
729
func (l *Lexer) peekIsDigit() bool {
730
	r := l.peek()
731
	return r >= '0' && r <= '9'
732
}
733
734
func (l *Lexer) isCommodityStart() bool {
735
	if l.ch == '$' || (l.ch >= 'A' && l.ch <= 'Z') {
736
		return true
737
	}
738
	if l.ch < utf8.RuneSelf {
739
		return false
740
	}
741
	return unicode.In(l.ch, unicode.Sc) || unicode.IsLetter(l.ch)
742
}
743
744
func (l *Lexer) isDate() bool {
745
	if !l.isDigit() {
746
		return false
747
	}
748
	// YYYY/M/D or YYYY/MM/DD
749
	if l.peekN(1) >= '0' && l.peekN(1) <= '9' &&
750
		l.peekN(2) >= '0' && l.peekN(2) <= '9' &&
751
		l.peekN(3) >= '0' && l.peekN(3) <= '9' {
752
		sep := l.peekN(4)
753
		if sep == '/' || sep == '-' || sep == '.' {
754
			if l.peekN(5) >= '0' && l.peekN(5) <= '9' {
755
				if l.peekN(6) == sep {
756
					return l.peekN(7) >= '0' && l.peekN(7) <= '9'
757
				}
758
				if l.peekN(7) == sep {
759
					return l.peekN(8) >= '0' && l.peekN(8) <= '9'
760
				}
761
			}
762
		}
763
		return false
764
	}
765
	// M/D or MM/DD(year inferred, only / and - separators; . is ambiguous with decimal numbers like 1.01)
766
	if (l.peekN(1) == '/' || l.peekN(1) == '-') &&
767
		l.peekN(2) >= '0' && l.peekN(2) <= '9' &&
768
		l.ch >= '1' && l.ch <= '9' {
769
		return validDay(l.peekN(2), l.peekN(3))
770
	}
771
	if (l.peekN(2) == '/' || l.peekN(2) == '-') &&
772
		l.peekN(3) >= '0' && l.peekN(3) <= '9' {
773
		m := int(l.ch-'0')*10 + int(l.peekN(1)-'0')
774
		return m >= 1 && m <= 12 && validDay(l.peekN(3), l.peekN(4))
775
	}
776
	return false
777
}
778
779
func validDay(first, second byte) bool {
780
	d := int(first - '0')
781
	if second >= '0' && second <= '9' {
782
		d = d*10 + int(second-'0')
783
	}
784
	return d >= 1 && d <= 31
785
}
786
787
func (l *Lexer) isTime() bool {
788
	if !l.isDigit() {
789
		return false
790
	}
791
	return l.peekN(2) == ':'
792
}
793
794
func (l *Lexer) lexTime() token.Token {
795
	s := l.save()
796
	for l.isDigit() || l.ch == ':' {
797
		l.advance()
798
	}
799
	return token.Token{Type: token.TIME, Literal: string(l.input[s.offset:l.pos]), Span: l.span(s)}
800
}
801
802
type savedPos struct{ offset, line, col int }
803
804
func (l *Lexer) save() savedPos {
805
	return savedPos{l.pos, l.line, l.col}
806
}
807
808
func (l *Lexer) span(s savedPos) token.Span {
809
	return token.Span{
810
		Start: token.Pos{File: l.file, Offset: s.offset, Line: s.line, Col: s.col},
811
		End:   token.Pos{File: l.file, Offset: l.pos, Line: l.line, Col: l.col},
812
	}
813
}
814
815
func (l *Lexer) token(kind token.Type, literal string) token.Token {
816
	s := savedPos{l.pos, l.line, l.col}
817
	return token.Token{Type: kind, Literal: literal, Span: l.span(s)}
818
}
819
820
func (l *Lexer) keyword(s string) token.Type {
821
	switch s {
822
	case "comment":
823
		return token.COMMENTKW
824
	case "account":
825
		return token.ACCOUNT
826
	case "commodity":
827
		return token.COMMODITY
828
	case "include":
829
		return token.INCLUDE
830
	case "alias":
831
		return token.ALIAS
832
	case "payee":
833
		return token.PAYEE
834
	case "tag":
835
		return token.TAG
836
	case "apply":
837
		return token.APPLY
838
	case "end":
839
		return token.END
840
	case "Y", "year":
841
		return token.YEAR
842
	case "decimal-mark":
843
		return token.DECIMALMARK
844
	case "D":
845
		return token.D
846
	case "P":
847
		return token.P
848
	case "N":
849
		return token.N
850
	case "C":
851
		return token.C
852
	default:
853
		return token.ILLEGAL
854
	}
855
}