all repos

clerk @ 368f35a

missing tooling for ledger/hledger

clerk/journal/parser/parser.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
lsp: semantic highlights, 2 months ago
1
package parser
2
3
import (
4
	"fmt"
5
	"strconv"
6
	"strings"
7
8
	"olexsmir.xyz/clerk/internal/decimal"
9
	"olexsmir.xyz/clerk/journal/ast"
10
	"olexsmir.xyz/clerk/journal/lexer"
11
	"olexsmir.xyz/clerk/journal/token"
12
)
13
14
type Parser struct {
15
	lexer  *lexer.Lexer
16
	errors []*ast.ParseError
17
	cur    token.Token
18
	peek   token.Token
19
20
	defaultYear int // set by year directive, used for short date inference
21
}
22
23
func New(lex *lexer.Lexer) *Parser {
24
	p := &Parser{lexer: lex}
25
	p.advance() // populate .peek
26
	p.advance() // populate .cur
27
	return p
28
}
29
30
func NewWithYear(lex *lexer.Lexer, year int) *Parser {
31
	p := &Parser{lexer: lex, defaultYear: year}
32
	p.advance() // populate .peek
33
	p.advance() // populate .cur
34
	return p
35
}
36
37
func (p *Parser) ParseJournal() *ast.Journal {
38
	f := &ast.Journal{}
39
	for p.cur.Type != token.EOF {
40
		if e := p.parseEntry(); e != nil {
41
			f.Entries = append(f.Entries, e)
42
		}
43
	}
44
	f.Errors = p.errors
45
	return f
46
}
47
48
func isDirectiveKeyword(t token.Type) bool {
49
	switch t {
50
	case token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE,
51
		token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END,
52
		token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C:
53
		return true
54
	}
55
	return false
56
}
57
58
func (p *Parser) parseEntry() ast.Entry {
59
	if p.got(token.BANG) || p.got(token.AT) {
60
		if isDirectiveKeyword(p.peek.Type) {
61
			p.advance() // consume prefix
62
		}
63
	}
64
	switch p.cur.Type {
65
	case token.ILLEGAL:
66
		p.errorf("illegal character %q", p.cur.Literal)
67
		p.advance()
68
		return nil
69
	case token.INDENT:
70
		p.errorf("unexpected indent")
71
		p.syncToNextline()
72
		return nil
73
	case token.DATE:
74
		return p.parseTransaction()
75
	case token.TILDE:
76
		return p.parsePeriodicTransaction()
77
	case token.EQ:
78
		return p.parseAutomatedTransaction()
79
	case token.NEWLINE:
80
		return p.parseBlankLine()
81
	case token.SEMICOLON, token.HASH, token.PERCENT, token.STAR:
82
		return p.parseComment()
83
	case token.ACCOUNT:
84
		return p.parseAccountDirective()
85
	case token.COMMODITY:
86
		return p.parseCommodityDirective()
87
	case token.INCLUDE:
88
		return p.parseIncludeDirective()
89
	case token.ALIAS:
90
		return p.parseAliasDirective()
91
	case token.PAYEE:
92
		return p.parsePayeeDirective()
93
	case token.TAG:
94
		return p.parseTagDirective()
95
	case token.YEAR:
96
		return p.parseYearDirective()
97
	case token.DECIMALMARK:
98
		return p.parseDecimalMarkDirective()
99
	case token.D:
100
		return p.parseDefaultCommodityDirective()
101
	case token.P:
102
		return p.parseMarketPriceDirective()
103
	case token.N:
104
		return p.parseIgnoredDirective()
105
	case token.C:
106
		return p.parseConversionDirective()
107
	case token.APPLY:
108
		return p.parseApplyDirective()
109
	case token.END:
110
		return p.parseEndDirective()
111
	case token.COMMENTKW:
112
		return p.parseCommentBlockDirective()
113
	default:
114
		p.errorf("unexpected token %s", p.cur.Type)
115
		p.sync()
116
		return nil
117
	}
118
}
119
120
func (p *Parser) parseTransaction() *ast.Transaction {
121
	s := p.cur.Span
122
	tx := &ast.Transaction{}
123
124
	tx.Date = p.parseDate()
125
126
	p.skipWhitespace()
127
128
	// optional secondary date
129
	if p.got(token.EQ) {
130
		p.advance()
131
		p.skipWhitespace()
132
		d := p.parseDate()
133
		tx.SecondDate = &d
134
	}
135
136
	p.skipWhitespace()
137
138
	// optional status
139
	tx.Status = p.parseStatus()
140
141
	// optional code
142
	if p.got(token.LPAREN) {
143
		cs := p.cur.Span
144
		p.advance()
145
		var code strings.Builder
146
		for p.cur.Type != token.RPAREN {
147
			_, _ = code.WriteString(p.cur.Literal)
148
			p.advance()
149
		}
150
		rp := p.cur.Span
151
		p.advance()
152
		tx.Code = &ast.Code{
153
			Value: code.String(),
154
			Span:  token.Span{Start: cs.Start, End: rp.End},
155
		}
156
		p.skipWhitespace()
157
	}
158
159
	// optional payee | note
160
	if p.got(token.TEXT) || p.got(token.STRING) {
161
		tx.Payee = p.parsePayee()
162
163
		// check for | separator
164
		if p.got(token.WHITESPACE) {
165
			p.skipWhitespace()
166
		}
167
168
		if p.got(token.PIPE) {
169
			p.advance()
170
			if p.got(token.TEXT) {
171
				sn := p.cur.Span
172
				n := p.cur.Literal
173
				p.advance()
174
				tx.Note = &ast.Note{Value: n, Span: p.span(sn)}
175
			}
176
		}
177
	}
178
179
	tx.Comment = p.parseOptInlineComment()
180
	p.expectNewline()
181
182
	// header comments — indented ; lines before first posting
183
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
184
		p.advance() // consume indent
185
		c := p.parseComment()
186
		tx.HeaderComments = append(tx.HeaderComments, c)
187
	}
188
189
	// postings
190
	for p.got(token.INDENT) {
191
		if p := p.parsePosting(); p != nil {
192
			tx.Postings = append(tx.Postings, p)
193
		}
194
	}
195
196
	tx.Span = p.span(s)
197
	return tx
198
}
199
200
func unquote(s string) string {
201
	if len(s) >= 2 && ((s[0] == '"' && s[len(s)-1] == '"') || (s[0] == '\'' && s[len(s)-1] == '\'')) {
202
		return s[1 : len(s)-1]
203
	}
204
	return s
205
}
206
207
func (p *Parser) parsePayee() *ast.Payee {
208
	s := p.cur.Span
209
210
	if p.got(token.STRING) {
211
		name := unquote(p.cur.Literal)
212
		p.advance()
213
		return &ast.Payee{Name: name, Span: p.span(s)}
214
	}
215
216
	// keep spaces/tags between text tokens; stop before trailing whitespace
217
	var name strings.Builder
218
	for p.got(token.TEXT) || p.got(token.INT) || p.got(token.DECIMAL) || (p.got(token.WHITESPACE) && (p.willGet(token.TEXT) || p.willGet(token.INT) || p.willGet(token.DECIMAL))) {
219
		_, _ = name.WriteString(p.cur.Literal)
220
		p.advance()
221
	}
222
	return &ast.Payee{Name: unquote(name.String()), Span: p.span(s)}
223
}
224
225
func (p *Parser) parsePeriodicTransaction() *ast.PeriodicTransaction {
226
	s := p.cur.Span
227
	p.expect(token.TILDE)
228
	p.skipWhitespace()
229
230
	pt := &ast.PeriodicTransaction{}
231
232
	pt.Period = p.parsePeriod()
233
234
	if desc, dspan := p.parseOptPeriodicDescription(); desc != "" {
235
		pt.Description = &ast.Description{Value: desc, Span: dspan}
236
	}
237
238
	comment := p.parseOptInlineComment()
239
	p.expectNewline()
240
241
	// header comment
242
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
243
		p.advance()
244
		pt.HeaderComments = append(pt.HeaderComments, p.parseComment())
245
	}
246
247
	// postings
248
	for p.got(token.INDENT) {
249
		if posting := p.parsePosting(); posting != nil {
250
			pt.Postings = append(pt.Postings, posting)
251
		}
252
	}
253
254
	pt.Span = p.span(s)
255
	pt.Comment = comment
256
	return pt
257
}
258
259
func (p *Parser) parseAutomatedTransaction() *ast.AutomatedTransaction {
260
	s := p.cur.Span
261
	p.expect(token.EQ)
262
	p.skipWhitespace()
263
264
	at := &ast.AutomatedTransaction{}
265
266
	// expression
267
	sd := p.cur.Span
268
	expr := p.parseDirectiveExpr()
269
	at.Expr = ast.Expr{Value: expr, Span: p.span(sd)}
270
	at.Comment = p.parseOptInlineComment()
271
	p.expectNewline()
272
273
	// header comments
274
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
275
		p.advance()
276
		at.HeaderComments = append(at.HeaderComments, p.parseComment())
277
	}
278
279
	// postings
280
	for p.got(token.INDENT) {
281
		if p := p.parsePosting(); p != nil {
282
			at.Postings = append(at.Postings, p)
283
		}
284
	}
285
286
	at.Span = p.span(s)
287
	return at
288
}
289
290
func (p *Parser) parsePeriod() ast.Period {
291
	s := p.cur.Span
292
293
	var periodBuf strings.Builder
294
295
	for !p.got(token.NEWLINE) && !p.got(token.EOF) &&
296
		!p.got(token.SEMICOLON) && !p.got(token.HASH) && !p.got(token.PERCENT) && !p.got(token.STAR) {
297
298
		if p.got(token.WHITESPACE) {
299
			if len(p.cur.Literal) >= 2 {
300
				break
301
			}
302
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) ||
303
				p.willGet(token.SEMICOLON) || p.willGet(token.HASH) ||
304
				p.willGet(token.PERCENT) || p.willGet(token.STAR) {
305
				p.advance()
306
				continue
307
			}
308
		}
309
310
		periodBuf.WriteString(p.cur.Literal)
311
		p.advance()
312
	}
313
314
	str := periodBuf.String()
315
	period := ast.Period{Raw: str, Span: p.span(s)}
316
317
	if _, after, ok := strings.Cut(str, " from "); ok {
318
		end := strings.Index(after, " ")
319
		dateStr := after
320
		if end >= 0 {
321
			dateStr = after[:end]
322
		}
323
		if d := parseSimpleDate(dateStr); d.Year > 0 {
324
			fromOff := strings.Index(str, dateStr)
325
			d.Span = periodDateSpan(period, str, dateStr, fromOff)
326
			period.From = &d
327
			rest := after
328
			if end >= 0 {
329
				rest = after[end:]
330
			}
331
			if _, toAfter, ok := strings.Cut(rest, " to "); ok {
332
				if toEnd := strings.Index(toAfter, " "); toEnd >= 0 {
333
					toAfter = toAfter[:toEnd]
334
				}
335
				if d := parseSimpleDate(toAfter); d.Year > 0 {
336
					d.Span = periodDateSpan(period, str, toAfter, fromOff+len(dateStr))
337
					period.To = &d
338
				}
339
			}
340
		}
341
	}
342
	return period
343
}
344
345
// periodDateSpan returns the source span of dateStr, which occurs in the
346
// period text at or after searchFrom. The period span and text cover the same
347
// bytes, so offsets line up 1:1.
348
func periodDateSpan(period ast.Period, text, dateStr string, searchFrom int) token.Span {
349
	off := strings.Index(text[searchFrom:], dateStr)
350
	abs := period.Span.Start.Offset + searchFrom + off
351
	return token.Span{
352
		Start: token.Pos{File: period.Span.Start.File, Offset: abs},
353
		End:   token.Pos{File: period.Span.Start.File, Offset: abs + len(dateStr)},
354
	}
355
}
356
357
func (p *Parser) parseComment() *ast.Comment {
358
	s := p.cur.Span
359
	marker := p.cur.Literal[0]
360
	p.advance()
361
	p.skipWhitespace()
362
363
	var text string
364
	if p.got(token.TEXT) {
365
		text = p.cur.Literal
366
		p.advance()
367
	}
368
369
	p.expectNewline()
370
371
	return &ast.Comment{
372
		Marker: marker,
373
		Text:   text,
374
		Span:   p.span(s),
375
	}
376
}
377
378
func (p *Parser) parseAccountDirective() *ast.AccountDirective {
379
	s := p.cur.Span
380
	p.expect(token.ACCOUNT)
381
	p.skipWhitespace()
382
383
	account := p.parseAccount()
384
	comment := p.parseOptInlineComment()
385
	p.expectNewline()
386
387
	for p.got(token.INDENT) {
388
		p.advance()
389
		for !p.got(token.NEWLINE) && !p.got(token.EOF) {
390
			p.advance()
391
		}
392
		p.expectNewline()
393
	}
394
395
	return &ast.AccountDirective{
396
		Account: account,
397
		Comment: comment,
398
		Span:    p.span(s),
399
	}
400
}
401
402
func (p *Parser) parseCommodityDirective() *ast.CommodityDirective {
403
	s := p.cur.Span
404
	p.expect(token.COMMODITY)
405
	p.skipWhitespace()
406
407
	var commodity string
408
	var commoditySpan token.Span
409
	var format *ast.Amount
410
411
	switch p.cur.Type {
412
	case token.COMMODITYMARK, token.TEXT, token.STRING:
413
		cs := p.cur.Span
414
		commodity = p.cur.Literal
415
		if p.got(token.STRING) {
416
			commodity = unquote(commodity)
417
		}
418
		p.advance()
419
		commoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
420
		hadSpace := p.got(token.WHITESPACE)
421
		p.skipWhitespace()
422
		if p.got(token.INT) || p.got(token.DECIMAL) || p.got(token.TEXT) {
423
			format = p.parseAmount()
424
			format.Commodity = commodity
425
			format.CommoditySpan = commoditySpan
426
			format.CommodityPos = ast.CommodityBefore
427
			format.HasSpace = hadSpace
428
		}
429
	case token.INT, token.DECIMAL:
430
		format = p.parseAmount()
431
		commodity = format.Commodity
432
		commoditySpan = format.CommoditySpan
433
	default:
434
		p.errorf("expected commodity name or amount, got %s", p.cur.Type)
435
	}
436
437
	if commodity == "" {
438
		p.errorf("expected commodity name, got %s", p.cur.Type)
439
	}
440
441
	comment := p.parseOptInlineComment()
442
	p.expectNewline()
443
444
	for p.got(token.INDENT) {
445
		p.advance()
446
		p.skipWhitespace()
447
		if p.got(token.COMMODITYMARK) && p.cur.Literal == "format" {
448
			p.advance()
449
			p.skipWhitespace()
450
			format = p.parseAmount()
451
		} else {
452
			for !p.got(token.NEWLINE) && !p.got(token.EOF) {
453
				p.advance()
454
			}
455
		}
456
		p.expectNewline()
457
	}
458
459
	cd := &ast.CommodityDirective{
460
		Commodity:     commodity,
461
		CommoditySpan: commoditySpan,
462
		Comment:       comment,
463
		Span:          p.span(s),
464
	}
465
	if format != nil {
466
		cd.Format = *format
467
	}
468
	return cd
469
}
470
471
func (p *Parser) parseIncludeDirective() *ast.IncludeDirective {
472
	s := p.cur.Span
473
	p.expect(token.INCLUDE)
474
	p.skipWhitespace()
475
476
	id := &ast.IncludeDirective{}
477
478
	if p.got(token.TEXT) {
479
		id.Path = p.cur.Literal
480
		p.advance()
481
	} else {
482
		p.errorf("expected file path, got %s", p.cur.Type)
483
	}
484
485
	p.skipWhitespace()
486
	id.Comment = p.parseOptInlineComment()
487
	p.expectNewline()
488
	id.Span = p.span(s)
489
	return id
490
}
491
492
func (p *Parser) parseAliasDirective() *ast.AliasDirective {
493
	s := p.cur.Span
494
	alias := &ast.AliasDirective{}
495
	p.expect(token.ALIAS)
496
	p.skipWhitespace()
497
	alias.From = p.parseAccount()
498
	p.skipWhitespace()
499
	p.expect(token.EQ)
500
	p.skipWhitespace()
501
	alias.To = p.parseAccount()
502
	p.skipWhitespace()
503
	alias.Comment = p.parseOptInlineComment()
504
	p.expectNewline()
505
	alias.Span = p.span(s)
506
	return alias
507
}
508
509
func (p *Parser) parsePayeeDirective() *ast.PayeeDirective {
510
	s := p.cur.Span
511
	p.expect(token.PAYEE)
512
	p.skipWhitespace()
513
514
	name := ""
515
	if p.got(token.TEXT) || p.got(token.STRING) {
516
		name = p.parsePayee().Name
517
	}
518
519
	comment := p.parseOptInlineComment()
520
	p.expectNewline()
521
522
	return &ast.PayeeDirective{
523
		Name:    name,
524
		Comment: comment,
525
		Span:    p.span(s),
526
	}
527
}
528
529
func (p *Parser) parseTagDirective() *ast.TagDirective {
530
	s := p.cur.Span
531
	p.expect(token.TAG)
532
	p.skipWhitespace()
533
534
	name := ""
535
	if p.got(token.TEXT) {
536
		name = p.cur.Literal
537
		p.advance()
538
	} else if p.got(token.STRING) {
539
		name = unquote(p.cur.Literal)
540
		p.advance()
541
	}
542
543
	comment := p.parseOptInlineComment()
544
	p.expectNewline()
545
546
	return &ast.TagDirective{
547
		Name:    name,
548
		Comment: comment,
549
		Span:    p.span(s),
550
	}
551
}
552
553
func (p *Parser) parseYearDirective() *ast.YearDirective {
554
	s := p.cur.Span
555
	year := &ast.YearDirective{}
556
	p.expect(token.YEAR)
557
	p.skipWhitespace()
558
559
	if p.got(token.INT) {
560
		year.Year, _ = strconv.Atoi(p.cur.Literal)
561
		p.defaultYear = year.Year
562
		p.advance()
563
	} else {
564
		p.errorf("expected year, got %s", p.cur.Type)
565
	}
566
567
	p.skipWhitespace()
568
	year.Comment = p.parseOptInlineComment()
569
	p.expectNewline()
570
	year.Span = p.span(s)
571
572
	return year
573
}
574
575
func (p *Parser) parseDecimalMarkDirective() *ast.DecimalMarkDirective {
576
	s := p.cur.Span
577
	mark := &ast.DecimalMarkDirective{}
578
	p.expect(token.DECIMALMARK)
579
	p.skipWhitespace()
580
581
	mark.Mark = byte('.')
582
	if p.got(token.TEXT) {
583
		if len(p.cur.Literal) > 0 {
584
			mark.Mark = p.cur.Literal[0]
585
		}
586
		p.advance()
587
	}
588
589
	p.skipWhitespace()
590
	mark.Comment = p.parseOptInlineComment()
591
	p.expectNewline()
592
	mark.Span = p.span(s)
593
	return mark
594
}
595
596
func (p *Parser) parseDefaultCommodityDirective() *ast.DefaultCommodityDirective {
597
	s := p.cur.Span
598
	com := &ast.DefaultCommodityDirective{}
599
	p.expect(token.D)
600
	p.skipWhitespace()
601
	com.Amount = *p.parseAmount()
602
	p.skipWhitespace()
603
	com.Comment = p.parseOptInlineComment()
604
	p.expectNewline()
605
	com.Span = p.span(s)
606
	return com
607
}
608
609
func (p *Parser) parseConversionDirective() *ast.ConversionDirective {
610
	s := p.cur.Span
611
	cd := &ast.ConversionDirective{}
612
	p.expect(token.C)
613
	p.skipWhitespace()
614
615
	if p.isAmountStart() {
616
		cd.From = *p.parseAmount()
617
	} else {
618
		p.errorf("expected amount, got %s", p.cur.Type)
619
	}
620
621
	p.skipWhitespace()
622
	if p.got(token.EQ) {
623
		p.advance()
624
		p.skipWhitespace()
625
		if p.isAmountStart() {
626
			cd.To = *p.parseAmount()
627
		} else {
628
			p.errorf("expected amount, got %s", p.cur.Type)
629
		}
630
	}
631
632
	p.skipWhitespace()
633
	cd.Comment = p.parseOptInlineComment()
634
	p.expectNewline()
635
	cd.Span = p.span(s)
636
	return cd
637
}
638
639
func (p *Parser) parseIgnoredDirective() *ast.IgnoredDirective {
640
	s := p.cur.Span
641
	p.expect(token.N)
642
	p.skipWhitespace()
643
644
	id := &ast.IgnoredDirective{}
645
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) {
646
		id.Text = p.cur.Literal
647
		p.advance()
648
	}
649
	p.skipWhitespace()
650
	id.Comment = p.parseOptInlineComment()
651
652
	p.expectNewline()
653
	id.Span = p.span(s)
654
	return id
655
}
656
657
func (p *Parser) parseMarketPriceDirective() *ast.MarketPriceDirective {
658
	s := p.cur.Span
659
	p.expect(token.P)
660
	p.skipWhitespace()
661
662
	mp := &ast.MarketPriceDirective{}
663
	mp.DateTime.Date = p.parseDate()
664
	p.skipWhitespace()
665
666
	if p.got(token.TIME) {
667
		mp.DateTime.Time = new(p.parseTime())
668
		p.skipWhitespace()
669
	}
670
671
	tok, _ := p.expect(token.COMMODITYMARK)
672
	mp.Commodity = tok.Literal
673
	p.advance()
674
675
	mp.Amount = *p.parseAmount()
676
677
	p.skipWhitespace()
678
	mp.Comment = p.parseOptInlineComment()
679
680
	p.expectNewline()
681
	mp.Span = p.span(s)
682
	return mp
683
}
684
685
func (p *Parser) parseTime() ast.Time {
686
	s := p.cur.Span
687
	tok, _ := p.expect(token.TIME)
688
	lit := tok.Literal
689
690
	parts := strings.Split(lit, ":")
691
	if len(parts) < 2 {
692
		p.errorf("invalid time format: %q", lit)
693
		return ast.Time{Span: p.span(s)}
694
	}
695
696
	hour, _ := strconv.Atoi(parts[0])
697
	minute, _ := strconv.Atoi(parts[1])
698
	second := 0
699
	if len(parts) > 2 {
700
		second, _ = strconv.Atoi(parts[2])
701
	}
702
703
	if hour < 0 || hour > 23 {
704
		p.errorf("invalid hour %d in time %q", hour, lit)
705
	}
706
	if minute < 0 || minute > 59 {
707
		p.errorf("invalid minute %d in time %q", minute, lit)
708
	}
709
	if second < 0 || second > 59 {
710
		p.errorf("invalid second %d in time %q", second, lit)
711
	}
712
713
	return ast.Time{
714
		Hour:   hour,
715
		Minute: minute,
716
		Second: second,
717
		Span:   p.span(s),
718
	}
719
}
720
721
func (p *Parser) parseApplyDirective() *ast.ApplyDirective {
722
	s := p.cur.Span
723
	p.expect(token.APPLY)
724
	p.skipWhitespace()
725
726
	expr := p.parseDirectiveExpr()
727
	comment := p.parseOptInlineComment()
728
	p.expectNewline()
729
730
	return &ast.ApplyDirective{
731
		Expr:    expr,
732
		Comment: comment,
733
		Span:    p.span(s),
734
	}
735
}
736
737
func (p *Parser) parseEndDirective() *ast.EndDirective {
738
	s := p.cur.Span
739
	p.expect(token.END)
740
	p.skipWhitespace()
741
742
	expr := p.parseDirectiveExpr()
743
	comment := p.parseOptInlineComment()
744
	p.expectNewline()
745
746
	return &ast.EndDirective{
747
		Expr:    expr,
748
		Comment: comment,
749
		Span:    p.span(s),
750
	}
751
}
752
753
func (p *Parser) parseCommentBlockDirective() *ast.CommentBlockDirective {
754
	start := p.cur.Span
755
	p.expect(token.COMMENTKW)
756
	p.skipWhitespace()
757
758
	header := p.parseDirectiveExpr()
759
	comment := p.parseOptInlineComment()
760
	p.expectNewline()
761
762
	var content strings.Builder
763
	for p.cur.Type != token.EOF {
764
		if p.got(token.END) {
765
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) {
766
				p.advance()
767
				p.expectNewline()
768
				break
769
			}
770
			if p.willGet(token.WHITESPACE) {
771
				endTok := p.cur
772
				p.advance()
773
				wsTok := p.cur
774
				p.advance()
775
				if p.got(token.TEXT) && p.cur.Literal == "comment" { // todo: this should check if it's an actual COMMENTKW token
776
					p.advance()
777
					p.parseDirectiveExpr()
778
					p.parseOptInlineComment()
779
					p.expectNewline()
780
					break
781
				}
782
				content.WriteString(endTok.Literal)
783
				content.WriteString(wsTok.Literal)
784
				continue
785
			}
786
		}
787
		content.WriteString(p.cur.Literal)
788
		p.advance()
789
	}
790
791
	return &ast.CommentBlockDirective{
792
		Header:  header,
793
		Content: content.String(),
794
		Comment: comment,
795
		Span:    p.span(start),
796
	}
797
}
798
799
func (p *Parser) parseStatus() ast.Status {
800
	s := p.cur.Span
801
	st := ast.Status{}
802
	switch p.cur.Type {
803
	case token.STAR:
804
		p.advance()
805
		p.skipWhitespace()
806
		st.Value = ast.StatusCleared
807
	case token.BANG:
808
		p.advance()
809
		p.skipWhitespace()
810
		st.Value = ast.StatusPending
811
	default:
812
		st.Value = ast.StatusNone
813
	}
814
	st.Span = p.span(s)
815
	return st
816
}
817
818
func (p *Parser) isAmountStart() bool {
819
	switch p.cur.Type {
820
	default:
821
		return false
822
	case token.COMMODITYMARK, token.STRING, token.INT, token.DECIMAL, token.MINUS, token.PLUS, token.PARENEXPR:
823
		return true
824
	}
825
}
826
827
func (p *Parser) parseAmount() *ast.Amount {
828
	s := p.cur.Span
829
	amt := &ast.Amount{
830
		QuantityFmt: ast.QuantityFormat{Decimal: '.'},
831
	}
832
	defer func() {
833
		// The span covers from the first token to the start of the next unconsumed token.
834
		// Since parseQuantityInto (and possible commodity consumption) advanced past the last
835
		// amount token, p.cur points to the next token after the amount — which is the correct end.
836
		amt.Span = p.span(s)
837
	}()
838
839
	// commodity before quantity: $10.00, eur 10.00
840
	if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
841
		cs := p.cur.Span
842
		amt.Commodity = unquote(p.cur.Literal)
843
		amt.CommodityPos = ast.CommodityBefore
844
		p.advance()
845
		amt.CommoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
846
		if p.got(token.WHITESPACE) {
847
			amt.HasSpace = true
848
			p.skipWhitespace()
849
		}
850
		switch p.cur.Type {
851
		case token.MINUS:
852
			amt.IsNegative = true
853
			p.advance()
854
		case token.PLUS:
855
			p.advance()
856
		}
857
		p.parseQuantityInto(amt)
858
	} else {
859
		// optional sign
860
		switch p.cur.Type {
861
		case token.MINUS:
862
			amt.IsNegative = true
863
			p.advance()
864
		case token.PLUS:
865
			p.advance()
866
		}
867
868
		// commodity before quantity: -$120, -eur 120:
869
		if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
870
			cs := p.cur.Span
871
			amt.Commodity = unquote(p.cur.Literal)
872
			amt.CommodityPos = ast.CommodityBefore
873
			p.advance()
874
			amt.CommoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
875
			if p.got(token.WHITESPACE) {
876
				amt.HasSpace = true
877
				p.skipWhitespace()
878
			}
879
		}
880
881
		p.parseQuantityInto(amt)
882
883
		// commodity after quantity: 10.00 UAH, 10.00 "EUR" (only if not set)
884
		if amt.Commodity == "" {
885
			switch p.cur.Type {
886
			case token.WHITESPACE:
887
				p.skipWhitespace()
888
				if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
889
					cs := p.cur.Span
890
					amt.HasSpace = true
891
					amt.Commodity = unquote(p.cur.Literal)
892
					amt.CommodityPos = ast.CommodityAfter
893
					p.advance()
894
					amt.CommoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
895
				}
896
			case token.COMMODITYMARK, token.TEXT, token.STRING:
897
				cs := p.cur.Span
898
				amt.Commodity = unquote(p.cur.Literal)
899
				amt.CommodityPos = ast.CommodityAfter
900
				p.advance()
901
				amt.CommoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
902
			}
903
		}
904
	}
905
906
	return amt
907
}
908
909
func (p *Parser) parseAmountWithOptExpr() *ast.Amount {
910
	if p.got(token.STAR) {
911
		p.advance()
912
		p.skipWhitespace()
913
		amt := p.parseAmount()
914
		if amt != nil {
915
			amt.IsExpr = true
916
		}
917
		return amt
918
	}
919
	if p.got(token.PARENEXPR) {
920
		lit := p.cur.Literal
921
		amt := &ast.Amount{
922
			IsExpr:      true,
923
			QuantityFmt: ast.QuantityFormat{Decimal: '.'},
924
		}
925
		if len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
926
			inner := lit[1 : len(lit)-1]
927
			i := 0
928
			for i < len(inner) && (inner[i] == ' ' || inner[i] == '\t') {
929
				i++
930
			}
931
			j := len(inner)
932
			for j > i && (inner[j-1] == ' ' || inner[j-1] == '\t') {
933
				j--
934
			}
935
			amt.Expr = inner[i:j]
936
		}
937
		amt.Span = p.cur.Span
938
		p.advance()
939
		return amt
940
	}
941
	return p.parseAmount()
942
}
943
944
func (p *Parser) parsePosting() *ast.Posting {
945
	s := p.cur.Span
946
	posting := &ast.Posting{}
947
	p.expect(token.INDENT)
948
949
	// exit if it's empty line
950
	if p.got(token.NEWLINE) || p.got(token.EOF) {
951
		p.syncToNextline()
952
		return nil
953
	}
954
955
	// optional status, outside of brackets, '! (account)'
956
	posting.Status = p.parseStatus()
957
958
	// detect virtual posting brackets
959
	switch p.cur.Type {
960
	case token.LPAREN:
961
		posting.Type = ast.PostingVirtualUnbalanced
962
		p.advance()
963
	case token.LBRACKET:
964
		posting.Type = ast.PostingVirtualBalanced
965
		p.advance()
966
	}
967
968
	// optional status, inside of brackets, '(* account)'
969
	if p.got(token.STAR) || p.got(token.BANG) {
970
		posting.Status = p.parseStatus()
971
	}
972
973
	// validate, must be account text
974
	if p.cur.Type != token.TEXT {
975
		p.errorf("expected account name, got %s", p.cur.Type)
976
		p.syncToNextline()
977
		return nil
978
	}
979
980
	posting.Account = p.parseAccount()
981
982
	// consume closing bracket
983
	switch p.cur.Type {
984
	case token.RPAREN:
985
		p.advance()
986
	case token.RBRACKET:
987
		p.advance()
988
	}
989
990
	// optional amount - after two spaces
991
	if p.got(token.WHITESPACE) {
992
		p.skipWhitespace()
993
		if p.isAmountStart() || p.got(token.STAR) {
994
			posting.Amount = p.parseAmountWithOptExpr()
995
		}
996
	}
997
998
	// optional cost '@' or '@@'
999
	if p.got(token.WHITESPACE) {
1000
		p.skipWhitespace()
1001
	}
1002
	if p.got(token.AT) || p.got(token.ATAT) {
1003
		posting.Cost = p.parseCost()
1004
	}
1005
1006
	// optional balance assertion
1007
	if p.got(token.WHITESPACE) {
1008
		p.skipWhitespace()
1009
	}
1010
	if p.got(token.EQ) || p.got(token.EQEQ) || p.got(token.EQEQEQ) {
1011
		posting.Balance = p.parseBalanceAssertion()
1012
	}
1013
1014
	posting.Comment = p.parseOptInlineComment()
1015
	p.expectNewline()
1016
1017
	// continuation comments
1018
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
1019
		p.advance()
1020
		c := p.parseComment()
1021
		posting.Comments = append(posting.Comments, *c)
1022
	}
1023
1024
	posting.Span = p.span(s)
1025
	return posting
1026
}
1027
1028
func (p *Parser) parseCost() *ast.Cost {
1029
	s := p.cur.Span
1030
	isTotal := p.got(token.ATAT)
1031
	p.advance() // consume '@' '@@'
1032
	p.skipWhitespace()
1033
	return &ast.Cost{
1034
		IsTotal: isTotal,
1035
		Amount:  *p.parseAmount(),
1036
		Span:    p.span(s),
1037
	}
1038
}
1039
1040
func (p *Parser) parseBalanceAssertion() *ast.BalanceAssertion {
1041
	s := p.cur.Span
1042
1043
	ba := &ast.BalanceAssertion{}
1044
	switch p.cur.Type {
1045
	case token.EQ: // basic assertion
1046
	case token.EQEQ:
1047
		ba.IsStrict = true
1048
	case token.EQEQEQ:
1049
		ba.IsStrict = true
1050
		ba.IsInclusive = true
1051
	}
1052
	p.advance()
1053
	p.skipWhitespace()
1054
1055
	ba.Amount = *p.parseAmount()
1056
	p.skipWhitespace()
1057
	if p.got(token.AT) || p.got(token.ATAT) {
1058
		c := p.parseCost()
1059
		ba.Cost = c
1060
	}
1061
	ba.Span = p.span(s)
1062
	return ba
1063
}
1064
1065
func (p *Parser) readAccountSegment() (ast.SubAccount, bool) {
1066
	switch p.cur.Type {
1067
	case token.TEXT:
1068
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
1069
		p.advance()
1070
1071
		// handle multi work segment, e.g: "credit card"
1072
		if p.got(token.WHITESPACE) && p.willGet(token.TEXT) && len(p.peek.Literal) > 0 && p.peek.Literal[0] != '(' {
1073
			sub.Name += " "
1074
			p.advance()
1075
			sub.Name += p.cur.Literal
1076
			p.advance()
1077
		}
1078
		return sub, true
1079
1080
	case token.COMMODITYMARK:
1081
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
1082
		p.advance()
1083
		// merge "EUR" + "-HRK" to "EUR-HRK"
1084
		for p.got(token.TEXT) {
1085
			sub.Name += p.cur.Literal
1086
			p.advance()
1087
		}
1088
		return sub, true
1089
1090
	default:
1091
		return ast.SubAccount{}, false
1092
	}
1093
}
1094
1095
func (p *Parser) parseAccount() ast.Account {
1096
	s := p.cur.Span
1097
	acc := ast.Account{}
1098
1099
	sub, ok := p.readAccountSegment()
1100
	if !ok {
1101
		p.errorf("expected account, got %s", p.cur.Type)
1102
		return ast.Account{}
1103
	}
1104
	acc.Name = append(acc.Name, sub)
1105
1106
	for p.got(token.COLON) {
1107
		p.advance()
1108
		sub, ok := p.readAccountSegment()
1109
		if !ok {
1110
			break
1111
		}
1112
		acc.Name = append(acc.Name, sub)
1113
	}
1114
1115
	acc.Span = p.span(s)
1116
	return acc
1117
}
1118
1119
func (p *Parser) parseDate() ast.Date {
1120
	s := p.cur.Span
1121
	tok, ok := p.expect(token.DATE)
1122
	if !ok {
1123
		return ast.Date{Span: p.span(s)}
1124
	}
1125
1126
	sep := byte(0)
1127
	lit := tok.Literal
1128
	for i := 0; i < len(lit); i++ {
1129
		if lit[i] == '/' || lit[i] == '-' || lit[i] == '.' {
1130
			sep = lit[i]
1131
			break
1132
		}
1133
	}
1134
	if sep == 0 {
1135
		p.errorf("invalid date format: %q", lit)
1136
		return ast.Date{Span: p.span(s)}
1137
	}
1138
1139
	parts := strings.Split(lit, string(sep))
1140
1141
	// M/D or MM/DD (year inferred)
1142
	if len(parts) == 2 {
1143
		month, err := strconv.Atoi(parts[0])
1144
		day, err2 := strconv.Atoi(parts[1])
1145
		if err != nil || err2 != nil {
1146
			p.errorf("invalid date literal: %q", lit)
1147
			return ast.Date{Span: p.span(s)}
1148
		}
1149
		if month < 1 || month > 12 {
1150
			p.errorf("invalid month %d in %q", month, lit)
1151
			return ast.Date{Span: p.span(s)}
1152
		}
1153
		if day < 1 || day > 31 {
1154
			p.errorf("invalid day %d in %q", day, lit)
1155
			return ast.Date{Span: p.span(s)}
1156
		}
1157
		return ast.Date{Year: p.defaultYear, Month: month, Day: day, Sep: sep, Span: p.span(s)}
1158
	}
1159
1160
	if len(parts) != 3 {
1161
		p.errorf("invalid date format: %q", lit)
1162
		return ast.Date{Span: p.span(s)}
1163
	}
1164
1165
	year, err := strconv.Atoi(parts[0])
1166
	month, err2 := strconv.Atoi(parts[1])
1167
	day, err3 := strconv.Atoi(parts[2])
1168
	if err != nil || err2 != nil || err3 != nil {
1169
		p.errorf("invalid date literal: %q", lit)
1170
		return ast.Date{Span: p.span(s)}
1171
	}
1172
	if month < 1 || month > 12 {
1173
		p.errorf("invalid month %d in %q", month, lit)
1174
		return ast.Date{Span: p.span(s)}
1175
	}
1176
	if day < 1 || day > 31 {
1177
		p.errorf("invalid day %d in %q", day, lit)
1178
		return ast.Date{Span: p.span(s)}
1179
	}
1180
1181
	return ast.Date{
1182
		Year:  year,
1183
		Month: month,
1184
		Day:   day,
1185
		Sep:   sep,
1186
		Span:  p.span(s),
1187
	}
1188
}
1189
1190
func (p *Parser) parseOptInlineComment() *ast.Comment {
1191
	p.skipWhitespace()
1192
	if p.cur.Type != token.SEMICOLON {
1193
		return nil
1194
	}
1195
1196
	s := p.cur.Span
1197
	marker := p.cur.Literal[0]
1198
	p.advance() // consume marker
1199
	p.skipWhitespace()
1200
1201
	text := ""
1202
	if p.got(token.TEXT) {
1203
		text = p.cur.Literal
1204
		p.advance()
1205
	}
1206
1207
	return &ast.Comment{
1208
		Marker: marker,
1209
		Text:   text,
1210
		Span:   p.span(s),
1211
	}
1212
}
1213
1214
func (p *Parser) parseOptPeriodicDescription() (string, token.Span) {
1215
	if p.cur.Type != token.WHITESPACE || len(p.cur.Literal) < 2 {
1216
		return "", token.Span{}
1217
	}
1218
1219
	p.skipWhitespace()
1220
1221
	if p.cur.Type != token.TEXT {
1222
		return "", token.Span{}
1223
	}
1224
1225
	s := p.cur.Span
1226
	desc := p.parseDescription()
1227
	return desc, p.span(s)
1228
}
1229
1230
func (p *Parser) parseDescription() string {
1231
	var desc strings.Builder
1232
	for p.got(token.TEXT) || (p.got(token.WHITESPACE) && p.willGet(token.TEXT)) {
1233
		_, _ = desc.WriteString(p.cur.Literal)
1234
		p.advance()
1235
	}
1236
	return desc.String()
1237
}
1238
1239
func (p *Parser) parseDirectiveExpr() string {
1240
	var b strings.Builder
1241
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF && p.cur.Type != token.SEMICOLON {
1242
		_, _ = b.WriteString(p.cur.Literal)
1243
		p.advance()
1244
	}
1245
	return b.String()
1246
}
1247
1248
func (p *Parser) parseQuantityInto(amt *ast.Amount) {
1249
	if p.cur.Type != token.INT && p.cur.Type != token.DECIMAL && p.cur.Type != token.TEXT {
1250
		p.errorf("expected quantity, got %s", p.cur.Type)
1251
		return
1252
	}
1253
1254
	lit := p.cur.Literal
1255
	p.advance()
1256
1257
	// detect format metadata before normalizing
1258
	amt.QuantityFmt = detectFormat(lit)
1259
1260
	// normalize for decimal.NewFromString
1261
	// remove thousands separators, replace decimal mark with '.'
1262
	normalized := normalizeLiteral(lit, amt.QuantityFmt.Thousands, amt.QuantityFmt.Decimal)
1263
1264
	q, err := decimal.FromString(normalized)
1265
	if err != nil {
1266
		p.errorf("invalid quantity %q: %v", lit, err)
1267
		return
1268
	}
1269
1270
	if amt.IsNegative {
1271
		q = q.Neg()
1272
	}
1273
	amt.Quantity = q
1274
}
1275
1276
func (p *Parser) parseBlankLine() *ast.BlankLine {
1277
	s := p.cur.Span
1278
	p.expectNewline()
1279
	return &ast.BlankLine{Span: s}
1280
}
1281
1282
func (p *Parser) expectNewline() {
1283
	if p.got(token.NEWLINE) || p.got(token.EOF) {
1284
		if p.got(token.NEWLINE) {
1285
			p.advance()
1286
		}
1287
		return
1288
	}
1289
	p.errorf("expected %s, got %s", token.NEWLINE, p.cur.Type)
1290
}
1291
1292
func (p *Parser) advance() token.Token {
1293
	prev := p.cur
1294
	p.cur = p.peek
1295
	p.peek = p.lexer.Next()
1296
	return prev
1297
}
1298
1299
func (p *Parser) got(kind token.Type) bool     { return p.cur.Type == kind }
1300
func (p *Parser) willGet(kind token.Type) bool { return p.peek.Type == kind }
1301
1302
func (p *Parser) expect(kind token.Type) (token.Token, bool) {
1303
	if p.got(kind) {
1304
		return p.advance(), true
1305
	}
1306
	p.errorf("expected %s, got %s", kind, p.cur.Type)
1307
	return p.cur, false
1308
}
1309
1310
func (p *Parser) errorf(format string, args ...any) {
1311
	p.errors = append(p.errors, &ast.ParseError{
1312
		Span:    p.cur.Span,
1313
		Message: fmt.Sprintf(format, args...),
1314
	})
1315
}
1316
1317
func (p *Parser) sync() {
1318
	for {
1319
		switch p.cur.Type {
1320
		case token.EOF:
1321
			return
1322
		case token.NEWLINE:
1323
			p.advance()
1324
			switch p.cur.Type {
1325
			case token.DATE, token.ACCOUNT, token.COMMODITY,
1326
				token.INCLUDE, token.ALIAS, token.PAYEE,
1327
				token.TAG, token.YEAR, token.D, token.P,
1328
				token.APPLY, token.END, token.COMMENTKW,
1329
				token.DECIMALMARK, token.TILDE, token.N, token.EQ:
1330
				return
1331
			}
1332
		default:
1333
			p.advance()
1334
		}
1335
	}
1336
}
1337
1338
func (p *Parser) syncToNextline() {
1339
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF {
1340
		p.advance()
1341
	}
1342
	if p.got(token.NEWLINE) {
1343
		p.advance()
1344
	}
1345
}
1346
1347
func (p *Parser) skipWhitespace() {
1348
	for p.got(token.WHITESPACE) {
1349
		p.advance()
1350
	}
1351
}
1352
1353
func (p *Parser) span(s token.Span) token.Span {
1354
	return token.Span{Start: s.Start, End: p.cur.Span.Start}
1355
}
1356
1357
func normalizeLiteral(lit string, thousands, decimal byte) string {
1358
	var b strings.Builder
1359
	for _, ch := range []byte(lit) {
1360
		if thousands != 0 && ch == thousands {
1361
			continue // skip thousands separator
1362
		}
1363
		if ch == decimal {
1364
			b.WriteByte('.')
1365
		} else {
1366
			b.WriteByte(ch)
1367
		}
1368
	}
1369
	return b.String()
1370
}
1371
1372
func detectFormat(lit string) ast.QuantityFormat {
1373
	var seps []int
1374
	for i, ch := range []byte(lit) {
1375
		if ch == '.' || ch == ',' || ch == ' ' || ch == '_' || ch == '\'' {
1376
			seps = append(seps, i)
1377
		}
1378
	}
1379
1380
	if len(seps) == 0 {
1381
		return ast.QuantityFormat{Decimal: '.', Thousands: 0, Precision: 0}
1382
	}
1383
1384
	last := seps[len(seps)-1]
1385
	dec := lit[last]
1386
	var thou byte
1387
	if len(seps) > 1 {
1388
		thou = lit[seps[0]]
1389
	} else if dec == ' ' || dec == '_' || dec == '\'' {
1390
		// single space/underscore/apostrophe is always thousands
1391
		thou = dec
1392
		dec = '.'
1393
	}
1394
1395
	// calculate precision when the last separator is a real decimal
1396
	prec := 0
1397
	if thou == 0 || len(seps) > 1 {
1398
		prec = len(lit) - last - 1
1399
	}
1400
1401
	return ast.QuantityFormat{Decimal: dec, Thousands: thou, Precision: prec}
1402
}
1403
1404
func parseSimpleDate(s string) ast.Date {
1405
	if len(s) < 8 {
1406
		return ast.Date{}
1407
	}
1408
	sep := byte('-')
1409
	if strings.Contains(s, "/") {
1410
		sep = byte('/')
1411
	} else if strings.Contains(s, ".") {
1412
		sep = byte('.')
1413
	}
1414
	parts := strings.Split(s, string(sep))
1415
	if len(parts) != 3 {
1416
		return ast.Date{}
1417
	}
1418
	year, _ := strconv.Atoi(parts[0])
1419
	month, _ := strconv.Atoi(parts[1])
1420
	day, _ := strconv.Atoi(parts[2])
1421
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep}
1422
}