all repos

clerk @ f58ca0d53763fda8f02b50cee150a3d30918fd1d

missing tooling for ledger/hledger

clerk/journal/parser/parser.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
parser: remove reduntant code, 2 months ago
1
package parser
2
3
import (
4
	"fmt"
5
	"strconv"
6
	"strings"
7
8
	"olexsmir.xyz/clerk/internal/decimal"
9
	"olexsmir.xyz/clerk/journal/ast"
10
	"olexsmir.xyz/clerk/journal/lexer"
11
	"olexsmir.xyz/clerk/journal/token"
12
)
13
14
type Parser struct {
15
	lexer  *lexer.Lexer
16
	errors []*ast.ParseError
17
	cur    token.Token
18
	peek   token.Token
19
20
	defaultYear int // set by year directive, used for short date inference
21
}
22
23
func New(lex *lexer.Lexer) *Parser {
24
	p := &Parser{lexer: lex}
25
	p.advance() // populate .peek
26
	p.advance() // populate .cur
27
	return p
28
}
29
30
func NewWithYear(lex *lexer.Lexer, year int) *Parser {
31
	p := &Parser{lexer: lex, defaultYear: year}
32
	p.advance() // populate .peek
33
	p.advance() // populate .cur
34
	return p
35
}
36
37
func (p *Parser) ParseJournal() *ast.Journal {
38
	f := &ast.Journal{}
39
	for p.cur.Type != token.EOF {
40
		if e := p.parseEntry(); e != nil {
41
			f.Entries = append(f.Entries, e)
42
		}
43
	}
44
	f.Errors = p.errors
45
	return f
46
}
47
48
func (p *Parser) parseEntry() ast.Entry {
49
	if p.got(token.BANG) || p.got(token.AT) {
50
		if isDirectiveKeyword(p.peek.Type) {
51
			p.advance() // consume prefix
52
		}
53
	}
54
55
	switch p.cur.Type {
56
	case token.ILLEGAL:
57
		p.errorf("illegal character %q", p.cur.Literal)
58
		p.advance()
59
		return nil
60
	case token.INDENT:
61
		p.errorf("unexpected indent")
62
		p.syncToNextline()
63
		return nil
64
	case token.DATE:
65
		return p.parseTransaction()
66
	case token.TILDE:
67
		return p.parsePeriodicTransaction()
68
	case token.EQ:
69
		return p.parseAutomatedTransaction()
70
	case token.NEWLINE:
71
		return p.parseBlankLine()
72
	case token.SEMICOLON, token.HASH, token.PERCENT, token.STAR:
73
		return p.parseComment()
74
	case token.ACCOUNT:
75
		return p.parseAccountDirective()
76
	case token.COMMODITY:
77
		return p.parseCommodityDirective()
78
	case token.INCLUDE:
79
		return p.parseIncludeDirective()
80
	case token.ALIAS:
81
		return p.parseAliasDirective()
82
	case token.PAYEE:
83
		return p.parsePayeeDirective()
84
	case token.TAG:
85
		return p.parseTagDirective()
86
	case token.YEAR:
87
		return p.parseYearDirective()
88
	case token.DECIMALMARK:
89
		return p.parseDecimalMarkDirective()
90
	case token.D:
91
		return p.parseDefaultCommodityDirective()
92
	case token.P:
93
		return p.parseMarketPriceDirective()
94
	case token.N:
95
		return p.parseIgnoredDirective()
96
	case token.C:
97
		return p.parseConversionDirective()
98
	case token.APPLY:
99
		return p.parseApplyDirective()
100
	case token.END:
101
		return p.parseEndDirective()
102
	case token.COMMENTKW:
103
		return p.parseCommentBlockDirective()
104
	default:
105
		p.errorf("unexpected token %s", p.cur.Type)
106
		p.sync()
107
		return nil
108
	}
109
}
110
111
func (p *Parser) parseTransaction() *ast.Transaction {
112
	s := p.cur.Span
113
	tx := &ast.Transaction{}
114
115
	tx.Date = p.parseDate()
116
117
	p.skipWhitespace()
118
119
	// optional secondary date
120
	if p.got(token.EQ) {
121
		p.advance()
122
		p.skipWhitespace()
123
		d := p.parseDate()
124
		tx.SecondDate = &d
125
	}
126
127
	p.skipWhitespace()
128
129
	// optional status
130
	tx.Status = p.parseStatus()
131
132
	// optional code - the lexer emits "(CODE)" as a single TEXT token; split it here
133
	if p.got(token.TEXT) {
134
		if lit := p.cur.Literal; len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
135
			tx.Code = &ast.Code{Value: lit[1 : len(lit)-1], Span: p.cur.Span}
136
			p.advance()
137
			p.skipWhitespace()
138
		}
139
	}
140
141
	// optional payee | note
142
	if p.got(token.TEXT) || p.got(token.STRING) {
143
		tx.Payee = p.parsePayee()
144
145
		// check for | separator
146
		p.skipWhitespace()
147
148
		if p.got(token.PIPE) {
149
			p.advance()
150
			if p.got(token.TEXT) {
151
				sn := p.cur.Span
152
				n := p.cur.Literal
153
				p.advance()
154
				tx.Note = &ast.Note{Value: n, Span: p.span(sn)}
155
			}
156
		}
157
	}
158
159
	tx.Comment = p.parseOptInlineComment()
160
	p.expectNewline()
161
162
	tx.HeaderComments, tx.Postings = p.parseHeaderCommentsAndPostings()
163
164
	tx.Span = p.span(s)
165
	return tx
166
}
167
168
func unquote(s string) string {
169
	if len(s) >= 2 && ((s[0] == '"' && s[len(s)-1] == '"') || (s[0] == '\'' && s[len(s)-1] == '\'')) {
170
		return s[1 : len(s)-1]
171
	}
172
	return s
173
}
174
175
func (p *Parser) parsePayee() *ast.Payee {
176
	s := p.cur.Span
177
178
	if p.got(token.STRING) {
179
		name := unquote(p.cur.Literal)
180
		p.advance()
181
		return &ast.Payee{Name: name, Span: p.span(s)}
182
	}
183
184
	// keep spaces/tags between text tokens; stop before trailing whitespace
185
	var name strings.Builder
186
	for payeeWord(p.cur.Type) || (payeeWord(p.peek.Type) && p.got(token.WHITESPACE)) {
187
		_, _ = name.WriteString(p.cur.Literal)
188
		p.advance()
189
	}
190
	return &ast.Payee{Name: unquote(name.String()), Span: p.span(s)}
191
}
192
193
func payeeWord(t token.Type) bool {
194
	switch t {
195
	case token.TEXT, token.INT, token.DECIMAL, token.COMMODITYMARK:
196
		return true
197
	}
198
	return false
199
}
200
201
func (p *Parser) parsePeriodicTransaction() *ast.PeriodicTransaction {
202
	s := p.cur.Span
203
	p.expect(token.TILDE)
204
	p.skipWhitespace()
205
206
	pt := &ast.PeriodicTransaction{}
207
208
	pt.Period = p.parsePeriod()
209
210
	if desc, dspan := p.parseOptPeriodicDescription(); desc != "" {
211
		pt.Description = &ast.Description{Value: desc, Span: dspan}
212
	}
213
214
	comment := p.parseOptInlineComment()
215
	p.expectNewline()
216
217
	pt.HeaderComments, pt.Postings = p.parseHeaderCommentsAndPostings()
218
219
	pt.Span = p.span(s)
220
	pt.Comment = comment
221
	return pt
222
}
223
224
func (p *Parser) parseAutomatedTransaction() *ast.AutomatedTransaction {
225
	s := p.cur.Span
226
	p.expect(token.EQ)
227
	p.skipWhitespace()
228
229
	at := &ast.AutomatedTransaction{}
230
231
	// expression
232
	sd := p.cur.Span
233
	expr := p.parseDirectiveExpr()
234
	at.Expr = ast.Expr{Value: expr, Span: p.span(sd)}
235
	at.Comment = p.parseOptInlineComment()
236
	p.expectNewline()
237
238
	at.HeaderComments, at.Postings = p.parseHeaderCommentsAndPostings()
239
240
	at.Span = p.span(s)
241
	return at
242
}
243
244
func (p *Parser) parseHeaderCommentsAndPostings() (comments []*ast.Comment, postings []*ast.Posting) {
245
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
246
		p.advance() // consume indent
247
		comments = append(comments, p.parseComment())
248
	}
249
250
	for p.got(token.INDENT) {
251
		if posting := p.parsePosting(); posting != nil {
252
			postings = append(postings, posting)
253
		}
254
	}
255
256
	return comments, postings
257
}
258
259
func (p *Parser) parsePeriod() ast.Period {
260
	s := p.cur.Span
261
262
	var periodBuf strings.Builder
263
264
	for !p.got(token.NEWLINE) && !p.got(token.EOF) &&
265
		!p.got(token.SEMICOLON) && !p.got(token.HASH) && !p.got(token.PERCENT) && !p.got(token.STAR) {
266
267
		if p.got(token.WHITESPACE) {
268
			if len(p.cur.Literal) >= 2 {
269
				break
270
			}
271
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) ||
272
				p.willGet(token.SEMICOLON) || p.willGet(token.HASH) ||
273
				p.willGet(token.PERCENT) || p.willGet(token.STAR) {
274
				p.advance()
275
				continue
276
			}
277
		}
278
279
		periodBuf.WriteString(p.cur.Literal)
280
		p.advance()
281
	}
282
283
	str := periodBuf.String()
284
	period := ast.Period{Raw: str, Span: p.span(s)}
285
286
	if _, after, ok := strings.Cut(str, " from "); ok {
287
		end := strings.Index(after, " ")
288
		dateStr := after
289
		if end >= 0 {
290
			dateStr = after[:end]
291
		}
292
		if d := parseSimpleDate(dateStr); d.Year > 0 {
293
			fromOff := strings.Index(str, dateStr)
294
			d.Span = periodDateSpan(period, str, dateStr, fromOff)
295
			period.From = &d
296
			rest := after
297
			if end >= 0 {
298
				rest = after[end:]
299
			}
300
			if _, toAfter, ok := strings.Cut(rest, " to "); ok {
301
				if toEnd := strings.Index(toAfter, " "); toEnd >= 0 {
302
					toAfter = toAfter[:toEnd]
303
				}
304
				if d := parseSimpleDate(toAfter); d.Year > 0 {
305
					d.Span = periodDateSpan(period, str, toAfter, fromOff+len(dateStr))
306
					period.To = &d
307
				}
308
			}
309
		}
310
	}
311
	return period
312
}
313
314
// periodDateSpan returns the source span of dateStr, which occurs in the
315
// period text at or after searchFrom. The period span and text cover the same
316
// bytes, so offsets line up 1:1.
317
func periodDateSpan(period ast.Period, text, dateStr string, searchFrom int) token.Span {
318
	off := strings.Index(text[searchFrom:], dateStr)
319
	abs := period.Span.Start.Offset + searchFrom + off
320
	return token.Span{
321
		Start: token.Pos{File: period.Span.Start.File, Offset: abs},
322
		End:   token.Pos{File: period.Span.Start.File, Offset: abs + len(dateStr)},
323
	}
324
}
325
326
func (p *Parser) parseComment() *ast.Comment {
327
	s := p.cur.Span
328
	c := p.parseCommentRest(s)
329
	p.expectNewline()
330
	c.Span = p.span(s) // comment spans its line through the newline
331
	return c
332
}
333
334
func (p *Parser) parseAccountDirective() *ast.AccountDirective {
335
	s := p.cur.Span
336
	p.expect(token.ACCOUNT)
337
	p.skipWhitespace()
338
339
	account := p.parseAccount()
340
	comment := p.parseOptInlineComment()
341
	p.expectNewline()
342
343
	for p.got(token.INDENT) {
344
		p.advance()
345
		for !p.got(token.NEWLINE) && !p.got(token.EOF) {
346
			p.advance()
347
		}
348
		p.expectNewline()
349
	}
350
351
	return &ast.AccountDirective{
352
		Account: account,
353
		Comment: comment,
354
		Span:    p.span(s),
355
	}
356
}
357
358
func (p *Parser) parseCommodityDirective() *ast.CommodityDirective {
359
	s := p.cur.Span
360
	p.expect(token.COMMODITY)
361
	p.skipWhitespace()
362
363
	var commodity string
364
	var commoditySpan token.Span
365
	var format *ast.Amount
366
367
	switch p.cur.Type {
368
	case token.COMMODITYMARK, token.TEXT, token.STRING:
369
		cs := p.cur.Span
370
		commodity = unquote(p.cur.Literal)
371
		p.advance()
372
		commoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
373
		hadSpace := p.got(token.WHITESPACE)
374
		p.skipWhitespace()
375
		if p.got(token.INT) || p.got(token.DECIMAL) || p.got(token.TEXT) {
376
			format = p.parseAmount()
377
			format.Commodity = commodity
378
			format.CommoditySpan = commoditySpan
379
			format.CommodityPos = ast.CommodityBefore
380
			format.HasSpace = hadSpace
381
		}
382
	case token.INT, token.DECIMAL:
383
		format = p.parseAmount()
384
		commodity = format.Commodity
385
		commoditySpan = format.CommoditySpan
386
	default:
387
		p.errorf("expected commodity name or amount, got %s", p.cur.Type)
388
	}
389
390
	if commodity == "" {
391
		p.errorf("expected commodity name, got %s", p.cur.Type)
392
	}
393
394
	comment := p.parseOptInlineComment()
395
	p.expectNewline()
396
397
	for p.got(token.INDENT) {
398
		p.advance()
399
		p.skipWhitespace()
400
		if p.got(token.TEXT) && p.cur.Literal == "format" {
401
			p.advance()
402
			p.skipWhitespace()
403
			format = p.parseAmount()
404
			p.expectNewline()
405
			continue
406
		}
407
		for !p.got(token.NEWLINE) && !p.got(token.EOF) {
408
			p.advance()
409
		}
410
		p.expectNewline()
411
	}
412
413
	cd := &ast.CommodityDirective{
414
		Commodity:     commodity,
415
		CommoditySpan: commoditySpan,
416
		Comment:       comment,
417
		Span:          p.span(s),
418
	}
419
	if format != nil {
420
		cd.Format = *format
421
	}
422
	return cd
423
}
424
425
func (p *Parser) parseIncludeDirective() *ast.IncludeDirective {
426
	s := p.cur.Span
427
	p.expect(token.INCLUDE)
428
	p.skipWhitespace()
429
430
	id := &ast.IncludeDirective{}
431
432
	if p.got(token.TEXT) {
433
		id.Path = p.cur.Literal
434
		p.advance()
435
	} else {
436
		p.errorf("expected file path, got %s", p.cur.Type)
437
	}
438
439
	id.Comment = p.parseOptInlineComment()
440
	p.expectNewline()
441
	id.Span = p.span(s)
442
	return id
443
}
444
445
func (p *Parser) parseAliasDirective() *ast.AliasDirective {
446
	s := p.cur.Span
447
	alias := &ast.AliasDirective{}
448
	p.expect(token.ALIAS)
449
	p.skipWhitespace()
450
	alias.From = p.parseAccount()
451
	p.skipWhitespace()
452
	p.expect(token.EQ)
453
	p.skipWhitespace()
454
	alias.To = p.parseAccount()
455
	alias.Comment = p.parseOptInlineComment()
456
	p.expectNewline()
457
	alias.Span = p.span(s)
458
	return alias
459
}
460
461
func (p *Parser) parsePayeeDirective() *ast.PayeeDirective {
462
	s := p.cur.Span
463
	p.expect(token.PAYEE)
464
	p.skipWhitespace()
465
466
	name := ""
467
	if p.got(token.TEXT) || p.got(token.STRING) || p.got(token.COMMODITYMARK) {
468
		name = p.parsePayee().Name
469
	}
470
471
	comment := p.parseOptInlineComment()
472
	p.expectNewline()
473
474
	return &ast.PayeeDirective{
475
		Name:    name,
476
		Comment: comment,
477
		Span:    p.span(s),
478
	}
479
}
480
481
func (p *Parser) parseTagDirective() *ast.TagDirective {
482
	s := p.cur.Span
483
	p.expect(token.TAG)
484
	p.skipWhitespace()
485
486
	name := ""
487
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) || p.got(token.STRING) {
488
		name = unquote(p.cur.Literal)
489
		p.advance()
490
	}
491
492
	comment := p.parseOptInlineComment()
493
	p.expectNewline()
494
495
	return &ast.TagDirective{
496
		Name:    name,
497
		Comment: comment,
498
		Span:    p.span(s),
499
	}
500
}
501
502
func (p *Parser) parseYearDirective() *ast.YearDirective {
503
	s := p.cur.Span
504
	year := &ast.YearDirective{}
505
	p.expect(token.YEAR)
506
	p.skipWhitespace()
507
508
	if p.got(token.INT) {
509
		year.Year, _ = strconv.Atoi(p.cur.Literal)
510
		p.defaultYear = year.Year
511
		p.advance()
512
	} else {
513
		p.errorf("expected year, got %s", p.cur.Type)
514
	}
515
516
	year.Comment = p.parseOptInlineComment()
517
	p.expectNewline()
518
	year.Span = p.span(s)
519
520
	return year
521
}
522
523
func (p *Parser) parseDecimalMarkDirective() *ast.DecimalMarkDirective {
524
	s := p.cur.Span
525
	mark := &ast.DecimalMarkDirective{}
526
	p.expect(token.DECIMALMARK)
527
	p.skipWhitespace()
528
529
	mark.Mark = byte('.')
530
	if p.got(token.TEXT) {
531
		if len(p.cur.Literal) > 0 {
532
			mark.Mark = p.cur.Literal[0]
533
		}
534
		p.advance()
535
	}
536
537
	mark.Comment = p.parseOptInlineComment()
538
	p.expectNewline()
539
	mark.Span = p.span(s)
540
	return mark
541
}
542
543
func (p *Parser) parseDefaultCommodityDirective() *ast.DefaultCommodityDirective {
544
	s := p.cur.Span
545
	com := &ast.DefaultCommodityDirective{}
546
	p.expect(token.D)
547
	p.skipWhitespace()
548
	com.Amount = *p.parseAmount()
549
	com.Comment = p.parseOptInlineComment()
550
	p.expectNewline()
551
	com.Span = p.span(s)
552
	return com
553
}
554
555
func (p *Parser) parseConversionDirective() *ast.ConversionDirective {
556
	s := p.cur.Span
557
	cd := &ast.ConversionDirective{}
558
	p.expect(token.C)
559
	p.skipWhitespace()
560
561
	if p.isAmountStart() {
562
		cd.From = *p.parseAmount()
563
	} else {
564
		p.errorf("expected amount, got %s", p.cur.Type)
565
	}
566
567
	p.skipWhitespace()
568
	if p.got(token.EQ) {
569
		p.advance()
570
		p.skipWhitespace()
571
		if p.isAmountStart() {
572
			cd.To = *p.parseAmount()
573
		} else {
574
			p.errorf("expected amount, got %s", p.cur.Type)
575
		}
576
	}
577
578
	cd.Comment = p.parseOptInlineComment()
579
	p.expectNewline()
580
	cd.Span = p.span(s)
581
	return cd
582
}
583
584
func (p *Parser) parseIgnoredDirective() *ast.IgnoredDirective {
585
	s := p.cur.Span
586
	p.expect(token.N)
587
	p.skipWhitespace()
588
589
	id := &ast.IgnoredDirective{}
590
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) {
591
		id.Text = p.cur.Literal
592
		p.advance()
593
	}
594
	id.Comment = p.parseOptInlineComment()
595
596
	p.expectNewline()
597
	id.Span = p.span(s)
598
	return id
599
}
600
601
func (p *Parser) parseMarketPriceDirective() *ast.MarketPriceDirective {
602
	s := p.cur.Span
603
	p.expect(token.P)
604
	p.skipWhitespace()
605
606
	mp := &ast.MarketPriceDirective{}
607
	mp.DateTime.Date = p.parseDate()
608
	p.skipWhitespace()
609
610
	if p.got(token.TIME) {
611
		mp.DateTime.Time = new(p.parseTime())
612
		p.skipWhitespace()
613
	}
614
615
	tok, _ := p.expect(token.COMMODITYMARK)
616
	mp.Commodity = tok.Literal
617
	p.skipWhitespace()
618
619
	mp.Amount = *p.parseAmount()
620
621
	mp.Comment = p.parseOptInlineComment()
622
623
	p.expectNewline()
624
	mp.Span = p.span(s)
625
	return mp
626
}
627
628
func (p *Parser) parseTime() ast.Time {
629
	s := p.cur.Span
630
	tok, _ := p.expect(token.TIME)
631
	lit := tok.Literal
632
633
	parts := strings.Split(lit, ":")
634
	if len(parts) < 2 {
635
		p.errorf("invalid time format: %q", lit)
636
		return ast.Time{Span: p.span(s)}
637
	}
638
639
	hour, _ := strconv.Atoi(parts[0])
640
	minute, _ := strconv.Atoi(parts[1])
641
	second := 0
642
	if len(parts) > 2 {
643
		second, _ = strconv.Atoi(parts[2])
644
	}
645
646
	if hour < 0 || hour > 23 {
647
		p.errorf("invalid hour %d in time %q", hour, lit)
648
	}
649
	if minute < 0 || minute > 59 {
650
		p.errorf("invalid minute %d in time %q", minute, lit)
651
	}
652
	if second < 0 || second > 59 {
653
		p.errorf("invalid second %d in time %q", second, lit)
654
	}
655
656
	return ast.Time{
657
		Hour:   hour,
658
		Minute: minute,
659
		Second: second,
660
		Span:   p.span(s),
661
	}
662
}
663
664
func (p *Parser) parseApplyDirective() *ast.ApplyDirective {
665
	s := p.cur.Span
666
	p.expect(token.APPLY)
667
	p.skipWhitespace()
668
669
	expr := p.parseDirectiveExpr()
670
	comment := p.parseOptInlineComment()
671
	p.expectNewline()
672
673
	return &ast.ApplyDirective{
674
		Expr:    expr,
675
		Comment: comment,
676
		Span:    p.span(s),
677
	}
678
}
679
680
func (p *Parser) parseEndDirective() *ast.EndDirective {
681
	s := p.cur.Span
682
	p.expect(token.END)
683
	p.skipWhitespace()
684
685
	expr := p.parseDirectiveExpr()
686
	comment := p.parseOptInlineComment()
687
	p.expectNewline()
688
689
	return &ast.EndDirective{
690
		Expr:    expr,
691
		Comment: comment,
692
		Span:    p.span(s),
693
	}
694
}
695
696
func (p *Parser) parseCommentBlockDirective() *ast.CommentBlockDirective {
697
	start := p.cur.Span
698
	p.expect(token.COMMENTKW)
699
	p.skipWhitespace()
700
701
	header := p.parseDirectiveExpr()
702
	comment := p.parseOptInlineComment()
703
	p.expectNewline()
704
705
	var content strings.Builder
706
	for p.cur.Type != token.EOF {
707
		if p.got(token.END) {
708
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) {
709
				p.advance()
710
				p.expectNewline()
711
				break
712
			}
713
			if p.willGet(token.WHITESPACE) {
714
				endTok := p.cur
715
				p.advance()
716
				wsTok := p.cur
717
				p.advance()
718
				if p.got(token.TEXT) && p.cur.Literal == "comment" { // todo: this should check if it's an actual COMMENTKW token
719
					p.advance()
720
					p.parseDirectiveExpr()
721
					p.parseOptInlineComment()
722
					p.expectNewline()
723
					break
724
				}
725
				content.WriteString(endTok.Literal)
726
				content.WriteString(wsTok.Literal)
727
				continue
728
			}
729
		}
730
		content.WriteString(p.cur.Literal)
731
		p.advance()
732
	}
733
734
	return &ast.CommentBlockDirective{
735
		Header:  header,
736
		Content: content.String(),
737
		Comment: comment,
738
		Span:    p.span(start),
739
	}
740
}
741
742
func (p *Parser) parseStatus() ast.Status {
743
	s := p.cur.Span
744
	st := ast.Status{}
745
	switch p.cur.Type {
746
	case token.STAR:
747
		st.Value = ast.StatusCleared
748
	case token.BANG:
749
		st.Value = ast.StatusPending
750
	}
751
	if st.Value != ast.StatusNone {
752
		p.advance()
753
		p.skipWhitespace()
754
	}
755
	st.Span = p.span(s)
756
	return st
757
}
758
759
func (p *Parser) isAmountStart() bool {
760
	switch p.cur.Type {
761
	default:
762
		return false
763
	case token.COMMODITYMARK, token.STRING, token.INT, token.DECIMAL, token.MINUS, token.PLUS, token.PARENEXPR, token.STAR:
764
		return true
765
	}
766
}
767
768
func (p *Parser) parseAmount() *ast.Amount {
769
	s := p.cur.Span
770
	amt := &ast.Amount{
771
		QuantityFmt: ast.QuantityFormat{Decimal: '.'},
772
	}
773
	defer func() {
774
		// The span covers from the first token to the start of the next unconsumed token.
775
		// Since parseQuantityInto (and possible commodity consumption) advanced past the last
776
		// amount token, p.cur points to the next token after the amount — which is the correct end.
777
		amt.Span = p.span(s)
778
	}()
779
780
	p.parseAmountSign(amt)
781
	p.skipWhitespace()
782
783
	// commodity before quantity: $10.00, eur 10.00
784
	if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
785
		cs := p.cur.Span
786
		amt.Commodity = unquote(p.cur.Literal)
787
		amt.CommodityPos = ast.CommodityBefore
788
		p.advance()
789
		amt.CommoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
790
		if p.got(token.WHITESPACE) {
791
			amt.HasSpace = true
792
			p.skipWhitespace()
793
		}
794
	}
795
796
	// optional sign after commodity: $ -10
797
	p.parseAmountSign(amt)
798
	p.skipWhitespace()
799
800
	p.parseQuantityInto(amt)
801
802
	// commodity after quantity: 10.00 UAH, 10.00 "EUR" (only if not set)
803
	if amt.Commodity == "" {
804
		switch p.cur.Type {
805
		case token.WHITESPACE:
806
			p.skipWhitespace()
807
			if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
808
				cs := p.cur.Span
809
				amt.HasSpace = true
810
				amt.Commodity = unquote(p.cur.Literal)
811
				amt.CommodityPos = ast.CommodityAfter
812
				p.advance()
813
				amt.CommoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
814
			}
815
		case token.COMMODITYMARK, token.TEXT, token.STRING:
816
			cs := p.cur.Span
817
			amt.Commodity = unquote(p.cur.Literal)
818
			amt.CommodityPos = ast.CommodityAfter
819
			p.advance()
820
			amt.CommoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
821
		}
822
	}
823
824
	return amt
825
}
826
827
// parseAmountSign consumes an optional leading +/- into IsNegative.
828
func (p *Parser) parseAmountSign(amt *ast.Amount) {
829
	switch p.cur.Type {
830
	case token.MINUS:
831
		amt.IsNegative = true
832
		p.advance()
833
	case token.PLUS:
834
		p.advance()
835
	}
836
}
837
838
func (p *Parser) parseAmountWithOptExpr() *ast.Amount {
839
	if p.got(token.STAR) {
840
		p.advance()
841
		p.skipWhitespace()
842
		amt := p.parseAmount()
843
		if amt != nil {
844
			amt.IsExpr = true
845
		}
846
		return amt
847
	}
848
	if p.got(token.PARENEXPR) {
849
		lit := p.cur.Literal
850
		amt := &ast.Amount{
851
			IsExpr:      true,
852
			QuantityFmt: ast.QuantityFormat{Decimal: '.'},
853
		}
854
		if len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
855
			amt.Expr = strings.Trim(lit[1:len(lit)-1], " \t")
856
		}
857
		amt.Span = p.cur.Span
858
		p.advance()
859
		return amt
860
	}
861
	return p.parseAmount()
862
}
863
864
func (p *Parser) parsePosting() *ast.Posting {
865
	s := p.cur.Span
866
	posting := &ast.Posting{}
867
	p.expect(token.INDENT)
868
869
	// exit if it's empty line
870
	if p.got(token.NEWLINE) || p.got(token.EOF) {
871
		p.syncToNextline()
872
		return nil
873
	}
874
875
	// optional status, outside of brackets, '! (account)'
876
	posting.Status = p.parseStatus()
877
878
	// detect virtual posting brackets
879
	switch p.cur.Type {
880
	case token.LPAREN:
881
		posting.Type = ast.PostingVirtualUnbalanced
882
		p.advance()
883
	case token.LBRACKET:
884
		posting.Type = ast.PostingVirtualBalanced
885
		p.advance()
886
	}
887
888
	// optional status, inside of brackets, '(* account)'
889
	if p.got(token.STAR) || p.got(token.BANG) {
890
		posting.Status = p.parseStatus()
891
	}
892
893
	// validate, must be account text
894
	if p.cur.Type != token.TEXT {
895
		p.errorf("expected account name, got %s", p.cur.Type)
896
		p.syncToNextline()
897
		return nil
898
	}
899
900
	posting.Account = p.parseAccount()
901
902
	// consume closing bracket
903
	switch p.cur.Type {
904
	case token.RPAREN:
905
		p.advance()
906
	case token.RBRACKET:
907
		p.advance()
908
	}
909
910
	// optional amount - after two spaces
911
	if p.got(token.WHITESPACE) {
912
		p.skipWhitespace()
913
		if p.isAmountStart() {
914
			posting.Amount = p.parseAmountWithOptExpr()
915
		}
916
	}
917
918
	// optional cost '@' or '@@'
919
	p.skipWhitespace()
920
	if p.got(token.AT) || p.got(token.ATAT) {
921
		posting.Cost = p.parseCost()
922
	}
923
924
	// optional balance assertion or assignment
925
	p.skipWhitespace()
926
	if p.got(token.COLON) && p.willGet(token.EQ) {
927
		p.advance() // consume ':' of ':='
928
		posting.Balance = p.parseBalanceAssertion()
929
		posting.Balance.IsAssignment = true
930
	} else if p.got(token.EQ) || p.got(token.EQEQ) || p.got(token.EQEQEQ) || p.got(token.EQSTAR) {
931
		posting.Balance = p.parseBalanceAssertion()
932
	}
933
934
	posting.Comment = p.parseOptInlineComment()
935
	p.expectNewline()
936
937
	// continuation comments
938
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
939
		p.advance()
940
		c := p.parseComment()
941
		posting.Comments = append(posting.Comments, *c)
942
	}
943
944
	posting.Span = p.span(s)
945
	return posting
946
}
947
948
func (p *Parser) parseCost() *ast.Cost {
949
	s := p.cur.Span
950
	isTotal := p.got(token.ATAT)
951
	p.advance() // consume '@' '@@'
952
	p.skipWhitespace()
953
	return &ast.Cost{
954
		IsTotal: isTotal,
955
		Amount:  *p.parseAmount(),
956
		Span:    p.span(s),
957
	}
958
}
959
960
func (p *Parser) parseBalanceAssertion() *ast.BalanceAssertion {
961
	s := p.cur.Span
962
963
	ba := &ast.BalanceAssertion{}
964
	switch p.cur.Type {
965
	case token.EQ: // basic assertion
966
	case token.EQSTAR: // inclusive assertion
967
		ba.IsInclusive = true
968
	case token.EQEQ: // strict assertion
969
		ba.IsStrict = true
970
	case token.EQEQEQ: // strict inclusive assertion
971
		ba.IsStrict = true
972
		ba.IsInclusive = true
973
	}
974
	p.advance()
975
	p.skipWhitespace()
976
977
	ba.Amount = *p.parseAmount()
978
	p.skipWhitespace()
979
	if p.got(token.AT) || p.got(token.ATAT) {
980
		c := p.parseCost()
981
		ba.Cost = c
982
	}
983
	ba.Span = p.span(s)
984
	return ba
985
}
986
987
func (p *Parser) readAccountSegment() (ast.SubAccount, bool) {
988
	switch p.cur.Type {
989
	case token.TEXT:
990
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
991
		p.advance()
992
993
		// handle multi work segment, e.g: "credit card"
994
		if p.got(token.WHITESPACE) && p.willGet(token.TEXT) && len(p.peek.Literal) > 0 && p.peek.Literal[0] != '(' {
995
			sub.Name += " "
996
			p.advance()
997
			sub.Name += p.cur.Literal
998
			p.advance()
999
		}
1000
		return sub, true
1001
1002
	case token.COMMODITYMARK:
1003
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
1004
		p.advance()
1005
		// merge "EUR" + "-HRK" to "EUR-HRK"
1006
		for p.got(token.TEXT) {
1007
			sub.Name += p.cur.Literal
1008
			p.advance()
1009
		}
1010
		return sub, true
1011
1012
	default:
1013
		return ast.SubAccount{}, false
1014
	}
1015
}
1016
1017
func (p *Parser) parseAccount() ast.Account {
1018
	s := p.cur.Span
1019
	acc := ast.Account{}
1020
1021
	sub, ok := p.readAccountSegment()
1022
	if !ok {
1023
		p.errorf("expected account, got %s", p.cur.Type)
1024
		return ast.Account{}
1025
	}
1026
	acc.Name = append(acc.Name, sub)
1027
1028
	for p.got(token.COLON) {
1029
		p.advance()
1030
		sub, ok := p.readAccountSegment()
1031
		if !ok {
1032
			break
1033
		}
1034
		acc.Name = append(acc.Name, sub)
1035
	}
1036
1037
	acc.Span = p.span(s)
1038
	return acc
1039
}
1040
1041
func (p *Parser) parseDate() ast.Date {
1042
	s := p.cur.Span
1043
	tok, ok := p.expect(token.DATE)
1044
	if !ok {
1045
		return ast.Date{Span: p.span(s)}
1046
	}
1047
1048
	year, month, day, sep, err := parseDateLiteral(tok.Literal)
1049
	if err != nil {
1050
		p.errorf("%v", err)
1051
		return ast.Date{Span: p.span(s)}
1052
	}
1053
	if year == 0 {
1054
		year = p.defaultYear
1055
	}
1056
1057
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep, Span: p.span(s)}
1058
}
1059
1060
func (p *Parser) parseOptInlineComment() *ast.Comment {
1061
	p.skipWhitespace()
1062
	if !p.got(token.SEMICOLON) {
1063
		return nil
1064
	}
1065
	return p.parseCommentRest(p.cur.Span)
1066
}
1067
1068
// parseCommentRest consumes a comment marker at p.cur, then optional text;
1069
// s anchors the span at the marker's start.
1070
func (p *Parser) parseCommentRest(s token.Span) *ast.Comment {
1071
	marker := p.cur.Literal[0]
1072
	p.advance()
1073
	p.skipWhitespace()
1074
1075
	text := ""
1076
	if p.got(token.TEXT) {
1077
		text = p.cur.Literal
1078
		p.advance()
1079
	}
1080
1081
	return &ast.Comment{
1082
		Marker: marker,
1083
		Text:   text,
1084
		Span:   p.span(s),
1085
	}
1086
}
1087
1088
func (p *Parser) parseOptPeriodicDescription() (string, token.Span) {
1089
	if p.cur.Type != token.WHITESPACE || len(p.cur.Literal) < 2 {
1090
		return "", token.Span{}
1091
	}
1092
1093
	p.skipWhitespace()
1094
1095
	if p.cur.Type != token.TEXT {
1096
		return "", token.Span{}
1097
	}
1098
1099
	s := p.cur.Span
1100
	desc := p.parseDescription()
1101
	return desc, p.span(s)
1102
}
1103
1104
func (p *Parser) parseDescription() string {
1105
	var desc strings.Builder
1106
	for p.got(token.TEXT) || (p.got(token.WHITESPACE) && p.willGet(token.TEXT)) {
1107
		_, _ = desc.WriteString(p.cur.Literal)
1108
		p.advance()
1109
	}
1110
	return desc.String()
1111
}
1112
1113
func (p *Parser) parseDirectiveExpr() string {
1114
	var b strings.Builder
1115
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF && p.cur.Type != token.SEMICOLON {
1116
		_, _ = b.WriteString(p.cur.Literal)
1117
		p.advance()
1118
	}
1119
	return b.String()
1120
}
1121
1122
func (p *Parser) parseQuantityInto(amt *ast.Amount) {
1123
	if p.cur.Type != token.INT && p.cur.Type != token.DECIMAL && p.cur.Type != token.TEXT {
1124
		p.errorf("expected quantity, got %s", p.cur.Type)
1125
		return
1126
	}
1127
1128
	lit := p.cur.Literal
1129
	p.advance()
1130
1131
	// detect format metadata before normalizing
1132
	amt.QuantityFmt = detectFormat(lit)
1133
1134
	// normalize for decimal.NewFromString
1135
	// remove thousands separators, replace decimal mark with '.'
1136
	normalized := normalizeLiteral(lit, amt.QuantityFmt.Thousands, amt.QuantityFmt.Decimal)
1137
1138
	q, err := decimal.FromString(normalized)
1139
	if err != nil {
1140
		p.errorf("invalid quantity %q: %v", lit, err)
1141
		return
1142
	}
1143
1144
	if amt.IsNegative {
1145
		q = q.Neg()
1146
	}
1147
	amt.Quantity = q
1148
}
1149
1150
func (p *Parser) parseBlankLine() *ast.BlankLine {
1151
	s := p.cur.Span
1152
	p.expectNewline()
1153
	return &ast.BlankLine{Span: s}
1154
}
1155
1156
func (p *Parser) expectNewline() {
1157
	if p.got(token.NEWLINE) || p.got(token.EOF) {
1158
		if p.got(token.NEWLINE) {
1159
			p.advance()
1160
		}
1161
		return
1162
	}
1163
	p.errorf("expected %s, got %s", token.NEWLINE, p.cur.Type)
1164
}
1165
1166
func (p *Parser) advance() token.Token {
1167
	prev := p.cur
1168
	p.cur = p.peek
1169
	p.peek = p.lexer.Next()
1170
	return prev
1171
}
1172
1173
func (p *Parser) got(kind token.Type) bool     { return p.cur.Type == kind }
1174
func (p *Parser) willGet(kind token.Type) bool { return p.peek.Type == kind }
1175
1176
func (p *Parser) expect(kind token.Type) (token.Token, bool) {
1177
	if p.got(kind) {
1178
		return p.advance(), true
1179
	}
1180
	p.errorf("expected %s, got %s", kind, p.cur.Type)
1181
	return p.cur, false
1182
}
1183
1184
func (p *Parser) errorf(format string, args ...any) {
1185
	p.errors = append(p.errors, &ast.ParseError{
1186
		Span:    p.cur.Span,
1187
		Message: fmt.Sprintf(format, args...),
1188
	})
1189
}
1190
1191
func isDirectiveKeyword(t token.Type) bool {
1192
	switch t {
1193
	case token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE,
1194
		token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END,
1195
		token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C:
1196
		return true
1197
	}
1198
	return false
1199
}
1200
1201
func (p *Parser) sync() {
1202
	for {
1203
		switch p.cur.Type {
1204
		case token.EOF:
1205
			return
1206
		case token.NEWLINE:
1207
			p.advance()
1208
			t := p.cur.Type
1209
			if isDirectiveKeyword(t) || t == token.DATE || t == token.TILDE || t == token.EQ {
1210
				return
1211
			}
1212
		default:
1213
			p.advance()
1214
		}
1215
	}
1216
}
1217
1218
func (p *Parser) syncToNextline() {
1219
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF {
1220
		p.advance()
1221
	}
1222
	if p.got(token.NEWLINE) {
1223
		p.advance()
1224
	}
1225
}
1226
1227
func (p *Parser) skipWhitespace() {
1228
	for p.got(token.WHITESPACE) {
1229
		p.advance()
1230
	}
1231
}
1232
1233
func (p *Parser) span(s token.Span) token.Span {
1234
	return token.Span{Start: s.Start, End: p.cur.Span.Start}
1235
}
1236
1237
func normalizeLiteral(lit string, thousands, decimal byte) string {
1238
	var b strings.Builder
1239
	for _, ch := range []byte(lit) {
1240
		if thousands != 0 && ch == thousands {
1241
			continue // skip thousands separator
1242
		}
1243
		if ch == decimal {
1244
			b.WriteByte('.')
1245
		} else {
1246
			b.WriteByte(ch)
1247
		}
1248
	}
1249
	return b.String()
1250
}
1251
1252
func detectFormat(lit string) ast.QuantityFormat {
1253
	var seps []int
1254
	for i, ch := range []byte(lit) {
1255
		if ch == '.' || ch == ',' || ch == ' ' || ch == '_' || ch == '\'' {
1256
			seps = append(seps, i)
1257
		}
1258
	}
1259
1260
	if len(seps) == 0 {
1261
		return ast.QuantityFormat{Decimal: '.', Thousands: 0, Precision: 0}
1262
	}
1263
1264
	last := seps[len(seps)-1]
1265
	dec := lit[last]
1266
	var thou byte
1267
	if len(seps) > 1 {
1268
		thou = lit[seps[0]]
1269
	} else if dec == ' ' || dec == '_' || dec == '\'' {
1270
		// single space/underscore/apostrophe is always thousands
1271
		thou = dec
1272
		dec = '.'
1273
	}
1274
1275
	// calculate precision when the last separator is a real decimal
1276
	prec := 0
1277
	if thou == 0 || len(seps) > 1 {
1278
		prec = len(lit) - last - 1
1279
	}
1280
1281
	return ast.QuantityFormat{Decimal: dec, Thousands: thou, Precision: prec}
1282
}
1283
1284
// parseSimpleDate  parses full YYYY/MM/DD date literal embedded in free text.
1285
func parseSimpleDate(s string) ast.Date {
1286
	year, month, day, sep, err := parseDateLiteral(s)
1287
	if err != nil {
1288
		return ast.Date{}
1289
	}
1290
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep}
1291
}
1292
1293
// parseDateLiteral parses and validates a date literal.
1294
func parseDateLiteral(lit string) (year, month, day int, sep byte, err error) {
1295
	sep = dateSeparator(lit)
1296
	if sep == 0 {
1297
		return 0, 0, 0, 0, fmt.Errorf("invalid date format: %q", lit)
1298
	}
1299
1300
	parts := strings.Split(lit, string(sep))
1301
	if len(parts) != 2 && len(parts) != 3 {
1302
		return 0, 0, 0, 0, fmt.Errorf("invalid date format: %q", lit)
1303
	}
1304
1305
	nums := make([]int, len(parts))
1306
	for i, part := range parts {
1307
		if nums[i], err = strconv.Atoi(part); err != nil {
1308
			return 0, 0, 0, 0, fmt.Errorf("invalid date literal: %q", lit)
1309
		}
1310
	}
1311
1312
	month = nums[len(parts)-2]
1313
	if month < 1 || month > 12 {
1314
		return 0, 0, 0, 0, fmt.Errorf("invalid month %d in %q", month, lit)
1315
	}
1316
1317
	day = nums[len(parts)-1]
1318
	if day < 1 || day > 31 {
1319
		return 0, 0, 0, 0, fmt.Errorf("invalid day %d in %q", day, lit)
1320
	}
1321
1322
	if len(parts) == 2 {
1323
		return 0, month, day, sep, nil
1324
	}
1325
	return nums[0], month, day, sep, nil
1326
}
1327
1328
func dateSeparator(lit string) byte {
1329
	for i := 0; i < len(lit); i++ {
1330
		if lit[i] == '/' || lit[i] == '-' || lit[i] == '.' {
1331
			return lit[i]
1332
		}
1333
	}
1334
	return 0
1335
}