all repos

clerk @ 2a7dd08

missing tooling for ledger/hledger

clerk/journal/parser/parser.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
lsp: go to definition, 1 month ago
1
package parser
2
3
import (
4
	"fmt"
5
	"strconv"
6
	"strings"
7
	"unicode"
8
	"unicode/utf8"
9
10
	"olexsmir.xyz/clerk/internal/decimal"
11
	"olexsmir.xyz/clerk/journal/ast"
12
	"olexsmir.xyz/clerk/journal/lexer"
13
	"olexsmir.xyz/clerk/journal/token"
14
)
15
16
type Parser struct {
17
	lexer  *lexer.Lexer
18
	errors []*ast.ParseError
19
	cur    token.Token
20
	peek   token.Token
21
22
	defaultYear int // set by year directive, used for short date inference
23
}
24
25
func New(lex *lexer.Lexer) *Parser {
26
	p := &Parser{lexer: lex}
27
	p.advance() // populate .peek
28
	p.advance() // populate .cur
29
	return p
30
}
31
32
func NewWithYear(lex *lexer.Lexer, year int) *Parser {
33
	p := &Parser{lexer: lex, defaultYear: year}
34
	p.advance() // populate .peek
35
	p.advance() // populate .cur
36
	return p
37
}
38
39
func (p *Parser) ParseJournal() *ast.Journal {
40
	f := &ast.Journal{}
41
	for p.cur.Type != token.EOF {
42
		if e := p.parseEntry(); e != nil {
43
			f.Entries = append(f.Entries, e)
44
		}
45
	}
46
	f.Errors = p.errors
47
	return f
48
}
49
50
func (p *Parser) parseEntry() ast.Entry {
51
	if p.got(token.BANG) || p.got(token.AT) {
52
		if isDirectiveKeyword(p.peek.Type) {
53
			p.advance() // consume prefix
54
		}
55
	}
56
57
	switch p.cur.Type {
58
	case token.ILLEGAL:
59
		p.errorf("illegal character %q", p.cur.Literal)
60
		p.advance()
61
		return nil
62
	case token.INDENT:
63
		p.errorf("unexpected indent")
64
		p.syncToNextline()
65
		return nil
66
	case token.DATE:
67
		return p.parseTransaction()
68
	case token.TILDE:
69
		return p.parsePeriodicTransaction()
70
	case token.EQ:
71
		return p.parseAutomatedTransaction()
72
	case token.NEWLINE:
73
		return p.parseBlankLine()
74
	case token.SEMICOLON, token.HASH, token.PERCENT, token.STAR:
75
		return p.parseComment()
76
	case token.ACCOUNT:
77
		return p.parseAccountDirective()
78
	case token.COMMODITY:
79
		return p.parseCommodityDirective()
80
	case token.INCLUDE:
81
		return p.parseIncludeDirective()
82
	case token.ALIAS:
83
		return p.parseAliasDirective()
84
	case token.PAYEE:
85
		return p.parsePayeeDirective()
86
	case token.TAG:
87
		return p.parseTagDirective()
88
	case token.YEAR:
89
		return p.parseYearDirective()
90
	case token.DECIMALMARK:
91
		return p.parseDecimalMarkDirective()
92
	case token.D:
93
		return p.parseDefaultCommodityDirective()
94
	case token.P:
95
		return p.parseMarketPriceDirective()
96
	case token.N:
97
		return p.parseIgnoredDirective()
98
	case token.C:
99
		return p.parseConversionDirective()
100
	case token.APPLY:
101
		return p.parseApplyDirective()
102
	case token.END:
103
		return p.parseEndDirective()
104
	case token.COMMENTKW:
105
		return p.parseCommentBlockDirective()
106
	default:
107
		p.errorf("unexpected token %s", p.cur.Type)
108
		p.sync()
109
		return nil
110
	}
111
}
112
113
func (p *Parser) parseTransaction() *ast.Transaction {
114
	s := p.cur.Span
115
	tx := &ast.Transaction{}
116
117
	tx.Date = p.parseDate()
118
119
	p.skipWhitespace()
120
121
	// optional secondary date
122
	if p.got(token.EQ) {
123
		p.advance()
124
		p.skipWhitespace()
125
		d := p.parseDate()
126
		tx.SecondDate = &d
127
	}
128
129
	p.skipWhitespace()
130
131
	// optional status
132
	tx.Status = p.parseStatus()
133
134
	// optional code - the lexer emits "(CODE)" as a single TEXT token; split it here
135
	if p.got(token.TEXT) {
136
		if lit := p.cur.Literal; len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
137
			tx.Code = &ast.Code{Value: lit[1 : len(lit)-1], Span: p.cur.Span}
138
			p.advance()
139
			p.skipWhitespace()
140
		}
141
	}
142
143
	// optional payee | note
144
	if p.got(token.TEXT) || p.got(token.STRING) {
145
		tx.Payee = p.parsePayee()
146
147
		// check for | separator
148
		p.skipWhitespace()
149
150
		if p.got(token.PIPE) {
151
			p.advance()
152
			if p.got(token.TEXT) {
153
				sn := p.cur.Span
154
				n := p.cur.Literal
155
				p.advance()
156
				tx.Note = &ast.Note{Value: n, Span: p.span(sn)}
157
			}
158
		}
159
	}
160
161
	tx.Comment = p.parseOptInlineComment()
162
	p.expectNewline()
163
164
	tx.HeaderComments, tx.Postings = p.parseHeaderCommentsAndPostings()
165
166
	tx.Span = p.span(s)
167
	return tx
168
}
169
170
func unquote(s string) string {
171
	if len(s) >= 2 && ((s[0] == '"' && s[len(s)-1] == '"') || (s[0] == '\'' && s[len(s)-1] == '\'')) {
172
		return s[1 : len(s)-1]
173
	}
174
	return s
175
}
176
177
func (p *Parser) parsePayee() *ast.Payee {
178
	s := p.cur.Span
179
180
	if p.got(token.STRING) {
181
		name := unquote(p.cur.Literal)
182
		p.advance()
183
		return &ast.Payee{Name: name, Span: p.span(s)}
184
	}
185
186
	// keep spaces/tags between text tokens; stop before trailing whitespace
187
	var name strings.Builder
188
	for payeeWord(p.cur.Type) || (payeeWord(p.peek.Type) && p.got(token.WHITESPACE)) {
189
		_, _ = name.WriteString(p.cur.Literal)
190
		p.advance()
191
	}
192
	return &ast.Payee{Name: unquote(name.String()), Span: p.span(s)}
193
}
194
195
func payeeWord(t token.Type) bool {
196
	switch t {
197
	case token.TEXT, token.INT, token.DECIMAL, token.COMMODITYMARK:
198
		return true
199
	}
200
	return false
201
}
202
203
func (p *Parser) parsePeriodicTransaction() *ast.PeriodicTransaction {
204
	s := p.cur.Span
205
	p.expect(token.TILDE)
206
	p.skipWhitespace()
207
208
	pt := &ast.PeriodicTransaction{}
209
210
	pt.Period = p.parsePeriod()
211
212
	if desc, dspan := p.parseOptPeriodicDescription(); desc != "" {
213
		pt.Description = &ast.Description{Value: desc, Span: dspan}
214
	}
215
216
	comment := p.parseOptInlineComment()
217
	p.expectNewline()
218
219
	pt.HeaderComments, pt.Postings = p.parseHeaderCommentsAndPostings()
220
221
	pt.Span = p.span(s)
222
	pt.Comment = comment
223
	return pt
224
}
225
226
func (p *Parser) parseAutomatedTransaction() *ast.AutomatedTransaction {
227
	s := p.cur.Span
228
	p.expect(token.EQ)
229
	p.skipWhitespace()
230
231
	at := &ast.AutomatedTransaction{}
232
233
	// expression
234
	sd := p.cur.Span
235
	expr := p.parseDirectiveExpr()
236
	at.Expr = ast.Expr{Value: expr, Span: p.span(sd)}
237
	at.Comment = p.parseOptInlineComment()
238
	p.expectNewline()
239
240
	at.HeaderComments, at.Postings = p.parseHeaderCommentsAndPostings()
241
242
	at.Span = p.span(s)
243
	return at
244
}
245
246
func (p *Parser) parseHeaderCommentsAndPostings() (comments []*ast.Comment, postings []*ast.Posting) {
247
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
248
		p.advance() // consume indent
249
		comments = append(comments, p.parseComment())
250
	}
251
252
	for p.got(token.INDENT) {
253
		if posting := p.parsePosting(); posting != nil {
254
			postings = append(postings, posting)
255
		}
256
	}
257
258
	return comments, postings
259
}
260
261
func (p *Parser) parsePeriod() ast.Period {
262
	s := p.cur.Span
263
264
	var periodBuf strings.Builder
265
266
	for !p.got(token.NEWLINE) && !p.got(token.EOF) &&
267
		!p.got(token.SEMICOLON) && !p.got(token.HASH) && !p.got(token.PERCENT) && !p.got(token.STAR) {
268
269
		if p.got(token.WHITESPACE) {
270
			if len(p.cur.Literal) >= 2 {
271
				break
272
			}
273
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) ||
274
				p.willGet(token.SEMICOLON) || p.willGet(token.HASH) ||
275
				p.willGet(token.PERCENT) || p.willGet(token.STAR) {
276
				p.advance()
277
				continue
278
			}
279
		}
280
281
		periodBuf.WriteString(p.cur.Literal)
282
		p.advance()
283
	}
284
285
	str := periodBuf.String()
286
	period := ast.Period{Raw: str, Span: p.span(s)}
287
288
	if _, after, ok := strings.Cut(str, " from "); ok {
289
		end := strings.Index(after, " ")
290
		dateStr := after
291
		if end >= 0 {
292
			dateStr = after[:end]
293
		}
294
		if d := parseSimpleDate(dateStr); d.Year > 0 {
295
			fromOff := strings.Index(str, dateStr)
296
			d.Span = periodDateSpan(period, str, dateStr, fromOff)
297
			period.From = &d
298
			rest := after
299
			if end >= 0 {
300
				rest = after[end:]
301
			}
302
			if _, toAfter, ok := strings.Cut(rest, " to "); ok {
303
				if toEnd := strings.Index(toAfter, " "); toEnd >= 0 {
304
					toAfter = toAfter[:toEnd]
305
				}
306
				if d := parseSimpleDate(toAfter); d.Year > 0 {
307
					d.Span = periodDateSpan(period, str, toAfter, fromOff+len(dateStr))
308
					period.To = &d
309
				}
310
			}
311
		}
312
	}
313
	return period
314
}
315
316
// periodDateSpan returns the source span of dateStr, which occurs in the
317
// period text at or after searchFrom. The period span and text cover the same
318
// bytes, so offsets line up 1:1.
319
func periodDateSpan(period ast.Period, text, dateStr string, searchFrom int) token.Span {
320
	off := strings.Index(text[searchFrom:], dateStr)
321
	abs := period.Span.Start.Offset + searchFrom + off
322
	return token.Span{
323
		Start: token.Pos{File: period.Span.Start.File, Offset: abs},
324
		End:   token.Pos{File: period.Span.Start.File, Offset: abs + len(dateStr)},
325
	}
326
}
327
328
func (p *Parser) parseComment() *ast.Comment {
329
	s := p.cur.Span
330
	c := p.parseCommentRest(s)
331
	p.expectNewline()
332
	c.Span = p.span(s) // comment spans its line through the newline
333
	return c
334
}
335
336
func (p *Parser) parseAccountDirective() *ast.AccountDirective {
337
	s := p.cur.Span
338
	p.expect(token.ACCOUNT)
339
	p.skipWhitespace()
340
341
	account := p.parseAccount()
342
	comment := p.parseOptInlineComment()
343
	p.expectNewline()
344
345
	for p.got(token.INDENT) {
346
		p.advance()
347
		for !p.got(token.NEWLINE) && !p.got(token.EOF) {
348
			p.advance()
349
		}
350
		p.expectNewline()
351
	}
352
353
	return &ast.AccountDirective{
354
		Account: account,
355
		Comment: comment,
356
		Span:    p.span(s),
357
	}
358
}
359
360
func (p *Parser) parseCommodityDirective() *ast.CommodityDirective {
361
	s := p.cur.Span
362
	p.expect(token.COMMODITY)
363
	p.skipWhitespace()
364
365
	var commodity string
366
	var commoditySpan token.Span
367
	var format *ast.Amount
368
369
	switch p.cur.Type {
370
	case token.COMMODITYMARK, token.TEXT, token.STRING:
371
		cs := p.cur.Span
372
		commodity = unquote(p.cur.Literal)
373
		p.advance()
374
		commoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
375
		hadSpace := p.got(token.WHITESPACE)
376
		p.skipWhitespace()
377
		if p.got(token.INT) || p.got(token.DECIMAL) || p.got(token.TEXT) {
378
			format = p.parseAmount()
379
			format.Commodity = commodity
380
			format.CommoditySpan = commoditySpan
381
			format.CommodityPos = ast.CommodityBefore
382
			format.HasSpace = hadSpace
383
		}
384
	case token.INT, token.DECIMAL:
385
		format = p.parseAmount()
386
		commodity = format.Commodity
387
		commoditySpan = format.CommoditySpan
388
	default:
389
		p.errorf("expected commodity name or amount, got %s", p.cur.Type)
390
	}
391
392
	if commodity == "" {
393
		p.errorf("expected commodity name, got %s", p.cur.Type)
394
	}
395
396
	comment := p.parseOptInlineComment()
397
	p.expectNewline()
398
399
	for p.got(token.INDENT) {
400
		p.advance()
401
		p.skipWhitespace()
402
		if p.got(token.TEXT) && p.cur.Literal == "format" {
403
			p.advance()
404
			p.skipWhitespace()
405
			format = p.parseAmount()
406
			p.expectNewline()
407
			continue
408
		}
409
		for !p.got(token.NEWLINE) && !p.got(token.EOF) {
410
			p.advance()
411
		}
412
		p.expectNewline()
413
	}
414
415
	cd := &ast.CommodityDirective{
416
		Commodity:     commodity,
417
		CommoditySpan: commoditySpan,
418
		Comment:       comment,
419
		Span:          p.span(s),
420
	}
421
	if format != nil {
422
		cd.Format = *format
423
	}
424
	return cd
425
}
426
427
func (p *Parser) parseIncludeDirective() *ast.IncludeDirective {
428
	s := p.cur.Span
429
	p.expect(token.INCLUDE)
430
	p.skipWhitespace()
431
432
	id := &ast.IncludeDirective{}
433
434
	if p.got(token.TEXT) {
435
		id.Path = p.cur.Literal
436
		p.advance()
437
	} else {
438
		p.errorf("expected file path, got %s", p.cur.Type)
439
	}
440
441
	id.Comment = p.parseOptInlineComment()
442
	p.expectNewline()
443
	id.Span = p.span(s)
444
	return id
445
}
446
447
func (p *Parser) parseAliasDirective() *ast.AliasDirective {
448
	s := p.cur.Span
449
	alias := &ast.AliasDirective{}
450
	p.expect(token.ALIAS)
451
	p.skipWhitespace()
452
	alias.From = p.parseAccount()
453
	p.skipWhitespace()
454
	p.expect(token.EQ)
455
	p.skipWhitespace()
456
	alias.To = p.parseAccount()
457
	alias.Comment = p.parseOptInlineComment()
458
	p.expectNewline()
459
	alias.Span = p.span(s)
460
	return alias
461
}
462
463
func (p *Parser) parsePayeeDirective() *ast.PayeeDirective {
464
	s := p.cur.Span
465
	p.expect(token.PAYEE)
466
	p.skipWhitespace()
467
468
	var name *ast.Payee
469
	if p.got(token.TEXT) || p.got(token.STRING) || p.got(token.COMMODITYMARK) {
470
		name = p.parsePayee()
471
	}
472
473
	comment := p.parseOptInlineComment()
474
	p.expectNewline()
475
476
	return &ast.PayeeDirective{
477
		Name:    name,
478
		Comment: comment,
479
		Span:    p.span(s),
480
	}
481
}
482
483
func (p *Parser) parseTagDirective() *ast.TagDirective {
484
	s := p.cur.Span
485
	p.expect(token.TAG)
486
	p.skipWhitespace()
487
488
	name := ""
489
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) || p.got(token.STRING) {
490
		name = unquote(p.cur.Literal)
491
		p.advance()
492
	}
493
494
	comment := p.parseOptInlineComment()
495
	p.expectNewline()
496
497
	return &ast.TagDirective{
498
		Name:    name,
499
		Comment: comment,
500
		Span:    p.span(s),
501
	}
502
}
503
504
func (p *Parser) parseYearDirective() *ast.YearDirective {
505
	s := p.cur.Span
506
	year := &ast.YearDirective{}
507
	p.expect(token.YEAR)
508
	p.skipWhitespace()
509
510
	if p.got(token.INT) {
511
		year.Year, _ = strconv.Atoi(p.cur.Literal)
512
		p.defaultYear = year.Year
513
		p.advance()
514
	} else {
515
		p.errorf("expected year, got %s", p.cur.Type)
516
	}
517
518
	year.Comment = p.parseOptInlineComment()
519
	p.expectNewline()
520
	year.Span = p.span(s)
521
522
	return year
523
}
524
525
func (p *Parser) parseDecimalMarkDirective() *ast.DecimalMarkDirective {
526
	s := p.cur.Span
527
	mark := &ast.DecimalMarkDirective{}
528
	p.expect(token.DECIMALMARK)
529
	p.skipWhitespace()
530
531
	mark.Mark = byte('.')
532
	if p.got(token.TEXT) {
533
		if len(p.cur.Literal) > 0 {
534
			mark.Mark = p.cur.Literal[0]
535
		}
536
		p.advance()
537
	}
538
539
	mark.Comment = p.parseOptInlineComment()
540
	p.expectNewline()
541
	mark.Span = p.span(s)
542
	return mark
543
}
544
545
func (p *Parser) parseDefaultCommodityDirective() *ast.DefaultCommodityDirective {
546
	s := p.cur.Span
547
	com := &ast.DefaultCommodityDirective{}
548
	p.expect(token.D)
549
	p.skipWhitespace()
550
	com.Amount = *p.parseAmount()
551
	com.Comment = p.parseOptInlineComment()
552
	p.expectNewline()
553
	com.Span = p.span(s)
554
	return com
555
}
556
557
func (p *Parser) parseConversionDirective() *ast.ConversionDirective {
558
	s := p.cur.Span
559
	cd := &ast.ConversionDirective{}
560
	p.expect(token.C)
561
	p.skipWhitespace()
562
563
	if p.isAmountStart() {
564
		cd.From = *p.parseAmount()
565
	} else {
566
		p.errorf("expected amount, got %s", p.cur.Type)
567
	}
568
569
	p.skipWhitespace()
570
	if p.got(token.EQ) {
571
		p.advance()
572
		p.skipWhitespace()
573
		if p.isAmountStart() {
574
			cd.To = *p.parseAmount()
575
		} else {
576
			p.errorf("expected amount, got %s", p.cur.Type)
577
		}
578
	}
579
580
	cd.Comment = p.parseOptInlineComment()
581
	p.expectNewline()
582
	cd.Span = p.span(s)
583
	return cd
584
}
585
586
func (p *Parser) parseIgnoredDirective() *ast.IgnoredDirective {
587
	s := p.cur.Span
588
	p.expect(token.N)
589
	p.skipWhitespace()
590
591
	id := &ast.IgnoredDirective{}
592
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) {
593
		id.Text = p.cur.Literal
594
		p.advance()
595
	}
596
	id.Comment = p.parseOptInlineComment()
597
598
	p.expectNewline()
599
	id.Span = p.span(s)
600
	return id
601
}
602
603
func (p *Parser) parseMarketPriceDirective() *ast.MarketPriceDirective {
604
	s := p.cur.Span
605
	p.expect(token.P)
606
	p.skipWhitespace()
607
608
	mp := &ast.MarketPriceDirective{}
609
	mp.DateTime.Date = p.parseDate()
610
	p.skipWhitespace()
611
612
	if p.got(token.TIME) {
613
		mp.DateTime.Time = new(p.parseTime())
614
		p.skipWhitespace()
615
	}
616
617
	tok, _ := p.expect(token.COMMODITYMARK)
618
	mp.Commodity = tok.Literal
619
	p.skipWhitespace()
620
621
	mp.Amount = *p.parseAmount()
622
623
	mp.Comment = p.parseOptInlineComment()
624
625
	p.expectNewline()
626
	mp.Span = p.span(s)
627
	return mp
628
}
629
630
func (p *Parser) parseTime() ast.Time {
631
	s := p.cur.Span
632
	tok, _ := p.expect(token.TIME)
633
	lit := tok.Literal
634
635
	parts := strings.Split(lit, ":")
636
	if len(parts) < 2 {
637
		p.errorf("invalid time format: %q", lit)
638
		return ast.Time{Span: p.span(s)}
639
	}
640
641
	hour, _ := strconv.Atoi(parts[0])
642
	minute, _ := strconv.Atoi(parts[1])
643
	second := 0
644
	if len(parts) > 2 {
645
		second, _ = strconv.Atoi(parts[2])
646
	}
647
648
	if hour < 0 || hour > 23 {
649
		p.errorf("invalid hour %d in time %q", hour, lit)
650
	}
651
	if minute < 0 || minute > 59 {
652
		p.errorf("invalid minute %d in time %q", minute, lit)
653
	}
654
	if second < 0 || second > 59 {
655
		p.errorf("invalid second %d in time %q", second, lit)
656
	}
657
658
	return ast.Time{
659
		Hour:   hour,
660
		Minute: minute,
661
		Second: second,
662
		Span:   p.span(s),
663
	}
664
}
665
666
func (p *Parser) parseApplyDirective() *ast.ApplyDirective {
667
	s := p.cur.Span
668
	p.expect(token.APPLY)
669
	p.skipWhitespace()
670
671
	expr := p.parseDirectiveExpr()
672
	comment := p.parseOptInlineComment()
673
	p.expectNewline()
674
675
	return &ast.ApplyDirective{
676
		Expr:    expr,
677
		Comment: comment,
678
		Span:    p.span(s),
679
	}
680
}
681
682
func (p *Parser) parseEndDirective() *ast.EndDirective {
683
	s := p.cur.Span
684
	p.expect(token.END)
685
	p.skipWhitespace()
686
687
	expr := p.parseDirectiveExpr()
688
	comment := p.parseOptInlineComment()
689
	p.expectNewline()
690
691
	return &ast.EndDirective{
692
		Expr:    expr,
693
		Comment: comment,
694
		Span:    p.span(s),
695
	}
696
}
697
698
func (p *Parser) parseCommentBlockDirective() *ast.CommentBlockDirective {
699
	start := p.cur.Span
700
	p.expect(token.COMMENTKW)
701
	p.skipWhitespace()
702
703
	header := p.parseDirectiveExpr()
704
	comment := p.parseOptInlineComment()
705
	p.expectNewline()
706
707
	var content strings.Builder
708
	for p.cur.Type != token.EOF {
709
		if p.got(token.END) {
710
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) {
711
				p.advance()
712
				p.expectNewline()
713
				break
714
			}
715
			if p.willGet(token.WHITESPACE) {
716
				endTok := p.cur
717
				p.advance()
718
				wsTok := p.cur
719
				p.advance()
720
				if p.got(token.TEXT) && p.cur.Literal == "comment" { // todo: this should check if it's an actual COMMENTKW token
721
					p.advance()
722
					p.parseDirectiveExpr()
723
					p.parseOptInlineComment()
724
					p.expectNewline()
725
					break
726
				}
727
				content.WriteString(endTok.Literal)
728
				content.WriteString(wsTok.Literal)
729
				continue
730
			}
731
		}
732
		content.WriteString(p.cur.Literal)
733
		p.advance()
734
	}
735
736
	return &ast.CommentBlockDirective{
737
		Header:  header,
738
		Content: content.String(),
739
		Comment: comment,
740
		Span:    p.span(start),
741
	}
742
}
743
744
func (p *Parser) parseStatus() ast.Status {
745
	s := p.cur.Span
746
	st := ast.Status{}
747
	switch p.cur.Type {
748
	case token.STAR:
749
		st.Value = ast.StatusCleared
750
	case token.BANG:
751
		st.Value = ast.StatusPending
752
	}
753
	if st.Value != ast.StatusNone {
754
		p.advance()
755
		p.skipWhitespace()
756
	}
757
	st.Span = p.span(s)
758
	return st
759
}
760
761
func (p *Parser) isAmountStart() bool {
762
	switch p.cur.Type {
763
	default:
764
		return false
765
	case token.COMMODITYMARK, token.STRING, token.INT, token.DECIMAL, token.MINUS, token.PLUS, token.PARENEXPR, token.STAR:
766
		return true
767
	}
768
}
769
770
func (p *Parser) parseAmount() *ast.Amount {
771
	s := p.cur.Span
772
	amt := &ast.Amount{
773
		QuantityFmt: ast.QuantityFormat{Decimal: '.'},
774
	}
775
	defer func() {
776
		// The span covers from the first token to the start of the next unconsumed token.
777
		// Since parseQuantityInto (and possible commodity consumption) advanced past the last
778
		// amount token, p.cur points to the next token after the amount — which is the correct end.
779
		amt.Span = p.span(s)
780
	}()
781
782
	p.parseAmountSign(amt)
783
	p.skipWhitespace()
784
785
	// commodity before quantity: $10.00, eur 10.00
786
	if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
787
		cs := p.cur.Span
788
		amt.Commodity = unquote(p.cur.Literal)
789
		amt.CommodityPos = ast.CommodityBefore
790
		p.advance()
791
		amt.CommoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
792
		if p.got(token.WHITESPACE) {
793
			amt.HasSpace = true
794
			p.skipWhitespace()
795
		}
796
	}
797
798
	// optional sign after commodity: $ -10
799
	p.parseAmountSign(amt)
800
	p.skipWhitespace()
801
802
	p.parseQuantityInto(amt)
803
804
	// commodity after quantity: 10.00 UAH, 10.00 "EUR" (only if not set)
805
	if amt.Commodity == "" {
806
		switch p.cur.Type {
807
		case token.WHITESPACE:
808
			p.skipWhitespace()
809
			if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
810
				cs := p.cur.Span
811
				amt.HasSpace = true
812
				amt.Commodity = unquote(p.cur.Literal)
813
				amt.CommodityPos = ast.CommodityAfter
814
				p.advance()
815
				amt.CommoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
816
			}
817
		case token.COMMODITYMARK, token.TEXT, token.STRING:
818
			cs := p.cur.Span
819
			amt.Commodity = unquote(p.cur.Literal)
820
			amt.CommodityPos = ast.CommodityAfter
821
			p.advance()
822
			amt.CommoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
823
		}
824
	}
825
826
	return amt
827
}
828
829
// parseAmountSign consumes an optional leading +/- into IsNegative.
830
func (p *Parser) parseAmountSign(amt *ast.Amount) {
831
	switch p.cur.Type {
832
	case token.MINUS:
833
		amt.IsNegative = true
834
		p.advance()
835
	case token.PLUS:
836
		p.advance()
837
	}
838
}
839
840
func (p *Parser) parseAmountWithOptExpr() *ast.Amount {
841
	if p.got(token.STAR) {
842
		p.advance()
843
		p.skipWhitespace()
844
		amt := p.parseAmount()
845
		if amt != nil {
846
			amt.IsExpr = true
847
		}
848
		return amt
849
	}
850
	if p.got(token.PARENEXPR) {
851
		lit := p.cur.Literal
852
		amt := &ast.Amount{
853
			IsExpr:      true,
854
			QuantityFmt: ast.QuantityFormat{Decimal: '.'},
855
		}
856
		if len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
857
			amt.Expr = strings.Trim(lit[1:len(lit)-1], " \t")
858
		}
859
		amt.Span = p.cur.Span
860
		p.advance()
861
		return amt
862
	}
863
	return p.parseAmount()
864
}
865
866
func (p *Parser) parsePosting() *ast.Posting {
867
	s := p.cur.Span
868
	posting := &ast.Posting{}
869
	p.expect(token.INDENT)
870
871
	// exit if it's empty line
872
	if p.got(token.NEWLINE) || p.got(token.EOF) {
873
		p.syncToNextline()
874
		return nil
875
	}
876
877
	// optional status, outside of brackets, '! (account)'
878
	posting.Status = p.parseStatus()
879
880
	// detect virtual posting brackets
881
	switch p.cur.Type {
882
	case token.LPAREN:
883
		posting.Type = ast.PostingVirtualUnbalanced
884
		p.advance()
885
	case token.LBRACKET:
886
		posting.Type = ast.PostingVirtualBalanced
887
		p.advance()
888
	}
889
890
	// optional status, inside of brackets, '(* account)'
891
	if p.got(token.STAR) || p.got(token.BANG) {
892
		posting.Status = p.parseStatus()
893
	}
894
895
	// validate, must be account text
896
	if p.cur.Type != token.TEXT {
897
		p.errorf("expected account name, got %s", p.cur.Type)
898
		p.syncToNextline()
899
		return nil
900
	}
901
902
	posting.Account = p.parseAccount()
903
904
	// consume closing bracket
905
	switch p.cur.Type {
906
	case token.RPAREN:
907
		p.advance()
908
	case token.RBRACKET:
909
		p.advance()
910
	}
911
912
	// optional amount - after two spaces
913
	if p.got(token.WHITESPACE) {
914
		p.skipWhitespace()
915
		if p.isAmountStart() {
916
			posting.Amount = p.parseAmountWithOptExpr()
917
		}
918
	}
919
920
	// optional cost '@' or '@@'
921
	p.skipWhitespace()
922
	if p.got(token.AT) || p.got(token.ATAT) {
923
		posting.Cost = p.parseCost()
924
	}
925
926
	// optional balance assertion or assignment
927
	p.skipWhitespace()
928
	if p.got(token.COLON) && p.willGet(token.EQ) {
929
		p.advance() // consume ':' of ':='
930
		posting.Balance = p.parseBalanceAssertion()
931
		posting.Balance.IsAssignment = true
932
	} else if p.got(token.EQ) || p.got(token.EQEQ) || p.got(token.EQEQEQ) || p.got(token.EQSTAR) {
933
		posting.Balance = p.parseBalanceAssertion()
934
	}
935
936
	posting.Comment = p.parseOptInlineComment()
937
	p.expectNewline()
938
939
	// continuation comments
940
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
941
		p.advance()
942
		c := p.parseComment()
943
		posting.Comments = append(posting.Comments, *c)
944
	}
945
946
	posting.Span = p.span(s)
947
	return posting
948
}
949
950
func (p *Parser) parseCost() *ast.Cost {
951
	s := p.cur.Span
952
	isTotal := p.got(token.ATAT)
953
	p.advance() // consume '@' '@@'
954
	p.skipWhitespace()
955
	return &ast.Cost{
956
		IsTotal: isTotal,
957
		Amount:  *p.parseAmount(),
958
		Span:    p.span(s),
959
	}
960
}
961
962
func (p *Parser) parseBalanceAssertion() *ast.BalanceAssertion {
963
	s := p.cur.Span
964
965
	ba := &ast.BalanceAssertion{}
966
	switch p.cur.Type {
967
	case token.EQ: // basic assertion
968
	case token.EQSTAR: // inclusive assertion
969
		ba.IsInclusive = true
970
	case token.EQEQ: // strict assertion
971
		ba.IsStrict = true
972
	case token.EQEQEQ: // strict inclusive assertion
973
		ba.IsStrict = true
974
		ba.IsInclusive = true
975
	}
976
	p.advance()
977
	p.skipWhitespace()
978
979
	ba.Amount = *p.parseAmount()
980
	p.skipWhitespace()
981
	if p.got(token.AT) || p.got(token.ATAT) {
982
		c := p.parseCost()
983
		ba.Cost = c
984
	}
985
	ba.Span = p.span(s)
986
	return ba
987
}
988
989
func (p *Parser) readAccountSegment() (ast.SubAccount, bool) {
990
	switch p.cur.Type {
991
	case token.TEXT:
992
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
993
		p.advance()
994
995
		// handle multi work segment, e.g: "credit card"
996
		if p.got(token.WHITESPACE) && p.willGet(token.TEXT) && len(p.peek.Literal) > 0 && p.peek.Literal[0] != '(' {
997
			sub.Name += " "
998
			p.advance()
999
			sub.Name += p.cur.Literal
1000
			p.advance()
1001
		}
1002
		return sub, true
1003
1004
	case token.COMMODITYMARK:
1005
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
1006
		p.advance()
1007
		// merge "EUR" + "-HRK" to "EUR-HRK"
1008
		for p.got(token.TEXT) {
1009
			sub.Name += p.cur.Literal
1010
			p.advance()
1011
		}
1012
		return sub, true
1013
1014
	default:
1015
		return ast.SubAccount{}, false
1016
	}
1017
}
1018
1019
func (p *Parser) parseAccount() ast.Account {
1020
	s := p.cur.Span
1021
	acc := ast.Account{}
1022
1023
	sub, ok := p.readAccountSegment()
1024
	if !ok {
1025
		p.errorf("expected account, got %s", p.cur.Type)
1026
		return ast.Account{}
1027
	}
1028
	acc.Name = append(acc.Name, sub)
1029
1030
	for p.got(token.COLON) {
1031
		p.advance()
1032
		sub, ok := p.readAccountSegment()
1033
		if !ok {
1034
			break
1035
		}
1036
		acc.Name = append(acc.Name, sub)
1037
	}
1038
1039
	acc.Span = p.span(s)
1040
	return acc
1041
}
1042
1043
func (p *Parser) parseDate() ast.Date {
1044
	s := p.cur.Span
1045
	tok, ok := p.expect(token.DATE)
1046
	if !ok {
1047
		return ast.Date{Span: p.span(s)}
1048
	}
1049
1050
	year, month, day, sep, err := ParseDateLiteral(tok.Literal)
1051
	if err != nil {
1052
		p.errorf("%v", err)
1053
		return ast.Date{Span: p.span(s)}
1054
	}
1055
	if year == 0 {
1056
		year = p.defaultYear
1057
	}
1058
1059
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep, Span: p.span(s)}
1060
}
1061
1062
func (p *Parser) parseOptInlineComment() *ast.Comment {
1063
	p.skipWhitespace()
1064
	if !p.got(token.SEMICOLON) {
1065
		return nil
1066
	}
1067
	return p.parseCommentRest(p.cur.Span)
1068
}
1069
1070
// parseCommentRest consumes a comment marker at p.cur, then optional text;
1071
// s anchors the span at the marker's start.
1072
func (p *Parser) parseCommentRest(s token.Span) *ast.Comment {
1073
	marker := p.cur.Literal[0]
1074
	p.advance()
1075
	p.skipWhitespace()
1076
1077
	var tags []ast.Tag
1078
	text := ""
1079
	if p.got(token.TEXT) {
1080
		text = p.cur.Literal
1081
		tags = parseCommentTags(text, p.cur.Span.Start)
1082
		p.advance()
1083
	}
1084
1085
	return &ast.Comment{
1086
		Marker: marker,
1087
		Tags:   tags,
1088
		Text:   text,
1089
		Span:   p.span(s),
1090
	}
1091
}
1092
1093
func (p *Parser) parseOptPeriodicDescription() (string, token.Span) {
1094
	if p.cur.Type != token.WHITESPACE || len(p.cur.Literal) < 2 {
1095
		return "", token.Span{}
1096
	}
1097
1098
	p.skipWhitespace()
1099
1100
	if p.cur.Type != token.TEXT {
1101
		return "", token.Span{}
1102
	}
1103
1104
	s := p.cur.Span
1105
	desc := p.parseDescription()
1106
	return desc, p.span(s)
1107
}
1108
1109
func (p *Parser) parseDescription() string {
1110
	var desc strings.Builder
1111
	for p.got(token.TEXT) || (p.got(token.WHITESPACE) && p.willGet(token.TEXT)) {
1112
		_, _ = desc.WriteString(p.cur.Literal)
1113
		p.advance()
1114
	}
1115
	return desc.String()
1116
}
1117
1118
func (p *Parser) parseDirectiveExpr() string {
1119
	var b strings.Builder
1120
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF && p.cur.Type != token.SEMICOLON {
1121
		_, _ = b.WriteString(p.cur.Literal)
1122
		p.advance()
1123
	}
1124
	return b.String()
1125
}
1126
1127
func (p *Parser) parseQuantityInto(amt *ast.Amount) {
1128
	if p.cur.Type != token.INT && p.cur.Type != token.DECIMAL && p.cur.Type != token.TEXT {
1129
		p.errorf("expected quantity, got %s", p.cur.Type)
1130
		return
1131
	}
1132
1133
	lit := p.cur.Literal
1134
	p.advance()
1135
1136
	// detect format metadata before normalizing
1137
	amt.QuantityFmt = detectFormat(lit)
1138
1139
	// normalize for decimal.NewFromString
1140
	// remove thousands separators, replace decimal mark with '.'
1141
	normalized := normalizeLiteral(lit, amt.QuantityFmt.Thousands, amt.QuantityFmt.Decimal)
1142
1143
	q, err := decimal.FromString(normalized)
1144
	if err != nil {
1145
		p.errorf("invalid quantity %q: %v", lit, err)
1146
		return
1147
	}
1148
1149
	if amt.IsNegative {
1150
		q = q.Neg()
1151
	}
1152
	amt.Quantity = q
1153
}
1154
1155
func (p *Parser) parseBlankLine() *ast.BlankLine {
1156
	s := p.cur.Span
1157
	p.expectNewline()
1158
	return &ast.BlankLine{Span: s}
1159
}
1160
1161
func (p *Parser) expectNewline() {
1162
	if p.got(token.NEWLINE) || p.got(token.EOF) {
1163
		if p.got(token.NEWLINE) {
1164
			p.advance()
1165
		}
1166
		return
1167
	}
1168
	p.errorf("expected %s, got %s", token.NEWLINE, p.cur.Type)
1169
}
1170
1171
func (p *Parser) advance() token.Token {
1172
	prev := p.cur
1173
	p.cur = p.peek
1174
	p.peek = p.lexer.Next()
1175
	return prev
1176
}
1177
1178
func (p *Parser) got(kind token.Type) bool     { return p.cur.Type == kind }
1179
func (p *Parser) willGet(kind token.Type) bool { return p.peek.Type == kind }
1180
1181
func (p *Parser) expect(kind token.Type) (token.Token, bool) {
1182
	if p.got(kind) {
1183
		return p.advance(), true
1184
	}
1185
	p.errorf("expected %s, got %s", kind, p.cur.Type)
1186
	return p.cur, false
1187
}
1188
1189
func (p *Parser) errorf(format string, args ...any) {
1190
	p.errors = append(p.errors, &ast.ParseError{
1191
		Span:    p.cur.Span,
1192
		Message: fmt.Sprintf(format, args...),
1193
	})
1194
}
1195
1196
func isDirectiveKeyword(t token.Type) bool {
1197
	switch t {
1198
	case token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE,
1199
		token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END,
1200
		token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C:
1201
		return true
1202
	}
1203
	return false
1204
}
1205
1206
func (p *Parser) sync() {
1207
	for {
1208
		switch p.cur.Type {
1209
		case token.EOF:
1210
			return
1211
		case token.NEWLINE:
1212
			p.advance()
1213
			t := p.cur.Type
1214
			if isDirectiveKeyword(t) || t == token.DATE || t == token.TILDE || t == token.EQ {
1215
				return
1216
			}
1217
		default:
1218
			p.advance()
1219
		}
1220
	}
1221
}
1222
1223
func (p *Parser) syncToNextline() {
1224
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF {
1225
		p.advance()
1226
	}
1227
	if p.got(token.NEWLINE) {
1228
		p.advance()
1229
	}
1230
}
1231
1232
func (p *Parser) skipWhitespace() {
1233
	for p.got(token.WHITESPACE) {
1234
		p.advance()
1235
	}
1236
}
1237
1238
func (p *Parser) span(s token.Span) token.Span {
1239
	return token.Span{Start: s.Start, End: p.cur.Span.Start}
1240
}
1241
1242
func normalizeLiteral(lit string, thousands, decimal byte) string {
1243
	var b strings.Builder
1244
	for _, ch := range []byte(lit) {
1245
		if thousands != 0 && ch == thousands {
1246
			continue // skip thousands separator
1247
		}
1248
		if ch == decimal {
1249
			b.WriteByte('.')
1250
		} else {
1251
			b.WriteByte(ch)
1252
		}
1253
	}
1254
	return b.String()
1255
}
1256
1257
func detectFormat(lit string) ast.QuantityFormat {
1258
	var seps []int
1259
	for i, ch := range []byte(lit) {
1260
		if ch == '.' || ch == ',' || ch == ' ' || ch == '_' || ch == '\'' {
1261
			seps = append(seps, i)
1262
		}
1263
	}
1264
1265
	if len(seps) == 0 {
1266
		return ast.QuantityFormat{Decimal: '.', Thousands: 0, Precision: 0}
1267
	}
1268
1269
	last := seps[len(seps)-1]
1270
	dec := lit[last]
1271
	var thou byte
1272
	if len(seps) > 1 {
1273
		thou = lit[seps[0]]
1274
	} else if dec == ' ' || dec == '_' || dec == '\'' {
1275
		// single space/underscore/apostrophe is always thousands
1276
		thou = dec
1277
		dec = '.'
1278
	}
1279
1280
	// calculate precision when the last separator is a real decimal
1281
	prec := 0
1282
	if thou == 0 || len(seps) > 1 {
1283
		prec = len(lit) - last - 1
1284
	}
1285
1286
	return ast.QuantityFormat{Decimal: dec, Thousands: thou, Precision: prec}
1287
}
1288
1289
// parseSimpleDate  parses full YYYY/MM/DD date literal embedded in free text.
1290
func parseSimpleDate(s string) ast.Date {
1291
	year, month, day, sep, err := ParseDateLiteral(s)
1292
	if err != nil {
1293
		return ast.Date{}
1294
	}
1295
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep}
1296
}
1297
1298
// ParseDateLiteral parses and validates a date literal.
1299
// It accepts full YYYY/MM/DD and partial MM/DD forms, with '-', '/' or '.' as separators.
1300
func ParseDateLiteral(lit string) (year, month, day int, sep byte, err error) {
1301
	sep = dateSeparator(lit)
1302
	if sep == 0 {
1303
		return 0, 0, 0, 0, fmt.Errorf("invalid date format: %q", lit)
1304
	}
1305
1306
	parts := strings.Split(lit, string(sep))
1307
	if len(parts) != 2 && len(parts) != 3 {
1308
		return 0, 0, 0, 0, fmt.Errorf("invalid date format: %q", lit)
1309
	}
1310
1311
	nums := make([]int, len(parts))
1312
	for i, part := range parts {
1313
		if nums[i], err = strconv.Atoi(part); err != nil {
1314
			return 0, 0, 0, 0, fmt.Errorf("invalid date literal: %q", lit)
1315
		}
1316
	}
1317
1318
	month = nums[len(parts)-2]
1319
	if month < 1 || month > 12 {
1320
		return 0, 0, 0, 0, fmt.Errorf("invalid month %d in %q", month, lit)
1321
	}
1322
1323
	day = nums[len(parts)-1]
1324
	if day < 1 || day > 31 {
1325
		return 0, 0, 0, 0, fmt.Errorf("invalid day %d in %q", day, lit)
1326
	}
1327
1328
	if len(parts) == 2 {
1329
		return 0, month, day, sep, nil
1330
	}
1331
	return nums[0], month, day, sep, nil
1332
}
1333
1334
func dateSeparator(lit string) byte {
1335
	for i := 0; i < len(lit); i++ {
1336
		if lit[i] == '/' || lit[i] == '-' || lit[i] == '.' {
1337
			return lit[i]
1338
		}
1339
	}
1340
	return 0
1341
}
1342
1343
// parseCommentTags extacts tags from comment text.
1344
// A tag is a word immediately followed by a ':', with an optional value that ends at a comma or the end of a line.
1345
// https://hledger.org/1.52/hledger.html?highlight=tags#tags
1346
func parseCommentTags(text string, base token.Pos) []ast.Tag {
1347
	var tags []ast.Tag
1348
	for i := 0; i < len(text); {
1349
		colon := strings.IndexByte(text[i:], ':')
1350
		if colon < 0 {
1351
			break
1352
		}
1353
		colon += i
1354
1355
		keyStart := colon
1356
		for keyStart > i {
1357
			r, size := utf8.DecodeLastRuneInString(text[:keyStart])
1358
			if unicode.IsSpace(r) {
1359
				break
1360
			}
1361
			keyStart -= size
1362
		}
1363
		if keyStart == colon { // nothing before the colon = not a tag
1364
			i = colon + 1
1365
			continue
1366
		}
1367
		key := text[keyStart:colon]
1368
1369
		valueEnd := colon + 1
1370
		for valueEnd < len(text) && text[valueEnd] != ',' {
1371
			valueEnd++
1372
		}
1373
		value := strings.TrimSpace(text[colon+1 : valueEnd])
1374
1375
		tags = append(tags, ast.Tag{
1376
			Key:   key,
1377
			Value: value,
1378
			Span: token.Span{
1379
				Start: tagPos(base, text, keyStart),
1380
				End:   tagPos(base, text, valueEnd),
1381
			},
1382
		})
1383
		i = valueEnd
1384
		if i < len(text) && text[i] == ',' {
1385
			i++
1386
		}
1387
	}
1388
1389
	return tags
1390
}
1391
1392
func tagPos(base token.Pos, text string, off int) token.Pos {
1393
	return token.Pos{
1394
		File:   base.File,
1395
		Offset: base.Offset + off,
1396
		Line:   base.Line,
1397
		Col:    base.Col + utf8.RuneCountInString(text[:off]),
1398
	}
1399
}