all repos

clerk @ 17e6766

missing tooling for ledger/hledger

clerk/journal/parser/parser.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
ast: store postings as values instead of pointers, 1 month ago
1
package parser
2
3
import (
4
	"fmt"
5
	"strconv"
6
	"strings"
7
	"unicode"
8
	"unicode/utf8"
9
10
	"olexsmir.xyz/clerk/internal/decimal"
11
	"olexsmir.xyz/clerk/journal/ast"
12
	"olexsmir.xyz/clerk/journal/lexer"
13
	"olexsmir.xyz/clerk/journal/token"
14
)
15
16
type Parser struct {
17
	lexer  *lexer.Lexer
18
	errors []*ast.ParseError
19
	cur    token.Token
20
	peek   token.Token
21
22
	defaultYear int // set by year directive, used for short date inference
23
}
24
25
func New(lex *lexer.Lexer) *Parser {
26
	p := &Parser{lexer: lex}
27
	p.advance() // populate .peek
28
	p.advance() // populate .cur
29
	return p
30
}
31
32
func NewWithYear(lex *lexer.Lexer, year int) *Parser {
33
	p := &Parser{lexer: lex, defaultYear: year}
34
	p.advance() // populate .peek
35
	p.advance() // populate .cur
36
	return p
37
}
38
39
func (p *Parser) ParseJournal() *ast.Journal {
40
	f := &ast.Journal{}
41
	for p.cur.Type != token.EOF {
42
		if e := p.parseEntry(); e != nil {
43
			f.Entries = append(f.Entries, e)
44
		}
45
	}
46
	f.Errors = p.errors
47
	return f
48
}
49
50
func (p *Parser) parseEntry() ast.Entry {
51
	if p.got(token.BANG) || p.got(token.AT) {
52
		if isDirectiveKeyword(p.peek.Type) {
53
			p.advance() // consume prefix
54
		}
55
	}
56
57
	switch p.cur.Type {
58
	case token.ILLEGAL:
59
		p.errorf("illegal character %q", p.cur.Literal)
60
		p.advance()
61
		return nil
62
	case token.INDENT:
63
		p.errorf("unexpected indent")
64
		p.syncToNextline()
65
		return nil
66
	case token.DATE:
67
		return p.parseTransaction()
68
	case token.TILDE:
69
		return p.parsePeriodicTransaction()
70
	case token.EQ:
71
		return p.parseAutomatedTransaction()
72
	case token.NEWLINE:
73
		return p.parseBlankLine()
74
	case token.SEMICOLON, token.HASH, token.PERCENT, token.STAR:
75
		return p.parseComment()
76
	case token.ACCOUNT:
77
		return p.parseAccountDirective()
78
	case token.COMMODITY:
79
		return p.parseCommodityDirective()
80
	case token.INCLUDE:
81
		return p.parseIncludeDirective()
82
	case token.ALIAS:
83
		return p.parseAliasDirective()
84
	case token.PAYEE:
85
		return p.parsePayeeDirective()
86
	case token.TAG:
87
		return p.parseTagDirective()
88
	case token.YEAR:
89
		return p.parseYearDirective()
90
	case token.DECIMALMARK:
91
		return p.parseDecimalMarkDirective()
92
	case token.D:
93
		return p.parseDefaultCommodityDirective()
94
	case token.P:
95
		return p.parseMarketPriceDirective()
96
	case token.N:
97
		return p.parseIgnoredDirective()
98
	case token.C:
99
		return p.parseConversionDirective()
100
	case token.APPLY:
101
		return p.parseApplyDirective()
102
	case token.END:
103
		return p.parseEndDirective()
104
	case token.COMMENTKW:
105
		return p.parseCommentBlockDirective()
106
	default:
107
		p.errorf("unexpected token %s", p.cur.Type)
108
		p.sync()
109
		return nil
110
	}
111
}
112
113
func (p *Parser) parseTransaction() *ast.Transaction {
114
	s := p.cur.Span
115
	tx := &ast.Transaction{}
116
117
	tx.Date = p.parseDate()
118
119
	p.skipWhitespace()
120
121
	// optional secondary date
122
	if p.got(token.EQ) {
123
		p.advance()
124
		p.skipWhitespace()
125
		d := p.parseDate()
126
		tx.SecondDate = &d
127
	}
128
129
	p.skipWhitespace()
130
131
	// optional status
132
	tx.Status = p.parseStatus()
133
134
	// optional code - the lexer emits "(CODE)" as a single TEXT token; split it here
135
	if p.got(token.TEXT) {
136
		if lit := p.cur.Literal; len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
137
			tx.Code = &ast.Code{Value: lit[1 : len(lit)-1], Span: p.cur.Span}
138
			p.advance()
139
			p.skipWhitespace()
140
		}
141
	}
142
143
	// optional payee | note
144
	if p.got(token.TEXT) || p.got(token.STRING) {
145
		tx.Payee = p.parsePayee()
146
147
		// check for | separator
148
		p.skipWhitespace()
149
150
		if p.got(token.PIPE) {
151
			p.advance()
152
			if p.got(token.TEXT) {
153
				sn := p.cur.Span
154
				n := p.cur.Literal
155
				p.advance()
156
				tx.Note = &ast.Note{Value: n, Span: p.span(sn)}
157
			}
158
		}
159
	}
160
161
	tx.Comment = p.parseOptInlineComment()
162
	p.expectNewline()
163
164
	tx.HeaderComments, tx.Postings = p.parseHeaderCommentsAndPostings()
165
166
	tx.Span = p.span(s)
167
	return tx
168
}
169
170
func unquote(s string) string {
171
	if len(s) >= 2 && ((s[0] == '"' && s[len(s)-1] == '"') || (s[0] == '\'' && s[len(s)-1] == '\'')) {
172
		return s[1 : len(s)-1]
173
	}
174
	return s
175
}
176
177
func (p *Parser) parsePayee() *ast.Payee {
178
	s := p.cur.Span
179
180
	if p.got(token.STRING) {
181
		name := unquote(p.cur.Literal)
182
		p.advance()
183
		return &ast.Payee{Name: name, Span: p.span(s)}
184
	}
185
186
	// keep spaces/tags between text tokens; stop before trailing whitespace
187
	var name strings.Builder
188
	for isPayeeWord(p.cur.Type) || (isPayeeWord(p.peek.Type) && p.got(token.WHITESPACE)) {
189
		_, _ = name.WriteString(p.cur.Literal)
190
		p.advance()
191
	}
192
	return &ast.Payee{Name: unquote(name.String()), Span: p.span(s)}
193
}
194
195
func isPayeeWord(t token.Type) bool {
196
	switch t {
197
	case token.TEXT, token.INT, token.DECIMAL, token.COMMODITYMARK:
198
		return true
199
	}
200
	return false
201
}
202
203
func (p *Parser) parsePeriodicTransaction() *ast.PeriodicTransaction {
204
	s := p.cur.Span
205
	p.expect(token.TILDE)
206
	p.skipWhitespace()
207
208
	pt := &ast.PeriodicTransaction{}
209
210
	pt.Period = p.parsePeriod()
211
212
	if desc, dspan := p.parseOptPeriodicDescription(); desc != "" {
213
		pt.Description = &ast.Description{Value: desc, Span: dspan}
214
	}
215
216
	comment := p.parseOptInlineComment()
217
	p.expectNewline()
218
219
	pt.HeaderComments, pt.Postings = p.parseHeaderCommentsAndPostings()
220
221
	pt.Span = p.span(s)
222
	pt.Comment = comment
223
	return pt
224
}
225
226
func (p *Parser) parseAutomatedTransaction() *ast.AutomatedTransaction {
227
	s := p.cur.Span
228
	p.expect(token.EQ)
229
	p.skipWhitespace()
230
231
	at := &ast.AutomatedTransaction{}
232
233
	// expression
234
	sd := p.cur.Span
235
	expr := p.parseDirectiveExpr()
236
	at.Expr = ast.Expr{Value: expr, Span: p.span(sd)}
237
	at.Comment = p.parseOptInlineComment()
238
	p.expectNewline()
239
240
	at.HeaderComments, at.Postings = p.parseHeaderCommentsAndPostings()
241
242
	at.Span = p.span(s)
243
	return at
244
}
245
246
func (p *Parser) parseHeaderCommentsAndPostings() (comments []*ast.Comment, postings []ast.Posting) {
247
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
248
		p.advance() // consume indent
249
		comments = append(comments, p.parseComment())
250
	}
251
252
	postings = make([]ast.Posting, 0, 2) // most transactions have 2 postings, small optimization
253
	for p.got(token.INDENT) {
254
		if posting, ok := p.parsePosting(); ok {
255
			postings = append(postings, posting)
256
		}
257
	}
258
259
	return comments, postings
260
}
261
262
func (p *Parser) parsePeriod() ast.Period {
263
	s := p.cur.Span
264
265
	var periodBuf strings.Builder
266
267
	for !p.got(token.NEWLINE) && !p.got(token.EOF) &&
268
		!p.got(token.SEMICOLON) && !p.got(token.HASH) && !p.got(token.PERCENT) && !p.got(token.STAR) {
269
270
		if p.got(token.WHITESPACE) {
271
			if len(p.cur.Literal) >= 2 {
272
				break
273
			}
274
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) ||
275
				p.willGet(token.SEMICOLON) || p.willGet(token.HASH) ||
276
				p.willGet(token.PERCENT) || p.willGet(token.STAR) {
277
				p.advance()
278
				continue
279
			}
280
		}
281
282
		periodBuf.WriteString(p.cur.Literal)
283
		p.advance()
284
	}
285
286
	str := periodBuf.String()
287
	period := ast.Period{Raw: str, Span: p.span(s)}
288
289
	if _, after, ok := strings.Cut(str, " from "); ok {
290
		end := strings.Index(after, " ")
291
		dateStr := after
292
		if end >= 0 {
293
			dateStr = after[:end]
294
		}
295
		if d := parseSimpleDate(dateStr); d.Year > 0 {
296
			fromOff := strings.Index(str, dateStr)
297
			d.Span = periodDateSpan(period, str, dateStr, fromOff)
298
			period.From = &d
299
			rest := after
300
			if end >= 0 {
301
				rest = after[end:]
302
			}
303
			if _, toAfter, ok := strings.Cut(rest, " to "); ok {
304
				if toEnd := strings.Index(toAfter, " "); toEnd >= 0 {
305
					toAfter = toAfter[:toEnd]
306
				}
307
				if d := parseSimpleDate(toAfter); d.Year > 0 {
308
					d.Span = periodDateSpan(period, str, toAfter, fromOff+len(dateStr))
309
					period.To = &d
310
				}
311
			}
312
		}
313
	}
314
	return period
315
}
316
317
// periodDateSpan returns the source span of dateStr, which occurs in the
318
// period text at or after searchFrom. The period span and text cover the same
319
// bytes, so offsets line up 1:1.
320
func periodDateSpan(period ast.Period, text, dateStr string, searchFrom int) token.Span {
321
	off := strings.Index(text[searchFrom:], dateStr)
322
	abs := period.Span.Start.Offset + searchFrom + off
323
	return token.Span{
324
		File:  period.Span.File,
325
		Start: token.Pos{Offset: abs},
326
		End:   token.Pos{Offset: abs + len(dateStr)},
327
	}
328
}
329
330
func (p *Parser) parseComment() *ast.Comment {
331
	s := p.cur.Span
332
	c := p.parseCommentRest(s)
333
	p.expectNewline()
334
	c.Span = p.span(s) // comment spans its line through the newline
335
	return c
336
}
337
338
func (p *Parser) parseAccountDirective() *ast.AccountDirective {
339
	s := p.cur.Span
340
	p.expect(token.ACCOUNT)
341
	p.skipWhitespace()
342
343
	account := p.parseAccount()
344
	comment := p.parseOptInlineComment()
345
	p.expectNewline()
346
347
	var subs []ast.AccountSubdirective
348
	for p.got(token.INDENT) {
349
		p.advance()
350
		p.skipWhitespace()
351
		if p.got(token.NEWLINE) || p.got(token.EOF) {
352
			// whitespace-only line: block continues
353
			p.expectNewline()
354
			continue
355
		}
356
		switch {
357
		case p.got(token.SEMICOLON):
358
			// comment line: directive mode lexes only ';' as a comment marker
359
			ns := p.cur.Span
360
			c := p.parseCommentRest(ns)
361
			p.expectNewline()
362
			subs = append(subs, ast.AccountSubdirective{Kind: ast.SubdirectiveComment, NameSpan: ns, Comment: c})
363
		case p.got(token.TEXT) && isCommentMarker(p.cur.Literal):
364
			// '#', '%' and '*' lex as TEXT in directive mode; treat them as comment lines
365
			c := p.parseTextComment()
366
			subs = append(subs, ast.AccountSubdirective{Kind: ast.SubdirectiveComment, NameSpan: c.Span, Comment: c})
367
		case p.got(token.TEXT):
368
			name := p.cur.Literal
369
			kind, ok := accountSubdirectiveKind(name)
370
			if !ok {
371
				p.errorf("unknown subdirective %q", name)
372
				p.skipToNewline()
373
				continue
374
			}
375
			kw := p.cur.Span
376
			p.advance()
377
			value, vspan := p.parseSubdirectiveValue()
378
			if value == "" {
379
				p.errorf("expected value for subdirective %q", name)
380
			}
381
			c := p.parseOptInlineComment()
382
			p.expectNewline()
383
			subs = append(subs, ast.AccountSubdirective{
384
				Kind:      kind,
385
				NameSpan:  kw,
386
				Value:     value,
387
				ValueSpan: vspan,
388
				Comment:   c,
389
			})
390
		default:
391
			p.errorf("expected subdirective name, got %s", p.cur.Type)
392
			p.skipToNewline()
393
		}
394
	}
395
396
	return &ast.AccountDirective{
397
		Account:       account,
398
		Subdirectives: subs,
399
		Comment:       comment,
400
		Span:          p.span(s),
401
	}
402
}
403
404
func (p *Parser) parseCommodityDirective() *ast.CommodityDirective {
405
	s := p.cur.Span
406
	p.expect(token.COMMODITY)
407
	p.skipWhitespace()
408
409
	var commodity string
410
	var commoditySpan token.Span
411
	var format *ast.FormatSubDirective
412
413
	switch p.cur.Type {
414
	case token.COMMODITYMARK, token.TEXT, token.STRING:
415
		cs := p.cur.Span
416
		commodity = unquote(p.cur.Literal)
417
		p.advance()
418
		commoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
419
		hadSpace := p.got(token.WHITESPACE)
420
		p.skipWhitespace()
421
		if p.got(token.INT) || p.got(token.DECIMAL) || p.got(token.TEXT) {
422
			amt := p.parseAmount()
423
			amt.Commodity = commodity
424
			amt.CommoditySpan = commoditySpan
425
			amt.CommodityPos = ast.CommodityBefore
426
			amt.HasSpace = hadSpace
427
			format = &ast.FormatSubDirective{Amount: amt}
428
		}
429
	case token.INT, token.DECIMAL:
430
		amt := p.parseAmount()
431
		commodity = amt.Commodity
432
		commoditySpan = amt.CommoditySpan
433
		format = &ast.FormatSubDirective{Amount: amt}
434
	default:
435
		p.errorf("expected commodity name or amount, got %s", p.cur.Type)
436
	}
437
438
	if commodity == "" {
439
		p.errorf("expected commodity name, got %s", p.cur.Type)
440
	}
441
442
	// hledger parity: an inline format amount must include a decimal mark
443
	if format != nil && format.Amount.QuantityFmt.Decimal == 0 {
444
		p.errorfAt(format.Amount.Span, "Please include a decimal point or decimal comma in commodity directives, to help us parse correctly. It may be followed by zero or more decimal digits.")
445
	}
446
447
	comment := p.parseOptInlineComment()
448
	p.expectNewline()
449
450
	var blockComments []*ast.Comment
451
	for p.got(token.INDENT) {
452
		p.advance()
453
		p.skipWhitespace()
454
		if p.got(token.NEWLINE) || p.got(token.EOF) {
455
			// whitespace-only line: block continues
456
			p.expectNewline()
457
			continue
458
		}
459
		switch {
460
		case p.got(token.TEXT) && p.cur.Literal == "format":
461
			kw := p.cur.Span
462
			p.advance()
463
			p.skipWhitespace()
464
			amt := p.parseAmount()
465
			// hledger parity: the format symbol must match the declared commodity,
466
			// and the amount must include a decimal mark; the node is kept either
467
			// way so the printer can round-trip the input.
468
			if amt.Commodity != commodity {
469
				p.errorfAt(amt.Span, "commodity directive symbol %q and format directive symbol %q should be the same", commodity, amt.Commodity)
470
			} else if amt.QuantityFmt.Decimal == 0 {
471
				p.errorfAt(amt.Span, "Please include a decimal point or decimal comma in commodity directives, to help us parse correctly. It may be followed by zero or more decimal digits.")
472
			}
473
			c := p.parseOptInlineComment()
474
			p.expectNewline()
475
			format = &ast.FormatSubDirective{KeywordSpan: kw, Amount: amt, Comment: c}
476
		case p.got(token.SEMICOLON): // comment line
477
			c := p.parseCommentRest(p.cur.Span)
478
			p.expectNewline()
479
			blockComments = append(blockComments, c)
480
		case p.got(token.TEXT) && isCommentMarker(p.cur.Literal):
481
			// '#', '%' and '*' lex as TEXT in directive mode; treat them as comment lines
482
			blockComments = append(blockComments, p.parseTextComment())
483
		case p.got(token.TEXT):
484
			p.errorf("unknown subdirective %q", p.cur.Literal)
485
			p.skipToNewline()
486
		default:
487
			p.errorf("expected subdirective name, got %s", p.cur.Type)
488
			p.skipToNewline()
489
		}
490
	}
491
492
	cd := &ast.CommodityDirective{
493
		Commodity:     commodity,
494
		CommoditySpan: commoditySpan,
495
		FormatSub:     format,
496
		BlockComments: blockComments,
497
		Comment:       comment,
498
		Span:          p.span(s),
499
	}
500
	return cd
501
}
502
503
func (p *Parser) parseIncludeDirective() *ast.IncludeDirective {
504
	s := p.cur.Span
505
	p.expect(token.INCLUDE)
506
	p.skipWhitespace()
507
508
	id := &ast.IncludeDirective{}
509
510
	if p.got(token.TEXT) {
511
		id.Path = p.cur.Literal
512
		p.advance()
513
	} else {
514
		p.errorf("expected file path, got %s", p.cur.Type)
515
	}
516
517
	id.Comment = p.parseOptInlineComment()
518
	p.expectNewline()
519
	id.Span = p.span(s)
520
	return id
521
}
522
523
func (p *Parser) parseAliasDirective() *ast.AliasDirective {
524
	s := p.cur.Span
525
	alias := &ast.AliasDirective{}
526
	p.expect(token.ALIAS)
527
	p.skipWhitespace()
528
	alias.From = p.parseAccount()
529
	p.skipWhitespace()
530
	p.expect(token.EQ)
531
	p.skipWhitespace()
532
	alias.To = p.parseAccount()
533
	alias.Comment = p.parseOptInlineComment()
534
	p.expectNewline()
535
	alias.Span = p.span(s)
536
	return alias
537
}
538
539
func (p *Parser) parsePayeeDirective() *ast.PayeeDirective {
540
	s := p.cur.Span
541
	p.expect(token.PAYEE)
542
	p.skipWhitespace()
543
544
	var name *ast.Payee
545
	if p.got(token.TEXT) || p.got(token.STRING) || p.got(token.COMMODITYMARK) {
546
		name = p.parsePayee()
547
	}
548
549
	comment := p.parseOptInlineComment()
550
	p.expectNewline()
551
552
	return &ast.PayeeDirective{
553
		Name:    name,
554
		Comment: comment,
555
		Span:    p.span(s),
556
	}
557
}
558
559
func (p *Parser) parseTagDirective() *ast.TagDirective {
560
	s := p.cur.Span
561
	p.expect(token.TAG)
562
	p.skipWhitespace()
563
564
	name := ""
565
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) || p.got(token.STRING) {
566
		name = unquote(p.cur.Literal)
567
		p.advance()
568
	}
569
570
	comment := p.parseOptInlineComment()
571
	p.expectNewline()
572
573
	return &ast.TagDirective{
574
		Name:    name,
575
		Comment: comment,
576
		Span:    p.span(s),
577
	}
578
}
579
580
func (p *Parser) parseYearDirective() *ast.YearDirective {
581
	s := p.cur.Span
582
	year := &ast.YearDirective{}
583
	p.expect(token.YEAR)
584
	p.skipWhitespace()
585
586
	if p.got(token.INT) {
587
		year.Year, _ = strconv.Atoi(p.cur.Literal)
588
		p.defaultYear = year.Year
589
		p.advance()
590
	} else {
591
		p.errorf("expected year, got %s", p.cur.Type)
592
	}
593
594
	year.Comment = p.parseOptInlineComment()
595
	p.expectNewline()
596
	year.Span = p.span(s)
597
598
	return year
599
}
600
601
func (p *Parser) parseDecimalMarkDirective() *ast.DecimalMarkDirective {
602
	s := p.cur.Span
603
	mark := &ast.DecimalMarkDirective{}
604
	p.expect(token.DECIMALMARK)
605
	p.skipWhitespace()
606
607
	mark.Mark = byte('.')
608
	if p.got(token.TEXT) {
609
		if len(p.cur.Literal) > 0 {
610
			mark.Mark = p.cur.Literal[0]
611
		}
612
		p.advance()
613
	}
614
615
	mark.Comment = p.parseOptInlineComment()
616
	p.expectNewline()
617
	mark.Span = p.span(s)
618
	return mark
619
}
620
621
func (p *Parser) parseDefaultCommodityDirective() *ast.DefaultCommodityDirective {
622
	s := p.cur.Span
623
	com := &ast.DefaultCommodityDirective{}
624
	p.expect(token.D)
625
	p.skipWhitespace()
626
	com.Amount = p.parseAmount()
627
	com.Comment = p.parseOptInlineComment()
628
	p.expectNewline()
629
	com.Span = p.span(s)
630
	return com
631
}
632
633
func (p *Parser) parseConversionDirective() *ast.ConversionDirective {
634
	s := p.cur.Span
635
	cd := &ast.ConversionDirective{}
636
	p.expect(token.C)
637
	p.skipWhitespace()
638
639
	if p.isAmountStart() {
640
		cd.From = p.parseAmount()
641
	} else {
642
		p.errorf("expected amount, got %s", p.cur.Type)
643
	}
644
645
	p.skipWhitespace()
646
	if p.got(token.EQ) {
647
		p.advance()
648
		p.skipWhitespace()
649
		if p.isAmountStart() {
650
			cd.To = p.parseAmount()
651
		} else {
652
			p.errorf("expected amount, got %s", p.cur.Type)
653
		}
654
	}
655
656
	cd.Comment = p.parseOptInlineComment()
657
	p.expectNewline()
658
	cd.Span = p.span(s)
659
	return cd
660
}
661
662
func (p *Parser) parseIgnoredDirective() *ast.IgnoredDirective {
663
	s := p.cur.Span
664
	p.expect(token.N)
665
	p.skipWhitespace()
666
667
	id := &ast.IgnoredDirective{}
668
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) || p.got(token.STRING) {
669
		id.Text = unquote(p.cur.Literal)
670
		p.advance()
671
	}
672
	id.Comment = p.parseOptInlineComment()
673
674
	p.expectNewline()
675
	id.Span = p.span(s)
676
	return id
677
}
678
679
func (p *Parser) parseMarketPriceDirective() *ast.MarketPriceDirective {
680
	s := p.cur.Span
681
	p.expect(token.P)
682
	p.skipWhitespace()
683
684
	mp := &ast.MarketPriceDirective{}
685
	mp.DateTime.Date = p.parseDate()
686
	p.skipWhitespace()
687
688
	if p.got(token.TIME) {
689
		mp.DateTime.Time = new(p.parseTime())
690
		p.skipWhitespace()
691
	}
692
693
	if p.got(token.COMMODITYMARK) || p.got(token.STRING) {
694
		mp.Commodity = unquote(p.cur.Literal)
695
		p.advance()
696
	} else {
697
		p.errorf("expected commodity symbol, got %s", p.cur.Type)
698
	}
699
	p.skipWhitespace()
700
701
	mp.Amount = p.parseAmount()
702
703
	mp.Comment = p.parseOptInlineComment()
704
705
	p.expectNewline()
706
	mp.Span = p.span(s)
707
	return mp
708
}
709
710
func (p *Parser) parseTime() ast.Time {
711
	s := p.cur.Span
712
	tok, _ := p.expect(token.TIME)
713
	lit := tok.Literal
714
715
	parts := strings.Split(lit, ":")
716
	if len(parts) < 2 {
717
		p.errorf("invalid time format: %q", lit)
718
		return ast.Time{Span: p.span(s)}
719
	}
720
721
	hour, _ := strconv.Atoi(parts[0])
722
	minute, _ := strconv.Atoi(parts[1])
723
	second := 0
724
	if len(parts) > 2 {
725
		second, _ = strconv.Atoi(parts[2])
726
	}
727
728
	if hour < 0 || hour > 23 {
729
		p.errorf("invalid hour %d in time %q", hour, lit)
730
	}
731
	if minute < 0 || minute > 59 {
732
		p.errorf("invalid minute %d in time %q", minute, lit)
733
	}
734
	if second < 0 || second > 59 {
735
		p.errorf("invalid second %d in time %q", second, lit)
736
	}
737
738
	return ast.Time{
739
		Hour:   hour,
740
		Minute: minute,
741
		Second: second,
742
		Span:   p.span(s),
743
	}
744
}
745
746
func (p *Parser) parseApplyDirective() *ast.ApplyDirective {
747
	s := p.cur.Span
748
	p.expect(token.APPLY)
749
	p.skipWhitespace()
750
751
	expr := p.parseDirectiveExpr()
752
	comment := p.parseOptInlineComment()
753
	p.expectNewline()
754
755
	return &ast.ApplyDirective{
756
		Expr:    expr,
757
		Comment: comment,
758
		Span:    p.span(s),
759
	}
760
}
761
762
func (p *Parser) parseEndDirective() *ast.EndDirective {
763
	s := p.cur.Span
764
	p.expect(token.END)
765
	p.skipWhitespace()
766
767
	expr := p.parseDirectiveExpr()
768
	comment := p.parseOptInlineComment()
769
	p.expectNewline()
770
771
	return &ast.EndDirective{
772
		Expr:    expr,
773
		Comment: comment,
774
		Span:    p.span(s),
775
	}
776
}
777
778
func (p *Parser) parseCommentBlockDirective() *ast.CommentBlockDirective {
779
	start := p.cur.Span
780
	p.expect(token.COMMENTKW)
781
	p.skipWhitespace()
782
783
	header := p.parseDirectiveExpr()
784
	comment := p.parseOptInlineComment()
785
	p.expectNewline()
786
787
	var content strings.Builder
788
	for p.cur.Type != token.EOF {
789
		if p.got(token.END) {
790
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) {
791
				p.advance()
792
				p.expectNewline()
793
				break
794
			}
795
			if p.willGet(token.WHITESPACE) {
796
				endTok := p.cur
797
				p.advance()
798
				wsTok := p.cur
799
				p.advance()
800
				if p.got(token.TEXT) && p.cur.Literal == "comment" { // todo: this should check if it's an actual COMMENTKW token
801
					p.advance()
802
					p.parseDirectiveExpr()
803
					p.parseOptInlineComment()
804
					p.expectNewline()
805
					break
806
				}
807
				content.WriteString(endTok.Literal)
808
				content.WriteString(wsTok.Literal)
809
				continue
810
			}
811
		}
812
		content.WriteString(p.cur.Literal)
813
		p.advance()
814
	}
815
816
	return &ast.CommentBlockDirective{
817
		Header:  header,
818
		Content: content.String(),
819
		Comment: comment,
820
		Span:    p.span(start),
821
	}
822
}
823
824
func (p *Parser) parseStatus() ast.Status {
825
	s := p.cur.Span
826
	st := ast.Status{}
827
	switch p.cur.Type {
828
	case token.STAR:
829
		st.Value = ast.StatusCleared
830
	case token.BANG:
831
		st.Value = ast.StatusPending
832
	}
833
	if st.Value != ast.StatusNone {
834
		p.advance()
835
		p.skipWhitespace()
836
	}
837
	st.Span = p.span(s)
838
	return st
839
}
840
841
func (p *Parser) isAmountStart() bool {
842
	switch p.cur.Type {
843
	default:
844
		return false
845
	case token.COMMODITYMARK, token.STRING, token.INT, token.DECIMAL, token.MINUS, token.PLUS, token.PARENEXPR, token.STAR:
846
		return true
847
	}
848
}
849
850
func (p *Parser) parseAmount() ast.Amount {
851
	s := p.cur.Span
852
	amt := ast.Amount{QuantityFmt: ast.QuantityFormat{}}
853
854
	p.parseAmountSign(&amt)
855
	p.skipWhitespace()
856
857
	// commodity before quantity: $10.00, eur 10.00
858
	if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
859
		cs := p.cur.Span
860
		amt.Commodity = unquote(p.cur.Literal)
861
		amt.CommodityPos = ast.CommodityBefore
862
		p.advance()
863
		amt.CommoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
864
		if p.got(token.WHITESPACE) {
865
			amt.HasSpace = true
866
			p.skipWhitespace()
867
		}
868
	}
869
870
	// optional sign after commodity: $ -10
871
	p.parseAmountSign(&amt)
872
	p.skipWhitespace()
873
874
	p.parseQuantityInto(&amt)
875
876
	// commodity after quantity: 10.00 UAH, 10.00 "EUR" (only if not set)
877
	if amt.Commodity == "" {
878
		switch p.cur.Type {
879
		case token.WHITESPACE:
880
			p.skipWhitespace()
881
			if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
882
				cs := p.cur.Span
883
				amt.HasSpace = true
884
				amt.Commodity = unquote(p.cur.Literal)
885
				amt.CommodityPos = ast.CommodityAfter
886
				p.advance()
887
				amt.CommoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
888
			}
889
		case token.COMMODITYMARK, token.TEXT, token.STRING:
890
			cs := p.cur.Span
891
			amt.Commodity = unquote(p.cur.Literal)
892
			amt.CommodityPos = ast.CommodityAfter
893
			p.advance()
894
			amt.CommoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
895
		}
896
	}
897
898
	amt.Span = p.span(s)
899
	return amt
900
}
901
902
// parseAmountSign consumes an optional leading +/- into IsNegative.
903
func (p *Parser) parseAmountSign(amt *ast.Amount) {
904
	switch p.cur.Type {
905
	case token.MINUS:
906
		amt.IsNegative = true
907
		p.advance()
908
	case token.PLUS:
909
		p.advance()
910
	}
911
}
912
913
func (p *Parser) parseAmountWithOptExpr() ast.Amount {
914
	if p.got(token.STAR) {
915
		p.advance()
916
		p.skipWhitespace()
917
		amt := p.parseAmount()
918
		amt.IsExpr = true
919
		return amt
920
	}
921
	if p.got(token.PARENEXPR) {
922
		lit := p.cur.Literal
923
		amt := ast.Amount{
924
			IsExpr:      true,
925
			QuantityFmt: ast.QuantityFormat{},
926
		}
927
		if len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
928
			amt.Expr = strings.Trim(lit[1:len(lit)-1], " \t")
929
		}
930
		amt.Span = p.cur.Span
931
		p.advance()
932
		return amt
933
	}
934
	return p.parseAmount()
935
}
936
937
func (p *Parser) parsePosting() (ast.Posting, bool) {
938
	s := p.cur.Span
939
	posting := ast.Posting{}
940
	p.expect(token.INDENT)
941
942
	// exit if it's empty line
943
	if p.got(token.NEWLINE) || p.got(token.EOF) {
944
		p.syncToNextline()
945
		return ast.Posting{}, false
946
	}
947
948
	// optional status, outside of brackets, '! (account)'
949
	posting.Status = p.parseStatus()
950
951
	// detect virtual posting brackets
952
	switch p.cur.Type {
953
	case token.LPAREN:
954
		posting.Type = ast.PostingVirtualUnbalanced
955
		p.advance()
956
	case token.LBRACKET:
957
		posting.Type = ast.PostingVirtualBalanced
958
		p.advance()
959
	}
960
961
	// optional status, inside of brackets, '(* account)'
962
	if p.got(token.STAR) || p.got(token.BANG) {
963
		posting.Status = p.parseStatus()
964
	}
965
966
	// validate, must be account text
967
	if p.cur.Type != token.TEXT {
968
		p.errorf("expected account name, got %s", p.cur.Type)
969
		p.syncToNextline()
970
		return ast.Posting{}, false
971
	}
972
973
	posting.Account = p.parseAccount()
974
975
	// consume closing bracket
976
	switch p.cur.Type {
977
	case token.RPAREN:
978
		p.advance()
979
	case token.RBRACKET:
980
		p.advance()
981
	}
982
983
	// optional amount - after two spaces
984
	if p.got(token.WHITESPACE) {
985
		p.skipWhitespace()
986
		if p.isAmountStart() {
987
			amt := p.parseAmountWithOptExpr()
988
			posting.Amount = &amt
989
		}
990
	}
991
992
	// optional cost '@' or '@@'
993
	p.skipWhitespace()
994
	if p.got(token.AT) || p.got(token.ATAT) {
995
		posting.Cost = p.parseCost()
996
	}
997
998
	// optional balance assertion or assignment
999
	p.skipWhitespace()
1000
	if p.got(token.COLON) && p.willGet(token.EQ) {
1001
		p.advance() // consume ':' of ':='
1002
		posting.Balance = p.parseBalanceAssertion()
1003
		posting.Balance.IsAssignment = true
1004
	} else if p.got(token.EQ) || p.got(token.EQEQ) || p.got(token.EQEQEQ) || p.got(token.EQSTAR) {
1005
		posting.Balance = p.parseBalanceAssertion()
1006
	}
1007
1008
	posting.Comment = p.parseOptInlineComment()
1009
	p.expectNewline()
1010
1011
	// continuation comments
1012
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
1013
		p.advance()
1014
		c := p.parseComment()
1015
		posting.Comments = append(posting.Comments, *c)
1016
	}
1017
1018
	posting.Span = p.span(s)
1019
	return posting, true
1020
}
1021
1022
func (p *Parser) parseCost() *ast.Cost {
1023
	s := p.cur.Span
1024
	isTotal := p.got(token.ATAT)
1025
	p.advance() // consume '@' '@@'
1026
	p.skipWhitespace()
1027
	return &ast.Cost{
1028
		IsTotal: isTotal,
1029
		Amount:  p.parseAmount(),
1030
		Span:    p.span(s),
1031
	}
1032
}
1033
1034
func (p *Parser) parseBalanceAssertion() *ast.BalanceAssertion {
1035
	s := p.cur.Span
1036
1037
	ba := &ast.BalanceAssertion{}
1038
	switch p.cur.Type {
1039
	case token.EQ: // basic assertion
1040
	case token.EQSTAR: // inclusive assertion
1041
		ba.IsInclusive = true
1042
	case token.EQEQ: // strict assertion
1043
		ba.IsStrict = true
1044
	case token.EQEQEQ: // strict inclusive assertion
1045
		ba.IsStrict = true
1046
		ba.IsInclusive = true
1047
	}
1048
	p.advance()
1049
	p.skipWhitespace()
1050
1051
	ba.Amount = p.parseAmount()
1052
	p.skipWhitespace()
1053
	if p.got(token.AT) || p.got(token.ATAT) {
1054
		c := p.parseCost()
1055
		ba.Cost = c
1056
	}
1057
	ba.Span = p.span(s)
1058
	return ba
1059
}
1060
1061
func (p *Parser) readAccountSegment() (ast.SubAccount, bool) {
1062
	switch p.cur.Type {
1063
	case token.TEXT:
1064
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
1065
		p.advance()
1066
1067
		// handle multi work segment, e.g: "credit card"
1068
		if p.got(token.WHITESPACE) && p.willGet(token.TEXT) && len(p.peek.Literal) > 0 && p.peek.Literal[0] != '(' {
1069
			sub.Name += " "
1070
			p.advance()
1071
			sub.Name += p.cur.Literal
1072
			p.advance()
1073
		}
1074
		return sub, true
1075
1076
	case token.COMMODITYMARK:
1077
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
1078
		p.advance()
1079
		// merge "EUR" + "-HRK" to "EUR-HRK"
1080
		for p.got(token.TEXT) {
1081
			sub.Name += p.cur.Literal
1082
			p.advance()
1083
		}
1084
		return sub, true
1085
1086
	default:
1087
		return ast.SubAccount{}, false
1088
	}
1089
}
1090
1091
func (p *Parser) parseAccount() ast.Account {
1092
	s := p.cur.Span
1093
	acc := ast.Account{Name: make([]ast.SubAccount, 0, 2)}
1094
1095
	sub, ok := p.readAccountSegment()
1096
	if !ok {
1097
		p.errorf("expected account, got %s", p.cur.Type)
1098
		return ast.Account{}
1099
	}
1100
	acc.Name = append(acc.Name, sub)
1101
1102
	for p.got(token.COLON) {
1103
		p.advance()
1104
		sub, ok := p.readAccountSegment()
1105
		if !ok {
1106
			break
1107
		}
1108
		acc.Name = append(acc.Name, sub)
1109
	}
1110
1111
	acc.Span = p.span(s)
1112
	return acc
1113
}
1114
1115
func (p *Parser) parseDate() ast.Date {
1116
	s := p.cur.Span
1117
	tok, ok := p.expect(token.DATE)
1118
	if !ok {
1119
		return ast.Date{Span: p.span(s)}
1120
	}
1121
1122
	year, month, day, sep, err := ParseDateLiteral(tok.Literal)
1123
	if err != nil {
1124
		p.errorf("%v", err)
1125
		return ast.Date{Span: p.span(s)}
1126
	}
1127
	if year == 0 {
1128
		year = p.defaultYear
1129
	}
1130
1131
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep, Span: p.span(s)}
1132
}
1133
1134
func (p *Parser) parseOptInlineComment() *ast.Comment {
1135
	p.skipWhitespace()
1136
	if !p.got(token.SEMICOLON) {
1137
		return nil
1138
	}
1139
	return p.parseCommentRest(p.cur.Span)
1140
}
1141
1142
// parseCommentRest consumes a comment marker at p.cur, then optional text;
1143
// s anchors the span at the marker's start.
1144
func (p *Parser) parseCommentRest(s token.Span) *ast.Comment {
1145
	marker := p.cur.Literal[0]
1146
	p.advance()
1147
	p.skipWhitespace()
1148
1149
	var tags []ast.Tag
1150
	text := ""
1151
	if p.got(token.TEXT) {
1152
		text = p.cur.Literal
1153
		tags = parseCommentTags(text, p.cur.Span)
1154
		p.advance()
1155
	}
1156
1157
	return &ast.Comment{
1158
		Marker: marker,
1159
		Tags:   tags,
1160
		Text:   text,
1161
		Span:   p.span(s),
1162
	}
1163
}
1164
1165
func (p *Parser) parseOptPeriodicDescription() (string, token.Span) {
1166
	if p.cur.Type != token.WHITESPACE || len(p.cur.Literal) < 2 {
1167
		return "", token.Span{}
1168
	}
1169
1170
	p.skipWhitespace()
1171
1172
	if p.cur.Type != token.TEXT {
1173
		return "", token.Span{}
1174
	}
1175
1176
	s := p.cur.Span
1177
	desc := p.parseDescription()
1178
	return desc, p.span(s)
1179
}
1180
1181
func (p *Parser) parseDescription() string {
1182
	var desc strings.Builder
1183
	for p.got(token.TEXT) || (p.got(token.WHITESPACE) && p.willGet(token.TEXT)) {
1184
		_, _ = desc.WriteString(p.cur.Literal)
1185
		p.advance()
1186
	}
1187
	return desc.String()
1188
}
1189
1190
func (p *Parser) parseDirectiveExpr() string {
1191
	var b strings.Builder
1192
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF && p.cur.Type != token.SEMICOLON {
1193
		_, _ = b.WriteString(p.cur.Literal)
1194
		p.advance()
1195
	}
1196
	return b.String()
1197
}
1198
1199
func (p *Parser) parseQuantityInto(amt *ast.Amount) {
1200
	if p.cur.Type != token.INT && p.cur.Type != token.DECIMAL && p.cur.Type != token.TEXT {
1201
		p.errorf("expected quantity, got %s", p.cur.Type)
1202
		return
1203
	}
1204
1205
	lit := p.cur.Literal
1206
	p.advance()
1207
1208
	// detect format metadata before normalizing
1209
	amt.QuantityFmt = detectFormat(lit)
1210
1211
	// normalize for decimal.NewFromString
1212
	// remove thousands separators, replace decimal mark with '.'
1213
	normalized := normalizeLiteral(lit, amt.QuantityFmt.Thousands, amt.QuantityFmt.Decimal)
1214
1215
	q, err := decimal.FromString(normalized)
1216
	if err != nil {
1217
		p.errorf("invalid quantity %q: %v", lit, err)
1218
		return
1219
	}
1220
1221
	if amt.IsNegative {
1222
		q = q.Neg()
1223
	}
1224
	amt.Quantity = q
1225
}
1226
1227
func (p *Parser) parseBlankLine() *ast.BlankLine {
1228
	s := p.cur.Span
1229
	p.expectNewline()
1230
	return &ast.BlankLine{Span: s}
1231
}
1232
1233
func (p *Parser) expectNewline() {
1234
	if p.got(token.NEWLINE) || p.got(token.EOF) {
1235
		if p.got(token.NEWLINE) {
1236
			p.advance()
1237
		}
1238
		return
1239
	}
1240
	p.errorf("expected %s, got %s", token.NEWLINE, p.cur.Type)
1241
}
1242
1243
func (p *Parser) advance() token.Token {
1244
	prev := p.cur
1245
	p.cur = p.peek
1246
	p.peek = p.lexer.Next()
1247
	return prev
1248
}
1249
1250
func (p *Parser) got(kind token.Type) bool     { return p.cur.Type == kind }
1251
func (p *Parser) willGet(kind token.Type) bool { return p.peek.Type == kind }
1252
1253
func (p *Parser) expect(kind token.Type) (token.Token, bool) {
1254
	if p.got(kind) {
1255
		return p.advance(), true
1256
	}
1257
	p.errorf("expected %s, got %s", kind, p.cur.Type)
1258
	return p.cur, false
1259
}
1260
1261
func (p *Parser) errorf(format string, args ...any) {
1262
	p.errors = append(p.errors, &ast.ParseError{
1263
		Span:    p.cur.Span,
1264
		Message: fmt.Sprintf(format, args...),
1265
	})
1266
}
1267
1268
// errorfAt records a parse error pointing at the start of span.
1269
func (p *Parser) errorfAt(span token.Span, format string, args ...any) {
1270
	p.errors = append(p.errors, &ast.ParseError{
1271
		Span:    token.Span{File: span.File, Start: span.Start, End: span.Start},
1272
		Message: fmt.Sprintf(format, args...),
1273
	})
1274
}
1275
1276
func accountSubdirectiveKind(name string) (ast.AccountSubdirectiveKind, bool) {
1277
	switch name {
1278
	case "alias":
1279
		return ast.SubdirectiveAlias, true
1280
	case "type":
1281
		return ast.SubdirectiveType, true
1282
	case "note":
1283
		return ast.SubdirectiveNote, true
1284
	}
1285
	return 0, false
1286
}
1287
1288
func isCommentMarker(s string) bool {
1289
	return len(s) > 0 && (s[0] == '#' || s[0] == '%' || s[0] == '*')
1290
}
1291
1292
// skipToNewline consumes the rest of the current line.
1293
func (p *Parser) skipToNewline() {
1294
	for !p.got(token.NEWLINE) && !p.got(token.EOF) {
1295
		p.advance()
1296
	}
1297
	p.expectNewline()
1298
}
1299
1300
func (p *Parser) parseSubdirectiveValue() (string, token.Span) {
1301
	var b strings.Builder
1302
	var first, last token.Span
1303
	var single, pendingWS string
1304
	for !p.got(token.SEMICOLON) && !p.got(token.NEWLINE) && !p.got(token.EOF) {
1305
		t := p.cur
1306
		if t.Type == token.WHITESPACE || t.Type == token.INDENT {
1307
			if first.Start.Offset > 0 {
1308
				pendingWS = t.Literal // only the last whitespace run matters
1309
			}
1310
			p.advance()
1311
			continue
1312
		}
1313
		if first.Start.Offset == 0 {
1314
			first, last = t.Span, t.Span
1315
			single = t.Literal
1316
		} else {
1317
			if single != "" {
1318
				b.WriteString(single)
1319
				single = ""
1320
			}
1321
			if pendingWS != "" {
1322
				b.WriteString(pendingWS)
1323
				pendingWS = ""
1324
			}
1325
			b.WriteString(t.Literal)
1326
			last = t.Span
1327
		}
1328
		p.advance()
1329
	}
1330
	if first.Start.Offset == 0 {
1331
		return "", token.Span{}
1332
	}
1333
	if single != "" {
1334
		return single, token.Span{File: first.File, Start: first.Start, End: last.End}
1335
	}
1336
	return strings.TrimSpace(b.String()), token.Span{File: first.File, Start: first.Start, End: last.End}
1337
}
1338
1339
func (p *Parser) parseTextComment() *ast.Comment {
1340
	s := p.cur.Span
1341
	marker := p.cur.Literal[0]
1342
	var b strings.Builder
1343
	b.WriteString(p.cur.Literal)
1344
	p.advance()
1345
	for !p.got(token.NEWLINE) && !p.got(token.EOF) {
1346
		b.WriteString(p.cur.Literal)
1347
		p.advance()
1348
	}
1349
	text := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(b.String()), string(marker)))
1350
	span := p.span(s) // marker to line end, without the newline
1351
	p.expectNewline()
1352
	return &ast.Comment{Marker: marker, Text: text, Span: span}
1353
}
1354
1355
func isDirectiveKeyword(t token.Type) bool {
1356
	switch t {
1357
	case token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE,
1358
		token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END,
1359
		token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C:
1360
		return true
1361
	}
1362
	return false
1363
}
1364
1365
func (p *Parser) sync() {
1366
	for {
1367
		switch p.cur.Type {
1368
		case token.EOF:
1369
			return
1370
		case token.NEWLINE:
1371
			p.advance()
1372
			t := p.cur.Type
1373
			if isDirectiveKeyword(t) || t == token.DATE || t == token.TILDE || t == token.EQ {
1374
				return
1375
			}
1376
		default:
1377
			p.advance()
1378
		}
1379
	}
1380
}
1381
1382
func (p *Parser) syncToNextline() {
1383
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF {
1384
		p.advance()
1385
	}
1386
	if p.got(token.NEWLINE) {
1387
		p.advance()
1388
	}
1389
}
1390
1391
func (p *Parser) skipWhitespace() {
1392
	for p.got(token.WHITESPACE) {
1393
		p.advance()
1394
	}
1395
}
1396
1397
func (p *Parser) span(s token.Span) token.Span {
1398
	return token.Span{File: s.File, Start: s.Start, End: p.cur.Span.Start}
1399
}
1400
1401
func normalizeLiteral(lit string, thousands, decimal byte) string {
1402
	var b strings.Builder
1403
	for _, ch := range []byte(lit) {
1404
		if thousands != 0 && ch == thousands {
1405
			continue // skip thousands separator
1406
		}
1407
		if ch == decimal {
1408
			b.WriteByte('.')
1409
		} else {
1410
			b.WriteByte(ch)
1411
		}
1412
	}
1413
	return b.String()
1414
}
1415
1416
func detectFormat(lit string) ast.QuantityFormat {
1417
	var seps []int
1418
	for i, ch := range []byte(lit) {
1419
		if ch == '.' || ch == ',' || ch == ' ' || ch == '_' || ch == '\'' {
1420
			seps = append(seps, i)
1421
		}
1422
	}
1423
1424
	if len(seps) == 0 {
1425
		return ast.QuantityFormat{}
1426
	}
1427
1428
	last := seps[len(seps)-1]
1429
	dec := lit[last]
1430
	if dec != '.' && dec != ',' {
1431
		// the last separator is a thousands mark; the literal has no decimal mark
1432
		dec = 0
1433
	}
1434
	var thou byte
1435
	if len(seps) > 1 {
1436
		thou = lit[seps[0]]
1437
	} else if dec == 0 {
1438
		// single space/underscore/apostrophe is always thousands
1439
		thou = lit[last]
1440
	}
1441
1442
	// calculate precision when the last separator is a real decimal
1443
	prec := 0
1444
	if thou == 0 || len(seps) > 1 {
1445
		prec = len(lit) - last - 1
1446
	}
1447
1448
	return ast.QuantityFormat{Decimal: dec, Thousands: thou, Precision: prec}
1449
}
1450
1451
// parseSimpleDate  parses full YYYY/MM/DD date literal embedded in free text.
1452
func parseSimpleDate(s string) ast.Date {
1453
	year, month, day, sep, err := ParseDateLiteral(s)
1454
	if err != nil {
1455
		return ast.Date{}
1456
	}
1457
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep}
1458
}
1459
1460
// ParseDateLiteral parses and validates a date literal.
1461
// It accepts full YYYY/MM/DD and partial MM/DD forms, with '-', '/' or '.' as separators.
1462
func ParseDateLiteral(lit string) (year, month, day int, sep byte, err error) {
1463
	sep = dateSeparator(lit)
1464
	if sep == 0 {
1465
		return 0, 0, 0, 0, fmt.Errorf("invalid date format: %q", lit)
1466
	}
1467
1468
	parts := strings.Split(lit, string(sep))
1469
	if len(parts) != 2 && len(parts) != 3 {
1470
		return 0, 0, 0, 0, fmt.Errorf("invalid date format: %q", lit)
1471
	}
1472
1473
	nums := make([]int, len(parts))
1474
	for i, part := range parts {
1475
		if nums[i], err = strconv.Atoi(part); err != nil {
1476
			return 0, 0, 0, 0, fmt.Errorf("invalid date literal: %q", lit)
1477
		}
1478
	}
1479
1480
	month = nums[len(parts)-2]
1481
	if month < 1 || month > 12 {
1482
		return 0, 0, 0, 0, fmt.Errorf("invalid month %d in %q", month, lit)
1483
	}
1484
1485
	day = nums[len(parts)-1]
1486
	if day < 1 || day > 31 {
1487
		return 0, 0, 0, 0, fmt.Errorf("invalid day %d in %q", day, lit)
1488
	}
1489
1490
	if len(parts) == 2 {
1491
		return 0, month, day, sep, nil
1492
	}
1493
	return nums[0], month, day, sep, nil
1494
}
1495
1496
func dateSeparator(lit string) byte {
1497
	for i := 0; i < len(lit); i++ {
1498
		if lit[i] == '/' || lit[i] == '-' || lit[i] == '.' {
1499
			return lit[i]
1500
		}
1501
	}
1502
	return 0
1503
}
1504
1505
// parseCommentTags extacts tags from comment text.
1506
// A tag is a word immediately followed by a ':', with an optional value that ends at a comma or the end of a line.
1507
// https://hledger.org/1.52/hledger.html?highlight=tags#tags
1508
func parseCommentTags(text string, base token.Span) []ast.Tag {
1509
	var tags []ast.Tag
1510
	for i := 0; i < len(text); {
1511
		colon := strings.IndexByte(text[i:], ':')
1512
		if colon < 0 {
1513
			break
1514
		}
1515
		colon += i
1516
1517
		keyStart := colon
1518
		for keyStart > i {
1519
			r, size := utf8.DecodeLastRuneInString(text[:keyStart])
1520
			if unicode.IsSpace(r) {
1521
				break
1522
			}
1523
			keyStart -= size
1524
		}
1525
		if keyStart == colon { // nothing before the colon = not a tag
1526
			i = colon + 1
1527
			continue
1528
		}
1529
		key := text[keyStart:colon]
1530
1531
		valueEnd := colon + 1
1532
		for valueEnd < len(text) && text[valueEnd] != ',' {
1533
			valueEnd++
1534
		}
1535
		value := strings.TrimSpace(text[colon+1 : valueEnd])
1536
1537
		tags = append(tags, ast.Tag{
1538
			Key:   key,
1539
			Value: value,
1540
			Span: token.Span{
1541
				File:  base.File,
1542
				Start: tagPos(base.Start, text, keyStart),
1543
				End:   tagPos(base.Start, text, valueEnd),
1544
			},
1545
		})
1546
		i = valueEnd
1547
		if i < len(text) && text[i] == ',' {
1548
			i++
1549
		}
1550
	}
1551
1552
	return tags
1553
}
1554
1555
func tagPos(base token.Pos, text string, off int) token.Pos {
1556
	return token.Pos{
1557
		Offset: base.Offset + off,
1558
		Line:   base.Line,
1559
		Col:    base.Col + utf8.RuneCountInString(text[:off]),
1560
	}
1561
}