all repos

clerk @ 6573acb6af84cdef8bdde568c92ebc1509eb600f

missing tooling for ledger/hledger

clerk/journal/parser/parser.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
ast: inline leaf value+span wrapper structs, 1 month ago
1
package parser
2
3
import (
4
	"fmt"
5
	"strconv"
6
	"strings"
7
	"unicode"
8
	"unicode/utf8"
9
10
	"olexsmir.xyz/clerk/internal/decimal"
11
	"olexsmir.xyz/clerk/journal/ast"
12
	"olexsmir.xyz/clerk/journal/lexer"
13
	"olexsmir.xyz/clerk/journal/token"
14
)
15
16
type Parser struct {
17
	lexer  *lexer.Lexer
18
	errors []*ast.ParseError
19
	cur    token.Token
20
	peek   token.Token
21
22
	defaultYear int // set by year directive, used for short date inference
23
}
24
25
func New(lex *lexer.Lexer) *Parser {
26
	p := &Parser{lexer: lex}
27
	p.advance() // populate .peek
28
	p.advance() // populate .cur
29
	return p
30
}
31
32
func NewWithYear(lex *lexer.Lexer, year int) *Parser {
33
	p := &Parser{lexer: lex, defaultYear: year}
34
	p.advance() // populate .peek
35
	p.advance() // populate .cur
36
	return p
37
}
38
39
func (p *Parser) ParseJournal() *ast.Journal {
40
	f := &ast.Journal{}
41
	for p.cur.Type != token.EOF {
42
		if e := p.parseEntry(); e != nil {
43
			f.Entries = append(f.Entries, e)
44
		}
45
	}
46
	f.Errors = p.errors
47
	return f
48
}
49
50
func (p *Parser) parseEntry() ast.Entry {
51
	if p.got(token.BANG) || p.got(token.AT) {
52
		if isDirectiveKeyword(p.peek.Type) {
53
			p.advance() // consume prefix
54
		}
55
	}
56
57
	switch p.cur.Type {
58
	case token.ILLEGAL:
59
		p.errorf("illegal character %q", p.cur.Literal)
60
		p.advance()
61
		return nil
62
	case token.INDENT:
63
		p.errorf("unexpected indent")
64
		p.syncToNextline()
65
		return nil
66
	case token.DATE:
67
		return p.parseTransaction()
68
	case token.TILDE:
69
		return p.parsePeriodicTransaction()
70
	case token.EQ:
71
		return p.parseAutomatedTransaction()
72
	case token.NEWLINE:
73
		return p.parseBlankLine()
74
	case token.SEMICOLON, token.HASH, token.PERCENT, token.STAR:
75
		return p.parseComment()
76
	case token.ACCOUNT:
77
		return p.parseAccountDirective()
78
	case token.COMMODITY:
79
		return p.parseCommodityDirective()
80
	case token.INCLUDE:
81
		return p.parseIncludeDirective()
82
	case token.ALIAS:
83
		return p.parseAliasDirective()
84
	case token.PAYEE:
85
		return p.parsePayeeDirective()
86
	case token.TAG:
87
		return p.parseTagDirective()
88
	case token.YEAR:
89
		return p.parseYearDirective()
90
	case token.DECIMALMARK:
91
		return p.parseDecimalMarkDirective()
92
	case token.D:
93
		return p.parseDefaultCommodityDirective()
94
	case token.P:
95
		return p.parseMarketPriceDirective()
96
	case token.N:
97
		return p.parseIgnoredDirective()
98
	case token.C:
99
		return p.parseConversionDirective()
100
	case token.APPLY:
101
		return p.parseApplyDirective()
102
	case token.END:
103
		return p.parseEndDirective()
104
	case token.COMMENTKW:
105
		return p.parseCommentBlockDirective()
106
	default:
107
		p.errorf("unexpected token %s", p.cur.Type)
108
		p.sync()
109
		return nil
110
	}
111
}
112
113
func (p *Parser) parseTransaction() *ast.Transaction {
114
	s := p.cur.Span
115
	tx := &ast.Transaction{}
116
117
	tx.Date = p.parseDate()
118
119
	p.skipWhitespace()
120
121
	// optional secondary date
122
	if p.got(token.EQ) {
123
		p.advance()
124
		p.skipWhitespace()
125
		d := p.parseDate()
126
		tx.SecondDate = &d
127
	}
128
129
	p.skipWhitespace()
130
131
	// optional status
132
	tx.Status, tx.StatusSpan = p.parseStatus()
133
134
	// optional code - the lexer emits "(CODE)" as a single TEXT token; split it here
135
	if p.got(token.TEXT) {
136
		if lit := p.cur.Literal; len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
137
			tx.Code = lit[1 : len(lit)-1]
138
			tx.CodeSpan = p.cur.Span
139
			p.advance()
140
			p.skipWhitespace()
141
		}
142
	}
143
144
	// optional payee | note
145
	if p.got(token.TEXT) || p.got(token.STRING) {
146
		tx.Payee, tx.PayeeSpan = p.parsePayee()
147
148
		// check for | separator
149
		p.skipWhitespace()
150
151
		if p.got(token.PIPE) {
152
			p.advance()
153
			if p.got(token.TEXT) {
154
				sn := p.cur.Span
155
				n := p.cur.Literal
156
				p.advance()
157
				tx.Note = n
158
				tx.NoteSpan = p.span(sn)
159
			}
160
		}
161
	}
162
163
	tx.Comment = p.parseOptInlineComment()
164
	p.expectNewline()
165
166
	tx.HeaderComments, tx.Postings = p.parseHeaderCommentsAndPostings()
167
168
	tx.Span = p.span(s)
169
	return tx
170
}
171
172
func unquote(s string) string {
173
	if len(s) >= 2 && ((s[0] == '"' && s[len(s)-1] == '"') || (s[0] == '\'' && s[len(s)-1] == '\'')) {
174
		return s[1 : len(s)-1]
175
	}
176
	return s
177
}
178
179
func (p *Parser) parsePayee() (string, token.Span) {
180
	s := p.cur.Span
181
182
	if p.got(token.STRING) {
183
		name := unquote(p.cur.Literal)
184
		p.advance()
185
		return name, p.span(s)
186
	}
187
188
	// keep spaces/tags between text tokens; stop before trailing whitespace
189
	var name strings.Builder
190
	for isPayeeWord(p.cur.Type) || (isPayeeWord(p.peek.Type) && p.got(token.WHITESPACE)) {
191
		_, _ = name.WriteString(p.cur.Literal)
192
		p.advance()
193
	}
194
	return unquote(name.String()), p.span(s)
195
}
196
197
func isPayeeWord(t token.Type) bool {
198
	switch t {
199
	case token.TEXT, token.INT, token.DECIMAL, token.COMMODITYMARK:
200
		return true
201
	}
202
	return false
203
}
204
205
func (p *Parser) parsePeriodicTransaction() *ast.PeriodicTransaction {
206
	s := p.cur.Span
207
	p.expect(token.TILDE)
208
	p.skipWhitespace()
209
210
	pt := &ast.PeriodicTransaction{}
211
212
	pt.Period = p.parsePeriod()
213
214
	if desc, dspan := p.parseOptPeriodicDescription(); desc != "" {
215
		pt.Description = desc
216
		pt.DescriptionSpan = dspan
217
	}
218
219
	comment := p.parseOptInlineComment()
220
	p.expectNewline()
221
222
	pt.HeaderComments, pt.Postings = p.parseHeaderCommentsAndPostings()
223
224
	pt.Span = p.span(s)
225
	pt.Comment = comment
226
	return pt
227
}
228
229
func (p *Parser) parseAutomatedTransaction() *ast.AutomatedTransaction {
230
	s := p.cur.Span
231
	p.expect(token.EQ)
232
	p.skipWhitespace()
233
234
	at := &ast.AutomatedTransaction{}
235
236
	// expression
237
	sd := p.cur.Span
238
	expr := p.parseDirectiveExpr()
239
	at.Expr = expr
240
	at.ExprSpan = p.span(sd)
241
	at.Comment = p.parseOptInlineComment()
242
	p.expectNewline()
243
244
	at.HeaderComments, at.Postings = p.parseHeaderCommentsAndPostings()
245
246
	at.Span = p.span(s)
247
	return at
248
}
249
250
func (p *Parser) parseHeaderCommentsAndPostings() (comments []*ast.Comment, postings []ast.Posting) {
251
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
252
		p.advance() // consume indent
253
		comments = append(comments, p.parseComment())
254
	}
255
256
	postings = make([]ast.Posting, 0, 2) // most transactions have 2 postings, small optimization
257
	for p.got(token.INDENT) {
258
		if posting, ok := p.parsePosting(); ok {
259
			postings = append(postings, posting)
260
		}
261
	}
262
263
	return comments, postings
264
}
265
266
func (p *Parser) parsePeriod() ast.Period {
267
	s := p.cur.Span
268
269
	var periodBuf strings.Builder
270
271
	for !p.got(token.NEWLINE) && !p.got(token.EOF) &&
272
		!p.got(token.SEMICOLON) && !p.got(token.HASH) && !p.got(token.PERCENT) && !p.got(token.STAR) {
273
274
		if p.got(token.WHITESPACE) {
275
			if len(p.cur.Literal) >= 2 {
276
				break
277
			}
278
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) ||
279
				p.willGet(token.SEMICOLON) || p.willGet(token.HASH) ||
280
				p.willGet(token.PERCENT) || p.willGet(token.STAR) {
281
				p.advance()
282
				continue
283
			}
284
		}
285
286
		periodBuf.WriteString(p.cur.Literal)
287
		p.advance()
288
	}
289
290
	str := periodBuf.String()
291
	period := ast.Period{Raw: str, Span: p.span(s)}
292
293
	if _, after, ok := strings.Cut(str, " from "); ok {
294
		end := strings.Index(after, " ")
295
		dateStr := after
296
		if end >= 0 {
297
			dateStr = after[:end]
298
		}
299
		if d := parseSimpleDate(dateStr); d.Year > 0 {
300
			fromOff := strings.Index(str, dateStr)
301
			d.Span = periodDateSpan(period, str, dateStr, fromOff)
302
			period.From = &d
303
			rest := after
304
			if end >= 0 {
305
				rest = after[end:]
306
			}
307
			if _, toAfter, ok := strings.Cut(rest, " to "); ok {
308
				if toEnd := strings.Index(toAfter, " "); toEnd >= 0 {
309
					toAfter = toAfter[:toEnd]
310
				}
311
				if d := parseSimpleDate(toAfter); d.Year > 0 {
312
					d.Span = periodDateSpan(period, str, toAfter, fromOff+len(dateStr))
313
					period.To = &d
314
				}
315
			}
316
		}
317
	}
318
	return period
319
}
320
321
// periodDateSpan returns the source span of dateStr, which occurs in the
322
// period text at or after searchFrom. The period span and text cover the same
323
// bytes, so offsets line up 1:1.
324
func periodDateSpan(period ast.Period, text, dateStr string, searchFrom int) token.Span {
325
	off := strings.Index(text[searchFrom:], dateStr)
326
	abs := period.Span.Start.Offset + searchFrom + off
327
	return token.Span{
328
		File:  period.Span.File,
329
		Start: token.Pos{Offset: abs},
330
		End:   token.Pos{Offset: abs + len(dateStr)},
331
	}
332
}
333
334
func (p *Parser) parseComment() *ast.Comment {
335
	s := p.cur.Span
336
	c := p.parseCommentRest(s)
337
	p.expectNewline()
338
	c.Span = p.span(s) // comment spans its line through the newline
339
	return c
340
}
341
342
func (p *Parser) parseAccountDirective() *ast.AccountDirective {
343
	s := p.cur.Span
344
	p.expect(token.ACCOUNT)
345
	p.skipWhitespace()
346
347
	account := p.parseAccount()
348
	comment := p.parseOptInlineComment()
349
	p.expectNewline()
350
351
	var subs []ast.AccountSubdirective
352
	for p.got(token.INDENT) {
353
		p.advance()
354
		p.skipWhitespace()
355
		if p.got(token.NEWLINE) || p.got(token.EOF) {
356
			// whitespace-only line: block continues
357
			p.expectNewline()
358
			continue
359
		}
360
		switch {
361
		case p.got(token.SEMICOLON):
362
			// comment line: directive mode lexes only ';' as a comment marker
363
			ns := p.cur.Span
364
			c := p.parseCommentRest(ns)
365
			p.expectNewline()
366
			subs = append(subs, ast.AccountSubdirective{Kind: ast.SubdirectiveComment, NameSpan: ns, Comment: c})
367
		case p.got(token.TEXT) && isCommentMarker(p.cur.Literal):
368
			// '#', '%' and '*' lex as TEXT in directive mode; treat them as comment lines
369
			c := p.parseTextComment()
370
			subs = append(subs, ast.AccountSubdirective{Kind: ast.SubdirectiveComment, NameSpan: c.Span, Comment: c})
371
		case p.got(token.TEXT):
372
			name := p.cur.Literal
373
			kind, ok := accountSubdirectiveKind(name)
374
			if !ok {
375
				p.errorf("unknown subdirective %q", name)
376
				p.skipToNewline()
377
				continue
378
			}
379
			kw := p.cur.Span
380
			p.advance()
381
			value, vspan := p.parseSubdirectiveValue()
382
			if value == "" {
383
				p.errorf("expected value for subdirective %q", name)
384
			}
385
			c := p.parseOptInlineComment()
386
			p.expectNewline()
387
			subs = append(subs, ast.AccountSubdirective{
388
				Kind:      kind,
389
				NameSpan:  kw,
390
				Value:     value,
391
				ValueSpan: vspan,
392
				Comment:   c,
393
			})
394
		default:
395
			p.errorf("expected subdirective name, got %s", p.cur.Type)
396
			p.skipToNewline()
397
		}
398
	}
399
400
	return &ast.AccountDirective{
401
		Account:       account,
402
		Subdirectives: subs,
403
		Comment:       comment,
404
		Span:          p.span(s),
405
	}
406
}
407
408
func (p *Parser) parseCommodityDirective() *ast.CommodityDirective {
409
	s := p.cur.Span
410
	p.expect(token.COMMODITY)
411
	p.skipWhitespace()
412
413
	var commodity string
414
	var commoditySpan token.Span
415
	var format *ast.FormatSubDirective
416
417
	switch p.cur.Type {
418
	case token.COMMODITYMARK, token.TEXT, token.STRING:
419
		cs := p.cur.Span
420
		commodity = unquote(p.cur.Literal)
421
		p.advance()
422
		commoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
423
		hadSpace := p.got(token.WHITESPACE)
424
		p.skipWhitespace()
425
		if p.got(token.INT) || p.got(token.DECIMAL) || p.got(token.TEXT) {
426
			amt := p.parseAmount()
427
			amt.Commodity = commodity
428
			amt.CommoditySpan = commoditySpan
429
			amt.CommodityPos = ast.CommodityBefore
430
			amt.HasSpace = hadSpace
431
			format = &ast.FormatSubDirective{Amount: amt}
432
		}
433
	case token.INT, token.DECIMAL:
434
		amt := p.parseAmount()
435
		commodity = amt.Commodity
436
		commoditySpan = amt.CommoditySpan
437
		format = &ast.FormatSubDirective{Amount: amt}
438
	default:
439
		p.errorf("expected commodity name or amount, got %s", p.cur.Type)
440
	}
441
442
	if commodity == "" {
443
		p.errorf("expected commodity name, got %s", p.cur.Type)
444
	}
445
446
	// hledger parity: an inline format amount must include a decimal mark
447
	if format != nil && format.Amount.QuantityFmt.Decimal == 0 {
448
		p.errorfAt(format.Amount.Span, "Please include a decimal point or decimal comma in commodity directives, to help us parse correctly. It may be followed by zero or more decimal digits.")
449
	}
450
451
	comment := p.parseOptInlineComment()
452
	p.expectNewline()
453
454
	var blockComments []*ast.Comment
455
	for p.got(token.INDENT) {
456
		p.advance()
457
		p.skipWhitespace()
458
		if p.got(token.NEWLINE) || p.got(token.EOF) {
459
			// whitespace-only line: block continues
460
			p.expectNewline()
461
			continue
462
		}
463
		switch {
464
		case p.got(token.TEXT) && p.cur.Literal == "format":
465
			kw := p.cur.Span
466
			p.advance()
467
			p.skipWhitespace()
468
			amt := p.parseAmount()
469
			// hledger parity: the format symbol must match the declared commodity,
470
			// and the amount must include a decimal mark; the node is kept either
471
			// way so the printer can round-trip the input.
472
			if amt.Commodity != commodity {
473
				p.errorfAt(amt.Span, "commodity directive symbol %q and format directive symbol %q should be the same", commodity, amt.Commodity)
474
			} else if amt.QuantityFmt.Decimal == 0 {
475
				p.errorfAt(amt.Span, "Please include a decimal point or decimal comma in commodity directives, to help us parse correctly. It may be followed by zero or more decimal digits.")
476
			}
477
			c := p.parseOptInlineComment()
478
			p.expectNewline()
479
			format = &ast.FormatSubDirective{KeywordSpan: kw, Amount: amt, Comment: c}
480
		case p.got(token.SEMICOLON): // comment line
481
			c := p.parseCommentRest(p.cur.Span)
482
			p.expectNewline()
483
			blockComments = append(blockComments, c)
484
		case p.got(token.TEXT) && isCommentMarker(p.cur.Literal):
485
			// '#', '%' and '*' lex as TEXT in directive mode; treat them as comment lines
486
			blockComments = append(blockComments, p.parseTextComment())
487
		case p.got(token.TEXT):
488
			p.errorf("unknown subdirective %q", p.cur.Literal)
489
			p.skipToNewline()
490
		default:
491
			p.errorf("expected subdirective name, got %s", p.cur.Type)
492
			p.skipToNewline()
493
		}
494
	}
495
496
	cd := &ast.CommodityDirective{
497
		Commodity:     commodity,
498
		CommoditySpan: commoditySpan,
499
		FormatSub:     format,
500
		BlockComments: blockComments,
501
		Comment:       comment,
502
		Span:          p.span(s),
503
	}
504
	return cd
505
}
506
507
func (p *Parser) parseIncludeDirective() *ast.IncludeDirective {
508
	s := p.cur.Span
509
	p.expect(token.INCLUDE)
510
	p.skipWhitespace()
511
512
	id := &ast.IncludeDirective{}
513
514
	if p.got(token.TEXT) {
515
		id.Path = p.cur.Literal
516
		p.advance()
517
	} else {
518
		p.errorf("expected file path, got %s", p.cur.Type)
519
	}
520
521
	id.Comment = p.parseOptInlineComment()
522
	p.expectNewline()
523
	id.Span = p.span(s)
524
	return id
525
}
526
527
func (p *Parser) parseAliasDirective() *ast.AliasDirective {
528
	s := p.cur.Span
529
	alias := &ast.AliasDirective{}
530
	p.expect(token.ALIAS)
531
	p.skipWhitespace()
532
	alias.From = p.parseAccount()
533
	p.skipWhitespace()
534
	p.expect(token.EQ)
535
	p.skipWhitespace()
536
	alias.To = p.parseAccount()
537
	alias.Comment = p.parseOptInlineComment()
538
	p.expectNewline()
539
	alias.Span = p.span(s)
540
	return alias
541
}
542
543
func (p *Parser) parsePayeeDirective() *ast.PayeeDirective {
544
	s := p.cur.Span
545
	p.expect(token.PAYEE)
546
	p.skipWhitespace()
547
548
	name, nameSpan := "", token.Span{}
549
	if p.got(token.TEXT) || p.got(token.STRING) || p.got(token.COMMODITYMARK) {
550
		name, nameSpan = p.parsePayee()
551
	}
552
553
	comment := p.parseOptInlineComment()
554
	p.expectNewline()
555
556
	return &ast.PayeeDirective{
557
		Name:     name,
558
		NameSpan: nameSpan,
559
		Comment:  comment,
560
		Span:     p.span(s),
561
	}
562
}
563
564
func (p *Parser) parseTagDirective() *ast.TagDirective {
565
	s := p.cur.Span
566
	p.expect(token.TAG)
567
	p.skipWhitespace()
568
569
	name := ""
570
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) || p.got(token.STRING) {
571
		name = unquote(p.cur.Literal)
572
		p.advance()
573
	}
574
575
	comment := p.parseOptInlineComment()
576
	p.expectNewline()
577
578
	return &ast.TagDirective{
579
		Name:    name,
580
		Comment: comment,
581
		Span:    p.span(s),
582
	}
583
}
584
585
func (p *Parser) parseYearDirective() *ast.YearDirective {
586
	s := p.cur.Span
587
	year := &ast.YearDirective{}
588
	p.expect(token.YEAR)
589
	p.skipWhitespace()
590
591
	if p.got(token.INT) {
592
		year.Year, _ = strconv.Atoi(p.cur.Literal)
593
		p.defaultYear = year.Year
594
		p.advance()
595
	} else {
596
		p.errorf("expected year, got %s", p.cur.Type)
597
	}
598
599
	year.Comment = p.parseOptInlineComment()
600
	p.expectNewline()
601
	year.Span = p.span(s)
602
603
	return year
604
}
605
606
func (p *Parser) parseDecimalMarkDirective() *ast.DecimalMarkDirective {
607
	s := p.cur.Span
608
	mark := &ast.DecimalMarkDirective{}
609
	p.expect(token.DECIMALMARK)
610
	p.skipWhitespace()
611
612
	mark.Mark = byte('.')
613
	if p.got(token.TEXT) {
614
		if len(p.cur.Literal) > 0 {
615
			mark.Mark = p.cur.Literal[0]
616
		}
617
		p.advance()
618
	}
619
620
	mark.Comment = p.parseOptInlineComment()
621
	p.expectNewline()
622
	mark.Span = p.span(s)
623
	return mark
624
}
625
626
func (p *Parser) parseDefaultCommodityDirective() *ast.DefaultCommodityDirective {
627
	s := p.cur.Span
628
	com := &ast.DefaultCommodityDirective{}
629
	p.expect(token.D)
630
	p.skipWhitespace()
631
	com.Amount = p.parseAmount()
632
	com.Comment = p.parseOptInlineComment()
633
	p.expectNewline()
634
	com.Span = p.span(s)
635
	return com
636
}
637
638
func (p *Parser) parseConversionDirective() *ast.ConversionDirective {
639
	s := p.cur.Span
640
	cd := &ast.ConversionDirective{}
641
	p.expect(token.C)
642
	p.skipWhitespace()
643
644
	if p.isAmountStart() {
645
		cd.From = p.parseAmount()
646
	} else {
647
		p.errorf("expected amount, got %s", p.cur.Type)
648
	}
649
650
	p.skipWhitespace()
651
	if p.got(token.EQ) {
652
		p.advance()
653
		p.skipWhitespace()
654
		if p.isAmountStart() {
655
			cd.To = p.parseAmount()
656
		} else {
657
			p.errorf("expected amount, got %s", p.cur.Type)
658
		}
659
	}
660
661
	cd.Comment = p.parseOptInlineComment()
662
	p.expectNewline()
663
	cd.Span = p.span(s)
664
	return cd
665
}
666
667
func (p *Parser) parseIgnoredDirective() *ast.IgnoredDirective {
668
	s := p.cur.Span
669
	p.expect(token.N)
670
	p.skipWhitespace()
671
672
	id := &ast.IgnoredDirective{}
673
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) || p.got(token.STRING) {
674
		id.Text = unquote(p.cur.Literal)
675
		p.advance()
676
	}
677
	id.Comment = p.parseOptInlineComment()
678
679
	p.expectNewline()
680
	id.Span = p.span(s)
681
	return id
682
}
683
684
func (p *Parser) parseMarketPriceDirective() *ast.MarketPriceDirective {
685
	s := p.cur.Span
686
	p.expect(token.P)
687
	p.skipWhitespace()
688
689
	mp := &ast.MarketPriceDirective{}
690
	mp.DateTime.Date = p.parseDate()
691
	p.skipWhitespace()
692
693
	if p.got(token.TIME) {
694
		mp.DateTime.Time = new(p.parseTime())
695
		p.skipWhitespace()
696
	}
697
698
	if p.got(token.COMMODITYMARK) || p.got(token.STRING) {
699
		mp.Commodity = unquote(p.cur.Literal)
700
		p.advance()
701
	} else {
702
		p.errorf("expected commodity symbol, got %s", p.cur.Type)
703
	}
704
	p.skipWhitespace()
705
706
	mp.Amount = p.parseAmount()
707
708
	mp.Comment = p.parseOptInlineComment()
709
710
	p.expectNewline()
711
	mp.Span = p.span(s)
712
	return mp
713
}
714
715
func (p *Parser) parseTime() ast.Time {
716
	s := p.cur.Span
717
	tok, _ := p.expect(token.TIME)
718
	lit := tok.Literal
719
720
	parts := strings.Split(lit, ":")
721
	if len(parts) < 2 {
722
		p.errorf("invalid time format: %q", lit)
723
		return ast.Time{Span: p.span(s)}
724
	}
725
726
	hour, _ := strconv.Atoi(parts[0])
727
	minute, _ := strconv.Atoi(parts[1])
728
	second := 0
729
	if len(parts) > 2 {
730
		second, _ = strconv.Atoi(parts[2])
731
	}
732
733
	if hour < 0 || hour > 23 {
734
		p.errorf("invalid hour %d in time %q", hour, lit)
735
	}
736
	if minute < 0 || minute > 59 {
737
		p.errorf("invalid minute %d in time %q", minute, lit)
738
	}
739
	if second < 0 || second > 59 {
740
		p.errorf("invalid second %d in time %q", second, lit)
741
	}
742
743
	return ast.Time{
744
		Hour:   hour,
745
		Minute: minute,
746
		Second: second,
747
		Span:   p.span(s),
748
	}
749
}
750
751
func (p *Parser) parseApplyDirective() *ast.ApplyDirective {
752
	s := p.cur.Span
753
	p.expect(token.APPLY)
754
	p.skipWhitespace()
755
756
	expr := p.parseDirectiveExpr()
757
	comment := p.parseOptInlineComment()
758
	p.expectNewline()
759
760
	return &ast.ApplyDirective{
761
		Expr:    expr,
762
		Comment: comment,
763
		Span:    p.span(s),
764
	}
765
}
766
767
func (p *Parser) parseEndDirective() *ast.EndDirective {
768
	s := p.cur.Span
769
	p.expect(token.END)
770
	p.skipWhitespace()
771
772
	expr := p.parseDirectiveExpr()
773
	comment := p.parseOptInlineComment()
774
	p.expectNewline()
775
776
	return &ast.EndDirective{
777
		Expr:    expr,
778
		Comment: comment,
779
		Span:    p.span(s),
780
	}
781
}
782
783
func (p *Parser) parseCommentBlockDirective() *ast.CommentBlockDirective {
784
	start := p.cur.Span
785
	p.expect(token.COMMENTKW)
786
	p.skipWhitespace()
787
788
	header := p.parseDirectiveExpr()
789
	comment := p.parseOptInlineComment()
790
	p.expectNewline()
791
792
	var content strings.Builder
793
	for p.cur.Type != token.EOF {
794
		if p.got(token.END) {
795
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) {
796
				p.advance()
797
				p.expectNewline()
798
				break
799
			}
800
			if p.willGet(token.WHITESPACE) {
801
				endTok := p.cur
802
				p.advance()
803
				wsTok := p.cur
804
				p.advance()
805
				if p.got(token.TEXT) && p.cur.Literal == "comment" { // todo: this should check if it's an actual COMMENTKW token
806
					p.advance()
807
					p.parseDirectiveExpr()
808
					p.parseOptInlineComment()
809
					p.expectNewline()
810
					break
811
				}
812
				content.WriteString(endTok.Literal)
813
				content.WriteString(wsTok.Literal)
814
				continue
815
			}
816
		}
817
		content.WriteString(p.cur.Literal)
818
		p.advance()
819
	}
820
821
	return &ast.CommentBlockDirective{
822
		Header:  header,
823
		Content: content.String(),
824
		Comment: comment,
825
		Span:    p.span(start),
826
	}
827
}
828
829
func (p *Parser) parseStatus() (ast.StatusType, token.Span) {
830
	s := p.cur.Span
831
	st := ast.StatusNone
832
	switch p.cur.Type {
833
	case token.STAR:
834
		st = ast.StatusCleared
835
	case token.BANG:
836
		st = ast.StatusPending
837
	}
838
	if st != ast.StatusNone {
839
		p.advance()
840
		p.skipWhitespace()
841
	}
842
	return st, p.span(s)
843
}
844
845
func (p *Parser) isAmountStart() bool {
846
	switch p.cur.Type {
847
	default:
848
		return false
849
	case token.COMMODITYMARK, token.STRING, token.INT, token.DECIMAL, token.MINUS, token.PLUS, token.PARENEXPR, token.STAR:
850
		return true
851
	}
852
}
853
854
func (p *Parser) parseAmount() ast.Amount {
855
	s := p.cur.Span
856
	amt := ast.Amount{QuantityFmt: ast.QuantityFormat{}}
857
858
	p.parseAmountSign(&amt)
859
	p.skipWhitespace()
860
861
	// commodity before quantity: $10.00, eur 10.00
862
	if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
863
		cs := p.cur.Span
864
		amt.Commodity = unquote(p.cur.Literal)
865
		amt.CommodityPos = ast.CommodityBefore
866
		p.advance()
867
		amt.CommoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
868
		if p.got(token.WHITESPACE) {
869
			amt.HasSpace = true
870
			p.skipWhitespace()
871
		}
872
	}
873
874
	// optional sign after commodity: $ -10
875
	p.parseAmountSign(&amt)
876
	p.skipWhitespace()
877
878
	p.parseQuantityInto(&amt)
879
880
	// commodity after quantity: 10.00 UAH, 10.00 "EUR" (only if not set)
881
	if amt.Commodity == "" {
882
		switch p.cur.Type {
883
		case token.WHITESPACE:
884
			p.skipWhitespace()
885
			if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
886
				cs := p.cur.Span
887
				amt.HasSpace = true
888
				amt.Commodity = unquote(p.cur.Literal)
889
				amt.CommodityPos = ast.CommodityAfter
890
				p.advance()
891
				amt.CommoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
892
			}
893
		case token.COMMODITYMARK, token.TEXT, token.STRING:
894
			cs := p.cur.Span
895
			amt.Commodity = unquote(p.cur.Literal)
896
			amt.CommodityPos = ast.CommodityAfter
897
			p.advance()
898
			amt.CommoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
899
		}
900
	}
901
902
	amt.Span = p.span(s)
903
	return amt
904
}
905
906
// parseAmountSign consumes an optional leading +/- into IsNegative.
907
func (p *Parser) parseAmountSign(amt *ast.Amount) {
908
	switch p.cur.Type {
909
	case token.MINUS:
910
		amt.IsNegative = true
911
		p.advance()
912
	case token.PLUS:
913
		p.advance()
914
	}
915
}
916
917
func (p *Parser) parseAmountWithOptExpr() ast.Amount {
918
	if p.got(token.STAR) {
919
		p.advance()
920
		p.skipWhitespace()
921
		amt := p.parseAmount()
922
		amt.IsExpr = true
923
		return amt
924
	}
925
	if p.got(token.PARENEXPR) {
926
		lit := p.cur.Literal
927
		amt := ast.Amount{
928
			IsExpr:      true,
929
			QuantityFmt: ast.QuantityFormat{},
930
		}
931
		if len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
932
			amt.Expr = strings.Trim(lit[1:len(lit)-1], " \t")
933
		}
934
		amt.Span = p.cur.Span
935
		p.advance()
936
		return amt
937
	}
938
	return p.parseAmount()
939
}
940
941
func (p *Parser) parsePosting() (ast.Posting, bool) {
942
	s := p.cur.Span
943
	posting := ast.Posting{}
944
	p.expect(token.INDENT)
945
946
	// exit if it's empty line
947
	if p.got(token.NEWLINE) || p.got(token.EOF) {
948
		p.syncToNextline()
949
		return ast.Posting{}, false
950
	}
951
952
	// optional status, outside of brackets, '! (account)'
953
	posting.Status, posting.StatusSpan = p.parseStatus()
954
955
	// detect virtual posting brackets
956
	switch p.cur.Type {
957
	case token.LPAREN:
958
		posting.Type = ast.PostingVirtualUnbalanced
959
		p.advance()
960
	case token.LBRACKET:
961
		posting.Type = ast.PostingVirtualBalanced
962
		p.advance()
963
	}
964
965
	// optional status, inside of brackets, '(* account)'
966
	if p.got(token.STAR) || p.got(token.BANG) {
967
		posting.Status, posting.StatusSpan = p.parseStatus()
968
	}
969
970
	// validate, must be account text
971
	if p.cur.Type != token.TEXT {
972
		p.errorf("expected account name, got %s", p.cur.Type)
973
		p.syncToNextline()
974
		return ast.Posting{}, false
975
	}
976
977
	posting.Account = p.parseAccount()
978
979
	// consume closing bracket
980
	switch p.cur.Type {
981
	case token.RPAREN:
982
		p.advance()
983
	case token.RBRACKET:
984
		p.advance()
985
	}
986
987
	// optional amount - after two spaces
988
	if p.got(token.WHITESPACE) {
989
		p.skipWhitespace()
990
		if p.isAmountStart() {
991
			amt := p.parseAmountWithOptExpr()
992
			posting.Amount = &amt
993
		}
994
	}
995
996
	// optional cost '@' or '@@'
997
	p.skipWhitespace()
998
	if p.got(token.AT) || p.got(token.ATAT) {
999
		posting.Cost = p.parseCost()
1000
	}
1001
1002
	// optional balance assertion or assignment
1003
	p.skipWhitespace()
1004
	if p.got(token.COLON) && p.willGet(token.EQ) {
1005
		p.advance() // consume ':' of ':='
1006
		posting.Balance = p.parseBalanceAssertion()
1007
		posting.Balance.IsAssignment = true
1008
	} else if p.got(token.EQ) || p.got(token.EQEQ) || p.got(token.EQEQEQ) || p.got(token.EQSTAR) {
1009
		posting.Balance = p.parseBalanceAssertion()
1010
	}
1011
1012
	posting.Comment = p.parseOptInlineComment()
1013
	p.expectNewline()
1014
1015
	// continuation comments
1016
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
1017
		p.advance()
1018
		c := p.parseComment()
1019
		posting.Comments = append(posting.Comments, *c)
1020
	}
1021
1022
	posting.Span = p.span(s)
1023
	return posting, true
1024
}
1025
1026
func (p *Parser) parseCost() *ast.Cost {
1027
	s := p.cur.Span
1028
	isTotal := p.got(token.ATAT)
1029
	p.advance() // consume '@' '@@'
1030
	p.skipWhitespace()
1031
	return &ast.Cost{
1032
		IsTotal: isTotal,
1033
		Amount:  p.parseAmount(),
1034
		Span:    p.span(s),
1035
	}
1036
}
1037
1038
func (p *Parser) parseBalanceAssertion() *ast.BalanceAssertion {
1039
	s := p.cur.Span
1040
1041
	ba := &ast.BalanceAssertion{}
1042
	switch p.cur.Type {
1043
	case token.EQ: // basic assertion
1044
	case token.EQSTAR: // inclusive assertion
1045
		ba.IsInclusive = true
1046
	case token.EQEQ: // strict assertion
1047
		ba.IsStrict = true
1048
	case token.EQEQEQ: // strict inclusive assertion
1049
		ba.IsStrict = true
1050
		ba.IsInclusive = true
1051
	}
1052
	p.advance()
1053
	p.skipWhitespace()
1054
1055
	ba.Amount = p.parseAmount()
1056
	p.skipWhitespace()
1057
	if p.got(token.AT) || p.got(token.ATAT) {
1058
		c := p.parseCost()
1059
		ba.Cost = c
1060
	}
1061
	ba.Span = p.span(s)
1062
	return ba
1063
}
1064
1065
func (p *Parser) readAccountSegment() (ast.SubAccount, bool) {
1066
	switch p.cur.Type {
1067
	case token.TEXT:
1068
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
1069
		p.advance()
1070
1071
		// handle multi work segment, e.g: "credit card"
1072
		if p.got(token.WHITESPACE) && p.willGet(token.TEXT) && len(p.peek.Literal) > 0 && p.peek.Literal[0] != '(' {
1073
			sub.Name += " "
1074
			p.advance()
1075
			sub.Name += p.cur.Literal
1076
			p.advance()
1077
		}
1078
		return sub, true
1079
1080
	case token.COMMODITYMARK:
1081
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
1082
		p.advance()
1083
		// merge "EUR" + "-HRK" to "EUR-HRK"
1084
		for p.got(token.TEXT) {
1085
			sub.Name += p.cur.Literal
1086
			p.advance()
1087
		}
1088
		return sub, true
1089
1090
	default:
1091
		return ast.SubAccount{}, false
1092
	}
1093
}
1094
1095
func (p *Parser) parseAccount() ast.Account {
1096
	s := p.cur.Span
1097
	acc := ast.Account{Name: make([]ast.SubAccount, 0, 6)}
1098
1099
	sub, ok := p.readAccountSegment()
1100
	if !ok {
1101
		p.errorf("expected account, got %s", p.cur.Type)
1102
		return ast.Account{}
1103
	}
1104
	acc.Name = append(acc.Name, sub)
1105
1106
	for p.got(token.COLON) {
1107
		p.advance()
1108
		sub, ok := p.readAccountSegment()
1109
		if !ok {
1110
			break
1111
		}
1112
		acc.Name = append(acc.Name, sub)
1113
	}
1114
1115
	acc.Span = p.span(s)
1116
	return acc
1117
}
1118
1119
func (p *Parser) parseDate() ast.Date {
1120
	s := p.cur.Span
1121
	tok, ok := p.expect(token.DATE)
1122
	if !ok {
1123
		return ast.Date{Span: p.span(s)}
1124
	}
1125
1126
	year, month, day, sep, err := ParseDateLiteral(tok.Literal)
1127
	if err != nil {
1128
		p.errorf("%v", err)
1129
		return ast.Date{Span: p.span(s)}
1130
	}
1131
	if year == 0 {
1132
		year = p.defaultYear
1133
	}
1134
1135
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep, Span: p.span(s)}
1136
}
1137
1138
func (p *Parser) parseOptInlineComment() *ast.Comment {
1139
	p.skipWhitespace()
1140
	if !p.got(token.SEMICOLON) {
1141
		return nil
1142
	}
1143
	return p.parseCommentRest(p.cur.Span)
1144
}
1145
1146
// parseCommentRest consumes a comment marker at p.cur, then optional text;
1147
// s anchors the span at the marker's start.
1148
func (p *Parser) parseCommentRest(s token.Span) *ast.Comment {
1149
	marker := p.cur.Literal[0]
1150
	p.advance()
1151
	p.skipWhitespace()
1152
1153
	var tags []ast.Tag
1154
	text := ""
1155
	if p.got(token.TEXT) {
1156
		text = p.cur.Literal
1157
		tags = parseCommentTags(text, p.cur.Span)
1158
		p.advance()
1159
	}
1160
1161
	return &ast.Comment{
1162
		Marker: marker,
1163
		Tags:   tags,
1164
		Text:   text,
1165
		Span:   p.span(s),
1166
	}
1167
}
1168
1169
func (p *Parser) parseOptPeriodicDescription() (string, token.Span) {
1170
	if p.cur.Type != token.WHITESPACE || len(p.cur.Literal) < 2 {
1171
		return "", token.Span{}
1172
	}
1173
1174
	p.skipWhitespace()
1175
1176
	if p.cur.Type != token.TEXT {
1177
		return "", token.Span{}
1178
	}
1179
1180
	s := p.cur.Span
1181
	desc := p.parseDescription()
1182
	return desc, p.span(s)
1183
}
1184
1185
func (p *Parser) parseDescription() string {
1186
	var desc strings.Builder
1187
	for p.got(token.TEXT) || (p.got(token.WHITESPACE) && p.willGet(token.TEXT)) {
1188
		_, _ = desc.WriteString(p.cur.Literal)
1189
		p.advance()
1190
	}
1191
	return desc.String()
1192
}
1193
1194
func (p *Parser) parseDirectiveExpr() string {
1195
	var b strings.Builder
1196
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF && p.cur.Type != token.SEMICOLON {
1197
		_, _ = b.WriteString(p.cur.Literal)
1198
		p.advance()
1199
	}
1200
	return b.String()
1201
}
1202
1203
func (p *Parser) parseQuantityInto(amt *ast.Amount) {
1204
	if p.cur.Type != token.INT && p.cur.Type != token.DECIMAL && p.cur.Type != token.TEXT {
1205
		p.errorf("expected quantity, got %s", p.cur.Type)
1206
		return
1207
	}
1208
1209
	lit := p.cur.Literal
1210
	p.advance()
1211
1212
	// detect format metadata before normalizing
1213
	amt.QuantityFmt = detectFormat(lit)
1214
1215
	// normalize for decimal.NewFromString
1216
	// remove thousands separators, replace decimal mark with '.'
1217
	normalized := normalizeLiteral(lit, amt.QuantityFmt.Thousands, amt.QuantityFmt.Decimal)
1218
1219
	q, err := decimal.FromString(normalized)
1220
	if err != nil {
1221
		p.errorf("invalid quantity %q: %v", lit, err)
1222
		return
1223
	}
1224
1225
	if amt.IsNegative {
1226
		q = q.Neg()
1227
	}
1228
	amt.Quantity = q
1229
}
1230
1231
func (p *Parser) parseBlankLine() *ast.BlankLine {
1232
	s := p.cur.Span
1233
	p.expectNewline()
1234
	return &ast.BlankLine{Span: s}
1235
}
1236
1237
func (p *Parser) expectNewline() {
1238
	if p.got(token.NEWLINE) || p.got(token.EOF) {
1239
		if p.got(token.NEWLINE) {
1240
			p.advance()
1241
		}
1242
		return
1243
	}
1244
	p.errorf("expected %s, got %s", token.NEWLINE, p.cur.Type)
1245
}
1246
1247
func (p *Parser) advance() token.Token {
1248
	prev := p.cur
1249
	p.cur = p.peek
1250
	p.peek = p.lexer.Next()
1251
	return prev
1252
}
1253
1254
func (p *Parser) got(kind token.Type) bool     { return p.cur.Type == kind }
1255
func (p *Parser) willGet(kind token.Type) bool { return p.peek.Type == kind }
1256
1257
func (p *Parser) expect(kind token.Type) (token.Token, bool) {
1258
	if p.got(kind) {
1259
		return p.advance(), true
1260
	}
1261
	p.errorf("expected %s, got %s", kind, p.cur.Type)
1262
	return p.cur, false
1263
}
1264
1265
func (p *Parser) errorf(format string, args ...any) {
1266
	p.errors = append(p.errors, &ast.ParseError{
1267
		Span:    p.cur.Span,
1268
		Message: fmt.Sprintf(format, args...),
1269
	})
1270
}
1271
1272
// errorfAt records a parse error pointing at the start of span.
1273
func (p *Parser) errorfAt(span token.Span, format string, args ...any) {
1274
	p.errors = append(p.errors, &ast.ParseError{
1275
		Span:    token.Span{File: span.File, Start: span.Start, End: span.Start},
1276
		Message: fmt.Sprintf(format, args...),
1277
	})
1278
}
1279
1280
func accountSubdirectiveKind(name string) (ast.AccountSubdirectiveKind, bool) {
1281
	switch name {
1282
	case "alias":
1283
		return ast.SubdirectiveAlias, true
1284
	case "type":
1285
		return ast.SubdirectiveType, true
1286
	case "note":
1287
		return ast.SubdirectiveNote, true
1288
	}
1289
	return 0, false
1290
}
1291
1292
func isCommentMarker(s string) bool {
1293
	return len(s) > 0 && (s[0] == '#' || s[0] == '%' || s[0] == '*')
1294
}
1295
1296
// skipToNewline consumes the rest of the current line.
1297
func (p *Parser) skipToNewline() {
1298
	for !p.got(token.NEWLINE) && !p.got(token.EOF) {
1299
		p.advance()
1300
	}
1301
	p.expectNewline()
1302
}
1303
1304
func (p *Parser) parseSubdirectiveValue() (string, token.Span) {
1305
	var b strings.Builder
1306
	var first, last token.Span
1307
	var single, pendingWS string
1308
	for !p.got(token.SEMICOLON) && !p.got(token.NEWLINE) && !p.got(token.EOF) {
1309
		t := p.cur
1310
		if t.Type == token.WHITESPACE || t.Type == token.INDENT {
1311
			if first.Start.Offset > 0 {
1312
				pendingWS = t.Literal // only the last whitespace run matters
1313
			}
1314
			p.advance()
1315
			continue
1316
		}
1317
		if first.Start.Offset == 0 {
1318
			first, last = t.Span, t.Span
1319
			single = t.Literal
1320
		} else {
1321
			if single != "" {
1322
				b.WriteString(single)
1323
				single = ""
1324
			}
1325
			if pendingWS != "" {
1326
				b.WriteString(pendingWS)
1327
				pendingWS = ""
1328
			}
1329
			b.WriteString(t.Literal)
1330
			last = t.Span
1331
		}
1332
		p.advance()
1333
	}
1334
	if first.Start.Offset == 0 {
1335
		return "", token.Span{}
1336
	}
1337
	if single != "" {
1338
		return single, token.Span{File: first.File, Start: first.Start, End: last.End}
1339
	}
1340
	return strings.TrimSpace(b.String()), token.Span{File: first.File, Start: first.Start, End: last.End}
1341
}
1342
1343
func (p *Parser) parseTextComment() *ast.Comment {
1344
	s := p.cur.Span
1345
	marker := p.cur.Literal[0]
1346
	var b strings.Builder
1347
	b.WriteString(p.cur.Literal)
1348
	p.advance()
1349
	for !p.got(token.NEWLINE) && !p.got(token.EOF) {
1350
		b.WriteString(p.cur.Literal)
1351
		p.advance()
1352
	}
1353
	text := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(b.String()), string(marker)))
1354
	span := p.span(s) // marker to line end, without the newline
1355
	p.expectNewline()
1356
	return &ast.Comment{Marker: marker, Text: text, Span: span}
1357
}
1358
1359
func isDirectiveKeyword(t token.Type) bool {
1360
	switch t {
1361
	case token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE,
1362
		token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END,
1363
		token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C:
1364
		return true
1365
	}
1366
	return false
1367
}
1368
1369
func (p *Parser) sync() {
1370
	for {
1371
		switch p.cur.Type {
1372
		case token.EOF:
1373
			return
1374
		case token.NEWLINE:
1375
			p.advance()
1376
			t := p.cur.Type
1377
			if isDirectiveKeyword(t) || t == token.DATE || t == token.TILDE || t == token.EQ {
1378
				return
1379
			}
1380
		default:
1381
			p.advance()
1382
		}
1383
	}
1384
}
1385
1386
func (p *Parser) syncToNextline() {
1387
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF {
1388
		p.advance()
1389
	}
1390
	if p.got(token.NEWLINE) {
1391
		p.advance()
1392
	}
1393
}
1394
1395
func (p *Parser) skipWhitespace() {
1396
	for p.got(token.WHITESPACE) {
1397
		p.advance()
1398
	}
1399
}
1400
1401
func (p *Parser) span(s token.Span) token.Span {
1402
	return token.Span{File: s.File, Start: s.Start, End: p.cur.Span.Start}
1403
}
1404
1405
func normalizeLiteral(lit string, thousands, decimal byte) string {
1406
	// fast path: no separators to strip and the decimal mark is already '.'
1407
	if thousands == 0 && (decimal == 0 || decimal == '.') {
1408
		return lit
1409
	}
1410
	var b strings.Builder
1411
	for _, ch := range []byte(lit) {
1412
		if thousands != 0 && ch == thousands {
1413
			continue // skip thousands separator
1414
		}
1415
		if ch == decimal {
1416
			b.WriteByte('.')
1417
		} else {
1418
			b.WriteByte(ch)
1419
		}
1420
	}
1421
	return b.String()
1422
}
1423
1424
func detectFormat(lit string) ast.QuantityFormat {
1425
	var seps []int
1426
	for i, ch := range []byte(lit) {
1427
		if ch == '.' || ch == ',' || ch == ' ' || ch == '_' || ch == '\'' {
1428
			seps = append(seps, i)
1429
		}
1430
	}
1431
1432
	if len(seps) == 0 {
1433
		return ast.QuantityFormat{}
1434
	}
1435
1436
	last := seps[len(seps)-1]
1437
	dec := lit[last]
1438
	if dec != '.' && dec != ',' {
1439
		// the last separator is a thousands mark; the literal has no decimal mark
1440
		dec = 0
1441
	}
1442
	var thou byte
1443
	if len(seps) > 1 {
1444
		thou = lit[seps[0]]
1445
	} else if dec == 0 {
1446
		// single space/underscore/apostrophe is always thousands
1447
		thou = lit[last]
1448
	}
1449
1450
	// calculate precision when the last separator is a real decimal
1451
	prec := 0
1452
	if thou == 0 || len(seps) > 1 {
1453
		prec = len(lit) - last - 1
1454
	}
1455
1456
	return ast.QuantityFormat{Decimal: dec, Thousands: thou, Precision: prec}
1457
}
1458
1459
// parseSimpleDate  parses full YYYY/MM/DD date literal embedded in free text.
1460
func parseSimpleDate(s string) ast.Date {
1461
	year, month, day, sep, err := ParseDateLiteral(s)
1462
	if err != nil {
1463
		return ast.Date{}
1464
	}
1465
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep}
1466
}
1467
1468
// ParseDateLiteral parses and validates a date literal.
1469
// It accepts full YYYY/MM/DD and partial MM/DD forms, with '-', '/' or '.' as separators.
1470
func ParseDateLiteral(lit string) (year, month, day int, sep byte, err error) {
1471
	sep = dateSeparator(lit)
1472
	if sep == 0 {
1473
		return 0, 0, 0, 0, fmt.Errorf("invalid date format: %q", lit)
1474
	}
1475
1476
	parts := strings.Split(lit, string(sep))
1477
	if len(parts) != 2 && len(parts) != 3 {
1478
		return 0, 0, 0, 0, fmt.Errorf("invalid date format: %q", lit)
1479
	}
1480
1481
	nums := make([]int, len(parts))
1482
	for i, part := range parts {
1483
		if nums[i], err = strconv.Atoi(part); err != nil {
1484
			return 0, 0, 0, 0, fmt.Errorf("invalid date literal: %q", lit)
1485
		}
1486
	}
1487
1488
	month = nums[len(parts)-2]
1489
	if month < 1 || month > 12 {
1490
		return 0, 0, 0, 0, fmt.Errorf("invalid month %d in %q", month, lit)
1491
	}
1492
1493
	day = nums[len(parts)-1]
1494
	if day < 1 || day > 31 {
1495
		return 0, 0, 0, 0, fmt.Errorf("invalid day %d in %q", day, lit)
1496
	}
1497
1498
	if len(parts) == 2 {
1499
		return 0, month, day, sep, nil
1500
	}
1501
	return nums[0], month, day, sep, nil
1502
}
1503
1504
func dateSeparator(lit string) byte {
1505
	for i := 0; i < len(lit); i++ {
1506
		if lit[i] == '/' || lit[i] == '-' || lit[i] == '.' {
1507
			return lit[i]
1508
		}
1509
	}
1510
	return 0
1511
}
1512
1513
// parseCommentTags extacts tags from comment text.
1514
// A tag is a word immediately followed by a ':', with an optional value that ends at a comma or the end of a line.
1515
// https://hledger.org/1.52/hledger.html?highlight=tags#tags
1516
func parseCommentTags(text string, base token.Span) []ast.Tag {
1517
	var tags []ast.Tag
1518
	for i := 0; i < len(text); {
1519
		colon := strings.IndexByte(text[i:], ':')
1520
		if colon < 0 {
1521
			break
1522
		}
1523
		colon += i
1524
1525
		keyStart := colon
1526
		for keyStart > i {
1527
			r, size := utf8.DecodeLastRuneInString(text[:keyStart])
1528
			if unicode.IsSpace(r) {
1529
				break
1530
			}
1531
			keyStart -= size
1532
		}
1533
		if keyStart == colon { // nothing before the colon = not a tag
1534
			i = colon + 1
1535
			continue
1536
		}
1537
		key := text[keyStart:colon]
1538
1539
		valueEnd := colon + 1
1540
		for valueEnd < len(text) && text[valueEnd] != ',' {
1541
			valueEnd++
1542
		}
1543
		value := strings.TrimSpace(text[colon+1 : valueEnd])
1544
1545
		tags = append(tags, ast.Tag{
1546
			Key:   key,
1547
			Value: value,
1548
			Span: token.Span{
1549
				File:  base.File,
1550
				Start: tagPos(base.Start, text, keyStart),
1551
				End:   tagPos(base.Start, text, valueEnd),
1552
			},
1553
		})
1554
		i = valueEnd
1555
		if i < len(text) && text[i] == ',' {
1556
			i++
1557
		}
1558
	}
1559
1560
	return tags
1561
}
1562
1563
func tagPos(base token.Pos, text string, off int) token.Pos {
1564
	return token.Pos{
1565
		Offset: base.Offset + off,
1566
		Line:   base.Line,
1567
		Col:    base.Col + utf8.RuneCountInString(text[:off]),
1568
	}
1569
}