all repos

clerk @ 78d7e88

missing tooling for ledger/hledger

clerk/journal/parser/parser.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
parser: some refactoring, 29 days ago
1
package parser
2
3
import (
4
	"fmt"
5
	"strconv"
6
	"strings"
7
	"unicode"
8
	"unicode/utf8"
9
10
	"olexsmir.xyz/clerk/internal/decimal"
11
	"olexsmir.xyz/clerk/journal/ast"
12
	"olexsmir.xyz/clerk/journal/lexer"
13
	"olexsmir.xyz/clerk/journal/token"
14
)
15
16
type Parser struct {
17
	lexer  *lexer.Lexer
18
	errors []*ast.ParseError
19
	cur    token.Token
20
	peek   token.Token
21
22
	defaultYear int // set by year directive, used for short date inference
23
}
24
25
func New(lex *lexer.Lexer) *Parser {
26
	p := &Parser{lexer: lex}
27
	p.advance() // populate .peek
28
	p.advance() // populate .cur
29
	return p
30
}
31
32
func NewWithYear(lex *lexer.Lexer, year int) *Parser {
33
	p := &Parser{lexer: lex, defaultYear: year}
34
	p.advance() // populate .peek
35
	p.advance() // populate .cur
36
	return p
37
}
38
39
func (p *Parser) ParseJournal() *ast.Journal {
40
	f := &ast.Journal{}
41
	for p.cur.Type != token.EOF {
42
		if e := p.parseEntry(); e != nil {
43
			f.Entries = append(f.Entries, e)
44
		}
45
	}
46
	f.Errors = p.errors
47
	return f
48
}
49
50
func (p *Parser) parseEntry() ast.Entry {
51
	if p.got(token.BANG) || p.got(token.AT) {
52
		if isDirectiveKeyword(p.peek.Type) {
53
			p.advance() // consume prefix
54
		}
55
	}
56
57
	switch p.cur.Type {
58
	case token.ILLEGAL:
59
		p.errorf("illegal character %q", p.cur.Literal)
60
		p.advance()
61
		return nil
62
	case token.INDENT:
63
		p.errorf("unexpected indent")
64
		p.syncToNextline()
65
		return nil
66
	case token.DATE:
67
		return p.parseTransaction()
68
	case token.TILDE:
69
		return p.parsePeriodicTransaction()
70
	case token.EQ:
71
		return p.parseAutomatedTransaction()
72
	case token.NEWLINE:
73
		return p.parseBlankLine()
74
	case token.SEMICOLON, token.HASH, token.PERCENT, token.STAR:
75
		return p.parseComment()
76
	case token.ACCOUNT:
77
		return p.parseAccountDirective()
78
	case token.COMMODITY:
79
		return p.parseCommodityDirective()
80
	case token.INCLUDE:
81
		return p.parseIncludeDirective()
82
	case token.ALIAS:
83
		return p.parseAliasDirective()
84
	case token.PAYEE:
85
		return p.parsePayeeDirective()
86
	case token.TAG:
87
		return p.parseTagDirective()
88
	case token.YEAR:
89
		return p.parseYearDirective()
90
	case token.DECIMALMARK:
91
		return p.parseDecimalMarkDirective()
92
	case token.D:
93
		return p.parseDefaultCommodityDirective()
94
	case token.P:
95
		return p.parseMarketPriceDirective()
96
	case token.N:
97
		return p.parseIgnoredDirective()
98
	case token.C:
99
		return p.parseConversionDirective()
100
	case token.APPLY:
101
		return p.parseApplyDirective()
102
	case token.END:
103
		return p.parseEndDirective()
104
	case token.COMMENTKW:
105
		return p.parseCommentBlockDirective()
106
	default:
107
		p.errorf("unexpected token %s", p.cur.Type)
108
		p.sync()
109
		return nil
110
	}
111
}
112
113
func (p *Parser) parseTransaction() *ast.Transaction {
114
	s := p.cur.Span
115
	tx := &ast.Transaction{}
116
117
	tx.Date = p.parseDate()
118
119
	p.skipWhitespace()
120
121
	// optional secondary date
122
	if p.got(token.EQ) {
123
		p.advance()
124
		p.skipWhitespace()
125
		d := p.parseDate()
126
		tx.SecondDate = &d
127
	}
128
129
	p.skipWhitespace()
130
131
	// optional status
132
	tx.Status, tx.StatusSpan = p.parseStatus()
133
134
	// optional code - the lexer emits "(CODE)" as a single TEXT token; split it here
135
	if p.got(token.TEXT) {
136
		if lit := p.cur.Literal; len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
137
			tx.Code = lit[1 : len(lit)-1]
138
			tx.CodeSpan = p.cur.Span
139
			p.advance()
140
			p.skipWhitespace()
141
		}
142
	}
143
144
	// optional payee | note
145
	if p.got(token.TEXT) || p.got(token.STRING) {
146
		tx.Payee, tx.PayeeSpan = p.parsePayee()
147
148
		// check for | separator
149
		p.skipWhitespace()
150
151
		if p.got(token.PIPE) {
152
			p.advance()
153
			if p.got(token.TEXT) {
154
				sn := p.cur.Span
155
				n := p.cur.Literal
156
				p.advance()
157
				tx.Note = n
158
				tx.NoteSpan = p.span(sn)
159
			}
160
		}
161
	}
162
163
	tx.Comment = p.parseOptInlineComment()
164
	p.expectNewline()
165
166
	tx.HeaderComments, tx.Postings = p.parseHeaderCommentsAndPostings()
167
168
	tx.Span = p.span(s)
169
	return tx
170
}
171
172
func unquote(s string) string {
173
	if len(s) >= 2 && ((s[0] == '"' && s[len(s)-1] == '"') || (s[0] == '\'' && s[len(s)-1] == '\'')) {
174
		return s[1 : len(s)-1]
175
	}
176
	return s
177
}
178
179
func (p *Parser) parsePayee() (string, token.Span) {
180
	s := p.cur.Span
181
182
	if p.got(token.STRING) {
183
		name := unquote(p.cur.Literal)
184
		p.advance()
185
		return name, p.span(s)
186
	}
187
188
	// keep spaces/tags between text tokens; stop before trailing whitespace
189
	var name strings.Builder
190
	for isPayeeWord(p.cur.Type) || (isPayeeWord(p.peek.Type) && p.got(token.WHITESPACE)) {
191
		_, _ = name.WriteString(p.cur.Literal)
192
		p.advance()
193
	}
194
	return unquote(name.String()), p.span(s)
195
}
196
197
func isPayeeWord(t token.Type) bool {
198
	switch t {
199
	case token.TEXT, token.INT, token.DECIMAL, token.COMMODITYMARK:
200
		return true
201
	}
202
	return false
203
}
204
205
func (p *Parser) parsePeriodicTransaction() *ast.PeriodicTransaction {
206
	s := p.cur.Span
207
	p.expect(token.TILDE)
208
	p.skipWhitespace()
209
210
	pt := &ast.PeriodicTransaction{}
211
	pt.Period = p.parsePeriod()
212
	if desc, dspan := p.parseOptPeriodicDescription(); desc != "" {
213
		pt.Description = desc
214
		pt.DescriptionSpan = dspan
215
	}
216
217
	pt.Comment = p.parseOptInlineComment()
218
	p.expectNewline()
219
220
	pt.HeaderComments, pt.Postings = p.parseHeaderCommentsAndPostings()
221
	pt.Span = p.span(s)
222
	return pt
223
}
224
225
func (p *Parser) parseAutomatedTransaction() *ast.AutomatedTransaction {
226
	s := p.cur.Span
227
	p.expect(token.EQ)
228
	p.skipWhitespace()
229
230
	at := &ast.AutomatedTransaction{}
231
232
	// expression
233
	expSpan := p.cur.Span
234
	expr := p.parseDirectiveExpr()
235
	at.Expr = expr
236
	at.ExprSpan = p.span(expSpan)
237
	at.Comment = p.parseOptInlineComment()
238
	p.expectNewline()
239
240
	at.HeaderComments, at.Postings = p.parseHeaderCommentsAndPostings()
241
242
	at.Span = p.span(s)
243
	return at
244
}
245
246
func (p *Parser) parseHeaderCommentsAndPostings() (comments []*ast.Comment, postings []ast.Posting) {
247
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
248
		p.advance() // consume indent
249
		comments = append(comments, p.parseComment())
250
	}
251
252
	postings = make([]ast.Posting, 0, 2) // most transactions have 2 postings, small optimization
253
	for p.got(token.INDENT) {
254
		if posting, ok := p.parsePosting(); ok {
255
			postings = append(postings, posting)
256
		}
257
	}
258
259
	return comments, postings
260
}
261
262
func (p *Parser) parsePeriod() ast.Period {
263
	s := p.cur.Span
264
265
	var periodBuf strings.Builder
266
	for !p.got(token.NEWLINE) && !p.got(token.EOF) &&
267
		!p.got(token.SEMICOLON) && !p.got(token.HASH) && !p.got(token.PERCENT) && !p.got(token.STAR) {
268
269
		if p.got(token.WHITESPACE) {
270
			if len(p.cur.Literal) >= 2 {
271
				break
272
			}
273
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) ||
274
				p.willGet(token.SEMICOLON) || p.willGet(token.HASH) ||
275
				p.willGet(token.PERCENT) || p.willGet(token.STAR) {
276
				p.advance()
277
				continue
278
			}
279
		}
280
281
		periodBuf.WriteString(p.cur.Literal)
282
		p.advance()
283
	}
284
285
	str := periodBuf.String()
286
	period := ast.Period{Raw: str, Span: p.span(s)}
287
	if _, after, ok := strings.Cut(str, " from "); ok {
288
		end := strings.Index(after, " ")
289
		dateStr := after
290
		if end >= 0 {
291
			dateStr = after[:end]
292
		}
293
		if d := parseSimpleDate(dateStr); d.Year > 0 {
294
			fromOff := strings.Index(str, dateStr)
295
			d.Span = periodDateSpan(period, str, dateStr, fromOff)
296
			period.From = &d
297
			rest := after
298
			if end >= 0 {
299
				rest = after[end:]
300
			}
301
			if _, toAfter, ok := strings.Cut(rest, " to "); ok {
302
				if toEnd := strings.Index(toAfter, " "); toEnd >= 0 {
303
					toAfter = toAfter[:toEnd]
304
				}
305
				if d := parseSimpleDate(toAfter); d.Year > 0 {
306
					d.Span = periodDateSpan(period, str, toAfter, fromOff+len(dateStr))
307
					period.To = &d
308
				}
309
			}
310
		}
311
	}
312
	return period
313
}
314
315
func periodDateSpan(period ast.Period, text, dateStr string, searchFrom int) token.Span {
316
	off := strings.Index(text[searchFrom:], dateStr)
317
	abs := period.Span.Start.Offset + searchFrom + off
318
	return token.Span{
319
		File:  period.Span.File,
320
		Start: token.Pos{Offset: abs},
321
		End:   token.Pos{Offset: abs + len(dateStr)},
322
	}
323
}
324
325
func (p *Parser) parseComment() *ast.Comment {
326
	s := p.cur.Span
327
	c := p.parseCommentRest(s)
328
	p.expectNewline()
329
	c.Span = p.span(s) // comment spans its line through the newline
330
	return c
331
}
332
333
func (p *Parser) parseAccountDirective() *ast.AccountDirective {
334
	s := p.cur.Span
335
	p.expect(token.ACCOUNT)
336
	p.skipWhitespace()
337
338
	account := p.parseAccount()
339
	comment := p.parseOptInlineComment()
340
	p.expectNewline()
341
342
	var subs []ast.AccountSubdirective
343
	for p.got(token.INDENT) {
344
		p.advance()
345
		p.skipWhitespace()
346
		if p.got(token.NEWLINE) || p.got(token.EOF) {
347
			// whitespace-only line: block continues
348
			p.expectNewline()
349
			continue
350
		}
351
		switch {
352
		case p.got(token.SEMICOLON):
353
			// comment line: directive mode lexes only ';' as a comment marker
354
			ns := p.cur.Span
355
			c := p.parseCommentRest(ns)
356
			p.expectNewline()
357
			subs = append(subs, ast.AccountSubdirective{Kind: ast.SubdirectiveComment, NameSpan: ns, Comment: c})
358
		case p.got(token.TEXT) && isCommentMarker(p.cur.Literal):
359
			// '#', '%' and '*' lex as TEXT in directive mode; treat them as comment lines
360
			c := p.parseTextComment()
361
			subs = append(subs, ast.AccountSubdirective{Kind: ast.SubdirectiveComment, NameSpan: c.Span, Comment: c})
362
		case p.got(token.TEXT):
363
			name := p.cur.Literal
364
			kind, ok := accountSubdirectiveKind(name)
365
			if !ok {
366
				p.errorf("unknown subdirective %q", name)
367
				p.skipToNewline()
368
				continue
369
			}
370
			kw := p.cur.Span
371
			p.advance()
372
			value, vspan := p.parseSubdirectiveValue()
373
			if value == "" {
374
				p.errorf("expected value for subdirective %q", name)
375
			}
376
			c := p.parseOptInlineComment()
377
			p.expectNewline()
378
			subs = append(subs, ast.AccountSubdirective{
379
				Kind:      kind,
380
				NameSpan:  kw,
381
				Value:     value,
382
				ValueSpan: vspan,
383
				Comment:   c,
384
			})
385
		default:
386
			p.errorf("expected subdirective name, got %s", p.cur.Type)
387
			p.skipToNewline()
388
		}
389
	}
390
391
	return &ast.AccountDirective{
392
		Account:       account,
393
		Subdirectives: subs,
394
		Comment:       comment,
395
		Span:          p.span(s),
396
	}
397
}
398
399
func (p *Parser) parseCommodityDirective() *ast.CommodityDirective {
400
	s := p.cur.Span
401
	p.expect(token.COMMODITY)
402
	p.skipWhitespace()
403
404
	var commodity string
405
	var commoditySpan token.Span
406
	var format *ast.FormatSubDirective
407
408
	switch p.cur.Type {
409
	case token.COMMODITYMARK, token.TEXT, token.STRING:
410
		cs := p.cur.Span
411
		commodity = unquote(p.cur.Literal)
412
		p.advance()
413
		commoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
414
		hadSpace := p.got(token.WHITESPACE)
415
		p.skipWhitespace()
416
		if p.got(token.INT) || p.got(token.DECIMAL) || p.got(token.TEXT) {
417
			amt := p.parseAmount()
418
			amt.Commodity = commodity
419
			amt.CommoditySpan = commoditySpan
420
			amt.CommodityPos = ast.CommodityBefore
421
			amt.HasSpace = hadSpace
422
			format = &ast.FormatSubDirective{Amount: amt}
423
		}
424
	case token.INT, token.DECIMAL:
425
		amt := p.parseAmount()
426
		commodity = amt.Commodity
427
		commoditySpan = amt.CommoditySpan
428
		format = &ast.FormatSubDirective{Amount: amt}
429
	default:
430
		p.errorf("expected commodity name or amount, got %s", p.cur.Type)
431
	}
432
433
	if commodity == "" {
434
		p.errorf("expected commodity name, got %s", p.cur.Type)
435
	}
436
437
	// hledger parity: an inline format amount must include a decimal mark
438
	if format != nil && format.Amount.QuantityFmt.Decimal == 0 {
439
		p.errorfAt(format.Amount.Span, "Please include a decimal point or decimal comma in commodity directives, to help us parse correctly. It may be followed by zero or more decimal digits.")
440
	}
441
442
	comment := p.parseOptInlineComment()
443
	p.expectNewline()
444
445
	var blockComments []*ast.Comment
446
	for p.got(token.INDENT) {
447
		p.advance()
448
		p.skipWhitespace()
449
		if p.got(token.NEWLINE) || p.got(token.EOF) {
450
			// whitespace-only line: block continues
451
			p.expectNewline()
452
			continue
453
		}
454
		switch {
455
		case p.got(token.TEXT) && p.cur.Literal == "format":
456
			kw := p.cur.Span
457
			p.advance()
458
			p.skipWhitespace()
459
			amt := p.parseAmount()
460
			// hledger parity: the format symbol must match the declared commodity,
461
			// and the amount must include a decimal mark; the node is kept either
462
			// way so the printer can round-trip the input.
463
			if amt.Commodity != commodity {
464
				p.errorfAt(amt.Span, "commodity directive symbol %q and format directive symbol %q should be the same", commodity, amt.Commodity)
465
			} else if amt.QuantityFmt.Decimal == 0 {
466
				p.errorfAt(amt.Span, "Please include a decimal point or decimal comma in commodity directives, to help us parse correctly. It may be followed by zero or more decimal digits.")
467
			}
468
			c := p.parseOptInlineComment()
469
			p.expectNewline()
470
			format = &ast.FormatSubDirective{KeywordSpan: kw, Amount: amt, Comment: c}
471
		case p.got(token.SEMICOLON): // comment line
472
			c := p.parseCommentRest(p.cur.Span)
473
			p.expectNewline()
474
			blockComments = append(blockComments, c)
475
		case p.got(token.TEXT) && isCommentMarker(p.cur.Literal):
476
			blockComments = append(blockComments, p.parseTextComment())
477
		case p.got(token.TEXT):
478
			p.errorf("unknown subdirective %q", p.cur.Literal)
479
			p.skipToNewline()
480
		default:
481
			p.errorf("expected subdirective name, got %s", p.cur.Type)
482
			p.skipToNewline()
483
		}
484
	}
485
486
	return &ast.CommodityDirective{
487
		Commodity:     commodity,
488
		CommoditySpan: commoditySpan,
489
		FormatSub:     format,
490
		BlockComments: blockComments,
491
		Comment:       comment,
492
		Span:          p.span(s),
493
	}
494
}
495
496
func (p *Parser) parseIncludeDirective() *ast.IncludeDirective {
497
	s := p.cur.Span
498
	p.expect(token.INCLUDE)
499
	p.skipWhitespace()
500
501
	id := &ast.IncludeDirective{}
502
	if p.got(token.TEXT) {
503
		id.Path = p.cur.Literal
504
		p.advance()
505
	} else {
506
		p.errorf("expected file path, got %s", p.cur.Type)
507
	}
508
	id.Comment = p.parseOptInlineComment()
509
	p.expectNewline()
510
	id.Span = p.span(s)
511
	return id
512
}
513
514
func (p *Parser) parseAliasDirective() *ast.AliasDirective {
515
	s := p.cur.Span
516
	alias := &ast.AliasDirective{}
517
	p.expect(token.ALIAS)
518
	p.skipWhitespace()
519
	alias.From = p.parseAccount()
520
	p.skipWhitespace()
521
	p.expect(token.EQ)
522
	p.skipWhitespace()
523
	alias.To = p.parseAccount()
524
	alias.Comment = p.parseOptInlineComment()
525
	p.expectNewline()
526
	alias.Span = p.span(s)
527
	return alias
528
}
529
530
func (p *Parser) parsePayeeDirective() *ast.PayeeDirective {
531
	s := p.cur.Span
532
	p.expect(token.PAYEE)
533
	p.skipWhitespace()
534
535
	pd := &ast.PayeeDirective{}
536
	if p.got(token.TEXT) || p.got(token.STRING) || p.got(token.COMMODITYMARK) {
537
		pd.Name, pd.NameSpan = p.parsePayee()
538
	}
539
	pd.Comment = p.parseOptInlineComment()
540
	p.expectNewline()
541
	pd.Span = p.span(s)
542
	return pd
543
}
544
545
func (p *Parser) parseTagDirective() *ast.TagDirective {
546
	s := p.cur.Span
547
	p.expect(token.TAG)
548
	p.skipWhitespace()
549
550
	td := &ast.TagDirective{}
551
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) || p.got(token.STRING) {
552
		td.Name = unquote(p.cur.Literal)
553
		p.advance()
554
	}
555
	td.Comment = p.parseOptInlineComment()
556
	p.expectNewline()
557
	td.Span = p.span(s)
558
	return td
559
}
560
561
func (p *Parser) parseYearDirective() *ast.YearDirective {
562
	s := p.cur.Span
563
	year := &ast.YearDirective{}
564
	p.expect(token.YEAR)
565
	p.skipWhitespace()
566
567
	if p.got(token.INT) {
568
		year.Year, _ = strconv.Atoi(p.cur.Literal)
569
		p.defaultYear = year.Year
570
		p.advance()
571
	} else {
572
		p.errorf("expected year, got %s", p.cur.Type)
573
	}
574
575
	year.Comment = p.parseOptInlineComment()
576
	p.expectNewline()
577
	year.Span = p.span(s)
578
579
	return year
580
}
581
582
func (p *Parser) parseDecimalMarkDirective() *ast.DecimalMarkDirective {
583
	s := p.cur.Span
584
	mark := &ast.DecimalMarkDirective{}
585
	p.expect(token.DECIMALMARK)
586
	p.skipWhitespace()
587
588
	mark.Mark = byte('.')
589
	if p.got(token.TEXT) {
590
		if len(p.cur.Literal) > 0 {
591
			mark.Mark = p.cur.Literal[0]
592
		}
593
		p.advance()
594
	}
595
596
	mark.Comment = p.parseOptInlineComment()
597
	p.expectNewline()
598
	mark.Span = p.span(s)
599
	return mark
600
}
601
602
func (p *Parser) parseDefaultCommodityDirective() *ast.DefaultCommodityDirective {
603
	s := p.cur.Span
604
	com := &ast.DefaultCommodityDirective{}
605
	p.expect(token.D)
606
	p.skipWhitespace()
607
	com.Amount = p.parseAmount()
608
	com.Comment = p.parseOptInlineComment()
609
	p.expectNewline()
610
	com.Span = p.span(s)
611
	return com
612
}
613
614
func (p *Parser) parseConversionDirective() *ast.ConversionDirective {
615
	s := p.cur.Span
616
	cd := &ast.ConversionDirective{}
617
	p.expect(token.C)
618
	p.skipWhitespace()
619
620
	if p.isAmountStart() {
621
		cd.From = p.parseAmount()
622
	} else {
623
		p.errorf("expected amount, got %s", p.cur.Type)
624
	}
625
626
	p.skipWhitespace()
627
	if p.got(token.EQ) {
628
		p.advance()
629
		p.skipWhitespace()
630
		if p.isAmountStart() {
631
			cd.To = p.parseAmount()
632
		} else {
633
			p.errorf("expected amount, got %s", p.cur.Type)
634
		}
635
	}
636
637
	cd.Comment = p.parseOptInlineComment()
638
	p.expectNewline()
639
	cd.Span = p.span(s)
640
	return cd
641
}
642
643
func (p *Parser) parseIgnoredDirective() *ast.IgnoredDirective {
644
	s := p.cur.Span
645
	p.expect(token.N)
646
	p.skipWhitespace()
647
648
	id := &ast.IgnoredDirective{}
649
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) || p.got(token.STRING) {
650
		id.Text = unquote(p.cur.Literal)
651
		p.advance()
652
	}
653
	id.Comment = p.parseOptInlineComment()
654
655
	p.expectNewline()
656
	id.Span = p.span(s)
657
	return id
658
}
659
660
func (p *Parser) parseMarketPriceDirective() *ast.MarketPriceDirective {
661
	s := p.cur.Span
662
	p.expect(token.P)
663
	p.skipWhitespace()
664
665
	mp := &ast.MarketPriceDirective{}
666
	mp.DateTime.Date = p.parseDate()
667
	p.skipWhitespace()
668
669
	if p.got(token.TIME) {
670
		mp.DateTime.Time = new(p.parseTime())
671
		p.skipWhitespace()
672
	}
673
674
	if p.got(token.COMMODITYMARK) || p.got(token.STRING) {
675
		mp.Commodity = unquote(p.cur.Literal)
676
		p.advance()
677
	} else {
678
		p.errorf("expected commodity symbol, got %s", p.cur.Type)
679
	}
680
	p.skipWhitespace()
681
682
	mp.Amount = p.parseAmount()
683
684
	mp.Comment = p.parseOptInlineComment()
685
686
	p.expectNewline()
687
	mp.Span = p.span(s)
688
	return mp
689
}
690
691
func (p *Parser) parseTime() ast.Time {
692
	s := p.cur.Span
693
	tok, _ := p.expect(token.TIME)
694
	lit := tok.Literal
695
696
	parts := strings.Split(lit, ":")
697
	if len(parts) < 2 {
698
		p.errorf("invalid time format: %q", lit)
699
		return ast.Time{Span: p.span(s)}
700
	}
701
702
	hour, _ := strconv.Atoi(parts[0])
703
	minute, _ := strconv.Atoi(parts[1])
704
	second := 0
705
	if len(parts) > 2 {
706
		second, _ = strconv.Atoi(parts[2])
707
	}
708
709
	if hour < 0 || hour > 23 {
710
		p.errorf("invalid hour %d in time %q", hour, lit)
711
	}
712
	if minute < 0 || minute > 59 {
713
		p.errorf("invalid minute %d in time %q", minute, lit)
714
	}
715
	if second < 0 || second > 59 {
716
		p.errorf("invalid second %d in time %q", second, lit)
717
	}
718
719
	return ast.Time{
720
		Hour:   hour,
721
		Minute: minute,
722
		Second: second,
723
		Span:   p.span(s),
724
	}
725
}
726
727
func (p *Parser) parseApplyDirective() *ast.ApplyDirective {
728
	s := p.cur.Span
729
	p.expect(token.APPLY)
730
	p.skipWhitespace()
731
732
	expr := p.parseDirectiveExpr()
733
	comment := p.parseOptInlineComment()
734
	p.expectNewline()
735
736
	return &ast.ApplyDirective{
737
		Expr:    expr,
738
		Comment: comment,
739
		Span:    p.span(s),
740
	}
741
}
742
743
func (p *Parser) parseEndDirective() *ast.EndDirective {
744
	s := p.cur.Span
745
	p.expect(token.END)
746
	p.skipWhitespace()
747
748
	expr := p.parseDirectiveExpr()
749
	comment := p.parseOptInlineComment()
750
	p.expectNewline()
751
752
	return &ast.EndDirective{
753
		Expr:    expr,
754
		Comment: comment,
755
		Span:    p.span(s),
756
	}
757
}
758
759
func (p *Parser) parseCommentBlockDirective() *ast.CommentBlockDirective {
760
	start := p.cur.Span
761
	p.expect(token.COMMENTKW)
762
	p.skipWhitespace()
763
764
	header := p.parseDirectiveExpr()
765
	comment := p.parseOptInlineComment()
766
	p.expectNewline()
767
768
	var content strings.Builder
769
	for p.cur.Type != token.EOF {
770
		if p.got(token.END) {
771
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) {
772
				p.advance()
773
				p.expectNewline()
774
				break
775
			}
776
			if p.willGet(token.WHITESPACE) {
777
				endTok := p.cur
778
				p.advance()
779
				wsTok := p.cur
780
				p.advance()
781
				if p.got(token.TEXT) && p.cur.Literal == "comment" { // todo: this should check if it's an actual COMMENTKW token
782
					p.advance()
783
					p.parseDirectiveExpr()
784
					p.parseOptInlineComment()
785
					p.expectNewline()
786
					break
787
				}
788
				content.WriteString(endTok.Literal)
789
				content.WriteString(wsTok.Literal)
790
				continue
791
			}
792
		}
793
		content.WriteString(p.cur.Literal)
794
		p.advance()
795
	}
796
797
	return &ast.CommentBlockDirective{
798
		Header:  header,
799
		Content: content.String(),
800
		Comment: comment,
801
		Span:    p.span(start),
802
	}
803
}
804
805
func (p *Parser) parseStatus() (ast.StatusType, token.Span) {
806
	s := p.cur.Span
807
	st := ast.StatusNone
808
	switch p.cur.Type {
809
	case token.STAR:
810
		st = ast.StatusCleared
811
	case token.BANG:
812
		st = ast.StatusPending
813
	}
814
	if st != ast.StatusNone {
815
		sp := p.cur.Span
816
		p.advance()
817
		p.skipWhitespace()
818
		return st, sp
819
	}
820
	return st, p.span(s)
821
}
822
823
func (p *Parser) isAmountStart() bool {
824
	switch p.cur.Type {
825
	default:
826
		return false
827
	case token.COMMODITYMARK, token.STRING, token.INT, token.DECIMAL, token.MINUS, token.PLUS, token.PARENEXPR, token.STAR:
828
		return true
829
	}
830
}
831
832
func (p *Parser) parseAmount() ast.Amount {
833
	s := p.cur.Span
834
	amt := ast.Amount{QuantityFmt: ast.QuantityFormat{}}
835
836
	p.parseAmountSign(&amt)
837
	p.skipWhitespace()
838
839
	// commodity before quantity: $10.00, eur 10.00
840
	if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
841
		cs := p.cur.Span
842
		amt.Commodity = unquote(p.cur.Literal)
843
		amt.CommodityPos = ast.CommodityBefore
844
		p.advance()
845
		amt.CommoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
846
		if p.got(token.WHITESPACE) {
847
			amt.HasSpace = true
848
			p.skipWhitespace()
849
		}
850
	}
851
852
	// optional sign after commodity: $ -10
853
	p.parseAmountSign(&amt)
854
	p.skipWhitespace()
855
856
	p.parseQuantityInto(&amt)
857
858
	// commodity after quantity: 10.00 UAH, 10.00 "EUR" (only if not set)
859
	if amt.Commodity == "" {
860
		switch p.cur.Type {
861
		case token.WHITESPACE:
862
			p.skipWhitespace()
863
			if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
864
				cs := p.cur.Span
865
				amt.HasSpace = true
866
				amt.Commodity = unquote(p.cur.Literal)
867
				amt.CommodityPos = ast.CommodityAfter
868
				p.advance()
869
				amt.CommoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
870
			}
871
		case token.COMMODITYMARK, token.TEXT, token.STRING:
872
			cs := p.cur.Span
873
			amt.Commodity = unquote(p.cur.Literal)
874
			amt.CommodityPos = ast.CommodityAfter
875
			p.advance()
876
			amt.CommoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
877
		}
878
	}
879
880
	amt.Span = p.span(s)
881
	return amt
882
}
883
884
// parseAmountSign consumes an optional leading +/- into IsNegative.
885
func (p *Parser) parseAmountSign(amt *ast.Amount) {
886
	switch p.cur.Type {
887
	case token.MINUS:
888
		amt.IsNegative = true
889
		p.advance()
890
	case token.PLUS:
891
		p.advance()
892
	}
893
}
894
895
func (p *Parser) parseAmountWithOptExpr() ast.Amount {
896
	if p.got(token.STAR) {
897
		p.advance()
898
		p.skipWhitespace()
899
		amt := p.parseAmount()
900
		amt.IsExpr = true
901
		return amt
902
	}
903
	if p.got(token.PARENEXPR) {
904
		lit := p.cur.Literal
905
		amt := ast.Amount{
906
			IsExpr:      true,
907
			QuantityFmt: ast.QuantityFormat{},
908
		}
909
		if len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
910
			amt.Expr = strings.Trim(lit[1:len(lit)-1], " \t")
911
		}
912
		amt.Span = p.cur.Span
913
		p.advance()
914
		return amt
915
	}
916
	return p.parseAmount()
917
}
918
919
func (p *Parser) parsePosting() (ast.Posting, bool) {
920
	s := p.cur.Span
921
	posting := ast.Posting{}
922
	p.expect(token.INDENT)
923
924
	// exit if it's empty line
925
	if p.got(token.NEWLINE) || p.got(token.EOF) {
926
		p.syncToNextline()
927
		return ast.Posting{}, false
928
	}
929
930
	// optional status, outside of brackets, '! (account)'
931
	posting.Status, posting.StatusSpan = p.parseStatus()
932
933
	// detect virtual posting brackets
934
	switch p.cur.Type {
935
	case token.LPAREN:
936
		posting.Type = ast.PostingVirtualUnbalanced
937
		p.advance()
938
	case token.LBRACKET:
939
		posting.Type = ast.PostingVirtualBalanced
940
		p.advance()
941
	}
942
943
	// optional status, inside of brackets, '(* account)'
944
	if p.got(token.STAR) || p.got(token.BANG) {
945
		posting.Status, posting.StatusSpan = p.parseStatus()
946
	}
947
948
	// validate, must be account text
949
	if p.cur.Type != token.TEXT {
950
		p.errorf("expected account name, got %s", p.cur.Type)
951
		p.syncToNextline()
952
		return ast.Posting{}, false
953
	}
954
955
	posting.Account = p.parseAccount()
956
957
	// consume closing bracket
958
	switch p.cur.Type {
959
	case token.RPAREN:
960
		p.advance()
961
	case token.RBRACKET:
962
		p.advance()
963
	}
964
965
	// optional amount - after two spaces
966
	if p.got(token.WHITESPACE) {
967
		p.skipWhitespace()
968
		if p.isAmountStart() {
969
			amt := p.parseAmountWithOptExpr()
970
			posting.Amount = &amt
971
		}
972
	}
973
974
	// optional cost '@' or '@@'
975
	p.skipWhitespace()
976
	if p.got(token.AT) || p.got(token.ATAT) {
977
		posting.Cost = p.parseCost()
978
	}
979
980
	// optional balance assertion or assignment
981
	p.skipWhitespace()
982
	if p.got(token.COLON) && p.willGet(token.EQ) {
983
		p.advance() // consume ':' of ':='
984
		posting.Balance = p.parseBalanceAssertion()
985
		posting.Balance.IsAssignment = true
986
	} else if p.got(token.EQ) || p.got(token.EQEQ) || p.got(token.EQEQEQ) || p.got(token.EQSTAR) {
987
		posting.Balance = p.parseBalanceAssertion()
988
	}
989
990
	posting.Comment = p.parseOptInlineComment()
991
	p.expectNewline()
992
993
	// continuation comments
994
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
995
		p.advance()
996
		c := p.parseComment()
997
		posting.Comments = append(posting.Comments, *c)
998
	}
999
1000
	posting.Span = p.span(s)
1001
	return posting, true
1002
}
1003
1004
func (p *Parser) parseCost() *ast.Cost {
1005
	s := p.cur.Span
1006
	isTotal := p.got(token.ATAT)
1007
	p.advance() // consume '@' '@@'
1008
	p.skipWhitespace()
1009
	return &ast.Cost{
1010
		IsTotal: isTotal,
1011
		Amount:  p.parseAmount(),
1012
		Span:    p.span(s),
1013
	}
1014
}
1015
1016
func (p *Parser) parseBalanceAssertion() *ast.BalanceAssertion {
1017
	s := p.cur.Span
1018
1019
	ba := &ast.BalanceAssertion{}
1020
	switch p.cur.Type {
1021
	case token.EQ: // basic assertion
1022
	case token.EQSTAR: // inclusive assertion
1023
		ba.IsInclusive = true
1024
	case token.EQEQ: // strict assertion
1025
		ba.IsStrict = true
1026
	case token.EQEQEQ: // strict inclusive assertion
1027
		ba.IsStrict = true
1028
		ba.IsInclusive = true
1029
	}
1030
	p.advance()
1031
	p.skipWhitespace()
1032
1033
	ba.Amount = p.parseAmount()
1034
	p.skipWhitespace()
1035
	if p.got(token.AT) || p.got(token.ATAT) {
1036
		c := p.parseCost()
1037
		ba.Cost = c
1038
	}
1039
	ba.Span = p.span(s)
1040
	return ba
1041
}
1042
1043
func (p *Parser) readAccountSegment() (ast.SubAccount, bool) {
1044
	switch p.cur.Type {
1045
	case token.TEXT:
1046
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
1047
		p.advance()
1048
1049
		// handle multi work segment, e.g: "credit card"
1050
		if p.got(token.WHITESPACE) && p.willGet(token.TEXT) && len(p.peek.Literal) > 0 && p.peek.Literal[0] != '(' {
1051
			sub.Name += " "
1052
			p.advance()
1053
			sub.Name += p.cur.Literal
1054
			p.advance()
1055
		}
1056
		return sub, true
1057
1058
	case token.COMMODITYMARK:
1059
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
1060
		p.advance()
1061
		// merge "EUR" + "-HRK" to "EUR-HRK"
1062
		for p.got(token.TEXT) {
1063
			sub.Name += p.cur.Literal
1064
			p.advance()
1065
		}
1066
		return sub, true
1067
1068
	default:
1069
		return ast.SubAccount{}, false
1070
	}
1071
}
1072
1073
func (p *Parser) parseAccount() ast.Account {
1074
	s := p.cur.Span
1075
	acc := ast.Account{Name: make([]ast.SubAccount, 0, 6)}
1076
1077
	seg, ok := p.readAccountSegment()
1078
	if !ok {
1079
		p.errorf("expected account, got %s", p.cur.Type)
1080
		return ast.Account{}
1081
	}
1082
	acc.Name = append(acc.Name, seg)
1083
1084
	for p.got(token.COLON) {
1085
		p.advance()
1086
		seg, ok := p.readAccountSegment()
1087
		if !ok {
1088
			break
1089
		}
1090
		acc.Name = append(acc.Name, seg)
1091
	}
1092
1093
	acc.Span = p.span(s)
1094
	return acc
1095
}
1096
1097
func (p *Parser) parseDate() ast.Date {
1098
	s := p.cur.Span
1099
	tok, ok := p.expect(token.DATE)
1100
	if !ok {
1101
		return ast.Date{Span: p.span(s)}
1102
	}
1103
1104
	year, month, day, sep, err := ParseDateLiteral(tok.Literal)
1105
	if err != nil {
1106
		p.errorf("%v", err)
1107
		return ast.Date{Span: p.span(s)}
1108
	}
1109
	if year == 0 {
1110
		year = p.defaultYear
1111
	}
1112
1113
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep, Span: p.span(s)}
1114
}
1115
1116
func (p *Parser) parseOptInlineComment() *ast.Comment {
1117
	p.skipWhitespace()
1118
	if !p.got(token.SEMICOLON) {
1119
		return nil
1120
	}
1121
	return p.parseCommentRest(p.cur.Span)
1122
}
1123
1124
func (p *Parser) parseCommentRest(s token.Span) *ast.Comment {
1125
	c := &ast.Comment{}
1126
	c.Marker = p.cur.Literal[0]
1127
	p.advance()
1128
	p.skipWhitespace()
1129
	if p.got(token.TEXT) {
1130
		c.Text = p.cur.Literal
1131
		c.Tags = parseCommentTags(c.Text, p.cur.Span)
1132
		p.advance()
1133
	}
1134
	c.Span = p.span(s)
1135
	return c
1136
}
1137
1138
func (p *Parser) parseOptPeriodicDescription() (string, token.Span) {
1139
	if p.cur.Type != token.WHITESPACE || len(p.cur.Literal) < 2 {
1140
		return "", token.Span{}
1141
	}
1142
1143
	p.skipWhitespace()
1144
	if p.cur.Type != token.TEXT {
1145
		return "", token.Span{}
1146
	}
1147
1148
	s := p.cur.Span
1149
	desc := p.parseDescription()
1150
	return desc, p.span(s)
1151
}
1152
1153
func (p *Parser) parseDescription() string {
1154
	var desc strings.Builder
1155
	for p.got(token.TEXT) || (p.got(token.WHITESPACE) && p.willGet(token.TEXT)) {
1156
		_, _ = desc.WriteString(p.cur.Literal)
1157
		p.advance()
1158
	}
1159
	return desc.String()
1160
}
1161
1162
func (p *Parser) parseDirectiveExpr() string {
1163
	var b strings.Builder
1164
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF && p.cur.Type != token.SEMICOLON {
1165
		_, _ = b.WriteString(p.cur.Literal)
1166
		p.advance()
1167
	}
1168
	return b.String()
1169
}
1170
1171
func (p *Parser) parseQuantityInto(amt *ast.Amount) {
1172
	if p.cur.Type != token.INT && p.cur.Type != token.DECIMAL && p.cur.Type != token.TEXT {
1173
		p.errorf("expected quantity, got %s", p.cur.Type)
1174
		return
1175
	}
1176
1177
	lit := p.cur.Literal
1178
	p.advance()
1179
1180
	amt.QuantityFmt = detectFormat(lit)
1181
	normalized := normalizeLiteral(lit, amt.QuantityFmt.Thousands, amt.QuantityFmt.Decimal)
1182
	q, err := decimal.FromString(normalized)
1183
	if err != nil {
1184
		p.errorf("invalid quantity %q: %v", lit, err)
1185
		return
1186
	}
1187
1188
	if amt.IsNegative {
1189
		q = q.Neg()
1190
	}
1191
	amt.Quantity = q
1192
}
1193
1194
func (p *Parser) parseBlankLine() *ast.BlankLine {
1195
	s := p.cur.Span
1196
	p.expectNewline()
1197
	return &ast.BlankLine{Span: s}
1198
}
1199
1200
func (p *Parser) expectNewline() {
1201
	if p.got(token.NEWLINE) || p.got(token.EOF) {
1202
		if p.got(token.NEWLINE) {
1203
			p.advance()
1204
		}
1205
		return
1206
	}
1207
	p.errorf("expected %s, got %s", token.NEWLINE, p.cur.Type)
1208
}
1209
1210
func (p *Parser) advance() token.Token {
1211
	prev := p.cur
1212
	p.cur = p.peek
1213
	p.peek = p.lexer.Next()
1214
	return prev
1215
}
1216
1217
func (p *Parser) got(kind token.Type) bool     { return p.cur.Type == kind }
1218
func (p *Parser) willGet(kind token.Type) bool { return p.peek.Type == kind }
1219
func (p *Parser) expect(kind token.Type) (token.Token, bool) {
1220
	if p.got(kind) {
1221
		return p.advance(), true
1222
	}
1223
	p.errorf("expected %s, got %s", kind, p.cur.Type)
1224
	return p.cur, false
1225
}
1226
1227
func (p *Parser) errorf(format string, args ...any) {
1228
	p.errors = append(p.errors, &ast.ParseError{
1229
		Span:    p.cur.Span,
1230
		Message: fmt.Sprintf(format, args...),
1231
	})
1232
}
1233
1234
func (p *Parser) errorfAt(span token.Span, format string, args ...any) {
1235
	p.errors = append(p.errors, &ast.ParseError{
1236
		Span:    token.Span{File: span.File, Start: span.Start, End: span.Start},
1237
		Message: fmt.Sprintf(format, args...),
1238
	})
1239
}
1240
1241
func accountSubdirectiveKind(name string) (ast.AccountSubdirectiveKind, bool) {
1242
	switch name {
1243
	case "alias":
1244
		return ast.SubdirectiveAlias, true
1245
	case "type":
1246
		return ast.SubdirectiveType, true
1247
	case "note":
1248
		return ast.SubdirectiveNote, true
1249
	}
1250
	return 0, false
1251
}
1252
1253
func isCommentMarker(s string) bool {
1254
	return len(s) > 0 && (s[0] == '#' || s[0] == '%' || s[0] == '*')
1255
}
1256
1257
// skipToNewline consumes the rest of the current line.
1258
func (p *Parser) skipToNewline() {
1259
	for !p.got(token.NEWLINE) && !p.got(token.EOF) {
1260
		p.advance()
1261
	}
1262
	p.expectNewline()
1263
}
1264
1265
func (p *Parser) parseSubdirectiveValue() (string, token.Span) {
1266
	var b strings.Builder
1267
	var first, last token.Span
1268
	var single, pendingWS string
1269
	for !p.got(token.SEMICOLON) && !p.got(token.NEWLINE) && !p.got(token.EOF) {
1270
		t := p.cur
1271
		if t.Type == token.WHITESPACE || t.Type == token.INDENT {
1272
			if first.Start.Offset > 0 {
1273
				pendingWS = t.Literal // only the last whitespace run matters
1274
			}
1275
			p.advance()
1276
			continue
1277
		}
1278
		if first.Start.Offset == 0 {
1279
			first, last = t.Span, t.Span
1280
			single = t.Literal
1281
		} else {
1282
			if single != "" {
1283
				b.WriteString(single)
1284
				single = ""
1285
			}
1286
			if pendingWS != "" {
1287
				b.WriteString(pendingWS)
1288
				pendingWS = ""
1289
			}
1290
			b.WriteString(t.Literal)
1291
			last = t.Span
1292
		}
1293
		p.advance()
1294
	}
1295
	if first.Start.Offset == 0 {
1296
		return "", token.Span{}
1297
	}
1298
	if single != "" {
1299
		return single, token.Span{File: first.File, Start: first.Start, End: last.End}
1300
	}
1301
	return strings.TrimSpace(b.String()), token.Span{File: first.File, Start: first.Start, End: last.End}
1302
}
1303
1304
func (p *Parser) parseTextComment() *ast.Comment {
1305
	s := p.cur.Span
1306
	marker := p.cur.Literal[0]
1307
	var b strings.Builder
1308
	b.WriteString(p.cur.Literal)
1309
	p.advance()
1310
	for !p.got(token.NEWLINE) && !p.got(token.EOF) {
1311
		b.WriteString(p.cur.Literal)
1312
		p.advance()
1313
	}
1314
	text := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(b.String()), string(marker)))
1315
	span := p.span(s) // marker to line end, without the newline
1316
	p.expectNewline()
1317
	return &ast.Comment{Marker: marker, Text: text, Span: span}
1318
}
1319
1320
func isDirectiveKeyword(t token.Type) bool {
1321
	switch t {
1322
	case token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE,
1323
		token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END,
1324
		token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C:
1325
		return true
1326
	}
1327
	return false
1328
}
1329
1330
func (p *Parser) sync() {
1331
	for {
1332
		switch p.cur.Type {
1333
		case token.EOF:
1334
			return
1335
		case token.NEWLINE:
1336
			p.advance()
1337
			t := p.cur.Type
1338
			if isDirectiveKeyword(t) || t == token.DATE || t == token.TILDE || t == token.EQ {
1339
				return
1340
			}
1341
		default:
1342
			p.advance()
1343
		}
1344
	}
1345
}
1346
1347
func (p *Parser) syncToNextline() {
1348
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF {
1349
		p.advance()
1350
	}
1351
	if p.got(token.NEWLINE) {
1352
		p.advance()
1353
	}
1354
}
1355
1356
func (p *Parser) skipWhitespace() {
1357
	for p.got(token.WHITESPACE) {
1358
		p.advance()
1359
	}
1360
}
1361
1362
func (p *Parser) span(s token.Span) token.Span {
1363
	return token.Span{File: s.File, Start: s.Start, End: p.cur.Span.Start}
1364
}
1365
1366
func normalizeLiteral(lit string, thousands, decimal byte) string {
1367
	// fast path: no separators to strip and the decimal mark is already '.'
1368
	if thousands == 0 && (decimal == 0 || decimal == '.') {
1369
		return lit
1370
	}
1371
	var b strings.Builder
1372
	for _, ch := range []byte(lit) {
1373
		if thousands != 0 && ch == thousands {
1374
			continue // skip thousands separator
1375
		}
1376
		if ch == decimal {
1377
			b.WriteByte('.')
1378
		} else {
1379
			b.WriteByte(ch)
1380
		}
1381
	}
1382
	return b.String()
1383
}
1384
1385
func detectFormat(lit string) ast.QuantityFormat {
1386
	var seps []int
1387
	for i, ch := range []byte(lit) {
1388
		if ch == '.' || ch == ',' || ch == ' ' || ch == '_' || ch == '\'' {
1389
			seps = append(seps, i)
1390
		}
1391
	}
1392
1393
	if len(seps) == 0 {
1394
		return ast.QuantityFormat{}
1395
	}
1396
1397
	last := seps[len(seps)-1]
1398
	dec := lit[last]
1399
	if dec != '.' && dec != ',' {
1400
		// the last separator is a thousands mark; the literal has no decimal mark
1401
		dec = 0
1402
	}
1403
	var thou byte
1404
	if len(seps) > 1 {
1405
		thou = lit[seps[0]]
1406
	} else if dec == 0 {
1407
		// single space/underscore/apostrophe is always thousands
1408
		thou = lit[last]
1409
	}
1410
1411
	// calculate precision when the last separator is a real decimal
1412
	prec := 0
1413
	if thou == 0 || len(seps) > 1 {
1414
		prec = len(lit) - last - 1
1415
	}
1416
1417
	return ast.QuantityFormat{Decimal: dec, Thousands: thou, Precision: prec}
1418
}
1419
1420
// parseSimpleDate  parses full YYYY/MM/DD date literal embedded in free text.
1421
func parseSimpleDate(s string) ast.Date {
1422
	year, month, day, sep, err := ParseDateLiteral(s)
1423
	if err != nil {
1424
		return ast.Date{}
1425
	}
1426
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep}
1427
}
1428
1429
// ParseDateLiteral parses and validates a date literal.
1430
// It accepts full YYYY/MM/DD and partial MM/DD forms, with '-', '/' or '.' as separators.
1431
func ParseDateLiteral(lit string) (year, month, day int, sep byte, err error) {
1432
	sep = dateSeparator(lit)
1433
	if sep == 0 {
1434
		return 0, 0, 0, 0, fmt.Errorf("invalid date format: %q", lit)
1435
	}
1436
1437
	parts := strings.Split(lit, string(sep))
1438
	if len(parts) != 2 && len(parts) != 3 {
1439
		return 0, 0, 0, 0, fmt.Errorf("invalid date format: %q", lit)
1440
	}
1441
1442
	nums := make([]int, len(parts))
1443
	for i, part := range parts {
1444
		if nums[i], err = strconv.Atoi(part); err != nil {
1445
			return 0, 0, 0, 0, fmt.Errorf("invalid date literal: %q", lit)
1446
		}
1447
	}
1448
1449
	month = nums[len(parts)-2]
1450
	if month < 1 || month > 12 {
1451
		return 0, 0, 0, 0, fmt.Errorf("invalid month %d in %q", month, lit)
1452
	}
1453
1454
	day = nums[len(parts)-1]
1455
	if day < 1 || day > 31 {
1456
		return 0, 0, 0, 0, fmt.Errorf("invalid day %d in %q", day, lit)
1457
	}
1458
1459
	if len(parts) == 2 {
1460
		return 0, month, day, sep, nil
1461
	}
1462
	return nums[0], month, day, sep, nil
1463
}
1464
1465
func dateSeparator(lit string) byte {
1466
	for i := 0; i < len(lit); i++ {
1467
		if lit[i] == '/' || lit[i] == '-' || lit[i] == '.' {
1468
			return lit[i]
1469
		}
1470
	}
1471
	return 0
1472
}
1473
1474
// parseCommentTags extacts tags from comment text.
1475
// A tag is a word immediately followed by a ':', with an optional value that ends at a comma or the end of a line.
1476
// https://hledger.org/1.52/hledger.html?highlight=tags#tags
1477
func parseCommentTags(text string, base token.Span) []ast.Tag {
1478
	var tags []ast.Tag
1479
	for i := 0; i < len(text); {
1480
		colon := strings.IndexByte(text[i:], ':')
1481
		if colon < 0 {
1482
			break
1483
		}
1484
		colon += i
1485
1486
		keyStart := colon
1487
		for keyStart > i {
1488
			r, size := utf8.DecodeLastRuneInString(text[:keyStart])
1489
			if unicode.IsSpace(r) {
1490
				break
1491
			}
1492
			keyStart -= size
1493
		}
1494
		if keyStart == colon { // nothing before the colon = not a tag
1495
			i = colon + 1
1496
			continue
1497
		}
1498
		key := text[keyStart:colon]
1499
1500
		valueEnd := colon + 1
1501
		for valueEnd < len(text) && text[valueEnd] != ',' {
1502
			valueEnd++
1503
		}
1504
		value := strings.TrimSpace(text[colon+1 : valueEnd])
1505
1506
		tags = append(tags, ast.Tag{
1507
			Key:   key,
1508
			Value: value,
1509
			Span: token.Span{
1510
				File:  base.File,
1511
				Start: tagPos(base.Start, text, keyStart),
1512
				End:   tagPos(base.Start, text, valueEnd),
1513
			},
1514
		})
1515
		i = valueEnd
1516
		if i < len(text) && text[i] == ',' {
1517
			i++
1518
		}
1519
	}
1520
1521
	return tags
1522
}
1523
1524
func tagPos(base token.Pos, text string, off int) token.Pos {
1525
	return token.Pos{
1526
		Offset: base.Offset + off,
1527
		Line:   base.Line,
1528
		Col:    base.Col + utf8.RuneCountInString(text[:off]),
1529
	}
1530
}