all repos

clerk @ b361473

missing tooling for ledger/hledger

clerk/journal/parser/parser.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
ast: add enum for subdirective kinds, 1 month ago
1
package parser
2
3
import (
4
	"fmt"
5
	"strconv"
6
	"strings"
7
	"unicode"
8
	"unicode/utf8"
9
10
	"olexsmir.xyz/clerk/internal/decimal"
11
	"olexsmir.xyz/clerk/journal/ast"
12
	"olexsmir.xyz/clerk/journal/lexer"
13
	"olexsmir.xyz/clerk/journal/token"
14
)
15
16
type Parser struct {
17
	lexer  *lexer.Lexer
18
	errors []*ast.ParseError
19
	cur    token.Token
20
	peek   token.Token
21
22
	defaultYear int // set by year directive, used for short date inference
23
}
24
25
func New(lex *lexer.Lexer) *Parser {
26
	p := &Parser{lexer: lex}
27
	p.advance() // populate .peek
28
	p.advance() // populate .cur
29
	return p
30
}
31
32
func NewWithYear(lex *lexer.Lexer, year int) *Parser {
33
	p := &Parser{lexer: lex, defaultYear: year}
34
	p.advance() // populate .peek
35
	p.advance() // populate .cur
36
	return p
37
}
38
39
func (p *Parser) ParseJournal() *ast.Journal {
40
	f := &ast.Journal{}
41
	for p.cur.Type != token.EOF {
42
		if e := p.parseEntry(); e != nil {
43
			f.Entries = append(f.Entries, e)
44
		}
45
	}
46
	f.Errors = p.errors
47
	return f
48
}
49
50
func (p *Parser) parseEntry() ast.Entry {
51
	if p.got(token.BANG) || p.got(token.AT) {
52
		if isDirectiveKeyword(p.peek.Type) {
53
			p.advance() // consume prefix
54
		}
55
	}
56
57
	switch p.cur.Type {
58
	case token.ILLEGAL:
59
		p.errorf("illegal character %q", p.cur.Literal)
60
		p.advance()
61
		return nil
62
	case token.INDENT:
63
		p.errorf("unexpected indent")
64
		p.syncToNextline()
65
		return nil
66
	case token.DATE:
67
		return p.parseTransaction()
68
	case token.TILDE:
69
		return p.parsePeriodicTransaction()
70
	case token.EQ:
71
		return p.parseAutomatedTransaction()
72
	case token.NEWLINE:
73
		return p.parseBlankLine()
74
	case token.SEMICOLON, token.HASH, token.PERCENT, token.STAR:
75
		return p.parseComment()
76
	case token.ACCOUNT:
77
		return p.parseAccountDirective()
78
	case token.COMMODITY:
79
		return p.parseCommodityDirective()
80
	case token.INCLUDE:
81
		return p.parseIncludeDirective()
82
	case token.ALIAS:
83
		return p.parseAliasDirective()
84
	case token.PAYEE:
85
		return p.parsePayeeDirective()
86
	case token.TAG:
87
		return p.parseTagDirective()
88
	case token.YEAR:
89
		return p.parseYearDirective()
90
	case token.DECIMALMARK:
91
		return p.parseDecimalMarkDirective()
92
	case token.D:
93
		return p.parseDefaultCommodityDirective()
94
	case token.P:
95
		return p.parseMarketPriceDirective()
96
	case token.N:
97
		return p.parseIgnoredDirective()
98
	case token.C:
99
		return p.parseConversionDirective()
100
	case token.APPLY:
101
		return p.parseApplyDirective()
102
	case token.END:
103
		return p.parseEndDirective()
104
	case token.COMMENTKW:
105
		return p.parseCommentBlockDirective()
106
	default:
107
		p.errorf("unexpected token %s", p.cur.Type)
108
		p.sync()
109
		return nil
110
	}
111
}
112
113
func (p *Parser) parseTransaction() *ast.Transaction {
114
	s := p.cur.Span
115
	tx := &ast.Transaction{}
116
117
	tx.Date = p.parseDate()
118
119
	p.skipWhitespace()
120
121
	// optional secondary date
122
	if p.got(token.EQ) {
123
		p.advance()
124
		p.skipWhitespace()
125
		d := p.parseDate()
126
		tx.SecondDate = &d
127
	}
128
129
	p.skipWhitespace()
130
131
	// optional status
132
	tx.Status = p.parseStatus()
133
134
	// optional code - the lexer emits "(CODE)" as a single TEXT token; split it here
135
	if p.got(token.TEXT) {
136
		if lit := p.cur.Literal; len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
137
			tx.Code = &ast.Code{Value: lit[1 : len(lit)-1], Span: p.cur.Span}
138
			p.advance()
139
			p.skipWhitespace()
140
		}
141
	}
142
143
	// optional payee | note
144
	if p.got(token.TEXT) || p.got(token.STRING) {
145
		tx.Payee = p.parsePayee()
146
147
		// check for | separator
148
		p.skipWhitespace()
149
150
		if p.got(token.PIPE) {
151
			p.advance()
152
			if p.got(token.TEXT) {
153
				sn := p.cur.Span
154
				n := p.cur.Literal
155
				p.advance()
156
				tx.Note = &ast.Note{Value: n, Span: p.span(sn)}
157
			}
158
		}
159
	}
160
161
	tx.Comment = p.parseOptInlineComment()
162
	p.expectNewline()
163
164
	tx.HeaderComments, tx.Postings = p.parseHeaderCommentsAndPostings()
165
166
	tx.Span = p.span(s)
167
	return tx
168
}
169
170
func unquote(s string) string {
171
	if len(s) >= 2 && ((s[0] == '"' && s[len(s)-1] == '"') || (s[0] == '\'' && s[len(s)-1] == '\'')) {
172
		return s[1 : len(s)-1]
173
	}
174
	return s
175
}
176
177
func (p *Parser) parsePayee() *ast.Payee {
178
	s := p.cur.Span
179
180
	if p.got(token.STRING) {
181
		name := unquote(p.cur.Literal)
182
		p.advance()
183
		return &ast.Payee{Name: name, Span: p.span(s)}
184
	}
185
186
	// keep spaces/tags between text tokens; stop before trailing whitespace
187
	var name strings.Builder
188
	for payeeWord(p.cur.Type) || (payeeWord(p.peek.Type) && p.got(token.WHITESPACE)) {
189
		_, _ = name.WriteString(p.cur.Literal)
190
		p.advance()
191
	}
192
	return &ast.Payee{Name: unquote(name.String()), Span: p.span(s)}
193
}
194
195
func payeeWord(t token.Type) bool {
196
	switch t {
197
	case token.TEXT, token.INT, token.DECIMAL, token.COMMODITYMARK:
198
		return true
199
	}
200
	return false
201
}
202
203
func (p *Parser) parsePeriodicTransaction() *ast.PeriodicTransaction {
204
	s := p.cur.Span
205
	p.expect(token.TILDE)
206
	p.skipWhitespace()
207
208
	pt := &ast.PeriodicTransaction{}
209
210
	pt.Period = p.parsePeriod()
211
212
	if desc, dspan := p.parseOptPeriodicDescription(); desc != "" {
213
		pt.Description = &ast.Description{Value: desc, Span: dspan}
214
	}
215
216
	comment := p.parseOptInlineComment()
217
	p.expectNewline()
218
219
	pt.HeaderComments, pt.Postings = p.parseHeaderCommentsAndPostings()
220
221
	pt.Span = p.span(s)
222
	pt.Comment = comment
223
	return pt
224
}
225
226
func (p *Parser) parseAutomatedTransaction() *ast.AutomatedTransaction {
227
	s := p.cur.Span
228
	p.expect(token.EQ)
229
	p.skipWhitespace()
230
231
	at := &ast.AutomatedTransaction{}
232
233
	// expression
234
	sd := p.cur.Span
235
	expr := p.parseDirectiveExpr()
236
	at.Expr = ast.Expr{Value: expr, Span: p.span(sd)}
237
	at.Comment = p.parseOptInlineComment()
238
	p.expectNewline()
239
240
	at.HeaderComments, at.Postings = p.parseHeaderCommentsAndPostings()
241
242
	at.Span = p.span(s)
243
	return at
244
}
245
246
func (p *Parser) parseHeaderCommentsAndPostings() (comments []*ast.Comment, postings []*ast.Posting) {
247
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
248
		p.advance() // consume indent
249
		comments = append(comments, p.parseComment())
250
	}
251
252
	for p.got(token.INDENT) {
253
		if posting := p.parsePosting(); posting != nil {
254
			postings = append(postings, posting)
255
		}
256
	}
257
258
	return comments, postings
259
}
260
261
func (p *Parser) parsePeriod() ast.Period {
262
	s := p.cur.Span
263
264
	var periodBuf strings.Builder
265
266
	for !p.got(token.NEWLINE) && !p.got(token.EOF) &&
267
		!p.got(token.SEMICOLON) && !p.got(token.HASH) && !p.got(token.PERCENT) && !p.got(token.STAR) {
268
269
		if p.got(token.WHITESPACE) {
270
			if len(p.cur.Literal) >= 2 {
271
				break
272
			}
273
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) ||
274
				p.willGet(token.SEMICOLON) || p.willGet(token.HASH) ||
275
				p.willGet(token.PERCENT) || p.willGet(token.STAR) {
276
				p.advance()
277
				continue
278
			}
279
		}
280
281
		periodBuf.WriteString(p.cur.Literal)
282
		p.advance()
283
	}
284
285
	str := periodBuf.String()
286
	period := ast.Period{Raw: str, Span: p.span(s)}
287
288
	if _, after, ok := strings.Cut(str, " from "); ok {
289
		end := strings.Index(after, " ")
290
		dateStr := after
291
		if end >= 0 {
292
			dateStr = after[:end]
293
		}
294
		if d := parseSimpleDate(dateStr); d.Year > 0 {
295
			fromOff := strings.Index(str, dateStr)
296
			d.Span = periodDateSpan(period, str, dateStr, fromOff)
297
			period.From = &d
298
			rest := after
299
			if end >= 0 {
300
				rest = after[end:]
301
			}
302
			if _, toAfter, ok := strings.Cut(rest, " to "); ok {
303
				if toEnd := strings.Index(toAfter, " "); toEnd >= 0 {
304
					toAfter = toAfter[:toEnd]
305
				}
306
				if d := parseSimpleDate(toAfter); d.Year > 0 {
307
					d.Span = periodDateSpan(period, str, toAfter, fromOff+len(dateStr))
308
					period.To = &d
309
				}
310
			}
311
		}
312
	}
313
	return period
314
}
315
316
// periodDateSpan returns the source span of dateStr, which occurs in the
317
// period text at or after searchFrom. The period span and text cover the same
318
// bytes, so offsets line up 1:1.
319
func periodDateSpan(period ast.Period, text, dateStr string, searchFrom int) token.Span {
320
	off := strings.Index(text[searchFrom:], dateStr)
321
	abs := period.Span.Start.Offset + searchFrom + off
322
	return token.Span{
323
		Start: token.Pos{File: period.Span.Start.File, Offset: abs},
324
		End:   token.Pos{File: period.Span.Start.File, Offset: abs + len(dateStr)},
325
	}
326
}
327
328
func (p *Parser) parseComment() *ast.Comment {
329
	s := p.cur.Span
330
	c := p.parseCommentRest(s)
331
	p.expectNewline()
332
	c.Span = p.span(s) // comment spans its line through the newline
333
	return c
334
}
335
336
func (p *Parser) parseAccountDirective() *ast.AccountDirective {
337
	s := p.cur.Span
338
	p.expect(token.ACCOUNT)
339
	p.skipWhitespace()
340
341
	account := p.parseAccount()
342
	comment := p.parseOptInlineComment()
343
	p.expectNewline()
344
345
	var subs []ast.AccountSubdirective
346
	for p.got(token.INDENT) {
347
		p.advance()
348
		p.skipWhitespace()
349
		if p.got(token.NEWLINE) || p.got(token.EOF) {
350
			// whitespace-only line: block continues
351
			p.expectNewline()
352
			continue
353
		}
354
		switch {
355
		case p.got(token.SEMICOLON):
356
			// comment line: directive mode lexes only ';' as a comment marker
357
			s := p.cur.Span
358
			c := p.parseCommentRest(s)
359
			p.expectNewline()
360
			subs = append(subs, ast.AccountSubdirective{Kind: ast.SubdirectiveComment, NameSpan: s, Comment: c})
361
		case p.got(token.TEXT) && isCommentMarker(p.cur.Literal):
362
			// '#', '%' and '*' lex as TEXT in directive mode; treat them as comment lines
363
			c := p.parseTextComment()
364
			subs = append(subs, ast.AccountSubdirective{Kind: ast.SubdirectiveComment, NameSpan: c.Span, Comment: c})
365
		case p.got(token.TEXT):
366
			name := p.cur.Literal
367
			kind, ok := accountSubdirectiveKind(name)
368
			if !ok {
369
				p.errorf("unknown subdirective %q", name)
370
				p.skipToNewline()
371
				continue
372
			}
373
			kw := p.cur.Span
374
			p.advance()
375
			value, vspan := p.parseSubdirectiveValue()
376
			if value == "" {
377
				p.errorf("expected value for subdirective %q", name)
378
			}
379
			c := p.parseOptInlineComment()
380
			p.expectNewline()
381
			subs = append(subs, ast.AccountSubdirective{
382
				Kind:      kind,
383
				NameSpan:  kw,
384
				Value:     value,
385
				ValueSpan: vspan,
386
				Comment:   c,
387
			})
388
		default:
389
			p.errorf("expected subdirective name, got %s", p.cur.Type)
390
			p.skipToNewline()
391
		}
392
	}
393
394
	return &ast.AccountDirective{
395
		Account:       account,
396
		Subdirectives: subs,
397
		Comment:       comment,
398
		Span:          p.span(s),
399
	}
400
}
401
402
func (p *Parser) parseCommodityDirective() *ast.CommodityDirective {
403
	s := p.cur.Span
404
	p.expect(token.COMMODITY)
405
	p.skipWhitespace()
406
407
	var commodity string
408
	var commoditySpan token.Span
409
	var format *ast.FormatSubDirective
410
411
	switch p.cur.Type {
412
	case token.COMMODITYMARK, token.TEXT, token.STRING:
413
		cs := p.cur.Span
414
		commodity = unquote(p.cur.Literal)
415
		p.advance()
416
		commoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
417
		hadSpace := p.got(token.WHITESPACE)
418
		p.skipWhitespace()
419
		if p.got(token.INT) || p.got(token.DECIMAL) || p.got(token.TEXT) {
420
			amt := p.parseAmount()
421
			amt.Commodity = commodity
422
			amt.CommoditySpan = commoditySpan
423
			amt.CommodityPos = ast.CommodityBefore
424
			amt.HasSpace = hadSpace
425
			format = &ast.FormatSubDirective{Amount: *amt}
426
		}
427
	case token.INT, token.DECIMAL:
428
		amt := p.parseAmount()
429
		commodity = amt.Commodity
430
		commoditySpan = amt.CommoditySpan
431
		format = &ast.FormatSubDirective{Amount: *amt}
432
	default:
433
		p.errorf("expected commodity name or amount, got %s", p.cur.Type)
434
	}
435
436
	if commodity == "" {
437
		p.errorf("expected commodity name, got %s", p.cur.Type)
438
	}
439
440
	// hledger parity: an inline format amount must include a decimal mark
441
	if format != nil && format.Amount.QuantityFmt.Decimal == 0 {
442
		p.errorfAt(format.Amount.Span.Start, "Please include a decimal point or decimal comma in commodity directives, to help us parse correctly. It may be followed by zero or more decimal digits.")
443
	}
444
445
	comment := p.parseOptInlineComment()
446
	p.expectNewline()
447
448
	var blockComments []*ast.Comment
449
	for p.got(token.INDENT) {
450
		p.advance()
451
		p.skipWhitespace()
452
		if p.got(token.NEWLINE) || p.got(token.EOF) {
453
			// whitespace-only line: block continues
454
			p.expectNewline()
455
			continue
456
		}
457
		switch {
458
		case p.got(token.TEXT) && p.cur.Literal == "format":
459
			kw := p.cur.Span
460
			p.advance()
461
			p.skipWhitespace()
462
			amt := p.parseAmount()
463
			// hledger parity: the format symbol must match the declared commodity,
464
			// and the amount must include a decimal mark; the node is kept either
465
			// way so the printer can round-trip the input.
466
			if amt.Commodity != commodity {
467
				p.errorfAt(amt.Span.Start, "commodity directive symbol %q and format directive symbol %q should be the same", commodity, amt.Commodity)
468
			} else if amt.QuantityFmt.Decimal == 0 {
469
				p.errorfAt(amt.Span.Start, "Please include a decimal point or decimal comma in commodity directives, to help us parse correctly. It may be followed by zero or more decimal digits.")
470
			}
471
			c := p.parseOptInlineComment()
472
			p.expectNewline()
473
			format = &ast.FormatSubDirective{KeywordSpan: kw, Amount: *amt, Comment: c}
474
		case p.got(token.SEMICOLON): // comment line
475
			c := p.parseCommentRest(p.cur.Span)
476
			p.expectNewline()
477
			blockComments = append(blockComments, c)
478
		case p.got(token.TEXT) && isCommentMarker(p.cur.Literal):
479
			// '#', '%' and '*' lex as TEXT in directive mode; treat them as comment lines
480
			blockComments = append(blockComments, p.parseTextComment())
481
		case p.got(token.TEXT):
482
			p.errorf("unknown subdirective %q", p.cur.Literal)
483
			p.skipToNewline()
484
		default:
485
			p.errorf("expected subdirective name, got %s", p.cur.Type)
486
			p.skipToNewline()
487
		}
488
	}
489
490
	cd := &ast.CommodityDirective{
491
		Commodity:     commodity,
492
		CommoditySpan: commoditySpan,
493
		FormatSub:     format,
494
		BlockComments: blockComments,
495
		Comment:       comment,
496
		Span:          p.span(s),
497
	}
498
	return cd
499
}
500
501
func (p *Parser) parseIncludeDirective() *ast.IncludeDirective {
502
	s := p.cur.Span
503
	p.expect(token.INCLUDE)
504
	p.skipWhitespace()
505
506
	id := &ast.IncludeDirective{}
507
508
	if p.got(token.TEXT) {
509
		id.Path = p.cur.Literal
510
		p.advance()
511
	} else {
512
		p.errorf("expected file path, got %s", p.cur.Type)
513
	}
514
515
	id.Comment = p.parseOptInlineComment()
516
	p.expectNewline()
517
	id.Span = p.span(s)
518
	return id
519
}
520
521
func (p *Parser) parseAliasDirective() *ast.AliasDirective {
522
	s := p.cur.Span
523
	alias := &ast.AliasDirective{}
524
	p.expect(token.ALIAS)
525
	p.skipWhitespace()
526
	alias.From = p.parseAccount()
527
	p.skipWhitespace()
528
	p.expect(token.EQ)
529
	p.skipWhitespace()
530
	alias.To = p.parseAccount()
531
	alias.Comment = p.parseOptInlineComment()
532
	p.expectNewline()
533
	alias.Span = p.span(s)
534
	return alias
535
}
536
537
func (p *Parser) parsePayeeDirective() *ast.PayeeDirective {
538
	s := p.cur.Span
539
	p.expect(token.PAYEE)
540
	p.skipWhitespace()
541
542
	var name *ast.Payee
543
	if p.got(token.TEXT) || p.got(token.STRING) || p.got(token.COMMODITYMARK) {
544
		name = p.parsePayee()
545
	}
546
547
	comment := p.parseOptInlineComment()
548
	p.expectNewline()
549
550
	return &ast.PayeeDirective{
551
		Name:    name,
552
		Comment: comment,
553
		Span:    p.span(s),
554
	}
555
}
556
557
func (p *Parser) parseTagDirective() *ast.TagDirective {
558
	s := p.cur.Span
559
	p.expect(token.TAG)
560
	p.skipWhitespace()
561
562
	name := ""
563
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) || p.got(token.STRING) {
564
		name = unquote(p.cur.Literal)
565
		p.advance()
566
	}
567
568
	comment := p.parseOptInlineComment()
569
	p.expectNewline()
570
571
	return &ast.TagDirective{
572
		Name:    name,
573
		Comment: comment,
574
		Span:    p.span(s),
575
	}
576
}
577
578
func (p *Parser) parseYearDirective() *ast.YearDirective {
579
	s := p.cur.Span
580
	year := &ast.YearDirective{}
581
	p.expect(token.YEAR)
582
	p.skipWhitespace()
583
584
	if p.got(token.INT) {
585
		year.Year, _ = strconv.Atoi(p.cur.Literal)
586
		p.defaultYear = year.Year
587
		p.advance()
588
	} else {
589
		p.errorf("expected year, got %s", p.cur.Type)
590
	}
591
592
	year.Comment = p.parseOptInlineComment()
593
	p.expectNewline()
594
	year.Span = p.span(s)
595
596
	return year
597
}
598
599
func (p *Parser) parseDecimalMarkDirective() *ast.DecimalMarkDirective {
600
	s := p.cur.Span
601
	mark := &ast.DecimalMarkDirective{}
602
	p.expect(token.DECIMALMARK)
603
	p.skipWhitespace()
604
605
	mark.Mark = byte('.')
606
	if p.got(token.TEXT) {
607
		if len(p.cur.Literal) > 0 {
608
			mark.Mark = p.cur.Literal[0]
609
		}
610
		p.advance()
611
	}
612
613
	mark.Comment = p.parseOptInlineComment()
614
	p.expectNewline()
615
	mark.Span = p.span(s)
616
	return mark
617
}
618
619
func (p *Parser) parseDefaultCommodityDirective() *ast.DefaultCommodityDirective {
620
	s := p.cur.Span
621
	com := &ast.DefaultCommodityDirective{}
622
	p.expect(token.D)
623
	p.skipWhitespace()
624
	com.Amount = *p.parseAmount()
625
	com.Comment = p.parseOptInlineComment()
626
	p.expectNewline()
627
	com.Span = p.span(s)
628
	return com
629
}
630
631
func (p *Parser) parseConversionDirective() *ast.ConversionDirective {
632
	s := p.cur.Span
633
	cd := &ast.ConversionDirective{}
634
	p.expect(token.C)
635
	p.skipWhitespace()
636
637
	if p.isAmountStart() {
638
		cd.From = *p.parseAmount()
639
	} else {
640
		p.errorf("expected amount, got %s", p.cur.Type)
641
	}
642
643
	p.skipWhitespace()
644
	if p.got(token.EQ) {
645
		p.advance()
646
		p.skipWhitespace()
647
		if p.isAmountStart() {
648
			cd.To = *p.parseAmount()
649
		} else {
650
			p.errorf("expected amount, got %s", p.cur.Type)
651
		}
652
	}
653
654
	cd.Comment = p.parseOptInlineComment()
655
	p.expectNewline()
656
	cd.Span = p.span(s)
657
	return cd
658
}
659
660
func (p *Parser) parseIgnoredDirective() *ast.IgnoredDirective {
661
	s := p.cur.Span
662
	p.expect(token.N)
663
	p.skipWhitespace()
664
665
	id := &ast.IgnoredDirective{}
666
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) {
667
		id.Text = p.cur.Literal
668
		p.advance()
669
	}
670
	id.Comment = p.parseOptInlineComment()
671
672
	p.expectNewline()
673
	id.Span = p.span(s)
674
	return id
675
}
676
677
func (p *Parser) parseMarketPriceDirective() *ast.MarketPriceDirective {
678
	s := p.cur.Span
679
	p.expect(token.P)
680
	p.skipWhitespace()
681
682
	mp := &ast.MarketPriceDirective{}
683
	mp.DateTime.Date = p.parseDate()
684
	p.skipWhitespace()
685
686
	if p.got(token.TIME) {
687
		mp.DateTime.Time = new(p.parseTime())
688
		p.skipWhitespace()
689
	}
690
691
	tok, _ := p.expect(token.COMMODITYMARK)
692
	mp.Commodity = tok.Literal
693
	p.skipWhitespace()
694
695
	mp.Amount = *p.parseAmount()
696
697
	mp.Comment = p.parseOptInlineComment()
698
699
	p.expectNewline()
700
	mp.Span = p.span(s)
701
	return mp
702
}
703
704
func (p *Parser) parseTime() ast.Time {
705
	s := p.cur.Span
706
	tok, _ := p.expect(token.TIME)
707
	lit := tok.Literal
708
709
	parts := strings.Split(lit, ":")
710
	if len(parts) < 2 {
711
		p.errorf("invalid time format: %q", lit)
712
		return ast.Time{Span: p.span(s)}
713
	}
714
715
	hour, _ := strconv.Atoi(parts[0])
716
	minute, _ := strconv.Atoi(parts[1])
717
	second := 0
718
	if len(parts) > 2 {
719
		second, _ = strconv.Atoi(parts[2])
720
	}
721
722
	if hour < 0 || hour > 23 {
723
		p.errorf("invalid hour %d in time %q", hour, lit)
724
	}
725
	if minute < 0 || minute > 59 {
726
		p.errorf("invalid minute %d in time %q", minute, lit)
727
	}
728
	if second < 0 || second > 59 {
729
		p.errorf("invalid second %d in time %q", second, lit)
730
	}
731
732
	return ast.Time{
733
		Hour:   hour,
734
		Minute: minute,
735
		Second: second,
736
		Span:   p.span(s),
737
	}
738
}
739
740
func (p *Parser) parseApplyDirective() *ast.ApplyDirective {
741
	s := p.cur.Span
742
	p.expect(token.APPLY)
743
	p.skipWhitespace()
744
745
	expr := p.parseDirectiveExpr()
746
	comment := p.parseOptInlineComment()
747
	p.expectNewline()
748
749
	return &ast.ApplyDirective{
750
		Expr:    expr,
751
		Comment: comment,
752
		Span:    p.span(s),
753
	}
754
}
755
756
func (p *Parser) parseEndDirective() *ast.EndDirective {
757
	s := p.cur.Span
758
	p.expect(token.END)
759
	p.skipWhitespace()
760
761
	expr := p.parseDirectiveExpr()
762
	comment := p.parseOptInlineComment()
763
	p.expectNewline()
764
765
	return &ast.EndDirective{
766
		Expr:    expr,
767
		Comment: comment,
768
		Span:    p.span(s),
769
	}
770
}
771
772
func (p *Parser) parseCommentBlockDirective() *ast.CommentBlockDirective {
773
	start := p.cur.Span
774
	p.expect(token.COMMENTKW)
775
	p.skipWhitespace()
776
777
	header := p.parseDirectiveExpr()
778
	comment := p.parseOptInlineComment()
779
	p.expectNewline()
780
781
	var content strings.Builder
782
	for p.cur.Type != token.EOF {
783
		if p.got(token.END) {
784
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) {
785
				p.advance()
786
				p.expectNewline()
787
				break
788
			}
789
			if p.willGet(token.WHITESPACE) {
790
				endTok := p.cur
791
				p.advance()
792
				wsTok := p.cur
793
				p.advance()
794
				if p.got(token.TEXT) && p.cur.Literal == "comment" { // todo: this should check if it's an actual COMMENTKW token
795
					p.advance()
796
					p.parseDirectiveExpr()
797
					p.parseOptInlineComment()
798
					p.expectNewline()
799
					break
800
				}
801
				content.WriteString(endTok.Literal)
802
				content.WriteString(wsTok.Literal)
803
				continue
804
			}
805
		}
806
		content.WriteString(p.cur.Literal)
807
		p.advance()
808
	}
809
810
	return &ast.CommentBlockDirective{
811
		Header:  header,
812
		Content: content.String(),
813
		Comment: comment,
814
		Span:    p.span(start),
815
	}
816
}
817
818
func (p *Parser) parseStatus() ast.Status {
819
	s := p.cur.Span
820
	st := ast.Status{}
821
	switch p.cur.Type {
822
	case token.STAR:
823
		st.Value = ast.StatusCleared
824
	case token.BANG:
825
		st.Value = ast.StatusPending
826
	}
827
	if st.Value != ast.StatusNone {
828
		p.advance()
829
		p.skipWhitespace()
830
	}
831
	st.Span = p.span(s)
832
	return st
833
}
834
835
func (p *Parser) isAmountStart() bool {
836
	switch p.cur.Type {
837
	default:
838
		return false
839
	case token.COMMODITYMARK, token.STRING, token.INT, token.DECIMAL, token.MINUS, token.PLUS, token.PARENEXPR, token.STAR:
840
		return true
841
	}
842
}
843
844
func (p *Parser) parseAmount() *ast.Amount {
845
	s := p.cur.Span
846
	amt := &ast.Amount{
847
		QuantityFmt: ast.QuantityFormat{},
848
	}
849
	defer func() {
850
		// The span covers from the first token to the start of the next unconsumed token.
851
		// Since parseQuantityInto (and possible commodity consumption) advanced past the last
852
		// amount token, p.cur points to the next token after the amount — which is the correct end.
853
		amt.Span = p.span(s)
854
	}()
855
856
	p.parseAmountSign(amt)
857
	p.skipWhitespace()
858
859
	// commodity before quantity: $10.00, eur 10.00
860
	if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
861
		cs := p.cur.Span
862
		amt.Commodity = unquote(p.cur.Literal)
863
		amt.CommodityPos = ast.CommodityBefore
864
		p.advance()
865
		amt.CommoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
866
		if p.got(token.WHITESPACE) {
867
			amt.HasSpace = true
868
			p.skipWhitespace()
869
		}
870
	}
871
872
	// optional sign after commodity: $ -10
873
	p.parseAmountSign(amt)
874
	p.skipWhitespace()
875
876
	p.parseQuantityInto(amt)
877
878
	// commodity after quantity: 10.00 UAH, 10.00 "EUR" (only if not set)
879
	if amt.Commodity == "" {
880
		switch p.cur.Type {
881
		case token.WHITESPACE:
882
			p.skipWhitespace()
883
			if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
884
				cs := p.cur.Span
885
				amt.HasSpace = true
886
				amt.Commodity = unquote(p.cur.Literal)
887
				amt.CommodityPos = ast.CommodityAfter
888
				p.advance()
889
				amt.CommoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
890
			}
891
		case token.COMMODITYMARK, token.TEXT, token.STRING:
892
			cs := p.cur.Span
893
			amt.Commodity = unquote(p.cur.Literal)
894
			amt.CommodityPos = ast.CommodityAfter
895
			p.advance()
896
			amt.CommoditySpan = token.Span{Start: cs.Start, End: p.cur.Span.Start}
897
		}
898
	}
899
900
	return amt
901
}
902
903
// parseAmountSign consumes an optional leading +/- into IsNegative.
904
func (p *Parser) parseAmountSign(amt *ast.Amount) {
905
	switch p.cur.Type {
906
	case token.MINUS:
907
		amt.IsNegative = true
908
		p.advance()
909
	case token.PLUS:
910
		p.advance()
911
	}
912
}
913
914
func (p *Parser) parseAmountWithOptExpr() *ast.Amount {
915
	if p.got(token.STAR) {
916
		p.advance()
917
		p.skipWhitespace()
918
		amt := p.parseAmount()
919
		if amt != nil {
920
			amt.IsExpr = true
921
		}
922
		return amt
923
	}
924
	if p.got(token.PARENEXPR) {
925
		lit := p.cur.Literal
926
		amt := &ast.Amount{
927
			IsExpr:      true,
928
			QuantityFmt: ast.QuantityFormat{},
929
		}
930
		if len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
931
			amt.Expr = strings.Trim(lit[1:len(lit)-1], " \t")
932
		}
933
		amt.Span = p.cur.Span
934
		p.advance()
935
		return amt
936
	}
937
	return p.parseAmount()
938
}
939
940
func (p *Parser) parsePosting() *ast.Posting {
941
	s := p.cur.Span
942
	posting := &ast.Posting{}
943
	p.expect(token.INDENT)
944
945
	// exit if it's empty line
946
	if p.got(token.NEWLINE) || p.got(token.EOF) {
947
		p.syncToNextline()
948
		return nil
949
	}
950
951
	// optional status, outside of brackets, '! (account)'
952
	posting.Status = p.parseStatus()
953
954
	// detect virtual posting brackets
955
	switch p.cur.Type {
956
	case token.LPAREN:
957
		posting.Type = ast.PostingVirtualUnbalanced
958
		p.advance()
959
	case token.LBRACKET:
960
		posting.Type = ast.PostingVirtualBalanced
961
		p.advance()
962
	}
963
964
	// optional status, inside of brackets, '(* account)'
965
	if p.got(token.STAR) || p.got(token.BANG) {
966
		posting.Status = p.parseStatus()
967
	}
968
969
	// validate, must be account text
970
	if p.cur.Type != token.TEXT {
971
		p.errorf("expected account name, got %s", p.cur.Type)
972
		p.syncToNextline()
973
		return nil
974
	}
975
976
	posting.Account = p.parseAccount()
977
978
	// consume closing bracket
979
	switch p.cur.Type {
980
	case token.RPAREN:
981
		p.advance()
982
	case token.RBRACKET:
983
		p.advance()
984
	}
985
986
	// optional amount - after two spaces
987
	if p.got(token.WHITESPACE) {
988
		p.skipWhitespace()
989
		if p.isAmountStart() {
990
			posting.Amount = p.parseAmountWithOptExpr()
991
		}
992
	}
993
994
	// optional cost '@' or '@@'
995
	p.skipWhitespace()
996
	if p.got(token.AT) || p.got(token.ATAT) {
997
		posting.Cost = p.parseCost()
998
	}
999
1000
	// optional balance assertion or assignment
1001
	p.skipWhitespace()
1002
	if p.got(token.COLON) && p.willGet(token.EQ) {
1003
		p.advance() // consume ':' of ':='
1004
		posting.Balance = p.parseBalanceAssertion()
1005
		posting.Balance.IsAssignment = true
1006
	} else if p.got(token.EQ) || p.got(token.EQEQ) || p.got(token.EQEQEQ) || p.got(token.EQSTAR) {
1007
		posting.Balance = p.parseBalanceAssertion()
1008
	}
1009
1010
	posting.Comment = p.parseOptInlineComment()
1011
	p.expectNewline()
1012
1013
	// continuation comments
1014
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
1015
		p.advance()
1016
		c := p.parseComment()
1017
		posting.Comments = append(posting.Comments, *c)
1018
	}
1019
1020
	posting.Span = p.span(s)
1021
	return posting
1022
}
1023
1024
func (p *Parser) parseCost() *ast.Cost {
1025
	s := p.cur.Span
1026
	isTotal := p.got(token.ATAT)
1027
	p.advance() // consume '@' '@@'
1028
	p.skipWhitespace()
1029
	return &ast.Cost{
1030
		IsTotal: isTotal,
1031
		Amount:  *p.parseAmount(),
1032
		Span:    p.span(s),
1033
	}
1034
}
1035
1036
func (p *Parser) parseBalanceAssertion() *ast.BalanceAssertion {
1037
	s := p.cur.Span
1038
1039
	ba := &ast.BalanceAssertion{}
1040
	switch p.cur.Type {
1041
	case token.EQ: // basic assertion
1042
	case token.EQSTAR: // inclusive assertion
1043
		ba.IsInclusive = true
1044
	case token.EQEQ: // strict assertion
1045
		ba.IsStrict = true
1046
	case token.EQEQEQ: // strict inclusive assertion
1047
		ba.IsStrict = true
1048
		ba.IsInclusive = true
1049
	}
1050
	p.advance()
1051
	p.skipWhitespace()
1052
1053
	ba.Amount = *p.parseAmount()
1054
	p.skipWhitespace()
1055
	if p.got(token.AT) || p.got(token.ATAT) {
1056
		c := p.parseCost()
1057
		ba.Cost = c
1058
	}
1059
	ba.Span = p.span(s)
1060
	return ba
1061
}
1062
1063
func (p *Parser) readAccountSegment() (ast.SubAccount, bool) {
1064
	switch p.cur.Type {
1065
	case token.TEXT:
1066
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
1067
		p.advance()
1068
1069
		// handle multi work segment, e.g: "credit card"
1070
		if p.got(token.WHITESPACE) && p.willGet(token.TEXT) && len(p.peek.Literal) > 0 && p.peek.Literal[0] != '(' {
1071
			sub.Name += " "
1072
			p.advance()
1073
			sub.Name += p.cur.Literal
1074
			p.advance()
1075
		}
1076
		return sub, true
1077
1078
	case token.COMMODITYMARK:
1079
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
1080
		p.advance()
1081
		// merge "EUR" + "-HRK" to "EUR-HRK"
1082
		for p.got(token.TEXT) {
1083
			sub.Name += p.cur.Literal
1084
			p.advance()
1085
		}
1086
		return sub, true
1087
1088
	default:
1089
		return ast.SubAccount{}, false
1090
	}
1091
}
1092
1093
func (p *Parser) parseAccount() ast.Account {
1094
	s := p.cur.Span
1095
	acc := ast.Account{}
1096
1097
	sub, ok := p.readAccountSegment()
1098
	if !ok {
1099
		p.errorf("expected account, got %s", p.cur.Type)
1100
		return ast.Account{}
1101
	}
1102
	acc.Name = append(acc.Name, sub)
1103
1104
	for p.got(token.COLON) {
1105
		p.advance()
1106
		sub, ok := p.readAccountSegment()
1107
		if !ok {
1108
			break
1109
		}
1110
		acc.Name = append(acc.Name, sub)
1111
	}
1112
1113
	acc.Span = p.span(s)
1114
	return acc
1115
}
1116
1117
func (p *Parser) parseDate() ast.Date {
1118
	s := p.cur.Span
1119
	tok, ok := p.expect(token.DATE)
1120
	if !ok {
1121
		return ast.Date{Span: p.span(s)}
1122
	}
1123
1124
	year, month, day, sep, err := ParseDateLiteral(tok.Literal)
1125
	if err != nil {
1126
		p.errorf("%v", err)
1127
		return ast.Date{Span: p.span(s)}
1128
	}
1129
	if year == 0 {
1130
		year = p.defaultYear
1131
	}
1132
1133
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep, Span: p.span(s)}
1134
}
1135
1136
func (p *Parser) parseOptInlineComment() *ast.Comment {
1137
	p.skipWhitespace()
1138
	if !p.got(token.SEMICOLON) {
1139
		return nil
1140
	}
1141
	return p.parseCommentRest(p.cur.Span)
1142
}
1143
1144
// parseCommentRest consumes a comment marker at p.cur, then optional text;
1145
// s anchors the span at the marker's start.
1146
func (p *Parser) parseCommentRest(s token.Span) *ast.Comment {
1147
	marker := p.cur.Literal[0]
1148
	p.advance()
1149
	p.skipWhitespace()
1150
1151
	var tags []ast.Tag
1152
	text := ""
1153
	if p.got(token.TEXT) {
1154
		text = p.cur.Literal
1155
		tags = parseCommentTags(text, p.cur.Span.Start)
1156
		p.advance()
1157
	}
1158
1159
	return &ast.Comment{
1160
		Marker: marker,
1161
		Tags:   tags,
1162
		Text:   text,
1163
		Span:   p.span(s),
1164
	}
1165
}
1166
1167
func (p *Parser) parseOptPeriodicDescription() (string, token.Span) {
1168
	if p.cur.Type != token.WHITESPACE || len(p.cur.Literal) < 2 {
1169
		return "", token.Span{}
1170
	}
1171
1172
	p.skipWhitespace()
1173
1174
	if p.cur.Type != token.TEXT {
1175
		return "", token.Span{}
1176
	}
1177
1178
	s := p.cur.Span
1179
	desc := p.parseDescription()
1180
	return desc, p.span(s)
1181
}
1182
1183
func (p *Parser) parseDescription() string {
1184
	var desc strings.Builder
1185
	for p.got(token.TEXT) || (p.got(token.WHITESPACE) && p.willGet(token.TEXT)) {
1186
		_, _ = desc.WriteString(p.cur.Literal)
1187
		p.advance()
1188
	}
1189
	return desc.String()
1190
}
1191
1192
func (p *Parser) parseDirectiveExpr() string {
1193
	var b strings.Builder
1194
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF && p.cur.Type != token.SEMICOLON {
1195
		_, _ = b.WriteString(p.cur.Literal)
1196
		p.advance()
1197
	}
1198
	return b.String()
1199
}
1200
1201
func (p *Parser) parseQuantityInto(amt *ast.Amount) {
1202
	if p.cur.Type != token.INT && p.cur.Type != token.DECIMAL && p.cur.Type != token.TEXT {
1203
		p.errorf("expected quantity, got %s", p.cur.Type)
1204
		return
1205
	}
1206
1207
	lit := p.cur.Literal
1208
	p.advance()
1209
1210
	// detect format metadata before normalizing
1211
	amt.QuantityFmt = detectFormat(lit)
1212
1213
	// normalize for decimal.NewFromString
1214
	// remove thousands separators, replace decimal mark with '.'
1215
	normalized := normalizeLiteral(lit, amt.QuantityFmt.Thousands, amt.QuantityFmt.Decimal)
1216
1217
	q, err := decimal.FromString(normalized)
1218
	if err != nil {
1219
		p.errorf("invalid quantity %q: %v", lit, err)
1220
		return
1221
	}
1222
1223
	if amt.IsNegative {
1224
		q = q.Neg()
1225
	}
1226
	amt.Quantity = q
1227
}
1228
1229
func (p *Parser) parseBlankLine() *ast.BlankLine {
1230
	s := p.cur.Span
1231
	p.expectNewline()
1232
	return &ast.BlankLine{Span: s}
1233
}
1234
1235
func (p *Parser) expectNewline() {
1236
	if p.got(token.NEWLINE) || p.got(token.EOF) {
1237
		if p.got(token.NEWLINE) {
1238
			p.advance()
1239
		}
1240
		return
1241
	}
1242
	p.errorf("expected %s, got %s", token.NEWLINE, p.cur.Type)
1243
}
1244
1245
func (p *Parser) advance() token.Token {
1246
	prev := p.cur
1247
	p.cur = p.peek
1248
	p.peek = p.lexer.Next()
1249
	return prev
1250
}
1251
1252
func (p *Parser) got(kind token.Type) bool     { return p.cur.Type == kind }
1253
func (p *Parser) willGet(kind token.Type) bool { return p.peek.Type == kind }
1254
1255
func (p *Parser) expect(kind token.Type) (token.Token, bool) {
1256
	if p.got(kind) {
1257
		return p.advance(), true
1258
	}
1259
	p.errorf("expected %s, got %s", kind, p.cur.Type)
1260
	return p.cur, false
1261
}
1262
1263
func (p *Parser) errorf(format string, args ...any) {
1264
	p.errors = append(p.errors, &ast.ParseError{
1265
		Span:    p.cur.Span,
1266
		Message: fmt.Sprintf(format, args...),
1267
	})
1268
}
1269
1270
// errorfAt records a parse error pointing at a single position.
1271
func (p *Parser) errorfAt(pos token.Pos, format string, args ...any) {
1272
	p.errors = append(p.errors, &ast.ParseError{
1273
		Span:    token.Span{Start: pos, End: pos},
1274
		Message: fmt.Sprintf(format, args...),
1275
	})
1276
}
1277
1278
func accountSubdirectiveKind(name string) (ast.AccountSubdirectiveKind, bool) {
1279
	switch name {
1280
	case "alias":
1281
		return ast.SubdirectiveAlias, true
1282
	case "type":
1283
		return ast.SubdirectiveType, true
1284
	case "note":
1285
		return ast.SubdirectiveNote, true
1286
	}
1287
	return 0, false
1288
}
1289
1290
func isCommentMarker(s string) bool {
1291
	return len(s) > 0 && (s[0] == '#' || s[0] == '%' || s[0] == '*')
1292
}
1293
1294
// skipToNewline consumes the rest of the current line.
1295
func (p *Parser) skipToNewline() {
1296
	for !p.got(token.NEWLINE) && !p.got(token.EOF) {
1297
		p.advance()
1298
	}
1299
	p.expectNewline()
1300
}
1301
1302
// parseSubdirectiveValue consumes the tokens after a subdirective keyword up
1303
// to an inline comment or the end of the line, returning the raw text and its
1304
// span (trimmed of surrounding whitespace). Single-token values return the
1305
// lexer's literal, sharing the source backing without a copy.
1306
func (p *Parser) parseSubdirectiveValue() (string, token.Span) {
1307
	var b strings.Builder
1308
	var first, last token.Span
1309
	var single, pendingWS string
1310
	for !p.got(token.SEMICOLON) && !p.got(token.NEWLINE) && !p.got(token.EOF) {
1311
		t := p.cur
1312
		if t.Type == token.WHITESPACE || t.Type == token.INDENT {
1313
			if first.Start.Offset > 0 {
1314
				pendingWS = t.Literal // only the last whitespace run matters
1315
			}
1316
			p.advance()
1317
			continue
1318
		}
1319
		if first.Start.Offset == 0 {
1320
			first, last = t.Span, t.Span
1321
			single = t.Literal
1322
		} else {
1323
			if single != "" {
1324
				b.WriteString(single)
1325
				single = ""
1326
			}
1327
			if pendingWS != "" {
1328
				b.WriteString(pendingWS)
1329
				pendingWS = ""
1330
			}
1331
			b.WriteString(t.Literal)
1332
			last = t.Span
1333
		}
1334
		p.advance()
1335
	}
1336
	if first.Start.Offset == 0 {
1337
		return "", token.Span{}
1338
	}
1339
	if single != "" {
1340
		return single, token.Span{Start: first.Start, End: last.End}
1341
	}
1342
	return strings.TrimSpace(b.String()), token.Span{Start: first.Start, End: last.End}
1343
}
1344
1345
// parseTextComment consumes a comment line whose marker ('#', '%' or '*')
1346
// lexed as TEXT inside a directive block, building the comment from token
1347
// literals. Tags are not extracted; the comment is one line.
1348
func (p *Parser) parseTextComment() *ast.Comment {
1349
	s := p.cur.Span
1350
	marker := p.cur.Literal[0]
1351
	var b strings.Builder
1352
	b.WriteString(p.cur.Literal)
1353
	p.advance()
1354
	for !p.got(token.NEWLINE) && !p.got(token.EOF) {
1355
		b.WriteString(p.cur.Literal)
1356
		p.advance()
1357
	}
1358
	text := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(b.String()), string(marker)))
1359
	span := p.span(s) // marker to line end, without the newline
1360
	p.expectNewline()
1361
	return &ast.Comment{Marker: marker, Text: text, Span: span}
1362
}
1363
1364
func isDirectiveKeyword(t token.Type) bool {
1365
	switch t {
1366
	case token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE,
1367
		token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END,
1368
		token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C:
1369
		return true
1370
	}
1371
	return false
1372
}
1373
1374
func (p *Parser) sync() {
1375
	for {
1376
		switch p.cur.Type {
1377
		case token.EOF:
1378
			return
1379
		case token.NEWLINE:
1380
			p.advance()
1381
			t := p.cur.Type
1382
			if isDirectiveKeyword(t) || t == token.DATE || t == token.TILDE || t == token.EQ {
1383
				return
1384
			}
1385
		default:
1386
			p.advance()
1387
		}
1388
	}
1389
}
1390
1391
func (p *Parser) syncToNextline() {
1392
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF {
1393
		p.advance()
1394
	}
1395
	if p.got(token.NEWLINE) {
1396
		p.advance()
1397
	}
1398
}
1399
1400
func (p *Parser) skipWhitespace() {
1401
	for p.got(token.WHITESPACE) {
1402
		p.advance()
1403
	}
1404
}
1405
1406
func (p *Parser) span(s token.Span) token.Span {
1407
	return token.Span{Start: s.Start, End: p.cur.Span.Start}
1408
}
1409
1410
func normalizeLiteral(lit string, thousands, decimal byte) string {
1411
	var b strings.Builder
1412
	for _, ch := range []byte(lit) {
1413
		if thousands != 0 && ch == thousands {
1414
			continue // skip thousands separator
1415
		}
1416
		if ch == decimal {
1417
			b.WriteByte('.')
1418
		} else {
1419
			b.WriteByte(ch)
1420
		}
1421
	}
1422
	return b.String()
1423
}
1424
1425
func detectFormat(lit string) ast.QuantityFormat {
1426
	var seps []int
1427
	for i, ch := range []byte(lit) {
1428
		if ch == '.' || ch == ',' || ch == ' ' || ch == '_' || ch == '\'' {
1429
			seps = append(seps, i)
1430
		}
1431
	}
1432
1433
	if len(seps) == 0 {
1434
		return ast.QuantityFormat{}
1435
	}
1436
1437
	last := seps[len(seps)-1]
1438
	dec := lit[last]
1439
	if dec != '.' && dec != ',' {
1440
		// the last separator is a thousands mark; the literal has no decimal mark
1441
		dec = 0
1442
	}
1443
	var thou byte
1444
	if len(seps) > 1 {
1445
		thou = lit[seps[0]]
1446
	} else if dec == 0 {
1447
		// single space/underscore/apostrophe is always thousands
1448
		thou = lit[last]
1449
	}
1450
1451
	// calculate precision when the last separator is a real decimal
1452
	prec := 0
1453
	if thou == 0 || len(seps) > 1 {
1454
		prec = len(lit) - last - 1
1455
	}
1456
1457
	return ast.QuantityFormat{Decimal: dec, Thousands: thou, Precision: prec}
1458
}
1459
1460
// parseSimpleDate  parses full YYYY/MM/DD date literal embedded in free text.
1461
func parseSimpleDate(s string) ast.Date {
1462
	year, month, day, sep, err := ParseDateLiteral(s)
1463
	if err != nil {
1464
		return ast.Date{}
1465
	}
1466
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep}
1467
}
1468
1469
// ParseDateLiteral parses and validates a date literal.
1470
// It accepts full YYYY/MM/DD and partial MM/DD forms, with '-', '/' or '.' as separators.
1471
func ParseDateLiteral(lit string) (year, month, day int, sep byte, err error) {
1472
	sep = dateSeparator(lit)
1473
	if sep == 0 {
1474
		return 0, 0, 0, 0, fmt.Errorf("invalid date format: %q", lit)
1475
	}
1476
1477
	parts := strings.Split(lit, string(sep))
1478
	if len(parts) != 2 && len(parts) != 3 {
1479
		return 0, 0, 0, 0, fmt.Errorf("invalid date format: %q", lit)
1480
	}
1481
1482
	nums := make([]int, len(parts))
1483
	for i, part := range parts {
1484
		if nums[i], err = strconv.Atoi(part); err != nil {
1485
			return 0, 0, 0, 0, fmt.Errorf("invalid date literal: %q", lit)
1486
		}
1487
	}
1488
1489
	month = nums[len(parts)-2]
1490
	if month < 1 || month > 12 {
1491
		return 0, 0, 0, 0, fmt.Errorf("invalid month %d in %q", month, lit)
1492
	}
1493
1494
	day = nums[len(parts)-1]
1495
	if day < 1 || day > 31 {
1496
		return 0, 0, 0, 0, fmt.Errorf("invalid day %d in %q", day, lit)
1497
	}
1498
1499
	if len(parts) == 2 {
1500
		return 0, month, day, sep, nil
1501
	}
1502
	return nums[0], month, day, sep, nil
1503
}
1504
1505
func dateSeparator(lit string) byte {
1506
	for i := 0; i < len(lit); i++ {
1507
		if lit[i] == '/' || lit[i] == '-' || lit[i] == '.' {
1508
			return lit[i]
1509
		}
1510
	}
1511
	return 0
1512
}
1513
1514
// parseCommentTags extacts tags from comment text.
1515
// A tag is a word immediately followed by a ':', with an optional value that ends at a comma or the end of a line.
1516
// https://hledger.org/1.52/hledger.html?highlight=tags#tags
1517
func parseCommentTags(text string, base token.Pos) []ast.Tag {
1518
	var tags []ast.Tag
1519
	for i := 0; i < len(text); {
1520
		colon := strings.IndexByte(text[i:], ':')
1521
		if colon < 0 {
1522
			break
1523
		}
1524
		colon += i
1525
1526
		keyStart := colon
1527
		for keyStart > i {
1528
			r, size := utf8.DecodeLastRuneInString(text[:keyStart])
1529
			if unicode.IsSpace(r) {
1530
				break
1531
			}
1532
			keyStart -= size
1533
		}
1534
		if keyStart == colon { // nothing before the colon = not a tag
1535
			i = colon + 1
1536
			continue
1537
		}
1538
		key := text[keyStart:colon]
1539
1540
		valueEnd := colon + 1
1541
		for valueEnd < len(text) && text[valueEnd] != ',' {
1542
			valueEnd++
1543
		}
1544
		value := strings.TrimSpace(text[colon+1 : valueEnd])
1545
1546
		tags = append(tags, ast.Tag{
1547
			Key:   key,
1548
			Value: value,
1549
			Span: token.Span{
1550
				Start: tagPos(base, text, keyStart),
1551
				End:   tagPos(base, text, valueEnd),
1552
			},
1553
		})
1554
		i = valueEnd
1555
		if i < len(text) && text[i] == ',' {
1556
			i++
1557
		}
1558
	}
1559
1560
	return tags
1561
}
1562
1563
func tagPos(base token.Pos, text string, off int) token.Pos {
1564
	return token.Pos{
1565
		File:   base.File,
1566
		Offset: base.Offset + off,
1567
		Line:   base.Line,
1568
		Col:    base.Col + utf8.RuneCountInString(text[:off]),
1569
	}
1570
}