all repos

clerk @ e76a121

missing tooling for ledger/hledger

clerk/journal/parser/parser.go (view raw)

Oleksandr Smirnov Oleksandr Smirnov
olexsmir@gmail.com
remove duplication of token.Pos.File in token.Span, 1 month ago
1
package parser
2
3
import (
4
	"fmt"
5
	"strconv"
6
	"strings"
7
	"unicode"
8
	"unicode/utf8"
9
10
	"olexsmir.xyz/clerk/internal/decimal"
11
	"olexsmir.xyz/clerk/journal/ast"
12
	"olexsmir.xyz/clerk/journal/lexer"
13
	"olexsmir.xyz/clerk/journal/token"
14
)
15
16
type Parser struct {
17
	lexer  *lexer.Lexer
18
	errors []*ast.ParseError
19
	cur    token.Token
20
	peek   token.Token
21
22
	defaultYear int // set by year directive, used for short date inference
23
}
24
25
func New(lex *lexer.Lexer) *Parser {
26
	p := &Parser{lexer: lex}
27
	p.advance() // populate .peek
28
	p.advance() // populate .cur
29
	return p
30
}
31
32
func NewWithYear(lex *lexer.Lexer, year int) *Parser {
33
	p := &Parser{lexer: lex, defaultYear: year}
34
	p.advance() // populate .peek
35
	p.advance() // populate .cur
36
	return p
37
}
38
39
func (p *Parser) ParseJournal() *ast.Journal {
40
	f := &ast.Journal{}
41
	for p.cur.Type != token.EOF {
42
		if e := p.parseEntry(); e != nil {
43
			f.Entries = append(f.Entries, e)
44
		}
45
	}
46
	f.Errors = p.errors
47
	return f
48
}
49
50
func (p *Parser) parseEntry() ast.Entry {
51
	if p.got(token.BANG) || p.got(token.AT) {
52
		if isDirectiveKeyword(p.peek.Type) {
53
			p.advance() // consume prefix
54
		}
55
	}
56
57
	switch p.cur.Type {
58
	case token.ILLEGAL:
59
		p.errorf("illegal character %q", p.cur.Literal)
60
		p.advance()
61
		return nil
62
	case token.INDENT:
63
		p.errorf("unexpected indent")
64
		p.syncToNextline()
65
		return nil
66
	case token.DATE:
67
		return p.parseTransaction()
68
	case token.TILDE:
69
		return p.parsePeriodicTransaction()
70
	case token.EQ:
71
		return p.parseAutomatedTransaction()
72
	case token.NEWLINE:
73
		return p.parseBlankLine()
74
	case token.SEMICOLON, token.HASH, token.PERCENT, token.STAR:
75
		return p.parseComment()
76
	case token.ACCOUNT:
77
		return p.parseAccountDirective()
78
	case token.COMMODITY:
79
		return p.parseCommodityDirective()
80
	case token.INCLUDE:
81
		return p.parseIncludeDirective()
82
	case token.ALIAS:
83
		return p.parseAliasDirective()
84
	case token.PAYEE:
85
		return p.parsePayeeDirective()
86
	case token.TAG:
87
		return p.parseTagDirective()
88
	case token.YEAR:
89
		return p.parseYearDirective()
90
	case token.DECIMALMARK:
91
		return p.parseDecimalMarkDirective()
92
	case token.D:
93
		return p.parseDefaultCommodityDirective()
94
	case token.P:
95
		return p.parseMarketPriceDirective()
96
	case token.N:
97
		return p.parseIgnoredDirective()
98
	case token.C:
99
		return p.parseConversionDirective()
100
	case token.APPLY:
101
		return p.parseApplyDirective()
102
	case token.END:
103
		return p.parseEndDirective()
104
	case token.COMMENTKW:
105
		return p.parseCommentBlockDirective()
106
	default:
107
		p.errorf("unexpected token %s", p.cur.Type)
108
		p.sync()
109
		return nil
110
	}
111
}
112
113
func (p *Parser) parseTransaction() *ast.Transaction {
114
	s := p.cur.Span
115
	tx := &ast.Transaction{}
116
117
	tx.Date = p.parseDate()
118
119
	p.skipWhitespace()
120
121
	// optional secondary date
122
	if p.got(token.EQ) {
123
		p.advance()
124
		p.skipWhitespace()
125
		d := p.parseDate()
126
		tx.SecondDate = &d
127
	}
128
129
	p.skipWhitespace()
130
131
	// optional status
132
	tx.Status = p.parseStatus()
133
134
	// optional code - the lexer emits "(CODE)" as a single TEXT token; split it here
135
	if p.got(token.TEXT) {
136
		if lit := p.cur.Literal; len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
137
			tx.Code = &ast.Code{Value: lit[1 : len(lit)-1], Span: p.cur.Span}
138
			p.advance()
139
			p.skipWhitespace()
140
		}
141
	}
142
143
	// optional payee | note
144
	if p.got(token.TEXT) || p.got(token.STRING) {
145
		tx.Payee = p.parsePayee()
146
147
		// check for | separator
148
		p.skipWhitespace()
149
150
		if p.got(token.PIPE) {
151
			p.advance()
152
			if p.got(token.TEXT) {
153
				sn := p.cur.Span
154
				n := p.cur.Literal
155
				p.advance()
156
				tx.Note = &ast.Note{Value: n, Span: p.span(sn)}
157
			}
158
		}
159
	}
160
161
	tx.Comment = p.parseOptInlineComment()
162
	p.expectNewline()
163
164
	tx.HeaderComments, tx.Postings = p.parseHeaderCommentsAndPostings()
165
166
	tx.Span = p.span(s)
167
	return tx
168
}
169
170
func unquote(s string) string {
171
	if len(s) >= 2 && ((s[0] == '"' && s[len(s)-1] == '"') || (s[0] == '\'' && s[len(s)-1] == '\'')) {
172
		return s[1 : len(s)-1]
173
	}
174
	return s
175
}
176
177
func (p *Parser) parsePayee() *ast.Payee {
178
	s := p.cur.Span
179
180
	if p.got(token.STRING) {
181
		name := unquote(p.cur.Literal)
182
		p.advance()
183
		return &ast.Payee{Name: name, Span: p.span(s)}
184
	}
185
186
	// keep spaces/tags between text tokens; stop before trailing whitespace
187
	var name strings.Builder
188
	for isPayeeWord(p.cur.Type) || (isPayeeWord(p.peek.Type) && p.got(token.WHITESPACE)) {
189
		_, _ = name.WriteString(p.cur.Literal)
190
		p.advance()
191
	}
192
	return &ast.Payee{Name: unquote(name.String()), Span: p.span(s)}
193
}
194
195
func isPayeeWord(t token.Type) bool {
196
	switch t {
197
	case token.TEXT, token.INT, token.DECIMAL, token.COMMODITYMARK:
198
		return true
199
	}
200
	return false
201
}
202
203
func (p *Parser) parsePeriodicTransaction() *ast.PeriodicTransaction {
204
	s := p.cur.Span
205
	p.expect(token.TILDE)
206
	p.skipWhitespace()
207
208
	pt := &ast.PeriodicTransaction{}
209
210
	pt.Period = p.parsePeriod()
211
212
	if desc, dspan := p.parseOptPeriodicDescription(); desc != "" {
213
		pt.Description = &ast.Description{Value: desc, Span: dspan}
214
	}
215
216
	comment := p.parseOptInlineComment()
217
	p.expectNewline()
218
219
	pt.HeaderComments, pt.Postings = p.parseHeaderCommentsAndPostings()
220
221
	pt.Span = p.span(s)
222
	pt.Comment = comment
223
	return pt
224
}
225
226
func (p *Parser) parseAutomatedTransaction() *ast.AutomatedTransaction {
227
	s := p.cur.Span
228
	p.expect(token.EQ)
229
	p.skipWhitespace()
230
231
	at := &ast.AutomatedTransaction{}
232
233
	// expression
234
	sd := p.cur.Span
235
	expr := p.parseDirectiveExpr()
236
	at.Expr = ast.Expr{Value: expr, Span: p.span(sd)}
237
	at.Comment = p.parseOptInlineComment()
238
	p.expectNewline()
239
240
	at.HeaderComments, at.Postings = p.parseHeaderCommentsAndPostings()
241
242
	at.Span = p.span(s)
243
	return at
244
}
245
246
func (p *Parser) parseHeaderCommentsAndPostings() (comments []*ast.Comment, postings []*ast.Posting) {
247
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
248
		p.advance() // consume indent
249
		comments = append(comments, p.parseComment())
250
	}
251
252
	for p.got(token.INDENT) {
253
		if posting := p.parsePosting(); posting != nil {
254
			postings = append(postings, posting)
255
		}
256
	}
257
258
	return comments, postings
259
}
260
261
func (p *Parser) parsePeriod() ast.Period {
262
	s := p.cur.Span
263
264
	var periodBuf strings.Builder
265
266
	for !p.got(token.NEWLINE) && !p.got(token.EOF) &&
267
		!p.got(token.SEMICOLON) && !p.got(token.HASH) && !p.got(token.PERCENT) && !p.got(token.STAR) {
268
269
		if p.got(token.WHITESPACE) {
270
			if len(p.cur.Literal) >= 2 {
271
				break
272
			}
273
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) ||
274
				p.willGet(token.SEMICOLON) || p.willGet(token.HASH) ||
275
				p.willGet(token.PERCENT) || p.willGet(token.STAR) {
276
				p.advance()
277
				continue
278
			}
279
		}
280
281
		periodBuf.WriteString(p.cur.Literal)
282
		p.advance()
283
	}
284
285
	str := periodBuf.String()
286
	period := ast.Period{Raw: str, Span: p.span(s)}
287
288
	if _, after, ok := strings.Cut(str, " from "); ok {
289
		end := strings.Index(after, " ")
290
		dateStr := after
291
		if end >= 0 {
292
			dateStr = after[:end]
293
		}
294
		if d := parseSimpleDate(dateStr); d.Year > 0 {
295
			fromOff := strings.Index(str, dateStr)
296
			d.Span = periodDateSpan(period, str, dateStr, fromOff)
297
			period.From = &d
298
			rest := after
299
			if end >= 0 {
300
				rest = after[end:]
301
			}
302
			if _, toAfter, ok := strings.Cut(rest, " to "); ok {
303
				if toEnd := strings.Index(toAfter, " "); toEnd >= 0 {
304
					toAfter = toAfter[:toEnd]
305
				}
306
				if d := parseSimpleDate(toAfter); d.Year > 0 {
307
					d.Span = periodDateSpan(period, str, toAfter, fromOff+len(dateStr))
308
					period.To = &d
309
				}
310
			}
311
		}
312
	}
313
	return period
314
}
315
316
// periodDateSpan returns the source span of dateStr, which occurs in the
317
// period text at or after searchFrom. The period span and text cover the same
318
// bytes, so offsets line up 1:1.
319
func periodDateSpan(period ast.Period, text, dateStr string, searchFrom int) token.Span {
320
	off := strings.Index(text[searchFrom:], dateStr)
321
	abs := period.Span.Start.Offset + searchFrom + off
322
	return token.Span{
323
		File:  period.Span.File,
324
		Start: token.Pos{Offset: abs},
325
		End:   token.Pos{Offset: abs + len(dateStr)},
326
	}
327
}
328
329
func (p *Parser) parseComment() *ast.Comment {
330
	s := p.cur.Span
331
	c := p.parseCommentRest(s)
332
	p.expectNewline()
333
	c.Span = p.span(s) // comment spans its line through the newline
334
	return c
335
}
336
337
func (p *Parser) parseAccountDirective() *ast.AccountDirective {
338
	s := p.cur.Span
339
	p.expect(token.ACCOUNT)
340
	p.skipWhitespace()
341
342
	account := p.parseAccount()
343
	comment := p.parseOptInlineComment()
344
	p.expectNewline()
345
346
	var subs []ast.AccountSubdirective
347
	for p.got(token.INDENT) {
348
		p.advance()
349
		p.skipWhitespace()
350
		if p.got(token.NEWLINE) || p.got(token.EOF) {
351
			// whitespace-only line: block continues
352
			p.expectNewline()
353
			continue
354
		}
355
		switch {
356
		case p.got(token.SEMICOLON):
357
			// comment line: directive mode lexes only ';' as a comment marker
358
			ns := p.cur.Span
359
			c := p.parseCommentRest(ns)
360
			p.expectNewline()
361
			subs = append(subs, ast.AccountSubdirective{Kind: ast.SubdirectiveComment, NameSpan: ns, Comment: c})
362
		case p.got(token.TEXT) && isCommentMarker(p.cur.Literal):
363
			// '#', '%' and '*' lex as TEXT in directive mode; treat them as comment lines
364
			c := p.parseTextComment()
365
			subs = append(subs, ast.AccountSubdirective{Kind: ast.SubdirectiveComment, NameSpan: c.Span, Comment: c})
366
		case p.got(token.TEXT):
367
			name := p.cur.Literal
368
			kind, ok := accountSubdirectiveKind(name)
369
			if !ok {
370
				p.errorf("unknown subdirective %q", name)
371
				p.skipToNewline()
372
				continue
373
			}
374
			kw := p.cur.Span
375
			p.advance()
376
			value, vspan := p.parseSubdirectiveValue()
377
			if value == "" {
378
				p.errorf("expected value for subdirective %q", name)
379
			}
380
			c := p.parseOptInlineComment()
381
			p.expectNewline()
382
			subs = append(subs, ast.AccountSubdirective{
383
				Kind:      kind,
384
				NameSpan:  kw,
385
				Value:     value,
386
				ValueSpan: vspan,
387
				Comment:   c,
388
			})
389
		default:
390
			p.errorf("expected subdirective name, got %s", p.cur.Type)
391
			p.skipToNewline()
392
		}
393
	}
394
395
	return &ast.AccountDirective{
396
		Account:       account,
397
		Subdirectives: subs,
398
		Comment:       comment,
399
		Span:          p.span(s),
400
	}
401
}
402
403
func (p *Parser) parseCommodityDirective() *ast.CommodityDirective {
404
	s := p.cur.Span
405
	p.expect(token.COMMODITY)
406
	p.skipWhitespace()
407
408
	var commodity string
409
	var commoditySpan token.Span
410
	var format *ast.FormatSubDirective
411
412
	switch p.cur.Type {
413
	case token.COMMODITYMARK, token.TEXT, token.STRING:
414
		cs := p.cur.Span
415
		commodity = unquote(p.cur.Literal)
416
		p.advance()
417
		commoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
418
		hadSpace := p.got(token.WHITESPACE)
419
		p.skipWhitespace()
420
		if p.got(token.INT) || p.got(token.DECIMAL) || p.got(token.TEXT) {
421
			amt := p.parseAmount()
422
			amt.Commodity = commodity
423
			amt.CommoditySpan = commoditySpan
424
			amt.CommodityPos = ast.CommodityBefore
425
			amt.HasSpace = hadSpace
426
			format = &ast.FormatSubDirective{Amount: *amt}
427
		}
428
	case token.INT, token.DECIMAL:
429
		amt := p.parseAmount()
430
		commodity = amt.Commodity
431
		commoditySpan = amt.CommoditySpan
432
		format = &ast.FormatSubDirective{Amount: *amt}
433
	default:
434
		p.errorf("expected commodity name or amount, got %s", p.cur.Type)
435
	}
436
437
	if commodity == "" {
438
		p.errorf("expected commodity name, got %s", p.cur.Type)
439
	}
440
441
	// hledger parity: an inline format amount must include a decimal mark
442
	if format != nil && format.Amount.QuantityFmt.Decimal == 0 {
443
		p.errorfAt(format.Amount.Span, "Please include a decimal point or decimal comma in commodity directives, to help us parse correctly. It may be followed by zero or more decimal digits.")
444
	}
445
446
	comment := p.parseOptInlineComment()
447
	p.expectNewline()
448
449
	var blockComments []*ast.Comment
450
	for p.got(token.INDENT) {
451
		p.advance()
452
		p.skipWhitespace()
453
		if p.got(token.NEWLINE) || p.got(token.EOF) {
454
			// whitespace-only line: block continues
455
			p.expectNewline()
456
			continue
457
		}
458
		switch {
459
		case p.got(token.TEXT) && p.cur.Literal == "format":
460
			kw := p.cur.Span
461
			p.advance()
462
			p.skipWhitespace()
463
			amt := p.parseAmount()
464
			// hledger parity: the format symbol must match the declared commodity,
465
			// and the amount must include a decimal mark; the node is kept either
466
			// way so the printer can round-trip the input.
467
			if amt.Commodity != commodity {
468
				p.errorfAt(amt.Span, "commodity directive symbol %q and format directive symbol %q should be the same", commodity, amt.Commodity)
469
			} else if amt.QuantityFmt.Decimal == 0 {
470
				p.errorfAt(amt.Span, "Please include a decimal point or decimal comma in commodity directives, to help us parse correctly. It may be followed by zero or more decimal digits.")
471
			}
472
			c := p.parseOptInlineComment()
473
			p.expectNewline()
474
			format = &ast.FormatSubDirective{KeywordSpan: kw, Amount: *amt, Comment: c}
475
		case p.got(token.SEMICOLON): // comment line
476
			c := p.parseCommentRest(p.cur.Span)
477
			p.expectNewline()
478
			blockComments = append(blockComments, c)
479
		case p.got(token.TEXT) && isCommentMarker(p.cur.Literal):
480
			// '#', '%' and '*' lex as TEXT in directive mode; treat them as comment lines
481
			blockComments = append(blockComments, p.parseTextComment())
482
		case p.got(token.TEXT):
483
			p.errorf("unknown subdirective %q", p.cur.Literal)
484
			p.skipToNewline()
485
		default:
486
			p.errorf("expected subdirective name, got %s", p.cur.Type)
487
			p.skipToNewline()
488
		}
489
	}
490
491
	cd := &ast.CommodityDirective{
492
		Commodity:     commodity,
493
		CommoditySpan: commoditySpan,
494
		FormatSub:     format,
495
		BlockComments: blockComments,
496
		Comment:       comment,
497
		Span:          p.span(s),
498
	}
499
	return cd
500
}
501
502
func (p *Parser) parseIncludeDirective() *ast.IncludeDirective {
503
	s := p.cur.Span
504
	p.expect(token.INCLUDE)
505
	p.skipWhitespace()
506
507
	id := &ast.IncludeDirective{}
508
509
	if p.got(token.TEXT) {
510
		id.Path = p.cur.Literal
511
		p.advance()
512
	} else {
513
		p.errorf("expected file path, got %s", p.cur.Type)
514
	}
515
516
	id.Comment = p.parseOptInlineComment()
517
	p.expectNewline()
518
	id.Span = p.span(s)
519
	return id
520
}
521
522
func (p *Parser) parseAliasDirective() *ast.AliasDirective {
523
	s := p.cur.Span
524
	alias := &ast.AliasDirective{}
525
	p.expect(token.ALIAS)
526
	p.skipWhitespace()
527
	alias.From = p.parseAccount()
528
	p.skipWhitespace()
529
	p.expect(token.EQ)
530
	p.skipWhitespace()
531
	alias.To = p.parseAccount()
532
	alias.Comment = p.parseOptInlineComment()
533
	p.expectNewline()
534
	alias.Span = p.span(s)
535
	return alias
536
}
537
538
func (p *Parser) parsePayeeDirective() *ast.PayeeDirective {
539
	s := p.cur.Span
540
	p.expect(token.PAYEE)
541
	p.skipWhitespace()
542
543
	var name *ast.Payee
544
	if p.got(token.TEXT) || p.got(token.STRING) || p.got(token.COMMODITYMARK) {
545
		name = p.parsePayee()
546
	}
547
548
	comment := p.parseOptInlineComment()
549
	p.expectNewline()
550
551
	return &ast.PayeeDirective{
552
		Name:    name,
553
		Comment: comment,
554
		Span:    p.span(s),
555
	}
556
}
557
558
func (p *Parser) parseTagDirective() *ast.TagDirective {
559
	s := p.cur.Span
560
	p.expect(token.TAG)
561
	p.skipWhitespace()
562
563
	name := ""
564
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) || p.got(token.STRING) {
565
		name = unquote(p.cur.Literal)
566
		p.advance()
567
	}
568
569
	comment := p.parseOptInlineComment()
570
	p.expectNewline()
571
572
	return &ast.TagDirective{
573
		Name:    name,
574
		Comment: comment,
575
		Span:    p.span(s),
576
	}
577
}
578
579
func (p *Parser) parseYearDirective() *ast.YearDirective {
580
	s := p.cur.Span
581
	year := &ast.YearDirective{}
582
	p.expect(token.YEAR)
583
	p.skipWhitespace()
584
585
	if p.got(token.INT) {
586
		year.Year, _ = strconv.Atoi(p.cur.Literal)
587
		p.defaultYear = year.Year
588
		p.advance()
589
	} else {
590
		p.errorf("expected year, got %s", p.cur.Type)
591
	}
592
593
	year.Comment = p.parseOptInlineComment()
594
	p.expectNewline()
595
	year.Span = p.span(s)
596
597
	return year
598
}
599
600
func (p *Parser) parseDecimalMarkDirective() *ast.DecimalMarkDirective {
601
	s := p.cur.Span
602
	mark := &ast.DecimalMarkDirective{}
603
	p.expect(token.DECIMALMARK)
604
	p.skipWhitespace()
605
606
	mark.Mark = byte('.')
607
	if p.got(token.TEXT) {
608
		if len(p.cur.Literal) > 0 {
609
			mark.Mark = p.cur.Literal[0]
610
		}
611
		p.advance()
612
	}
613
614
	mark.Comment = p.parseOptInlineComment()
615
	p.expectNewline()
616
	mark.Span = p.span(s)
617
	return mark
618
}
619
620
func (p *Parser) parseDefaultCommodityDirective() *ast.DefaultCommodityDirective {
621
	s := p.cur.Span
622
	com := &ast.DefaultCommodityDirective{}
623
	p.expect(token.D)
624
	p.skipWhitespace()
625
	com.Amount = *p.parseAmount()
626
	com.Comment = p.parseOptInlineComment()
627
	p.expectNewline()
628
	com.Span = p.span(s)
629
	return com
630
}
631
632
func (p *Parser) parseConversionDirective() *ast.ConversionDirective {
633
	s := p.cur.Span
634
	cd := &ast.ConversionDirective{}
635
	p.expect(token.C)
636
	p.skipWhitespace()
637
638
	if p.isAmountStart() {
639
		cd.From = *p.parseAmount()
640
	} else {
641
		p.errorf("expected amount, got %s", p.cur.Type)
642
	}
643
644
	p.skipWhitespace()
645
	if p.got(token.EQ) {
646
		p.advance()
647
		p.skipWhitespace()
648
		if p.isAmountStart() {
649
			cd.To = *p.parseAmount()
650
		} else {
651
			p.errorf("expected amount, got %s", p.cur.Type)
652
		}
653
	}
654
655
	cd.Comment = p.parseOptInlineComment()
656
	p.expectNewline()
657
	cd.Span = p.span(s)
658
	return cd
659
}
660
661
func (p *Parser) parseIgnoredDirective() *ast.IgnoredDirective {
662
	s := p.cur.Span
663
	p.expect(token.N)
664
	p.skipWhitespace()
665
666
	id := &ast.IgnoredDirective{}
667
	if p.got(token.TEXT) || p.got(token.COMMODITYMARK) || p.got(token.STRING) {
668
		id.Text = unquote(p.cur.Literal)
669
		p.advance()
670
	}
671
	id.Comment = p.parseOptInlineComment()
672
673
	p.expectNewline()
674
	id.Span = p.span(s)
675
	return id
676
}
677
678
func (p *Parser) parseMarketPriceDirective() *ast.MarketPriceDirective {
679
	s := p.cur.Span
680
	p.expect(token.P)
681
	p.skipWhitespace()
682
683
	mp := &ast.MarketPriceDirective{}
684
	mp.DateTime.Date = p.parseDate()
685
	p.skipWhitespace()
686
687
	if p.got(token.TIME) {
688
		mp.DateTime.Time = new(p.parseTime())
689
		p.skipWhitespace()
690
	}
691
692
	if p.got(token.COMMODITYMARK) || p.got(token.STRING) {
693
		mp.Commodity = unquote(p.cur.Literal)
694
		p.advance()
695
	} else {
696
		p.errorf("expected commodity symbol, got %s", p.cur.Type)
697
	}
698
	p.skipWhitespace()
699
700
	mp.Amount = *p.parseAmount()
701
702
	mp.Comment = p.parseOptInlineComment()
703
704
	p.expectNewline()
705
	mp.Span = p.span(s)
706
	return mp
707
}
708
709
func (p *Parser) parseTime() ast.Time {
710
	s := p.cur.Span
711
	tok, _ := p.expect(token.TIME)
712
	lit := tok.Literal
713
714
	parts := strings.Split(lit, ":")
715
	if len(parts) < 2 {
716
		p.errorf("invalid time format: %q", lit)
717
		return ast.Time{Span: p.span(s)}
718
	}
719
720
	hour, _ := strconv.Atoi(parts[0])
721
	minute, _ := strconv.Atoi(parts[1])
722
	second := 0
723
	if len(parts) > 2 {
724
		second, _ = strconv.Atoi(parts[2])
725
	}
726
727
	if hour < 0 || hour > 23 {
728
		p.errorf("invalid hour %d in time %q", hour, lit)
729
	}
730
	if minute < 0 || minute > 59 {
731
		p.errorf("invalid minute %d in time %q", minute, lit)
732
	}
733
	if second < 0 || second > 59 {
734
		p.errorf("invalid second %d in time %q", second, lit)
735
	}
736
737
	return ast.Time{
738
		Hour:   hour,
739
		Minute: minute,
740
		Second: second,
741
		Span:   p.span(s),
742
	}
743
}
744
745
func (p *Parser) parseApplyDirective() *ast.ApplyDirective {
746
	s := p.cur.Span
747
	p.expect(token.APPLY)
748
	p.skipWhitespace()
749
750
	expr := p.parseDirectiveExpr()
751
	comment := p.parseOptInlineComment()
752
	p.expectNewline()
753
754
	return &ast.ApplyDirective{
755
		Expr:    expr,
756
		Comment: comment,
757
		Span:    p.span(s),
758
	}
759
}
760
761
func (p *Parser) parseEndDirective() *ast.EndDirective {
762
	s := p.cur.Span
763
	p.expect(token.END)
764
	p.skipWhitespace()
765
766
	expr := p.parseDirectiveExpr()
767
	comment := p.parseOptInlineComment()
768
	p.expectNewline()
769
770
	return &ast.EndDirective{
771
		Expr:    expr,
772
		Comment: comment,
773
		Span:    p.span(s),
774
	}
775
}
776
777
func (p *Parser) parseCommentBlockDirective() *ast.CommentBlockDirective {
778
	start := p.cur.Span
779
	p.expect(token.COMMENTKW)
780
	p.skipWhitespace()
781
782
	header := p.parseDirectiveExpr()
783
	comment := p.parseOptInlineComment()
784
	p.expectNewline()
785
786
	var content strings.Builder
787
	for p.cur.Type != token.EOF {
788
		if p.got(token.END) {
789
			if p.willGet(token.NEWLINE) || p.willGet(token.EOF) {
790
				p.advance()
791
				p.expectNewline()
792
				break
793
			}
794
			if p.willGet(token.WHITESPACE) {
795
				endTok := p.cur
796
				p.advance()
797
				wsTok := p.cur
798
				p.advance()
799
				if p.got(token.TEXT) && p.cur.Literal == "comment" { // todo: this should check if it's an actual COMMENTKW token
800
					p.advance()
801
					p.parseDirectiveExpr()
802
					p.parseOptInlineComment()
803
					p.expectNewline()
804
					break
805
				}
806
				content.WriteString(endTok.Literal)
807
				content.WriteString(wsTok.Literal)
808
				continue
809
			}
810
		}
811
		content.WriteString(p.cur.Literal)
812
		p.advance()
813
	}
814
815
	return &ast.CommentBlockDirective{
816
		Header:  header,
817
		Content: content.String(),
818
		Comment: comment,
819
		Span:    p.span(start),
820
	}
821
}
822
823
func (p *Parser) parseStatus() ast.Status {
824
	s := p.cur.Span
825
	st := ast.Status{}
826
	switch p.cur.Type {
827
	case token.STAR:
828
		st.Value = ast.StatusCleared
829
	case token.BANG:
830
		st.Value = ast.StatusPending
831
	}
832
	if st.Value != ast.StatusNone {
833
		p.advance()
834
		p.skipWhitespace()
835
	}
836
	st.Span = p.span(s)
837
	return st
838
}
839
840
func (p *Parser) isAmountStart() bool {
841
	switch p.cur.Type {
842
	default:
843
		return false
844
	case token.COMMODITYMARK, token.STRING, token.INT, token.DECIMAL, token.MINUS, token.PLUS, token.PARENEXPR, token.STAR:
845
		return true
846
	}
847
}
848
849
func (p *Parser) parseAmount() *ast.Amount {
850
	s := p.cur.Span
851
	amt := &ast.Amount{
852
		QuantityFmt: ast.QuantityFormat{},
853
	}
854
	defer func() {
855
		// The span covers from the first token to the start of the next unconsumed token.
856
		// Since parseQuantityInto (and possible commodity consumption) advanced past the last
857
		// amount token, p.cur points to the next token after the amount — which is the correct end.
858
		amt.Span = p.span(s)
859
	}()
860
861
	p.parseAmountSign(amt)
862
	p.skipWhitespace()
863
864
	// commodity before quantity: $10.00, eur 10.00
865
	if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
866
		cs := p.cur.Span
867
		amt.Commodity = unquote(p.cur.Literal)
868
		amt.CommodityPos = ast.CommodityBefore
869
		p.advance()
870
		amt.CommoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
871
		if p.got(token.WHITESPACE) {
872
			amt.HasSpace = true
873
			p.skipWhitespace()
874
		}
875
	}
876
877
	// optional sign after commodity: $ -10
878
	p.parseAmountSign(amt)
879
	p.skipWhitespace()
880
881
	p.parseQuantityInto(amt)
882
883
	// commodity after quantity: 10.00 UAH, 10.00 "EUR" (only if not set)
884
	if amt.Commodity == "" {
885
		switch p.cur.Type {
886
		case token.WHITESPACE:
887
			p.skipWhitespace()
888
			if p.got(token.COMMODITYMARK) || p.got(token.TEXT) || p.got(token.STRING) {
889
				cs := p.cur.Span
890
				amt.HasSpace = true
891
				amt.Commodity = unquote(p.cur.Literal)
892
				amt.CommodityPos = ast.CommodityAfter
893
				p.advance()
894
				amt.CommoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
895
			}
896
		case token.COMMODITYMARK, token.TEXT, token.STRING:
897
			cs := p.cur.Span
898
			amt.Commodity = unquote(p.cur.Literal)
899
			amt.CommodityPos = ast.CommodityAfter
900
			p.advance()
901
			amt.CommoditySpan = token.Span{File: cs.File, Start: cs.Start, End: p.cur.Span.Start}
902
		}
903
	}
904
905
	return amt
906
}
907
908
// parseAmountSign consumes an optional leading +/- into IsNegative.
909
func (p *Parser) parseAmountSign(amt *ast.Amount) {
910
	switch p.cur.Type {
911
	case token.MINUS:
912
		amt.IsNegative = true
913
		p.advance()
914
	case token.PLUS:
915
		p.advance()
916
	}
917
}
918
919
func (p *Parser) parseAmountWithOptExpr() *ast.Amount {
920
	if p.got(token.STAR) {
921
		p.advance()
922
		p.skipWhitespace()
923
		amt := p.parseAmount()
924
		if amt != nil {
925
			amt.IsExpr = true
926
		}
927
		return amt
928
	}
929
	if p.got(token.PARENEXPR) {
930
		lit := p.cur.Literal
931
		amt := &ast.Amount{
932
			IsExpr:      true,
933
			QuantityFmt: ast.QuantityFormat{},
934
		}
935
		if len(lit) >= 2 && lit[0] == '(' && lit[len(lit)-1] == ')' {
936
			amt.Expr = strings.Trim(lit[1:len(lit)-1], " \t")
937
		}
938
		amt.Span = p.cur.Span
939
		p.advance()
940
		return amt
941
	}
942
	return p.parseAmount()
943
}
944
945
func (p *Parser) parsePosting() *ast.Posting {
946
	s := p.cur.Span
947
	posting := &ast.Posting{}
948
	p.expect(token.INDENT)
949
950
	// exit if it's empty line
951
	if p.got(token.NEWLINE) || p.got(token.EOF) {
952
		p.syncToNextline()
953
		return nil
954
	}
955
956
	// optional status, outside of brackets, '! (account)'
957
	posting.Status = p.parseStatus()
958
959
	// detect virtual posting brackets
960
	switch p.cur.Type {
961
	case token.LPAREN:
962
		posting.Type = ast.PostingVirtualUnbalanced
963
		p.advance()
964
	case token.LBRACKET:
965
		posting.Type = ast.PostingVirtualBalanced
966
		p.advance()
967
	}
968
969
	// optional status, inside of brackets, '(* account)'
970
	if p.got(token.STAR) || p.got(token.BANG) {
971
		posting.Status = p.parseStatus()
972
	}
973
974
	// validate, must be account text
975
	if p.cur.Type != token.TEXT {
976
		p.errorf("expected account name, got %s", p.cur.Type)
977
		p.syncToNextline()
978
		return nil
979
	}
980
981
	posting.Account = p.parseAccount()
982
983
	// consume closing bracket
984
	switch p.cur.Type {
985
	case token.RPAREN:
986
		p.advance()
987
	case token.RBRACKET:
988
		p.advance()
989
	}
990
991
	// optional amount - after two spaces
992
	if p.got(token.WHITESPACE) {
993
		p.skipWhitespace()
994
		if p.isAmountStart() {
995
			posting.Amount = p.parseAmountWithOptExpr()
996
		}
997
	}
998
999
	// optional cost '@' or '@@'
1000
	p.skipWhitespace()
1001
	if p.got(token.AT) || p.got(token.ATAT) {
1002
		posting.Cost = p.parseCost()
1003
	}
1004
1005
	// optional balance assertion or assignment
1006
	p.skipWhitespace()
1007
	if p.got(token.COLON) && p.willGet(token.EQ) {
1008
		p.advance() // consume ':' of ':='
1009
		posting.Balance = p.parseBalanceAssertion()
1010
		posting.Balance.IsAssignment = true
1011
	} else if p.got(token.EQ) || p.got(token.EQEQ) || p.got(token.EQEQEQ) || p.got(token.EQSTAR) {
1012
		posting.Balance = p.parseBalanceAssertion()
1013
	}
1014
1015
	posting.Comment = p.parseOptInlineComment()
1016
	p.expectNewline()
1017
1018
	// continuation comments
1019
	for p.got(token.INDENT) && p.willGet(token.SEMICOLON) {
1020
		p.advance()
1021
		c := p.parseComment()
1022
		posting.Comments = append(posting.Comments, *c)
1023
	}
1024
1025
	posting.Span = p.span(s)
1026
	return posting
1027
}
1028
1029
func (p *Parser) parseCost() *ast.Cost {
1030
	s := p.cur.Span
1031
	isTotal := p.got(token.ATAT)
1032
	p.advance() // consume '@' '@@'
1033
	p.skipWhitespace()
1034
	return &ast.Cost{
1035
		IsTotal: isTotal,
1036
		Amount:  *p.parseAmount(),
1037
		Span:    p.span(s),
1038
	}
1039
}
1040
1041
func (p *Parser) parseBalanceAssertion() *ast.BalanceAssertion {
1042
	s := p.cur.Span
1043
1044
	ba := &ast.BalanceAssertion{}
1045
	switch p.cur.Type {
1046
	case token.EQ: // basic assertion
1047
	case token.EQSTAR: // inclusive assertion
1048
		ba.IsInclusive = true
1049
	case token.EQEQ: // strict assertion
1050
		ba.IsStrict = true
1051
	case token.EQEQEQ: // strict inclusive assertion
1052
		ba.IsStrict = true
1053
		ba.IsInclusive = true
1054
	}
1055
	p.advance()
1056
	p.skipWhitespace()
1057
1058
	ba.Amount = *p.parseAmount()
1059
	p.skipWhitespace()
1060
	if p.got(token.AT) || p.got(token.ATAT) {
1061
		c := p.parseCost()
1062
		ba.Cost = c
1063
	}
1064
	ba.Span = p.span(s)
1065
	return ba
1066
}
1067
1068
func (p *Parser) readAccountSegment() (ast.SubAccount, bool) {
1069
	switch p.cur.Type {
1070
	case token.TEXT:
1071
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
1072
		p.advance()
1073
1074
		// handle multi work segment, e.g: "credit card"
1075
		if p.got(token.WHITESPACE) && p.willGet(token.TEXT) && len(p.peek.Literal) > 0 && p.peek.Literal[0] != '(' {
1076
			sub.Name += " "
1077
			p.advance()
1078
			sub.Name += p.cur.Literal
1079
			p.advance()
1080
		}
1081
		return sub, true
1082
1083
	case token.COMMODITYMARK:
1084
		sub := ast.SubAccount{Name: p.cur.Literal, Span: p.cur.Span}
1085
		p.advance()
1086
		// merge "EUR" + "-HRK" to "EUR-HRK"
1087
		for p.got(token.TEXT) {
1088
			sub.Name += p.cur.Literal
1089
			p.advance()
1090
		}
1091
		return sub, true
1092
1093
	default:
1094
		return ast.SubAccount{}, false
1095
	}
1096
}
1097
1098
func (p *Parser) parseAccount() ast.Account {
1099
	s := p.cur.Span
1100
	acc := ast.Account{}
1101
1102
	sub, ok := p.readAccountSegment()
1103
	if !ok {
1104
		p.errorf("expected account, got %s", p.cur.Type)
1105
		return ast.Account{}
1106
	}
1107
	acc.Name = append(acc.Name, sub)
1108
1109
	for p.got(token.COLON) {
1110
		p.advance()
1111
		sub, ok := p.readAccountSegment()
1112
		if !ok {
1113
			break
1114
		}
1115
		acc.Name = append(acc.Name, sub)
1116
	}
1117
1118
	acc.Span = p.span(s)
1119
	return acc
1120
}
1121
1122
func (p *Parser) parseDate() ast.Date {
1123
	s := p.cur.Span
1124
	tok, ok := p.expect(token.DATE)
1125
	if !ok {
1126
		return ast.Date{Span: p.span(s)}
1127
	}
1128
1129
	year, month, day, sep, err := ParseDateLiteral(tok.Literal)
1130
	if err != nil {
1131
		p.errorf("%v", err)
1132
		return ast.Date{Span: p.span(s)}
1133
	}
1134
	if year == 0 {
1135
		year = p.defaultYear
1136
	}
1137
1138
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep, Span: p.span(s)}
1139
}
1140
1141
func (p *Parser) parseOptInlineComment() *ast.Comment {
1142
	p.skipWhitespace()
1143
	if !p.got(token.SEMICOLON) {
1144
		return nil
1145
	}
1146
	return p.parseCommentRest(p.cur.Span)
1147
}
1148
1149
// parseCommentRest consumes a comment marker at p.cur, then optional text;
1150
// s anchors the span at the marker's start.
1151
func (p *Parser) parseCommentRest(s token.Span) *ast.Comment {
1152
	marker := p.cur.Literal[0]
1153
	p.advance()
1154
	p.skipWhitespace()
1155
1156
	var tags []ast.Tag
1157
	text := ""
1158
	if p.got(token.TEXT) {
1159
		text = p.cur.Literal
1160
		tags = parseCommentTags(text, p.cur.Span)
1161
		p.advance()
1162
	}
1163
1164
	return &ast.Comment{
1165
		Marker: marker,
1166
		Tags:   tags,
1167
		Text:   text,
1168
		Span:   p.span(s),
1169
	}
1170
}
1171
1172
func (p *Parser) parseOptPeriodicDescription() (string, token.Span) {
1173
	if p.cur.Type != token.WHITESPACE || len(p.cur.Literal) < 2 {
1174
		return "", token.Span{}
1175
	}
1176
1177
	p.skipWhitespace()
1178
1179
	if p.cur.Type != token.TEXT {
1180
		return "", token.Span{}
1181
	}
1182
1183
	s := p.cur.Span
1184
	desc := p.parseDescription()
1185
	return desc, p.span(s)
1186
}
1187
1188
func (p *Parser) parseDescription() string {
1189
	var desc strings.Builder
1190
	for p.got(token.TEXT) || (p.got(token.WHITESPACE) && p.willGet(token.TEXT)) {
1191
		_, _ = desc.WriteString(p.cur.Literal)
1192
		p.advance()
1193
	}
1194
	return desc.String()
1195
}
1196
1197
func (p *Parser) parseDirectiveExpr() string {
1198
	var b strings.Builder
1199
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF && p.cur.Type != token.SEMICOLON {
1200
		_, _ = b.WriteString(p.cur.Literal)
1201
		p.advance()
1202
	}
1203
	return b.String()
1204
}
1205
1206
func (p *Parser) parseQuantityInto(amt *ast.Amount) {
1207
	if p.cur.Type != token.INT && p.cur.Type != token.DECIMAL && p.cur.Type != token.TEXT {
1208
		p.errorf("expected quantity, got %s", p.cur.Type)
1209
		return
1210
	}
1211
1212
	lit := p.cur.Literal
1213
	p.advance()
1214
1215
	// detect format metadata before normalizing
1216
	amt.QuantityFmt = detectFormat(lit)
1217
1218
	// normalize for decimal.NewFromString
1219
	// remove thousands separators, replace decimal mark with '.'
1220
	normalized := normalizeLiteral(lit, amt.QuantityFmt.Thousands, amt.QuantityFmt.Decimal)
1221
1222
	q, err := decimal.FromString(normalized)
1223
	if err != nil {
1224
		p.errorf("invalid quantity %q: %v", lit, err)
1225
		return
1226
	}
1227
1228
	if amt.IsNegative {
1229
		q = q.Neg()
1230
	}
1231
	amt.Quantity = q
1232
}
1233
1234
func (p *Parser) parseBlankLine() *ast.BlankLine {
1235
	s := p.cur.Span
1236
	p.expectNewline()
1237
	return &ast.BlankLine{Span: s}
1238
}
1239
1240
func (p *Parser) expectNewline() {
1241
	if p.got(token.NEWLINE) || p.got(token.EOF) {
1242
		if p.got(token.NEWLINE) {
1243
			p.advance()
1244
		}
1245
		return
1246
	}
1247
	p.errorf("expected %s, got %s", token.NEWLINE, p.cur.Type)
1248
}
1249
1250
func (p *Parser) advance() token.Token {
1251
	prev := p.cur
1252
	p.cur = p.peek
1253
	p.peek = p.lexer.Next()
1254
	return prev
1255
}
1256
1257
func (p *Parser) got(kind token.Type) bool     { return p.cur.Type == kind }
1258
func (p *Parser) willGet(kind token.Type) bool { return p.peek.Type == kind }
1259
1260
func (p *Parser) expect(kind token.Type) (token.Token, bool) {
1261
	if p.got(kind) {
1262
		return p.advance(), true
1263
	}
1264
	p.errorf("expected %s, got %s", kind, p.cur.Type)
1265
	return p.cur, false
1266
}
1267
1268
func (p *Parser) errorf(format string, args ...any) {
1269
	p.errors = append(p.errors, &ast.ParseError{
1270
		Span:    p.cur.Span,
1271
		Message: fmt.Sprintf(format, args...),
1272
	})
1273
}
1274
1275
// errorfAt records a parse error pointing at the start of span.
1276
func (p *Parser) errorfAt(span token.Span, format string, args ...any) {
1277
	p.errors = append(p.errors, &ast.ParseError{
1278
		Span:    token.Span{File: span.File, Start: span.Start, End: span.Start},
1279
		Message: fmt.Sprintf(format, args...),
1280
	})
1281
}
1282
1283
func accountSubdirectiveKind(name string) (ast.AccountSubdirectiveKind, bool) {
1284
	switch name {
1285
	case "alias":
1286
		return ast.SubdirectiveAlias, true
1287
	case "type":
1288
		return ast.SubdirectiveType, true
1289
	case "note":
1290
		return ast.SubdirectiveNote, true
1291
	}
1292
	return 0, false
1293
}
1294
1295
func isCommentMarker(s string) bool {
1296
	return len(s) > 0 && (s[0] == '#' || s[0] == '%' || s[0] == '*')
1297
}
1298
1299
// skipToNewline consumes the rest of the current line.
1300
func (p *Parser) skipToNewline() {
1301
	for !p.got(token.NEWLINE) && !p.got(token.EOF) {
1302
		p.advance()
1303
	}
1304
	p.expectNewline()
1305
}
1306
1307
func (p *Parser) parseSubdirectiveValue() (string, token.Span) {
1308
	var b strings.Builder
1309
	var first, last token.Span
1310
	var single, pendingWS string
1311
	for !p.got(token.SEMICOLON) && !p.got(token.NEWLINE) && !p.got(token.EOF) {
1312
		t := p.cur
1313
		if t.Type == token.WHITESPACE || t.Type == token.INDENT {
1314
			if first.Start.Offset > 0 {
1315
				pendingWS = t.Literal // only the last whitespace run matters
1316
			}
1317
			p.advance()
1318
			continue
1319
		}
1320
		if first.Start.Offset == 0 {
1321
			first, last = t.Span, t.Span
1322
			single = t.Literal
1323
		} else {
1324
			if single != "" {
1325
				b.WriteString(single)
1326
				single = ""
1327
			}
1328
			if pendingWS != "" {
1329
				b.WriteString(pendingWS)
1330
				pendingWS = ""
1331
			}
1332
			b.WriteString(t.Literal)
1333
			last = t.Span
1334
		}
1335
		p.advance()
1336
	}
1337
	if first.Start.Offset == 0 {
1338
		return "", token.Span{}
1339
	}
1340
	if single != "" {
1341
		return single, token.Span{File: first.File, Start: first.Start, End: last.End}
1342
	}
1343
	return strings.TrimSpace(b.String()), token.Span{File: first.File, Start: first.Start, End: last.End}
1344
}
1345
1346
func (p *Parser) parseTextComment() *ast.Comment {
1347
	s := p.cur.Span
1348
	marker := p.cur.Literal[0]
1349
	var b strings.Builder
1350
	b.WriteString(p.cur.Literal)
1351
	p.advance()
1352
	for !p.got(token.NEWLINE) && !p.got(token.EOF) {
1353
		b.WriteString(p.cur.Literal)
1354
		p.advance()
1355
	}
1356
	text := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(b.String()), string(marker)))
1357
	span := p.span(s) // marker to line end, without the newline
1358
	p.expectNewline()
1359
	return &ast.Comment{Marker: marker, Text: text, Span: span}
1360
}
1361
1362
func isDirectiveKeyword(t token.Type) bool {
1363
	switch t {
1364
	case token.COMMENTKW, token.ACCOUNT, token.COMMODITY, token.INCLUDE,
1365
		token.ALIAS, token.PAYEE, token.TAG, token.APPLY, token.END,
1366
		token.YEAR, token.DECIMALMARK, token.D, token.P, token.N, token.C:
1367
		return true
1368
	}
1369
	return false
1370
}
1371
1372
func (p *Parser) sync() {
1373
	for {
1374
		switch p.cur.Type {
1375
		case token.EOF:
1376
			return
1377
		case token.NEWLINE:
1378
			p.advance()
1379
			t := p.cur.Type
1380
			if isDirectiveKeyword(t) || t == token.DATE || t == token.TILDE || t == token.EQ {
1381
				return
1382
			}
1383
		default:
1384
			p.advance()
1385
		}
1386
	}
1387
}
1388
1389
func (p *Parser) syncToNextline() {
1390
	for p.cur.Type != token.NEWLINE && p.cur.Type != token.EOF {
1391
		p.advance()
1392
	}
1393
	if p.got(token.NEWLINE) {
1394
		p.advance()
1395
	}
1396
}
1397
1398
func (p *Parser) skipWhitespace() {
1399
	for p.got(token.WHITESPACE) {
1400
		p.advance()
1401
	}
1402
}
1403
1404
func (p *Parser) span(s token.Span) token.Span {
1405
	return token.Span{File: s.File, Start: s.Start, End: p.cur.Span.Start}
1406
}
1407
1408
func normalizeLiteral(lit string, thousands, decimal byte) string {
1409
	var b strings.Builder
1410
	for _, ch := range []byte(lit) {
1411
		if thousands != 0 && ch == thousands {
1412
			continue // skip thousands separator
1413
		}
1414
		if ch == decimal {
1415
			b.WriteByte('.')
1416
		} else {
1417
			b.WriteByte(ch)
1418
		}
1419
	}
1420
	return b.String()
1421
}
1422
1423
func detectFormat(lit string) ast.QuantityFormat {
1424
	var seps []int
1425
	for i, ch := range []byte(lit) {
1426
		if ch == '.' || ch == ',' || ch == ' ' || ch == '_' || ch == '\'' {
1427
			seps = append(seps, i)
1428
		}
1429
	}
1430
1431
	if len(seps) == 0 {
1432
		return ast.QuantityFormat{}
1433
	}
1434
1435
	last := seps[len(seps)-1]
1436
	dec := lit[last]
1437
	if dec != '.' && dec != ',' {
1438
		// the last separator is a thousands mark; the literal has no decimal mark
1439
		dec = 0
1440
	}
1441
	var thou byte
1442
	if len(seps) > 1 {
1443
		thou = lit[seps[0]]
1444
	} else if dec == 0 {
1445
		// single space/underscore/apostrophe is always thousands
1446
		thou = lit[last]
1447
	}
1448
1449
	// calculate precision when the last separator is a real decimal
1450
	prec := 0
1451
	if thou == 0 || len(seps) > 1 {
1452
		prec = len(lit) - last - 1
1453
	}
1454
1455
	return ast.QuantityFormat{Decimal: dec, Thousands: thou, Precision: prec}
1456
}
1457
1458
// parseSimpleDate  parses full YYYY/MM/DD date literal embedded in free text.
1459
func parseSimpleDate(s string) ast.Date {
1460
	year, month, day, sep, err := ParseDateLiteral(s)
1461
	if err != nil {
1462
		return ast.Date{}
1463
	}
1464
	return ast.Date{Year: year, Month: month, Day: day, Sep: sep}
1465
}
1466
1467
// ParseDateLiteral parses and validates a date literal.
1468
// It accepts full YYYY/MM/DD and partial MM/DD forms, with '-', '/' or '.' as separators.
1469
func ParseDateLiteral(lit string) (year, month, day int, sep byte, err error) {
1470
	sep = dateSeparator(lit)
1471
	if sep == 0 {
1472
		return 0, 0, 0, 0, fmt.Errorf("invalid date format: %q", lit)
1473
	}
1474
1475
	parts := strings.Split(lit, string(sep))
1476
	if len(parts) != 2 && len(parts) != 3 {
1477
		return 0, 0, 0, 0, fmt.Errorf("invalid date format: %q", lit)
1478
	}
1479
1480
	nums := make([]int, len(parts))
1481
	for i, part := range parts {
1482
		if nums[i], err = strconv.Atoi(part); err != nil {
1483
			return 0, 0, 0, 0, fmt.Errorf("invalid date literal: %q", lit)
1484
		}
1485
	}
1486
1487
	month = nums[len(parts)-2]
1488
	if month < 1 || month > 12 {
1489
		return 0, 0, 0, 0, fmt.Errorf("invalid month %d in %q", month, lit)
1490
	}
1491
1492
	day = nums[len(parts)-1]
1493
	if day < 1 || day > 31 {
1494
		return 0, 0, 0, 0, fmt.Errorf("invalid day %d in %q", day, lit)
1495
	}
1496
1497
	if len(parts) == 2 {
1498
		return 0, month, day, sep, nil
1499
	}
1500
	return nums[0], month, day, sep, nil
1501
}
1502
1503
func dateSeparator(lit string) byte {
1504
	for i := 0; i < len(lit); i++ {
1505
		if lit[i] == '/' || lit[i] == '-' || lit[i] == '.' {
1506
			return lit[i]
1507
		}
1508
	}
1509
	return 0
1510
}
1511
1512
// parseCommentTags extacts tags from comment text.
1513
// A tag is a word immediately followed by a ':', with an optional value that ends at a comma or the end of a line.
1514
// https://hledger.org/1.52/hledger.html?highlight=tags#tags
1515
func parseCommentTags(text string, base token.Span) []ast.Tag {
1516
	var tags []ast.Tag
1517
	for i := 0; i < len(text); {
1518
		colon := strings.IndexByte(text[i:], ':')
1519
		if colon < 0 {
1520
			break
1521
		}
1522
		colon += i
1523
1524
		keyStart := colon
1525
		for keyStart > i {
1526
			r, size := utf8.DecodeLastRuneInString(text[:keyStart])
1527
			if unicode.IsSpace(r) {
1528
				break
1529
			}
1530
			keyStart -= size
1531
		}
1532
		if keyStart == colon { // nothing before the colon = not a tag
1533
			i = colon + 1
1534
			continue
1535
		}
1536
		key := text[keyStart:colon]
1537
1538
		valueEnd := colon + 1
1539
		for valueEnd < len(text) && text[valueEnd] != ',' {
1540
			valueEnd++
1541
		}
1542
		value := strings.TrimSpace(text[colon+1 : valueEnd])
1543
1544
		tags = append(tags, ast.Tag{
1545
			Key:   key,
1546
			Value: value,
1547
			Span: token.Span{
1548
				File:  base.File,
1549
				Start: tagPos(base.Start, text, keyStart),
1550
				End:   tagPos(base.Start, text, valueEnd),
1551
			},
1552
		})
1553
		i = valueEnd
1554
		if i < len(text) && text[i] == ',' {
1555
			i++
1556
		}
1557
	}
1558
1559
	return tags
1560
}
1561
1562
func tagPos(base token.Pos, text string, off int) token.Pos {
1563
	return token.Pos{
1564
		Offset: base.Offset + off,
1565
		Line:   base.Line,
1566
		Col:    base.Col + utf8.RuneCountInString(text[:off]),
1567
	}
1568
}