1
2
3
4
5 package modfile
6
7 import (
8 "bytes"
9 "errors"
10 "fmt"
11 "os"
12 "slices"
13 "strconv"
14 "strings"
15 "unicode"
16 "unicode/utf8"
17 )
18
19
20
21 type Position struct {
22 Line int
23 LineRune int
24 Byte int
25 }
26
27
28 func (p Position) add(s string) Position {
29 p.Byte += len(s)
30 if n := strings.Count(s, "\n"); n > 0 {
31 p.Line += n
32 s = s[strings.LastIndex(s, "\n")+1:]
33 p.LineRune = 1
34 }
35 p.LineRune += utf8.RuneCountInString(s)
36 return p
37 }
38
39
40 type Expr interface {
41
42
43 Span() (start, end Position)
44
45
46
47
48 Comment() *Comments
49 }
50
51
52 type Comment struct {
53 Start Position
54 Token string
55 Suffix bool
56 }
57
58
59 type Comments struct {
60 Before []Comment
61 Suffix []Comment
62
63
64
65 After []Comment
66 }
67
68
69
70
71
72 func (c *Comments) Comment() *Comments {
73 return c
74 }
75
76
77 type FileSyntax struct {
78 Name string
79 Comments
80 Stmt []Expr
81 }
82
83 func (x *FileSyntax) Span() (start, end Position) {
84 if len(x.Stmt) == 0 {
85 return
86 }
87 start, _ = x.Stmt[0].Span()
88 _, end = x.Stmt[len(x.Stmt)-1].Span()
89 return start, end
90 }
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105 func (x *FileSyntax) addLine(hint Expr, tokens ...string) *Line {
106 if hint == nil {
107
108 Loop:
109 for _, stmt := range slices.Backward(x.Stmt) {
110 switch stmt := stmt.(type) {
111 case *Line:
112 if stmt.Token != nil && stmt.Token[0] == tokens[0] {
113 hint = stmt
114 break Loop
115 }
116 case *LineBlock:
117 if stmt.Token[0] == tokens[0] {
118 hint = stmt
119 break Loop
120 }
121 }
122 }
123 }
124
125 newLineAfter := func(i int) *Line {
126 new := &Line{Token: tokens}
127 if i == len(x.Stmt) {
128 x.Stmt = append(x.Stmt, new)
129 } else {
130 x.Stmt = append(x.Stmt, nil)
131 copy(x.Stmt[i+2:], x.Stmt[i+1:])
132 x.Stmt[i+1] = new
133 }
134 return new
135 }
136
137 if hint != nil {
138 for i, stmt := range x.Stmt {
139 switch stmt := stmt.(type) {
140 case *Line:
141 if stmt == hint {
142 if stmt.Token == nil || stmt.Token[0] != tokens[0] {
143 return newLineAfter(i)
144 }
145
146
147 stmt.InBlock = true
148 block := &LineBlock{Token: stmt.Token[:1], Line: []*Line{stmt}}
149 stmt.Token = stmt.Token[1:]
150 x.Stmt[i] = block
151 new := &Line{Token: tokens[1:], InBlock: true}
152 block.Line = append(block.Line, new)
153 return new
154 }
155
156 case *LineBlock:
157 if stmt == hint {
158 if stmt.Token[0] != tokens[0] {
159 return newLineAfter(i)
160 }
161
162 new := &Line{Token: tokens[1:], InBlock: true}
163 stmt.Line = append(stmt.Line, new)
164 return new
165 }
166
167 for j, line := range stmt.Line {
168 if line == hint {
169 if stmt.Token[0] != tokens[0] {
170 return newLineAfter(i)
171 }
172
173
174 stmt.Line = append(stmt.Line, nil)
175 copy(stmt.Line[j+2:], stmt.Line[j+1:])
176 new := &Line{Token: tokens[1:], InBlock: true}
177 stmt.Line[j+1] = new
178 return new
179 }
180 }
181 }
182 }
183 }
184
185 new := &Line{Token: tokens}
186 x.Stmt = append(x.Stmt, new)
187 return new
188 }
189
190 func (x *FileSyntax) updateLine(line *Line, tokens ...string) {
191 if line.InBlock {
192 tokens = tokens[1:]
193 }
194 line.Token = tokens
195 }
196
197
198
199 func (line *Line) markRemoved() {
200 line.Token = nil
201 line.Comments.Suffix = nil
202 }
203
204
205
206
207
208
209 func (x *FileSyntax) Cleanup() {
210 w := 0
211 for _, stmt := range x.Stmt {
212 switch stmt := stmt.(type) {
213 case *Line:
214 if stmt.Token == nil {
215 continue
216 }
217 case *LineBlock:
218 ww := 0
219 for _, line := range stmt.Line {
220 if line.Token != nil {
221 stmt.Line[ww] = line
222 ww++
223 }
224 }
225 if ww == 0 {
226 continue
227 }
228 if ww == 1 && len(stmt.RParen.Comments.Before) == 0 {
229
230
231 *stmt.Line[0] = Line{
232 Comments: Comments{
233 Before: commentsAdd(stmt.Before, stmt.Line[0].Before),
234 Suffix: commentsAdd(stmt.Line[0].Suffix, stmt.Suffix),
235 After: commentsAdd(stmt.Line[0].After, stmt.After),
236 },
237 Token: stringsAdd(stmt.Token, stmt.Line[0].Token),
238 }
239 x.Stmt[w] = stmt.Line[0]
240 w++
241 continue
242 }
243 stmt.Line = stmt.Line[:ww]
244 }
245 x.Stmt[w] = stmt
246 w++
247 }
248 x.Stmt = x.Stmt[:w]
249 }
250
251 func commentsAdd(x, y []Comment) []Comment {
252 return append(x[:len(x):len(x)], y...)
253 }
254
255 func stringsAdd(x, y []string) []string {
256 return append(x[:len(x):len(x)], y...)
257 }
258
259
260
261 type CommentBlock struct {
262 Comments
263 Start Position
264 }
265
266 func (x *CommentBlock) Span() (start, end Position) {
267 return x.Start, x.Start
268 }
269
270
271 type Line struct {
272 Comments
273 Start Position
274 Token []string
275 InBlock bool
276 End Position
277 }
278
279 func (x *Line) Span() (start, end Position) {
280 return x.Start, x.End
281 }
282
283
284
285
286
287
288
289 type LineBlock struct {
290 Comments
291 Start Position
292 LParen LParen
293 Token []string
294 Line []*Line
295 RParen RParen
296 }
297
298 func (x *LineBlock) Span() (start, end Position) {
299 return x.Start, x.RParen.Pos.add(")")
300 }
301
302
303
304 type LParen struct {
305 Comments
306 Pos Position
307 }
308
309 func (x *LParen) Span() (start, end Position) {
310 return x.Pos, x.Pos.add(")")
311 }
312
313
314
315 type RParen struct {
316 Comments
317 Pos Position
318 }
319
320 func (x *RParen) Span() (start, end Position) {
321 return x.Pos, x.Pos.add(")")
322 }
323
324
325 type input struct {
326
327 filename string
328 complete []byte
329 remaining []byte
330 tokenStart []byte
331 token token
332 pos Position
333 comments []Comment
334
335
336 file *FileSyntax
337 parseErrors ErrorList
338
339
340 pre []Expr
341 post []Expr
342 }
343
344 func newInput(filename string, data []byte) *input {
345 return &input{
346 filename: filename,
347 complete: data,
348 remaining: data,
349 pos: Position{Line: 1, LineRune: 1, Byte: 0},
350 }
351 }
352
353
354 func parse(file string, data []byte) (f *FileSyntax, err error) {
355
356
357
358
359 in := newInput(file, data)
360 defer func() {
361 if e := recover(); e != nil && e != &in.parseErrors {
362 in.parseErrors = append(in.parseErrors, Error{
363 Filename: in.filename,
364 Pos: in.pos,
365 Err: fmt.Errorf("internal error: %v", e),
366 })
367 }
368 if err == nil && len(in.parseErrors) > 0 {
369 err = in.parseErrors
370 }
371 }()
372
373
374
375 in.readToken()
376
377
378 in.parseFile()
379 if len(in.parseErrors) > 0 {
380 return nil, in.parseErrors
381 }
382 in.file.Name = in.filename
383
384
385 in.assignComments()
386
387 return in.file, nil
388 }
389
390
391
392 func (in *input) Error(s string) {
393 in.parseErrors = append(in.parseErrors, Error{
394 Filename: in.filename,
395 Pos: in.pos,
396 Err: errors.New(s),
397 })
398 panic(&in.parseErrors)
399 }
400
401
402 func (in *input) eof() bool {
403 return len(in.remaining) == 0
404 }
405
406
407 func (in *input) peekRune() int {
408 if len(in.remaining) == 0 {
409 return 0
410 }
411 r, _ := utf8.DecodeRune(in.remaining)
412 return int(r)
413 }
414
415
416 func (in *input) peekPrefix(prefix string) bool {
417
418
419 for i := 0; i < len(prefix); i++ {
420 if i >= len(in.remaining) || in.remaining[i] != prefix[i] {
421 return false
422 }
423 }
424 return true
425 }
426
427
428 func (in *input) readRune() int {
429 if len(in.remaining) == 0 {
430 in.Error("internal lexer error: readRune at EOF")
431 }
432 r, size := utf8.DecodeRune(in.remaining)
433 in.remaining = in.remaining[size:]
434 if r == '\n' {
435 in.pos.Line++
436 in.pos.LineRune = 1
437 } else {
438 in.pos.LineRune++
439 }
440 in.pos.Byte += size
441 return int(r)
442 }
443
444 type token struct {
445 kind tokenKind
446 pos Position
447 endPos Position
448 text string
449 }
450
451 type tokenKind int
452
453 const (
454 _EOF tokenKind = -(iota + 1)
455 _EOLCOMMENT
456 _IDENT
457 _STRING
458 _COMMENT
459
460
461 )
462
463 func (k tokenKind) isComment() bool {
464 return k == _COMMENT || k == _EOLCOMMENT
465 }
466
467
468 func (k tokenKind) isEOL() bool {
469 return k == _EOF || k == _EOLCOMMENT || k == '\n'
470 }
471
472
473
474
475 func (in *input) startToken() {
476 in.tokenStart = in.remaining
477 in.token.text = ""
478 in.token.pos = in.pos
479 }
480
481
482
483
484 func (in *input) endToken(kind tokenKind) {
485 in.token.kind = kind
486 text := string(in.tokenStart[:len(in.tokenStart)-len(in.remaining)])
487 if kind.isComment() {
488 if strings.HasSuffix(text, "\r\n") {
489 text = text[:len(text)-2]
490 } else {
491 text = strings.TrimSuffix(text, "\n")
492 }
493 }
494 in.token.text = text
495 in.token.endPos = in.pos
496 }
497
498
499 func (in *input) peek() tokenKind {
500 return in.token.kind
501 }
502
503
504 func (in *input) lex() token {
505 tok := in.token
506 in.readToken()
507 return tok
508 }
509
510
511 func (in *input) readToken() {
512
513 for !in.eof() {
514 c := in.peekRune()
515 if c == ' ' || c == '\t' || c == '\r' {
516 in.readRune()
517 continue
518 }
519
520
521 if in.peekPrefix("//") {
522 in.startToken()
523
524
525
526
527 i := bytes.LastIndex(in.complete[:in.pos.Byte], []byte("\n"))
528 suffix := len(bytes.TrimSpace(in.complete[i+1:in.pos.Byte])) > 0
529 in.readRune()
530 in.readRune()
531
532
533 for len(in.remaining) > 0 && in.readRune() != '\n' {
534 }
535
536
537
538
539 if !suffix {
540 in.endToken(_COMMENT)
541 return
542 }
543
544
545 in.endToken(_EOLCOMMENT)
546 in.comments = append(in.comments, Comment{in.token.pos, in.token.text, suffix})
547 return
548 }
549
550 if in.peekPrefix("/*") {
551 in.Error("mod files must use // comments (not /* */ comments)")
552 }
553
554
555 break
556 }
557
558
559 in.startToken()
560
561
562 if in.eof() {
563 in.endToken(_EOF)
564 return
565 }
566
567
568 switch c := in.peekRune(); c {
569 case '\n', '(', ')', '[', ']', '{', '}', ',':
570 in.readRune()
571 in.endToken(tokenKind(c))
572 return
573
574 case '"', '`':
575 quote := c
576 in.readRune()
577 for {
578 if in.eof() {
579 in.pos = in.token.pos
580 in.Error("unexpected EOF in string")
581 }
582 if in.peekRune() == '\n' {
583 in.Error("unexpected newline in string")
584 }
585 c := in.readRune()
586 if c == quote {
587 break
588 }
589 if c == '\\' && quote != '`' {
590 if in.eof() {
591 in.pos = in.token.pos
592 in.Error("unexpected EOF in string")
593 }
594 in.readRune()
595 }
596 }
597 in.endToken(_STRING)
598 return
599 }
600
601
602 if c := in.peekRune(); !isIdent(c) {
603 in.Error(fmt.Sprintf("unexpected input character %#q", rune(c)))
604 }
605
606
607 for isIdent(in.peekRune()) {
608 if in.peekPrefix("//") {
609 break
610 }
611 if in.peekPrefix("/*") {
612 in.Error("mod files must use // comments (not /* */ comments)")
613 }
614 in.readRune()
615 }
616 in.endToken(_IDENT)
617 }
618
619
620
621
622 func isIdent(c int) bool {
623 switch r := rune(c); r {
624 case ' ', '(', ')', '[', ']', '{', '}', ',':
625 return false
626 default:
627 return !unicode.IsSpace(r) && unicode.IsPrint(r)
628 }
629 }
630
631
632
633
634
635
636
637
638
639
640
641 func (in *input) order(x Expr) {
642 if x != nil {
643 in.pre = append(in.pre, x)
644 }
645 switch x := x.(type) {
646 default:
647 panic(fmt.Errorf("order: unexpected type %T", x))
648 case nil:
649
650 case *LParen, *RParen:
651
652 case *CommentBlock:
653
654 case *Line:
655
656 case *FileSyntax:
657 for _, stmt := range x.Stmt {
658 in.order(stmt)
659 }
660 case *LineBlock:
661 in.order(&x.LParen)
662 for _, l := range x.Line {
663 in.order(l)
664 }
665 in.order(&x.RParen)
666 }
667 if x != nil {
668 in.post = append(in.post, x)
669 }
670 }
671
672
673 func (in *input) assignComments() {
674 const debug = false
675
676
677 in.order(in.file)
678
679
680 var line, suffix []Comment
681 for _, com := range in.comments {
682 if com.Suffix {
683 suffix = append(suffix, com)
684 } else {
685 line = append(line, com)
686 }
687 }
688
689 if debug {
690 for _, c := range line {
691 fmt.Fprintf(os.Stderr, "LINE %q :%d:%d #%d\n", c.Token, c.Start.Line, c.Start.LineRune, c.Start.Byte)
692 }
693 }
694
695
696 for _, x := range in.pre {
697 start, _ := x.Span()
698 if debug {
699 fmt.Fprintf(os.Stderr, "pre %T :%d:%d #%d\n", x, start.Line, start.LineRune, start.Byte)
700 }
701 xcom := x.Comment()
702 for len(line) > 0 && start.Byte >= line[0].Start.Byte {
703 if debug {
704 fmt.Fprintf(os.Stderr, "ASSIGN LINE %q #%d\n", line[0].Token, line[0].Start.Byte)
705 }
706 xcom.Before = append(xcom.Before, line[0])
707 line = line[1:]
708 }
709 }
710
711
712 in.file.After = append(in.file.After, line...)
713
714 if debug {
715 for _, c := range suffix {
716 fmt.Fprintf(os.Stderr, "SUFFIX %q :%d:%d #%d\n", c.Token, c.Start.Line, c.Start.LineRune, c.Start.Byte)
717 }
718 }
719
720
721 for _, x := range slices.Backward(in.post) {
722 start, end := x.Span()
723 if debug {
724 fmt.Fprintf(os.Stderr, "post %T :%d:%d #%d :%d:%d #%d\n", x, start.Line, start.LineRune, start.Byte, end.Line, end.LineRune, end.Byte)
725 }
726
727
728
729 switch x.(type) {
730 case *FileSyntax:
731 continue
732 }
733
734
735
736
737
738
739
740
741 if start.Line != end.Line {
742 continue
743 }
744 xcom := x.Comment()
745 for len(suffix) > 0 && end.Byte <= suffix[len(suffix)-1].Start.Byte {
746 if debug {
747 fmt.Fprintf(os.Stderr, "ASSIGN SUFFIX %q #%d\n", suffix[len(suffix)-1].Token, suffix[len(suffix)-1].Start.Byte)
748 }
749 xcom.Suffix = append(xcom.Suffix, suffix[len(suffix)-1])
750 suffix = suffix[:len(suffix)-1]
751 }
752 }
753
754
755
756
757 for _, x := range in.post {
758 reverseComments(x.Comment().Suffix)
759 }
760
761
762 in.file.Before = append(in.file.Before, suffix...)
763 }
764
765
766 func reverseComments(list []Comment) {
767 for i, j := 0, len(list)-1; i < j; i, j = i+1, j-1 {
768 list[i], list[j] = list[j], list[i]
769 }
770 }
771
772 func (in *input) parseFile() {
773 in.file = new(FileSyntax)
774 var cb *CommentBlock
775 for {
776 switch in.peek() {
777 case '\n':
778 in.lex()
779 if cb != nil {
780 in.file.Stmt = append(in.file.Stmt, cb)
781 cb = nil
782 }
783 case _COMMENT:
784 tok := in.lex()
785 if cb == nil {
786 cb = &CommentBlock{Start: tok.pos}
787 }
788 com := cb.Comment()
789 com.Before = append(com.Before, Comment{Start: tok.pos, Token: tok.text})
790 case _EOF:
791 if cb != nil {
792 in.file.Stmt = append(in.file.Stmt, cb)
793 }
794 return
795 default:
796 in.parseStmt()
797 if cb != nil {
798 in.file.Stmt[len(in.file.Stmt)-1].Comment().Before = cb.Before
799 cb = nil
800 }
801 }
802 }
803 }
804
805 func (in *input) parseStmt() {
806 tok := in.lex()
807 start := tok.pos
808 end := tok.endPos
809 tokens := []string{tok.text}
810 for {
811 tok := in.lex()
812 switch {
813 case tok.kind.isEOL():
814 in.file.Stmt = append(in.file.Stmt, &Line{
815 Start: start,
816 Token: tokens,
817 End: end,
818 })
819 return
820
821 case tok.kind == '(':
822 if next := in.peek(); next.isEOL() {
823
824 in.file.Stmt = append(in.file.Stmt, in.parseLineBlock(start, tokens, tok))
825 return
826 } else if next == ')' {
827 rparen := in.lex()
828 if in.peek().isEOL() {
829
830 in.lex()
831 in.file.Stmt = append(in.file.Stmt, &LineBlock{
832 Start: start,
833 Token: tokens,
834 LParen: LParen{Pos: tok.pos},
835 RParen: RParen{Pos: rparen.pos},
836 })
837 return
838 }
839
840 tokens = append(tokens, tok.text, rparen.text)
841 } else {
842
843 tokens = append(tokens, tok.text)
844 }
845
846 default:
847 tokens = append(tokens, tok.text)
848 end = tok.endPos
849 }
850 }
851 }
852
853 func (in *input) parseLineBlock(start Position, token []string, lparen token) *LineBlock {
854 x := &LineBlock{
855 Start: start,
856 Token: token,
857 LParen: LParen{Pos: lparen.pos},
858 }
859 var comments []Comment
860 for {
861 switch in.peek() {
862 case _EOLCOMMENT:
863
864 in.lex()
865 case '\n':
866
867 in.lex()
868 if len(comments) == 0 && len(x.Line) > 0 || len(comments) > 0 && comments[len(comments)-1].Token != "" {
869 comments = append(comments, Comment{})
870 }
871 case _COMMENT:
872 tok := in.lex()
873 comments = append(comments, Comment{Start: tok.pos, Token: tok.text})
874 case _EOF:
875 in.Error(fmt.Sprintf("syntax error (unterminated block started at %s:%d:%d)", in.filename, x.Start.Line, x.Start.LineRune))
876 case ')':
877 rparen := in.lex()
878
879
880 if len(comments) == 1 && comments[0] == (Comment{}) {
881 comments = nil
882 }
883 x.RParen.Before = comments
884 x.RParen.Pos = rparen.pos
885 if !in.peek().isEOL() {
886 in.Error("syntax error (expected newline after closing paren)")
887 }
888 in.lex()
889 return x
890 default:
891 l := in.parseLine()
892 x.Line = append(x.Line, l)
893 l.Comment().Before = comments
894 comments = nil
895 }
896 }
897 }
898
899 func (in *input) parseLine() *Line {
900 tok := in.lex()
901 if tok.kind.isEOL() {
902 in.Error("internal parse error: parseLine at end of line")
903 }
904 start := tok.pos
905 end := tok.endPos
906 tokens := []string{tok.text}
907 for {
908 tok := in.lex()
909 if tok.kind.isEOL() {
910 return &Line{
911 Start: start,
912 Token: tokens,
913 End: end,
914 InBlock: true,
915 }
916 }
917 tokens = append(tokens, tok.text)
918 end = tok.endPos
919 }
920 }
921
922 var (
923 slashSlash = []byte("//")
924 moduleStr = []byte("module")
925 )
926
927
928
929
930 func ModulePath(mod []byte) string {
931 for len(mod) > 0 {
932 line := mod
933 mod = nil
934 if i := bytes.IndexByte(line, '\n'); i >= 0 {
935 line, mod = line[:i], line[i+1:]
936 }
937 if i := bytes.Index(line, slashSlash); i >= 0 {
938 line = line[:i]
939 }
940 line = bytes.TrimSpace(line)
941 if !bytes.HasPrefix(line, moduleStr) {
942 continue
943 }
944 line = line[len(moduleStr):]
945 n := len(line)
946 line = bytes.TrimSpace(line)
947 if len(line) == n || len(line) == 0 {
948 continue
949 }
950
951 if line[0] == '"' || line[0] == '`' {
952 p, err := strconv.Unquote(string(line))
953 if err != nil {
954 return ""
955 }
956 return p
957 }
958
959 return string(line)
960 }
961 return ""
962 }
963
View as plain text