commit 46cb98c0bbecb439ad744f61b53eed9c70455e33
parent c25ceb5bdd7937504c28ba68e24090b38efdbd51
Author: weetat <wtenison@protonmail.com>
Date: Sun, 2 Mar 2025 01:25:51 +0000
Support escaping commands and backslashes in fmd
Diffstat:
5 files changed, 114 insertions(+), 72 deletions(-)
diff --git a/src/fmd/fmd.ebnf b/src/fmd/fmd.ebnf
@@ -4,15 +4,20 @@
<fmd> ::= (<command> | <block> | <body>)*
-<command> ::= (<keyword>)+ ":" ( <word> | <block> )
+<command> ::= (<command_id>)+ ":" <word_or_block>
+<command_id> ::= <keyword> | <keyword> "(" <params> ")"
<keyword> ::= "b" | "i" | "h" [1-6] | "@" | "#" | "col" | "img" | "_"
+<params> ::= ("id=" | "class=" | "ref=") <word_or_block>
-<block> ::= "{" <body> "}"
+<word_or_block> ::= <word> | <block>
+<block> ::= "{" <body> "}"
<body> := (<word> | <space>)+
<word> ::= (<letter> | <number> | <symbol>)+
+
<space> ::= " "
<letter> ::= [a-z] | [A-Z]
<number> ::= [0-9]+ | (<number> "." <number>)
+
<symbol> ::= "." | "," | ";" | "-" | "!" | "#" | "$" | "%" | "^" | "&" | "*" | "(" | ")"
\ No newline at end of file
diff --git a/src/fmd/fmd_test.go b/src/fmd/fmd_test.go
@@ -1,29 +1,31 @@
package fmd
import (
+ "os"
"testing"
)
func TestFMD(t *testing.T) {
- _path := ("/home/weetat/projects/miki/data/README.fmd")
+ _path := ("/home/weetat/projects/miki/data/fmd/home.fmd")
t.Log("Path:", _path)
- WriteFMDToHTMLFile(_path, "/home/weetat/projects/miki/README.md")
+ // WriteFMDToHTMLFile(_path, "/home/weetat/projects/miki/README.md")
- // _input, _err := os.ReadFile(_path)
- // if _err != nil {
- // t.Error("_ERROR\tcannot open file:", _err.Error())
- // return
- // }
+ _input, _err := os.ReadFile(_path)
+ if _err != nil {
+ t.Error("_ERROR\tcannot open file:", _err.Error())
+ return
+ }
- // _l := MakeLexer(_input)
- // _tokens := _l.Exec()
+ _l := MakeLexer(_input)
+ _tokens := _l.Exec()
- // t.Log("Tokens:", _tokens)
+ for _, _t := range _tokens {
+ t.Log("Tokens:", _t)
+ }
- // _p := MakeParser(_tokens)
- // _ast := _p.Exec()
- // t.Log("AST:", _ast)
+ _p := MakeParser(_tokens)
+ _ast := _p.Exec()
- // t.Log("HTML:\n", interpret(_ast))
+ t.Log("HTML:\n", interpret(_ast))
}
diff --git a/src/fmd/interpreter.go b/src/fmd/interpreter.go
@@ -62,7 +62,7 @@ func interpret(n node) string {
case nSpacing:
return n.value
case nNewline:
- if n.pNode.nType != nNewline {
+ if n.nParent.nType != nNewline {
return "<br>\n"
} else {
return "\n"
diff --git a/src/fmd/lexer.go b/src/fmd/lexer.go
@@ -1,6 +1,7 @@
package fmd
import (
+ "log"
"slices"
)
@@ -28,17 +29,18 @@ const (
tLParen // 9. (
tRParen // 10. )
- tError // 11
- tEmpty // 12
+ // UTILS
+ tError // 11. error
+ tEmpty // 12. nil
)
type token struct {
tType tokenType
- Value string
+ value string
}
var operatorSymbols = []string{"\\", ":", "{", "}", "(", ")"}
-var spacingSymbols = []string{" ", "\n", "\t", ","}
+var spacingSymbols = []string{" ", "\n", "\t"}
var keywords = []string{
"b", // bold
"i", // italic
@@ -51,9 +53,9 @@ var keywords = []string{
}
type lexer struct {
- input []byte // text input to scan
- pos int // the start position for scanning
- tokens []token
+ input []byte // text input to scan
+ pos int // the start position for scanning
+ tokens []token // the output token stream
}
func MakeLexer(input []byte) *lexer {
@@ -61,6 +63,9 @@ func MakeLexer(input []byte) *lexer {
}
func (l *lexer) Exec() []token {
+ var _symbols []string
+ _symbols = append(_symbols, spacingSymbols...)
+ _symbols = append(_symbols, operatorSymbols...)
for l.pos < len(l.input) {
_currentSymbol := l.input[l.pos]
@@ -76,7 +81,11 @@ func (l *lexer) Exec() []token {
l.tokens = append(l.tokens, token{tTab, "\t"})
l.pos++
case _currentSymbol == '\\':
- l.tokens = append(l.tokens, token{tEscape, "\\"})
+ if l.peekTokenType(-1) == tEscape {
+ l.tokens = append(l.tokens, token{tWord, "\\"})
+ } else {
+ l.tokens = append(l.tokens, token{tEscape, "\\"})
+ }
l.pos++
case _currentSymbol == ':':
l.tokens = append(l.tokens, token{tColon, ":"})
@@ -93,17 +102,10 @@ func (l *lexer) Exec() []token {
case _currentSymbol == ')':
l.tokens = append(l.tokens, token{tRParen, ")"})
l.pos++
- case _currentSymbol == ',':
- l.tokens = append(l.tokens, token{tWord, ","})
- l.pos++
default:
{ // NOTE: inlined check until good reason to abstract arrives.
_start := l.pos
- var _symbols []string
- _symbols = append(_symbols, spacingSymbols...)
- _symbols = append(_symbols, operatorSymbols...)
-
for l.pos < len(l.input) && !slices.Contains(_symbols, string(l.input[l.pos])) {
l.pos += 1
}
@@ -111,7 +113,7 @@ func (l *lexer) Exec() []token {
_tokenStr := string(l.input[_start:l.pos])
if slices.Contains(keywords, _tokenStr) {
- if _start == 0 || l.tokens[len(l.tokens)-1].tType != tEscape {
+ if _start == 0 || l.peekTokenType(-1) != tEscape {
l.tokens = append(l.tokens, token{tKeyword, _tokenStr})
} else {
l.tokens = append(l.tokens, token{tWord, _tokenStr})
@@ -126,3 +128,15 @@ func (l *lexer) Exec() []token {
return l.tokens
}
+
+func (l *lexer) peekTokenType(posOffset int) tokenType {
+
+ _index := len(l.tokens) + posOffset
+
+ if posOffset < 0 && _index > -1 {
+ return l.tokens[_index].tType
+ }
+
+ log.Printf("_ERROR:peekTokenType %t, %t at %d", l.pos+posOffset > -1, l.pos+posOffset < len(l.tokens), l.pos+posOffset)
+ return tError
+}
diff --git a/src/fmd/parser.go b/src/fmd/parser.go
@@ -10,9 +10,10 @@ type nodeType string
const (
nRoot nodeType = "root"
- nCommand nodeType = "command"
- nBlock nodeType = "block"
- nBody nodeType = "body"
+ nCommand nodeType = "command"
+ nParameter nodeType = "parameter"
+ nBlock nodeType = "block"
+ nBody nodeType = "body"
nWord nodeType = "word"
nOperator nodeType = "operator"
@@ -20,34 +21,37 @@ const (
nNewline nodeType = "newline"
nError nodeType = "error"
+ nEmpty nodeType = "nil"
)
type node struct {
nType nodeType
value string
- pNode *node
+ nParent *node
children []node
}
type parser struct {
- tokens []token
- pos int
- depth int
- pNode *node
- errorNodes []node
+ tokens []token // input tokens
+ nErrors []node // output error nodes
+
+ pos int // current 'head' position
+ depth int // nest depth
+ nPrev *node // previous node reference
}
func MakeParser(tokens []token) *parser {
if tokens == nil {
log.Printf("_DEBUG:MakeParser:token array is empty")
- _tokens := []token{{tType: tError, Value: "_ERROR:MakeParser:empty token string"}}
+ _tokens := []token{{tType: tError, value: "_ERROR:MakeParser:empty token string"}}
return &parser{tokens: _tokens}
}
- return &parser{tokens: tokens, pos: 0}
+
+ return &parser{tokens: tokens}
}
func (p *parser) Exec() node {
- _fmd := p.makeNode(nRoot, token{tType: tEmpty, Value: "_fmd_"})
+ _fmd := p.makeNode(nRoot, token{tType: tEmpty, value: "_fmd-root_"})
for p.pos < len(p.tokens) {
if p.isNextCommand() {
@@ -64,27 +68,42 @@ func (p *parser) Exec() node {
func (p *parser) parseCommand() node {
- // TODO: Added function for parseing multiple keywords; parseKeyword() (e.g. bi@:... )
_command := p.makeNode(nCommand, p.consume(tKeyword))
p.consume(tColon)
if p.isNextCommand() {
- _command.children = []node{p.parseCommand()}
+ _command.children = append(_command.children, p.parseCommand())
+ } else if p.isTokenType(tLParen, 0) {
+ _command.children = append(_command.children, p.parseParam())
} else if p.isTokenType(tLBrace, 0) {
- _command.children = []node{p.parseBlock()}
+ _command.children = append(_command.nParent.children, p.parseBlock())
} else {
- _command.children = []node{p.makeNode(nWord, p.consume(tWord))}
+ _command.children = append(_command.children, p.makeNode(nWord, p.consume(tWord)))
}
return _command
}
+func (p *parser) parseParam() node {
+ var _n node
+
+ p.consume(tLParen)
+
+ for p.pos < len(p.tokens) && p.peekToken(0).tType != tRParen {
+
+ }
+
+ p.consume(tRParen)
+
+ return _n
+}
+
func (p *parser) parseBlock() node {
p.depth += 1
- _block := p.makeNode(nBlock, token{tType: tEmpty, Value: ""})
+ _block := p.makeNode(nBlock, token{tType: tEmpty, value: ""})
p.consume(tLBrace)
@@ -106,7 +125,7 @@ func (p *parser) parseBlock() node {
}
func (p *parser) parseBody() node {
- _body := p.makeNode(nBody, token{tType: tEmpty, Value: ""})
+ _body := p.makeNode(nBody, token{tType: tEmpty, value: ""})
for p.pos < len(p.tokens) {
@@ -141,7 +160,7 @@ func (p *parser) parseBody() node {
if p.isNextCommand() {
_body.children = append(_body.children, p.parseCommand())
} else {
- _body.children = append(_body.children, p.makeNode(nError, token{tType: tError, Value: "unknown token;"}))
+ _body.children = append(_body.children, p.makeNode(nError, token{tType: tError, value: "unknown token;"}))
p.pos += 1
log.Printf("INFO:block:error:%v", _body.children[len(_body.children)-1])
}
@@ -154,24 +173,24 @@ func (p *parser) parseBody() node {
//// UTILITY FUNCTIONS
func (p *parser) makeNode(nType nodeType, iToken token) node {
- _n := node{nType: nType, value: iToken.Value}
+ _n := node{nType: nType, value: iToken.value}
if iToken.tType == tError {
- _n = p.errorNode(iToken.Value)
+ _n = p.errorNode(iToken.value)
}
- _n.pNode = p.pNode
- p.pNode = &_n
+ _n.nParent = p.nPrev
+ p.nPrev = &_n
return _n
}
func (p *parser) errorNode(message string) node {
- _errorValue := fmt.Sprintf("_ERROR:on %s at pos %d:%s", p.tokens[p.pos].Value, p.pos, message)
- p.errorNodes = append(p.errorNodes, node{nType: nError, value: _errorValue})
+ _errorValue := fmt.Sprintf("_ERROR:errorNode:on %s at pos %d:%s", p.tokens[p.pos].value, p.pos, message)
+ p.nErrors = append(p.nErrors, node{nType: nError, value: _errorValue})
- return p.errorNodes[len(p.errorNodes)-1]
+ return p.nErrors[len(p.nErrors)-1]
}
func (p *parser) consume(expectedTType tokenType) token {
@@ -179,35 +198,37 @@ func (p *parser) consume(expectedTType tokenType) token {
_token := p.tokens[p.pos]
if !p.isTokenType(expectedTType, 0) {
- _errorValue := fmt.Sprintf("_ERROR:on %s at pos %d:%s", p.tokens[p.pos].Value, p.pos, "invalid token type")
+ _errorValue := fmt.Sprintf("_ERROR:consume:on %s at pos %d:expected type %d got %d", p.tokens[p.pos].value, p.pos, expectedTType, _token.tType)
_token = token{tError, _errorValue}
}
p.pos += 1
- // log.Printf("INFO:consume:%v", _token.Value)
+ log.Printf("INFO:consume %d,%s", _token.tType, _token.value)
return _token
}
-func (p *parser) peekToken(numOfTokensAhead int) token {
+func (p *parser) peekToken(posOffset int) token {
- if p.pos+numOfTokensAhead < len(p.tokens) {
+ if p.pos+posOffset < len(p.tokens) {
return p.tokens[p.pos]
}
return token{tError, "_ERROR:peekToken:attempting to peek beyond array size"}
}
-func (p *parser) isTokenType(expectedTType tokenType, indexOffset int) bool {
- return p.pos+indexOffset > -1 &&
- p.pos+indexOffset < len(p.tokens) &&
- p.tokens[p.pos+indexOffset].tType == expectedTType
+func (p *parser) isTokenType(expectedTType tokenType, posOffset int) bool {
+ return p.pos+posOffset > -1 &&
+ p.pos+posOffset < len(p.tokens) &&
+ p.tokens[p.pos+posOffset].tType == expectedTType
}
func (p *parser) isNextCommand() bool {
- return !p.isTokenType(tEscape, -1) &&
- p.isTokenType(tKeyword, 0) &&
- p.isTokenType(tColon, 1) &&
- p.isTokenType(tWord, 2) ||
- p.isTokenType(tLBrace, 2)
+ if p.isTokenType(tEscape, 0) {
+ p.consume(tEscape)
+ return false
+ }
+
+ return p.isTokenType(tKeyword, 0) && p.isTokenType(tColon, 1)
+
}