miki

Miki: personal wiki with the fmd markup format
git clone https://wtenison.com/repo/miki.git
Log | Files | Refs | README | LICENSE

commit c7dd8655b9e6a17c63ba4f49867d47ff6c034129
parent d7a479f32ada12f0aa1a61bb6e4e94b3dc6d59de
Author: weetat <wtenison@protonmail.com>
Date:   Mon,  3 Mar 2025 17:59:13 +0000

Moved 'fmd' to own project/repo.

Diffstat:
Dsrc/fmd/fmd.ebnf | 24------------------------
Dsrc/fmd/fmd.go | 65-----------------------------------------------------------------
Dsrc/fmd/fmd_test.go | 31-------------------------------
Dsrc/fmd/interpreter.go | 84-------------------------------------------------------------------------------
Dsrc/fmd/lexer.go | 146-------------------------------------------------------------------------------
Dsrc/fmd/parser.go | 235-------------------------------------------------------------------------------
6 files changed, 0 insertions(+), 585 deletions(-)

diff --git a/src/fmd/fmd.ebnf b/src/fmd/fmd.ebnf @@ -1,23 +0,0 @@ -# This grammar is a work-in-progress and represents the ideal form -# of the lanuage. It may not be fully implemented and is subject -# to change as I learn. - -<fmd> ::= (<command> | <block> | <body>)* - -<command> ::= (<command_id>)+ ":" <word_or_block> -<command_id> ::= <keyword> | <keyword> "(" <params> ")" -<keyword> ::= "b" | "i" | "h" [1-6] | "@" | "#" | "col" | "img" | "_" -<params> ::= ("id=" | "class=" | "ref=") <word_or_block> - -<word_or_block> ::= <word> | <block> - -<block> ::= "{" <body> "}" -<body> := (<word> | <space>)+ - -<word> ::= (<letter> | <number> | <symbol>)+ - -<space> ::= " " -<letter> ::= [a-z] | [A-Z] -<number> ::= [0-9]+ | (<number> "." <number>) - -<symbol> ::= "." | "," | ";" | "-" | "!" | "#" | "$" | "%" | "^" | "&" | "*" | "(" | ")" -\ No newline at end of file diff --git a/src/fmd/fmd.go b/src/fmd/fmd.go @@ -1,65 +0,0 @@ -package fmd - -import ( - "bytes" - "log" - "os" - "path/filepath" -) - -func ReadFMDStreamtoHTMLString(inputStreamFMD []byte) string { - - _l := MakeLexer(inputStreamFMD) - - _tokens := _l.Exec() - - _p := MakeParser(_tokens) - - _ast := _p.Exec() - - return interpret(_ast) -} - -func WriteFMDToHTMLFile(inputFilePath string, outputFilePath string) { - _fileName := filepath.Base(inputFilePath) - if filepath.Ext(_fileName) != ".fmd" { - log.Printf("WriteFMDToHTMLFile:filepath.Ext:%s:invalid file extension", _fileName) - return - } - - _fileBytes, _err := os.ReadFile(inputFilePath) - if _err != nil { - log.Println("_ERROR\tfmd:WriteFMDToHTMLFile:ReadFile:", inputFilePath, ":", _err) - } - - _l := MakeLexer(_fileBytes) - - _tokens := _l.Exec() - - _p := MakeParser(_tokens) - - _ast := _p.Exec() - - _outputBytes := bytes.NewBufferString(interpret(_ast)) - - if _err = os.WriteFile(outputFilePath, _outputBytes.Bytes(), 0666); _err != nil { - log.Println("_ERROR\tfmd:WriteFMDToHTMLFile:os.WriteFile:", inputFilePath, ":", _err) - } -} - -func WritFMDToHTMLString(inputFilePath string) string { - _fileBytes, _err := os.ReadFile(inputFilePath) - if _err != nil { - log.Println("_ERROR\tfmd:WriteFMDToHTMLString:ReadFile:", inputFilePath, ":", _err) - } - - _l := MakeLexer(_fileBytes) - - _tokens := _l.Exec() - - _p := MakeParser(_tokens) - - _ast := _p.Exec() - - return interpret(_ast) -} diff --git a/src/fmd/fmd_test.go b/src/fmd/fmd_test.go @@ -1,31 +0,0 @@ -package fmd - -import ( - "os" - "testing" -) - -func TestFMD(t *testing.T) { - _path := ("/home/weetat/projects/miki/data/README.fmd") - t.Log("Path:", _path) - - // WriteFMDToHTMLFile(_path, "/home/weetat/projects/miki/README.md") - - _input, _err := os.ReadFile(_path) - if _err != nil { - t.Error("_ERROR\tcannot open file:", _err.Error()) - return - } - - _l := MakeLexer(_input) - _tokens := _l.Exec() - - // for _, _t := range _tokens { - // t.Log("Tokens:", _t) - // } - - _p := MakeParser(_tokens) - _ast := _p.Exec() - - t.Log("HTML:\n", interpret(_ast)) -} diff --git a/src/fmd/interpreter.go b/src/fmd/interpreter.go @@ -1,84 +0,0 @@ -package fmd - -import ( - "slices" - "strings" -) - -// TODO: Implment better solution to sanitization. -// TODO: ALternative; move interpreter to client side via Web Assembly. -// NOTE: Solution; move sanitization to separate package and let the interpreter work fast, then clean? -var invalidHTML = []string{"<script>", "</script>", "<object>", "</object>", "<embed", "<link>", "</link>"} - -func interpret(n node) string { - var _res strings.Builder - - switch n.nType { - case nRoot: - for _, _child := range n.children { - _res.WriteString(interpret(_child)) - } - case nCommand: - switch n.value { - case "b": - return "<b>" + interpret(n.children[0]) + "</b>" - case "i": - return "<i>" + interpret(n.children[0]) + "</i>" - case "h1": - return "<h1>" + interpret(n.children[0]) + "</h1>" - case "h2": - return "<h2>" + interpret(n.children[0]) + "</h2>" - case "h3": - return "<h3>" + interpret(n.children[0]) + "</h3>" - case "h4": - return "<h4>" + interpret(n.children[0]) + "</h4>" - case "h5": - return "<h5>" + interpret(n.children[0]) + "</h5>" - case "h6": - return "<h6>" + interpret(n.children[0]) + "</h6>" - case "code": - return "<code>" + interpret(n.children[0]) + "</code>" - case "@": - return "<a>" + interpret(n.children[0]) + "</a>" - case "img": - { - _child := interpret(n.children[0]) - _src := strings.Split(_child, "/") - _alt := _src[len(_src)-1] - return "<img src=\"" + _child + "\" alt=\"" + _alt + "\">" + "</img>" - } - case "col": - return "<div class=\"__mcol\">" + interpret(n.children[0]) + "</div>" - default: - return "<div>" + interpret(n.children[0]) + "</div>" - } - case nBlock: - _res.WriteString("<div class=\"__b\">") - for _, _child := range n.children { - _res.WriteString(interpret(_child)) - } - _res.WriteString("</div>") - case nBody: - for _, _child := range n.children { - _res.WriteString(interpret(_child)) - } - case nSpacing: - return n.value - case nNewline: - if n.nPrev.nType != nNewline { - return "<br>\n" - } else { - return "\n" - } - case nWord: - if slices.Contains(invalidHTML, n.value) { - return "" - } - - return n.value - default: - return n.value - } - - return _res.String() -} diff --git a/src/fmd/lexer.go b/src/fmd/lexer.go @@ -1,146 +0,0 @@ -package fmd - -import ( - "slices" -) - -/* Reference source: Lexical Scanning in Go - Rob Pike, -https://www.youtube.com/watch?v=HxaD_trXwRE */ - -type tokenType int - -const ( - - // WORDS - tWord tokenType = iota // 0. (letters, numbers, symbols)+ - tKeyword // 1. (valid keywords) - - // SPACING - tSpace // 2. " " - tNewline // 3. \n - tTab // 4. \t - - // OPERATORS - tEscape // 5. \ - tColon // 6. : - tLBrace // 7. { - tRBrace // 8. } - tLParen // 9. ( - tRParen // 10. ) - - // UTILS - tError // 11. error - tEmpty // 12. nil -) - -type token struct { - tType tokenType - value string -} - -var operatorSymbols = []string{"\\", ":", "{", "}", "(", ")"} -var spacingSymbols = []string{" ", "\n", "\t"} -var keywords = []string{ - "b", // bold - "i", // italic - "h1", "h2", "h3", "h4", "h5", "h6", // header - "code", // code element - "@", // link - "#", // tag - "col", // column container - "img", // image - "_", // empty container (<div>) -} - -type lexer struct { - input []byte // text input to scan - pos int // the start position for scanning - tokens []token // the output token stream -} - -func MakeLexer(input []byte) *lexer { - return &lexer{input: input, pos: 0, tokens: []token{}} -} - -func (l *lexer) Exec() []token { - var _symbols []string - _symbols = append(_symbols, spacingSymbols...) - _symbols = append(_symbols, operatorSymbols...) - - for l.pos < len(l.input) { - _currentSymbol := l.input[l.pos] - - switch { - case _currentSymbol == ' ': - l.tokens = append(l.tokens, token{tSpace, " "}) - l.pos++ - case _currentSymbol == '\n': - l.tokens = append(l.tokens, token{tNewline, "\n"}) - l.pos++ - case _currentSymbol == '\t': - l.tokens = append(l.tokens, token{tTab, "\t"}) - l.pos++ - case _currentSymbol == '\\': - if l.pos != 0 && l.peekTokenType(-1) == tEscape { - l.tokens = append(l.tokens, token{tWord, "\\"}) - } else { - l.tokens = append(l.tokens, token{tEscape, "\\"}) - } - l.pos++ - case _currentSymbol == ':': - l.tokens = append(l.tokens, token{tColon, ":"}) - l.pos++ - case _currentSymbol == '{': - l.tokens = append(l.tokens, token{tLBrace, "{"}) - l.pos++ - case _currentSymbol == '}': - l.tokens = append(l.tokens, token{tRBrace, "}"}) - l.pos++ - case _currentSymbol == '(': - l.tokens = append(l.tokens, token{tLParen, "("}) - l.pos++ - case _currentSymbol == ')': - l.tokens = append(l.tokens, token{tRParen, ")"}) - l.pos++ - default: - { // NOTE: inlined check until good reason to abstract arrives. - _start := l.pos - - for l.pos < len(l.input) && !slices.Contains(_symbols, string(l.input[l.pos])) { - l.pos += 1 - } - - _tokenStr := string(l.input[_start:l.pos]) - - if slices.Contains(keywords, _tokenStr) && l.peekInput(1) == ':' { - if _start == 0 || l.peekTokenType(-1) != tEscape { - l.tokens = append(l.tokens, token{tKeyword, _tokenStr}) - } else { - l.tokens = append(l.tokens, token{tWord, _tokenStr}) - l.tokens = append(l.tokens, token{tWord, ":"}) - l.pos += 1 - } - } else if _tokenStr != "" { - l.tokens = append(l.tokens, token{tWord, _tokenStr}) - } - } - } - } - - return l.tokens -} - -func (l *lexer) peekInput(posOffset int) byte { - if l.pos+posOffset > -1 && l.pos+posOffset < len(l.input) { - return l.input[l.pos] - } - panic("__PANIC:peekInput:check out of bounds") -} - -func (l *lexer) peekTokenType(posOffset int) tokenType { - _index := len(l.tokens) + posOffset - if posOffset < 0 && _index > -1 { - return l.tokens[_index].tType - } - panic("__PANIC:peekTokenType:check out of bounds") -} diff --git a/src/fmd/parser.go b/src/fmd/parser.go @@ -1,235 +0,0 @@ -package fmd - -import ( - "fmt" - "log" -) - -type nodeType string - -const ( - nRoot nodeType = "root" - - nCommand nodeType = "command" - nParameter nodeType = "parameter" - nBlock nodeType = "block" - nBody nodeType = "body" - - nWord nodeType = "word" - nOperator nodeType = "operator" - nSpacing nodeType = "spacing" - nNewline nodeType = "newline" - - nError nodeType = "error" - nEmpty nodeType = "nil" -) - -type node struct { - nType nodeType - value string - nPrev *node - children []node -} - -type parser struct { - tokens []token // input tokens - nErrors []node // output error nodes - - pos int // current 'head' position - depth int // nest depth - nPrev *node // previous node reference -} - -func MakeParser(tokens []token) *parser { - if tokens == nil { - log.Printf("_DEBUG:MakeParser:token array is empty") - _tokens := []token{{tType: tError, value: "_ERROR:MakeParser:empty token string"}} - return &parser{tokens: _tokens} - } - - return &parser{tokens: tokens} -} - -func (p *parser) Exec() node { - _fmd := p.makeNode(nRoot, token{tType: tEmpty, value: "_fmd-root_"}) - - for p.pos < len(p.tokens) { - if p.isNextCommand() { - _fmd.children = append(_fmd.children, p.parseCommand()) - } else if p.peekToken(0).tType == tLBrace { - _fmd.children = append(_fmd.children, p.parseBlock()) - } else { - _fmd.children = append(_fmd.children, p.parseBody()) - } - } - - return _fmd -} - -func (p *parser) parseCommand() node { - - _command := p.makeNode(nCommand, p.consume(tKeyword)) - - p.consume(tColon) - - if p.isNextCommand() { - _command.children = append(_command.children, p.parseCommand()) - } else if p.isTokenType(tLParen, 0) { - _command.children = append(_command.children, p.parseParam()) - } else if p.isTokenType(tLBrace, 0) { - _command.children = append(_command.nPrev.children, p.parseBlock()) - } else { - _command.children = append(_command.children, p.makeNode(nWord, p.consume(tWord))) - } - - return _command -} - -func (p *parser) parseParam() node { - var _n node - - p.consume(tLParen) - - // TODO: Implement parameter parsing. - for p.pos < len(p.tokens) && p.peekToken(0).tType != tRParen { - p.pos += 1 - } - - p.consume(tRParen) - - return _n -} - -func (p *parser) parseBlock() node { - - p.depth += 1 - - _block := p.makeNode(nBlock, token{tType: tEmpty, value: ""}) - - p.consume(tLBrace) - - for p.pos < len(p.tokens) && p.peekToken(0).tType != tRBrace { - - switch { - case p.isNextCommand(): - _block.children = append(_block.children, p.parseCommand()) - default: - _block.children = append(_block.children, p.parseBody()) - } - } - - p.depth -= 1 - - p.consume(tRBrace) - - return _block -} - -func (p *parser) parseBody() node { - _body := p.makeNode(nBody, token{tType: tEmpty, value: ""}) - - for p.pos < len(p.tokens) { - - if p.peekToken(0).tType == tRBrace && p.depth > 0 { - break - } - - _pToken := p.peekToken(0) - - switch _pToken.tType { - case tWord: - _body.children = append(_body.children, p.makeNode(nWord, p.consume(tWord))) - case tSpace: - _body.children = append(_body.children, p.makeNode(nSpacing, p.consume(tSpace))) - case tTab: - _body.children = append(_body.children, p.makeNode(nSpacing, p.consume(tTab))) - case tNewline: - _body.children = append(_body.children, p.makeNode(nNewline, p.consume(tNewline))) - case tLParen: - _body.children = append(_body.children, p.makeNode(nWord, p.consume(tLParen))) - case tRParen: - _body.children = append(_body.children, p.makeNode(nWord, p.consume(tRParen))) - case tEscape: - p.consume(tEscape) - case tColon: - p.consume(tColon) - case tLBrace: - _body.children = append(_body.children, p.parseBlock()) - case tRBrace: - _body.children = append(_body.children, p.errorNode("no matching LBrace")) - default: - if p.isNextCommand() { - _body.children = append(_body.children, p.parseCommand()) - } else { - _body.children = append(_body.children, p.makeNode(nError, token{tType: tError, value: "unknown token;"})) - p.pos += 1 - log.Printf("INFO:block:error:%v", _body.children[len(_body.children)-1]) - } - } - } - - return _body -} - -//// UTILITY FUNCTIONS - -func (p *parser) makeNode(nType nodeType, iToken token) node { - _n := node{nType: nType, value: iToken.value} - - if iToken.tType == tError { - _n = p.errorNode(iToken.value) - } - - _n.nPrev = p.nPrev - p.nPrev = &_n - - return _n -} - -func (p *parser) errorNode(message string) node { - - _errorValue := fmt.Sprintf("_ERROR:errorNode:on %s at pos %d:%s", p.tokens[p.pos].value, p.pos, message) - p.nErrors = append(p.nErrors, node{nType: nError, value: _errorValue}) - - return p.nErrors[len(p.nErrors)-1] -} - -func (p *parser) consume(expectedTType tokenType) token { - - _token := p.tokens[p.pos] - - if !p.isTokenType(expectedTType, 0) { - _errorValue := fmt.Sprintf("_ERROR:consume:on %s at pos %d:expected type %d got %d", p.tokens[p.pos].value, p.pos, expectedTType, _token.tType) - _token = token{tError, _errorValue} - } - - p.pos += 1 - - log.Printf("INFO:consume %d,%s", _token.tType, _token.value) - return _token -} - -func (p *parser) peekToken(posOffset int) token { - - if p.pos+posOffset < len(p.tokens) { - return p.tokens[p.pos] - } - - return token{tError, "_ERROR:peekToken:attempting to peek beyond array size"} -} - -func (p *parser) isTokenType(expectedTType tokenType, posOffset int) bool { - return p.pos+posOffset > -1 && - p.pos+posOffset < len(p.tokens) && - p.tokens[p.pos+posOffset].tType == expectedTType -} - -func (p *parser) isNextCommand() bool { - if p.isTokenType(tEscape, 0) { - p.consume(tEscape) - return false - } - - return p.isTokenType(tKeyword, 0) && p.isTokenType(tColon, 1) - -}