1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
|
// markdown.go — note text is markdown. Agents write prose about code:
// backticked identifiers, bullet lists, the occasional fenced snippet, and
// links out. Rendering it as one escaped <p> threw all of that away.
//
// Two things make markdown fit text nobody wrote for a browser:
//
// - angle brackets are never markup. In a note they are code —
// RwLock<SysvarCache> is a type — so the HTML parsers come out and the
// brackets survive as the characters they were typed as.
// - fenced code goes through the same tree-sitter highlighter as the code
// view, so a snippet in a note is painted by the reader's theme exactly
// like the file it was copied from.
package main
import (
"bytes"
stdhtml "html"
"html/template"
"net/url"
"path"
"regexp"
"strconv"
"strings"
"github.com/yuin/goldmark"
"github.com/yuin/goldmark/ast"
"github.com/yuin/goldmark/extension"
mdparser "github.com/yuin/goldmark/parser"
"github.com/yuin/goldmark/renderer"
"github.com/yuin/goldmark/renderer/html"
mdtext "github.com/yuin/goldmark/text"
"github.com/yuin/goldmark/util"
)
// markdown renders note text. goldmark is safe for concurrent use, which the
// server (a goroutine per request) and the export (a goroutine per file) both
// rely on.
type markdown struct{ md goldmark.Markdown }
// newMarkdown builds the note renderer. GFM adds the things people actually
// type — tables, ~~strikethrough~~, bare URLs, task lists — and hard wraps
// keep a newline a newline, since a note is typed into a textarea or passed on
// a command line, not authored as a document.
func newMarkdown(hl *highlighter) *markdown {
md := goldmark.New(
goldmark.WithParser(notesParser()),
goldmark.WithExtensions(extension.GFM),
goldmark.WithRendererOptions(html.WithHardWraps()),
)
// lower priority number wins: this one replaces the stock code-block
// renderers, everything else stays goldmark's
md.Renderer().AddOptions(renderer.WithNodeRenderers(
util.Prioritized(¬eRenderer{hl: hl}, 100)))
return &markdown{md: md}
}
// goldmark's priorities for the two parsers that read angle brackets as
// markup, from mdparser.DefaultBlockParsers and mdparser.DefaultInlineParsers.
const (
htmlBlockPriority = 900
rawHTMLPriority = 400
)
// notesParser is goldmark's default parser with those two dropped. Leaving
// them in costs the note either way: goldmark's safe default deletes raw HTML
// outright, so "RwLock<SysvarCache>" loses a word, and the block form
// additionally stops parsing markdown for the rest of the paragraph — a note
// that opens with a quoted <div> would render its links and lists as source.
// Without the parsers a bracket is just a character the text renderer escapes,
// which is both what the writer meant and the one thing a browser cannot act
// on. Dropping by priority keeps every other default, including any upstream
// adds later.
func notesParser() mdparser.Parser {
return mdparser.NewParser(
mdparser.WithBlockParsers(without(mdparser.DefaultBlockParsers(), htmlBlockPriority)...),
mdparser.WithInlineParsers(without(mdparser.DefaultInlineParsers(), rawHTMLPriority)...),
mdparser.WithParagraphTransformers(mdparser.DefaultParagraphTransformers()...),
)
}
func without(vs []util.PrioritizedValue, priority int) []util.PrioritizedValue {
out := vs[:0]
for _, v := range vs {
if v.Priority != priority {
out = append(out, v)
}
}
return out
}
// newNoteLinkParser is shared across one model build. It uses the same parser
// and GFM extensions as rendered note bodies, so only links the reader can
// actually click become mentions (code spans and plain prose do not).
func newNoteLinkParser() mdparser.Parser {
return goldmark.New(
goldmark.WithParser(notesParser()),
goldmark.WithExtensions(extension.GFM),
).Parser()
}
func noteLinks(p mdparser.Parser, source string) []int {
if !strings.Contains(source, "note") {
return nil
}
src := []byte(source)
doc := p.Parse(mdtext.NewReader(src))
seen := map[int]bool{}
var ids []int
_ = ast.Walk(doc, func(n ast.Node, entering bool) (ast.WalkStatus, error) {
if !entering || n.Kind() != ast.KindLink {
return ast.WalkContinue, nil
}
id := noteLinkID(string(n.(*ast.Link).Destination))
if id > 0 && !seen[id] {
seen[id] = true
ids = append(ids, id)
}
return ast.WalkContinue, nil
})
return ids
}
// noteLinkID recognizes local links ending in note/N. Leading /, ./ and ../
// components are all allowed; schemes and protocol-relative hosts are not.
func noteLinkID(destination string) int {
u, err := url.Parse(strings.TrimSpace(destination))
if err != nil || u.Scheme != "" || u.Host != "" {
return 0
}
clean := path.Clean("/" + strings.TrimSuffix(u.Path, ".html"))
parts := strings.Split(strings.Trim(clean, "/"), "/")
if len(parts) < 2 || parts[len(parts)-2] != "note" {
return 0
}
id, err := strconv.Atoi(parts[len(parts)-1])
if err != nil || id < 1 {
return 0
}
return id
}
// render turns one note's text into HTML. A note is never worth losing to a
// renderer error, so a failure falls back to the text as written.
func (m *markdown) render(text string) template.HTML {
if strings.TrimSpace(text) == "" {
return ""
}
var b bytes.Buffer
b.Grow(len(text) + len(text)/2)
if err := m.md.Convert([]byte(text), &b); err != nil {
return template.HTML("<p>" + template.HTMLEscapeString(text) + "</p>")
}
return template.HTML(b.String())
}
var noteHrefRE = regexp.MustCompile(`href="([^"]+)"`)
// renderAt canonicalizes note links for the page carrying the card. A note is
// shown at several URL depths (/notes, /code/x, /note/N, and static variants),
// so leaving its relative Markdown href untouched would make the mention work
// in only one of those places.
func (m *markdown) renderAt(text string, p *page) template.HTML {
body := m.render(text)
if body == "" || p == nil {
return body
}
out := noteHrefRE.ReplaceAllStringFunc(string(body), func(attr string) string {
destination := stdhtml.UnescapeString(attr[len(`href="`) : len(attr)-1])
id := noteLinkID(destination)
if id == 0 {
return attr
}
if !p.Live && p.Site != nil && p.Site.NoteByID(id) == nil {
return attr
}
href := p.Href("note", strconv.Itoa(id))
if href == "" {
return attr
}
return `href="` + template.HTMLEscapeString(href) + `"`
})
return template.HTML(out)
}
// noteRenderer overrides the two node kinds the stock renderer emits without
// syntax colors. Everything else it renders is fine as goldmark writes it.
type noteRenderer struct{ hl *highlighter }
func (nr *noteRenderer) RegisterFuncs(reg renderer.NodeRendererFuncRegisterer) {
reg.Register(ast.KindFencedCodeBlock, nr.renderFenced)
reg.Register(ast.KindCodeBlock, nr.renderIndented)
}
func (nr *noteRenderer) renderFenced(w util.BufWriter, src []byte, node ast.Node, entering bool) (ast.WalkStatus, error) {
if !entering {
return ast.WalkSkipChildren, nil
}
n := node.(*ast.FencedCodeBlock)
nr.code(w, blockLines(src, n), string(n.Language(src)))
return ast.WalkSkipChildren, nil
}
func (nr *noteRenderer) renderIndented(w util.BufWriter, src []byte, node ast.Node, entering bool) (ast.WalkStatus, error) {
if !entering {
return ast.WalkSkipChildren, nil
}
nr.code(w, blockLines(src, node), "")
return ast.WalkSkipChildren, nil
}
// blockLines joins a block node's source lines. They are contiguous in the
// source, but only the segments are authoritative about where the block's
// indentation was stripped.
func blockLines(src []byte, n ast.Node) []byte {
lines := n.Lines()
var b bytes.Buffer
for i := 0; i < lines.Len(); i++ {
seg := lines.At(i)
b.Write(seg.Value(src))
}
return bytes.TrimRight(b.Bytes(), "\n")
}
// code writes one code block, syntax-painted when a grammar claims the fence's
// language and plain when none does. It emits the same <span class="s-…">
// runs as the code view, so the stylesheet already knows how to color it; the
// <li>-per-line scaffolding does not come along, since a snippet has no line
// numbers, no coverage and no notes of its own.
func (nr *noteRenderer) code(w util.BufWriter, src []byte, lang string) {
w.WriteString(`<pre class="mdcode"`)
if lang != "" {
w.WriteString(` data-lang="`)
w.WriteString(template.HTMLEscapeString(lang))
w.WriteByte('"')
}
w.WriteString("><code>")
var b strings.Builder
b.Grow(len(src) + len(src)/4)
classes := nr.hl.classifyLang(lang, src)
names := []string(nil)
if nr.hl != nil {
names = nr.hl.classNames
}
for start := 0; start <= len(src); {
end := start
for end < len(src) && src[end] != '\n' {
end++
}
emitLine(&b, src[start:end], classes[start:end], names)
if end >= len(src) {
break
}
b.WriteByte('\n')
start = end + 1
}
w.WriteString(b.String())
w.WriteString("</code></pre>\n")
}
|