Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions CLAUDE.md

Large diffs are not rendered by default.

333 changes: 333 additions & 0 deletions _plans/056_escape-text-on-read.md

Large diffs are not rendered by default.

20 changes: 20 additions & 0 deletions docs/markdown-file.md
Original file line number Diff line number Diff line change
Expand Up @@ -635,6 +635,26 @@ gets to the stored page (measured; see
Don't use HTML comments to leave notes on published pages that you will `read`
or `export` in the future.

### Backslash escapes

A backslash before a punctuation character makes the character literal, as in
any Markdown: `\*not emphasis\*` publishes as `*not emphasis*`, and `1\. not a
list` publishes as a paragraph.

`read` and `export` write these escapes wherever page text would otherwise read
as Markdown. Without them, text such as `1. not a list`, `# not a heading`,
`> 90 days` or `*not emphasis*` would turn into a list, a heading, a quote or
emphasis the next time that you publish the file. markfluence escapes a
character only where it can have an effect, so `snake_case`, `a * b`,
`about ~5 min` and `AT&T` stay as they are. In a few places a character
reference does the same job as a backslash, for example `~` for a `~`
next to strikethrough.

A URL or an email address that is plain text on the page, and not a link, is
written as `https\://example.com` or `ops\@example.com`. Without the
backslash, publishing would turn it into a link. You can remove the backslash
if you want a link.

### Raw Confluence storage format

You can paste Confluence
Expand Down
92 changes: 32 additions & 60 deletions internal/convert/aclink.go
Original file line number Diff line number Diff line change
Expand Up @@ -218,7 +218,7 @@ func (r *mdRenderer) renderACLink(n *snode) string {
return r.renderAnchorLink(n, anchor)
}
// No target and no anchor: there is nothing to point at.
return serialize(n)
return r.serializeInline(n)
case target.name == "ri:page":
return r.renderPageLink(n, target, anchor)
case target.name == "ri:space":
Expand All @@ -230,7 +230,7 @@ func (r *mdRenderer) renderACLink(n *snode) string {
// images are uploaded (images.go); ri:blog-post cannot be resolved to
// an id, because SearchPagesByTitle does not see blog posts. Both
// round-trip exactly as storage.
return serialize(n)
return r.serializeInline(n)
}
}

Expand All @@ -249,7 +249,7 @@ func acLinkTarget(n *snode) *snode {
func (r *mdRenderer) renderPageLink(n, target *snode, anchor string) string {
href, ok := r.pageLinks[pageTarget(target)]
if !ok {
return serialize(n)
return r.serializeInline(n)
}
if anchor != "" {
// Appended verbatim, still percent-encoded: it is a URL fragment, and
Expand All @@ -264,7 +264,7 @@ func (r *mdRenderer) renderPageLink(n, target *snode, anchor string) string {
func (r *mdRenderer) renderSpaceLink(n, target *snode) string {
key := target.attrs["ri:space-key"]
if key == "" || r.siteURL == "" {
return serialize(n)
return r.serializeInline(n)
}
return mdLink(r.acLinkText(n, key), r.siteURL+"/wiki/spaces/"+key)
}
Expand Down Expand Up @@ -348,11 +348,11 @@ const unknownUserName = "Unlicensed user"
func (r *mdRenderer) renderUserMention(n, target *snode) string {
id := target.attrs["ri:account-id"]
if id == "" {
return serialize(n)
return r.serializeInline(n)
}
name, known := r.userNames[id]
if !known {
return serialize(n)
return r.serializeInline(n)
}
if name == "" {
name = unknownUserName
Expand All @@ -366,7 +366,7 @@ func (r *mdRenderer) renderAnchorLink(n *snode, anchor string) string {
if !ok {
// A "#slug" matching no heading publishes as a dead relative href, and
// the forward path says nothing about it. Keep the storage, which works.
return serialize(n)
return r.serializeInline(n)
}
return mdLink(r.acLinkText(n, anchor), "#"+slug)
}
Expand All @@ -376,14 +376,10 @@ func (r *mdRenderer) renderAnchorLink(n *snode, anchor string) string {
// is that name.
//
// A body comes in two spellings -- ac:link-body holds rich text, and
// ac:plain-text-link-body holds CDATA -- and both occur on real pages.
//
// Only the *raw* sources are escaped, and which is which is the whole point of
// the split below. An ac:link-body has already been rendered to Markdown by
// renderInlineChildren, so escaping it would turn a bold link body into a
// literal "\*\*bold\*\*". The CDATA body and the fallback are plain text
// straight off the server -- a page title, a space key, an anchor -- and a "]"
// in any of them ends the link early.
// ac:plain-text-link-body holds CDATA -- and both occur on real pages. The rich
// body's text nodes are escaped as they render; the CDATA body and the fallback
// are plain text straight off the server -- a page title, a space key, an
// anchor -- and are escaped whole.
func (r *mdRenderer) acLinkText(n *snode, fallback string) string {
if b := findChild(n, "ac:link-body"); b != nil {
if s := r.inlineTextForLink(b); s != "" {
Expand All @@ -398,71 +394,47 @@ func (r *mdRenderer) acLinkText(n *snode, fallback string) string {
return escapeLinkText(fallback)
}

// inlineTextForLink renders a node's children as a Markdown link's text,
// escaping the result when it is nothing but plain text.
//
// The distinction matters both ways, and an earlier version got it wrong in one
// direction. Escaping a *rendered* body turns "<strong>bold</strong>" into a
// literal "\*\*bold\*\*", which is why the escaping was first applied only to
// the raw sources. But the common case for a link body is plain text, and
// leaving it unescaped loses the link outright: a page titled "Q1 Draft]"
// rendered as "[Q1 Draft] notes](url)", which CommonMark reads as literal text,
// so the next update publishes no link at all. Found in review.
// inlineTextForLink renders a node's children as a Markdown link's text.
//
// "Nothing but plain text" is checkable rather than guessable: a text node is
// an snode with an empty name, so a body whose every descendant is one carries
// no markup for escaping to damage.
// Its text nodes are escaped as they render, with a "]" escaped too while
// linkDepth is raised: unescaped, a page titled "Q1 Draft]" rendered as
// "[Q1 Draft] notes](url)", which CommonMark reads as literal text, so the next
// update publishes no link at all. Escaping at the text node rather than over
// the rendered result is what lets a body holding markup be escaped at all:
// escaping "<strong>bold</strong>" after rendering turns it into a literal
// "\*\*bold\*\*", which is why this used to escape plain-text bodies only.
//
// Whitespace at the text's edges is kept inside the brackets, where Markdown
// allows it and publishes it back; trimming it joined "<a>see </a>here" into
// "[see](url)here" (#204).
func (r *mdRenderer) inlineTextForLink(n *snode) string {
r.linkDepth++
defer func() { r.linkDepth-- }()
rendered := r.renderInlineRun(n)
if strings.TrimSpace(rendered) == "" {
return ""
}
if !onlyText(n) {
return rendered
}
return escapeLinkText(rendered)
return rendered
}

// onlyText reports whether every descendant of n is a text node, so rendering
// it produced no Markdown syntax of its own.
func onlyText(n *snode) bool {
for _, k := range n.kids {
if k.name != "" || !onlyText(k) {
return false
}
}
return true
}

// escapeLinkText makes plain text safe to use as a Markdown link's text.
//
// The set is deliberately the one that *breaks* a link rather than everything
// Markdown reads specially: an unescaped "]" ends the text early and leaves the
// rest of the line as literal junk, and a backslash has to go first or it would
// escape the escapes. A title like "*Foo*" is a different problem -- it renders
// as emphasis instead of as asterisks, losing fidelity without breaking the
// link -- and is knowingly not handled here, since escaping every Markdown
// indicator in every recovered title is a larger change with its own round-trip
// consequences.
// escapeLinkText makes plain text -- a string rather than a rendered node --
// safe to use as a Markdown link's text: escapeText with a "]" escaped too.
//
// Before this, mdLink was a bare Sprintf: any page title holding a bracket
// exported as a broken link, which mentions turned from theoretical into likely
// because display names carry them.
// It escapes everything Markdown would read, not only what breaks the link.
// It began as "\", "[" and "]", the set that breaks a link: before that,
// mdLink was a bare Sprintf and any page title holding a bracket exported as a
// broken link, which mentions turned from theoretical into likely because
// display names carry them. A title like "*Foo*" then still rendered as
// emphasis, and #203 closed that the way it closed it for all text.
func escapeLinkText(s string) string {
s = strings.ReplaceAll(s, `\`, `\\`)
s = strings.ReplaceAll(s, "[", `\[`)
return strings.ReplaceAll(s, "]", `\]`)
return escapeText(s, true)
}

// mdLink renders an inline Markdown link, falling back to showing the
// destination when there is no text for it.
func mdLink(text, dest string) string {
if text == "" {
text = dest
text = escapeLinkText(dest)
}
return fmt.Sprintf("[%s](%s)", text, dest)
}
16 changes: 11 additions & 5 deletions internal/convert/aclink_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -267,11 +267,17 @@ func TestLinkTextDoesNotEscapeARenderedBody(t *testing.T) {
}
}

// TestLinkTextEscapesABackslashFirst: the backslash pass has to run before the
// bracket passes, or it would escape the escapes they add.
func TestLinkTextEscapesABackslashFirst(t *testing.T) {
if got := convert.EscapeLinkTextForTest(`a\b]c`); got != `a\\b\]c` {
t.Errorf("EscapeLinkText = %q, want %q", got, `a\\b\]c`)
// TestLinkTextEscapesABackslashBeforeABracket: a backslash before the "]" it
// escapes must itself be escaped, or it would escape the escape. One before a
// letter is literal in CommonMark and is left alone.
func TestLinkTextEscapesABackslashBeforeABracket(t *testing.T) {
for in, want := range map[string]string{
`a\]c`: `a\\\]c`,
`a\b]c`: `a\b\]c`,
} {
if got := convert.EscapeLinkTextForTest(in); got != want {
t.Errorf("EscapeLinkText(%q) = %q, want %q", in, got, want)
}
}
}

Expand Down
Loading
Loading