Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 8 additions & 1 deletion internal/convert/aclink.go
Original file line number Diff line number Diff line change
Expand Up @@ -412,8 +412,15 @@ func (r *mdRenderer) acLinkText(n *snode, fallback string) string {
// "Nothing but plain text" is checkable rather than guessable: a text node is
// an snode with an empty name, so a body whose every descendant is one carries
// no markup for escaping to damage.
//
// Whitespace at the text's edges is kept inside the brackets, where Markdown
// allows it and publishes it back; trimming it joined "<a>see </a>here" into
// "[see](url)here" (#204).
func (r *mdRenderer) inlineTextForLink(n *snode) string {
rendered := r.renderInlineChildren(n)
rendered := r.renderInlineRun(n)
if strings.TrimSpace(rendered) == "" {
return ""
}
if !onlyText(n) {
return rendered
}
Expand Down
80 changes: 71 additions & 9 deletions internal/convert/storage_to_md.go
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,7 @@ package convert
// aclink.go.

import (
"bytes"
"encoding/json"
"encoding/xml"
"fmt"
Expand All @@ -24,6 +25,7 @@ import (
"regexp"
"sort"
"strings"
"unicode"
)

// calloutMacroInverse maps a Confluence callout macro back to a GitHub alert
Expand Down Expand Up @@ -288,7 +290,10 @@ func (r *mdRenderer) renderBlock(n *snode, listIndent string) string {
switch n.name {
case "h1", "h2", "h3", "h4", "h5", "h6":
level := int(n.name[1] - '0')
return strings.Repeat("#", level) + " " + r.renderInlineChildren(n)
// A heading is one line, so a hard break would end it and publish the
// rest as a paragraph. The break stays as the <br /> it was.
text := strings.ReplaceAll(r.renderInlineChildren(n), hardBreak, "<br />")
return strings.Repeat("#", level) + " " + text
case "p":
return r.renderInlineChildren(n)
case "ul":
Expand Down Expand Up @@ -638,9 +643,31 @@ func (r *mdRenderer) renderCallout(n *snode, alert string) string {

// renderInlineChildren renders a node's children as a single inline string.
func (r *mdRenderer) renderInlineChildren(n *snode) string {
var b strings.Builder
return strings.TrimSpace(r.renderInlineRun(n))
}

// renderInlineRun is renderInlineChildren without the trim, for a mark, which
// must see the whitespace at its own edges to move it outside its delimiters.
func (r *mdRenderer) renderInlineRun(n *snode) string {
var buf []byte
for _, k := range coalesceSplitMarks(n.kids) {
part := r.renderInline(k)
// A mark moves its edge whitespace outside its delimiters, so a space
// there can meet a space in the text beside it. One is kept: storage
// collapses a run of spaces anyway, so the second would only reach the
// Markdown on this read and be gone on the next, and the Markdown would
// not be a fixed point.
if bytes.HasSuffix(buf, []byte(" ")) && strings.HasPrefix(part, " ") && !opensWithBreak(part) {
part = strings.TrimLeft(part, " ")
}
// Likewise, whitespace before a hard break is gone on the next read,
// since publishing ends the line at the break: "a <br />" read as
// "a " (three spaces) and read back as two. The break is written as
// exactly the two spaces that make it one.
if opensWithBreak(part) {
buf = bytes.TrimRight(buf, " \t")
part = hardBreak + strings.TrimLeft(part, " \t")[1:]
}
// Storage is XHTML and its newlines are insignificant, so
// "<br />\nSecond" is the ordinary spelling -- and that newline
// normalizes to a space, landing immediately after the two-space hard
Expand All @@ -651,12 +678,43 @@ func (r *mdRenderer) renderInlineChildren(n *snode) string {
// *are* the next hard break. Found by the round-trip property test.
// Never trim a hard break itself: two of them in a row are two blank
// line-endings, and its own leading spaces are what make it one.
if k.name != "br" && strings.HasSuffix(b.String(), hardBreak) {
if !opensWithBreak(part) && bytes.HasSuffix(buf, []byte(hardBreak)) {
part = strings.TrimLeft(part, " \t")
}
b.WriteString(part)
buf = append(buf, part...)
}
return strings.TrimSpace(b.String())
return string(buf)
}

// opensWithBreak reports whether an inline part begins with a hard break --
// a <br /> itself, or a mark that moved one out from its leading edge -- whose
// leading spaces are what make it a break and must never be trimmed.
func opensWithBreak(part string) bool {
return strings.HasPrefix(strings.TrimLeft(part, " \t"), "\n")
}

// renderMark renders a bold, italic or strikethrough span, moving whitespace
// at its edges outside the delimiters (#204). It cannot stay inside, since
// CommonMark refuses "**bold **" as emphasis -- a closing delimiter preceded by
// whitespace does not close -- and it cannot be dropped, since the editor
// leaves a space typed at the end of a bold run inside the run, and dropping
// it published "**bold**next" as one word. Unicode whitespace counts, because
// CommonMark's flanking rule counts it: a trailing no-break space refuses the
// delimiter as surely as a space does. A mark holding only whitespace has
// nothing Markdown can mark, so it renders as that whitespace.
//
// This is the whitespace half of the flanking rule only. A mark whose text
// starts or ends with punctuation against a letter outside it ("a**(b)**")
// still fails the other half, and did before.
func (r *mdRenderer) renderMark(n *snode, delim string) string {
s := r.renderInlineRun(n)
body := strings.TrimLeftFunc(s, unicode.IsSpace)
if body == "" {
return s
}
lead := s[:len(s)-len(body)]
inner := strings.TrimRightFunc(body, unicode.IsSpace)
return lead + delim + inner + delim + body[len(inner):]
}

// hardBreak is Markdown's two-space line break, as renderInline emits it for a
Expand Down Expand Up @@ -821,13 +879,13 @@ func (r *mdRenderer) renderInline(n *snode) string {
}
switch n.name {
case "strong", "b":
return "**" + r.renderInlineChildren(n) + "**"
return r.renderMark(n, "**")
case "em", "i":
return "*" + r.renderInlineChildren(n) + "*"
return r.renderMark(n, "*")
case "code":
return "`" + textContent(n) + "`"
case "del", "s", "strike":
return "~~" + r.renderInlineChildren(n) + "~~"
return r.renderMark(n, "~~")
case "br":
return hardBreak
case "a":
Expand All @@ -845,7 +903,11 @@ func (r *mdRenderer) renderInline(n *snode) string {
// would otherwise render the node and the fallback one after the other.
return serialize(adfPassthrough(n))
default:
return r.renderInlineChildren(n)
// A wrapper Markdown has no syntax for (a coloured <span>, <u>, <sup>)
// keeps its edge whitespace for the run around it to settle: trimming it
// here joined "<span><strong>bold </strong></span>next" into one word
// after renderMark had moved the space out.
return r.renderInlineRun(n)
}
}

Expand Down
2 changes: 1 addition & 1 deletion internal/convert/storage_to_md_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -350,7 +350,7 @@ func TestStorageToMarkdownCoalescesSplitMarks(t *testing.T) {
},
"link only partly bold does not merge": {
in: `<p><strong>a </strong><a href="https://example.com">b<strong>c</strong></a></p>`,
want: "**a**[b**c**](https://example.com)\n",
want: "**a** [b**c**](https://example.com)\n",
},
}
for name, tc := range tests {
Expand Down
53 changes: 47 additions & 6 deletions internal/convert/table_property_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -106,7 +106,7 @@ func newPropertyEnv(t testing.TB) *propertyEnv {
}

func (e *propertyEnv) check(seed uint64) string {
g := &tableGen{r: rand.New(rand.NewPCG(seed, 0x55))}
g := &tableGen{r: rand.New(rand.NewPCG(seed, 0x55)), edge: rand.New(rand.NewPCG(seed, 0x204))}
return e.checkStorage(fmt.Sprintf("seed %d", seed), g.table(0))
}

Expand Down Expand Up @@ -416,10 +416,27 @@ func modelInline(kids []*snode) string {
if a, ok := inlineAlias[name]; ok {
name = a
}
fmt.Fprintf(&b, "<%s>%s</%s>", name, modelInline(k.kids), name)
inner := modelInline(k.kids)
// A space at the edge of a mark separates the same words inside it
// as outside, and Markdown can only put it outside (#204). Only the
// marks: a <code> span's spaces are its text.
var lead, trail string
if name == "strong" || name == "em" || name == "del" {
body := strings.TrimLeft(inner, " ")
lead = inner[:len(inner)-len(body)]
inner = strings.TrimRight(body, " ")
trail = body[len(inner):]
}
fmt.Fprintf(&b, "%s<%s>%s</%s>%s", lead, name, inner, name, trail)
}
}
s := whitespaceRunRE.ReplaceAllString(b.String(), " ")
// Two runs of one mark with nothing but a space between them read as one
// run: "<em>a</em> <em>b</em>" and "<em>a b</em>" look the same.
for _, m := range []string{"strong", "em", "del"} {
s = strings.ReplaceAll(s, "</"+m+"><"+m+">", "")
s = strings.ReplaceAll(s, "</"+m+"> <"+m+">", " ")
}
// Whitespace beside a line break shows nothing, and publishing writes a
// newline after every <br />.
return strings.ReplaceAll(strings.ReplaceAll(s, " ⏎", "⏎"), "⏎ ", "⏎")
Expand Down Expand Up @@ -541,7 +558,10 @@ func modelHasElement(n *snode) bool {
// which #203 covers, and inline tags read has no Markdown for (<time>, <u>),
// which read drops in every paragraph, and two lists side by side, which
// Markdown reads back as one (#205) -- all gaps older than #55.
type tableGen struct{ r *rand.Rand }
// tableGen builds table storage from a seed. edge is a second stream, for the
// spaces markText adds at a mark's edge (#204), so that adding them left every
// seed's table otherwise as it was.
type tableGen struct{ r, edge *rand.Rand }

func (g *tableGen) chance(p float64) bool { return g.r.Float64() < p }

Expand Down Expand Up @@ -680,12 +700,25 @@ func (g *tableGen) text() string {
return strings.Join(ws, " ")
}

// markText is text for inside a mark, sometimes with a space at an edge: the
// editor leaves a space typed at the end of a bold run inside the run (#204).
func (g *tableGen) markText() string {
s := g.text()
if g.edge.Float64() < 0.2 {
s += " "
}
if g.edge.Float64() < 0.1 {
s = " " + s
}
return s
}

func (g *tableGen) inline() string {
switch g.r.IntN(8) {
case 0:
return "<strong>" + g.text() + "</strong>"
return "<strong>" + g.markText() + "</strong>"
case 1:
return "<em>" + g.text() + "</em>"
return "<em>" + g.markText() + "</em>"
case 2:
return "<code>" + g.text() + "</code>"
case 3:
Expand Down Expand Up @@ -714,7 +747,15 @@ func (g *tableGen) paragraph(align string, cellStyle bool) string {
}
body := g.inline()
if g.chance(0.3) {
body += " " + g.inline()
// A mark that already ends in a space is followed directly, which is
// the shape #204 joined into one word; otherwise a space separates.
next := g.inline()
sep := " "
if strings.HasSuffix(body, " </strong>") || strings.HasSuffix(body, " </em>") ||
strings.HasPrefix(next, "<strong> ") || strings.HasPrefix(next, "<em> ") {
sep = ""
}
body += sep + next
}
if g.chance(0.1) {
body += "<br />" + g.text()
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
<p>A trailing space inside bold: <strong>bold </strong>next.</p>
<p>Inside italic: <em>x </em>y, and strikethrough: <del>lost for </del>13.5h.</p>
<p>A leading space: a<strong> bold</strong> word.</p>
<p>A space on both sides of the edge: <strong>bold </strong> next.</p>
<p>A no-break space: <strong>a&nbsp;</strong>b.</p>
<p>Nested marks: <strong><em>n </em></strong>x.</p>
<p>A mark holding only a space: x<strong> </strong>y.</p>
<p>A hard break at the end of a mark: <strong>a<br /></strong>b.</p>
<p>A hard break at the start of a mark, after a space: a <strong><br />b</strong>.</p>
<p>A space before a hard break: a <br />b.</p>
<p>Coloured bold: <span style="color: rgb(255,0,0);"><strong>bold </strong></span>next, and underline: <u>under </u>next.</p>
<p>Link text: <a href="https://example.com">see </a>here, and <a href="https://example.com"><strong>bold </strong></a>next.</p>
<h2><strong>A heading with a break at a mark&apos;s edge<br /></strong>continued</h2>
<h3>A heading with a bare break<br />continued</h3>
<table data-layout="align-start"><thead><tr><th>Item</th><th>Time</th></tr></thead><tbody><tr><td><del>lost for </del>13.5h</td><td>1</td></tr></tbody></table>
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
A trailing space inside bold: **bold** next.

Inside italic: *x* y, and strikethrough: ~~lost for~~ 13.5h.

A leading space: a **bold** word.

A space on both sides of the edge: **bold** next.

A no-break space: **a** b.

Nested marks: ***n*** x.

A mark holding only a space: x y.

A hard break at the end of a mark: **a**
b.

A hard break at the start of a mark, after a space: a
**b**.

A space before a hard break: a
b.

Coloured bold: **bold** next, and underline: under next.

Link text: [see ](https://example.com)here, and [**bold** ](https://example.com)next.

## **A heading with a break at a mark's edge**<br />continued

### A heading with a bare break<br />continued

| Item | Time |
| --- | --- |
| ~~lost for~~ 13.5h | 1 |
2 changes: 1 addition & 1 deletion internal/convert/testdata/storage2md/split-marks/output.md
Original file line number Diff line number Diff line change
Expand Up @@ -4,4 +4,4 @@ Editor-split italic link then text: *[x](https://example.com) more text*.

Adjacent same-tag runs with no link nearby: **ab**.

A link only partly bold does not merge: **a**[b**c**](https://example.com).
A link only partly bold does not merge: **a** [b**c**](https://example.com).
Loading