Files

429 lines
12 KiB
Go

// Package importer brings sites in from elsewhere: WordPress (a live site or
// an export file), Hugo, Jekyll and Ghost. Pages become Markdown where that's
// faithful, keep their HTML where it isn't, bring their pictures through the
// same re-encoding as uploads, and leave a redirect at every old address.
package importer
import (
"fmt"
"regexp"
"strings"
"golang.org/x/net/html"
"golang.org/x/net/html/atom"
)
// Converted is a body turned into Markdown, and what couldn't be.
type Converted struct {
Markdown string
Lossy []string // why Markdown would lose something; empty when it's faithful
Images []string // every image address, in order
}
// ToMarkdown converts an HTML body. When something can't be said in Markdown
// (an embed, a layout block, a table with merged cells), Lossy says what, and
// the caller keeps the HTML instead.
func ToMarkdown(src string) Converted {
c := &conv{}
nodes, err := html.ParseFragment(strings.NewReader(src), &html.Node{Type: html.ElementNode, Data: "body", DataAtom: atom.Body})
if err != nil {
return Converted{Lossy: []string{"HTML that couldn't be read"}}
}
var b strings.Builder
for _, n := range nodes {
c.block(&b, n, "")
}
md := regexp.MustCompile(`\n{3,}`).ReplaceAllString(strings.TrimSpace(b.String()), "\n\n")
return Converted{Markdown: md + "\n", Lossy: dedupe(c.lossy), Images: c.images}
}
type conv struct {
lossy []string
images []string
}
func dedupe(xs []string) []string {
seen := map[string]bool{}
var out []string
for _, x := range xs {
if !seen[x] {
seen[x] = true
out = append(out, x)
}
}
return out
}
func attr(n *html.Node, k string) string {
for _, a := range n.Attr {
if a.Key == k {
return a.Val
}
}
return ""
}
// Elements that are layout or behavior, never words: Markdown can't hold them.
var lossyTags = map[string]string{
"iframe": "an embed", "video": "a video", "audio": "audio", "script": "a script", "style": "inline styles",
"form": "a form", "input": "a form", "select": "a form", "textarea": "a form", "button": "a button",
"object": "an embed", "embed": "an embed", "svg": "a drawing (SVG)", "canvas": "a canvas", "map": "an image map",
"details": "a collapsible section", "sup": "superscript", "sub": "subscript", "u": "underlining", "mark": "highlighting",
}
// WordPress block classes that are layout.
var layoutClass = regexp.MustCompile(`\bwp-block-(columns|column|cover|media-text|buttons|button|embed|table|pullquote|spacer|separator-dots|social-links)\b|\belementor|\bvc_|\bet_pb_|\bfl-row|\bwp-block-gallery\b.*\bcolumns-`)
func (c *conv) block(b *strings.Builder, n *html.Node, indent string) {
switch n.Type {
case html.TextNode:
if t := strings.TrimSpace(n.Data); t != "" {
b.WriteString(indent + escapeText(collapse(n.Data), true) + "\n\n")
}
return
case html.CommentNode:
return // WordPress block markers
case html.ElementNode:
default:
return
}
if why, ok := lossyTags[n.Data]; ok {
c.lossy = append(c.lossy, why)
}
if cls := attr(n, "class"); layoutClass.MatchString(cls) {
c.lossy = append(c.lossy, "a layout block ("+strings.Fields(layoutClass.FindString(cls))[0]+")")
}
if attr(n, "style") != "" {
c.lossy = append(c.lossy, "inline styles")
}
switch n.Data {
case "p":
if t := strings.TrimSpace(c.inline(n)); t != "" {
t = escapeLineStart(t)
b.WriteString(indent + strings.ReplaceAll(t, "\n", "\n"+indent) + "\n\n")
}
case "h1", "h2", "h3", "h4", "h5", "h6":
level := int(n.Data[1] - '0')
b.WriteString(indent + strings.Repeat("#", level) + " " + strings.TrimSpace(strings.ReplaceAll(c.inline(n), "\n", " ")) + "\n\n")
case "ul", "ol":
c.list(b, n, indent)
b.WriteString("\n")
case "blockquote":
var inner strings.Builder
for ch := n.FirstChild; ch != nil; ch = ch.NextSibling {
c.blockOrInline(&inner, ch, "")
}
for _, l := range strings.Split(strings.TrimSpace(inner.String()), "\n") {
b.WriteString(indent + strings.TrimRight("> "+l, " ") + "\n")
}
b.WriteString("\n")
case "pre":
code := textOf(n)
lang := ""
for ch := n.FirstChild; ch != nil; ch = ch.NextSibling {
if ch.Type == html.ElementNode && ch.Data == "code" {
if m := regexp.MustCompile(`language-([\w+-]+)`).FindStringSubmatch(attr(ch, "class")); m != nil {
lang = m[1]
}
}
}
fence := "```"
for strings.Contains(code, fence) {
fence += "`"
}
b.WriteString(indent + fence + lang + "\n" + strings.TrimRight(code, "\n") + "\n" + indent + fence + "\n\n")
case "hr":
b.WriteString(indent + "---\n\n")
case "figure":
var caption string
for ch := n.FirstChild; ch != nil; ch = ch.NextSibling {
if ch.Type == html.ElementNode && ch.Data == "figcaption" {
caption = strings.TrimSpace(c.inline(ch))
continue
}
c.blockOrInline(b, ch, indent)
}
if caption != "" {
b.WriteString(indent + "*" + caption + "*\n\n")
}
case "table":
c.table(b, n, indent)
case "img":
b.WriteString(indent + c.image(n) + "\n\n")
case "br":
default:
// div, section, article, span and the like: their content, in place.
for ch := n.FirstChild; ch != nil; ch = ch.NextSibling {
c.blockOrInline(b, ch, indent)
}
}
}
// blockOrInline handles a child that may be a block or loose inline content.
func (c *conv) blockOrInline(b *strings.Builder, n *html.Node, indent string) {
if n.Type == html.ElementNode && isInline(n.Data) {
if t := strings.TrimSpace(c.inline(&html.Node{Type: html.ElementNode, Data: "span", FirstChild: n, LastChild: n})); t != "" {
b.WriteString(indent + t + "\n\n")
}
return
}
c.block(b, n, indent)
}
func isInline(tag string) bool {
switch tag {
case "a", "strong", "b", "em", "i", "code", "span", "del", "s", "strike", "abbr", "cite", "q", "small", "time", "kbd", "var", "samp", "sup", "sub", "u", "mark", "br":
return true
}
return false
}
func (c *conv) list(b *strings.Builder, n *html.Node, indent string) {
i := 1
if s := attr(n, "start"); s != "" {
fmt.Sscanf(s, "%d", &i)
}
for li := n.FirstChild; li != nil; li = li.NextSibling {
if li.Type != html.ElementNode || li.Data != "li" {
continue
}
marker := "- "
if n.Data == "ol" {
marker = fmt.Sprintf("%d. ", i)
i++
}
pad := indent + strings.Repeat(" ", len(marker))
var text strings.Builder
var nested strings.Builder
for ch := li.FirstChild; ch != nil; ch = ch.NextSibling {
if ch.Type == html.ElementNode && (ch.Data == "ul" || ch.Data == "ol") {
c.list(&nested, ch, pad)
continue
}
if ch.Type == html.ElementNode && (ch.Data == "p" || ch.Data == "div") {
text.WriteString(" " + c.inline(ch))
continue
}
text.WriteString(c.inlineNode(ch))
}
line := strings.TrimSpace(collapseKeepBreaks(text.String()))
b.WriteString(indent + marker + strings.ReplaceAll(line, "\n", "\n"+pad) + "\n")
b.WriteString(nested.String())
}
}
func (c *conv) table(b *strings.Builder, n *html.Node, indent string) {
var rows [][]string
head := -1
var walk func(*html.Node)
simple := true
walk = func(x *html.Node) {
for ch := x.FirstChild; ch != nil; ch = ch.NextSibling {
if ch.Type != html.ElementNode {
continue
}
switch ch.Data {
case "tr":
var cells []string
for td := ch.FirstChild; td != nil; td = td.NextSibling {
if td.Type != html.ElementNode || (td.Data != "td" && td.Data != "th") {
continue
}
if attr(td, "colspan") != "" || attr(td, "rowspan") != "" {
simple = false
}
for k := td.FirstChild; k != nil; k = k.NextSibling {
if k.Type == html.ElementNode && !isInline(k.Data) && k.Data != "p" {
simple = false
}
}
if td.Data == "th" && head < 0 {
head = len(rows)
}
cells = append(cells, strings.ReplaceAll(strings.TrimSpace(strings.ReplaceAll(c.inline(td), "\n", " ")), "|", `\|`))
}
rows = append(rows, cells)
default:
walk(ch)
}
}
}
walk(n)
if !simple || len(rows) == 0 {
c.lossy = append(c.lossy, "a table with merged or complex cells")
return
}
width := 0
for _, r := range rows {
width = max(width, len(r))
}
if head != 0 { // GitHub tables need a header row: an empty one
rows = append([][]string{make([]string, width)}, rows...)
}
for i, r := range rows {
for len(r) < width {
r = append(r, "")
}
b.WriteString(indent + "| " + strings.Join(r, " | ") + " |\n")
if i == 0 {
b.WriteString(indent + "|" + strings.Repeat(" --- |", width) + "\n")
}
}
b.WriteString("\n")
}
func (c *conv) image(n *html.Node) string {
src := attr(n, "src")
if src == "" {
if s := attr(n, "data-src"); s != "" {
src = s
}
}
c.images = append(c.images, src)
alt := strings.ReplaceAll(strings.ReplaceAll(attr(n, "alt"), "[", ""), "]", "")
title := attr(n, "title")
if title != "" {
return fmt.Sprintf(`![%s](%s "%s")`, alt, src, strings.ReplaceAll(title, `"`, "'"))
}
return fmt.Sprintf("![%s](%s)", alt, src)
}
func (c *conv) inline(n *html.Node) string {
var b strings.Builder
for ch := n.FirstChild; ch != nil; ch = ch.NextSibling {
b.WriteString(c.inlineNode(ch))
}
return collapseKeepBreaks(b.String())
}
func (c *conv) inlineNode(n *html.Node) string {
switch n.Type {
case html.TextNode:
return escapeText(n.Data, false)
case html.ElementNode:
default:
return ""
}
if why, ok := lossyTags[n.Data]; ok {
c.lossy = append(c.lossy, why)
}
inner := func() string { return c.inline(n) }
wrap := func(mark string) string {
t := inner()
lead, trail := "", ""
if strings.HasPrefix(t, " ") {
lead = " "
}
if strings.HasSuffix(t, " ") {
trail = " "
}
t = strings.TrimSpace(t)
if t == "" {
return lead + trail
}
return lead + mark + t + mark + trail
}
switch n.Data {
case "strong", "b":
return wrap("**")
case "em", "i", "cite":
return wrap("*")
case "del", "s", "strike":
return wrap("~~")
case "code", "kbd", "samp":
t := textOf(n)
tick := "`"
for strings.Contains(t, tick) {
tick += "`"
}
if strings.HasPrefix(t, "`") || strings.HasSuffix(t, "`") {
t = " " + t + " "
}
return tick + t + tick
case "a":
href := attr(n, "href")
t := strings.TrimSpace(inner())
if href == "" {
return t
}
if t == "" {
return ""
}
if title := attr(n, "title"); title != "" && title != t {
return fmt.Sprintf(`[%s](%s "%s")`, t, href, strings.ReplaceAll(title, `"`, "'"))
}
return "[" + t + "](" + href + ")"
case "img":
return c.image(n)
case "br":
return "\\\n"
case "p", "div":
return " " + inner() + " "
default:
return inner()
}
}
func textOf(n *html.Node) string {
var b strings.Builder
var walk func(*html.Node)
walk = func(x *html.Node) {
if x.Type == html.TextNode {
b.WriteString(x.Data)
}
for ch := x.FirstChild; ch != nil; ch = ch.NextSibling {
walk(ch)
}
}
walk(n)
return b.String()
}
var spaces = regexp.MustCompile(`[ \t\r\n]+`)
func collapse(s string) string { return spaces.ReplaceAllString(s, " ") }
// collapseKeepBreaks collapses whitespace but keeps the hard breaks (\ then
// a newline) inline conversion produced.
func collapseKeepBreaks(s string) string {
parts := strings.Split(s, "\\\n")
for i, p := range parts {
parts[i] = collapse(p)
if i > 0 {
parts[i] = strings.TrimLeft(parts[i], " ")
}
}
return strings.Join(parts, "\\\n")
}
var (
mdSpecial = strings.NewReplacer(`\`, `\\`, "`", "\\`", "*", `\*`, "[", `\[`, "]", `\]`, "<", `\<`)
// An underscore inside a word (snake_case) can't start emphasis; only
// one at a word's edge needs escaping.
edgeUnderscore = regexp.MustCompile(`(^|[^\pL\pN])_|_([^\pL\pN]|$)`)
lineStart = regexp.MustCompile(`^(\s*)([#>+-]|\d+\.)(\s)`)
)
// escapeText keeps text from being read as Markdown.
func escapeText(s string, block bool) string {
s = mdSpecial.Replace(s)
s = edgeUnderscore.ReplaceAllStringFunc(s, func(m string) string { return strings.ReplaceAll(m, "_", `\_`) })
if block {
s = escapeLineStart(s)
}
return s
}
// escapeLineStart keeps a line from opening a heading, quote or list: a
// backslash before the marker, or before the dot of "1." (a digit itself
// can't be escaped).
func escapeLineStart(s string) string {
return lineStart.ReplaceAllStringFunc(s, func(m string) string {
sub := lineStart.FindStringSubmatch(m)
if strings.HasSuffix(sub[2], ".") {
return sub[1] + strings.TrimSuffix(sub[2], ".") + `\.` + sub[3]
}
return sub[1] + `\` + sub[2] + sub[3]
})
}