470 lines
13 KiB
Go
470 lines
13 KiB
Go
package importer
|
|
|
|
import (
|
|
"bytes"
|
|
"fmt"
|
|
"io"
|
|
"net/http"
|
|
"net/url"
|
|
"os"
|
|
"path"
|
|
"path/filepath"
|
|
"regexp"
|
|
"sort"
|
|
"strings"
|
|
"time"
|
|
|
|
"gopkg.in/yaml.v3"
|
|
|
|
"git.coffeylabs.org/coffey-labs/hotdog-cms/internal/media"
|
|
"git.coffeylabs.org/coffey-labs/hotdog-cms/internal/site"
|
|
"git.coffeylabs.org/coffey-labs/hotdog-cms/internal/siteyaml"
|
|
)
|
|
|
|
// Item is one post or page, from wherever it came.
|
|
type Item struct {
|
|
Kind string // "post" or "page"
|
|
Title string
|
|
Slug string
|
|
Date time.Time
|
|
Updated time.Time
|
|
Draft bool
|
|
Summary string
|
|
Tags []string
|
|
Author string
|
|
HTML string // the body as HTML (WordPress, Ghost)
|
|
Markdown string // the body as Markdown already (Hugo, Jekyll)
|
|
Extra map[string]any // other front matter worth keeping
|
|
OldURL string // where it lived, so its address keeps working
|
|
Aliases []string // other old addresses (Hugo aliases, Jekyll redirect_from)
|
|
Dir string // where under content/ it goes ("articles", "about/team"); "" for the top
|
|
Index bool // a page with pages under it: it becomes its folder's own page (_index.md)
|
|
Image string // its featured picture
|
|
ImageAlt string
|
|
Notes []string // what the source already knows didn't come across (shortcodes…)
|
|
}
|
|
|
|
// Options says where and how items are written.
|
|
type Options struct {
|
|
SiteDir string
|
|
Overwrite bool // replace pages already there
|
|
Fetch func(u string) ([]byte, error) // download a picture; default plain HTTP
|
|
Base string // the old site's address, for relative image links
|
|
LocalRoot string // file:// pictures are read only from inside this folder (Hugo, Jekyll)
|
|
Log io.Writer
|
|
}
|
|
|
|
// Report is what happened, page by page.
|
|
type Report struct {
|
|
Written []string
|
|
Skipped []string // already there
|
|
KeptHTML map[string][]string
|
|
Notes map[string][]string
|
|
Pictures int
|
|
Failed map[string]string // picture address → why
|
|
Redirects []site.Redirect
|
|
}
|
|
|
|
// Write puts items into a site.
|
|
func Write(items []Item, opt Options) (*Report, error) {
|
|
cfgPath := filepath.Join(opt.SiteDir, "site.yaml")
|
|
cfgRaw, err := os.ReadFile(cfgPath)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("no site at %s (run hotdog-cms new first): %w", opt.SiteDir, err)
|
|
}
|
|
cfg, err := site.LoadConfig(opt.SiteDir)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
if opt.Fetch == nil {
|
|
opt.Fetch = func(u string) ([]byte, error) { return fetch(u, opt.LocalRoot) }
|
|
}
|
|
if opt.Log == nil {
|
|
opt.Log = io.Discard
|
|
}
|
|
rep := &Report{KeptHTML: map[string][]string{}, Notes: map[string][]string{}, Failed: map[string]string{}}
|
|
pictures := map[string]string{} // old address → new
|
|
taken := func(p string) bool {
|
|
_, err := os.Stat(filepath.Join(opt.SiteDir, filepath.FromSlash(p)))
|
|
return err == nil
|
|
}
|
|
keptHTML := false
|
|
redirects := map[string]string{}
|
|
for _, it := range items {
|
|
// Slugs and folders come from the export or the old site, so they're
|
|
// made safe here: one plain name each, never "..", never a path.
|
|
slug := slugify(it.Slug)
|
|
if slug == "" {
|
|
slug = slugify(it.Title)
|
|
}
|
|
it.Dir = safeDir(it.Dir)
|
|
if slug == "" {
|
|
slug = "untitled"
|
|
}
|
|
rel := path.Join("content", it.Dir, slug+".md")
|
|
if it.Index {
|
|
rel = path.Join("content", it.Dir, slug, "_index.md")
|
|
}
|
|
newURL := "/" + path.Join(it.Dir, slug) + "/"
|
|
if !opt.Overwrite && taken(rel) {
|
|
rep.Skipped = append(rep.Skipped, rel)
|
|
continue
|
|
}
|
|
month := it.Date
|
|
if month.IsZero() {
|
|
month = time.Now()
|
|
}
|
|
mediaDir := month.Format("2006/01")
|
|
// Pictures: brought in through the same re-encoding as uploads.
|
|
bring := func(u string) string {
|
|
if u == "" || strings.HasPrefix(u, "data:") {
|
|
return u
|
|
}
|
|
abs := u
|
|
if base, err := url.Parse(opt.Base); err == nil && opt.Base != "" {
|
|
if ref, err := url.Parse(u); err == nil {
|
|
abs = base.ResolveReference(ref).String()
|
|
}
|
|
}
|
|
if done, ok := pictures[abs]; ok {
|
|
return done
|
|
}
|
|
data, err := opt.Fetch(abs)
|
|
if err != nil {
|
|
rep.Failed[abs] = err.Error()
|
|
return u
|
|
}
|
|
name := path.Base(strings.SplitN(abs, "?", 2)[0])
|
|
img, err := media.Process(data, name, mediaDir, taken)
|
|
if err != nil {
|
|
rep.Failed[abs] = err.Error()
|
|
return u
|
|
}
|
|
for _, f := range img.Files {
|
|
if err := writeInSite(opt.SiteDir, path.Join("static", f.Path), f.Data); err != nil {
|
|
rep.Failed[abs] = err.Error()
|
|
return u
|
|
}
|
|
}
|
|
rep.Pictures++
|
|
pictures[abs] = img.URL
|
|
return img.URL
|
|
}
|
|
|
|
fm := map[string]any{}
|
|
var body string
|
|
format := ""
|
|
switch {
|
|
case it.HTML != "":
|
|
conv := ToMarkdown(it.HTML)
|
|
if len(conv.Lossy) > 0 {
|
|
body = cleanHTML(it.HTML)
|
|
format = "html"
|
|
keptHTML = true
|
|
rep.KeptHTML[rel] = conv.Lossy
|
|
for _, u := range imageAddresses(body) {
|
|
body = strings.ReplaceAll(body, u, bring(u))
|
|
}
|
|
} else {
|
|
body = conv.Markdown
|
|
for _, u := range conv.Images {
|
|
if n := bring(u); n != u {
|
|
body = strings.ReplaceAll(body, "]("+u, "]("+n)
|
|
}
|
|
}
|
|
}
|
|
default:
|
|
body = it.Markdown
|
|
for _, m := range mdImage.FindAllStringSubmatch(body, -1) {
|
|
if n := bring(m[2]); n != m[2] {
|
|
body = strings.ReplaceAll(body, "]("+m[2], "]("+n)
|
|
}
|
|
}
|
|
}
|
|
if len(it.Notes) > 0 {
|
|
rep.Notes[rel] = it.Notes
|
|
}
|
|
fm["title"] = it.Title
|
|
if !it.Date.IsZero() {
|
|
fm["date"] = it.Date.Format("2006-01-02")
|
|
}
|
|
if !it.Updated.IsZero() && it.Updated.Format("2006-01-02") != it.Date.Format("2006-01-02") {
|
|
fm["updated"] = it.Updated.Format("2006-01-02")
|
|
}
|
|
if it.Summary != "" {
|
|
fm["summary"] = it.Summary
|
|
}
|
|
if len(it.Tags) > 0 {
|
|
fm["tags"] = it.Tags
|
|
}
|
|
if it.Author != "" {
|
|
fm["author"] = it.Author
|
|
}
|
|
if it.Draft {
|
|
fm["draft"] = true
|
|
}
|
|
if it.Image != "" {
|
|
fm["image"] = bring(it.Image)
|
|
if it.ImageAlt != "" {
|
|
fm["image_alt"] = it.ImageAlt
|
|
}
|
|
}
|
|
if format != "" {
|
|
fm["format"] = format
|
|
}
|
|
for k, v := range it.Extra {
|
|
if _, ok := fm[k]; !ok {
|
|
fm[k] = v
|
|
}
|
|
}
|
|
doc, err := frontMatter(fm)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
if err := writeInSite(opt.SiteDir, rel, []byte("---\n"+doc+"---\n\n"+strings.TrimLeft(body, "\n"))); err != nil {
|
|
return nil, err
|
|
}
|
|
rep.Written = append(rep.Written, rel)
|
|
fmt.Fprintf(opt.Log, " %s\n", rel)
|
|
for _, old := range append([]string{it.OldURL}, it.Aliases...) {
|
|
if from := oldPath(old); from != "" && from != newURL {
|
|
redirects[from] = newURL
|
|
}
|
|
}
|
|
}
|
|
|
|
// Every old address keeps working.
|
|
text := string(cfgRaw)
|
|
if len(redirects) > 0 {
|
|
all := map[string]string{}
|
|
for _, r := range cfg.Redirects {
|
|
all[r.From] = r.To
|
|
}
|
|
for from, to := range redirects {
|
|
if _, ok := all[from]; !ok {
|
|
all[from] = to
|
|
rep.Redirects = append(rep.Redirects, site.Redirect{From: from, To: to})
|
|
}
|
|
}
|
|
froms := make([]string, 0, len(all))
|
|
for f := range all {
|
|
froms = append(froms, f)
|
|
}
|
|
sort.Strings(froms)
|
|
list := []map[string]string{}
|
|
for _, f := range froms {
|
|
list = append(list, map[string]string{"from": f, "to": all[f]})
|
|
}
|
|
if text, err = siteyaml.SetTopLevel(text, "redirects", list); err != nil {
|
|
return nil, err
|
|
}
|
|
}
|
|
// Pages kept as HTML need the site to allow (sanitized) HTML.
|
|
if keptHTML && !cfg.Markdown.UnsafeHTML && !cfg.Markdown.TrustHTML {
|
|
md := map[string]any{}
|
|
var doc yaml.Node
|
|
if yaml.Unmarshal([]byte(text), &doc) == nil && len(doc.Content) == 1 {
|
|
m := doc.Content[0]
|
|
for i := 0; i+1 < len(m.Content); i += 2 {
|
|
if m.Content[i].Value == "markdown" {
|
|
_ = m.Content[i+1].Decode(&md)
|
|
}
|
|
}
|
|
}
|
|
md["unsafe_html"] = true
|
|
if text, err = siteyaml.SetTopLevel(text, "markdown", md); err != nil {
|
|
return nil, err
|
|
}
|
|
}
|
|
if text != string(cfgRaw) {
|
|
if err := os.WriteFile(cfgPath, []byte(text), 0o644); err != nil {
|
|
return nil, err
|
|
}
|
|
}
|
|
sort.Slice(rep.Redirects, func(i, j int) bool { return rep.Redirects[i].From < rep.Redirects[j].From })
|
|
return rep, nil
|
|
}
|
|
|
|
var mdImage = regexp.MustCompile(`!\[([^\]]*)\]\(\s*<?([^)\s>]+)`)
|
|
|
|
func imageAddresses(h string) []string {
|
|
var out []string
|
|
for _, m := range regexp.MustCompile(`<img[^>]+src="([^"]+)"`).FindAllStringSubmatch(h, -1) {
|
|
out = append(out, m[1])
|
|
}
|
|
return out
|
|
}
|
|
|
|
// cleanHTML drops WordPress's block markers from HTML that's kept as it is.
|
|
func cleanHTML(h string) string {
|
|
h = regexp.MustCompile(`<!-- /?wp:[^>]*-->\n?`).ReplaceAllString(h, "")
|
|
return strings.TrimSpace(h) + "\n"
|
|
}
|
|
|
|
func oldPath(u string) string {
|
|
if u == "" {
|
|
return ""
|
|
}
|
|
p, err := url.Parse(u)
|
|
if err != nil {
|
|
return ""
|
|
}
|
|
if p.Path == "" {
|
|
return ""
|
|
}
|
|
return p.Path
|
|
}
|
|
|
|
func frontMatter(fm map[string]any) (string, error) {
|
|
order := []string{"title", "date", "updated", "summary", "tags", "author", "draft", "image", "image_alt", "format"}
|
|
m := &yaml.Node{Kind: yaml.MappingNode}
|
|
add := func(k string, v any) error {
|
|
var n yaml.Node
|
|
if err := n.Encode(v); err != nil {
|
|
return err
|
|
}
|
|
if _, ok := v.([]string); ok {
|
|
n.Style = yaml.FlowStyle
|
|
}
|
|
m.Content = append(m.Content, &yaml.Node{Kind: yaml.ScalarNode, Value: k}, &n)
|
|
return nil
|
|
}
|
|
for _, k := range order {
|
|
if v, ok := fm[k]; ok {
|
|
if err := add(k, v); err != nil {
|
|
return "", err
|
|
}
|
|
delete(fm, k)
|
|
}
|
|
}
|
|
rest := make([]string, 0, len(fm))
|
|
for k := range fm {
|
|
rest = append(rest, k)
|
|
}
|
|
sort.Strings(rest)
|
|
for _, k := range rest {
|
|
if err := add(k, fm[k]); err != nil {
|
|
return "", err
|
|
}
|
|
}
|
|
var b bytes.Buffer
|
|
enc := yaml.NewEncoder(&b)
|
|
enc.SetIndent(2)
|
|
if err := enc.Encode(m); err != nil {
|
|
return "", err
|
|
}
|
|
return b.String(), nil
|
|
}
|
|
|
|
var slugRe = regexp.MustCompile(`[^\p{Ll}\p{Lo}\p{N}]+`)
|
|
|
|
// safeDir keeps a folder from an import to plain names: "../../x" becomes
|
|
// "x", "/a//b/" becomes "a/b".
|
|
func safeDir(d string) string {
|
|
var parts []string
|
|
for _, p := range strings.Split(strings.ReplaceAll(d, "\\", "/"), "/") {
|
|
if s := slugify(p); s != "" {
|
|
parts = append(parts, s)
|
|
}
|
|
}
|
|
return strings.Join(parts, "/")
|
|
}
|
|
|
|
// writeInSite writes rel inside the site, never outside it.
|
|
func writeInSite(siteDir, rel string, data []byte) error {
|
|
root, err := os.OpenRoot(siteDir)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
defer root.Close()
|
|
name := filepath.FromSlash(path.Clean("/" + rel))[1:]
|
|
if err := root.MkdirAll(filepath.Dir(name), 0o755); err != nil {
|
|
return err
|
|
}
|
|
return root.WriteFile(name, data, 0o644)
|
|
}
|
|
|
|
func slugify(s string) string {
|
|
return strings.Trim(slugRe.ReplaceAllString(strings.ToLower(s), "-"), "-")
|
|
}
|
|
|
|
// fetch downloads a picture over http(s), or reads it from beside the page
|
|
// being imported (file://, only inside localRoot), at most media.MaxBytes.
|
|
func fetch(u, localRoot string) ([]byte, error) {
|
|
p, err := url.Parse(u)
|
|
if err == nil && p.Scheme == "file" {
|
|
abs, err1 := filepath.Abs(filepath.FromSlash(p.Path))
|
|
root, err2 := filepath.Abs(localRoot)
|
|
if localRoot == "" || err1 != nil || err2 != nil || !strings.HasPrefix(abs, root+string(filepath.Separator)) {
|
|
return nil, fmt.Errorf("a local file outside the folder being imported")
|
|
}
|
|
f, err := os.Open(abs)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
defer f.Close()
|
|
return io.ReadAll(io.LimitReader(f, media.MaxBytes+1))
|
|
}
|
|
if err != nil || (p.Scheme != "https" && p.Scheme != "http") {
|
|
return nil, fmt.Errorf("not a web address")
|
|
}
|
|
res, err := (&http.Client{Timeout: 30 * time.Second}).Get(u)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
defer res.Body.Close()
|
|
if res.StatusCode != http.StatusOK {
|
|
return nil, fmt.Errorf("%s", res.Status)
|
|
}
|
|
return io.ReadAll(io.LimitReader(res.Body, media.MaxBytes+1))
|
|
}
|
|
|
|
// WriteReport writes what happened as Markdown, for the person who ran it.
|
|
func (r *Report) Markdown(source string) string {
|
|
var b strings.Builder
|
|
fmt.Fprintf(&b, "# Import from %s\n\n", source)
|
|
fmt.Fprintf(&b, "%d page(s) written, %d skipped (already there), %d picture(s) brought in, %d redirect(s) added.\n\n", len(r.Written), len(r.Skipped), r.Pictures, len(r.Redirects))
|
|
if len(r.KeptHTML) > 0 {
|
|
b.WriteString("## Kept as HTML\n\nMarkdown would have lost something in these, so their HTML was kept (`format: html`). It's sanitized when the site is built, so embeds and scripts are dropped; check how they look.\n\n")
|
|
for _, k := range sortedKeys(r.KeptHTML) {
|
|
fmt.Fprintf(&b, "- `%s`: %s\n", k, strings.Join(r.KeptHTML[k], ", "))
|
|
}
|
|
b.WriteString("\n")
|
|
}
|
|
if len(r.Notes) > 0 {
|
|
b.WriteString("## To look at\n\n")
|
|
for _, k := range sortedKeys(r.Notes) {
|
|
fmt.Fprintf(&b, "- `%s`: %s\n", k, strings.Join(r.Notes[k], "; "))
|
|
}
|
|
b.WriteString("\n")
|
|
}
|
|
if len(r.Failed) > 0 {
|
|
b.WriteString("## Pictures that didn't come across\n\nThese still point at the old site.\n\n")
|
|
keys := make([]string, 0, len(r.Failed))
|
|
for k := range r.Failed {
|
|
keys = append(keys, k)
|
|
}
|
|
sort.Strings(keys)
|
|
for _, k := range keys {
|
|
fmt.Fprintf(&b, "- %s: %s\n", k, r.Failed[k])
|
|
}
|
|
b.WriteString("\n")
|
|
}
|
|
if len(r.Skipped) > 0 {
|
|
b.WriteString("## Skipped\n\nA page was already there; run again with -overwrite to replace them.\n\n")
|
|
for _, s := range r.Skipped {
|
|
fmt.Fprintf(&b, "- `%s`\n", s)
|
|
}
|
|
}
|
|
return b.String()
|
|
}
|
|
|
|
func sortedKeys(m map[string][]string) []string {
|
|
out := make([]string, 0, len(m))
|
|
for k := range m {
|
|
out = append(out, k)
|
|
}
|
|
sort.Strings(out)
|
|
return out
|
|
}
|