436 lines
13 KiB
Go
436 lines
13 KiB
Go
package importer
|
|
|
|
import (
|
|
"encoding/json"
|
|
"fmt"
|
|
"io"
|
|
"io/fs"
|
|
"os"
|
|
"path"
|
|
"path/filepath"
|
|
"regexp"
|
|
"strconv"
|
|
"strings"
|
|
"time"
|
|
|
|
"gopkg.in/yaml.v3"
|
|
)
|
|
|
|
// Ghost, Hugo and Jekyll. Hugo and Jekyll are already Markdown with front
|
|
// matter, so it's mostly mapping fields and working out the old addresses;
|
|
// Ghost's export carries each post's HTML.
|
|
|
|
// FromGhost reads a Ghost export (Settings → Labs → Export).
|
|
func FromGhost(r io.Reader, siteURL, posts string) ([]Item, error) {
|
|
if posts == "" {
|
|
posts = "articles"
|
|
}
|
|
var doc struct {
|
|
DB []struct {
|
|
Data struct {
|
|
Posts []struct {
|
|
ID, Title, Slug, Status, Type string
|
|
HTML *string `json:"html"`
|
|
PublishedAt string `json:"published_at"`
|
|
UpdatedAt string `json:"updated_at"`
|
|
CustomExcerpt *string `json:"custom_excerpt"`
|
|
FeatureImage *string `json:"feature_image"`
|
|
FeatureImageAlt *string `json:"feature_image_alt"`
|
|
} `json:"posts"`
|
|
Tags []struct {
|
|
ID, Name string
|
|
} `json:"tags"`
|
|
PostsTags []struct {
|
|
PostID string `json:"post_id"`
|
|
TagID string `json:"tag_id"`
|
|
} `json:"posts_tags"`
|
|
Users []struct {
|
|
ID, Name string
|
|
} `json:"users"`
|
|
PostsAuthors []struct {
|
|
PostID string `json:"post_id"`
|
|
AuthorID string `json:"author_id"`
|
|
} `json:"posts_authors"`
|
|
} `json:"data"`
|
|
} `json:"db"`
|
|
}
|
|
if err := json.NewDecoder(r).Decode(&doc); err != nil || len(doc.DB) == 0 {
|
|
return nil, fmt.Errorf("not a Ghost export: %v", err)
|
|
}
|
|
d := doc.DB[0].Data
|
|
tags := map[string]string{}
|
|
for _, t := range d.Tags {
|
|
tags[t.ID] = t.Name
|
|
}
|
|
users := map[string]string{}
|
|
for _, u := range d.Users {
|
|
users[u.ID] = u.Name
|
|
}
|
|
postTags := map[string][]string{}
|
|
for _, pt := range d.PostsTags {
|
|
if n := tags[pt.TagID]; n != "" && !strings.HasPrefix(n, "#") { // #internal tags are Ghost's own
|
|
postTags[pt.PostID] = append(postTags[pt.PostID], n)
|
|
}
|
|
}
|
|
author := map[string]string{}
|
|
for _, pa := range d.PostsAuthors {
|
|
if _, ok := author[pa.PostID]; !ok {
|
|
author[pa.PostID] = users[pa.AuthorID]
|
|
}
|
|
}
|
|
str := func(p *string) string {
|
|
if p == nil {
|
|
return ""
|
|
}
|
|
return strings.ReplaceAll(*p, "__GHOST_URL__", strings.TrimRight(siteURL, "/"))
|
|
}
|
|
var items []Item
|
|
for _, p := range d.Posts {
|
|
it := Item{Kind: p.Type, Title: p.Title, Slug: p.Slug, Draft: p.Status != "published", HTML: str(p.HTML), Summary: str(p.CustomExcerpt),
|
|
Image: str(p.FeatureImage), ImageAlt: str(p.FeatureImageAlt), Tags: postTags[p.ID], Author: author[p.ID], OldURL: "/" + p.Slug + "/"}
|
|
it.Date, _ = time.Parse(time.RFC3339, strings.Replace(p.PublishedAt, ".000Z", "Z", 1))
|
|
it.Updated, _ = time.Parse(time.RFC3339, strings.Replace(p.UpdatedAt, ".000Z", "Z", 1))
|
|
if p.Type == "post" {
|
|
it.Dir = posts
|
|
}
|
|
if p.HTML == nil {
|
|
it.Notes = append(it.Notes, "no HTML in the export (exported from an editor format only): the body is empty")
|
|
}
|
|
items = append(items, it)
|
|
}
|
|
return items, nil
|
|
}
|
|
|
|
// splitFront separates front matter (YAML ---, TOML +++ or JSON {}) from a body.
|
|
func splitFront(raw string) (map[string]any, string, error) {
|
|
raw = strings.TrimPrefix(strings.ReplaceAll(raw, "\r\n", "\n"), "\ufeff")
|
|
fm := map[string]any{}
|
|
switch {
|
|
case strings.HasPrefix(raw, "---\n"):
|
|
end := strings.Index(raw[4:], "\n---")
|
|
if end < 0 {
|
|
return fm, raw, nil
|
|
}
|
|
err := yaml.Unmarshal([]byte(raw[4:4+end]), &fm)
|
|
return fm, strings.TrimPrefix(raw[4+end+4:], "\n"), err
|
|
case strings.HasPrefix(raw, "+++\n"):
|
|
end := strings.Index(raw[4:], "\n+++")
|
|
if end < 0 {
|
|
return fm, raw, nil
|
|
}
|
|
return parseTOML(raw[4 : 4+end]), strings.TrimPrefix(raw[4+end+4:], "\n"), nil
|
|
case strings.HasPrefix(raw, "{"):
|
|
dec := json.NewDecoder(strings.NewReader(raw))
|
|
if err := dec.Decode(&fm); err != nil {
|
|
return fm, raw, err
|
|
}
|
|
rest, _ := io.ReadAll(dec.Buffered())
|
|
return fm, strings.TrimLeft(string(rest)+raw[dec.InputOffset():], "\n"), nil
|
|
}
|
|
return fm, raw, nil
|
|
}
|
|
|
|
// parseTOML reads the flat part of TOML front matter: strings, numbers,
|
|
// booleans, dates and arrays of them. Tables ([params]) are skipped.
|
|
func parseTOML(s string) map[string]any {
|
|
out := map[string]any{}
|
|
inTable := false
|
|
for _, line := range strings.Split(s, "\n") {
|
|
line = strings.TrimSpace(line)
|
|
if line == "" || strings.HasPrefix(line, "#") {
|
|
continue
|
|
}
|
|
if strings.HasPrefix(line, "[") {
|
|
inTable = true
|
|
continue
|
|
}
|
|
if inTable {
|
|
continue
|
|
}
|
|
k, v, ok := strings.Cut(line, "=")
|
|
if !ok {
|
|
continue
|
|
}
|
|
out[strings.Trim(strings.TrimSpace(k), `"`)] = tomlValue(strings.TrimSpace(v))
|
|
}
|
|
return out
|
|
}
|
|
|
|
func tomlValue(v string) any {
|
|
switch {
|
|
case strings.HasPrefix(v, "["):
|
|
var list []any
|
|
for _, part := range regexp.MustCompile(`"[^"]*"|'[^']*'|[^,\[\]\s]+`).FindAllString(v, -1) {
|
|
list = append(list, tomlValue(part))
|
|
}
|
|
return list
|
|
case strings.HasPrefix(v, `"`) || strings.HasPrefix(v, `'`):
|
|
if u, err := strconv.Unquote(`"` + strings.Trim(v, `"'`) + `"`); err == nil {
|
|
return u
|
|
}
|
|
return strings.Trim(v, `"'`)
|
|
case v == "true" || v == "false":
|
|
return v == "true"
|
|
}
|
|
if n, err := strconv.Atoi(v); err == nil {
|
|
return n
|
|
}
|
|
if t, err := time.Parse(time.RFC3339, v); err == nil {
|
|
return t
|
|
}
|
|
if t, err := time.Parse("2006-01-02", v); err == nil {
|
|
return t
|
|
}
|
|
return v
|
|
}
|
|
|
|
func fmString(fm map[string]any, keys ...string) string {
|
|
for _, k := range keys {
|
|
switch v := fm[k].(type) {
|
|
case string:
|
|
if v != "" {
|
|
return v
|
|
}
|
|
case time.Time:
|
|
return v.Format(time.RFC3339)
|
|
}
|
|
}
|
|
return ""
|
|
}
|
|
|
|
func fmTime(fm map[string]any, keys ...string) time.Time {
|
|
for _, k := range keys {
|
|
switch v := fm[k].(type) {
|
|
case time.Time:
|
|
return v
|
|
case string:
|
|
for _, layout := range []string{time.RFC3339, "2006-01-02 15:04:05 -0700", "2006-01-02 15:04:05", "2006-01-02T15:04:05", "2006-01-02"} {
|
|
if t, err := time.Parse(layout, v); err == nil {
|
|
return t
|
|
}
|
|
}
|
|
}
|
|
}
|
|
return time.Time{}
|
|
}
|
|
|
|
func fmList(fm map[string]any, keys ...string) []string {
|
|
var out []string
|
|
for _, k := range keys {
|
|
switch v := fm[k].(type) {
|
|
case []any:
|
|
for _, x := range v {
|
|
out = append(out, fmt.Sprint(x))
|
|
}
|
|
case string:
|
|
if v != "" {
|
|
out = append(out, strings.Fields(strings.ReplaceAll(v, ",", " "))...)
|
|
}
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
|
|
var (
|
|
hugoShortcode = regexp.MustCompile(`\{\{[<%]\s*/?\s*([\w-]+)`)
|
|
liquidTag = regexp.MustCompile(`\{%\s*(\w+)|\{\{\s*[\w.]+`)
|
|
relImage = regexp.MustCompile(`(!\[[^\]]*\]\()([^)\s:]+)(\s*(?:"[^"]*")?\))`)
|
|
)
|
|
|
|
// localImages points relative image links at the file beside the page, so
|
|
// the writer can bring them in.
|
|
func localImages(body, dir string) string {
|
|
return relImage.ReplaceAllStringFunc(body, func(m string) string {
|
|
sub := relImage.FindStringSubmatch(m)
|
|
if strings.HasPrefix(sub[2], "/") || strings.HasPrefix(sub[2], "#") {
|
|
return m
|
|
}
|
|
return sub[1] + "file://" + filepath.ToSlash(filepath.Join(dir, sub[2])) + sub[3]
|
|
})
|
|
}
|
|
|
|
// FromHugo reads a Hugo site's content/ folder. rename maps a section to a
|
|
// collection here ("posts" to "articles"); its old addresses then redirect.
|
|
func FromHugo(siteDir string, rename map[string]string) ([]Item, error) {
|
|
root := filepath.Join(siteDir, "content")
|
|
var items []Item
|
|
err := filepath.WalkDir(root, func(p string, d fs.DirEntry, err error) error {
|
|
if err != nil || d.IsDir() || !strings.HasSuffix(p, ".md") {
|
|
return err
|
|
}
|
|
raw, err := os.ReadFile(p)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
fm, body, err := splitFront(string(raw))
|
|
if err != nil {
|
|
return fmt.Errorf("%s: %w", p, err)
|
|
}
|
|
rel, _ := filepath.Rel(root, p)
|
|
rel = filepath.ToSlash(rel)
|
|
dir, name := path.Split(rel)
|
|
dir = strings.TrimSuffix(dir, "/")
|
|
slug := strings.TrimSuffix(name, ".md")
|
|
if slug == "index" { // a page bundle: the folder is the page
|
|
dir, slug = path.Split(dir)
|
|
dir = strings.TrimSuffix(dir, "/")
|
|
}
|
|
if slug == "_index" {
|
|
return nil // a section's own page: the collection list here makes its own
|
|
}
|
|
if s := fmString(fm, "slug"); s != "" {
|
|
slug = s
|
|
}
|
|
section := strings.SplitN(dir, "/", 2)[0]
|
|
oldURL := "/" + path.Join(dir, slug) + "/"
|
|
if u := fmString(fm, "url"); u != "" {
|
|
oldURL = u
|
|
}
|
|
newDir := dir
|
|
if to, ok := rename[section]; ok {
|
|
newDir = to + strings.TrimPrefix(dir, section)
|
|
}
|
|
it := Item{Kind: "page", Title: fmString(fm, "title"), Slug: slug, Date: fmTime(fm, "date", "publishDate"), Updated: fmTime(fm, "lastmod"),
|
|
Summary: fmString(fm, "description", "summary"), Tags: append(fmList(fm, "tags"), fmList(fm, "categories")...),
|
|
Markdown: localImages(body, filepath.Dir(p)), OldURL: oldURL, Aliases: fmList(fm, "aliases"), Dir: newDir}
|
|
if dir != "" {
|
|
it.Kind = "post"
|
|
}
|
|
if b, ok := fm["draft"].(bool); ok && b {
|
|
it.Draft = true
|
|
}
|
|
if imgs := fmList(fm, "images"); len(imgs) > 0 {
|
|
it.Image = imgs[0]
|
|
}
|
|
for _, m := range hugoShortcode.FindAllStringSubmatch(body, -1) {
|
|
it.Notes = append(it.Notes, "the Hugo shortcode "+m[1]+", which only Hugo could run; it's left as text")
|
|
}
|
|
it.Notes = dedupe(it.Notes)
|
|
items = append(items, it)
|
|
return nil
|
|
})
|
|
return items, err
|
|
}
|
|
|
|
// FromJekyll reads a Jekyll site: _posts/, _drafts/ and its pages.
|
|
func FromJekyll(siteDir, posts string) ([]Item, error) {
|
|
if posts == "" {
|
|
posts = "articles"
|
|
}
|
|
permalink := "date"
|
|
var conf map[string]any
|
|
if raw, err := os.ReadFile(filepath.Join(siteDir, "_config.yml")); err == nil {
|
|
_ = yaml.Unmarshal(raw, &conf)
|
|
if p, ok := conf["permalink"].(string); ok && p != "" {
|
|
permalink = p
|
|
}
|
|
}
|
|
pattern := map[string]string{
|
|
"date": "/:categories/:year/:month/:day/:title:output_ext",
|
|
"pretty": "/:categories/:year/:month/:day/:title/",
|
|
"ordinal": "/:categories/:year/:y_day/:title:output_ext",
|
|
"none": "/:categories/:title:output_ext",
|
|
}[permalink]
|
|
if pattern == "" {
|
|
pattern = permalink
|
|
}
|
|
dated := regexp.MustCompile(`^(\d{4})-(\d{2})-(\d{2})-(.+)\.(md|markdown|html)$`)
|
|
var items []Item
|
|
read := func(p string, draft bool) error {
|
|
raw, err := os.ReadFile(p)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
fm, body, err := splitFront(string(raw))
|
|
if err != nil {
|
|
return fmt.Errorf("%s: %w", p, err)
|
|
}
|
|
name := filepath.Base(p)
|
|
it := Item{Kind: "post", Dir: posts, Draft: draft, Title: fmString(fm, "title"), Summary: fmString(fm, "excerpt", "description"),
|
|
Tags: append(fmList(fm, "tags"), fmList(fm, "categories", "category")...), Aliases: fmList(fm, "redirect_from"), Author: fmString(fm, "author")}
|
|
if m := dated.FindStringSubmatch(name); m != nil {
|
|
it.Slug = m[4]
|
|
it.Date, _ = time.Parse("2006-01-02", m[1]+"-"+m[2]+"-"+m[3])
|
|
} else {
|
|
it.Slug = strings.TrimSuffix(name, filepath.Ext(name))
|
|
}
|
|
if t := fmTime(fm, "date"); !t.IsZero() {
|
|
it.Date = t
|
|
}
|
|
if b, ok := fm["published"].(bool); ok && !b {
|
|
it.Draft = true
|
|
}
|
|
if strings.HasSuffix(name, ".html") {
|
|
it.HTML = body
|
|
} else {
|
|
it.Markdown = localImages(body, filepath.Dir(p))
|
|
}
|
|
cats := strings.Join(fmList(fm, "categories", "category"), "/")
|
|
old := fmString(fm, "permalink")
|
|
if old == "" && !draft {
|
|
r := strings.NewReplacer(":categories", cats, ":year", it.Date.Format("2006"), ":month", it.Date.Format("01"), ":day", it.Date.Format("02"),
|
|
":y_day", strconv.Itoa(it.Date.YearDay()), ":title", it.Slug, ":slug", it.Slug, ":output_ext", ".html")
|
|
old = regexp.MustCompile(`/+`).ReplaceAllString(r.Replace(pattern), "/")
|
|
}
|
|
it.OldURL = old
|
|
for _, m := range liquidTag.FindAllStringSubmatch(body, -1) {
|
|
tag := strings.TrimSpace(strings.Trim(m[0], "{%"))
|
|
it.Notes = append(it.Notes, "Liquid ("+tag+"), which only Jekyll could run; it's left as text")
|
|
}
|
|
it.Notes = dedupe(it.Notes)
|
|
items = append(items, it)
|
|
return nil
|
|
}
|
|
for _, folder := range []struct {
|
|
dir string
|
|
draft bool
|
|
}{{"_posts", false}, {"_drafts", true}} {
|
|
entries, _ := os.ReadDir(filepath.Join(siteDir, folder.dir))
|
|
for _, e := range entries {
|
|
if !e.IsDir() && regexp.MustCompile(`\.(md|markdown|html)$`).MatchString(e.Name()) {
|
|
if err := read(filepath.Join(siteDir, folder.dir, e.Name()), folder.draft); err != nil {
|
|
return nil, err
|
|
}
|
|
}
|
|
}
|
|
}
|
|
// Pages: Markdown files outside the _folders.
|
|
err := filepath.WalkDir(siteDir, func(p string, d fs.DirEntry, err error) error {
|
|
if err != nil {
|
|
return err
|
|
}
|
|
rel, _ := filepath.Rel(siteDir, p)
|
|
if d.IsDir() {
|
|
if rel != "." && (strings.HasPrefix(d.Name(), "_") || strings.HasPrefix(d.Name(), ".") || d.Name() == "assets" || d.Name() == "vendor" || d.Name() == "node_modules") {
|
|
return fs.SkipDir
|
|
}
|
|
return nil
|
|
}
|
|
if !strings.HasSuffix(p, ".md") || strings.EqualFold(d.Name(), "README.md") {
|
|
return nil
|
|
}
|
|
raw, err := os.ReadFile(p)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
fm, body, err := splitFront(string(raw))
|
|
if err != nil || len(fm) == 0 {
|
|
return nil // without front matter Jekyll doesn't publish it
|
|
}
|
|
dir, name := path.Split(filepath.ToSlash(rel))
|
|
slug := strings.TrimSuffix(name, ".md")
|
|
it := Item{Kind: "page", Title: fmString(fm, "title"), Slug: slug, Dir: strings.TrimSuffix(dir, "/"), Markdown: localImages(body, filepath.Dir(p)),
|
|
Summary: fmString(fm, "description"), Aliases: fmList(fm, "redirect_from")}
|
|
it.OldURL = fmString(fm, "permalink")
|
|
if it.OldURL == "" {
|
|
it.OldURL = "/" + path.Join(it.Dir, slug) + ".html"
|
|
}
|
|
if slug == "index" && it.Dir == "" {
|
|
return nil // the home page: the site's own front page stays
|
|
}
|
|
items = append(items, it)
|
|
return nil
|
|
})
|
|
return items, err
|
|
}
|