242 lines
12 KiB
Go
242 lines
12 KiB
Go
package importer
|
|
|
|
import (
|
|
"bytes"
|
|
"encoding/json"
|
|
"image"
|
|
"image/color"
|
|
"image/jpeg"
|
|
"io/fs"
|
|
"net/http"
|
|
"net/http/httptest"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"testing"
|
|
|
|
"git.coffeylabs.org/coffey-labs/hotdog-cms/internal/build"
|
|
"git.coffeylabs.org/coffey-labs/hotdog-cms/internal/check"
|
|
"git.coffeylabs.org/coffey-labs/hotdog-cms/starter"
|
|
)
|
|
|
|
func starterSite(t *testing.T) string {
|
|
t.Helper()
|
|
dir := t.TempDir()
|
|
err := fs.WalkDir(starter.Files, "site", func(p string, d fs.DirEntry, err error) error {
|
|
if err != nil {
|
|
return err
|
|
}
|
|
target := filepath.Join(dir, strings.TrimPrefix(p, "site"))
|
|
if d.IsDir() {
|
|
return os.MkdirAll(target, 0o755)
|
|
}
|
|
data, _ := starter.Files.ReadFile(p)
|
|
return os.WriteFile(target, []byte(strings.ReplaceAll(string(data), "{{SITE_NAME}}", "Test")), 0o644)
|
|
})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
return dir
|
|
}
|
|
|
|
func jpg() []byte {
|
|
img := image.NewRGBA(image.Rect(0, 0, 800, 500))
|
|
for i := range img.Pix {
|
|
img.Pix[i] = uint8(i)
|
|
}
|
|
img.Set(1, 1, color.RGBA{200, 10, 10, 255})
|
|
var b bytes.Buffer
|
|
jpeg.Encode(&b, img, nil)
|
|
return b.Bytes()
|
|
}
|
|
|
|
func read(t *testing.T, p string) string {
|
|
t.Helper()
|
|
b, err := os.ReadFile(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
return string(b)
|
|
}
|
|
|
|
func fakeWordPress(t *testing.T, password string) *httptest.Server {
|
|
var srv *httptest.Server
|
|
srv = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
if password != "" && strings.HasPrefix(r.URL.Path, "/wp-json/") {
|
|
if u, p, ok := r.BasicAuth(); !ok || u != "ed" || p != password {
|
|
w.WriteHeader(401)
|
|
return
|
|
}
|
|
}
|
|
img := srv.URL + "/wp-content/uploads/2026/09/dock.jpg"
|
|
post := func(id int, slug, title, content string, status string) map[string]any {
|
|
return map[string]any{"id": id, "type": "post", "slug": slug, "status": status, "link": srv.URL + "/2026/09/" + slug + "/",
|
|
"date_gmt": "2026-09-14T08:30:00", "modified_gmt": "2026-09-20T10:00:00", "parent": 0,
|
|
"title": map[string]string{"rendered": title}, "content": map[string]string{"rendered": content},
|
|
"excerpt": map[string]string{"rendered": "<p>A short summary […]</p>"},
|
|
"_embedded": map[string]any{"author": []map[string]string{{"name": "Ed Writer"}},
|
|
"wp:featuredmedia": []map[string]string{{"source_url": img, "alt_text": "The dock"}},
|
|
"wp:term": [][]map[string]string{{{"name": "Uncategorized", "taxonomy": "category"}, {"name": "Boats", "taxonomy": "category"}}, {{"name": "Harbor & sea", "taxonomy": "post_tag"}}}}}
|
|
}
|
|
switch r.URL.Path {
|
|
case "/wp-json/wp/v2/posts":
|
|
w.Header().Set("X-WP-TotalPages", "1")
|
|
json.NewEncoder(w).Encode([]any{
|
|
post(1, "hello-dock", "Hello “Dock”", `<!-- wp:paragraph --><p>Morning at the <strong>dock</strong>.</p><!-- /wp:paragraph --><figure class="wp-block-image"><img src="`+img+`" alt="Boats"><figcaption>Calm water</figcaption></figure>`, "publish"),
|
|
post(2, "with-video", "With a video", `<p>Watch:</p><figure class="wp-block-embed"><iframe src="https://www.youtube.com/embed/x"></iframe></figure>`, "publish"),
|
|
})
|
|
case "/wp-json/wp/v2/pages":
|
|
w.Header().Set("X-WP-TotalPages", "1")
|
|
json.NewEncoder(w).Encode([]any{
|
|
map[string]any{"id": 10, "type": "page", "slug": "company", "status": "publish", "link": srv.URL + "/company/", "parent": 0, "date_gmt": "2026-01-01T00:00:00",
|
|
"title": map[string]string{"rendered": "Company"}, "content": map[string]string{"rendered": "<p>Who we are.</p>"}, "excerpt": map[string]string{"rendered": ""}},
|
|
map[string]any{"id": 11, "type": "page", "slug": "team", "status": "publish", "link": srv.URL + "/company/team/", "parent": 10, "date_gmt": "2026-01-02T00:00:00",
|
|
"title": map[string]string{"rendered": "Team"}, "content": map[string]string{"rendered": "<ul><li>Ed</li><li>Jo</li></ul>"}, "excerpt": map[string]string{"rendered": ""}},
|
|
})
|
|
case "/wp-content/uploads/2026/09/dock.jpg":
|
|
w.Write(jpg())
|
|
default:
|
|
w.WriteHeader(404)
|
|
}
|
|
}))
|
|
t.Cleanup(srv.Close)
|
|
return srv
|
|
}
|
|
|
|
func TestWordPressImport(t *testing.T) {
|
|
srv := fakeWordPress(t, "app pass word")
|
|
if _, err := FromWordPress(WordPressOptions{URL: srv.URL, User: "ed", Password: "wrong"}); err == nil || !strings.Contains(err.Error(), "Application Password") {
|
|
t.Errorf("bad password: %v", err)
|
|
}
|
|
items, err := FromWordPress(WordPressOptions{URL: srv.URL, User: "ed", Password: "app pass word"})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(items) != 4 || items[0].Title != "Hello “Dock”" || items[0].Summary != "A short summary" || strings.Join(items[0].Tags, ",") != "Boats,Harbor & sea" || items[3].Dir != "company" {
|
|
t.Fatalf("items: %+v", items)
|
|
}
|
|
site := starterSite(t)
|
|
rep, err := Write(items, Options{SiteDir: site, Base: srv.URL})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
post := read(t, filepath.Join(site, "content", "articles", "hello-dock.md"))
|
|
for _, want := range []string{"title: Hello “Dock”", "date: \"2026-09-14\"", "updated: \"2026-09-20\"", "tags: [Boats, Harbor & sea]", "author: Ed Writer", "image: /media/2026/09/dock.jpg", "image_alt: The dock", "Morning at the **dock**.", "", "*Calm water*"} {
|
|
if !strings.Contains(post, want) {
|
|
t.Errorf("post missing %q:\n%s", want, post)
|
|
}
|
|
}
|
|
if !strings.Contains(read(t, filepath.Join(site, "content", "articles", "with-video.md")), "format: html") || rep.KeptHTML["content/articles/with-video.md"] == nil {
|
|
t.Error("the embed page wasn't kept as HTML")
|
|
}
|
|
if team := read(t, filepath.Join(site, "content", "company", "team.md")); !strings.Contains(team, "- Ed\n- Jo") {
|
|
t.Errorf("team:\n%s", team)
|
|
}
|
|
if _, err := os.Stat(filepath.Join(site, "static", "media", "2026", "09", "dock.webp")); err != nil || rep.Pictures != 1 {
|
|
t.Errorf("picture: %v, %d brought", err, rep.Pictures)
|
|
}
|
|
cfg := read(t, filepath.Join(site, "site.yaml"))
|
|
for _, want := range []string{" - from: /2026/09/hello-dock/\n to: /articles/hello-dock/", "markdown:", "unsafe_html: true"} {
|
|
if !strings.Contains(cfg, want) {
|
|
t.Errorf("site.yaml missing %q", want)
|
|
}
|
|
}
|
|
if strings.Contains(cfg, "from: /company/\n") {
|
|
t.Error("a redirect from an address that didn't change")
|
|
}
|
|
// The imported site builds and passes the checks.
|
|
res, err := build.Run(build.Options{SiteDir: site})
|
|
if err != nil {
|
|
t.Fatalf("build: %v", err)
|
|
}
|
|
ck, _ := check.Run(site, res.Out, res.Site.Config)
|
|
if ck.Errors() > 0 {
|
|
t.Errorf("check errors: %v", ck.Problems)
|
|
}
|
|
if md := rep.Markdown(srv.URL); !strings.Contains(md, "with-video.md`: ") || !strings.Contains(md, "an embed") {
|
|
t.Errorf("report:\n%s", md)
|
|
}
|
|
// Importing again skips what's there.
|
|
rep, _ = Write(items, Options{SiteDir: site, Base: srv.URL})
|
|
if len(rep.Skipped) != 4 || len(rep.Written) != 0 {
|
|
t.Errorf("second import: %d skipped, %d written", len(rep.Skipped), len(rep.Written))
|
|
}
|
|
}
|
|
|
|
func TestWXR(t *testing.T) {
|
|
x := `<?xml version="1.0"?><rss xmlns:content="http://purl.org/rss/1.0/modules/content/" xmlns:excerpt="http://wordpress.org/export/1.2/excerpt/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:wp="http://wordpress.org/export/1.2/"><channel><link>https://old.example</link>
|
|
<item><title>Classic post</title><link>https://old.example/2020/05/classic-post/</link><dc:creator>ed</dc:creator>
|
|
<content:encoded><![CDATA[First paragraph
|
|
still the first.
|
|
|
|
[caption id="a" align="alignnone"]<img src="file:///etc/passwd" alt="x"> A caption[/caption]
|
|
|
|
[gallery ids="1,2"]]]></content:encoded><excerpt:encoded><![CDATA[]]></excerpt:encoded>
|
|
<wp:post_id>5</wp:post_id><wp:post_date_gmt>2020-05-01 10:00:00</wp:post_date_gmt><wp:post_name>classic-post</wp:post_name><wp:status>publish</wp:status><wp:post_parent>0</wp:post_parent><wp:post_type>post</wp:post_type>
|
|
<category domain="post_tag" nicename="old"><![CDATA[Old]]></category></item>
|
|
<item><title>Trashed</title><wp:post_id>6</wp:post_id><wp:status>trash</wp:status><wp:post_type>post</wp:post_type></item>
|
|
</channel></rss>`
|
|
items, link, err := FromWXR(strings.NewReader(x), "")
|
|
if err != nil || len(items) != 1 || link != "https://old.example" {
|
|
t.Fatalf("wxr: %v %d %s", err, len(items), link)
|
|
}
|
|
it := items[0]
|
|
if !strings.Contains(it.HTML, "<p>First paragraph<br>still the first.</p>") || !strings.Contains(it.HTML, "<figcaption>A caption</figcaption>") || len(it.Notes) != 1 || !strings.Contains(it.Notes[0], "[gallery]") {
|
|
t.Fatalf("wxr item: %q %v", it.HTML, it.Notes)
|
|
}
|
|
site := starterSite(t)
|
|
rep, err := Write(items, Options{SiteDir: site, Base: link})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if rep.Failed["file:///etc/passwd"] == "" {
|
|
t.Errorf("a file:// picture outside the import was read: %v", rep.Failed)
|
|
}
|
|
if md := read(t, filepath.Join(site, "content", "articles", "classic-post.md")); !strings.Contains(md, "First paragraph\\\nstill the first.") {
|
|
t.Errorf("classic post:\n%s", md)
|
|
}
|
|
}
|
|
|
|
func TestHugoJekyllGhost(t *testing.T) {
|
|
// Hugo: TOML front matter, aliases, a page bundle with its picture, a shortcode.
|
|
hugo := t.TempDir()
|
|
os.MkdirAll(filepath.Join(hugo, "content", "posts", "trip"), 0o755)
|
|
os.WriteFile(filepath.Join(hugo, "content", "posts", "trip", "index.md"), []byte("+++\ntitle = \"A trip\"\ndate = 2025-06-01\ntags = [\"travel\", \"boats\"]\naliases = [\"/old/trip/\"]\n[params]\nhidden = true\n+++\n\n\n\n{{< youtube abc >}}\n"), 0o644)
|
|
os.WriteFile(filepath.Join(hugo, "content", "posts", "trip", "boat.jpg"), jpg(), 0o644)
|
|
os.WriteFile(filepath.Join(hugo, "content", "about.md"), []byte("---\ntitle: About\n---\nHi.\n"), 0o644)
|
|
items, err := FromHugo(hugo, map[string]string{"posts": "articles"})
|
|
if err != nil || len(items) != 2 {
|
|
t.Fatalf("hugo: %v %d", err, len(items))
|
|
}
|
|
site := starterSite(t)
|
|
rep, err := Write(items, Options{SiteDir: site, LocalRoot: hugo, Overwrite: true})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
trip := read(t, filepath.Join(site, "content", "articles", "trip.md"))
|
|
if !strings.Contains(trip, "") || !strings.Contains(trip, "tags: [travel, boats]") || rep.Notes["content/articles/trip.md"] == nil {
|
|
t.Errorf("hugo trip:\n%s %v", trip, rep.Notes)
|
|
}
|
|
cfg := read(t, filepath.Join(site, "site.yaml"))
|
|
if !strings.Contains(cfg, "from: /old/trip/\n to: /articles/trip/") || !strings.Contains(cfg, "from: /posts/trip/\n to: /articles/trip/") {
|
|
t.Errorf("hugo redirects:\n%s", cfg)
|
|
}
|
|
|
|
// Jekyll: the old permalink from the date, redirect_from, Liquid.
|
|
jk := t.TempDir()
|
|
os.MkdirAll(filepath.Join(jk, "_posts"), 0o755)
|
|
os.WriteFile(filepath.Join(jk, "_config.yml"), []byte("permalink: date\n"), 0o644)
|
|
os.WriteFile(filepath.Join(jk, "_posts", "2024-03-05-spring.md"), []byte("---\ntitle: Spring\ncategories: [news]\nredirect_from: [/spring/]\n---\nIt's {{ site.title }} time.\n"), 0o644)
|
|
items, err = FromJekyll(jk, "")
|
|
if err != nil || len(items) != 1 || items[0].OldURL != "/news/2024/03/05/spring.html" || items[0].Date.Format("2006-01-02") != "2024-03-05" || len(items[0].Notes) == 0 {
|
|
t.Fatalf("jekyll: %v %+v", err, items)
|
|
}
|
|
|
|
// Ghost: placeholders and tags.
|
|
g := `{"db":[{"data":{"posts":[{"id":"p1","title":"Ghostly","slug":"ghostly","status":"published","type":"post","html":"<p>Hi <img src=\"__GHOST_URL__/content/images/a.jpg\" alt=\"a\"></p>","published_at":"2025-01-02T03:04:05.000Z","updated_at":"2025-01-03T03:04:05.000Z","custom_excerpt":"Boo"}],"tags":[{"id":"t1","name":"spooky"},{"id":"t2","name":"#hidden"}],"posts_tags":[{"post_id":"p1","tag_id":"t1"},{"post_id":"p1","tag_id":"t2"}],"users":[{"id":"u1","name":"Casper"}],"posts_authors":[{"post_id":"p1","author_id":"u1"}]}}]}`
|
|
items, err = FromGhost(strings.NewReader(g), "https://old.ghost", "")
|
|
if err != nil || len(items) != 1 || !strings.Contains(items[0].HTML, "https://old.ghost/content/images/a.jpg") || strings.Join(items[0].Tags, ",") != "spooky" || items[0].Author != "Casper" || items[0].Dir != "articles" {
|
|
t.Fatalf("ghost: %v %+v", err, items)
|
|
}
|
|
}
|