Files
petal/internal/docs/export_test.go
T
prosolis 69bf3ffde1 Close the door the edge gate used to hold
A security review of the whole repo. The queries were already scoped, the
OIDC flow already did state and nonce and PKCE, the session tokens were
already stored as hashes. What it found was mostly the seam between the
code and the deployment — and one place where the deployment quietly
undid the code.

The one that matters: with any AUTHENTIK_* variable missing, Petal fell
back to resolving every request to the single `local` user. That is right
on a laptop and a catastrophe on a public host, and Phase 16 removed the
Traefik basic-auth gate that used to stand behind the mistake. A typo in
the client secret would have served her journals to the open internet and
said so only in a log line nobody reads. It now refuses to start, guarded
by default for any BASE_URL that isn't loopback.

Then the one that would have been fixed and wasn't: stored images now
serve under `default-src 'none'; sandbox`, so an SVG pasted into a
document can't run as a page on Petal's own origin. Traefik's
customresponseheaders *overwrites*, so the CSP declared in the compose
labels would have silently replaced that per-route policy in production.
The whole header block moved into the binary, where a route can tighten
its own and a test can prove it; only HSTS stays at the edge, where TLS
actually terminates.

The rest, smaller:

  - PETAL_ALLOWED_SUBS empty means everyone authentik authenticates, and
    authentik here fronts half a dozen applications. Still legal, now
    said out loud every boot, and set in both env examples.
  - LLM failures relayed err.Error() to the browser, which carries the
    address of the inference box on the far side of the VPN. Logged
    instead; the client only ever rendered "the helper is resting".
  - Exports scheme-check their links. Escaping makes a URL safe to sit
    in an attribute and says nothing about following it, and an export
    is the one artifact here meant to leave. Writing the test found the
    markdown image src, which I'd missed reading it.
  - The draft rescue is namespaced per account and cleared on sign-out.
    Everything else in localStorage is a preference; this is her unsaved
    writing, sitting in a profile two people share.
  - /auth/logout is POST-only. With SameSite=Lax a GET route lets any
    page on the internet sign her out mid-draft.
  - Image uploads get a per-account allowance and the TTS cache a size
    cap. Both share the encrypted volume the database is on, and a full
    disk is SQLite failing to write, not a feature degrading.
  - The session cookie takes the __Host- prefix over https, so nothing
    else under parodia.dev can plant one. Old cookies still resolve;
    nobody is signed out to get there.
  - npm audit: linkify-it and postcss.

Verified: go build, go vet, the full Go suite, tsc, 195 frontend tests,
npm audit clean. The startup guard and both CSPs checked against a
running server rather than only asserted.

Claude-Session: https://claude.ai/code/session_016y6gyuHkQXPiEuW8RGQyua
2026-07-27 18:24:47 -07:00

318 lines
11 KiB
Go

package docs
import (
"archive/zip"
"bytes"
"encoding/json"
"io"
"net/http"
"strings"
"testing"
"gitea.parodia.dev/drwily/petal/internal/db"
)
// richDocJSON is a Tiptap document exercising headings, marks, and a list —
// including CJK text, which every format must carry through intact.
const richDocJSON = `{"type":"doc","content":[` +
`{"type":"heading","attrs":{"level":2},"content":[{"type":"text","text":"My Section"}]},` +
`{"type":"paragraph","content":[{"type":"text","text":"Hello "},{"type":"text","marks":[{"type":"bold"}],"text":"world"},{"type":"text","text":" 你好"}]},` +
`{"type":"bulletList","content":[` +
`{"type":"listItem","content":[{"type":"paragraph","content":[{"type":"text","text":"first"}]}]},` +
`{"type":"listItem","content":[{"type":"paragraph","content":[{"type":"text","text":"second"}]}]}` +
`]}` +
`]}`
func seedRichDoc(t *testing.T, srv http.Handler) string {
t.Helper()
id := newDoc(t, srv)
body := `{"title":"日记 Diary","content":` + jsonString(richDocJSON) + `,"content_text":"My Section\nHello world 你好\nfirst\nsecond","word_count":6}`
if rec := do(t, srv, http.MethodPut, "/"+id, body); rec.Code != http.StatusOK {
t.Fatalf("seed rich doc: code=%d body=%s", rec.Code, rec.Body)
}
return id
}
// jsonString quotes a string as a JSON string literal (so the Tiptap JSON can be
// embedded as the "content" field value).
func jsonString(s string) string {
b, _ := json.Marshal(s)
return string(b)
}
func TestExportMarkdown(t *testing.T) {
srv := newTestServer(t)
id := seedRichDoc(t, srv)
rec := do(t, srv, http.MethodGet, "/"+id+"/export?format=md", "")
if rec.Code != http.StatusOK {
t.Fatalf("export md: code=%d body=%s", rec.Code, rec.Body)
}
if ct := rec.Header().Get("Content-Type"); !strings.HasPrefix(ct, "text/markdown") {
t.Fatalf("unexpected content-type: %q", ct)
}
out := rec.Body.String()
for _, want := range []string{"## My Section", "**world**", "你好", "- first", "- second"} {
if !strings.Contains(out, want) {
t.Fatalf("markdown missing %q in:\n%s", want, out)
}
}
// CJK filename must survive in the RFC 5987 form.
if cd := rec.Header().Get("Content-Disposition"); !strings.Contains(cd, "filename*=UTF-8''") {
t.Fatalf("expected RFC 5987 filename, got %q", cd)
}
}
func TestExportAll(t *testing.T) {
srv := newTestServer(t)
// Two docs, one with a duplicate title to exercise filename de-duplication.
seedRichDoc(t, srv)
id2 := newDoc(t, srv)
if rec := do(t, srv, http.MethodPut, "/"+id2, `{"title":"日记 Diary","content":`+jsonString(richDocJSON)+`,"content_text":"x","word_count":1}`); rec.Code != http.StatusOK {
t.Fatalf("seed second doc: %d %s", rec.Code, rec.Body)
}
rec := do(t, srv, http.MethodGet, "/export-all?format=md", "")
if rec.Code != http.StatusOK {
t.Fatalf("export-all: code=%d body=%s", rec.Code, rec.Body)
}
if ct := rec.Header().Get("Content-Type"); ct != "application/zip" {
t.Fatalf("unexpected content-type: %q", ct)
}
zr, err := zip.NewReader(bytes.NewReader(rec.Body.Bytes()), int64(rec.Body.Len()))
if err != nil {
t.Fatalf("zip open: %v", err)
}
if len(zr.File) != 2 {
t.Fatalf("expected 2 files in backup, got %d", len(zr.File))
}
names := map[string]bool{}
for _, f := range zr.File {
names[f.Name] = true
rc, _ := f.Open()
b, _ := io.ReadAll(rc)
rc.Close()
if !strings.Contains(string(b), "你好") {
t.Fatalf("backup entry %q missing CJK body", f.Name)
}
}
// The duplicate title must have been disambiguated, not overwritten.
if !names["日记 Diary.md"] || !names["日记 Diary (1).md"] {
t.Fatalf("expected de-duplicated filenames, got %v", names)
}
}
func TestExportHTML(t *testing.T) {
srv := newTestServer(t)
id := seedRichDoc(t, srv)
rec := do(t, srv, http.MethodGet, "/"+id+"/export?format=html", "")
out := rec.Body.String()
for _, want := range []string{"<!doctype html>", "<h2>My Section</h2>", "<strong>world</strong>", "你好", "<li>first</li>"} {
if !strings.Contains(out, want) {
t.Fatalf("html missing %q", want)
}
}
}
func TestExportPlainText(t *testing.T) {
srv := newTestServer(t)
id := seedRichDoc(t, srv)
rec := do(t, srv, http.MethodGet, "/"+id+"/export?format=txt", "")
out := rec.Body.String()
for _, want := range []string{"My Section", "Hello world 你好", "• first"} {
if !strings.Contains(out, want) {
t.Fatalf("txt missing %q in:\n%s", want, out)
}
}
}
func TestExportDocx(t *testing.T) {
srv := newTestServer(t)
id := seedRichDoc(t, srv)
rec := do(t, srv, http.MethodGet, "/"+id+"/export?format=docx", "")
if rec.Code != http.StatusOK {
t.Fatalf("export docx: code=%d", rec.Code)
}
raw := rec.Body.Bytes()
zr, err := zip.NewReader(bytes.NewReader(raw), int64(len(raw)))
if err != nil {
t.Fatalf("docx is not a valid zip: %v", err)
}
want := map[string]bool{"[Content_Types].xml": false, "word/document.xml": false}
var docXML string
for _, f := range zr.File {
if _, ok := want[f.Name]; ok {
want[f.Name] = true
}
if f.Name == "word/document.xml" {
rc, _ := f.Open()
b, _ := io.ReadAll(rc)
rc.Close()
docXML = string(b)
}
}
for name, found := range want {
if !found {
t.Fatalf("docx missing part %q", name)
}
}
for _, w := range []string{"My Section", "你好", "<w:b/>", "Heading2"} {
if !strings.Contains(docXML, w) {
t.Fatalf("document.xml missing %q", w)
}
}
}
// richDocJSON2 exercises the newer node/mark types: links, highlight, an image,
// and a table (with a header row). Each export format should carry them through.
const richDocJSON2 = `{"type":"doc","content":[` +
`{"type":"paragraph","content":[` +
`{"type":"text","marks":[{"type":"link","attrs":{"href":"https://petal.test"}}],"text":"site"},` +
`{"type":"text","text":" and "},` +
`{"type":"text","marks":[{"type":"highlight","attrs":{"color":"#FFF1A8"}}],"text":"lit"}` +
`]},` +
`{"type":"image","attrs":{"src":"/api/images/abc123.png","alt":"a cat"}},` +
`{"type":"table","content":[` +
`{"type":"tableRow","content":[` +
`{"type":"tableHeader","content":[{"type":"paragraph","content":[{"type":"text","text":"Name"}]}]},` +
`{"type":"tableHeader","content":[{"type":"paragraph","content":[{"type":"text","text":"年龄"}]}]}` +
`]},` +
`{"type":"tableRow","content":[` +
`{"type":"tableCell","content":[{"type":"paragraph","content":[{"type":"text","text":"Mei"}]}]},` +
`{"type":"tableCell","content":[{"type":"paragraph","content":[{"type":"text","text":"30"}]}]}` +
`]}` +
`]}` +
`]}`
func seedRichDoc2(t *testing.T, srv http.Handler) string {
t.Helper()
id := newDoc(t, srv)
body := `{"title":"Rich2","content":` + jsonString(richDocJSON2) + `,"content_text":"site and lit\nName 年龄 Mei 30","word_count":6}`
if rec := do(t, srv, http.MethodPut, "/"+id, body); rec.Code != http.StatusOK {
t.Fatalf("seed rich doc 2: code=%d body=%s", rec.Code, rec.Body)
}
return id
}
func TestExportLinksImagesTables(t *testing.T) {
srv := newTestServer(t)
id := seedRichDoc2(t, srv)
md := do(t, srv, http.MethodGet, "/"+id+"/export?format=md", "").Body.String()
for _, want := range []string{"[site](https://petal.test)", "![a cat](/api/images/abc123.png)", "| Name | 年龄 |", "| --- | --- |", "| Mei | 30 |"} {
if !strings.Contains(md, want) {
t.Fatalf("markdown missing %q in:\n%s", want, md)
}
}
html := do(t, srv, http.MethodGet, "/"+id+"/export?format=html", "").Body.String()
for _, want := range []string{`<a href="https://petal.test">site</a>`, "<mark>lit</mark>", `<img src="/api/images/abc123.png" alt="a cat">`, "<th>Name</th>", "<td>Mei</td>"} {
if !strings.Contains(html, want) {
t.Fatalf("html missing %q in:\n%s", want, html)
}
}
docx := do(t, srv, http.MethodGet, "/"+id+"/export?format=docx", "")
raw := docx.Body.Bytes()
zr, err := zip.NewReader(bytes.NewReader(raw), int64(len(raw)))
if err != nil {
t.Fatalf("docx not a zip: %v", err)
}
var docXML string
for _, f := range zr.File {
if f.Name == "word/document.xml" {
rc, _ := f.Open()
b, _ := io.ReadAll(rc)
rc.Close()
docXML = string(b)
}
}
for _, want := range []string{"<w:tbl>", "Name", "年龄", "[a cat]", "<w:highlight"} {
if !strings.Contains(docXML, want) {
t.Fatalf("docx document.xml missing %q", want)
}
}
}
func TestExportUnsupportedFormat(t *testing.T) {
srv := newTestServer(t)
id := newDoc(t, srv)
if rec := do(t, srv, http.MethodGet, "/"+id+"/export?format=pdf", ""); rec.Code != http.StatusBadRequest {
t.Fatalf("expected 400 for unsupported format, got %d", rec.Code)
}
}
// Escaping makes a URL safe to sit inside an attribute; it says nothing about
// what happens when the attribute is followed. An export is the one artifact
// here meant to leave — the file handed to a teacher, opened on a machine with
// no reason to trust it — so a destination that runs code is dropped.
func TestExportDropsUnsafeLinkSchemes(t *testing.T) {
unsafe := []string{
"javascript:alert(1)",
"JaVaScRiPt:alert(1)",
"java\nscript:alert(1)", // browsers strip control characters first
" javascript:alert(1)",
"data:text/html;base64,PHNjcmlwdD5hbGVydCgxKTwvc2NyaXB0Pg==",
"vbscript:msgbox(1)",
}
for _, href := range unsafe {
if got := safeURL(href); got != "" {
t.Errorf("safeURL(%q) = %q, want it dropped", href, got)
}
}
safe := []string{
"https://example.com/a?b=1#c",
"http://example.com",
"mailto:her@example.com",
"/api/images/abc.png",
"#a-heading",
"notes/chapter:one.md", // a colon that isn't a scheme
}
for _, href := range safe {
if got := safeURL(href); got != href {
t.Errorf("safeURL(%q) = %q, want it kept", href, got)
}
}
}
// End to end through the renderers: an unsafe href loses its destination, never
// its words.
func TestRenderedExportsCarryNoScriptURLs(t *testing.T) {
doc := db.Document{
Title: "Notes",
Content: `{"type":"doc","content":[{"type":"paragraph","content":[
{"type":"text","text":"click me","marks":[{"type":"link","attrs":{"href":"javascript:alert(1)"}}]}]},
{"type":"image","attrs":{"src":"javascript:alert(2)","alt":"a drawing"}}]}`,
}
html, err := renderHTMLFile(doc)
if err != nil {
t.Fatal(err)
}
if strings.Contains(strings.ToLower(string(html)), "javascript:") {
t.Fatalf("html export carried a javascript: URL:\n%s", html)
}
if !strings.Contains(string(html), "click me") {
t.Fatal("html export dropped the link text along with the href")
}
if !strings.Contains(string(html), "a drawing") {
t.Fatal("html export dropped the alt text of the rejected image")
}
md, err := renderMarkdown(doc)
if err != nil {
t.Fatal(err)
}
if strings.Contains(strings.ToLower(string(md)), "javascript:") {
t.Fatalf("markdown export carried a javascript: URL:\n%s", md)
}
if !strings.Contains(string(md), "click me") {
t.Fatal("markdown export dropped the link text along with the href")
}
}