diff --git a/comments.go b/comments.go
index 6e41f38..c65364e 100644
--- a/comments.go
+++ b/comments.go
@@ -2,6 +2,7 @@ package devianter
import (
"encoding/json"
+ "html"
"net/url"
"strconv"
"strings"
@@ -25,6 +26,10 @@ type Thread struct {
// Comment is the comment's plain text, which [GetComments] extracts from
// TextContent. Prefer it; TextContent is the unprocessed original.
+ //
+ // Text is all it holds: a comment is a rich document, and its images,
+ // emotes, mentions, and link targets are dropped in the flattening. Read
+ // TextContent for those.
Comment string
TextContent Text
@@ -71,46 +76,105 @@ func GetComments(postid string, cursor string, page int, typ int) (cmmts Comment
cursor = cmmts.Cursor
for i := 0; i < len(cmmts.Thread); i++ {
- cmmts.Thread[i].Comment = flattenComment(cmmts.Thread[i].TextContent.Html.Markup)
+ cmmts.Thread[i].Comment = flattenMarkup(cmmts.Thread[i].TextContent.Html.Markup)
}
}
return
}
-// flattenComment renders a body of user-written markup as plain text, be it a
-// comment or a deviation's description. Bodies are JSON inside JSON: newer ones
-// are a Draft.js document encoded into the markup string, older ones are plain
-// HTML, which passes through unchanged. Markup that does not parse, and empty
-// markup, also pass through.
+// flattenMarkup renders a body of user-written markup as plain text, be it a
+// comment or a deviation's description. Bodies are JSON inside JSON, and
+// DeviantArt still serves all three formats it has used over the years:
//
-// A Draft.js document is a list of blocks, which are block-level elements
-// (paragraphs, list items); they are joined with newlines, one block per line.
-func flattenComment(m string) string {
+// - tiptap, current: {"version":1,"document":{"type":"doc","content":[...]}}
+// - Draft.js, legacy: {"blocks":[{"text":"..."}]}
+// - plain HTML, oldest, which passes through unchanged
+//
+// Block-level elements (paragraphs, headings) are joined with newlines, one per
+// line, and a hard break inside one becomes a newline too. HTML entities in the
+// text are decoded, so a body reads as ’ on the wire but an apostrophe
+// here. Markup matching no known format passes through unchanged rather than
+// being replaced by an empty string.
+func flattenMarkup(m string) string {
l := len(m)
if l == 0 || m[0] != '{' || m[l-1] != '}' {
return m
}
+ if text, ok := flattenTiptap(m); ok {
+ return text
+ }
+ if text, ok := flattenDraftJS(m); ok {
+ return text
+ }
+ return m
+}
+
+// tiptapNode is one node of a tiptap (ProseMirror) document tree.
+//
+// The document's "version" field is deliberately not modelled: DeviantArt sends
+// it as a number on some bodies and a string on others, so any typed field for
+// it fails to unmarshal on half of them.
+type tiptapNode struct {
+ Type string `json:"type"`
+ Text string `json:"text"`
+ Content []tiptapNode `json:"content"`
+}
+
+// flattenTiptap renders a tiptap document, reporting false if the markup is not
+// one.
+func flattenTiptap(m string) (string, bool) {
+ var doc struct {
+ Document tiptapNode `json:"document"`
+ }
+ if json.Unmarshal([]byte(m), &doc) != nil || doc.Document.Type != "doc" {
+ return "", false
+ }
+
+ lines := make([]string, 0, len(doc.Document.Content))
+ for _, block := range doc.Document.Content {
+ var b strings.Builder
+ writeTiptapText(block, &b)
+ lines = append(lines, b.String())
+ }
+
+ return html.UnescapeString(strings.Join(lines, "\n")), true
+}
+
+// writeTiptapText collects the text of a node and everything nested inside it.
+// Nodes carrying no text of their own ā images, galleries, emotes ā contribute
+// nothing.
+func writeTiptapText(n tiptapNode, b *strings.Builder) {
+ switch n.Type {
+ case "text":
+ b.WriteString(n.Text)
+ return
+ case "hardBreak":
+ b.WriteString("\n")
+ return
+ }
+ for _, c := range n.Content {
+ writeTiptapText(c, b)
+ }
+}
+
+// flattenDraftJS renders a legacy Draft.js document, reporting false if the
+// markup is not one.
+func flattenDraftJS(m string) (string, bool) {
var content struct {
Blocks []struct {
Text string
}
}
-
- e := json.Unmarshal([]byte(m), &content)
- try(e)
-
- if len(content.Blocks) == 0 {
- return m
+ if json.Unmarshal([]byte(m), &content) != nil || len(content.Blocks) == 0 {
+ return "", false
}
- var b strings.Builder
- for i, a := range content.Blocks {
- if i > 0 {
- b.WriteString("\n")
- }
- b.WriteString(a.Text)
+ lines := make([]string, 0, len(content.Blocks))
+ for _, blk := range content.Blocks {
+ lines = append(lines, blk.Text)
}
- return b.String()
+
+ return html.UnescapeString(strings.Join(lines, "\n")), true
}
diff --git a/comments_test.go b/comments_test.go
index a6c5a47..60d38e4 100644
--- a/comments_test.go
+++ b/comments_test.go
@@ -2,50 +2,145 @@ package devianter
import "testing"
-// Regression: flattenComment's shape check used to read m[0] and m[len(m)-1]
-// without a length check, so a comment with an empty markup body panicked with
-// index out of range and killed the caller's process.
-func TestFlattenCommentEmptyMarkup(t *testing.T) {
- if got := flattenComment(""); got != "" {
+// Regression: the shape check used to read m[0] and m[len(m)-1] without a length
+// check, so a comment with an empty markup body panicked with index out of range
+// and killed the caller's process.
+func TestFlattenMarkupEmptyMarkup(t *testing.T) {
+ if got := flattenMarkup(""); got != "" {
t.Errorf("want an empty comment for empty markup, got %q", got)
}
}
-func TestFlattenComment(t *testing.T) {
- // A newer, Draft.js-encoded body is flattened to its text.
- draft := `{"blocks":[{"text":"hello there"}]}`
- if got := flattenComment(draft); got != "hello there" {
- t.Errorf("want the Draft.js block text, got %q", got)
+// tiptap is what DeviantArt actually serves today; the fixtures below are shaped
+// like responses captured from the live API.
+func TestFlattenMarkupTiptap(t *testing.T) {
+ one := `{"version":1,"document":{"type":"doc","content":[` +
+ `{"type":"paragraph","attrs":{"textAlign":"left"},"content":[` +
+ `{"type":"text","text":"Nice artwork!"}]}]}}`
+ if got := flattenMarkup(one); got != "Nice artwork!" {
+ t.Errorf("want the paragraph's text, got %q", got)
}
- // An older, plain-HTML body passes through untouched.
- html := "hello there"
- if got := flattenComment(html); got != html {
- t.Errorf("want plain HTML passed through, got %q", got)
+ // Regression: DeviantArt sends "version" as a number on some bodies and a
+ // string on others. Modelling it as either type fails to unmarshal half of
+ // them, so it must not be modelled at all.
+ stringVersion := `{"version":"1","document":{"type":"doc","content":[` +
+ `{"type":"paragraph","content":[{"type":"text","text":"hello"}]}]}}`
+ if got := flattenMarkup(stringVersion); got != "hello" {
+ t.Errorf(`want a string "version" handled the same as a numeric one, got %q`, got)
+ }
+
+ // Each block is its own line.
+ two := `{"version":1,"document":{"type":"doc","content":[` +
+ `{"type":"paragraph","content":[{"type":"text","text":"first"}]},` +
+ `{"type":"paragraph","content":[{"type":"text","text":"second"}]}]}}`
+ if got, want := flattenMarkup(two), "first\nsecond"; got != want {
+ t.Errorf("want one line per block:\n got %q\nwant %q", got, want)
+ }
+
+ // An empty paragraph is a blank line, not something to skip.
+ blank := `{"version":1,"document":{"type":"doc","content":[` +
+ `{"type":"paragraph","content":[{"type":"text","text":"a"}]},` +
+ `{"type":"paragraph"},` +
+ `{"type":"paragraph","content":[{"type":"text","text":"b"}]}]}}`
+ if got, want := flattenMarkup(blank), "a\n\nb"; got != want {
+ t.Errorf("want an empty paragraph preserved as a blank line:\n got %q\nwant %q", got, want)
+ }
+
+ // A hard break is a newline within its block.
+ brk := `{"version":1,"document":{"type":"doc","content":[` +
+ `{"type":"paragraph","content":[{"type":"text","text":"up"},` +
+ `{"type":"hardBreak"},{"type":"text","text":"down"}]}]}}`
+ if got, want := flattenMarkup(brk), "up\ndown"; got != want {
+ t.Errorf("want a hardBreak as a newline:\n got %q\nwant %q", got, want)
+ }
+
+ // Marked-up runs are separate text nodes and must be concatenated, not
+ // separated.
+ marks := `{"version":1,"document":{"type":"doc","content":[` +
+ `{"type":"paragraph","content":[` +
+ `{"type":"text","text":"plain "},` +
+ `{"type":"text","marks":[{"type":"bold"}],"text":"bold"},` +
+ `{"type":"text","text":" tail"}]}]}}`
+ if got, want := flattenMarkup(marks), "plain bold tail"; got != want {
+ t.Errorf("want marked runs concatenated:\n got %q\nwant %q", got, want)
+ }
+
+ // Text carrying a link mark still surfaces; the href does not.
+ link := `{"version":1,"document":{"type":"doc","content":[` +
+ `{"type":"paragraph","content":[{"type":"text","marks":[` +
+ `{"type":"link","attrs":{"href":"https://example.com"}}],"text":"click here"}]}]}}`
+ if got, want := flattenMarkup(link), "click here"; got != want {
+ t.Errorf("want the link text without the href:\n got %q\nwant %q", got, want)
+ }
+
+ // Headings are blocks like any other.
+ heading := `{"version":1,"document":{"type":"doc","content":[` +
+ `{"type":"heading","attrs":{"level":2},"content":[{"type":"text","text":"Title"}]},` +
+ `{"type":"paragraph","content":[{"type":"text","text":"body"}]}]}}`
+ if got, want := flattenMarkup(heading), "Title\nbody"; got != want {
+ t.Errorf("want a heading flattened as a block:\n got %q\nwant %q", got, want)
+ }
+
+ // Bodies carry HTML entities on the wire; plain text should not.
+ entity := `{"version":1,"document":{"type":"doc","content":[` +
+ `{"type":"paragraph","content":[{"type":"text","text":"I’d love & more"}]}]}}`
+ if got, want := flattenMarkup(entity), "Iād love & more"; got != want {
+ t.Errorf("want HTML entities decoded:\n got %q\nwant %q", got, want)
+ }
+
+ // A node carrying no text of its own contributes nothing.
+ emote := `{"version":1,"document":{"type":"doc","content":[` +
+ `{"type":"paragraph","content":[{"type":"text","text":"hi "},` +
+ `{"type":"da-emote","attrs":{"name":":happy:"}}]}]}}`
+ if got, want := flattenMarkup(emote), "hi "; got != want {
+ t.Errorf("want a textless node to contribute nothing:\n got %q\nwant %q", got, want)
+ }
+}
+
+// Draft.js is legacy but still served on old bodies, so the path stays.
+func TestFlattenMarkupDraftJS(t *testing.T) {
+ draft := `{"blocks":[{"text":"hello there"}]}`
+ if got := flattenMarkup(draft); got != "hello there" {
+ t.Errorf("want the Draft.js block text, got %q", got)
}
// Regression: the block loop used to assign rather than accumulate, so every
// block but the last was silently dropped and a multi-paragraph comment came
// back as its closing line only.
multi := `{"blocks":[{"text":"first"},{"text":"second"},{"text":"third"}]}`
- if got, want := flattenComment(multi), "first\nsecond\nthird"; got != want {
+ if got, want := flattenMarkup(multi), "first\nsecond\nthird"; got != want {
t.Errorf("want every block, one per line:\n got %q\nwant %q", got, want)
}
- // An empty block is a blank line in the comment, not something to skip.
+ // An empty block is a blank line, not something to skip.
blank := `{"blocks":[{"text":"first"},{"text":""},{"text":"third"}]}`
- if got, want := flattenComment(blank), "first\n\nthird"; got != want {
+ if got, want := flattenMarkup(blank), "first\n\nthird"; got != want {
t.Errorf("want an empty block preserved as a blank line:\n got %q\nwant %q", got, want)
}
+}
+
+func TestFlattenMarkupPassthrough(t *testing.T) {
+ // An older, plain-HTML body passes through untouched.
+ html := "hello there"
+ if got := flattenMarkup(html); got != html {
+ t.Errorf("want plain HTML passed through, got %q", got)
+ }
+
+ // Brace-shaped markup in no known format falls back to itself rather than to
+ // an empty string.
+ if got := flattenMarkup("{}"); got != "{}" {
+ t.Errorf("want the original markup when nothing parses, got %q", got)
+ }
- // Brace-shaped markup that isn't a Draft.js document falls back to itself
- // rather than to an empty string.
- if got := flattenComment("{}"); got != "{}" {
- t.Errorf("want the original markup when there are no blocks, got %q", got)
+ // Well-formed JSON that is neither format is still not silently eaten.
+ other := `{"something":"else"}`
+ if got := flattenMarkup(other); got != other {
+ t.Errorf("want unrecognised JSON passed through, got %q", got)
}
// A single brace satisfies neither end of the shape check.
- if got := flattenComment("{"); got != "{" {
+ if got := flattenMarkup("{"); got != "{" {
t.Errorf("want a lone brace passed through, got %q", got)
}
}
diff --git a/deviantion.go b/deviantion.go
index 5b6df39..f3f9103 100644
--- a/deviantion.go
+++ b/deviantion.go
@@ -79,9 +79,14 @@ type Media struct {
}
// Text is a block of user-written text ā a description, a comment, a group's
-// about page. Markup holds either HTML or a JSON-encoded Draft.js document,
-// distinguished by Type; the functions that return a Text generally extract the
-// plain text into a neighbouring field, which is easier to use.
+// about page.
+//
+// Markup is a rich document rather than a string of prose, in whichever format
+// DeviantArt stored it: tiptap JSON on anything recent (Type is "tiptap"),
+// Draft.js JSON on older bodies, or plain HTML on the oldest. The functions
+// returning a Text generally flatten it to plain text in a neighbouring field,
+// which is what most callers want; read Markup itself for the formatting,
+// images, and links that flattening discards.
type Text struct {
Excerpt string
Html struct {
@@ -91,8 +96,9 @@ type Text struct {
// Post is a deviation together with its comment metadata, as returned by
// [GetDeviation]. IMG and Description are conveniences that GetDeviation derives
-// from the Deviation, so callers need not assemble a URL or decode Draft.js
-// markup themselves.
+// from the Deviation, so callers need not assemble a URL or flatten a rich-text
+// document themselves. Description is empty for the many deviations that have
+// none.
//
// Comments holds only a total and a cursor. To retrieve the comments, pass them
// to [GetComments] with type 1.
@@ -175,7 +181,14 @@ func GetDeviation(id string, user string) (st Post, err Error) {
st.IMG, _ = UrlFromMedia(st.Deviation.Media)
- st.Description = flattenComment(st.Deviation.TextContent.Html.Markup)
+ // The description lives in Extended.DescriptionText on the great majority of
+ // deviations; TextContent carries it on only a small minority. Prefer the
+ // former and fall back, since either may be the populated one.
+ desc := st.Deviation.Extended.DescriptionText.Html.Markup
+ if desc == "" {
+ desc = st.Deviation.TextContent.Html.Markup
+ }
+ st.Description = flattenMarkup(desc)
return
}