- docker-compose.yml: BODY_SIZE_LIMIT=52428800 (50MB) on UI service — adapter-node was rejecting PDFs >512KB with 'Content-length exceeds limit' - storage/import.go: add AnalyzeFile() public function using real PDF/EPUB parsers - handlers_import.go: analyzeImportFile() now calls storage.AnalyzeFile() instead of the file-size stub, so chapter count is accurate in the preview
569 lines
15 KiB
Go
569 lines
15 KiB
Go
package storage
|
|
|
|
import (
|
|
"archive/zip"
|
|
"bytes"
|
|
"context"
|
|
"fmt"
|
|
"io"
|
|
"regexp"
|
|
"sort"
|
|
"strconv"
|
|
"strings"
|
|
|
|
"github.com/dslipak/pdf"
|
|
"github.com/libnovel/backend/internal/bookstore"
|
|
"github.com/libnovel/backend/internal/domain"
|
|
minio "github.com/minio/minio-go/v7"
|
|
"golang.org/x/net/html"
|
|
)
|
|
|
|
// chapterHeadingRE matches common chapter heading patterns:
|
|
// "Chapter 1", "Chapter 1:", "Chapter 1 -", "CHAPTER ONE", "1.", "Part 1", etc.
|
|
var chapterHeadingRE = regexp.MustCompile(
|
|
`(?i)^(?:chapter|ch\.?|part|episode|book)\s+(\d+|[ivxlcdm]+)\b|^\d{1,4}[\.\)]\s+\S`)
|
|
|
|
type importer struct {
|
|
mc *minioClient
|
|
}
|
|
|
|
// NewBookImporter creates a BookImporter that reads files from MinIO.
|
|
func NewBookImporter(s *Store) bookstore.BookImporter {
|
|
return &importer{mc: s.mc}
|
|
}
|
|
|
|
func (i *importer) Import(ctx context.Context, objectKey, fileType string) ([]bookstore.Chapter, error) {
|
|
if fileType != "pdf" && fileType != "epub" {
|
|
return nil, fmt.Errorf("unsupported file type: %s", fileType)
|
|
}
|
|
|
|
obj, err := i.mc.client.GetObject(ctx, "imports", objectKey, minio.GetObjectOptions{})
|
|
if err != nil {
|
|
return nil, fmt.Errorf("get object from minio: %w", err)
|
|
}
|
|
defer obj.Close()
|
|
|
|
data, err := io.ReadAll(obj)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("read object: %w", err)
|
|
}
|
|
|
|
if fileType == "pdf" {
|
|
return parsePDF(data)
|
|
}
|
|
return parseEPUB(data)
|
|
}
|
|
|
|
// AnalyzeFile parses the given PDF or EPUB data and returns the detected
|
|
// chapter count and up to 3 preview lines (first non-empty line of each of
|
|
// the first 3 chapters). It is used by the analyze-only endpoint so users
|
|
// can preview chapter count before committing the import.
|
|
func AnalyzeFile(data []byte, fileType string) (chapterCount int, firstLines []string, err error) {
|
|
var chapters []bookstore.Chapter
|
|
switch fileType {
|
|
case "pdf":
|
|
chapters, err = parsePDF(data)
|
|
case "epub":
|
|
chapters, err = parseEPUB(data)
|
|
default:
|
|
return 0, nil, fmt.Errorf("unsupported file type: %s", fileType)
|
|
}
|
|
if err != nil {
|
|
return 0, nil, err
|
|
}
|
|
chapterCount = len(chapters)
|
|
for i, ch := range chapters {
|
|
if i >= 3 {
|
|
break
|
|
}
|
|
line := strings.TrimSpace(ch.Content)
|
|
if nl := strings.Index(line, "\n"); nl > 0 {
|
|
line = line[:nl]
|
|
}
|
|
if len(line) > 120 {
|
|
line = line[:120] + "…"
|
|
}
|
|
firstLines = append(firstLines, line)
|
|
}
|
|
return chapterCount, firstLines, nil
|
|
}
|
|
|
|
|
|
|
|
func parsePDF(data []byte) ([]bookstore.Chapter, error) {
|
|
r, err := pdf.NewReader(bytes.NewReader(data), int64(len(data)))
|
|
if err != nil {
|
|
return nil, fmt.Errorf("open PDF: %w", err)
|
|
}
|
|
|
|
// Extract per-page text so we can detect chapter boundaries.
|
|
numPages := r.NumPage()
|
|
if numPages == 0 {
|
|
return nil, fmt.Errorf("PDF has no pages")
|
|
}
|
|
|
|
// Collect full text first with page markers so we can split by chapter.
|
|
var sb strings.Builder
|
|
fonts := make(map[string]*pdf.Font)
|
|
for i := 1; i <= numPages; i++ {
|
|
page := r.Page(i)
|
|
if page.V.IsNull() {
|
|
continue
|
|
}
|
|
text, err := page.GetPlainText(fonts)
|
|
if err != nil {
|
|
continue
|
|
}
|
|
sb.WriteString(text)
|
|
sb.WriteByte('\n')
|
|
}
|
|
|
|
return extractChaptersFromText(sb.String()), nil
|
|
}
|
|
|
|
// ── EPUB parsing ──────────────────────────────────────────────────────────────
|
|
|
|
func parseEPUB(data []byte) ([]bookstore.Chapter, error) {
|
|
zr, err := zip.NewReader(bytes.NewReader(data), int64(len(data)))
|
|
if err != nil {
|
|
return nil, fmt.Errorf("open EPUB zip: %w", err)
|
|
}
|
|
|
|
// 1. Read META-INF/container.xml → find rootfile (content.opf path).
|
|
opfPath, err := epubRootfilePath(zr)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("epub container: %w", err)
|
|
}
|
|
|
|
// 2. Parse content.opf → spine order of chapter files.
|
|
spineFiles, titleMap, err := epubSpine(zr, opfPath)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("epub spine: %w", err)
|
|
}
|
|
|
|
if len(spineFiles) == 0 {
|
|
return nil, fmt.Errorf("EPUB spine is empty")
|
|
}
|
|
|
|
// Base directory of the OPF file for resolving relative hrefs.
|
|
opfDir := ""
|
|
if idx := strings.LastIndex(opfPath, "/"); idx >= 0 {
|
|
opfDir = opfPath[:idx+1]
|
|
}
|
|
|
|
var chapters []bookstore.Chapter
|
|
for i, href := range spineFiles {
|
|
fullPath := opfDir + href
|
|
content, err := epubFileContent(zr, fullPath)
|
|
if err != nil {
|
|
continue
|
|
}
|
|
text := htmlToText(content)
|
|
if strings.TrimSpace(text) == "" {
|
|
continue
|
|
}
|
|
title := titleMap[href]
|
|
if title == "" {
|
|
title = fmt.Sprintf("Chapter %d", i+1)
|
|
}
|
|
chapters = append(chapters, bookstore.Chapter{
|
|
Number: i + 1,
|
|
Title: title,
|
|
Content: text,
|
|
})
|
|
}
|
|
|
|
if len(chapters) == 0 {
|
|
return nil, fmt.Errorf("no readable chapters found in EPUB")
|
|
}
|
|
return chapters, nil
|
|
}
|
|
|
|
// epubRootfilePath parses META-INF/container.xml and returns the full-path
|
|
// of the OPF package document.
|
|
func epubRootfilePath(zr *zip.Reader) (string, error) {
|
|
f := zipFile(zr, "META-INF/container.xml")
|
|
if f == nil {
|
|
return "", fmt.Errorf("META-INF/container.xml not found")
|
|
}
|
|
rc, err := f.Open()
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
defer rc.Close()
|
|
|
|
doc, err := html.Parse(rc)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
|
|
var path string
|
|
var walk func(*html.Node)
|
|
walk = func(n *html.Node) {
|
|
if n.Type == html.ElementNode && strings.EqualFold(n.Data, "rootfile") {
|
|
for _, a := range n.Attr {
|
|
if strings.EqualFold(a.Key, "full-path") {
|
|
path = a.Val
|
|
return
|
|
}
|
|
}
|
|
}
|
|
for c := n.FirstChild; c != nil; c = c.NextSibling {
|
|
walk(c)
|
|
}
|
|
}
|
|
walk(doc)
|
|
|
|
if path == "" {
|
|
return "", fmt.Errorf("rootfile full-path not found in container.xml")
|
|
}
|
|
return path, nil
|
|
}
|
|
|
|
// epubSpine parses the OPF document and returns the spine item hrefs in order,
|
|
// plus a map from href → nav title (if available from NCX/NAV).
|
|
func epubSpine(zr *zip.Reader, opfPath string) ([]string, map[string]string, error) {
|
|
f := zipFile(zr, opfPath)
|
|
if f == nil {
|
|
return nil, nil, fmt.Errorf("OPF file %q not found in EPUB", opfPath)
|
|
}
|
|
rc, err := f.Open()
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
defer rc.Close()
|
|
|
|
opfData, err := io.ReadAll(rc)
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
|
|
// Build id→href map from <manifest>.
|
|
idToHref := make(map[string]string)
|
|
// Also keep a href→navTitle map (populated from NCX later).
|
|
hrefTitle := make(map[string]string)
|
|
|
|
// Parse OPF XML with html.Parse (handles malformed XML too).
|
|
doc, _ := html.Parse(bytes.NewReader(opfData))
|
|
|
|
var manifestItems []struct{ id, href, mediaType string }
|
|
var spineIdrefs []string
|
|
var ncxID string
|
|
|
|
var walk func(*html.Node)
|
|
walk = func(n *html.Node) {
|
|
if n.Type == html.ElementNode {
|
|
tag := strings.ToLower(n.Data)
|
|
switch tag {
|
|
case "item":
|
|
var id, href, mt string
|
|
for _, a := range n.Attr {
|
|
switch strings.ToLower(a.Key) {
|
|
case "id":
|
|
id = a.Val
|
|
case "href":
|
|
href = a.Val
|
|
case "media-type":
|
|
mt = a.Val
|
|
}
|
|
}
|
|
if id != "" && href != "" {
|
|
manifestItems = append(manifestItems, struct{ id, href, mediaType string }{id, href, mt})
|
|
idToHref[id] = href
|
|
}
|
|
case "itemref":
|
|
for _, a := range n.Attr {
|
|
if strings.ToLower(a.Key) == "idref" {
|
|
spineIdrefs = append(spineIdrefs, a.Val)
|
|
}
|
|
}
|
|
case "spine":
|
|
for _, a := range n.Attr {
|
|
if strings.ToLower(a.Key) == "toc" {
|
|
ncxID = a.Val
|
|
}
|
|
}
|
|
}
|
|
}
|
|
for c := n.FirstChild; c != nil; c = c.NextSibling {
|
|
walk(c)
|
|
}
|
|
}
|
|
walk(doc)
|
|
|
|
// Build ordered spine href list.
|
|
var spineHrefs []string
|
|
for _, idref := range spineIdrefs {
|
|
if href, ok := idToHref[idref]; ok {
|
|
spineHrefs = append(spineHrefs, href)
|
|
}
|
|
}
|
|
|
|
// If no explicit spine, fall back to all XHTML items in manifest order.
|
|
if len(spineHrefs) == 0 {
|
|
sort.Slice(manifestItems, func(i, j int) bool {
|
|
return manifestItems[i].href < manifestItems[j].href
|
|
})
|
|
for _, it := range manifestItems {
|
|
mt := strings.ToLower(it.mediaType)
|
|
if strings.Contains(mt, "html") || strings.HasSuffix(strings.ToLower(it.href), ".html") || strings.HasSuffix(strings.ToLower(it.href), ".xhtml") {
|
|
spineHrefs = append(spineHrefs, it.href)
|
|
}
|
|
}
|
|
}
|
|
|
|
// Try to get chapter titles from NCX (toc.ncx).
|
|
opfDir := ""
|
|
if idx := strings.LastIndex(opfPath, "/"); idx >= 0 {
|
|
opfDir = opfPath[:idx+1]
|
|
}
|
|
if ncxHref, ok := idToHref[ncxID]; ok {
|
|
ncxPath := opfDir + ncxHref
|
|
if ncxFile := zipFile(zr, ncxPath); ncxFile != nil {
|
|
if ncxRC, err := ncxFile.Open(); err == nil {
|
|
defer ncxRC.Close()
|
|
parseNCXTitles(ncxRC, hrefTitle)
|
|
}
|
|
}
|
|
}
|
|
|
|
return spineHrefs, hrefTitle, nil
|
|
}
|
|
|
|
// parseNCXTitles extracts navPoint label→src mappings from a toc.ncx.
|
|
func parseNCXTitles(r io.Reader, out map[string]string) {
|
|
doc, err := html.Parse(r)
|
|
if err != nil {
|
|
return
|
|
}
|
|
|
|
// Collect navPoints: each has a <navLabel><text>…</text></navLabel> and
|
|
// a <content src="…"/> child.
|
|
var walk func(*html.Node)
|
|
walk = func(n *html.Node) {
|
|
if n.Type == html.ElementNode && strings.EqualFold(n.Data, "navpoint") {
|
|
var label, src string
|
|
var inner func(*html.Node)
|
|
inner = func(c *html.Node) {
|
|
if c.Type == html.ElementNode {
|
|
if strings.EqualFold(c.Data, "text") && label == "" {
|
|
if c.FirstChild != nil && c.FirstChild.Type == html.TextNode {
|
|
label = strings.TrimSpace(c.FirstChild.Data)
|
|
}
|
|
}
|
|
if strings.EqualFold(c.Data, "content") {
|
|
for _, a := range c.Attr {
|
|
if strings.EqualFold(a.Key, "src") {
|
|
// Strip fragment identifier (#...).
|
|
src = strings.SplitN(a.Val, "#", 2)[0]
|
|
}
|
|
}
|
|
}
|
|
}
|
|
for child := c.FirstChild; child != nil; child = child.NextSibling {
|
|
inner(child)
|
|
}
|
|
}
|
|
inner(n)
|
|
if label != "" && src != "" {
|
|
out[src] = label
|
|
}
|
|
}
|
|
for c := n.FirstChild; c != nil; c = c.NextSibling {
|
|
walk(c)
|
|
}
|
|
}
|
|
walk(doc)
|
|
}
|
|
|
|
// epubFileContent returns the raw bytes of a file inside the EPUB zip.
|
|
func epubFileContent(zr *zip.Reader, path string) ([]byte, error) {
|
|
f := zipFile(zr, path)
|
|
if f == nil {
|
|
return nil, fmt.Errorf("file %q not in EPUB", path)
|
|
}
|
|
rc, err := f.Open()
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
defer rc.Close()
|
|
return io.ReadAll(rc)
|
|
}
|
|
|
|
// zipFile finds a file by name (case-insensitive) in a zip.Reader.
|
|
func zipFile(zr *zip.Reader, name string) *zip.File {
|
|
nameLower := strings.ToLower(name)
|
|
for _, f := range zr.File {
|
|
if strings.ToLower(f.Name) == nameLower {
|
|
return f
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// htmlToText converts HTML/XHTML content to plain text suitable for storage.
|
|
func htmlToText(data []byte) string {
|
|
doc, err := html.Parse(bytes.NewReader(data))
|
|
if err != nil {
|
|
return string(data)
|
|
}
|
|
|
|
var sb strings.Builder
|
|
var walk func(*html.Node)
|
|
walk = func(n *html.Node) {
|
|
if n.Type == html.TextNode {
|
|
text := strings.TrimSpace(n.Data)
|
|
if text != "" {
|
|
sb.WriteString(text)
|
|
sb.WriteByte(' ')
|
|
}
|
|
}
|
|
if n.Type == html.ElementNode {
|
|
switch strings.ToLower(n.Data) {
|
|
case "p", "div", "br", "h1", "h2", "h3", "h4", "h5", "h6", "li", "tr":
|
|
// Block-level: ensure newline before content.
|
|
if sb.Len() > 0 {
|
|
s := sb.String()
|
|
if s[len(s)-1] != '\n' {
|
|
sb.WriteByte('\n')
|
|
}
|
|
}
|
|
case "script", "style", "head":
|
|
// Skip entirely.
|
|
return
|
|
}
|
|
}
|
|
for c := n.FirstChild; c != nil; c = c.NextSibling {
|
|
walk(c)
|
|
}
|
|
if n.Type == html.ElementNode {
|
|
switch strings.ToLower(n.Data) {
|
|
case "p", "div", "h1", "h2", "h3", "h4", "h5", "h6", "li", "tr":
|
|
sb.WriteByte('\n')
|
|
}
|
|
}
|
|
}
|
|
walk(doc)
|
|
|
|
// Collapse multiple blank lines.
|
|
lines := strings.Split(sb.String(), "\n")
|
|
var out []string
|
|
blanks := 0
|
|
for _, l := range lines {
|
|
l = strings.TrimSpace(l)
|
|
if l == "" {
|
|
blanks++
|
|
if blanks <= 1 {
|
|
out = append(out, "")
|
|
}
|
|
} else {
|
|
blanks = 0
|
|
out = append(out, l)
|
|
}
|
|
}
|
|
return strings.TrimSpace(strings.Join(out, "\n"))
|
|
}
|
|
|
|
// ── Chapter segmentation (shared by PDF and plain-text paths) ─────────────────
|
|
|
|
// extractChaptersFromText splits a block of plain text into chapters by
|
|
// detecting heading lines that match chapterHeadingRE.
|
|
// Falls back to paragraph-splitting when no headings are found.
|
|
func extractChaptersFromText(text string) []bookstore.Chapter {
|
|
lines := strings.Split(text, "\n")
|
|
|
|
type segment struct {
|
|
title string
|
|
number int
|
|
lines []string
|
|
}
|
|
|
|
var segments []segment
|
|
var cur *segment
|
|
chNum := 0
|
|
|
|
for _, line := range lines {
|
|
line = strings.TrimSpace(line)
|
|
if chapterHeadingRE.MatchString(line) {
|
|
if cur != nil {
|
|
segments = append(segments, *cur)
|
|
}
|
|
chNum++
|
|
// Try to parse the explicit chapter number from the heading.
|
|
if m := regexp.MustCompile(`\d+`).FindString(line); m != "" {
|
|
if n, err := strconv.Atoi(m); err == nil && n > 0 && n < 100000 {
|
|
chNum = n
|
|
}
|
|
}
|
|
cur = &segment{title: line, number: chNum}
|
|
} else if cur != nil && line != "" {
|
|
cur.lines = append(cur.lines, line)
|
|
}
|
|
}
|
|
if cur != nil {
|
|
segments = append(segments, *cur)
|
|
}
|
|
|
|
// Require segments to have meaningful content (>= 100 chars).
|
|
var chapters []bookstore.Chapter
|
|
for _, seg := range segments {
|
|
content := strings.Join(seg.lines, "\n")
|
|
if len(strings.TrimSpace(content)) < 50 {
|
|
continue
|
|
}
|
|
chapters = append(chapters, bookstore.Chapter{
|
|
Number: seg.number,
|
|
Title: seg.title,
|
|
Content: content,
|
|
})
|
|
}
|
|
|
|
// Fallback: no headings found — split by double newlines (paragraph blocks).
|
|
if len(chapters) == 0 {
|
|
paragraphs := strings.Split(text, "\n\n")
|
|
n := 0
|
|
for _, para := range paragraphs {
|
|
para = strings.TrimSpace(para)
|
|
if len(para) > 100 {
|
|
n++
|
|
chapters = append(chapters, bookstore.Chapter{
|
|
Number: n,
|
|
Title: fmt.Sprintf("Chapter %d", n),
|
|
Content: para,
|
|
})
|
|
}
|
|
}
|
|
}
|
|
|
|
return chapters
|
|
}
|
|
|
|
// ── Chapter ingestion ─────────────────────────────────────────────────────────
|
|
|
|
// IngestChapters stores extracted chapters for a book.
|
|
// Each chapter is written as a markdown file in the chapters MinIO bucket
|
|
// and its index record is upserted in PocketBase via WriteChapter.
|
|
func (s *Store) IngestChapters(ctx context.Context, slug string, chapters []bookstore.Chapter) error {
|
|
for _, ch := range chapters {
|
|
var mdContent string
|
|
if ch.Title != "" && ch.Title != fmt.Sprintf("Chapter %d", ch.Number) {
|
|
mdContent = fmt.Sprintf("# %s\n\n%s", ch.Title, ch.Content)
|
|
} else {
|
|
mdContent = fmt.Sprintf("# Chapter %d\n\n%s", ch.Number, ch.Content)
|
|
}
|
|
domainCh := domain.Chapter{
|
|
Ref: domain.ChapterRef{Number: ch.Number, Title: ch.Title},
|
|
Text: mdContent,
|
|
}
|
|
if err := s.WriteChapter(ctx, slug, domainCh); err != nil {
|
|
return fmt.Errorf("ingest chapter %d: %w", ch.Number, err)
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// GetImportObjectKey returns the MinIO object key for an uploaded import file.
|
|
func GetImportObjectKey(filename string) string {
|
|
return fmt.Sprintf("imports/%s", filename)
|
|
}
|