package storage import ( "archive/zip" "bytes" "context" "fmt" "io" "regexp" "sort" "strconv" "strings" "github.com/dslipak/pdf" "github.com/libnovel/backend/internal/bookstore" "github.com/libnovel/backend/internal/domain" minio "github.com/minio/minio-go/v7" "golang.org/x/net/html" ) // chapterHeadingRE matches common chapter heading patterns: // "Chapter 1", "Chapter 1:", "Chapter 1 -", "CHAPTER ONE", "1.", "Part 1", etc. var chapterHeadingRE = regexp.MustCompile( `(?i)^(?:chapter|ch\.?|part|episode|book)\s+(\d+|[ivxlcdm]+)\b|^\d{1,4}[\.\)]\s+\S`) type importer struct { mc *minioClient } // NewBookImporter creates a BookImporter that reads files from MinIO. func NewBookImporter(s *Store) bookstore.BookImporter { return &importer{mc: s.mc} } func (i *importer) Import(ctx context.Context, objectKey, fileType string) ([]bookstore.Chapter, error) { if fileType != "pdf" && fileType != "epub" { return nil, fmt.Errorf("unsupported file type: %s", fileType) } obj, err := i.mc.client.GetObject(ctx, "imports", objectKey, minio.GetObjectOptions{}) if err != nil { return nil, fmt.Errorf("get object from minio: %w", err) } defer obj.Close() data, err := io.ReadAll(obj) if err != nil { return nil, fmt.Errorf("read object: %w", err) } if fileType == "pdf" { return parsePDF(data) } return parseEPUB(data) } // AnalyzeFile parses the given PDF or EPUB data and returns the detected // chapter count and up to 3 preview lines (first non-empty line of each of // the first 3 chapters). It is used by the analyze-only endpoint so users // can preview chapter count before committing the import. func AnalyzeFile(data []byte, fileType string) (chapterCount int, firstLines []string, err error) { var chapters []bookstore.Chapter switch fileType { case "pdf": chapters, err = parsePDF(data) case "epub": chapters, err = parseEPUB(data) default: return 0, nil, fmt.Errorf("unsupported file type: %s", fileType) } if err != nil { return 0, nil, err } chapterCount = len(chapters) for i, ch := range chapters { if i >= 3 { break } line := strings.TrimSpace(ch.Content) if nl := strings.Index(line, "\n"); nl > 0 { line = line[:nl] } if len(line) > 120 { line = line[:120] + "…" } firstLines = append(firstLines, line) } return chapterCount, firstLines, nil } func parsePDF(data []byte) ([]bookstore.Chapter, error) { r, err := pdf.NewReader(bytes.NewReader(data), int64(len(data))) if err != nil { return nil, fmt.Errorf("open PDF: %w", err) } // Extract per-page text so we can detect chapter boundaries. numPages := r.NumPage() if numPages == 0 { return nil, fmt.Errorf("PDF has no pages") } // Collect full text first with page markers so we can split by chapter. var sb strings.Builder fonts := make(map[string]*pdf.Font) for i := 1; i <= numPages; i++ { page := r.Page(i) if page.V.IsNull() { continue } text, err := page.GetPlainText(fonts) if err != nil { continue } sb.WriteString(text) sb.WriteByte('\n') } return extractChaptersFromText(sb.String()), nil } // ── EPUB parsing ────────────────────────────────────────────────────────────── func parseEPUB(data []byte) ([]bookstore.Chapter, error) { zr, err := zip.NewReader(bytes.NewReader(data), int64(len(data))) if err != nil { return nil, fmt.Errorf("open EPUB zip: %w", err) } // 1. Read META-INF/container.xml → find rootfile (content.opf path). opfPath, err := epubRootfilePath(zr) if err != nil { return nil, fmt.Errorf("epub container: %w", err) } // 2. Parse content.opf → spine order of chapter files. spineFiles, titleMap, err := epubSpine(zr, opfPath) if err != nil { return nil, fmt.Errorf("epub spine: %w", err) } if len(spineFiles) == 0 { return nil, fmt.Errorf("EPUB spine is empty") } // Base directory of the OPF file for resolving relative hrefs. opfDir := "" if idx := strings.LastIndex(opfPath, "/"); idx >= 0 { opfDir = opfPath[:idx+1] } var chapters []bookstore.Chapter for i, href := range spineFiles { fullPath := opfDir + href content, err := epubFileContent(zr, fullPath) if err != nil { continue } text := htmlToText(content) if strings.TrimSpace(text) == "" { continue } title := titleMap[href] if title == "" { title = fmt.Sprintf("Chapter %d", i+1) } chapters = append(chapters, bookstore.Chapter{ Number: i + 1, Title: title, Content: text, }) } if len(chapters) == 0 { return nil, fmt.Errorf("no readable chapters found in EPUB") } return chapters, nil } // epubRootfilePath parses META-INF/container.xml and returns the full-path // of the OPF package document. func epubRootfilePath(zr *zip.Reader) (string, error) { f := zipFile(zr, "META-INF/container.xml") if f == nil { return "", fmt.Errorf("META-INF/container.xml not found") } rc, err := f.Open() if err != nil { return "", err } defer rc.Close() doc, err := html.Parse(rc) if err != nil { return "", err } var path string var walk func(*html.Node) walk = func(n *html.Node) { if n.Type == html.ElementNode && strings.EqualFold(n.Data, "rootfile") { for _, a := range n.Attr { if strings.EqualFold(a.Key, "full-path") { path = a.Val return } } } for c := n.FirstChild; c != nil; c = c.NextSibling { walk(c) } } walk(doc) if path == "" { return "", fmt.Errorf("rootfile full-path not found in container.xml") } return path, nil } // epubSpine parses the OPF document and returns the spine item hrefs in order, // plus a map from href → nav title (if available from NCX/NAV). func epubSpine(zr *zip.Reader, opfPath string) ([]string, map[string]string, error) { f := zipFile(zr, opfPath) if f == nil { return nil, nil, fmt.Errorf("OPF file %q not found in EPUB", opfPath) } rc, err := f.Open() if err != nil { return nil, nil, err } defer rc.Close() opfData, err := io.ReadAll(rc) if err != nil { return nil, nil, err } // Build id→href map from . idToHref := make(map[string]string) // Also keep a href→navTitle map (populated from NCX later). hrefTitle := make(map[string]string) // Parse OPF XML with html.Parse (handles malformed XML too). doc, _ := html.Parse(bytes.NewReader(opfData)) var manifestItems []struct{ id, href, mediaType string } var spineIdrefs []string var ncxID string var walk func(*html.Node) walk = func(n *html.Node) { if n.Type == html.ElementNode { tag := strings.ToLower(n.Data) switch tag { case "item": var id, href, mt string for _, a := range n.Attr { switch strings.ToLower(a.Key) { case "id": id = a.Val case "href": href = a.Val case "media-type": mt = a.Val } } if id != "" && href != "" { manifestItems = append(manifestItems, struct{ id, href, mediaType string }{id, href, mt}) idToHref[id] = href } case "itemref": for _, a := range n.Attr { if strings.ToLower(a.Key) == "idref" { spineIdrefs = append(spineIdrefs, a.Val) } } case "spine": for _, a := range n.Attr { if strings.ToLower(a.Key) == "toc" { ncxID = a.Val } } } } for c := n.FirstChild; c != nil; c = c.NextSibling { walk(c) } } walk(doc) // Build ordered spine href list. var spineHrefs []string for _, idref := range spineIdrefs { if href, ok := idToHref[idref]; ok { spineHrefs = append(spineHrefs, href) } } // If no explicit spine, fall back to all XHTML items in manifest order. if len(spineHrefs) == 0 { sort.Slice(manifestItems, func(i, j int) bool { return manifestItems[i].href < manifestItems[j].href }) for _, it := range manifestItems { mt := strings.ToLower(it.mediaType) if strings.Contains(mt, "html") || strings.HasSuffix(strings.ToLower(it.href), ".html") || strings.HasSuffix(strings.ToLower(it.href), ".xhtml") { spineHrefs = append(spineHrefs, it.href) } } } // Try to get chapter titles from NCX (toc.ncx). opfDir := "" if idx := strings.LastIndex(opfPath, "/"); idx >= 0 { opfDir = opfPath[:idx+1] } if ncxHref, ok := idToHref[ncxID]; ok { ncxPath := opfDir + ncxHref if ncxFile := zipFile(zr, ncxPath); ncxFile != nil { if ncxRC, err := ncxFile.Open(); err == nil { defer ncxRC.Close() parseNCXTitles(ncxRC, hrefTitle) } } } return spineHrefs, hrefTitle, nil } // parseNCXTitles extracts navPoint label→src mappings from a toc.ncx. func parseNCXTitles(r io.Reader, out map[string]string) { doc, err := html.Parse(r) if err != nil { return } // Collect navPoints: each has a and // a child. var walk func(*html.Node) walk = func(n *html.Node) { if n.Type == html.ElementNode && strings.EqualFold(n.Data, "navpoint") { var label, src string var inner func(*html.Node) inner = func(c *html.Node) { if c.Type == html.ElementNode { if strings.EqualFold(c.Data, "text") && label == "" { if c.FirstChild != nil && c.FirstChild.Type == html.TextNode { label = strings.TrimSpace(c.FirstChild.Data) } } if strings.EqualFold(c.Data, "content") { for _, a := range c.Attr { if strings.EqualFold(a.Key, "src") { // Strip fragment identifier (#...). src = strings.SplitN(a.Val, "#", 2)[0] } } } } for child := c.FirstChild; child != nil; child = child.NextSibling { inner(child) } } inner(n) if label != "" && src != "" { out[src] = label } } for c := n.FirstChild; c != nil; c = c.NextSibling { walk(c) } } walk(doc) } // epubFileContent returns the raw bytes of a file inside the EPUB zip. func epubFileContent(zr *zip.Reader, path string) ([]byte, error) { f := zipFile(zr, path) if f == nil { return nil, fmt.Errorf("file %q not in EPUB", path) } rc, err := f.Open() if err != nil { return nil, err } defer rc.Close() return io.ReadAll(rc) } // zipFile finds a file by name (case-insensitive) in a zip.Reader. func zipFile(zr *zip.Reader, name string) *zip.File { nameLower := strings.ToLower(name) for _, f := range zr.File { if strings.ToLower(f.Name) == nameLower { return f } } return nil } // htmlToText converts HTML/XHTML content to plain text suitable for storage. func htmlToText(data []byte) string { doc, err := html.Parse(bytes.NewReader(data)) if err != nil { return string(data) } var sb strings.Builder var walk func(*html.Node) walk = func(n *html.Node) { if n.Type == html.TextNode { text := strings.TrimSpace(n.Data) if text != "" { sb.WriteString(text) sb.WriteByte(' ') } } if n.Type == html.ElementNode { switch strings.ToLower(n.Data) { case "p", "div", "br", "h1", "h2", "h3", "h4", "h5", "h6", "li", "tr": // Block-level: ensure newline before content. if sb.Len() > 0 { s := sb.String() if s[len(s)-1] != '\n' { sb.WriteByte('\n') } } case "script", "style", "head": // Skip entirely. return } } for c := n.FirstChild; c != nil; c = c.NextSibling { walk(c) } if n.Type == html.ElementNode { switch strings.ToLower(n.Data) { case "p", "div", "h1", "h2", "h3", "h4", "h5", "h6", "li", "tr": sb.WriteByte('\n') } } } walk(doc) // Collapse multiple blank lines. lines := strings.Split(sb.String(), "\n") var out []string blanks := 0 for _, l := range lines { l = strings.TrimSpace(l) if l == "" { blanks++ if blanks <= 1 { out = append(out, "") } } else { blanks = 0 out = append(out, l) } } return strings.TrimSpace(strings.Join(out, "\n")) } // ── Chapter segmentation (shared by PDF and plain-text paths) ───────────────── // extractChaptersFromText splits a block of plain text into chapters by // detecting heading lines that match chapterHeadingRE. // Falls back to paragraph-splitting when no headings are found. func extractChaptersFromText(text string) []bookstore.Chapter { lines := strings.Split(text, "\n") type segment struct { title string number int lines []string } var segments []segment var cur *segment chNum := 0 for _, line := range lines { line = strings.TrimSpace(line) if chapterHeadingRE.MatchString(line) { if cur != nil { segments = append(segments, *cur) } chNum++ // Try to parse the explicit chapter number from the heading. if m := regexp.MustCompile(`\d+`).FindString(line); m != "" { if n, err := strconv.Atoi(m); err == nil && n > 0 && n < 100000 { chNum = n } } cur = &segment{title: line, number: chNum} } else if cur != nil && line != "" { cur.lines = append(cur.lines, line) } } if cur != nil { segments = append(segments, *cur) } // Require segments to have meaningful content (>= 100 chars). var chapters []bookstore.Chapter for _, seg := range segments { content := strings.Join(seg.lines, "\n") if len(strings.TrimSpace(content)) < 50 { continue } chapters = append(chapters, bookstore.Chapter{ Number: seg.number, Title: seg.title, Content: content, }) } // Fallback: no headings found — split by double newlines (paragraph blocks). if len(chapters) == 0 { paragraphs := strings.Split(text, "\n\n") n := 0 for _, para := range paragraphs { para = strings.TrimSpace(para) if len(para) > 100 { n++ chapters = append(chapters, bookstore.Chapter{ Number: n, Title: fmt.Sprintf("Chapter %d", n), Content: para, }) } } } return chapters } // ── Chapter ingestion ───────────────────────────────────────────────────────── // IngestChapters stores extracted chapters for a book. // Each chapter is written as a markdown file in the chapters MinIO bucket // and its index record is upserted in PocketBase via WriteChapter. func (s *Store) IngestChapters(ctx context.Context, slug string, chapters []bookstore.Chapter) error { for _, ch := range chapters { var mdContent string if ch.Title != "" && ch.Title != fmt.Sprintf("Chapter %d", ch.Number) { mdContent = fmt.Sprintf("# %s\n\n%s", ch.Title, ch.Content) } else { mdContent = fmt.Sprintf("# Chapter %d\n\n%s", ch.Number, ch.Content) } domainCh := domain.Chapter{ Ref: domain.ChapterRef{Number: ch.Number, Title: ch.Title}, Text: mdContent, } if err := s.WriteChapter(ctx, slug, domainCh); err != nil { return fmt.Errorf("ingest chapter %d: %w", ch.Number, err) } } return nil } // GetImportObjectKey returns the MinIO object key for an uploaded import file. func GetImportObjectKey(filename string) string { return fmt.Sprintf("imports/%s", filename) }