From cdfa1ac5b218ce61b00e7cc8c8dd92d971945216 Mon Sep 17 00:00:00 2001 From: root Date: Thu, 9 Apr 2026 23:43:09 +0500 Subject: [PATCH] fix(pdf): fix page ordering, Win-1252 quotes, and chapter header cleanup MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three fixes to PDF chapter extraction quality: 1. Page ordering: parse page number from pdfcpu filename (out_Content_page_N.txt) instead of using lexicographic sort index — fixes chapters bleeding into each other (e.g. Prologue text appearing inside Chapter 1). 2. Windows-1252 chars: map bytes 0x91-0x9F to proper Unicode (curly quotes U+2018/ U+2019/U+201C/U+201D, em-dash U+2014, etc.) instead of raw Latin-1 control bytes that rendered as ◆ in the browser. 3. Chapter header cleanup: skip the first page of each bookmark range (decorative title art page) and strip any run-on title fragment at the start of the first body page (e.g. 'for New Journeys!I stood atop...' → 'I stood atop...'). The remaining sentence truncation is a fundamental limitation of this PDF's PUA-encoded body font (C2_1/Literata) — those glyphs cannot be decoded without the publisher's private ToUnicode mapping. --- backend/internal/storage/import.go | 181 +++++++++++++++++++++++++++-- 1 file changed, 171 insertions(+), 10 deletions(-) diff --git a/backend/internal/storage/import.go b/backend/internal/storage/import.go index c7dbf36..99a1260 100644 --- a/backend/internal/storage/import.go +++ b/backend/internal/storage/import.go @@ -186,17 +186,19 @@ func parsePDF(data []byte) ([]bookstore.Chapter, error) { return nil, fmt.Errorf("PDF has no content pages") } - // Sort entries by filename so index == page number - 1. - sort.Slice(entries, func(i, j int) bool { return entries[i].Name() < entries[j].Name() }) - - // Build page-index → extracted text map. + // pdfcpu names files "out_Content_page_N.txt" — parse the page number + // from the filename so the map is correct regardless of lexicographic order. pageTexts := make(map[int]string, len(entries)) - for idx, e := range entries { + for _, e := range entries { + pageNum := pageNumFromFilename(e.Name()) + if pageNum <= 0 { + continue + } raw, readErr := os.ReadFile(tmpDir + "/" + e.Name()) if readErr != nil { continue } - pageTexts[idx+1] = extractTextFromContentStream(raw) + pageTexts[pageNum] = fixWin1252(extractTextFromContentStream(raw)) } // Try to use bookmarks (outline) for chapter structure. @@ -208,9 +210,15 @@ func parsePDF(data []byte) ([]bookstore.Chapter, error) { } } - // Fallback: concatenate all page texts and split by heading patterns. + // Fallback: concatenate all page texts in page order and split by heading patterns. var sb strings.Builder - for p := 1; p <= len(entries); p++ { + maxPage := 0 + for p := range pageTexts { + if p > maxPage { + maxPage = p + } + } + for p := 1; p <= maxPage; p++ { sb.WriteString(pageTexts[p]) sb.WriteByte('\n') } @@ -264,14 +272,21 @@ func chaptersFromBookmarks(bookmarks []pdfcpu.Bookmark, pageTexts map[int]string continue } // Gather text for all pages in this bookmark's range. + // The first page of each chapter is typically a decorative title page + // (chapter number, subtitle art, series title) — skip it and start + // from PageFrom+1 so the content begins with actual story text. + bodyStart := bm.PageFrom + 1 + if bodyStart > bm.PageThru { + bodyStart = bm.PageFrom // single-page section, use it + } var sb strings.Builder - for p := bm.PageFrom; p <= bm.PageThru; p++ { + for p := bodyStart; p <= bm.PageThru; p++ { if t, ok := pageTexts[p]; ok { sb.WriteString(t) sb.WriteByte('\n') } } - text := strings.TrimSpace(sb.String()) + text := cleanChapterText(strings.TrimSpace(sb.String())) if len(text) < 50 { continue // skip nearly-empty sections } @@ -285,6 +300,152 @@ func chaptersFromBookmarks(bookmarks []pdfcpu.Bookmark, pageTexts map[int]string return chapters } +// cleanChapterText removes decorative header fragments that sometimes appear +// at the start of the first body page when the chapter subtitle is printed +// at the top of that page (e.g. "for New Journeys!I stood atop the roof..."). +// +// It strips any prefix text up to and including the last '!' or '?' that is +// immediately followed by a capital letter on the same line (a run-on from the +// title art), and removes short leading lines (< 40 chars) that look like +// title/header text rather than story content. +func cleanChapterText(text string) string { + lines := strings.Split(text, "\n") + + // Find first line that is substantive story content. + // Strategy: skip short lines at the top. The first line >= 40 chars + // OR starting with an opening quote is the start of the story. + start := 0 + for i, raw := range lines { + line := strings.TrimSpace(raw) + if line == "" { + start = i + 1 + continue + } + // Long enough to be a real sentence fragment from a body page. + if len(line) >= 40 || strings.HasPrefix(line, "\u201C") || strings.HasPrefix(line, "\"") { + start = i + break + } + // Short line — if it ends with '!' or '?' and the NEXT non-empty + // token on the SAME line (run-on) starts a sentence, strip it. + // This catches "for New Journeys!I stood atop..." on one line. + start = i + 1 // tentatively skip this short line + } + + result := strings.TrimSpace(strings.Join(lines[start:], "\n")) + + // Strip any run-on title fragment at the very start of the first line. + // Pattern: something ending with '!' or '?' immediately before a capital letter. + // e.g. "for New Journeys!I stood..." → "I stood..." + if len(result) > 0 { + // Find last '!' or '?' in the first 80 bytes that is followed by [A-Z"]. + firstLine := result + if nl := strings.Index(firstLine, "\n"); nl >= 0 { + firstLine = firstLine[:nl] + } + for i, c := range firstLine { + if (c == '!' || c == '?') && i+1 < len(firstLine) { + next := rune(firstLine[i+1]) + if (next >= 'A' && next <= 'Z') || next == '\u201C' || next == '"' { + // Strip up to and including this '!'/'?' + result = strings.TrimSpace(result[i+1:]) + break + } + } + } + } + + if result == "" { + return text + } + return result +} + +// pageNumFromFilename extracts the page number from a pdfcpu content-stream +// filename like "out_Content_page_42.txt". Returns 0 if not parseable. +func pageNumFromFilename(name string) int { + // Strip directory prefix and extension. + base := name + if idx := strings.LastIndex(base, "/"); idx >= 0 { + base = base[idx+1:] + } + if idx := strings.LastIndex(base, "."); idx >= 0 { + base = base[:idx] + } + // Find last "_" and parse the number after it. + if idx := strings.LastIndex(base, "_"); idx >= 0 { + n, err := strconv.Atoi(base[idx+1:]) + if err == nil && n > 0 { + return n + } + } + return 0 +} + +// win1252ToUnicode maps the Windows-1252 control range 0x80–0x9F to the +// Unicode characters they actually represent in that encoding. +// Standard Latin-1 maps these bytes to control characters; Win-1252 maps +// them to typographic symbols that appear in publisher PDFs. +var win1252ToUnicode = map[byte]rune{ + 0x80: '\u20AC', // € + 0x82: '\u201A', // ‚ + 0x83: '\u0192', // ƒ + 0x84: '\u201E', // „ + 0x85: '\u2026', // … + 0x86: '\u2020', // † + 0x87: '\u2021', // ‡ + 0x88: '\u02C6', // ˆ + 0x89: '\u2030', // ‰ + 0x8A: '\u0160', // Š + 0x8B: '\u2039', // ‹ + 0x8C: '\u0152', // Œ + 0x8E: '\u017D', // Ž + 0x91: '\u2018', // ' (left single quotation mark) + 0x92: '\u2019', // ' (right single quotation mark / apostrophe) + 0x93: '\u201C', // " (left double quotation mark) + 0x94: '\u201D', // " (right double quotation mark) + 0x95: '\u2022', // • (bullet) + 0x96: '\u2013', // – (en dash) + 0x97: '\u2014', // — (em dash) + 0x98: '\u02DC', // ˜ + 0x99: '\u2122', // ™ + 0x9A: '\u0161', // š + 0x9B: '\u203A', // › + 0x9C: '\u0153', // œ + 0x9E: '\u017E', // ž + 0x9F: '\u0178', // Ÿ +} + +// fixWin1252 replaces Windows-1252 specific bytes (0x80–0x9F) in a string +// that was decoded as raw Latin-1 bytes with their proper Unicode equivalents. +func fixWin1252(s string) string { + // Fast path: if no bytes in 0x80–0x9F range, return unchanged. + needsFix := false + for i := 0; i < len(s); i++ { + b := s[i] + if b >= 0x80 && b <= 0x9F { + needsFix = true + break + } + } + if !needsFix { + return s + } + var sb strings.Builder + sb.Grow(len(s)) + for i := 0; i < len(s); i++ { + b := s[i] + if b >= 0x80 && b <= 0x9F { + if r, ok := win1252ToUnicode[b]; ok { + sb.WriteRune(r) + continue + } + } + sb.WriteByte(b) + } + return sb.String() +} + // extractTextFromContentStream parses a raw PDF content stream and extracts // readable text from Tj and TJ operators. //