- Remove dead code: browser cdp/content_scrape strategies, writer package, printUsage, downloadAndStoreCoverCLI in main.go - Fix bugs: defer-in-loop in pocketbase deleteWhere, listAll() pagination hard cap removed, splitChapterTitle off-by-one in date extraction - Split server.go (~1700 lines) into focused handler files: handlers_audio, handlers_browse, handlers_progress, handlers_ranking, handlers_scrape - Export htmlutil.AttrVal/TextContent/ResolveURL; add storage/coverutil.go to consolidate duplicate helpers - Flatten deeply nested conditionals: voices() early-return guards, ScrapeCatalogue next-link double attr scan, chapterNumberFromKey dead strings.Cut line, splitChapterTitle double-nested unit/suffix loop - Add unit tests: htmlutil (9 funcs), novelfire ScrapeMetadata (3 cases), orchestrator Run (5 cases), storage chapterNumberFromKey/splitChapterTitle (22 cases); all pass with go build/vet/test clean
149 lines
5.3 KiB
Go
149 lines
5.3 KiB
Go
package novelfire
|
||
|
||
import (
|
||
"context"
|
||
"testing"
|
||
|
||
"github.com/libnovel/scraper/internal/scraper"
|
||
)
|
||
|
||
// rankingPage1HTML is a realistic mock of the popular genre listing page
|
||
// (novelfire.net/genre-all/sort-popular/status-all/all-novel?page=1).
|
||
// It uses the real novelfire.net DOM: <li class="novel-item"> cards with
|
||
// <h4 class="novel-title"> and a rel="next" pagination link.
|
||
func rankingPage1HTML() string {
|
||
return `<!DOCTYPE html>
|
||
<html><body>
|
||
<ul class="list-novel">
|
||
<li class="novel-item">
|
||
<a title="The Iron Throne" href="/book/the-iron-throne">
|
||
<figure class="novel-cover"><img class="lazy" src="data:image/gif;base64,R0lG" data-src="/covers/iron-throne.jpg" alt="The Iron Throne"></figure>
|
||
<h4 class="novel-title text2row">The Iron Throne</h4>
|
||
</a>
|
||
<div class="novel-stats"><i class="icon-book-open"></i> 500 Chapters</div>
|
||
</li>
|
||
<li class="novel-item">
|
||
<a title="Shadow Mage" href="/book/shadow-mage">
|
||
<figure class="novel-cover"><img class="lazy" src="data:image/gif;base64,R0lG" data-src="/covers/shadow-mage.jpg" alt="Shadow Mage"></figure>
|
||
<h4 class="novel-title text2row">Shadow Mage</h4>
|
||
</a>
|
||
<div class="novel-stats"><i class="icon-book-open"></i> 200 Chapters</div>
|
||
</li>
|
||
</ul>
|
||
<ul class="pagination">
|
||
<li class="page-item active"><span class="page-link">1</span></li>
|
||
<li class="page-item"><a class="page-link" href="/genre-all/sort-popular/status-all/all-novel?page=2" rel="next" aria-label="Next">›</a></li>
|
||
</ul>
|
||
</body></html>`
|
||
}
|
||
|
||
func rankingPage2HTML() string {
|
||
return `<!DOCTYPE html>
|
||
<html><body>
|
||
<ul class="list-novel">
|
||
<li class="novel-item">
|
||
<a title="Void Hunter" href="/book/void-hunter">
|
||
<figure class="novel-cover"><img class="lazy" src="data:image/gif;base64,R0lG" data-src="/covers/void-hunter.jpg" alt="Void Hunter"></figure>
|
||
<h4 class="novel-title text2row">Void Hunter</h4>
|
||
</a>
|
||
<div class="novel-stats"><i class="icon-book-open"></i> 100 Chapters</div>
|
||
</li>
|
||
</ul>
|
||
<!-- no rel="next" link → last page -->
|
||
<ul class="pagination">
|
||
<li class="page-item"><a class="page-link" href="?page=1" rel="prev">‹</a></li>
|
||
<li class="page-item active"><span class="page-link">2</span></li>
|
||
</ul>
|
||
</body></html>`
|
||
}
|
||
|
||
// drainRanking collects all entries from a ScrapeRanking call without
|
||
// deadlocking. It uses "for A != nil || B != nil" — nil channels are never
|
||
// selected, so setting one to nil effectively removes it from the select.
|
||
func drainRanking(t *testing.T, entryCh <-chan scraper.BookMeta, errCh <-chan error) []scraper.BookMeta {
|
||
t.Helper()
|
||
var entries []scraper.BookMeta
|
||
for entryCh != nil || errCh != nil {
|
||
select {
|
||
case meta, ok := <-entryCh:
|
||
if !ok {
|
||
entryCh = nil
|
||
} else {
|
||
entries = append(entries, meta)
|
||
}
|
||
case err, ok := <-errCh:
|
||
if !ok {
|
||
errCh = nil
|
||
} else if err != nil {
|
||
t.Fatalf("unexpected scrape error: %v", err)
|
||
}
|
||
}
|
||
}
|
||
return entries
|
||
}
|
||
|
||
// TestScrapeRanking_SinglePage verifies a single page is parsed into entries
|
||
// with sequential Ranking numbers using a stub client.
|
||
// ScrapeRanking uses s.client (the main client, not urlClient) because the
|
||
// ranking page is fully server-rendered.
|
||
func TestScrapeRanking_SinglePage(t *testing.T) {
|
||
// newScraper passes the stub as s.client — exactly what ScrapeRanking uses.
|
||
s := newScraper(rankingPage1HTML())
|
||
entryCh, errCh := s.ScrapeRanking(context.Background(), 1)
|
||
entries := drainRanking(t, entryCh, errCh)
|
||
|
||
if len(entries) != 2 {
|
||
t.Fatalf("expected 2 entries, got %d", len(entries))
|
||
}
|
||
if entries[0].Ranking != 1 || entries[0].Title != "The Iron Throne" {
|
||
t.Errorf("entry[0]: got rank=%d title=%q, want rank=1 title=%q",
|
||
entries[0].Ranking, entries[0].Title, "The Iron Throne")
|
||
}
|
||
if entries[1].Ranking != 2 || entries[1].Title != "Shadow Mage" {
|
||
t.Errorf("entry[1]: got rank=%d title=%q, want rank=2 title=%q",
|
||
entries[1].Ranking, entries[1].Title, "Shadow Mage")
|
||
}
|
||
}
|
||
|
||
// TestScrapeRanking_MultiPage verifies pagination across two pages yields
|
||
// contiguous rank numbers (1, 2, 3).
|
||
func TestScrapeRanking_MultiPage(t *testing.T) {
|
||
// Use pagedStubClient for s.client so each GetContent call returns the
|
||
// next page. ScrapeRanking now calls s.client directly.
|
||
urlClient := &pagedStubClient{pages: []string{rankingPage1HTML(), rankingPage2HTML()}}
|
||
s := New(urlClient, nil, nil, nil, nil) // nil cache — no disk I/O in tests
|
||
|
||
entryCh, errCh := s.ScrapeRanking(context.Background(), 0) // 0 = all pages
|
||
entries := drainRanking(t, entryCh, errCh)
|
||
|
||
if len(entries) != 3 {
|
||
t.Fatalf("expected 3 entries across 2 pages, got %d", len(entries))
|
||
}
|
||
want := []struct {
|
||
rank int
|
||
title string
|
||
}{
|
||
{1, "The Iron Throne"},
|
||
{2, "Shadow Mage"},
|
||
{3, "Void Hunter"},
|
||
}
|
||
for i, w := range want {
|
||
if entries[i].Ranking != w.rank || entries[i].Title != w.title {
|
||
t.Errorf("entry[%d]: got rank=%d title=%q, want rank=%d title=%q",
|
||
i, entries[i].Ranking, entries[i].Title, w.rank, w.title)
|
||
}
|
||
}
|
||
}
|
||
|
||
// TestScrapeRanking_EmptyPage verifies that a page with no .novel-item
|
||
// cards produces zero entries and closes channels cleanly (no deadlock).
|
||
func TestScrapeRanking_EmptyPage(t *testing.T) {
|
||
s := newScraper(`<!DOCTYPE html><html><body><div class="no-rankings"></div></body></html>`)
|
||
entryCh, errCh := s.ScrapeRanking(context.Background(), 1)
|
||
entries := drainRanking(t, entryCh, errCh)
|
||
|
||
if len(entries) != 0 {
|
||
t.Errorf("expected 0 entries for empty page, got %d", len(entries))
|
||
}
|
||
}
|