feat: paginated ranking scrape with lazy page fetching

- ScrapeRanking now accepts a maxPages int parameter (0 = all pages).
  Each page is fetched strictly sequentially; the next page is only
  requested after every entry from the current page has been sent,
  so there is no pre-fetching or look-ahead.
  Pagination stops automatically when no next-page link is present
  or when the rank-novels container is absent/empty.

- The ranking URL pattern follows the existing catalogue convention:
  /ranking?page=N (next-page link detection as the stop condition).

- Server: handleRankingRefresh reads an optional 'pages' form field
  and passes it to ScrapeRanking. Timeout scales at 90 s/page.

- UI: Refresh Rankings button is now a small form with a numeric
  'Pages' input (default 1), letting the user choose how many pages
  to pull in one refresh without touching the server config.
This commit is contained in:
Admin
2026-03-01 16:54:52 +05:00
parent e9f880f7f7
commit 1469e49190
3 changed files with 61 additions and 20 deletions

View File

@@ -365,25 +365,29 @@ func (s *Scraper) ScrapeChapterList(ctx context.Context, bookURL string) ([]scra
// ─── RankingProvider ─────────────────────────────────────────────────────────── // ─── RankingProvider ───────────────────────────────────────────────────────────
func (s *Scraper) ScrapeRanking(ctx context.Context) (<-chan scraper.BookMeta, <-chan error) { // ScrapeRanking pages through up to maxPages ranking pages on novelfire.net/ranking.
entries := make(chan scraper.BookMeta, 64) // Pages are fetched one at a time, strictly sequentially: the next page is only
// requested after every entry from the current page has been sent to the channel.
// maxPages <= 0 means "fetch all pages until no more are found".
func (s *Scraper) ScrapeRanking(ctx context.Context, maxPages int) (<-chan scraper.BookMeta, <-chan error) {
entries := make(chan scraper.BookMeta, 32)
errs := make(chan error, 16) errs := make(chan error, 16)
go func() { go func() {
defer close(entries) defer close(entries)
defer close(errs) defer close(errs)
pageURL := baseURL + rankingPath
rank := 1 rank := 1
for pageURL != "" { for page := 1; maxPages <= 0 || page <= maxPages; page++ {
select { select {
case <-ctx.Done(): case <-ctx.Done():
return return
default: default:
} }
s.log.Info("scraping ranking page", "url", pageURL) pageURL := fmt.Sprintf("%s%s?page=%d", baseURL, rankingPath, page)
s.log.Info("scraping ranking page", "page", page, "url", pageURL)
// Always use the urlClient for the ranking page — it requires JS rendering. // Always use the urlClient for the ranking page — it requires JS rendering.
raw, err := s.urlClient.GetContent(ctx, browser.ContentRequest{ raw, err := s.urlClient.GetContent(ctx, browser.ContentRequest{
@@ -394,23 +398,29 @@ func (s *Scraper) ScrapeRanking(ctx context.Context) (<-chan scraper.BookMeta, <
BestAttempt: true, BestAttempt: true,
}) })
if err != nil { if err != nil {
s.log.Debug("ranking page fetch failed", "url", pageURL, "err", err) s.log.Debug("ranking page fetch failed", "page", page, "url", pageURL, "err", err)
errs <- fmt.Errorf("ranking page: %w", err) errs <- fmt.Errorf("ranking page %d: %w", page, err)
return return
} }
root, err := htmlutil.ParseHTML(raw) root, err := htmlutil.ParseHTML(raw)
if err != nil { if err != nil {
errs <- fmt.Errorf("ranking page parse: %w", err) errs <- fmt.Errorf("ranking page %d parse: %w", page, err)
return return
} }
rankList := htmlutil.FindFirst(root, scraper.Selector{Class: "rank-novels"}) rankList := htmlutil.FindFirst(root, scraper.Selector{Class: "rank-novels"})
if rankList == nil { if rankList == nil {
s.log.Debug("rank-novels container not found, stopping pagination", "page", page)
break break
} }
items := htmlutil.FindAll(rankList, scraper.Selector{Tag: "li", Class: "novel-item"}) items := htmlutil.FindAll(rankList, scraper.Selector{Tag: "li", Class: "novel-item"})
if len(items) == 0 {
s.log.Debug("no ranking items on page, stopping pagination", "page", page)
break
}
for _, item := range items { for _, item := range items {
// Cover: <figure class="cover"><a href="/book/slug"><img data-src="..."></a></figure> // Cover: <figure class="cover"><a href="/book/slug"><img data-src="..."></a></figure>
var cover string var cover string
@@ -436,7 +446,7 @@ func (s *Scraper) ScrapeRanking(ctx context.Context) (<-chan scraper.BookMeta, <
// Status: <span class="status"> Ongoing/Completed </span> // Status: <span class="status"> Ongoing/Completed </span>
status := htmlutil.ExtractFirst(item, scraper.Selector{Tag: "span", Class: "status"}) status := htmlutil.ExtractFirst(item, scraper.Selector{Tag: "span", Class: "status"})
// Genres: <div class="categories"><div class="scroll"><span>Genre1</span><span>Genre2</span>...</div></div> // Genres: <div class="categories"><div class="scroll"><span>Genre1</span>...</div></div>
var genres []string var genres []string
categoriesNode := htmlutil.FindFirst(item, scraper.Selector{Tag: "div", Class: "categories"}) categoriesNode := htmlutil.FindFirst(item, scraper.Selector{Tag: "div", Class: "categories"})
if categoriesNode != nil { if categoriesNode != nil {
@@ -463,8 +473,12 @@ func (s *Scraper) ScrapeRanking(ctx context.Context) (<-chan scraper.BookMeta, <
} }
} }
// Next page - ranking pages use different pagination, just get first page for now // Stop if no next-page link exists (natural end of ranking list).
break nextHref := htmlutil.ExtractFirst(root, scraper.Selector{Tag: "a", Class: "next", Attr: "href"})
if nextHref == "" {
s.log.Debug("no next-page link found, stopping pagination", "page", page)
break
}
} }
}() }()

View File

@@ -112,9 +112,12 @@ type ChapterTextProvider interface {
// RankingProvider can enumerate novels from a ranking page. // RankingProvider can enumerate novels from a ranking page.
type RankingProvider interface { type RankingProvider interface {
// ScrapeRanking pages through the ranking list, sending BookMeta values // ScrapeRanking pages through up to maxPages ranking pages, sending BookMeta
// (with basic info like title, cover, genres, status, sourceURL) to the returned channel. // values (with basic info like title, cover, genres, status, sourceURL) to
ScrapeRanking(ctx context.Context) (<-chan BookMeta, <-chan error) // the returned channel. Pages are fetched sequentially and lazily: the next
// page is only requested once all entries from the current page have been
// sent. maxPages <= 0 means "all pages".
ScrapeRanking(ctx context.Context, maxPages int) (<-chan BookMeta, <-chan error)
} }
// NovelScraper is the full interface that a concrete novel source must implement. // NovelScraper is the full interface that a concrete novel source must implement.

View File

@@ -222,13 +222,21 @@ const rankingTmpl = `
class="text-sm px-3 py-1.5 rounded-lg bg-zinc-700 hover:bg-zinc-600 text-white inline-flex items-center gap-1"> class="text-sm px-3 py-1.5 rounded-lg bg-zinc-700 hover:bg-zinc-600 text-white inline-flex items-center gap-1">
View Markdown View Markdown
</a> </a>
<button <form
hx-post="/ranking/refresh" hx-post="/ranking/refresh"
hx-target="#ranking-refresh-status" hx-target="#ranking-refresh-status"
hx-swap="innerHTML" hx-swap="innerHTML"
class="text-sm px-3 py-1.5 rounded-lg bg-amber-700 hover:bg-amber-600 text-white inline-flex items-center gap-2"> class="flex items-center gap-2">
Refresh Rankings <label class="flex items-center gap-1.5">
</button> <span class="text-xs text-zinc-400 whitespace-nowrap">Pages</span>
<input type="number" name="pages" value="1" min="1"
class="w-16 rounded-lg bg-zinc-800 border border-zinc-700 px-2 py-1 text-sm text-zinc-100 text-center focus:outline-none focus:border-amber-500 transition-colors" />
</label>
<button type="submit"
class="text-sm px-3 py-1.5 rounded-lg bg-amber-700 hover:bg-amber-600 text-white inline-flex items-center gap-2">
Refresh Rankings
</button>
</form>
</div> </div>
</div> </div>
<div id="ranking-refresh-status" class="mt-2"></div> <div id="ranking-refresh-status" class="mt-2"></div>
@@ -353,7 +361,18 @@ func (s *Server) handleRanking(w http.ResponseWriter, r *http.Request) {
// handleRankingRefresh starts an async scrape of novelfire.net/ranking and // handleRankingRefresh starts an async scrape of novelfire.net/ranking and
// immediately returns a polling badge. The browser polls /ui/ranking/status // immediately returns a polling badge. The browser polls /ui/ranking/status
// until the job finishes, then follows an HX-Redirect back to /ranking. // until the job finishes, then follows an HX-Redirect back to /ranking.
//
// Accepts an optional form field "pages" (integer ≥ 1). 0 or absent means
// fetch all pages; otherwise at most that many pages are scraped.
func (s *Server) handleRankingRefresh(w http.ResponseWriter, r *http.Request) { func (s *Server) handleRankingRefresh(w http.ResponseWriter, r *http.Request) {
_ = r.ParseForm()
maxPages := 0
if p := strings.TrimSpace(r.FormValue("pages")); p != "" {
if n, err := strconv.Atoi(p); err == nil && n > 0 {
maxPages = n
}
}
s.mu.Lock() s.mu.Lock()
if s.rankingRunning { if s.rankingRunning {
s.mu.Unlock() s.mu.Unlock()
@@ -370,10 +389,15 @@ func (s *Server) handleRankingRefresh(w http.ResponseWriter, r *http.Request) {
s.mu.Unlock() s.mu.Unlock()
}() }()
ctx, cancel := context.WithTimeout(context.Background(), 120*time.Second) // Allow ~90 s per page; minimum 120 s for a single page.
timeout := 120 * time.Second
if maxPages > 1 {
timeout = time.Duration(maxPages) * 90 * time.Second
}
ctx, cancel := context.WithTimeout(context.Background(), timeout)
defer cancel() defer cancel()
rankingCh, errCh := s.novel.ScrapeRanking(ctx) rankingCh, errCh := s.novel.ScrapeRanking(ctx, maxPages)
var rankingItems []writer.RankingItem var rankingItems []writer.RankingItem
for { for {