Some checks failed
CI / v3 / Check ui (pull_request) Failing after 15s
CI / v3 / Test backend (pull_request) Failing after 16s
CI / v3 / Docker / backend (pull_request) Has been skipped
CI / v3 / Docker / runner (pull_request) Has been skipped
CI / v3 / Docker / ui (pull_request) Has been skipped
- Remove all pre-v3 code: scraper, ui-v2, backend v1, ios v1+v2, legacy CI workflows - Flatten v3/ contents to repo root - Add Doppler secrets management (project=libnovel, config=prd) - Add justfile with doppler run wrappers for all docker compose commands - Strip hardcoded env fallbacks from docker-compose.yml - Add minimal README.md - Clean up .gitignore
186 lines
5.8 KiB
Go
186 lines
5.8 KiB
Go
package runner
|
||
|
||
// catalogue_refresh.go — independent loop that walks the full novelfire.net
|
||
// catalogue, scrapes per-book metadata, downloads cover images to MinIO, and
|
||
// indexes every book in Meilisearch.
|
||
//
|
||
// Design:
|
||
// - Runs on its own ticker (CatalogueRefreshInterval, default 24h) inside Run().
|
||
// - Also fires once on startup.
|
||
// - ScrapeCatalogue streams CatalogueEntry values over a channel — we iterate
|
||
// and call ScrapeMetadata for each entry.
|
||
// - Per-request random jitter (1–3s) prevents hammering novelfire.net.
|
||
// - Cover images are fetched from the URL embedded in BookMeta.Cover and
|
||
// stored in MinIO (browse bucket, key: covers/{slug}.jpg).
|
||
// - WriteMetadata + UpsertBook are called for every successfully scraped book.
|
||
// - Errors for individual books are logged and skipped; the loop continues.
|
||
// - The cover URL stored in BookMeta.Cover is rewritten to the internal proxy
|
||
// path (/api/cover/novelfire.net/{slug}) so the UI always fetches via the
|
||
// backend, which will serve from MinIO.
|
||
|
||
import (
|
||
"context"
|
||
"fmt"
|
||
"io"
|
||
"math/rand"
|
||
"net/http"
|
||
"time"
|
||
)
|
||
|
||
// runCatalogueRefresh performs one full catalogue walk: scrapes metadata for
|
||
// every book on novelfire.net, downloads covers to MinIO, and upserts to
|
||
// Meilisearch. Errors for individual books are logged and skipped.
|
||
func (r *Runner) runCatalogueRefresh(ctx context.Context) {
|
||
if r.deps.Novel == nil {
|
||
r.deps.Log.Warn("runner: catalogue refresh skipped — Novel scraper not configured")
|
||
return
|
||
}
|
||
if r.deps.BookWriter == nil {
|
||
r.deps.Log.Warn("runner: catalogue refresh skipped — BookWriter not configured")
|
||
return
|
||
}
|
||
|
||
log := r.deps.Log.With("op", "catalogue_refresh")
|
||
log.Info("runner: catalogue refresh starting")
|
||
|
||
entries, errCh := r.deps.Novel.ScrapeCatalogue(ctx)
|
||
|
||
ok, skipped, errCount := 0, 0, 0
|
||
for entry := range entries {
|
||
if ctx.Err() != nil {
|
||
break
|
||
}
|
||
|
||
// Skip books already present in Meilisearch — they were indexed on a
|
||
// previous run. Re-indexing only happens when a scrape task is
|
||
// explicitly enqueued (e.g. via the admin UI or API).
|
||
if r.deps.SearchIndex.BookExists(ctx, entry.Slug) {
|
||
skipped++
|
||
continue
|
||
}
|
||
|
||
// Random jitter between books to avoid rate-limiting.
|
||
jitter := time.Duration(1000+rand.Intn(2000)) * time.Millisecond
|
||
select {
|
||
case <-ctx.Done():
|
||
break
|
||
case <-time.After(jitter):
|
||
}
|
||
|
||
meta, err := r.deps.Novel.ScrapeMetadata(ctx, entry.URL)
|
||
if err != nil {
|
||
log.Warn("runner: catalogue refresh: metadata scrape failed",
|
||
"url", entry.URL, "err", err)
|
||
errCount++
|
||
continue
|
||
}
|
||
|
||
// Rewrite cover URL to backend proxy path so UI never hits CDN directly.
|
||
originalCover := meta.Cover
|
||
meta.Cover = fmt.Sprintf("/api/cover/novelfire.net/%s", meta.Slug)
|
||
|
||
// Persist to PocketBase.
|
||
if err := r.deps.BookWriter.WriteMetadata(ctx, meta); err != nil {
|
||
log.Warn("runner: catalogue refresh: WriteMetadata failed",
|
||
"slug", meta.Slug, "err", err)
|
||
errCount++
|
||
continue
|
||
}
|
||
|
||
// Index in Meilisearch.
|
||
if err := r.deps.SearchIndex.UpsertBook(ctx, meta); err != nil {
|
||
log.Warn("runner: catalogue refresh: UpsertBook failed",
|
||
"slug", meta.Slug, "err", err)
|
||
// non-fatal — continue
|
||
}
|
||
|
||
// Download and store cover image in MinIO if we have a cover URL
|
||
// and a CoverStore is wired in.
|
||
if r.deps.CoverStore != nil && originalCover != "" {
|
||
if !r.deps.CoverStore.CoverExists(ctx, meta.Slug) {
|
||
if err := r.downloadCover(ctx, meta.Slug, originalCover); err != nil {
|
||
log.Warn("runner: catalogue refresh: cover download failed",
|
||
"slug", meta.Slug, "url", originalCover, "err", err)
|
||
// non-fatal
|
||
}
|
||
}
|
||
}
|
||
|
||
ok++
|
||
if ok%100 == 0 {
|
||
log.Info("runner: catalogue refresh progress",
|
||
"scraped", ok, "errors", errCount)
|
||
}
|
||
}
|
||
|
||
if err := <-errCh; err != nil {
|
||
log.Warn("runner: catalogue refresh: catalogue stream error", "err", err)
|
||
}
|
||
|
||
log.Info("runner: catalogue refresh finished",
|
||
"ok", ok, "skipped", skipped, "errors", errCount)
|
||
}
|
||
|
||
// downloadCover fetches the cover image from coverURL and stores it in MinIO
|
||
// under covers/{slug}.jpg. It retries up to 3 times with exponential backoff
|
||
// on transient errors (5xx, network failures).
|
||
func (r *Runner) downloadCover(ctx context.Context, slug, coverURL string) error {
|
||
const maxRetries = 3
|
||
delay := 2 * time.Second
|
||
|
||
var lastErr error
|
||
for attempt := 0; attempt < maxRetries; attempt++ {
|
||
if ctx.Err() != nil {
|
||
return ctx.Err()
|
||
}
|
||
if attempt > 0 {
|
||
select {
|
||
case <-ctx.Done():
|
||
return ctx.Err()
|
||
case <-time.After(delay):
|
||
}
|
||
delay *= 2
|
||
}
|
||
|
||
data, err := fetchCoverBytes(ctx, coverURL)
|
||
if err != nil {
|
||
lastErr = err
|
||
continue
|
||
}
|
||
|
||
if err := r.deps.CoverStore.PutCover(ctx, slug, data, ""); err != nil {
|
||
return fmt.Errorf("put cover: %w", err)
|
||
}
|
||
return nil
|
||
}
|
||
return fmt.Errorf("download cover after %d retries: %w", maxRetries, lastErr)
|
||
}
|
||
|
||
// fetchCoverBytes performs a single HTTP GET for coverURL and returns the body.
|
||
func fetchCoverBytes(ctx context.Context, coverURL string) ([]byte, error) {
|
||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, coverURL, nil)
|
||
if err != nil {
|
||
return nil, fmt.Errorf("build request: %w", err)
|
||
}
|
||
req.Header.Set("User-Agent", "Mozilla/5.0 (compatible; libnovel-runner/2)")
|
||
req.Header.Set("Referer", "https://novelfire.net/")
|
||
|
||
client := &http.Client{Timeout: 30 * time.Second}
|
||
resp, err := client.Do(req)
|
||
if err != nil {
|
||
return nil, fmt.Errorf("http get: %w", err)
|
||
}
|
||
defer resp.Body.Close()
|
||
|
||
if resp.StatusCode >= 500 {
|
||
_, _ = io.Copy(io.Discard, resp.Body)
|
||
return nil, fmt.Errorf("upstream %d for %s", resp.StatusCode, coverURL)
|
||
}
|
||
if resp.StatusCode != http.StatusOK {
|
||
_, _ = io.Copy(io.Discard, resp.Body)
|
||
return nil, fmt.Errorf("unexpected status %d for %s", resp.StatusCode, coverURL)
|
||
}
|
||
|
||
return io.ReadAll(io.LimitReader(resp.Body, 5<<20)) // 5 MiB cap
|
||
}
|