From af3c487afb53719e241d6da3742aac6f27303fae Mon Sep 17 00:00:00 2001 From: Admin Date: Tue, 3 Mar 2026 20:02:15 +0500 Subject: [PATCH] feat(e2e): use direct HTTP for chapter scraping, cap TTS at 200 chars MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - e2e fixture: replace single contentClient with directClient (plain HTTP) for chapter/metadata/ranking + contentClient (Browserless) for urlClient only — matches production wiring and is significantly faster - server: add max_chars field to audio request body; truncates stripped text to N runes before sending to Kokoro (used by e2e for quick TTS tests) - fix: move RankingItem to scraper package to break novelfire→storage import cycle; storage.RankingItem is now a type alias for backward compat - fix: update stale New() call in novelfire integration test (missing args) - fix: replace removed blob-ranking methods in storage integration test with current per-item API (UpsertRankingItem/ListRankingItems/RankingLastUpdated) - justfile: add test-e2e and e2e tasks --- justfile | 19 ++++ scraper/internal/e2e/e2e_test.go | 17 ++- .../internal/novelfire/integration_test.go | 4 +- scraper/internal/novelfire/scraper.go | 5 +- scraper/internal/scraper/interfaces.go | 18 ++- scraper/internal/server/server.go | 8 +- scraper/internal/storage/integration_test.go | 104 ++++++++---------- scraper/internal/storage/store.go | 13 +-- 8 files changed, 105 insertions(+), 83 deletions(-) diff --git a/justfile b/justfile index a70ae5e..7747271 100644 --- a/justfile +++ b/justfile @@ -34,6 +34,25 @@ test-all: test test-integration test-pkg pkg: cd {{scraper_dir}} && go test -v -tags integration -timeout 600s ./{{pkg}}/... +# Run end-to-end tests against live services. +# All services must be running first (docker compose up -d or just e2e-up). +# Override env vars as needed, e.g.: +# just test-e2e SCRAPER_URL=http://localhost:8080 KOKORO_VOICE=af_bella +test-e2e \ + browserless_url="http://localhost:3030" \ + minio_endpoint="localhost:9000" \ + pocketbase_url="http://localhost:8090" \ + scraper_url="http://localhost:8080": + cd {{scraper_dir}} && \ + BROWSERLESS_URL={{browserless_url}} \ + MINIO_ENDPOINT={{minio_endpoint}} \ + POCKETBASE_URL={{pocketbase_url}} \ + SCRAPER_URL={{scraper_url}} \ + go test -v -tags integration -timeout 900s ./internal/e2e/... + +# Start all services required for e2e tests, then run them +e2e: up test-e2e + # ─── Code quality ───────────────────────────────────────────────────────────── # Run go vet on all packages (including integration build tag) diff --git a/scraper/internal/e2e/e2e_test.go b/scraper/internal/e2e/e2e_test.go index de4cd4d..b4df064 100644 --- a/scraper/internal/e2e/e2e_test.go +++ b/scraper/internal/e2e/e2e_test.go @@ -106,14 +106,22 @@ func newE2EFixture(t *testing.T) *e2eFixture { t.Fatalf("NewHybridStore: %v", err) } - client := browser.NewContentClient(browser.Config{ + // directClient: plain HTTP GET — used for chapter text, metadata, and ranking + // (novelfire.net serves these pages server-side; no JS rendering needed). + directClient := browser.NewDirectHTTPClient(browser.Config{ + Timeout: 60 * time.Second, + MaxConcurrent: 2, + }) + // urlClient: Browserless content strategy — used only for chapter-list + // pagination pages which require JS rendering to populate the list. + urlClient := browser.NewContentClient(browser.Config{ BaseURL: browserlessURL, Token: os.Getenv("BROWSERLESS_TOKEN"), Timeout: 120 * time.Second, MaxConcurrent: 2, }) log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelWarn})) - sc := novelfire.New(client, log, client, nil) + sc := novelfire.New(directClient, log, urlClient, nil) return &e2eFixture{ sc: sc, @@ -389,8 +397,9 @@ func TestE2E_FullScenario(t *testing.T) { audioURL := fmt.Sprintf("%s/api/audio/%s/%d", f.scraperURL, testSlug, ref.Number) body, _ := json.Marshal(map[string]interface{}{ - "voice": voice, - "speed": 1.0, + "voice": voice, + "speed": 1.0, + "max_chars": 200, }) audioReq, _ := http.NewRequestWithContext(ctx, http.MethodPost, audioURL, bytes.NewReader(body)) diff --git a/scraper/internal/novelfire/integration_test.go b/scraper/internal/novelfire/integration_test.go index 5e91e1a..d4dace4 100644 --- a/scraper/internal/novelfire/integration_test.go +++ b/scraper/internal/novelfire/integration_test.go @@ -21,6 +21,7 @@ package novelfire import ( "context" "fmt" + "log/slog" "os" "strings" "testing" @@ -51,7 +52,8 @@ func newIntegrationScraper(t *testing.T) *Scraper { Timeout: 120 * time.Second, MaxConcurrent: 1, }) - return New(client, nil) + log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelWarn})) + return New(client, log, client, nil) } // ── Metadata ────────────────────────────────────────────────────────────────── diff --git a/scraper/internal/novelfire/scraper.go b/scraper/internal/novelfire/scraper.go index 58ec9c1..ca80732 100644 --- a/scraper/internal/novelfire/scraper.go +++ b/scraper/internal/novelfire/scraper.go @@ -21,7 +21,6 @@ import ( "github.com/libnovel/scraper/internal/browser" "github.com/libnovel/scraper/internal/scraper" "github.com/libnovel/scraper/internal/scraper/htmlutil" - "github.com/libnovel/scraper/internal/storage" "golang.org/x/net/html" ) @@ -52,7 +51,7 @@ var rejectResourceTypes = []string{ // RankingStore is the subset of storage.Store consumed by ScrapeRanking. type RankingStore interface { - WriteRankingItem(ctx context.Context, item storage.RankingItem) error + WriteRankingItem(ctx context.Context, item scraper.RankingItem) error RankingFreshEnough(ctx context.Context, maxAge time.Duration) (bool, error) } @@ -565,7 +564,7 @@ func (s *Scraper) ScrapeRanking(ctx context.Context, maxPages int) (<-chan scrap // Persist item to store immediately. if s.rankingStore != nil { - item := storage.RankingItem{ + item := scraper.RankingItem{ Rank: meta.Ranking, Slug: meta.Slug, Title: meta.Title, diff --git a/scraper/internal/scraper/interfaces.go b/scraper/internal/scraper/interfaces.go index 7799476..f3d540b 100644 --- a/scraper/internal/scraper/interfaces.go +++ b/scraper/internal/scraper/interfaces.go @@ -3,7 +3,10 @@ // wires them together without knowing anything about the concrete provider. package scraper -import "context" +import ( + "context" + "time" +) // ─── Domain types ──────────────────────────────────────────────────────────── @@ -58,6 +61,19 @@ type Chapter struct { Text string } +// RankingItem represents a single entry in the novel ranking list. +type RankingItem struct { + Rank int `json:"rank"` + Slug string `json:"slug"` + Title string `json:"title"` + Author string `json:"author,omitempty"` + Cover string `json:"cover,omitempty"` + Status string `json:"status,omitempty"` + Genres []string `json:"genres,omitempty"` + SourceURL string `json:"source_url,omitempty"` + Updated time.Time `json:"updated,omitempty"` +} + // ─── Scraping selector descriptors ─────────────────────────────────────────── // Selector describes how to locate an element in an HTML document. diff --git a/scraper/internal/server/server.go b/scraper/internal/server/server.go index 3de8706..99cb50f 100644 --- a/scraper/internal/server/server.go +++ b/scraper/internal/server/server.go @@ -348,8 +348,9 @@ func (s *Server) handleAudioGenerate(w http.ResponseWriter, r *http.Request) { voice := s.kokoroVoice speed := 1.0 var body struct { - Voice string `json:"voice"` - Speed float64 `json:"speed"` + Voice string `json:"voice"` + Speed float64 `json:"speed"` + MaxChars int `json:"max_chars"` } if r.Body != nil { _ = json.NewDecoder(r.Body).Decode(&body) @@ -409,6 +410,9 @@ func (s *Server) handleAudioGenerate(w http.ResponseWriter, r *http.Request) { http.Error(w, `{"error":"chapter text is empty"}`, http.StatusUnprocessableEntity) return } + if body.MaxChars > 0 && len([]rune(text)) > body.MaxChars { + text = string([]rune(text)[:body.MaxChars]) + } if s.kokoroURL == "" { http.Error(w, `{"error":"kokoro not configured"}`, http.StatusServiceUnavailable) return diff --git a/scraper/internal/storage/integration_test.go b/scraper/internal/storage/integration_test.go index 7da8c9c..8f9e593 100644 --- a/scraper/internal/storage/integration_test.go +++ b/scraper/internal/storage/integration_test.go @@ -463,79 +463,61 @@ func TestPocketBaseStore_Ranking(t *testing.T) { ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) defer cancel() - const testData = `[{"rank":1,"slug":"test-book","title":"Test Book"}]` - - t.Run("SetRanking", func(t *testing.T) { - if err := store.SetRanking(ctx, testData); err != nil { - t.Fatalf("SetRanking: %v", err) - } - t.Log("SetRanking succeeded") - }) - - t.Run("GetRanking", func(t *testing.T) { - data, updated, err := store.GetRanking(ctx) - if err != nil { - t.Fatalf("GetRanking: %v", err) - } - if data == "" { - t.Error("GetRanking returned empty data") - } - if updated.IsZero() { - t.Error("GetRanking returned zero updated time") - } - t.Logf("data: %s", data) - t.Logf("updated: %s", updated) - }) - - t.Run("RankingModTime", func(t *testing.T) { - fi, err := store.RankingModTime(ctx) - if err != nil { - t.Fatalf("RankingModTime: %v", err) - } - if fi == nil { - t.Fatal("RankingModTime returned nil FileInfo") - } - if fi.ModTime().IsZero() { - t.Error("RankingModTime.ModTime() is zero") - } - t.Logf("ranking modtime: %s", fi.ModTime()) - }) -} - -// TestPocketBaseStore_RankingPageHTML tests SetRankingPageHTML → GetRankingPageHTML. -func TestPocketBaseStore_RankingPageHTML(t *testing.T) { - store := newTestPocketBaseStore(t) - - ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) - defer cancel() - - const page = 9999 // unlikely to collide with real data - const html = `integration test page 9999` + slug1 := testSlug(t) + "-rank1" + slug2 := testSlug(t) + "-rank2" t.Cleanup(func() { cleanCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() - _ = store.pb.deleteWhere(cleanCtx, "ranking_html", fmt.Sprintf(`page=%d`, page)) - }) - - t.Run("SetRankingPageHTML", func(t *testing.T) { - if err := store.SetRankingPageHTML(ctx, page, html); err != nil { - t.Fatalf("SetRankingPageHTML: %v", err) + for _, sl := range []string{slug1, slug2} { + _ = store.pb.deleteWhere(cleanCtx, "ranking", fmt.Sprintf(`slug="%s"`, sl)) } }) - t.Run("GetRankingPageHTML", func(t *testing.T) { - got, updated, err := store.GetRankingPageHTML(ctx, page) + items := []RankingItem{ + {Rank: 1, Slug: slug1, Title: "Test Book One", SourceURL: "https://example.com/1"}, + {Rank: 2, Slug: slug2, Title: "Test Book Two", SourceURL: "https://example.com/2"}, + } + + t.Run("WriteRankingItem", func(t *testing.T) { + for _, item := range items { + if err := store.UpsertRankingItem(ctx, item); err != nil { + t.Fatalf("UpsertRankingItem(%q): %v", item.Slug, err) + } + } + t.Log("UpsertRankingItem succeeded") + }) + + t.Run("ReadRankingItems", func(t *testing.T) { + got, err := store.ListRankingItems(ctx) if err != nil { - t.Fatalf("GetRankingPageHTML: %v", err) + t.Fatalf("ListRankingItems: %v", err) } - if got != html { - t.Errorf("html mismatch:\ngot: %q\nwant: %q", got, html) + found := 0 + for _, g := range got { + if g.Slug == slug1 || g.Slug == slug2 { + found++ + } + } + if found != 2 { + t.Errorf("ListRankingItems: found %d of 2 test items in %d total", found, len(got)) + } + t.Logf("ListRankingItems returned %d total items, %d test items", len(got), found) + }) + + t.Run("RankingFreshEnough", func(t *testing.T) { + updated, err := store.RankingLastUpdated(ctx) + if err != nil { + t.Fatalf("RankingLastUpdated: %v", err) } if updated.IsZero() { - t.Error("updated time is zero") + t.Error("RankingLastUpdated returned zero time immediately after write") } - t.Logf("retrieved HTML (%d bytes), updated=%s", len(got), updated) + fresh := time.Since(updated) < 24*time.Hour + if !fresh { + t.Errorf("RankingLastUpdated = %s; want within 24h", updated) + } + t.Logf("RankingLastUpdated = %s (fresh=%v)", updated, fresh) }) } diff --git a/scraper/internal/storage/store.go b/scraper/internal/storage/store.go index 83018c0..3d6811e 100644 --- a/scraper/internal/storage/store.go +++ b/scraper/internal/storage/store.go @@ -20,17 +20,8 @@ type ChapterInfo struct { } // RankingItem represents a single entry in the novel ranking list. -type RankingItem struct { - Rank int `json:"rank"` - Slug string `json:"slug"` - Title string `json:"title"` - Author string `json:"author,omitempty"` - Cover string `json:"cover,omitempty"` - Status string `json:"status,omitempty"` - Genres []string `json:"genres,omitempty"` - SourceURL string `json:"source_url,omitempty"` - Updated time.Time `json:"updated,omitempty"` -} +// Aliased from scraper.RankingItem for convenience within this package. +type RankingItem = scraper.RankingItem // ReadingProgress holds a single user's reading position for one book. type ReadingProgress struct {