feat(e2e): use direct HTTP for chapter scraping, cap TTS at 200 chars

- e2e fixture: replace single contentClient with directClient (plain HTTP)
  for chapter/metadata/ranking + contentClient (Browserless) for urlClient
  only — matches production wiring and is significantly faster
- server: add max_chars field to audio request body; truncates stripped text
  to N runes before sending to Kokoro (used by e2e for quick TTS tests)
- fix: move RankingItem to scraper package to break novelfire→storage import
  cycle; storage.RankingItem is now a type alias for backward compat
- fix: update stale New() call in novelfire integration test (missing args)
- fix: replace removed blob-ranking methods in storage integration test with
  current per-item API (UpsertRankingItem/ListRankingItems/RankingLastUpdated)
- justfile: add test-e2e and e2e tasks
This commit is contained in:
Admin
2026-03-03 20:02:15 +05:00
parent b8d4d94b18
commit af3c487afb
8 changed files with 105 additions and 83 deletions

View File

@@ -34,6 +34,25 @@ test-all: test test-integration
test-pkg pkg: test-pkg pkg:
cd {{scraper_dir}} && go test -v -tags integration -timeout 600s ./{{pkg}}/... cd {{scraper_dir}} && go test -v -tags integration -timeout 600s ./{{pkg}}/...
# Run end-to-end tests against live services.
# All services must be running first (docker compose up -d or just e2e-up).
# Override env vars as needed, e.g.:
# just test-e2e SCRAPER_URL=http://localhost:8080 KOKORO_VOICE=af_bella
test-e2e \
browserless_url="http://localhost:3030" \
minio_endpoint="localhost:9000" \
pocketbase_url="http://localhost:8090" \
scraper_url="http://localhost:8080":
cd {{scraper_dir}} && \
BROWSERLESS_URL={{browserless_url}} \
MINIO_ENDPOINT={{minio_endpoint}} \
POCKETBASE_URL={{pocketbase_url}} \
SCRAPER_URL={{scraper_url}} \
go test -v -tags integration -timeout 900s ./internal/e2e/...
# Start all services required for e2e tests, then run them
e2e: up test-e2e
# ─── Code quality ───────────────────────────────────────────────────────────── # ─── Code quality ─────────────────────────────────────────────────────────────
# Run go vet on all packages (including integration build tag) # Run go vet on all packages (including integration build tag)

View File

@@ -106,14 +106,22 @@ func newE2EFixture(t *testing.T) *e2eFixture {
t.Fatalf("NewHybridStore: %v", err) t.Fatalf("NewHybridStore: %v", err)
} }
client := browser.NewContentClient(browser.Config{ // directClient: plain HTTP GET — used for chapter text, metadata, and ranking
// (novelfire.net serves these pages server-side; no JS rendering needed).
directClient := browser.NewDirectHTTPClient(browser.Config{
Timeout: 60 * time.Second,
MaxConcurrent: 2,
})
// urlClient: Browserless content strategy — used only for chapter-list
// pagination pages which require JS rendering to populate the list.
urlClient := browser.NewContentClient(browser.Config{
BaseURL: browserlessURL, BaseURL: browserlessURL,
Token: os.Getenv("BROWSERLESS_TOKEN"), Token: os.Getenv("BROWSERLESS_TOKEN"),
Timeout: 120 * time.Second, Timeout: 120 * time.Second,
MaxConcurrent: 2, MaxConcurrent: 2,
}) })
log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelWarn})) log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelWarn}))
sc := novelfire.New(client, log, client, nil) sc := novelfire.New(directClient, log, urlClient, nil)
return &e2eFixture{ return &e2eFixture{
sc: sc, sc: sc,
@@ -391,6 +399,7 @@ func TestE2E_FullScenario(t *testing.T) {
body, _ := json.Marshal(map[string]interface{}{ body, _ := json.Marshal(map[string]interface{}{
"voice": voice, "voice": voice,
"speed": 1.0, "speed": 1.0,
"max_chars": 200,
}) })
audioReq, _ := http.NewRequestWithContext(ctx, http.MethodPost, audioURL, bytes.NewReader(body)) audioReq, _ := http.NewRequestWithContext(ctx, http.MethodPost, audioURL, bytes.NewReader(body))

View File

@@ -21,6 +21,7 @@ package novelfire
import ( import (
"context" "context"
"fmt" "fmt"
"log/slog"
"os" "os"
"strings" "strings"
"testing" "testing"
@@ -51,7 +52,8 @@ func newIntegrationScraper(t *testing.T) *Scraper {
Timeout: 120 * time.Second, Timeout: 120 * time.Second,
MaxConcurrent: 1, MaxConcurrent: 1,
}) })
return New(client, nil) log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelWarn}))
return New(client, log, client, nil)
} }
// ── Metadata ────────────────────────────────────────────────────────────────── // ── Metadata ──────────────────────────────────────────────────────────────────

View File

@@ -21,7 +21,6 @@ import (
"github.com/libnovel/scraper/internal/browser" "github.com/libnovel/scraper/internal/browser"
"github.com/libnovel/scraper/internal/scraper" "github.com/libnovel/scraper/internal/scraper"
"github.com/libnovel/scraper/internal/scraper/htmlutil" "github.com/libnovel/scraper/internal/scraper/htmlutil"
"github.com/libnovel/scraper/internal/storage"
"golang.org/x/net/html" "golang.org/x/net/html"
) )
@@ -52,7 +51,7 @@ var rejectResourceTypes = []string{
// RankingStore is the subset of storage.Store consumed by ScrapeRanking. // RankingStore is the subset of storage.Store consumed by ScrapeRanking.
type RankingStore interface { type RankingStore interface {
WriteRankingItem(ctx context.Context, item storage.RankingItem) error WriteRankingItem(ctx context.Context, item scraper.RankingItem) error
RankingFreshEnough(ctx context.Context, maxAge time.Duration) (bool, error) RankingFreshEnough(ctx context.Context, maxAge time.Duration) (bool, error)
} }
@@ -565,7 +564,7 @@ func (s *Scraper) ScrapeRanking(ctx context.Context, maxPages int) (<-chan scrap
// Persist item to store immediately. // Persist item to store immediately.
if s.rankingStore != nil { if s.rankingStore != nil {
item := storage.RankingItem{ item := scraper.RankingItem{
Rank: meta.Ranking, Rank: meta.Ranking,
Slug: meta.Slug, Slug: meta.Slug,
Title: meta.Title, Title: meta.Title,

View File

@@ -3,7 +3,10 @@
// wires them together without knowing anything about the concrete provider. // wires them together without knowing anything about the concrete provider.
package scraper package scraper
import "context" import (
"context"
"time"
)
// ─── Domain types ──────────────────────────────────────────────────────────── // ─── Domain types ────────────────────────────────────────────────────────────
@@ -58,6 +61,19 @@ type Chapter struct {
Text string Text string
} }
// RankingItem represents a single entry in the novel ranking list.
type RankingItem struct {
Rank int `json:"rank"`
Slug string `json:"slug"`
Title string `json:"title"`
Author string `json:"author,omitempty"`
Cover string `json:"cover,omitempty"`
Status string `json:"status,omitempty"`
Genres []string `json:"genres,omitempty"`
SourceURL string `json:"source_url,omitempty"`
Updated time.Time `json:"updated,omitempty"`
}
// ─── Scraping selector descriptors ─────────────────────────────────────────── // ─── Scraping selector descriptors ───────────────────────────────────────────
// Selector describes how to locate an element in an HTML document. // Selector describes how to locate an element in an HTML document.

View File

@@ -350,6 +350,7 @@ func (s *Server) handleAudioGenerate(w http.ResponseWriter, r *http.Request) {
var body struct { var body struct {
Voice string `json:"voice"` Voice string `json:"voice"`
Speed float64 `json:"speed"` Speed float64 `json:"speed"`
MaxChars int `json:"max_chars"`
} }
if r.Body != nil { if r.Body != nil {
_ = json.NewDecoder(r.Body).Decode(&body) _ = json.NewDecoder(r.Body).Decode(&body)
@@ -409,6 +410,9 @@ func (s *Server) handleAudioGenerate(w http.ResponseWriter, r *http.Request) {
http.Error(w, `{"error":"chapter text is empty"}`, http.StatusUnprocessableEntity) http.Error(w, `{"error":"chapter text is empty"}`, http.StatusUnprocessableEntity)
return return
} }
if body.MaxChars > 0 && len([]rune(text)) > body.MaxChars {
text = string([]rune(text)[:body.MaxChars])
}
if s.kokoroURL == "" { if s.kokoroURL == "" {
http.Error(w, `{"error":"kokoro not configured"}`, http.StatusServiceUnavailable) http.Error(w, `{"error":"kokoro not configured"}`, http.StatusServiceUnavailable)
return return

View File

@@ -463,79 +463,61 @@ func TestPocketBaseStore_Ranking(t *testing.T) {
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
defer cancel() defer cancel()
const testData = `[{"rank":1,"slug":"test-book","title":"Test Book"}]` slug1 := testSlug(t) + "-rank1"
slug2 := testSlug(t) + "-rank2"
t.Run("SetRanking", func(t *testing.T) {
if err := store.SetRanking(ctx, testData); err != nil {
t.Fatalf("SetRanking: %v", err)
}
t.Log("SetRanking succeeded")
})
t.Run("GetRanking", func(t *testing.T) {
data, updated, err := store.GetRanking(ctx)
if err != nil {
t.Fatalf("GetRanking: %v", err)
}
if data == "" {
t.Error("GetRanking returned empty data")
}
if updated.IsZero() {
t.Error("GetRanking returned zero updated time")
}
t.Logf("data: %s", data)
t.Logf("updated: %s", updated)
})
t.Run("RankingModTime", func(t *testing.T) {
fi, err := store.RankingModTime(ctx)
if err != nil {
t.Fatalf("RankingModTime: %v", err)
}
if fi == nil {
t.Fatal("RankingModTime returned nil FileInfo")
}
if fi.ModTime().IsZero() {
t.Error("RankingModTime.ModTime() is zero")
}
t.Logf("ranking modtime: %s", fi.ModTime())
})
}
// TestPocketBaseStore_RankingPageHTML tests SetRankingPageHTML → GetRankingPageHTML.
func TestPocketBaseStore_RankingPageHTML(t *testing.T) {
store := newTestPocketBaseStore(t)
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
defer cancel()
const page = 9999 // unlikely to collide with real data
const html = `<html><body>integration test page 9999</body></html>`
t.Cleanup(func() { t.Cleanup(func() {
cleanCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second) cleanCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
defer cancel() defer cancel()
_ = store.pb.deleteWhere(cleanCtx, "ranking_html", fmt.Sprintf(`page=%d`, page)) for _, sl := range []string{slug1, slug2} {
}) _ = store.pb.deleteWhere(cleanCtx, "ranking", fmt.Sprintf(`slug="%s"`, sl))
t.Run("SetRankingPageHTML", func(t *testing.T) {
if err := store.SetRankingPageHTML(ctx, page, html); err != nil {
t.Fatalf("SetRankingPageHTML: %v", err)
} }
}) })
t.Run("GetRankingPageHTML", func(t *testing.T) { items := []RankingItem{
got, updated, err := store.GetRankingPageHTML(ctx, page) {Rank: 1, Slug: slug1, Title: "Test Book One", SourceURL: "https://example.com/1"},
{Rank: 2, Slug: slug2, Title: "Test Book Two", SourceURL: "https://example.com/2"},
}
t.Run("WriteRankingItem", func(t *testing.T) {
for _, item := range items {
if err := store.UpsertRankingItem(ctx, item); err != nil {
t.Fatalf("UpsertRankingItem(%q): %v", item.Slug, err)
}
}
t.Log("UpsertRankingItem succeeded")
})
t.Run("ReadRankingItems", func(t *testing.T) {
got, err := store.ListRankingItems(ctx)
if err != nil { if err != nil {
t.Fatalf("GetRankingPageHTML: %v", err) t.Fatalf("ListRankingItems: %v", err)
} }
if got != html { found := 0
t.Errorf("html mismatch:\ngot: %q\nwant: %q", got, html) for _, g := range got {
if g.Slug == slug1 || g.Slug == slug2 {
found++
}
}
if found != 2 {
t.Errorf("ListRankingItems: found %d of 2 test items in %d total", found, len(got))
}
t.Logf("ListRankingItems returned %d total items, %d test items", len(got), found)
})
t.Run("RankingFreshEnough", func(t *testing.T) {
updated, err := store.RankingLastUpdated(ctx)
if err != nil {
t.Fatalf("RankingLastUpdated: %v", err)
} }
if updated.IsZero() { if updated.IsZero() {
t.Error("updated time is zero") t.Error("RankingLastUpdated returned zero time immediately after write")
} }
t.Logf("retrieved HTML (%d bytes), updated=%s", len(got), updated) fresh := time.Since(updated) < 24*time.Hour
if !fresh {
t.Errorf("RankingLastUpdated = %s; want within 24h", updated)
}
t.Logf("RankingLastUpdated = %s (fresh=%v)", updated, fresh)
}) })
} }

View File

@@ -20,17 +20,8 @@ type ChapterInfo struct {
} }
// RankingItem represents a single entry in the novel ranking list. // RankingItem represents a single entry in the novel ranking list.
type RankingItem struct { // Aliased from scraper.RankingItem for convenience within this package.
Rank int `json:"rank"` type RankingItem = scraper.RankingItem
Slug string `json:"slug"`
Title string `json:"title"`
Author string `json:"author,omitempty"`
Cover string `json:"cover,omitempty"`
Status string `json:"status,omitempty"`
Genres []string `json:"genres,omitempty"`
SourceURL string `json:"source_url,omitempty"`
Updated time.Time `json:"updated,omitempty"`
}
// ReadingProgress holds a single user's reading position for one book. // ReadingProgress holds a single user's reading position for one book.
type ReadingProgress struct { type ReadingProgress struct {