feat(scraper): harden browser headers and add proxy support
All checks were successful
CI / Scraper / Lint (push) Successful in 8s
CI / Scraper / Lint (pull_request) Successful in 12s
CI / Scraper / Test (pull_request) Successful in 8s
CI / Scraper / Test (push) Successful in 16s
Release / Scraper / Test (push) Successful in 8s
Release / UI / Build (push) Successful in 24s
CI / Scraper / Docker Push (pull_request) Has been skipped
CI / UI / Build (pull_request) Successful in 30s
CI / Scraper / Docker Push (push) Successful in 39s
CI / UI / Docker Push (pull_request) Has been skipped
Release / UI / Docker (push) Successful in 40s
Release / Scraper / Docker (push) Successful in 50s
iOS CI / Build (pull_request) Successful in 3m7s
iOS CI / Test (pull_request) Successful in 5m23s

Upgrade DirectHTTPClient to send a full Chrome 124 header set
(Sec-Fetch-*, Accept-Encoding with gzip decompression, Referer,
Cache-Control) to reduce bot-detection false positives on WAFs.

Add SCRAPER_PROXY env var to route all outbound scrape requests
through a configurable proxy (residential or otherwise); falls back
to the standard HTTP_PROXY / HTTPS_PROXY env vars.
This commit is contained in:
Admin
2026-03-14 18:40:32 +05:00
parent 02705dc6ed
commit 1642434a79
2 changed files with 78 additions and 8 deletions

View File

@@ -1,10 +1,14 @@
package browser
import (
"compress/gzip"
"context"
"fmt"
"io"
"net/http"
"net/url"
"os"
"strings"
"time"
)
@@ -18,11 +22,41 @@ func NewDirectHTTPClient(cfg Config) BrowserClient {
if cfg.Timeout == 0 {
cfg.Timeout = 30 * time.Second
}
return &httpClient{
cfg: cfg,
http: &http.Client{Timeout: cfg.Timeout},
sem: makeSem(cfg.MaxConcurrent),
transport := http.DefaultTransport.(*http.Transport).Clone()
// Wire in proxy from environment (HTTP_PROXY / HTTPS_PROXY / NO_PROXY).
// This lets operators route traffic through a residential proxy by simply
// setting HTTPS_PROXY=http://user:pass@proxy-host:port without any code
// changes — the standard approach for bypassing datacenter IP blocks.
if proxyURL := proxyFromEnv(); proxyURL != nil {
transport.Proxy = http.ProxyURL(proxyURL)
} else {
transport.Proxy = http.ProxyFromEnvironment
}
return &httpClient{
cfg: cfg,
http: &http.Client{
Timeout: cfg.Timeout,
Transport: transport,
},
sem: makeSem(cfg.MaxConcurrent),
}
}
// proxyFromEnv returns an explicit proxy URL if SCRAPER_PROXY is set, otherwise
// nil (and http.ProxyFromEnvironment handles the standard HTTP_PROXY / HTTPS_PROXY).
func proxyFromEnv() *url.URL {
raw := os.Getenv("SCRAPER_PROXY")
if raw == "" {
return nil
}
u, err := url.Parse(raw)
if err != nil || u.Host == "" {
return nil
}
return u
}
func (c *httpClient) Strategy() Strategy { return StrategyDirect }
@@ -37,9 +71,25 @@ func (c *httpClient) GetContent(ctx context.Context, req ContentRequest) (string
if err != nil {
return "", fmt.Errorf("http: build request: %w", err)
}
httpReq.Header.Set("User-Agent", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36")
httpReq.Header.Set("Accept", "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8")
httpReq.Header.Set("Accept-Language", "en-US,en;q=0.5")
// Mimic a real Chrome browser request to reduce bot-detection likelihood.
// These headers match what Chrome 124 sends for a top-level navigation.
httpReq.Header.Set("User-Agent", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36")
httpReq.Header.Set("Accept", "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7")
httpReq.Header.Set("Accept-Language", "en-US,en;q=0.9")
httpReq.Header.Set("Accept-Encoding", "gzip, deflate, br")
httpReq.Header.Set("Connection", "keep-alive")
httpReq.Header.Set("Upgrade-Insecure-Requests", "1")
httpReq.Header.Set("Sec-Fetch-Dest", "document")
httpReq.Header.Set("Sec-Fetch-Mode", "navigate")
httpReq.Header.Set("Sec-Fetch-Site", "none")
httpReq.Header.Set("Sec-Fetch-User", "?1")
httpReq.Header.Set("Cache-Control", "max-age=0")
// Set Referer for subsequent page requests (anything that is not the root).
if parsed, pErr := url.Parse(req.URL); pErr == nil && parsed.Path != "" && parsed.Path != "/" {
httpReq.Header.Set("Referer", parsed.Scheme+"://"+parsed.Host+"/")
}
resp, err := c.http.Do(httpReq)
if err != nil {
@@ -52,7 +102,22 @@ func (c *httpClient) GetContent(ctx context.Context, req ContentRequest) (string
return "", fmt.Errorf("http: unexpected status %d: %s", resp.StatusCode, b)
}
raw, err := io.ReadAll(resp.Body)
// Decompress gzip/br responses when the server honours Accept-Encoding.
// net/http decompresses gzip automatically only when it sets the header
// itself; since we set Accept-Encoding explicitly we must do it ourselves.
body := resp.Body
if strings.EqualFold(resp.Header.Get("Content-Encoding"), "gzip") {
gr, gzErr := gzip.NewReader(resp.Body)
if gzErr != nil {
return "", fmt.Errorf("http: gzip reader: %w", gzErr)
}
defer gr.Close()
body = gr
}
// br (Brotli) decompression requires an external package; skip for now —
// the server will fall back to gzip or plain text for unknown encodings.
raw, err := io.ReadAll(body)
if err != nil {
return "", fmt.Errorf("http: read body: %w", err)
}