This commit is contained in:
toom1996
2026-09-20 00:44:47 +08:00
parent d8f667e391
commit fde57984e7
16 changed files with 3908 additions and 123 deletions

155
internal/fetch/fetch.go Normal file
View File

@ -0,0 +1,155 @@
// Package fetch 提供「全局并发限流 + 自动重试」的 HTTP GET,供各爬虫共用。
//
// 为什么需要它:
// - 此前 vogue / theimpression 各自实现了一份 request(),逻辑重复、错误被吞掉(失败只返回 ("", 0));
// - 更严重的是并发失控:vogue 在「品牌级」与「详情级」各开一层信号量,实际并发是 MaxCo²
// (如 5×5=25),与「最多 5 个并发」的本意不符。
//
// 本包用一个「全局信号量」统一约束进程内所有请求的并发上限,各爬虫共享同一个 Client
// 即可保证整体并发受控,不会因嵌套而放大。
package fetch
import (
"context"
"fmt"
"io"
"log"
"net/http"
"time"
)
const (
// DefaultTimeout 单次请求默认超时。
DefaultTimeout = 30 * time.Second
// DefaultMaxConcurrency 默认全局并发上限。
DefaultMaxConcurrency = 5
// DefaultRetries 默认重试次数(不含首次)。
DefaultRetries = 2
userAgent = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " +
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
acceptHeader = "text/html,application/json,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
)
// Options 构造 Client 的参数,零值字段自动回落默认。
type Options struct {
MaxConcurrency int // 全局并发上限(<=0 回落 5)
MinInterval time.Duration // 两次请求之间的最小间隔,全局限速(0 = 不限速)
Retries int // 失败重试次数,不含首次(<0 回落 2;0 表示不重试)
Backoff time.Duration // 重试基础退避,第 n 次等待 n*Backoff(<=0 回落 1s)
Timeout time.Duration // 单次请求超时(<=0 回落 30s)
}
// Client 带并发限流与重试的 HTTP 客户端。
type Client struct {
hc *http.Client
sem chan struct{} // 全局并发槽(跨所有调用方共享)
tick *time.Ticker // 非 nil 时按 MinInterval 全局限速
retries int
backoff time.Duration
}
// New 构造 Client。返回值实现了一个进程级共享的限流器,多个爬虫应复用同一个实例。
func New(opt Options) *Client {
if opt.MaxConcurrency <= 0 {
opt.MaxConcurrency = DefaultMaxConcurrency
}
if opt.Timeout <= 0 {
opt.Timeout = DefaultTimeout
}
if opt.Backoff <= 0 {
opt.Backoff = time.Second
}
if opt.Retries < 0 {
opt.Retries = DefaultRetries
}
c := &Client{
hc: &http.Client{Timeout: opt.Timeout},
sem: make(chan struct{}, opt.MaxConcurrency),
retries: opt.Retries,
backoff: opt.Backoff,
}
if opt.MinInterval > 0 {
c.tick = time.NewTicker(opt.MinInterval)
}
return c
}
// Close 停止内部限速计时器(未开启限速时为空操作)。
func (c *Client) Close() {
if c.tick != nil {
c.tick.Stop()
}
}
// Get 发起带限流与重试的 GET,返回响应体与最终 HTTP 状态码。
//
// 重试策略:网络错误 / 5xx / 429 会按「线性退避」重试(第 n 次等 n*Backoff);
// 其余 4xx 视为客户端错误(如 403 被反爬、404 页面不存在),不重试,直接返回。
// 调用方拿到非 nil error 时即可判定最终失败,无需再自行区分网络错误与状态码。
func (c *Client) Get(ctx context.Context, url string) ([]byte, int, error) {
// 全局限速:每次请求前消费一个 tick(ticker 通道容量为 1,不会无限堆积)。
if c.tick != nil {
select {
case <-c.tick.C:
case <-ctx.Done():
return nil, 0, ctx.Err()
}
}
// 全局并发上限:超过则在此排队,保证总量受控。
select {
case c.sem <- struct{}{}:
defer func() { <-c.sem }()
case <-ctx.Done():
return nil, 0, ctx.Err()
}
var lastErr error
var lastStatus int
for attempt := 0; attempt <= c.retries; attempt++ {
if attempt > 0 {
delay := c.backoff * time.Duration(attempt)
log.Printf("[fetch] 第 %d/%d 次重试 %s(等待 %s)", attempt, c.retries, url, delay)
select {
case <-time.After(delay):
case <-ctx.Done():
return nil, 0, ctx.Err()
}
}
body, status, err := c.do(ctx, url)
if err == nil && status == http.StatusOK {
return body, status, nil
}
lastErr, lastStatus = err, status
// 4xx(除 429 限流)不重试:重试也只会再次失败。
if status >= 400 && status < 500 && status != http.StatusTooManyRequests {
break
}
}
if lastErr == nil {
lastErr = fmt.Errorf("http %d", lastStatus)
}
return nil, lastStatus, lastErr
}
// do 执行单次请求。
func (c *Client) do(ctx context.Context, url string) ([]byte, int, error) {
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
if err != nil {
return nil, 0, err
}
req.Header.Set("User-Agent", userAgent)
req.Header.Set("Accept", acceptHeader)
resp, err := c.hc.Do(req)
if err != nil {
return nil, 0, err
}
defer resp.Body.Close()
body, err := io.ReadAll(resp.Body)
if err != nil {
return nil, resp.StatusCode, err
}
return body, resp.StatusCode, nil
}

View File

@ -0,0 +1,128 @@
package fetch
import (
"context"
"net/http"
"net/http/httptest"
"sync"
"sync/atomic"
"testing"
"time"
)
func TestGetSuccess(t *testing.T) {
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
_, _ = w.Write([]byte("ok"))
}))
defer srv.Close()
c := New(Options{MaxConcurrency: 1, Retries: 0})
defer c.Close()
body, code, err := c.Get(context.Background(), srv.URL)
if err != nil || code != http.StatusOK || string(body) != "ok" {
t.Fatalf("期望 200/ok,得到 code=%d body=%q err=%v", code, body, err)
}
}
// TestGetRetriesOn5xx 5xx 应重试到上限后失败(首次 + 2 次重试 = 3 次请求)。
func TestGetRetriesOn5xx(t *testing.T) {
var calls int32
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
atomic.AddInt32(&calls, 1)
w.WriteHeader(http.StatusInternalServerError)
}))
defer srv.Close()
c := New(Options{MaxConcurrency: 1, Retries: 2, Backoff: time.Millisecond})
defer c.Close()
_, code, err := c.Get(context.Background(), srv.URL)
if err == nil {
t.Fatal("500 最终应返回 error")
}
if code != http.StatusInternalServerError {
t.Fatalf("最终状态码应为 500,得到 %d", code)
}
if got := atomic.LoadInt32(&calls); got != 3 {
t.Fatalf("500 应请求 3 次(首次+2 重试),实际 %d", got)
}
}
// TestGetNoRetryOn4xx 4xx(非 429)不应重试。
func TestGetNoRetryOn4xx(t *testing.T) {
var calls int32
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
atomic.AddInt32(&calls, 1)
w.WriteHeader(http.StatusNotFound)
}))
defer srv.Close()
c := New(Options{MaxConcurrency: 1, Retries: 3, Backoff: time.Millisecond})
defer c.Close()
_, code, err := c.Get(context.Background(), srv.URL)
if err == nil || code != http.StatusNotFound {
t.Fatalf("404 应报错且状态码 404,得到 code=%d err=%v", code, err)
}
if got := atomic.LoadInt32(&calls); got != 1 {
t.Fatalf("404 不应重试(只请求 1 次),实际 %d", got)
}
}
// TestGetRetriesOn429 429(限流)应重试,后续成功则返回成功。
func TestGetRetriesOn429(t *testing.T) {
var calls int32
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if atomic.AddInt32(&calls, 1) == 1 {
w.WriteHeader(http.StatusTooManyRequests)
return
}
_, _ = w.Write([]byte("ok"))
}))
defer srv.Close()
c := New(Options{MaxConcurrency: 1, Retries: 2, Backoff: time.Millisecond})
defer c.Close()
body, code, err := c.Get(context.Background(), srv.URL)
if err != nil || code != http.StatusOK || string(body) != "ok" {
t.Fatalf("429 后重试应成功,得到 code=%d body=%q err=%v", code, body, err)
}
}
// TestConcurrencyLimit 并发上限不得被突破(修复 MaxCo² 回归的守门测试)。
func TestConcurrencyLimit(t *testing.T) {
var inflight, peak int32
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
n := atomic.AddInt32(&inflight, 1)
for {
old := atomic.LoadInt32(&peak)
if n <= old || atomic.CompareAndSwapInt32(&peak, old, n) {
break
}
}
time.Sleep(20 * time.Millisecond)
atomic.AddInt32(&inflight, -1)
_, _ = w.Write([]byte("ok"))
}))
defer srv.Close()
const limit = 2
c := New(Options{MaxConcurrency: limit, Retries: 0})
defer c.Close()
var wg sync.WaitGroup
for i := 0; i < 8; i++ {
wg.Add(1)
go func() {
defer wg.Done()
_, _, _ = c.Get(context.Background(), srv.URL)
}()
}
wg.Wait()
if m := atomic.LoadInt32(&peak); m > limit {
t.Fatalf("并发峰值不应超过 %d,实际 %d", limit, m)
}
}

147
internal/ingest/client.go Normal file
View File

@ -0,0 +1,147 @@
package ingest
import (
"bytes"
"context"
"crypto/rand"
"encoding/hex"
"encoding/json"
"fmt"
"io"
"log"
"net/http"
"strconv"
"strings"
"time"
)
// Client 向后台 ingest 接口发送 HMAC 签名上报的客户端。
type Client struct {
Endpoint string
Secret string
HTTPClient *http.Client
}
// NewClient 创建上报客户端。endpoint 形如 http://host:8092/admin/internal/ingest。
func NewClient(endpoint, secret string) *Client {
return &Client{
Endpoint: endpoint,
Secret: secret,
HTTPClient: &http.Client{Timeout: 30 * time.Second},
}
}
// Submit 把一条走秀上报签名后 POST 到后台,返回 job_id(202 入队即返回)。
// 失败返回 error(网络/校验/服务端拒绝),调用方应记日志并继续,不中断整轮抓取。
func (c *Client) Submit(ctx context.Context, p RunwayIngest) (uint32, error) {
if c == nil || c.Endpoint == "" {
return 0, fmt.Errorf("ingest client 未配置")
}
body, err := json.Marshal(p)
if err != nil {
return 0, err
}
ts := strconv.FormatInt(time.Now().Unix(), 10)
nonce, err := newNonce()
if err != nil {
return 0, err
}
sig := Sign(c.Secret, ts, nonce, string(body))
req, err := http.NewRequestWithContext(ctx, http.MethodPost, c.Endpoint, bytes.NewReader(body))
if err != nil {
return 0, err
}
req.Header.Set("Content-Type", "application/json")
req.Header.Set("X-Signature", sig)
req.Header.Set("X-Timestamp", ts)
req.Header.Set("X-Nonce", nonce)
resp, err := c.HTTPClient.Do(req)
if err != nil {
return 0, err
}
defer resp.Body.Close()
raw, _ := io.ReadAll(resp.Body)
if resp.StatusCode != http.StatusAccepted {
return 0, fmt.Errorf("ingest http %d: %s", resp.StatusCode, string(raw))
}
var out struct {
JobID uint32 `json:"job_id"`
Status string `json:"status"`
}
if err := json.Unmarshal(raw, &out); err != nil {
// 202 但响应结构非预期:任务已入队,记日志后当作成功
log.Printf("[Ingest] 响应解析失败但已被接受: %s", string(raw))
return 0, nil
}
return out.JobID, nil
}
// newNonce 生成 16 字节随机十六进制串,作为一次性 X-Nonce 防重放。
func newNonce() (string, error) {
b := make([]byte, 16)
if _, err := rand.Read(b); err != nil {
return "", err
}
return hex.EncodeToString(b), nil
}
// CrawlBrand 后端抓取任务接口返回的单个品牌(brand_uid 已编码,name 为英文名)。
type CrawlBrand struct {
BrandUID string `json:"brand_uid"`
Name string `json:"name"`
}
// crawlBrandsURL 由 ingest endpoint 推导抓取任务接口地址(把末尾 /ingest 换成 /crawl/brands)。
func (c *Client) crawlBrandsURL() string {
base := c.Endpoint
if i := strings.LastIndex(base, "/ingest"); i != -1 {
base = base[:i]
}
return strings.TrimRight(base, "/") + "/crawl/brands"
}
// GetCrawlBrands 拉取爬虫要抓取的品牌任务列表(HMAC 签名 GET,复用与上报相同的验签算法)。
// brandID>0 时只取该品牌(对应 --brand 单品牌调试)。返回 brand_uid(直接上送 ingest)+ 英文名。
func (c *Client) GetCrawlBrands(ctx context.Context, brandID uint) ([]CrawlBrand, error) {
if c == nil || c.Endpoint == "" {
return nil, fmt.Errorf("ingest client 未配置")
}
url := c.crawlBrandsURL()
if brandID > 0 {
url += fmt.Sprintf("?brand=%d", brandID)
}
// GET 无请求体,签名串的 body 部分为空(与后台 IngestAuth 验签一致)。
ts := NowTS()
nonce, err := newNonce()
if err != nil {
return nil, err
}
sig := Sign(c.Secret, ts, nonce, "")
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
if err != nil {
return nil, err
}
req.Header.Set("X-Signature", sig)
req.Header.Set("X-Timestamp", ts)
req.Header.Set("X-Nonce", nonce)
resp, err := c.HTTPClient.Do(req)
if err != nil {
return nil, err
}
defer resp.Body.Close()
raw, _ := io.ReadAll(resp.Body)
if resp.StatusCode != http.StatusOK {
return nil, fmt.Errorf("crawl brands http %d: %s", resp.StatusCode, string(raw))
}
var out struct {
Brands []CrawlBrand `json:"brands"`
}
if err := json.Unmarshal(raw, &out); err != nil {
return nil, fmt.Errorf("crawl brands 解析失败: %w", err)
}
return out.Brands, nil
}

81
internal/ingest/hashid.go Normal file
View File

@ -0,0 +1,81 @@
package ingest
import (
"crypto/sha256"
"encoding/binary"
"strings"
)
// 以下 Feistel + base62 编码与后台 internal/pkg/hashid 完全一致,
// 仅实现编码侧(爬虫只需把品牌数字主键编码为 brand_uid 上送,无需解码)。
// 盐值必须 == 后台 HASHID_SECRET,否则生成的 brand_uid 后台解码不出正确主键。
const (
hidAlphabet = "0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ"
hidBase = 62
hidMinLen = 8
)
var hidKeys [4]uint32
// InitHashID 用与后台 HASHID_SECRET 相同的盐值初始化编码密钥。
// secret 为空时回落后台内置默认盐(仅防顺序枚举,不安全)。
func InitHashID(secret string) {
if secret == "" {
secret = "fashion-archive-default-salt-change-me"
}
h := sha256.Sum256([]byte(secret))
for i := 0; i < 4; i++ {
hidKeys[i] = binary.BigEndian.Uint32(h[i*4 : i*4+4])
}
}
// hidFeistel 32-bit 平衡 Feistel 网络,encrypt=true 加密。
func hidFeistel(v uint32, encrypt bool) uint32 {
const rounds = 8
l, r := uint16(v>>16), uint16(v&0xffff)
for i := 0; i < rounds; i++ {
idx := i
if !encrypt {
idx = rounds - 1 - i
}
round := func(h uint16) uint16 {
f := uint32(h)*0x9E3779B1 + hidKeys[idx%4]
return uint16((f ^ (f >> 16)) & 0xffff)
}
if encrypt {
nl, nr := r, uint16(l)^round(r)
l, r = nl, nr
} else {
nl, nr := uint16(l)^round(l), l
l, r = nl, nr
}
}
return uint32(l)<<16 | uint32(r)
}
func encodeNum(n uint32) string {
x := hidFeistel(n, true)
var sb strings.Builder
for x > 0 {
sb.WriteByte(hidAlphabet[x%hidBase])
x /= hidBase
}
if sb.Len() == 0 {
sb.WriteByte(hidAlphabet[0])
}
runes := []rune(sb.String())
for i, j := 0, len(runes)-1; i < j; i, j = i+1, j-1 {
runes[i], runes[j] = runes[j], runes[i]
}
out := string(runes)
if len(out) < hidMinLen {
out = strings.Repeat("0", hidMinLen-len(out)) + out
}
return out
}
// EncodeBrand 把品牌数字主键编码为 hashid 串(与后台 brand_uid 同算法)。
func EncodeBrand(id uint32) string {
return encodeNum(id)
}

27
internal/ingest/hmac.go Normal file
View File

@ -0,0 +1,27 @@
package ingest
import (
"crypto/hmac"
"crypto/sha256"
"encoding/hex"
"strconv"
"time"
)
// Sign 与后台 internal/pkg/hmac.Sign 完全一致:
// HMAC_SHA256(secret, timestamp + "." + nonce + "." + body) 的十六进制串。
// bodyRaw 必须是请求体原始字节,确保与后台收到的字节严格一致。
func Sign(secret, timestamp, nonce, body string) string {
mac := hmac.New(sha256.New, []byte(secret))
mac.Write([]byte(timestamp))
mac.Write([]byte("."))
mac.Write([]byte(nonce))
mac.Write([]byte("."))
mac.Write([]byte(body))
return hex.EncodeToString(mac.Sum(nil))
}
// NowTS 返回当前 Unix 秒字符串(X-Timestamp 用)。
func NowTS() string {
return strconv.FormatInt(time.Now().Unix(), 10)
}

View File

@ -0,0 +1,77 @@
package ingest
import (
"crypto/hmac"
"crypto/sha256"
"encoding/hex"
"strconv"
"testing"
"time"
)
// verify 是后台 internal/pkg/hmac.Verify 的逐字镜像,仅用于本测试验证「爬虫签名能被
// 后台同算法校验通过」这条契约——算法必须与后台保持一致。
func verify(secret, body, sig, ts, nonce string, ttl int) bool {
if sig == "" || ts == "" || nonce == "" {
return false
}
if ttl <= 0 {
ttl = 300
}
tsN, err := strconv.ParseInt(ts, 10, 64)
if err != nil {
return false
}
now := time.Now().Unix()
if diff := now - tsN; diff > int64(ttl) || diff < -int64(ttl) {
return false
}
mac := hmac.New(sha256.New, []byte(secret))
mac.Write([]byte(ts))
mac.Write([]byte("."))
mac.Write([]byte(nonce))
mac.Write([]byte("."))
mac.Write([]byte(body))
expected := hex.EncodeToString(mac.Sum(nil))
return hmac.Equal([]byte(expected), []byte(sig))
}
func TestSignVerifyRoundtrip(t *testing.T) {
secret := "test-secret"
body := `{"brand_uid":"001DESke","title_en":"Fall 2024 Ready-to-Wear"}`
ts := strconv.FormatInt(time.Now().Unix(), 10)
nonce := "abc123nonce"
sig := Sign(secret, ts, nonce, body)
if !verify(secret, body, sig, ts, nonce, 300) {
t.Fatal("后台应当校验通过爬虫签名,但未通过")
}
// 篡改 body
if verify(secret, body+"x", sig, ts, nonce, 300) {
t.Fatal("body 被篡改后不应通过校验")
}
// 错误密钥
if verify("wrong", body, sig, ts, nonce, 300) {
t.Fatal("错误密钥不应通过校验")
}
// 过期时间戳
old := strconv.FormatInt(time.Now().Unix()-600, 10)
if verify(secret, body, Sign(secret, old, nonce, body), old, nonce, 300) {
t.Fatal("过期时间戳不应通过校验")
}
}
func TestEncodeBrandNonSequential(t *testing.T) {
// 相邻主键编码结果应互不相关(防顺序枚举),且同值稳定
a := EncodeBrand(108)
b := EncodeBrand(109)
if a == b {
t.Fatal("相邻 id 编码必须不同")
}
if EncodeBrand(108) != a {
t.Fatal("编码必须稳定(同输入同输出)")
}
if len(a) < 8 {
t.Fatalf("编码长度不应短于 minLen=8,实际 %d", len(a))
}
}

View File

@ -0,0 +1,74 @@
// Package ingest 爬虫侧入库客户端:把走秀元数据以 HMAC 签名 POST 到后台
// ingest 接口(:8092/admin/internal/ingest),替代原先直写 brand_runway 表。
//
// 与后台 internal/pkg/hmac、internal/pkg/hashid 保持同算法,使爬虫模块完全解耦
// 于后端代码(不再 import 后端 internal 包),双方仅通过「签名 + JSON 载荷」契约通信。
package ingest
import "strings"
// 入库类型(与后台 dto 保持一致)。
const (
KindRunway = "runway" // 走秀(默认,需 brand_uid)
KindStreet = "street" // 街拍(无品牌,需 city/title)
)
// RunwayIngest 与后台 dto.RunwayIngest 的 JSON 字段一一对应。
// 爬虫只传元数据 + 图片 URL 列表,图下载/OSS 由后台 worker 完成。
type RunwayIngest struct {
Kind string `json:"kind"` // runway | street,缺省 runway
BrandUID string `json:"brand_uid"` // 品牌 hashid(EncodeBrand 生成),runway 必填
TitleEn string `json:"title_en"` // 英文标题(优先;street 作为单标题)
TitleCn string `json:"title_cn"` // 中文标题(可空)
DescriptionEn string `json:"description_en"` // 英文描述(可空)
DescriptionCn string `json:"description_cn"` // 中文描述(可空)
Year uint16 `json:"year"` // 年份
Season string `json:"season"` // spring / fall
CollectionType string `json:"collection_type"` // rtw / menswear / couture / resort / pre_fall
City string `json:"city"` // 地区/城市(street 用)
Images []string `json:"images"` // 兼容旧 worker:铺平的主图 URL 列表
// Looks 结构化「主图 + 细节图」分组(Vogue 一个 gallery item = 1 主图 + N 细节图)。
// 结构化上报时优先用 Looks;为空时 worker 回退到 Images(全部视为主图)。
Looks []RunwayLook `json:"looks,omitempty"`
}
// RunwayLook 一场秀中的一个 look:1 张主图 + 0..N 张细节图。
type RunwayLook struct {
Main string `json:"main"` // 主图(look)原始 URL
Details []string `json:"details"` // 细节图原始 URL 列表
}
// ParseCollection 从 Vogue 标题推导 collection_type 与 season,供后台补 season_code。
//
// "Fall 2024 Ready-to-Wear" -> ("rtw", "fall")
// "Spring 2025 Menswear" -> ("menswear", "spring")
// "Resort 2024" -> ("resort", "")
// "Pre-Fall 2024" -> ("pre_fall", "")
// "Couture Fall 2024" -> ("couture", "fall")
func ParseCollection(title string) (collectionType, season string) {
t := strings.ToLower(title)
switch {
case strings.Contains(t, "resort"):
collectionType = "resort"
case strings.Contains(t, "pre-fall"), strings.Contains(t, "pre fall"):
collectionType = "pre_fall"
case strings.Contains(t, "couture"):
collectionType = "couture"
case strings.Contains(t, "menswear"), strings.Contains(t, "men's"):
collectionType = "menswear"
case strings.Contains(t, "ready"):
collectionType = "rtw"
default:
// 仅出现 spring/fall 而无明确品类词时,回落为 rtw(绝大多数季场秀)
if strings.Contains(t, "spring") || strings.Contains(t, "fall") {
collectionType = "rtw"
}
}
switch {
case strings.Contains(t, "spring"):
season = "spring"
case strings.Contains(t, "fall"):
season = "fall"
}
return collectionType, season
}

7
internal/spider/regex.go Normal file
View File

@ -0,0 +1,7 @@
package spider
import "regexp"
// yearRe 预编译的 4 位年份正则,供 vogue / theimpression 共用
// (预编译避免在每次解析时重新编译)。
var yearRe = regexp.MustCompile(`\b(19|20)\d{2}\b`)

View File

@ -4,10 +4,7 @@ import (
"context"
"encoding/json"
"fmt"
"io"
"log"
"net/http"
"regexp"
"strconv"
"strings"
"sync"
@ -15,6 +12,7 @@ import (
"github.com/PuerkitoBio/goquery"
"my-spiders/internal/fetch"
"my-spiders/internal/ingest"
)
@ -28,10 +26,11 @@ const (
// TheImpressionSpider The Impression 街拍采集器(对应 PHP 的 TheImpressionStreetCommand)
type TheImpressionSpider struct {
Client *http.Client
Ingest *ingest.Client // 入库管线客户端(nil = 无法入库,仅 Debug 输出)
MaxCo int // 最大并发数
Debug bool // Debug 模式:仅输出不入库
Fetcher *fetch.Client // 带全局并发限流 + 重试的 HTTP 客户端
Ingest *ingest.Client // 入库管线客户端(nil = 无法入库,仅 Debug 输出)
MaxCo int // 最大并发数
MaxArticles int // 小批量模式:最多抓取的街拍任务数(0=不限制)
Debug bool // Debug 模式:仅输出不入库
}
func NewTheImpressionSpider(maxCo int, ingestClient *ingest.Client) *TheImpressionSpider {
@ -41,9 +40,13 @@ func NewTheImpressionSpider(maxCo int, ingestClient *ingest.Client) *TheImpressi
return &TheImpressionSpider{
Ingest: ingestClient,
MaxCo: maxCo,
Client: &http.Client{
Timeout: 30 * time.Second,
},
// 全局并发上限 = maxCo;叠加 200ms 最小间隔做全局限速(替代原先的 time.Sleep)。
Fetcher: fetch.New(fetch.Options{
MaxConcurrency: maxCo,
MinInterval: 200 * time.Millisecond,
Retries: fetch.DefaultRetries,
Backoff: time.Second,
}),
}
}
@ -59,13 +62,13 @@ func (s *TheImpressionSpider) Run() {
// 1. 抓取 street-style 首页 banner(.parallax .mask-overlay)
bannerURL := TheImpressionBaseURL + "/street-style"
body, httpCode := s.request(bannerURL)
if httpCode != 200 || body == "" {
log.Printf("[错误] %s 请求失败, code: %d", bannerURL, httpCode)
bannerBody, bannerCode, bannerErr := s.Fetcher.Get(context.Background(), bannerURL)
if bannerErr != nil || bannerCode != 200 || len(bannerBody) == 0 {
log.Printf("[错误] %s 请求失败, code: %d, err: %v", bannerURL, bannerCode, bannerErr)
return
}
doc, err := goquery.NewDocumentFromReader(strings.NewReader(body))
doc, err := goquery.NewDocumentFromReader(strings.NewReader(string(bannerBody)))
if err != nil {
log.Printf("[错误] 解析 %s 失败: %v", bannerURL, err)
return
@ -89,15 +92,15 @@ func (s *TheImpressionSpider) Run() {
"%s/wp-json/codetipi-zeen/v1/block?paged=%d&type=1&data%%5Bargs%%5D%%5Bcat%%5D=1",
TheImpressionBaseURL, i,
)
body, httpCode := s.request(listURL)
if httpCode != 200 || body == "" {
log.Printf("[错误] %s 请求失败, code: %d", listURL, httpCode)
listBody, listCode, listErr := s.Fetcher.Get(context.Background(), listURL)
if listErr != nil || listCode != 200 || len(listBody) == 0 {
log.Printf("[错误] %s 请求失败, code: %d, err: %v", listURL, listCode, listErr)
return
}
// 接口返回的是 JSON 数组,第 2 个元素(index 1)是包含文章的 HTML 片段
var arr []interface{}
if err := json.Unmarshal([]byte(body), &arr); err != nil {
if err := json.Unmarshal(listBody, &arr); err != nil {
log.Printf("[错误] %s JSON 解析失败: %v", listURL, err)
continue
}
@ -128,21 +131,21 @@ func (s *TheImpressionSpider) Run() {
})
}
// 小批量模式:仅抓取前 N 个街拍(用于试抓 / 控制入库量)。
if s.MaxArticles > 0 && len(tasks) > s.MaxArticles {
log.Printf("[Info] 小批量模式:仅抓取前 %d / %d 个街拍", s.MaxArticles, len(tasks))
tasks = tasks[:s.MaxArticles]
}
log.Printf("[Info] 共收集到 %d 个待抓取任务,开始执行...", len(tasks))
// 3. 并发抓取详情页
sem := make(chan struct{}, s.MaxCo)
// 3. 并发抓取详情页:不再自建信号量,HTTP 并发与频控统一由 s.Fetcher 承担。
var wg sync.WaitGroup
for _, t := range tasks {
sem <- struct{}{}
wg.Add(1)
go func(tk streetTask) {
defer func() {
<-sem
wg.Done()
}()
defer wg.Done()
s.getDetail(tk)
time.Sleep(200 * time.Millisecond) // 频控,防封 IP
}(t)
}
wg.Wait()
@ -158,13 +161,13 @@ func (s *TheImpressionSpider) getDetail(tk streetTask) {
return
}
body, httpCode := s.request(url)
if httpCode != 200 || body == "" {
log.Printf("[错误] %s 请求失败, code: %d", url, httpCode)
body, code, err := s.Fetcher.Get(context.Background(), url)
if err != nil || code != 200 || len(body) == 0 {
log.Printf("[错误] %s 请求失败, code: %d, err: %v", url, code, err)
return
}
doc, err := goquery.NewDocumentFromReader(strings.NewReader(body))
doc, err := goquery.NewDocumentFromReader(strings.NewReader(string(body)))
if err != nil {
log.Printf("[错误] 解析 %s 失败: %v", url, err)
return
@ -233,37 +236,15 @@ func (s *TheImpressionSpider) getDetail(tk streetTask) {
}
}
// request HTTP 请求封装(带浏览器 UA)
func (s *TheImpressionSpider) request(url string) (string, int) {
req, err := http.NewRequest("GET", url, nil)
if err != nil {
return "", 0
}
req.Header.Set("User-Agent", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36")
req.Header.Set("Accept", "text/html,application/json,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8")
resp, err := s.Client.Do(req)
if err != nil {
return "", 0
}
defer resp.Body.Close()
bodyBytes, err := io.ReadAll(resp.Body)
if err != nil {
return "", resp.StatusCode
}
return string(bodyBytes), resp.StatusCode
}
// parseYear 从标题中提取 4 位年份数字(对应 PHP 的 AppHelper::getYear)
// parseYear 从标题中提取 4 位年份(对应 PHP 的 AppHelper::getYear)。
// 提取不到时返回 0(未知年份),不再回退到「当前年份」——那会把旧街拍伪造成今年。
func (s *TheImpressionSpider) parseYear(title string) int {
re := regexp.MustCompile(`\b(19|20)\d{2}\b`)
match := re.FindString(title)
if match != "" {
y, _ := strconv.Atoi(match)
if m := yearRe.FindString(title); m != "" {
y, _ := strconv.Atoi(m)
return y
}
return time.Now().Year()
log.Printf("[Warning] 标题未含年份,按未知(0)处理: %s", title)
return 0
}
// streetCities 常见时装周城市(用于从街拍标题中粗提 city 字段)。

View File

@ -2,9 +2,7 @@ package spider
import (
"context"
"io"
"log"
"net/http"
"regexp"
"strconv"
"strings"
@ -13,6 +11,7 @@ import (
"github.com/tidwall/gjson"
"my-spiders/internal/fetch"
"my-spiders/internal/ingest"
)
@ -21,8 +20,15 @@ const (
VoguePlatform = "vogue"
)
// 预编译正则(避免每次调用都重新编译)。
// 列表页与详情页的内联 state 结束标记不同,故保留两条,勿轻易合并。
var (
showsStateRe = regexp.MustCompile(`(?s)window\.__PRELOADED_STATE__\s*=\s*(.*?);</script>`)
detailStateRe = regexp.MustCompile(`(?s)window\.__PRELOADED_STATE__\s*=\s*(.*?);<`)
)
type VogueSpider struct {
Client *http.Client
Fetcher *fetch.Client // 带全局并发限流 + 重试的 HTTP 客户端
Ingest *ingest.Client // 入库管线客户端(nil = 离线/Mock 模式)
MaxCo int // 最大并发数
BrandID uint // 客户端指定的品牌 ID(--brand,单品牌调试用)
@ -41,9 +47,14 @@ func NewVogueSpider(maxCo int, brandID uint, maxShows int, ingestClient *ingest.
MaxCo: maxCo,
BrandID: brandID,
MaxShows: maxShows,
Client: &http.Client{
Timeout: 30 * time.Second,
},
// 全局并发上限 = maxCo(单一信号量,避免「品牌级 × 详情级」叠加成 maxCo²);
// 再叠加 200ms 最小间隔做全局限速,替代原先散落在 goroutine 里的 time.Sleep。
Fetcher: fetch.New(fetch.Options{
MaxConcurrency: maxCo,
MinInterval: 200 * time.Millisecond,
Retries: fetch.DefaultRetries,
Backoff: time.Second,
}),
}
}
@ -57,7 +68,8 @@ func (s *VogueSpider) Run() {
log.Printf("[Info] 成功获取 %d 个品牌任务,开始执行...", len(tasks))
// 品牌级别的并发控制
// 品牌级并发:仅限制「同时在跑的品牌 goroutine 数」,避免一次性对巨量品牌开 goroutine;
// 真正的 HTTP 并发由 Fetcher 的全局信号量统一控制(修正此前两层信号量叠加成 MaxCo² 的问题)。
brandSem := make(chan struct{}, s.MaxCo)
var wg sync.WaitGroup
@ -140,24 +152,16 @@ func (s *VogueSpider) SpiderStart(task ingest.CrawlBrand) {
showsList = showsList[:s.MaxShows]
}
// 核心修复:限制该品牌下发布会详情页的抓取并发数
detailSem := make(chan struct{}, s.MaxCo)
// 详情页并发抓取:这里不再自建信号量,HTTP 并发与频控统一由 s.Fetcher 承担
// (避免与品牌级信号量叠加导致并发放大)。
var wg sync.WaitGroup
for _, list := range showsList {
detailSem <- struct{}{} // 超过 MaxCo 时会自动卡住等待
wg.Add(1)
go func(info gjson.Result) {
defer func() {
<-detailSem // 释放槽位
wg.Done()
}()
defer wg.Done()
s.getDetail(task.BrandUID, info)
// 频控:每次下载后稍作停顿,保护带宽并防封 IP
time.Sleep(200 * time.Millisecond)
}(list)
}
@ -166,17 +170,19 @@ func (s *VogueSpider) SpiderStart(task ingest.CrawlBrand) {
// getShowsList 获取品牌的发布会列表 JSON 数据
func (s *VogueSpider) getShowsList(url string) []gjson.Result {
body, httpCode := s.request(url)
if httpCode == 200 && body != "" {
re := regexp.MustCompile(`(?s)window\.__PRELOADED_STATE__\s*=\s*(.*?);</script>`)
matches := re.FindStringSubmatch(body)
if len(matches) > 1 {
collections := gjson.Get(matches[1], "transformed.runwayDesignerContent.designerCollections")
return collections.Array()
}
body, code, err := s.Fetcher.Get(context.Background(), url)
if err != nil {
log.Printf("[错误] 获取发布会列表失败 %s: %v", url, err)
return nil
}
if code != 200 || len(body) == 0 {
log.Printf("[Info] %s 返回异常, code: %d", url, code)
return nil
}
if m := showsStateRe.FindSubmatch(body); len(m) > 1 {
collections := gjson.GetBytes(m[1], "transformed.runwayDesignerContent.designerCollections")
return collections.Array()
}
log.Printf("[Info] %s 未找到数据.", url)
return nil
}
@ -191,20 +197,22 @@ func (s *VogueSpider) getDetail(brandUID string, info gjson.Result) {
requestURL := s.sourceURL(info)
log.Printf("正在匹配发布会详情 %s", requestURL)
body, httpCode := s.request(requestURL)
if httpCode != 200 || body == "" {
log.Printf("[Warning] %s 请求失败.", requestURL)
body, code, err := s.Fetcher.Get(context.Background(), requestURL)
if err != nil {
log.Printf("[Warning] %s 请求失败: %v", requestURL, err)
return
}
if code != 200 || len(body) == 0 {
log.Printf("[Warning] %s 返回异常, code: %d", requestURL, code)
return
}
re := regexp.MustCompile(`(?s)window\.__PRELOADED_STATE__\s*=\s*(.*?);<`)
matches := re.FindStringSubmatch(body)
matches := detailStateRe.FindSubmatch(body)
if len(matches) <= 1 {
return
}
imagesResult := gjson.Get(matches[1], "transformed.runwayGalleries.galleries.0.items")
imagesResult := gjson.GetBytes(matches[1], "transformed.runwayGalleries.galleries.0.items")
if !imagesResult.Exists() {
log.Printf("[Warning] %s 获取图片失败.", requestURL)
@ -279,36 +287,14 @@ func (s *VogueSpider) getTaskName(name string) string {
return strings.ToLower(r.Replace(name))
}
// HTTP 请求封装
func (s *VogueSpider) request(url string) (string, int) {
req, err := http.NewRequest("GET", url, nil)
if err != nil {
return "", 0
}
req.Header.Set("User-Agent", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36")
resp, err := s.Client.Do(req)
if err != nil {
return "", 0
}
defer resp.Body.Close()
bodyBytes, err := io.ReadAll(resp.Body)
if err != nil {
return "", resp.StatusCode
}
return string(bodyBytes), resp.StatusCode
}
// parseYear 从标题中提取 4 位年份数字(如 Fall 2024 Ready-to-Wear -> 2024)
// parseYear 从标题中提取 4 位年份(如 Fall 2024 Ready-to-Wear -> 2024)。
// 提取不到时返回 0(未知年份),交由后端 season.Derive 处理(year=0 → season_code 为空);
// 绝不回退到「当前年份」——那会把旧秀伪造成今年,污染排序与去重。
func (s *VogueSpider) parseYear(title string) int {
re := regexp.MustCompile(`\b(19|20)\d{2}\b`)
match := re.FindString(title)
if match != "" {
y, _ := strconv.Atoi(match)
if m := yearRe.FindString(title); m != "" {
y, _ := strconv.Atoi(m)
return y
}
return time.Now().Year()
log.Printf("[Warning] 标题未含年份,按未知(0)处理: %s", title)
return 0
}