This commit is contained in:
toom1996
2026-09-07 00:04:01 +08:00
parent 6a5a4378ab
commit 10d8a96e8c
29 changed files with 5349 additions and 0 deletions

23
internal/dto/crawl.go Normal file
View File

@ -0,0 +1,23 @@
package dto
// CrawlBrand 爬虫取任务接口返回的单个品牌。
// brand_uid 为 hashid 编码串(与对外一致),spider 直接拿它上送 ingest,无需自己编码;
// name 为英文品牌名,供 spider 拼接到 vogue 抓取 URL(如 /fashion-shows/designer/<slug>)。
type CrawlBrand struct {
BrandUID string `json:"brand_uid"`
Name string `json:"name"`
}
// CrawlExistsRequest 图集预检请求:一次问一批 source_url 是否已爬取过。
// 设计为批量而非逐个,是因为爬虫在拿到列表页后能一次拼出几十上百个图集链接,
// 逐个问会退化成 N 次 HTTP 往返,反而比直接抓取更慢。
type CrawlExistsRequest struct {
SourceURLs []string `json:"source_urls"`
}
// CrawlExistsResponse 图集预检响应。existing 为「已爬取过、无需再抓」的 source_url 列表;
// 未出现在其中的即认为需要抓取。
type CrawlExistsResponse struct {
Existing []string `json:"existing"` // 已爬取过的 source_url,爬虫应跳过
ExistingCount int `json:"existing_count"` // 命中数量,便于爬虫打日志观察命中率
}