package handler import ( "errors" "net/http" "sort" "strconv" "fashionapi/internal/dto" "fashionapi/internal/service" "github.com/gin-gonic/gin" ) // IngestHandler 爬虫上报入口(:8092 的 /admin/internal/ingest)。 // // 安全:不挂后台登录 cookie 中间件,改走 HMAC 验签中间件(middleware.IngestAuth); // 因此多节点爬虫可直连上报,无需回环绑定。接口本身很薄:验签 → Submit 入队 → 立即 202。 type IngestHandler struct { ingest *service.IngestService brands service.BrandService } // NewIngestHandler 创建入库 handler。 func NewIngestHandler(ingest *service.IngestService, brands service.BrandService) *IngestHandler { return &IngestHandler{ingest: ingest, brands: brands} } // Submit 接收单场走秀上报,入队后返回 202 Accepted(异步处理)。 func (h *IngestHandler) Submit(c *gin.Context) { var p dto.RunwayIngest if err := c.ShouldBindJSON(&p); err != nil { c.JSON(http.StatusBadRequest, gin.H{"error": "invalid json: " + err.Error()}) return } id, err := h.ingest.Submit(c.Request.Context(), p) if err != nil { // 业务错误(缺字段 / 未知品牌等)映射为 4xx,其余 5xx if errors.Is(err, service.ErrBrandNotFound) || isClientErr(err) { c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()}) return } c.JSON(http.StatusInternalServerError, gin.H{"error": "enqueue failed"}) return } c.JSON(http.StatusAccepted, gin.H{"job_id": id, "status": "queued"}) } // CrawlBrands 爬虫取任务接口(GET /admin/internal/crawl/brands,复用 IngestAuth HMAC 验签)。 // 返回所有可抓取品牌的 brand_uid(hashid)+ 英文品牌名,供 spider 拼 vogue 抓取 URL 并上送 ingest。 func (h *IngestHandler) CrawlBrands(c *gin.Context) { // 支持 ?brand= 单品牌调试:解析失败或非正数时回落全量。 var brandID uint32 if v := c.Query("brand"); v != "" { if id, err := strconv.ParseUint(v, 10, 32); err == nil { brandID = uint32(id) } } brands, err := h.brands.CrawlTasks(c.Request.Context(), brandID) if err != nil { c.JSON(http.StatusInternalServerError, gin.H{"error": "fetch crawl tasks failed"}) return } c.JSON(http.StatusOK, gin.H{"brands": brands}) } // CrawlExists 图集预检接口(POST /admin/internal/crawl/exists,复用 IngestAuth HMAC 验签)。 // // 爬虫在抓详情页之前批量问「这些 source_url 是否已爬取过」,跳过已存在的图集, // 省掉「抓详情页 → 提取图片 URL → 上报 → worker 判重后丢弃」这一整套无效动作。 // 判定条件与 worker 判重一致,因此跳过是安全的:不会漏抓,也不会重复抓。 func (h *IngestHandler) CrawlExists(c *gin.Context) { var req dto.CrawlExistsRequest if err := c.ShouldBindJSON(&req); err != nil { c.JSON(http.StatusBadRequest, gin.H{"error": "invalid json: " + err.Error()}) return } exists, err := h.ingest.ExistsSourceURLs(c.Request.Context(), req.SourceURLs) if err != nil { c.JSON(http.StatusInternalServerError, gin.H{"error": "query failed"}) return } existing := make([]string, 0, len(exists)) for u := range exists { existing = append(existing, u) } // 排序保证输出稳定,便于日志比对与测试断言。 sort.Strings(existing) c.JSON(http.StatusOK, dto.CrawlExistsResponse{ Existing: existing, ExistingCount: len(existing), }) } func isClientErr(err error) bool { // service 层 NewError 带 http 状态码;这里简单按消息前缀判断,避免暴露内部类型。 if err == nil { return false } msg := err.Error() return len(msg) > 0 && (msg == "source_url required" || msg == "brand_uid required") }